switchroom 0.18.10 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/dist/agent-scheduler/index.js +29 -5
  2. package/dist/auth-broker/index.js +53 -13
  3. package/dist/cli/hindsight-mental-model-pretool.mjs +39 -0
  4. package/dist/cli/notion-write-pretool.mjs +29 -5
  5. package/dist/cli/switchroom.js +2636 -1369
  6. package/dist/cli/ui/apple-touch-icon.png +0 -0
  7. package/dist/cli/ui/favicon-32.png +0 -0
  8. package/dist/cli/ui/favicon.ico +0 -0
  9. package/dist/cli/ui/index.html +163 -17
  10. package/dist/host-control/main.js +1248 -342
  11. package/dist/vault/approvals/kernel-server.js +54 -13
  12. package/dist/vault/broker/server.js +163 -114
  13. package/package.json +3 -4
  14. package/profiles/_base/start.sh.hbs +65 -0
  15. package/profiles/_shared/vault-protocol.md.hbs +3 -1
  16. package/profiles/coding/CLAUDE.md.hbs +1 -1
  17. package/profiles/default/CLAUDE.md.hbs +2 -2
  18. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  19. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  20. package/telegram-plugin/bridge/bridge.ts +37 -0
  21. package/telegram-plugin/bridge/inbound-dedup.ts +101 -0
  22. package/telegram-plugin/dist/bridge/bridge.js +73 -1
  23. package/telegram-plugin/dist/gateway/gateway.js +3603 -1007
  24. package/telegram-plugin/dist/server.js +74 -2
  25. package/telegram-plugin/flood-circuit-breaker.ts +493 -21
  26. package/telegram-plugin/gateway/approval-hold.ts +583 -0
  27. package/telegram-plugin/gateway/auth-command.ts +92 -2
  28. package/telegram-plugin/gateway/auth-loopback-relay.ts +670 -0
  29. package/telegram-plugin/gateway/boot-card.ts +12 -5
  30. package/telegram-plugin/gateway/callback-query-handlers.ts +76 -1
  31. package/telegram-plugin/gateway/config-approval-handler.ts +6 -1
  32. package/telegram-plugin/gateway/disconnect-flush.ts +19 -0
  33. package/telegram-plugin/gateway/dm-pin-sweep.test.ts +251 -0
  34. package/telegram-plugin/gateway/dm-pin-sweep.ts +178 -0
  35. package/telegram-plugin/gateway/gateway.ts +1482 -165
  36. package/telegram-plugin/gateway/hostd-dispatch.ts +23 -0
  37. package/telegram-plugin/gateway/idle-clear.ts +90 -6
  38. package/telegram-plugin/gateway/inbound-delivery-machine-shadow.ts +26 -5
  39. package/telegram-plugin/gateway/inject-handler.ts +8 -0
  40. package/telegram-plugin/gateway/ipc-protocol.ts +46 -3
  41. package/telegram-plugin/gateway/ipc-server.ts +43 -0
  42. package/telegram-plugin/gateway/mental-model-propose-resolve.ts +145 -37
  43. package/telegram-plugin/gateway/model-command.ts +9 -3
  44. package/telegram-plugin/gateway/pending-session-command.ts +13 -1
  45. package/telegram-plugin/gateway/permission-ttl-sweep.ts +66 -0
  46. package/telegram-plugin/gateway/pre-approval-check.ts +74 -0
  47. package/telegram-plugin/gateway/queued-card-store.ts +217 -0
  48. package/telegram-plugin/gateway/session-model-file.ts +26 -1
  49. package/telegram-plugin/gateway/turn-end-gate-backstop.ts +59 -0
  50. package/telegram-plugin/gateway/turn-end-gate.ts +95 -0
  51. package/telegram-plugin/gateway/turn-typing-loop.ts +10 -2
  52. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +13 -0
  53. package/telegram-plugin/hooks/dispatch-claim-scan.mjs +259 -0
  54. package/telegram-plugin/hooks/dispatch-claim-stop.mjs +129 -0
  55. package/telegram-plugin/hooks/hooks.json +9 -0
  56. package/telegram-plugin/inline-keyboard-callbacks.ts +209 -2
  57. package/telegram-plugin/operator-events.ts +23 -0
  58. package/telegram-plugin/package.json +0 -1
  59. package/telegram-plugin/permission-rule.ts +1 -0
  60. package/telegram-plugin/permission-title.ts +1 -0
  61. package/telegram-plugin/retry-api-call.ts +212 -2
  62. package/telegram-plugin/send-gate-degraded.test.ts +443 -0
  63. package/telegram-plugin/send-gate-observability.test.ts +470 -0
  64. package/telegram-plugin/send-gate-observability.ts +355 -0
  65. package/telegram-plugin/send-gate.test.ts +698 -0
  66. package/telegram-plugin/send-gate.ts +982 -0
  67. package/telegram-plugin/shared/bot-runtime.ts +17 -5
  68. package/telegram-plugin/shared/gw-trace-gate.ts +105 -0
  69. package/telegram-plugin/status-pin-driver.ts +52 -7
  70. package/telegram-plugin/status-pin.ts +81 -0
  71. package/telegram-plugin/subagent-watcher.ts +102 -2
  72. package/telegram-plugin/tests/activity-card-wiring.test.ts +18 -5
  73. package/telegram-plugin/tests/approval-hold-harness.ts +425 -0
  74. package/telegram-plugin/tests/approval-hold-outcome.test.ts +296 -0
  75. package/telegram-plugin/tests/approval-hold-record.test.ts +531 -0
  76. package/telegram-plugin/tests/approval-hold-redeliver.test.ts +602 -0
  77. package/telegram-plugin/tests/auth-loopback-relay.test.ts +533 -0
  78. package/telegram-plugin/tests/boot-card-flood-suppress.test.ts +53 -7
  79. package/telegram-plugin/tests/busy-key-reaper.test.ts +1 -0
  80. package/telegram-plugin/tests/dispatch-claim-scan.test.ts +250 -0
  81. package/telegram-plugin/tests/flood-breaker-blindness.test.ts +213 -0
  82. package/telegram-plugin/tests/flood-windows-persistence.test.ts +224 -0
  83. package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
  84. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +29 -1
  85. package/telegram-plugin/tests/gateway-loopback-paste-redact.test.ts +66 -0
  86. package/telegram-plugin/tests/gw-trace-gate.test.ts +105 -0
  87. package/telegram-plugin/tests/idle-clear.test.ts +233 -3
  88. package/telegram-plugin/tests/inbound-dedup.test.ts +93 -0
  89. package/telegram-plugin/tests/inline-keyboard-callbacks.test.ts +284 -0
  90. package/telegram-plugin/tests/ipc-server-check-pre-approved.test.ts +194 -0
  91. package/telegram-plugin/tests/mental-model-propose-resolve.test.ts +123 -0
  92. package/telegram-plugin/tests/missed-approvals-wiring.test.ts +1 -1
  93. package/telegram-plugin/tests/model-command.test.ts +14 -0
  94. package/telegram-plugin/tests/pending-session-command.test.ts +21 -0
  95. package/telegram-plugin/tests/permission-card-routing.test.ts +30 -5
  96. package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +8 -7
  97. package/telegram-plugin/tests/permission-rearm-wiring.test.ts +1 -1
  98. package/telegram-plugin/tests/pre-approval-check.test.ts +148 -0
  99. package/telegram-plugin/tests/queued-card-store.test.ts +232 -0
  100. package/telegram-plugin/tests/reaction-flush-turn-gated.test.ts +100 -0
  101. package/telegram-plugin/tests/retry-api-call.test.ts +398 -0
  102. package/telegram-plugin/tests/session-model-file.test.ts +50 -0
  103. package/telegram-plugin/tests/status-pin-boot-recovery.test.ts +35 -14
  104. package/telegram-plugin/tests/status-pin.test.ts +275 -1
  105. package/telegram-plugin/tests/subagent-watcher-deferral-log-ratelimit.test.ts +316 -0
  106. package/telegram-plugin/tests/turn-end-gate-backstop.test.ts +92 -0
  107. package/telegram-plugin/tests/turn-end-gate.test.ts +137 -0
  108. package/telegram-plugin/tests/typing-emitter.test.ts +586 -0
  109. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +20 -0
  110. package/telegram-plugin/typing-emitter.ts +224 -0
  111. package/telegram-plugin/uat/scenarios/jtbd-feel-like-a-colleague-dm.test.ts +136 -0
  112. package/telegram-plugin/welcome-text.ts +42 -0
  113. package/vendor/hindsight-memory/scripts/drain_pending.py +22 -6
  114. package/vendor/hindsight-memory/scripts/lib/client.py +12 -5
  115. package/vendor/hindsight-memory/scripts/lib/directives.py +38 -3
  116. package/vendor/hindsight-memory/scripts/lib/pending.py +36 -9
  117. package/vendor/hindsight-memory/scripts/session_end.py +14 -3
  118. package/vendor/hindsight-memory/scripts/session_start.py +21 -0
  119. package/vendor/hindsight-memory/scripts/tests/test_directives.py +38 -0
  120. package/vendor/hindsight-memory/tests/test_drain_pending.py +68 -0
  121. package/vendor/hindsight-memory/tests/test_pending.py +44 -0
  122. package/vendor/hindsight-memory/tests/test_session_end_pending.py +38 -0
  123. package/vendor/hindsight-memory/tests/test_session_start_drain.py +155 -0
  124. package/telegram-plugin/channel-envelope-safety.test.ts +0 -56
  125. package/telegram-plugin/channel-envelope-safety.ts +0 -56
@@ -0,0 +1,224 @@
1
+ /**
2
+ * The single gate every `sendChatAction` in the gateway passes through (#3084).
3
+ *
4
+ * WHY THIS EXISTS — the 2026-07-11 flood ban
5
+ * ------------------------------------------
6
+ * The typing indicator used to be "a 4 s interval", which sounds rate-limited
7
+ * and is not. Both typing loops (`startTypingLoop` in gateway.ts and the
8
+ * turn-level `createTurnTypingLoop`) are restart-safe by design: a re-start
9
+ * clears the old interval and fires ONE action immediately so "typing…" lands
10
+ * instantly. The tool-use wrapper restarts the loop on every tool call, so the
11
+ * ping rate tracked the AGENT'S TOOL-CALL RATE, not the 4 s cadence — the
12
+ * interval was decorative. On 2026-07-11 `overlord` emitted 8,729
13
+ * sendChatAction calls (55% of all outbound volume, bursting at 200-300/min
14
+ * into ONE DM) to deliver 203 messages, and earned a per-bot-token flood ban:
15
+ * `429 retry_after=16739s` — 4.6 hours with every outbound reply rejected.
16
+ *
17
+ * The old code comment said the redundant pings were "harmless — same action,
18
+ * and sendChatAction is cheap." They are not cheap. They spend the per-bot
19
+ * flood budget the REPLIES need.
20
+ *
21
+ * WHAT THIS ENFORCES
22
+ * ------------------
23
+ * 1. A per-chat-key emission FLOOR. At most one chat action per key per
24
+ * `floorMs` (~4 s refresh window), no matter how many loops restart, from
25
+ * which caller, on which surface. Both loops share ONE emitter, so the
26
+ * floor holds ACROSS them — they target the same chat key and neither can
27
+ * out-shout the other. This is what structurally decouples ping rate from
28
+ * tool-call rate.
29
+ * 2. The UX intent survives: a COLD start (no ping for this key inside the
30
+ * floor) still fires immediately, so "typing…" lands the moment a turn
31
+ * begins. Only the REDUNDANT restarts are dropped — and a dropped tick is
32
+ * never silently lost: it arms a COALESCED catch-up at the instant the
33
+ * window opens, so the worst-case gap between two emissions is `floorMs`
34
+ * (3.5 s), comfortably inside Telegram's ~5 s action expiry. Without it,
35
+ * an eaten tick on a fixed-cadence loop pushes the next emission out to
36
+ * `floorMs + refreshMs` = 7.5 s and the chat goes DARK mid-turn — trading
37
+ * a flood ban for a dead indicator, which is not a trade we make.
38
+ * 3. NON-ESSENTIAL by definition: while a flood-wait window is open
39
+ * (`isSuppressed`, wired to the #2923 circuit breaker) no typing is
40
+ * emitted at all. It cannot succeed, and sending into an open window can
41
+ * extend the ban.
42
+ *
43
+ * The window is claimed on every attempt that PASSES the floor, including one
44
+ * suppressed by a flood window. That is deliberate: it bounds the flood-state
45
+ * read to once per floor per chat instead of once per tool call.
46
+ *
47
+ * Pure + injectable (clock, send, chat-key, suppression) so the whole gate is
48
+ * unit-testable on a fake clock without a bot — `tests/typing-emitter.test.ts`.
49
+ */
50
+
51
+ /** Refresh cadence of the typing loops. Telegram's `typing` expires at ~5 s. */
52
+ export const TYPING_REFRESH_MS = 4000
53
+
54
+ /**
55
+ * Minimum gap between two chat actions on one chat key. Deliberately a hair
56
+ * UNDER `TYPING_REFRESH_MS` so a loop's own on-time refresh is never eaten by
57
+ * timer jitter (a 3999 ms tick would be dropped by a 4000 ms floor, and the
58
+ * indicator would go dark for a whole window). The gap between successive
59
+ * emissions therefore stays under Telegram's ~5 s action expiry, while the
60
+ * worst-case rate falls from the observed ~300/min to ~17/min per chat.
61
+ */
62
+ export const TYPING_FLOOR_MS = 3500
63
+
64
+ export interface TypingEmitterDeps {
65
+ /** Perform the actual chat action. Errors are the caller's concern — the
66
+ * emitter never throws and never awaits. */
67
+ send: (chatId: string, threadId: number | null, action: string) => void
68
+ /** Canonical chat:thread key (the gateway's `chatKey`) — a supergroup topic
69
+ * is its own lane and gets its own floor. */
70
+ chatKey: (chatId: string, threadId: number | null) => string
71
+ /** True while a flood-wait window is open. Typing is non-essential: while
72
+ * this is true nothing is emitted. Defaults to "never suppressed". */
73
+ isSuppressed?: () => boolean
74
+ /** Injected clock (tests pass a fake). */
75
+ now?: () => number
76
+ /** Per-key emission floor in ms. */
77
+ floorMs?: number
78
+ /** Injected timer for the catch-up (tests pass fake timers). */
79
+ schedule?: (fn: () => void, ms: number) => unknown
80
+ /** Cancel a handle returned by `schedule`. */
81
+ cancel?: (handle: unknown) => void
82
+ /** Observability hook — fires for each dropped emission. */
83
+ onDrop?: (info: { key: string; reason: 'floor' | 'flood' }) => void
84
+ }
85
+
86
+ export interface TypingEmitter {
87
+ /**
88
+ * Emit one chat action for (chatId, threadId), subject to the floor and the
89
+ * flood gate. Returns true iff the action was actually handed to `send`.
90
+ * A floor-dropped emission arms a coalesced catch-up (see below), so the
91
+ * caller's cadence is never silently lost.
92
+ */
93
+ emit: (chatId: string, threadId?: number | null, action?: string) => boolean
94
+ /**
95
+ * Cancel any pending catch-up for a chat key. The canonical turn-end calls
96
+ * this so a dropped tick can't resurrect "typing…" after the turn is over.
97
+ * Does NOT clear the floor — the floor must survive loop stops, because the
98
+ * tool loop stops on every tool result and that is precisely the churn the
99
+ * floor exists to absorb.
100
+ */
101
+ cancelPending: (chatId: string, threadId?: number | null) => void
102
+ /** Test/observability: number of chat keys currently tracked. */
103
+ trackedKeys: () => number
104
+ /** Test/observability: number of catch-ups currently armed (0 ⇒ none leaked). */
105
+ pendingCatchUps: () => number
106
+ /** Drop all floor state and cancel every catch-up (shutdown / tests). */
107
+ reset: () => void
108
+ }
109
+
110
+ /** Keep the floor map from growing without bound in a long-lived gateway. */
111
+ const PRUNE_AFTER_FACTOR = 10
112
+ const PRUNE_SIZE_THRESHOLD = 64
113
+
114
+ export function createTypingEmitter(deps: TypingEmitterDeps): TypingEmitter {
115
+ const floorMs = deps.floorMs ?? TYPING_FLOOR_MS
116
+ const now = deps.now ?? Date.now
117
+ const isSuppressed = deps.isSuppressed ?? (() => false)
118
+ const schedule =
119
+ deps.schedule ??
120
+ ((fn: () => void, ms: number) => {
121
+ const h = setTimeout(fn, ms)
122
+ ;(h as { unref?: () => void }).unref?.()
123
+ return h
124
+ })
125
+ const cancel =
126
+ deps.cancel ?? ((h: unknown) => clearTimeout(h as ReturnType<typeof setTimeout>))
127
+
128
+ /** chat key → epoch ms at which the current emission window was claimed. */
129
+ const windowClaimedAt = new Map<string, number>()
130
+ /** chat key → the single armed catch-up timer (coalesced; never two). */
131
+ const catchUps = new Map<string, unknown>()
132
+
133
+ function prune(t: number): void {
134
+ if (windowClaimedAt.size <= PRUNE_SIZE_THRESHOLD) return
135
+ const cutoff = t - floorMs * PRUNE_AFTER_FACTOR
136
+ for (const [k, ts] of [...windowClaimedAt.entries()]) {
137
+ if (ts < cutoff && !catchUps.has(k)) windowClaimedAt.delete(k)
138
+ }
139
+ }
140
+
141
+ function clearCatchUp(key: string): void {
142
+ const h = catchUps.get(key)
143
+ if (h !== undefined) {
144
+ cancel(h)
145
+ catchUps.delete(key)
146
+ }
147
+ }
148
+
149
+ /**
150
+ * A dropped tick MUST NOT become silence. Both loops re-arm a FIXED-cadence
151
+ * interval, so a tick eaten by the floor is simply lost: the next emission
152
+ * for that key could land as late as `floorMs + refreshMs` (7.5 s) after the
153
+ * last one — past Telegram's ~5 s chat-action expiry, and the chat goes dark
154
+ * mid-turn. That would trade a flood ban for a dead "typing…" indicator,
155
+ * which breaches `know-what-my-agent-is-doing` ("stays present from receipt
156
+ * to turn end").
157
+ *
158
+ * So a floor drop arms ONE coalesced catch-up at the exact moment the window
159
+ * opens (`claimedAt + floorMs`). Effect: the worst-case gap between two
160
+ * emissions is bounded by `floorMs`, while the emission CAP is untouched (the
161
+ * catch-up fires only once the floor has expired, so it can never emit early).
162
+ * A second drop inside the same window must not arm a second timer — hence
163
+ * the coalesce.
164
+ */
165
+ function armCatchUp(
166
+ key: string,
167
+ chatId: string,
168
+ threadId: number | null,
169
+ action: string,
170
+ delayMs: number,
171
+ ): void {
172
+ if (catchUps.has(key)) return // coalesced — one timer per key, always
173
+ const h = schedule(() => {
174
+ catchUps.delete(key)
175
+ // Re-enters `emit`, which re-checks the floor (now expired) and the flood
176
+ // gate. A send here cancels nothing — there is nothing left to cancel.
177
+ emit(chatId, threadId, action)
178
+ }, Math.max(0, delayMs))
179
+ catchUps.set(key, h)
180
+ }
181
+
182
+ function emit(chatId: string, threadId: number | null = null, action = 'typing'): boolean {
183
+ const key = deps.chatKey(chatId, threadId ?? null)
184
+ const t = now()
185
+ const claimed = windowClaimedAt.get(key)
186
+ // `t >= claimed` guards a backwards clock step (NTP): a stale future
187
+ // timestamp must not wedge the indicator off, so treat it as expired.
188
+ if (claimed != null && t >= claimed && t - claimed < floorMs) {
189
+ armCatchUp(key, chatId, threadId ?? null, action, claimed + floorMs - t)
190
+ deps.onDrop?.({ key, reason: 'floor' })
191
+ return false
192
+ }
193
+ // Claim the window BEFORE the flood check so a burst of restarts costs
194
+ // one flood-state read per floor, not one per restart.
195
+ windowClaimedAt.set(key, t)
196
+ prune(t)
197
+ if (isSuppressed()) {
198
+ // A flood window is open. Don't arm a catch-up — the goal is silence, not
199
+ // a deferred ping into the ban.
200
+ clearCatchUp(key)
201
+ deps.onDrop?.({ key, reason: 'flood' })
202
+ return false
203
+ }
204
+ // This send satisfies whatever a pending catch-up was going to deliver.
205
+ // Cancelling it here is what keeps a catch-up from re-arming itself into a
206
+ // self-sustaining heartbeat after the loops have stopped.
207
+ clearCatchUp(key)
208
+ deps.send(chatId, threadId ?? null, action)
209
+ return true
210
+ }
211
+
212
+ return {
213
+ emit,
214
+ cancelPending(chatId, threadId = null) {
215
+ clearCatchUp(deps.chatKey(chatId, threadId ?? null))
216
+ },
217
+ trackedKeys: () => windowClaimedAt.size,
218
+ pendingCatchUps: () => catchUps.size,
219
+ reset: () => {
220
+ for (const key of [...catchUps.keys()]) clearCatchUp(key)
221
+ windowClaimedAt.clear()
222
+ },
223
+ }
224
+ }
@@ -0,0 +1,136 @@
1
+ /**
2
+ * JTBD scenario — feel like a colleague, not a chatbot (DM).
3
+ *
4
+ * Serves: `reference/jobs/feel-like-a-colleague.md`. The colleague posture
5
+ * is shipped fleet-wide as lane-2 model-visible text (`~/.switchroom/fleet/
6
+ * CLAUDE.md`, seeded by `renderFleetDefaultsClaudeMd()` in
7
+ * `src/agents/fleet-defaults.ts`, epic #1850 / issue #1855). One bullet of
8
+ * "Good looks like" reads:
9
+ *
10
+ * "The agent asks at most one good clarifying question, and skips it when
11
+ * intent is clear, stating its assumption inline as it acts."
12
+ *
13
+ * This scenario sends a genuinely ambiguous request and asserts the agent
14
+ * asks EXACTLY ONE clarifying question rather than a question avalanche
15
+ * (the "Bad looks like" failure) or a confident guess that ignores the
16
+ * ambiguity.
17
+ *
18
+ * Gating: like every file under `telegram-plugin/uat/scenarios/`, this hits
19
+ * real Telegram + real Claude and only runs under `vitest.uat.config.ts`
20
+ * (`bun run test:uat`). The default vitest config excludes this directory,
21
+ * so CI stays green without live credentials.
22
+ */
23
+
24
+ import { describe, it, expect } from "vitest";
25
+ import { spinUp } from "../harness.js";
26
+
27
+ const AGENT = "test-harness";
28
+
29
+ // Ambiguous on purpose: "the report" has no referent, no format, no
30
+ // destination. A colleague asks which report / what for; it does not
31
+ // silently invent one, and it does not fire three questions.
32
+ const AMBIGUOUS_PROMPT = "can you send over the report when you get a sec?";
33
+
34
+ /**
35
+ * Count clarifying questions the agent itself asks. We count sentences
36
+ * ending in a question mark AFTER stripping spans where a `?` is not a
37
+ * question the agent is asking: URLs (query strings), inline/fenced code,
38
+ * and quoted echoes of the user's own words. Rhetorical framing
39
+ * ("sure, which one?") counts as one question; the invariant is at most one.
40
+ */
41
+ function countQuestions(text: string): number {
42
+ const stripped = text
43
+ // Fenced code blocks, then inline code spans.
44
+ .replace(/```[\s\S]*?```/g, " ")
45
+ .replace(/`[^`\n]*`/g, " ")
46
+ // URLs — `?` here is a query string, not a question.
47
+ .replace(/\bhttps?:\/\/\S+/gi, " ")
48
+ .replace(/\bwww\.\S+/gi, " ")
49
+ // Quoted spans — the agent echoing the user ("you said 'which report?'")
50
+ // is not the agent asking a question.
51
+ .replace(/"[^"\n]*"/g, " ")
52
+ // Single-quoted spans must OPEN at a word boundary so apostrophes in
53
+ // contractions ("don't", "user's") are not misread as quote delimiters.
54
+ .replace(/(^|[\s([{])'[^'\n]*'/g, "$1 ")
55
+ .replace(/[“”][^“”\n]*[“”]/g, " ")
56
+ // Markdown blockquote lines are quoted material, not the agent's voice.
57
+ .replace(/^\s*>.*$/gm, " ");
58
+ const matches = stripped.match(/[^.!?\n]*\?/g);
59
+ if (matches == null) return 0;
60
+ // Filter out trivial fragments (a lone "?" or whitespace) that are not
61
+ // real questions.
62
+ return matches.filter((m) => m.replace(/[^a-z0-9]/gi, "").length >= 3).length;
63
+ }
64
+
65
+ /**
66
+ * `true` for bot messages that are infrastructure cards rather than a
67
+ * conversational reply: the boot/greeting card (always delivered with the
68
+ * Telegram `silent` flag — see boot-card.ts "Boot cards are ALWAYS
69
+ * delivered silently"), edits of earlier messages, and anything matching
70
+ * the known boot-card header shape (`✅ <agent> back up · <version>`).
71
+ */
72
+ function isInfrastructureCard(m: {
73
+ text: string;
74
+ silent: boolean;
75
+ edited: boolean;
76
+ }): boolean {
77
+ if (m.silent || m.edited) return true;
78
+ return /back up ·|^✅ /u.test(m.text.trim());
79
+ }
80
+
81
+ describe("uat: feel like a colleague — one clarifying question when ambiguous", () => {
82
+ it(
83
+ "ambiguous ask → agent asks exactly one clarifying question",
84
+ async () => {
85
+ const sc = await spinUp({ agent: AGENT });
86
+ try {
87
+ await sc.sendDM(AMBIGUOUS_PROMPT);
88
+
89
+ // Substantive replies only: skip the boot/greeting card and any
90
+ // silent interim edits so the assertion runs against the agent's
91
+ // actual conversational answer to the prompt.
92
+ const reply = await sc.expectMessage(
93
+ (m) => /\S/.test(m.text) && !isInfrastructureCard(m),
94
+ {
95
+ from: "bot",
96
+ timeout: 90_000,
97
+ },
98
+ );
99
+
100
+ expect(reply.text.length).toBeGreaterThan(0);
101
+
102
+ const questionCount = countQuestions(reply.text);
103
+
104
+ // Invariant: at least one clarifying question (it did not guess
105
+ // blindly) and at most one (no question avalanche).
106
+ if (questionCount === 0) {
107
+ throw new Error(
108
+ `[colleague] ambiguous ask got no clarifying question — the ` +
109
+ `agent either guessed a referent or ignored the ambiguity. ` +
110
+ `Reply: ${JSON.stringify(reply.text.slice(0, 300))}`,
111
+ );
112
+ }
113
+ if (questionCount > 1) {
114
+ throw new Error(
115
+ `[colleague] question avalanche: ${questionCount} questions ` +
116
+ `where one would do. Reply: ${JSON.stringify(reply.text.slice(0, 300))}`,
117
+ );
118
+ }
119
+ expect(questionCount).toBe(1);
120
+
121
+ // Posture bonus: no sycophantic preamble. A soft forensic warn,
122
+ // not a hard fail (voice specifics are audited in
123
+ // fuzz-voice-scrub-dm).
124
+ if (/^(great question|i'?d be happy to|absolutely!|of course!)/i.test(reply.text.trim())) {
125
+ console.warn(
126
+ `[colleague] reply opens with sycophantic preamble: ` +
127
+ `${JSON.stringify(reply.text.slice(0, 120))}`,
128
+ );
129
+ }
130
+ } finally {
131
+ await sc.tearDown();
132
+ }
133
+ },
134
+ 120_000,
135
+ );
136
+ });
@@ -93,6 +93,27 @@ export type AgentMetadata = {
93
93
  * asked for current state, so terseness loses to completeness here.
94
94
  */
95
95
  live?: StatusProbeRow[];
96
+ /**
97
+ * Send-gate state (#3084 PR 3, part3-design §6). Present only when the gate
98
+ * feature flag is ON; omitted entirely when off so `/status` looks exactly
99
+ * as it did before the gate existed.
100
+ */
101
+ sendGate?: SendGateStatus;
102
+ };
103
+
104
+ /**
105
+ * `/status` view of the deterministic send gate (#3084). Queued / shed totals
106
+ * plus any open flood windows with their expiry. Only populated when the gate
107
+ * is enabled.
108
+ */
109
+ export type SendGateStatus = {
110
+ queued: number;
111
+ shed: number;
112
+ expired: number;
113
+ failedFast: number;
114
+ dropped: number;
115
+ /** Currently-open flood windows (expired already pruned). */
116
+ openWindows: { scopeKey: string; untilTs: number }[];
96
117
  };
97
118
 
98
119
  // Markdown escaper for dynamic values interpolated into bold/plain card
@@ -231,6 +252,27 @@ export function statusPairedText(params: {
231
252
  }
232
253
  }
233
254
 
255
+ // Send-gate block (#3084 PR 3) — only when the gate flag is on, so a fleet
256
+ // running with the gate OFF sees the identical pre-gate /status.
257
+ if (meta.sendGate) {
258
+ const sg = meta.sendGate;
259
+ lines.push("");
260
+ lines.push("**Send gate**");
261
+ lines.push(
262
+ `queued ${sg.queued} · shed ${sg.shed} · expired ${sg.expired} · ` +
263
+ `fail-fast ${sg.failedFast} · dropped ${sg.dropped}`,
264
+ );
265
+ if (sg.openWindows.length > 0) {
266
+ const now = Date.now();
267
+ for (const w of sg.openWindows) {
268
+ const secs = Math.max(0, Math.round((w.untilTs - now) / 1000));
269
+ lines.push(`⏳ flood window \`${escapeHtml(w.scopeKey)}\` — clears in ${secs}s`);
270
+ }
271
+ } else {
272
+ lines.push("no open flood windows");
273
+ }
274
+ }
275
+
234
276
  const audit = meta.audit;
235
277
  if (audit) {
236
278
  // Blank separator before the audit block so the reply reads as two
@@ -9,15 +9,22 @@ the queue no longer drains it but the operator can still inspect via
9
9
 
10
10
  Boundaries
11
11
  ----------
12
- * Per-entry HTTP timeout: ``HINDSIGHT_DRAIN_TIMEOUT`` (default 5s).
12
+ * Per-entry HTTP timeout: ``HINDSIGHT_DRAIN_TIMEOUT`` (default 5s), but
13
+ clamped per entry to the budget still remaining (see below) so a
14
+ single slow entry can never overshoot the wall-clock cap. The default
15
+ timeout (5s) intentionally exceeds the default budget (4s): the clamp,
16
+ not the raw timeout, is what bounds a slow entry.
17
+ * Total wall-clock cap: ``HINDSIGHT_DRAIN_BUDGET_S`` (default 4s) so
18
+ drain never blocks SessionStart longer than the upstream hook timeout
19
+ permits. This is the authoritative bound; the per-entry timeout is
20
+ clamped down to ``max(1, remaining budget)`` before each request, so
21
+ even one slow upstream entry overshoots the budget by at most the
22
+ clamp floor (~1s), not by ``HINDSIGHT_DRAIN_TIMEOUT - budget``.
13
23
  * Stall guard: if ``STALL_THRESHOLD`` (3) consecutive entries fail with
14
24
  the same error class, we stop draining for this session — that's a
15
25
  systemic outage, not a transient flake, and continuing would only
16
26
  burn the SessionStart timeout budget. The remaining entries stay
17
27
  queued for the next session.
18
- * Total wall-clock cap: ``HINDSIGHT_DRAIN_BUDGET_S`` (default 4s) so
19
- drain never blocks SessionStart longer than the upstream
20
- hook timeout permits.
21
28
 
22
29
  Standalone usage::
23
30
 
@@ -113,13 +120,22 @@ def drain(config: dict | None = None) -> dict:
113
120
  last_error_class: str | None = None
114
121
 
115
122
  for path, entry in entries:
116
- if time.monotonic() - started > budget:
123
+ elapsed = time.monotonic() - started
124
+ if elapsed > budget:
117
125
  summary["budget_exceeded"] = True
118
126
  debug_log(config, "drain_pending: total budget exceeded, stopping")
119
127
  break
120
128
 
129
+ # Clamp the per-entry HTTP timeout to the budget still remaining
130
+ # (#1094 item 2). Without this, a single slow entry using the full
131
+ # HINDSIGHT_DRAIN_TIMEOUT (default 5s) overshoots the total budget
132
+ # (default 4s). Floor at 1s so we still give a near-exhausted
133
+ # budget one bounded shot rather than a 0s (instant-fail) request.
134
+ remaining = budget - elapsed
135
+ effective_timeout = max(1, min(timeout, int(remaining) if remaining >= 1 else 1))
136
+
121
137
  try:
122
- _retry_one(entry, timeout=timeout)
138
+ _retry_one(entry, timeout=effective_timeout)
123
139
  except Exception as e:
124
140
  err_class = type(e).__name__
125
141
  if err_class == last_error_class:
@@ -93,15 +93,22 @@ class HindsightClient:
93
93
  pass
94
94
  raise RuntimeError(f"HTTP {e.code} from {url}: {body_text}") from e
95
95
 
96
- def health_check(self, timeout: int = 5) -> bool:
96
+ def health_check(self, timeout: int = 5, retries: int = HEALTH_CHECK_RETRIES) -> bool:
97
97
  """Check if the Hindsight server is reachable.
98
98
 
99
- Mirrors Openclaw's checkExternalApiHealth: retries up to 3 times
100
- with 2s delay between attempts.
99
+ Mirrors Openclaw's checkExternalApiHealth: retries up to
100
+ ``retries`` times (default ``HEALTH_CHECK_RETRIES`` = 3) with
101
+ ``HEALTH_CHECK_DELAY`` (2s) between attempts.
102
+
103
+ Time-budgeted callers (e.g. the SessionStart drain gate,
104
+ #1094) should pass ``retries=1``: against a HUNG server the
105
+ default loop costs ~retries*timeout + (retries-1)*delay of wall
106
+ clock, which blows a 5s hook budget.
101
107
  """
102
108
  import time
103
109
 
104
- for attempt in range(1, HEALTH_CHECK_RETRIES + 1):
110
+ retries = max(1, retries)
111
+ for attempt in range(1, retries + 1):
105
112
  try:
106
113
  url = f"{self.api_url}/health"
107
114
  req = urllib.request.Request(url, headers=self._headers(), method="GET")
@@ -110,7 +117,7 @@ class HindsightClient:
110
117
  return True
111
118
  except Exception:
112
119
  pass
113
- if attempt < HEALTH_CHECK_RETRIES:
120
+ if attempt < retries:
114
121
  time.sleep(HEALTH_CHECK_DELAY)
115
122
  return False
116
123
 
@@ -183,6 +183,27 @@ def parse_active_directives_block(text: str) -> list:
183
183
  return contents
184
184
 
185
185
 
186
+ # Length-aware guard for terse rules (#2912). Forward coverage alone
187
+ # (|rule ∩ directive| / |rule|) is easy to satisfy when the rule has only a
188
+ # couple of significant tokens: any long directive that happens to contain
189
+ # those few words scores ~1.0 and silently swallows a genuinely-new short
190
+ # rule, skipping a legitimate re-prompt. For such short rules we additionally
191
+ # require REVERSE coverage — the matched directive's significant-token set must
192
+ # also be substantially covered by the rule's — which is a Jaccard-style
193
+ # bidirectional check. That fails precisely the "few tokens diluted inside a
194
+ # long unrelated directive" case while still passing a terse rule that restates
195
+ # a comparably terse directive.
196
+ #
197
+ # Constants chosen against the real tokenizer output:
198
+ # - < 4 significant tokens is "short" (1-3 tokens: the regime where a single
199
+ # incidental word swings forward coverage past 0.6).
200
+ # - reverse coverage >= 0.5 keeps near-equal-size restatements deduped
201
+ # (e.g. a 2-token rule vs a 4-token directive → 2/4 = 0.5) but rejects a
202
+ # 2-token rule diluted inside a 6+-token directive (2/6 ≈ 0.33 < 0.5).
203
+ _SHORT_RULE_TOKEN_LIMIT = 4
204
+ _SHORT_RULE_REVERSE_COVERAGE = 0.5
205
+
206
+
186
207
  def rule_already_captured(
187
208
  rule_text: str, directive_contents: list, threshold: float = 0.6
188
209
  ) -> bool:
@@ -193,15 +214,29 @@ def rule_already_captured(
193
214
  best-matching directive. A high coverage ratio means the restated rule adds
194
215
  (almost) no new significant words over one already stored — i.e. a
195
216
  duplicate. Deterministic; no model/API call.
217
+
218
+ For terse rules (fewer than ``_SHORT_RULE_TOKEN_LIMIT`` significant tokens)
219
+ forward coverage is not sufficient — see the module comment above — so a
220
+ reverse-coverage guard is also required before declaring a match.
196
221
  """
197
222
  rule_tokens = _dedup_tokens(rule_text)
198
223
  if not rule_tokens:
199
224
  return False
225
+ is_short_rule = len(rule_tokens) < _SHORT_RULE_TOKEN_LIMIT
200
226
  for content in directive_contents:
201
227
  d_tokens = _dedup_tokens(content)
202
228
  if not d_tokens:
203
229
  continue
204
- covered = len(rule_tokens & d_tokens) / len(rule_tokens)
205
- if covered >= threshold:
206
- return True
230
+ intersection = len(rule_tokens & d_tokens)
231
+ covered = intersection / len(rule_tokens)
232
+ if covered < threshold:
233
+ continue
234
+ if is_short_rule:
235
+ # Bidirectional guard: the directive must not be much larger than
236
+ # the rule, or those few shared tokens are incidental overlap
237
+ # rather than a true restatement.
238
+ reverse_covered = intersection / len(d_tokens)
239
+ if reverse_covered < _SHORT_RULE_REVERSE_COVERAGE:
240
+ continue
241
+ return True
207
242
  return False
@@ -197,22 +197,49 @@ def update_attempt(path: str, entry: dict, error: BaseException) -> bool:
197
197
 
198
198
  def mark_dead(path: str, entry: dict) -> Optional[str]:
199
199
  """Convert an entry that exceeded ``MAX_ATTEMPTS`` into a permanent
200
- failure marker. Renames ``<path>`` to ``<path>.dead`` so the queue
201
- no longer drains it but operators can still inspect.
202
-
203
- Returns the marker path, or ``None`` if the rename failed.
200
+ failure marker at ``<path>.dead`` so the queue no longer drains it
201
+ but operators can still inspect.
202
+
203
+ Returns the marker path, or ``None`` if it failed.
204
+
205
+ Crash-window invariant (#1094 item 3): **a live ``<path>.json`` entry
206
+ must never carry a ``dead_at`` stamp.** The old two-step form violated
207
+ this — it wrote the dead_at-stamped payload back to the *live* path
208
+ (rename tmp -> path) and only then renamed path -> path.dead, so a
209
+ crash between the two renames left a live entry with ``dead_at`` set
210
+ that the drainer would re-enter and re-bump. Here we instead:
211
+
212
+ 1. write the dead_at-stamped payload to ``<path>.tmp``
213
+ 2. ``os.replace(tmp, dead_path)`` — the .dead marker appears in one
214
+ atomic step (never drained: the drainer only lists ``*.json``)
215
+ 3. ``os.unlink(path)`` — drop the original live entry
216
+
217
+ At every crash point the invariant holds: the ``dead_at`` stamp only
218
+ ever lands on ``<path>.dead``. A crash after step 2 leaves both the
219
+ (stale, no-dead_at) live entry and the .dead marker; the next drain
220
+ re-marks it dead (os.replace overwrites the marker idempotently),
221
+ never observing a live entry with dead_at.
204
222
  """
205
223
  entry["dead_at"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
206
224
  dead_path = path + ".dead"
225
+ tmp = path + ".tmp"
207
226
  try:
208
- # Best-effort: write the final state first so the marker shows
209
- # the death timestamp + last error.
210
- tmp = path + ".tmp"
211
227
  with open(tmp, "w", encoding="utf-8") as f:
212
228
  json.dump(entry, f, ensure_ascii=False)
213
229
  os.chmod(tmp, 0o600)
214
- os.rename(tmp, path)
215
- os.rename(path, dead_path)
230
+ os.replace(tmp, dead_path)
231
+ # Marker is durable now; removing the original never resurrects a
232
+ # dead_at-stamped live entry. Best-effort — a leftover live entry
233
+ # is self-healing (re-marked dead on the next pass).
234
+ try:
235
+ os.unlink(path)
236
+ except OSError:
237
+ pass
216
238
  return dead_path
217
239
  except OSError:
240
+ # Clean up a possibly-orphaned tmp so it doesn't linger.
241
+ try:
242
+ os.unlink(tmp)
243
+ except OSError:
244
+ pass
218
245
  return None
@@ -40,7 +40,16 @@ from lib.pending import MAX_ENTRIES, count as pending_count, enqueue as pending_
40
40
  # Exit codes:
41
41
  # 0 — success (or retain skipped for benign reasons)
42
42
  # 1 — retain failed AND was queued to pending-retains (recoverable)
43
- # 2 — retain failed AND the queue rejected it (chronic backlog)
43
+ # 2 — retain failed but nothing was queued — either the queue rejected
44
+ # it (chronic backlog) or there was no payload to queue (the
45
+ # failure happened before one was built). Not recoverable via the
46
+ # pending-retains drain, so it's semantically "dropped", not
47
+ # "queued".
48
+ # Only the sign of the exit code is load-bearing downstream: Claude
49
+ # Code's hook runner routes any non-zero SessionEnd exit to the issue
50
+ # sink (bin/run-hook.sh / #424); nothing distinguishes 1 from 2
51
+ # programmatically, so this is a correctness/clarity fix, not a
52
+ # contract change.
44
53
  EXIT_OK = 0
45
54
  EXIT_QUEUED = 1
46
55
  EXIT_DROPPED = 2
@@ -98,8 +107,10 @@ def main() -> int:
98
107
  exit_code = EXIT_QUEUED
99
108
  else:
100
109
  # No payload to queue — the failure happened before we
101
- # finished building one (e.g. URL resolution).
102
- exit_code = EXIT_QUEUED
110
+ # finished building one (e.g. URL resolution). Nothing
111
+ # landed in pending-retains, so this is a drop, not a
112
+ # queue (#1094 item 5): EXIT_DROPPED, not EXIT_QUEUED.
113
+ exit_code = EXIT_DROPPED
103
114
 
104
115
  # Stop daemon if we started it. Always runs, even on retain failure,
105
116
  # so we don't leak a daemon process.