switchroom 0.18.15 → 0.18.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +16 -0
- package/dist/auth-broker/index.js +445 -10
- package/dist/cli/notion-write-pretool.mjs +16 -0
- package/dist/cli/switchroom.js +654 -479
- package/dist/host-control/main.js +20 -1
- package/dist/vault/approvals/kernel-server.js +16 -0
- package/dist/vault/broker/server.js +16 -0
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +81 -139
- package/telegram-plugin/bridge/bridge.ts +7 -1
- package/telegram-plugin/dist/bridge/bridge.js +26 -1
- package/telegram-plugin/dist/gateway/gateway.js +1758 -661
- package/telegram-plugin/dist/server.js +26 -1
- package/telegram-plugin/draft-stream.ts +78 -3
- package/telegram-plugin/fleet-fallback-resume.ts +26 -3
- package/telegram-plugin/gateway/approval-hold.ts +49 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
- package/telegram-plugin/gateway/effort-command.ts +9 -7
- package/telegram-plugin/gateway/gateway.ts +627 -291
- package/telegram-plugin/gateway/linear-activity.ts +20 -4
- package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
- package/telegram-plugin/gateway/model-command.ts +96 -18
- package/telegram-plugin/gateway/pending-session-command.ts +10 -8
- package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
- package/telegram-plugin/gateway/session-model-file.ts +141 -172
- package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
- package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
- package/telegram-plugin/litellm-local-notice.ts +189 -0
- package/telegram-plugin/llm-error-present.ts +436 -0
- package/telegram-plugin/operator-events.ts +7 -1
- package/telegram-plugin/permission-title.ts +172 -10
- package/telegram-plugin/premium-recovery.ts +101 -0
- package/telegram-plugin/quota-watch.ts +16 -4
- package/telegram-plugin/raw-error-scrub.ts +73 -0
- package/telegram-plugin/retry-api-call.ts +8 -2
- package/telegram-plugin/runtime-metrics.ts +16 -0
- package/telegram-plugin/send-gate-degraded.test.ts +161 -8
- package/telegram-plugin/send-gate-observability.test.ts +140 -0
- package/telegram-plugin/send-gate-observability.ts +65 -20
- package/telegram-plugin/send-gate.test.ts +143 -1
- package/telegram-plugin/send-gate.ts +246 -23
- package/telegram-plugin/session-tail.ts +16 -0
- package/telegram-plugin/shared/local-time.ts +69 -0
- package/telegram-plugin/stream-controller.ts +143 -20
- package/telegram-plugin/stream-reply-handler.ts +12 -2
- package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
- package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
- package/telegram-plugin/tests/bot-api.harness.ts +7 -2
- package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
- package/telegram-plugin/tests/draft-stream.test.ts +110 -1
- package/telegram-plugin/tests/effort-command.test.ts +4 -4
- package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
- package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
- package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
- package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
- package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
- package/telegram-plugin/tests/model-command.test.ts +84 -1
- package/telegram-plugin/tests/permission-title.test.ts +167 -4
- package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
- package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
- package/telegram-plugin/tests/quota-watch.test.ts +21 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
- package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
- package/telegram-plugin/tests/session-model-file.test.ts +7 -155
- package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
- package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
- package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
- package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
- package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
- package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
- package/telegram-plugin/tier-downgrade.ts +198 -0
- package/telegram-plugin/tool-activity-summary.ts +99 -0
- package/telegram-plugin/worker-activity-feed.ts +543 -368
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Integration tests: `createStreamController` driven through a REAL
|
|
3
|
+
* `createSendGate` (#3110).
|
|
4
|
+
*
|
|
5
|
+
* Production wires stream-controller's `retry` to the gateway's
|
|
6
|
+
* `robustApiCall`, which is the send gate wrapped over the retry policy —
|
|
7
|
+
* so what the gate does to the opts a call site passes IS the production
|
|
8
|
+
* behavior. Before #3110 the controller passed only `{ threadId, chat_id }`,
|
|
9
|
+
* the gate saw ordinary sends, and the draft stream's own 400 ms DM throttle
|
|
10
|
+
* drove same-message editMessageText straight past the >=1.5s per-message
|
|
11
|
+
* edit floor — the #1 documented flood-ban trigger (part3-design §4;
|
|
12
|
+
* production ~6h ban 2026-07-12).
|
|
13
|
+
*
|
|
14
|
+
* These tests pin the wiring end to end:
|
|
15
|
+
* - draft edits inside the floor never reach the API; intermediates
|
|
16
|
+
* collapse; the latest snapshot lands floor-paced
|
|
17
|
+
* - the floor is per-message (stream A does not delay stream B)
|
|
18
|
+
* - the gate's no-op skip drops a repeat payload for the same message
|
|
19
|
+
* - an open flood window sheds draft edits with ZERO API calls, and the
|
|
20
|
+
* stream recovers with full state after the window closes
|
|
21
|
+
* - a shed draft is NOT recorded as delivered — a later flush of the
|
|
22
|
+
* SAME text (the completed answer) still lands
|
|
23
|
+
* - the finalize flush is `critical`: never shed; waits out a short
|
|
24
|
+
* window; fails fast (structured, logged) on a long one
|
|
25
|
+
* - regression pin: the controller passes messageId / editPayload /
|
|
26
|
+
* priorityClass through the retry policy on every edit
|
|
27
|
+
*
|
|
28
|
+
* Two clocks, deliberately separate: vitest fake timers drive draft-stream's
|
|
29
|
+
* internal `setTimeout` throttle (only `setTimeout`/`clearTimeout` are faked
|
|
30
|
+
* so the gate's microtask pump below stays real); the gate runs on the same
|
|
31
|
+
* injectable fake `Clock` used by send-gate.test.ts.
|
|
32
|
+
*/
|
|
33
|
+
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'
|
|
34
|
+
import { createStreamController, type RetryPolicy } from '../stream-controller.js'
|
|
35
|
+
import { createSendGate, type Clock, type SendGateConfig } from '../send-gate.js'
|
|
36
|
+
import { isFloodWaitActiveError } from '../retry-api-call.js'
|
|
37
|
+
import { renderOutboundChunks } from '../render/rich-render.js'
|
|
38
|
+
import { createMockBot, installBotResetHook } from './bot-api.harness.js'
|
|
39
|
+
|
|
40
|
+
/** Deterministic fake clock — same contract as send-gate.test.ts's. */
|
|
41
|
+
class FakeClock implements Clock {
|
|
42
|
+
private cur = 0
|
|
43
|
+
private seq = 0
|
|
44
|
+
private timers: { at: number; id: number; resolve: () => void }[] = []
|
|
45
|
+
|
|
46
|
+
now(): number {
|
|
47
|
+
return this.cur
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
sleep(ms: number): Promise<void> {
|
|
51
|
+
return new Promise<void>((resolve) => {
|
|
52
|
+
this.timers.push({ at: this.cur + ms, id: this.seq++, resolve })
|
|
53
|
+
})
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
async advance(ms: number): Promise<void> {
|
|
57
|
+
const target = this.cur + ms
|
|
58
|
+
for (;;) {
|
|
59
|
+
const due = this.timers
|
|
60
|
+
.filter((t) => t.at <= target)
|
|
61
|
+
.sort((a, b) => a.at - b.at || a.id - b.id)
|
|
62
|
+
if (due.length === 0) break
|
|
63
|
+
const t = due[0]
|
|
64
|
+
this.timers = this.timers.filter((x) => x !== t)
|
|
65
|
+
this.cur = t.at
|
|
66
|
+
t.resolve()
|
|
67
|
+
await flush()
|
|
68
|
+
}
|
|
69
|
+
this.cur = target
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Flush pending microtasks via a REAL macrotask hop (setImmediate is not faked). */
|
|
74
|
+
function flush(): Promise<void> {
|
|
75
|
+
return new Promise<void>((r) => setImmediate(r))
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** A real send gate on a fake clock, exposed as a stream-controller RetryPolicy. */
|
|
79
|
+
function makeGatedRetry(cfg: Partial<SendGateConfig> = {}) {
|
|
80
|
+
const clock = new FakeClock()
|
|
81
|
+
const gate = createSendGate({ enabled: true, clock, jitter: () => 0, ...cfg })
|
|
82
|
+
const retry: RetryPolicy = (fn, opts) => gate.gate(fn, opts)
|
|
83
|
+
return { clock, gate, retry }
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
describe('stream-controller × send gate (#3110)', () => {
|
|
87
|
+
const bot = createMockBot()
|
|
88
|
+
installBotResetHook(bot)
|
|
89
|
+
|
|
90
|
+
// Fake ONLY draft-stream's throttle timers. The gate's fake clock is
|
|
91
|
+
// driven explicitly; its microtask pump (setImmediate) must stay real.
|
|
92
|
+
beforeEach(() => vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] }))
|
|
93
|
+
afterEach(() => vi.useRealTimers())
|
|
94
|
+
|
|
95
|
+
/** Release draft-stream's local throttle, then settle microtasks. */
|
|
96
|
+
async function tick(throttleMs = 250): Promise<void> {
|
|
97
|
+
vi.advanceTimersByTime(throttleMs)
|
|
98
|
+
await flush()
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const editBodies = () =>
|
|
102
|
+
bot.api.editMessageText.mock.calls.map(([, , text]) =>
|
|
103
|
+
typeof text === 'object' && text != null ? text.markdown : text,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
it('draft edits inside the floor never hit the API; intermediates collapse; latest lands floor-paced', async () => {
|
|
107
|
+
const { clock, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
108
|
+
// Record the FAKE-clock time of every editMessageText that reaches the API.
|
|
109
|
+
const editTimes: number[] = []
|
|
110
|
+
bot.api.editMessageText.mockImplementation(async () => {
|
|
111
|
+
editTimes.push(clock.now())
|
|
112
|
+
return true as const
|
|
113
|
+
})
|
|
114
|
+
const stream = createStreamController({ bot, chatId: '1', throttleMs: 250, retry })
|
|
115
|
+
|
|
116
|
+
// Anchor send, then the first edit — no prior edit on the message, so the
|
|
117
|
+
// gate admits it immediately.
|
|
118
|
+
void stream.update('draft 1')
|
|
119
|
+
await flush()
|
|
120
|
+
expect(bot.api.sendRichMessage).toHaveBeenCalledTimes(1)
|
|
121
|
+
void stream.update('draft 2')
|
|
122
|
+
await tick()
|
|
123
|
+
expect(editBodies()).toEqual(['draft 2'])
|
|
124
|
+
|
|
125
|
+
// Three rapid snapshots inside the 1.5s floor: d3 reaches the gate (edit
|
|
126
|
+
// floor holds it); d4 and d5 arrive while that edit is in flight, so
|
|
127
|
+
// draft-stream's own last-write-wins collapses d4 — it must NEVER reach
|
|
128
|
+
// the API.
|
|
129
|
+
void stream.update('draft 3')
|
|
130
|
+
await tick()
|
|
131
|
+
void stream.update('draft 4')
|
|
132
|
+
void stream.update('draft 5')
|
|
133
|
+
await flush()
|
|
134
|
+
// Still inside the floor: nothing new hit the API.
|
|
135
|
+
expect(editBodies()).toEqual(['draft 2'])
|
|
136
|
+
|
|
137
|
+
await clock.advance(1500) // floor clears → d3 lands; d5 flushes into the gate
|
|
138
|
+
await clock.advance(1500) // next floor clears → d5 lands
|
|
139
|
+
expect(editBodies()).toEqual(['draft 2', 'draft 3', 'draft 5'])
|
|
140
|
+
// Wire spacing is the gate floor, not the 250 ms local throttle.
|
|
141
|
+
expect(editTimes).toEqual([0, 1500, 3000])
|
|
142
|
+
})
|
|
143
|
+
|
|
144
|
+
it('the edit floor is per-message: stream B is not delayed by stream A holding its floor', async () => {
|
|
145
|
+
const { clock, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
146
|
+
const editTimes = new Map<number, number[]>()
|
|
147
|
+
bot.api.editMessageText.mockImplementation(async (_chat, id) => {
|
|
148
|
+
const arr = editTimes.get(id) ?? []
|
|
149
|
+
arr.push(clock.now())
|
|
150
|
+
editTimes.set(id, arr)
|
|
151
|
+
return true as const
|
|
152
|
+
})
|
|
153
|
+
// Two chats → independent per-chat buckets; ONE shared gate (as in the
|
|
154
|
+
// gateway, where every surface transits the same robustApiCall).
|
|
155
|
+
const a = createStreamController({ bot, chatId: '100', throttleMs: 250, retry })
|
|
156
|
+
const b = createStreamController({ bot, chatId: '200', throttleMs: 250, retry })
|
|
157
|
+
|
|
158
|
+
void a.update('a1')
|
|
159
|
+
void b.update('b1')
|
|
160
|
+
await flush()
|
|
161
|
+
void a.update('a2')
|
|
162
|
+
void b.update('b2')
|
|
163
|
+
await tick()
|
|
164
|
+
// First edit per message admits immediately on each message's own floor.
|
|
165
|
+
const idA = a.getMessageId() as number
|
|
166
|
+
const idB = b.getMessageId() as number
|
|
167
|
+
expect(editTimes.get(idA)).toEqual([0])
|
|
168
|
+
expect(editTimes.get(idB)).toEqual([0])
|
|
169
|
+
|
|
170
|
+
// Refill the global bucket a little so the cosmetic shed check (token
|
|
171
|
+
// pressure) doesn't fire — this test isolates the FLOOR.
|
|
172
|
+
await clock.advance(400)
|
|
173
|
+
void a.update('a3')
|
|
174
|
+
void b.update('b3')
|
|
175
|
+
await tick()
|
|
176
|
+
expect(editTimes.get(idA)).toEqual([0]) // held by A's floor…
|
|
177
|
+
expect(editTimes.get(idB)).toEqual([0]) // …and B by B's own, not A's.
|
|
178
|
+
|
|
179
|
+
await clock.advance(1100) // 400 + 1100 = 1500 → BOTH floors clear together
|
|
180
|
+
expect(editTimes.get(idA)).toEqual([0, 1500])
|
|
181
|
+
expect(editTimes.get(idB)).toEqual([0, 1500])
|
|
182
|
+
})
|
|
183
|
+
|
|
184
|
+
it('gate no-op skip: a repeated payload is dropped BENIGNLY — no API call, treated as delivered (F4)', async () => {
|
|
185
|
+
const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
186
|
+
// Two controllers re-attached to the SAME anchor message (the #626
|
|
187
|
+
// initialMessageId path — e.g. a stream re-created after a restart).
|
|
188
|
+
const first = createStreamController({
|
|
189
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
|
|
190
|
+
})
|
|
191
|
+
void first.update('**status: done**')
|
|
192
|
+
await flush()
|
|
193
|
+
expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
|
|
194
|
+
|
|
195
|
+
await clock.advance(1500) // clear the floor — isolate the no-op skip
|
|
196
|
+
const onEdit = vi.fn()
|
|
197
|
+
const logs: string[] = []
|
|
198
|
+
const second = createStreamController({
|
|
199
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777, onEdit,
|
|
200
|
+
log: (m) => logs.push(m),
|
|
201
|
+
})
|
|
202
|
+
void second.update('**status: done**')
|
|
203
|
+
await flush()
|
|
204
|
+
|
|
205
|
+
// Identical rendered payload for the same chat+message → dropped by the
|
|
206
|
+
// gate before the API…
|
|
207
|
+
expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
|
|
208
|
+
expect(gate.stats().global.dropped).toBe(1)
|
|
209
|
+
// …and the drop is BENIGN (the payload IS on screen): the controller
|
|
210
|
+
// reports it as delivered — onEdit fires, no shed marker, no failure log
|
|
211
|
+
// (F4: pre-sentinel this test passed via the wrong mechanism — the
|
|
212
|
+
// drop's `undefined` was misread as a shed and onEdit stayed silent).
|
|
213
|
+
expect(onEdit).toHaveBeenCalledTimes(1)
|
|
214
|
+
expect(logs.filter((m) => m.includes('shed') || m.includes('edit failed'))).toEqual([])
|
|
215
|
+
|
|
216
|
+
// Because the drop was recorded as delivered, a THIRD identical update
|
|
217
|
+
// dedupes inside draft-stream itself — it never even reaches the gate.
|
|
218
|
+
void second.update('**status: done**')
|
|
219
|
+
await tick() // release the local throttle so the dedupe check runs
|
|
220
|
+
expect(gate.stats().global.dropped).toBe(1) // unchanged — gate never consulted
|
|
221
|
+
expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
|
|
222
|
+
})
|
|
223
|
+
|
|
224
|
+
it('open flood window: draft edits shed with ZERO API calls; full state lands after it closes', async () => {
|
|
225
|
+
const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
226
|
+
const stream = createStreamController({
|
|
227
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
|
|
228
|
+
})
|
|
229
|
+
|
|
230
|
+
gate.openFloodWindow('global', clock.now() + 30_000)
|
|
231
|
+
|
|
232
|
+
void stream.update('draft a')
|
|
233
|
+
await flush()
|
|
234
|
+
void stream.update('draft b')
|
|
235
|
+
await tick()
|
|
236
|
+
// Both drafts shed as cosmetic — nothing reached the API.
|
|
237
|
+
expect(bot.api.editMessageText).not.toHaveBeenCalled()
|
|
238
|
+
expect(gate.stats().global.shed).toBe(2)
|
|
239
|
+
|
|
240
|
+
await clock.advance(30_000) // window closes
|
|
241
|
+
void stream.update('draft c — full state')
|
|
242
|
+
await tick()
|
|
243
|
+
expect(editBodies()).toEqual(['draft c — full state'])
|
|
244
|
+
})
|
|
245
|
+
|
|
246
|
+
it('a shed draft is NOT recorded as delivered: a later finalize of the SAME text still lands', async () => {
|
|
247
|
+
const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
248
|
+
const stream = createStreamController({
|
|
249
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
|
|
250
|
+
})
|
|
251
|
+
|
|
252
|
+
gate.openFloodWindow('global', clock.now() + 30_000)
|
|
253
|
+
void stream.update('the completed answer')
|
|
254
|
+
await flush()
|
|
255
|
+
expect(bot.api.editMessageText).not.toHaveBeenCalled()
|
|
256
|
+
expect(gate.stats().global.shed).toBe(1)
|
|
257
|
+
|
|
258
|
+
await clock.advance(30_000)
|
|
259
|
+
// stream_reply done=true with the same text → finalize(text). If the shed
|
|
260
|
+
// draft had been recorded as on-screen, draft-stream's dedupe would skip
|
|
261
|
+
// this flush and the completed answer would never render.
|
|
262
|
+
await stream.finalize('the completed answer')
|
|
263
|
+
expect(editBodies()).toEqual(['the completed answer'])
|
|
264
|
+
})
|
|
265
|
+
|
|
266
|
+
it('finalize is critical: not shed by an open short window — waits it out, then lands', async () => {
|
|
267
|
+
const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
268
|
+
const editTimes: number[] = []
|
|
269
|
+
bot.api.editMessageText.mockImplementation(async () => {
|
|
270
|
+
editTimes.push(clock.now())
|
|
271
|
+
return true as const
|
|
272
|
+
})
|
|
273
|
+
const stream = createStreamController({
|
|
274
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
|
|
275
|
+
})
|
|
276
|
+
|
|
277
|
+
gate.openFloodWindow('global', clock.now() + 30_000) // short: <= 60s fail-fast ceiling
|
|
278
|
+
void stream.update('draft while banned')
|
|
279
|
+
await flush()
|
|
280
|
+
expect(bot.api.editMessageText).not.toHaveBeenCalled() // draft shed
|
|
281
|
+
|
|
282
|
+
const fin = stream.finalize('the answer')
|
|
283
|
+
await flush()
|
|
284
|
+
// Critical: NOT shed — queued against the window, still zero API calls.
|
|
285
|
+
expect(bot.api.editMessageText).not.toHaveBeenCalled()
|
|
286
|
+
|
|
287
|
+
await clock.advance(30_000)
|
|
288
|
+
await fin
|
|
289
|
+
expect(editBodies()).toEqual(['the answer'])
|
|
290
|
+
expect(editTimes).toEqual([30_000])
|
|
291
|
+
expect(gate.stats().global.shed).toBe(1) // only the draft
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
it('finalize under a LONG window fails fast (structured FLOOD_WAIT_ACTIVE, logged) — no API call, no hang', async () => {
|
|
295
|
+
const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
|
|
296
|
+
const logs: string[] = []
|
|
297
|
+
const stream = createStreamController({
|
|
298
|
+
bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
|
|
299
|
+
log: (m) => logs.push(m),
|
|
300
|
+
})
|
|
301
|
+
|
|
302
|
+
gate.openFloodWindow('global', clock.now() + 21_397_000) // the 2026-07-12 ban: ~5.9h
|
|
303
|
+
const fin = stream.finalize('the answer')
|
|
304
|
+
await flush()
|
|
305
|
+
await fin // resolves — draft-stream surfaces the failure via log, not a hang
|
|
306
|
+
|
|
307
|
+
expect(bot.api.editMessageText).not.toHaveBeenCalled()
|
|
308
|
+
expect(gate.stats().global.failedFast).toBe(1)
|
|
309
|
+
expect(logs.some((m) => m.includes('FLOOD_WAIT_ACTIVE'))).toBe(true)
|
|
310
|
+
// Sanity: the gate's structured error shape is what the reply path keys on.
|
|
311
|
+
try {
|
|
312
|
+
await retry(async () => true, { chat_id: '1', priorityClass: 'critical' })
|
|
313
|
+
expect.unreachable('critical send during a long window must fail fast')
|
|
314
|
+
} catch (err) {
|
|
315
|
+
expect(isFloodWaitActiveError(err)).toBe(true)
|
|
316
|
+
}
|
|
317
|
+
})
|
|
318
|
+
|
|
319
|
+
it('REGRESSION PIN: every edit passes messageId / editPayload / priorityClass to the retry policy', async () => {
|
|
320
|
+
// Spy retry with NO gate — pins exactly what the controller hands to
|
|
321
|
+
// robustApiCall (the #3110 bypass was these fields being absent).
|
|
322
|
+
const seen: Array<Record<string, unknown> | undefined> = []
|
|
323
|
+
const retry: RetryPolicy = (fn, opts) => {
|
|
324
|
+
seen.push(opts as Record<string, unknown> | undefined)
|
|
325
|
+
return fn()
|
|
326
|
+
}
|
|
327
|
+
const stream = createStreamController({ bot, chatId: '9', threadId: 3, throttleMs: 250, retry })
|
|
328
|
+
|
|
329
|
+
void stream.update('draft')
|
|
330
|
+
await flush()
|
|
331
|
+
// The anchor SEND stays untagged: the gate admits untagged non-edit
|
|
332
|
+
// sends as critical, and a shed send would break message-id capture.
|
|
333
|
+
expect(seen[0]).toEqual({ threadId: 3, chat_id: '9' })
|
|
334
|
+
|
|
335
|
+
void stream.update('draft v2')
|
|
336
|
+
await tick()
|
|
337
|
+
const id = stream.getMessageId() as number
|
|
338
|
+
expect(seen[1]).toEqual({
|
|
339
|
+
threadId: 3,
|
|
340
|
+
chat_id: '9',
|
|
341
|
+
messageId: id,
|
|
342
|
+
editPayload: { markdown: 'draft v2' },
|
|
343
|
+
priorityClass: 'cosmetic',
|
|
344
|
+
})
|
|
345
|
+
|
|
346
|
+
await stream.finalize('the final answer')
|
|
347
|
+
expect(seen[2]).toEqual({
|
|
348
|
+
threadId: 3,
|
|
349
|
+
chat_id: '9',
|
|
350
|
+
messageId: id,
|
|
351
|
+
editPayload: { markdown: 'the final answer' },
|
|
352
|
+
priorityClass: 'critical',
|
|
353
|
+
})
|
|
354
|
+
})
|
|
355
|
+
|
|
356
|
+
it('REGRESSION PIN: the plain parse-entities fallback edit carries the gate fields too', async () => {
|
|
357
|
+
const seen: Array<Record<string, unknown> | undefined> = []
|
|
358
|
+
const retry: RetryPolicy = (fn, opts) => {
|
|
359
|
+
seen.push(opts as Record<string, unknown> | undefined)
|
|
360
|
+
return fn()
|
|
361
|
+
}
|
|
362
|
+
const stream = createStreamController({ bot, chatId: '9', throttleMs: 250, retry })
|
|
363
|
+
void stream.update('draft')
|
|
364
|
+
await flush()
|
|
365
|
+
const id = stream.getMessageId() as number
|
|
366
|
+
|
|
367
|
+
// Rich edit 400s with a parse-entities rejection → same-id plain fallback.
|
|
368
|
+
const { GrammyError } = await import('grammy')
|
|
369
|
+
bot.api.editMessageText.mockRejectedValueOnce(
|
|
370
|
+
new GrammyError(
|
|
371
|
+
"Bad Request: can't parse entities",
|
|
372
|
+
{ ok: false, error_code: 400, description: "Bad Request: can't parse entities" },
|
|
373
|
+
'editMessageText',
|
|
374
|
+
{},
|
|
375
|
+
),
|
|
376
|
+
)
|
|
377
|
+
void stream.update('draft *broken')
|
|
378
|
+
await tick()
|
|
379
|
+
|
|
380
|
+
expect(bot.api.editMessageText).toHaveBeenCalledTimes(2)
|
|
381
|
+
// The fallback is an EDIT of the same message: gate fields present, and
|
|
382
|
+
// the payload is the PLAIN body (hashes differently from the rich form,
|
|
383
|
+
// so the gate never no-op-drops the recovery edit).
|
|
384
|
+
expect(seen[2]).toEqual({
|
|
385
|
+
threadId: undefined,
|
|
386
|
+
chat_id: '9',
|
|
387
|
+
messageId: id,
|
|
388
|
+
editPayload: 'draft *broken',
|
|
389
|
+
priorityClass: 'cosmetic',
|
|
390
|
+
})
|
|
391
|
+
})
|
|
392
|
+
|
|
393
|
+
/**
|
|
394
|
+
* The reviewer's F1 production-faithful probe: a multi-piece body (the
|
|
395
|
+
* escaped-markdown expansion splits it across several wire-cap pieces)
|
|
396
|
+
* where only the TAIL grows across flushes. The anchor piece stops
|
|
397
|
+
* changing, so on a perfectly healthy chat its edit resolves `undefined`
|
|
398
|
+
* every flush — first via robustApiCall's swallowed "message is not
|
|
399
|
+
* modified" 400, then via the gate's no-op drop. Pre-sentinel, that
|
|
400
|
+
* `undefined` was misread as a shed and thrown BEFORE the tail loop: no
|
|
401
|
+
* tail edit ever landed (screen frozen at "tail v1") and a false "shed"
|
|
402
|
+
* line hit stderr every flush.
|
|
403
|
+
*/
|
|
404
|
+
it('multi-piece body: a benign anchor outcome never starves the tail edits (F1)', async () => {
|
|
405
|
+
const { clock, gate, retry } = makeGatedRetry({
|
|
406
|
+
editFloorMs: 1500,
|
|
407
|
+
// Generous buckets: this test isolates the benign-undefined path, not
|
|
408
|
+
// token pressure.
|
|
409
|
+
globalPerSec: 1000, globalBurst: 100, perChatPerSec: 1000, perChatBurst: 100,
|
|
410
|
+
})
|
|
411
|
+
const base = ('a_b_c_d_e ').repeat(3000)
|
|
412
|
+
const b1 = `${base}tail v1`
|
|
413
|
+
const b2 = `${base}tail v2`
|
|
414
|
+
const b3 = `${base}tail v3`
|
|
415
|
+
// Sanity-pin the repro geometry: several pieces, prefix-stable chunking,
|
|
416
|
+
// only the LAST piece changes as the tail grows.
|
|
417
|
+
const p1 = renderOutboundChunks(b1)
|
|
418
|
+
const p2 = renderOutboundChunks(b2)
|
|
419
|
+
expect(p1.length).toBeGreaterThan(1)
|
|
420
|
+
expect(p2.length).toBe(p1.length)
|
|
421
|
+
for (let i = 0; i < p1.length - 1; i++) expect(p2[i].text).toBe(p1[i].text)
|
|
422
|
+
expect(p2[p1.length - 1].text).not.toBe(p1[p1.length - 1].text)
|
|
423
|
+
|
|
424
|
+
// Faithful edit mock: repeating a message's current text 400s with
|
|
425
|
+
// "message is not modified", which the production retry policy swallows
|
|
426
|
+
// to `undefined` — simulate the post-swallow result directly.
|
|
427
|
+
const currentText = new Map<number, string>()
|
|
428
|
+
bot.api.sendMessage.mockImplementation(async (_c, text) => {
|
|
429
|
+
const id = bot.nextMessageId++
|
|
430
|
+
currentText.set(id, text)
|
|
431
|
+
return { message_id: id }
|
|
432
|
+
})
|
|
433
|
+
bot.api.editMessageText.mockImplementation(async (_c, id, text) => {
|
|
434
|
+
const body = typeof text === 'object' && text != null ? text.markdown : text
|
|
435
|
+
if (currentText.get(id) === body) return undefined // swallowed not-modified
|
|
436
|
+
currentText.set(id, body)
|
|
437
|
+
return true as const
|
|
438
|
+
})
|
|
439
|
+
|
|
440
|
+
const logs: string[] = []
|
|
441
|
+
const stream = createStreamController({
|
|
442
|
+
bot, chatId: '1', throttleMs: 250, retry,
|
|
443
|
+
log: (m) => logs.push(m), warn: (m) => logs.push(m),
|
|
444
|
+
})
|
|
445
|
+
|
|
446
|
+
void stream.update(b1)
|
|
447
|
+
await flush()
|
|
448
|
+
const anchorId = stream.getMessageId() as number
|
|
449
|
+
const lastTailId = anchorId + p1.length - 1
|
|
450
|
+
expect(bot.api.sendMessage).toHaveBeenCalledTimes(p1.length)
|
|
451
|
+
expect(currentText.get(lastTailId)).toContain('tail v1')
|
|
452
|
+
|
|
453
|
+
// Flush 2: anchor unchanged → its edit resolves undefined via the
|
|
454
|
+
// not-modified swallow. The tail edit MUST still land.
|
|
455
|
+
void stream.update(b2)
|
|
456
|
+
await tick()
|
|
457
|
+
expect(currentText.get(lastTailId)).toContain('tail v2')
|
|
458
|
+
|
|
459
|
+
// Flush 3: anchor unchanged again → this time the GATE no-op-drops it
|
|
460
|
+
// (its hash is recorded from flush 2). The tail edit MUST still land.
|
|
461
|
+
await clock.advance(1600) // clear the tail message's own edit floor
|
|
462
|
+
void stream.update(b3)
|
|
463
|
+
await tick()
|
|
464
|
+
expect(currentText.get(lastTailId)).toContain('tail v3')
|
|
465
|
+
expect(gate.stats().global.dropped).toBeGreaterThanOrEqual(1)
|
|
466
|
+
|
|
467
|
+
// The whole run was healthy: no shed marker, no failure lines.
|
|
468
|
+
expect(logs.filter((m) => m.includes('shed') || m.includes('FAILED') || m.includes('edit failed'))).toEqual([])
|
|
469
|
+
|
|
470
|
+
// And the flushes were recorded as delivered: an identical finalize
|
|
471
|
+
// dedupes inside draft-stream — zero further API calls.
|
|
472
|
+
const editCalls = bot.api.editMessageText.mock.calls.length
|
|
473
|
+
await stream.finalize(b3)
|
|
474
|
+
expect(bot.api.editMessageText.mock.calls.length).toBe(editCalls)
|
|
475
|
+
})
|
|
476
|
+
|
|
477
|
+
it('a shed TAIL is not recorded as delivered: argument-less finalize() re-flushes and lands it (F2+F3)', async () => {
|
|
478
|
+
const { clock, gate, retry } = makeGatedRetry({
|
|
479
|
+
editFloorMs: 1500,
|
|
480
|
+
globalPerSec: 1000, globalBurst: 100, perChatPerSec: 1000, perChatBurst: 100,
|
|
481
|
+
})
|
|
482
|
+
const base = ('a_b_c_d_e ').repeat(3000)
|
|
483
|
+
const b1 = `${base}tail v1`
|
|
484
|
+
const b2 = `${base}tail v2 — the completed answer`
|
|
485
|
+
const pieceCount = renderOutboundChunks(b1).length
|
|
486
|
+
expect(pieceCount).toBeGreaterThan(1)
|
|
487
|
+
|
|
488
|
+
const logs: string[] = []
|
|
489
|
+
const stream = createStreamController({
|
|
490
|
+
bot, chatId: '1', throttleMs: 250, retry, log: (m) => logs.push(m),
|
|
491
|
+
})
|
|
492
|
+
void stream.update(b1)
|
|
493
|
+
await flush()
|
|
494
|
+
const anchorId = stream.getMessageId() as number
|
|
495
|
+
const lastTailId = anchorId + pieceCount - 1
|
|
496
|
+
|
|
497
|
+
// Flood window scoped to the LAST tail message only (H1 msg-edit scope):
|
|
498
|
+
// its edit sheds; the anchor and other pieces are unaffected.
|
|
499
|
+
gate.openFloodWindow(`msg-edit:1:${lastTailId}`, clock.now() + 30_000)
|
|
500
|
+
|
|
501
|
+
void stream.update(b2)
|
|
502
|
+
await tick()
|
|
503
|
+
// The changed piece is the suppressed tail → shed, zero edits landed on
|
|
504
|
+
// it; the flush is reported shed and the snapshot preserved (F2), NOT
|
|
505
|
+
// recorded as delivered (F3).
|
|
506
|
+
const tailEdits = () =>
|
|
507
|
+
bot.api.editMessageText.mock.calls.filter(([, id]) => id === lastTailId)
|
|
508
|
+
expect(tailEdits()).toHaveLength(0)
|
|
509
|
+
expect(gate.stats().global.shed).toBe(1)
|
|
510
|
+
expect(logs.some((m) => m.includes('shed by send gate'))).toBe(true)
|
|
511
|
+
|
|
512
|
+
// Window closes; the gateway-style ARGUMENT-LESS finalize (the
|
|
513
|
+
// disconnect-flush / turn-end cleanup path) must re-deliver the shed
|
|
514
|
+
// snapshot — pre-F2 the content was silently lost here.
|
|
515
|
+
await clock.advance(31_000)
|
|
516
|
+
await stream.finalize()
|
|
517
|
+
expect(tailEdits()).toHaveLength(1)
|
|
518
|
+
const [, , tailBody] = tailEdits()[0]
|
|
519
|
+
expect(String(tailBody)).toContain('tail v2')
|
|
520
|
+
})
|
|
521
|
+
})
|
|
@@ -207,6 +207,50 @@ describe('handleStreamReply', () => {
|
|
|
207
207
|
expect(state.activeDraftStreams.size).toBe(0)
|
|
208
208
|
})
|
|
209
209
|
|
|
210
|
+
it('done=true routes the final text through finalize(text) — the answer edit is CRITICAL for the send gate, drafts cosmetic (#3110)', async () => {
|
|
211
|
+
const state = makeState()
|
|
212
|
+
// Spy retry: pins exactly what reaches robustApiCall (= the send gate).
|
|
213
|
+
const seen: Array<Record<string, unknown> | undefined> = []
|
|
214
|
+
const retry = (<T,>(fn: () => Promise<T>, opts?: unknown) => {
|
|
215
|
+
seen.push(opts as Record<string, unknown> | undefined)
|
|
216
|
+
return fn()
|
|
217
|
+
}) as StreamReplyDeps['retry']
|
|
218
|
+
const deps = makeDeps(bot, { retry })
|
|
219
|
+
|
|
220
|
+
// Turn start → anchor send.
|
|
221
|
+
let pending = handleStreamReply({ chat_id: '1', text: 'thinking…' }, state, deps)
|
|
222
|
+
await microtaskFlush()
|
|
223
|
+
await pending
|
|
224
|
+
|
|
225
|
+
// Intermediate snapshot → a sheddable (cosmetic) draft edit.
|
|
226
|
+
pending = handleStreamReply({ chat_id: '1', text: 'half the answer' }, state, deps)
|
|
227
|
+
await microtaskFlush()
|
|
228
|
+
vi.advanceTimersByTime(600) // release the local throttle
|
|
229
|
+
await microtaskFlush()
|
|
230
|
+
await pending
|
|
231
|
+
|
|
232
|
+
// done=true → the completed answer flushes via finalize(text): the edit
|
|
233
|
+
// runs with the stream already final and is tagged critical — never shed
|
|
234
|
+
// as a draft under flood pressure.
|
|
235
|
+
pending = handleStreamReply(
|
|
236
|
+
{ chat_id: '1', text: 'the full answer', done: true },
|
|
237
|
+
state,
|
|
238
|
+
deps,
|
|
239
|
+
)
|
|
240
|
+
await microtaskFlush()
|
|
241
|
+
const result = await pending
|
|
242
|
+
|
|
243
|
+
expect(result.status).toBe('finalized')
|
|
244
|
+
expect(richEditMarkdown(bot, 1)).toBe('the full answer')
|
|
245
|
+
expect(seen[0]).toEqual({ threadId: undefined, chat_id: '1' }) // anchor send: untagged
|
|
246
|
+
expect(seen[1]).toMatchObject({ messageId: 500, priorityClass: 'cosmetic' })
|
|
247
|
+
expect(seen[2]).toMatchObject({
|
|
248
|
+
messageId: 500,
|
|
249
|
+
priorityClass: 'critical',
|
|
250
|
+
editPayload: { markdown: 'the full answer' },
|
|
251
|
+
})
|
|
252
|
+
})
|
|
253
|
+
|
|
210
254
|
it('done=true on a named lane does NOT fire terminal 👍', async () => {
|
|
211
255
|
const state = makeState()
|
|
212
256
|
const deps = makeDeps(bot)
|