switchroom 0.19.14 → 0.19.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +1 -1
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/bridge/bridge.ts +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +1 -1
- package/telegram-plugin/dist/gateway/gateway.js +1002 -504
- package/telegram-plugin/dist/server.js +1 -1
- package/telegram-plugin/gateway/forward-origin.ts +6 -1
- package/telegram-plugin/gateway/gateway.ts +4 -0
- package/telegram-plugin/gateway/narrative-lane.ts +11 -0
- package/telegram-plugin/gateway/outbox-sweep.ts +32 -2
- package/telegram-plugin/gateway/rich-message-handler.ts +235 -0
- package/telegram-plugin/gateway/stream-render.ts +107 -15
- package/telegram-plugin/gateway/unhandled-message.ts +14 -0
- package/telegram-plugin/hooks/narration-classify.d.mts +23 -0
- package/telegram-plugin/hooks/narration-classify.mjs +210 -0
- package/telegram-plugin/hooks/silent-end-scan.mjs +136 -82
- package/telegram-plugin/narrative-flush.ts +35 -0
- package/telegram-plugin/outbox.ts +73 -3
- package/telegram-plugin/shown-ledger.ts +145 -0
- package/telegram-plugin/silent-end.ts +42 -0
- package/telegram-plugin/tests/backstop-exactly-once.test.ts +335 -0
- package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +14 -0
- package/telegram-plugin/tests/forward-origin.test.ts +20 -0
- package/telegram-plugin/tests/forwarded-rich-message.test.ts +305 -0
- package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +1 -0
- package/telegram-plugin/tests/narration-leak-3513.test.ts +352 -0
- package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +42 -13
- package/telegram-plugin/tests/silent-end.test.ts +7 -1
- package/telegram-plugin/tests/turn-flush-safety.test.ts +35 -3
- package/telegram-plugin/turn-flush-safety.ts +66 -53
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* narration-leak-3513.test.ts — outcome-asserting regression suite for
|
|
3
|
+
* switchroom#3513 ("agent intent-narration leaks into the final Telegram reply
|
|
4
|
+
* as a real chat message").
|
|
5
|
+
*
|
|
6
|
+
* The single-surface invariant: every trailing plain-text block is assigned once
|
|
7
|
+
* to exactly one surface — delivered chat answer / ephemeral progress card /
|
|
8
|
+
* suppressed — by ONE shared classifier, and NO delivery path (E1 turn-flush,
|
|
9
|
+
* E2 quiescence, E3 captured-prose bridge, E4 outbox sweep) ever emits a block
|
|
10
|
+
* classified as structural narration; the one genuine unsent answer still
|
|
11
|
+
* delivers exactly once.
|
|
12
|
+
*
|
|
13
|
+
* Every test here asserts an OUTCOME (was the narration delivered? was the real
|
|
14
|
+
* answer delivered?), not merely that a code path ran, and each is written to be
|
|
15
|
+
* RED on the pre-fix code (which had two hand-synced regex classifiers and no
|
|
16
|
+
* structural/ledger guard).
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { describe, it, expect, vi } from 'vitest'
|
|
20
|
+
import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
|
21
|
+
import { tmpdir } from 'node:os'
|
|
22
|
+
import { join } from 'node:path'
|
|
23
|
+
|
|
24
|
+
import {
|
|
25
|
+
isStructuralNarration,
|
|
26
|
+
isNarrationBlock,
|
|
27
|
+
ledgerHashHex,
|
|
28
|
+
SUBSTANTIVE_MIN_CHARS,
|
|
29
|
+
} from '../hooks/narration-classify.mjs'
|
|
30
|
+
import { selectFlushDeliveryText, FLUSH_SUBSTANTIVE_MIN_CHARS } from '../turn-flush-safety.js'
|
|
31
|
+
import { decideOutboxSweep } from '../outbox.js'
|
|
32
|
+
import { decideCapturedProseDelivery } from '../silent-end.js'
|
|
33
|
+
import { appendShownBlock, isShownBlock, readShownHashes } from '../shown-ledger.js'
|
|
34
|
+
import { NarrativeFlushController } from '../narrative-flush.js'
|
|
35
|
+
|
|
36
|
+
const NARRATION = 'Now let me check the gateway logs to see what happened.'
|
|
37
|
+
const REAL_ANSWER =
|
|
38
|
+
'The gateway crashed because the outbox sweep tried to deliver a record whose ' +
|
|
39
|
+
'turnNonce had already been journaled, and the dedup cache had been evicted, so ' +
|
|
40
|
+
'the exactly-once guard fell back to the text-dedup path which returned false. ' +
|
|
41
|
+
'The fix is to check the shown-ledger before the quiet window, not after.'
|
|
42
|
+
|
|
43
|
+
// A substantive real message that the model happens to precede a NON-delivering
|
|
44
|
+
// tool with (react / pin / typing). Must NOT be treated as narration.
|
|
45
|
+
const SUBSTANTIVE_BEFORE_REACT = REAL_ANSWER
|
|
46
|
+
|
|
47
|
+
describe('shared classifier — the ONE source of truth (correction 2 + 3)', () => {
|
|
48
|
+
it('short/narration-shaped block followed by a tool IS structural narration', () => {
|
|
49
|
+
expect(isStructuralNarration(NARRATION, true)).toBe(true)
|
|
50
|
+
expect(isStructuralNarration('On it — pulling the numbers…', true)).toBe(true)
|
|
51
|
+
expect(isStructuralNarration('', true)).toBe(true)
|
|
52
|
+
})
|
|
53
|
+
|
|
54
|
+
it('SUBSTANTIVE block followed by a tool is NOT narration (react/pin/typing still delivers)', () => {
|
|
55
|
+
// The #3237 asymmetry: a real answer preceding react/pin/typing must deliver.
|
|
56
|
+
expect(SUBSTANTIVE_BEFORE_REACT.length).toBeGreaterThan(SUBSTANTIVE_MIN_CHARS)
|
|
57
|
+
expect(isStructuralNarration(SUBSTANTIVE_BEFORE_REACT, true)).toBe(false)
|
|
58
|
+
})
|
|
59
|
+
|
|
60
|
+
it('a terminal block (nothing followed it) is NEVER structural narration', () => {
|
|
61
|
+
// followedByToolUse false OR absent → never structural (could be the answer).
|
|
62
|
+
expect(isStructuralNarration(NARRATION, false)).toBe(false)
|
|
63
|
+
expect(isStructuralNarration(NARRATION, undefined)).toBe(false)
|
|
64
|
+
expect(isStructuralNarration('Yes.', false)).toBe(false)
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
it('FLUSH_SUBSTANTIVE_MIN_CHARS is wired to the shared constant (no drift)', () => {
|
|
68
|
+
expect(FLUSH_SUBSTANTIVE_MIN_CHARS).toBe(SUBSTANTIVE_MIN_CHARS)
|
|
69
|
+
})
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
describe('E1/E2 turn-flush — narration never becomes the delivered answer', () => {
|
|
73
|
+
it('drops a trailing narration block that a tool followed, keeps the real answer', () => {
|
|
74
|
+
const out = selectFlushDeliveryText([REAL_ANSWER, NARRATION], [false, true])
|
|
75
|
+
expect(out).toContain(REAL_ANSWER)
|
|
76
|
+
expect(out).not.toContain(NARRATION)
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
it('drops a SHORT non-regex block a tool followed (the class the old regex missed)', () => {
|
|
80
|
+
// "The logs show three errors." matches NO wording heuristic (no opener, no
|
|
81
|
+
// trailing ellipsis/colon) — the pre-fix regex-only classifier would have
|
|
82
|
+
// DELIVERED it. It is structural narration only because a tool followed it
|
|
83
|
+
// AND it is under the substantive floor. This is the core #3513 discriminator.
|
|
84
|
+
const shortNonRegex = 'The logs show three errors.'
|
|
85
|
+
expect(isNarrationBlock(shortNonRegex)).toBe(false)
|
|
86
|
+
expect(shortNonRegex.length).toBeLessThan(SUBSTANTIVE_MIN_CHARS)
|
|
87
|
+
const out = selectFlushDeliveryText([REAL_ANSWER, shortNonRegex], [false, true])
|
|
88
|
+
expect(out).toContain(REAL_ANSWER)
|
|
89
|
+
expect(out).not.toContain(shortNonRegex)
|
|
90
|
+
})
|
|
91
|
+
|
|
92
|
+
it('zero-reply, narration-only turn (all blocks followed by tools) flushes NOTHING', () => {
|
|
93
|
+
const out = selectFlushDeliveryText(['Let me pull the logs.', NARRATION], [true, true])
|
|
94
|
+
expect(out).toBe('')
|
|
95
|
+
})
|
|
96
|
+
|
|
97
|
+
it('a genuine terminal answer with no provenance still flushes (fail-open)', () => {
|
|
98
|
+
// Provenance-less single block — conservative: deliver it.
|
|
99
|
+
expect(selectFlushDeliveryText(['Here are the results:'])).toBe('Here are the results:')
|
|
100
|
+
})
|
|
101
|
+
|
|
102
|
+
it('a SUBSTANTIVE answer followed by a non-delivering tool STILL flushes', () => {
|
|
103
|
+
const out = selectFlushDeliveryText([SUBSTANTIVE_BEFORE_REACT], [true])
|
|
104
|
+
expect(out).toBe(SUBSTANTIVE_BEFORE_REACT)
|
|
105
|
+
})
|
|
106
|
+
})
|
|
107
|
+
|
|
108
|
+
describe('E4 outbox sweep — a ledger-marked block is never re-delivered (correction 1)', () => {
|
|
109
|
+
const base = {
|
|
110
|
+
now: 10_000_000,
|
|
111
|
+
deliveredNonces: new Set<string>(),
|
|
112
|
+
textAlreadyDelivered: false,
|
|
113
|
+
routable: true,
|
|
114
|
+
quietMs: 0,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
it('skips a record whose text was shown on the ephemeral card', () => {
|
|
118
|
+
const d = decideOutboxSweep({
|
|
119
|
+
...base,
|
|
120
|
+
record: { turnNonce: 'c:_#5', text: NARRATION, createdAt: 0 },
|
|
121
|
+
shownLedgerHit: true,
|
|
122
|
+
})
|
|
123
|
+
expect(d.action).toBe('skip-ephemeral-shown')
|
|
124
|
+
expect(d.text).toBeUndefined()
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
it('delivers a real answer that was NOT shown on the card', () => {
|
|
128
|
+
const d = decideOutboxSweep({
|
|
129
|
+
...base,
|
|
130
|
+
record: { turnNonce: 'c:_#5', text: REAL_ANSWER, createdAt: base.now },
|
|
131
|
+
shownLedgerHit: false,
|
|
132
|
+
})
|
|
133
|
+
expect(d.action).toBe('send')
|
|
134
|
+
expect(d.text).toContain(REAL_ANSWER)
|
|
135
|
+
})
|
|
136
|
+
|
|
137
|
+
it('the ledger hit is checked AFTER journaled (exactly-once wins) but BEFORE quiet', () => {
|
|
138
|
+
// Already-journaled record short-circuits regardless of ledger state.
|
|
139
|
+
const journaled = decideOutboxSweep({
|
|
140
|
+
...base,
|
|
141
|
+
deliveredNonces: new Set(['c:_#5']),
|
|
142
|
+
record: { turnNonce: 'c:_#5', text: NARRATION, createdAt: 0 },
|
|
143
|
+
shownLedgerHit: true,
|
|
144
|
+
})
|
|
145
|
+
expect(journaled.action).toBe('skip-journaled')
|
|
146
|
+
})
|
|
147
|
+
})
|
|
148
|
+
|
|
149
|
+
describe('E4 outbox sweep × durable shown-ledger — end-to-end wiring (correction 1 + 4)', () => {
|
|
150
|
+
// Integration test: exercise the REAL ledger round-trip (append/markShown +
|
|
151
|
+
// isShownBlock/isBlockShown) keyed by turnNonce THROUGH the outbox-sweep
|
|
152
|
+
// decision, mirroring the production composition at
|
|
153
|
+
// gateway/outbox-sweep.ts:122 (`shownLedgerHit: isShownBlock(record.turnNonce,
|
|
154
|
+
// record.text, deps.stateDir)`). Unlike the unit tests above that pass a
|
|
155
|
+
// hand-set `shownLedgerHit` boolean, this drives the flag FROM the durable
|
|
156
|
+
// ledger file — so if the ledger wiring is removed (append becomes a no-op, or
|
|
157
|
+
// decideOutboxSweep stops consulting the hit), the skip assertion goes RED.
|
|
158
|
+
const base = {
|
|
159
|
+
now: 10_000_000,
|
|
160
|
+
deliveredNonces: new Set<string>(),
|
|
161
|
+
textAlreadyDelivered: false,
|
|
162
|
+
routable: true,
|
|
163
|
+
quietMs: 0,
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
it('a block marked shown for its turnNonce is skipped; a non-shown block delivers', () => {
|
|
167
|
+
const dir = mkdtempSync(join(tmpdir(), 'outbox-ledger-'))
|
|
168
|
+
try {
|
|
169
|
+
const nonce = 'c:_#42'
|
|
170
|
+
// markShown: the mid-turn narrative paint appended this block to the
|
|
171
|
+
// durable ledger for THIS turn's nonce.
|
|
172
|
+
appendShownBlock(nonce, NARRATION, dir)
|
|
173
|
+
|
|
174
|
+
// Sweep the SAME (nonce, text): the production wiring derives the flag from
|
|
175
|
+
// the ledger, so the record must be skipped as ephemeral-shown.
|
|
176
|
+
const shown = decideOutboxSweep({
|
|
177
|
+
...base,
|
|
178
|
+
record: { turnNonce: nonce, text: NARRATION, createdAt: 0 },
|
|
179
|
+
shownLedgerHit: isShownBlock(nonce, NARRATION, dir),
|
|
180
|
+
})
|
|
181
|
+
expect(isShownBlock(nonce, NARRATION, dir)).toBe(true)
|
|
182
|
+
expect(shown.action).toBe('skip-ephemeral-shown')
|
|
183
|
+
expect(shown.text).toBeUndefined()
|
|
184
|
+
|
|
185
|
+
// A NON-shown block (the genuine unsent answer, never appended) still
|
|
186
|
+
// delivers exactly once — the ledger miss falls through to send.
|
|
187
|
+
const notShown = decideOutboxSweep({
|
|
188
|
+
...base,
|
|
189
|
+
record: { turnNonce: nonce, text: REAL_ANSWER, createdAt: base.now },
|
|
190
|
+
shownLedgerHit: isShownBlock(nonce, REAL_ANSWER, dir),
|
|
191
|
+
})
|
|
192
|
+
expect(isShownBlock(nonce, REAL_ANSWER, dir)).toBe(false)
|
|
193
|
+
expect(notShown.action).toBe('send')
|
|
194
|
+
expect(notShown.text).toContain(REAL_ANSWER)
|
|
195
|
+
|
|
196
|
+
// Cross-turn isolation: the SAME text under a DIFFERENT turnNonce is not a
|
|
197
|
+
// ledger hit, so it delivers (no cross-turn suppression through the sweep).
|
|
198
|
+
const otherTurn = decideOutboxSweep({
|
|
199
|
+
...base,
|
|
200
|
+
record: { turnNonce: 'c:_#43', text: NARRATION, createdAt: base.now },
|
|
201
|
+
shownLedgerHit: isShownBlock('c:_#43', NARRATION, dir),
|
|
202
|
+
})
|
|
203
|
+
expect(otherTurn.action).toBe('send')
|
|
204
|
+
} finally {
|
|
205
|
+
rmSync(dir, { recursive: true, force: true })
|
|
206
|
+
}
|
|
207
|
+
})
|
|
208
|
+
})
|
|
209
|
+
|
|
210
|
+
describe('E3 captured-prose bridge — a shown block is refused (correction 1)', () => {
|
|
211
|
+
/** Write a real silent-end-pending.json into a temp stateDir. */
|
|
212
|
+
function withState(text: string, run: (deps: { stateDir: string }) => void): void {
|
|
213
|
+
const dir = mkdtempSync(join(tmpdir(), 'silent-end-'))
|
|
214
|
+
try {
|
|
215
|
+
writeFileSync(
|
|
216
|
+
join(dir, 'silent-end-pending.json'),
|
|
217
|
+
JSON.stringify({
|
|
218
|
+
chatId: 'c',
|
|
219
|
+
threadId: null,
|
|
220
|
+
turnKey: 'c:_',
|
|
221
|
+
turnId: 'c:_#5',
|
|
222
|
+
pendingText: text,
|
|
223
|
+
retryCount: 0,
|
|
224
|
+
timestamp: Date.now(),
|
|
225
|
+
}),
|
|
226
|
+
)
|
|
227
|
+
run({ stateDir: dir })
|
|
228
|
+
} finally {
|
|
229
|
+
rmSync(dir, { recursive: true, force: true })
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
it('refuses to bridge prose already surfaced on the card', () => {
|
|
234
|
+
withState(REAL_ANSWER, ({ stateDir }) => {
|
|
235
|
+
const d = decideCapturedProseDelivery(
|
|
236
|
+
{ turnKey: 'c:_', turnId: 'c:_#5', minChars: 10 },
|
|
237
|
+
{ stateDir, isBlockShown: () => true },
|
|
238
|
+
)
|
|
239
|
+
expect(d.deliver).toBe(false)
|
|
240
|
+
expect(d.reason).toBe('ephemeral-shown')
|
|
241
|
+
})
|
|
242
|
+
})
|
|
243
|
+
|
|
244
|
+
it('bridges a genuine unsent answer that was NOT shown (empty-capture divergence)', () => {
|
|
245
|
+
withState(REAL_ANSWER, ({ stateDir }) => {
|
|
246
|
+
const d = decideCapturedProseDelivery(
|
|
247
|
+
{ turnKey: 'c:_', turnId: 'c:_#5', minChars: 10 },
|
|
248
|
+
{ stateDir, isBlockShown: () => false },
|
|
249
|
+
)
|
|
250
|
+
expect(d.deliver).toBe(true)
|
|
251
|
+
if (d.deliver) expect(d.text).toContain(REAL_ANSWER)
|
|
252
|
+
})
|
|
253
|
+
})
|
|
254
|
+
})
|
|
255
|
+
|
|
256
|
+
describe('durable shown-ledger — round trip + envelope-less no-op (correction 4)', () => {
|
|
257
|
+
it('marks then recognises a structural narration block for its turnNonce', () => {
|
|
258
|
+
const dir = mkdtempSync(join(tmpdir(), 'shown-ledger-'))
|
|
259
|
+
try {
|
|
260
|
+
appendShownBlock('c:_#7', NARRATION, dir)
|
|
261
|
+
expect(isShownBlock('c:_#7', NARRATION, dir)).toBe(true)
|
|
262
|
+
// A different turn's nonce must NOT match (no cross-turn suppression).
|
|
263
|
+
expect(isShownBlock('c:_#8', NARRATION, dir)).toBe(false)
|
|
264
|
+
// The stored hash matches the shared classifier hash.
|
|
265
|
+
expect(readShownHashes('c:_#7', dir).has(ledgerHashHex(NARRATION))).toBe(true)
|
|
266
|
+
} finally {
|
|
267
|
+
rmSync(dir, { recursive: true, force: true })
|
|
268
|
+
}
|
|
269
|
+
})
|
|
270
|
+
|
|
271
|
+
it('a null turnNonce (envelope-less handback/cron) is never written', () => {
|
|
272
|
+
const dir = mkdtempSync(join(tmpdir(), 'shown-ledger-'))
|
|
273
|
+
try {
|
|
274
|
+
appendShownBlock(null, NARRATION, dir)
|
|
275
|
+
expect(isShownBlock(null, NARRATION, dir)).toBe(false)
|
|
276
|
+
expect(readShownHashes(null, dir).size).toBe(0)
|
|
277
|
+
} finally {
|
|
278
|
+
rmSync(dir, { recursive: true, force: true })
|
|
279
|
+
}
|
|
280
|
+
})
|
|
281
|
+
})
|
|
282
|
+
|
|
283
|
+
/** Fake at-most-one scheduler mirroring the real unref'd setTimeout wiring. */
|
|
284
|
+
class FakeScheduler {
|
|
285
|
+
private fn: (() => void) | null = null
|
|
286
|
+
arm(fn: () => void): void {
|
|
287
|
+
this.fn = fn
|
|
288
|
+
}
|
|
289
|
+
disarm(): void {
|
|
290
|
+
this.fn = null
|
|
291
|
+
}
|
|
292
|
+
fire(): void {
|
|
293
|
+
const fn = this.fn
|
|
294
|
+
this.fn = null
|
|
295
|
+
fn?.()
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
describe('narrative controller — durable-mark ONLY mid-turn narration (correction 4)', () => {
|
|
300
|
+
function make() {
|
|
301
|
+
const show = vi.fn<[string], void>()
|
|
302
|
+
const retractShown = vi.fn<[string], void>()
|
|
303
|
+
const markDurableNarration = vi.fn<[string], void>()
|
|
304
|
+
const scheduler = new FakeScheduler()
|
|
305
|
+
const ctrl = new NarrativeFlushController(
|
|
306
|
+
{ show, retractShown, markDurableNarration },
|
|
307
|
+
scheduler,
|
|
308
|
+
250,
|
|
309
|
+
)
|
|
310
|
+
return { ctrl, show, markDurableNarration, scheduler }
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
it('marks a block SHOWN mid-turn because another narration block followed it (stage)', () => {
|
|
314
|
+
const { ctrl, show, markDurableNarration } = make()
|
|
315
|
+
ctrl.stage('First, let me check the logs…')
|
|
316
|
+
ctrl.stage('Now scanning the outbox…') // lookahead → shows + marks the first
|
|
317
|
+
expect(show).toHaveBeenCalledWith('First, let me check the logs…')
|
|
318
|
+
expect(markDurableNarration).toHaveBeenCalledWith('First, let me check the logs…')
|
|
319
|
+
})
|
|
320
|
+
|
|
321
|
+
it('marks a block SHOWN mid-turn because a tool followed it (resolveOnTool)', () => {
|
|
322
|
+
const { ctrl, markDurableNarration } = make()
|
|
323
|
+
ctrl.stage(NARRATION)
|
|
324
|
+
ctrl.resolveOnTool('Bash', { command: 'grep' }) // non-reply tool lookahead
|
|
325
|
+
expect(markDurableNarration).toHaveBeenCalledWith(NARRATION)
|
|
326
|
+
})
|
|
327
|
+
|
|
328
|
+
it('NEVER marks the timer-painted terminal block (could be the real answer)', () => {
|
|
329
|
+
const { ctrl, show, markDurableNarration, scheduler } = make()
|
|
330
|
+
ctrl.stage(REAL_ANSWER)
|
|
331
|
+
scheduler.fire() // timer paints the LAST block early
|
|
332
|
+
expect(show).toHaveBeenCalledWith(REAL_ANSWER)
|
|
333
|
+
expect(markDurableNarration).not.toHaveBeenCalled()
|
|
334
|
+
})
|
|
335
|
+
|
|
336
|
+
it('NEVER marks the turn-end paint (terminal, could be the real answer)', () => {
|
|
337
|
+
const { ctrl, show, markDurableNarration } = make()
|
|
338
|
+
ctrl.stage(REAL_ANSWER)
|
|
339
|
+
ctrl.flushAtTurnEnd('') // no reply delivered → shows trailing prose, but never marks
|
|
340
|
+
expect(show).toHaveBeenCalledWith(REAL_ANSWER)
|
|
341
|
+
expect(markDurableNarration).not.toHaveBeenCalled()
|
|
342
|
+
})
|
|
343
|
+
|
|
344
|
+
it('does NOT mark a SUBSTANTIVE block even mid-turn (substance-gated, correction 3)', () => {
|
|
345
|
+
const { ctrl, markDurableNarration } = make()
|
|
346
|
+
ctrl.stage(SUBSTANTIVE_BEFORE_REACT)
|
|
347
|
+
ctrl.resolveOnTool('react', { emoji: '👍' }) // non-delivering tool
|
|
348
|
+
// Followed by a tool, but substantive → not structural narration → not marked,
|
|
349
|
+
// so a backstop could still deliver it if it were the genuine unsent answer.
|
|
350
|
+
expect(markDurableNarration).not.toHaveBeenCalled()
|
|
351
|
+
})
|
|
352
|
+
})
|
|
@@ -404,10 +404,13 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
|
|
|
404
404
|
const NARRATION_B = "Now let me query the second source; it is slower than expected, hang tight…" // opener + …
|
|
405
405
|
const NARRATION_C = "I'll cross-reference the last set of figures against the ledger before I summarise…" // opener + …
|
|
406
406
|
|
|
407
|
-
it('
|
|
408
|
-
//
|
|
409
|
-
//
|
|
410
|
-
//
|
|
407
|
+
it('narration interleaved with work-tools, ending on a terminal narration line → only the terminal block, never the join (#3513 structural coalescer)', () => {
|
|
408
|
+
// #3513 follow-up — A and B are each followed by a turn-continuing tool_use
|
|
409
|
+
// (Bash / Read) → structurally intra-turn → UNCONDITIONALLY suppressed.
|
|
410
|
+
// C is the terminal block (no tool after it), so the deterministic coalescer
|
|
411
|
+
// delivers C ALONE. The #3228 Finding-2 join-masquerade (concatenating short
|
|
412
|
+
// narration to cross the 200-char floor) is STILL prevented: the 200-floor
|
|
413
|
+
// `pendingText` stays undefined, and the delivered text is never the join.
|
|
411
414
|
expect((NARRATION_A + '\n\n' + NARRATION_B + '\n\n' + NARRATION_C).length)
|
|
412
415
|
.toBeGreaterThanOrEqual(200)
|
|
413
416
|
const text = jsonl(
|
|
@@ -421,8 +424,12 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
|
|
|
421
424
|
const r = scanTurnForFinalReply(text)
|
|
422
425
|
expect(r.decided).toBe('block')
|
|
423
426
|
expect(r.reason).toBe('no-final-reply')
|
|
424
|
-
//
|
|
425
|
-
expect(r.pendingText).
|
|
427
|
+
// Never the join of the suppressed A/B blocks (masquerade guard preserved).
|
|
428
|
+
expect(r.pendingText).not.toContain(NARRATION_A)
|
|
429
|
+
expect(r.pendingText).not.toContain(NARRATION_B)
|
|
430
|
+
// The terminal block is the coalescer's selection (delivered via the
|
|
431
|
+
// zero-reply lowered floor); the 200-char masquerade path yields nothing.
|
|
432
|
+
expect(r.pendingText).toBe(NARRATION_C)
|
|
426
433
|
})
|
|
427
434
|
|
|
428
435
|
it('zero-delivery turn of MULTIPLE sub-200 REAL-CONTENT blocks → JOINED pendingText (review item 3 drop fix)', () => {
|
|
@@ -445,15 +452,34 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
|
|
|
445
452
|
expect(r.hasTrailingProse).toBe(true)
|
|
446
453
|
})
|
|
447
454
|
|
|
448
|
-
it('leading narration + a real sub-200 answer block → only the answer is delivered (
|
|
455
|
+
it('leading narration FOLLOWED BY A TOOL + a real sub-200 answer block → only the answer is delivered (structural strip)', () => {
|
|
456
|
+
// #3513 follow-up — the narration is stripped by the STRUCTURAL signal (a
|
|
457
|
+
// work-tool followed it), not by wording. The terminal answer block (no tool
|
|
458
|
+
// after it) is the coalescer's selection.
|
|
449
459
|
const NARR = 'Let me pull the numbers first…' // narration opener + …
|
|
450
460
|
const ANSWER = 'Revenue was up 12% quarter-over-quarter, driven mostly by the new enterprise tier.' // ~85, real
|
|
451
|
-
const text = jsonl(
|
|
461
|
+
const text = jsonl(
|
|
462
|
+
ENQUEUE,
|
|
463
|
+
assistantText(NARR),
|
|
464
|
+
assistantToolUse('Bash', { command: 'psql -c "select ..."' }),
|
|
465
|
+
assistantText(ANSWER),
|
|
466
|
+
)
|
|
452
467
|
const r = scanTurnForFinalReply(text)
|
|
453
468
|
expect(r.decided).toBe('block')
|
|
454
469
|
expect(r.pendingText).toBe(ANSWER)
|
|
455
470
|
})
|
|
456
471
|
|
|
472
|
+
it('leading narration with NO tool + a real answer → both delivered joined (fail-open; no wording strip on the backstop path, #3513)', () => {
|
|
473
|
+
// Without a structural continuation signal the coalescer delivers the full
|
|
474
|
+
// terminal run — the wording-based strip is gone (it was provably incomplete).
|
|
475
|
+
const NARR = 'Let me pull the numbers first…'
|
|
476
|
+
const ANSWER = 'Revenue was up 12% quarter-over-quarter, driven mostly by the new enterprise tier.'
|
|
477
|
+
const text = jsonl(ENQUEUE, assistantText(NARR), assistantText(ANSWER))
|
|
478
|
+
const r = scanTurnForFinalReply(text)
|
|
479
|
+
expect(r.decided).toBe('block')
|
|
480
|
+
expect(r.pendingText).toBe(`${NARR}\n\n${ANSWER}`)
|
|
481
|
+
})
|
|
482
|
+
|
|
457
483
|
it('trailing narration after a delivered reply, all sub-floor → block only if a real ≥floor block exists; here → allow, no pendingText', () => {
|
|
458
484
|
// Each trailing block is sub-floor, so `sawUndeliveredTextAfterAllow` is
|
|
459
485
|
// false → allow. (The old joined-floor logic never affected the block
|
|
@@ -469,18 +495,21 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
|
|
|
469
495
|
expect(r.pendingText).toBeUndefined()
|
|
470
496
|
})
|
|
471
497
|
|
|
472
|
-
it('a genuine ≥floor answer followed by a SHORT closer →
|
|
473
|
-
// "big answer then short closer"
|
|
474
|
-
//
|
|
498
|
+
it('a genuine ≥floor answer followed by a SHORT closer → both delivered as the terminal run (answer never dropped, #3513)', () => {
|
|
499
|
+
// #3513 follow-up — "big answer then short closer", NO tool after either
|
|
500
|
+
// block, so both are the terminal run and are delivered JOINED. The #3228
|
|
501
|
+
// concern (a short closer DISPLACING the answer) cannot happen: the coalescer
|
|
502
|
+
// never drops the answer — it delivers the whole terminal run.
|
|
475
503
|
const answer = 'Here is the real answer you were waiting for: ' + 'A'.repeat(300)
|
|
504
|
+
const closer = 'Let me know if you need anything else.'
|
|
476
505
|
const text = jsonl(
|
|
477
506
|
ENQUEUE,
|
|
478
507
|
assistantText(answer),
|
|
479
|
-
assistantText(
|
|
508
|
+
assistantText(closer),
|
|
480
509
|
)
|
|
481
510
|
const r = scanTurnForFinalReply(text)
|
|
482
511
|
expect(r.decided).toBe('block')
|
|
483
|
-
expect(r.pendingText).toBe(answer)
|
|
512
|
+
expect(r.pendingText).toBe(`${answer}\n\n${closer}`)
|
|
484
513
|
})
|
|
485
514
|
|
|
486
515
|
it('substance-floor boundary: a single trailing block at 199/200/201 chars', () => {
|
|
@@ -787,11 +787,17 @@ describe('silent-end-interrupt-stop hook — integration (#1775: transcript-scan
|
|
|
787
787
|
expect(lowered.text).toBe(joined)
|
|
788
788
|
})
|
|
789
789
|
|
|
790
|
-
it('review item 3:
|
|
790
|
+
it('review item 3: intra-turn narration blocks each FOLLOWED BY A TOOL → NO pendingText (structural suppression, #3513)', () => {
|
|
791
|
+
// #3513 follow-up — each narration block is followed by a turn-continuing
|
|
792
|
+
// tool_use, so the deterministic coalescer suppresses BOTH unconditionally.
|
|
793
|
+
// The terminal run is empty and the last (tool-followed) block is sub-200 →
|
|
794
|
+
// the empty-terminal corner returns null → no pendingText, no masquerade.
|
|
791
795
|
const transcript = writeTranscript([
|
|
792
796
|
ENQUEUE,
|
|
793
797
|
{ type: 'assistant', message: { content: [{ type: 'text', text: 'Let me check the first source now, scanning the rows…' }] } },
|
|
798
|
+
{ type: 'assistant', message: { content: [{ type: 'tool_use', id: 't1', name: 'Bash', input: { command: 'ls' } }] } },
|
|
794
799
|
{ type: 'assistant', message: { content: [{ type: 'text', text: "Now let me query the second source; hang tight…" }] } },
|
|
800
|
+
{ type: 'assistant', message: { content: [{ type: 'tool_use', id: 't2', name: 'Read', input: { file_path: '/tmp/x' } }] } },
|
|
795
801
|
])
|
|
796
802
|
const r = runHook({ session_id: 's', transcript_path: transcript, hook_event_name: 'Stop' })
|
|
797
803
|
expect(JSON.parse(r.stdout.trim()).decision).toBe('block')
|
|
@@ -683,13 +683,36 @@ describe('selectFlushDeliveryText — deliver the terminal answer, strip only na
|
|
|
683
683
|
expect(out).toBe('Here are the results:')
|
|
684
684
|
})
|
|
685
685
|
|
|
686
|
-
|
|
686
|
+
// #3513 follow-up — decideTurnFlush now routes through the deterministic
|
|
687
|
+
// `selectBackstopDelivery` coalescer, NOT the wording heuristic. With NO
|
|
688
|
+
// structural provenance (no `capturedBlockMeta`, so no block was followed by a
|
|
689
|
+
// turn-continuing tool_use) EVERY block is part of the terminal run and is
|
|
690
|
+
// delivered joined — fail OPEN, never drop. The wording-based narration strip
|
|
691
|
+
// is gone on the backstop path (it was provably incomplete; #3513). When the
|
|
692
|
+
// structural signal IS present it, and only it, suppresses.
|
|
693
|
+
it('decideTurnFlush delivers the full terminal run when NO tool-follow provenance is present (fail-open)', () => {
|
|
687
694
|
const decision = decideTurnFlush({
|
|
688
695
|
chatId: 'chat1',
|
|
689
696
|
replyCalled: false,
|
|
690
697
|
capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
|
|
691
698
|
})
|
|
692
699
|
expect(decision.kind).toBe('flush')
|
|
700
|
+
if (decision.kind === 'flush') {
|
|
701
|
+
expect(decision.text).toBe(`Let me check.\n\nNow let me compose the reply.\n\n${answer}`)
|
|
702
|
+
}
|
|
703
|
+
})
|
|
704
|
+
|
|
705
|
+
it('decideTurnFlush suppresses tool-followed blocks UNCONDITIONALLY (structural provenance drives the coalescer)', () => {
|
|
706
|
+
// The first two blocks were each followed by a turn-continuing tool_use →
|
|
707
|
+
// intra-turn narration → suppressed with NO length/wording gate. Only the
|
|
708
|
+
// terminal answer (not tool-followed) survives.
|
|
709
|
+
const decision = decideTurnFlush({
|
|
710
|
+
chatId: 'chat1',
|
|
711
|
+
replyCalled: false,
|
|
712
|
+
capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
|
|
713
|
+
capturedBlockMeta: [true, true, false],
|
|
714
|
+
})
|
|
715
|
+
expect(decision.kind).toBe('flush')
|
|
693
716
|
if (decision.kind === 'flush') {
|
|
694
717
|
expect(decision.text).toBe(answer)
|
|
695
718
|
expect(decision.text).not.toContain('Let me check')
|
|
@@ -832,11 +855,20 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
|
|
|
832
855
|
// trailing ellipsis/colon — so only the structural flag could strip it.
|
|
833
856
|
const wrapUp = 'Saved that to memory.'
|
|
834
857
|
// meta[0]=true (a memory tool_use followed the real answer), [1]=false.
|
|
858
|
+
// The STANDALONE `selectFlushDeliveryText` (the old #3237 substance-gated
|
|
859
|
+
// function, unchanged and still exported) keeps the substantial block.
|
|
835
860
|
const out = selectFlushDeliveryText([realAnswer, wrapUp], [true, false])
|
|
836
861
|
expect(out).toBe(`${realAnswer}\n\n${wrapUp}`)
|
|
837
862
|
expect(out).toContain('The migration completed cleanly')
|
|
838
863
|
|
|
839
|
-
//
|
|
864
|
+
// #3513 follow-up — decideTurnFlush now routes through
|
|
865
|
+
// `selectBackstopDelivery`, which suppresses a tool-followed block
|
|
866
|
+
// UNCONDITIONALLY (no substantive-floor carve-out). This DELIBERATELY
|
|
867
|
+
// overturns #3237's substance-gate ON THE BACKSTOP PATH: a real answer the
|
|
868
|
+
// model followed with a turn-continuing tool_use is treated as intra-turn
|
|
869
|
+
// and excluded; only the terminal (non-tool-followed) block is delivered.
|
|
870
|
+
// The deterministic contract is "the terminal run wins" — the model keeps
|
|
871
|
+
// its answer by ending on it or calling the reply tool.
|
|
840
872
|
const decision = decideTurnFlush({
|
|
841
873
|
chatId: 'chat1',
|
|
842
874
|
replyCalled: false,
|
|
@@ -845,7 +877,7 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
|
|
|
845
877
|
})
|
|
846
878
|
expect(decision.kind).toBe('flush')
|
|
847
879
|
if (decision.kind === 'flush') {
|
|
848
|
-
expect(decision.text).toBe(
|
|
880
|
+
expect(decision.text).toBe(wrapUp)
|
|
849
881
|
}
|
|
850
882
|
})
|
|
851
883
|
|