switchroom 0.19.14 → 0.19.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +1 -1
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/bridge/bridge.ts +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +31 -2
- package/telegram-plugin/dist/gateway/gateway.js +1690 -932
- package/telegram-plugin/dist/server.js +31 -2
- package/telegram-plugin/gateway/background-shell-liveness.ts +65 -0
- package/telegram-plugin/gateway/forward-origin.ts +6 -1
- package/telegram-plugin/gateway/gateway.ts +10 -57
- package/telegram-plugin/gateway/narrative-lane.ts +11 -0
- package/telegram-plugin/gateway/outbound-send-path.ts +25 -23
- package/telegram-plugin/gateway/outbox-listen-markup.ts +67 -0
- package/telegram-plugin/gateway/outbox-sweep.ts +124 -20
- package/telegram-plugin/gateway/rich-message-handler.ts +241 -0
- package/telegram-plugin/gateway/silence-poke-session-event.ts +89 -0
- package/telegram-plugin/gateway/stream-render.ts +107 -15
- package/telegram-plugin/gateway/unhandled-message.ts +14 -0
- package/telegram-plugin/hooks/narration-classify.d.mts +23 -0
- package/telegram-plugin/hooks/narration-classify.mjs +210 -0
- package/telegram-plugin/hooks/silent-end-scan.mjs +136 -82
- package/telegram-plugin/narrative-flush.ts +35 -0
- package/telegram-plugin/outbox.ts +73 -3
- package/telegram-plugin/session-tail.ts +88 -1
- package/telegram-plugin/shown-ledger.ts +145 -0
- package/telegram-plugin/silence-poke.ts +118 -1
- package/telegram-plugin/silent-end.ts +42 -0
- package/telegram-plugin/tests/background-shell-liveness.test.ts +72 -0
- package/telegram-plugin/tests/backstop-exactly-once.test.ts +335 -0
- package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +14 -0
- package/telegram-plugin/tests/feed-survival.test.ts +7 -1
- package/telegram-plugin/tests/fixtures/bg-shell-liveness-3519.jsonl +3 -0
- package/telegram-plugin/tests/forward-origin.test.ts +20 -0
- package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +290 -0
- package/telegram-plugin/tests/forwarded-rich-message.test.ts +305 -0
- package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +1 -0
- package/telegram-plugin/tests/narration-leak-3513.test.ts +352 -0
- package/telegram-plugin/tests/outbox-sweep-listen-button.test.ts +253 -0
- package/telegram-plugin/tests/session-tail.test.ts +91 -1
- package/telegram-plugin/tests/silence-poke.test.ts +280 -0
- package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +42 -13
- package/telegram-plugin/tests/silent-end.test.ts +7 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +66 -0
- package/telegram-plugin/tests/turn-flush-safety.test.ts +35 -3
- package/telegram-plugin/tests/voice-normalize-text.test.ts +82 -1
- package/telegram-plugin/tts-normalize.ts +12 -0
- package/telegram-plugin/turn-flush-safety.ts +66 -53
- package/telegram-plugin/voice-normalize-text.ts +100 -0
- package/telegram-plugin/voice-ondemand.ts +71 -0
|
@@ -683,13 +683,36 @@ describe('selectFlushDeliveryText — deliver the terminal answer, strip only na
|
|
|
683
683
|
expect(out).toBe('Here are the results:')
|
|
684
684
|
})
|
|
685
685
|
|
|
686
|
-
|
|
686
|
+
// #3513 follow-up — decideTurnFlush now routes through the deterministic
|
|
687
|
+
// `selectBackstopDelivery` coalescer, NOT the wording heuristic. With NO
|
|
688
|
+
// structural provenance (no `capturedBlockMeta`, so no block was followed by a
|
|
689
|
+
// turn-continuing tool_use) EVERY block is part of the terminal run and is
|
|
690
|
+
// delivered joined — fail OPEN, never drop. The wording-based narration strip
|
|
691
|
+
// is gone on the backstop path (it was provably incomplete; #3513). When the
|
|
692
|
+
// structural signal IS present it, and only it, suppresses.
|
|
693
|
+
it('decideTurnFlush delivers the full terminal run when NO tool-follow provenance is present (fail-open)', () => {
|
|
687
694
|
const decision = decideTurnFlush({
|
|
688
695
|
chatId: 'chat1',
|
|
689
696
|
replyCalled: false,
|
|
690
697
|
capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
|
|
691
698
|
})
|
|
692
699
|
expect(decision.kind).toBe('flush')
|
|
700
|
+
if (decision.kind === 'flush') {
|
|
701
|
+
expect(decision.text).toBe(`Let me check.\n\nNow let me compose the reply.\n\n${answer}`)
|
|
702
|
+
}
|
|
703
|
+
})
|
|
704
|
+
|
|
705
|
+
it('decideTurnFlush suppresses tool-followed blocks UNCONDITIONALLY (structural provenance drives the coalescer)', () => {
|
|
706
|
+
// The first two blocks were each followed by a turn-continuing tool_use →
|
|
707
|
+
// intra-turn narration → suppressed with NO length/wording gate. Only the
|
|
708
|
+
// terminal answer (not tool-followed) survives.
|
|
709
|
+
const decision = decideTurnFlush({
|
|
710
|
+
chatId: 'chat1',
|
|
711
|
+
replyCalled: false,
|
|
712
|
+
capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
|
|
713
|
+
capturedBlockMeta: [true, true, false],
|
|
714
|
+
})
|
|
715
|
+
expect(decision.kind).toBe('flush')
|
|
693
716
|
if (decision.kind === 'flush') {
|
|
694
717
|
expect(decision.text).toBe(answer)
|
|
695
718
|
expect(decision.text).not.toContain('Let me check')
|
|
@@ -832,11 +855,20 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
|
|
|
832
855
|
// trailing ellipsis/colon — so only the structural flag could strip it.
|
|
833
856
|
const wrapUp = 'Saved that to memory.'
|
|
834
857
|
// meta[0]=true (a memory tool_use followed the real answer), [1]=false.
|
|
858
|
+
// The STANDALONE `selectFlushDeliveryText` (the old #3237 substance-gated
|
|
859
|
+
// function, unchanged and still exported) keeps the substantial block.
|
|
835
860
|
const out = selectFlushDeliveryText([realAnswer, wrapUp], [true, false])
|
|
836
861
|
expect(out).toBe(`${realAnswer}\n\n${wrapUp}`)
|
|
837
862
|
expect(out).toContain('The migration completed cleanly')
|
|
838
863
|
|
|
839
|
-
//
|
|
864
|
+
// #3513 follow-up — decideTurnFlush now routes through
|
|
865
|
+
// `selectBackstopDelivery`, which suppresses a tool-followed block
|
|
866
|
+
// UNCONDITIONALLY (no substantive-floor carve-out). This DELIBERATELY
|
|
867
|
+
// overturns #3237's substance-gate ON THE BACKSTOP PATH: a real answer the
|
|
868
|
+
// model followed with a turn-continuing tool_use is treated as intra-turn
|
|
869
|
+
// and excluded; only the terminal (non-tool-followed) block is delivered.
|
|
870
|
+
// The deterministic contract is "the terminal run wins" — the model keeps
|
|
871
|
+
// its answer by ending on it or calling the reply tool.
|
|
840
872
|
const decision = decideTurnFlush({
|
|
841
873
|
chatId: 'chat1',
|
|
842
874
|
replyCalled: false,
|
|
@@ -845,7 +877,7 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
|
|
|
845
877
|
})
|
|
846
878
|
expect(decision.kind).toBe('flush')
|
|
847
879
|
if (decision.kind === 'flush') {
|
|
848
|
-
expect(decision.text).toBe(
|
|
880
|
+
expect(decision.text).toBe(wrapUp)
|
|
849
881
|
}
|
|
850
882
|
})
|
|
851
883
|
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import { describe, it, expect } from 'bun:test'
|
|
9
|
-
import { normalizeForSpeech } from '../voice-normalize-text.js'
|
|
9
|
+
import { normalizeForSpeech, decodeHtmlEntities } from '../voice-normalize-text.js'
|
|
10
10
|
|
|
11
11
|
describe('normalizeForSpeech — reported markdown/symbol cases', () => {
|
|
12
12
|
it('drops stray tildes (never spoken as "tilde")', () => {
|
|
@@ -254,3 +254,84 @@ describe('normalizeForSpeech — conservative, does not mangle real words', () =
|
|
|
254
254
|
expect(normalizeForSpeech(' \n ')).toBe('')
|
|
255
255
|
})
|
|
256
256
|
})
|
|
257
|
+
|
|
258
|
+
describe('normalizeForSpeech — backslash escapes & HTML entities (voice trash)', () => {
|
|
259
|
+
it('strips a literal backslash-b (\\b) so it is never spoken as "backslash b"', () => {
|
|
260
|
+
// The exact operator report: the regex/escape "\b" was read as "slash b".
|
|
261
|
+
const out = normalizeForSpeech('the regex \\b word boundary')
|
|
262
|
+
expect(out).toBe('the regex b word boundary')
|
|
263
|
+
expect(out).not.toContain('\\')
|
|
264
|
+
})
|
|
265
|
+
|
|
266
|
+
it('unescapes MarkdownV2 punctuation escapes without leaving backslashes', () => {
|
|
267
|
+
expect(normalizeForSpeech('done\\. next\\! wait\\-')).toBe('done. next! wait-')
|
|
268
|
+
expect(normalizeForSpeech('a \\* b')).not.toContain('\\')
|
|
269
|
+
})
|
|
270
|
+
|
|
271
|
+
it('an escaped emphasis marker collapses to the inner word, not a backslash', () => {
|
|
272
|
+
expect(normalizeForSpeech('Use \\*literal\\* here')).toBe('Use literal here')
|
|
273
|
+
})
|
|
274
|
+
|
|
275
|
+
it('drops a Windows-path-style backslash run (C:\\build)', () => {
|
|
276
|
+
expect(normalizeForSpeech('path C:\\build\\out done')).toBe('path C:buildout done')
|
|
277
|
+
})
|
|
278
|
+
|
|
279
|
+
it('decodes HTML entities so & / < are not read as "amp" / "lt"', () => {
|
|
280
|
+
expect(normalizeForSpeech('Tom & Jerry')).toBe('Tom and Jerry')
|
|
281
|
+
expect(normalizeForSpeech('5 < 10 > 3')).toBe('5 < 10 > 3')
|
|
282
|
+
expect(normalizeForSpeech('it's here')).toBe("it's here")
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
it('produces clean plain speech for a rich mixed reply (end-to-end)', () => {
|
|
286
|
+
const reply =
|
|
287
|
+
'**Bold** and `code\\b` and a [label](https://example.com/x) ' +
|
|
288
|
+
'with Tom & Jerry and a regex \\b\\.'
|
|
289
|
+
const out = normalizeForSpeech(reply)
|
|
290
|
+
expect(out).not.toContain('\\')
|
|
291
|
+
expect(out).not.toContain('`')
|
|
292
|
+
expect(out).not.toContain('*')
|
|
293
|
+
expect(out).not.toContain('&')
|
|
294
|
+
expect(out).not.toContain('example.com')
|
|
295
|
+
expect(out).toContain('label')
|
|
296
|
+
expect(out).toContain('Tom and Jerry')
|
|
297
|
+
})
|
|
298
|
+
|
|
299
|
+
it('is idempotent: a second pass finds no backslashes/entities to change', () => {
|
|
300
|
+
const reply = 'a \\b and Tom & Jerry \\. end'
|
|
301
|
+
const once = normalizeForSpeech(reply)
|
|
302
|
+
expect(normalizeForSpeech(once)).toBe(once)
|
|
303
|
+
})
|
|
304
|
+
})
|
|
305
|
+
|
|
306
|
+
describe('normalizeForSpeech — review findings (fixpoint decode, metachar, nits)', () => {
|
|
307
|
+
it('L2 parity: a double-encoded entity decodes to a fixpoint (depth-independent)', () => {
|
|
308
|
+
// Single decode and double decode must land on the SAME spoken char so the
|
|
309
|
+
// immediate voice-out and the lazy Listen tap never diverge.
|
|
310
|
+
const single = normalizeForSpeech('&amp;lt;')
|
|
311
|
+
const doubled = normalizeForSpeech(normalizeForSpeech('&amp;lt;'))
|
|
312
|
+
expect(single).toBe('<')
|
|
313
|
+
expect(doubled).toBe('<')
|
|
314
|
+
expect(normalizeForSpeech('&amp;amp;')).toBe('&')
|
|
315
|
+
})
|
|
316
|
+
|
|
317
|
+
it('L1: an entity that decodes to a line-leading metachar keeps a spoken form', () => {
|
|
318
|
+
// # → '#'. Naively re-fed to the heading stripper it would vanish; the
|
|
319
|
+
// user escaped it on purpose, so it must survive as spoken "hash".
|
|
320
|
+
expect(normalizeForSpeech('# Heading')).toBe('hash Heading')
|
|
321
|
+
expect(normalizeForSpeech('2 * 3')).toBe('2 asterisk 3')
|
|
322
|
+
})
|
|
323
|
+
|
|
324
|
+
it('nit: a dangling trailing backslash is dropped, never spoken', () => {
|
|
325
|
+
expect(normalizeForSpeech('ends here\\')).toBe('ends here')
|
|
326
|
+
expect(normalizeForSpeech('ends here\\')).not.toContain('\\')
|
|
327
|
+
})
|
|
328
|
+
|
|
329
|
+
it('nit: \ decodes to a backslash which is then stripped (no trash)', () => {
|
|
330
|
+
expect(normalizeForSpeech('X\Y')).toBe('XY')
|
|
331
|
+
expect(normalizeForSpeech('X\Y')).not.toContain('\\')
|
|
332
|
+
})
|
|
333
|
+
|
|
334
|
+
it('fixpoint does not over-decode entity-less text (Q&A stays literal)', () => {
|
|
335
|
+
expect(decodeHtmlEntities('Q&A test')).toBe('Q&A test')
|
|
336
|
+
})
|
|
337
|
+
})
|
|
@@ -38,6 +38,8 @@
|
|
|
38
38
|
* ~5 → "about 5", > blockquote markers dropped.
|
|
39
39
|
*/
|
|
40
40
|
|
|
41
|
+
import { decodeHtmlEntities, stripBackslashEscapes } from './voice-normalize-text'
|
|
42
|
+
|
|
41
43
|
const NULL = '\x00'
|
|
42
44
|
const INLINE_PH = `${NULL}TN_INLINE`
|
|
43
45
|
|
|
@@ -197,6 +199,16 @@ export function normalizeForTts(text: string): string {
|
|
|
197
199
|
|
|
198
200
|
let s = text.replace(/\r\n?/g, '\n')
|
|
199
201
|
|
|
202
|
+
// -- HTML entities → char, then backslash escapes → the escaped char. The
|
|
203
|
+
// last line of defence at the /tts body build: the Listen lazy path and
|
|
204
|
+
// the pre-synth queue can synthesize from cache entries that predate the
|
|
205
|
+
// normalizeForSpeech coverage, so these must be stripped here too. Both
|
|
206
|
+
// are idempotent — if normalizeForSpeech already ran there is nothing
|
|
207
|
+
// left to decode/unescape. Without this the engine speaks `\b` as
|
|
208
|
+
// "backslash b" and `&` as "amp".
|
|
209
|
+
s = decodeHtmlEntities(s)
|
|
210
|
+
s = stripBackslashEscapes(s)
|
|
211
|
+
|
|
200
212
|
// -- Code fences → spoken placeholder (before anything can see contents).
|
|
201
213
|
s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*?\n[ \t]*\2[ \t]*(?=\n|$)/g, '$1code block omitted.')
|
|
202
214
|
s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*$/g, '$1code block omitted.')
|
|
@@ -19,6 +19,13 @@
|
|
|
19
19
|
* and can be set to `0` / `false` / `off` to disable without a rebuild.
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
|
+
import {
|
|
23
|
+
isNarrationBlock,
|
|
24
|
+
isStructuralNarration,
|
|
25
|
+
selectBackstopDelivery,
|
|
26
|
+
SUBSTANTIVE_MIN_CHARS,
|
|
27
|
+
} from './hooks/narration-classify.mjs'
|
|
28
|
+
|
|
22
29
|
const SILENT_MARKERS = new Set(['NO_REPLY', 'HEARTBEAT_OK'])
|
|
23
30
|
// Small buffer so `NO_REPLY.` with a stray period still counts as silent.
|
|
24
31
|
const SILENT_MARKER_MAX_LEN = Math.max(
|
|
@@ -134,7 +141,7 @@ export function endsWithSilentMarker(text: string | undefined): boolean {
|
|
|
134
141
|
* `hooks/silent-end-scan.mjs` — the same bar the codebase uses everywhere to
|
|
135
142
|
* recognise "this text block is a real answer, not a short narration/closer".
|
|
136
143
|
*/
|
|
137
|
-
export const FLUSH_SUBSTANTIVE_MIN_CHARS =
|
|
144
|
+
export const FLUSH_SUBSTANTIVE_MIN_CHARS = SUBSTANTIVE_MIN_CHARS
|
|
138
145
|
|
|
139
146
|
/**
|
|
140
147
|
* Pick the text a turn-flush should actually DELIVER from the captured
|
|
@@ -203,9 +210,33 @@ export function selectFlushDeliveryText(
|
|
|
203
210
|
.map((b, i) => ({ text: b.trim(), followedByToolUse: followedByToolUse?.[i] }))
|
|
204
211
|
.filter(c => c.text.length > 0)
|
|
205
212
|
if (candidates.length === 0) return ''
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
213
|
+
|
|
214
|
+
// #3513 — the STRUCTURAL rule now applies to the TERMINAL block too, not only
|
|
215
|
+
// preceding blocks. A lone trailing block that a `tool_use` FOLLOWED (the
|
|
216
|
+
// model kept acting after writing it) is intra-turn narration by construction,
|
|
217
|
+
// never the terminal answer — so it must not be delivered as one. This is the
|
|
218
|
+
// primary fix for the observed leak ("Now checking the gateway logs…"),
|
|
219
|
+
// wording-independent where the opener regex fails. It stays substance-gated
|
|
220
|
+
// (correction 3): only a SHORT or narration-shaped block that a tool followed
|
|
221
|
+
// is suppressed — a SUBSTANTIVE, non-narration block followed by a
|
|
222
|
+
// non-delivering tool (react / pin / typing / edit) is NOT structural
|
|
223
|
+
// narration and still delivers.
|
|
224
|
+
//
|
|
225
|
+
// Only the STRUCTURAL signal (`followedByToolUse === true`) may strip the
|
|
226
|
+
// terminal block; the opener/trailer heuristic remains a residual tie-breaker
|
|
227
|
+
// for PRECEDING blocks only, so a provenance-less single block still delivers
|
|
228
|
+
// verbatim (#3276 finding 7 — a colon-terminated line that IS the whole answer
|
|
229
|
+
// is never dropped).
|
|
230
|
+
let cs = candidates
|
|
231
|
+
while (cs.length > 1 && isStructuralNarration(cs[cs.length - 1].text, cs[cs.length - 1].followedByToolUse)) {
|
|
232
|
+
cs = cs.slice(0, -1)
|
|
233
|
+
}
|
|
234
|
+
if (cs.length === 1) {
|
|
235
|
+
return isStructuralNarration(cs[0].text, cs[0].followedByToolUse) ? '' : cs[0].text
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
const answer = cs[cs.length - 1].text
|
|
239
|
+
const preceding = cs.slice(0, -1)
|
|
209
240
|
// Deliver only the terminal answer when every earlier block is
|
|
210
241
|
// intent-narration. Per block:
|
|
211
242
|
// - structural flag TRUE ⇒ a tool_use followed this block, but that alone
|
|
@@ -228,50 +259,15 @@ export function selectFlushDeliveryText(
|
|
|
228
259
|
? false
|
|
229
260
|
: isNarrationBlock(c.text),
|
|
230
261
|
)
|
|
231
|
-
return allNarration ? answer :
|
|
262
|
+
return allNarration ? answer : cs.map(c => c.text).join('\n\n')
|
|
232
263
|
}
|
|
233
264
|
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
* legitimate multi-paragraph answer content — so gating on length either drops a
|
|
241
|
-
* short real answer or keeps a long narration. When earlier blocks don't match
|
|
242
|
-
* this opener we keep the full joined text (never drop content we can't
|
|
243
|
-
* confidently attribute to narration).
|
|
244
|
-
*/
|
|
245
|
-
const NARRATION_OPENER =
|
|
246
|
-
/^(let me\b|lemme\b|i'?ll\b|i will\b|i am going to\b|i'?m going to\b|i'?m about to\b|going to\b|first,?\s+(?:let me|i'?ll|i will)\b|now,?\s+(?:let me|i'?ll|i will)\b|next,?\s+(?:let me|i'?ll|i will)\b|let'?s\b)/i
|
|
247
|
-
|
|
248
|
-
/**
|
|
249
|
-
* #3276 guard 8 — the `NARRATION_OPENER` regex alone misses common progress
|
|
250
|
-
* narration that opens with a gerund/present-continuous verb and trails off
|
|
251
|
-
* into an ellipsis or colon: "Checking now…", "Pulling the numbers:",
|
|
252
|
-
* "Looking into that…". Relying on the opener regex leaked those blocks into
|
|
253
|
-
* the delivered answer. This recognises a NON-terminal narration line
|
|
254
|
-
* deterministically, WITHOUT re-gating on a min-char floor (a short real
|
|
255
|
-
* answer like "Yes, done." must still deliver): a single short line that ends
|
|
256
|
-
* with an ellipsis or a colon is progress narration, not the terminal answer.
|
|
257
|
-
*
|
|
258
|
-
* Kept conservative on purpose — only a SINGLE-line block (no internal
|
|
259
|
-
* paragraph) under the substantive floor, ending in `…` / `...` / `:`, so a
|
|
260
|
-
* genuine multi-paragraph answer that happens to end a paragraph with a colon
|
|
261
|
-
* is never mistaken for narration.
|
|
262
|
-
*/
|
|
263
|
-
const NARRATION_TRAILER = /(?:\.{3}|…|:)\s*$/
|
|
264
|
-
|
|
265
|
-
function isTrailingNarrationLine(block: string): boolean {
|
|
266
|
-
const t = block.trim()
|
|
267
|
-
if (t.length === 0 || t.length >= FLUSH_SUBSTANTIVE_MIN_CHARS) return false
|
|
268
|
-
if (t.includes('\n')) return false
|
|
269
|
-
return NARRATION_TRAILER.test(t)
|
|
270
|
-
}
|
|
271
|
-
|
|
272
|
-
function isNarrationBlock(block: string): boolean {
|
|
273
|
-
return NARRATION_OPENER.test(block.trimStart()) || isTrailingNarrationLine(block)
|
|
274
|
-
}
|
|
265
|
+
// The narration heuristic (`isNarrationBlock` / openers / trailer) and the
|
|
266
|
+
// structural rule (`isStructuralNarration`) now live in the ONE shared
|
|
267
|
+
// classifier `hooks/narration-classify.mjs`, imported at the top of this file
|
|
268
|
+
// and by the unbundled Stop hook (`hooks/silent-end-scan.mjs`) alike — so the
|
|
269
|
+
// TS gateway side and the `.mjs` side can never drift (switchroom#3513
|
|
270
|
+
// correction 2; retires the old "MUST stay in sync" hand-synced copies).
|
|
275
271
|
|
|
276
272
|
export type FlushDecision =
|
|
277
273
|
| { kind: 'flush'; text: string }
|
|
@@ -283,6 +279,11 @@ export type FlushSkipReason =
|
|
|
283
279
|
| 'no-inbound-chat'
|
|
284
280
|
| 'empty-text'
|
|
285
281
|
| 'silent-marker'
|
|
282
|
+
// A prior BACKSTOP (E2 answer-ready flush, or an out-of-process E3/E4 machine
|
|
283
|
+
// across a crash/restart) already delivered this turn's answer — the durable
|
|
284
|
+
// exactly-once-among-backstops guard (#3513 follow-up, MF2). Set by the
|
|
285
|
+
// caller, not `decideTurnFlush`.
|
|
286
|
+
| 'already-delivered'
|
|
286
287
|
|
|
287
288
|
export interface FlushDecisionInput {
|
|
288
289
|
/** Inbound chat the turn was servicing. `null` means system-initiated /
|
|
@@ -375,14 +376,26 @@ export function decideTurnFlush(input: FlushDecisionInput): FlushDecision {
|
|
|
375
376
|
// sentinel — treat the whole turn as intentionally silent rather than
|
|
376
377
|
// flush the prose with the sentinel glued on.
|
|
377
378
|
if (endsWithSilentMarker(joined)) return { kind: 'skip', reason: 'silent-marker' }
|
|
378
|
-
//
|
|
379
|
-
//
|
|
380
|
-
//
|
|
381
|
-
//
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
379
|
+
// #3513 follow-up — deterministic, model-discipline-free coalescing. The
|
|
380
|
+
// DELIVERED text is the TERMINAL RUN (the suffix of blocks NOT followed by a
|
|
381
|
+
// turn-continuing tool_use); every tool-followed block is suppressed
|
|
382
|
+
// UNCONDITIONALLY (no length/wording gate), replacing #3515's substance-gated
|
|
383
|
+
// heuristic on this backstop path. `selectBackstopDelivery` returns null when
|
|
384
|
+
// nothing terminal survives (a short tool-followed fragment) — treat that as
|
|
385
|
+
// 'empty-text' so the turn routes to the Stop-hook re-prompt ladder rather
|
|
386
|
+
// than delivering an interim narration line. The silent-marker / empty guards
|
|
387
|
+
// above still run on the full `joined` string so a partly-silent turn is
|
|
388
|
+
// classified correctly.
|
|
389
|
+
const selected = selectBackstopDelivery(
|
|
390
|
+
input.capturedText.map((text, i) => ({
|
|
391
|
+
text,
|
|
392
|
+
followedByToolUse: input.capturedBlockMeta?.[i],
|
|
393
|
+
})),
|
|
394
|
+
)
|
|
395
|
+
if (selected == null || selected.text.trim().length === 0) {
|
|
396
|
+
return { kind: 'skip', reason: 'empty-text' }
|
|
385
397
|
}
|
|
398
|
+
return { kind: 'flush', text: selected.text }
|
|
386
399
|
}
|
|
387
400
|
|
|
388
401
|
/**
|
|
@@ -53,6 +53,89 @@
|
|
|
53
53
|
/** Replace a fenced code block with a spoken placeholder. */
|
|
54
54
|
const CODE_BLOCK_PLACEHOLDER = 'code block omitted'
|
|
55
55
|
|
|
56
|
+
/** Named HTML entities the reply text realistically carries. */
|
|
57
|
+
const HTML_ENTITIES: Record<string, string> = {
|
|
58
|
+
amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ',
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Markdown metacharacters that the block/emphasis/table stripper would
|
|
63
|
+
* silently consume at line-start or as a pair. When such a char arrives via
|
|
64
|
+
* an entity escape the user meant it LITERALLY (that is the whole point of
|
|
65
|
+
* escaping it), so instead of emitting the raw char — which the downstream
|
|
66
|
+
* stripper would then eat, losing the intent — we emit a neutral spoken form
|
|
67
|
+
* that survives every later pass. Deterministic; the spoken form contains no
|
|
68
|
+
* `&`/`;` so it can never re-enter the entity decoder.
|
|
69
|
+
*/
|
|
70
|
+
const METACHAR_SPOKEN: Record<string, string> = {
|
|
71
|
+
'#': ' hash ',
|
|
72
|
+
'*': ' asterisk ',
|
|
73
|
+
'_': ' underscore ',
|
|
74
|
+
'~': ' tilde ',
|
|
75
|
+
'`': ' backtick ',
|
|
76
|
+
'|': ' bar ',
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** One decode pass: named + numeric entities → char (or spoken metachar). */
|
|
80
|
+
function decodeHtmlEntitiesOnce(input: string): string {
|
|
81
|
+
const toChar = (cp: number, raw: string): string => {
|
|
82
|
+
if (!(cp > 0 && cp <= 0x10ffff)) return raw
|
|
83
|
+
const ch = String.fromCodePoint(cp)
|
|
84
|
+
return METACHAR_SPOKEN[ch] ?? ch
|
|
85
|
+
}
|
|
86
|
+
return input
|
|
87
|
+
.replace(/&#x([0-9a-f]+);/gi, (m, hex: string) => toChar(parseInt(hex, 16), m))
|
|
88
|
+
.replace(/&#(\d+);/g, (m, dec: string) => toChar(Number(dec), m))
|
|
89
|
+
.replace(/&([a-z][a-z0-9]*);/gi, (m, name: string) => {
|
|
90
|
+
const ch = HTML_ENTITIES[name.toLowerCase()]
|
|
91
|
+
if (ch === undefined) return m
|
|
92
|
+
return METACHAR_SPOKEN[ch] ?? ch
|
|
93
|
+
})
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Decode HTML entities (named + numeric) to their character so a TTS engine
|
|
98
|
+
* never reads `&` as "amp". Unknown named entities are left untouched.
|
|
99
|
+
* Pure + deterministic.
|
|
100
|
+
*
|
|
101
|
+
* Iterates to a FIXPOINT: a double-encoded entity (`&amp;lt;`) is decoded
|
|
102
|
+
* repeatedly until no entity remains, so this pass is depth-idempotent —
|
|
103
|
+
* applying it once yields the same result as applying it twice. That keeps
|
|
104
|
+
* every voice callsite in lockstep: the immediate voice-out path runs
|
|
105
|
+
* normalizeForSpeech THEN normalizeForTts, and the value the lazy Listen tap /
|
|
106
|
+
* pre-synth queue reads is itself already normalizeForSpeech'd before its own
|
|
107
|
+
* normalizeForTts — fixpoint decoding guarantees both speak an identical
|
|
108
|
+
* string regardless of how deep the original encoding was. The loop strictly
|
|
109
|
+
* shrinks the entity count each turn (and is capped) so it always terminates.
|
|
110
|
+
* Text WITHOUT a trailing `;` (e.g. `Q&A`) matches nothing and is returned
|
|
111
|
+
* untouched.
|
|
112
|
+
*/
|
|
113
|
+
export function decodeHtmlEntities(input: string): string {
|
|
114
|
+
let s = input
|
|
115
|
+
// A fully-decodable chain shrinks by at least one entity per pass; the cap
|
|
116
|
+
// is a belt-and-braces guard against any pathological crafted input.
|
|
117
|
+
for (let i = 0; i < 10; i++) {
|
|
118
|
+
const next = decodeHtmlEntitiesOnce(s)
|
|
119
|
+
if (next === s) break
|
|
120
|
+
s = next
|
|
121
|
+
}
|
|
122
|
+
return s
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Remove markdown/MarkdownV2 backslash escapes so the spoken text carries no
|
|
127
|
+
* literal backslashes. A backslash before ANY single character is dropped,
|
|
128
|
+
* keeping the character (`\.` → ".", `\*` → "*", `\b` → "b"); a dangling
|
|
129
|
+
* trailing backslash is dropped. Pure + deterministic + idempotent (a second
|
|
130
|
+
* pass finds no backslashes). A real newline is preserved (only the escaping
|
|
131
|
+
* backslash is consumed).
|
|
132
|
+
*/
|
|
133
|
+
export function stripBackslashEscapes(input: string): string {
|
|
134
|
+
// `\X` → `X` for any following char (including an escaped `\\`), then drop
|
|
135
|
+
// any lone backslash the first pass left (an escaped backslash's survivor).
|
|
136
|
+
return input.replace(/\\([\s\S])/g, '$1').replace(/\\/g, '')
|
|
137
|
+
}
|
|
138
|
+
|
|
56
139
|
// ---------------------------------------------------------------------------
|
|
57
140
|
// Number → words helpers (small, deterministic, English cardinal only).
|
|
58
141
|
// Used by the numbers/units pass. Supports 0..999_999_999 which is far more
|
|
@@ -168,6 +251,23 @@ export function normalizeForSpeech(input: string): string {
|
|
|
168
251
|
if (!input) return ''
|
|
169
252
|
let s = input.replace(/\r\n?/g, '\n')
|
|
170
253
|
|
|
254
|
+
// 0a. HTML entities → their character. The reply text can carry entity
|
|
255
|
+
// escapes (`&`, `<`, `'`) that a TTS engine would otherwise
|
|
256
|
+
// read as "amp" / "lt" / a digit run. Decode BEFORE markdown/symbol
|
|
257
|
+
// passes so the recovered char is then handled naturally (e.g. a
|
|
258
|
+
// decoded `&` becomes "and" in the symbols pass).
|
|
259
|
+
s = decodeHtmlEntities(s)
|
|
260
|
+
|
|
261
|
+
// 0b. Backslash escapes → the escaped character. Telegram MarkdownV2 and
|
|
262
|
+
// CommonMark escape literal punctuation with a leading backslash
|
|
263
|
+
// (`\.`, `\-`, `\*`), and a backslash before a non-punctuation char
|
|
264
|
+
// (`\b`) is a literal backslash. Left in place the engine speaks
|
|
265
|
+
// "backslash b" / "slash b" — exactly the operator's "trash" report.
|
|
266
|
+
// Unescaping here (before the emphasis pass) restores the literal text
|
|
267
|
+
// so genuine `*emphasis*` markers are still stripped downstream while
|
|
268
|
+
// an escaped `\*` collapses to nothing spoken. Runs once; idempotent.
|
|
269
|
+
s = stripBackslashEscapes(s)
|
|
270
|
+
|
|
171
271
|
// 0. Emoji & pictographs → dropped entirely, then whitespace collapsed.
|
|
172
272
|
// TTS reads an emoji as its long CLDR name ("grinning face"), which is
|
|
173
273
|
// noise. We also drop `:shortcode:` forms so nothing is read as
|
|
@@ -311,3 +311,74 @@ export function mayInjectListenButton(
|
|
|
311
311
|
if (rawKeyboard == null) return true
|
|
312
312
|
return !rawKeyboard.some((row) => Array.isArray(row) && row.length > 0)
|
|
313
313
|
}
|
|
314
|
+
|
|
315
|
+
/** The voice-out plan fields the Listen-button decision depends on. A subset of
|
|
316
|
+
* the gateway's full `VoiceOutPlan` so this pure helper carries no gateway
|
|
317
|
+
* import. */
|
|
318
|
+
export interface ListenButtonVoiceOutPlan {
|
|
319
|
+
engine: 'kokoro' | 'openai'
|
|
320
|
+
voice?: string
|
|
321
|
+
speed: number
|
|
322
|
+
replyMode: 'voice+text' | 'voice-only' | 'on-demand'
|
|
323
|
+
ttsChunks: string[]
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
/** The decision to inject a 🔊 Listen button on a message, plus the cache
|
|
327
|
+
* payload the caller must persist so a tap can resynthesize. */
|
|
328
|
+
export interface ListenButtonPlan {
|
|
329
|
+
/** The single-row Listen keyboard to attach as `reply_markup`. */
|
|
330
|
+
replyMarkup: {
|
|
331
|
+
inline_keyboard: Array<Array<{ text: string; callback_data: string }>>
|
|
332
|
+
}
|
|
333
|
+
/** The freshly minted token keying the keyboard callback + cache entry. */
|
|
334
|
+
token: string
|
|
335
|
+
/** The payload the caller stores under `token` in the VoiceOnDemandCache
|
|
336
|
+
* (and enqueues for eager pre-synth). */
|
|
337
|
+
payload: VoiceOnDemandPayload
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* The SINGLE source of truth for "should this delivered message carry a 🔊
|
|
342
|
+
* Listen button, and if so which token/keyboard/cache payload?" — shared by
|
|
343
|
+
* BOTH the normal reply path (`sendReply` in outbound-send-path.ts) and the
|
|
344
|
+
* durable-outbox safety-net delivery (`outbox-sweep.ts` wiring), so a
|
|
345
|
+
* net-delivered final answer is indistinguishable from a normally-delivered
|
|
346
|
+
* one (switchroom #3502 regression: the sweep dropped the button).
|
|
347
|
+
*
|
|
348
|
+
* Returns null — no button — when any gate fails:
|
|
349
|
+
* - no voice-out plan, or the plan is not kokoro on-demand (openai on-demand
|
|
350
|
+
* synthesizes immediately and its taps would dead-end on the local sidecar);
|
|
351
|
+
* - the TTS text is empty (nothing to speak — the empty-TTS guard);
|
|
352
|
+
* - the agent supplied its own inline keyboard (single_use collision gate —
|
|
353
|
+
* see mayInjectListenButton).
|
|
354
|
+
*
|
|
355
|
+
* Pure: it mints a token and builds the keyboard/payload but performs NO side
|
|
356
|
+
* effects (no cache write, no eager enqueue). The caller owns those so the
|
|
357
|
+
* token is only persisted when it decides to actually attach the button.
|
|
358
|
+
*/
|
|
359
|
+
export function planListenButton(params: {
|
|
360
|
+
voiceOutPlan: ListenButtonVoiceOutPlan | null
|
|
361
|
+
rawKeyboard: unknown[][] | undefined | null
|
|
362
|
+
}): ListenButtonPlan | null {
|
|
363
|
+
const plan = params.voiceOutPlan
|
|
364
|
+
// on-demand Listen button is a LOCAL-engine (kokoro) feature only.
|
|
365
|
+
const useOnDemandButton =
|
|
366
|
+
plan != null && plan.replyMode === 'on-demand' && plan.engine === 'kokoro'
|
|
367
|
+
if (!useOnDemandButton) return null
|
|
368
|
+
// Empty-TTS guard: nothing to speak → no button.
|
|
369
|
+
if (!(plan.ttsChunks.length > 0 && plan.ttsChunks[0]!.length > 0)) return null
|
|
370
|
+
// Collision gate: agent supplied its own keyboard → skip.
|
|
371
|
+
if (!mayInjectListenButton(params.rawKeyboard)) return null
|
|
372
|
+
|
|
373
|
+
const token = mintVoiceOnDemandToken()
|
|
374
|
+
return {
|
|
375
|
+
replyMarkup: buildListenKeyboard(token),
|
|
376
|
+
token,
|
|
377
|
+
payload: {
|
|
378
|
+
// ttsChunks[0] is already normalizeForSpeech(reply) (kokoro path).
|
|
379
|
+
text: plan.ttsChunks[0]!,
|
|
380
|
+
...(plan.voice != null ? { voice: plan.voice } : {}),
|
|
381
|
+
speed: plan.speed,
|
|
382
|
+
},
|
|
383
|
+
}
|
|
384
|
+
}
|