switchroom 0.18.31 → 0.18.33
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +4 -2
- package/dist/auth-broker/index.js +21 -3
- package/dist/cli/notion-write-pretool.mjs +4 -2
- package/dist/cli/switchroom.js +1410 -852
- package/dist/host-control/main.js +22 -4
- package/dist/vault/approvals/kernel-server.js +21 -3
- package/dist/vault/broker/server.js +48 -4
- package/package.json +4 -3
- package/profiles/_base/start.sh.hbs +148 -23
- package/telegram-plugin/dist/gateway/gateway.js +62726 -58282
- package/telegram-plugin/gateway/agent-button-callback-handler.ts +237 -0
- package/telegram-plugin/gateway/ask-callback-handler.ts +92 -0
- package/telegram-plugin/gateway/attachment-message-handlers.ts +152 -0
- package/telegram-plugin/gateway/backstop-delivery.ts +223 -23
- package/telegram-plugin/gateway/boot-card.ts +169 -1
- package/telegram-plugin/gateway/bot-commands-model-effort.ts +209 -0
- package/telegram-plugin/gateway/bot-commands-start-info.ts +108 -0
- package/telegram-plugin/gateway/callback-query-handlers.ts +124 -0
- package/telegram-plugin/gateway/captured-answer-resume.ts +259 -0
- package/telegram-plugin/gateway/card-approval-keyboards.test.ts +28 -0
- package/telegram-plugin/gateway/card-tool-handlers.ts +639 -0
- package/telegram-plugin/gateway/checklist-message-handler.ts +107 -0
- package/telegram-plugin/gateway/delivery-confirm-wiring.ts +133 -0
- package/telegram-plugin/gateway/disconnect-flush.ts +6 -44
- package/telegram-plugin/gateway/gateway-import-clean.test.ts +188 -0
- package/telegram-plugin/gateway/gateway.ts +6403 -13456
- package/telegram-plugin/gateway/inbound-delivery-machine-dispatch.ts +7 -15
- package/telegram-plugin/gateway/inbound-delivery-machine-shadow.ts +35 -68
- package/telegram-plugin/gateway/inbound-interceptors.ts +1133 -0
- package/telegram-plugin/gateway/inbound-router.ts +400 -0
- package/telegram-plugin/gateway/liveness-wiring.ts +440 -0
- package/telegram-plugin/gateway/media-message-handlers.ts +256 -0
- package/telegram-plugin/gateway/mental-model-propose-card.ts +16 -0
- package/telegram-plugin/gateway/model-command.ts +23 -0
- package/telegram-plugin/gateway/narrative-lane.ts +865 -0
- package/telegram-plugin/gateway/obligation-ledger.ts +42 -0
- package/telegram-plugin/gateway/obligation-store.ts +37 -1
- package/telegram-plugin/gateway/obligation-wiring.ts +333 -0
- package/telegram-plugin/gateway/outbound-send-path.ts +2012 -0
- package/telegram-plugin/gateway/photo-message-handler.ts +80 -0
- package/telegram-plugin/gateway/pinned-message-handler.ts +86 -0
- package/telegram-plugin/gateway/secret-request-card.test.ts +46 -0
- package/telegram-plugin/gateway/secret-request-card.ts +45 -0
- package/telegram-plugin/gateway/stream-render.ts +2166 -0
- package/telegram-plugin/gateway/turn-end.ts +606 -0
- package/telegram-plugin/gateway/turn-start-surfaces.ts +298 -0
- package/telegram-plugin/gateway/vault-request-access-card.ts +16 -0
- package/telegram-plugin/gateway/vault-request-save-card.test.ts +49 -0
- package/telegram-plugin/gateway/vault-request-save-card.ts +52 -0
- package/telegram-plugin/gateway/voice-message-handler.ts +123 -0
- package/telegram-plugin/gateway/voice-ondemand-callback-handler.ts +204 -0
- package/telegram-plugin/gateway/worker-feed-dispatch.ts +40 -0
- package/telegram-plugin/narrative-dedup.ts +24 -1
- package/telegram-plugin/narrative-flush.ts +2 -2
- package/telegram-plugin/pending-user-notice.ts +59 -13
- package/telegram-plugin/render/render.ts +25 -1
- package/telegram-plugin/status-no-truncate.ts +13 -0
- package/telegram-plugin/subagent-watcher.ts +297 -31
- package/telegram-plugin/tests/activity-card-wiring.test.ts +8 -3
- package/telegram-plugin/tests/activity-ever-opened-sticky.test.ts +18 -3
- package/telegram-plugin/tests/agent-button-callback-handler.test.ts +149 -0
- package/telegram-plugin/tests/ask-callback-handler.test.ts +118 -0
- package/telegram-plugin/tests/attachment-message-handlers.test.ts +135 -0
- package/telegram-plugin/tests/backstop-delivery.test.ts +167 -0
- package/telegram-plugin/tests/backstop-readback-probe.test.ts +144 -0
- package/telegram-plugin/tests/boot-card-routing.test.ts +139 -0
- package/telegram-plugin/tests/bot-commands-model-effort.test.ts +189 -0
- package/telegram-plugin/tests/bot-commands-start-info.test.ts +240 -0
- package/telegram-plugin/tests/buffer-gate-broadened.test.ts +28 -9
- package/telegram-plugin/tests/busy-ack-wiring.test.ts +6 -1
- package/telegram-plugin/tests/button-tap-turn-gated.test.ts +21 -12
- package/telegram-plugin/tests/callback-query-handlers.test.ts +101 -0
- package/telegram-plugin/tests/captured-answer-resume.test.ts +358 -0
- package/telegram-plugin/tests/card-tool-handlers.test.ts +497 -0
- package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +5 -2
- package/telegram-plugin/tests/checklist-message-handler.test.ts +160 -0
- package/telegram-plugin/tests/emission-authority-facade.test.ts +76 -29
- package/telegram-plugin/tests/emission-authority-ping-gate.test.ts +4 -1
- package/telegram-plugin/tests/emission-determinism-wiring.test.ts +45 -16
- package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +30 -7
- package/telegram-plugin/tests/gateway-boot-side-effect-gating.test.ts +270 -0
- package/telegram-plugin/tests/gateway-boot-smoke.test.ts +150 -0
- package/telegram-plugin/tests/gateway-bot-construction-deferral.test.ts +251 -0
- package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +5 -128
- package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +299 -0
- package/telegram-plugin/tests/gateway-loopback-paste-redact.test.ts +44 -29
- package/telegram-plugin/tests/gateway-outbound-redact.test.ts +18 -5
- package/telegram-plugin/tests/gateway-request-secret.test.ts +7 -3
- package/telegram-plugin/tests/gateway-secret-detect.test.ts +20 -10
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +8 -2
- package/telegram-plugin/tests/inbound-delivery-cutover-flip.test.ts +54 -150
- package/telegram-plugin/tests/inbound-delivery-cutover-gate.test.ts +10 -14
- package/telegram-plugin/tests/inbound-delivery-dispatch-equivalence.test.ts +6 -7
- package/telegram-plugin/tests/inbound-delivery-machine-dispatch.test.ts +0 -16
- package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +18 -7
- package/telegram-plugin/tests/inbound-message-types.test.ts +52 -16
- package/telegram-plugin/tests/litellm-proxy-auth-misconfig.test.ts +69 -14
- package/telegram-plugin/tests/media-message-handlers.test.ts +276 -0
- package/telegram-plugin/tests/mental-model-propose-callback-gate.test.ts +8 -4
- package/telegram-plugin/tests/model-command.test.ts +30 -0
- package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +36 -12
- package/telegram-plugin/tests/narrative-dedup.test.ts +32 -0
- package/telegram-plugin/tests/narrative-flush.test.ts +6 -2
- package/telegram-plugin/tests/narrative-lane-golden.test.ts +458 -0
- package/telegram-plugin/tests/no-reply-bounded-drain.test.ts +14 -3
- package/telegram-plugin/tests/obligation-ledger.test.ts +40 -0
- package/telegram-plugin/tests/obligation-store.test.ts +43 -0
- package/telegram-plugin/tests/pending-card-durability-wiring.test.ts +18 -8
- package/telegram-plugin/tests/per-topic-current-turn.test.ts +32 -8
- package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +9 -6
- package/telegram-plugin/tests/photo-message-handler.test.ts +114 -0
- package/telegram-plugin/tests/photo-reroute-wiring.test.ts +5 -2
- package/telegram-plugin/tests/pinned-message-handler.test.ts +108 -0
- package/telegram-plugin/tests/render/render.test.ts +42 -0
- package/telegram-plugin/tests/reply-terminal-reaction.test.ts +6 -2
- package/telegram-plugin/tests/secret-detect-delete-must-surface-failures.test.ts +8 -4
- package/telegram-plugin/tests/secret-detect-fail-closed.test.ts +38 -28
- package/telegram-plugin/tests/secret-detect-oauth-code.test.ts +31 -20
- package/telegram-plugin/tests/send-reply-golden.test.ts +571 -0
- package/telegram-plugin/tests/silence-liveness-wiring.test.ts +22 -8
- package/telegram-plugin/tests/status-pin-service-message-suppression.test.ts +42 -49
- package/telegram-plugin/tests/stop-command.test.ts +22 -12
- package/telegram-plugin/tests/stream-render-golden.test.ts +424 -0
- package/telegram-plugin/tests/subagent-watcher-boot-skip-dead.test.ts +218 -0
- package/telegram-plugin/tests/subagent-watcher-resume-reregister.test.ts +305 -0
- package/telegram-plugin/tests/subagent-watcher-resurrection.test.ts +32 -0
- package/telegram-plugin/tests/subagent-watcher.test.ts +35 -3
- package/telegram-plugin/tests/turn-end-gate-backstop.test.ts +8 -12
- package/telegram-plugin/tests/turn-flush-safety.test.ts +191 -11
- package/telegram-plugin/tests/turn-flush-suppression-wiring.test.ts +117 -0
- package/telegram-plugin/tests/vault-approval-posture.test.ts +8 -2
- package/telegram-plugin/tests/vault-grant-inbound-builders.test.ts +2 -2
- package/telegram-plugin/tests/vault-grant-union.test.ts +4 -1
- package/telegram-plugin/tests/vault-key-regex-allows-slash.test.ts +16 -5
- package/telegram-plugin/tests/vault-request-access-tool.test.ts +10 -5
- package/telegram-plugin/tests/vault-request-access-unlock-resume.test.ts +4 -1
- package/telegram-plugin/tests/vault-subcommands.test.ts +6 -1
- package/telegram-plugin/tests/voice-message-handler.test.ts +111 -0
- package/telegram-plugin/tests/voice-ondemand-callback-handler.test.ts +140 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +86 -19
- package/telegram-plugin/tests/worker-feed-coalesce.test.ts +236 -20
- package/telegram-plugin/tests/worker-feed-resume-guard.test.ts +86 -0
- package/telegram-plugin/tool-activity-summary.ts +110 -38
- package/telegram-plugin/turn-flush-safety.ts +80 -14
- package/telegram-plugin/uat/restart-capability.ts +76 -0
- package/telegram-plugin/uat/scenarios/bg-sub-agent-dispatch-dm.test.ts +14 -4
- package/telegram-plugin/uat/scenarios/bridge-flap-resilience-dm.test.ts +11 -1
- package/telegram-plugin/uat/scenarios/cross-turn-pending-progress-dm.test.ts +19 -2
- package/telegram-plugin/uat/scenarios/jtbd-always-on-after-restart-dm.test.ts +6 -12
- package/telegram-plugin/uat/scenarios/jtbd-deliberate-restart-resumes-dm.test.ts +6 -12
- package/telegram-plugin/uat/scenarios/jtbd-interrupted-turn-resumes-dm.test.ts +6 -12
- package/telegram-plugin/uat/scenarios/jtbd-multipart-render-dm.test.ts +47 -13
- package/telegram-plugin/worker-activity-feed.ts +34 -4
- package/telegram-plugin/gateway/busy-key-reaper.ts +0 -113
- package/telegram-plugin/gateway/gate-parity-probe.ts +0 -102
- package/telegram-plugin/tests/busy-key-reaper.test.ts +0 -192
- package/telegram-plugin/tests/fixtures/cutover-killswitch-probe.ts +0 -75
- package/telegram-plugin/tests/gate-parity-probe.test.ts +0 -171
- package/telegram-plugin/tests/parallel-turns-deadlock-fix.test.ts +0 -217
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Issue #3373 follow-up (PR #3376 adversarial review) — guards for the
|
|
3
|
+
* SendMessage-resume → feed re-surface seam.
|
|
4
|
+
*
|
|
5
|
+
* The #3376 fix has three links: the watcher fires `onResume` on jsonl growth
|
|
6
|
+
* past a genuine terminal (pinned by subagent-watcher-resume-reregister.test.ts),
|
|
7
|
+
* the gateway's `onResume` handler delegates to `handleWorkerResume`, and
|
|
8
|
+
* `handleWorkerResume` calls `feed.resurrect(agentId)` to clear the terminal
|
|
9
|
+
* `finalized` latch. The review found the middle + last links untested: deleting
|
|
10
|
+
* the gateway handler (or gutting its body) left every test green. This file
|
|
11
|
+
* closes both:
|
|
12
|
+
*
|
|
13
|
+
* 1. UNIT — `handleWorkerResume` calls `feed.resurrect` with the agentId,
|
|
14
|
+
* never throws when resurrect throws or the feed is absent, and always
|
|
15
|
+
* emits the re-surface audit line.
|
|
16
|
+
* 2. SOURCE SCAN (mirrors permission-verdict-resume-guard.test.ts) — the
|
|
17
|
+
* gateway wires an `onResume:` callback into the watcher config AND that
|
|
18
|
+
* callback delegates to `handleWorkerResume`. Deleting the handler, or
|
|
19
|
+
* replacing the delegation with a no-op body, reds this suite.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { describe, it, expect } from 'vitest'
|
|
23
|
+
import { readFileSync } from 'node:fs'
|
|
24
|
+
import { fileURLToPath } from 'node:url'
|
|
25
|
+
import { dirname, resolve } from 'node:path'
|
|
26
|
+
import { handleWorkerResume } from '../gateway/worker-feed-dispatch.js'
|
|
27
|
+
|
|
28
|
+
describe('handleWorkerResume (issue #3373 seam unit)', () => {
|
|
29
|
+
it('calls feed.resurrect with the agentId and logs the re-surface line', () => {
|
|
30
|
+
const calls: string[] = []
|
|
31
|
+
const logs: string[] = []
|
|
32
|
+
handleWorkerResume({ resurrect: (id) => calls.push(id) }, 'w1', (m) => logs.push(m))
|
|
33
|
+
expect(calls).toEqual(['w1'])
|
|
34
|
+
expect(logs.some((l) => l.includes('w1') && l.includes('RE-SURFACED'))).toBe(true)
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
it('a throwing resurrect is logged, never propagated (watcher poll loop safety)', () => {
|
|
38
|
+
const logs: string[] = []
|
|
39
|
+
expect(() =>
|
|
40
|
+
handleWorkerResume(
|
|
41
|
+
{ resurrect: () => { throw new Error('boom') } },
|
|
42
|
+
'w2',
|
|
43
|
+
(m) => logs.push(m),
|
|
44
|
+
),
|
|
45
|
+
).not.toThrow()
|
|
46
|
+
expect(logs.some((l) => l.includes('boom') && l.includes('w2'))).toBe(true)
|
|
47
|
+
// The audit line still lands after the failure.
|
|
48
|
+
expect(logs.some((l) => l.includes('RE-SURFACED'))).toBe(true)
|
|
49
|
+
})
|
|
50
|
+
|
|
51
|
+
it('a null/undefined feed (feed disabled) is a safe no-op that still audits', () => {
|
|
52
|
+
const logs: string[] = []
|
|
53
|
+
expect(() => handleWorkerResume(null, 'w3', (m) => logs.push(m))).not.toThrow()
|
|
54
|
+
expect(logs.some((l) => l.includes('w3') && l.includes('RE-SURFACED'))).toBe(true)
|
|
55
|
+
})
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
describe('gateway onResume wiring (issue #3373 source-scan guard)', () => {
|
|
59
|
+
const __dirname = dirname(fileURLToPath(import.meta.url))
|
|
60
|
+
const GATEWAY_SRC = readFileSync(
|
|
61
|
+
resolve(__dirname, '..', 'gateway', 'gateway.ts'),
|
|
62
|
+
'utf8',
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
it('the watcher config carries an onResume callback', () => {
|
|
66
|
+
expect(/\bonResume:\s*\(/.test(GATEWAY_SRC)).toBe(true)
|
|
67
|
+
})
|
|
68
|
+
|
|
69
|
+
it('the onResume callback delegates to handleWorkerResume (not an inline no-op)', () => {
|
|
70
|
+
// Match the `onResume: (…) => { … }` callback body and require the
|
|
71
|
+
// delegation call inside it. A gutted handler (empty body / dropped
|
|
72
|
+
// resurrect) fails here even though tsc stays green.
|
|
73
|
+
const m = GATEWAY_SRC.match(/\bonResume:\s*\([^)]*\)\s*=>\s*\{([\s\S]*?)\n\s{14}\},/)
|
|
74
|
+
expect(m, 'onResume callback not found in gateway.ts').not.toBeNull()
|
|
75
|
+
expect(m![1]).toMatch(/\bhandleWorkerResume\s*\(/)
|
|
76
|
+
expect(m![1]).toMatch(/\bworkerActivityFeed\b/)
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
it('handleWorkerResume is imported from worker-feed-dispatch (the tested seam)', () => {
|
|
80
|
+
expect(
|
|
81
|
+
/import\s*\{[^}]*\bhandleWorkerResume\b[^}]*\}\s*from\s*'\.\/worker-feed-dispatch\.js'/.test(
|
|
82
|
+
GATEWAY_SRC,
|
|
83
|
+
),
|
|
84
|
+
).toBe(true)
|
|
85
|
+
})
|
|
86
|
+
})
|
|
@@ -86,6 +86,7 @@ export function describeToolUse(
|
|
|
86
86
|
import {
|
|
87
87
|
STATUS_CARD_CHAR_BUDGET,
|
|
88
88
|
STATUS_ROLLING_LINES,
|
|
89
|
+
WORKER_HISTORY_MAX,
|
|
89
90
|
STATUS_LINE_MAX,
|
|
90
91
|
NESTED_PREFIX,
|
|
91
92
|
} from './status-no-truncate.js'
|
|
@@ -286,7 +287,8 @@ function escapeStepLine(raw: string): string {
|
|
|
286
287
|
|
|
287
288
|
/**
|
|
288
289
|
* Shared step-feed emitter. Appends `✓`/`→` bullet lines to `out` for the
|
|
289
|
-
* given ALREADY-ESCAPED step strings, windowing to
|
|
290
|
+
* given ALREADY-ESCAPED step strings, windowing to `window` (default
|
|
291
|
+
* STATUS_ROLLING_LINES; the worker surfaces pass a deeper window) and
|
|
290
292
|
* prepending a `+N earlier…` header when the feed overflows the window (on
|
|
291
293
|
* BOTH surfaces now). The worker feed imports this directly.
|
|
292
294
|
*
|
|
@@ -301,9 +303,10 @@ export function renderStepFeed(
|
|
|
301
303
|
steps: string[],
|
|
302
304
|
allDone: boolean,
|
|
303
305
|
liveSuffix = '',
|
|
306
|
+
window: number = STATUS_ROLLING_LINES,
|
|
304
307
|
): void {
|
|
305
308
|
if (steps.length === 0) return
|
|
306
|
-
const shown = steps.slice(-
|
|
309
|
+
const shown = steps.slice(-Math.max(1, window))
|
|
307
310
|
const hidden = steps.length - shown.length
|
|
308
311
|
if (hidden > 0) out.push(`_✓ +${hidden} earlier…_`)
|
|
309
312
|
const lastIdx = shown.length - 1
|
|
@@ -351,6 +354,13 @@ export interface StatusCardOpts {
|
|
|
351
354
|
stepCount?: number
|
|
352
355
|
/** Optional terminal result block (worker recap), already-cleaned text + emoji. */
|
|
353
356
|
result?: { emoji: string; text: string }
|
|
357
|
+
/**
|
|
358
|
+
* How many trailing step/child lines the rolling window shows. Defaults to
|
|
359
|
+
* STATUS_ROLLING_LINES (the 🤖 agent card). The 🛠 single-worker card passes
|
|
360
|
+
* a deeper window (`workerHistoryDepth(1)` = 6) so a lone worker can show its
|
|
361
|
+
* full recent trail — see WORKER_HISTORY_MAX.
|
|
362
|
+
*/
|
|
363
|
+
historyWindow?: number
|
|
354
364
|
}
|
|
355
365
|
|
|
356
366
|
/**
|
|
@@ -364,6 +374,7 @@ export interface StatusCardOpts {
|
|
|
364
374
|
*/
|
|
365
375
|
export function renderStatusCard(opts: StatusCardOpts): string | null {
|
|
366
376
|
const { header, final = false, liveSuffix = '', stepCount, result } = opts
|
|
377
|
+
const window = Math.max(1, opts.historyWindow ?? STATUS_ROLLING_LINES)
|
|
367
378
|
const rawSteps = opts.steps.filter((s) => s != null)
|
|
368
379
|
const rawChildren = (opts.childSteps ?? []).map((s) => s.trim()).filter((s) => s.length > 0)
|
|
369
380
|
const hasChildren = rawChildren.length > 0
|
|
@@ -393,12 +404,12 @@ export function renderStatusCard(opts: StatusCardOpts): string | null {
|
|
|
393
404
|
|
|
394
405
|
if (hasChildren) {
|
|
395
406
|
// Parent lines all render done — the live → step lives in the nested block.
|
|
396
|
-
const shownParent = steps.slice(-
|
|
407
|
+
const shownParent = steps.slice(-window)
|
|
397
408
|
const hiddenParent = steps.length - shownParent.length
|
|
398
409
|
if (hiddenParent > 0) out.push(`_✓ +${hiddenParent} earlier…_`)
|
|
399
410
|
for (const s of shownParent) out.push(`~~_✓ ${s}_~~`)
|
|
400
411
|
// Child block.
|
|
401
|
-
const shownChild = children.slice(-
|
|
412
|
+
const shownChild = children.slice(-window)
|
|
402
413
|
const hiddenChild = children.length - shownChild.length
|
|
403
414
|
if (hiddenChild > 0) out.push(`${NESTED_PREFIX}_+${hiddenChild} earlier…_`)
|
|
404
415
|
const lastChildIdx = shownChild.length - 1
|
|
@@ -410,7 +421,7 @@ export function renderStatusCard(opts: StatusCardOpts): string | null {
|
|
|
410
421
|
)
|
|
411
422
|
})
|
|
412
423
|
} else {
|
|
413
|
-
renderStepFeed(out, steps, final, liveSuffix)
|
|
424
|
+
renderStepFeed(out, steps, final, liveSuffix, window)
|
|
414
425
|
}
|
|
415
426
|
|
|
416
427
|
if (final && stepCount != null && stepCount > 0) {
|
|
@@ -646,54 +657,96 @@ export interface CombinedWorkerRow {
|
|
|
646
657
|
/** Running total tokens for this worker — rendered as `· {N} tok` on the row
|
|
647
658
|
* header. Omitted (0/undefined) → no token segment. */
|
|
648
659
|
totalTokens?: number
|
|
660
|
+
/**
|
|
661
|
+
* Stable per-card ordinal (1-based), assigned when the worker joins its feed
|
|
662
|
+
* group and KEPT for the card's lifetime — survivors keep their numbers when
|
|
663
|
+
* an earlier worker finishes (the card may show `2.`/`3.` with no `1.`; the
|
|
664
|
+
* `N running` chrome carries the count). Rendered as a `{ordinal}. ` prefix
|
|
665
|
+
* inside the bold header when the card has 2+ rows. Omitted → unnumbered
|
|
666
|
+
* (back-compat for direct callers).
|
|
667
|
+
*/
|
|
668
|
+
ordinal?: number
|
|
649
669
|
}
|
|
650
670
|
|
|
651
671
|
export interface CombinedWorkerFeedOpts {
|
|
652
672
|
/** Max worker rows rendered before the `+M more working…` spill line.
|
|
653
|
-
*
|
|
654
|
-
*
|
|
673
|
+
* Rows are kept head-first (`rows.slice(0, visibleCount)`): the earliest
|
|
674
|
+
* supplied (oldest dispatch-order) workers stay visible and the
|
|
675
|
+
* newest/trailing rows spill. */
|
|
655
676
|
maxRows: number
|
|
656
677
|
}
|
|
657
678
|
|
|
679
|
+
/** Each visible worker costs one header line before any history. */
|
|
680
|
+
const PER_WORKER_HEADER_COST = 1
|
|
681
|
+
|
|
682
|
+
/**
|
|
683
|
+
* Floor of the per-worker depth curve — every SHOWN worker renders at least
|
|
684
|
+
* this many recent steps, no matter how large the fan-out (Ken's "4+ → 3 each").
|
|
685
|
+
*/
|
|
686
|
+
const MIN_WORKER_DEPTH = 3
|
|
687
|
+
|
|
688
|
+
/**
|
|
689
|
+
* Design fan-out: the largest concurrent-worker count whose full Ken curve is
|
|
690
|
+
* allowed to render before the total-line backstop starts collapsing the
|
|
691
|
+
* NEWEST (trailing) rows into `+M more working…`. Chosen at 6 — beyond six live workers a
|
|
692
|
+
* per-worker trail is no longer a glanceable card, so extra rows spill rather
|
|
693
|
+
* than every shown worker losing depth. Every worker that IS shown keeps its
|
|
694
|
+
* full curve depth; the ceiling trims row COUNT, never per-worker depth.
|
|
695
|
+
*/
|
|
696
|
+
const DESIGN_FANOUT = 6
|
|
697
|
+
|
|
658
698
|
/**
|
|
659
699
|
* Total per-worker BODY line budget for the combined feed — the sum, across all
|
|
660
700
|
* visible workers, of (one header line + that worker's history lines). The top
|
|
661
701
|
* `🛠 Workers · N running` line and the `+M more working…` spill are OUTSIDE
|
|
662
|
-
* this budget (fixed chrome).
|
|
663
|
-
*
|
|
664
|
-
*
|
|
665
|
-
*
|
|
666
|
-
*
|
|
702
|
+
* this budget (fixed chrome).
|
|
703
|
+
*
|
|
704
|
+
* Re-derived (#3349) to fit Ken's per-worker curve `max(3, 7 − w)` rather than
|
|
705
|
+
* to DRIVE the depth: at the flat tail (w ≥ 4, depth = MIN_WORKER_DEPTH) each
|
|
706
|
+
* worker costs `PER_WORKER_HEADER_COST + MIN_WORKER_DEPTH` = 4 lines, so a
|
|
707
|
+
* DESIGN_FANOUT of 6 fully-rendered workers needs 6 × 4 = 24 lines. That fully
|
|
708
|
+
* fits every fan-out through 6 workers (w1=7, w2=12, w3=15, w4=16, w5=20,
|
|
709
|
+
* w6=24); a 7th/8th concurrent worker overflows and the backstop drops the
|
|
710
|
+
* newest (trailing) visible rows to the spill — bounding the card at 24 body lines so a
|
|
711
|
+
* big swarm never explodes it. The per-worker curve wins on DEPTH; this ceiling
|
|
712
|
+
* wins on ROW COUNT.
|
|
667
713
|
*/
|
|
668
|
-
const MAX_COMBINED_BODY_LINES =
|
|
669
|
-
/** Each visible worker costs one header line before any history. */
|
|
670
|
-
const PER_WORKER_HEADER_COST = 1
|
|
714
|
+
const MAX_COMBINED_BODY_LINES = DESIGN_FANOUT * (PER_WORKER_HEADER_COST + MIN_WORKER_DEPTH)
|
|
671
715
|
|
|
672
716
|
/**
|
|
673
717
|
* Deterministic per-worker history depth for `w` visible workers:
|
|
674
|
-
* clamp(
|
|
675
|
-
* So
|
|
676
|
-
*
|
|
677
|
-
*
|
|
718
|
+
* clamp( max(MIN_WORKER_DEPTH, 7 − w), 1, WORKER_HISTORY_MAX )
|
|
719
|
+
* So 1 worker → 6, 2 → 5, 3 → 4, 4 → 3, ≥4 → 3 (Ken's 6/5/4/3 curve, #3349).
|
|
720
|
+
* Pure function of the visible worker count — no model input, consistent with
|
|
721
|
+
* deterministic controls. Replaces the former budget-driven divide (which
|
|
722
|
+
* yielded 5/5/3/2/… and could never reach 6 for a lone worker).
|
|
678
723
|
*/
|
|
679
724
|
export function combinedHistoryDepth(w: number): number {
|
|
680
725
|
if (w <= 0) return 1
|
|
681
|
-
|
|
682
|
-
return Math.max(1, Math.min(STATUS_ROLLING_LINES, raw))
|
|
726
|
+
return Math.min(WORKER_HISTORY_MAX, Math.max(MIN_WORKER_DEPTH, 7 - w))
|
|
683
727
|
}
|
|
684
728
|
|
|
729
|
+
/** Alias for readability at the single-worker call site — same curve. */
|
|
730
|
+
export const workerHistoryDepth = combinedHistoryDepth
|
|
731
|
+
|
|
685
732
|
/**
|
|
686
733
|
* Render N≥1 live workers into ONE combined feed body (ready Telegram
|
|
687
734
|
* markdown; callers send verbatim — do NOT re-escape). Layout:
|
|
688
735
|
*
|
|
689
736
|
* 🛠 **Workers** · _N running_
|
|
690
|
-
* **{desc1}** _· {elapsed} · {n} tools_
|
|
737
|
+
* **1. {desc1}** _· {elapsed} · {n} tools_
|
|
691
738
|
* ~~_✓ {earlier step}_~~
|
|
692
739
|
* **→ {newest step}**
|
|
693
|
-
* **{desc2}** _· {elapsed} · {n} tools_
|
|
740
|
+
* **2. {desc2}** _· {elapsed} · {n} tools_
|
|
694
741
|
* **→ {newest step}**
|
|
695
742
|
* _+M more working…_
|
|
696
743
|
*
|
|
744
|
+
* NUMBERING (#3298): when the card tracks 2+ rows AND a row carries `ordinal`,
|
|
745
|
+
* its header gets a stable `{ordinal}. ` prefix. Ordinals are assigned by the
|
|
746
|
+
* caller at dispatch and kept for the card's life — after an earlier worker
|
|
747
|
+
* finishes the survivors keep their numbers (`2.`, `3.` with no `1.`). A lone
|
|
748
|
+
* row, or rows without ordinals, render unnumbered.
|
|
749
|
+
*
|
|
697
750
|
* ADAPTIVE DENSITY: each visible worker renders its last-K narrative lines as a
|
|
698
751
|
* `✓`/`→` trail (prior steps struck, newest bold in-progress) — the single-
|
|
699
752
|
* worker card's idiom — where K = `combinedHistoryDepth(visibleCount)` splits a
|
|
@@ -705,7 +758,7 @@ export function combinedHistoryDepth(w: number): number {
|
|
|
705
758
|
* Pure. Rows are rendered in the order supplied (the manager passes them
|
|
706
759
|
* dispatch-order, oldest first). `maxRows` caps the visible rows; the hidden
|
|
707
760
|
* remainder collapses to a single `+M more working…` line. A total-budget
|
|
708
|
-
* backstop drops the
|
|
761
|
+
* backstop drops the NEWEST (trailing) visible rows one at a time (growing the spill)
|
|
709
762
|
* until the body fits STATUS_CARD_CHAR_BUDGET, so a burst of long descriptions
|
|
710
763
|
* can never overflow the wire limit. Returns null only when `rows` is empty.
|
|
711
764
|
*/
|
|
@@ -716,6 +769,11 @@ export function renderCombinedWorkerFeed(
|
|
|
716
769
|
if (rows.length === 0) return null
|
|
717
770
|
const maxRows = Math.max(1, Math.floor(opts.maxRows))
|
|
718
771
|
|
|
772
|
+
// Number the workers only when the CARD tracks 2+ (a lone worker stays
|
|
773
|
+
// unnumbered). Uses the total row count, not the visible count, so ordinals
|
|
774
|
+
// don't appear/vanish as the overflow backstop shrinks the visible set.
|
|
775
|
+
const numbered = rows.length >= 2
|
|
776
|
+
|
|
719
777
|
const rowHeader = (r: CombinedWorkerRow): string => {
|
|
720
778
|
const desc = escapeMarkdown(
|
|
721
779
|
truncate(stripMarkdown(r.description).replace(/\s+/g, ' ').trim() || 'background task', COMBINED_ROW_DESC_MAX),
|
|
@@ -724,7 +782,11 @@ export function renderCombinedWorkerFeed(
|
|
|
724
782
|
const tokPart = tokenSegment(r.totalTokens)
|
|
725
783
|
const modelLabel = formatModelLabel(r.model)
|
|
726
784
|
const modelPart = modelLabel != null ? ` · ${escapeMarkdown(modelLabel)}` : ''
|
|
727
|
-
|
|
785
|
+
// Stable ordinal prefix INSIDE the bold span, before the already-escaped
|
|
786
|
+
// description — no new escaping surface, and the gateway md→HTML conversion
|
|
787
|
+
// has no ordered-list auto-formatting on bolded text.
|
|
788
|
+
const num = numbered && r.ordinal != null ? `${r.ordinal}. ` : ''
|
|
789
|
+
return `**${num}${desc}** _· ${formatFeedElapsed(r.elapsedMs)} · ${r.toolCount} ${toolWord}${tokPart}${modelPart}_`
|
|
728
790
|
}
|
|
729
791
|
|
|
730
792
|
// Raw (unescaped) history for a worker, oldest→newest, empty lines stripped.
|
|
@@ -734,37 +796,47 @@ export function renderCombinedWorkerFeed(
|
|
|
734
796
|
return src.filter((s) => s != null && stripMarkdown(s).replace(/\s+/g, ' ').trim().length > 0)
|
|
735
797
|
}
|
|
736
798
|
|
|
737
|
-
const compose = (visibleCount: number): string => {
|
|
799
|
+
const compose = (visibleCount: number): { body: string; bodyLines: number } => {
|
|
738
800
|
const shown = rows.slice(0, visibleCount)
|
|
739
801
|
const hidden = rows.length - shown.length
|
|
740
|
-
//
|
|
741
|
-
//
|
|
802
|
+
// Per-worker depth follows Ken's deterministic curve max(3, 7−w) (#3349):
|
|
803
|
+
// the curve drives DEPTH; the total-line budget below drives ROW COUNT.
|
|
742
804
|
const depth = combinedHistoryDepth(shown.length)
|
|
743
|
-
const
|
|
805
|
+
const chrome: string[] = [`🛠 **Workers** · _${rows.length} running_`]
|
|
806
|
+
const bodyOut: string[] = []
|
|
744
807
|
for (const r of shown) {
|
|
745
|
-
|
|
808
|
+
bodyOut.push(rowHeader(r))
|
|
746
809
|
const hist = rowHistory(r)
|
|
747
810
|
if (hist.length === 0) {
|
|
748
|
-
|
|
811
|
+
bodyOut.push('→ _starting…_')
|
|
749
812
|
continue
|
|
750
813
|
}
|
|
751
814
|
// Paint the last-K history lines with the SAME `✓`/`→` idiom as the
|
|
752
815
|
// single-worker card: escape each raw line through the shared per-line
|
|
753
816
|
// pipeline (escapeStepLine), then renderStepFeed strikes the prior steps
|
|
754
|
-
// and bolds the newest in-progress step.
|
|
817
|
+
// and bolds the newest in-progress step. The window equals the depth so a
|
|
818
|
+
// per-worker `+N earlier…` marker never appears inside the combined feed.
|
|
755
819
|
const esc = hist.slice(-depth).map(escapeStepLine)
|
|
756
|
-
renderStepFeed(
|
|
820
|
+
renderStepFeed(bodyOut, esc, false, '', depth)
|
|
757
821
|
}
|
|
822
|
+
const out = [...chrome, ...bodyOut]
|
|
758
823
|
if (hidden > 0) out.push(`_+${hidden} more working…_`)
|
|
759
|
-
return stackCardLines(out)
|
|
824
|
+
return { body: stackCardLines(out), bodyLines: bodyOut.length }
|
|
760
825
|
}
|
|
761
826
|
|
|
762
|
-
// Cap to maxRows first, then shrink
|
|
827
|
+
// Cap to maxRows first, then shrink the visible set while EITHER the total
|
|
828
|
+
// body-line budget (#3349: bounds a big swarm without stealing depth from the
|
|
829
|
+
// shown workers) OR the wire char budget is exceeded. Newest (trailing) rows
|
|
830
|
+
// collapse into the `+M more working…` spill (`rows.slice(0, visibleCount)`
|
|
831
|
+
// keeps the head of the list).
|
|
763
832
|
let visible = Math.min(rows.length, maxRows)
|
|
764
|
-
let body = compose(visible)
|
|
765
|
-
while (
|
|
833
|
+
let { body, bodyLines } = compose(visible)
|
|
834
|
+
while (
|
|
835
|
+
(bodyLines > MAX_COMBINED_BODY_LINES || body.length > STATUS_CARD_CHAR_BUDGET) &&
|
|
836
|
+
visible > 1
|
|
837
|
+
) {
|
|
766
838
|
visible -= 1
|
|
767
|
-
body = compose(visible)
|
|
839
|
+
;({ body, bodyLines } = compose(visible))
|
|
768
840
|
}
|
|
769
841
|
return body
|
|
770
842
|
}
|
|
@@ -155,28 +155,80 @@ export const FLUSH_SUBSTANTIVE_MIN_CHARS = 200
|
|
|
155
155
|
* `[verboseNarration(250), realAnswer(150)]` a reversed length scan returns the
|
|
156
156
|
* 250-char narration and DROPS the 150-char real answer. Instead we take the
|
|
157
157
|
* last non-empty block as the answer and strip only the EARLIER blocks — and
|
|
158
|
-
* only when they look like intent-narration
|
|
159
|
-
*
|
|
160
|
-
*
|
|
161
|
-
*
|
|
158
|
+
* only when they look like intent-narration. If the earlier blocks are
|
|
159
|
+
* themselves substantial (a genuine multi-paragraph answer written as several
|
|
160
|
+
* blocks) we keep the whole thing joined, so we never truncate a real long
|
|
161
|
+
* answer down to its last paragraph.
|
|
162
|
+
*
|
|
163
|
+
* #3237 — the STRUCTURAL discriminator. The opener heuristic below
|
|
164
|
+
* (`isNarrationBlock`) cannot tell a narration preamble ("Let me pull the
|
|
165
|
+
* numbers…" followed by a separate reply) from a real answer paragraph that
|
|
166
|
+
* merely OPENS with "Let me explain…": both match the same regex, and length
|
|
167
|
+
* cannot separate them (the observed narration was itself ≥200 chars). The one
|
|
168
|
+
* signal that DOES separate them is structural — did a `tool_use` follow this
|
|
169
|
+
* block in the model's actual message? A narration preamble is drafted, then
|
|
170
|
+
* the model ACTS (a tool call follows it); a terminal answer paragraph is not
|
|
171
|
+
* followed by any tool call. That per-block flag (`lastInMessage`) is computed
|
|
172
|
+
* upstream by `projectAssistantTextBlocks` (session-tail.ts) and, when the
|
|
173
|
+
* caller plumbs it through as `followedByToolUse`, we consult STRUCTURE, but
|
|
174
|
+
* asymmetrically:
|
|
175
|
+
* - present-and-FALSE (no tool_use followed) is purely additive — it can only
|
|
176
|
+
* RESCUE a block the opener regex would have mis-stripped (a terminal answer
|
|
177
|
+
* opening "Let me explain…"); it never drops a block the heuristic kept.
|
|
178
|
+
* - present-and-TRUE (a tool_use followed) is NOT taken as narration on its
|
|
179
|
+
* own: a substantial real-content paragraph can precede a tool call, so the
|
|
180
|
+
* strip is GATED by the substance check (narration only if it also matches
|
|
181
|
+
* the opener/trailer heuristic OR falls below the substantive floor). This
|
|
182
|
+
* is deliberately not "strictly additive" over the opener-only strip — it
|
|
183
|
+
* both rescues real content the old flag-alone path would have dropped and
|
|
184
|
+
* stays truncation-free.
|
|
185
|
+
* Where the structural flag is ABSENT for a block (legacy `string[]` caller, or
|
|
186
|
+
* the accumulator lost provenance) we fall back to the opener/trailer heuristic
|
|
187
|
+
* for that block.
|
|
188
|
+
*
|
|
189
|
+
* `followedByToolUse` is a parallel array aligned to `blocks` (index `i` ⇒
|
|
190
|
+
* `blocks[i]`); it is zipped BEFORE the empty-block filter so alignment holds
|
|
191
|
+
* even if the caller passes empty/whitespace blocks. `undefined` at an index
|
|
192
|
+
* (or a missing/short array) means "no reliable structural signal for this
|
|
193
|
+
* block" → opener-heuristic fallback.
|
|
162
194
|
*
|
|
163
195
|
* `blocks` are already trimmed/non-empty candidates (silent markers removed by
|
|
164
196
|
* the caller's guards). Returns the chosen delivery text.
|
|
165
197
|
*/
|
|
166
|
-
export function selectFlushDeliveryText(
|
|
198
|
+
export function selectFlushDeliveryText(
|
|
199
|
+
blocks: string[],
|
|
200
|
+
followedByToolUse?: ReadonlyArray<boolean | undefined>,
|
|
201
|
+
): string {
|
|
167
202
|
const candidates = blocks
|
|
168
|
-
.map(b => b.trim())
|
|
169
|
-
.filter(
|
|
203
|
+
.map((b, i) => ({ text: b.trim(), followedByToolUse: followedByToolUse?.[i] }))
|
|
204
|
+
.filter(c => c.text.length > 0)
|
|
170
205
|
if (candidates.length === 0) return ''
|
|
171
|
-
if (candidates.length === 1) return candidates[0]
|
|
172
|
-
const answer = candidates[candidates.length - 1]
|
|
206
|
+
if (candidates.length === 1) return candidates[0].text
|
|
207
|
+
const answer = candidates[candidates.length - 1].text
|
|
173
208
|
const preceding = candidates.slice(0, -1)
|
|
174
209
|
// Deliver only the terminal answer when every earlier block is
|
|
175
|
-
// intent-narration
|
|
210
|
+
// intent-narration. Per block:
|
|
211
|
+
// - structural flag TRUE ⇒ a tool_use followed this block, but that alone
|
|
212
|
+
// is NOT sufficient to drop it: a substantial real-content paragraph can
|
|
213
|
+
// legitimately precede a tool call (the model writes a real answer, then
|
|
214
|
+
// calls a memory/verify tool, then a short wrap-up). So gate the
|
|
215
|
+
// structural strip with the substantive floor — narration only if the
|
|
216
|
+
// block ALSO looks like narration OR is below the substantive floor.
|
|
217
|
+
// - structural flag FALSE ⇒ present-and-false definitively overrides the
|
|
218
|
+
// opener regex: no tool_use followed, so it is a terminal-style block,
|
|
219
|
+
// never narration (#3237).
|
|
220
|
+
// - structural flag ABSENT ⇒ fall back to the opener/trailer heuristic.
|
|
176
221
|
// Otherwise the earlier blocks carry real content — keep the full joined text
|
|
177
|
-
// so a legitimate multi-block answer is never truncated to its last
|
|
178
|
-
|
|
179
|
-
|
|
222
|
+
// so a legitimate multi-block answer is never truncated to its last
|
|
223
|
+
// paragraph (#3237).
|
|
224
|
+
const allNarration = preceding.every(c =>
|
|
225
|
+
c.followedByToolUse === true
|
|
226
|
+
? isNarrationBlock(c.text) || c.text.trim().length < FLUSH_SUBSTANTIVE_MIN_CHARS
|
|
227
|
+
: c.followedByToolUse === false
|
|
228
|
+
? false
|
|
229
|
+
: isNarrationBlock(c.text),
|
|
230
|
+
)
|
|
231
|
+
return allNarration ? answer : candidates.map(c => c.text).join('\n\n')
|
|
180
232
|
}
|
|
181
233
|
|
|
182
234
|
/**
|
|
@@ -244,6 +296,17 @@ export interface FlushDecisionInput {
|
|
|
244
296
|
* is false — once the model has called reply / stream_reply the turn is
|
|
245
297
|
* served and trailing terminal text is dropped (see `decideTurnFlush`). */
|
|
246
298
|
capturedText: string[]
|
|
299
|
+
/** Optional per-block structural provenance, aligned to `capturedText`
|
|
300
|
+
* (index `i` describes `capturedText[i]`). `true` ⇒ a `tool_use` followed
|
|
301
|
+
* this text block in its assistant message (the draft-then-send narration
|
|
302
|
+
* signal — the negation of `projectAssistantTextBlocks`' `lastInMessage`).
|
|
303
|
+
* Consumed by `selectFlushDeliveryText` to separate a narration preamble from
|
|
304
|
+
* a real answer paragraph that merely opens with a narration phrase (#3237).
|
|
305
|
+
* When absent (legacy caller / lost provenance) the strip falls back to the
|
|
306
|
+
* opener/trailer heuristic. Present-and-false is additive (only rescues a
|
|
307
|
+
* mis-stripped terminal answer); present-and-true is gated by the substantive
|
|
308
|
+
* floor rather than trusted alone. */
|
|
309
|
+
capturedBlockMeta?: boolean[]
|
|
247
310
|
/** Feature flag — defaults to true. Pass `false` to force skip everywhere. */
|
|
248
311
|
flushEnabled?: boolean
|
|
249
312
|
}
|
|
@@ -316,7 +379,10 @@ export function decideTurnFlush(input: FlushDecisionInput): FlushDecision {
|
|
|
316
379
|
// blob (see `selectFlushDeliveryText`). The silent-marker / empty guards above
|
|
317
380
|
// still run on the full `joined` string so a partly-silent turn is classified
|
|
318
381
|
// correctly; only the DELIVERED text is narrowed to the answer.
|
|
319
|
-
return {
|
|
382
|
+
return {
|
|
383
|
+
kind: 'flush',
|
|
384
|
+
text: selectFlushDeliveryText(input.capturedText, input.capturedBlockMeta),
|
|
385
|
+
}
|
|
320
386
|
}
|
|
321
387
|
|
|
322
388
|
/**
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared restart-capability probe + loud-skip announcer for the
|
|
3
|
+
* restart/resume UAT scenarios.
|
|
4
|
+
*
|
|
5
|
+
* These scenarios (`jtbd-deliberate-restart-resumes-dm`,
|
|
6
|
+
* `jtbd-interrupted-turn-resumes-dm`, `jtbd-always-on-after-restart-dm`)
|
|
7
|
+
* can only exercise a REAL restart when the runner has NOPASSWD `sudo` and
|
|
8
|
+
* the `switchroom` CLI on PATH. On a sandboxed runner that lacks those,
|
|
9
|
+
* `vitest`'s `describe.skip` marks them skipped — but a bare skip reads as
|
|
10
|
+
* a plain green in a summarised board, which is exactly how the v0.18.32
|
|
11
|
+
* UAT canary over-claimed "#3315 restart/resume validated" when in fact
|
|
12
|
+
* these three self-skipped (#3334 item d).
|
|
13
|
+
*
|
|
14
|
+
* The durable fix has two parts. Part 2 (a runner that can actually
|
|
15
|
+
* restart) is infrastructure and is tracked separately. Part 1 — this file
|
|
16
|
+
* — makes the skip LOUD and unmistakable: a single-source-of-truth probe
|
|
17
|
+
* plus a banner printed to stderr at collection time, so a green that is
|
|
18
|
+
* really a skip can never be silently indistinguishable from a green that
|
|
19
|
+
* ran live.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { execSync } from "node:child_process";
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* True only when the runner can drive a real agent restart: NOPASSWD
|
|
26
|
+
* `sudo` is available (the scenarios shell out to
|
|
27
|
+
* `sudo -n switchroom agent restart …`). A sandboxed CI runner returns
|
|
28
|
+
* false, which routes the scenario to `describe.skip`.
|
|
29
|
+
*/
|
|
30
|
+
export function canShellSudo(): boolean {
|
|
31
|
+
try {
|
|
32
|
+
execSync("sudo -n true", { stdio: "ignore", timeout: 2_000 });
|
|
33
|
+
return true;
|
|
34
|
+
} catch {
|
|
35
|
+
return false;
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Print a loud, unmissable banner to stderr when a restart/resume scenario
|
|
41
|
+
* is about to self-skip because the runner cannot exercise a real restart.
|
|
42
|
+
* Call this at module-collection time (top level of the scenario file) so
|
|
43
|
+
* the banner lands in CI stdout/stderr regardless of the reporter's
|
|
44
|
+
* skip-count rendering.
|
|
45
|
+
*
|
|
46
|
+
* The banner deliberately spells out that a GREEN here is a SKIP, not live
|
|
47
|
+
* proof — the exact confusion #3334 item d flags as "the dangerous one".
|
|
48
|
+
*/
|
|
49
|
+
export function announceRestartSkip(scenarioTitle: string): void {
|
|
50
|
+
console.warn(
|
|
51
|
+
"\n" +
|
|
52
|
+
"════════════════════════════════════════════════════════════════════\n" +
|
|
53
|
+
" ⚠️ UAT SKIPPED — NOT LIVE-VALIDATED (restart capability absent)\n" +
|
|
54
|
+
` scenario: ${scenarioTitle}\n` +
|
|
55
|
+
" reason: this runner has no NOPASSWD sudo + switchroom CLI, so\n" +
|
|
56
|
+
" it cannot exercise a real agent restart. The scenario\n" +
|
|
57
|
+
" self-skips. Its GREEN is a SKIP, not live proof.\n" +
|
|
58
|
+
" see: #3334 item (d) / #3315 — restart/resume assurance from\n" +
|
|
59
|
+
" this run rests on the unit/vitest layer, not this UAT.\n" +
|
|
60
|
+
"════════════════════════════════════════════════════════════════════\n",
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Convenience: probe capability once and, when absent, emit the loud-skip
|
|
66
|
+
* banner. Returns the capability boolean so the caller can gate
|
|
67
|
+
* `describe` vs `describe.skip`:
|
|
68
|
+
*
|
|
69
|
+
* const sudoOk = restartCapableOrAnnounceSkip(TITLE);
|
|
70
|
+
* (sudoOk ? describe : describe.skip)(TITLE, () => { … });
|
|
71
|
+
*/
|
|
72
|
+
export function restartCapableOrAnnounceSkip(scenarioTitle: string): boolean {
|
|
73
|
+
const ok = canShellSudo();
|
|
74
|
+
if (!ok) announceRestartSkip(scenarioTitle);
|
|
75
|
+
return ok;
|
|
76
|
+
}
|
|
@@ -89,7 +89,17 @@ const BG_DISPATCH_PROMPT =
|
|
|
89
89
|
`brief reply saying you've kicked off the background worker so I can ` +
|
|
90
90
|
`watch the progress feed.`;
|
|
91
91
|
|
|
92
|
-
|
|
92
|
+
// Single-worker RUNNING headers show NO literal "running" — they render
|
|
93
|
+
// `<elapsed> · <n> tools[ · <tok> tok][ · <model>]`
|
|
94
|
+
// (`tool-activity-summary.ts` renderActivityHeader, running branch); the
|
|
95
|
+
// word "running" only appears on the MULTI-worker combined header
|
|
96
|
+
// (`🛠 Workers · N running`). The old `/running\s*·/i` oracle matched only
|
|
97
|
+
// by accident when the multi-worker header happened to render (failed live
|
|
98
|
+
// 2026-07-18, ci-uat run 29634400115, against a healthy in-flight card).
|
|
99
|
+
// In-flight signal = either header shape's live metric, paired with the
|
|
100
|
+
// `not.toMatch(WORKER_DONE_RE)` terminal exclusion below — a skeleton that
|
|
101
|
+
// paints only `🛠 Worker · <name>` with no metric line still fails.
|
|
102
|
+
const WORKER_RUNNING_RE = /\brunning\b|·\s*\d+\s+tools?\b/i;
|
|
93
103
|
const WORKER_DONE_RE = /finished\s*·\s*(completed|failed)/i;
|
|
94
104
|
|
|
95
105
|
describe("uat: background sub-agent visibility (#709/#776/#782/#788)", () => {
|
|
@@ -117,9 +127,9 @@ describe("uat: background sub-agent visibility (#709/#776/#782/#788)", () => {
|
|
|
117
127
|
expect(feed.messageId).toBeGreaterThan(0);
|
|
118
128
|
expect(feed.text).toMatch(WORKER_FEED_RE);
|
|
119
129
|
|
|
120
|
-
// AC-2 step 1: feed body MUST show
|
|
121
|
-
//
|
|
122
|
-
// completed yet.
|
|
130
|
+
// AC-2 step 1: feed body MUST show an in-flight signal (live
|
|
131
|
+
// metric line, or multi-worker "N running"), NOT the terminal
|
|
132
|
+
// "finished ·" — the worker hasn't completed yet.
|
|
123
133
|
expect(feed.text).toMatch(WORKER_RUNNING_RE);
|
|
124
134
|
expect(feed.text).not.toMatch(WORKER_DONE_RE);
|
|
125
135
|
|
|
@@ -123,8 +123,18 @@ describe("uat: bridge-flap resilience — agent stays responsive, gateway does n
|
|
|
123
123
|
`overall deadline hit before DM ${i} — earlier turns were too slow`,
|
|
124
124
|
).toBeGreaterThan(0);
|
|
125
125
|
|
|
126
|
+
// Skip empty-text observations: the Bot API cannot send an
|
|
127
|
+
// empty text message (Telegram rejects it), so an empty-text
|
|
128
|
+
// fromBot observation is by construction a SERVICE message —
|
|
129
|
+
// e.g. the `[pinned_message]` event from the progress-card pin.
|
|
130
|
+
// One of those latched here as "the reply" on 2026-07-18
|
|
131
|
+
// (ci-uat run 29634400115: pinChatMessage 06:54:36.420Z → rx
|
|
132
|
+
// [pinned_message] 06:54:37.012Z, exactly at the DM-2 failure).
|
|
133
|
+
// A genuinely eaten turn_end still fails loudly: no non-empty
|
|
134
|
+
// reply arrives and this expectMessage times out.
|
|
126
135
|
const reply = await sc.expectMessage(
|
|
127
|
-
(m: ObservedMessage) =>
|
|
136
|
+
(m: ObservedMessage) =>
|
|
137
|
+
m.fromBot && !m.edited && m.text.length > 0,
|
|
128
138
|
{ from: "bot", timeout: remaining },
|
|
129
139
|
);
|
|
130
140
|
expect(
|