switchroom 0.17.10 → 0.18.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/workspace-dynamic-hook.sh +12 -13
- package/dist/agent-scheduler/index.js +29 -2
- package/dist/auth-broker/index.js +6163 -152
- package/dist/cli/notion-write-pretool.mjs +31 -3
- package/dist/cli/switchroom.js +695 -526
- package/dist/host-control/main.js +6184 -173
- package/dist/vault/approvals/kernel-server.js +5893 -165
- package/dist/vault/broker/server.js +6666 -921
- package/package.json +1 -1
- package/profiles/_base/settings.json.hbs +2 -2
- package/profiles/_base/start.sh.hbs +170 -21
- package/profiles/coding/CLAUDE.md.hbs +1 -1
- package/profiles/default/CLAUDE.md.hbs +2 -2
- package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
- package/profiles/health-coach/CLAUDE.md.hbs +1 -1
- package/skills/switchroom-release/SKILL.md +78 -0
- package/telegram-plugin/auth-snapshot-format.ts +37 -25
- package/telegram-plugin/context-exhaustion.ts +124 -0
- package/telegram-plugin/dist/gateway/gateway.js +25025 -9203
- package/telegram-plugin/gateway/activity-card-store.ts +76 -0
- package/telegram-plugin/gateway/gateway.ts +740 -106
- package/telegram-plugin/gateway/inbound-delivery-gate.ts +26 -0
- package/telegram-plugin/gateway/model-command.ts +70 -10
- package/telegram-plugin/gateway/resolve-person.ts +304 -0
- package/telegram-plugin/gateway/unhandled-rejection-policy.ts +21 -1
- package/telegram-plugin/hooks/silent-end-scan.mjs +164 -40
- package/telegram-plugin/operator-events.ts +21 -0
- package/telegram-plugin/package.json +6 -0
- package/telegram-plugin/pending-work-progress.ts +42 -7
- package/telegram-plugin/quota-bar-format.ts +360 -0
- package/telegram-plugin/quota-watch.ts +4 -6
- package/telegram-plugin/registry/turns-schema.test.ts +97 -0
- package/telegram-plugin/registry/turns-schema.ts +78 -0
- package/telegram-plugin/render/ir.ts +209 -0
- package/telegram-plugin/render/parse.ts +363 -0
- package/telegram-plugin/render/render.ts +440 -0
- package/telegram-plugin/render/rich-render.ts +72 -0
- package/telegram-plugin/stream-controller.ts +14 -3
- package/telegram-plugin/subagent-watcher.ts +27 -9
- package/telegram-plugin/tests/activity-card-store.test.ts +94 -0
- package/telegram-plugin/tests/auth-command-format2.test.ts +1 -1
- package/telegram-plugin/tests/auth-snapshot-format.test.ts +51 -16
- package/telegram-plugin/tests/claude-code-event-contract.test.ts +48 -0
- package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +11 -0
- package/telegram-plugin/tests/feed-survival.test.ts +39 -0
- package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +81 -0
- package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +82 -0
- package/telegram-plugin/tests/liveness-tracker.test.ts +228 -0
- package/telegram-plugin/tests/model-command.test.ts +193 -16
- package/telegram-plugin/tests/narrative-render.test.ts +125 -0
- package/telegram-plugin/tests/operator-events.test.ts +16 -0
- package/telegram-plugin/tests/orphaned-reply-rearm.test.ts +123 -163
- package/telegram-plugin/tests/pending-work-progress.test.ts +116 -3
- package/telegram-plugin/tests/quota-bar-format.test.ts +444 -0
- package/telegram-plugin/tests/quota-watch.test.ts +1 -4
- package/telegram-plugin/tests/rapid-fire-delivery-ordering.test.ts +149 -0
- package/telegram-plugin/tests/render/parse-torture.test.ts +136 -0
- package/telegram-plugin/tests/render/parse.test.ts +393 -0
- package/telegram-plugin/tests/render/render.test.ts +436 -0
- package/telegram-plugin/tests/render/rich-render.test.ts +85 -0
- package/telegram-plugin/tests/resolve-person.test.ts +290 -0
- package/telegram-plugin/tests/silent-end-interrupt-stop-integration.test.ts +53 -0
- package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +138 -0
- package/telegram-plugin/tests/subagent-watcher.test.ts +61 -0
- package/telegram-plugin/tests/telegram-activity-visibility-integration.test.ts +155 -1
- package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +19 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +97 -0
- package/telegram-plugin/tests/worktree-watch-cwds.test.ts +98 -3
- package/telegram-plugin/turn-liveness-floor.ts +35 -1
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +99 -7
- package/telegram-plugin/worker-activity-feed.ts +220 -15
- package/telegram-plugin/worktree-watch-cwds.ts +92 -17
- package/vendor/hindsight-memory/scripts/lib/client.py +11 -1
- package/vendor/hindsight-memory/scripts/lib/config.py +9 -2
- package/vendor/hindsight-memory/scripts/recall.py +64 -6
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +1 -0
- package/vendor/hindsight-memory/tests/test_client.py +43 -0
- package/vendor/hindsight-memory/tests/test_recall_precision.py +114 -0
- package/profiles/default/CLAUDE.md +0 -116
- package/telegram-plugin/node_modules/.vite/vitest/da39a3ee5e6b4b0d3255bfef95601890afd80709/results.json +0 -1
- package/vendor/hindsight-memory/scripts/__pycache__/directive_verify.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/drain_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/recall.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/retain.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/session_end.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/__init__.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/bank.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/client.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/config.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/content.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/daemon.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/directives.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/gateway_ipc.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/llm.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/state.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/switchroom_envelope.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/__init__.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_switchroom_envelope.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.0.3.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.0.3.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313.pyc +0 -0
|
@@ -33,6 +33,28 @@
|
|
|
33
33
|
* delivery obligation.
|
|
34
34
|
*/
|
|
35
35
|
|
|
36
|
+
// Verified complete (2026-07-09, adversarial-review follow-up): `reply`
|
|
37
|
+
// and `stream_reply` are the ONLY two MCP tools whose payload is the
|
|
38
|
+
// model's free-text final-answer content reaching the user — the exact
|
|
39
|
+
// scope `final-answer-detect.ts`'s own docstring claims ("plain assistant
|
|
40
|
+
// transcript text instead of a `reply` / `stream_reply` tool call").
|
|
41
|
+
// Cross-checked the full tool surface in `telegram-plugin/bridge/bridge.ts`
|
|
42
|
+
// (`TOOL_SCHEMAS`, kept in sync with `gateway/gateway.ts`): `edit_message`
|
|
43
|
+
// explicitly does NOT ping/deliver a fresh answer (its own description says
|
|
44
|
+
// "send a new reply when a long task completes"); `react`, `pin_message`,
|
|
45
|
+
// `delete_message`, `forward_message`, `send_typing`, `download_attachment`,
|
|
46
|
+
// `get_recent_messages` carry no model-authored answer text at all;
|
|
47
|
+
// `send_checklist` / `send_sticker` / `send_gif` / `ask_user` /
|
|
48
|
+
// `update_checklist` deliver structured/templated content, not the turn's
|
|
49
|
+
// prose answer, and are intentionally a different interaction pattern (a
|
|
50
|
+
// question or a fixed artifact, not "the answer"). `stream_reply` sends its
|
|
51
|
+
// FULL cumulative text snapshot on every call (not incremental chunks —
|
|
52
|
+
// see `stream-reply-handler.ts` docstring), and each call is its own
|
|
53
|
+
// `tool_use` block in the transcript in chronological order, so the
|
|
54
|
+
// "last delivery event wins" walk below already treats a stream's final
|
|
55
|
+
// (`done:true`) call as the qualifying one regardless of how many
|
|
56
|
+
// intermediate non-final `stream_reply` calls preceded it. No gap found;
|
|
57
|
+
// re-verify only if a new outbound-delivery tool is added to bridge.ts.
|
|
36
58
|
const REPLY_TOOLS = new Set([
|
|
37
59
|
'mcp__switchroom-telegram__reply',
|
|
38
60
|
'mcp__switchroom-telegram__stream_reply',
|
|
@@ -126,24 +148,47 @@ function buildTurnKey(chatId, threadId) {
|
|
|
126
148
|
return `${chatId}:${threadId == null || threadId === 0 ? '_' : threadId}`
|
|
127
149
|
}
|
|
128
150
|
|
|
151
|
+
/**
|
|
152
|
+
* Build the `{ decided: 'block', ... }` result shape, populating
|
|
153
|
+
* `turnKey`/`chatId`/`threadId` from the enqueue envelope when
|
|
154
|
+
* available. Shared by both block branches below.
|
|
155
|
+
*
|
|
156
|
+
* @param {ReturnType<typeof parseChannelEnvelope>} envelope
|
|
157
|
+
* @param {string} reason
|
|
158
|
+
*/
|
|
159
|
+
function buildBlockResult(envelope, reason) {
|
|
160
|
+
const block = { decided: 'block', reason }
|
|
161
|
+
if (envelope.chatId) {
|
|
162
|
+
block.chatId = envelope.chatId
|
|
163
|
+
block.threadId = envelope.threadId
|
|
164
|
+
block.turnKey = buildTurnKey(envelope.chatId, envelope.threadId)
|
|
165
|
+
}
|
|
166
|
+
return block
|
|
167
|
+
}
|
|
168
|
+
|
|
129
169
|
/**
|
|
130
170
|
* Scan a JSONL transcript and decide whether the current turn ended
|
|
131
171
|
* with a final reply delivered.
|
|
132
172
|
*
|
|
133
173
|
* Returns:
|
|
134
|
-
* { decided: 'allow', reason } — qualifying reply OR silent marker found
|
|
174
|
+
* { decided: 'allow', reason } — qualifying reply OR silent marker found,
|
|
175
|
+
* and nothing undelivered was written
|
|
176
|
+
* after it
|
|
135
177
|
* { decided: 'block', reason, turnKey?, chatId?, threadId? }
|
|
136
|
-
* — turn-start found, no qualifying reply
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
*
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
*
|
|
178
|
+
* — turn-start found, no qualifying reply
|
|
179
|
+
* delivered (or a qualifying reply
|
|
180
|
+
* happened, but the model kept writing
|
|
181
|
+
* plain-text content afterward that was
|
|
182
|
+
* never sent through a delivery tool).
|
|
183
|
+
* `turnKey`/`chatId`/`threadId` populated
|
|
184
|
+
* from the enqueue's channel envelope so
|
|
185
|
+
* the hook can write a state file shape
|
|
186
|
+
* that matches what the gateway's
|
|
187
|
+
* `recordSilentTurnEnd` would write —
|
|
188
|
+
* keeping the retry-count preservation
|
|
189
|
+
* gate at `silent-end.ts:114` happy when
|
|
190
|
+
* the gateway's later write reads back
|
|
191
|
+
* the hook's state.
|
|
147
192
|
* { decided: 'unknown', reason } — couldn't locate turn-start; caller fail-open
|
|
148
193
|
*
|
|
149
194
|
* Turn-start anchor: the most recent `queue-operation`/`enqueue` line
|
|
@@ -154,6 +199,38 @@ function buildTurnKey(chatId, threadId) {
|
|
|
154
199
|
* edge case where the model replied combined ahead of the second
|
|
155
200
|
* enqueue's append; accepted residual.)
|
|
156
201
|
*
|
|
202
|
+
* IMPORTANT — trailing-content check (fixes the "at least once" bug):
|
|
203
|
+
* a naive scan that returns 'allow' on the FIRST qualifying reply it
|
|
204
|
+
* finds is wrong. A turn can legitimately call `reply` early (e.g. a
|
|
205
|
+
* notification-bearing interim ack — `disable_notification` unset/false
|
|
206
|
+
* always qualifies as "final" under `isFinalAnswerReply`, regardless of
|
|
207
|
+
* how short the text is) and then keep working, eventually writing a
|
|
208
|
+
* SUBSTANTIVE plain-text verdict that never goes through `reply` again.
|
|
209
|
+
* The early ack satisfied "reply was called somewhere in this turn",
|
|
210
|
+
* but the user never saw the actual answer. So this scan does NOT
|
|
211
|
+
* short-circuit on the first match: it walks the ENTIRE turn in
|
|
212
|
+
* chronological order, remembers the position of the LAST qualifying
|
|
213
|
+
* delivery event (a final-answer reply/stream_reply call, or an
|
|
214
|
+
* explicit silent-marker), and then checks whether any plain assistant
|
|
215
|
+
* text block appears AFTER that position. If one does, that content was
|
|
216
|
+
* written but never delivered — block, same as the zero-reply case.
|
|
217
|
+
*
|
|
218
|
+
* This deliberately does NOT flag a turn that ends on a delivery
|
|
219
|
+
* tool_use with nothing after it (the normal, healthy shape), nor a
|
|
220
|
+
* turn where all assistant text precedes the final delivering reply
|
|
221
|
+
* (the model narrating before it sends) — only text that comes AFTER
|
|
222
|
+
* the last delivery event counts as a drop.
|
|
223
|
+
*
|
|
224
|
+
* Substance floor (#2956 review): the trailing-text check only BLOCKS
|
|
225
|
+
* when the trailing text is SUBSTANTIVE — at least FINAL_ANSWER_MIN_CHARS
|
|
226
|
+
* (the same bar `isFinalAnswerReply` uses to recognise a real answer). A
|
|
227
|
+
* SHORT trailing pleasantry / closer after a delivered reply ("Let me
|
|
228
|
+
* know if you need anything else.") is not a dropped answer and must not
|
|
229
|
+
* trigger a re-prompt (no-spam / single-answer invariant). A long
|
|
230
|
+
* trailing verdict the model forgot to send still blocks. The floor keeps
|
|
231
|
+
* the "at least once" guarantee for real dropped answers while stopping a
|
|
232
|
+
* false-positive that burned retry budget on healthy turns.
|
|
233
|
+
*
|
|
157
234
|
* @param {string} jsonl
|
|
158
235
|
* @returns {{ decided: 'allow' | 'block' | 'unknown', reason: string, turnKey?: string, chatId?: string, threadId?: number | null }}
|
|
159
236
|
*/
|
|
@@ -178,8 +255,13 @@ export function scanTurnForFinalReply(jsonl) {
|
|
|
178
255
|
return { decided: 'unknown', reason: 'no-turn-start' }
|
|
179
256
|
}
|
|
180
257
|
|
|
181
|
-
// 2.
|
|
182
|
-
//
|
|
258
|
+
// 2. Flatten every assistant content block (text and tool_use) from
|
|
259
|
+
// the turn into a single chronologically-ordered list. Classify
|
|
260
|
+
// each block as it's collected: a "delivery" event (qualifying
|
|
261
|
+
// final-answer reply/stream_reply, or an explicit silent marker —
|
|
262
|
+
// whether emitted as plain text or as a reply-tool payload) or
|
|
263
|
+
// plain undelivered text.
|
|
264
|
+
const blocks = []
|
|
183
265
|
for (let i = startIdx + 1; i < lines.length; i++) {
|
|
184
266
|
const line = lines[i]
|
|
185
267
|
if (!line || line[0] !== '{') continue
|
|
@@ -193,16 +275,28 @@ export function scanTurnForFinalReply(jsonl) {
|
|
|
193
275
|
const content = obj?.message?.content
|
|
194
276
|
if (!Array.isArray(content)) continue
|
|
195
277
|
for (const c of content) {
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
278
|
+
if (c?.type === 'text') {
|
|
279
|
+
// Plain assistant text carve-out (#2053): a turn that ends with
|
|
280
|
+
// a trailing bare NO_REPLY / HEARTBEAT_OK line — emitted as
|
|
281
|
+
// plain transcript text, NOT through the reply tool — is the
|
|
282
|
+
// model explicitly signalling "intentionally silent". Treat a
|
|
283
|
+
// trailing-marker text block as a delivery/silence event;
|
|
284
|
+
// anything else is candidate undelivered content.
|
|
285
|
+
if (endsWithSilentMarker(String(c.text ?? ''))) {
|
|
286
|
+
blocks.push({ kind: 'deliver', reason: 'silent-marker-text' })
|
|
287
|
+
} else if (String(c.text ?? '').trim().length > 0) {
|
|
288
|
+
// Carry the trimmed char count so the trailing-content check
|
|
289
|
+
// (step 3) can apply a substance floor: a SHORT trailing text
|
|
290
|
+
// after a delivered reply (a pleasantry / closer like "Let me
|
|
291
|
+
// know if you need anything else.") is NOT a dropped answer and
|
|
292
|
+
// must not trigger a re-prompt (no-spam invariant). Only
|
|
293
|
+
// SUBSTANTIVE trailing text — at least FINAL_ANSWER_MIN_CHARS,
|
|
294
|
+
// the same bar `isFinalAnswerReply` uses to recognise a real
|
|
295
|
+
// answer — counts as "undelivered content the user was waiting
|
|
296
|
+
// on". #2956 review finding.
|
|
297
|
+
blocks.push({ kind: 'text', chars: String(c.text ?? '').trim().length })
|
|
298
|
+
}
|
|
299
|
+
continue
|
|
206
300
|
}
|
|
207
301
|
if (c?.type !== 'tool_use') continue
|
|
208
302
|
if (!REPLY_TOOLS.has(c.name)) continue
|
|
@@ -214,33 +308,63 @@ export function scanTurnForFinalReply(jsonl) {
|
|
|
214
308
|
// prose+trailing-marker shape (#2053). Same posture as the
|
|
215
309
|
// gateway's silent-marker suppression at gateway.ts:6692.
|
|
216
310
|
if (SILENT_MARKER_RE.test(text.trim()) || endsWithSilentMarker(text)) {
|
|
217
|
-
|
|
311
|
+
blocks.push({ kind: 'deliver', reason: 'silent-marker' })
|
|
312
|
+
continue
|
|
218
313
|
}
|
|
219
314
|
if (isFinalAnswerReply({
|
|
220
315
|
text,
|
|
221
316
|
disableNotification: input.disable_notification === true,
|
|
222
317
|
done: input.done === true,
|
|
223
318
|
})) {
|
|
224
|
-
|
|
319
|
+
blocks.push({ kind: 'deliver', reason: 'final-reply' })
|
|
320
|
+
continue
|
|
225
321
|
}
|
|
322
|
+
// Non-qualifying reply call (interim ack) — delivered to the
|
|
323
|
+
// user, but not a "final answer". It's neither a delivery event
|
|
324
|
+
// nor undelivered text, so it doesn't affect the decision either
|
|
325
|
+
// way; simply not pushed.
|
|
226
326
|
}
|
|
227
327
|
}
|
|
228
328
|
|
|
229
|
-
//
|
|
230
|
-
//
|
|
231
|
-
//
|
|
232
|
-
//
|
|
233
|
-
//
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
329
|
+
// 3. Find the LAST delivery event's position, then check whether any
|
|
330
|
+
// plain-text block appears strictly after it. This is the fix for
|
|
331
|
+
// the "at least once" bug: a naive scan that stops at the FIRST
|
|
332
|
+
// qualifying reply misses substantive content the model wrote
|
|
333
|
+
// afterward and never (re-)sent.
|
|
334
|
+
let lastAllowBlockIdx = -1
|
|
335
|
+
let lastAllowReason = null
|
|
336
|
+
for (let i = 0; i < blocks.length; i++) {
|
|
337
|
+
if (blocks[i].kind === 'deliver') {
|
|
338
|
+
lastAllowBlockIdx = i
|
|
339
|
+
lastAllowReason = blocks[i].reason
|
|
340
|
+
}
|
|
237
341
|
}
|
|
342
|
+
const sawUndeliveredTextAfterAllow = blocks
|
|
343
|
+
.slice(lastAllowBlockIdx + 1)
|
|
344
|
+
.some((b) => b.kind === 'text' && (b.chars ?? 0) >= FINAL_ANSWER_MIN_CHARS)
|
|
238
345
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
346
|
+
if (lastAllowBlockIdx === -1) {
|
|
347
|
+
// No qualifying delivery/silence event anywhere in the turn.
|
|
348
|
+
// Cron-fired turns (#2053): a scheduled turn that produced no
|
|
349
|
+
// qualifying reply is NOT a delivery failure the user is waiting
|
|
350
|
+
// on — nagging it only pushes the model to escape the loop by
|
|
351
|
+
// shoving a NO_REPLY sentinel through the reply tool, which leaks
|
|
352
|
+
// to chat. A cron turn that genuinely needs to speak will have
|
|
353
|
+
// called reply (caught above); otherwise let it end silently.
|
|
354
|
+
if (envelope.source === 'cron') {
|
|
355
|
+
return { decided: 'allow', reason: 'cron-source' }
|
|
356
|
+
}
|
|
357
|
+
return buildBlockResult(envelope, 'no-final-reply')
|
|
244
358
|
}
|
|
245
|
-
|
|
359
|
+
|
|
360
|
+
if (sawUndeliveredTextAfterAllow) {
|
|
361
|
+
// A qualifying delivery DID happen somewhere in the turn, but the
|
|
362
|
+
// model kept writing after it and that trailing content was never
|
|
363
|
+
// sent through a delivery tool. This is the "at least once" bug:
|
|
364
|
+
// an early ack (or any qualifying reply) must not amnesty
|
|
365
|
+
// everything written afterward.
|
|
366
|
+
return buildBlockResult(envelope, 'trailing-text-after-reply')
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
return { decided: 'allow', reason: lastAllowReason }
|
|
246
370
|
}
|
|
@@ -26,6 +26,7 @@ export type OperatorEventKind =
|
|
|
26
26
|
| 'agent-restarted-unexpectedly'
|
|
27
27
|
| 'unknown-4xx'
|
|
28
28
|
| 'unknown-5xx'
|
|
29
|
+
| 'config-warning'
|
|
29
30
|
|
|
30
31
|
export interface OperatorEvent {
|
|
31
32
|
kind: OperatorEventKind
|
|
@@ -375,6 +376,26 @@ export function renderOperatorEvent(ev: OperatorEvent): RenderResult {
|
|
|
375
376
|
],
|
|
376
377
|
},
|
|
377
378
|
}
|
|
379
|
+
|
|
380
|
+
// Deliberately low-severity framing (ℹ️, Dismiss-only — no Restart /
|
|
381
|
+
// Reauth / Show-logs actions): config-warning is for boot-time config
|
|
382
|
+
// problems (e.g. a dropped `person_id` entry) that must be visible to
|
|
383
|
+
// the operator but MUST NOT read like a real outage or page anyone.
|
|
384
|
+
case 'config-warning':
|
|
385
|
+
return {
|
|
386
|
+
text: [
|
|
387
|
+
`ℹ️ **Config warning** for **${agent}**.`,
|
|
388
|
+
detail ? `_${detail}_` : '',
|
|
389
|
+
`Non-urgent — config will keep working with today's fallback behavior.`,
|
|
390
|
+
]
|
|
391
|
+
.filter(Boolean)
|
|
392
|
+
.join('\n'),
|
|
393
|
+
keyboard: {
|
|
394
|
+
inline_keyboard: [
|
|
395
|
+
[{ text: '❌ Dismiss', callback_data: `op:dismiss:${encodeURIComponent(ev.agent)}` }],
|
|
396
|
+
],
|
|
397
|
+
},
|
|
398
|
+
}
|
|
378
399
|
}
|
|
379
400
|
}
|
|
380
401
|
|
|
@@ -33,8 +33,14 @@
|
|
|
33
33
|
"@secretlint/types": "^12.2.0",
|
|
34
34
|
"@xterm/headless": "^6.0.0",
|
|
35
35
|
"grammy": "^1.44",
|
|
36
|
+
"mdast-util-from-markdown": "^2.0.2",
|
|
37
|
+
"mdast-util-gfm": "^3.0.0",
|
|
38
|
+
"micromark-extension-gfm": "^3.0.0",
|
|
36
39
|
"posthog-node": "^5.29.2"
|
|
37
40
|
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@types/mdast": "^4.0.4"
|
|
43
|
+
},
|
|
38
44
|
"engines": {
|
|
39
45
|
"node": ">=20.11.0"
|
|
40
46
|
},
|
|
@@ -419,9 +419,15 @@ function tick(now: number): void {
|
|
|
419
419
|
newText,
|
|
420
420
|
literalText: s.anchorLiteralText,
|
|
421
421
|
}
|
|
422
|
-
// Fire-and-forget so a slow edit doesn't block the tick loop.
|
|
423
|
-
//
|
|
424
|
-
// /
|
|
422
|
+
// Fire-and-forget so a slow edit doesn't block the tick loop. The
|
|
423
|
+
// production `editMessage` dep is `swallowingApiCall`-wrapped, which
|
|
424
|
+
// catches every transport outcome (not-modified / not-found / 429 /
|
|
425
|
+
// network) upstream and never rejects — so in production this `.catch`
|
|
426
|
+
// is a backstop that rarely fires. It is kept as a contract-level guard
|
|
427
|
+
// (a throwing dep, or a future non-swallowing wiring, must not log a
|
|
428
|
+
// scary "edit failed" warning for a best-effort liveness surface nor
|
|
429
|
+
// keep hammering a dead anchor). Only a genuinely unexpected error
|
|
430
|
+
// reaches the fallthrough; transport classes are silent.
|
|
425
431
|
void Promise.resolve()
|
|
426
432
|
.then(() => activeDeps!.editMessage(editCtx))
|
|
427
433
|
.then(() => {
|
|
@@ -432,10 +438,39 @@ function tick(now: number): void {
|
|
|
432
438
|
})
|
|
433
439
|
})
|
|
434
440
|
.catch((err) => {
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
441
|
+
const desc =
|
|
442
|
+
err instanceof Error ? err.message : err != null && typeof err === 'object' && 'description' in err
|
|
443
|
+
? String((err as { description?: unknown }).description)
|
|
444
|
+
: String(err)
|
|
445
|
+
const low = desc.toLowerCase()
|
|
446
|
+
// "message is not modified" — the anchor already shows this suffix.
|
|
447
|
+
// The card is correct; count it as an edit and move on silently.
|
|
448
|
+
if (low.includes('not modified')) {
|
|
449
|
+
activeDeps!.emitMetric?.({
|
|
450
|
+
kind: 'pending_progress_edited',
|
|
451
|
+
chatKey: key,
|
|
452
|
+
elapsedMs: elapsed,
|
|
453
|
+
})
|
|
454
|
+
return
|
|
455
|
+
}
|
|
456
|
+
// Message / chat gone, or the 48h edit window closed. The anchor
|
|
457
|
+
// is dead — stop retrying it (clear state) so the tick isn't
|
|
458
|
+
// hammering a non-existent message every EDIT_INTERVAL_MS. The
|
|
459
|
+
// next outbound reply re-establishes a fresh anchor. Silent: no
|
|
460
|
+
// card to update is not a liveness-logic error.
|
|
461
|
+
if (
|
|
462
|
+
low.includes('not found') ||
|
|
463
|
+
low.includes("can't be edited") ||
|
|
464
|
+
low.includes('cannot be edited') ||
|
|
465
|
+
low.includes('not enough rights')
|
|
466
|
+
) {
|
|
467
|
+
clearPending(key, 'stale_turn')
|
|
468
|
+
return
|
|
469
|
+
}
|
|
470
|
+
// 429 / transient network blip — leave state intact; the next tick
|
|
471
|
+
// retries. No stderr: a transport hiccup on a best-effort card is
|
|
472
|
+
// not a logic error, and the production wiring already applies
|
|
473
|
+
// retry_after backoff upstream.
|
|
439
474
|
})
|
|
440
475
|
}
|
|
441
476
|
}
|