switchroom 0.17.10 → 0.18.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. package/bin/workspace-dynamic-hook.sh +12 -13
  2. package/dist/agent-scheduler/index.js +29 -2
  3. package/dist/auth-broker/index.js +6163 -152
  4. package/dist/cli/notion-write-pretool.mjs +31 -3
  5. package/dist/cli/switchroom.js +695 -526
  6. package/dist/host-control/main.js +6184 -173
  7. package/dist/vault/approvals/kernel-server.js +5893 -165
  8. package/dist/vault/broker/server.js +6666 -921
  9. package/package.json +1 -1
  10. package/profiles/_base/settings.json.hbs +2 -2
  11. package/profiles/_base/start.sh.hbs +170 -21
  12. package/profiles/coding/CLAUDE.md.hbs +1 -1
  13. package/profiles/default/CLAUDE.md.hbs +2 -2
  14. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  15. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  16. package/skills/switchroom-release/SKILL.md +78 -0
  17. package/telegram-plugin/auth-snapshot-format.ts +37 -25
  18. package/telegram-plugin/context-exhaustion.ts +124 -0
  19. package/telegram-plugin/dist/gateway/gateway.js +25025 -9203
  20. package/telegram-plugin/gateway/activity-card-store.ts +76 -0
  21. package/telegram-plugin/gateway/gateway.ts +740 -106
  22. package/telegram-plugin/gateway/inbound-delivery-gate.ts +26 -0
  23. package/telegram-plugin/gateway/model-command.ts +70 -10
  24. package/telegram-plugin/gateway/resolve-person.ts +304 -0
  25. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +21 -1
  26. package/telegram-plugin/hooks/silent-end-scan.mjs +164 -40
  27. package/telegram-plugin/operator-events.ts +21 -0
  28. package/telegram-plugin/package.json +6 -0
  29. package/telegram-plugin/pending-work-progress.ts +42 -7
  30. package/telegram-plugin/quota-bar-format.ts +360 -0
  31. package/telegram-plugin/quota-watch.ts +4 -6
  32. package/telegram-plugin/registry/turns-schema.test.ts +97 -0
  33. package/telegram-plugin/registry/turns-schema.ts +78 -0
  34. package/telegram-plugin/render/ir.ts +209 -0
  35. package/telegram-plugin/render/parse.ts +363 -0
  36. package/telegram-plugin/render/render.ts +440 -0
  37. package/telegram-plugin/render/rich-render.ts +72 -0
  38. package/telegram-plugin/stream-controller.ts +14 -3
  39. package/telegram-plugin/subagent-watcher.ts +27 -9
  40. package/telegram-plugin/tests/activity-card-store.test.ts +94 -0
  41. package/telegram-plugin/tests/auth-command-format2.test.ts +1 -1
  42. package/telegram-plugin/tests/auth-snapshot-format.test.ts +51 -16
  43. package/telegram-plugin/tests/claude-code-event-contract.test.ts +48 -0
  44. package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +11 -0
  45. package/telegram-plugin/tests/feed-survival.test.ts +39 -0
  46. package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
  47. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +81 -0
  48. package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +82 -0
  49. package/telegram-plugin/tests/liveness-tracker.test.ts +228 -0
  50. package/telegram-plugin/tests/model-command.test.ts +193 -16
  51. package/telegram-plugin/tests/narrative-render.test.ts +125 -0
  52. package/telegram-plugin/tests/operator-events.test.ts +16 -0
  53. package/telegram-plugin/tests/orphaned-reply-rearm.test.ts +123 -163
  54. package/telegram-plugin/tests/pending-work-progress.test.ts +116 -3
  55. package/telegram-plugin/tests/quota-bar-format.test.ts +444 -0
  56. package/telegram-plugin/tests/quota-watch.test.ts +1 -4
  57. package/telegram-plugin/tests/rapid-fire-delivery-ordering.test.ts +149 -0
  58. package/telegram-plugin/tests/render/parse-torture.test.ts +136 -0
  59. package/telegram-plugin/tests/render/parse.test.ts +393 -0
  60. package/telegram-plugin/tests/render/render.test.ts +436 -0
  61. package/telegram-plugin/tests/render/rich-render.test.ts +85 -0
  62. package/telegram-plugin/tests/resolve-person.test.ts +290 -0
  63. package/telegram-plugin/tests/silent-end-interrupt-stop-integration.test.ts +53 -0
  64. package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +138 -0
  65. package/telegram-plugin/tests/subagent-watcher.test.ts +61 -0
  66. package/telegram-plugin/tests/telegram-activity-visibility-integration.test.ts +155 -1
  67. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +19 -0
  68. package/telegram-plugin/tests/worker-activity-feed.test.ts +97 -0
  69. package/telegram-plugin/tests/worktree-watch-cwds.test.ts +98 -3
  70. package/telegram-plugin/turn-liveness-floor.ts +35 -1
  71. package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +99 -7
  72. package/telegram-plugin/worker-activity-feed.ts +220 -15
  73. package/telegram-plugin/worktree-watch-cwds.ts +92 -17
  74. package/vendor/hindsight-memory/scripts/lib/client.py +11 -1
  75. package/vendor/hindsight-memory/scripts/lib/config.py +9 -2
  76. package/vendor/hindsight-memory/scripts/recall.py +64 -6
  77. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +1 -0
  78. package/vendor/hindsight-memory/tests/test_client.py +43 -0
  79. package/vendor/hindsight-memory/tests/test_recall_precision.py +114 -0
  80. package/profiles/default/CLAUDE.md +0 -116
  81. package/telegram-plugin/node_modules/.vite/vitest/da39a3ee5e6b4b0d3255bfef95601890afd80709/results.json +0 -1
  82. package/vendor/hindsight-memory/scripts/__pycache__/directive_verify.cpython-313.pyc +0 -0
  83. package/vendor/hindsight-memory/scripts/__pycache__/drain_pending.cpython-313.pyc +0 -0
  84. package/vendor/hindsight-memory/scripts/__pycache__/recall.cpython-313.pyc +0 -0
  85. package/vendor/hindsight-memory/scripts/__pycache__/retain.cpython-313.pyc +0 -0
  86. package/vendor/hindsight-memory/scripts/__pycache__/session_end.cpython-313.pyc +0 -0
  87. package/vendor/hindsight-memory/scripts/lib/__pycache__/__init__.cpython-313.pyc +0 -0
  88. package/vendor/hindsight-memory/scripts/lib/__pycache__/bank.cpython-313.pyc +0 -0
  89. package/vendor/hindsight-memory/scripts/lib/__pycache__/client.cpython-313.pyc +0 -0
  90. package/vendor/hindsight-memory/scripts/lib/__pycache__/config.cpython-313.pyc +0 -0
  91. package/vendor/hindsight-memory/scripts/lib/__pycache__/content.cpython-313.pyc +0 -0
  92. package/vendor/hindsight-memory/scripts/lib/__pycache__/daemon.cpython-313.pyc +0 -0
  93. package/vendor/hindsight-memory/scripts/lib/__pycache__/directives.cpython-313.pyc +0 -0
  94. package/vendor/hindsight-memory/scripts/lib/__pycache__/gateway_ipc.cpython-313.pyc +0 -0
  95. package/vendor/hindsight-memory/scripts/lib/__pycache__/llm.cpython-313.pyc +0 -0
  96. package/vendor/hindsight-memory/scripts/lib/__pycache__/pending.cpython-313.pyc +0 -0
  97. package/vendor/hindsight-memory/scripts/lib/__pycache__/state.cpython-313.pyc +0 -0
  98. package/vendor/hindsight-memory/scripts/lib/__pycache__/switchroom_envelope.cpython-313.pyc +0 -0
  99. package/vendor/hindsight-memory/scripts/tests/__pycache__/__init__.cpython-313.pyc +0 -0
  100. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313-pytest-9.1.1.pyc +0 -0
  101. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313.pyc +0 -0
  102. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313-pytest-9.1.1.pyc +0 -0
  103. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313.pyc +0 -0
  104. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313-pytest-9.1.1.pyc +0 -0
  105. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313.pyc +0 -0
  106. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313-pytest-9.1.1.pyc +0 -0
  107. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313.pyc +0 -0
  108. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313-pytest-9.1.1.pyc +0 -0
  109. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313.pyc +0 -0
  110. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313-pytest-9.1.1.pyc +0 -0
  111. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313.pyc +0 -0
  112. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313-pytest-9.1.1.pyc +0 -0
  113. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313.pyc +0 -0
  114. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313-pytest-9.1.1.pyc +0 -0
  115. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313.pyc +0 -0
  116. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313-pytest-9.1.1.pyc +0 -0
  117. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313.pyc +0 -0
  118. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313-pytest-9.1.1.pyc +0 -0
  119. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313.pyc +0 -0
  120. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313-pytest-9.1.1.pyc +0 -0
  121. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313.pyc +0 -0
  122. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313-pytest-9.1.1.pyc +0 -0
  123. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313.pyc +0 -0
  124. package/vendor/hindsight-memory/scripts/tests/__pycache__/test_switchroom_envelope.cpython-313-pytest-9.1.1.pyc +0 -0
  125. package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.0.3.pyc +0 -0
  126. package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.1.1.pyc +0 -0
  127. package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313-pytest-9.1.1.pyc +0 -0
  128. package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313.pyc +0 -0
  129. package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313-pytest-9.1.1.pyc +0 -0
  130. package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313.pyc +0 -0
  131. package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.0.3.pyc +0 -0
  132. package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.1.1.pyc +0 -0
  133. package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313.pyc +0 -0
  134. package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313-pytest-9.1.1.pyc +0 -0
  135. package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313.pyc +0 -0
  136. package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313-pytest-9.1.1.pyc +0 -0
  137. package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313.pyc +0 -0
  138. package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313-pytest-9.1.1.pyc +0 -0
  139. package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313.pyc +0 -0
  140. package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313-pytest-9.1.1.pyc +0 -0
  141. package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313.pyc +0 -0
  142. package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313-pytest-9.1.1.pyc +0 -0
  143. package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313.pyc +0 -0
  144. package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313-pytest-9.1.1.pyc +0 -0
  145. package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313.pyc +0 -0
  146. package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313-pytest-9.1.1.pyc +0 -0
  147. package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313.pyc +0 -0
  148. package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313-pytest-9.1.1.pyc +0 -0
  149. package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313.pyc +0 -0
@@ -220,6 +220,28 @@ interface WorkerHandle {
220
220
  chain: Promise<void>
221
221
  /** Last view rendered into the message (drives the heartbeat re-render). */
222
222
  lastView: WorkerActivityView | null
223
+ /**
224
+ * A terminal (`finish`) view whose edit could not land yet — most often
225
+ * because a 429 cooldown was in effect when `doFinish` ran. The heartbeat
226
+ * re-drives `doFinish` with this view once the cooldown expires so a
227
+ * transport hiccup can't leave the card stuck on its last running render
228
+ * ("worker done, card says running"). Cleared on a successful terminal
229
+ * edit, on a permanent failure (message gone), or when the handle is
230
+ * deleted. Null when no finalize is pending.
231
+ */
232
+ pendingFinish: WorkerActivityView | null
233
+ /**
234
+ * Latched in `doFinish` before the terminal edit. A late watcher
235
+ * `onProgress` tick that arrives after `finish()` queued its chain (but
236
+ * before the `.finally(handles.delete)` microtask drains) must NOT
237
+ * resurrect the handle and paint a fresh `running` message on an
238
+ * already-finalized worker. The heartbeat's orphan-paint guard
239
+ * (`if (!handles.has(h.agentId)) continue`) only covers the heartbeat
240
+ * tick — this flag covers the `update` entry point. Set synchronously
241
+ * inside `doFinish` (runs on the chain), checked synchronously in
242
+ * `update` before handle creation.
243
+ */
244
+ finished: boolean
223
245
  /**
224
246
  * Wall-clock ms the worker was dispatched, derived from `now - view.elapsedMs`
225
247
  * on the first update. The heartbeat computes a live elapsed from this so the
@@ -246,6 +268,47 @@ function extractRetryAfterSecs(err: unknown): number | null {
246
268
  return null
247
269
  }
248
270
 
271
+ /**
272
+ * Classify a card-edit transport error. Card edits are a best-effort
273
+ * liveness surface — a recoverable hiccup must never freeze the card, and
274
+ * a permanent failure must never log a scary warning for something with
275
+ * nothing to update.
276
+ *
277
+ * 'not_modified' — content identical to what's already shown. The card
278
+ * already reads correctly; treat as SUCCESS (no retry, no warning).
279
+ * 'rate_limited' — 429 with retry_after. Back off; the heartbeat re-drives
280
+ * the edit after cooldown (running renders + deferred terminal edits).
281
+ * 'gone' — message/chat deleted or edit window expired. Nothing to
282
+ * update; drop the handle silently (no warning — there is no card).
283
+ * 'transient' — anything else (network blip, 5xx). Retry on the next
284
+ * heartbeat tick; don't spam stderr.
285
+ */
286
+ type EditOutcome = 'not_modified' | 'rate_limited' | 'gone' | 'transient'
287
+ function classifyEditError(err: unknown): EditOutcome {
288
+ const retryAfter = extractRetryAfterSecs(err)
289
+ if (retryAfter != null) return 'rate_limited'
290
+ const desc =
291
+ err instanceof Error ? err.message : err != null && typeof err === 'object' && 'description' in err
292
+ ? String((err as { description?: unknown }).description)
293
+ : String(err)
294
+ const low = desc.toLowerCase()
295
+ // "message is not modified" / "message was not modified" — Telegram's
296
+ // identical-content signal. The card already shows the right thing.
297
+ if (low.includes('not modified')) return 'not_modified'
298
+ // Message or chat no longer exists, or the edit window (48h) has closed.
299
+ // "message to edit not found" / "message to delete not found" / "chat not
300
+ // found" / "message can't be edited". Nothing to update.
301
+ if (
302
+ low.includes('not found') ||
303
+ low.includes("can't be edited") ||
304
+ low.includes('cannot be edited') ||
305
+ low.includes('not enough rights')
306
+ ) {
307
+ return 'gone'
308
+ }
309
+ return 'transient'
310
+ }
311
+
249
312
  /**
250
313
  * Manager owning one live message per background worker. Keyed by jsonl
251
314
  * agent id. The gateway calls `update` on each watcher activity cue and
@@ -294,6 +357,28 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
294
357
  })
295
358
  const clearIntervalFn = opts.clearInterval ?? ((handle: unknown) => clearInterval(handle as ReturnType<typeof setInterval>))
296
359
  const handles = new Map<string, WorkerHandle>()
360
+ /**
361
+ * Agent ids that have been finalized (`doFinish` latched). Survives handle
362
+ * deletion so a LATE watcher `onProgress` tick — which can arrive after
363
+ * `finish()`'s chain has fully settled and the handle was deleted — cannot
364
+ * resurrect a fresh handle and paint a running card on a worker that is
365
+ * already done. The per-handle `finished` flag only covers the narrow
366
+ * window between latch and delete; this set is the durable gate. A late
367
+ * tick arrives within seconds of finish (watcher poll cadence), so the set
368
+ * only needs to cover recent finalizations — capped at FINALIZED_CAP and
369
+ * trimmed FIFO to stay bounded across a long gateway lifetime.
370
+ */
371
+ const finalized = new Set<string>()
372
+ const FINALIZED_CAP = 256
373
+ function markFinalized(agentId: string): void {
374
+ if (finalized.has(agentId)) return
375
+ finalized.add(agentId)
376
+ if (finalized.size > FINALIZED_CAP) {
377
+ // Map-free FIFO trim: Set iterates in insertion order; drop the oldest.
378
+ const oldest = finalized.values().next().value
379
+ if (oldest != null) finalized.delete(oldest)
380
+ }
381
+ }
297
382
  let heartbeatTimer: unknown = null
298
383
 
299
384
  function sendOptsFor(h: WorkerHandle): Record<string, unknown> {
@@ -384,39 +469,99 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
384
469
  `thread=${h.threadId ?? '-'} msgId=${h.messageId} bytes=${body.length}`,
385
470
  )
386
471
  } catch (err) {
387
- noteRateLimited(h, err, 'edit')
388
- // Stale message_id (manually deleted / edit window gone). Re-post
389
- // on the next tick rather than now, so we don't double-down inside
390
- // a cooldown.
391
- log(`worker-feed: edit failed, will re-post: ${(err as Error).message}`)
392
- h.messageId = null
393
- h.lastBody = null
472
+ const outcome = classifyEditError(err)
473
+ if (outcome === 'rate_limited') {
474
+ noteRateLimited(h, err, 'edit')
475
+ return
476
+ }
477
+ if (outcome === 'not_modified') {
478
+ // Card already shows this body — record it as landed and move on.
479
+ h.lastBody = body
480
+ h.lastEditAt = nowFn()
481
+ return
482
+ }
483
+ if (outcome === 'gone') {
484
+ // Message/chat deleted or edit window closed — there is no card to
485
+ // update. Drop the handle silently; a fresh first-paint on the next
486
+ // running tick re-establishes one if the worker is still active. No
487
+ // warning: "no card" is not a liveness-logic error.
488
+ h.messageId = null
489
+ h.lastBody = null
490
+ return
491
+ }
492
+ // 'transient' — network blip / 5xx. Leave the handle intact; the
493
+ // heartbeat re-attempts on its next tick. Log at debug, not stderr-warn:
494
+ // a transport hiccup on a best-effort card is not "shit code", it's a
495
+ // retryable blip the framework rides out deterministically.
496
+ log(`worker-feed: edit transient error agent=${h.agentId}: ${(err as Error).message}`)
394
497
  }
395
498
  }
396
499
 
397
500
  async function doFinish(h: WorkerHandle, view: WorkerActivityView): Promise<void> {
501
+ // Latch FIRST, before any early return. A `running`-cue tick arriving
502
+ // after `finish()` queued this chain (but before its `.finally(delete)`
503
+ // drains) would otherwise resurrect a handle via `update()` and paint a
504
+ // fresh running message on a finalized worker. Setting this synchronously
505
+ // on the chain — ahead of the cooldown/no-message guards — makes the
506
+ // gate in `update()` authoritative regardless of which guard path runs.
507
+ // The durable `finalized` set survives the subsequent handle deletion so
508
+ // a tick arriving AFTER the full settle still can't resurrect.
509
+ h.finished = true
510
+ markFinalized(h.agentId)
398
511
  // No message ever posted → nothing to finalize. The worker's result
399
512
  // reaches the user via the handback reply; a bare "done" recap with
400
513
  // no preceding activity would be noise.
401
- if (h.messageId == null) return
514
+ if (h.messageId == null) {
515
+ h.pendingFinish = null
516
+ return
517
+ }
402
518
  if (nowFn() < h.cooldownUntil) {
403
- // Honour the flood-wait; a terminal edit isn't worth a ban. The
404
- // message is left at its last running render — stale but harmless.
519
+ // Honour the flood-wait; a terminal edit isn't worth a ban. But
520
+ // unlike the prior "stale but harmless" surrender, STAGE the terminal
521
+ // view so the heartbeat re-drives the finalize edit the instant the
522
+ // cooldown expires — a transport hiccup can no longer leave a
523
+ // finished worker's card stuck on its last running render.
524
+ h.pendingFinish = view
405
525
  return
406
526
  }
407
527
  const body = renderWorkerActivity({ ...view, narrativeLines: h.narrative })
408
- if (body === h.lastBody) return
528
+ if (body === h.lastBody) {
529
+ h.pendingFinish = null
530
+ return
531
+ }
409
532
  try {
410
533
  await opts.bot.editMessageText(h.chatId, h.messageId, body, sendOptsFor(h))
411
534
  h.lastBody = body
412
535
  h.lastEditAt = nowFn()
536
+ h.pendingFinish = null
413
537
  log(
414
538
  `worker-feed: finish agent=${h.agentId} chat=${h.chatId} ` +
415
539
  `thread=${h.threadId ?? '-'} msgId=${h.messageId} state=${view.state} bytes=${body.length}`,
416
540
  )
417
541
  } catch (err) {
418
- noteRateLimited(h, err, 'finish')
419
- log(`worker-feed: finish edit failed: ${(err as Error).message}`)
542
+ const outcome = classifyEditError(err)
543
+ if (outcome === 'rate_limited') {
544
+ noteRateLimited(h, err, 'finish')
545
+ // Re-stage for the heartbeat to re-drive after cooldown.
546
+ h.pendingFinish = view
547
+ return
548
+ }
549
+ if (outcome === 'not_modified') {
550
+ // Card already shows the finalized body — terminal edit succeeded.
551
+ h.lastBody = body
552
+ h.lastEditAt = nowFn()
553
+ h.pendingFinish = null
554
+ return
555
+ }
556
+ if (outcome === 'gone') {
557
+ // Message/chat gone — no card to finalize. Drop silently; the
558
+ // handback reply carries the result regardless.
559
+ h.pendingFinish = null
560
+ return
561
+ }
562
+ // 'transient' — re-stage for a heartbeat retry; log at debug.
563
+ h.pendingFinish = view
564
+ log(`worker-feed: finish transient error agent=${h.agentId}: ${(err as Error).message}`)
420
565
  }
421
566
  }
422
567
 
@@ -456,6 +601,36 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
456
601
  // (which would orphan a card that never finalizes). Restores the
457
602
  // structural safety the pre-first-paint `messageId == null` skip gave.
458
603
  if (!handles.has(h.agentId)) continue
604
+
605
+ // Deferred-finalize re-drive: a terminal edit that hit a 429 cooldown
606
+ // (or a transient error) was staged on `pendingFinish` by `doFinish`.
607
+ // Re-drive it once the cooldown has expired so a finished worker's card
608
+ // can't get stuck on its last running render. This is the deterministic
609
+ // backstop that replaces the old "stale but harmless" surrender — the
610
+ // framework owns ALIVE-and-done, wall-clock driven, no model in the loop.
611
+ // (The handle is still in the map because `finish()`'s `.finally(delete)`
612
+ // is chained AFTER `doFinish` and won't drain while a re-drive keeps the
613
+ // chain busy; once the terminal edit lands, `pendingFinish` is cleared
614
+ // and the `.finally` runs on the next chain settle.)
615
+ if (h.pendingFinish != null && now >= h.cooldownUntil) {
616
+ const view = h.pendingFinish
617
+ h.chain = h.chain
618
+ .then(() => doFinish(h, view))
619
+ .catch((err) => {
620
+ log(`worker-feed: heartbeat finalize re-drive error ${h.agentId}: ${(err as Error).message}`)
621
+ })
622
+ .finally(() => {
623
+ // Mirror `finish()`'s teardown: once the re-driven `doFinish`
624
+ // clears `pendingFinish` (terminal edit landed OR permanently
625
+ // failed), drop the handle. If it re-staged (another 429), the
626
+ // handle survives for the next heartbeat tick to retry.
627
+ if (handles.get(h.agentId)?.pendingFinish == null) {
628
+ handles.delete(h.agentId)
629
+ }
630
+ })
631
+ continue
632
+ }
633
+
459
634
  if (h.lastView == null) continue
460
635
  if (h.lastView.state !== 'running') continue
461
636
  if (now < h.cooldownUntil) continue
@@ -523,7 +698,19 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
523
698
  // No chat to post to (owner DM unconfigured) — don't create a
524
699
  // handle that would retry a failing send('') every tick.
525
700
  if (chatId.length === 0) return Promise.resolve()
526
- let h = handles.get(agentId)
701
+ // Resurrection guard: a worker that has already been finalized
702
+ // (`doFinish` latched `finalized`) must not get a fresh running cue.
703
+ // A late watcher `onProgress` tick can arrive after `finish()`'s chain
704
+ // has fully settled and the handle was deleted — without this durable
705
+ // gate the tick would create a brand-new handle and paint a fresh
706
+ // `running` message on an already-done worker (the card lies). The
707
+ // heartbeat's orphan-paint guard covers the heartbeat tick only; the
708
+ // per-handle `finished` flag covers the pre-delete window; this set
709
+ // covers the post-delete window.
710
+ if (finalized.has(agentId)) return Promise.resolve()
711
+ const existing = handles.get(agentId)
712
+ if (existing?.finished === true) return Promise.resolve()
713
+ let h = existing
527
714
  if (h == null) {
528
715
  h = {
529
716
  agentId,
@@ -538,6 +725,8 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
538
725
  lastView: null,
539
726
  dispatchAtMs: null,
540
727
  stepStartedAtMs: null,
728
+ finished: false,
729
+ pendingFinish: null,
541
730
  }
542
731
  handles.set(agentId, h)
543
732
  }
@@ -556,11 +745,27 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
556
745
  log(`worker-feed: finish chain error ${agentId}: ${(err as Error).message}`)
557
746
  })
558
747
  .finally(() => {
559
- handles.delete(agentId)
748
+ // Only tear down the handle once the terminal edit has actually
749
+ // landed (or permanently failed). If `doFinish` staged the edit on
750
+ // `pendingFinish` (a 429 cooldown / transient error was in effect),
751
+ // the handle must survive so the heartbeat can re-drive the
752
+ // finalize after cooldown. The heartbeat's re-drive chain ends by
753
+ // re-entering `doFinish`, which clears `pendingFinish` on success
754
+ // or permanent-failure — so this `.finally` deletes on the NEXT
755
+ // chain settle once there is nothing left to finalize. Without this
756
+ // guard, the `.finally` would delete the handle (and its staged
757
+ // pendingFinish) immediately after the first staged doFinish,
758
+ // stranding the card on its last running render.
759
+ if (handles.get(agentId)?.pendingFinish == null) {
760
+ handles.delete(agentId)
761
+ }
560
762
  })
561
763
  return h.chain
562
764
  },
563
765
  drop(agentId) {
766
+ // A dropped worker is also done — mark finalized so a late watcher
767
+ // tick can't resurrect a running card on it (same gate as `finish`).
768
+ markFinalized(agentId)
564
769
  handles.delete(agentId)
565
770
  },
566
771
  heartbeatTick,
@@ -1,29 +1,52 @@
1
1
  /**
2
2
  * Ownership filter for the worktree-isolated cwds the subagent-watcher should
3
3
  * additionally watch (deterministic-turn-liveness.md Known Gap 2 + the #2893
4
- * ownership-predicate review fix).
4
+ * ownership-predicate review fix + the #1116 / #2893 durable-identity fix).
5
5
  *
6
6
  * A sub-agent dispatched into a `switchroom worktree claim` cwd runs under a
7
7
  * different project-dir slug than the agent's own `agentCwd`, so the #1116
8
8
  * foreign-slug filter would skip it forever unless the watcher also watches
9
9
  * the slugs of worktrees THIS agent owns. This helper derives that set from
10
- * the host-global worktree registry, and it is deliberately fail-CLOSED:
10
+ * the host-global worktree registry, filtered by the agent's own identity.
11
11
  *
12
- * - Unset/empty identity ⇒ contribute NOTHING. `ownerAgent` is optional in
13
- * the registry, so a naive `r.ownerAgent === process.env.SWITCHROOM_AGENT_NAME`
14
- * with the env var unset becomes `undefined === undefined` and matches
15
- * every OWNERLESS record in the host-global registry — including other
16
- * agents' worktrees. That is precisely the #1116 leak the filter exists to
17
- * prevent, reintroduced by failing OPEN. With no identity we cannot prove
18
- * ownership of anything, so we return `[]`.
19
- * - A registry read failure ⇒ `[]` (best-effort; never disturb the base
20
- * agentCwd watch).
21
- * - Owner match ⇒ include, realpath'd. Claude Code mints the project slug off
22
- * the process's PHYSICAL cwd, so a symlinked base (macOS `/tmp` →
12
+ * Identity resolution is two-tier (durable fix for the gap where a worktree
13
+ * worker whose identity can't be attributed gets NO live progress feed):
14
+ *
15
+ * 1. FAST PATH — `self` (`process.env.SWITCHROOM_AGENT_NAME`). Set
16
+ * authoritatively by compose env (compose.ts) AND hoisted in start.sh
17
+ * before the gateway fork, so this is present in the overwhelming
18
+ * majority of runs.
19
+ * 2. DURABLE FALLBACK — when `self` is unset/empty, derive the identity
20
+ * from `agentDir` (the agent's own directory, itself derived from
21
+ * `TELEGRAM_STATE_DIR` = `<agentDir>/telegram`, which the gateway
22
+ * already requires to be present before it even starts the watcher).
23
+ * The basename of `agentDir` is `resolve(agents_dir, <name>)`'s leaf —
24
+ * i.e. this agent's OWN name. This can only ever resolve to THIS
25
+ * agent's identity, never another agent's, so it cannot mis-attribute:
26
+ * a wrong basename matches zero registry records (fail-closed), it
27
+ * never matches a DIFFERENT owner. Env is just the fast path; ownership
28
+ * resolves correctly from durable config when env is missing.
29
+ *
30
+ * Fail-CLOSED, deliberately, and never mis-attributing:
31
+ *
32
+ * - Owner match ⇒ include, realpath'd. Claude Code mints the project slug
33
+ * off the process's PHYSICAL cwd, so a symlinked base (macOS `/tmp` →
23
34
  * `/private/tmp`) would otherwise derive a slug that misses the physical
24
35
  * one; realpath best-effort, falling back to the raw path.
36
+ * - Ownerless registry records (`ownerAgent` undefined) are NEVER matched,
37
+ * even with identity set — a naive `undefined === undefined` would leak
38
+ * every other agent's ownerless worktree (the #1116 leak this exists to
39
+ * prevent).
40
+ * - A registry read failure ⇒ `[]` (best-effort; never disturb the base
41
+ * agentCwd watch).
42
+ * - BOTH env AND agentDir-derived identity unavailable ⇒ `[]` (same
43
+ * fail-closed contract as before this fix — we never guess) but escalate
44
+ * the log from the #2893 one-shot warn to a clear ERROR naming that
45
+ * identity resolution fully failed, so the lost live feed is diagnosable.
46
+ * Never throws, never mis-attributes.
25
47
  */
26
48
  import { realpathSync } from "node:fs";
49
+ import { basename } from "node:path";
27
50
 
28
51
  export interface WorktreeOwnershipRecord {
29
52
  path: string;
@@ -31,22 +54,74 @@ export interface WorktreeOwnershipRecord {
31
54
  }
32
55
 
33
56
  export interface OwnedWorktreeCwdsOptions {
34
- /** The agent's identity — `process.env.SWITCHROOM_AGENT_NAME`. */
57
+ /** The agent's identity — `process.env.SWITCHROOM_AGENT_NAME` (fast path). */
35
58
  self: string | undefined;
36
59
  /** The host-global registry read (`listRecords` from src/worktree/registry). */
37
60
  listRecords: () => WorktreeOwnershipRecord[];
61
+ /**
62
+ * Durable, non-env fallback source for identity: the agent's OWN directory
63
+ * (`resolveAgentDirFromEnv()` in the gateway). When `self` is unset/empty,
64
+ * the identity is derived as `basename(agentDir)`. Omit to disable the
65
+ * fallback (the pre-fix, env-only behaviour — used by the kill-switch).
66
+ */
67
+ agentDir?: string | null;
38
68
  /** Injectable for tests; defaults to `fs.realpathSync`. */
39
69
  realpath?: (p: string) => string;
70
+ /**
71
+ * Injectable derivation of the agent name from `agentDir`. Defaults to
72
+ * `path.basename`. Returns "" when it cannot derive a usable name.
73
+ */
74
+ deriveName?: (agentDir: string) => string;
75
+ /** Escalated-failure sink (both identity sources unavailable). */
76
+ log?: (msg: string) => void;
77
+ }
78
+
79
+ // One-shot guard so the escalated "identity fully unresolved" ERROR is emitted
80
+ // ONCE per process rather than every rescan tick (the provider is re-invoked on
81
+ // every tick). Mirrors the #2893 one-shot-warn ethos; exported reset for tests.
82
+ let identityEscalated = false;
83
+ export function __resetIdentityEscalationForTests(): void {
84
+ identityEscalated = false;
85
+ }
86
+
87
+ function defaultDeriveName(agentDir: string): string {
88
+ if (!agentDir || agentDir.trim().length === 0) return "";
89
+ const leaf = basename(agentDir).trim();
90
+ return leaf;
40
91
  }
41
92
 
42
93
  export function ownedWorktreeCwds(opts: OwnedWorktreeCwdsOptions): string[] {
43
- const { self } = opts;
44
- if (self == null || self === "") return [];
94
+ // Tier 1: env fast path. Tier 2: durable agentDir-derived fallback.
95
+ let resolved: string = opts.self != null ? opts.self : "";
96
+ if (resolved === "" && opts.agentDir != null && opts.agentDir !== "") {
97
+ const derive = opts.deriveName ?? defaultDeriveName;
98
+ resolved = derive(opts.agentDir) || "";
99
+ }
100
+
101
+ if (resolved === "") {
102
+ // Both env and durable config unavailable. Keep the historical
103
+ // fail-closed contract (return [] — never guess, never mis-attribute) but
104
+ // ESCALATE past the #2893 one-shot warn: name that identity resolution
105
+ // fully failed and the live worktree-worker feed is lost for this run.
106
+ if (!identityEscalated) {
107
+ identityEscalated = true;
108
+ opts.log?.(
109
+ "ERROR: worktree identity resolution FAILED — both " +
110
+ "SWITCHROOM_AGENT_NAME and the agentDir-derived fallback are " +
111
+ "unavailable. Worktree ownership cannot be attributed; a " +
112
+ "worktree-isolated background sub-agent will get NO live progress " +
113
+ "feed this run (its registry row is still reaped by the 1h safety " +
114
+ "net). This is a configuration fault, not a transient error.",
115
+ );
116
+ }
117
+ return [];
118
+ }
119
+
45
120
  const rp = opts.realpath ?? realpathSync;
46
121
  try {
47
122
  return opts
48
123
  .listRecords()
49
- .filter((r) => r.ownerAgent === self)
124
+ .filter((r) => r.ownerAgent === resolved)
50
125
  .map((r) => {
51
126
  try {
52
127
  return rp(r.path);
@@ -124,11 +124,19 @@ class HindsightClient:
124
124
  tags: Optional[list] = None,
125
125
  tags_match: Optional[str] = None,
126
126
  tag_groups: Optional[object] = None,
127
+ prefer_observations: Optional[bool] = None,
127
128
  timeout: int = 10,
128
129
  ) -> dict:
129
130
  """Recall memories from a bank.
130
131
 
131
- Returns the raw API response dict with 'results' list.
132
+ Returns the raw API response dict with 'results' list. Each result
133
+ carries a `scores` object (`RecallScores`) whose `final` field is the
134
+ engine's combined ranking score — callers sort the merged multi-bank
135
+ set by it before applying any count cap.
136
+
137
+ `prefer_observations=True` asks the engine to prefer deduped
138
+ observation statements over the raw facts they supersede, backfilling
139
+ the freed slots — denser coverage inside the same token/count budget.
132
140
  """
133
141
  path = f"/v1/default/banks/{urllib.parse.quote(bank_id, safe='')}/memories/recall"
134
142
  body = {
@@ -145,6 +153,8 @@ class HindsightClient:
145
153
  body["tags_match"] = tags_match
146
154
  if tag_groups:
147
155
  body["tag_groups"] = tag_groups
156
+ if prefer_observations is not None:
157
+ body["prefer_observations"] = prefer_observations
148
158
  return self._request("POST", path, body, timeout=timeout)
149
159
 
150
160
  def retain(
@@ -33,10 +33,17 @@ DEFAULTS = {
33
33
  # user's query terms and a memory's text terms. Memories below this
34
34
  # threshold are dropped before formatting. 0.0 disables the gate
35
35
  # (current behaviour: inject everything Hindsight returns up to the
36
- # count cap). Hindsight's HTTP API does not expose similarity
37
- # scores, so this is the switchroom-side quality filter — see #475.
36
+ # count cap). NOTE: Hindsight's HTTP recall API DOES return per-result
37
+ # relevance scores (`scores.final`, plus `.semantic`/`.keyword`/
38
+ # `.reranker`) — verified at runtime — and recall.py now reads and
39
+ # sorts the merged set by `scores.final`. This Jaccard gate is a
40
+ # separate lexical-overlap quality filter layered on top — see #475.
38
41
  "recallMinOverlap": 0.0,
39
42
  "recallTypes": ["world", "experience"],
43
+ # Switchroom-local: when True (default; Ken-approved ON) recall biases
44
+ # toward synthesized `observation`-tier facts. Escape hatch: pin off via
45
+ # `recallPreferObservations: false` in the user config — read in recall.py.
46
+ "recallPreferObservations": True,
40
47
  # Switchroom #2848 Stage B/C — deterministic directive capture.
41
48
  # When on (switchroom default; pinned true in the copied plugin
42
49
  # settings.json by applyHindsightSettingsOverrides), TWO deterministic
@@ -373,12 +373,18 @@ def _is_demoted_memory(memory) -> bool:
373
373
 
374
374
  # Switchroom #475 — lexical-overlap relevance gate.
375
375
  #
376
- # Hindsight's HTTP API does not return similarity scores. Without a
377
- # score the existing `recallMaxMemories` cap acts as a *floor* on
378
- # low-relevance prompts: weak matches still fill the slot up to N,
379
- # mis-steering the model. This gate computes Jaccard overlap between
380
- # the user's query terms and each memory's text terms, and drops
381
- # memories below a configurable threshold.
376
+ # Hindsight's HTTP recall API DOES return per-result relevance scores
377
+ # (`RecallResult.scores.final`, plus `.semantic`/`.keyword`/`.reranker`);
378
+ # the merged multi-bank set is now sorted by `scores.final` before the
379
+ # `recallMaxMemories` cap (see the sort just before the cap in
380
+ # process_recall) so the most relevant memories survive the head-slice
381
+ # regardless of which bank they came from. This gate is a *complementary*,
382
+ # opt-in absolute precision floor: `scores.final` is a relative rank that
383
+ # still orders weakly-matching memories rather than excluding them, so on a
384
+ # low-relevance prompt the top-N could still be low-signal. The Jaccard
385
+ # overlap between the user's query terms and each memory's text terms is a
386
+ # query-independent absolute measure that drops memories below a
387
+ # configurable threshold outright — something the relative sort does not do.
382
388
  #
383
389
  # Threshold default is 0.0 (disabled) so the gate is opt-in initially.
384
390
  # Operators tune via `memory.recall.min_overlap` in switchroom.yaml or
@@ -469,6 +475,40 @@ def _filter_by_overlap(results, query: str, threshold: float):
469
475
  return kept, dropped
470
476
 
471
477
 
478
+ def _result_final_score(m) -> float:
479
+ """Return a result's engine relevance score (`scores.final`).
480
+
481
+ Switchroom Phase-1 precision. The Hindsight recall response attaches a
482
+ `scores` object to every result whose required `final` field is the
483
+ engine's combined ranking score (reranker + recency/temporal/proof
484
+ boosts). Results missing a usable score sort last so a malformed or
485
+ score-less entry can never starve a properly-ranked one.
486
+ """
487
+ if isinstance(m, dict):
488
+ scores = m.get("scores")
489
+ if isinstance(scores, dict):
490
+ val = scores.get("final")
491
+ if isinstance(val, (int, float)) and not isinstance(val, bool):
492
+ return float(val)
493
+ return float("-inf")
494
+
495
+
496
+ def _sort_by_final_score(results):
497
+ """Sort merged multi-bank results by `scores.final` descending, in place.
498
+
499
+ Switchroom Phase-1 bank-starvation fix. The recall path appends
500
+ additional-bank (profile / shared / sender) results after the own-bank
501
+ results, then head-slices at `recallMaxMemories`. Before this sort, a
502
+ full own-bank result set silently dropped every additional-bank memory
503
+ at the cap regardless of relevance. Sorting by the engine's real
504
+ relevance score before the cap means the cap keeps the most relevant
505
+ memories cross-bank. Python's sort is stable, so ties preserve the
506
+ prior own-bank-first insertion order.
507
+ """
508
+ results.sort(key=_result_final_score, reverse=True)
509
+ return results
510
+
511
+
472
512
  def _write_recall_log(entry: dict) -> None:
473
513
  """Append a JSONL line to recall_log.jsonl. Bounded by line count.
474
514
 
@@ -986,6 +1026,11 @@ def main():
986
1026
  tags=recall_tags,
987
1027
  tags_match=tags_match,
988
1028
  tag_groups=tag_groups,
1029
+ # Switchroom Phase-1 precision — prefer deduped observation
1030
+ # statements over the raw facts they supersede, backfilling freed
1031
+ # slots for denser coverage inside the same budget. On by default;
1032
+ # operators can pin off via `recallPreferObservations: false`.
1033
+ prefer_observations=config.get("recallPreferObservations", True),
989
1034
  # 8s in-script timeout leaves 4s headroom inside the 12s
990
1035
  # UserPromptSubmit hook ceiling (see hooks.json:20) for cache
991
1036
  # write + block formatting. Tightened from 10s in switchroom
@@ -1031,6 +1076,10 @@ def main():
1031
1076
  tags=extra_tags,
1032
1077
  tags_match=extra_tags_match,
1033
1078
  tag_groups=extra_tag_groups,
1079
+ # Switchroom Phase-1 precision — prefer deduped observation
1080
+ # statements here too so additional banks contribute their
1081
+ # densest statements to the merged, score-sorted set.
1082
+ prefer_observations=config.get("recallPreferObservations", True),
1034
1083
  # 8s in-script timeout leaves 4s headroom inside the 12s
1035
1084
  # UserPromptSubmit hook ceiling (see hooks.json:20) for cache
1036
1085
  # write + block formatting. Tightened from 10s in switchroom
@@ -1099,6 +1148,15 @@ def main():
1099
1148
  else:
1100
1149
  overlap_dropped = 0
1101
1150
 
1151
+ # Switchroom Phase-1 precision — sort the merged primary + additional-bank
1152
+ # result set by the engine's relevance score (`scores.final`) descending
1153
+ # BEFORE the head-slice cap below. Previously additional-bank results were
1154
+ # appended after own-bank results and sliced off, silently starving
1155
+ # profile / shared / sender banks whenever own-bank filled the cap. Sorting
1156
+ # by real relevance first means the cap keeps the most relevant memories
1157
+ # regardless of source bank. Stable sort: ties keep own-bank-first order.
1158
+ _sort_by_final_score(results)
1159
+
1102
1160
  # Switchroom-local: client-side count cap. Plugin v0.4.0 has no
1103
1161
  # `recallTopK` in the Claude Code integration (Openclaw-only), and a
1104
1162
  # token budget alone doesn't bound count — a single long memory can
@@ -74,6 +74,7 @@ class _FakeClient:
74
74
  tags=None,
75
75
  tags_match=None,
76
76
  tag_groups=None,
77
+ prefer_observations=None,
77
78
  timeout=10,
78
79
  ):
79
80
  self.recall_calls.append(