switchroom 0.17.6 → 0.17.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +38 -4
- package/dist/auth-broker/index.js +302 -203
- package/dist/cli/notion-write-pretool.mjs +35 -2
- package/dist/cli/switchroom.js +1178 -576
- package/dist/host-control/main.js +148 -14
- package/dist/vault/approvals/kernel-server.js +140 -55
- package/dist/vault/broker/server.js +142 -57
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +50 -6
- package/profiles/default/CLAUDE.md +116 -0
- package/skills/mental-model-curator/SKILL.md +162 -0
- package/telegram-plugin/bridge/bridge.ts +80 -1
- package/telegram-plugin/bridge/ipc-client.ts +19 -0
- package/telegram-plugin/bridge/permission-ledger.ts +61 -0
- package/telegram-plugin/consolidation-legibility.ts +279 -0
- package/telegram-plugin/dist/bridge/bridge.js +85 -1
- package/telegram-plugin/dist/gateway/gateway.js +2565 -610
- package/telegram-plugin/dist/server.js +86 -2
- package/telegram-plugin/feed-heartbeat-climb.ts +206 -0
- package/telegram-plugin/gateway/activity-card-store.ts +293 -0
- package/telegram-plugin/gateway/gateway.ts +1376 -82
- package/telegram-plugin/gateway/inbound-spool.ts +22 -0
- package/telegram-plugin/gateway/mental-model-propose-card.ts +69 -0
- package/telegram-plugin/gateway/mental-model-propose-diff.ts +171 -0
- package/telegram-plugin/gateway/mental-model-propose-inbound-builders.ts +147 -0
- package/telegram-plugin/gateway/mental-model-propose-resolve.ts +201 -0
- package/telegram-plugin/gateway/missed-approvals-card.ts +161 -0
- package/telegram-plugin/gateway/missed-approvals-store.ts +167 -0
- package/telegram-plugin/gateway/permission-rearm.ts +115 -0
- package/telegram-plugin/gateway/scoped-grant-store.ts +89 -0
- package/telegram-plugin/memory-legibility.ts +217 -0
- package/telegram-plugin/node_modules/.vite/vitest/da39a3ee5e6b4b0d3255bfef95601890afd80709/results.json +1 -0
- package/telegram-plugin/scoped-approval.ts +59 -0
- package/telegram-plugin/silent-end.ts +78 -0
- package/telegram-plugin/subagent-watcher.ts +60 -6
- package/telegram-plugin/tests/activity-card-store.test.ts +436 -0
- package/telegram-plugin/tests/activity-card-wiring.test.ts +88 -0
- package/telegram-plugin/tests/consolidation-legibility.test.ts +224 -0
- package/telegram-plugin/tests/emission-authority-facade.test.ts +25 -10
- package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +33 -9
- package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
- package/telegram-plugin/tests/inbound-spool.test.ts +105 -0
- package/telegram-plugin/tests/memory-legibility.test.ts +216 -0
- package/telegram-plugin/tests/mental-model-propose-callback-gate.test.ts +67 -0
- package/telegram-plugin/tests/mental-model-propose-card.test.ts +56 -0
- package/telegram-plugin/tests/mental-model-propose-diff.test.ts +201 -0
- package/telegram-plugin/tests/mental-model-propose-inbound-builders.test.ts +68 -0
- package/telegram-plugin/tests/mental-model-propose-resolve.test.ts +157 -0
- package/telegram-plugin/tests/missed-approvals-card.test.ts +145 -0
- package/telegram-plugin/tests/missed-approvals-store.test.ts +147 -0
- package/telegram-plugin/tests/missed-approvals-wiring.test.ts +89 -0
- package/telegram-plugin/tests/permission-ledger.test.ts +166 -0
- package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +1 -1
- package/telegram-plugin/tests/permission-rearm-wiring.test.ts +175 -0
- package/telegram-plugin/tests/permission-rearm.test.ts +126 -0
- package/telegram-plugin/tests/scoped-grant-persist.test.ts +223 -0
- package/telegram-plugin/tests/silent-end-transport.test.ts +290 -0
- package/telegram-plugin/tests/silent-turn-climb-transport.test.ts +337 -0
- package/telegram-plugin/tests/subagent-watcher.test.ts +139 -0
- package/telegram-plugin/tests/worktree-watch-cwds.test.ts +103 -0
- package/telegram-plugin/uat/assertions.ts +88 -4
- package/telegram-plugin/uat/feed-matcher.test.ts +69 -0
- package/telegram-plugin/uat/scenarios/fuzz-liveness-climb-dm.test.ts +155 -0
- package/telegram-plugin/uat/scenarios/jtbd-directive-capture-nudge-dm.test.ts +185 -0
- package/telegram-plugin/uat/scenarios/jtbd-liveness-climb-channel.test.ts +192 -0
- package/telegram-plugin/uat/scenarios/jtbd-liveness-climb-dm.test.ts +220 -0
- package/telegram-plugin/uat/scenarios/jtbd-liveness-narration-channel.test.ts +137 -0
- package/telegram-plugin/uat/scenarios/jtbd-liveness-narration-dm.test.ts +148 -0
- package/telegram-plugin/uat/scenarios/jtbd-memory-legibility-channel.test.ts +66 -0
- package/telegram-plugin/uat/scenarios/jtbd-memory-legibility-dm.test.ts +61 -0
- package/telegram-plugin/uat/scenarios/silent-end-recovery-channel.test.ts +136 -0
- package/telegram-plugin/uat/scenarios/silent-end-recovery-dm.test.ts +24 -2
- package/telegram-plugin/worktree-watch-cwds.ts +60 -0
- package/vendor/hindsight-memory/hooks/hooks.json +9 -0
- package/vendor/hindsight-memory/scripts/__pycache__/directive_verify.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/drain_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/recall.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/retain.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/__pycache__/session_end.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/directive_verify.py +445 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/__init__.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/bank.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/client.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/config.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/content.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/daemon.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/directives.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/gateway_ipc.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/llm.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/state.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/__pycache__/switchroom_envelope.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +37 -0
- package/vendor/hindsight-memory/scripts/lib/directives.py +88 -0
- package/vendor/hindsight-memory/scripts/lib/switchroom_envelope.py +77 -0
- package/vendor/hindsight-memory/scripts/recall.py +153 -4
- package/vendor/hindsight-memory/scripts/retain.py +17 -0
- package/vendor/hindsight-memory/scripts/setup_hooks.py +9 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/__init__.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_config_client_casts.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_capture_nudge.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directive_verify.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_directives.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_gateway_ipc.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_context_slice.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_integration.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_tag_filters.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_topic_filter.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_recall_trivial_skip.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_retain_window.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_sender_routing.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/__pycache__/test_switchroom_envelope.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/scripts/tests/test_directive_capture_nudge.py +185 -0
- package/vendor/hindsight-memory/scripts/tests/test_directive_verify.py +516 -0
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +49 -0
- package/vendor/hindsight-memory/scripts/tests/test_retain_window.py +66 -1
- package/vendor/hindsight-memory/scripts/tests/test_switchroom_envelope.py +69 -0
- package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.0.3.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/conftest.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_bank.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_client.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.0.3.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_config.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_content.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_drain_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_hooks.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_manifest.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_recall_exit_codes.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_session_end_pending.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313-pytest-9.1.1.pyc +0 -0
- package/vendor/hindsight-memory/tests/__pycache__/test_state.cpython-313.pyc +0 -0
- package/vendor/hindsight-memory/tests/test_recall_exit_codes.py +49 -2
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sparse, chat-legible memory surface — hindsight Phase 4 (#2849).
|
|
3
|
+
*
|
|
4
|
+
* When the interactive `claude` session materially changes what it
|
|
5
|
+
* remembers during a turn — stores a new standing directive, or
|
|
6
|
+
* invalidates / demotes an existing memory — this module surfaces ONE
|
|
7
|
+
* terse line in the ORIGINATING Telegram chat/topic:
|
|
8
|
+
*
|
|
9
|
+
* 📌 <i>remembered:</i> "Always prefer TypeScript for this user"
|
|
10
|
+
* ✂️ <i>forgot:</i> superseded deploy runbook
|
|
11
|
+
*
|
|
12
|
+
* Why so sparse: the `remember-across-sessions` job spec lists
|
|
13
|
+
* "regurgitating old facts unprompted just to prove it remembered" as a
|
|
14
|
+
* top anti-pattern. A per-turn line would *become* that anti-pattern, so
|
|
15
|
+
* this surface fires ONLY on a genuine store/correct — never on ordinary
|
|
16
|
+
* recall, and never on routine consolidation. Detection is a
|
|
17
|
+
* deterministic tool-call observation (no model call, no polling): we
|
|
18
|
+
* watch the main-agent turn stream for the specific hindsight memory
|
|
19
|
+
* tools that represent a material change.
|
|
20
|
+
*
|
|
21
|
+
* Design notes / references:
|
|
22
|
+
* - RFC Phase 4: `reference/rfcs/hindsight-synthesis-layers.md`
|
|
23
|
+
* - Job spec: `reference/jobs/remember-across-sessions.md`
|
|
24
|
+
* - The `consolidation.completed` webhook the RFC names as the poll-free
|
|
25
|
+
* driver for the "updated what I know about Y" side does NOT exist in
|
|
26
|
+
* the pinned hindsight image (v0.8.4) — it is RFC-only. This v1 ships
|
|
27
|
+
* the tool-observation path; the webhook stays a follow-up.
|
|
28
|
+
*
|
|
29
|
+
* Reuse: `classifyMemoryToolCall` / `detectMemoryLegibilityEvent` are the
|
|
30
|
+
* single, reusable "did a directive-create / invalidate happen in this
|
|
31
|
+
* turn?" primitive on the TypeScript gateway side. Phase 3 Stage B (#2848)
|
|
32
|
+
* runs as a vendored Python hook (`recall.py`) and CANNOT import this TS
|
|
33
|
+
* module — it mirrors the same hindsight tool-name matching independently.
|
|
34
|
+
* Keep the two in sync by hand when the tool surface changes.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import { stripMarkdown, truncate } from './card-format.js'
|
|
38
|
+
|
|
39
|
+
/** ON by default; an operator opts out with SWITCHROOM_MEMORY_LEGIBILITY=0.
|
|
40
|
+
* Mirrors the SWITCHROOM_WORKER_ACTIVITY_FEED convention (only the literal
|
|
41
|
+
* string '0' disables; unset / anything else → enabled). */
|
|
42
|
+
export function isMemoryLegibilityEnabled(envVal: string | undefined): boolean {
|
|
43
|
+
return envVal !== '0'
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* The material memory operations we surface. Deliberately narrow — an
|
|
48
|
+
* ordinary `recall` / `reflect` / benign `update_memory` (a non-demote
|
|
49
|
+
* edit) is NOT material and returns null from the classifier.
|
|
50
|
+
*/
|
|
51
|
+
export type MemoryToolClass = 'directive-create' | 'memory-invalidate'
|
|
52
|
+
|
|
53
|
+
/** The two hindsight tools that represent a material store / correct.
|
|
54
|
+
* Fully-qualified names as they appear in the turn stream (`ev.toolName`). */
|
|
55
|
+
export const CREATE_DIRECTIVE_TOOL = 'mcp__hindsight__create_directive'
|
|
56
|
+
export const INVALIDATE_MEMORY_TOOL = 'mcp__hindsight__invalidate_memory'
|
|
57
|
+
/** The demote path (`switchroom memory demote`) adds this client-side tag
|
|
58
|
+
* via the `update_memory` tool; that specific tagging is the "forgot"
|
|
59
|
+
* signal, whereas a plain `update_memory` edit is not material. */
|
|
60
|
+
export const UPDATE_MEMORY_TOOL = 'mcp__hindsight__update_memory'
|
|
61
|
+
export const DEMOTE_FROM_RECALL_MARKER = 'demote-from-recall'
|
|
62
|
+
|
|
63
|
+
export type MemoryLegibilityKind = 'remembered' | 'forgot'
|
|
64
|
+
|
|
65
|
+
export interface MemoryLegibilityEvent {
|
|
66
|
+
kind: MemoryLegibilityKind
|
|
67
|
+
/** Best-effort human detail (directive body / invalidate reason). May be
|
|
68
|
+
* empty when the tool carries no legible detail — the render then falls
|
|
69
|
+
* back to a bare "forgot a memory" line. Raw (unescaped, unstripped);
|
|
70
|
+
* `renderMemoryLegibilityLine` does the cleanup. */
|
|
71
|
+
detail: string
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function asString(v: unknown): string {
|
|
75
|
+
return typeof v === 'string' ? v : ''
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** True when an `update_memory` call's tags carry the demote-from-recall
|
|
79
|
+
* marker — the only shape of `update_memory` that is a material correction.
|
|
80
|
+
* Bracket- and case-tolerant (`[demote-from-recall]` / `demote-from-recall`). */
|
|
81
|
+
function hasDemoteTag(input: Record<string, unknown> | undefined): boolean {
|
|
82
|
+
if (input == null) return false
|
|
83
|
+
const pools: unknown[] = []
|
|
84
|
+
for (const key of ['add_tags', 'tags']) {
|
|
85
|
+
const val = input[key]
|
|
86
|
+
if (Array.isArray(val)) pools.push(...val)
|
|
87
|
+
else if (typeof val === 'string') pools.push(val)
|
|
88
|
+
}
|
|
89
|
+
return pools.some(
|
|
90
|
+
(t) => typeof t === 'string' && t.toLowerCase().includes(DEMOTE_FROM_RECALL_MARKER),
|
|
91
|
+
)
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Classify a single tool call as a material memory operation, or null if
|
|
96
|
+
* it isn't one. This is the reusable detection primitive: it answers "did a
|
|
97
|
+
* directive-create / invalidate happen in this turn?" with no rendering
|
|
98
|
+
* concern. #2848 Stage B (the vendored Python `recall.py` hook) mirrors this
|
|
99
|
+
* matching independently — it cannot import this TS module.
|
|
100
|
+
*/
|
|
101
|
+
export function classifyMemoryToolCall(
|
|
102
|
+
toolName: string,
|
|
103
|
+
input: Record<string, unknown> | undefined,
|
|
104
|
+
): MemoryToolClass | null {
|
|
105
|
+
if (toolName === CREATE_DIRECTIVE_TOOL) return 'directive-create'
|
|
106
|
+
if (toolName === INVALIDATE_MEMORY_TOOL) return 'memory-invalidate'
|
|
107
|
+
// `update_memory` is material ONLY when it applies the demote-from-recall
|
|
108
|
+
// tag; any other edit is routine and must not surface a line.
|
|
109
|
+
if (toolName === UPDATE_MEMORY_TOOL && hasDemoteTag(input)) return 'memory-invalidate'
|
|
110
|
+
return null
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Detect a surfaceable memory-legibility event from one tool call, or null.
|
|
115
|
+
* Pure — no I/O. Extracts the best-available human detail from the tool
|
|
116
|
+
* input so the caller can render + route it.
|
|
117
|
+
*/
|
|
118
|
+
export function detectMemoryLegibilityEvent(
|
|
119
|
+
toolName: string,
|
|
120
|
+
input: Record<string, unknown> | undefined,
|
|
121
|
+
): MemoryLegibilityEvent | null {
|
|
122
|
+
const klass = classifyMemoryToolCall(toolName, input)
|
|
123
|
+
if (klass == null) return null
|
|
124
|
+
if (klass === 'directive-create') {
|
|
125
|
+
// The live server requires `content` (directive body); the profile
|
|
126
|
+
// guidance example writes `create_directive(text)`, so tolerate both,
|
|
127
|
+
// then fall back to the directive `name`.
|
|
128
|
+
const detail =
|
|
129
|
+
asString(input?.content).trim() ||
|
|
130
|
+
asString(input?.text).trim() ||
|
|
131
|
+
asString(input?.name).trim()
|
|
132
|
+
return { kind: 'remembered', detail }
|
|
133
|
+
}
|
|
134
|
+
// memory-invalidate: prefer a human `reason`; the opaque memory_id is not
|
|
135
|
+
// user-legible, so omit it and let the render use the bare fallback.
|
|
136
|
+
const detail = asString(input?.reason).trim()
|
|
137
|
+
return { kind: 'forgot', detail }
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Confirm-before-legibility stager (Fix 1.3, #2903).
|
|
142
|
+
*
|
|
143
|
+
* A material memory op is detected on the `tool_use` event, but the write is
|
|
144
|
+
* not yet confirmed — the hindsight `tools/call` can still fail (engine down,
|
|
145
|
+
* `isError` envelope) and return an error `tool_result`. Sending the 📌/✂️ line
|
|
146
|
+
* on the tool_use would claim "remembered" for a write that then failed.
|
|
147
|
+
*
|
|
148
|
+
* This stager holds detected events keyed by `toolUseId`. `confirm` returns the
|
|
149
|
+
* event to send ONLY on a successful result; a failed result (`isError`) drops
|
|
150
|
+
* it and returns null. Pure and I/O-free so the send/no-send decision is unit-
|
|
151
|
+
* testable independent of the gateway's Telegram plumbing.
|
|
152
|
+
*
|
|
153
|
+
* `M` is caller-side routing metadata (chat/thread) carried opaquely.
|
|
154
|
+
*/
|
|
155
|
+
export class MemoryLegibilityStager<M> {
|
|
156
|
+
private pending = new Map<string, { event: MemoryLegibilityEvent; meta: M }>()
|
|
157
|
+
/** Bound the map so a turn that never emits a matching tool_result (crash
|
|
158
|
+
* mid-tool) can't leak entries unboundedly. */
|
|
159
|
+
constructor(private readonly cap = 256) {}
|
|
160
|
+
|
|
161
|
+
/** Stage a detected event awaiting result confirmation. */
|
|
162
|
+
stage(toolUseId: string, event: MemoryLegibilityEvent, meta: M): void {
|
|
163
|
+
if (this.pending.size >= this.cap) {
|
|
164
|
+
const oldest = this.pending.keys().next().value
|
|
165
|
+
if (oldest != null) this.pending.delete(oldest)
|
|
166
|
+
}
|
|
167
|
+
this.pending.set(toolUseId, { event, meta })
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Resolve a staged event on its matching tool_result. Returns the event +
|
|
172
|
+
* meta to send on CONFIRMED success; returns null when the write errored,
|
|
173
|
+
* when nothing was staged for this id, or when the id is empty. Always
|
|
174
|
+
* consumes the staged entry.
|
|
175
|
+
*/
|
|
176
|
+
confirm(
|
|
177
|
+
toolUseId: string | null | undefined,
|
|
178
|
+
isError: boolean | undefined,
|
|
179
|
+
): { event: MemoryLegibilityEvent; meta: M } | null {
|
|
180
|
+
if (toolUseId == null || toolUseId.length === 0) return null
|
|
181
|
+
const staged = this.pending.get(toolUseId)
|
|
182
|
+
if (staged == null) return null
|
|
183
|
+
this.pending.delete(toolUseId)
|
|
184
|
+
if (isError === true) return null
|
|
185
|
+
return staged
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** Test/introspection helper: number of currently-staged events. */
|
|
189
|
+
get size(): number {
|
|
190
|
+
return this.pending.size
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** HTML-escape for parse_mode:'HTML' (escape the 3 entity-significant chars). */
|
|
195
|
+
function escapeHtml(s: string): string {
|
|
196
|
+
return s.replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>')
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** Max chars of directive/reason detail shown on the one-liner. */
|
|
200
|
+
const DETAIL_MAX = 160
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Render the terse one-line surface (Telegram HTML). Callers send this as a
|
|
204
|
+
* real `sendMessage` (not a draft/reaction) so it's observable + durable.
|
|
205
|
+
*
|
|
206
|
+
* 📌 <i>remembered:</i> "<directive, cleaned + truncated>"
|
|
207
|
+
* ✂️ <i>forgot:</i> <reason> (or "✂️ <i>forgot a memory.</i>")
|
|
208
|
+
*/
|
|
209
|
+
export function renderMemoryLegibilityLine(ev: MemoryLegibilityEvent): string {
|
|
210
|
+
const clean = truncate(stripMarkdown(ev.detail).replace(/\s+/g, ' ').trim(), DETAIL_MAX)
|
|
211
|
+
if (ev.kind === 'remembered') {
|
|
212
|
+
if (clean.length === 0) return `📌 <i>remembered a new directive.</i>`
|
|
213
|
+
return `📌 <i>remembered:</i> "${escapeHtml(clean)}"`
|
|
214
|
+
}
|
|
215
|
+
if (clean.length === 0) return `✂️ <i>forgot a memory.</i>`
|
|
216
|
+
return `✂️ <i>forgot:</i> ${escapeHtml(clean)}`
|
|
217
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":"3.2.4","results":[[":tests/worker-activity-feed.test.ts",{"duration":604.6399409999995,"failed":true}]]}
|
|
@@ -214,6 +214,65 @@ export function sweepScopedGrants(store: ScopedGrantStore, now: number): void {
|
|
|
214
214
|
}
|
|
215
215
|
}
|
|
216
216
|
|
|
217
|
+
/** Total number of live-or-not grant entries across all agents. Cheap change
|
|
218
|
+
* signal for "did a sweep remove anything?" (sweeps only ever remove). */
|
|
219
|
+
export function countScopedGrants(store: ScopedGrantStore): number {
|
|
220
|
+
let n = 0;
|
|
221
|
+
for (const list of store.values()) n += list.length;
|
|
222
|
+
return n;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* A single grant flattened for on-disk persistence. Carries the agent it
|
|
227
|
+
* belongs to (the store is keyed by agent), the narrow rule/signature, and
|
|
228
|
+
* the ABSOLUTE wall-clock expiry — never a relative TTL, so a reload can
|
|
229
|
+
* never extend a window (the restart-doesn't-extend invariant). No grant
|
|
230
|
+
* time is stored because nothing needs it: expiry is the only gate.
|
|
231
|
+
*/
|
|
232
|
+
export interface SerializedScopedGrant {
|
|
233
|
+
readonly agent: string;
|
|
234
|
+
readonly rule: string;
|
|
235
|
+
readonly expiresAt: number;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/** Flatten the per-agent store to a plain array for JSON persistence. */
|
|
239
|
+
export function serializeScopedGrants(store: ScopedGrantStore): SerializedScopedGrant[] {
|
|
240
|
+
const out: SerializedScopedGrant[] = [];
|
|
241
|
+
for (const [agent, list] of store) {
|
|
242
|
+
for (const g of list) out.push({ agent, rule: g.rule, expiresAt: g.expiresAt });
|
|
243
|
+
}
|
|
244
|
+
return out;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* Rebuild the store from persisted entries at gateway boot. Entries already
|
|
249
|
+
* at/past `now` are DROPPED (expiry is absolute — a restart never revives or
|
|
250
|
+
* extends a window). Malformed rows are skipped defensively (fail-closed:
|
|
251
|
+
* an unparseable grant simply doesn't exist → the action re-cards). A
|
|
252
|
+
* duplicate rule for one agent collapses to the latest expiry, mirroring
|
|
253
|
+
* `recordScopedGrant`'s replace-not-accumulate behaviour.
|
|
254
|
+
*/
|
|
255
|
+
export function deserializeScopedGrants(
|
|
256
|
+
data: readonly unknown[],
|
|
257
|
+
now: number,
|
|
258
|
+
): ScopedGrantStore {
|
|
259
|
+
const store: ScopedGrantStore = new Map();
|
|
260
|
+
if (!Array.isArray(data)) return store;
|
|
261
|
+
for (const raw of data) {
|
|
262
|
+
if (!raw || typeof raw !== "object") continue;
|
|
263
|
+
const entry = raw as Partial<SerializedScopedGrant>;
|
|
264
|
+
if (typeof entry.agent !== "string" || entry.agent.length === 0) continue;
|
|
265
|
+
if (typeof entry.rule !== "string" || entry.rule.length === 0) continue;
|
|
266
|
+
if (typeof entry.expiresAt !== "number" || !Number.isFinite(entry.expiresAt)) continue;
|
|
267
|
+
if (entry.expiresAt <= now) continue; // absolute expiry already elapsed → drop
|
|
268
|
+
const list = store.get(entry.agent) ?? [];
|
|
269
|
+
const others = list.filter((g) => g.rule !== entry.rule);
|
|
270
|
+
others.push({ rule: entry.rule, expiresAt: entry.expiresAt });
|
|
271
|
+
store.set(entry.agent, others);
|
|
272
|
+
}
|
|
273
|
+
return store;
|
|
274
|
+
}
|
|
275
|
+
|
|
217
276
|
/**
|
|
218
277
|
* Heuristic destructive/irreversible-command detector. FAIL-CLOSED: when
|
|
219
278
|
* a command can't be read or looks risky, return `true` so it is never
|
|
@@ -49,6 +49,17 @@ export interface SilentEndDeps {
|
|
|
49
49
|
stateDir?: string
|
|
50
50
|
/** stderr writer (defaults to `process.stderr.write`). */
|
|
51
51
|
log?: (line: string) => void
|
|
52
|
+
/**
|
|
53
|
+
* Has a genuine assistant reply been delivered to this chat (optionally
|
|
54
|
+
* scoped to thread) at or after `sinceMs`? Same predicate shape as
|
|
55
|
+
* `history.hasOutboundDeliveredSince` / the represent-guard's dep
|
|
56
|
+
* (`gateway/represent-guard.ts`) — injected here so the exhaustion check
|
|
57
|
+
* below is a pure, testable decision. When omitted (history unavailable,
|
|
58
|
+
* or callers that don't wire it), the check is skipped and the guard
|
|
59
|
+
* falls back to the pre-existing turnKey/retryCount bookkeeping only —
|
|
60
|
+
* never suppress on doubt.
|
|
61
|
+
*/
|
|
62
|
+
hasOutboundDeliveredSince?: (chatId: string, sinceMs: number, threadId?: number | null) => boolean
|
|
52
63
|
}
|
|
53
64
|
|
|
54
65
|
/**
|
|
@@ -69,6 +80,35 @@ export interface SilentEndDeps {
|
|
|
69
80
|
*/
|
|
70
81
|
export const SILENT_END_MAX_RETRIES = 2
|
|
71
82
|
|
|
83
|
+
/**
|
|
84
|
+
* User-facing fallback text delivered when a user-message turn ends with no
|
|
85
|
+
* final answer AND the deterministic Stop-hook re-prompt has already been
|
|
86
|
+
* exhausted (#1161). Without this the user only sees the progress card
|
|
87
|
+
* vanish; silence must never be the failure mode.
|
|
88
|
+
*
|
|
89
|
+
* PR #2892 (reference/rfcs/deterministic-turn-liveness.md Phase 2
|
|
90
|
+
* hardening): include the turn's elapsed so the fallback is honest about
|
|
91
|
+
* how long the user actually waited, instead of a generic apology with no
|
|
92
|
+
* timing. A degenerate/unknown duration (missing, non-finite, or <= 0 —
|
|
93
|
+
* e.g. a turn whose `startedAt` was never stamped) omits the waited clause
|
|
94
|
+
* rather than printing a nonsensical "(waited 0s)".
|
|
95
|
+
*
|
|
96
|
+
* Lives here (not gateway.ts) so the transport-boundary tests exercise the
|
|
97
|
+
* REAL string (tests/silent-end-transport.test.ts), not a hand-maintained
|
|
98
|
+
* copy — gateway.ts is not importable in tests.
|
|
99
|
+
*/
|
|
100
|
+
export function silentEndFallbackText(turnDurationMs: number | undefined): string {
|
|
101
|
+
const elapsed =
|
|
102
|
+
typeof turnDurationMs === 'number' && Number.isFinite(turnDurationMs) && turnDurationMs > 0
|
|
103
|
+
? ` (waited ${Math.round(turnDurationMs / 1000)}s)`
|
|
104
|
+
: ''
|
|
105
|
+
return (
|
|
106
|
+
'⚠️ The agent finished working but didn’t send a reply' +
|
|
107
|
+
elapsed +
|
|
108
|
+
' — your last message may not have been answered. Please try asking again.'
|
|
109
|
+
)
|
|
110
|
+
}
|
|
111
|
+
|
|
72
112
|
function resolveStateDir(deps?: SilentEndDeps): string {
|
|
73
113
|
if (deps?.stateDir != null) return deps.stateDir
|
|
74
114
|
const env = process.env.TELEGRAM_STATE_DIR
|
|
@@ -255,6 +295,44 @@ export function recordSilentTurnEnd(
|
|
|
255
295
|
prev.turnKey === args.turnKey &&
|
|
256
296
|
prev.retryCount >= SILENT_END_MAX_RETRIES
|
|
257
297
|
) {
|
|
298
|
+
// Staleness guard (mirrors the represent-guard's #2472 fix,
|
|
299
|
+
// `gateway/represent-guard.ts:shouldSuppressRepresent`): `turnKey` here
|
|
300
|
+
// is `statusKey(chatId, threadId)` — stable across every turn on the
|
|
301
|
+
// same chat/thread, NOT a per-turn nonce (unlike the obligation
|
|
302
|
+
// ledger's `originTurnId`). The whole mechanism depends on a reply
|
|
303
|
+
// ALWAYS clearing this state file via `clearSilentEndState` at the
|
|
304
|
+
// send-site. `clearSilentEndState` is fail-silent by design (state
|
|
305
|
+
// corruption / a write race must never crash the gateway), so if a
|
|
306
|
+
// clear is ever missed, a stale `retryCount >= MAX_RETRIES` record
|
|
307
|
+
// from an OLD, already-answered turn would be misread as "this
|
|
308
|
+
// brand-new dark turn already exhausted its re-prompt budget" —
|
|
309
|
+
// firing the user-facing fallback immediately, without the turn ever
|
|
310
|
+
// going through the Stop-hook re-prompt ladder. Before trusting that
|
|
311
|
+
// reading, verify against real delivery history: has a genuine reply
|
|
312
|
+
// landed on this chat/thread since the stale record was last written?
|
|
313
|
+
// If so, the record is satisfied-but-misdetected — drop it silently
|
|
314
|
+
// and let this dark turn start its OWN fresh retry cycle instead of
|
|
315
|
+
// inheriting someone else's spent budget.
|
|
316
|
+
if (deps?.hasOutboundDeliveredSince?.(args.chatId, prev.timestamp, args.threadId)) {
|
|
317
|
+
emitLog(
|
|
318
|
+
deps,
|
|
319
|
+
`silent-end: stale exhausted record for turnKey=${args.turnKey} ` +
|
|
320
|
+
`(retryCount=${prev.retryCount}) but a reply was delivered since ` +
|
|
321
|
+
`${prev.timestamp} — treating as satisfied-but-misdetected, not exhausted\n`,
|
|
322
|
+
)
|
|
323
|
+
// MUST clear before writing (adversarial review of #2892):
|
|
324
|
+
// writeSilentEndState re-inherits retryCount whenever the on-disk
|
|
325
|
+
// turnKey matches — which it ALWAYS does here, since turnKey is the
|
|
326
|
+
// stable statusKey(chatId, threadId). Writing over the stale record
|
|
327
|
+
// directly would start the "fresh" ladder at retryCount=MAX: the
|
|
328
|
+
// Stop hook would see retryCount >= MAX_RETRIES and never re-prompt,
|
|
329
|
+
// while this call just returned exhausted:false so no fallback fires
|
|
330
|
+
// either — pure silence, strictly worse than the pre-fix behaviour.
|
|
331
|
+
// Clearing first makes the new record genuinely start at retryCount=0.
|
|
332
|
+
clearSilentEndState(args.turnKey, deps)
|
|
333
|
+
writeSilentEndState(args, deps)
|
|
334
|
+
return { exhausted: false }
|
|
335
|
+
}
|
|
258
336
|
clearSilentEndState(args.turnKey, deps)
|
|
259
337
|
emitLog(
|
|
260
338
|
deps,
|
|
@@ -211,6 +211,24 @@ export interface SubagentWatcherConfig {
|
|
|
211
211
|
* an agent's home pollutes the watcher with phantom registrations).
|
|
212
212
|
*/
|
|
213
213
|
agentCwd?: string
|
|
214
|
+
/**
|
|
215
|
+
* Additional cwds (beyond `agentCwd`) whose Claude-Code project-dir slug
|
|
216
|
+
* should also be watched. Gap 2 of `deterministic-turn-liveness.md`: a
|
|
217
|
+
* sub-agent dispatched with worktree isolation (`switchroom worktree
|
|
218
|
+
* claim`) runs with a *different* cwd than the parent agent, so its
|
|
219
|
+
* `agent-*.jsonl` lands under a different `.claude/projects/<slug>/`
|
|
220
|
+
* tree than `expectedProjectSlug` — the #1116 foreign-slug filter then
|
|
221
|
+
* silently skips it forever (no activity stamp, no `🛠 Worker` feed).
|
|
222
|
+
*
|
|
223
|
+
* Called fresh on every rescan tick (cheap — reads a handful of small
|
|
224
|
+
* JSON records from the worktree registry) so a worktree claimed or
|
|
225
|
+
* released mid-run is picked up/dropped without a watcher restart.
|
|
226
|
+
* Deterministic guarantee this preserves: only cwds this agent's own
|
|
227
|
+
* sub-agents can legitimately run in are ever added — a worktree record
|
|
228
|
+
* not owned by this agent is never included, so this does not widen the
|
|
229
|
+
* #1116 foreign-project protection into a wildcard watch.
|
|
230
|
+
*/
|
|
231
|
+
extraWatchCwdsProvider?: () => string[]
|
|
214
232
|
/**
|
|
215
233
|
* How often to re-scan for new subagent dirs (ms). Default 1000.
|
|
216
234
|
*/
|
|
@@ -1173,6 +1191,7 @@ export function startSubagentWatcher(config: SubagentWatcherConfig): SubagentWat
|
|
|
1173
1191
|
const expectedProjectSlug = config.agentCwd != null
|
|
1174
1192
|
? sanitizeCwdToProjectName(config.agentCwd)
|
|
1175
1193
|
: null
|
|
1194
|
+
const extraWatchCwdsProvider = config.extraWatchCwdsProvider ?? null
|
|
1176
1195
|
// One-shot logging: warn the first time a foreign slug is observed
|
|
1177
1196
|
// so silent regressions are visible without re-running with debug.
|
|
1178
1197
|
const warnedForeignSlugs = new Set<string>()
|
|
@@ -1706,17 +1725,52 @@ export function startSubagentWatcher(config: SubagentWatcherConfig): SubagentWat
|
|
|
1706
1725
|
projectDirs = fs.readdirSync(projectsRoot) as string[]
|
|
1707
1726
|
} catch { return }
|
|
1708
1727
|
|
|
1728
|
+
// Gap 2 (deterministic-turn-liveness.md): re-derive the set of
|
|
1729
|
+
// worktree-isolated slugs this agent's own sub-agents may legitimately
|
|
1730
|
+
// run in, fresh on every tick. Best-effort — a registry read hiccup
|
|
1731
|
+
// (e.g. the worktree dir not existing on an agent that never uses
|
|
1732
|
+
// worktrees) must never break the tail loop, so it just yields no
|
|
1733
|
+
// extra slugs for this tick.
|
|
1734
|
+
let allowedSlugs: Set<string> | null = null
|
|
1735
|
+
// Did the extra-cwd provider run cleanly this tick? A transient failure
|
|
1736
|
+
// (registry read hiccup) must NOT permanently mislabel an owned worktree
|
|
1737
|
+
// slug as "foreign": we still skip it this tick (it isn't provably ours
|
|
1738
|
+
// right now), but we do NOT latch the one-shot warning, so the next clean
|
|
1739
|
+
// tick re-includes and re-derives it instead of staying silently excluded.
|
|
1740
|
+
let providerOk = true
|
|
1741
|
+
if (expectedProjectSlug != null) {
|
|
1742
|
+
allowedSlugs = new Set([expectedProjectSlug])
|
|
1743
|
+
if (extraWatchCwdsProvider != null) {
|
|
1744
|
+
try {
|
|
1745
|
+
for (const cwd of extraWatchCwdsProvider()) {
|
|
1746
|
+
allowedSlugs.add(sanitizeCwdToProjectName(cwd))
|
|
1747
|
+
}
|
|
1748
|
+
} catch (err) {
|
|
1749
|
+
providerOk = false
|
|
1750
|
+
log?.(`subagent-watcher: extraWatchCwdsProvider error: ${(err as Error).message}`)
|
|
1751
|
+
}
|
|
1752
|
+
}
|
|
1753
|
+
}
|
|
1754
|
+
|
|
1709
1755
|
for (const pDir of projectDirs) {
|
|
1710
|
-
// Issue #1116: filter to the agent's own slug
|
|
1711
|
-
//
|
|
1712
|
-
//
|
|
1713
|
-
|
|
1714
|
-
|
|
1756
|
+
// Issue #1116: filter to the agent's own slug (plus, per Gap 2, any
|
|
1757
|
+
// worktree-isolated slugs this agent's own sub-agents may run in).
|
|
1758
|
+
// Skip foreign project dirs so their stale subagent JSONLs (which
|
|
1759
|
+
// Claude Code reaps mid-session) don't pollute the watcher's registry.
|
|
1760
|
+
if (allowedSlugs != null && !allowedSlugs.has(pDir)) {
|
|
1761
|
+
// A slug that is now allowed must clear any stale "foreign" latch, so a
|
|
1762
|
+
// slug that was transiently excluded (or later genuinely goes foreign)
|
|
1763
|
+
// re-warns rather than staying mislabeled forever.
|
|
1764
|
+
if (providerOk && !warnedForeignSlugs.has(pDir)) {
|
|
1715
1765
|
warnedForeignSlugs.add(pDir)
|
|
1716
|
-
|
|
1766
|
+
const allowed = [...allowedSlugs].join(', ')
|
|
1767
|
+
log?.(`subagent-watcher: skipping foreign project dir ${pDir} (allowed: ${allowed})`)
|
|
1717
1768
|
}
|
|
1718
1769
|
continue
|
|
1719
1770
|
}
|
|
1771
|
+
// Owned/allowed this tick — drop any prior foreign latch so a future
|
|
1772
|
+
// genuine foreign appearance of the same slug re-warns.
|
|
1773
|
+
warnedForeignSlugs.delete(pDir)
|
|
1720
1774
|
const projectPath = join(projectsRoot, pDir)
|
|
1721
1775
|
let sessionDirs: string[]
|
|
1722
1776
|
try {
|