talon-agent 3.8.0 → 3.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.8.0",
3
+ "version": "3.8.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -29,7 +29,6 @@ import {
29
29
  recordTokens,
30
30
  finalizeResponseText,
31
31
  formatUserPrompt,
32
- formatPromptWithRetrievedMemory,
33
32
  prepareSystemPrompt,
34
33
  extractSessionName,
35
34
  summarizeUsage,
@@ -259,18 +258,12 @@ export async function handleMessage(
259
258
  sessionEpoch: session.createdAt,
260
259
  });
261
260
 
262
- // Retrieved memory wraps the FORMATTED live prompt (Phase B): it stays
263
- // outside the frozen system prompt, so the first-turn concatenation below
264
- // keeps the boundary "cached systemPrompt, separator, live prompt wrapper".
265
- const prompt = formatPromptWithRetrievedMemory(
266
- formatUserPrompt({
267
- text,
268
- senderName: senderName ?? "user",
269
- isGroup,
270
- messageId,
271
- }),
272
- params.retrievedMemory,
273
- );
261
+ const prompt = formatUserPrompt({
262
+ text,
263
+ senderName: senderName ?? "user",
264
+ isGroup,
265
+ messageId,
266
+ });
274
267
 
275
268
  log("agent", `[${chatId}] <- (${text.length} chars)`);
276
269
  traceMessage(chatId, "in", text, { senderName, isGroup });
@@ -39,7 +39,6 @@ import {
39
39
  recordTokens,
40
40
  finalizeResponseText,
41
41
  formatUserPrompt,
42
- formatPromptWithRetrievedMemory,
43
42
  prepareSystemPrompt,
44
43
  extractSessionName,
45
44
  summarizeUsage,
@@ -90,18 +89,13 @@ export async function handleMessage(
90
89
  await ensureChatMcpServer(oc, chatId);
91
90
  await ensurePluginMcpServers(oc, chatId);
92
91
 
93
- // Build the prompt (time tag + sender + msg_id reference), then wrap it
94
- // with any retrieved memory (Phase B). The wrapper goes into the live user
95
- // part only — `system` below stays byte-identical to the prepared prompt.
96
- const prompt = formatPromptWithRetrievedMemory(
97
- formatUserPrompt({
98
- text,
99
- senderName: senderName ?? "user",
100
- isGroup,
101
- messageId,
102
- }),
103
- params.retrievedMemory,
104
- );
92
+ // Build the prompt (time tag + sender + msg_id reference)
93
+ const prompt = formatUserPrompt({
94
+ text,
95
+ senderName: senderName ?? "user",
96
+ isGroup,
97
+ messageId,
98
+ });
105
99
 
106
100
  // Per-session frozen prompt + OpenCode-specific delivery suffix
107
101
  const { text: systemPrompt } = prepareSystemPrompt({
@@ -84,7 +84,6 @@ export async function* handlerToEvents(
84
84
  senderName: params.senderName,
85
85
  isGroup: params.isGroup,
86
86
  messageId: params.messageId,
87
- retrievedMemory: params.retrievedMemory,
88
87
  onStreamDelta: (accumulated) => {
89
88
  if (typeof accumulated !== "string" || accumulated.length === 0) {
90
89
  return;
@@ -17,8 +17,6 @@
17
17
 
18
18
  // ── Query lifecycle (backend-internal) ──────────────────────────────────────
19
19
 
20
- import type { RetrievedMemory } from "../../core/agent-runtime/capabilities.js";
21
-
22
20
  /** Parameters for a backend AI query. */
23
21
  export type QueryParams = {
24
22
  chatId: string;
@@ -35,12 +33,6 @@ export type QueryParams = {
35
33
  * Provider message ID. Telegram is numeric; Discord snowflakes are strings.
36
34
  */
37
35
  messageId?: number | string;
38
- /**
39
- * Optional pre-retrieved memory slice for this turn (Phase B). Handlers
40
- * fold it into the live user prompt via `formatPromptWithRetrievedMemory`;
41
- * it must never reach `prepareSystemPrompt()` or a backend `system` field.
42
- */
43
- retrievedMemory?: RetrievedMemory;
44
36
  onStreamDelta?: (accumulated: string, phase?: "thinking" | "text") => void;
45
37
  onTextBlock?: (text: string) => Promise<void>;
46
38
  /**
@@ -47,12 +47,7 @@ export {
47
47
 
48
48
  export { registerTurnInterrupt, interruptChatTurn } from "./turn-interrupt.js";
49
49
 
50
- export {
51
- formatUserPrompt,
52
- formatPromptWithRetrievedMemory,
53
- RETRIEVED_MEMORY_DEFAULT_MAX_CHARS,
54
- type PromptFormatInputs,
55
- } from "./prompt-format.js";
50
+ export { formatUserPrompt, type PromptFormatInputs } from "./prompt-format.js";
56
51
 
57
52
  export {
58
53
  buildDeliveryContract,
@@ -15,7 +15,6 @@
15
15
  * DM (no msg_id): "[2026-05-15 11:01:23] actual text"
16
16
  */
17
17
 
18
- import type { RetrievedMemory } from "../../core/agent-runtime/capabilities.js";
19
18
  import { formatFullDatetime } from "../../util/time.js";
20
19
 
21
20
  // ── Public API ──────────────────────────────────────────────────────────────
@@ -62,74 +61,6 @@ export function formatUserPrompt(inputs: PromptFormatInputs): string {
62
61
  return joinNonEmpty(timeTag, inputs.text);
63
62
  }
64
63
 
65
- // ── Retrieved-memory wrapper (Phase B pre-retrieval) ────────────────────────
66
-
67
- /** Default hard cap on the injected memory block, provenance labels included. */
68
- export const RETRIEVED_MEMORY_DEFAULT_MAX_CHARS = 3000;
69
-
70
- /**
71
- * Wrap an already-formatted live user prompt with a bounded retrieved-memory
72
- * block. This is the ONLY place retrieved memory enters a prompt, and it
73
- * wraps the whole `formatUserPrompt(...)` output rather than rebuilding its
74
- * internals — the existing sender/time/msg_id wrapper stays intact inside the
75
- * `User message:` section.
76
- *
77
- * Contract (see docs/memory-phase-b-pre-retrieval.md):
78
- * - `memory` undefined or empty items → the prompt is returned
79
- * BYTE-IDENTICAL. Prompt-cache and prompt-format tests stay valid.
80
- * - Non-empty → emit `Relevant memory:` with one provenance-labelled line
81
- * per item, a blank line, `User message:`, then the original prompt.
82
- * - The memory block (labels included) is capped at `maxChars`; item text
83
- * is truncated deterministically with an ellipsis marker. The user
84
- * message itself is NEVER dropped or truncated.
85
- * - This block is dynamic turn context: callers must keep it out of
86
- * `prepareSystemPrompt()`, prompt additions, and backend `system` fields.
87
- */
88
- export function formatPromptWithRetrievedMemory(
89
- prompt: string,
90
- memory?: RetrievedMemory,
91
- maxChars: number = RETRIEVED_MEMORY_DEFAULT_MAX_CHARS,
92
- ): string {
93
- if (!memory || memory.items.length === 0) return prompt;
94
-
95
- const header = "Relevant memory:";
96
- const footer = "User message:";
97
- // Budget applies to the memory block only (header + item lines), so the
98
- // user message can never be squeezed out.
99
- let budget = Math.max(0, maxChars) - header.length - 1; // "\n" after header
100
- const lines: string[] = [];
101
- for (const item of memory.items) {
102
- const label = provenanceLabel(item.wing, item.room, item.sourceFile);
103
- const prefix = `- ${label} `;
104
- if (prefix.length >= budget) break;
105
- const text = sanitizeInline(item.text);
106
- const room = budget - prefix.length - 1; // "\n" for this line
107
- const body =
108
- text.length <= room ? text : `${text.slice(0, Math.max(0, room - 1))}…`;
109
- if (body.length === 0) break;
110
- const line = `${prefix}${body}`;
111
- lines.push(line);
112
- budget -= line.length + 1;
113
- }
114
- if (lines.length === 0) return prompt;
115
-
116
- return `${header}\n${lines.join("\n")}\n\n${footer}\n${prompt}`;
117
- }
118
-
119
- function provenanceLabel(
120
- wing: string,
121
- room?: string,
122
- sourceFile?: string,
123
- ): string {
124
- const path = room ? `${wing}/${room}` : wing;
125
- return sourceFile ? `[${path} ${sourceFile}]` : `[${path}]`;
126
- }
127
-
128
- /** Collapse newlines/control whitespace so one item stays one labelled line. */
129
- function sanitizeInline(text: string): string {
130
- return text.replace(/\s+/g, " ").trim();
131
- }
132
-
133
64
  // ── Helpers ─────────────────────────────────────────────────────────────────
134
65
 
135
66
  function joinNonEmpty(...parts: string[]): string {
@@ -35,54 +35,16 @@ import type {
35
35
 
36
36
  // ── Run parameters ──────────────────────────────────────────────────────────
37
37
 
38
- /**
39
- * Provenance trust level of a retrieved memory item, per the memory-poisoning
40
- * threat model (#373). Only the first three levels are ever eligible for
41
- * automatic injection; `user_claim` and `group_chat` content must stay
42
- * pull-only (explicit search), never auto-injected.
43
- */
44
- export type RetrievedMemoryTrustLevel =
45
- | "dylan_direct" // stated by the operator in a verified DM
46
- | "bot_inferred" // inferred by the bot from code/docs/verified primary source
47
- | "heartbeat_synthesis" // synthesized in a background run, no external input
48
- | "user_claim" // claimed by a non-operator user, unverified
49
- | "group_chat"; // sourced from group chat content
50
-
51
- /** One retrieved memory fragment with its provenance. */
52
- export interface RetrievedMemoryItem {
53
- /** Palace wing (top-level category), e.g. "technical". */
54
- wing: string;
55
- /** Palace room within the wing, when known. */
56
- room?: string;
57
- /** Source file locator, when known (e.g. "memory-phase-b.md"). */
58
- sourceFile?: string;
59
- /** The retrieved text fragment. */
60
- text: string;
61
- /** Retrieval relevance score, when the retriever provides one. */
62
- score?: number;
63
- /** Provenance trust level; absent means unknown (treat as untrusted). */
64
- trustLevel?: RetrievedMemoryTrustLevel;
65
- }
66
-
67
- /**
68
- * A bounded, sanitized slice of long-term memory retrieved for one turn.
69
- * This is DYNAMIC turn context: it is injected into the live user prompt by
70
- * the backend prompt formatter and must never enter `prepareSystemPrompt()`
71
- * output, frozen prompt snapshots, plugin prompt additions, or backend
72
- * `system` fields — that would break the prompt-cache contract.
73
- */
74
- export interface RetrievedMemory {
75
- source: "mempalace";
76
- /** The (possibly trimmed) query the retriever ran. */
77
- query: string;
78
- items: RetrievedMemoryItem[];
79
- }
80
-
81
38
  /**
82
39
  * Parameters for a chat turn. `model` is a resolved `ModelRef`,
83
40
  * carrying everything the backend needs to identify the model and
84
41
  * render the resulting reply. Streaming callbacks aren't part of
85
42
  * this shape — backends emit `AgentEvent`s.
43
+ *
44
+ * Long-term memory is deliberately NOT part of this shape. `memory.md` is
45
+ * loaded once into the cached system prompt (`core/prompt/assemble.ts`);
46
+ * anything deeper the model searches for itself with the MemPalace tools.
47
+ * Nothing is auto-injected into a turn.
86
48
  */
87
49
  export interface ChatRunParams {
88
50
  chatId: string;
@@ -92,12 +54,6 @@ export interface ChatRunParams {
92
54
  isGroup?: boolean;
93
55
  /** Provider message ID. Telegram is numeric; Discord snowflakes are strings. */
94
56
  messageId?: number | string;
95
- /**
96
- * Optional pre-retrieved memory slice for this turn. Backends fold it into
97
- * the live user prompt (after the cached system prompt), never into
98
- * `system` — see `formatPromptWithRetrievedMemory`.
99
- */
100
- retrievedMemory?: RetrievedMemory;
101
57
  }
102
58
 
103
59
  // ── Catalog types ───────────────────────────────────────────────────────────
@@ -60,6 +60,68 @@ export class TalonError extends Error {
60
60
  }
61
61
  }
62
62
 
63
+ // ── Transport-failure detection ─────────────────────────────────────────────
64
+
65
+ /**
66
+ * Node / undici error codes that mean "the connection failed, try again".
67
+ *
68
+ * Codes beat message text: they're stable across Node versions and
69
+ * locales, and undici in particular throws bare `TypeError: terminated`
70
+ * / `TypeError: fetch failed` whose message says nothing while the
71
+ * `code` on the cause chain says everything.
72
+ *
73
+ * `ENOTFOUND` is deliberately absent — a DNS name that doesn't resolve
74
+ * is a config error, not a blip. `EAI_AGAIN` (temporary resolver
75
+ * failure) IS here, because that one does clear.
76
+ */
77
+ const RETRYABLE_TRANSPORT_CODES: ReadonlySet<string> = new Set([
78
+ "ECONNRESET",
79
+ "ECONNREFUSED",
80
+ "ECONNABORTED",
81
+ "EPIPE",
82
+ "ETIMEDOUT",
83
+ "EHOSTUNREACH",
84
+ "ENETUNREACH",
85
+ "ENETDOWN",
86
+ "EAI_AGAIN",
87
+ "UND_ERR_CONNECT_TIMEOUT",
88
+ "UND_ERR_HEADERS_TIMEOUT",
89
+ "UND_ERR_BODY_TIMEOUT",
90
+ "UND_ERR_SOCKET",
91
+ "ERR_STREAM_PREMATURE_CLOSE",
92
+ "ERR_SOCKET_CONNECTION_TIMEOUT",
93
+ ]);
94
+
95
+ /**
96
+ * Transient transport failures that surface as prose with no usable
97
+ * `code` — Node's own `socket hang up`, undici's `other side closed` /
98
+ * `Premature close`, and the named 5xx bodies proxies return as text.
99
+ *
100
+ * `timeout`/`timed out` is here but bare `abort` deliberately is NOT: a
101
+ * user interrupt (`controller.abort()` → "This operation was aborted")
102
+ * must stay non-retryable, or cancelling a turn would restart it.
103
+ * `AbortSignal.timeout()` says "aborted due to timeout" and is caught by
104
+ * the timeout half, which is the distinction we want.
105
+ */
106
+ const TRANSIENT_TRANSPORT_RE =
107
+ /socket hang up|other side closed|premature close|connection (?:closed|reset|lost)|timed out|time-?out|bad gateway|service unavailable|gateway time-?out|internal server error|upstream connect error|server disconnected/i;
108
+
109
+ /**
110
+ * Walk the `cause` chain collecting `code` / `errno` / `name` values.
111
+ * undici nests the real fault one or two levels down (`TypeError: fetch
112
+ * failed` → `cause: Error { code: 'ECONNREFUSED' }`), so a shallow look
113
+ * at the thrown object misses it entirely.
114
+ */
115
+ function errorCodes(err: unknown, depth = 0): string[] {
116
+ if (depth > 5 || err === null || typeof err !== "object") return [];
117
+ const e = err as { code?: unknown; errno?: unknown; cause?: unknown };
118
+ const codes: string[] = [];
119
+ if (typeof e.code === "string") codes.push(e.code);
120
+ if (typeof e.errno === "string") codes.push(e.errno);
121
+ codes.push(...errorCodes(e.cause, depth + 1));
122
+ return codes;
123
+ }
124
+
63
125
  // ── Classify any error ──────────────────────────────────────────────────────
64
126
 
65
127
  /**
@@ -115,6 +177,27 @@ export function classify(err: unknown): TalonError {
115
177
  });
116
178
  }
117
179
 
180
+ // Transport failure by CODE, before any prose matching. undici wraps
181
+ // the real fault ("TypeError: fetch failed" → cause.code) so the
182
+ // message alone is often empty of signal while the code is decisive.
183
+ //
184
+ // `TimeoutError` (what AbortSignal.timeout throws) is a transient
185
+ // deadline and retries. `AbortError` is NOT handled here on purpose —
186
+ // it means a user interrupt, and retrying a cancelled turn would
187
+ // restart work the user just stopped.
188
+ const errName = err instanceof Error ? err.name : "";
189
+ const transportCode = errorCodes(err).find((code) =>
190
+ RETRYABLE_TRANSPORT_CODES.has(code),
191
+ );
192
+ if (transportCode !== undefined || errName === "TimeoutError") {
193
+ return new TalonError(msg, {
194
+ reason: "network",
195
+ retryable: true,
196
+ retryAfterMs: 2_000,
197
+ cause,
198
+ });
199
+ }
200
+
118
201
  // Overloaded / capacity
119
202
  if (/overloaded|503|capacity/i.test(msg)) {
120
203
  return new TalonError(msg, {
@@ -126,11 +209,19 @@ export function classify(err: unknown): TalonError {
126
209
  });
127
210
  }
128
211
 
129
- // Network errors
212
+ // Network errors — codes echoed into the message text (a stringified
213
+ // cause, a provider wrapping the errno into prose), plus the
214
+ // code-less transient shapes in TRANSIENT_TRANSPORT_RE.
215
+ //
216
+ // The prose half only applies when NO HTTP status was found. A real
217
+ // status is the stronger signal and its own branches below own it:
218
+ // "500 Internal Server Error" must stay `overloaded` carrying
219
+ // status 500, not become a status-less `network`.
130
220
  if (
131
- /network|ECONNREFUSED|ECONNRESET|ECONNABORTED|ETIMEDOUT|ENOTFOUND|fetch failed|connection reset/i.test(
221
+ /network|ECONNREFUSED|ECONNRESET|ECONNABORTED|ETIMEDOUT|ENOTFOUND|EAI_AGAIN|EPIPE|UND_ERR_|fetch failed|connection reset/i.test(
132
222
  msg,
133
- )
223
+ ) ||
224
+ (status === undefined && TRANSIENT_TRANSPORT_RE.test(msg))
134
225
  ) {
135
226
  return new TalonError(msg, {
136
227
  reason: "network",
@@ -140,6 +231,19 @@ export function classify(err: unknown): TalonError {
140
231
  });
141
232
  }
142
233
 
234
+ // 408 Request Timeout — a transient deadline like any other, but it
235
+ // is neither 4xx-terminal nor 5xx, so it used to fall through to
236
+ // `unknown`/non-retryable and strand the request.
237
+ if (status === 408) {
238
+ return new TalonError(msg, {
239
+ reason: "network",
240
+ retryable: true,
241
+ status: 408,
242
+ retryAfterMs: 2_000,
243
+ cause,
244
+ });
245
+ }
246
+
143
247
  // Session expired
144
248
  if (/session.*expired|expired.*session|invalid.*resume/i.test(msg)) {
145
249
  return new TalonError(msg, {
@@ -3,7 +3,6 @@ export { ThreadSession, type SessionSummary } from "./thread-session.js";
3
3
  export { Loom, type ContextRegistry } from "./loom.js";
4
4
  export { carryTurnEvents, type EventSink } from "./shuttle.js";
5
5
  export { startTypingLoop, TYPING_REFRESH_MS } from "./typing-loop.js";
6
- export { prefetchMemory } from "./memory-prefetch.js";
7
6
  export {
8
7
  resolveWarp,
9
8
  type WarpResolution,
@@ -7,8 +7,6 @@
7
7
  * null-model guard and per-run override fallback;
8
8
  * - `startTypingLoop` (typing-loop.ts) — keeps the frontend's typing
9
9
  * indicator alive for the duration of the turn;
10
- * - `prefetchMemory` (memory-prefetch.ts) — optional fail-closed
11
- * palace pre-retrieval for live user messages;
12
10
  * - `carryTurnEvents` (shuttle.ts) — pumps the backend's AgentEvent
13
11
  * stream into the frontend sink, settles delivery acks, captures
14
12
  * the result and rethrows error terminators.
@@ -22,13 +20,11 @@ import { randomBytes } from "node:crypto";
22
20
  import type { Backend } from "../agent-runtime/capabilities.js";
23
21
  import type { AgentResult } from "../agent-runtime/events.js";
24
22
  import type { ModelRef } from "../agent-runtime/model-ref.js";
25
- import type { MemoryRetriever } from "../memory/retrieval.js";
26
23
  import type { ContextManager, ExecuteParams, ExecuteResult } from "../types.js";
27
24
  import { bus } from "../bus/index.js";
28
25
  import { taskTable, type TaskHandle } from "../tasks/index.js";
29
26
  import { log, logDebug, logWarn } from "../../util/log.js";
30
27
  import { Loom } from "./loom.js";
31
- import { prefetchMemory } from "./memory-prefetch.js";
32
28
  import { carryTurnEvents } from "./shuttle.js";
33
29
  import type { Thread, ThreadSnapshot } from "./thread.js";
34
30
  import { startTypingLoop } from "./typing-loop.js";
@@ -52,13 +48,6 @@ export type WeaverDeps = {
52
48
  ) => Promise<ModelRef | null>;
53
49
  context: ContextManager;
54
50
  sendTyping: (chatId: number, stringId?: string) => Promise<void>;
55
- /**
56
- * Optional memory pre-retrieval (Phase B). Called for `source: "message"`
57
- * turns after model/backend resolution and context acquisition, before
58
- * `runChatTurn(...)`. Fail-closed: errors are logged and the turn runs
59
- * without injected memory. Absent dep ⇒ prompts byte-identical to before.
60
- */
61
- retrieveMemory?: MemoryRetriever;
62
51
  };
63
52
 
64
53
  export class Weaver {
@@ -163,7 +152,7 @@ export class Weaver {
163
152
  params: ExecuteParams,
164
153
  task: TaskHandle,
165
154
  ): Promise<ExecuteResult> {
166
- const { context, retrieveMemory } = this.deps;
155
+ const { context } = this.deps;
167
156
  const backend = this.deps.getBackend(params.chatId);
168
157
  const reqId = randomBytes(4).toString("hex");
169
158
 
@@ -237,17 +226,6 @@ export class Weaver {
237
226
  );
238
227
  }
239
228
 
240
- const retrievedMemory =
241
- retrieveMemory && params.source === "message"
242
- ? await prefetchMemory(retrieveMemory, {
243
- chatId: params.chatId,
244
- text: params.prompt,
245
- senderName: params.senderName,
246
- isGroup: params.isGroup,
247
- reqId,
248
- })
249
- : undefined;
250
-
251
229
  const stream = backend.chat.runChatTurn({
252
230
  chatId: params.chatId,
253
231
  model: warp.ref,
@@ -255,7 +233,6 @@ export class Weaver {
255
233
  senderName: params.senderName,
256
234
  isGroup: params.isGroup,
257
235
  messageId: params.messageId,
258
- retrievedMemory,
259
236
  });
260
237
  const agentResult = await carryTurnEvents(stream, params.onEvent);
261
238
 
@@ -7,6 +7,7 @@
7
7
  * - `settings` — `settings:*` (effort/proactive + stale-picker handling)
8
8
  * - `pulse` — `pulse:*`
9
9
  * - `effort` — `effort:*`
10
+ * - `metrics` — `metrics:*` (today ↔ all-time panel grain)
10
11
  * - `model` — `model:*` (menu / backend / browse controller)
11
12
  *
12
13
  * `registerCallbacks` installs one `callback_query:data` listener that
@@ -22,6 +23,7 @@ import type { CallbackDeps } from "./shared.js";
22
23
  import { handleSettingsCallback } from "./settings.js";
23
24
  import { handlePulseCallback } from "./pulse.js";
24
25
  import { handleEffortCallback } from "./effort.js";
26
+ import { handleMetricsCallback } from "./metrics.js";
25
27
  import { handleModelCallback } from "./model.js";
26
28
 
27
29
  export { answerCallbackQuerySafe } from "./shared.js";
@@ -56,6 +58,12 @@ export function registerCallbacks(
56
58
  return;
57
59
  }
58
60
 
61
+ // Handle /metrics grain switching (today ↔ all time).
62
+ if (data.startsWith("metrics:")) {
63
+ await handleMetricsCallback(ctx, data);
64
+ return;
65
+ }
66
+
59
67
  // Handle /model callbacks via the pure parser + menu controller.
60
68
  if (data.startsWith("model:")) {
61
69
  await handleModelCallback(ctx, data, cid, deps);
@@ -0,0 +1,36 @@
1
+ /**
2
+ * `metrics:*` callbacks — swap the /metrics panel between the today and
3
+ * all-time grains by editing the message in place.
4
+ *
5
+ * Admin-gated like the command that posts the panel: in a group chat the
6
+ * buttons sit on a message anyone can tap.
7
+ */
8
+
9
+ import type { Context } from "grammy";
10
+ import { getMetrics, getTodayMetrics } from "../../../util/metrics.js";
11
+ import {
12
+ renderMetricsKeyboard,
13
+ renderMetricsPanel,
14
+ type MetricsView,
15
+ } from "../helpers/index.js";
16
+ import { isAuthorizedAdmin } from "../commands/state.js";
17
+ import { answerCallbackQuerySafe, editOrIgnoreSame } from "./shared.js";
18
+
19
+ export async function handleMetricsCallback(
20
+ ctx: Context,
21
+ data: string,
22
+ ): Promise<void> {
23
+ if (!isAuthorizedAdmin(ctx)) {
24
+ await answerCallbackQuerySafe(ctx, { text: "Not authorized." });
25
+ return;
26
+ }
27
+
28
+ const view: MetricsView = data === "metrics:all" ? "all" : "today";
29
+ await answerCallbackQuerySafe(ctx);
30
+ const metrics = view === "all" ? getMetrics() : getTodayMetrics();
31
+ await editOrIgnoreSame(
32
+ ctx,
33
+ renderMetricsPanel(metrics, view),
34
+ renderMetricsKeyboard(view),
35
+ );
36
+ }
@@ -19,11 +19,12 @@ import { closestMatch } from "../../../native/strsim.js";
19
19
  import {
20
20
  formatDuration,
21
21
  renderDoctorMessage,
22
- renderMetricsMessages,
22
+ renderMetricsKeyboard,
23
+ renderMetricsPanel,
23
24
  } from "../helpers/index.js";
24
25
  import { collectDoctorReport } from "../../../core/doctor.js";
25
26
  import { handleAdminCommand } from "../admin.js";
26
- import { getMetrics, getTodayMetrics } from "../../../util/metrics.js";
27
+ import { getTodayMetrics } from "../../../util/metrics.js";
27
28
  import { isAuthorizedAdmin, type RegisterDeps } from "./state.js";
28
29
  import { telegramCommandMenu } from "./definitions.js";
29
30
 
@@ -44,17 +45,12 @@ export function registerAdminCommands(
44
45
  await ctx.reply("Not authorized.");
45
46
  return;
46
47
  }
47
- const messages = [
48
- ...renderMetricsMessages(getMetrics()),
49
- ...renderMetricsMessages(
50
- getTodayMetrics(),
51
- undefined,
52
- "📊 Metrics — today (UTC)",
53
- ),
54
- ];
55
- for (const message of messages) {
56
- await ctx.reply(message, { parse_mode: "HTML" });
57
- }
48
+ // One message, two grains. Today opens first — it is the smaller,
49
+ // more actionable view; All time is a tap away on the same message.
50
+ await ctx.reply(renderMetricsPanel(getTodayMetrics(), "today"), {
51
+ parse_mode: "HTML",
52
+ reply_markup: { inline_keyboard: renderMetricsKeyboard("today") },
53
+ });
58
54
  });
59
55
 
60
56
  bot.command("doctor", async (ctx) => {
@@ -5,7 +5,7 @@
5
5
  import type { Bot } from "grammy";
6
6
  import { isUserClientReady } from "../userbot.js";
7
7
  import { escapeHtml } from "../formatting.js";
8
- import { formatDuration } from "../helpers/index.js";
8
+ import { formatDuration, renderMeshReport } from "../helpers/index.js";
9
9
  import { getLoadedPlugins } from "../../../core/plugin/index.js";
10
10
  import { getMeshService } from "../../../core/mesh/index.js";
11
11
  import type { MeshPingResult } from "../../../core/mesh/service.js";
@@ -104,7 +104,7 @@ export function registerInfoCommands(bot: Bot): void {
104
104
  });
105
105
 
106
106
  bot.command("mesh", async (ctx) => {
107
- const sent = await ctx.reply("🛰️ Pinging mesh devices…");
107
+ const sent = await ctx.reply("Pinging mesh devices…");
108
108
  let results: MeshPingResult[];
109
109
  try {
110
110
  results = await getMeshService().pingAll();
@@ -117,33 +117,11 @@ export function registerInfoCommands(bot: Bot): void {
117
117
  );
118
118
  return;
119
119
  }
120
- if (results.length === 0) {
121
- await editOrReply(
122
- bot,
123
- ctx.chat.id,
124
- sent.message_id,
125
- "<b>🛰️ Mesh</b>\n\nNo devices have registered yet.",
126
- );
127
- return;
128
- }
129
- // Online (reachable first, by latency) before offline; a stable, useful
130
- // order for a glance at the fleet.
131
- const sorted = [...results].sort((a, b) => {
132
- if (a.device.online !== b.device.online) return a.device.online ? -1 : 1;
133
- return (a.latencyMs ?? Infinity) - (b.latencyMs ?? Infinity);
134
- });
135
- const online = results.filter((r) => r.device.online).length;
136
- const reachable = results.filter((r) => r.reachable).length;
137
- const lines = sorted.map((r) => meshLine(r));
138
120
  await editOrReply(
139
121
  bot,
140
122
  ctx.chat.id,
141
123
  sent.message_id,
142
- [
143
- `<b>🛰️ Mesh</b> — ${results.length} device(s), ${online} online, ${reachable} responding`,
144
- "",
145
- ...lines,
146
- ].join("\n"),
124
+ renderMeshReport(results),
147
125
  );
148
126
  });
149
127
 
@@ -171,25 +149,6 @@ export function registerInfoCommands(bot: Bot): void {
171
149
  });
172
150
  }
173
151
 
174
- /** One `/mesh` line: presence + reachability + latency + platform. */
175
- function meshLine(r: MeshPingResult): string {
176
- const d = r.device;
177
- const dot = r.reachable ? "🟢" : d.online ? "🟡" : "⚪";
178
- const name = `<b>${escapeHtml(d.name)}</b>`;
179
- const bits: string[] = [`${d.platform}`];
180
- if (r.reachable && typeof r.latencyMs === "number") {
181
- bits.push(`${r.latencyMs}ms`);
182
- } else if (d.online && r.error) {
183
- bits.push(escapeHtml(r.error));
184
- } else if (!d.online) {
185
- bits.push(`offline, last seen ${formatDuration(Date.now() - d.lastSeen)}`);
186
- }
187
- if (typeof d.battery === "number") {
188
- bits.push(`${d.battery}%${d.charging ? "⚡" : ""}`);
189
- }
190
- return `${dot} ${name} — ${bits.join(" · ")}`;
191
- }
192
-
193
152
  /** Edit the placeholder in place, falling back to a fresh reply. */
194
153
  async function editOrReply(
195
154
  bot: Bot,
@@ -1,13 +1,24 @@
1
1
  /**
2
- * Metrics + doctor report rendering for Telegram (HTML messages).
2
+ * Metrics, doctor, and mesh report rendering for Telegram (HTML messages).
3
3
  */
4
4
 
5
5
  import { escapeHtml } from "../formatting.js";
6
6
  import type { DoctorReport } from "../../../core/doctor.js";
7
+ import type { MeshPingResult } from "../../../core/mesh/service.js";
8
+ import type { SettingsButton } from "./menu.js";
7
9
  import { formatDuration, formatBytes } from "./format.js";
8
10
 
9
11
  const DEFAULT_METRICS_MESSAGE_MAX = 3800;
10
12
 
13
+ /**
14
+ * Rows shown in the `tool_calls` leaderboard before the tail collapses
15
+ * into a "…and N more" line. A long-lived bot accumulates a lifetime
16
+ * tool-call entry per distinct tool name (easily 100+ once plugin and
17
+ * MCP tools are counted), which is what used to push /metrics past
18
+ * Telegram's message limit and split it across messages.
19
+ */
20
+ const TOOL_CALLS_TOP_N = 12;
21
+
11
22
  type MetricsSnapshot = {
12
23
  counters: Record<string, number>;
13
24
  histograms: Record<
@@ -16,18 +27,26 @@ type MetricsSnapshot = {
16
27
  >;
17
28
  };
18
29
 
30
+ /** Which grain the /metrics panel is currently showing. */
31
+ export type MetricsView = "today" | "all";
32
+
33
+ const VIEW_TITLES: Record<MetricsView, string> = {
34
+ today: "Metrics — today (UTC)",
35
+ all: "Metrics — all time",
36
+ };
37
+
38
+ /**
39
+ * One titled block of the panel. `hidden` counts rows dropped to make
40
+ * the panel fit one message; it renders as a trailing "…and N more".
41
+ */
42
+ type Section = { title: string; rows: string[]; hidden: number };
43
+
19
44
  function truncateMetricLabel(label: string, max = 80): string {
20
45
  return label.length <= max ? label : `${label.slice(0, max - 3)}...`;
21
46
  }
22
47
 
23
- export function renderMetricsMessages(
24
- metrics: MetricsSnapshot,
25
- maxLen = DEFAULT_METRICS_MESSAGE_MAX,
26
- title = "📊 Metrics",
27
- ): string[] {
28
- const firstHeader = `<b>${escapeHtml(title)}</b>`;
29
- const continuationHeader = `<b>${escapeHtml(title)} (cont.)</b>`;
30
- const sections: string[][] = [];
48
+ function buildMetricsSections(metrics: MetricsSnapshot): Section[] {
49
+ const sections: Section[] = [];
31
50
 
32
51
  // Histograms come in two flavours: durations (keys ending in `_ms`,
33
52
  // rendered as human times) and plain counts like `tool_calls_per_turn`
@@ -44,16 +63,18 @@ export function renderMetricsMessages(
44
63
  );
45
64
  };
46
65
  if (durationKeys.length > 0) {
47
- sections.push([
48
- "<b>Latency</b>",
49
- ...durationKeys.map((key) => histLine(key, formatDuration)),
50
- ]);
66
+ sections.push({
67
+ title: "<b>Latency</b>",
68
+ rows: durationKeys.map((key) => histLine(key, formatDuration)),
69
+ hidden: 0,
70
+ });
51
71
  }
52
72
  if (countKeys.length > 0) {
53
- sections.push([
54
- "<b>Distributions</b>",
55
- ...countKeys.map((key) => histLine(key, String)),
56
- ]);
73
+ sections.push({
74
+ title: "<b>Distributions</b>",
75
+ rows: countKeys.map((key) => histLine(key, String)),
76
+ hidden: 0,
77
+ });
57
78
  }
58
79
 
59
80
  const counterKeys = Object.keys(metrics.counters).sort();
@@ -68,17 +89,20 @@ export function renderMetricsMessages(
68
89
  for (const prefix of [...groups.keys()].sort()) {
69
90
  // tool_calls reads best as a leaderboard — busiest tools first.
70
91
  // Other groups keep alphabetical order (stable lookup by name).
71
- const keys =
72
- prefix === "tool_calls"
73
- ? [...groups.get(prefix)!].sort(
74
- (a, b) =>
75
- metrics.counters[b]! - metrics.counters[a]! ||
76
- a.localeCompare(b),
77
- )
78
- : groups.get(prefix)!;
79
- sections.push([
80
- `<b>${escapeHtml(prefix)}</b>`,
81
- ...keys.map((key) => {
92
+ const isToolCalls = prefix === "tool_calls";
93
+ const keys = isToolCalls
94
+ ? [...groups.get(prefix)!].sort(
95
+ (a, b) =>
96
+ metrics.counters[b]! - metrics.counters[a]! || a.localeCompare(b),
97
+ )
98
+ : groups.get(prefix)!;
99
+ // The leaderboard is pre-capped: past the top N the tail is a long
100
+ // list of one-call tools nobody reads, and on the all-time view it
101
+ // is what made the panel overflow.
102
+ const shown = isToolCalls ? keys.slice(0, TOOL_CALLS_TOP_N) : keys;
103
+ sections.push({
104
+ title: `<b>${escapeHtml(prefix)}</b>`,
105
+ rows: shown.map((key) => {
82
106
  const label = key.includes(".")
83
107
  ? key.split(".").slice(1).join(".")
84
108
  : key;
@@ -87,59 +111,88 @@ export function renderMetricsMessages(
87
111
  `${metrics.counters[key]!.toLocaleString()}`
88
112
  );
89
113
  }),
90
- ]);
114
+ hidden: keys.length - shown.length,
115
+ });
91
116
  }
92
117
  }
93
118
 
94
- if (sections.length === 0) {
95
- return [`${firstHeader}\n\n<i>No metrics recorded yet.</i>`];
96
- }
97
-
98
- const chunks: string[] = [];
99
- let header = firstHeader;
100
- let current = header;
101
-
102
- const flush = () => {
103
- chunks.push(current);
104
- header = continuationHeader;
105
- current = header;
106
- };
119
+ return sections;
120
+ }
107
121
 
108
- const appendLine = (line: string) => {
109
- if (!line && current === header) return;
122
+ function renderSection(section: Section): string {
123
+ const lines = [section.title, ...section.rows];
124
+ if (section.hidden > 0) {
125
+ lines.push(` <i>…and ${section.hidden} more</i>`);
126
+ }
127
+ return lines.join("\n");
128
+ }
110
129
 
111
- const candidate = `${current}\n${line}`;
112
- if (candidate.length <= maxLen) {
113
- current = candidate;
114
- return;
115
- }
130
+ /**
131
+ * Render the sections under `header`, shrinking to fit `maxLen`.
132
+ *
133
+ * Telegram hard-caps a message at 4096 characters. Rather than split the
134
+ * report across messages — which is what the panel replaced — rows are
135
+ * dropped from the longest section first (each drop bumping that
136
+ * section's "…and N more") until the whole thing fits. Sections always
137
+ * keep at least their first row, so every group stays visible.
138
+ */
139
+ function fitPanel(header: string, sections: Section[], maxLen: number): string {
140
+ const render = () => [header, ...sections.map(renderSection)].join("\n\n");
116
141
 
117
- if (current !== header) {
118
- flush();
119
- if (!line) return;
142
+ let out = render();
143
+ while (out.length > maxLen) {
144
+ let target: Section | undefined;
145
+ for (const section of sections) {
146
+ if (section.rows.length <= 1) continue;
147
+ if (!target || section.rows.length > target.rows.length) target = section;
120
148
  }
121
-
122
- const available = maxLen - header.length - 1;
123
- if (available < 0) return; // header alone already fills maxLen — skip line
124
- const safeLine =
125
- line.length <= available
126
- ? line
127
- : available >= 4
128
- ? `${line.slice(0, available - 3)}...`
129
- : line.slice(0, available); // not enough room for ellipsis — just truncate
130
- current = `${current}\n${safeLine}`;
131
- };
132
-
133
- for (const section of sections) {
134
- appendLine("");
135
- for (const line of section) appendLine(line);
149
+ if (!target) break;
150
+ target.rows.pop();
151
+ target.hidden += 1;
152
+ out = render();
136
153
  }
137
154
 
138
- if (current !== header || chunks.length === 0) {
139
- chunks.push(current);
155
+ if (out.length <= maxLen) return out;
156
+ // Nothing left to shed (a pathologically small maxLen). Cut on a line
157
+ // boundary so we never slice through an HTML tag and fail the parse.
158
+ const cut = out.lastIndexOf("\n", maxLen - 1);
159
+ return cut > 0 ? out.slice(0, cut) : out.slice(0, maxLen);
160
+ }
161
+
162
+ /**
163
+ * Render the /metrics panel for one grain as a SINGLE Telegram message.
164
+ * Pair it with `renderMetricsKeyboard(view)` — the Today / All time
165
+ * buttons swap grains by editing this message in place.
166
+ */
167
+ export function renderMetricsPanel(
168
+ metrics: MetricsSnapshot,
169
+ view: MetricsView,
170
+ maxLen = DEFAULT_METRICS_MESSAGE_MAX,
171
+ ): string {
172
+ const header = `<b>${escapeHtml(VIEW_TITLES[view])}</b>`;
173
+ const sections = buildMetricsSections(metrics);
174
+ if (sections.length === 0) {
175
+ return `${header}\n\n<i>No metrics recorded yet.</i>`;
140
176
  }
177
+ return fitPanel(header, sections, maxLen);
178
+ }
141
179
 
142
- return chunks;
180
+ /** Grain-switch buttons for the /metrics panel. */
181
+ export function renderMetricsKeyboard(
182
+ view: MetricsView,
183
+ ): Array<Array<SettingsButton>> {
184
+ return [
185
+ [
186
+ {
187
+ text: view === "today" ? "✓ Today" : "Today",
188
+ callback_data: "metrics:today",
189
+ },
190
+ {
191
+ text: view === "all" ? "✓ All time" : "All time",
192
+ callback_data: "metrics:all",
193
+ },
194
+ ],
195
+ ];
143
196
  }
144
197
 
145
198
  const DOCTOR_ICONS: Record<string, string> = {
@@ -187,3 +240,70 @@ export function renderDoctorMessage(report: DoctorReport): string {
187
240
 
188
241
  return lines.join("\n");
189
242
  }
243
+
244
+ // ── /mesh ───────────────────────────────────────────────────────────────────
245
+
246
+ function meshDeviceLine(r: MeshPingResult, now: number): string {
247
+ const d = r.device;
248
+ const bits: string[] = [escapeHtml(d.platform)];
249
+ if (r.reachable && typeof r.latencyMs === "number") {
250
+ bits.push(`${r.latencyMs} ms`);
251
+ } else if (d.online && r.error) {
252
+ bits.push(escapeHtml(r.error));
253
+ } else if (!d.online) {
254
+ bits.push(`last seen ${formatDuration(now - d.lastSeen)} ago`);
255
+ }
256
+ if (typeof d.battery === "number") {
257
+ bits.push(`${d.battery}%${d.charging ? " charging" : ""}`);
258
+ }
259
+ return ` <b>${escapeHtml(d.name)}</b> — ${bits.join(" · ")}`;
260
+ }
261
+
262
+ /**
263
+ * Render the `/mesh` fleet report as one HTML message.
264
+ *
265
+ * Devices group under a state heading — Responding, Unreachable, Offline
266
+ * — rather than carrying a coloured status glyph per row: the grouping
267
+ * already says what the glyph said, and the report stays readable when
268
+ * the fleet grows. Empty groups are omitted entirely.
269
+ */
270
+ export function renderMeshReport(
271
+ results: MeshPingResult[],
272
+ now = Date.now(),
273
+ ): string {
274
+ if (results.length === 0) {
275
+ return "<b>Mesh</b>\n\n<i>No devices have registered yet.</i>";
276
+ }
277
+
278
+ const responding = results
279
+ .filter((r) => r.reachable)
280
+ .sort((a, b) => (a.latencyMs ?? Infinity) - (b.latencyMs ?? Infinity));
281
+ const unreachable = results
282
+ .filter((r) => !r.reachable && r.device.online)
283
+ .sort((a, b) => a.device.name.localeCompare(b.device.name));
284
+ const offline = results
285
+ .filter((r) => !r.reachable && !r.device.online)
286
+ .sort((a, b) => b.device.lastSeen - a.device.lastSeen);
287
+
288
+ const summary = [
289
+ `${results.length} device${results.length === 1 ? "" : "s"}`,
290
+ `${responding.length} responding`,
291
+ ...(unreachable.length > 0 ? [`${unreachable.length} unreachable`] : []),
292
+ ...(offline.length > 0 ? [`${offline.length} offline`] : []),
293
+ ].join(" · ");
294
+
295
+ const lines = ["<b>Mesh</b>", summary];
296
+ const section = (title: string, entries: MeshPingResult[]): void => {
297
+ if (entries.length === 0) return;
298
+ lines.push(
299
+ "",
300
+ `<b>${title}</b>`,
301
+ ...entries.map((r) => meshDeviceLine(r, now)),
302
+ );
303
+ };
304
+ section("Responding", responding);
305
+ section("Unreachable", unreachable);
306
+ section("Offline", offline);
307
+
308
+ return lines.join("\n");
309
+ }
@@ -256,23 +256,6 @@ const mem0SettingsSchema = z.object({
256
256
  userId: z.string().min(1).optional(),
257
257
  });
258
258
 
259
- /**
260
- * Memory pre-retrieval (Phase B) — automatic per-turn palace retrieval.
261
- * Ships INERT: the flag gates wiring a real retriever into the dispatcher.
262
- * With `enabled: false` (default) prompts are byte-identical to previous
263
- * behavior. See docs in core/memory/retrieval.ts for the trust policy.
264
- */
265
- const memoryPreRetrievalSchema = z.object({
266
- /** Master switch. Default off — plumbing merges disabled. */
267
- enabled: z.boolean().default(false),
268
- /** Maximum retrieved items injected per turn. */
269
- maxResults: z.number().int().min(1).max(10).default(3),
270
- /** Hard cap on injected characters, provenance labels included. */
271
- maxChars: z.number().int().min(200).max(20000).default(3000),
272
- /** In groups, filter private-user drawers before injection. */
273
- groupPrivateUserFiltering: z.boolean().default(true),
274
- });
275
-
276
259
  const configSchema = z.object({
277
260
  frontend: z.union([frontendEnum, z.array(frontendEnum)]).default("telegram"),
278
261
  botToken: z.string().optional(),
@@ -351,8 +334,6 @@ const configSchema = z.object({
351
334
  pulseIntervalMs: z.number().int().min(60000).default(300000),
352
335
  /** Background memory-consolidation (dream) runs. Mirrors `pulse`/`heartbeat`. */
353
336
  dream: z.boolean().default(true),
354
- /** Memory pre-retrieval (Phase B). Optional; absent = disabled. */
355
- memoryPreRetrieval: memoryPreRetrievalSchema.optional(),
356
337
  /**
357
338
  * Periodic background agent (default: on, hourly). Advances open
358
339
  * goals, runs user-defined maintenance, and proactively messages
@@ -1,92 +0,0 @@
1
- /**
2
- * Memory pre-retrieval (Phase B) — injectable retriever boundary.
3
- *
4
- * Phase A made `memory.md` the compact always-loaded layer; Phase B closes the
5
- * recall gap by handing each chat turn a small, relevant palace slice without
6
- * depending on the model to call `mempalace_search` itself. This module owns
7
- * the CORE side of that boundary: a typed retriever dependency the Weaver can
8
- * call before `runChatTurn(...)`, returning plain data
9
- * (`RetrievedMemory | undefined`) that backends fold into the live user
10
- * prompt via `formatPromptWithRetrievedMemory`.
11
- *
12
- * Deliberate constraints (see docs/memory-phase-b-pre-retrieval.md):
13
- *
14
- * - The retriever receives a small context object and returns plain data.
15
- * It never imports backend handlers, and retrieval output never enters
16
- * `prepareSystemPrompt()` / frozen prompt snapshots — cache safety first.
17
- * - Fail closed: a broken retriever must never block chat delivery. The
18
- * Weaver catches errors, logs one warning, and runs the turn without
19
- * injected memory.
20
- * - The PRODUCTION retriever is intentionally absent in this first pass.
21
- * There is no clean in-process MemPalace call surface yet (the plugin is
22
- * an MCP server, not a typed module), and shelling out per message from
23
- * the dispatch hot path is explicitly forbidden. Until a deliberate
24
- * bridge exists, deployments get `noopMemoryRetriever` — plumbing lands
25
- * inert, prompts stay byte-identical.
26
- * - Trust-aware injection (#373): when a real adapter lands, only items
27
- * with `trustLevel` in `AUTO_INJECT_TRUST_LEVELS` may be auto-injected.
28
- * `user_claim` / `group_chat` content stays pull-only, or a poisoned
29
- * drawer becomes a persistent prompt injection in every session.
30
- */
31
-
32
- import type {
33
- RetrievedMemory,
34
- RetrievedMemoryTrustLevel,
35
- } from "../agent-runtime/capabilities.js";
36
-
37
- // ── Retriever contract ──────────────────────────────────────────────────────
38
-
39
- /** Context handed to a retriever for one chat turn. */
40
- export type MemoryRetrievalContext = {
41
- runKind: "chat";
42
- chatId: string;
43
- /** The raw incoming message text (retrievers trim/cap it themselves). */
44
- text: string;
45
- senderName: string;
46
- isGroup?: boolean;
47
- };
48
-
49
- /**
50
- * A memory retriever: small context in, bounded plain data out.
51
- * `undefined` means "inject nothing" — the turn proceeds unchanged.
52
- */
53
- export type MemoryRetriever = (
54
- context: MemoryRetrievalContext,
55
- ) => Promise<RetrievedMemory | undefined>;
56
-
57
- // ── Trust policy (#373) ─────────────────────────────────────────────────────
58
-
59
- /**
60
- * Trust levels eligible for automatic injection. Anything else (including an
61
- * absent `trustLevel`) must be filtered by adapters before returning items.
62
- */
63
- export const AUTO_INJECT_TRUST_LEVELS: ReadonlySet<RetrievedMemoryTrustLevel> =
64
- new Set(["dylan_direct", "bot_inferred", "heartbeat_synthesis"]);
65
-
66
- /** True when an item's trust level allows automatic injection. */
67
- export function isAutoInjectTrusted(
68
- trustLevel: RetrievedMemoryTrustLevel | undefined,
69
- ): boolean {
70
- return trustLevel !== undefined && AUTO_INJECT_TRUST_LEVELS.has(trustLevel);
71
- }
72
-
73
- /**
74
- * Enforce the #373 trust policy on a retrieval result: drop every item that
75
- * is not explicitly auto-inject trusted. Adapters should call this as their
76
- * last step, and the Weaver's `prefetchMemory` applies it again to whatever
77
- * a retriever returns — a buggy adapter cannot leak low-trust items into
78
- * the prompt.
79
- */
80
- export function filterAutoInjectable(
81
- memory: RetrievedMemory | undefined,
82
- ): RetrievedMemory | undefined {
83
- if (!memory) return undefined;
84
- const items = memory.items.filter((i) => isAutoInjectTrusted(i.trustLevel));
85
- if (items.length === 0) return undefined;
86
- return { ...memory, items };
87
- }
88
-
89
- // ── Default (inert) retriever ───────────────────────────────────────────────
90
-
91
- /** The no-op retriever: Phase B plumbing present, injection disabled. */
92
- export const noopMemoryRetriever: MemoryRetriever = async () => undefined;
@@ -1,65 +0,0 @@
1
- /**
2
- * Memory prefetch (Phase B) — optional pre-retrieval of palace memory
3
- * for live user messages, strictly fail-closed: a broken palace must
4
- * never block chat delivery. The result is dynamic turn context passed
5
- * through ChatRunParams; it never touches the frozen prompt.
6
- *
7
- * The #373 trust policy is enforced HERE, not just in adapters: whatever a
8
- * retriever returns is passed through `filterAutoInjectable` before it can
9
- * reach the prompt, so a buggy or future adapter cannot leak `user_claim` /
10
- * `group_chat` items into auto-injection.
11
- */
12
-
13
- import type { RetrievedMemory } from "../agent-runtime/capabilities.js";
14
- import type { MemoryRetriever } from "../memory/retrieval.js";
15
- import { filterAutoInjectable } from "../memory/retrieval.js";
16
- import { logDebug, logWarn } from "../../util/log.js";
17
-
18
- export type PrefetchMemoryInput = {
19
- chatId: string;
20
- text: string;
21
- senderName: string;
22
- isGroup: boolean;
23
- /** Request id for log correlation. */
24
- reqId: string;
25
- };
26
-
27
- export async function prefetchMemory(
28
- retrieve: MemoryRetriever,
29
- input: PrefetchMemoryInput,
30
- ): Promise<RetrievedMemory | undefined> {
31
- try {
32
- const raw = await retrieve({
33
- runKind: "chat",
34
- chatId: input.chatId,
35
- text: input.text,
36
- senderName: input.senderName,
37
- isGroup: input.isGroup,
38
- });
39
- const retrieved = filterAutoInjectable(raw);
40
- const droppedCount =
41
- (raw?.items.length ?? 0) - (retrieved?.items.length ?? 0);
42
- if (droppedCount > 0) {
43
- logWarn(
44
- "dispatcher",
45
- `[${input.reqId}] memory pre-retrieval: dropped ${droppedCount} ` +
46
- `low-trust item(s) the retriever failed to filter (#373)`,
47
- );
48
- }
49
- if (retrieved) {
50
- logDebug(
51
- "dispatcher",
52
- `[${input.reqId}] memory pre-retrieval: ${retrieved.items.length} item(s), ` +
53
- `${retrieved.items.reduce((n, i) => n + i.text.length, 0)} chars`,
54
- );
55
- }
56
- return retrieved;
57
- } catch (err) {
58
- logWarn(
59
- "dispatcher",
60
- `[${input.reqId}] memory pre-retrieval failed (running turn without it): ` +
61
- (err instanceof Error ? err.message : String(err)),
62
- );
63
- return undefined;
64
- }
65
- }