talon-agent 3.12.4 → 3.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.12.4",
3
+ "version": "3.13.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -3,6 +3,8 @@
3
3
  The following is your memory file. Reference it naturally. Update it via the Write tool when you learn important new information.
4
4
  File: ~/.talon/workspace/memory/memory.md
5
5
 
6
- {{content}}{% if truncated %}
6
+ {{content}}{% if omitted %}
7
+
8
+ Sections held back to keep session start lean — Read the file above when you need them: {{omitted}}{% elsif truncated %}
7
9
 
8
10
  …(memory file truncated here to keep session start lean — Read the file above for the rest){% endif %}
@@ -64,6 +64,10 @@ import {
64
64
  recordTurnMetrics,
65
65
  recordFailedTurnAccounting,
66
66
  recordFlowViolation,
67
+ formatTurnCache,
68
+ exceedsLookbackWindow,
69
+ estimateTurnBlocks,
70
+ CACHE_LOOKBACK_BLOCKS,
67
71
  } from "../shared/index.js";
68
72
 
69
73
  // ── Post-result watchdog ────────────────────────────────────────────────────
@@ -577,6 +581,19 @@ export async function* runChatTurn(
577
581
 
578
582
  state.allResponseText += state.currentBlockText;
579
583
 
584
+ // The aggregate `cache=NN%` can't distinguish a turn that reused the
585
+ // previous turn's prefix from one that re-wrote it — see
586
+ // shared/cache-telemetry.ts. Append the cross-turn verdict when the
587
+ // provider gave us per-request usage to derive it from.
588
+ if (state.cacheStats && exceedsLookbackWindow(state.toolCalls)) {
589
+ logWarn(
590
+ "agent",
591
+ `[${chatId}] turn emitted ~${estimateTurnBlocks(state.toolCalls)} content ` +
592
+ `blocks (> ${CACHE_LOOKBACK_BLOCKS} lookback) — the next turn's cache ` +
593
+ `breakpoint may not find this turn's prefix`,
594
+ );
595
+ }
596
+
580
597
  log(
581
598
  "agent",
582
599
  `[${chatId}] -> (${summarizeUsage(
@@ -586,7 +603,13 @@ export async function* runChatTurn(
586
603
  cacheRead: state.sdkCacheRead,
587
604
  cacheWrite: state.sdkCacheWrite,
588
605
  },
589
- { durationMs, toolCalls: state.toolCalls },
606
+ {
607
+ durationMs,
608
+ toolCalls: state.toolCalls,
609
+ ...(state.cacheStats
610
+ ? { suffix: formatTurnCache(state.cacheStats) }
611
+ : {}),
612
+ },
590
613
  )})`,
591
614
  );
592
615
  traceMessage(chatId, "out", state.allResponseText, {
@@ -19,6 +19,7 @@ import { log, logWarn } from "../../util/log.js";
19
19
  import { ALLOWED_TOOLS_BACKGROUND } from "../../core/constants.js";
20
20
  import { EFFORT_MAP } from "./constants.js";
21
21
  import { buildMcpServers, buildPluginMcpServers } from "./options.js";
22
+ import { warnIfBelowCacheMinimum } from "../shared/cache-telemetry.js";
22
23
 
23
24
  const DEFAULT_SUBPROCESS_KILL_GRACE_MS = 5 * 1000;
24
25
 
@@ -111,6 +112,11 @@ export async function runOneShotAgent(
111
112
  );
112
113
  }
113
114
 
115
+ // Background runs are the one path whose prompt can be small enough to
116
+ // fall under the model's cacheable floor — where nothing is cached and the
117
+ // API reports no error at all. Chat prompts always clear it.
118
+ warnIfBelowCacheMinimum(contextLabel, model, `${systemPrompt}\n${prompt}`);
119
+
114
120
  const qi = query({
115
121
  prompt,
116
122
  options: options as Parameters<typeof query>[0]["options"],
@@ -26,6 +26,10 @@ import {
26
26
  hubPluginServerNames,
27
27
  } from "../../core/mcp-hub/index.js";
28
28
  import { nonTerminalFrontends, frontendsForChat } from "../shared/frontends.js";
29
+ import {
30
+ noteToolFingerprint,
31
+ toolFingerprint,
32
+ } from "../shared/cache-telemetry.js";
29
33
  import { log, logError } from "../../util/log.js";
30
34
  import { getConfig, getBridgePort } from "./state.js";
31
35
  import { ALLOWED_TOOLS_CHAT, EFFORT_MAP } from "./constants.js";
@@ -378,6 +382,24 @@ export function buildSdkOptions(
378
382
  const { postToolUseFailureHook, postToolBatchHook } =
379
383
  buildTurnTerminatorHooks();
380
384
 
385
+ const builtinTools = config.nativeTools
386
+ ? ALLOWED_TOOLS_CHAT.filter((t) => !NATIVE_REPLACED_BUILTINS.has(t))
387
+ : [...ALLOWED_TOOLS_CHAT];
388
+
389
+ const mcpServers = {
390
+ ...buildMcpServers(chatId),
391
+ ...buildPluginMcpServers(chatId),
392
+ };
393
+
394
+ // Tool definitions render BEFORE the system prompt, so a set that shifts
395
+ // mid-session invalidates the system prompt and every cached message after
396
+ // it — the most expensive cache event there is, and one the aggregate
397
+ // hit-rate can't show. Plugin-provided MCP servers are the mutable part.
398
+ noteToolFingerprint(
399
+ chatId,
400
+ toolFingerprint(builtinTools, Object.keys(mcpServers)),
401
+ );
402
+
381
403
  const options: Options = {
382
404
  model: resolvedActiveModel,
383
405
  // Prefer the caller's frozen per-session prompt; fall back to the
@@ -424,16 +446,9 @@ export function buildSdkOptions(
424
446
  // teleport onto companion devices), and Agent (sub-agent dispatch) is
425
447
  // removed too — the owner prefers the native surface without nested
426
448
  // agents. Flip the flag back off to restore the built-ins instantly.
427
- tools: config.nativeTools
428
- ? ALLOWED_TOOLS_CHAT.filter(
429
- (t) => !NATIVE_REPLACED_BUILTINS.has(t as string),
430
- )
431
- : [...ALLOWED_TOOLS_CHAT],
449
+ tools: builtinTools,
432
450
  ...thinkingConfig,
433
- mcpServers: {
434
- ...buildMcpServers(chatId),
435
- ...buildPluginMcpServers(chatId),
436
- },
451
+ mcpServers,
437
452
  hooks: {
438
453
  PostToolUseFailure: [{ hooks: [postToolUseFailureHook] }],
439
454
  PostToolBatch: [{ hooks: [postToolBatchHook] }],
@@ -19,6 +19,10 @@ import type { BetaRawContentBlockDeltaEvent } from "@anthropic-ai/sdk/resources/
19
19
  import { STREAM_INTERVAL } from "./constants.js";
20
20
  import { log } from "../../util/log.js";
21
21
  import { checkModelDrift } from "./model-drift.js";
22
+ import {
23
+ turnCacheStats,
24
+ type TurnCacheStats,
25
+ } from "../shared/cache-telemetry.js";
22
26
 
23
27
  // ── Stream state accumulator ────────────────────────────────────────────────
24
28
 
@@ -35,6 +39,14 @@ export type StreamState = {
35
39
  sdkOutputTokens: number;
36
40
  sdkCacheRead: number;
37
41
  sdkCacheWrite: number;
42
+ /**
43
+ * Per-turn cache behaviour derived from the result message's per-request
44
+ * `usage.iterations`. Undefined when the provider reported none — the
45
+ * aggregate totals above cannot distinguish a turn that read the previous
46
+ * turn's prefix from one that re-wrote it, and that distinction is the
47
+ * whole cost signal (see shared/cache-telemetry.ts).
48
+ */
49
+ cacheStats: TurnCacheStats | undefined;
38
50
  lastStreamUpdate: number;
39
51
  /**
40
52
  * Trailing text from the most recent assistant message — text after all
@@ -101,6 +113,7 @@ export function createStreamState(): StreamState {
101
113
  sdkOutputTokens: 0,
102
114
  sdkCacheRead: 0,
103
115
  sdkCacheWrite: 0,
116
+ cacheStats: undefined,
104
117
  lastStreamUpdate: 0,
105
118
  lastTrailingText: "",
106
119
  deliveredTextNorms: [],
@@ -332,6 +345,9 @@ export function processResultMessage(
332
345
  (last.input_tokens ?? 0) +
333
346
  (last.cache_read_input_tokens ?? 0) +
334
347
  (last.cache_creation_input_tokens ?? 0);
348
+ // The same array answers a question the per-turn totals can't: whether
349
+ // the FIRST request of this turn read a cache or paid to write one.
350
+ state.cacheStats = turnCacheStats(usage.iterations);
335
351
  }
336
352
 
337
353
  // Read token counts from the ACTIVE model's usage only.
@@ -0,0 +1,297 @@
1
+ /**
2
+ * Prompt-cache telemetry — the numbers needed to tell whether Talon's
3
+ * prompt cache is actually working, and why not when it isn't.
4
+ *
5
+ * Motivation (see docs/memory-persona-plan.md §3.6): the obvious lever for a
6
+ * sparse-traffic chat bot would be a 1-hour cache TTL, but
7
+ * `@anthropic-ai/claude-agent-sdk` exposes no cache-control surface at all —
8
+ * it owns `cache_control` placement internally. So the only levers Talon has
9
+ * are (a) keeping the prompt prefix byte-stable and (b) keeping it small,
10
+ * and the only way to know which one is costing money is to measure.
11
+ *
12
+ * Four things are measured here, each answering a question the aggregate
13
+ * `cache=NN%` on the accounting line cannot:
14
+ *
15
+ * 1. **Cross-turn vs within-turn hits.** An agentic turn makes many model
16
+ * requests; every one after the first reads the prefix the first one
17
+ * just paid for. So a 97% aggregate hit rate is compatible with *every
18
+ * turn* re-writing the whole prefix. The turn's FIRST request is the
19
+ * only one that reports whether the previous turn's cache survived —
20
+ * that is the number that tracks cost.
21
+ * 2. **Tool-set churn.** Tool definitions render BEFORE the system prompt,
22
+ * so any change to the tool array invalidates the system prompt and the
23
+ * whole message history with it. A stable prompt behind an unstable
24
+ * tool list buys nothing, and Talon's per-chat MCP servers are exactly
25
+ * the shape that can shift mid-session.
26
+ * 3. **Lookback-window risk.** A cache breakpoint searches back at most
27
+ * `CACHE_LOOKBACK_BLOCKS` content blocks for a prior entry. A turn that
28
+ * emits more blocks than that can leave the *next* turn's breakpoint
29
+ * unable to find anything — a silent miss with no error.
30
+ * 4. **Sub-minimum prompts.** Below a model-specific token floor nothing
31
+ * caches, and the API reports no error. Talon's chat prompts clear every
32
+ * floor; its one-shot prompts (dream, heartbeat, cron) may not.
33
+ *
34
+ * Everything here is pure except `noteToolFingerprint`, which keeps a small
35
+ * per-chat map so it can name what changed rather than just that something
36
+ * did.
37
+ */
38
+
39
+ import { logWarn } from "../../util/log.js";
40
+
41
+ // ── Per-turn cache stats ────────────────────────────────────────────────────
42
+
43
+ /**
44
+ * One model request's usage, as the SDK reports it in
45
+ * `result.usage.iterations`. Fields are optional because a given provider
46
+ * may omit any of them; absent is treated as zero.
47
+ */
48
+ export type CacheIteration = {
49
+ readonly input_tokens?: number;
50
+ readonly cache_read_input_tokens?: number;
51
+ readonly cache_creation_input_tokens?: number;
52
+ };
53
+
54
+ /** Cache behaviour for one user-visible turn. */
55
+ export type TurnCacheStats = {
56
+ /** Model requests the SDK made inside this turn. */
57
+ readonly modelRequests: number;
58
+ /** Cache read on the turn's FIRST request — did the last turn's prefix live? */
59
+ readonly firstRead: number;
60
+ /** Cache written by the turn's first request. */
61
+ readonly firstWrite: number;
62
+ /** Cache read across every request in the turn. */
63
+ readonly totalRead: number;
64
+ /** Cache written across every request in the turn. */
65
+ readonly totalWrite: number;
66
+ };
67
+
68
+ /**
69
+ * What the turn's first request says about the previous turn's cache.
70
+ *
71
+ * - `hit` — the prefix survived; this turn read it instead of paying for it.
72
+ * - `miss` — the prefix was gone and had to be re-written. The expensive case,
73
+ * and the one a TTL shorter than the gap between turns produces.
74
+ * - `none` — nothing was cached either way: below the model's cacheable
75
+ * minimum, or a provider that doesn't cache at all.
76
+ */
77
+ export type CrossTurnVerdict = "hit" | "miss" | "none";
78
+
79
+ /**
80
+ * Fold a turn's per-request usage into cache stats. Returns undefined when
81
+ * the provider reported no iterations — the caller then logs nothing rather
82
+ * than inventing a verdict from aggregate totals it can't attribute.
83
+ */
84
+ export function turnCacheStats(
85
+ iterations: readonly CacheIteration[] | undefined,
86
+ ): TurnCacheStats | undefined {
87
+ if (!iterations || iterations.length === 0) return undefined;
88
+ const first = iterations[0]!;
89
+ let totalRead = 0;
90
+ let totalWrite = 0;
91
+ for (const it of iterations) {
92
+ totalRead += it.cache_read_input_tokens ?? 0;
93
+ totalWrite += it.cache_creation_input_tokens ?? 0;
94
+ }
95
+ return {
96
+ modelRequests: iterations.length,
97
+ firstRead: first.cache_read_input_tokens ?? 0,
98
+ firstWrite: first.cache_creation_input_tokens ?? 0,
99
+ totalRead,
100
+ totalWrite,
101
+ };
102
+ }
103
+
104
+ /** Classify the turn's first request. See {@link CrossTurnVerdict}. */
105
+ export function crossTurnVerdict(stats: TurnCacheStats): CrossTurnVerdict {
106
+ if (stats.firstRead > 0) return "hit";
107
+ if (stats.firstWrite > 0) return "miss";
108
+ return "none";
109
+ }
110
+
111
+ /**
112
+ * Compact suffix for the per-turn accounting line, e.g.
113
+ * `xturn=miss reqs=11 rw=8.0`. Deliberately terse and space-delimited so a
114
+ * week of logs can be parsed without a schema:
115
+ *
116
+ * - `xturn` — the cross-turn verdict; the field that tracks cost.
117
+ * - `reqs` — model requests in the turn; explains a high aggregate hit
118
+ * rate that isn't saving anything.
119
+ * - `rw` — total read ÷ total write. Above ~1 the turn amortised its
120
+ * write; at or below it, the write dominated.
121
+ */
122
+ export function formatTurnCache(stats: TurnCacheStats): string {
123
+ const parts = [
124
+ `xturn=${crossTurnVerdict(stats)}`,
125
+ `reqs=${stats.modelRequests}`,
126
+ ];
127
+ if (stats.totalWrite > 0) {
128
+ parts.push(`rw=${(stats.totalRead / stats.totalWrite).toFixed(1)}`);
129
+ }
130
+ return parts.join(" ");
131
+ }
132
+
133
+ // ── Lookback window ────────────────────────────────────────────────────────
134
+
135
+ /**
136
+ * How far back a cache breakpoint searches for a prior entry, in content
137
+ * blocks. A turn that emits more than this can prevent the NEXT turn from
138
+ * finding any cache to read.
139
+ */
140
+ export const CACHE_LOOKBACK_BLOCKS = 20;
141
+
142
+ /**
143
+ * Rough content-block count for a turn from its tool-call count. Each tool
144
+ * call is a `tool_use` block plus a `tool_result` block, and the assistant
145
+ * text around them adds at least one more — so `2n + 1` is a floor, not an
146
+ * estimate. Used only to decide whether to warn, never to report a number.
147
+ */
148
+ export function estimateTurnBlocks(toolCalls: number): number {
149
+ return Math.max(0, toolCalls) * 2 + 1;
150
+ }
151
+
152
+ /** True when a turn plausibly emitted more blocks than a breakpoint looks back. */
153
+ export function exceedsLookbackWindow(toolCalls: number): boolean {
154
+ return estimateTurnBlocks(toolCalls) > CACHE_LOOKBACK_BLOCKS;
155
+ }
156
+
157
+ // ── Cacheable minimum ──────────────────────────────────────────────────────
158
+
159
+ /**
160
+ * Minimum cacheable prefix, in tokens, by model. Below this nothing is
161
+ * cached and the API reports no error — `cache_creation_input_tokens` is
162
+ * simply 0.
163
+ *
164
+ * The floor is NOT monotonic across generations (512 on the newest models,
165
+ * 4096 on Opus 4.6 and Haiku 4.5), so it can't be inferred from a version
166
+ * number. Longest match wins, so `sonnet-4-6` is checked before `sonnet`.
167
+ *
168
+ * Bare aliases resolve only where Talon's catalog leaves no ambiguity:
169
+ * `haiku` is Haiku 4.5 (and carries the largest floor, making it the most
170
+ * likely to silently not cache). Ambiguous aliases — `opus`, `sonnet`,
171
+ * `default` — deliberately return undefined: a warning that fires on the
172
+ * wrong model teaches people to ignore warnings.
173
+ */
174
+ const CACHE_MINIMUMS: readonly (readonly [string, number])[] = [
175
+ ["fable-5", 512],
176
+ ["mythos-5", 512],
177
+ ["opus-5", 512],
178
+ ["opus-4-8", 1024],
179
+ ["sonnet-5", 1024],
180
+ ["sonnet-4-6", 1024],
181
+ ["sonnet-4-5", 1024],
182
+ ["opus-4-1", 1024],
183
+ ["opus-4-7", 2048],
184
+ ["mythos-preview", 2048],
185
+ ["opus-4-6", 4096],
186
+ ["opus-4-5", 4096],
187
+ ["haiku-4-5", 4096],
188
+ ["haiku", 4096],
189
+ ];
190
+
191
+ /**
192
+ * The cacheable-prefix floor for a model, or undefined when the model string
193
+ * doesn't unambiguously identify one.
194
+ */
195
+ export function cacheMinimumTokens(model: string): number | undefined {
196
+ const m = model.toLowerCase();
197
+ let best: { key: string; min: number } | undefined;
198
+ for (const [key, min] of CACHE_MINIMUMS) {
199
+ if (!m.includes(key)) continue;
200
+ if (!best || key.length > best.key.length) best = { key, min };
201
+ }
202
+ return best?.min;
203
+ }
204
+
205
+ /** Cheap tokenizer-free estimate (~4 chars/token), matching soul/projector. */
206
+ function estimateTokens(text: string): number {
207
+ return Math.ceil(text.length / 4);
208
+ }
209
+
210
+ /**
211
+ * Warn when a prompt is too small to be cacheable on its model. No-op when
212
+ * the model's floor is unknown or the prompt clears it. `label` names the
213
+ * caller (e.g. `"dream"`) so the warning points somewhere.
214
+ */
215
+ export function warnIfBelowCacheMinimum(
216
+ label: string,
217
+ model: string,
218
+ prompt: string,
219
+ ): void {
220
+ const min = cacheMinimumTokens(model);
221
+ if (min === undefined) return;
222
+ const estimated = estimateTokens(prompt);
223
+ if (estimated >= min) return;
224
+ logWarn(
225
+ "agent",
226
+ `[${label}] prompt ~${estimated} tokens is below ${model}'s ${min}-token ` +
227
+ `cacheable minimum — nothing will be cached for this run`,
228
+ );
229
+ }
230
+
231
+ // ── Tool-set fingerprint ───────────────────────────────────────────────────
232
+
233
+ /**
234
+ * Per-chat fingerprint of the last tool set seen. Bounded so a long-lived
235
+ * process with many chats can't grow it without limit; eviction is
236
+ * insertion-ordered, and a false "changed" warning after eviction is
237
+ * cheaper than unbounded retention.
238
+ */
239
+ const MAX_TRACKED_CHATS = 256;
240
+ const lastToolSets = new Map<string, readonly string[]>();
241
+
242
+ /** Stable fingerprint for a tool set: sorted, deduped names. */
243
+ export function toolFingerprint(
244
+ builtinTools: readonly string[],
245
+ mcpServerNames: readonly string[],
246
+ ): readonly string[] {
247
+ return [
248
+ ...new Set([...builtinTools, ...mcpServerNames.map((n) => `mcp:${n}`)]),
249
+ ].sort();
250
+ }
251
+
252
+ /**
253
+ * Record this turn's tool set for a chat and warn when it differs from the
254
+ * previous turn's. Returns true when a change was detected.
255
+ *
256
+ * Tools render before the system prompt, so a mid-session change invalidates
257
+ * the system prompt and every cached message after it — the most expensive
258
+ * cache event available, and invisible in aggregate hit-rate numbers.
259
+ */
260
+ export function noteToolFingerprint(
261
+ chatId: string,
262
+ fingerprint: readonly string[],
263
+ ): boolean {
264
+ const previous = lastToolSets.get(chatId);
265
+
266
+ if (lastToolSets.size >= MAX_TRACKED_CHATS && !previous) {
267
+ const oldest = lastToolSets.keys().next().value;
268
+ if (oldest !== undefined) lastToolSets.delete(oldest);
269
+ }
270
+ lastToolSets.set(chatId, fingerprint);
271
+
272
+ if (!previous) return false;
273
+ if (
274
+ previous.length === fingerprint.length &&
275
+ previous.every((t, i) => t === fingerprint[i])
276
+ ) {
277
+ return false;
278
+ }
279
+
280
+ const before = new Set(previous);
281
+ const after = new Set(fingerprint);
282
+ const added = fingerprint.filter((t) => !before.has(t));
283
+ const removed = previous.filter((t) => !after.has(t));
284
+ logWarn(
285
+ "agent",
286
+ `[${chatId}] tool set changed mid-session — invalidates the whole prompt ` +
287
+ `cache for this chat` +
288
+ (added.length ? ` (+${added.join(",")})` : "") +
289
+ (removed.length ? ` (-${removed.join(",")})` : ""),
290
+ );
291
+ return true;
292
+ }
293
+
294
+ /** Drop all tracked fingerprints (tests / explicit reset). */
295
+ export function resetToolFingerprints(): void {
296
+ lastToolSets.clear();
297
+ }
@@ -68,6 +68,16 @@ export {
68
68
  type TokenUsageSnapshot,
69
69
  } from "./usage.js";
70
70
 
71
+ // Only what is consumed THROUGH the barrel. Everything else in
72
+ // cache-telemetry.ts is imported from the module directly, matching the
73
+ // barrel discipline the prompt/ barrel was just trimmed to.
74
+ export {
75
+ formatTurnCache,
76
+ estimateTurnBlocks,
77
+ exceedsLookbackWindow,
78
+ CACHE_LOOKBACK_BLOCKS,
79
+ } from "./cache-telemetry.js";
80
+
71
81
  export {
72
82
  prepareSystemPrompt,
73
83
  appendBackendSuffix,
@@ -16,8 +16,9 @@
16
16
  * 2. Core behaviour ~/.talon/prompts/custom.md,
17
17
  * else base.md, else fallback
18
18
  * 3. Frontend capabilities ~/.talon/prompts/<frontend>.md
19
- * 4. Persistent memory (size-capped) prompts/system/persistent-memory.md
19
+ * 4. Persistent memory (ranked, capped) prompts/system/persistent-memory.md
20
20
  * wrapping ~/.talon/workspace/memory/memory.md
21
+ * via memory-view.ts
21
22
  * 5. Memory recall + capability docs prompts/system/{memory-recall,workspace,...}.md
22
23
  * 6. Plugin additions plugin.systemPrompt() contributions
23
24
  * (7. Delivery contract — appended by the backend as its suffix,
@@ -57,6 +58,7 @@ import { dirs, files as pathFiles } from "../../util/paths.js";
57
58
  import { todayAndYesterday } from "../../util/time.js";
58
59
  import { log } from "../../util/log.js";
59
60
  import { loadSystemTemplate } from "./templates.js";
61
+ import { renderMemoryView } from "./memory-view.js";
60
62
  import { renderWorkspaceListing } from "./workspace-listing.js";
61
63
  import { renderSkillsPrompt } from "../../storage/skill-store.js";
62
64
  import { renderStickerLibraryPrompt } from "../../storage/sticker-store.js";
@@ -92,17 +94,6 @@ export function joinSystemPromptParts(parts: SystemPromptParts): string {
92
94
  return `${parts.staticText}\n\n---\n\n${parts.dynamicText}`;
93
95
  }
94
96
 
95
- // ── Tunables ────────────────────────────────────────────────────────────────
96
-
97
- /**
98
- * Cap on how much of `memory.md` is injected into the static prompt.
99
- * Memory files grow without bound over months of use; injecting all
100
- * of it bloats EVERY session from its very first turn. 12k chars is
101
- * roughly 3k tokens — past that, the model gets the head of the file
102
- * plus a truncation pointer and can Read the rest on demand.
103
- */
104
- export const MEMORY_INJECT_MAX_CHARS = 12_000;
105
-
106
97
  // ── Helpers ─────────────────────────────────────────────────────────────────
107
98
 
108
99
  function readOptionalFile(path: string): string {
@@ -114,23 +105,6 @@ function readOptionalFile(path: string): string {
114
105
  return "";
115
106
  }
116
107
 
117
- /**
118
- * Truncate memory content at the cap, snapping back to the previous
119
- * newline so the cut never lands mid-sentence. Returns the (possibly
120
- * shortened) content and whether truncation happened.
121
- */
122
- function capMemory(content: string): { text: string; truncated: boolean } {
123
- if (content.length <= MEMORY_INJECT_MAX_CHARS) {
124
- return { text: content, truncated: false };
125
- }
126
- const head = content.slice(0, MEMORY_INJECT_MAX_CHARS);
127
- const lastNewline = head.lastIndexOf("\n");
128
- return {
129
- text: (lastNewline > 0 ? head.slice(0, lastNewline) : head).trimEnd(),
130
- truncated: true,
131
- };
132
- }
133
-
134
108
  let lastLoggedPromptKey = "";
135
109
 
136
110
  // ── Assembly ────────────────────────────────────────────────────────────────
@@ -194,17 +168,21 @@ export function assembleSystemPrompt(
194
168
  }
195
169
 
196
170
  // 4. Persistent memory — size-capped so a memory file that has grown
197
- // for months can't bloat every session from turn 0.
171
+ // for months can't bloat every session from turn 0. Over the cap the
172
+ // view ranks sections rather than head-slicing, so durable knowledge
173
+ // isn't evicted by whatever happens to sit at the top of the file
174
+ // (see prompt/memory-view.ts).
198
175
  const memory = readOptionalFile(pathFiles.memory);
199
176
  if (memory) {
200
- const { text, truncated } = capMemory(memory);
177
+ const { text, truncated, omitted } = renderMemoryView(memory);
201
178
  staticParts.push(
202
179
  loadSystemTemplate("persistent-memory", {
203
180
  content: text,
204
181
  truncated: truncated ? "yes" : undefined,
182
+ omitted: omitted || undefined,
205
183
  }),
206
184
  );
207
- loaded.push(truncated ? "memory(capped)" : "memory");
185
+ loaded.push(truncated ? "memory(ranked)" : "memory");
208
186
  }
209
187
 
210
188
  // 5. Package-owned behavioural and capability docs. The memory policy
@@ -3,6 +3,7 @@
3
3
  *
4
4
  * One stop for everything system-prompt:
5
5
  * - `assemble` — section pipeline + static/dynamic split
6
+ * - `memory-view` — ranked selection of `memory.md` under a budget
6
7
  * - `templates` — package-owned `prompts/system/*.md` loader
7
8
  * - `workspace-listing` — lazy workspace tree for the dynamic tail
8
9
  *
@@ -0,0 +1,304 @@
1
+ /**
2
+ * Persistent-memory view — what the model actually sees of `memory.md`.
3
+ *
4
+ * `memory.md` grows without bound and only the first
5
+ * `MEMORY_INJECT_MAX_CHARS` reach the prompt. Head-slicing that file makes
6
+ * the cut positional while the content is not priority-ordered, so the
7
+ * *least* durable material evicts the most durable. Observed on the live
8
+ * deployment: a 23.6k-char file cut at line 46 of 113, where everything
9
+ * above the line was one facts block plus three near-duplicate
10
+ * `## Inbox / CI Watch (as of …, Run #N)` status snapshots, and
11
+ * `## Active Investigations` — with the root-cause analysis in it — fell
12
+ * below the cut and was never injected at all.
13
+ *
14
+ * So this module makes truncation a *selection* problem:
15
+ *
16
+ * 1. Split the file into `## ` sections (an `h3` stays with its parent).
17
+ * 2. Collapse "state families" — sections whose headings differ only by a
18
+ * trailing timestamp / run number — down to the newest member. Three
19
+ * CI-watch snapshots describing the same recurring failures are one
20
+ * fact, not three.
21
+ * 3. Order by tier (directives → people → active work → general → status
22
+ * → historical), stable within a tier.
23
+ * 4. Emit whole sections until the budget is spent, and name the ones
24
+ * that didn't make it so the model knows to `Read` for them.
25
+ *
26
+ * Three deliberate constraints:
27
+ *
28
+ * - **Under the cap, output is byte-identical to the input.** Reordering
29
+ * only happens when the alternative is losing content, so deployments
30
+ * whose memory still fits see no behaviour change and no prompt-cache
31
+ * churn.
32
+ * - **Surviving sections are emitted in file order**, not tier order. The
33
+ * ranking decides *what* survives, not how it reads — and a
34
+ * tier-ordered body would rewrite the prompt prefix every time a
35
+ * section's heading changed tier.
36
+ * - **If nothing fits, fall back to head-slicing.** A single section
37
+ * larger than the whole budget must still give the model something
38
+ * rather than an empty memory block.
39
+ */
40
+
41
+ // ── Tunables ────────────────────────────────────────────────────────────────
42
+
43
+ /**
44
+ * Cap on how much of `memory.md` is injected into the static prompt.
45
+ * Memory files grow without bound over months of use; injecting all of it
46
+ * bloats EVERY session from its very first turn. 12k chars is roughly 3k
47
+ * tokens — past that, the model gets the ranked selection below plus a
48
+ * pointer, and can Read the file on demand.
49
+ */
50
+ export const MEMORY_INJECT_MAX_CHARS = 12_000;
51
+
52
+ /** Cap on how many omitted section titles are named before eliding the rest. */
53
+ const MAX_OMITTED_NAMED = 8;
54
+
55
+ // ── Tiers ───────────────────────────────────────────────────────────────────
56
+
57
+ /**
58
+ * Section priority, most-durable first. `status` sits second-to-last because
59
+ * a snapshot is true *now* and false later — it is the content most safely
60
+ * dropped and most cheaply re-derived. `historical` is last for the mirror
61
+ * reason: it will never change again, so it is the least urgent to carry.
62
+ */
63
+ const TIER_ORDER = [
64
+ "directive",
65
+ "people",
66
+ "active",
67
+ "general",
68
+ "status",
69
+ "historical",
70
+ ] as const;
71
+
72
+ type Tier = (typeof TIER_ORDER)[number];
73
+
74
+ const tier = (name: Tier): number => TIER_ORDER.indexOf(name);
75
+
76
+ /**
77
+ * Heading classifiers. First match wins, so order is the policy: a heading
78
+ * reading "Active investigation status" is active work, not a snapshot.
79
+ * `general` is the unmatched default and ranks above `status` — a section
80
+ * nobody labelled is more likely durable knowledge than a dated snapshot.
81
+ */
82
+ const MATCHERS: readonly { readonly tier: Tier; readonly match: RegExp }[] = [
83
+ {
84
+ tier: "historical",
85
+ match:
86
+ /\b(historical|history|archived?|superseded|resolved|closed|completed|past|old)\b/i,
87
+ },
88
+ {
89
+ tier: "directive",
90
+ match: /\b(directives?|preferences?|instructions?|rules?)\b/i,
91
+ },
92
+ {
93
+ tier: "active",
94
+ match:
95
+ /\b(active|current|investigations?|open|pending|todo|follow.?ups?|blocked|in.progress|priorit(?:y|ies)|goals?)\b/i,
96
+ },
97
+ {
98
+ tier: "people",
99
+ match: /\b(users?|people|person|contacts?|team|about)\b/i,
100
+ },
101
+ {
102
+ tier: "status",
103
+ match: /\b(as of|run #|status|health|watch|inbox|snapshot|report)\b/i,
104
+ },
105
+ ];
106
+
107
+ const STATUS_TIER = tier("status");
108
+ const GENERAL_TIER = tier("general");
109
+
110
+ function classify(title: string): number {
111
+ for (const m of MATCHERS) {
112
+ if (m.match.test(title)) return tier(m.tier);
113
+ }
114
+ return GENERAL_TIER;
115
+ }
116
+
117
+ // ── Parsing ─────────────────────────────────────────────────────────────────
118
+
119
+ type Section = {
120
+ /** Heading with markup and trailing qualifiers stripped — the display name. */
121
+ readonly title: string;
122
+ /** Whole section including its heading and any `###` children. */
123
+ readonly body: string;
124
+ /** Index in the original file, for stable ordering within a tier. */
125
+ readonly order: number;
126
+ readonly tier: number;
127
+ /** Key shared by every member of a state family (see `familyKey`). */
128
+ readonly family: string;
129
+ /** Recency score used to pick a family's survivor; higher is newer. */
130
+ readonly recency: string;
131
+ };
132
+
133
+ /** Strip `## ` markup and bold markers from a heading line. */
134
+ function headingTitle(heading: string): string {
135
+ return heading
136
+ .replace(/^#+\s*/, "")
137
+ .replace(/\*\*/g, "")
138
+ .trim();
139
+ }
140
+
141
+ /**
142
+ * The family a section belongs to: its title minus temporal qualifiers.
143
+ * `Inbox / CI Watch (as of 2026-07-03, Run #134)` → `inbox / ci watch`. That
144
+ * parenthetical is exactly what makes each snapshot look unique while the
145
+ * underlying topic repeats, so removing it is what lets a family collapse.
146
+ */
147
+ function familyKey(title: string): string {
148
+ return title
149
+ .replace(/\s*\([^)]*\)\s*$/, "")
150
+ .replace(/\s*[—–-]\s*(?:as of|run #).*$/i, "")
151
+ .trim()
152
+ .toLowerCase();
153
+ }
154
+
155
+ /**
156
+ * A sortable recency key: `<date>|<run>`, both fixed-width so a plain string
157
+ * compare orders correctly. Both fields are zero-filled when absent rather
158
+ * than left empty — an empty field would make the `|` separator the first
159
+ * character compared, and `|` sorts *above* every digit, which would rank an
160
+ * undated section as newer than a dated one. A heading with neither field
161
+ * scores lowest and falls back to file order, where the dream agent puts the
162
+ * newest section first.
163
+ */
164
+ function recencyKey(heading: string): string {
165
+ const date = /(\d{4}-\d{2}-\d{2})/.exec(heading)?.[1] ?? "0000-00-00";
166
+ const run = /\brun\s*#\s*(\d+)/i.exec(heading)?.[1] ?? "0";
167
+ return `${date}|${run.padStart(10, "0")}`;
168
+ }
169
+
170
+ /**
171
+ * Split content into a leading preamble (the `#` title and anything before
172
+ * the first `## `) plus one entry per `## ` section. The split is on `## `
173
+ * only, so an `h3` travels with the section it belongs to.
174
+ */
175
+ function parseSections(content: string): {
176
+ preamble: string;
177
+ sections: Section[];
178
+ } {
179
+ const parts = content.split(/^(?=## )/m);
180
+ const first = parts[0] ?? "";
181
+ const preamble = first.startsWith("## ") ? "" : (parts.shift() ?? "");
182
+ const sections: Section[] = [];
183
+ for (const body of parts) {
184
+ if (!body.trim()) continue;
185
+ const heading = body.split("\n", 1)[0] ?? "";
186
+ const title = headingTitle(heading);
187
+ sections.push({
188
+ title,
189
+ body,
190
+ order: sections.length,
191
+ tier: classify(title),
192
+ family: familyKey(title),
193
+ recency: recencyKey(heading),
194
+ });
195
+ }
196
+ return { preamble, sections };
197
+ }
198
+
199
+ /**
200
+ * Keep one section per state family: the newest by `recency`, and on a tie
201
+ * the one that appeared first. Only the `status` tier collapses — two
202
+ * sections about people are two different people, not two snapshots of one.
203
+ */
204
+ function collapseFamilies(sections: readonly Section[]): {
205
+ kept: Section[];
206
+ dropped: Section[];
207
+ } {
208
+ const winners = new Map<string, Section>();
209
+ for (const s of sections) {
210
+ if (s.tier !== STATUS_TIER) continue;
211
+ const held = winners.get(s.family);
212
+ if (!held || s.recency > held.recency) winners.set(s.family, s);
213
+ }
214
+ const kept: Section[] = [];
215
+ const dropped: Section[] = [];
216
+ for (const s of sections) {
217
+ if (s.tier !== STATUS_TIER || winners.get(s.family) === s) kept.push(s);
218
+ else dropped.push(s);
219
+ }
220
+ return { kept, dropped };
221
+ }
222
+
223
+ // ── Public API ──────────────────────────────────────────────────────────────
224
+
225
+ /** The memory block to inject, plus what had to be left out of it. */
226
+ export type MemoryView = {
227
+ /** Text to render into the persistent-memory prompt section. */
228
+ text: string;
229
+ /** True when the file did not fit whole — drives the "Read for more" note. */
230
+ truncated: boolean;
231
+ /** Human-readable list of ranked-out sections, or "" when none. */
232
+ omitted: string;
233
+ };
234
+
235
+ /** Head-slice at the cap, snapping back to a newline so the cut isn't mid-line. */
236
+ function headSlice(content: string, budget: number): string {
237
+ const head = content.slice(0, budget);
238
+ const lastNewline = head.lastIndexOf("\n");
239
+ return (lastNewline > 0 ? head.slice(0, lastNewline) : head).trimEnd();
240
+ }
241
+
242
+ /** Render omitted titles as one prose list, eliding a long tail. */
243
+ function formatOmitted(titles: readonly string[]): string {
244
+ if (titles.length === 0) return "";
245
+ if (titles.length <= MAX_OMITTED_NAMED) return titles.join("; ");
246
+ const named = titles.slice(0, MAX_OMITTED_NAMED).join("; ");
247
+ return `${named}; and ${titles.length - MAX_OMITTED_NAMED} more`;
248
+ }
249
+
250
+ /**
251
+ * Render the injectable view of `memory.md`.
252
+ *
253
+ * Under the cap this is the identity function — same bytes in, same bytes
254
+ * out. Over the cap, sections are collapsed and ranked as described in the
255
+ * module docstring, and anything that didn't fit is named in `omitted`.
256
+ */
257
+ export function renderMemoryView(
258
+ content: string,
259
+ budget: number = MEMORY_INJECT_MAX_CHARS,
260
+ ): MemoryView {
261
+ if (content.length <= budget) {
262
+ return { text: content, truncated: false, omitted: "" };
263
+ }
264
+
265
+ const { preamble, sections } = parseSections(content);
266
+ const { kept, dropped } = collapseFamilies(sections);
267
+
268
+ // Tier first, then original file order — stable within a tier so the
269
+ // author's own ordering survives wherever priority doesn't decide.
270
+ const ranked = [...kept].sort((a, b) => a.tier - b.tier || a.order - b.order);
271
+
272
+ const head = preamble.trimEnd();
273
+ let spent = head.length;
274
+ const chosen: Section[] = [];
275
+ const omitted: Section[] = [...dropped];
276
+ for (const s of ranked) {
277
+ const cost = s.body.trimEnd().length + 2; // body + section separator
278
+ if (spent + cost <= budget) {
279
+ spent += cost;
280
+ chosen.push(s);
281
+ } else {
282
+ omitted.push(s);
283
+ }
284
+ }
285
+
286
+ // Nothing fit — a single section is bigger than the whole budget. Degrade
287
+ // to the old behaviour rather than injecting an empty memory block.
288
+ if (chosen.length === 0) {
289
+ return { text: headSlice(content, budget), truncated: true, omitted: "" };
290
+ }
291
+
292
+ chosen.sort((a, b) => a.order - b.order);
293
+ const body = chosen.map((s) => s.body.trimEnd()).join("\n\n");
294
+
295
+ return {
296
+ text: head ? `${head}\n\n${body}` : body,
297
+ truncated: true,
298
+ omitted: formatOmitted(
299
+ omitted
300
+ .sort((a, b) => a.tier - b.tier || a.order - b.order)
301
+ .map((s) => s.title),
302
+ ),
303
+ };
304
+ }