talon-agent 3.12.5 → 3.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/prompts/discord.md +2 -2
- package/prompts/dream.md +29 -9
- package/prompts/heartbeat.md +16 -4
- package/prompts/identity.md +45 -22
- package/prompts/native.md +2 -2
- package/prompts/system/heartbeat-agent.md +10 -0
- package/prompts/system/live-state.md +8 -0
- package/prompts/system/memory-recall.md +12 -0
- package/prompts/system/persistent-memory.md +4 -2
- package/prompts/system/workspace.md +2 -0
- package/prompts/teams.md +2 -2
- package/prompts/telegram.md +2 -2
- package/src/backend/claude-sdk/handler.ts +24 -1
- package/src/backend/claude-sdk/one-shot.ts +6 -0
- package/src/backend/claude-sdk/options.ts +24 -9
- package/src/backend/claude-sdk/stream.ts +16 -0
- package/src/backend/shared/cache-telemetry.ts +297 -0
- package/src/backend/shared/index.ts +10 -0
- package/src/core/background/dream.ts +1 -0
- package/src/core/background/heartbeat/agent.ts +21 -0
- package/src/core/prompt/assemble.ts +40 -27
- package/src/core/prompt/embedded-prompts.ts +18 -16
- package/src/core/prompt/index.ts +1 -0
- package/src/core/prompt/memory-view.ts +304 -0
- package/src/util/paths.ts +20 -0
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Prompt-cache telemetry — the numbers needed to tell whether Talon's
|
|
3
|
+
* prompt cache is actually working, and why not when it isn't.
|
|
4
|
+
*
|
|
5
|
+
* Motivation (see docs/memory-persona-plan.md §3.6): the obvious lever for a
|
|
6
|
+
* sparse-traffic chat bot would be a 1-hour cache TTL, but
|
|
7
|
+
* `@anthropic-ai/claude-agent-sdk` exposes no cache-control surface at all —
|
|
8
|
+
* it owns `cache_control` placement internally. So the only levers Talon has
|
|
9
|
+
* are (a) keeping the prompt prefix byte-stable and (b) keeping it small,
|
|
10
|
+
* and the only way to know which one is costing money is to measure.
|
|
11
|
+
*
|
|
12
|
+
* Four things are measured here, each answering a question the aggregate
|
|
13
|
+
* `cache=NN%` on the accounting line cannot:
|
|
14
|
+
*
|
|
15
|
+
* 1. **Cross-turn vs within-turn hits.** An agentic turn makes many model
|
|
16
|
+
* requests; every one after the first reads the prefix the first one
|
|
17
|
+
* just paid for. So a 97% aggregate hit rate is compatible with *every
|
|
18
|
+
* turn* re-writing the whole prefix. The turn's FIRST request is the
|
|
19
|
+
* only one that reports whether the previous turn's cache survived —
|
|
20
|
+
* that is the number that tracks cost.
|
|
21
|
+
* 2. **Tool-set churn.** Tool definitions render BEFORE the system prompt,
|
|
22
|
+
* so any change to the tool array invalidates the system prompt and the
|
|
23
|
+
* whole message history with it. A stable prompt behind an unstable
|
|
24
|
+
* tool list buys nothing, and Talon's per-chat MCP servers are exactly
|
|
25
|
+
* the shape that can shift mid-session.
|
|
26
|
+
* 3. **Lookback-window risk.** A cache breakpoint searches back at most
|
|
27
|
+
* `CACHE_LOOKBACK_BLOCKS` content blocks for a prior entry. A turn that
|
|
28
|
+
* emits more blocks than that can leave the *next* turn's breakpoint
|
|
29
|
+
* unable to find anything — a silent miss with no error.
|
|
30
|
+
* 4. **Sub-minimum prompts.** Below a model-specific token floor nothing
|
|
31
|
+
* caches, and the API reports no error. Talon's chat prompts clear every
|
|
32
|
+
* floor; its one-shot prompts (dream, heartbeat, cron) may not.
|
|
33
|
+
*
|
|
34
|
+
* Everything here is pure except `noteToolFingerprint`, which keeps a small
|
|
35
|
+
* per-chat map so it can name what changed rather than just that something
|
|
36
|
+
* did.
|
|
37
|
+
*/
|
|
38
|
+
|
|
39
|
+
import { logWarn } from "../../util/log.js";
|
|
40
|
+
|
|
41
|
+
// ── Per-turn cache stats ────────────────────────────────────────────────────
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* One model request's usage, as the SDK reports it in
|
|
45
|
+
* `result.usage.iterations`. Fields are optional because a given provider
|
|
46
|
+
* may omit any of them; absent is treated as zero.
|
|
47
|
+
*/
|
|
48
|
+
export type CacheIteration = {
|
|
49
|
+
readonly input_tokens?: number;
|
|
50
|
+
readonly cache_read_input_tokens?: number;
|
|
51
|
+
readonly cache_creation_input_tokens?: number;
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
/** Cache behaviour for one user-visible turn. */
|
|
55
|
+
export type TurnCacheStats = {
|
|
56
|
+
/** Model requests the SDK made inside this turn. */
|
|
57
|
+
readonly modelRequests: number;
|
|
58
|
+
/** Cache read on the turn's FIRST request — did the last turn's prefix live? */
|
|
59
|
+
readonly firstRead: number;
|
|
60
|
+
/** Cache written by the turn's first request. */
|
|
61
|
+
readonly firstWrite: number;
|
|
62
|
+
/** Cache read across every request in the turn. */
|
|
63
|
+
readonly totalRead: number;
|
|
64
|
+
/** Cache written across every request in the turn. */
|
|
65
|
+
readonly totalWrite: number;
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* What the turn's first request says about the previous turn's cache.
|
|
70
|
+
*
|
|
71
|
+
* - `hit` — the prefix survived; this turn read it instead of paying for it.
|
|
72
|
+
* - `miss` — the prefix was gone and had to be re-written. The expensive case,
|
|
73
|
+
* and the one a TTL shorter than the gap between turns produces.
|
|
74
|
+
* - `none` — nothing was cached either way: below the model's cacheable
|
|
75
|
+
* minimum, or a provider that doesn't cache at all.
|
|
76
|
+
*/
|
|
77
|
+
export type CrossTurnVerdict = "hit" | "miss" | "none";
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Fold a turn's per-request usage into cache stats. Returns undefined when
|
|
81
|
+
* the provider reported no iterations — the caller then logs nothing rather
|
|
82
|
+
* than inventing a verdict from aggregate totals it can't attribute.
|
|
83
|
+
*/
|
|
84
|
+
export function turnCacheStats(
|
|
85
|
+
iterations: readonly CacheIteration[] | undefined,
|
|
86
|
+
): TurnCacheStats | undefined {
|
|
87
|
+
if (!iterations || iterations.length === 0) return undefined;
|
|
88
|
+
const first = iterations[0]!;
|
|
89
|
+
let totalRead = 0;
|
|
90
|
+
let totalWrite = 0;
|
|
91
|
+
for (const it of iterations) {
|
|
92
|
+
totalRead += it.cache_read_input_tokens ?? 0;
|
|
93
|
+
totalWrite += it.cache_creation_input_tokens ?? 0;
|
|
94
|
+
}
|
|
95
|
+
return {
|
|
96
|
+
modelRequests: iterations.length,
|
|
97
|
+
firstRead: first.cache_read_input_tokens ?? 0,
|
|
98
|
+
firstWrite: first.cache_creation_input_tokens ?? 0,
|
|
99
|
+
totalRead,
|
|
100
|
+
totalWrite,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** Classify the turn's first request. See {@link CrossTurnVerdict}. */
|
|
105
|
+
export function crossTurnVerdict(stats: TurnCacheStats): CrossTurnVerdict {
|
|
106
|
+
if (stats.firstRead > 0) return "hit";
|
|
107
|
+
if (stats.firstWrite > 0) return "miss";
|
|
108
|
+
return "none";
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Compact suffix for the per-turn accounting line, e.g.
|
|
113
|
+
* `xturn=miss reqs=11 rw=8.0`. Deliberately terse and space-delimited so a
|
|
114
|
+
* week of logs can be parsed without a schema:
|
|
115
|
+
*
|
|
116
|
+
* - `xturn` — the cross-turn verdict; the field that tracks cost.
|
|
117
|
+
* - `reqs` — model requests in the turn; explains a high aggregate hit
|
|
118
|
+
* rate that isn't saving anything.
|
|
119
|
+
* - `rw` — total read ÷ total write. Above ~1 the turn amortised its
|
|
120
|
+
* write; at or below it, the write dominated.
|
|
121
|
+
*/
|
|
122
|
+
export function formatTurnCache(stats: TurnCacheStats): string {
|
|
123
|
+
const parts = [
|
|
124
|
+
`xturn=${crossTurnVerdict(stats)}`,
|
|
125
|
+
`reqs=${stats.modelRequests}`,
|
|
126
|
+
];
|
|
127
|
+
if (stats.totalWrite > 0) {
|
|
128
|
+
parts.push(`rw=${(stats.totalRead / stats.totalWrite).toFixed(1)}`);
|
|
129
|
+
}
|
|
130
|
+
return parts.join(" ");
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// ── Lookback window ────────────────────────────────────────────────────────
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* How far back a cache breakpoint searches for a prior entry, in content
|
|
137
|
+
* blocks. A turn that emits more than this can prevent the NEXT turn from
|
|
138
|
+
* finding any cache to read.
|
|
139
|
+
*/
|
|
140
|
+
export const CACHE_LOOKBACK_BLOCKS = 20;
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* Rough content-block count for a turn from its tool-call count. Each tool
|
|
144
|
+
* call is a `tool_use` block plus a `tool_result` block, and the assistant
|
|
145
|
+
* text around them adds at least one more — so `2n + 1` is a floor, not an
|
|
146
|
+
* estimate. Used only to decide whether to warn, never to report a number.
|
|
147
|
+
*/
|
|
148
|
+
export function estimateTurnBlocks(toolCalls: number): number {
|
|
149
|
+
return Math.max(0, toolCalls) * 2 + 1;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** True when a turn plausibly emitted more blocks than a breakpoint looks back. */
|
|
153
|
+
export function exceedsLookbackWindow(toolCalls: number): boolean {
|
|
154
|
+
return estimateTurnBlocks(toolCalls) > CACHE_LOOKBACK_BLOCKS;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// ── Cacheable minimum ──────────────────────────────────────────────────────
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Minimum cacheable prefix, in tokens, by model. Below this nothing is
|
|
161
|
+
* cached and the API reports no error — `cache_creation_input_tokens` is
|
|
162
|
+
* simply 0.
|
|
163
|
+
*
|
|
164
|
+
* The floor is NOT monotonic across generations (512 on the newest models,
|
|
165
|
+
* 4096 on Opus 4.6 and Haiku 4.5), so it can't be inferred from a version
|
|
166
|
+
* number. Longest match wins, so `sonnet-4-6` is checked before `sonnet`.
|
|
167
|
+
*
|
|
168
|
+
* Bare aliases resolve only where Talon's catalog leaves no ambiguity:
|
|
169
|
+
* `haiku` is Haiku 4.5 (and carries the largest floor, making it the most
|
|
170
|
+
* likely to silently not cache). Ambiguous aliases — `opus`, `sonnet`,
|
|
171
|
+
* `default` — deliberately return undefined: a warning that fires on the
|
|
172
|
+
* wrong model teaches people to ignore warnings.
|
|
173
|
+
*/
|
|
174
|
+
const CACHE_MINIMUMS: readonly (readonly [string, number])[] = [
|
|
175
|
+
["fable-5", 512],
|
|
176
|
+
["mythos-5", 512],
|
|
177
|
+
["opus-5", 512],
|
|
178
|
+
["opus-4-8", 1024],
|
|
179
|
+
["sonnet-5", 1024],
|
|
180
|
+
["sonnet-4-6", 1024],
|
|
181
|
+
["sonnet-4-5", 1024],
|
|
182
|
+
["opus-4-1", 1024],
|
|
183
|
+
["opus-4-7", 2048],
|
|
184
|
+
["mythos-preview", 2048],
|
|
185
|
+
["opus-4-6", 4096],
|
|
186
|
+
["opus-4-5", 4096],
|
|
187
|
+
["haiku-4-5", 4096],
|
|
188
|
+
["haiku", 4096],
|
|
189
|
+
];
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* The cacheable-prefix floor for a model, or undefined when the model string
|
|
193
|
+
* doesn't unambiguously identify one.
|
|
194
|
+
*/
|
|
195
|
+
export function cacheMinimumTokens(model: string): number | undefined {
|
|
196
|
+
const m = model.toLowerCase();
|
|
197
|
+
let best: { key: string; min: number } | undefined;
|
|
198
|
+
for (const [key, min] of CACHE_MINIMUMS) {
|
|
199
|
+
if (!m.includes(key)) continue;
|
|
200
|
+
if (!best || key.length > best.key.length) best = { key, min };
|
|
201
|
+
}
|
|
202
|
+
return best?.min;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** Cheap tokenizer-free estimate (~4 chars/token), matching soul/projector. */
|
|
206
|
+
function estimateTokens(text: string): number {
|
|
207
|
+
return Math.ceil(text.length / 4);
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Warn when a prompt is too small to be cacheable on its model. No-op when
|
|
212
|
+
* the model's floor is unknown or the prompt clears it. `label` names the
|
|
213
|
+
* caller (e.g. `"dream"`) so the warning points somewhere.
|
|
214
|
+
*/
|
|
215
|
+
export function warnIfBelowCacheMinimum(
|
|
216
|
+
label: string,
|
|
217
|
+
model: string,
|
|
218
|
+
prompt: string,
|
|
219
|
+
): void {
|
|
220
|
+
const min = cacheMinimumTokens(model);
|
|
221
|
+
if (min === undefined) return;
|
|
222
|
+
const estimated = estimateTokens(prompt);
|
|
223
|
+
if (estimated >= min) return;
|
|
224
|
+
logWarn(
|
|
225
|
+
"agent",
|
|
226
|
+
`[${label}] prompt ~${estimated} tokens is below ${model}'s ${min}-token ` +
|
|
227
|
+
`cacheable minimum — nothing will be cached for this run`,
|
|
228
|
+
);
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
// ── Tool-set fingerprint ───────────────────────────────────────────────────
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Per-chat fingerprint of the last tool set seen. Bounded so a long-lived
|
|
235
|
+
* process with many chats can't grow it without limit; eviction is
|
|
236
|
+
* insertion-ordered, and a false "changed" warning after eviction is
|
|
237
|
+
* cheaper than unbounded retention.
|
|
238
|
+
*/
|
|
239
|
+
const MAX_TRACKED_CHATS = 256;
|
|
240
|
+
const lastToolSets = new Map<string, readonly string[]>();
|
|
241
|
+
|
|
242
|
+
/** Stable fingerprint for a tool set: sorted, deduped names. */
|
|
243
|
+
export function toolFingerprint(
|
|
244
|
+
builtinTools: readonly string[],
|
|
245
|
+
mcpServerNames: readonly string[],
|
|
246
|
+
): readonly string[] {
|
|
247
|
+
return [
|
|
248
|
+
...new Set([...builtinTools, ...mcpServerNames.map((n) => `mcp:${n}`)]),
|
|
249
|
+
].sort();
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
/**
|
|
253
|
+
* Record this turn's tool set for a chat and warn when it differs from the
|
|
254
|
+
* previous turn's. Returns true when a change was detected.
|
|
255
|
+
*
|
|
256
|
+
* Tools render before the system prompt, so a mid-session change invalidates
|
|
257
|
+
* the system prompt and every cached message after it — the most expensive
|
|
258
|
+
* cache event available, and invisible in aggregate hit-rate numbers.
|
|
259
|
+
*/
|
|
260
|
+
export function noteToolFingerprint(
|
|
261
|
+
chatId: string,
|
|
262
|
+
fingerprint: readonly string[],
|
|
263
|
+
): boolean {
|
|
264
|
+
const previous = lastToolSets.get(chatId);
|
|
265
|
+
|
|
266
|
+
if (lastToolSets.size >= MAX_TRACKED_CHATS && !previous) {
|
|
267
|
+
const oldest = lastToolSets.keys().next().value;
|
|
268
|
+
if (oldest !== undefined) lastToolSets.delete(oldest);
|
|
269
|
+
}
|
|
270
|
+
lastToolSets.set(chatId, fingerprint);
|
|
271
|
+
|
|
272
|
+
if (!previous) return false;
|
|
273
|
+
if (
|
|
274
|
+
previous.length === fingerprint.length &&
|
|
275
|
+
previous.every((t, i) => t === fingerprint[i])
|
|
276
|
+
) {
|
|
277
|
+
return false;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
const before = new Set(previous);
|
|
281
|
+
const after = new Set(fingerprint);
|
|
282
|
+
const added = fingerprint.filter((t) => !before.has(t));
|
|
283
|
+
const removed = previous.filter((t) => !after.has(t));
|
|
284
|
+
logWarn(
|
|
285
|
+
"agent",
|
|
286
|
+
`[${chatId}] tool set changed mid-session — invalidates the whole prompt ` +
|
|
287
|
+
`cache for this chat` +
|
|
288
|
+
(added.length ? ` (+${added.join(",")})` : "") +
|
|
289
|
+
(removed.length ? ` (-${removed.join(",")})` : ""),
|
|
290
|
+
);
|
|
291
|
+
return true;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/** Drop all tracked fingerprints (tests / explicit reset). */
|
|
295
|
+
export function resetToolFingerprints(): void {
|
|
296
|
+
lastToolSets.clear();
|
|
297
|
+
}
|
|
@@ -68,6 +68,16 @@ export {
|
|
|
68
68
|
type TokenUsageSnapshot,
|
|
69
69
|
} from "./usage.js";
|
|
70
70
|
|
|
71
|
+
// Only what is consumed THROUGH the barrel. Everything else in
|
|
72
|
+
// cache-telemetry.ts is imported from the module directly, matching the
|
|
73
|
+
// barrel discipline the prompt/ barrel was just trimmed to.
|
|
74
|
+
export {
|
|
75
|
+
formatTurnCache,
|
|
76
|
+
estimateTurnBlocks,
|
|
77
|
+
exceedsLookbackWindow,
|
|
78
|
+
CACHE_LOOKBACK_BLOCKS,
|
|
79
|
+
} from "./cache-telemetry.js";
|
|
80
|
+
|
|
71
81
|
export {
|
|
72
82
|
prepareSystemPrompt,
|
|
73
83
|
appendBackendSuffix,
|
|
@@ -235,6 +235,7 @@ If commands fail, log the error and continue — this stage is optional.`
|
|
|
235
235
|
.replace(/\{\{lastRunIso\}\}/g, lastRunIso)
|
|
236
236
|
.replace(/\{\{memoryFile\}\}/g, memoryFile)
|
|
237
237
|
.replace(/\{\{dailyMemoryDir\}\}/g, dirs.dailyMemory)
|
|
238
|
+
.replace(/\{\{memoryArchiveDir\}\}/g, dirs.memoryArchive)
|
|
238
239
|
.replace(/\{\{mempalaceSection\}\}/g, mempalaceSection);
|
|
239
240
|
} catch {
|
|
240
241
|
throw new Error("Failed to read dream prompt (dream.md)");
|
|
@@ -127,14 +127,17 @@ export async function runHeartbeatAgent(
|
|
|
127
127
|
|
|
128
128
|
let prompt: string;
|
|
129
129
|
let hadGoalsVar: boolean;
|
|
130
|
+
let hadStateVar: boolean;
|
|
130
131
|
try {
|
|
131
132
|
const raw = readFileSync(promptPath, "utf-8");
|
|
132
133
|
hadGoalsVar = raw.includes("{{goals}}");
|
|
134
|
+
hadStateVar = raw.includes("{{stateFile}}");
|
|
133
135
|
prompt = raw
|
|
134
136
|
.replace(/\{\{workspace\}\}/g, workspace)
|
|
135
137
|
.replace(/\{\{logsDir\}\}/g, logsDir)
|
|
136
138
|
.replace(/\{\{lastRunIso\}\}/g, lastRunIso)
|
|
137
139
|
.replace(/\{\{memoryFile\}\}/g, memoryFile)
|
|
140
|
+
.replace(/\{\{stateFile\}\}/g, pathFiles.state)
|
|
138
141
|
.replace(/\{\{instructionsFile\}\}/g, instructionsFile)
|
|
139
142
|
.replace(/\{\{dailyMemoryFile\}\}/g, dailyMemoryFile)
|
|
140
143
|
.replace(/\{\{runCount\}\}/g, String(runCount))
|
|
@@ -155,6 +158,24 @@ export async function runHeartbeatAgent(
|
|
|
155
158
|
}).trim()}`;
|
|
156
159
|
}
|
|
157
160
|
|
|
161
|
+
// Same vintage problem for the memory/state split: a seeded heartbeat.md
|
|
162
|
+
// from before it still instructs the agent to write memory.md, which is
|
|
163
|
+
// how status snapshots accreted in the durable store in the first place.
|
|
164
|
+
// Append the ownership rules so the split holds regardless of template
|
|
165
|
+
// vintage — the seeded copy is never rewritten once the user owns it.
|
|
166
|
+
if (!hadStateVar) {
|
|
167
|
+
logWarn(
|
|
168
|
+
"heartbeat",
|
|
169
|
+
`Seeded ${promptPath} predates the memory/state split — appending ` +
|
|
170
|
+
`file-ownership rules. Delete that file to re-seed the current prompt.`,
|
|
171
|
+
);
|
|
172
|
+
prompt += `\n\n${loadSystemTemplate("heartbeat-agent", {
|
|
173
|
+
mode: "state-fallback",
|
|
174
|
+
stateFile: pathFiles.state,
|
|
175
|
+
memoryFile,
|
|
176
|
+
}).trim()}`;
|
|
177
|
+
}
|
|
178
|
+
|
|
158
179
|
const model = config.heartbeatModel ?? config.model ?? getDefaultModel();
|
|
159
180
|
|
|
160
181
|
const backend = config.getBackend?.() ?? null;
|
|
@@ -16,8 +16,12 @@
|
|
|
16
16
|
* 2. Core behaviour ~/.talon/prompts/custom.md,
|
|
17
17
|
* else base.md, else fallback
|
|
18
18
|
* 3. Frontend capabilities ~/.talon/prompts/<frontend>.md
|
|
19
|
-
* 4. Persistent memory (
|
|
19
|
+
* 4. Persistent memory (ranked, capped) prompts/system/persistent-memory.md
|
|
20
20
|
* wrapping ~/.talon/workspace/memory/memory.md
|
|
21
|
+
* via memory-view.ts
|
|
22
|
+
* 4.5 Live state (capped) prompts/system/live-state.md
|
|
23
|
+
* wrapping ~/.talon/workspace/memory/state.md
|
|
24
|
+
* (heartbeat-owned, rewritten whole)
|
|
21
25
|
* 5. Memory recall + capability docs prompts/system/{memory-recall,workspace,...}.md
|
|
22
26
|
* 6. Plugin additions plugin.systemPrompt() contributions
|
|
23
27
|
* (7. Delivery contract — appended by the backend as its suffix,
|
|
@@ -57,6 +61,7 @@ import { dirs, files as pathFiles } from "../../util/paths.js";
|
|
|
57
61
|
import { todayAndYesterday } from "../../util/time.js";
|
|
58
62
|
import { log } from "../../util/log.js";
|
|
59
63
|
import { loadSystemTemplate } from "./templates.js";
|
|
64
|
+
import { renderMemoryView } from "./memory-view.js";
|
|
60
65
|
import { renderWorkspaceListing } from "./workspace-listing.js";
|
|
61
66
|
import { renderSkillsPrompt } from "../../storage/skill-store.js";
|
|
62
67
|
import { renderStickerLibraryPrompt } from "../../storage/sticker-store.js";
|
|
@@ -95,13 +100,14 @@ export function joinSystemPromptParts(parts: SystemPromptParts): string {
|
|
|
95
100
|
// ── Tunables ────────────────────────────────────────────────────────────────
|
|
96
101
|
|
|
97
102
|
/**
|
|
98
|
-
* Cap on
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
+
* Cap on the injected `state.md` block. Deliberately much tighter than the
|
|
104
|
+
* memory cap: this is a status snapshot the heartbeat rewrites every run, so
|
|
105
|
+
* anything past a couple of thousand chars means the heartbeat is
|
|
106
|
+
* accumulating history in a file that is supposed to be replaced — the
|
|
107
|
+
* failure the memory/state split exists to prevent. Truncating loudly is the
|
|
108
|
+
* signal that it is happening.
|
|
103
109
|
*/
|
|
104
|
-
export const
|
|
110
|
+
export const STATE_INJECT_MAX_CHARS = 2_000;
|
|
105
111
|
|
|
106
112
|
// ── Helpers ─────────────────────────────────────────────────────────────────
|
|
107
113
|
|
|
@@ -114,23 +120,6 @@ function readOptionalFile(path: string): string {
|
|
|
114
120
|
return "";
|
|
115
121
|
}
|
|
116
122
|
|
|
117
|
-
/**
|
|
118
|
-
* Truncate memory content at the cap, snapping back to the previous
|
|
119
|
-
* newline so the cut never lands mid-sentence. Returns the (possibly
|
|
120
|
-
* shortened) content and whether truncation happened.
|
|
121
|
-
*/
|
|
122
|
-
function capMemory(content: string): { text: string; truncated: boolean } {
|
|
123
|
-
if (content.length <= MEMORY_INJECT_MAX_CHARS) {
|
|
124
|
-
return { text: content, truncated: false };
|
|
125
|
-
}
|
|
126
|
-
const head = content.slice(0, MEMORY_INJECT_MAX_CHARS);
|
|
127
|
-
const lastNewline = head.lastIndexOf("\n");
|
|
128
|
-
return {
|
|
129
|
-
text: (lastNewline > 0 ? head.slice(0, lastNewline) : head).trimEnd(),
|
|
130
|
-
truncated: true,
|
|
131
|
-
};
|
|
132
|
-
}
|
|
133
|
-
|
|
134
123
|
let lastLoggedPromptKey = "";
|
|
135
124
|
|
|
136
125
|
// ── Assembly ────────────────────────────────────────────────────────────────
|
|
@@ -194,17 +183,41 @@ export function assembleSystemPrompt(
|
|
|
194
183
|
}
|
|
195
184
|
|
|
196
185
|
// 4. Persistent memory — size-capped so a memory file that has grown
|
|
197
|
-
// for months can't bloat every session from turn 0.
|
|
186
|
+
// for months can't bloat every session from turn 0. Over the cap the
|
|
187
|
+
// view ranks sections rather than head-slicing, so durable knowledge
|
|
188
|
+
// isn't evicted by whatever happens to sit at the top of the file
|
|
189
|
+
// (see prompt/memory-view.ts).
|
|
198
190
|
const memory = readOptionalFile(pathFiles.memory);
|
|
199
191
|
if (memory) {
|
|
200
|
-
const { text, truncated } =
|
|
192
|
+
const { text, truncated, omitted } = renderMemoryView(memory);
|
|
201
193
|
staticParts.push(
|
|
202
194
|
loadSystemTemplate("persistent-memory", {
|
|
203
195
|
content: text,
|
|
204
196
|
truncated: truncated ? "yes" : undefined,
|
|
197
|
+
omitted: omitted || undefined,
|
|
198
|
+
}),
|
|
199
|
+
);
|
|
200
|
+
loaded.push(truncated ? "memory(ranked)" : "memory");
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// 4.5. Live state — the heartbeat's rewritten-whole status snapshot, kept
|
|
204
|
+
// OUT of memory.md so "as of Run #N" sections can't accrete in the
|
|
205
|
+
// durable store and push real knowledge past the cap. Capped hard:
|
|
206
|
+
// this is the most volatile content in the static prompt, and a
|
|
207
|
+
// status file that grows is the exact failure this split exists to
|
|
208
|
+
// prevent.
|
|
209
|
+
const state = readOptionalFile(pathFiles.state);
|
|
210
|
+
if (state) {
|
|
211
|
+
const truncated = state.length > STATE_INJECT_MAX_CHARS;
|
|
212
|
+
staticParts.push(
|
|
213
|
+
loadSystemTemplate("live-state", {
|
|
214
|
+
content: truncated
|
|
215
|
+
? state.slice(0, STATE_INJECT_MAX_CHARS).trimEnd()
|
|
216
|
+
: state,
|
|
217
|
+
truncated: truncated ? "yes" : undefined,
|
|
205
218
|
}),
|
|
206
219
|
);
|
|
207
|
-
loaded.push(truncated ? "
|
|
220
|
+
loaded.push(truncated ? "state(capped)" : "state");
|
|
208
221
|
}
|
|
209
222
|
|
|
210
223
|
// 5. Package-owned behavioural and capability docs. The memory policy
|
|
@@ -26,14 +26,15 @@ import asset12 from "../../../prompts/system/cron.md" with { type: "file" };
|
|
|
26
26
|
import asset13 from "../../../prompts/system/daily-memory.md" with { type: "file" };
|
|
27
27
|
import asset14 from "../../../prompts/system/goals.md" with { type: "file" };
|
|
28
28
|
import asset15 from "../../../prompts/system/heartbeat-agent.md" with { type: "file" };
|
|
29
|
-
import asset16 from "../../../prompts/system/
|
|
30
|
-
import asset17 from "../../../prompts/system/
|
|
31
|
-
import asset18 from "../../../prompts/system/
|
|
32
|
-
import asset19 from "../../../prompts/system/
|
|
33
|
-
import asset20 from "../../../prompts/system/
|
|
34
|
-
import asset21 from "../../../prompts/
|
|
35
|
-
import asset22 from "../../../prompts/
|
|
36
|
-
import asset23 from "../../../prompts/
|
|
29
|
+
import asset16 from "../../../prompts/system/live-state.md" with { type: "file" };
|
|
30
|
+
import asset17 from "../../../prompts/system/memory-recall.md" with { type: "file" };
|
|
31
|
+
import asset18 from "../../../prompts/system/persistent-memory.md" with { type: "file" };
|
|
32
|
+
import asset19 from "../../../prompts/system/skills.md" with { type: "file" };
|
|
33
|
+
import asset20 from "../../../prompts/system/triggers.md" with { type: "file" };
|
|
34
|
+
import asset21 from "../../../prompts/system/workspace.md" with { type: "file" };
|
|
35
|
+
import asset22 from "../../../prompts/teams.md" with { type: "file" };
|
|
36
|
+
import asset23 from "../../../prompts/telegram.md" with { type: "file" };
|
|
37
|
+
import asset24 from "../../../prompts/terminal.md" with { type: "file" };
|
|
37
38
|
|
|
38
39
|
/** rel path (posix, under prompts/) → embedded file path (/$bunfs/… when compiled). */
|
|
39
40
|
const ASSETS: Record<string, string> = {
|
|
@@ -53,14 +54,15 @@ const ASSETS: Record<string, string> = {
|
|
|
53
54
|
"system/daily-memory.md": asset13,
|
|
54
55
|
"system/goals.md": asset14,
|
|
55
56
|
"system/heartbeat-agent.md": asset15,
|
|
56
|
-
"system/
|
|
57
|
-
"system/
|
|
58
|
-
"system/
|
|
59
|
-
"system/
|
|
60
|
-
"system/
|
|
61
|
-
"
|
|
62
|
-
"
|
|
63
|
-
"
|
|
57
|
+
"system/live-state.md": asset16,
|
|
58
|
+
"system/memory-recall.md": asset17,
|
|
59
|
+
"system/persistent-memory.md": asset18,
|
|
60
|
+
"system/skills.md": asset19,
|
|
61
|
+
"system/triggers.md": asset20,
|
|
62
|
+
"system/workspace.md": asset21,
|
|
63
|
+
"teams.md": asset22,
|
|
64
|
+
"telegram.md": asset23,
|
|
65
|
+
"terminal.md": asset24,
|
|
64
66
|
};
|
|
65
67
|
|
|
66
68
|
/** Read an embedded prompt by its rel path (e.g. "system/cron.md"). */
|
package/src/core/prompt/index.ts
CHANGED
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
*
|
|
4
4
|
* One stop for everything system-prompt:
|
|
5
5
|
* - `assemble` — section pipeline + static/dynamic split
|
|
6
|
+
* - `memory-view` — ranked selection of `memory.md` under a budget
|
|
6
7
|
* - `templates` — package-owned `prompts/system/*.md` loader
|
|
7
8
|
* - `workspace-listing` — lazy workspace tree for the dynamic tail
|
|
8
9
|
*
|