talon-agent 3.7.0 → 3.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +16 -0
  2. package/package.json +2 -2
  3. package/prompts/README.md +10 -2
  4. package/prompts/dream.md +5 -3
  5. package/prompts/heartbeat.md +1 -1
  6. package/prompts/identity.md +1 -1
  7. package/prompts/mem0.md +7 -5
  8. package/prompts/mempalace.md +7 -5
  9. package/prompts/system/memory-recall.md +49 -0
  10. package/prompts/system/workspace.md +2 -2
  11. package/src/backend/claude-sdk/one-shot.ts +31 -1
  12. package/src/backend/codex/effort.ts +30 -0
  13. package/src/backend/codex/handler/message.ts +12 -22
  14. package/src/backend/codex/one-shot.ts +20 -0
  15. package/src/backend/opencode/handler/message.ts +7 -13
  16. package/src/backend/remote-server/one-shot.ts +6 -0
  17. package/src/backend/shared/handler-to-events.ts +0 -1
  18. package/src/backend/shared/handler-types.ts +0 -8
  19. package/src/backend/shared/index.ts +1 -6
  20. package/src/backend/shared/prompt-format.ts +0 -69
  21. package/src/bootstrap.ts +2 -0
  22. package/src/core/agent-runtime/capabilities.ts +5 -49
  23. package/src/core/background/dream.ts +27 -2
  24. package/src/core/background/effort.ts +80 -0
  25. package/src/core/background/heartbeat/agent.ts +22 -1
  26. package/src/core/background/heartbeat/state.ts +7 -0
  27. package/src/core/engine/model-audit.ts +69 -2
  28. package/src/core/errors.ts +107 -3
  29. package/src/core/models/reasoning-levels.ts +14 -3
  30. package/src/core/prompt/assemble.ts +8 -5
  31. package/src/core/prompt/embedded-prompts.ts +16 -14
  32. package/src/core/types.ts +10 -0
  33. package/src/core/weaver/index.ts +0 -1
  34. package/src/core/weaver/weaver.ts +1 -24
  35. package/src/frontend/telegram/callbacks/index.ts +8 -0
  36. package/src/frontend/telegram/callbacks/metrics.ts +36 -0
  37. package/src/frontend/telegram/commands/admin.ts +9 -13
  38. package/src/frontend/telegram/commands/info.ts +3 -44
  39. package/src/frontend/telegram/helpers/diagnostics.ts +190 -70
  40. package/src/util/config.ts +34 -19
  41. package/src/core/memory/retrieval.ts +0 -92
  42. package/src/core/weaver/memory-prefetch.ts +0 -65
package/README.md CHANGED
@@ -385,6 +385,10 @@ Config file: `~/.talon/config.json`
385
385
  | `pulse` | `true` | Periodic group engagement |
386
386
  | `heartbeat` | `false` | Background maintenance agent |
387
387
  | `heartbeatIntervalMinutes` | `60` | Heartbeat interval |
388
+ | `heartbeatModel` | --- | Model for the heartbeat agent (falls back to `model`) |
389
+ | `heartbeatEffort` | --- | Reasoning effort for the heartbeat agent: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Unset = the model's own default |
390
+ | `dreamModel` | --- | Model for dream / memory consolidation (falls back to `model`) |
391
+ | `dreamEffort` | --- | Reasoning effort for the dream agent — same levels as `heartbeatEffort` |
388
392
  | `braveApiKey` | --- | Brave Search API key |
389
393
  | `timezone` | --- | IANA timezone (e.g. `"Europe/London"`) |
390
394
  | `plugins` | `[]` | External plugin packages |
@@ -398,6 +402,18 @@ Config file: `~/.talon/config.json`
398
402
  | `mempalace` | --- | Legacy MemPalace plugin config (prefer `memory`) |
399
403
  | `playwright` | --- | Playwright plugin config (see above) |
400
404
 
405
+ ### Background reasoning effort
406
+
407
+ `heartbeatEffort` / `dreamEffort` set how hard the background agents think —
408
+ useful when you want unattended goal work to reason harder than a chat turn,
409
+ or hourly heartbeats to stay cheap. Chat effort stays per-chat (`/settings`).
410
+
411
+ Which levels a model accepts comes from its catalog entry, so the usable set
412
+ differs per model (`max` is Claude's ceiling, `xhigh` is Codex's). A level the
413
+ model doesn't offer is dropped — the run proceeds on the model default, the
414
+ reason is written to the run log, and the boot-time model audit warns about it.
415
+ Backends with no reasoning knob at all (Kilo, OpenCode) ignore the setting.
416
+
401
417
  ---
402
418
 
403
419
  ## Terminal Mode
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.7.0",
3
+ "version": "3.8.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -105,7 +105,7 @@
105
105
  "@openai/agents": "^0.13.0",
106
106
  "@openai/codex-sdk": "^0.145.0",
107
107
  "@opencode-ai/sdk": "^1.17.4",
108
- "@playwright/mcp": "0.0.56",
108
+ "@playwright/mcp": "0.0.78",
109
109
  "@types/cross-spawn": "^6.0.6",
110
110
  "big-integer": "^1.6.52",
111
111
  "cheerio": "^1.2.0",
package/prompts/README.md CHANGED
@@ -16,7 +16,7 @@ placed after the cache boundary):
16
16
  | 2 | Core behaviour | `custom.md` (replaces `base.md` when present) | static |
17
17
  | 3 | Frontend capabilities | `<frontend>.md` (telegram / discord / teams / terminal / native) | static |
18
18
  | 4 | Persistent memory | `system/persistent-memory.md` wrapping `memory/memory.md`, size-capped | static |
19
- | 5 | Capability docs | `system/workspace.md`, `system/cron.md`, `system/triggers.md` | static |
19
+ | 5 | Memory + capabilities | `system/memory-recall.md`, workspace, cron, triggers, goals, skills | static |
20
20
  | 6 | Plugin additions | each plugin's `systemPrompt()` contribution | static |
21
21
  | 7 | **Delivery contract** | `system/contract-*.md`, appended by the **backend** as its suffix | static (tail) |
22
22
  | 8 | Daily-memory pointer | `system/daily-memory.md` (names today's file) | dynamic |
@@ -42,7 +42,7 @@ when a backend has no native "skills" feature.
42
42
  **User-editable prompts** (everything at the top level of this directory:
43
43
  `identity.md`, `base.md`, `custom.md`, `telegram.md`, `discord.md`,
44
44
  `teams.md`, `terminal.md`, `native.md`, `heartbeat.md`, `dream.md`,
45
- `mempalace.md`) are
45
+ `mempalace.md`, `mem0.md`) are
46
46
  seeded into `~/.talon/prompts/` on first run and read from there.
47
47
  Seeding is upgrade-aware (dpkg-conffile semantics, tracked via a
48
48
  `.seeded.json` hash manifest next to the seeded files): a file the user
@@ -70,6 +70,14 @@ User-editable prompts are NOT Liquid: their consumers substitute
70
70
  `{{var}}` placeholders with plain string replacement, so a user edit
71
71
  can never break prompt assembly with a template syntax error.
72
72
 
73
+ `system/memory-recall.md` is the provider-neutral continuity policy. It guides
74
+ agents to make a proportionate, thorough attempt across relevant memory,
75
+ workspace, log, and connected sources before asking a user to repeat
76
+ information, and to persist new information. When a memory plugin is enabled,
77
+ its prompt addition follows this section and names the preferred provider and
78
+ exact tools; without one, the policy falls back to `memory/memory.md` plus
79
+ daily notes.
80
+
73
81
  ## The delivery contract (response flow)
74
82
 
75
83
  How a reply reaches the user is a property of the **backend**, not the
package/prompts/dream.md CHANGED
@@ -23,7 +23,8 @@ You primarily use filesystem tools (Read, Write, Edit, Bash, Glob, Grep). Do NOT
23
23
  - Corrections to previously held beliefs
24
24
  - Operational patterns (e.g. who stays up late, who prefers what tools)
25
25
  - Project context changes inferred from the conversation (e.g. new repos, shifted priorities)
26
- - Be selective — only extract genuinely new or updated information
26
+ - Capture every genuinely new or updated piece of information; avoid
27
+ duplicating facts already represented in memory
27
28
 
28
29
  ### Stage 3 — Consolidate
29
30
 
@@ -38,8 +39,9 @@ You primarily use filesystem tools (Read, Write, Edit, Bash, Glob, Grep). Do NOT
38
39
  ### Stage 4 — Prune
39
40
 
40
41
  - Remove entries that have been explicitly contradicted
41
- - Remove entries that are clearly stale or irrelevant
42
- - Do NOT remove entries just because they're old — only remove if wrong or superseded
42
+ - Remove entries that are clearly superseded
43
+ - Do NOT remove entries just because they're old or seem unimportant — only
44
+ remove information that is wrong or replaced by a newer version
43
45
  - Write the updated memory.md back to `{{memoryFile}}`
44
46
 
45
47
  ### Stage 5 — Mine to MemPalace & Write Diary (optional)
@@ -32,7 +32,7 @@ For each goal:
32
32
  3. Record every advance with `update_goal(goal_id=..., progress_note=..., chat_id=<the goal's chat>)`. The `chat_id` parameter is REQUIRED in heartbeat mode — use the chat id shown next to the goal. Keep notes short and concrete: what was done, what was learned, what's blocked.
33
33
  4. When a goal's objective is achieved, set `status="completed"` and send a short, high-signal message to the goal's chat (explicit `chat_id` required). If a goal has become impossible or moot, set `status="abandoned"` with a note explaining why.
34
34
  5. If nothing can be done on a goal right now, skip it silently — do not write filler progress notes.
35
- 6. If MemPalace tools are available: `mempalace_search` for context relevant to a goal before working on it, and store durable learnings afterward (`mempalace_add_drawer` / `mempalace_kg_add`).
35
+ 6. If MemPalace tools are available: `mempalace_search` for context relevant to a goal before working on it, and store new information learned while working (`mempalace_add_drawer` / `mempalace_kg_add`).
36
36
 
37
37
  ## Instructions
38
38
 
@@ -33,4 +33,4 @@ When a filesystem-capable tool is available, persist the answers to `~/.talon/wo
33
33
 
34
34
  ## Memory
35
35
 
36
- When you learn something worth keeping — who people are, how they like to work, what they're building, decisions and facts that should outlive this session — persist it to `~/.talon/workspace/memory/memory.md` (when a filesystem-capable tool is available for this backend; otherwise hold it in working memory for the conversation and don't pretend to save). The test is simple: would future-you be glad this was written down? Update memory quietly as conversations happen — no announcements — and keep the file organized, current, and free of trivia.
36
+ When you learn new information — who people are, how they like to work, what they're building, decisions, facts, and surrounding context — follow the Memory and Recall policy in this prompt. Use the configured long-term-memory provider when one is available; otherwise use the workspace memory files.
package/prompts/mem0.md CHANGED
@@ -2,12 +2,14 @@
2
2
 
3
3
  You have access to mem0 long-term memory via MCP tools. mem0 extracts durable facts from what you store and retrieves them by semantic search. All memories are filed under the entity id `{{userId}}`.
4
4
 
5
- ### Protocol — FOLLOW EVERY SESSION
5
+ mem0 is the preferred durable-memory store while its tools are available. Workspace daily notes can still hold concise chronological context; `memory.md` remains a fallback if the mem0 tools are unavailable.
6
6
 
7
- 1. **BEFORE RESPONDING** about any person, project, or past event: call `mem0_search_memory` FIRST. Never guess — verify from memory.
8
- 2. **IF UNSURE** about a fact (name, age, relationship, preference): search memory. Wrong is worse than slow.
9
- 3. **WHEN FACTS CHANGE**: store the new fact with `mem0_add_memory` (mem0 supersedes contradicted memories itself); delete plainly wrong entries with `mem0_delete_memory`.
10
- 4. **AFTER LEARNING** something important: store it with `mem0_add_memory`. Pass natural conversational text — mem0 extracts the durable facts.
7
+ ### How to use it well
8
+
9
+ 1. Search with `mem0_search_memory` when prior context about a person, project, or past event could materially improve the answer.
10
+ 2. If a fact such as a name, relationship, or preference is uncertain, checking memory is usually better than guessing or asking the user to repeat it.
11
+ 3. When a fact changes, store the new version with `mem0_add_memory` (mem0 supersedes contradicted memories itself); use `mem0_delete_memory` for plainly wrong entries.
12
+ 4. When you learn new information, pass natural conversational text to `mem0_add_memory` so mem0 can extract and retain the facts.
11
13
 
12
14
  ### Tools
13
15
 
@@ -2,6 +2,8 @@
2
2
 
3
3
  You have access to a local memory palace via MCP tools. The palace stores verbatim conversation history and a temporal knowledge graph — all local, zero cloud, zero API calls.
4
4
 
5
+ MemPalace is the preferred durable-memory store while its tools are available. Workspace daily notes can still hold concise chronological context; `memory.md` remains a fallback if the MemPalace tools are unavailable.
6
+
5
7
  ### Architecture
6
8
 
7
9
  - **Wings** = top-level categories (people, projects, topics)
@@ -10,12 +12,12 @@ You have access to a local memory palace via MCP tools. The palace stores verbat
10
12
  - **Tunnels** = cross-wing links between related rooms (auto-created in mempalace 3.3.4+ when topics overlap, plus manual)
11
13
  - **Knowledge Graph** = entity-relationship facts with temporal validity
12
14
 
13
- ### Protocol — FOLLOW EVERY SESSION
15
+ ### How to use it well
14
16
 
15
- 1. **BEFORE RESPONDING** about any person, project, or past event: call `mempalace_search` or `mempalace_kg_query` FIRST. Never guess — verify from the palace.
16
- 2. **IF UNSURE** about a fact (name, age, relationship, preference): query the palace. Wrong is worse than slow.
17
- 3. **WHEN FACTS CHANGE**: Call `mempalace_kg_invalidate` on the old fact, then `mempalace_kg_add` for the new one.
18
- 4. **AFTER LEARNING** something important: store it. Use `mempalace_add_drawer` for rich context, `mempalace_kg_add` for structured facts.
17
+ 1. Search with `mempalace_search` or `mempalace_kg_query` when prior context about a person, project, or past event could materially improve the answer.
18
+ 2. If a fact such as a name, relationship, or preference is uncertain, checking the palace is usually better than guessing or asking the user to repeat it.
19
+ 3. When a fact changes, keep its history accurate with `mempalace_kg_invalidate` followed by `mempalace_kg_add`.
20
+ 4. When you learn new information, use `mempalace_add_drawer` for rich context or `mempalace_kg_add` for a structured fact.
19
21
 
20
22
  ### Tools
21
23
 
@@ -0,0 +1,49 @@
1
+ ## Memory and Recall
2
+
3
+ ### Recall before asking
4
+
5
+ Protect continuity. If a request relies on information the user reasonably
6
+ expects you to already have, or you are unsure about a prior fact, take that
7
+ as a cue to recover the context before asking them to repeat it. Make a
8
+ proportionate but thorough attempt across the relevant sources available to
9
+ you:
10
+
11
+ - the current conversation and memory already included in this prompt;
12
+ - enabled long-term-memory providers, including browse/fetch tools when a
13
+ search hit needs more context;
14
+ - `memory/memory.md`, including the rest when its prompt excerpt is
15
+ truncated, and relevant recent files in `memory/daily/`;
16
+ - workspace files, interaction logs, and connected sources that the request
17
+ suggests may contain the answer.
18
+
19
+ Start with the most likely source, then broaden rather than stopping after one
20
+ empty result. Alternate names, keywords, dates, or scopes can recover memories
21
+ that a first query misses; follow promising results to their full source and
22
+ weigh conflicts by recency and authority. Keep the effort relevant to the
23
+ request—ordinary questions about the current turn do not call for rummaging
24
+ through unrelated history.
25
+
26
+ If a meaningful search still leaves the answer missing, inaccessible, or
27
+ genuinely ambiguous, ask for the smallest piece of information needed and
28
+ briefly explain the gap.
29
+
30
+ ### Save new information
31
+
32
+ Proactively persist new information so future conversations can draw on it.
33
+ This includes preferences, relationships, decisions, corrections, project
34
+ context, durable facts, and details that may only become relevant later. Do
35
+ this naturally as you learn it, without interrupting the conversation. Keep
36
+ memory organized, update stale facts, and avoid duplicate copies.
37
+
38
+ - If a dedicated long-term-memory provider is described elsewhere in this
39
+ prompt and its tools are available, prefer it as the canonical durable store
40
+ and use its guidance for searching, adding, updating, and deduplicating
41
+ memories. Daily notes can still preserve useful chronological context.
42
+ - Otherwise, when filesystem tools are available, keep durable knowledge
43
+ organized and current in `~/.talon/workspace/memory/memory.md`, and use
44
+ today's `memory/daily/YYYY-MM-DD.md` for concise dated observations,
45
+ corrections, and follow-ups.
46
+ - If neither a memory provider nor filesystem tools are available, retain the
47
+ information only for the current conversation and never claim it was saved.
48
+
49
+ Memory updates should usually be quiet unless the user asks about them.
@@ -2,8 +2,8 @@
2
2
 
3
3
  You have a workspace directory at `~/.talon/workspace/`. This is your home — organize it however you want.
4
4
 
5
- - `memory/memory.md` — your persistent memory file. Update it (Write tool) when you learn important things.
6
- - `memory/daily/YYYY-MM-DD.md` — your daily notes: observations, learnings, corrections, follow-ups. Keep entries concise.
5
+ - `memory/memory.md` — your file-based persistent-memory fallback when no dedicated memory provider is available.
6
+ - `memory/daily/YYYY-MM-DD.md` — concise chronological notes: observations, learnings, corrections, and follow-ups.
7
7
  - `logs/` — daily interaction logs, written automatically.
8
8
  - `uploads/` — files users send you (photos, docs, voice) land here.
9
9
  - Everything else is yours to create and organize as you see fit.
@@ -15,8 +15,9 @@ import { readdir, readFile } from "node:fs/promises";
15
15
  import { query } from "@anthropic-ai/claude-agent-sdk";
16
16
  import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk";
17
17
  import type { OneShotAgentParams, OneShotUsage } from "../../core/types.js";
18
- import { log } from "../../util/log.js";
18
+ import { log, logWarn } from "../../util/log.js";
19
19
  import { ALLOWED_TOOLS_BACKGROUND } from "../../core/constants.js";
20
+ import { EFFORT_MAP } from "./constants.js";
20
21
  import { buildMcpServers, buildPluginMcpServers } from "./options.js";
21
22
 
22
23
  const DEFAULT_SUBPROCESS_KILL_GRACE_MS = 5 * 1000;
@@ -59,14 +60,26 @@ export async function runOneShotAgent(
59
60
  systemPrompt,
60
61
  workspace,
61
62
  model,
63
+ reasoningEffort,
62
64
  contextLabel,
63
65
  abortController,
64
66
  appendLog,
65
67
  } = params;
66
68
 
69
+ // Reasoning effort is opt-in for background runs (config `heartbeatEffort`
70
+ // / `dreamEffort`). Unset → omit the thinking options entirely so the SDK
71
+ // keeps whatever default the model ships with, which is what these runs
72
+ // did before the knob existed. The chat path applies an explicit
73
+ // `{ thinking: { type: "adaptive" } }` fallback instead because a chat has
74
+ // a persisted per-chat setting to honour; a one-shot has none.
75
+ const thinkingConfig = reasoningEffort
76
+ ? EFFORT_MAP[reasoningEffort]
77
+ : undefined;
78
+
67
79
  const options = {
68
80
  model,
69
81
  systemPrompt,
82
+ ...thinkingConfig,
70
83
  cwd: workspace,
71
84
  permissionMode: "bypassPermissions" as const,
72
85
  allowDangerouslySkipPermissions: true,
@@ -81,6 +94,23 @@ export async function runOneShotAgent(
81
94
  tools: [...ALLOWED_TOOLS_BACKGROUND],
82
95
  };
83
96
 
97
+ if (reasoningEffort && !thinkingConfig) {
98
+ // `minimal` / `xhigh` are Codex-side vocabulary with no Claude
99
+ // equivalent in EFFORT_MAP — the run proceeds on the model default
100
+ // rather than failing, but say so in the log so a configured knob that
101
+ // does nothing isn't silent.
102
+ logWarn(
103
+ "agent",
104
+ `[${contextLabel}] Claude one-shot: effort "${reasoningEffort}" has no ` +
105
+ `Claude mapping — using the model default`,
106
+ );
107
+ } else if (thinkingConfig) {
108
+ log(
109
+ "agent",
110
+ `[${contextLabel}] Claude one-shot effort: ${reasoningEffort}`,
111
+ );
112
+ }
113
+
84
114
  const qi = query({
85
115
  prompt,
86
116
  options: options as Parameters<typeof query>[0]["options"],
@@ -0,0 +1,30 @@
1
+ /**
2
+ * Codex reasoning-effort vocabulary mapping.
3
+ *
4
+ * Talon's canonical `ReasoningEffortLevel` is a superset of what Codex's
5
+ * `modelReasoningEffort` thread option accepts: `off` isn't expressible on a
6
+ * reasoning model, and `max` is Claude-only. Both simply fall through to the
7
+ * model's own default.
8
+ *
9
+ * This is pure vocabulary translation — the "does this model offer that
10
+ * level?" question is answered by the caller (per-chat settings for an
11
+ * interactive turn, `core/background/effort.ts` for heartbeat/dream) against
12
+ * the model catalog's `supportedReasoningLevels`. Keeping the mapping here
13
+ * means the chat path and the one-shot path can't drift apart.
14
+ */
15
+
16
+ import type { ReasoningEffortLevel } from "../../core/types.js";
17
+
18
+ /** The levels Codex's `modelReasoningEffort` thread option accepts. */
19
+ export type CodexReasoningEffort = Exclude<ReasoningEffortLevel, "off" | "max">;
20
+
21
+ /**
22
+ * Map a canonical level onto Codex's thread option, or undefined when Codex
23
+ * has no way to express it (`off`, `max`, or nothing requested).
24
+ */
25
+ export function toCodexReasoningEffort(
26
+ level: ReasoningEffortLevel | undefined,
27
+ ): CodexReasoningEffort | undefined {
28
+ if (!level || level === "off" || level === "max") return undefined;
29
+ return level;
30
+ }
@@ -29,7 +29,6 @@ import {
29
29
  recordTokens,
30
30
  finalizeResponseText,
31
31
  formatUserPrompt,
32
- formatPromptWithRetrievedMemory,
33
32
  prepareSystemPrompt,
34
33
  extractSessionName,
35
34
  summarizeUsage,
@@ -66,6 +65,7 @@ import {
66
65
  isCodexOAuthIncompat,
67
66
  } from "../models.js";
68
67
  import { supportsReasoningLevel } from "../../../core/models/reasoning-levels.js";
68
+ import { toCodexReasoningEffort } from "../effort.js";
69
69
  import { markOAuthIncompat } from "../oauth-incompat.js";
70
70
  import { readLastRolloutSnapshot } from "../token-usage.js";
71
71
  import { activeAborts } from "./state.js";
@@ -258,18 +258,12 @@ export async function handleMessage(
258
258
  sessionEpoch: session.createdAt,
259
259
  });
260
260
 
261
- // Retrieved memory wraps the FORMATTED live prompt (Phase B): it stays
262
- // outside the frozen system prompt, so the first-turn concatenation below
263
- // keeps the boundary "cached systemPrompt, separator, live prompt wrapper".
264
- const prompt = formatPromptWithRetrievedMemory(
265
- formatUserPrompt({
266
- text,
267
- senderName: senderName ?? "user",
268
- isGroup,
269
- messageId,
270
- }),
271
- params.retrievedMemory,
272
- );
261
+ const prompt = formatUserPrompt({
262
+ text,
263
+ senderName: senderName ?? "user",
264
+ isGroup,
265
+ messageId,
266
+ });
273
267
 
274
268
  log("agent", `[${chatId}] <- (${text.length} chars)`);
275
269
  traceMessage(chatId, "in", text, { senderName, isGroup });
@@ -283,22 +277,18 @@ export async function handleMessage(
283
277
  const supportedReasoningLevels =
284
278
  activeModelInfo?.supportedReasoningLevels ?? [];
285
279
  const requestedEffort = chatSettings.effort;
280
+ // Availability check (does this model offer the level?) then vocabulary
281
+ // translation (can Codex express it?) — the latter is shared with the
282
+ // one-shot path via `toCodexReasoningEffort` so the two can't drift.
286
283
  const modelReasoningEffort =
287
284
  requestedEffort &&
288
- requestedEffort !== "off" &&
289
- requestedEffort !== "max" &&
290
285
  supportsReasoningLevel(requestedEffort, supportedReasoningLevels)
291
- ? requestedEffort
286
+ ? toCodexReasoningEffort(requestedEffort)
292
287
  : undefined;
293
288
  const threadOptions = {
294
289
  model: activeModel,
295
290
  skipGitRepoCheck: true,
296
- ...(modelReasoningEffort
297
- ? {
298
- modelReasoningEffort: modelReasoningEffort as
299
- "minimal" | "low" | "medium" | "high" | "xhigh",
300
- }
301
- : {}),
291
+ ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
302
292
  ...CODEX_THREAD_PERMISSIONS,
303
293
  };
304
294
  const thread: Thread = session.sessionId
@@ -29,6 +29,7 @@ import {
29
29
  import { isChatGptModelMismatchError } from "./auth.js";
30
30
  import { chatGptFallbackFor, isCodexOAuthIncompat } from "./models.js";
31
31
  import { markOAuthIncompat } from "./oauth-incompat.js";
32
+ import { toCodexReasoningEffort } from "./effort.js";
32
33
 
33
34
  /**
34
35
  * Resolve the effective model for a one-shot run, applying the same
@@ -74,6 +75,7 @@ export async function runOneShotAgent(
74
75
  prompt,
75
76
  systemPrompt,
76
77
  model: requestedModel,
78
+ reasoningEffort,
77
79
  contextLabel,
78
80
  abortController,
79
81
  appendLog,
@@ -103,9 +105,27 @@ export async function runOneShotAgent(
103
105
  }
104
106
  log("agent", `[${contextLabel}] Codex one-shot model: ${activeModel}`);
105
107
 
108
+ // Availability was already checked by the caller against the model
109
+ // catalog (core/background/effort.ts); all that's left is Codex's own
110
+ // vocabulary, which can't express `off` / `max`.
111
+ const modelReasoningEffort = toCodexReasoningEffort(reasoningEffort);
112
+ if (reasoningEffort && !modelReasoningEffort) {
113
+ logWarn(
114
+ "agent",
115
+ `[${contextLabel}] Codex one-shot: effort "${reasoningEffort}" has no ` +
116
+ `Codex equivalent — using the model default`,
117
+ );
118
+ } else if (modelReasoningEffort) {
119
+ log(
120
+ "agent",
121
+ `[${contextLabel}] Codex one-shot effort: ${modelReasoningEffort}`,
122
+ );
123
+ }
124
+
106
125
  const thread = codex.startThread({
107
126
  model: activeModel,
108
127
  skipGitRepoCheck: true,
128
+ ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
109
129
  ...CODEX_THREAD_PERMISSIONS,
110
130
  });
111
131
 
@@ -39,7 +39,6 @@ import {
39
39
  recordTokens,
40
40
  finalizeResponseText,
41
41
  formatUserPrompt,
42
- formatPromptWithRetrievedMemory,
43
42
  prepareSystemPrompt,
44
43
  extractSessionName,
45
44
  summarizeUsage,
@@ -90,18 +89,13 @@ export async function handleMessage(
90
89
  await ensureChatMcpServer(oc, chatId);
91
90
  await ensurePluginMcpServers(oc, chatId);
92
91
 
93
- // Build the prompt (time tag + sender + msg_id reference), then wrap it
94
- // with any retrieved memory (Phase B). The wrapper goes into the live user
95
- // part only — `system` below stays byte-identical to the prepared prompt.
96
- const prompt = formatPromptWithRetrievedMemory(
97
- formatUserPrompt({
98
- text,
99
- senderName: senderName ?? "user",
100
- isGroup,
101
- messageId,
102
- }),
103
- params.retrievedMemory,
104
- );
92
+ // Build the prompt (time tag + sender + msg_id reference)
93
+ const prompt = formatUserPrompt({
94
+ text,
95
+ senderName: senderName ?? "user",
96
+ isGroup,
97
+ messageId,
98
+ });
105
99
 
106
100
  // Per-session frozen prompt + OpenCode-specific delivery suffix
107
101
  const { text: systemPrompt } = prepareSystemPrompt({
@@ -19,6 +19,12 @@
19
19
  * subprocesses to evict — `evictOrphanSubprocesses` is intentionally
20
20
  * not implemented for this family.
21
21
  *
22
+ * Note on reasoning effort: `params.reasoningEffort` (config
23
+ * `heartbeatEffort` / `dreamEffort`) is deliberately unused here. Neither
24
+ * server's `session.prompt` exposes a reasoning knob — the level is baked
25
+ * into the provider's model id when it's selectable at all — so the field
26
+ * is ignored rather than half-honoured.
27
+ *
22
28
  * Note on abort semantics: both SDKs expose `session.abort` but the
23
29
  * REST `prompt` endpoint blocks until the underlying provider
24
30
  * returns. The heartbeat module's outer abort grace is what actually
@@ -84,7 +84,6 @@ export async function* handlerToEvents(
84
84
  senderName: params.senderName,
85
85
  isGroup: params.isGroup,
86
86
  messageId: params.messageId,
87
- retrievedMemory: params.retrievedMemory,
88
87
  onStreamDelta: (accumulated) => {
89
88
  if (typeof accumulated !== "string" || accumulated.length === 0) {
90
89
  return;
@@ -17,8 +17,6 @@
17
17
 
18
18
  // ── Query lifecycle (backend-internal) ──────────────────────────────────────
19
19
 
20
- import type { RetrievedMemory } from "../../core/agent-runtime/capabilities.js";
21
-
22
20
  /** Parameters for a backend AI query. */
23
21
  export type QueryParams = {
24
22
  chatId: string;
@@ -35,12 +33,6 @@ export type QueryParams = {
35
33
  * Provider message ID. Telegram is numeric; Discord snowflakes are strings.
36
34
  */
37
35
  messageId?: number | string;
38
- /**
39
- * Optional pre-retrieved memory slice for this turn (Phase B). Handlers
40
- * fold it into the live user prompt via `formatPromptWithRetrievedMemory`;
41
- * it must never reach `prepareSystemPrompt()` or a backend `system` field.
42
- */
43
- retrievedMemory?: RetrievedMemory;
44
36
  onStreamDelta?: (accumulated: string, phase?: "thinking" | "text") => void;
45
37
  onTextBlock?: (text: string) => Promise<void>;
46
38
  /**
@@ -47,12 +47,7 @@ export {
47
47
 
48
48
  export { registerTurnInterrupt, interruptChatTurn } from "./turn-interrupt.js";
49
49
 
50
- export {
51
- formatUserPrompt,
52
- formatPromptWithRetrievedMemory,
53
- RETRIEVED_MEMORY_DEFAULT_MAX_CHARS,
54
- type PromptFormatInputs,
55
- } from "./prompt-format.js";
50
+ export { formatUserPrompt, type PromptFormatInputs } from "./prompt-format.js";
56
51
 
57
52
  export {
58
53
  buildDeliveryContract,
@@ -15,7 +15,6 @@
15
15
  * DM (no msg_id): "[2026-05-15 11:01:23] actual text"
16
16
  */
17
17
 
18
- import type { RetrievedMemory } from "../../core/agent-runtime/capabilities.js";
19
18
  import { formatFullDatetime } from "../../util/time.js";
20
19
 
21
20
  // ── Public API ──────────────────────────────────────────────────────────────
@@ -62,74 +61,6 @@ export function formatUserPrompt(inputs: PromptFormatInputs): string {
62
61
  return joinNonEmpty(timeTag, inputs.text);
63
62
  }
64
63
 
65
- // ── Retrieved-memory wrapper (Phase B pre-retrieval) ────────────────────────
66
-
67
- /** Default hard cap on the injected memory block, provenance labels included. */
68
- export const RETRIEVED_MEMORY_DEFAULT_MAX_CHARS = 3000;
69
-
70
- /**
71
- * Wrap an already-formatted live user prompt with a bounded retrieved-memory
72
- * block. This is the ONLY place retrieved memory enters a prompt, and it
73
- * wraps the whole `formatUserPrompt(...)` output rather than rebuilding its
74
- * internals — the existing sender/time/msg_id wrapper stays intact inside the
75
- * `User message:` section.
76
- *
77
- * Contract (see docs/memory-phase-b-pre-retrieval.md):
78
- * - `memory` undefined or empty items → the prompt is returned
79
- * BYTE-IDENTICAL. Prompt-cache and prompt-format tests stay valid.
80
- * - Non-empty → emit `Relevant memory:` with one provenance-labelled line
81
- * per item, a blank line, `User message:`, then the original prompt.
82
- * - The memory block (labels included) is capped at `maxChars`; item text
83
- * is truncated deterministically with an ellipsis marker. The user
84
- * message itself is NEVER dropped or truncated.
85
- * - This block is dynamic turn context: callers must keep it out of
86
- * `prepareSystemPrompt()`, prompt additions, and backend `system` fields.
87
- */
88
- export function formatPromptWithRetrievedMemory(
89
- prompt: string,
90
- memory?: RetrievedMemory,
91
- maxChars: number = RETRIEVED_MEMORY_DEFAULT_MAX_CHARS,
92
- ): string {
93
- if (!memory || memory.items.length === 0) return prompt;
94
-
95
- const header = "Relevant memory:";
96
- const footer = "User message:";
97
- // Budget applies to the memory block only (header + item lines), so the
98
- // user message can never be squeezed out.
99
- let budget = Math.max(0, maxChars) - header.length - 1; // "\n" after header
100
- const lines: string[] = [];
101
- for (const item of memory.items) {
102
- const label = provenanceLabel(item.wing, item.room, item.sourceFile);
103
- const prefix = `- ${label} `;
104
- if (prefix.length >= budget) break;
105
- const text = sanitizeInline(item.text);
106
- const room = budget - prefix.length - 1; // "\n" for this line
107
- const body =
108
- text.length <= room ? text : `${text.slice(0, Math.max(0, room - 1))}…`;
109
- if (body.length === 0) break;
110
- const line = `${prefix}${body}`;
111
- lines.push(line);
112
- budget -= line.length + 1;
113
- }
114
- if (lines.length === 0) return prompt;
115
-
116
- return `${header}\n${lines.join("\n")}\n\n${footer}\n${prompt}`;
117
- }
118
-
119
- function provenanceLabel(
120
- wing: string,
121
- room?: string,
122
- sourceFile?: string,
123
- ): string {
124
- const path = room ? `${wing}/${room}` : wing;
125
- return sourceFile ? `[${path} ${sourceFile}]` : `[${path}]`;
126
- }
127
-
128
- /** Collapse newlines/control whitespace so one item stays one labelled line. */
129
- function sanitizeInline(text: string): string {
130
- return text.replace(/\s+/g, " ").trim();
131
- }
132
-
133
64
  // ── Helpers ─────────────────────────────────────────────────────────────────
134
65
 
135
66
  function joinNonEmpty(...parts: string[]): string {
package/src/bootstrap.ts CHANGED
@@ -467,6 +467,7 @@ export async function initBackendAndDispatcher(
467
467
  initDream({
468
468
  model: config.model,
469
469
  dreamModel: config.dreamModel,
470
+ dreamEffort: config.dreamEffort,
470
471
  workspace: config.workspace,
471
472
  enabled: config.dream,
472
473
  getBackend: () => getBackendForRole("dream"),
@@ -482,6 +483,7 @@ export async function initBackendAndDispatcher(
482
483
  initHeartbeat({
483
484
  model: config.model,
484
485
  heartbeatModel: config.heartbeatModel,
486
+ heartbeatEffort: config.heartbeatEffort,
485
487
  workspace: config.workspace,
486
488
  getBackend: () => getBackendForRole("heartbeat"),
487
489
  frontends: frontendNames,