talon-agent 3.6.3 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -0
- package/package.json +2 -2
- package/prompts/README.md +10 -2
- package/prompts/dream.md +5 -3
- package/prompts/heartbeat.md +1 -1
- package/prompts/identity.md +1 -1
- package/prompts/mem0.md +7 -5
- package/prompts/mempalace.md +7 -5
- package/prompts/system/memory-recall.md +49 -0
- package/prompts/system/workspace.md +2 -2
- package/src/backend/claude-sdk/one-shot.ts +31 -1
- package/src/backend/codex/effort.ts +30 -0
- package/src/backend/codex/handler/message.ts +6 -9
- package/src/backend/codex/one-shot.ts +20 -0
- package/src/backend/remote-server/one-shot.ts +6 -0
- package/src/bootstrap.ts +2 -0
- package/src/core/background/dream.ts +27 -2
- package/src/core/background/effort.ts +80 -0
- package/src/core/background/heartbeat/agent.ts +22 -1
- package/src/core/background/heartbeat/state.ts +7 -0
- package/src/core/engine/model-audit.ts +69 -2
- package/src/core/models/reasoning-levels.ts +14 -3
- package/src/core/prompt/assemble.ts +8 -5
- package/src/core/prompt/embedded-prompts.ts +16 -14
- package/src/core/types.ts +10 -0
- package/src/util/config.ts +34 -0
package/README.md
CHANGED
|
@@ -385,6 +385,10 @@ Config file: `~/.talon/config.json`
|
|
|
385
385
|
| `pulse` | `true` | Periodic group engagement |
|
|
386
386
|
| `heartbeat` | `false` | Background maintenance agent |
|
|
387
387
|
| `heartbeatIntervalMinutes` | `60` | Heartbeat interval |
|
|
388
|
+
| `heartbeatModel` | --- | Model for the heartbeat agent (falls back to `model`) |
|
|
389
|
+
| `heartbeatEffort` | --- | Reasoning effort for the heartbeat agent: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Unset = the model's own default |
|
|
390
|
+
| `dreamModel` | --- | Model for dream / memory consolidation (falls back to `model`) |
|
|
391
|
+
| `dreamEffort` | --- | Reasoning effort for the dream agent — same levels as `heartbeatEffort` |
|
|
388
392
|
| `braveApiKey` | --- | Brave Search API key |
|
|
389
393
|
| `timezone` | --- | IANA timezone (e.g. `"Europe/London"`) |
|
|
390
394
|
| `plugins` | `[]` | External plugin packages |
|
|
@@ -398,6 +402,18 @@ Config file: `~/.talon/config.json`
|
|
|
398
402
|
| `mempalace` | --- | Legacy MemPalace plugin config (prefer `memory`) |
|
|
399
403
|
| `playwright` | --- | Playwright plugin config (see above) |
|
|
400
404
|
|
|
405
|
+
### Background reasoning effort
|
|
406
|
+
|
|
407
|
+
`heartbeatEffort` / `dreamEffort` set how hard the background agents think —
|
|
408
|
+
useful when you want unattended goal work to reason harder than a chat turn,
|
|
409
|
+
or hourly heartbeats to stay cheap. Chat effort stays per-chat (`/settings`).
|
|
410
|
+
|
|
411
|
+
Which levels a model accepts comes from its catalog entry, so the usable set
|
|
412
|
+
differs per model (`max` is Claude's ceiling, `xhigh` is Codex's). A level the
|
|
413
|
+
model doesn't offer is dropped — the run proceeds on the model default, the
|
|
414
|
+
reason is written to the run log, and the boot-time model audit warns about it.
|
|
415
|
+
Backends with no reasoning knob at all (Kilo, OpenCode) ignore the setting.
|
|
416
|
+
|
|
401
417
|
---
|
|
402
418
|
|
|
403
419
|
## Terminal Mode
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "talon-agent",
|
|
3
|
-
"version": "3.
|
|
3
|
+
"version": "3.8.0",
|
|
4
4
|
"description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
|
|
5
5
|
"author": "Dylan Neve",
|
|
6
6
|
"license": "MIT",
|
|
@@ -105,7 +105,7 @@
|
|
|
105
105
|
"@openai/agents": "^0.13.0",
|
|
106
106
|
"@openai/codex-sdk": "^0.145.0",
|
|
107
107
|
"@opencode-ai/sdk": "^1.17.4",
|
|
108
|
-
"@playwright/mcp": "0.0.
|
|
108
|
+
"@playwright/mcp": "0.0.78",
|
|
109
109
|
"@types/cross-spawn": "^6.0.6",
|
|
110
110
|
"big-integer": "^1.6.52",
|
|
111
111
|
"cheerio": "^1.2.0",
|
package/prompts/README.md
CHANGED
|
@@ -16,7 +16,7 @@ placed after the cache boundary):
|
|
|
16
16
|
| 2 | Core behaviour | `custom.md` (replaces `base.md` when present) | static |
|
|
17
17
|
| 3 | Frontend capabilities | `<frontend>.md` (telegram / discord / teams / terminal / native) | static |
|
|
18
18
|
| 4 | Persistent memory | `system/persistent-memory.md` wrapping `memory/memory.md`, size-capped | static |
|
|
19
|
-
| 5 |
|
|
19
|
+
| 5 | Memory + capabilities | `system/memory-recall.md`, workspace, cron, triggers, goals, skills | static |
|
|
20
20
|
| 6 | Plugin additions | each plugin's `systemPrompt()` contribution | static |
|
|
21
21
|
| 7 | **Delivery contract** | `system/contract-*.md`, appended by the **backend** as its suffix | static (tail) |
|
|
22
22
|
| 8 | Daily-memory pointer | `system/daily-memory.md` (names today's file) | dynamic |
|
|
@@ -42,7 +42,7 @@ when a backend has no native "skills" feature.
|
|
|
42
42
|
**User-editable prompts** (everything at the top level of this directory:
|
|
43
43
|
`identity.md`, `base.md`, `custom.md`, `telegram.md`, `discord.md`,
|
|
44
44
|
`teams.md`, `terminal.md`, `native.md`, `heartbeat.md`, `dream.md`,
|
|
45
|
-
`mempalace.md`) are
|
|
45
|
+
`mempalace.md`, `mem0.md`) are
|
|
46
46
|
seeded into `~/.talon/prompts/` on first run and read from there.
|
|
47
47
|
Seeding is upgrade-aware (dpkg-conffile semantics, tracked via a
|
|
48
48
|
`.seeded.json` hash manifest next to the seeded files): a file the user
|
|
@@ -70,6 +70,14 @@ User-editable prompts are NOT Liquid: their consumers substitute
|
|
|
70
70
|
`{{var}}` placeholders with plain string replacement, so a user edit
|
|
71
71
|
can never break prompt assembly with a template syntax error.
|
|
72
72
|
|
|
73
|
+
`system/memory-recall.md` is the provider-neutral continuity policy. It guides
|
|
74
|
+
agents to make a proportionate, thorough attempt across relevant memory,
|
|
75
|
+
workspace, log, and connected sources before asking a user to repeat
|
|
76
|
+
information, and to persist new information. When a memory plugin is enabled,
|
|
77
|
+
its prompt addition follows this section and names the preferred provider and
|
|
78
|
+
exact tools; without one, the policy falls back to `memory/memory.md` plus
|
|
79
|
+
daily notes.
|
|
80
|
+
|
|
73
81
|
## The delivery contract (response flow)
|
|
74
82
|
|
|
75
83
|
How a reply reaches the user is a property of the **backend**, not the
|
package/prompts/dream.md
CHANGED
|
@@ -23,7 +23,8 @@ You primarily use filesystem tools (Read, Write, Edit, Bash, Glob, Grep). Do NOT
|
|
|
23
23
|
- Corrections to previously held beliefs
|
|
24
24
|
- Operational patterns (e.g. who stays up late, who prefers what tools)
|
|
25
25
|
- Project context changes inferred from the conversation (e.g. new repos, shifted priorities)
|
|
26
|
-
-
|
|
26
|
+
- Capture every genuinely new or updated piece of information; avoid
|
|
27
|
+
duplicating facts already represented in memory
|
|
27
28
|
|
|
28
29
|
### Stage 3 — Consolidate
|
|
29
30
|
|
|
@@ -38,8 +39,9 @@ You primarily use filesystem tools (Read, Write, Edit, Bash, Glob, Grep). Do NOT
|
|
|
38
39
|
### Stage 4 — Prune
|
|
39
40
|
|
|
40
41
|
- Remove entries that have been explicitly contradicted
|
|
41
|
-
- Remove entries that are clearly
|
|
42
|
-
- Do NOT remove entries just because they're old
|
|
42
|
+
- Remove entries that are clearly superseded
|
|
43
|
+
- Do NOT remove entries just because they're old or seem unimportant — only
|
|
44
|
+
remove information that is wrong or replaced by a newer version
|
|
43
45
|
- Write the updated memory.md back to `{{memoryFile}}`
|
|
44
46
|
|
|
45
47
|
### Stage 5 — Mine to MemPalace & Write Diary (optional)
|
package/prompts/heartbeat.md
CHANGED
|
@@ -32,7 +32,7 @@ For each goal:
|
|
|
32
32
|
3. Record every advance with `update_goal(goal_id=..., progress_note=..., chat_id=<the goal's chat>)`. The `chat_id` parameter is REQUIRED in heartbeat mode — use the chat id shown next to the goal. Keep notes short and concrete: what was done, what was learned, what's blocked.
|
|
33
33
|
4. When a goal's objective is achieved, set `status="completed"` and send a short, high-signal message to the goal's chat (explicit `chat_id` required). If a goal has become impossible or moot, set `status="abandoned"` with a note explaining why.
|
|
34
34
|
5. If nothing can be done on a goal right now, skip it silently — do not write filler progress notes.
|
|
35
|
-
6. If MemPalace tools are available: `mempalace_search` for context relevant to a goal before working on it, and store
|
|
35
|
+
6. If MemPalace tools are available: `mempalace_search` for context relevant to a goal before working on it, and store new information learned while working (`mempalace_add_drawer` / `mempalace_kg_add`).
|
|
36
36
|
|
|
37
37
|
## Instructions
|
|
38
38
|
|
package/prompts/identity.md
CHANGED
|
@@ -33,4 +33,4 @@ When a filesystem-capable tool is available, persist the answers to `~/.talon/wo
|
|
|
33
33
|
|
|
34
34
|
## Memory
|
|
35
35
|
|
|
36
|
-
When you learn
|
|
36
|
+
When you learn new information — who people are, how they like to work, what they're building, decisions, facts, and surrounding context — follow the Memory and Recall policy in this prompt. Use the configured long-term-memory provider when one is available; otherwise use the workspace memory files.
|
package/prompts/mem0.md
CHANGED
|
@@ -2,12 +2,14 @@
|
|
|
2
2
|
|
|
3
3
|
You have access to mem0 long-term memory via MCP tools. mem0 extracts durable facts from what you store and retrieves them by semantic search. All memories are filed under the entity id `{{userId}}`.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
mem0 is the preferred durable-memory store while its tools are available. Workspace daily notes can still hold concise chronological context; `memory.md` remains a fallback if the mem0 tools are unavailable.
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
7
|
+
### How to use it well
|
|
8
|
+
|
|
9
|
+
1. Search with `mem0_search_memory` when prior context about a person, project, or past event could materially improve the answer.
|
|
10
|
+
2. If a fact such as a name, relationship, or preference is uncertain, checking memory is usually better than guessing or asking the user to repeat it.
|
|
11
|
+
3. When a fact changes, store the new version with `mem0_add_memory` (mem0 supersedes contradicted memories itself); use `mem0_delete_memory` for plainly wrong entries.
|
|
12
|
+
4. When you learn new information, pass natural conversational text to `mem0_add_memory` so mem0 can extract and retain the facts.
|
|
11
13
|
|
|
12
14
|
### Tools
|
|
13
15
|
|
package/prompts/mempalace.md
CHANGED
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
You have access to a local memory palace via MCP tools. The palace stores verbatim conversation history and a temporal knowledge graph — all local, zero cloud, zero API calls.
|
|
4
4
|
|
|
5
|
+
MemPalace is the preferred durable-memory store while its tools are available. Workspace daily notes can still hold concise chronological context; `memory.md` remains a fallback if the MemPalace tools are unavailable.
|
|
6
|
+
|
|
5
7
|
### Architecture
|
|
6
8
|
|
|
7
9
|
- **Wings** = top-level categories (people, projects, topics)
|
|
@@ -10,12 +12,12 @@ You have access to a local memory palace via MCP tools. The palace stores verbat
|
|
|
10
12
|
- **Tunnels** = cross-wing links between related rooms (auto-created in mempalace 3.3.4+ when topics overlap, plus manual)
|
|
11
13
|
- **Knowledge Graph** = entity-relationship facts with temporal validity
|
|
12
14
|
|
|
13
|
-
###
|
|
15
|
+
### How to use it well
|
|
14
16
|
|
|
15
|
-
1.
|
|
16
|
-
2.
|
|
17
|
-
3.
|
|
18
|
-
4.
|
|
17
|
+
1. Search with `mempalace_search` or `mempalace_kg_query` when prior context about a person, project, or past event could materially improve the answer.
|
|
18
|
+
2. If a fact such as a name, relationship, or preference is uncertain, checking the palace is usually better than guessing or asking the user to repeat it.
|
|
19
|
+
3. When a fact changes, keep its history accurate with `mempalace_kg_invalidate` followed by `mempalace_kg_add`.
|
|
20
|
+
4. When you learn new information, use `mempalace_add_drawer` for rich context or `mempalace_kg_add` for a structured fact.
|
|
19
21
|
|
|
20
22
|
### Tools
|
|
21
23
|
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
## Memory and Recall
|
|
2
|
+
|
|
3
|
+
### Recall before asking
|
|
4
|
+
|
|
5
|
+
Protect continuity. If a request relies on information the user reasonably
|
|
6
|
+
expects you to already have, or you are unsure about a prior fact, take that
|
|
7
|
+
as a cue to recover the context before asking them to repeat it. Make a
|
|
8
|
+
proportionate but thorough attempt across the relevant sources available to
|
|
9
|
+
you:
|
|
10
|
+
|
|
11
|
+
- the current conversation and memory already included in this prompt;
|
|
12
|
+
- enabled long-term-memory providers, including browse/fetch tools when a
|
|
13
|
+
search hit needs more context;
|
|
14
|
+
- `memory/memory.md`, including the rest when its prompt excerpt is
|
|
15
|
+
truncated, and relevant recent files in `memory/daily/`;
|
|
16
|
+
- workspace files, interaction logs, and connected sources that the request
|
|
17
|
+
suggests may contain the answer.
|
|
18
|
+
|
|
19
|
+
Start with the most likely source, then broaden rather than stopping after one
|
|
20
|
+
empty result. Alternate names, keywords, dates, or scopes can recover memories
|
|
21
|
+
that a first query misses; follow promising results to their full source and
|
|
22
|
+
weigh conflicts by recency and authority. Keep the effort relevant to the
|
|
23
|
+
request—ordinary questions about the current turn do not call for rummaging
|
|
24
|
+
through unrelated history.
|
|
25
|
+
|
|
26
|
+
If a meaningful search still leaves the answer missing, inaccessible, or
|
|
27
|
+
genuinely ambiguous, ask for the smallest piece of information needed and
|
|
28
|
+
briefly explain the gap.
|
|
29
|
+
|
|
30
|
+
### Save new information
|
|
31
|
+
|
|
32
|
+
Proactively persist new information so future conversations can draw on it.
|
|
33
|
+
This includes preferences, relationships, decisions, corrections, project
|
|
34
|
+
context, durable facts, and details that may only become relevant later. Do
|
|
35
|
+
this naturally as you learn it, without interrupting the conversation. Keep
|
|
36
|
+
memory organized, update stale facts, and avoid duplicate copies.
|
|
37
|
+
|
|
38
|
+
- If a dedicated long-term-memory provider is described elsewhere in this
|
|
39
|
+
prompt and its tools are available, prefer it as the canonical durable store
|
|
40
|
+
and use its guidance for searching, adding, updating, and deduplicating
|
|
41
|
+
memories. Daily notes can still preserve useful chronological context.
|
|
42
|
+
- Otherwise, when filesystem tools are available, keep durable knowledge
|
|
43
|
+
organized and current in `~/.talon/workspace/memory/memory.md`, and use
|
|
44
|
+
today's `memory/daily/YYYY-MM-DD.md` for concise dated observations,
|
|
45
|
+
corrections, and follow-ups.
|
|
46
|
+
- If neither a memory provider nor filesystem tools are available, retain the
|
|
47
|
+
information only for the current conversation and never claim it was saved.
|
|
48
|
+
|
|
49
|
+
Memory updates should usually be quiet unless the user asks about them.
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
You have a workspace directory at `~/.talon/workspace/`. This is your home — organize it however you want.
|
|
4
4
|
|
|
5
|
-
- `memory/memory.md` — your persistent
|
|
6
|
-
- `memory/daily/YYYY-MM-DD.md` —
|
|
5
|
+
- `memory/memory.md` — your file-based persistent-memory fallback when no dedicated memory provider is available.
|
|
6
|
+
- `memory/daily/YYYY-MM-DD.md` — concise chronological notes: observations, learnings, corrections, and follow-ups.
|
|
7
7
|
- `logs/` — daily interaction logs, written automatically.
|
|
8
8
|
- `uploads/` — files users send you (photos, docs, voice) land here.
|
|
9
9
|
- Everything else is yours to create and organize as you see fit.
|
|
@@ -15,8 +15,9 @@ import { readdir, readFile } from "node:fs/promises";
|
|
|
15
15
|
import { query } from "@anthropic-ai/claude-agent-sdk";
|
|
16
16
|
import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk";
|
|
17
17
|
import type { OneShotAgentParams, OneShotUsage } from "../../core/types.js";
|
|
18
|
-
import { log } from "../../util/log.js";
|
|
18
|
+
import { log, logWarn } from "../../util/log.js";
|
|
19
19
|
import { ALLOWED_TOOLS_BACKGROUND } from "../../core/constants.js";
|
|
20
|
+
import { EFFORT_MAP } from "./constants.js";
|
|
20
21
|
import { buildMcpServers, buildPluginMcpServers } from "./options.js";
|
|
21
22
|
|
|
22
23
|
const DEFAULT_SUBPROCESS_KILL_GRACE_MS = 5 * 1000;
|
|
@@ -59,14 +60,26 @@ export async function runOneShotAgent(
|
|
|
59
60
|
systemPrompt,
|
|
60
61
|
workspace,
|
|
61
62
|
model,
|
|
63
|
+
reasoningEffort,
|
|
62
64
|
contextLabel,
|
|
63
65
|
abortController,
|
|
64
66
|
appendLog,
|
|
65
67
|
} = params;
|
|
66
68
|
|
|
69
|
+
// Reasoning effort is opt-in for background runs (config `heartbeatEffort`
|
|
70
|
+
// / `dreamEffort`). Unset → omit the thinking options entirely so the SDK
|
|
71
|
+
// keeps whatever default the model ships with, which is what these runs
|
|
72
|
+
// did before the knob existed. The chat path applies an explicit
|
|
73
|
+
// `{ thinking: { type: "adaptive" } }` fallback instead because a chat has
|
|
74
|
+
// a persisted per-chat setting to honour; a one-shot has none.
|
|
75
|
+
const thinkingConfig = reasoningEffort
|
|
76
|
+
? EFFORT_MAP[reasoningEffort]
|
|
77
|
+
: undefined;
|
|
78
|
+
|
|
67
79
|
const options = {
|
|
68
80
|
model,
|
|
69
81
|
systemPrompt,
|
|
82
|
+
...thinkingConfig,
|
|
70
83
|
cwd: workspace,
|
|
71
84
|
permissionMode: "bypassPermissions" as const,
|
|
72
85
|
allowDangerouslySkipPermissions: true,
|
|
@@ -81,6 +94,23 @@ export async function runOneShotAgent(
|
|
|
81
94
|
tools: [...ALLOWED_TOOLS_BACKGROUND],
|
|
82
95
|
};
|
|
83
96
|
|
|
97
|
+
if (reasoningEffort && !thinkingConfig) {
|
|
98
|
+
// `minimal` / `xhigh` are Codex-side vocabulary with no Claude
|
|
99
|
+
// equivalent in EFFORT_MAP — the run proceeds on the model default
|
|
100
|
+
// rather than failing, but say so in the log so a configured knob that
|
|
101
|
+
// does nothing isn't silent.
|
|
102
|
+
logWarn(
|
|
103
|
+
"agent",
|
|
104
|
+
`[${contextLabel}] Claude one-shot: effort "${reasoningEffort}" has no ` +
|
|
105
|
+
`Claude mapping — using the model default`,
|
|
106
|
+
);
|
|
107
|
+
} else if (thinkingConfig) {
|
|
108
|
+
log(
|
|
109
|
+
"agent",
|
|
110
|
+
`[${contextLabel}] Claude one-shot effort: ${reasoningEffort}`,
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
|
|
84
114
|
const qi = query({
|
|
85
115
|
prompt,
|
|
86
116
|
options: options as Parameters<typeof query>[0]["options"],
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Codex reasoning-effort vocabulary mapping.
|
|
3
|
+
*
|
|
4
|
+
* Talon's canonical `ReasoningEffortLevel` is a superset of what Codex's
|
|
5
|
+
* `modelReasoningEffort` thread option accepts: `off` isn't expressible on a
|
|
6
|
+
* reasoning model, and `max` is Claude-only. Both simply fall through to the
|
|
7
|
+
* model's own default.
|
|
8
|
+
*
|
|
9
|
+
* This is pure vocabulary translation — the "does this model offer that
|
|
10
|
+
* level?" question is answered by the caller (per-chat settings for an
|
|
11
|
+
* interactive turn, `core/background/effort.ts` for heartbeat/dream) against
|
|
12
|
+
* the model catalog's `supportedReasoningLevels`. Keeping the mapping here
|
|
13
|
+
* means the chat path and the one-shot path can't drift apart.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import type { ReasoningEffortLevel } from "../../core/types.js";
|
|
17
|
+
|
|
18
|
+
/** The levels Codex's `modelReasoningEffort` thread option accepts. */
|
|
19
|
+
export type CodexReasoningEffort = Exclude<ReasoningEffortLevel, "off" | "max">;
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Map a canonical level onto Codex's thread option, or undefined when Codex
|
|
23
|
+
* has no way to express it (`off`, `max`, or nothing requested).
|
|
24
|
+
*/
|
|
25
|
+
export function toCodexReasoningEffort(
|
|
26
|
+
level: ReasoningEffortLevel | undefined,
|
|
27
|
+
): CodexReasoningEffort | undefined {
|
|
28
|
+
if (!level || level === "off" || level === "max") return undefined;
|
|
29
|
+
return level;
|
|
30
|
+
}
|
|
@@ -66,6 +66,7 @@ import {
|
|
|
66
66
|
isCodexOAuthIncompat,
|
|
67
67
|
} from "../models.js";
|
|
68
68
|
import { supportsReasoningLevel } from "../../../core/models/reasoning-levels.js";
|
|
69
|
+
import { toCodexReasoningEffort } from "../effort.js";
|
|
69
70
|
import { markOAuthIncompat } from "../oauth-incompat.js";
|
|
70
71
|
import { readLastRolloutSnapshot } from "../token-usage.js";
|
|
71
72
|
import { activeAborts } from "./state.js";
|
|
@@ -283,22 +284,18 @@ export async function handleMessage(
|
|
|
283
284
|
const supportedReasoningLevels =
|
|
284
285
|
activeModelInfo?.supportedReasoningLevels ?? [];
|
|
285
286
|
const requestedEffort = chatSettings.effort;
|
|
287
|
+
// Availability check (does this model offer the level?) then vocabulary
|
|
288
|
+
// translation (can Codex express it?) — the latter is shared with the
|
|
289
|
+
// one-shot path via `toCodexReasoningEffort` so the two can't drift.
|
|
286
290
|
const modelReasoningEffort =
|
|
287
291
|
requestedEffort &&
|
|
288
|
-
requestedEffort !== "off" &&
|
|
289
|
-
requestedEffort !== "max" &&
|
|
290
292
|
supportsReasoningLevel(requestedEffort, supportedReasoningLevels)
|
|
291
|
-
? requestedEffort
|
|
293
|
+
? toCodexReasoningEffort(requestedEffort)
|
|
292
294
|
: undefined;
|
|
293
295
|
const threadOptions = {
|
|
294
296
|
model: activeModel,
|
|
295
297
|
skipGitRepoCheck: true,
|
|
296
|
-
...(modelReasoningEffort
|
|
297
|
-
? {
|
|
298
|
-
modelReasoningEffort: modelReasoningEffort as
|
|
299
|
-
"minimal" | "low" | "medium" | "high" | "xhigh",
|
|
300
|
-
}
|
|
301
|
-
: {}),
|
|
298
|
+
...(modelReasoningEffort ? { modelReasoningEffort } : {}),
|
|
302
299
|
...CODEX_THREAD_PERMISSIONS,
|
|
303
300
|
};
|
|
304
301
|
const thread: Thread = session.sessionId
|
|
@@ -29,6 +29,7 @@ import {
|
|
|
29
29
|
import { isChatGptModelMismatchError } from "./auth.js";
|
|
30
30
|
import { chatGptFallbackFor, isCodexOAuthIncompat } from "./models.js";
|
|
31
31
|
import { markOAuthIncompat } from "./oauth-incompat.js";
|
|
32
|
+
import { toCodexReasoningEffort } from "./effort.js";
|
|
32
33
|
|
|
33
34
|
/**
|
|
34
35
|
* Resolve the effective model for a one-shot run, applying the same
|
|
@@ -74,6 +75,7 @@ export async function runOneShotAgent(
|
|
|
74
75
|
prompt,
|
|
75
76
|
systemPrompt,
|
|
76
77
|
model: requestedModel,
|
|
78
|
+
reasoningEffort,
|
|
77
79
|
contextLabel,
|
|
78
80
|
abortController,
|
|
79
81
|
appendLog,
|
|
@@ -103,9 +105,27 @@ export async function runOneShotAgent(
|
|
|
103
105
|
}
|
|
104
106
|
log("agent", `[${contextLabel}] Codex one-shot model: ${activeModel}`);
|
|
105
107
|
|
|
108
|
+
// Availability was already checked by the caller against the model
|
|
109
|
+
// catalog (core/background/effort.ts); all that's left is Codex's own
|
|
110
|
+
// vocabulary, which can't express `off` / `max`.
|
|
111
|
+
const modelReasoningEffort = toCodexReasoningEffort(reasoningEffort);
|
|
112
|
+
if (reasoningEffort && !modelReasoningEffort) {
|
|
113
|
+
logWarn(
|
|
114
|
+
"agent",
|
|
115
|
+
`[${contextLabel}] Codex one-shot: effort "${reasoningEffort}" has no ` +
|
|
116
|
+
`Codex equivalent — using the model default`,
|
|
117
|
+
);
|
|
118
|
+
} else if (modelReasoningEffort) {
|
|
119
|
+
log(
|
|
120
|
+
"agent",
|
|
121
|
+
`[${contextLabel}] Codex one-shot effort: ${modelReasoningEffort}`,
|
|
122
|
+
);
|
|
123
|
+
}
|
|
124
|
+
|
|
106
125
|
const thread = codex.startThread({
|
|
107
126
|
model: activeModel,
|
|
108
127
|
skipGitRepoCheck: true,
|
|
128
|
+
...(modelReasoningEffort ? { modelReasoningEffort } : {}),
|
|
109
129
|
...CODEX_THREAD_PERMISSIONS,
|
|
110
130
|
});
|
|
111
131
|
|
|
@@ -19,6 +19,12 @@
|
|
|
19
19
|
* subprocesses to evict — `evictOrphanSubprocesses` is intentionally
|
|
20
20
|
* not implemented for this family.
|
|
21
21
|
*
|
|
22
|
+
* Note on reasoning effort: `params.reasoningEffort` (config
|
|
23
|
+
* `heartbeatEffort` / `dreamEffort`) is deliberately unused here. Neither
|
|
24
|
+
* server's `session.prompt` exposes a reasoning knob — the level is baked
|
|
25
|
+
* into the provider's model id when it's selectable at all — so the field
|
|
26
|
+
* is ignored rather than half-honoured.
|
|
27
|
+
*
|
|
22
28
|
* Note on abort semantics: both SDKs expose `session.abort` but the
|
|
23
29
|
* REST `prompt` endpoint blocks until the underlying provider
|
|
24
30
|
* returns. The heartbeat module's outer abort grace is what actually
|
package/src/bootstrap.ts
CHANGED
|
@@ -467,6 +467,7 @@ export async function initBackendAndDispatcher(
|
|
|
467
467
|
initDream({
|
|
468
468
|
model: config.model,
|
|
469
469
|
dreamModel: config.dreamModel,
|
|
470
|
+
dreamEffort: config.dreamEffort,
|
|
470
471
|
workspace: config.workspace,
|
|
471
472
|
enabled: config.dream,
|
|
472
473
|
getBackend: () => getBackendForRole("dream"),
|
|
@@ -482,6 +483,7 @@ export async function initBackendAndDispatcher(
|
|
|
482
483
|
initHeartbeat({
|
|
483
484
|
model: config.model,
|
|
484
485
|
heartbeatModel: config.heartbeatModel,
|
|
486
|
+
heartbeatEffort: config.heartbeatEffort,
|
|
485
487
|
workspace: config.workspace,
|
|
486
488
|
getBackend: () => getBackendForRole("heartbeat"),
|
|
487
489
|
frontends: frontendNames,
|
|
@@ -19,10 +19,11 @@ import { importLegacyJson } from "../../storage/legacy-import.js";
|
|
|
19
19
|
import { readPromptAsset } from "#prompt-assets";
|
|
20
20
|
import { log, logError, logWarn } from "../../util/log.js";
|
|
21
21
|
import { getDefaultModel } from "../models/catalog.js";
|
|
22
|
-
import type { OneShotAgentParams } from "../types.js";
|
|
22
|
+
import type { OneShotAgentParams, ReasoningEffortLevel } from "../types.js";
|
|
23
23
|
import type { Backend } from "../agent-runtime/capabilities.js";
|
|
24
24
|
import { getSoul } from "../soul/service.js";
|
|
25
25
|
import { taskTable } from "../tasks/index.js";
|
|
26
|
+
import { resolveBackgroundEffort } from "./effort.js";
|
|
26
27
|
import { FailureBackoff } from "./failure-backoff.js";
|
|
27
28
|
|
|
28
29
|
// ── Types ────────────────────────────────────────────────────────────────────
|
|
@@ -65,6 +66,8 @@ export const dreamFailureBackoff = new FailureBackoff();
|
|
|
65
66
|
let configRef: {
|
|
66
67
|
model?: string;
|
|
67
68
|
dreamModel?: string;
|
|
69
|
+
/** Reasoning effort for dream runs. Undefined = backend/model default. */
|
|
70
|
+
dreamEffort?: ReasoningEffortLevel;
|
|
68
71
|
workspace?: string;
|
|
69
72
|
/** When false, `maybeStartDream` never fires (config `dream: false`). */
|
|
70
73
|
enabled?: boolean;
|
|
@@ -86,6 +89,12 @@ export function initDream(cfg: {
|
|
|
86
89
|
model?: string;
|
|
87
90
|
/** Override model for dream consolidation (e.g. a cheaper model). Falls back to main model. */
|
|
88
91
|
dreamModel?: string;
|
|
92
|
+
/**
|
|
93
|
+
* Reasoning effort for dream consolidation (config `dreamEffort`). Unset
|
|
94
|
+
* leaves the backend/model default in place. Ignored by backends with no
|
|
95
|
+
* reasoning knob (Kilo, OpenCode).
|
|
96
|
+
*/
|
|
97
|
+
dreamEffort?: ReasoningEffortLevel;
|
|
89
98
|
workspace?: string;
|
|
90
99
|
/** Gate for automatic dream runs — config `dream` flag. Defaults to enabled. */
|
|
91
100
|
enabled?: boolean;
|
|
@@ -242,13 +251,28 @@ If commands fail, log the error and continue — this stage is optional.`
|
|
|
242
251
|
);
|
|
243
252
|
}
|
|
244
253
|
|
|
254
|
+
// Resolved against the dream backend's catalog — an effort level the model
|
|
255
|
+
// doesn't offer is dropped with a reason instead of reaching the SDK.
|
|
256
|
+
const effort = await resolveBackgroundEffort({
|
|
257
|
+
requested: configRef.dreamEffort,
|
|
258
|
+
model,
|
|
259
|
+
backend,
|
|
260
|
+
});
|
|
261
|
+
if (effort.dropped) {
|
|
262
|
+
logWarn("dream", effort.dropped);
|
|
263
|
+
}
|
|
264
|
+
|
|
245
265
|
// Set up dream log file
|
|
246
266
|
const dreamLogFile = createDreamLogFile();
|
|
247
267
|
appendDreamLog(dreamLogFile, `# Dream Run — ${new Date().toISOString()}\n`);
|
|
248
268
|
appendDreamLog(
|
|
249
269
|
dreamLogFile,
|
|
250
|
-
`**Trigger:** last_run=${lastRunIso}, model=${model}
|
|
270
|
+
`**Trigger:** last_run=${lastRunIso}, model=${model}` +
|
|
271
|
+
`${effort.effort ? `, effort=${effort.effort}` : ""}\n`,
|
|
251
272
|
);
|
|
273
|
+
if (effort.dropped) {
|
|
274
|
+
appendDreamLog(dreamLogFile, `**Effort:** ${effort.dropped}\n`);
|
|
275
|
+
}
|
|
252
276
|
appendDreamLog(
|
|
253
277
|
dreamLogFile,
|
|
254
278
|
`**Prompt:**\n\`\`\`\n${prompt}\n\`\`\`\n\n---\n`,
|
|
@@ -272,6 +296,7 @@ If commands fail, log the error and continue — this stage is optional.`
|
|
|
272
296
|
systemPrompt,
|
|
273
297
|
workspace,
|
|
274
298
|
model,
|
|
299
|
+
...(effort.effort ? { reasoningEffort: effort.effort } : {}),
|
|
275
300
|
contextLabel: "dream",
|
|
276
301
|
abortController,
|
|
277
302
|
// appendDreamLog is sync (writeFileSync) — wrap to satisfy the async
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reasoning-effort resolution for background runs (heartbeat, dream).
|
|
3
|
+
*
|
|
4
|
+
* Both background agents take an effort level from config
|
|
5
|
+
* (`heartbeatEffort` / `dreamEffort`). Two separate questions hide behind
|
|
6
|
+
* "apply that level", and they belong to different layers:
|
|
7
|
+
*
|
|
8
|
+
* 1. *Is the level available on this model?* — a model-capability
|
|
9
|
+
* question. Every backend catalog already answers it through the
|
|
10
|
+
* `models.getRawModelInfo` capability (`supportedReasoningLevels`),
|
|
11
|
+
* which is the same channel the frontends' effort pickers read. So it
|
|
12
|
+
* is answered here, once, against the `Backend` abstraction.
|
|
13
|
+
*
|
|
14
|
+
* 2. *How is the level expressed to the provider?* — Claude thinking
|
|
15
|
+
* options, Codex `modelReasoningEffort`, nothing at all for
|
|
16
|
+
* Kilo/OpenCode. That is adapter work and stays inside each backend's
|
|
17
|
+
* one-shot runner.
|
|
18
|
+
*
|
|
19
|
+
* Keeping (1) here is what stops four one-shot runners from each growing
|
|
20
|
+
* their own copy of the same catalog lookup.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import type { Backend } from "../agent-runtime/capabilities.js";
|
|
24
|
+
import type { ReasoningEffortLevel } from "../types.js";
|
|
25
|
+
import {
|
|
26
|
+
normalizeReasoningLevels,
|
|
27
|
+
supportsReasoningLevel,
|
|
28
|
+
} from "../models/reasoning-levels.js";
|
|
29
|
+
|
|
30
|
+
export type BackgroundEffortResolution = {
|
|
31
|
+
/** The level to pass to the backend, or undefined to use its default. */
|
|
32
|
+
effort?: ReasoningEffortLevel;
|
|
33
|
+
/**
|
|
34
|
+
* Set when a configured level was discarded — a log-ready explanation of
|
|
35
|
+
* what was asked for and why the run proceeds without it. A dropped level
|
|
36
|
+
* never fails the run: a stale effort setting shouldn't cost an operator
|
|
37
|
+
* their hourly heartbeat.
|
|
38
|
+
*/
|
|
39
|
+
dropped?: string;
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Resolve the effort level a background run should ask for.
|
|
44
|
+
*
|
|
45
|
+
* Unset config → `{}` (backend/model default, i.e. what background runs did
|
|
46
|
+
* before the knob existed). A backend with no catalog capability, an
|
|
47
|
+
* unreachable catalog, or a model that reports no level metadata all pass
|
|
48
|
+
* the level through untouched: absent metadata is not evidence that the
|
|
49
|
+
* level is unsupported, and the adapter still has the final say.
|
|
50
|
+
*/
|
|
51
|
+
export async function resolveBackgroundEffort(params: {
|
|
52
|
+
requested: ReasoningEffortLevel | undefined;
|
|
53
|
+
model: string;
|
|
54
|
+
backend: Backend | null;
|
|
55
|
+
}): Promise<BackgroundEffortResolution> {
|
|
56
|
+
const { requested, model, backend } = params;
|
|
57
|
+
if (!requested) return {};
|
|
58
|
+
|
|
59
|
+
const catalog = backend?.models;
|
|
60
|
+
if (!catalog) return { effort: requested };
|
|
61
|
+
|
|
62
|
+
let info;
|
|
63
|
+
try {
|
|
64
|
+
info = await catalog.getRawModelInfo(model);
|
|
65
|
+
} catch {
|
|
66
|
+
return { effort: requested }; // catalog unavailable — don't second-guess
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const levels = normalizeReasoningLevels(info?.supportedReasoningLevels);
|
|
70
|
+
if (levels.length === 0) return { effort: requested };
|
|
71
|
+
|
|
72
|
+
if (!supportsReasoningLevel(requested, levels)) {
|
|
73
|
+
return {
|
|
74
|
+
dropped:
|
|
75
|
+
`configured effort "${requested}" is not available on "${model}" ` +
|
|
76
|
+
`(supports: ${levels.join(", ")}) — running the model default`,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
return { effort: requested };
|
|
80
|
+
}
|
|
@@ -14,6 +14,7 @@ import { loadSystemTemplate } from "../../prompt/templates.js";
|
|
|
14
14
|
import { formatGoal, getOpenGoals } from "../../../storage/goal-store.js";
|
|
15
15
|
import { taskTable } from "../../tasks/index.js";
|
|
16
16
|
import type { OneShotAgentParams } from "../../types.js";
|
|
17
|
+
import { resolveBackgroundEffort } from "../effort.js";
|
|
17
18
|
import { hb } from "./state.js";
|
|
18
19
|
|
|
19
20
|
const DEFAULT_HEARTBEAT_TIMEOUT_MS = 10 * 60 * 1000; // 10-minute soft cap
|
|
@@ -164,6 +165,18 @@ export async function runHeartbeatAgent(
|
|
|
164
165
|
);
|
|
165
166
|
}
|
|
166
167
|
|
|
168
|
+
// Effort is resolved against the heartbeat backend's catalog, not just
|
|
169
|
+
// copied from config — a level the model doesn't offer is dropped with a
|
|
170
|
+
// reason rather than handed to the SDK.
|
|
171
|
+
const effort = await resolveBackgroundEffort({
|
|
172
|
+
requested: config.heartbeatEffort,
|
|
173
|
+
model,
|
|
174
|
+
backend,
|
|
175
|
+
});
|
|
176
|
+
if (effort.dropped) {
|
|
177
|
+
logWarn("heartbeat", effort.dropped);
|
|
178
|
+
}
|
|
179
|
+
|
|
167
180
|
// Set up heartbeat log file
|
|
168
181
|
const heartbeatLogFile = await createHeartbeatLogFile();
|
|
169
182
|
await appendHeartbeatLog(
|
|
@@ -172,8 +185,15 @@ export async function runHeartbeatAgent(
|
|
|
172
185
|
);
|
|
173
186
|
await appendHeartbeatLog(
|
|
174
187
|
heartbeatLogFile,
|
|
175
|
-
`**Trigger:** ${lastRunIso === "never" ? "first run" : `last_run=${lastRunIso}`}, model=${model}
|
|
188
|
+
`**Trigger:** ${lastRunIso === "never" ? "first run" : `last_run=${lastRunIso}`}, model=${model}` +
|
|
189
|
+
`${effort.effort ? `, effort=${effort.effort}` : ""}\n`,
|
|
176
190
|
);
|
|
191
|
+
if (effort.dropped) {
|
|
192
|
+
await appendHeartbeatLog(
|
|
193
|
+
heartbeatLogFile,
|
|
194
|
+
`**Effort:** ${effort.dropped}\n`,
|
|
195
|
+
);
|
|
196
|
+
}
|
|
177
197
|
await appendHeartbeatLog(
|
|
178
198
|
heartbeatLogFile,
|
|
179
199
|
`**Prompt:**\n\`\`\`\n${prompt}\n\`\`\`\n\n---\n`,
|
|
@@ -197,6 +217,7 @@ export async function runHeartbeatAgent(
|
|
|
197
217
|
systemPrompt: buildHeartbeatSystemPrompt(),
|
|
198
218
|
workspace,
|
|
199
219
|
model,
|
|
220
|
+
...(effort.effort ? { reasoningEffort: effort.effort } : {}),
|
|
200
221
|
contextLabel: "heartbeat",
|
|
201
222
|
abortController,
|
|
202
223
|
appendLog: (text) => appendHeartbeatLog(heartbeatLogFile, text),
|
|
@@ -10,6 +10,7 @@ import { files as pathFiles } from "../../../util/paths.js";
|
|
|
10
10
|
import { kvGet, kvSet } from "../../../storage/kv.js";
|
|
11
11
|
import { importLegacyJson } from "../../../storage/legacy-import.js";
|
|
12
12
|
import type { Backend } from "../../agent-runtime/capabilities.js";
|
|
13
|
+
import type { ReasoningEffortLevel } from "../../types.js";
|
|
13
14
|
import { FailureBackoff } from "../failure-backoff.js";
|
|
14
15
|
|
|
15
16
|
export type HeartbeatState = {
|
|
@@ -28,6 +29,12 @@ export type HeartbeatState = {
|
|
|
28
29
|
export type HeartbeatConfig = {
|
|
29
30
|
model?: string;
|
|
30
31
|
heartbeatModel?: string;
|
|
32
|
+
/**
|
|
33
|
+
* Reasoning effort for heartbeat runs (config `heartbeatEffort`).
|
|
34
|
+
* Undefined = backend/model default. Passed straight through to the
|
|
35
|
+
* one-shot params; backends without a reasoning knob ignore it.
|
|
36
|
+
*/
|
|
37
|
+
heartbeatEffort?: ReasoningEffortLevel;
|
|
31
38
|
workspace?: string;
|
|
32
39
|
/**
|
|
33
40
|
* Accessor for the active backend — invoked each time a heartbeat fires so
|
|
@@ -20,14 +20,20 @@
|
|
|
20
20
|
|
|
21
21
|
import type { Backend } from "../agent-runtime/capabilities.js";
|
|
22
22
|
import type { TalonConfig } from "../../util/config.js";
|
|
23
|
+
import type { ReasoningEffortLevel } from "../types.js";
|
|
24
|
+
import {
|
|
25
|
+
normalizeReasoningLevels,
|
|
26
|
+
supportsReasoningLevel,
|
|
27
|
+
} from "../models/reasoning-levels.js";
|
|
23
28
|
|
|
24
29
|
export type ModelAuditRole = "chat" | "heartbeat" | "dream";
|
|
25
30
|
|
|
26
31
|
export type ModelAuditFinding = {
|
|
27
32
|
role: ModelAuditRole;
|
|
28
33
|
backendId: string;
|
|
34
|
+
/** The config value the finding is about — a model id, or an effort level for `unsupported-effort`. */
|
|
29
35
|
configured: string;
|
|
30
|
-
kind: "missing" | "ambiguous";
|
|
36
|
+
kind: "missing" | "ambiguous" | "unsupported-effort";
|
|
31
37
|
/** Human-readable, log-ready description with the fix. */
|
|
32
38
|
message: string;
|
|
33
39
|
};
|
|
@@ -39,6 +45,15 @@ const CONFIG_KEY: Record<ModelAuditRole, string> = {
|
|
|
39
45
|
dream: "dreamModel",
|
|
40
46
|
};
|
|
41
47
|
|
|
48
|
+
/**
|
|
49
|
+
* Effort config keys, for the roles that have one. Chat effort is a
|
|
50
|
+
* per-chat setting (`/settings`), not config, so it isn't audited here.
|
|
51
|
+
*/
|
|
52
|
+
const EFFORT_KEY: Partial<Record<ModelAuditRole, string>> = {
|
|
53
|
+
heartbeat: "heartbeatEffort",
|
|
54
|
+
dream: "dreamEffort",
|
|
55
|
+
};
|
|
56
|
+
|
|
42
57
|
/**
|
|
43
58
|
* Audit each role's configured model against its backend's catalog.
|
|
44
59
|
*
|
|
@@ -55,22 +70,25 @@ export async function auditConfiguredModels(
|
|
|
55
70
|
role: ModelAuditRole;
|
|
56
71
|
model: string | undefined;
|
|
57
72
|
backendId: string;
|
|
73
|
+
effort?: ReasoningEffortLevel;
|
|
58
74
|
}> = [
|
|
59
75
|
{ role: "chat", model: config.model, backendId: config.backend },
|
|
60
76
|
{
|
|
61
77
|
role: "heartbeat",
|
|
62
78
|
model: config.heartbeatModel,
|
|
63
79
|
backendId: config.heartbeatBackend ?? config.backend,
|
|
80
|
+
effort: config.heartbeatEffort,
|
|
64
81
|
},
|
|
65
82
|
{
|
|
66
83
|
role: "dream",
|
|
67
84
|
model: config.dreamModel,
|
|
68
85
|
backendId: config.dreamBackend ?? config.backend,
|
|
86
|
+
effort: config.dreamEffort,
|
|
69
87
|
},
|
|
70
88
|
];
|
|
71
89
|
|
|
72
90
|
const findings: ModelAuditFinding[] = [];
|
|
73
|
-
for (const { role, model, backendId } of targets) {
|
|
91
|
+
for (const { role, model, backendId, effort } of targets) {
|
|
74
92
|
if (!model || model === "default") continue;
|
|
75
93
|
|
|
76
94
|
let backend: Backend | undefined;
|
|
@@ -115,7 +133,56 @@ export async function auditConfiguredModels(
|
|
|
115
133
|
`"${backendId}" (matches: ${names}) — pin an exact id in ` +
|
|
116
134
|
`"${CONFIG_KEY[role]}" in config.json.`,
|
|
117
135
|
});
|
|
136
|
+
} else if (resolution.kind === "exact") {
|
|
137
|
+
const finding = auditEffort(
|
|
138
|
+
role,
|
|
139
|
+
backendId,
|
|
140
|
+
effort,
|
|
141
|
+
resolution.model.supportedReasoningLevels,
|
|
142
|
+
);
|
|
143
|
+
if (finding) findings.push(finding);
|
|
118
144
|
}
|
|
119
145
|
}
|
|
120
146
|
return findings;
|
|
121
147
|
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Check a role's configured effort level against what the resolved model
|
|
151
|
+
* advertises.
|
|
152
|
+
*
|
|
153
|
+
* Same "silently runs something else" failure mode as a withdrawn model:
|
|
154
|
+
* an effort the model doesn't offer is dropped at run time, and without
|
|
155
|
+
* this the operator only learns that from a heartbeat log up to an hour
|
|
156
|
+
* later (up to twelve, for dream).
|
|
157
|
+
*
|
|
158
|
+
* Returns undefined — no finding — whenever the answer isn't knowable:
|
|
159
|
+
* no effort configured, no role-level effort key, or a model that reports
|
|
160
|
+
* no level metadata at all (absent metadata is not evidence of absence).
|
|
161
|
+
* Note this only runs for roles with a PINNED model; an effort set against
|
|
162
|
+
* an unpinned model can't be checked because there's no id to resolve.
|
|
163
|
+
*/
|
|
164
|
+
function auditEffort(
|
|
165
|
+
role: ModelAuditRole,
|
|
166
|
+
backendId: string,
|
|
167
|
+
effort: ReasoningEffortLevel | undefined,
|
|
168
|
+
advertised: readonly ReasoningEffortLevel[] | undefined,
|
|
169
|
+
): ModelAuditFinding | undefined {
|
|
170
|
+
const key = EFFORT_KEY[role];
|
|
171
|
+
if (!effort || !key) return undefined;
|
|
172
|
+
|
|
173
|
+
const levels = normalizeReasoningLevels(advertised);
|
|
174
|
+
if (levels.length === 0) return undefined;
|
|
175
|
+
if (supportsReasoningLevel(effort, levels)) return undefined;
|
|
176
|
+
|
|
177
|
+
return {
|
|
178
|
+
role,
|
|
179
|
+
backendId,
|
|
180
|
+
configured: effort,
|
|
181
|
+
kind: "unsupported-effort",
|
|
182
|
+
message:
|
|
183
|
+
`${role}: configured effort "${effort}" is NOT available on the ` +
|
|
184
|
+
`model pinned for this role on backend "${backendId}" ` +
|
|
185
|
+
`(supports: ${levels.join(", ")}) — runs will use the model default. ` +
|
|
186
|
+
`Update "${key}" in config.json.`,
|
|
187
|
+
};
|
|
188
|
+
}
|
|
@@ -1,13 +1,24 @@
|
|
|
1
1
|
import type { ReasoningEffortLevel } from "../types.js";
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Ascending ladder, weakest reasoning first. This is the ONLY ordering in
|
|
5
|
+
* the codebase: `normalizeReasoningLevels` re-sorts a model's advertised
|
|
6
|
+
* levels through it, so every effort picker (Telegram, Discord, native)
|
|
7
|
+
* renders in this sequence.
|
|
8
|
+
*
|
|
9
|
+
* `xhigh` (Codex's ceiling) sits below `max` (Claude's ceiling) so the row
|
|
10
|
+
* reads low → medium → high → xhigh → max. A single model advertises one
|
|
11
|
+
* ceiling or the other, never both, so their relative position only ever
|
|
12
|
+
* matters for how the ladder reads.
|
|
13
|
+
*/
|
|
3
14
|
export const REASONING_LEVEL_ORDER: ReasoningEffortLevel[] = [
|
|
4
15
|
"off",
|
|
5
16
|
"minimal",
|
|
6
17
|
"low",
|
|
7
18
|
"medium",
|
|
8
19
|
"high",
|
|
9
|
-
"max",
|
|
10
20
|
"xhigh",
|
|
21
|
+
"max",
|
|
11
22
|
];
|
|
12
23
|
|
|
13
24
|
export const REASONING_LEVEL_LABELS: Record<ReasoningEffortLevel, string> = {
|
|
@@ -16,8 +27,8 @@ export const REASONING_LEVEL_LABELS: Record<ReasoningEffortLevel, string> = {
|
|
|
16
27
|
low: "Low",
|
|
17
28
|
medium: "Med",
|
|
18
29
|
high: "High",
|
|
19
|
-
max: "Max",
|
|
20
30
|
xhigh: "XHigh",
|
|
31
|
+
max: "Max",
|
|
21
32
|
};
|
|
22
33
|
|
|
23
34
|
export const REASONING_LEVEL_DESCRIPTIONS: Record<
|
|
@@ -29,8 +40,8 @@ export const REASONING_LEVEL_DESCRIPTIONS: Record<
|
|
|
29
40
|
low: "short reasoning pass",
|
|
30
41
|
medium: "balanced reasoning",
|
|
31
42
|
high: "deeper reasoning, slower",
|
|
32
|
-
max: "maximum Claude reasoning budget",
|
|
33
43
|
xhigh: "maximum Codex reasoning budget",
|
|
44
|
+
max: "maximum Claude reasoning budget",
|
|
34
45
|
adaptive: "use the model/backend default",
|
|
35
46
|
};
|
|
36
47
|
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
* 3. Frontend capabilities ~/.talon/prompts/<frontend>.md
|
|
19
19
|
* 4. Persistent memory (size-capped) prompts/system/persistent-memory.md
|
|
20
20
|
* wrapping ~/.talon/workspace/memory/memory.md
|
|
21
|
-
* 5.
|
|
21
|
+
* 5. Memory recall + capability docs prompts/system/{memory-recall,workspace,...}.md
|
|
22
22
|
* 6. Plugin additions plugin.systemPrompt() contributions
|
|
23
23
|
* (7. Delivery contract — appended by the backend as its suffix,
|
|
24
24
|
* AFTER plugins, so it is the last thing the model reads.
|
|
@@ -207,11 +207,14 @@ export function assembleSystemPrompt(
|
|
|
207
207
|
loaded.push(truncated ? "memory(capped)" : "memory");
|
|
208
208
|
}
|
|
209
209
|
|
|
210
|
-
// 5.
|
|
211
|
-
//
|
|
212
|
-
//
|
|
213
|
-
// in
|
|
210
|
+
// 5. Package-owned behavioural and capability docs. The memory policy
|
|
211
|
+
// is deliberately package-owned so custom identity/base prompts cannot
|
|
212
|
+
// remove recall-before-asking or adaptive persistence behaviour.
|
|
213
|
+
// Provider-specific additions follow in step 6 and become canonical
|
|
214
|
+
// when their tools are available; otherwise the policy falls back to
|
|
215
|
+
// memory.md + daily notes.
|
|
214
216
|
staticParts.push(
|
|
217
|
+
loadSystemTemplate("memory-recall"),
|
|
215
218
|
loadSystemTemplate("workspace"),
|
|
216
219
|
loadSystemTemplate("cron"),
|
|
217
220
|
loadSystemTemplate("triggers"),
|
|
@@ -26,13 +26,14 @@ import asset12 from "../../../prompts/system/cron.md" with { type: "file" };
|
|
|
26
26
|
import asset13 from "../../../prompts/system/daily-memory.md" with { type: "file" };
|
|
27
27
|
import asset14 from "../../../prompts/system/goals.md" with { type: "file" };
|
|
28
28
|
import asset15 from "../../../prompts/system/heartbeat-agent.md" with { type: "file" };
|
|
29
|
-
import asset16 from "../../../prompts/system/
|
|
30
|
-
import asset17 from "../../../prompts/system/
|
|
31
|
-
import asset18 from "../../../prompts/system/
|
|
32
|
-
import asset19 from "../../../prompts/system/
|
|
33
|
-
import asset20 from "../../../prompts/
|
|
34
|
-
import asset21 from "../../../prompts/
|
|
35
|
-
import asset22 from "../../../prompts/
|
|
29
|
+
import asset16 from "../../../prompts/system/memory-recall.md" with { type: "file" };
|
|
30
|
+
import asset17 from "../../../prompts/system/persistent-memory.md" with { type: "file" };
|
|
31
|
+
import asset18 from "../../../prompts/system/skills.md" with { type: "file" };
|
|
32
|
+
import asset19 from "../../../prompts/system/triggers.md" with { type: "file" };
|
|
33
|
+
import asset20 from "../../../prompts/system/workspace.md" with { type: "file" };
|
|
34
|
+
import asset21 from "../../../prompts/teams.md" with { type: "file" };
|
|
35
|
+
import asset22 from "../../../prompts/telegram.md" with { type: "file" };
|
|
36
|
+
import asset23 from "../../../prompts/terminal.md" with { type: "file" };
|
|
36
37
|
|
|
37
38
|
/** rel path (posix, under prompts/) → embedded file path (/$bunfs/… when compiled). */
|
|
38
39
|
const ASSETS: Record<string, string> = {
|
|
@@ -52,13 +53,14 @@ const ASSETS: Record<string, string> = {
|
|
|
52
53
|
"system/daily-memory.md": asset13,
|
|
53
54
|
"system/goals.md": asset14,
|
|
54
55
|
"system/heartbeat-agent.md": asset15,
|
|
55
|
-
"system/
|
|
56
|
-
"system/
|
|
57
|
-
"system/
|
|
58
|
-
"system/
|
|
59
|
-
"
|
|
60
|
-
"
|
|
61
|
-
"
|
|
56
|
+
"system/memory-recall.md": asset16,
|
|
57
|
+
"system/persistent-memory.md": asset17,
|
|
58
|
+
"system/skills.md": asset18,
|
|
59
|
+
"system/triggers.md": asset19,
|
|
60
|
+
"system/workspace.md": asset20,
|
|
61
|
+
"teams.md": asset21,
|
|
62
|
+
"telegram.md": asset22,
|
|
63
|
+
"terminal.md": asset23,
|
|
62
64
|
};
|
|
63
65
|
|
|
64
66
|
/** Read an embedded prompt by its rel path (e.g. "system/cron.md"). */
|
package/src/core/types.ts
CHANGED
|
@@ -141,6 +141,16 @@ export type OneShotAgentParams = {
|
|
|
141
141
|
workspace: string;
|
|
142
142
|
/** Model id (interpretation is backend-specific). */
|
|
143
143
|
model: string;
|
|
144
|
+
/**
|
|
145
|
+
* Reasoning effort for the run (config `heartbeatEffort` / `dreamEffort`).
|
|
146
|
+
* Undefined = let the backend/model pick its own default, which is what
|
|
147
|
+
* every background run did before the knob existed.
|
|
148
|
+
*
|
|
149
|
+
* Honoured by the backends that expose a reasoning knob (Claude SDK
|
|
150
|
+
* thinking/effort, Codex `modelReasoningEffort`); ignored by the ones
|
|
151
|
+
* that don't (Kilo, OpenCode).
|
|
152
|
+
*/
|
|
153
|
+
reasoningEffort?: ReasoningEffortLevel;
|
|
144
154
|
/**
|
|
145
155
|
* Sentinel chat ID for outbound MCP tool calls (e.g. "heartbeat", "dream").
|
|
146
156
|
* Frontend MCP servers use this to enforce explicit `chat_id` on outbound
|
package/src/util/config.ts
CHANGED
|
@@ -5,6 +5,7 @@ import { dirs, files as pathFiles } from "./paths.js";
|
|
|
5
5
|
import { hardenTalonPermissions } from "./harden.js";
|
|
6
6
|
import { setTimezone } from "./time.js";
|
|
7
7
|
import { BACKEND_IDS } from "../core/agent-runtime/model-ref.js";
|
|
8
|
+
import { REASONING_LEVEL_ORDER } from "../core/models/reasoning-levels.js";
|
|
8
9
|
import {
|
|
9
10
|
assembleSystemPrompt,
|
|
10
11
|
joinSystemPromptParts,
|
|
@@ -28,6 +29,20 @@ const BACKEND_ID_ENUM = [...BACKEND_IDS] as [
|
|
|
28
29
|
...(typeof BACKEND_IDS)[number][],
|
|
29
30
|
];
|
|
30
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Reasoning-effort literal source for the background-agent knobs
|
|
34
|
+
* (`heartbeatEffort` / `dreamEffort`). Same trick as `BACKEND_ID_ENUM`:
|
|
35
|
+
* reuse the single source of truth (`REASONING_LEVEL_ORDER`) so adding a
|
|
36
|
+
* level to the vocabulary doesn't need a second edit here.
|
|
37
|
+
*
|
|
38
|
+
* There is no `"adaptive"` member — leaving the field unset IS adaptive
|
|
39
|
+
* (the backend/model default), matching how per-chat `effort` behaves.
|
|
40
|
+
*/
|
|
41
|
+
const REASONING_EFFORT_ENUM = [...REASONING_LEVEL_ORDER] as [
|
|
42
|
+
(typeof REASONING_LEVEL_ORDER)[number],
|
|
43
|
+
...(typeof REASONING_LEVEL_ORDER)[number][],
|
|
44
|
+
];
|
|
45
|
+
|
|
31
46
|
// ── Config schema ───────────────────────────────────────────────────────────
|
|
32
47
|
|
|
33
48
|
/** Path-based Talon plugin (loaded as a Node module). */
|
|
@@ -317,6 +332,15 @@ const configSchema = z.object({
|
|
|
317
332
|
*/
|
|
318
333
|
backendDefaults: z.record(z.string(), z.string()).optional(),
|
|
319
334
|
dreamModel: z.string().optional(), // Model used for background memory consolidation (defaults to main model)
|
|
335
|
+
/**
|
|
336
|
+
* Reasoning effort for the dream / memory-consolidation agent. Unset =
|
|
337
|
+
* the backend/model default. Pair with `dreamModel` when you want a
|
|
338
|
+
* cheap model that still thinks hard (or an expensive one that doesn't).
|
|
339
|
+
*
|
|
340
|
+
* Honoured by backends with a reasoning knob (Claude SDK, Codex);
|
|
341
|
+
* silently ignored by Kilo / OpenCode, which have none.
|
|
342
|
+
*/
|
|
343
|
+
dreamEffort: z.enum(REASONING_EFFORT_ENUM).optional(),
|
|
320
344
|
maxMessageLength: z.number().int().min(100).default(4000),
|
|
321
345
|
concurrency: z.number().int().min(1).max(20).default(1),
|
|
322
346
|
apiId: z.number().int().optional(),
|
|
@@ -339,6 +363,16 @@ const configSchema = z.object({
|
|
|
339
363
|
heartbeat: z.boolean().default(true),
|
|
340
364
|
heartbeatIntervalMinutes: z.number().int().min(5).default(60),
|
|
341
365
|
heartbeatModel: z.string().optional(), // Model for heartbeat agent (defaults to main model)
|
|
366
|
+
/**
|
|
367
|
+
* Reasoning effort for the heartbeat agent. Unset = the backend/model
|
|
368
|
+
* default. Pair with `heartbeatModel` — e.g. `"high"` so unattended
|
|
369
|
+
* goal work reasons harder than a chat turn, or `"low"` to keep hourly
|
|
370
|
+
* runs cheap.
|
|
371
|
+
*
|
|
372
|
+
* Honoured by backends with a reasoning knob (Claude SDK, Codex);
|
|
373
|
+
* silently ignored by Kilo / OpenCode, which have none.
|
|
374
|
+
*/
|
|
375
|
+
heartbeatEffort: z.enum(REASONING_EFFORT_ENUM).optional(),
|
|
342
376
|
braveApiKey: z.string().optional(),
|
|
343
377
|
/**
|
|
344
378
|
* Codex-specific OpenAI API key. Prefer this, CODEX_API_KEY, or
|