talon-agent 3.14.0 → 3.15.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app.ts +24 -10
- package/src/backend/kilo/factory.ts +26 -0
- package/src/backend/kilo/handler/message.ts +12 -6
- package/src/backend/kilo/handler/turn.ts +43 -40
- package/src/backend/kilo/index.ts +7 -1
- package/src/backend/kilo/server.ts +28 -32
- package/src/backend/kilo/sessions.ts +17 -0
- package/src/backend/opencode/factory.ts +26 -0
- package/src/backend/opencode/handler/message.ts +13 -5
- package/src/backend/opencode/handler/turn.ts +41 -32
- package/src/backend/opencode/index.ts +7 -1
- package/src/backend/opencode/server.ts +27 -16
- package/src/backend/opencode/sessions.ts +17 -0
- package/src/backend/remote-server/client.ts +1 -1
- package/src/backend/remote-server/events.ts +19 -6
- package/src/backend/remote-server/index.ts +16 -1
- package/src/backend/remote-server/lifecycle.ts +5 -1
- package/src/backend/remote-server/mcp.ts +196 -110
- package/src/backend/remote-server/one-shot.ts +53 -6
- package/src/backend/remote-server/providers.ts +12 -0
- package/src/backend/remote-server/session-helpers.ts +69 -0
- package/src/backend/remote-server/sessions.ts +67 -7
- package/src/backend/remote-server/sse-stream.ts +17 -10
- package/src/backend/remote-server/state.ts +6 -0
- package/src/backend/remote-server/turn-timeout.ts +71 -0
- package/src/core/mcp-hub/index.ts +22 -0
- package/src/frontend/shared/status-context.ts +161 -0
- package/src/frontend/telegram/callbacks/model.ts +48 -0
- package/src/frontend/telegram/commands/settings.ts +8 -2
- package/src/frontend/telegram/helpers/menu.ts +5 -2
- package/src/frontend/terminal/commands.ts +152 -0
- package/src/util/respawn.ts +66 -38
|
@@ -17,7 +17,12 @@ import {
|
|
|
17
17
|
import { log, logWarn } from "../../util/log.js";
|
|
18
18
|
import type { RemoteAgentClient, RemotePermissionRule } from "./client.js";
|
|
19
19
|
import type { RemoteServerState } from "./state.js";
|
|
20
|
-
import {
|
|
20
|
+
import {
|
|
21
|
+
TALON_MCP_SERVER_NAME,
|
|
22
|
+
TALON_PLUGIN_MCP_SERVER_NAME,
|
|
23
|
+
getChatMcpServerName,
|
|
24
|
+
safeMcpNamePart,
|
|
25
|
+
} from "./mcp.js";
|
|
21
26
|
|
|
22
27
|
/**
|
|
23
28
|
* Build the per-session permission ruleset Talon installs on every fresh
|
|
@@ -33,14 +38,13 @@ import { TALON_MCP_SERVER_NAME, getChatMcpServerName } from "./mcp.js";
|
|
|
33
38
|
* "No active chat context". Or in the cross-chat case where chat B
|
|
34
39
|
* IS active, the model in chat A could leak content into chat B.
|
|
35
40
|
* Deny pattern blocks both. (Visibility is also blocked at the
|
|
36
|
-
*
|
|
37
|
-
* defense in depth.)
|
|
41
|
+
* prompt layer by `buildToolOverrides`; this rule is defense in depth.)
|
|
38
42
|
*
|
|
39
|
-
* 2. Auto-allow built-in tools (`tool *`, `edit *`, `bash *`)
|
|
43
|
+
* 2. Auto-allow built-in tools (`tool *`, `edit *`, `bash *`) and
|
|
44
|
+
* workspace paths outside the server process's launch directory so they
|
|
40
45
|
* don't sit in `permission.asked` waiting for a reply that never
|
|
41
|
-
* arrives
|
|
42
|
-
*
|
|
43
|
-
* first `read` call hangs the entire turn.
|
|
46
|
+
* arrives. The permission watchdog is a fallback for new categories;
|
|
47
|
+
* this rule resolves the known path case without a polling round trip.
|
|
44
48
|
*
|
|
45
49
|
* Rules are evaluated in order; first match wins. (See upstream's
|
|
46
50
|
* `PermissionRule` type — `permission` is the rule category, `pattern`
|
|
@@ -48,6 +52,7 @@ import { TALON_MCP_SERVER_NAME, getChatMcpServerName } from "./mcp.js";
|
|
|
48
52
|
*/
|
|
49
53
|
export function buildPermissionRuleset(chatId: string): RemotePermissionRule[] {
|
|
50
54
|
const ourServerName = getChatMcpServerName(chatId);
|
|
55
|
+
const ourPluginPrefix = `${TALON_PLUGIN_MCP_SERVER_NAME}-${safeMcpNamePart(chatId, "chat")}-`;
|
|
51
56
|
return [
|
|
52
57
|
{ permission: "tool", pattern: `${ourServerName}_*`, action: "allow" },
|
|
53
58
|
{
|
|
@@ -55,9 +60,16 @@ export function buildPermissionRuleset(chatId: string): RemotePermissionRule[] {
|
|
|
55
60
|
pattern: `${TALON_MCP_SERVER_NAME}-*`,
|
|
56
61
|
action: "deny",
|
|
57
62
|
},
|
|
63
|
+
{ permission: "tool", pattern: `${ourPluginPrefix}*`, action: "allow" },
|
|
64
|
+
{
|
|
65
|
+
permission: "tool",
|
|
66
|
+
pattern: `${TALON_PLUGIN_MCP_SERVER_NAME}-*`,
|
|
67
|
+
action: "deny",
|
|
68
|
+
},
|
|
58
69
|
{ permission: "tool", pattern: "*", action: "allow" },
|
|
59
70
|
{ permission: "edit", pattern: "*", action: "allow" },
|
|
60
71
|
{ permission: "bash", pattern: "*", action: "allow" },
|
|
72
|
+
{ permission: "external_directory", pattern: "*", action: "allow" },
|
|
61
73
|
];
|
|
62
74
|
}
|
|
63
75
|
|
|
@@ -106,3 +118,51 @@ export async function ensureRemoteSession<TClient extends RemoteAgentClient>(
|
|
|
106
118
|
|
|
107
119
|
return newId;
|
|
108
120
|
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* The per-turn setup a warm-up front-loads. Both remote-server backends
|
|
124
|
+
* expose these under identical signatures; the shape lets the helper stay
|
|
125
|
+
* backend-agnostic without importing either SDK.
|
|
126
|
+
*/
|
|
127
|
+
export interface RemoteWarmDeps<TClient extends RemoteAgentClient> {
|
|
128
|
+
ensureServer(): Promise<TClient>;
|
|
129
|
+
ensureSession(client: TClient, chatId: string): Promise<string>;
|
|
130
|
+
ensureChatMcpServer(client: TClient, chatId: string): Promise<string>;
|
|
131
|
+
ensurePluginMcpServers(client: TClient, chatId: string): Promise<string[]>;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Pre-pay a chat's cold start: spawn the server if it isn't up, create (or
|
|
136
|
+
* resume) the session, and register the chat + plugin MCP servers.
|
|
137
|
+
*
|
|
138
|
+
* This is the remote-server analogue of the Claude backend's `warmSession`,
|
|
139
|
+
* and closes the last `sessions` capability gap between the two families.
|
|
140
|
+
* `performSessionReset` and the native frontend call it right after a reset,
|
|
141
|
+
* so the first turn on a fresh session doesn't serially pay session creation
|
|
142
|
+
* plus a full plugin-MCP registration sweep — the dominant cold-start cost
|
|
143
|
+
* here, since each plugin server is a separate connect.
|
|
144
|
+
*
|
|
145
|
+
* Best-effort by contract: `/reset` has already succeeded by the time this
|
|
146
|
+
* runs, and the same work is idempotent and repeated at the head of every
|
|
147
|
+
* turn. A failure must degrade to a slow first turn, never surface as a
|
|
148
|
+
* failed reset — so everything is caught and logged, not rethrown.
|
|
149
|
+
*/
|
|
150
|
+
export async function warmRemoteSession<TClient extends RemoteAgentClient>(
|
|
151
|
+
state: RemoteServerState<TClient>,
|
|
152
|
+
chatId: string,
|
|
153
|
+
deps: RemoteWarmDeps<TClient>,
|
|
154
|
+
): Promise<void> {
|
|
155
|
+
try {
|
|
156
|
+
const client = await deps.ensureServer();
|
|
157
|
+
await deps.ensureSession(client, chatId);
|
|
158
|
+
await deps.ensureChatMcpServer(client, chatId);
|
|
159
|
+
await deps.ensurePluginMcpServers(client, chatId);
|
|
160
|
+
log("agent", `[${chatId}] Warmed ${state.label} session`);
|
|
161
|
+
} catch (err) {
|
|
162
|
+
logWarn(
|
|
163
|
+
"agent",
|
|
164
|
+
`[${chatId}] ${state.label} warm-up skipped (first turn pays cold start): ` +
|
|
165
|
+
`${err instanceof Error ? err.message : String(err)}`,
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
}
|
|
@@ -44,17 +44,24 @@ export async function subscribeSseStream(
|
|
|
44
44
|
client: SseSubscribableClient,
|
|
45
45
|
chatId: string,
|
|
46
46
|
): Promise<AsyncIterable<unknown> | undefined> {
|
|
47
|
-
let
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
"
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
47
|
+
let lastError: unknown;
|
|
48
|
+
for (let attempt = 1; attempt <= 3; attempt++) {
|
|
49
|
+
try {
|
|
50
|
+
const stream = narrowSseResult(await client.global.event());
|
|
51
|
+
if (stream) return stream;
|
|
52
|
+
lastError = new Error("response did not contain an async event stream");
|
|
53
|
+
} catch (err) {
|
|
54
|
+
lastError = err;
|
|
55
|
+
}
|
|
56
|
+
if (attempt < 3) {
|
|
57
|
+
await new Promise((resolve) => setTimeout(resolve, 150 * attempt));
|
|
58
|
+
}
|
|
56
59
|
}
|
|
57
|
-
|
|
60
|
+
logWarn(
|
|
61
|
+
"agent",
|
|
62
|
+
`[${chatId}] SSE subscribe failed after 3 attempts: ${lastError instanceof Error ? lastError.message : String(lastError)}`,
|
|
63
|
+
);
|
|
64
|
+
return undefined;
|
|
58
65
|
}
|
|
59
66
|
|
|
60
67
|
/**
|
|
@@ -52,6 +52,10 @@ export interface RemoteServerState<TClient extends RemoteAgentClient> {
|
|
|
52
52
|
* actual state, so we cache locally instead of trusting the server.
|
|
53
53
|
*/
|
|
54
54
|
readonly registeredMcpServers: Set<string>;
|
|
55
|
+
/** Exact tool names exposed by each registered MCP server. */
|
|
56
|
+
readonly registeredMcpTools: Map<string, readonly string[]>;
|
|
57
|
+
/** Plugin-name → registered server-name mapping for each chat context. */
|
|
58
|
+
readonly pluginMcpServersByChat: Map<string, Map<string, string>>;
|
|
55
59
|
}
|
|
56
60
|
|
|
57
61
|
/** Inputs for {@link createRemoteServerState}. */
|
|
@@ -84,6 +88,8 @@ export function createRemoteServerState<TClient extends RemoteAgentClient>(
|
|
|
84
88
|
serverHandle: null,
|
|
85
89
|
modelProviderCache: new Map(),
|
|
86
90
|
registeredMcpServers: new Set(),
|
|
91
|
+
registeredMcpTools: new Map(),
|
|
92
|
+
pluginMcpServersByChat: new Map(),
|
|
87
93
|
};
|
|
88
94
|
}
|
|
89
95
|
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
/** Bounded turn execution for OpenCode/Kilo's long-lived HTTP servers. */
|
|
2
|
+
|
|
3
|
+
import { logWarn } from "../../util/log.js";
|
|
4
|
+
|
|
5
|
+
const DEFAULT_REMOTE_TURN_TIMEOUT_MS = 10 * 60_000;
|
|
6
|
+
|
|
7
|
+
export function remoteTurnTimeoutMs(): number {
|
|
8
|
+
const configured = Number(process.env.TALON_REMOTE_TURN_TIMEOUT_MS);
|
|
9
|
+
return Number.isFinite(configured) && configured > 0
|
|
10
|
+
? configured
|
|
11
|
+
: DEFAULT_REMOTE_TURN_TIMEOUT_MS;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export class RemoteTurnTimeoutError extends Error {
|
|
15
|
+
constructor(label: string, timeoutMs: number) {
|
|
16
|
+
super(`${label} turn timed out after ${Math.round(timeoutMs / 1000)}s`);
|
|
17
|
+
// Core error classification treats TimeoutError as a retryable transport
|
|
18
|
+
// failure, so the normal fallback-model path remains available.
|
|
19
|
+
this.name = "TimeoutError";
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export interface RemoteTurnAbortClient {
|
|
24
|
+
session: {
|
|
25
|
+
abort(args: { sessionID: string }): Promise<unknown>;
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Race a prompt+SSE turn against a hard deadline. On expiry, abort upstream
|
|
31
|
+
* generation before rejecting so the provider stops spending tokens and the
|
|
32
|
+
* session can be safely reset by the shared retry path.
|
|
33
|
+
*/
|
|
34
|
+
export async function awaitRemoteTurn<T>(
|
|
35
|
+
turn: Promise<T>,
|
|
36
|
+
inputs: {
|
|
37
|
+
client: RemoteTurnAbortClient;
|
|
38
|
+
sessionId: string;
|
|
39
|
+
chatId: string;
|
|
40
|
+
label: string;
|
|
41
|
+
timeoutMs?: number;
|
|
42
|
+
},
|
|
43
|
+
): Promise<T> {
|
|
44
|
+
const timeoutMs = inputs.timeoutMs ?? remoteTurnTimeoutMs();
|
|
45
|
+
let timer: ReturnType<typeof setTimeout> | null = null;
|
|
46
|
+
try {
|
|
47
|
+
return await Promise.race([
|
|
48
|
+
turn,
|
|
49
|
+
new Promise<never>((_, reject) => {
|
|
50
|
+
timer = setTimeout(() => {
|
|
51
|
+
logWarn(
|
|
52
|
+
"agent",
|
|
53
|
+
`[${inputs.chatId}] ${inputs.label} turn exceeded ${Math.round(timeoutMs / 1000)}s; aborting`,
|
|
54
|
+
);
|
|
55
|
+
inputs.client.session
|
|
56
|
+
.abort({ sessionID: inputs.sessionId })
|
|
57
|
+
.catch((err) =>
|
|
58
|
+
logWarn(
|
|
59
|
+
"agent",
|
|
60
|
+
`[${inputs.chatId}] ${inputs.label} timeout abort failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
61
|
+
),
|
|
62
|
+
);
|
|
63
|
+
reject(new RemoteTurnTimeoutError(inputs.label, timeoutMs));
|
|
64
|
+
}, timeoutMs);
|
|
65
|
+
timer.unref?.();
|
|
66
|
+
}),
|
|
67
|
+
]);
|
|
68
|
+
} finally {
|
|
69
|
+
if (timer) clearTimeout(timer);
|
|
70
|
+
}
|
|
71
|
+
}
|
|
@@ -104,6 +104,28 @@ export function hubPluginServerNames(only?: string[]): string[] {
|
|
|
104
104
|
return Object.keys(getPluginMcpServers("", "hub-enum", only));
|
|
105
105
|
}
|
|
106
106
|
|
|
107
|
+
/**
|
|
108
|
+
* Enumerate one plugin server's tools through the same hub-managed child that
|
|
109
|
+
* serves MCP requests. Remote agent servers do not include dynamically-added
|
|
110
|
+
* MCP tools in their `/experimental/tool/ids` response, so OpenCode/Kilo need
|
|
111
|
+
* this authoritative list to build per-chat visibility overrides.
|
|
112
|
+
*/
|
|
113
|
+
export async function listHubPluginToolNames(
|
|
114
|
+
serverName: string,
|
|
115
|
+
chatId: string,
|
|
116
|
+
bridgeUrl: string,
|
|
117
|
+
): Promise<string[]> {
|
|
118
|
+
const key =
|
|
119
|
+
serverName === "brave-search"
|
|
120
|
+
? "brave-search"
|
|
121
|
+
: `${serverName}\u0000${chatId}`;
|
|
122
|
+
const child = await acquireChild(key, () =>
|
|
123
|
+
pluginSpec(serverName, chatId, bridgeUrl),
|
|
124
|
+
);
|
|
125
|
+
child.touch();
|
|
126
|
+
return (await child.listTools()).map((tool) => tool.name);
|
|
127
|
+
}
|
|
128
|
+
|
|
107
129
|
// ── Server construction per session ─────────────────────────────────────────
|
|
108
130
|
|
|
109
131
|
type HubTarget =
|
|
@@ -1,5 +1,166 @@
|
|
|
1
1
|
import type { CacheMetricsSupport } from "../../core/types.js";
|
|
2
2
|
|
|
3
|
+
// ── /context breakdown ────────────────────────────────────────────────────────
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Rough token estimate — ~4 chars/token, the house heuristic used everywhere
|
|
7
|
+
* (soul/projector, cache-telemetry). No real tokenizer is wired, so every
|
|
8
|
+
* measured figure in the breakdown below is an estimate at this fidelity.
|
|
9
|
+
*/
|
|
10
|
+
export function estimateContextTokens(text: string): number {
|
|
11
|
+
return Math.ceil(text.length / 4);
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export type ContextSegmentKey = "system" | "tools" | "conversation";
|
|
15
|
+
|
|
16
|
+
interface ContextSegment {
|
|
17
|
+
key: ContextSegmentKey;
|
|
18
|
+
label: string;
|
|
19
|
+
tokens: number;
|
|
20
|
+
/** Share of the window (0–100) when the window is known, else share of used. */
|
|
21
|
+
pct: number;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export interface ContextBreakdown {
|
|
25
|
+
/** True once there is anything to show (a system prompt or a reported fill). */
|
|
26
|
+
known: boolean;
|
|
27
|
+
/** True when the window size is known — only then can free space be shown. */
|
|
28
|
+
windowKnown: boolean;
|
|
29
|
+
used: number;
|
|
30
|
+
max: number;
|
|
31
|
+
usedPct: number;
|
|
32
|
+
free: number;
|
|
33
|
+
freePct: number;
|
|
34
|
+
/** Fixed → variable order: System, Tools, Conversation. */
|
|
35
|
+
segments: ContextSegment[];
|
|
36
|
+
warn: boolean;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function posInt(n: number | undefined): number {
|
|
40
|
+
return typeof n === "number" && Number.isFinite(n) && n > 0
|
|
41
|
+
? Math.round(n)
|
|
42
|
+
: 0;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function round1(n: number): number {
|
|
46
|
+
return Math.round(n * 10) / 10;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Decompose the context window into System / Tools / Conversation + free.
|
|
51
|
+
*
|
|
52
|
+
* The honesty this has to preserve: the backend reports exactly one
|
|
53
|
+
* authoritative number, `contextTokens` (the real last-turn fill). Nothing
|
|
54
|
+
* reports a per-category split, so:
|
|
55
|
+
*
|
|
56
|
+
* - **System** is measured from the actual (frozen) system prompt — accurate,
|
|
57
|
+
* and the part a user can act on.
|
|
58
|
+
* - **Conversation** is estimated from stored history. It can overshoot what
|
|
59
|
+
* is really in-window after compaction; when it does, it is clamped to fit.
|
|
60
|
+
* - **Tools** is the residual: `used − system − conversation`. Tool schemas
|
|
61
|
+
* (invisible to us — they live inside the SDK) dominate it, but it also
|
|
62
|
+
* absorbs message-formatting overhead and estimation slack. Labelled as the
|
|
63
|
+
* remainder, not claimed as exact.
|
|
64
|
+
*
|
|
65
|
+
* When the backend reports no fill, tools cannot be derived, so only the two
|
|
66
|
+
* measured/estimated parts are shown and the window's free space (if known) is
|
|
67
|
+
* whatever is left of it.
|
|
68
|
+
*/
|
|
69
|
+
export function buildContextBreakdown(input: {
|
|
70
|
+
contextTokens?: number;
|
|
71
|
+
contextWindow?: number;
|
|
72
|
+
systemTokens: number;
|
|
73
|
+
conversationTokens: number;
|
|
74
|
+
}): ContextBreakdown {
|
|
75
|
+
const max = posInt(input.contextWindow);
|
|
76
|
+
const system = Math.max(0, Math.round(input.systemTokens));
|
|
77
|
+
let conversation = Math.max(0, Math.round(input.conversationTokens));
|
|
78
|
+
const fill = posInt(input.contextTokens);
|
|
79
|
+
|
|
80
|
+
const segments: ContextSegment[] = [];
|
|
81
|
+
let used: number;
|
|
82
|
+
|
|
83
|
+
if (fill > 0) {
|
|
84
|
+
used = fill;
|
|
85
|
+
// System is sent in full every turn; if our estimate exceeds the real fill
|
|
86
|
+
// that is estimation slack, not reality — clamp so parts never exceed used.
|
|
87
|
+
const sys = Math.min(system, used);
|
|
88
|
+
let tools = used - sys - conversation;
|
|
89
|
+
if (tools < 0) {
|
|
90
|
+
// Conversation overshot the real fill (compaction dropped in-window
|
|
91
|
+
// messages) — give the remainder back to conversation, zero the residual.
|
|
92
|
+
conversation = Math.max(0, used - sys);
|
|
93
|
+
tools = 0;
|
|
94
|
+
}
|
|
95
|
+
segments.push({ key: "system", label: "System", tokens: sys, pct: 0 });
|
|
96
|
+
segments.push({ key: "tools", label: "Tools", tokens: tools, pct: 0 });
|
|
97
|
+
segments.push({
|
|
98
|
+
key: "conversation",
|
|
99
|
+
label: "Conversation",
|
|
100
|
+
tokens: conversation,
|
|
101
|
+
pct: 0,
|
|
102
|
+
});
|
|
103
|
+
} else {
|
|
104
|
+
// No authoritative fill — show what we can measure; tools isn't derivable.
|
|
105
|
+
used = system + conversation;
|
|
106
|
+
segments.push({ key: "system", label: "System", tokens: system, pct: 0 });
|
|
107
|
+
segments.push({
|
|
108
|
+
key: "conversation",
|
|
109
|
+
label: "Conversation",
|
|
110
|
+
tokens: conversation,
|
|
111
|
+
pct: 0,
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
const windowKnown = max > 0;
|
|
116
|
+
const free = windowKnown ? Math.max(0, max - used) : 0;
|
|
117
|
+
const denom = windowKnown ? max : used;
|
|
118
|
+
for (const s of segments) {
|
|
119
|
+
s.pct = denom > 0 ? round1((s.tokens / denom) * 100) : 0;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
return {
|
|
123
|
+
known: used > 0,
|
|
124
|
+
windowKnown,
|
|
125
|
+
used,
|
|
126
|
+
max,
|
|
127
|
+
usedPct: windowKnown ? Math.min(100, round1((used / max) * 100)) : 0,
|
|
128
|
+
free,
|
|
129
|
+
freePct: windowKnown ? round1((free / max) * 100) : 0,
|
|
130
|
+
segments,
|
|
131
|
+
warn: windowKnown && used / max >= 0.8,
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Distribute `width` integer cells across `weights` proportionally, by the
|
|
137
|
+
* largest-remainder method — the cells sum to exactly `width` and no positive
|
|
138
|
+
* weight is systematically rounded to nothing before its peers. Zero weights
|
|
139
|
+
* get zero cells. Used to lay out the segmented bar so its coloured runs sum to
|
|
140
|
+
* the bar width regardless of rounding.
|
|
141
|
+
*/
|
|
142
|
+
export function apportionCells(
|
|
143
|
+
weights: readonly number[],
|
|
144
|
+
width: number,
|
|
145
|
+
): number[] {
|
|
146
|
+
const total = weights.reduce((a, b) => a + b, 0);
|
|
147
|
+
if (total <= 0 || width <= 0) return weights.map(() => 0);
|
|
148
|
+
const exact = weights.map((w) => (Math.max(0, w) / total) * width);
|
|
149
|
+
const cells = exact.map((x) => Math.floor(x));
|
|
150
|
+
let remaining = width - cells.reduce((a, b) => a + b, 0);
|
|
151
|
+
const byRemainder = exact
|
|
152
|
+
.map((x, i) => ({ i, frac: x - Math.floor(x) }))
|
|
153
|
+
.sort((a, b) => b.frac - a.frac);
|
|
154
|
+
for (const { i } of byRemainder) {
|
|
155
|
+
if (remaining <= 0) break;
|
|
156
|
+
if (weights[i]! > 0) {
|
|
157
|
+
cells[i]!++;
|
|
158
|
+
remaining--;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
return cells;
|
|
162
|
+
}
|
|
163
|
+
|
|
3
164
|
export interface ContextDisplay {
|
|
4
165
|
known: boolean;
|
|
5
166
|
used: number;
|
|
@@ -44,6 +44,47 @@ import {
|
|
|
44
44
|
editOrIgnoreSame,
|
|
45
45
|
type CallbackDeps,
|
|
46
46
|
} from "./shared.js";
|
|
47
|
+
import { logWarn } from "../../../util/log.js";
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Run the backend-side half of a chat's session handoff.
|
|
51
|
+
*
|
|
52
|
+
* A backend switch already clears Talon's own stores (session row,
|
|
53
|
+
* history, pulse checkpoint), but those are only half the state.
|
|
54
|
+
* `performSessionReset` also drives the backend capability slots, and
|
|
55
|
+
* the switch path skipped them entirely:
|
|
56
|
+
*
|
|
57
|
+
* - the OUTGOING backend keeps in-process per-chat state that no
|
|
58
|
+
* amount of clearing Talon's stores touches. openai-agents holds a
|
|
59
|
+
* `MemorySession` map keyed by chat id, so switching away and back
|
|
60
|
+
* resurrected the old conversation the operator meant to drop.
|
|
61
|
+
* - the INCOMING backend was never warmed, so the first turn after a
|
|
62
|
+
* switch paid the full cold start — on OpenCode/Kilo that is session
|
|
63
|
+
* creation plus a per-plugin MCP registration sweep.
|
|
64
|
+
*
|
|
65
|
+
* The warm is deliberately fire-and-forget: it can take seconds, and the
|
|
66
|
+
* callback still has a toast to answer and a menu to redraw. Telegram
|
|
67
|
+
* expires an unanswered callback query, so blocking here would trade a
|
|
68
|
+
* cold first turn for a visibly stuck button.
|
|
69
|
+
*/
|
|
70
|
+
function handOffBackendSession(
|
|
71
|
+
chatId: string,
|
|
72
|
+
previousBackend: ReturnType<typeof resolveBackendForChat>,
|
|
73
|
+
gateway: CallbackDeps["gateway"],
|
|
74
|
+
): void {
|
|
75
|
+
previousBackend?.sessions?.resetChat?.(chatId);
|
|
76
|
+
const nextBackend = resolveBackendForChat(chatId, gateway);
|
|
77
|
+
// Same instance on a no-op switch — warming it again is harmless
|
|
78
|
+
// (every step is idempotent) but pointless, so skip it.
|
|
79
|
+
if (!nextBackend || nextBackend === previousBackend) return;
|
|
80
|
+
void Promise.resolve(nextBackend.sessions?.warmSession?.(chatId)).catch(
|
|
81
|
+
(err) =>
|
|
82
|
+
logWarn(
|
|
83
|
+
"bot",
|
|
84
|
+
`[${chatId}] warm after backend switch failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
85
|
+
),
|
|
86
|
+
);
|
|
87
|
+
}
|
|
47
88
|
|
|
48
89
|
export async function handleModelCallback(
|
|
49
90
|
ctx: Context,
|
|
@@ -171,6 +212,10 @@ export async function handleModelCallback(
|
|
|
171
212
|
});
|
|
172
213
|
return;
|
|
173
214
|
}
|
|
215
|
+
// Resolve the outgoing backend BEFORE rebinding — afterwards this
|
|
216
|
+
// chat already points at the new one and the old in-process state
|
|
217
|
+
// would be unreachable.
|
|
218
|
+
const previousBackend = resolveBackendForChat(cid, gateway);
|
|
174
219
|
const result = await rebindChat(cid, action.backendId, config);
|
|
175
220
|
if (!result.ok) {
|
|
176
221
|
await answerCallbackQuerySafe(ctx, {
|
|
@@ -188,6 +233,7 @@ export async function handleModelCallback(
|
|
|
188
233
|
resetSession(cid);
|
|
189
234
|
clearHistory(cid);
|
|
190
235
|
resetPulseCheckpoint(cid);
|
|
236
|
+
handOffBackendSession(cid, previousBackend, gateway);
|
|
191
237
|
const label =
|
|
192
238
|
available.find((b) => b.id === action.backendId)?.label ??
|
|
193
239
|
action.backendId;
|
|
@@ -210,11 +256,13 @@ export async function handleModelCallback(
|
|
|
210
256
|
// global chat-role backend. Per-backend model picks are
|
|
211
257
|
// preserved (modelByBackend stays intact) so reverting and
|
|
212
258
|
// switching back later still restores prior choices.
|
|
259
|
+
const previousBackend = resolveBackendForChat(cid, gateway);
|
|
213
260
|
await releaseChat(cid);
|
|
214
261
|
setChatBackend(cid, undefined);
|
|
215
262
|
resetSession(cid);
|
|
216
263
|
clearHistory(cid);
|
|
217
264
|
resetPulseCheckpoint(cid);
|
|
265
|
+
handOffBackendSession(cid, previousBackend, gateway);
|
|
218
266
|
// Resolve the now-default backend's model for the toast.
|
|
219
267
|
const defaultBackend = resolveBackendForChat(cid, gateway);
|
|
220
268
|
const defaultBackendId = getBackendIdForChat(cid);
|
|
@@ -112,10 +112,16 @@ export function registerSettingsCommands(
|
|
|
112
112
|
if (be?.models?.resolveModelInfo) {
|
|
113
113
|
const resolution = await be.models?.resolveModelInfo(arg);
|
|
114
114
|
if (resolution.kind !== "exact") {
|
|
115
|
+
// Every backend's formatModelError returns plain text and
|
|
116
|
+
// interpolates the raw query, so escape at this boundary — the
|
|
117
|
+
// same fix as the model menu's status lines. `/model <name>`
|
|
118
|
+
// (which the OpenCode/Kilo hint literally tells you to type)
|
|
119
|
+
// otherwise reaches Telegram as an unsupported `<name>` tag and
|
|
120
|
+
// the whole reply 400s, leaving the command looking dead.
|
|
115
121
|
const msg =
|
|
116
122
|
be.models?.formatModelError?.(arg, resolution) ??
|
|
117
|
-
`No model matched "${
|
|
118
|
-
await ctx.reply(msg, { parse_mode: "HTML" });
|
|
123
|
+
`No model matched "${arg}".`;
|
|
124
|
+
await ctx.reply(escapeHtml(msg), { parse_mode: "HTML" });
|
|
119
125
|
return;
|
|
120
126
|
}
|
|
121
127
|
if (!resolution.model.selectable) {
|
|
@@ -370,9 +370,12 @@ export function renderModelMenuText(state: ModelMenuState): string {
|
|
|
370
370
|
`<i>Use the picker below to choose one — sending a message before picking will be refused.</i>`,
|
|
371
371
|
);
|
|
372
372
|
} else {
|
|
373
|
-
lines.push(`<b>Model:</b> <code>${state.activeDisplay}</code>`);
|
|
373
|
+
lines.push(`<b>Model:</b> <code>${escapeHtml(state.activeDisplay)}</code>`);
|
|
374
374
|
}
|
|
375
|
-
|
|
375
|
+
// Display names and status lines are plain text from backend catalogs —
|
|
376
|
+
// a literal `<name>` in a hint (or a `<` in a model id) is otherwise
|
|
377
|
+
// parsed as an HTML tag and Telegram rejects the whole send with 400.
|
|
378
|
+
for (const l of state.statusLines) lines.push(escapeHtml(l));
|
|
376
379
|
if (state.freeOnly && state.showFreeToggle) {
|
|
377
380
|
lines.push("<i>Filtering to free-tier models when browsing.</i>");
|
|
378
381
|
}
|