talon-agent 3.15.2 → 3.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/backend/claude-sdk/factory.ts +9 -0
- package/src/backend/claude-sdk/handler.ts +11 -0
- package/src/backend/claude-sdk/plan-usage.ts +170 -0
- package/src/backend/claude-sdk/stream.ts +13 -0
- package/src/backend/remote-server/events.ts +5 -0
- package/src/backend/shared/delivery.ts +37 -12
- package/src/backend/shared/index.ts +2 -0
- package/src/backend/shared/stream-state.ts +36 -0
- package/src/core/agent-runtime/capabilities.ts +31 -5
- package/src/core/agent-runtime/contract-tests.ts +5 -5
- package/src/core/engine/gateway-actions/models.ts +33 -0
- package/src/core/tools/models.ts +9 -0
- package/src/frontend/discord/commands/session.ts +12 -1
- package/src/frontend/discord/helpers.ts +1 -0
- package/src/frontend/shared/session-status.ts +17 -0
- package/src/frontend/shared/status-context.ts +62 -0
- package/src/frontend/telegram/commands/session.ts +12 -1
- package/src/frontend/telegram/helpers/format.ts +1 -0
- package/src/frontend/terminal/commands.ts +20 -0
package/package.json
CHANGED
|
@@ -22,7 +22,9 @@ import {
|
|
|
22
22
|
type SessionBackend,
|
|
23
23
|
type SystemControl,
|
|
24
24
|
type ToolRuntime,
|
|
25
|
+
type UsageTelemetry,
|
|
25
26
|
} from "../../core/agent-runtime/capabilities.js";
|
|
27
|
+
import { getPlanUsage } from "./plan-usage.js";
|
|
26
28
|
|
|
27
29
|
import {
|
|
28
30
|
initAgent as initClaudeAgent,
|
|
@@ -117,6 +119,12 @@ const claudeSdkFactory: BackendFactory = {
|
|
|
117
119
|
updateSystemPrompt: (prompt) => claudeUpdateSystemPrompt(prompt),
|
|
118
120
|
};
|
|
119
121
|
|
|
122
|
+
// No per-session snapshot to offer (each turn is a fresh subprocess),
|
|
123
|
+
// but the subscription's rate-limit windows are readable.
|
|
124
|
+
const usage: UsageTelemetry = {
|
|
125
|
+
getPlanUsage: () => getPlanUsage(),
|
|
126
|
+
};
|
|
127
|
+
|
|
120
128
|
const backend = composeBackend({
|
|
121
129
|
id: "claude",
|
|
122
130
|
label: "Anthropic",
|
|
@@ -126,6 +134,7 @@ const claudeSdkFactory: BackendFactory = {
|
|
|
126
134
|
models,
|
|
127
135
|
sessions,
|
|
128
136
|
tools,
|
|
137
|
+
usage,
|
|
129
138
|
control,
|
|
130
139
|
});
|
|
131
140
|
|
|
@@ -36,6 +36,7 @@ import { applyRetryDecisionStream } from "../shared/handle-retry.js";
|
|
|
36
36
|
import { getConfig } from "./state.js";
|
|
37
37
|
import { buildSdkOptions, getActiveFrontends } from "./options.js";
|
|
38
38
|
import { waitForMcpServersReady } from "./mcp-ready.js";
|
|
39
|
+
import { invalidatePlanUsage } from "./plan-usage.js";
|
|
39
40
|
import { frontendsForChat } from "../shared/frontends.js";
|
|
40
41
|
import {
|
|
41
42
|
createStreamState,
|
|
@@ -43,6 +44,7 @@ import {
|
|
|
43
44
|
isStreamEvent,
|
|
44
45
|
isAssistant,
|
|
45
46
|
isResult,
|
|
47
|
+
isRateLimitEvent,
|
|
46
48
|
isUserMessage,
|
|
47
49
|
extractToolResults,
|
|
48
50
|
processStreamDelta,
|
|
@@ -384,6 +386,14 @@ export async function* runChatTurn(
|
|
|
384
386
|
continue;
|
|
385
387
|
}
|
|
386
388
|
|
|
389
|
+
// The turn just moved the plan's usage — drop the cached windows so
|
|
390
|
+
// the next /status reads them again instead of showing pre-turn
|
|
391
|
+
// figures.
|
|
392
|
+
if (isRateLimitEvent(message)) {
|
|
393
|
+
invalidatePlanUsage();
|
|
394
|
+
continue;
|
|
395
|
+
}
|
|
396
|
+
|
|
387
397
|
if (isResult(message)) {
|
|
388
398
|
processResultMessage(message, state, options.model ?? activeModel);
|
|
389
399
|
armPostResultWatchdog();
|
|
@@ -515,6 +525,7 @@ export async function* runChatTurn(
|
|
|
515
525
|
contextTokens: state.contextTokens,
|
|
516
526
|
contextWindow: state.contextWindow,
|
|
517
527
|
numApiCalls: state.numApiCalls,
|
|
528
|
+
costUsd: state.costUsd,
|
|
518
529
|
});
|
|
519
530
|
|
|
520
531
|
// Set a descriptive session name from the first message.
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Claude.ai subscription rate-limit windows for /status.
|
|
3
|
+
*
|
|
4
|
+
* The 5-hour, weekly, and per-model utilisation percentages come from the
|
|
5
|
+
* same OAuth endpoint Claude Code's own usage panel reads. The Agent SDK
|
|
6
|
+
* also exposes them through a control request, but that path additionally
|
|
7
|
+
* builds a local-session behaviour report and costs seconds per call; the
|
|
8
|
+
* endpoint alone answers in well under a second.
|
|
9
|
+
*
|
|
10
|
+
* Everything degrades to `undefined`: no credentials, an API-key session
|
|
11
|
+
* (plan limits don't apply), or any transport failure. /status hides the
|
|
12
|
+
* section rather than rendering zeroes.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { readFile } from "node:fs/promises";
|
|
16
|
+
import { homedir } from "node:os";
|
|
17
|
+
import { join } from "node:path";
|
|
18
|
+
import { logWarn } from "../../util/log.js";
|
|
19
|
+
import type {
|
|
20
|
+
PlanUsage,
|
|
21
|
+
PlanWindow,
|
|
22
|
+
} from "../../core/agent-runtime/capabilities.js";
|
|
23
|
+
|
|
24
|
+
const USAGE_ENDPOINT = "https://api.anthropic.com/api/oauth/usage";
|
|
25
|
+
const REQUEST_TIMEOUT_MS = 5_000;
|
|
26
|
+
const CACHE_TTL_MS = 60_000;
|
|
27
|
+
|
|
28
|
+
let cache: { value: PlanUsage; fetchedAt: number } | undefined;
|
|
29
|
+
let inFlight: Promise<PlanUsage | undefined> | undefined;
|
|
30
|
+
|
|
31
|
+
function credentialsPath(): string {
|
|
32
|
+
const configDir = process.env.CLAUDE_CONFIG_DIR?.trim();
|
|
33
|
+
return join(
|
|
34
|
+
configDir && configDir.length > 0 ? configDir : join(homedir(), ".claude"),
|
|
35
|
+
".credentials.json",
|
|
36
|
+
);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
interface OAuthCredentials {
|
|
40
|
+
accessToken?: string;
|
|
41
|
+
subscriptionType?: string;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
async function readCredentials(): Promise<OAuthCredentials | undefined> {
|
|
45
|
+
try {
|
|
46
|
+
const parsed = JSON.parse(await readFile(credentialsPath(), "utf8")) as {
|
|
47
|
+
claudeAiOauth?: OAuthCredentials;
|
|
48
|
+
};
|
|
49
|
+
const oauth = parsed.claudeAiOauth;
|
|
50
|
+
return oauth?.accessToken ? oauth : undefined;
|
|
51
|
+
} catch {
|
|
52
|
+
return undefined;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
interface RawLimit {
|
|
57
|
+
kind?: string;
|
|
58
|
+
percent?: number;
|
|
59
|
+
resets_at?: string | null;
|
|
60
|
+
scope?: { model?: { display_name?: string | null } | null } | null;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Short display label for one window, or undefined to skip the row.
|
|
65
|
+
*
|
|
66
|
+
* Only the three documented kinds are rendered. The response also carries
|
|
67
|
+
* internal codenamed windows; skipping unknown kinds keeps those out of the
|
|
68
|
+
* user-facing panel.
|
|
69
|
+
*/
|
|
70
|
+
function windowLabel(limit: RawLimit): string | undefined {
|
|
71
|
+
if (limit.kind === "session") return "5h";
|
|
72
|
+
if (limit.kind === "weekly_all") return "7d";
|
|
73
|
+
if (limit.kind === "weekly_scoped") {
|
|
74
|
+
const model = limit.scope?.model?.display_name?.trim();
|
|
75
|
+
return model && model.length > 0 ? model : undefined;
|
|
76
|
+
}
|
|
77
|
+
return undefined;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export function parsePlanUsage(
|
|
81
|
+
body: unknown,
|
|
82
|
+
subscriptionType?: string,
|
|
83
|
+
): PlanUsage | undefined {
|
|
84
|
+
const limits = (body as { limits?: unknown } | null)?.limits;
|
|
85
|
+
if (!Array.isArray(limits)) return undefined;
|
|
86
|
+
|
|
87
|
+
const windows: PlanWindow[] = [];
|
|
88
|
+
for (const limit of limits as RawLimit[]) {
|
|
89
|
+
const label = windowLabel(limit);
|
|
90
|
+
if (!label) continue;
|
|
91
|
+
const raw = limit.percent;
|
|
92
|
+
const percent =
|
|
93
|
+
typeof raw === "number" && Number.isFinite(raw)
|
|
94
|
+
? Math.max(0, Math.min(100, Math.round(raw)))
|
|
95
|
+
: 0;
|
|
96
|
+
windows.push({
|
|
97
|
+
label,
|
|
98
|
+
percent,
|
|
99
|
+
...(typeof limit.resets_at === "string"
|
|
100
|
+
? { resetsAt: limit.resets_at }
|
|
101
|
+
: {}),
|
|
102
|
+
});
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
if (windows.length === 0) return undefined;
|
|
106
|
+
return {
|
|
107
|
+
...(subscriptionType ? { plan: subscriptionType } : {}),
|
|
108
|
+
windows,
|
|
109
|
+
fetchedAt: Date.now(),
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async function load(): Promise<PlanUsage | undefined> {
|
|
114
|
+
const creds = await readCredentials();
|
|
115
|
+
if (!creds?.accessToken) return undefined;
|
|
116
|
+
|
|
117
|
+
try {
|
|
118
|
+
const res = await fetch(USAGE_ENDPOINT, {
|
|
119
|
+
headers: {
|
|
120
|
+
Authorization: `Bearer ${creds.accessToken}`,
|
|
121
|
+
"anthropic-beta": "oauth-2025-04-20",
|
|
122
|
+
},
|
|
123
|
+
signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
|
|
124
|
+
});
|
|
125
|
+
if (!res.ok) {
|
|
126
|
+
logWarn("agent", `plan usage: endpoint returned ${res.status}`);
|
|
127
|
+
return undefined;
|
|
128
|
+
}
|
|
129
|
+
return parsePlanUsage(await res.json(), creds.subscriptionType);
|
|
130
|
+
} catch (err) {
|
|
131
|
+
logWarn(
|
|
132
|
+
"agent",
|
|
133
|
+
`plan usage: ${err instanceof Error ? err.message : String(err)}`,
|
|
134
|
+
);
|
|
135
|
+
return undefined;
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Plan windows for /status, cached for a minute so a burst of commands
|
|
141
|
+
* makes one request. A failed refresh falls back to the last known values
|
|
142
|
+
* — `fetchedAt` lets the caller age them.
|
|
143
|
+
*/
|
|
144
|
+
export async function getPlanUsage(): Promise<PlanUsage | undefined> {
|
|
145
|
+
// An API-key session bills against the key, not the subscription, so the
|
|
146
|
+
// stored OAuth credentials would describe limits that don't apply here.
|
|
147
|
+
if (process.env.ANTHROPIC_API_KEY) return undefined;
|
|
148
|
+
|
|
149
|
+
if (cache && Date.now() - cache.fetchedAt < CACHE_TTL_MS) return cache.value;
|
|
150
|
+
|
|
151
|
+
inFlight ??= load().finally(() => {
|
|
152
|
+
inFlight = undefined;
|
|
153
|
+
});
|
|
154
|
+
const loaded = await inFlight;
|
|
155
|
+
if (loaded) cache = { value: loaded, fetchedAt: loaded.fetchedAt };
|
|
156
|
+
return loaded ?? cache?.value;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Expire the cache after the SDK reports a rate-limit change, so the next
|
|
161
|
+
* /status re-reads instead of showing figures from before the turn.
|
|
162
|
+
*/
|
|
163
|
+
export function invalidatePlanUsage(): void {
|
|
164
|
+
if (cache) cache.fetchedAt = 0;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export function resetPlanUsageForTest(): void {
|
|
168
|
+
cache = undefined;
|
|
169
|
+
inFlight = undefined;
|
|
170
|
+
}
|
|
@@ -39,6 +39,8 @@ export type StreamState = {
|
|
|
39
39
|
sdkOutputTokens: number;
|
|
40
40
|
sdkCacheRead: number;
|
|
41
41
|
sdkCacheWrite: number;
|
|
42
|
+
/** Cost of this turn in USD, as reported by the result message. */
|
|
43
|
+
costUsd: number;
|
|
42
44
|
/**
|
|
43
45
|
* Per-turn cache behaviour derived from the result message's per-request
|
|
44
46
|
* `usage.iterations`. Undefined when the provider reported none — the
|
|
@@ -113,6 +115,7 @@ export function createStreamState(): StreamState {
|
|
|
113
115
|
sdkOutputTokens: 0,
|
|
114
116
|
sdkCacheRead: 0,
|
|
115
117
|
sdkCacheWrite: 0,
|
|
118
|
+
costUsd: 0,
|
|
116
119
|
cacheStats: undefined,
|
|
117
120
|
lastStreamUpdate: 0,
|
|
118
121
|
lastTrailingText: "",
|
|
@@ -148,6 +151,11 @@ export function isResult(msg: SDKMessage): msg is SDKResultMessage {
|
|
|
148
151
|
return msg.type === "result";
|
|
149
152
|
}
|
|
150
153
|
|
|
154
|
+
/** Emitted when the subscription's rate-limit state changes mid-turn. */
|
|
155
|
+
export function isRateLimitEvent(msg: SDKMessage): boolean {
|
|
156
|
+
return msg.type === "rate_limit_event";
|
|
157
|
+
}
|
|
158
|
+
|
|
151
159
|
// ── Message processors ──────────────────────────────────────────────────────
|
|
152
160
|
|
|
153
161
|
/** Output of `processStreamDelta` when the throttle interval has elapsed. */
|
|
@@ -337,6 +345,11 @@ export function processResultMessage(
|
|
|
337
345
|
): void {
|
|
338
346
|
state.numApiCalls = msg.num_turns ?? 0;
|
|
339
347
|
|
|
348
|
+
// Per-turn, not cumulative across a resumed session — safe to add up.
|
|
349
|
+
if (typeof msg.total_cost_usd === "number" && msg.total_cost_usd > 0) {
|
|
350
|
+
state.costUsd = msg.total_cost_usd;
|
|
351
|
+
}
|
|
352
|
+
|
|
340
353
|
// Context fill from last API iteration
|
|
341
354
|
const usage = msg.usage;
|
|
342
355
|
if (usage && Array.isArray(usage.iterations) && usage.iterations.length > 0) {
|
|
@@ -28,6 +28,7 @@
|
|
|
28
28
|
import {
|
|
29
29
|
appendText,
|
|
30
30
|
closeCurrentSegment,
|
|
31
|
+
markProgressDelivered,
|
|
31
32
|
recordTokens,
|
|
32
33
|
recordToolUse,
|
|
33
34
|
recordToolCall,
|
|
@@ -279,6 +280,10 @@ async function processPartUpdate(
|
|
|
279
280
|
if (progress && ctx.onTextBlock) {
|
|
280
281
|
try {
|
|
281
282
|
await ctx.onTextBlock(progress);
|
|
283
|
+
// Only now is this text actually with the user. `closeCurrentSegment`
|
|
284
|
+
// folded it into `allResponseText`, so without this the end-of-turn
|
|
285
|
+
// delivery would ship it a second time inside the full transcript.
|
|
286
|
+
markProgressDelivered(ctx.state);
|
|
282
287
|
} catch (err) {
|
|
283
288
|
// Non-fatal — never break the stream loop on a UI callback. Logged
|
|
284
289
|
// at debug level so a repeatedly-failing frontend is visible to
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Unified delivery routing for the remote-server backend family.
|
|
3
3
|
*
|
|
4
|
-
* Both OpenCode and Kilo can reach the user through one of
|
|
4
|
+
* Both OpenCode and Kilo can reach the user through one of several paths
|
|
5
5
|
* after a turn completes. The decision tree below is identical between
|
|
6
6
|
* them, so it lives here as a shared helper instead of being duplicated
|
|
7
7
|
* in each handler.
|
|
@@ -19,7 +19,12 @@
|
|
|
19
19
|
* marker, or analogous failure path). Surface as a Talon error
|
|
20
20
|
* instead of shipping the raw upstream string as a reply.
|
|
21
21
|
* - `text-part` — model emitted plain assistant text and didn't call
|
|
22
|
-
* a delivery tool. Ship
|
|
22
|
+
* a delivery tool. Ship the part the user hasn't already seen through
|
|
23
|
+
* `onTextBlock` (see `progress` below).
|
|
24
|
+
* - `progress` — every line of the reply already reached the user as a
|
|
25
|
+
* mid-turn progress message, so there is nothing left to send. Recorded
|
|
26
|
+
* distinctly rather than as `empty`, which means "the model said
|
|
27
|
+
* nothing at all".
|
|
23
28
|
* - `empty` — no tool, no text, no synthetic-error. Surface a concise
|
|
24
29
|
* notice so the user isn't left staring at silence.
|
|
25
30
|
*
|
|
@@ -29,11 +34,13 @@
|
|
|
29
34
|
*/
|
|
30
35
|
|
|
31
36
|
import type { StreamState } from "./stream-state.js";
|
|
37
|
+
import { undeliveredResponseText } from "./stream-state.js";
|
|
32
38
|
import { logWarn } from "../../util/log.js";
|
|
33
39
|
import { incrementCounter } from "../../storage/metrics.js";
|
|
34
40
|
|
|
35
41
|
/** Route the delivery decision selected. */
|
|
36
|
-
export type DeliveryRoute =
|
|
42
|
+
export type DeliveryRoute =
|
|
43
|
+
"tool" | "text-part" | "progress" | "synthetic-error" | "empty";
|
|
37
44
|
|
|
38
45
|
export class TextBlockDeliveryError extends Error {
|
|
39
46
|
readonly route: DeliveryRoute;
|
|
@@ -163,23 +170,41 @@ export async function routeDelivery(
|
|
|
163
170
|
};
|
|
164
171
|
}
|
|
165
172
|
|
|
166
|
-
// Route 3 — plain text part. Ship
|
|
167
|
-
|
|
173
|
+
// Route 3 — plain text part. Ship only what the user hasn't already seen.
|
|
174
|
+
//
|
|
175
|
+
// The remote-server backends flush each pre-tool segment through
|
|
176
|
+
// `onTextBlock` as a progress message, and `closeCurrentSegment` also folds
|
|
177
|
+
// that segment into `allResponseText`. Shipping `responseText` wholesale
|
|
178
|
+
// therefore re-sent every narration line a second time, concatenated — the
|
|
179
|
+
// doubled-message symptom on a tool-heavy OpenCode/Kilo turn.
|
|
180
|
+
//
|
|
181
|
+
// The subtraction is gated on a progress send having actually happened.
|
|
182
|
+
// `responseText` is an explicit input and callers are not required to derive
|
|
183
|
+
// it from `state` (several pass a literal), so reading the remainder off
|
|
184
|
+
// `allResponseText` unconditionally would silently drop their reply.
|
|
185
|
+
const pending =
|
|
186
|
+
state.progressDeliveredLen > 0
|
|
187
|
+
? undeliveredResponseText(state)
|
|
188
|
+
: responseText;
|
|
189
|
+
if (pending && !state.turnTerminated) {
|
|
168
190
|
if (onTextBlock) {
|
|
169
191
|
try {
|
|
170
|
-
await onTextBlock(
|
|
192
|
+
await onTextBlock(pending);
|
|
171
193
|
} catch (err) {
|
|
172
194
|
logWarn("agent", `[${chatId}] onTextBlock failed: ${errMsg(err)}`);
|
|
173
195
|
if (propagateDeliveryFailure) {
|
|
174
|
-
throw new TextBlockDeliveryError(
|
|
175
|
-
"text-part",
|
|
176
|
-
responseText.length,
|
|
177
|
-
err,
|
|
178
|
-
);
|
|
196
|
+
throw new TextBlockDeliveryError("text-part", pending.length, err);
|
|
179
197
|
}
|
|
180
198
|
}
|
|
181
199
|
}
|
|
182
|
-
return { route: "text-part", chars:
|
|
200
|
+
return { route: "text-part", chars: pending.length };
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// Route 3b — the whole reply already reached the user as progress messages.
|
|
204
|
+
// Nothing left to send; recorded distinctly so the turn isn't misfiled as an
|
|
205
|
+
// empty completion in the logs or the empty-turn counter.
|
|
206
|
+
if (responseText && !state.turnTerminated) {
|
|
207
|
+
return { route: "progress", chars: responseText.length };
|
|
183
208
|
}
|
|
184
209
|
|
|
185
210
|
// Route 4 — empty turn. The model produced no text (and didn't end_turn):
|
|
@@ -44,6 +44,21 @@ export interface StreamState {
|
|
|
44
44
|
allResponseText: string;
|
|
45
45
|
/** Text *after* the last tool call (or the entire response if no tools). */
|
|
46
46
|
lastTrailingText: string;
|
|
47
|
+
/**
|
|
48
|
+
* How much of `allResponseText` has already reached the user as a
|
|
49
|
+
* mid-turn progress message.
|
|
50
|
+
*
|
|
51
|
+
* The remote-server backends flush the pending segment through
|
|
52
|
+
* `onTextBlock` at each tool boundary, but `closeCurrentSegment` also
|
|
53
|
+
* folds that segment into `allResponseText` — so the end-of-turn
|
|
54
|
+
* delivery would ship every narration line a second time, concatenated.
|
|
55
|
+
* Delivery ships only `allResponseText.slice(progressDeliveredLen)`.
|
|
56
|
+
*
|
|
57
|
+
* Advanced only after a progress send actually succeeds, so a flush that
|
|
58
|
+
* throws (e.g. Telegram's 4096-char limit) leaves its text pending and
|
|
59
|
+
* the end-of-turn delivery still carries it.
|
|
60
|
+
*/
|
|
61
|
+
progressDeliveredLen: number;
|
|
47
62
|
|
|
48
63
|
// ── Session ───────────────────────────────────────────────────────────────
|
|
49
64
|
/** Provider-assigned session id, if one was announced during the stream. */
|
|
@@ -133,6 +148,7 @@ export function createStreamState(chatId?: string): StreamState {
|
|
|
133
148
|
currentBlockText: "",
|
|
134
149
|
allResponseText: "",
|
|
135
150
|
lastTrailingText: "",
|
|
151
|
+
progressDeliveredLen: 0,
|
|
136
152
|
newSessionId: undefined,
|
|
137
153
|
toolCalls: 0,
|
|
138
154
|
turnTerminated: false,
|
|
@@ -279,3 +295,23 @@ export function finalizeResponseText(state: StreamState): string {
|
|
|
279
295
|
}
|
|
280
296
|
return state.allResponseText.trim();
|
|
281
297
|
}
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* Record that everything accumulated so far has been shipped to the user
|
|
301
|
+
* as a progress message. Call only after the send succeeds.
|
|
302
|
+
*/
|
|
303
|
+
export function markProgressDelivered(state: StreamState): void {
|
|
304
|
+
state.progressDeliveredLen = state.allResponseText.length;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* The portion of the turn's text that has NOT already been shipped as a
|
|
309
|
+
* mid-turn progress message — i.e. what end-of-turn delivery still owes
|
|
310
|
+
* the user.
|
|
311
|
+
*
|
|
312
|
+
* Backends that never flush progress leave `progressDeliveredLen` at 0,
|
|
313
|
+
* so this is the whole response and their behaviour is unchanged.
|
|
314
|
+
*/
|
|
315
|
+
export function undeliveredResponseText(state: StreamState): string {
|
|
316
|
+
return state.allResponseText.slice(state.progressDeliveredLen).trim();
|
|
317
|
+
}
|
|
@@ -199,10 +199,11 @@ export interface ToolRuntime {
|
|
|
199
199
|
}
|
|
200
200
|
|
|
201
201
|
/**
|
|
202
|
-
* `/status` enrichment.
|
|
203
|
-
*
|
|
204
|
-
*
|
|
205
|
-
*
|
|
202
|
+
* `/status` enrichment. Both members are optional — a backend
|
|
203
|
+
* implements whichever it can answer. Backends that track per-session
|
|
204
|
+
* usage (Codex, OpenAI Agents) supply `getSessionSnapshot`; a backend
|
|
205
|
+
* with no per-session model (Claude SDK on a fresh subprocess per
|
|
206
|
+
* turn) omits it and may still report plan limits.
|
|
206
207
|
*
|
|
207
208
|
* The snapshot's `contextModelId` carries the resolved-this-turn
|
|
208
209
|
* model id when the SDK can surface it. Frontend `/status` reads
|
|
@@ -211,7 +212,7 @@ export interface ToolRuntime {
|
|
|
211
212
|
* ChatGPT-OAuth).
|
|
212
213
|
*/
|
|
213
214
|
export interface UsageTelemetry {
|
|
214
|
-
getSessionSnapshot(sessionId: string): Promise<
|
|
215
|
+
getSessionSnapshot?(sessionId: string): Promise<
|
|
215
216
|
| {
|
|
216
217
|
inputTokens?: number;
|
|
217
218
|
outputTokens?: number;
|
|
@@ -221,6 +222,31 @@ export interface UsageTelemetry {
|
|
|
221
222
|
}
|
|
222
223
|
| undefined
|
|
223
224
|
>;
|
|
225
|
+
/**
|
|
226
|
+
* Subscription rate-limit windows. Account-level rather than
|
|
227
|
+
* per-chat: `/status` renders whichever backend can answer, even
|
|
228
|
+
* when another one is serving the chat. Absent on backends with no
|
|
229
|
+
* plan concept; resolves `undefined` when the data can't be read.
|
|
230
|
+
*/
|
|
231
|
+
getPlanUsage?(): Promise<PlanUsage | undefined>;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/** One subscription rate-limit window, as `/status` renders it. */
|
|
235
|
+
export interface PlanWindow {
|
|
236
|
+
/** Short label — `5h`, `7d`, or the scoped model's display name. */
|
|
237
|
+
label: string;
|
|
238
|
+
/** Window utilisation, 0-100. */
|
|
239
|
+
percent: number;
|
|
240
|
+
/** ISO timestamp of the next reset, when the plan reports one. */
|
|
241
|
+
resetsAt?: string;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
export interface PlanUsage {
|
|
245
|
+
/** Subscription tier (`max`, `pro`, …) when known. */
|
|
246
|
+
plan?: string;
|
|
247
|
+
windows: PlanWindow[];
|
|
248
|
+
/** Epoch ms of the read, so renderers can flag figures as aged. */
|
|
249
|
+
fetchedAt: number;
|
|
224
250
|
}
|
|
225
251
|
|
|
226
252
|
/**
|
|
@@ -272,16 +272,16 @@ export async function assertModelCatalogDefaultShape(
|
|
|
272
272
|
}
|
|
273
273
|
|
|
274
274
|
/**
|
|
275
|
-
* Backends that
|
|
276
|
-
*
|
|
277
|
-
*
|
|
278
|
-
*
|
|
275
|
+
* Backends that implement `getSessionSnapshot` must answer it with
|
|
276
|
+
* either a `UsageSnapshot` object whose counters are non-negative
|
|
277
|
+
* numbers, or `undefined`. Negative counters or `NaN`s are contract
|
|
278
|
+
* violations. Backends that only report plan limits are skipped.
|
|
279
279
|
*/
|
|
280
280
|
export async function assertUsageTelemetryShape(
|
|
281
281
|
backend: Backend,
|
|
282
282
|
sessionId = "contract-test-session",
|
|
283
283
|
): Promise<void> {
|
|
284
|
-
if (!backend.usage) return;
|
|
284
|
+
if (!backend.usage?.getSessionSnapshot) return;
|
|
285
285
|
const snapshot = await backend.usage.getSessionSnapshot(sessionId);
|
|
286
286
|
if (snapshot === undefined) return;
|
|
287
287
|
const fields: (keyof typeof snapshot)[] = [
|
|
@@ -98,6 +98,39 @@ export const modelHandlers: SharedActionHandlers = {
|
|
|
98
98
|
}
|
|
99
99
|
},
|
|
100
100
|
|
|
101
|
+
// Account-level, so it falls back to the pooled Claude backend when
|
|
102
|
+
// another provider is serving this chat.
|
|
103
|
+
plan_usage: async (_body, chatId) => {
|
|
104
|
+
const current = getPooledBackend(getBackendIdForChat(String(chatId)));
|
|
105
|
+
const source = current?.usage?.getPlanUsage
|
|
106
|
+
? current
|
|
107
|
+
: getPooledBackend("claude");
|
|
108
|
+
if (!source?.usage?.getPlanUsage)
|
|
109
|
+
return {
|
|
110
|
+
ok: false,
|
|
111
|
+
error: "No configured backend reports subscription rate limits.",
|
|
112
|
+
};
|
|
113
|
+
|
|
114
|
+
const usage = await source.usage.getPlanUsage();
|
|
115
|
+
if (!usage)
|
|
116
|
+
return {
|
|
117
|
+
ok: false,
|
|
118
|
+
error:
|
|
119
|
+
"Plan usage is unavailable — no subscription credentials, or this session authenticates with an API key.",
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
const lines = usage.windows.map(
|
|
123
|
+
(w) =>
|
|
124
|
+
`- ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
|
|
125
|
+
);
|
|
126
|
+
return {
|
|
127
|
+
ok: true,
|
|
128
|
+
plan: usage.plan ?? null,
|
|
129
|
+
windows: usage.windows,
|
|
130
|
+
text: `Plan usage${usage.plan ? ` (${usage.plan})` : ""}:\n${lines.join("\n")}`,
|
|
131
|
+
};
|
|
132
|
+
},
|
|
133
|
+
|
|
101
134
|
list_backends: (body, chatId) => {
|
|
102
135
|
const currentId = getBackendIdForChat(String(chatId));
|
|
103
136
|
const backends = getAvailableBackends().map((b) => ({
|
package/src/core/tools/models.ts
CHANGED
|
@@ -28,6 +28,15 @@ export const modelTools: ToolDefinition[] = [
|
|
|
28
28
|
tag: "models",
|
|
29
29
|
},
|
|
30
30
|
|
|
31
|
+
{
|
|
32
|
+
name: "plan_usage",
|
|
33
|
+
description:
|
|
34
|
+
"Read your own subscription usage: how much of the 5-hour, weekly, and per-model rate-limit windows is spent, and when each resets. Use it before starting long or expensive work, or when deciding whether to defer something. Only answers on a subscription-backed Anthropic backend; other providers report no plan limits.",
|
|
35
|
+
schema: {},
|
|
36
|
+
execute: (_params, bridge) => bridge("plan_usage", {}),
|
|
37
|
+
tag: "models",
|
|
38
|
+
},
|
|
39
|
+
|
|
31
40
|
{
|
|
32
41
|
name: "list_backends",
|
|
33
42
|
description:
|
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
formatDuration,
|
|
14
14
|
formatTokenCount,
|
|
15
15
|
formatBytes,
|
|
16
|
+
formatUsd,
|
|
16
17
|
} from "../helpers.js";
|
|
17
18
|
import {
|
|
18
19
|
getBackendIdForChat,
|
|
@@ -78,8 +79,18 @@ export async function handleStatus(
|
|
|
78
79
|
` Read ${formatTokenCount(s.cache.read)}${s.cache.showsWrite ? ` Write ${formatTokenCount(s.cache.write)}` : ""}`,
|
|
79
80
|
]
|
|
80
81
|
: []),
|
|
81
|
-
` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}`,
|
|
82
|
+
` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}${s.costUsd > 0 ? ` Cost ${formatUsd(s.costUsd)}` : ""}`,
|
|
82
83
|
"",
|
|
84
|
+
...(s.plan
|
|
85
|
+
? [
|
|
86
|
+
`**Plan**${s.plan.plan ? ` ${s.plan.plan}` : ""}${s.plan.ageLabel ? ` *(${s.plan.ageLabel})*` : ""}`,
|
|
87
|
+
...s.plan.windows.map(
|
|
88
|
+
(w) =>
|
|
89
|
+
` \`${w.label.padEnd(6)}${w.bar} ${String(w.percent).padStart(3)}%\`${w.resetLabel ? ` reset ${w.resetLabel}` : ""}`,
|
|
90
|
+
),
|
|
91
|
+
"",
|
|
92
|
+
]
|
|
93
|
+
: []),
|
|
83
94
|
`**Pulse** ${s.pulseOn ? "on" : "off"}`,
|
|
84
95
|
`**Workspace** ${formatBytes(s.diskBytes)}`,
|
|
85
96
|
`**Session** ${s.sessionName ? `"${s.sessionName}" ` : ""}${s.sessionId ? "`" + s.sessionId.slice(0, 8) + "...`" : "_(new)_"} · ${s.sessionAge} old`,
|
|
@@ -23,11 +23,14 @@ import { isPulseEnabled } from "../../core/background/pulse.js";
|
|
|
23
23
|
import { getWorkspaceDiskUsage } from "../../util/workspace.js";
|
|
24
24
|
import { appendDailyLog } from "../../storage/daily-log.js";
|
|
25
25
|
import { resolveActiveModelForChat } from "../../core/models/active-model.js";
|
|
26
|
+
import { getPooledBackend } from "../../core/engine/backend-controller/index.js";
|
|
26
27
|
import {
|
|
27
28
|
buildCacheDisplay,
|
|
28
29
|
buildContextDisplay,
|
|
30
|
+
buildPlanDisplay,
|
|
29
31
|
type CacheDisplay,
|
|
30
32
|
type ContextDisplay,
|
|
33
|
+
type PlanDisplay,
|
|
31
34
|
} from "./status-context.js";
|
|
32
35
|
import { formatDuration } from "./format.js";
|
|
33
36
|
|
|
@@ -70,8 +73,10 @@ export interface SessionStatusData {
|
|
|
70
73
|
pulseOn: boolean;
|
|
71
74
|
context: ContextDisplay;
|
|
72
75
|
cache: CacheDisplay | null;
|
|
76
|
+
plan: PlanDisplay | null;
|
|
73
77
|
inputTokens: number;
|
|
74
78
|
outputTokens: number;
|
|
79
|
+
costUsd: number;
|
|
75
80
|
turns: number;
|
|
76
81
|
turnsModelLabel: string | undefined;
|
|
77
82
|
lastResponseMs: number;
|
|
@@ -157,6 +162,16 @@ export async function collectSessionStatus(
|
|
|
157
162
|
contextWindow: ctxMax,
|
|
158
163
|
});
|
|
159
164
|
|
|
165
|
+
// Plan limits belong to the account, not the chat: fall back to the pooled
|
|
166
|
+
// Claude backend so the section still shows while another provider serves
|
|
167
|
+
// this chat.
|
|
168
|
+
const planSource = backend?.usage?.getPlanUsage
|
|
169
|
+
? backend
|
|
170
|
+
: getPooledBackend("claude");
|
|
171
|
+
const plan = buildPlanDisplay(
|
|
172
|
+
await planSource?.usage?.getPlanUsage?.().catch(() => undefined),
|
|
173
|
+
);
|
|
174
|
+
|
|
160
175
|
const avgResponseMs =
|
|
161
176
|
info.turns > 0 && u.totalResponseMs
|
|
162
177
|
? Math.round(u.totalResponseMs / info.turns)
|
|
@@ -170,8 +185,10 @@ export async function collectSessionStatus(
|
|
|
170
185
|
pulseOn: isPulseEnabled(chatId),
|
|
171
186
|
context,
|
|
172
187
|
cache,
|
|
188
|
+
plan,
|
|
173
189
|
inputTokens,
|
|
174
190
|
outputTokens,
|
|
191
|
+
costUsd: u.estimatedCostUsd,
|
|
175
192
|
turns: info.turns,
|
|
176
193
|
turnsModelLabel,
|
|
177
194
|
lastResponseMs: u.lastResponseMs || 0,
|
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import type { CacheMetricsSupport } from "../../core/types.js";
|
|
2
|
+
import type { PlanUsage } from "../../core/agent-runtime/capabilities.js";
|
|
3
|
+
import { formatSmartTimestamp, formatRelativeAge } from "../../util/time.js";
|
|
2
4
|
|
|
3
5
|
// ── /context breakdown ────────────────────────────────────────────────────────
|
|
4
6
|
|
|
@@ -284,3 +286,63 @@ export function buildCacheDisplay(input: {
|
|
|
284
286
|
showsWrite: mode === "readwrite",
|
|
285
287
|
};
|
|
286
288
|
}
|
|
289
|
+
|
|
290
|
+
// ── Plan limits ─────────────────────────────────────────────────────────────
|
|
291
|
+
|
|
292
|
+
/** Figures older than this are labelled with their age in /status. */
|
|
293
|
+
const PLAN_STALE_AFTER_MS = 5 * 60_000;
|
|
294
|
+
|
|
295
|
+
export interface PlanWindowDisplay {
|
|
296
|
+
label: string;
|
|
297
|
+
percent: number;
|
|
298
|
+
bar: string;
|
|
299
|
+
/** Local-time reset, absent for windows the plan reports no reset for. */
|
|
300
|
+
resetLabel: string | undefined;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
export interface PlanDisplay {
|
|
304
|
+
plan: string | undefined;
|
|
305
|
+
windows: PlanWindowDisplay[];
|
|
306
|
+
/** Set only once the figures have aged, e.g. "12m ago". */
|
|
307
|
+
ageLabel: string | undefined;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
function planResetLabel(iso: string | undefined): string | undefined {
|
|
311
|
+
if (!iso) return undefined;
|
|
312
|
+
const ts = Date.parse(iso);
|
|
313
|
+
if (!Number.isFinite(ts)) return undefined;
|
|
314
|
+
// Windows are reported a second short of the boundary (20:59:59); round to
|
|
315
|
+
// the minute so the panel reads 21:00, as the plan's own dashboards do.
|
|
316
|
+
return formatSmartTimestamp(Math.round(ts / 60_000) * 60_000);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Lay out the subscription's rate-limit windows for /status. Returns null
|
|
321
|
+
* when the backend has nothing to report, so the section disappears rather
|
|
322
|
+
* than rendering empty bars.
|
|
323
|
+
*/
|
|
324
|
+
export function buildPlanDisplay(
|
|
325
|
+
usage: PlanUsage | undefined,
|
|
326
|
+
barLen = 20,
|
|
327
|
+
): PlanDisplay | null {
|
|
328
|
+
if (!usage || usage.windows.length === 0) return null;
|
|
329
|
+
|
|
330
|
+
const age = Date.now() - usage.fetchedAt;
|
|
331
|
+
return {
|
|
332
|
+
plan: usage.plan,
|
|
333
|
+
ageLabel:
|
|
334
|
+
age > PLAN_STALE_AFTER_MS
|
|
335
|
+
? formatRelativeAge(usage.fetchedAt)
|
|
336
|
+
: undefined,
|
|
337
|
+
windows: usage.windows.map((w) => {
|
|
338
|
+
const percent = Math.max(0, Math.min(100, Math.round(w.percent)));
|
|
339
|
+
const filled = Math.round((percent / 100) * barLen);
|
|
340
|
+
return {
|
|
341
|
+
label: w.label,
|
|
342
|
+
percent,
|
|
343
|
+
bar: "█".repeat(filled) + "░".repeat(barLen - filled),
|
|
344
|
+
resetLabel: planResetLabel(w.resetsAt),
|
|
345
|
+
};
|
|
346
|
+
}),
|
|
347
|
+
};
|
|
348
|
+
}
|
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
formatDuration,
|
|
13
13
|
formatTokenCount,
|
|
14
14
|
formatBytes,
|
|
15
|
+
formatUsd,
|
|
15
16
|
} from "../helpers/index.js";
|
|
16
17
|
import { resolveBackendForChat } from "../model-menu.js";
|
|
17
18
|
import { getBackendIdForChat } from "../../../core/engine/backend-controller/index.js";
|
|
@@ -65,8 +66,18 @@ export function registerSessionCommands(
|
|
|
65
66
|
` Read ${formatTokenCount(s.cache.read)}${s.cache.showsWrite ? ` Write ${formatTokenCount(s.cache.write)}` : ""}`,
|
|
66
67
|
]
|
|
67
68
|
: []),
|
|
68
|
-
` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}`,
|
|
69
|
+
` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}${s.costUsd > 0 ? ` Cost ${formatUsd(s.costUsd)}` : ""}`,
|
|
69
70
|
"",
|
|
71
|
+
...(s.plan
|
|
72
|
+
? [
|
|
73
|
+
`<b>Plan</b>${s.plan.plan ? ` ${escapeHtml(s.plan.plan)}` : ""}${s.plan.ageLabel ? ` <i>(${s.plan.ageLabel})</i>` : ""}`,
|
|
74
|
+
...s.plan.windows.map(
|
|
75
|
+
(w) =>
|
|
76
|
+
` <code>${escapeHtml(w.label.padEnd(6))}${w.bar} ${String(w.percent).padStart(3)}%</code>${w.resetLabel ? ` reset ${w.resetLabel}` : ""}`,
|
|
77
|
+
),
|
|
78
|
+
"",
|
|
79
|
+
]
|
|
80
|
+
: []),
|
|
70
81
|
`<b>Pulse</b> ${s.pulseOn ? "on" : "off"}`,
|
|
71
82
|
`<b>Workspace</b> ${formatBytes(s.diskBytes)}`,
|
|
72
83
|
`<b>Session</b> ${s.sessionName ? `"${escapeHtml(s.sessionName)}" ` : ""}${s.sessionId ? "<code>" + escapeHtml(s.sessionId.slice(0, 8)) + "...</code>" : "<i>(new)</i>"} · ${s.sessionAge} old`,
|
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
import {
|
|
19
19
|
buildCacheDisplay,
|
|
20
20
|
buildContextDisplay,
|
|
21
|
+
buildPlanDisplay,
|
|
21
22
|
buildContextBreakdown,
|
|
22
23
|
estimateContextTokens,
|
|
23
24
|
apportionCells,
|
|
@@ -43,6 +44,7 @@ import {
|
|
|
43
44
|
setSessionName,
|
|
44
45
|
} from "../../storage/sessions.js";
|
|
45
46
|
import { getLoadedPlugins } from "../../core/plugin/index.js";
|
|
47
|
+
import { getPooledBackend } from "../../core/engine/backend-controller/index.js";
|
|
46
48
|
|
|
47
49
|
// ── Types ────────────────────────────────────────────────────────────────────
|
|
48
50
|
|
|
@@ -430,6 +432,24 @@ export function registerBuiltinCommands(): void {
|
|
|
430
432
|
` ${pc.dim(`estimated session cost ${formatUsd(u.estimatedCostUsd)}`)}`,
|
|
431
433
|
);
|
|
432
434
|
}
|
|
435
|
+
|
|
436
|
+
const planSource = be?.usage?.getPlanUsage
|
|
437
|
+
? be
|
|
438
|
+
: getPooledBackend("claude");
|
|
439
|
+
const plan = buildPlanDisplay(
|
|
440
|
+
await planSource?.usage?.getPlanUsage?.().catch(() => undefined),
|
|
441
|
+
);
|
|
442
|
+
if (plan) {
|
|
443
|
+
ctx.renderer.writeln();
|
|
444
|
+
ctx.renderer.writeln(
|
|
445
|
+
` ${pc.bold("Plan")}${plan.plan ? ` ${plan.plan}` : ""}${plan.ageLabel ? pc.dim(` (${plan.ageLabel})`) : ""}`,
|
|
446
|
+
);
|
|
447
|
+
for (const w of plan.windows) {
|
|
448
|
+
ctx.renderer.writeln(
|
|
449
|
+
` ${w.label.padEnd(6)}${pc.dim(w.bar)} ${String(w.percent).padStart(3)}%${w.resetLabel ? pc.dim(` reset ${w.resetLabel}`) : ""}`,
|
|
450
|
+
);
|
|
451
|
+
}
|
|
452
|
+
}
|
|
433
453
|
if (backendModelLine) {
|
|
434
454
|
ctx.renderer.writeln();
|
|
435
455
|
ctx.renderer.writeln(backendModelLine);
|