talon-agent 3.15.2 → 3.15.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.15.2",
3
+ "version": "3.15.3",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -22,7 +22,9 @@ import {
22
22
  type SessionBackend,
23
23
  type SystemControl,
24
24
  type ToolRuntime,
25
+ type UsageTelemetry,
25
26
  } from "../../core/agent-runtime/capabilities.js";
27
+ import { getPlanUsage } from "./plan-usage.js";
26
28
 
27
29
  import {
28
30
  initAgent as initClaudeAgent,
@@ -117,6 +119,12 @@ const claudeSdkFactory: BackendFactory = {
117
119
  updateSystemPrompt: (prompt) => claudeUpdateSystemPrompt(prompt),
118
120
  };
119
121
 
122
+ // No per-session snapshot to offer (each turn is a fresh subprocess),
123
+ // but the subscription's rate-limit windows are readable.
124
+ const usage: UsageTelemetry = {
125
+ getPlanUsage: () => getPlanUsage(),
126
+ };
127
+
120
128
  const backend = composeBackend({
121
129
  id: "claude",
122
130
  label: "Anthropic",
@@ -126,6 +134,7 @@ const claudeSdkFactory: BackendFactory = {
126
134
  models,
127
135
  sessions,
128
136
  tools,
137
+ usage,
129
138
  control,
130
139
  });
131
140
 
@@ -36,6 +36,7 @@ import { applyRetryDecisionStream } from "../shared/handle-retry.js";
36
36
  import { getConfig } from "./state.js";
37
37
  import { buildSdkOptions, getActiveFrontends } from "./options.js";
38
38
  import { waitForMcpServersReady } from "./mcp-ready.js";
39
+ import { invalidatePlanUsage } from "./plan-usage.js";
39
40
  import { frontendsForChat } from "../shared/frontends.js";
40
41
  import {
41
42
  createStreamState,
@@ -43,6 +44,7 @@ import {
43
44
  isStreamEvent,
44
45
  isAssistant,
45
46
  isResult,
47
+ isRateLimitEvent,
46
48
  isUserMessage,
47
49
  extractToolResults,
48
50
  processStreamDelta,
@@ -384,6 +386,14 @@ export async function* runChatTurn(
384
386
  continue;
385
387
  }
386
388
 
389
+ // The turn just moved the plan's usage — drop the cached windows so
390
+ // the next /status reads them again instead of showing pre-turn
391
+ // figures.
392
+ if (isRateLimitEvent(message)) {
393
+ invalidatePlanUsage();
394
+ continue;
395
+ }
396
+
387
397
  if (isResult(message)) {
388
398
  processResultMessage(message, state, options.model ?? activeModel);
389
399
  armPostResultWatchdog();
@@ -515,6 +525,7 @@ export async function* runChatTurn(
515
525
  contextTokens: state.contextTokens,
516
526
  contextWindow: state.contextWindow,
517
527
  numApiCalls: state.numApiCalls,
528
+ costUsd: state.costUsd,
518
529
  });
519
530
 
520
531
  // Set a descriptive session name from the first message.
@@ -0,0 +1,170 @@
1
+ /**
2
+ * Claude.ai subscription rate-limit windows for /status.
3
+ *
4
+ * The 5-hour, weekly, and per-model utilisation percentages come from the
5
+ * same OAuth endpoint Claude Code's own usage panel reads. The Agent SDK
6
+ * also exposes them through a control request, but that path additionally
7
+ * builds a local-session behaviour report and costs seconds per call; the
8
+ * endpoint alone answers in well under a second.
9
+ *
10
+ * Everything degrades to `undefined`: no credentials, an API-key session
11
+ * (plan limits don't apply), or any transport failure. /status hides the
12
+ * section rather than rendering zeroes.
13
+ */
14
+
15
+ import { readFile } from "node:fs/promises";
16
+ import { homedir } from "node:os";
17
+ import { join } from "node:path";
18
+ import { logWarn } from "../../util/log.js";
19
+ import type {
20
+ PlanUsage,
21
+ PlanWindow,
22
+ } from "../../core/agent-runtime/capabilities.js";
23
+
24
+ const USAGE_ENDPOINT = "https://api.anthropic.com/api/oauth/usage";
25
+ const REQUEST_TIMEOUT_MS = 5_000;
26
+ const CACHE_TTL_MS = 60_000;
27
+
28
+ let cache: { value: PlanUsage; fetchedAt: number } | undefined;
29
+ let inFlight: Promise<PlanUsage | undefined> | undefined;
30
+
31
+ function credentialsPath(): string {
32
+ const configDir = process.env.CLAUDE_CONFIG_DIR?.trim();
33
+ return join(
34
+ configDir && configDir.length > 0 ? configDir : join(homedir(), ".claude"),
35
+ ".credentials.json",
36
+ );
37
+ }
38
+
39
+ interface OAuthCredentials {
40
+ accessToken?: string;
41
+ subscriptionType?: string;
42
+ }
43
+
44
+ async function readCredentials(): Promise<OAuthCredentials | undefined> {
45
+ try {
46
+ const parsed = JSON.parse(await readFile(credentialsPath(), "utf8")) as {
47
+ claudeAiOauth?: OAuthCredentials;
48
+ };
49
+ const oauth = parsed.claudeAiOauth;
50
+ return oauth?.accessToken ? oauth : undefined;
51
+ } catch {
52
+ return undefined;
53
+ }
54
+ }
55
+
56
+ interface RawLimit {
57
+ kind?: string;
58
+ percent?: number;
59
+ resets_at?: string | null;
60
+ scope?: { model?: { display_name?: string | null } | null } | null;
61
+ }
62
+
63
+ /**
64
+ * Short display label for one window, or undefined to skip the row.
65
+ *
66
+ * Only the three documented kinds are rendered. The response also carries
67
+ * internal codenamed windows; skipping unknown kinds keeps those out of the
68
+ * user-facing panel.
69
+ */
70
+ function windowLabel(limit: RawLimit): string | undefined {
71
+ if (limit.kind === "session") return "5h";
72
+ if (limit.kind === "weekly_all") return "7d";
73
+ if (limit.kind === "weekly_scoped") {
74
+ const model = limit.scope?.model?.display_name?.trim();
75
+ return model && model.length > 0 ? model : undefined;
76
+ }
77
+ return undefined;
78
+ }
79
+
80
+ export function parsePlanUsage(
81
+ body: unknown,
82
+ subscriptionType?: string,
83
+ ): PlanUsage | undefined {
84
+ const limits = (body as { limits?: unknown } | null)?.limits;
85
+ if (!Array.isArray(limits)) return undefined;
86
+
87
+ const windows: PlanWindow[] = [];
88
+ for (const limit of limits as RawLimit[]) {
89
+ const label = windowLabel(limit);
90
+ if (!label) continue;
91
+ const raw = limit.percent;
92
+ const percent =
93
+ typeof raw === "number" && Number.isFinite(raw)
94
+ ? Math.max(0, Math.min(100, Math.round(raw)))
95
+ : 0;
96
+ windows.push({
97
+ label,
98
+ percent,
99
+ ...(typeof limit.resets_at === "string"
100
+ ? { resetsAt: limit.resets_at }
101
+ : {}),
102
+ });
103
+ }
104
+
105
+ if (windows.length === 0) return undefined;
106
+ return {
107
+ ...(subscriptionType ? { plan: subscriptionType } : {}),
108
+ windows,
109
+ fetchedAt: Date.now(),
110
+ };
111
+ }
112
+
113
+ async function load(): Promise<PlanUsage | undefined> {
114
+ const creds = await readCredentials();
115
+ if (!creds?.accessToken) return undefined;
116
+
117
+ try {
118
+ const res = await fetch(USAGE_ENDPOINT, {
119
+ headers: {
120
+ Authorization: `Bearer ${creds.accessToken}`,
121
+ "anthropic-beta": "oauth-2025-04-20",
122
+ },
123
+ signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
124
+ });
125
+ if (!res.ok) {
126
+ logWarn("agent", `plan usage: endpoint returned ${res.status}`);
127
+ return undefined;
128
+ }
129
+ return parsePlanUsage(await res.json(), creds.subscriptionType);
130
+ } catch (err) {
131
+ logWarn(
132
+ "agent",
133
+ `plan usage: ${err instanceof Error ? err.message : String(err)}`,
134
+ );
135
+ return undefined;
136
+ }
137
+ }
138
+
139
+ /**
140
+ * Plan windows for /status, cached for a minute so a burst of commands
141
+ * makes one request. A failed refresh falls back to the last known values
142
+ * — `fetchedAt` lets the caller age them.
143
+ */
144
+ export async function getPlanUsage(): Promise<PlanUsage | undefined> {
145
+ // An API-key session bills against the key, not the subscription, so the
146
+ // stored OAuth credentials would describe limits that don't apply here.
147
+ if (process.env.ANTHROPIC_API_KEY) return undefined;
148
+
149
+ if (cache && Date.now() - cache.fetchedAt < CACHE_TTL_MS) return cache.value;
150
+
151
+ inFlight ??= load().finally(() => {
152
+ inFlight = undefined;
153
+ });
154
+ const loaded = await inFlight;
155
+ if (loaded) cache = { value: loaded, fetchedAt: loaded.fetchedAt };
156
+ return loaded ?? cache?.value;
157
+ }
158
+
159
+ /**
160
+ * Expire the cache after the SDK reports a rate-limit change, so the next
161
+ * /status re-reads instead of showing figures from before the turn.
162
+ */
163
+ export function invalidatePlanUsage(): void {
164
+ if (cache) cache.fetchedAt = 0;
165
+ }
166
+
167
+ export function resetPlanUsageForTest(): void {
168
+ cache = undefined;
169
+ inFlight = undefined;
170
+ }
@@ -39,6 +39,8 @@ export type StreamState = {
39
39
  sdkOutputTokens: number;
40
40
  sdkCacheRead: number;
41
41
  sdkCacheWrite: number;
42
+ /** Cost of this turn in USD, as reported by the result message. */
43
+ costUsd: number;
42
44
  /**
43
45
  * Per-turn cache behaviour derived from the result message's per-request
44
46
  * `usage.iterations`. Undefined when the provider reported none — the
@@ -113,6 +115,7 @@ export function createStreamState(): StreamState {
113
115
  sdkOutputTokens: 0,
114
116
  sdkCacheRead: 0,
115
117
  sdkCacheWrite: 0,
118
+ costUsd: 0,
116
119
  cacheStats: undefined,
117
120
  lastStreamUpdate: 0,
118
121
  lastTrailingText: "",
@@ -148,6 +151,11 @@ export function isResult(msg: SDKMessage): msg is SDKResultMessage {
148
151
  return msg.type === "result";
149
152
  }
150
153
 
154
+ /** Emitted when the subscription's rate-limit state changes mid-turn. */
155
+ export function isRateLimitEvent(msg: SDKMessage): boolean {
156
+ return msg.type === "rate_limit_event";
157
+ }
158
+
151
159
  // ── Message processors ──────────────────────────────────────────────────────
152
160
 
153
161
  /** Output of `processStreamDelta` when the throttle interval has elapsed. */
@@ -337,6 +345,11 @@ export function processResultMessage(
337
345
  ): void {
338
346
  state.numApiCalls = msg.num_turns ?? 0;
339
347
 
348
+ // Per-turn, not cumulative across a resumed session — safe to add up.
349
+ if (typeof msg.total_cost_usd === "number" && msg.total_cost_usd > 0) {
350
+ state.costUsd = msg.total_cost_usd;
351
+ }
352
+
340
353
  // Context fill from last API iteration
341
354
  const usage = msg.usage;
342
355
  if (usage && Array.isArray(usage.iterations) && usage.iterations.length > 0) {
@@ -28,6 +28,7 @@
28
28
  import {
29
29
  appendText,
30
30
  closeCurrentSegment,
31
+ markProgressDelivered,
31
32
  recordTokens,
32
33
  recordToolUse,
33
34
  recordToolCall,
@@ -279,6 +280,10 @@ async function processPartUpdate(
279
280
  if (progress && ctx.onTextBlock) {
280
281
  try {
281
282
  await ctx.onTextBlock(progress);
283
+ // Only now is this text actually with the user. `closeCurrentSegment`
284
+ // folded it into `allResponseText`, so without this the end-of-turn
285
+ // delivery would ship it a second time inside the full transcript.
286
+ markProgressDelivered(ctx.state);
282
287
  } catch (err) {
283
288
  // Non-fatal — never break the stream loop on a UI callback. Logged
284
289
  // at debug level so a repeatedly-failing frontend is visible to
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Unified delivery routing for the remote-server backend family.
3
3
  *
4
- * Both OpenCode and Kilo can reach the user through one of four paths
4
+ * Both OpenCode and Kilo can reach the user through one of several paths
5
5
  * after a turn completes. The decision tree below is identical between
6
6
  * them, so it lives here as a shared helper instead of being duplicated
7
7
  * in each handler.
@@ -19,7 +19,12 @@
19
19
  * marker, or analogous failure path). Surface as a Talon error
20
20
  * instead of shipping the raw upstream string as a reply.
21
21
  * - `text-part` — model emitted plain assistant text and didn't call
22
- * a delivery tool. Ship it through `onTextBlock`.
22
+ * a delivery tool. Ship the part the user hasn't already seen through
23
+ * `onTextBlock` (see `progress` below).
24
+ * - `progress` — every line of the reply already reached the user as a
25
+ * mid-turn progress message, so there is nothing left to send. Recorded
26
+ * distinctly rather than as `empty`, which means "the model said
27
+ * nothing at all".
23
28
  * - `empty` — no tool, no text, no synthetic-error. Surface a concise
24
29
  * notice so the user isn't left staring at silence.
25
30
  *
@@ -29,11 +34,13 @@
29
34
  */
30
35
 
31
36
  import type { StreamState } from "./stream-state.js";
37
+ import { undeliveredResponseText } from "./stream-state.js";
32
38
  import { logWarn } from "../../util/log.js";
33
39
  import { incrementCounter } from "../../storage/metrics.js";
34
40
 
35
41
  /** Route the delivery decision selected. */
36
- export type DeliveryRoute = "tool" | "text-part" | "synthetic-error" | "empty";
42
+ export type DeliveryRoute =
43
+ "tool" | "text-part" | "progress" | "synthetic-error" | "empty";
37
44
 
38
45
  export class TextBlockDeliveryError extends Error {
39
46
  readonly route: DeliveryRoute;
@@ -163,23 +170,41 @@ export async function routeDelivery(
163
170
  };
164
171
  }
165
172
 
166
- // Route 3 — plain text part. Ship it.
167
- if (responseText && !state.turnTerminated) {
173
+ // Route 3 — plain text part. Ship only what the user hasn't already seen.
174
+ //
175
+ // The remote-server backends flush each pre-tool segment through
176
+ // `onTextBlock` as a progress message, and `closeCurrentSegment` also folds
177
+ // that segment into `allResponseText`. Shipping `responseText` wholesale
178
+ // therefore re-sent every narration line a second time, concatenated — the
179
+ // doubled-message symptom on a tool-heavy OpenCode/Kilo turn.
180
+ //
181
+ // The subtraction is gated on a progress send having actually happened.
182
+ // `responseText` is an explicit input and callers are not required to derive
183
+ // it from `state` (several pass a literal), so reading the remainder off
184
+ // `allResponseText` unconditionally would silently drop their reply.
185
+ const pending =
186
+ state.progressDeliveredLen > 0
187
+ ? undeliveredResponseText(state)
188
+ : responseText;
189
+ if (pending && !state.turnTerminated) {
168
190
  if (onTextBlock) {
169
191
  try {
170
- await onTextBlock(responseText);
192
+ await onTextBlock(pending);
171
193
  } catch (err) {
172
194
  logWarn("agent", `[${chatId}] onTextBlock failed: ${errMsg(err)}`);
173
195
  if (propagateDeliveryFailure) {
174
- throw new TextBlockDeliveryError(
175
- "text-part",
176
- responseText.length,
177
- err,
178
- );
196
+ throw new TextBlockDeliveryError("text-part", pending.length, err);
179
197
  }
180
198
  }
181
199
  }
182
- return { route: "text-part", chars: responseText.length };
200
+ return { route: "text-part", chars: pending.length };
201
+ }
202
+
203
+ // Route 3b — the whole reply already reached the user as progress messages.
204
+ // Nothing left to send; recorded distinctly so the turn isn't misfiled as an
205
+ // empty completion in the logs or the empty-turn counter.
206
+ if (responseText && !state.turnTerminated) {
207
+ return { route: "progress", chars: responseText.length };
183
208
  }
184
209
 
185
210
  // Route 4 — empty turn. The model produced no text (and didn't end_turn):
@@ -96,6 +96,8 @@ export {
96
96
  createStreamState,
97
97
  appendText,
98
98
  closeCurrentSegment,
99
+ markProgressDelivered,
100
+ undeliveredResponseText,
99
101
  recordToolUse,
100
102
  recordTokens,
101
103
  pushLiveUsage,
@@ -44,6 +44,21 @@ export interface StreamState {
44
44
  allResponseText: string;
45
45
  /** Text *after* the last tool call (or the entire response if no tools). */
46
46
  lastTrailingText: string;
47
+ /**
48
+ * How much of `allResponseText` has already reached the user as a
49
+ * mid-turn progress message.
50
+ *
51
+ * The remote-server backends flush the pending segment through
52
+ * `onTextBlock` at each tool boundary, but `closeCurrentSegment` also
53
+ * folds that segment into `allResponseText` — so the end-of-turn
54
+ * delivery would ship every narration line a second time, concatenated.
55
+ * Delivery ships only `allResponseText.slice(progressDeliveredLen)`.
56
+ *
57
+ * Advanced only after a progress send actually succeeds, so a flush that
58
+ * throws (e.g. Telegram's 4096-char limit) leaves its text pending and
59
+ * the end-of-turn delivery still carries it.
60
+ */
61
+ progressDeliveredLen: number;
47
62
 
48
63
  // ── Session ───────────────────────────────────────────────────────────────
49
64
  /** Provider-assigned session id, if one was announced during the stream. */
@@ -133,6 +148,7 @@ export function createStreamState(chatId?: string): StreamState {
133
148
  currentBlockText: "",
134
149
  allResponseText: "",
135
150
  lastTrailingText: "",
151
+ progressDeliveredLen: 0,
136
152
  newSessionId: undefined,
137
153
  toolCalls: 0,
138
154
  turnTerminated: false,
@@ -279,3 +295,23 @@ export function finalizeResponseText(state: StreamState): string {
279
295
  }
280
296
  return state.allResponseText.trim();
281
297
  }
298
+
299
+ /**
300
+ * Record that everything accumulated so far has been shipped to the user
301
+ * as a progress message. Call only after the send succeeds.
302
+ */
303
+ export function markProgressDelivered(state: StreamState): void {
304
+ state.progressDeliveredLen = state.allResponseText.length;
305
+ }
306
+
307
+ /**
308
+ * The portion of the turn's text that has NOT already been shipped as a
309
+ * mid-turn progress message — i.e. what end-of-turn delivery still owes
310
+ * the user.
311
+ *
312
+ * Backends that never flush progress leave `progressDeliveredLen` at 0,
313
+ * so this is the whole response and their behaviour is unchanged.
314
+ */
315
+ export function undeliveredResponseText(state: StreamState): string {
316
+ return state.allResponseText.slice(state.progressDeliveredLen).trim();
317
+ }
@@ -199,10 +199,11 @@ export interface ToolRuntime {
199
199
  }
200
200
 
201
201
  /**
202
- * `/status` enrichment. Backends that track per-session usage
203
- * (Codex, OpenAI Agents) implement this. Backends without a
204
- * per-session model (Claude SDK on a fresh subprocess per turn)
205
- * return `undefined`.
202
+ * `/status` enrichment. Both members are optional — a backend
203
+ * implements whichever it can answer. Backends that track per-session
204
+ * usage (Codex, OpenAI Agents) supply `getSessionSnapshot`; a backend
205
+ * with no per-session model (Claude SDK on a fresh subprocess per
206
+ * turn) omits it and may still report plan limits.
206
207
  *
207
208
  * The snapshot's `contextModelId` carries the resolved-this-turn
208
209
  * model id when the SDK can surface it. Frontend `/status` reads
@@ -211,7 +212,7 @@ export interface ToolRuntime {
211
212
  * ChatGPT-OAuth).
212
213
  */
213
214
  export interface UsageTelemetry {
214
- getSessionSnapshot(sessionId: string): Promise<
215
+ getSessionSnapshot?(sessionId: string): Promise<
215
216
  | {
216
217
  inputTokens?: number;
217
218
  outputTokens?: number;
@@ -221,6 +222,31 @@ export interface UsageTelemetry {
221
222
  }
222
223
  | undefined
223
224
  >;
225
+ /**
226
+ * Subscription rate-limit windows. Account-level rather than
227
+ * per-chat: `/status` renders whichever backend can answer, even
228
+ * when another one is serving the chat. Absent on backends with no
229
+ * plan concept; resolves `undefined` when the data can't be read.
230
+ */
231
+ getPlanUsage?(): Promise<PlanUsage | undefined>;
232
+ }
233
+
234
+ /** One subscription rate-limit window, as `/status` renders it. */
235
+ export interface PlanWindow {
236
+ /** Short label — `5h`, `7d`, or the scoped model's display name. */
237
+ label: string;
238
+ /** Window utilisation, 0-100. */
239
+ percent: number;
240
+ /** ISO timestamp of the next reset, when the plan reports one. */
241
+ resetsAt?: string;
242
+ }
243
+
244
+ export interface PlanUsage {
245
+ /** Subscription tier (`max`, `pro`, …) when known. */
246
+ plan?: string;
247
+ windows: PlanWindow[];
248
+ /** Epoch ms of the read, so renderers can flag figures as aged. */
249
+ fetchedAt: number;
224
250
  }
225
251
 
226
252
  /**
@@ -272,16 +272,16 @@ export async function assertModelCatalogDefaultShape(
272
272
  }
273
273
 
274
274
  /**
275
- * Backends that expose `UsageTelemetry` must answer
276
- * `getSessionSnapshot` with either a `UsageSnapshot` object whose
277
- * counters are non-negative numbers, or `undefined`. Negative
278
- * counters or `NaN`s are contract violations.
275
+ * Backends that implement `getSessionSnapshot` must answer it with
276
+ * either a `UsageSnapshot` object whose counters are non-negative
277
+ * numbers, or `undefined`. Negative counters or `NaN`s are contract
278
+ * violations. Backends that only report plan limits are skipped.
279
279
  */
280
280
  export async function assertUsageTelemetryShape(
281
281
  backend: Backend,
282
282
  sessionId = "contract-test-session",
283
283
  ): Promise<void> {
284
- if (!backend.usage) return;
284
+ if (!backend.usage?.getSessionSnapshot) return;
285
285
  const snapshot = await backend.usage.getSessionSnapshot(sessionId);
286
286
  if (snapshot === undefined) return;
287
287
  const fields: (keyof typeof snapshot)[] = [
@@ -98,6 +98,39 @@ export const modelHandlers: SharedActionHandlers = {
98
98
  }
99
99
  },
100
100
 
101
+ // Account-level, so it falls back to the pooled Claude backend when
102
+ // another provider is serving this chat.
103
+ plan_usage: async (_body, chatId) => {
104
+ const current = getPooledBackend(getBackendIdForChat(String(chatId)));
105
+ const source = current?.usage?.getPlanUsage
106
+ ? current
107
+ : getPooledBackend("claude");
108
+ if (!source?.usage?.getPlanUsage)
109
+ return {
110
+ ok: false,
111
+ error: "No configured backend reports subscription rate limits.",
112
+ };
113
+
114
+ const usage = await source.usage.getPlanUsage();
115
+ if (!usage)
116
+ return {
117
+ ok: false,
118
+ error:
119
+ "Plan usage is unavailable — no subscription credentials, or this session authenticates with an API key.",
120
+ };
121
+
122
+ const lines = usage.windows.map(
123
+ (w) =>
124
+ `- ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
125
+ );
126
+ return {
127
+ ok: true,
128
+ plan: usage.plan ?? null,
129
+ windows: usage.windows,
130
+ text: `Plan usage${usage.plan ? ` (${usage.plan})` : ""}:\n${lines.join("\n")}`,
131
+ };
132
+ },
133
+
101
134
  list_backends: (body, chatId) => {
102
135
  const currentId = getBackendIdForChat(String(chatId));
103
136
  const backends = getAvailableBackends().map((b) => ({
@@ -28,6 +28,15 @@ export const modelTools: ToolDefinition[] = [
28
28
  tag: "models",
29
29
  },
30
30
 
31
+ {
32
+ name: "plan_usage",
33
+ description:
34
+ "Read your own subscription usage: how much of the 5-hour, weekly, and per-model rate-limit windows is spent, and when each resets. Use it before starting long or expensive work, or when deciding whether to defer something. Only answers on a subscription-backed Anthropic backend; other providers report no plan limits.",
35
+ schema: {},
36
+ execute: (_params, bridge) => bridge("plan_usage", {}),
37
+ tag: "models",
38
+ },
39
+
31
40
  {
32
41
  name: "list_backends",
33
42
  description:
@@ -13,6 +13,7 @@ import {
13
13
  formatDuration,
14
14
  formatTokenCount,
15
15
  formatBytes,
16
+ formatUsd,
16
17
  } from "../helpers.js";
17
18
  import {
18
19
  getBackendIdForChat,
@@ -78,8 +79,18 @@ export async function handleStatus(
78
79
  ` Read ${formatTokenCount(s.cache.read)}${s.cache.showsWrite ? ` Write ${formatTokenCount(s.cache.write)}` : ""}`,
79
80
  ]
80
81
  : []),
81
- ` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}`,
82
+ ` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}${s.costUsd > 0 ? ` Cost ${formatUsd(s.costUsd)}` : ""}`,
82
83
  "",
84
+ ...(s.plan
85
+ ? [
86
+ `**Plan**${s.plan.plan ? ` ${s.plan.plan}` : ""}${s.plan.ageLabel ? ` *(${s.plan.ageLabel})*` : ""}`,
87
+ ...s.plan.windows.map(
88
+ (w) =>
89
+ ` \`${w.label.padEnd(6)}${w.bar} ${String(w.percent).padStart(3)}%\`${w.resetLabel ? ` reset ${w.resetLabel}` : ""}`,
90
+ ),
91
+ "",
92
+ ]
93
+ : []),
83
94
  `**Pulse** ${s.pulseOn ? "on" : "off"}`,
84
95
  `**Workspace** ${formatBytes(s.diskBytes)}`,
85
96
  `**Session** ${s.sessionName ? `"${s.sessionName}" ` : ""}${s.sessionId ? "`" + s.sessionId.slice(0, 8) + "...`" : "_(new)_"} · ${s.sessionAge} old`,
@@ -20,6 +20,7 @@ export {
20
20
  formatDuration,
21
21
  formatTokenCount,
22
22
  formatBytes,
23
+ formatUsd,
23
24
  formatModelLabel,
24
25
  } from "../shared/format.js";
25
26
 
@@ -23,11 +23,14 @@ import { isPulseEnabled } from "../../core/background/pulse.js";
23
23
  import { getWorkspaceDiskUsage } from "../../util/workspace.js";
24
24
  import { appendDailyLog } from "../../storage/daily-log.js";
25
25
  import { resolveActiveModelForChat } from "../../core/models/active-model.js";
26
+ import { getPooledBackend } from "../../core/engine/backend-controller/index.js";
26
27
  import {
27
28
  buildCacheDisplay,
28
29
  buildContextDisplay,
30
+ buildPlanDisplay,
29
31
  type CacheDisplay,
30
32
  type ContextDisplay,
33
+ type PlanDisplay,
31
34
  } from "./status-context.js";
32
35
  import { formatDuration } from "./format.js";
33
36
 
@@ -70,8 +73,10 @@ export interface SessionStatusData {
70
73
  pulseOn: boolean;
71
74
  context: ContextDisplay;
72
75
  cache: CacheDisplay | null;
76
+ plan: PlanDisplay | null;
73
77
  inputTokens: number;
74
78
  outputTokens: number;
79
+ costUsd: number;
75
80
  turns: number;
76
81
  turnsModelLabel: string | undefined;
77
82
  lastResponseMs: number;
@@ -157,6 +162,16 @@ export async function collectSessionStatus(
157
162
  contextWindow: ctxMax,
158
163
  });
159
164
 
165
+ // Plan limits belong to the account, not the chat: fall back to the pooled
166
+ // Claude backend so the section still shows while another provider serves
167
+ // this chat.
168
+ const planSource = backend?.usage?.getPlanUsage
169
+ ? backend
170
+ : getPooledBackend("claude");
171
+ const plan = buildPlanDisplay(
172
+ await planSource?.usage?.getPlanUsage?.().catch(() => undefined),
173
+ );
174
+
160
175
  const avgResponseMs =
161
176
  info.turns > 0 && u.totalResponseMs
162
177
  ? Math.round(u.totalResponseMs / info.turns)
@@ -170,8 +185,10 @@ export async function collectSessionStatus(
170
185
  pulseOn: isPulseEnabled(chatId),
171
186
  context,
172
187
  cache,
188
+ plan,
173
189
  inputTokens,
174
190
  outputTokens,
191
+ costUsd: u.estimatedCostUsd,
175
192
  turns: info.turns,
176
193
  turnsModelLabel,
177
194
  lastResponseMs: u.lastResponseMs || 0,
@@ -1,4 +1,6 @@
1
1
  import type { CacheMetricsSupport } from "../../core/types.js";
2
+ import type { PlanUsage } from "../../core/agent-runtime/capabilities.js";
3
+ import { formatSmartTimestamp, formatRelativeAge } from "../../util/time.js";
2
4
 
3
5
  // ── /context breakdown ────────────────────────────────────────────────────────
4
6
 
@@ -284,3 +286,63 @@ export function buildCacheDisplay(input: {
284
286
  showsWrite: mode === "readwrite",
285
287
  };
286
288
  }
289
+
290
+ // ── Plan limits ─────────────────────────────────────────────────────────────
291
+
292
+ /** Figures older than this are labelled with their age in /status. */
293
+ const PLAN_STALE_AFTER_MS = 5 * 60_000;
294
+
295
+ export interface PlanWindowDisplay {
296
+ label: string;
297
+ percent: number;
298
+ bar: string;
299
+ /** Local-time reset, absent for windows the plan reports no reset for. */
300
+ resetLabel: string | undefined;
301
+ }
302
+
303
+ export interface PlanDisplay {
304
+ plan: string | undefined;
305
+ windows: PlanWindowDisplay[];
306
+ /** Set only once the figures have aged, e.g. "12m ago". */
307
+ ageLabel: string | undefined;
308
+ }
309
+
310
+ function planResetLabel(iso: string | undefined): string | undefined {
311
+ if (!iso) return undefined;
312
+ const ts = Date.parse(iso);
313
+ if (!Number.isFinite(ts)) return undefined;
314
+ // Windows are reported a second short of the boundary (20:59:59); round to
315
+ // the minute so the panel reads 21:00, as the plan's own dashboards do.
316
+ return formatSmartTimestamp(Math.round(ts / 60_000) * 60_000);
317
+ }
318
+
319
+ /**
320
+ * Lay out the subscription's rate-limit windows for /status. Returns null
321
+ * when the backend has nothing to report, so the section disappears rather
322
+ * than rendering empty bars.
323
+ */
324
+ export function buildPlanDisplay(
325
+ usage: PlanUsage | undefined,
326
+ barLen = 20,
327
+ ): PlanDisplay | null {
328
+ if (!usage || usage.windows.length === 0) return null;
329
+
330
+ const age = Date.now() - usage.fetchedAt;
331
+ return {
332
+ plan: usage.plan,
333
+ ageLabel:
334
+ age > PLAN_STALE_AFTER_MS
335
+ ? formatRelativeAge(usage.fetchedAt)
336
+ : undefined,
337
+ windows: usage.windows.map((w) => {
338
+ const percent = Math.max(0, Math.min(100, Math.round(w.percent)));
339
+ const filled = Math.round((percent / 100) * barLen);
340
+ return {
341
+ label: w.label,
342
+ percent,
343
+ bar: "█".repeat(filled) + "░".repeat(barLen - filled),
344
+ resetLabel: planResetLabel(w.resetsAt),
345
+ };
346
+ }),
347
+ };
348
+ }
@@ -12,6 +12,7 @@ import {
12
12
  formatDuration,
13
13
  formatTokenCount,
14
14
  formatBytes,
15
+ formatUsd,
15
16
  } from "../helpers/index.js";
16
17
  import { resolveBackendForChat } from "../model-menu.js";
17
18
  import { getBackendIdForChat } from "../../../core/engine/backend-controller/index.js";
@@ -65,8 +66,18 @@ export function registerSessionCommands(
65
66
  ` Read ${formatTokenCount(s.cache.read)}${s.cache.showsWrite ? ` Write ${formatTokenCount(s.cache.write)}` : ""}`,
66
67
  ]
67
68
  : []),
68
- ` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}`,
69
+ ` Input ${formatTokenCount(s.inputTokens)} Output ${formatTokenCount(s.outputTokens)}${s.costUsd > 0 ? ` Cost ${formatUsd(s.costUsd)}` : ""}`,
69
70
  "",
71
+ ...(s.plan
72
+ ? [
73
+ `<b>Plan</b>${s.plan.plan ? ` ${escapeHtml(s.plan.plan)}` : ""}${s.plan.ageLabel ? ` <i>(${s.plan.ageLabel})</i>` : ""}`,
74
+ ...s.plan.windows.map(
75
+ (w) =>
76
+ ` <code>${escapeHtml(w.label.padEnd(6))}${w.bar} ${String(w.percent).padStart(3)}%</code>${w.resetLabel ? ` reset ${w.resetLabel}` : ""}`,
77
+ ),
78
+ "",
79
+ ]
80
+ : []),
70
81
  `<b>Pulse</b> ${s.pulseOn ? "on" : "off"}`,
71
82
  `<b>Workspace</b> ${formatBytes(s.diskBytes)}`,
72
83
  `<b>Session</b> ${s.sessionName ? `"${escapeHtml(s.sessionName)}" ` : ""}${s.sessionId ? "<code>" + escapeHtml(s.sessionId.slice(0, 8)) + "...</code>" : "<i>(new)</i>"} · ${s.sessionAge} old`,
@@ -13,6 +13,7 @@ export {
13
13
  formatDuration,
14
14
  formatTokenCount,
15
15
  formatBytes,
16
+ formatUsd,
16
17
  formatModelLabel,
17
18
  } from "../../shared/format.js";
18
19
 
@@ -18,6 +18,7 @@ import {
18
18
  import {
19
19
  buildCacheDisplay,
20
20
  buildContextDisplay,
21
+ buildPlanDisplay,
21
22
  buildContextBreakdown,
22
23
  estimateContextTokens,
23
24
  apportionCells,
@@ -43,6 +44,7 @@ import {
43
44
  setSessionName,
44
45
  } from "../../storage/sessions.js";
45
46
  import { getLoadedPlugins } from "../../core/plugin/index.js";
47
+ import { getPooledBackend } from "../../core/engine/backend-controller/index.js";
46
48
 
47
49
  // ── Types ────────────────────────────────────────────────────────────────────
48
50
 
@@ -430,6 +432,24 @@ export function registerBuiltinCommands(): void {
430
432
  ` ${pc.dim(`estimated session cost ${formatUsd(u.estimatedCostUsd)}`)}`,
431
433
  );
432
434
  }
435
+
436
+ const planSource = be?.usage?.getPlanUsage
437
+ ? be
438
+ : getPooledBackend("claude");
439
+ const plan = buildPlanDisplay(
440
+ await planSource?.usage?.getPlanUsage?.().catch(() => undefined),
441
+ );
442
+ if (plan) {
443
+ ctx.renderer.writeln();
444
+ ctx.renderer.writeln(
445
+ ` ${pc.bold("Plan")}${plan.plan ? ` ${plan.plan}` : ""}${plan.ageLabel ? pc.dim(` (${plan.ageLabel})`) : ""}`,
446
+ );
447
+ for (const w of plan.windows) {
448
+ ctx.renderer.writeln(
449
+ ` ${w.label.padEnd(6)}${pc.dim(w.bar)} ${String(w.percent).padStart(3)}%${w.resetLabel ? pc.dim(` reset ${w.resetLabel}`) : ""}`,
450
+ );
451
+ }
452
+ }
433
453
  if (backendModelLine) {
434
454
  ctx.renderer.writeln();
435
455
  ctx.renderer.writeln(backendModelLine);