talon-agent 3.14.0 → 3.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.14.0",
3
+ "version": "3.15.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -1,5 +1,166 @@
1
1
  import type { CacheMetricsSupport } from "../../core/types.js";
2
2
 
3
+ // ── /context breakdown ────────────────────────────────────────────────────────
4
+
5
+ /**
6
+ * Rough token estimate — ~4 chars/token, the house heuristic used everywhere
7
+ * (soul/projector, cache-telemetry). No real tokenizer is wired, so every
8
+ * measured figure in the breakdown below is an estimate at this fidelity.
9
+ */
10
+ export function estimateContextTokens(text: string): number {
11
+ return Math.ceil(text.length / 4);
12
+ }
13
+
14
+ export type ContextSegmentKey = "system" | "tools" | "conversation";
15
+
16
+ interface ContextSegment {
17
+ key: ContextSegmentKey;
18
+ label: string;
19
+ tokens: number;
20
+ /** Share of the window (0–100) when the window is known, else share of used. */
21
+ pct: number;
22
+ }
23
+
24
+ export interface ContextBreakdown {
25
+ /** True once there is anything to show (a system prompt or a reported fill). */
26
+ known: boolean;
27
+ /** True when the window size is known — only then can free space be shown. */
28
+ windowKnown: boolean;
29
+ used: number;
30
+ max: number;
31
+ usedPct: number;
32
+ free: number;
33
+ freePct: number;
34
+ /** Fixed → variable order: System, Tools, Conversation. */
35
+ segments: ContextSegment[];
36
+ warn: boolean;
37
+ }
38
+
39
+ function posInt(n: number | undefined): number {
40
+ return typeof n === "number" && Number.isFinite(n) && n > 0
41
+ ? Math.round(n)
42
+ : 0;
43
+ }
44
+
45
+ function round1(n: number): number {
46
+ return Math.round(n * 10) / 10;
47
+ }
48
+
49
+ /**
50
+ * Decompose the context window into System / Tools / Conversation + free.
51
+ *
52
+ * The honesty this has to preserve: the backend reports exactly one
53
+ * authoritative number, `contextTokens` (the real last-turn fill). Nothing
54
+ * reports a per-category split, so:
55
+ *
56
+ * - **System** is measured from the actual (frozen) system prompt — accurate,
57
+ * and the part a user can act on.
58
+ * - **Conversation** is estimated from stored history. It can overshoot what
59
+ * is really in-window after compaction; when it does, it is clamped to fit.
60
+ * - **Tools** is the residual: `used − system − conversation`. Tool schemas
61
+ * (invisible to us — they live inside the SDK) dominate it, but it also
62
+ * absorbs message-formatting overhead and estimation slack. Labelled as the
63
+ * remainder, not claimed as exact.
64
+ *
65
+ * When the backend reports no fill, tools cannot be derived, so only the two
66
+ * measured/estimated parts are shown and the window's free space (if known) is
67
+ * whatever is left of it.
68
+ */
69
+ export function buildContextBreakdown(input: {
70
+ contextTokens?: number;
71
+ contextWindow?: number;
72
+ systemTokens: number;
73
+ conversationTokens: number;
74
+ }): ContextBreakdown {
75
+ const max = posInt(input.contextWindow);
76
+ const system = Math.max(0, Math.round(input.systemTokens));
77
+ let conversation = Math.max(0, Math.round(input.conversationTokens));
78
+ const fill = posInt(input.contextTokens);
79
+
80
+ const segments: ContextSegment[] = [];
81
+ let used: number;
82
+
83
+ if (fill > 0) {
84
+ used = fill;
85
+ // System is sent in full every turn; if our estimate exceeds the real fill
86
+ // that is estimation slack, not reality — clamp so parts never exceed used.
87
+ const sys = Math.min(system, used);
88
+ let tools = used - sys - conversation;
89
+ if (tools < 0) {
90
+ // Conversation overshot the real fill (compaction dropped in-window
91
+ // messages) — give the remainder back to conversation, zero the residual.
92
+ conversation = Math.max(0, used - sys);
93
+ tools = 0;
94
+ }
95
+ segments.push({ key: "system", label: "System", tokens: sys, pct: 0 });
96
+ segments.push({ key: "tools", label: "Tools", tokens: tools, pct: 0 });
97
+ segments.push({
98
+ key: "conversation",
99
+ label: "Conversation",
100
+ tokens: conversation,
101
+ pct: 0,
102
+ });
103
+ } else {
104
+ // No authoritative fill — show what we can measure; tools isn't derivable.
105
+ used = system + conversation;
106
+ segments.push({ key: "system", label: "System", tokens: system, pct: 0 });
107
+ segments.push({
108
+ key: "conversation",
109
+ label: "Conversation",
110
+ tokens: conversation,
111
+ pct: 0,
112
+ });
113
+ }
114
+
115
+ const windowKnown = max > 0;
116
+ const free = windowKnown ? Math.max(0, max - used) : 0;
117
+ const denom = windowKnown ? max : used;
118
+ for (const s of segments) {
119
+ s.pct = denom > 0 ? round1((s.tokens / denom) * 100) : 0;
120
+ }
121
+
122
+ return {
123
+ known: used > 0,
124
+ windowKnown,
125
+ used,
126
+ max,
127
+ usedPct: windowKnown ? Math.min(100, round1((used / max) * 100)) : 0,
128
+ free,
129
+ freePct: windowKnown ? round1((free / max) * 100) : 0,
130
+ segments,
131
+ warn: windowKnown && used / max >= 0.8,
132
+ };
133
+ }
134
+
135
+ /**
136
+ * Distribute `width` integer cells across `weights` proportionally, by the
137
+ * largest-remainder method — the cells sum to exactly `width` and no positive
138
+ * weight is systematically rounded to nothing before its peers. Zero weights
139
+ * get zero cells. Used to lay out the segmented bar so its coloured runs sum to
140
+ * the bar width regardless of rounding.
141
+ */
142
+ export function apportionCells(
143
+ weights: readonly number[],
144
+ width: number,
145
+ ): number[] {
146
+ const total = weights.reduce((a, b) => a + b, 0);
147
+ if (total <= 0 || width <= 0) return weights.map(() => 0);
148
+ const exact = weights.map((w) => (Math.max(0, w) / total) * width);
149
+ const cells = exact.map((x) => Math.floor(x));
150
+ let remaining = width - cells.reduce((a, b) => a + b, 0);
151
+ const byRemainder = exact
152
+ .map((x, i) => ({ i, frac: x - Math.floor(x) }))
153
+ .sort((a, b) => b.frac - a.frac);
154
+ for (const { i } of byRemainder) {
155
+ if (remaining <= 0) break;
156
+ if (weights[i]! > 0) {
157
+ cells[i]!++;
158
+ remaining--;
159
+ }
160
+ }
161
+ return cells;
162
+ }
163
+
3
164
  export interface ContextDisplay {
4
165
  known: boolean;
5
166
  used: number;
@@ -18,7 +18,13 @@ import {
18
18
  import {
19
19
  buildCacheDisplay,
20
20
  buildContextDisplay,
21
+ buildContextBreakdown,
22
+ estimateContextTokens,
23
+ apportionCells,
24
+ type ContextBreakdown,
25
+ type ContextSegmentKey,
21
26
  } from "../shared/status-context.js";
27
+ import { getRecentHistory } from "../../storage/history.js";
22
28
  import {
23
29
  formatDuration,
24
30
  formatTokenCount,
@@ -113,6 +119,63 @@ export function clearCommands(): void {
113
119
  nameIndex.clear();
114
120
  }
115
121
 
122
+ // ── /context rendering ───────────────────────────────────────────────────────
123
+
124
+ /** Each used segment gets its own colour; free is a dim hatch. */
125
+ const CONTEXT_SEGMENT_COLOR: Record<ContextSegmentKey, (s: string) => string> =
126
+ {
127
+ system: pc.blue,
128
+ tools: pc.yellow,
129
+ conversation: pc.cyan,
130
+ };
131
+
132
+ const CONTEXT_BAR_WIDTH = 42;
133
+
134
+ /**
135
+ * The segmented bar: one coloured run per segment (proportional to the window),
136
+ * then the free space as a dim `░` hatch. Cell counts come from
137
+ * `apportionCells`, so the runs always sum to exactly the bar width.
138
+ */
139
+ function renderContextBar(bd: ContextBreakdown): string {
140
+ const weights = bd.segments.map((s) => s.tokens);
141
+ if (bd.windowKnown) weights.push(bd.free);
142
+ const cells = apportionCells(weights, CONTEXT_BAR_WIDTH);
143
+ let bar = "";
144
+ bd.segments.forEach((s, i) => {
145
+ bar += CONTEXT_SEGMENT_COLOR[s.key]("█".repeat(cells[i] ?? 0));
146
+ });
147
+ if (bd.windowKnown) {
148
+ bar += pc.dim("░".repeat(cells[bd.segments.length] ?? 0));
149
+ }
150
+ return bar;
151
+ }
152
+
153
+ /** One aligned legend row per segment (and Free), colour-matched to the bar. */
154
+ function renderContextLegend(bd: ContextBreakdown): string[] {
155
+ const rows = bd.segments.map((s) => ({
156
+ dot: CONTEXT_SEGMENT_COLOR[s.key]("●"),
157
+ label: s.label,
158
+ tokens: s.tokens,
159
+ pct: s.pct,
160
+ }));
161
+ if (bd.windowKnown) {
162
+ rows.push({
163
+ dot: pc.dim("░"),
164
+ label: "Free",
165
+ tokens: bd.free,
166
+ pct: bd.freePct,
167
+ });
168
+ }
169
+ const labelW = Math.max(...rows.map((r) => r.label.length));
170
+ const tokW = Math.max(...rows.map((r) => formatTokenCount(r.tokens).length));
171
+ return rows.map((r) => {
172
+ const label = r.label.padEnd(labelW);
173
+ const tok = formatTokenCount(r.tokens).padStart(tokW);
174
+ const pct = `${r.pct}%`.padStart(6);
175
+ return `${r.dot} ${label} ${pc.dim(tok)} ${pc.dim(pct)}`;
176
+ });
177
+ }
178
+
116
179
  // ── Built-in commands ────────────────────────────────────────────────────────
117
180
 
118
181
  export function registerBuiltinCommands(): void {
@@ -393,6 +456,95 @@ export function registerBuiltinCommands(): void {
393
456
  },
394
457
  });
395
458
 
459
+ registerCommand({
460
+ name: "context",
461
+ aliases: ["ctx"],
462
+ description: "Context-window usage, broken down",
463
+ async handler(_args, ctx) {
464
+ const chatId = ctx.chatId();
465
+ const u = getSessionInfo(chatId).usage;
466
+ const be = ctx.backend;
467
+ const activeModel = getChatSettings(chatId).model ?? ctx.config.model;
468
+
469
+ // Window + a friendly model name, enriched from the backend like /status.
470
+ let contextWindow = u.contextWindow;
471
+ let modelName = resolveModelName(activeModel);
472
+ if (be?.models?.getRawModelInfo) {
473
+ const mi = await be.models
474
+ .getRawModelInfo(activeModel)
475
+ .catch(() => undefined);
476
+ if (mi) {
477
+ if (mi.contextWindow) contextWindow ||= mi.contextWindow;
478
+ if (mi.displayName) modelName = mi.displayName;
479
+ }
480
+ }
481
+
482
+ // System = the actual frozen prompt the model is running with. Measure
483
+ // it directly rather than rebuilding, so the number matches what was
484
+ // really sent (the prompt is frozen per session by design).
485
+ const parts = ctx.config.systemPromptParts;
486
+ const systemText = parts
487
+ ? [parts.staticText, parts.dynamicText].filter(Boolean).join("\n")
488
+ : (ctx.config.systemPrompt ?? "");
489
+ const systemTokens = estimateContextTokens(systemText);
490
+
491
+ // Conversation = stored history for this chat. An estimate: the model's
492
+ // real in-window history may be smaller after compaction, which the
493
+ // breakdown clamps against the authoritative fill.
494
+ const history = getRecentHistory(chatId, 2000);
495
+ const conversationTokens = estimateContextTokens(
496
+ history.map((m) => m.text ?? "").join("\n"),
497
+ );
498
+
499
+ const bd = buildContextBreakdown({
500
+ contextTokens: u.contextTokens,
501
+ contextWindow,
502
+ systemTokens,
503
+ conversationTokens,
504
+ });
505
+
506
+ ctx.renderer.writeln();
507
+ if (!bd.known) {
508
+ ctx.renderer.writeln(
509
+ ` ${pc.bold("Context")} ${pc.dim("no usage yet — send a message first")}`,
510
+ );
511
+ ctx.reprompt();
512
+ return;
513
+ }
514
+
515
+ const windowStr = bd.windowKnown
516
+ ? `${formatTokenCount(bd.max)} window`
517
+ : pc.dim("window unknown");
518
+ ctx.renderer.writeln(
519
+ ` ${pc.bold("Context")} ${modelName} · ${windowStr}`,
520
+ );
521
+
522
+ const usedStr = bd.windowKnown
523
+ ? `${formatTokenCount(bd.used)} / ${formatTokenCount(bd.max)} · ${bd.usedPct}% used`
524
+ : `${formatTokenCount(bd.used)} used`;
525
+ ctx.renderer.writeln(
526
+ ` ${bd.warn ? pc.yellow(usedStr) : pc.dim(usedStr)}${bd.warn ? pc.yellow(" · nearing limit") : ""}`,
527
+ );
528
+ ctx.renderer.writeln();
529
+ ctx.renderer.writeln(` ${renderContextBar(bd)}`);
530
+ ctx.renderer.writeln();
531
+ for (const line of renderContextLegend(bd)) {
532
+ ctx.renderer.writeln(` ${line}`);
533
+ }
534
+ // Explain only what's on screen: Tools is derivable (and shown) only
535
+ // when the backend reported a real fill.
536
+ const hasTools = bd.segments.some((s) => s.key === "tools");
537
+ ctx.renderer.writeln(
538
+ ` ${pc.dim(
539
+ hasTools
540
+ ? "System measured; Conversation estimated; Tools is the remainder."
541
+ : "System measured; Conversation estimated from stored history.",
542
+ )}`,
543
+ );
544
+ ctx.reprompt();
545
+ },
546
+ });
547
+
396
548
  registerCommand({
397
549
  name: "reset",
398
550
  description: "Start a fresh session",