@monotykamary/pi-tps 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -72,6 +72,11 @@ interface ToolExecutionStartEvent {
72
72
  /** Minimum gap between token updates to count as a stall (ms) */
73
73
  const STALL_THRESHOLD_MS = 500;
74
74
 
75
+ /** Event name emitted by pi-neuralwatt-provider per turn with energy-billed cost data. */
76
+ const NEURALWATT_ENERGY_EVENT = 'neuralwatt:turn-energy';
77
+ /** Cap the per-turn billed-cost cache (turnIndex → costUsd) to avoid unbounded growth in long sessions. */
78
+ const NEURALWATT_ENERGY_CACHE_MAX = 32;
79
+
75
80
  // ─── Data types ─────────────────────────────────────────────────────────────
76
81
 
77
82
  /** Structured telemetry persisted per turn in the session JSONL */
@@ -96,6 +101,15 @@ interface TurnTelemetry {
96
101
  cacheWrite: number;
97
102
  total: number;
98
103
  } | null;
104
+ /**
105
+ * Blended $/M-tokens rate shown in the notification banner: the per-turn
106
+ * cost divided by (totalTokens / 1_000_000). Cost source is the Neuralwatt
107
+ * billed cost when available (energy-based), otherwise the list-price
108
+ * compute cost from `cost.total`. null when neither is usable or tokens
109
+ * are zero. Purely derived — recomputed in composeDisplayString, not a
110
+ * duplicate of the cost block.
111
+ */
112
+ rateUsdPerMTokens: number | null;
99
113
  timestamp: number;
100
114
  }
101
115
 
@@ -118,10 +132,59 @@ interface TurnTiming {
118
132
  messageCount: number;
119
133
  isToolCall: boolean; // tool_execution_start fired during this turn
120
134
  isPrimaryBranch: boolean; // TPS came from primary-branch (reliable) measurement
135
+ turnStartTimestamp: number; // wall-clock ms at turn_start, for correlating session energy entries
121
136
  }
122
137
 
123
138
  // ─── Helpers ────────────────────────────────────────────────────────────────
124
139
 
140
+ /**
141
+ * Compute the blended $/M-tokens rate: costUsd / (totalTokens / 1_000_000).
142
+ * The single calculation backing the banner's $/M-tokens field. Returns null
143
+ * when the inputs can't produce a meaningful rate (zero/negative tokens,
144
+ * non-finite or negative cost).
145
+ */
146
+ function computeRateUsdPerM(costUsd: number | null, totalTokens: number): number | null {
147
+ if (costUsd === null || !Number.isFinite(costUsd) || costUsd < 0) return null;
148
+ if (!Number.isFinite(totalTokens) || totalTokens <= 0) return null;
149
+ const rate = costUsd / (totalTokens / 1_000_000);
150
+ if (!Number.isFinite(rate) || rate < 0) return null;
151
+ return Math.round(rate * 100) / 100;
152
+ }
153
+
154
+ /**
155
+ * Fallback: find a Neuralwatt energy entry that was persisted in this session
156
+ * during the current turn. This covers the case where pi-neuralwatt-provider
157
+ * loads before pi-tps, so the energy entry exists before our turn_end runs,
158
+ * but the live event cache missed (e.g. malformed payload, event ordering, or
159
+ * a race).
160
+ *
161
+ * Scans backward from the most recent entry and requires the entry timestamp
162
+ * to be >= turnStartTimestamp so a previous turn's energy data is not reused.
163
+ */
164
+ function findEnergyCostFromSession(
165
+ ctx: ExtensionContext,
166
+ turnStartTimestamp: number
167
+ ): number | null {
168
+ if (!ctx.sessionManager?.getEntries) return null;
169
+
170
+ const entries = ctx.sessionManager.getEntries();
171
+ for (let i = entries.length - 1; i >= 0; i--) {
172
+ const e = entries[i];
173
+ if (e.type !== 'custom' || e.customType !== 'neuralwatt-energy') continue;
174
+
175
+ const entryTimestamp = typeof e.timestamp === 'number' ? e.timestamp : 0;
176
+ if (entryTimestamp < turnStartTimestamp) return null;
177
+
178
+ const data = e.data as Record<string, unknown> | null | undefined;
179
+ if (!data) continue;
180
+
181
+ const costUsd = typeof data.cost_usd === 'number' ? data.cost_usd : null;
182
+ if (costUsd !== null && Number.isFinite(costUsd) && costUsd >= 0) return costUsd;
183
+ }
184
+
185
+ return null;
186
+ }
187
+
125
188
  function isAssistantMessage(message: unknown): message is AssistantMessage {
126
189
  if (!message || typeof message !== 'object') return false;
127
190
  const msg = message as Record<string, unknown>;
@@ -236,6 +299,9 @@ function composeDisplayString(t: TurnTelemetry): string {
236
299
  const stallStr = formatDuration(t.timing.stallMs / 1000);
237
300
  parts.push(`stall ${stallStr}×${t.timing.stallCount}`);
238
301
  }
302
+ if (t.rateUsdPerMTokens != null) {
303
+ parts.push(`$${t.rateUsdPerMTokens.toFixed(2)}/M`);
304
+ }
239
305
  return parts.join(' · ');
240
306
  }
241
307
 
@@ -243,7 +309,11 @@ function composeDisplayString(t: TurnTelemetry): string {
243
309
  * Build structured TurnTelemetry from accumulated turn timing.
244
310
  * Returns null if the turn had no meaningful LLM output.
245
311
  */
246
- function buildTelemetry(timing: TurnTiming, turnEndMs: number): TurnTelemetry | null {
312
+ function buildTelemetry(
313
+ timing: TurnTiming,
314
+ turnEndMs: number,
315
+ billedCost: number | null = null
316
+ ): TurnTelemetry | null {
247
317
  let input = 0;
248
318
  let output = 0;
249
319
  let cacheRead = 0;
@@ -415,6 +485,13 @@ function buildTelemetry(timing: TurnTiming, turnEndMs: number): TurnTelemetry |
415
485
  isPrimaryBranch = false;
416
486
  }
417
487
 
488
+ // Single blended $/M-tokens rate for the banner. Cost source: Neuralwatt's
489
+ // billed cost when present (energy-based, what the user actually pays),
490
+ // otherwise the list-price compute cost from message.usage.cost. Never both
491
+ // — billedCost wins outright when present, so there's no double-counting.
492
+ const effectiveCost = billedCost ?? (hasCost ? costTotal : null);
493
+ const rateUsdPerMTokens = computeRateUsdPerM(effectiveCost, totalTokens);
494
+
418
495
  return {
419
496
  model,
420
497
  tokens: { input, output, cacheRead, cacheWrite, total: totalTokens },
@@ -438,6 +515,7 @@ function buildTelemetry(timing: TurnTiming, turnEndMs: number): TurnTelemetry |
438
515
  total: costTotal,
439
516
  }
440
517
  : null,
518
+ rateUsdPerMTokens,
441
519
  timestamp: Date.now(),
442
520
  };
443
521
  }
@@ -455,6 +533,34 @@ export default function tpsExtension(pi: ExtensionAPI) {
455
533
  // Cached session entries for argument completion (captured on session_start / session_tree)
456
534
  let cachedEntries: Array<{ type?: string; customType?: string; data?: unknown }> = [];
457
535
 
536
+ // ── Neuralwatt per-turn billed-cost integration ─────────────────────────
537
+ // pi-neuralwatt-provider emits NEURALWATT_ENERGY_EVENT in its turn_end handler
538
+ // (after its tee reader drains) with { costUsd, energyJoules, turnIndex }. We stash
539
+ // only the one number we need — billed costUsd — keyed by turnIndex, and read it
540
+ // synchronously in our own turn_end. pi awaits turn_end handlers sequentially in
541
+ // registration order, so if the neuralwatt provider was registered before us the
542
+ // event has already fired and the cache hit is immediate; if after us we miss it
543
+ // for that one turn and fall back to the list-price compute rate (the neuralwatt
544
+ // widget still shows billing separately). Zero latency, no awaiting, no duplicate
545
+ // capture — the raw energy/cost records stay solely in the neuralwatt provider's
546
+ // own session entries. Non-Neuralwatt turns never emit, so this stays empty.
547
+ const neuralwattBilledCostByTurn = new Map<number, number>();
548
+
549
+ pi.events?.on(NEURALWATT_ENERGY_EVENT, (payload: unknown) => {
550
+ if (!payload || typeof payload !== 'object') return;
551
+ const p = payload as Record<string, unknown>;
552
+ const turnIndex = typeof p.turnIndex === 'number' ? p.turnIndex : null;
553
+ const costUsd = typeof p.costUsd === 'number' ? p.costUsd : null;
554
+ if (turnIndex === null || costUsd === null) return; // require turnIndex correlation
555
+ neuralwattBilledCostByTurn.set(turnIndex, costUsd);
556
+ // Bound the cache (oldest = lowest turnIndex)
557
+ if (neuralwattBilledCostByTurn.size > NEURALWATT_ENERGY_CACHE_MAX) {
558
+ let oldest = Infinity;
559
+ for (const k of neuralwattBilledCostByTurn.keys()) if (k < oldest) oldest = k;
560
+ if (oldest !== Infinity) neuralwattBilledCostByTurn.delete(oldest);
561
+ }
562
+ });
563
+
458
564
  // ── Rehydration ─────────────────────────────────────────────────────────
459
565
 
460
566
  /**
@@ -508,6 +614,7 @@ export default function tpsExtension(pi: ExtensionAPI) {
508
614
  pi.on('turn_start', (_event: TurnStartEvent) => {
509
615
  currentTiming = {
510
616
  turnStartMs: performance.now(),
617
+ turnStartTimestamp: typeof _event.timestamp === 'number' ? _event.timestamp : Date.now(),
511
618
  lastUpdateMs: performance.now(),
512
619
  firstTokenMs: null,
513
620
  currentMessageStartMs: null,
@@ -614,15 +721,33 @@ export default function tpsExtension(pi: ExtensionAPI) {
614
721
 
615
722
  // ── Persist telemetry ───────────────────────────────────────────────────
616
723
 
617
- // Calculate, display, and persist telemetry at the end of each LLM turn
618
- pi.on('turn_end', (_event: TurnEndEvent, ctx: ExtensionContext) => {
724
+ // Calculate, display, and persist telemetry at the end of each LLM turn.
725
+ // Synchronous: the Neuralwatt billed cost (if any) was stashed by our
726
+ // NEURALWATT_ENERGY_EVENT listener, keyed by turnIndex, and is read here
727
+ // synchronously. No awaiting, no added latency to turn dispatch.
728
+ pi.on('turn_end', (event: TurnEndEvent, ctx: ExtensionContext) => {
619
729
  if (!currentTiming) return;
620
730
 
621
731
  const timing = currentTiming;
622
732
  currentTiming = null;
623
733
 
624
734
  const turnEndMs = performance.now();
625
- const telemetry = buildTelemetry(timing, turnEndMs);
735
+
736
+ // Pick the effective cost for the blended $/M-tokens rate: Neuralwatt's
737
+ // billed cost when present, otherwise the list-price compute cost from
738
+ // message.usage.cost. The live event is the primary source; if it missed,
739
+ // we read the persisted neuralwatt-energy session entry. Only if neither
740
+ // energy source is available do we fall back to the compute rate.
741
+ // Prefer the live energy event, fall back to the persisted energy entry,
742
+ // and finally to the list-price compute cost.
743
+ let billedCost = neuralwattBilledCostByTurn.get(event.turnIndex) ?? null;
744
+ neuralwattBilledCostByTurn.delete(event.turnIndex);
745
+
746
+ if (billedCost === null && timing.turnStartTimestamp) {
747
+ billedCost = findEnergyCostFromSession(ctx, timing.turnStartTimestamp);
748
+ }
749
+
750
+ const telemetry = buildTelemetry(timing, turnEndMs, billedCost);
626
751
  if (!telemetry) return;
627
752
 
628
753
  // ── Dynamic TPS cap ────────────────────────────────────────────────