pennyrouter 0.3.19 → 0.3.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,575 @@
1
+ """Reads token counts out of the transcripts Claude Code and Codex write as they run, plus
2
+ the per-step records OpenCode stores in SQLite.
3
+
4
+ The JSONL transcripts and SQLite store land ahead of any gateway round trip and need no proxy in
5
+ front of the session. Only usage numbers and model metadata are read; message content is never
6
+ parsed.
7
+
8
+ The two formats differ in kind: Claude writes per-request counts and repeats a message's usage
9
+ block on every streamed line, so totals are summed over distinct message ids. Codex writes a
10
+ running cumulative total, so its last event wins.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ import sqlite3
18
+ import time
19
+ from dataclasses import dataclass, field
20
+ from pathlib import Path
21
+
22
+ import pricing
23
+
24
+ # Set this to False to restore the original parent-transcript-only behavior if a harness
25
+ # changes its child-session format or a comparison needs to exclude subagent work.
26
+ INCLUDE_SUBAGENT_USAGE = True
27
+
28
+
29
+ @dataclass
30
+ class Usage:
31
+ cache_read: int = 0
32
+ cache_write: int = 0
33
+ # Split for cost accuracy; cache_write above is their sum, shown as one row in the HUD.
34
+ write_5m: int = 0
35
+ write_1h: int = 0
36
+ fresh_input: int = 0
37
+ output: int = 0
38
+ context_used: int = 0
39
+ context_window: int | None = None
40
+ model: str = ""
41
+
42
+ @property
43
+ def total(self) -> int:
44
+ return self.cache_read + self.cache_write + self.fresh_input + self.output
45
+
46
+ @property
47
+ def context_fraction(self) -> float | None:
48
+ if not self.context_window:
49
+ return None
50
+ return min(1.0, self.context_used / self.context_window)
51
+
52
+
53
+ @dataclass
54
+ class UsageCall:
55
+ """One billable model call as recorded by a harness transcript."""
56
+
57
+ timestamp: str | None
58
+ recorded_at: float
59
+ model: str
60
+ cache_read: int
61
+ cache_write: int
62
+ write_5m: int
63
+ write_1h: int
64
+ fresh_input: int
65
+ output: int
66
+ reasoning_output: int = 0
67
+
68
+ @property
69
+ def total(self) -> int:
70
+ # Codex reports reasoning_output_tokens as a subset of output_tokens, so it is useful
71
+ # detail but must not be added to the billable total a second time.
72
+ return self.cache_read + self.cache_write + self.fresh_input + self.output
73
+
74
+
75
+ @dataclass
76
+ class _Progress:
77
+ offset: int = 0
78
+ usage: Usage = field(default_factory=Usage)
79
+ counted: set = field(default_factory=set)
80
+
81
+
82
+ class UsageReader:
83
+ """Follows one transcript, parsing only what was appended since the last read."""
84
+
85
+ def __init__(self, harness: str, path: str | None = None, on_call=None):
86
+ self.harness = harness
87
+ self.path = path
88
+ self.on_call = on_call
89
+ self._progress: dict[str, _Progress] = {}
90
+
91
+ def _emit(self, call: UsageCall) -> None:
92
+ if self.on_call:
93
+ self.on_call(call)
94
+
95
+ def read(self, path: str | None = None) -> Usage | None:
96
+ target = path or self.path
97
+ if self.harness == "opencode" and target and "#" in target:
98
+ state = self._progress.setdefault(target, _Progress())
99
+ return self._read_opencode(target, state)
100
+ if not target or not os.path.exists(target):
101
+ return None
102
+ state = self._progress.setdefault(target, _Progress())
103
+
104
+ size = os.path.getsize(target)
105
+ # A shorter file is a different conversation reusing the path; start over.
106
+ if size < state.offset:
107
+ self._progress[target] = state = _Progress()
108
+
109
+ with open(target, "rb") as handle:
110
+ handle.seek(state.offset)
111
+ chunk = handle.read()
112
+
113
+ if chunk:
114
+ # Stop at the last newline so a partially written line is re-read once complete.
115
+ cut = chunk.rfind(b"\n")
116
+ if cut >= 0:
117
+ state.offset += cut + 1
118
+ apply = self._apply_claude if self.harness == "claude-code" else self._apply_codex
119
+ for raw in chunk[: cut + 1].splitlines():
120
+ if raw.strip():
121
+ apply(raw, state)
122
+ return state.usage
123
+
124
+ def _read_opencode(self, target: str, state: _Progress) -> Usage | None:
125
+ """Read OpenCode's per-step usage records from its SQLite store."""
126
+ database, session_id = target.split("#", 1)
127
+ try:
128
+ db = sqlite3.connect(f"file:{database}?mode=ro", uri=True, timeout=0.2)
129
+ row = db.execute("select model from session where id = ?", (session_id,)).fetchone()
130
+ if not row:
131
+ db.close()
132
+ return None
133
+ try:
134
+ model = str((json.loads(row[0] or "{}") or {}).get("id") or "")
135
+ except ValueError:
136
+ model = ""
137
+ state.usage.model = model
138
+ for part_id, raw in db.execute(
139
+ "select id, data from part where session_id = ? order by time_created, id",
140
+ (session_id,),
141
+ ):
142
+ if part_id in state.counted:
143
+ continue
144
+ try:
145
+ obj = json.loads(raw)
146
+ except (TypeError, ValueError):
147
+ state.counted.add(part_id)
148
+ continue
149
+ if obj.get("type") != "step-finish" or not isinstance(obj.get("tokens"), dict):
150
+ continue
151
+ tokens = obj["tokens"]
152
+ cache = tokens.get("cache") or {}
153
+ read = int(cache.get("read") or 0)
154
+ write = int(cache.get("write") or 0)
155
+ fresh = int(tokens.get("input") or 0)
156
+ output = int(tokens.get("output") or 0)
157
+ state.usage.cache_read += read
158
+ state.usage.cache_write += write
159
+ state.usage.write_5m += write
160
+ state.usage.fresh_input += fresh
161
+ state.usage.output += output
162
+ state.usage.context_used = fresh + read + write
163
+ state.counted.add(part_id)
164
+ self._emit(UsageCall(
165
+ timestamp=None, recorded_at=time.time(), model=model,
166
+ cache_read=read, cache_write=write, write_5m=write, write_1h=0,
167
+ fresh_input=fresh, output=output,
168
+ reasoning_output=int(tokens.get("reasoning") or 0),
169
+ ))
170
+ db.close()
171
+ return state.usage
172
+ except (OSError, sqlite3.Error, ValueError):
173
+ return state.usage if state.usage.model else None
174
+
175
+ def read_many(self, paths: list[str]) -> Usage | None:
176
+ """Read a root transcript and any related child transcripts as one usage stream."""
177
+ if not paths:
178
+ return None
179
+ usages = [self.read(path) for path in paths]
180
+ present = [usage for usage in usages if usage is not None]
181
+ if not present:
182
+ return None
183
+ combined = Usage(model=present[-1].model,
184
+ context_window=present[-1].context_window)
185
+ for usage in present:
186
+ combined.cache_read += usage.cache_read
187
+ combined.cache_write += usage.cache_write
188
+ combined.write_5m += usage.write_5m
189
+ combined.write_1h += usage.write_1h
190
+ combined.fresh_input += usage.fresh_input
191
+ combined.output += usage.output
192
+ combined.context_used = max(combined.context_used, usage.context_used)
193
+ if usage.model:
194
+ combined.model = usage.model
195
+ if usage.context_window:
196
+ combined.context_window = usage.context_window
197
+ return combined
198
+
199
+ def _apply_claude(self, raw: bytes, state: _Progress) -> None:
200
+ try:
201
+ obj = json.loads(raw)
202
+ except (ValueError, UnicodeDecodeError):
203
+ return
204
+ message = obj.get("message")
205
+ if not isinstance(message, dict):
206
+ return
207
+ usage = message.get("usage")
208
+ if not isinstance(usage, dict):
209
+ return
210
+
211
+ mid = message.get("id")
212
+ if mid:
213
+ if mid in state.counted:
214
+ return
215
+ state.counted.add(mid)
216
+
217
+ read = int(usage.get("cache_read_input_tokens") or 0)
218
+ write = int(usage.get("cache_creation_input_tokens") or 0)
219
+ creation = usage.get("cache_creation") or {}
220
+ write_5m = int(creation.get("ephemeral_5m_input_tokens") or 0)
221
+ write_1h = int(creation.get("ephemeral_1h_input_tokens") or 0)
222
+ # Older responses omit the split; treat an unsplit write as the (cheaper) 5-minute tier
223
+ # rather than silently dropping it, since 0/0 there would undercount the total.
224
+ if not write_5m and not write_1h and write:
225
+ write_5m = write
226
+ fresh = int(usage.get("input_tokens") or 0)
227
+
228
+ state.usage.cache_read += read
229
+ state.usage.cache_write += write
230
+ state.usage.write_5m += write_5m
231
+ state.usage.write_1h += write_1h
232
+ state.usage.fresh_input += fresh
233
+ state.usage.output += int(usage.get("output_tokens") or 0)
234
+ state.usage.context_used = read + write + fresh
235
+ if message.get("model"):
236
+ state.usage.model = message["model"]
237
+ # Anthropic records no window in the transcript, so it is inferred from the model.
238
+ if state.usage.context_window is None:
239
+ state.usage.context_window = pricing.context_window(message["model"])
240
+ # The long-context variants are a runtime setting rather than a distinct model id, so an
241
+ # observed context larger than the assumed window is the only signal one is in use.
242
+ if state.usage.context_window and state.usage.context_used > state.usage.context_window:
243
+ state.usage.context_window = pricing.next_window_above(state.usage.context_used)
244
+
245
+ self._emit(UsageCall(
246
+ timestamp=obj.get("timestamp"), recorded_at=time.time(),
247
+ model=str(message.get("model") or state.usage.model),
248
+ cache_read=read, cache_write=write, write_5m=write_5m,
249
+ write_1h=write_1h, fresh_input=fresh,
250
+ output=int(usage.get("output_tokens") or 0),
251
+ reasoning_output=0,
252
+ ))
253
+
254
+ def _apply_codex(self, raw: bytes, state: _Progress) -> None:
255
+ try:
256
+ obj = json.loads(raw)
257
+ except (ValueError, UnicodeDecodeError):
258
+ return
259
+ payload = obj.get("payload")
260
+ if not isinstance(payload, dict):
261
+ return
262
+ # Codex records its line kind on the envelope; only usage carries it in the payload.
263
+ if obj.get("type") == "turn_context" and payload.get("model"):
264
+ state.usage.model = payload["model"]
265
+ return
266
+ if payload.get("type") != "token_count":
267
+ return
268
+ info = payload.get("info")
269
+ if not isinstance(info, dict):
270
+ return
271
+ totals = info.get("total_token_usage")
272
+ if not isinstance(totals, dict):
273
+ return
274
+
275
+ # Codex occasionally emits the same cumulative snapshot again with an empty
276
+ # last_token_usage block. It is status, not another model call.
277
+ fingerprint = tuple(int(totals.get(key) or 0) for key in (
278
+ "input_tokens", "cached_input_tokens", "cache_write_input_tokens",
279
+ "output_tokens", "reasoning_output_tokens",
280
+ ))
281
+ if fingerprint in state.counted:
282
+ return
283
+ state.counted.add(fingerprint)
284
+
285
+ cached = int(totals.get("cached_input_tokens") or 0)
286
+ write = int(totals.get("cache_write_input_tokens") or 0)
287
+ # Codex reports cached reads and writes inside input_tokens; the HUD keeps
288
+ # them as separate, non-overlapping buckets.
289
+ state.usage.cache_read = cached
290
+ state.usage.cache_write = write
291
+ # Codex has no 1-hour cache tier, so its entire write is the 5-minute rate.
292
+ state.usage.write_5m = write
293
+ state.usage.fresh_input = max(
294
+ 0, int(totals.get("input_tokens") or 0) - cached - write
295
+ )
296
+ # reasoning_output_tokens is a subset of output_tokens, not an extra bucket.
297
+ state.usage.output = int(totals.get("output_tokens") or 0)
298
+
299
+ window = info.get("model_context_window")
300
+ if isinstance(window, int) and window > 0:
301
+ state.usage.context_window = window
302
+ last = info.get("last_token_usage")
303
+ if isinstance(last, dict):
304
+ used = int(last.get("input_tokens") or 0)
305
+ # A duplicate event can report zero; it would otherwise blank the gauge.
306
+ if used > 0:
307
+ state.usage.context_used = used
308
+ last_cached = int(last.get("cached_input_tokens") or 0)
309
+ last_write = int(last.get("cache_write_input_tokens") or 0)
310
+ self._emit(UsageCall(
311
+ timestamp=obj.get("timestamp"), recorded_at=time.time(),
312
+ model=state.usage.model,
313
+ cache_read=last_cached, cache_write=last_write,
314
+ write_5m=last_write, write_1h=0,
315
+ # last_token_usage.input_tokens includes both cached reads and
316
+ # cache writes, just like total_token_usage.input_tokens.
317
+ fresh_input=max(0, used - last_cached - last_write),
318
+ output=int(last.get("output_tokens") or 0),
319
+ reasoning_output=int(last.get("reasoning_output_tokens") or 0),
320
+ ))
321
+
322
+
323
+ def session_root(harness: str) -> Path | None:
324
+ home = Path.home()
325
+ if harness == "claude-code":
326
+ return home / ".claude/projects"
327
+ if harness == "codex":
328
+ return home / ".codex/sessions"
329
+ if harness == "opencode":
330
+ return home / ".local/share/opencode/opencode.db"
331
+ return None
332
+
333
+
334
+ def flattened(directory: str) -> str:
335
+ """Claude Code names a project directory after its cwd with separators turned to dashes."""
336
+ return str(directory).replace("/", "-")
337
+
338
+
339
+ def newest_session(harness: str, project_dir: str | None = None,
340
+ since: float | None = None, exclude: set | None = None,
341
+ provider: bool | None = None) -> str | None:
342
+ """The transcript to follow.
343
+
344
+ A project directory narrows the search to one working directory. `since` restricts it to
345
+ sessions created after a moment in time, and `exclude` rules out transcripts another pane has
346
+ already claimed, which is what separates two panes running the same harness there.
347
+
348
+ `provider` (Codex only) selects by how the session was launched: True for one routed through
349
+ PennyRouter, False for a native one. OpenCode owns provider selection in its own config, so
350
+ its sessions are selected by directory/time regardless of provider."""
351
+ root = session_root(harness)
352
+ if harness == "opencode":
353
+ if not root or not root.exists():
354
+ return None
355
+ excluded = exclude or set()
356
+ try:
357
+ db = sqlite3.connect(f"file:{root}?mode=ro", uri=True, timeout=0.2)
358
+ query = "select id, directory, time_created, model from session"
359
+ params = ()
360
+ if project_dir:
361
+ query += " where directory = ?"
362
+ params = (str(project_dir),)
363
+ query += " order by time_created desc"
364
+ for session_id, directory, created_ms, model_raw in db.execute(query, params):
365
+ target = f"{root}#{session_id}"
366
+ if target in excluded or (since is not None and created_ms / 1000 <= since):
367
+ continue
368
+ try:
369
+ model = json.loads(model_raw or "{}") or {}
370
+ except ValueError:
371
+ model = {}
372
+ is_penny = str(model.get("providerID") or "").lower() == "pennyrouter"
373
+ if provider is not None and is_penny != provider:
374
+ continue
375
+ db.close()
376
+ return target
377
+ db.close()
378
+ except sqlite3.Error:
379
+ return None
380
+ return None
381
+ if not root or not root.exists():
382
+ return None
383
+ if project_dir and harness == "claude-code":
384
+ scoped = root / flattened(project_dir)
385
+ if scoped.exists():
386
+ root = scoped
387
+ newest, newest_mtime = None, -1.0
388
+ excluded = exclude or set()
389
+ for path in root.rglob("*.jsonl"):
390
+ if str(path) in excluded:
391
+ continue
392
+ try:
393
+ mtime = path.stat().st_mtime
394
+ except OSError:
395
+ continue
396
+ if mtime <= newest_mtime:
397
+ continue
398
+ if since is not None:
399
+ try:
400
+ if path.stat().st_ctime < since:
401
+ continue
402
+ except OSError:
403
+ continue
404
+ if harness == "claude-code" and provider is False and _claude_is_penny(path):
405
+ # Legacy fallback only: compare normally assigns Claude's session id before launch.
406
+ # If a caller does use discovery, never adopt a transcript Penny explicitly claimed.
407
+ continue
408
+ if harness == "codex" and (project_dir or provider is not None):
409
+ # Codex keeps every session in one tree, so its working directory comes from the
410
+ # session header rather than the path. That header also names the provider, which
411
+ # is what separates a penny pane's session from a native one's in the same
412
+ # directory — neither timing nor the path can tell those apart.
413
+ meta = codex_session_meta(path)
414
+ if project_dir and meta.get("cwd") != str(project_dir):
415
+ continue
416
+ if provider is not None and _is_penny(meta) != provider:
417
+ continue
418
+ newest, newest_mtime = str(path), mtime
419
+ return newest
420
+
421
+
422
+ def _is_penny(meta: dict) -> bool:
423
+ """Whether a Codex session was launched through PennyRouter."""
424
+ if str(meta.get("model_provider") or "").lower() == "pennyrouter":
425
+ return True
426
+ return _codex_is_penny(meta)
427
+
428
+
429
+ def _codex_is_penny(meta: dict) -> bool:
430
+ """Whether the gateway claimed this Codex session as Penny-routed.
431
+
432
+ A subscription session reaches Codex's built-in OpenAI provider, which is what gives it the
433
+ native tool runtime, so its header names `openai` exactly like a native session's. Codex
434
+ sends the rollout's own conversation UUID on the WebSocket handshake, and the gateway
435
+ records it there; matching on that id keeps the two apart without timing or ordering.
436
+ """
437
+ session_id = str(meta.get("session_id") or "")
438
+ if not session_id:
439
+ return False
440
+ data = Path.home() / ".local" / "share" / "pennyrouter"
441
+ try:
442
+ return (data / f"codex-session-{session_id}").exists()
443
+ except OSError:
444
+ return False
445
+
446
+
447
+ def _claude_is_penny(path) -> bool:
448
+ """Whether Penny's exact Claude session-id handoff claims this transcript.
449
+
450
+ Transcript text is not an ownership marker: native sessions can mention PennyRouter through
451
+ MCP instructions, and native proxy sessions can use provider-qualified model names. Penny's
452
+ title-session files contain the actual Claude session ids it launched.
453
+ """
454
+ session_id = Path(path).stem
455
+ data = Path.home() / ".local" / "share" / "pennyrouter"
456
+ try:
457
+ entries = data.glob("title-session-*")
458
+ for entry in entries:
459
+ try:
460
+ if entry.read_text().strip() == session_id:
461
+ return True
462
+ except OSError:
463
+ continue
464
+ except OSError:
465
+ pass
466
+ return False
467
+
468
+
469
+ def transcript_for_session(session_id: str, project_dir: str | None = None) -> str | None:
470
+ """The transcript file belonging to a Claude Code session id.
471
+
472
+ Claude Code names a transcript after its session id and records the same id in every line,
473
+ so the path is derived directly and the field is only consulted if that file is absent.
474
+ """
475
+ root = session_root("claude-code")
476
+ if not root or not session_id:
477
+ return None
478
+ search = root
479
+ if project_dir:
480
+ scoped = root / flattened(project_dir)
481
+ if scoped.exists():
482
+ search = scoped
483
+ direct = search / f"{session_id}.jsonl"
484
+ if direct.exists():
485
+ return str(direct)
486
+ for path in sorted(search.rglob("*.jsonl"), key=lambda p: -p.stat().st_mtime):
487
+ try:
488
+ with open(path) as handle:
489
+ first = handle.readline()
490
+ except OSError:
491
+ continue
492
+ try:
493
+ if json.loads(first).get("sessionId") == session_id:
494
+ return str(path)
495
+ except ValueError:
496
+ continue
497
+ return None
498
+
499
+
500
+ def codex_session_meta(path) -> dict:
501
+ """The session_meta payload Codex writes as a session's first line.
502
+
503
+ It carries both the working directory and the provider the session was launched against,
504
+ which is what tells a penny pane's rollout apart from a native one's in the same directory.
505
+ """
506
+ try:
507
+ with open(path, "rb") as handle:
508
+ first = handle.readline()
509
+ return json.loads(first).get("payload") or {}
510
+ except (OSError, ValueError):
511
+ return {}
512
+
513
+
514
+ def related_transcripts(harness: str, root_path: str | None) -> list[str]:
515
+ """Return a root transcript plus its known subagent descendants.
516
+
517
+ Claude stores sidechains below the root transcript. Codex stores each thread beside the
518
+ others and records the parent id in session_meta, so lineage must be resolved explicitly.
519
+ """
520
+ if not root_path or not INCLUDE_SUBAGENT_USAGE:
521
+ return [root_path] if root_path else []
522
+ root = Path(root_path)
523
+ if harness == "claude-code":
524
+ child_dir = root.with_suffix("") / "subagents"
525
+ children = sorted(str(path) for path in child_dir.rglob("*.jsonl")) if child_dir.exists() else []
526
+ return [root_path, *children]
527
+ if harness != "codex":
528
+ return [root_path]
529
+
530
+ root_meta = codex_session_meta(root_path)
531
+ root_ids = {str(root_meta.get("id") or root_meta.get("session_id") or "")}
532
+ root_ids.discard("")
533
+ discovered = [root_path]
534
+ session_root_path = session_root("codex")
535
+ if not session_root_path or not session_root_path.exists() or not root_ids:
536
+ return discovered
537
+ candidates = []
538
+ for path in session_root_path.rglob("*.jsonl"):
539
+ if str(path) == root_path:
540
+ continue
541
+ meta = codex_session_meta(path)
542
+ session_id = str(meta.get("id") or meta.get("session_id") or "")
543
+ parent_id = str(meta.get("parent_thread_id") or meta.get("forked_from_id") or "")
544
+ candidates.append((str(path), session_id, parent_id))
545
+ while True:
546
+ added = False
547
+ for path, session_id, parent_id in candidates:
548
+ if path in discovered or parent_id not in root_ids:
549
+ continue
550
+ discovered.append(path)
551
+ if session_id:
552
+ root_ids.add(session_id)
553
+ added = True
554
+ if not added:
555
+ break
556
+ return discovered
557
+
558
+
559
+ def codex_cwd(path) -> str | None:
560
+ """The working directory Codex recorded in a session's first line."""
561
+ return codex_session_meta(path).get("cwd")
562
+
563
+
564
+ def session_id_for_transcript(harness: str, path: str | None) -> str | None:
565
+ """Return the stable client session id stored by either harness."""
566
+ if not path:
567
+ return None
568
+ if harness == "claude-code":
569
+ return Path(path).stem
570
+ if harness == "codex":
571
+ meta = codex_session_meta(path)
572
+ return str(meta.get("session_id") or meta.get("id") or Path(path).stem.replace("rollout-", ""))
573
+ if harness == "opencode" and "#" in path:
574
+ return path.split("#", 1)[1]
575
+ return None