pennyrouter 0.3.19 → 0.3.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,430 @@
1
+ #!/usr/bin/env python3
2
+ """Runs one harness with a token readout pinned above it.
3
+
4
+ Invoked inside a tmux pane by penny-compare. The readout owns the top rows via a scroll region,
5
+ so the harness below scrolls normally and its own rendering is untouched.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ from dataclasses import asdict
12
+ import json
13
+ import os
14
+ import shutil
15
+ import subprocess
16
+ import sys
17
+ import threading
18
+ import time
19
+ import traceback
20
+ import uuid
21
+
22
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
23
+
24
+ import hud
25
+ import pricing
26
+ from usage import (UsageReader, newest_session, session_id_for_transcript, session_root,
27
+ related_transcripts, transcript_for_session)
28
+
29
+ # Each variant is the command that already launches that session correctly. The penny variants
30
+ # delegate to `penny`, which owns gateway auth and routing; reproducing that here is what made a
31
+ # pane prompt about a foreign upstream key.
32
+ VARIANTS = {
33
+ "claude": ("claude-code", ["claude"], "NATIVE"),
34
+ "penny-claude": ("claude-code", ["penny", "claude"], "PENNY"),
35
+ "codex": ("codex", ["codex"], "NATIVE"),
36
+ "penny-codex": ("codex", ["penny", "codex"], "PENNY"),
37
+ "opencode": ("opencode", ["opencode"], "NATIVE"),
38
+ "penny-opencode": ("opencode", ["penny", "opencode"], "PENNY"),
39
+ }
40
+
41
+ # Upstream env a NATIVE pane never inherits. The ambient shell may already point every
42
+ # Anthropic client somewhere; a NATIVE pane decides its own upstream, so it starts from a
43
+ # clean slate rather than whatever happens to be exported.
44
+ NATIVE_STRIP_ENV = ("ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN", "ANTHROPIC_API_KEY")
45
+
46
+ # With --native-proxy, the NATIVE pane routes through these instead of going direct.
47
+ NATIVE_PROXY_URL_ENV = "PENNY_COMPARE_NATIVE_BASE_URL"
48
+ NATIVE_PROXY_TOKEN_ENV = "PENNY_COMPARE_NATIVE_AUTH_TOKEN"
49
+
50
+
51
+ class Readout:
52
+ """Paints the fixed block at the top of the pane."""
53
+
54
+ def __init__(self, harness, title, delta_path=None, provider=None, label=None,
55
+ session_id=None, side=None, variant=None, pricing_profile="standard"):
56
+ self.harness = harness
57
+ # The pane's fixed role ("NATIVE"/"PENNY"). It keys the files the two panes use to
58
+ # coordinate, so it stays constant no matter what the readout is labelled.
59
+ self.title = title
60
+ # What the readout actually shows, which the user may rename.
61
+ self.label = label or title
62
+ self.delta_path = delta_path
63
+ # For Codex, whether this pane's session should be the PennyRouter-routed one.
64
+ self.provider = provider
65
+ # Claude accepts a caller-supplied session id, so its transcript can be selected exactly
66
+ # instead of inferred from timing, model names, or Penny's global title handoff files.
67
+ self.session_id = session_id
68
+ self.side = side or title.lower()
69
+ self.variant = variant
70
+ self.pricing_profile = pricing_profile
71
+ self.reader = UsageReader(harness, on_call=self._record_call)
72
+ self.call_cost = 0.0
73
+ self.call_cost_baseline = 0.0
74
+ self.counter = hud.CountUp()
75
+ # Counts recorded before the last reset; subtracted from what is shown.
76
+ self.baseline = (0, 0, 0, 0)
77
+ self.write_5m_baseline = 0
78
+ self.write_1h_baseline = 0
79
+ self.reset_pending = False
80
+ self._reset_at = 0.0
81
+ self.reported_error = False
82
+ self.path = None
83
+ # Only sessions created after the pane launched can be this pane's.
84
+ self.started = time.time()
85
+ self._preexisting = self._existing_sessions()
86
+ self.stop = threading.Event()
87
+ self.last = time.monotonic()
88
+ self._publish_metadata()
89
+
90
+ def _state_path(self, name):
91
+ state = os.environ.get("PENNY_COMPARE_STATE")
92
+ return os.path.join(state, name) if state else None
93
+
94
+ def _publish_metadata(self):
95
+ path = self._state_path(f"pane-{self.side}.json")
96
+ if not path:
97
+ return
98
+ payload = {
99
+ "side": self.side, "role": self.title, "label": self.label,
100
+ "variant": self.variant, "harness": self.harness, "cwd": os.getcwd(),
101
+ "pricing_profile": self.pricing_profile,
102
+ "session_id": self.session_id,
103
+ "transcript_path": self.path,
104
+ }
105
+ try:
106
+ with open(path, "w") as handle:
107
+ json.dump(payload, handle)
108
+ except OSError:
109
+ pass
110
+
111
+ def _record_call(self, call):
112
+ # The readout's cost is this running total, so it accrues for every call. Writing the
113
+ # per-call log needs a state directory; a session started without one still shows cost.
114
+ context_tokens = call.fresh_input + call.cache_read + call.cache_write
115
+ cost = pricing.cost(
116
+ call.model, call.fresh_input, call.cache_read, call.output,
117
+ write_5m=call.write_5m, write_1h=call.write_1h,
118
+ profile=self.pricing_profile, context_tokens=context_tokens,
119
+ )
120
+ self.call_cost += cost
121
+
122
+ path = self._state_path(f"calls-{self.side}.jsonl")
123
+ if not path:
124
+ return
125
+ payload = asdict(call)
126
+ payload["total"] = call.total
127
+ payload["pricing_profile"] = self.pricing_profile
128
+ rate_names = ("input", "cache_read", "cache_write_5m", "cache_write_1h", "output")
129
+ payload["context_tokens"] = context_tokens
130
+ payload["rates_per_million"] = dict(zip(
131
+ rate_names,
132
+ (rate * 1_000_000 for rate in pricing.rates_for(
133
+ call.model, self.pricing_profile, context_tokens)),
134
+ ))
135
+ payload["cost"] = cost
136
+ try:
137
+ with open(path, "a") as handle:
138
+ handle.write(json.dumps(payload, separators=(",", ":")) + "\n")
139
+ except OSError:
140
+ pass
141
+
142
+ def resolve(self):
143
+ """Pin this pane to the transcript its own harness created.
144
+
145
+ Both panes run in the same working directory, so the newest session there is ambiguous
146
+ between them, and resolving by timing alone can label the two backwards.
147
+
148
+ Claude is launched with an explicit session id, making both its native and Penny
149
+ transcript paths deterministic. Codex has no equivalent launch flag, but records the
150
+ provider in its session header, so each Codex pane takes the session matching how it was
151
+ started.
152
+ """
153
+ if self.path and os.path.exists(self.path):
154
+ return self.path
155
+
156
+ if self.session_id:
157
+ self.path = transcript_for_session(self.session_id, os.getcwd())
158
+ if self.path:
159
+ self._publish(self.path)
160
+ return self.path
161
+
162
+ # A transcript that already existed when the pane launched belongs to some other
163
+ # session, so require one created afterwards rather than adopting the newest on disk.
164
+ taken = self._peers() | self._preexisting
165
+ candidate = newest_session(self.harness, os.getcwd(),
166
+ since=self.started, exclude=taken,
167
+ provider=self.provider)
168
+ if candidate and not self._claim(candidate):
169
+ return None
170
+ self.path = candidate
171
+ if candidate:
172
+ self._publish(candidate)
173
+ return self.path
174
+
175
+ def _existing_sessions(self) -> set:
176
+ """Transcripts already on disk, which cannot belong to a session starting now."""
177
+ root = session_root(self.harness)
178
+ if not root or not root.exists():
179
+ return set()
180
+ return {str(p) for p in root.rglob("*.jsonl")}
181
+
182
+ def _publish(self, path):
183
+ """Record which transcript this pane resolved, so the other can rule it out."""
184
+ state = os.environ.get("PENNY_COMPARE_STATE")
185
+ if not state:
186
+ return
187
+ try:
188
+ self.session_id = session_id_for_transcript(self.harness, path)
189
+ with open(os.path.join(state, f"session-{self.side}"), "w") as handle:
190
+ handle.write(path)
191
+ self._publish_metadata()
192
+ except OSError:
193
+ pass
194
+
195
+ def _peers(self) -> set:
196
+ """Transcripts the other pane has already resolved."""
197
+ state = os.environ.get("PENNY_COMPARE_STATE")
198
+ if not state:
199
+ return set()
200
+ found = set()
201
+ try:
202
+ for name in os.listdir(state):
203
+ if not name.startswith("session-") or name.endswith(self.side):
204
+ continue
205
+ with open(os.path.join(state, name)) as handle:
206
+ value = handle.read().strip()
207
+ if value:
208
+ found.add(value)
209
+ except OSError:
210
+ pass
211
+ return found
212
+
213
+ def _claim(self, path):
214
+ """Record this transcript as taken, so the other pane looks past it."""
215
+ state = os.environ.get("PENNY_COMPARE_STATE")
216
+ if not state:
217
+ return True
218
+ marker = os.path.join(state, "claimed")
219
+ try:
220
+ with open(marker, "a+") as handle:
221
+ handle.seek(0)
222
+ taken = {line.strip() for line in handle if line.strip()}
223
+ if path in taken:
224
+ return False
225
+ handle.write(path + "\n")
226
+ return True
227
+ except OSError:
228
+ return True
229
+
230
+ def _check_reset(self):
231
+ """Zero the counters when the launcher signals a reset.
232
+
233
+ A comparison is only fair once both sides have a warm cache, so the run that matters is
234
+ usually not the first one. Rebasing on the counts already recorded lets the readout start
235
+ from zero without restarting either session.
236
+ """
237
+ state = os.environ.get("PENNY_COMPARE_STATE")
238
+ if not state:
239
+ return
240
+ marker = os.path.join(state, "reset")
241
+ try:
242
+ stamp = os.stat(marker).st_mtime
243
+ except OSError:
244
+ return
245
+ if stamp <= self._reset_at:
246
+ return
247
+ self._reset_at = stamp
248
+ self.reset_pending = True
249
+ self.counter.reset()
250
+
251
+ def paint(self):
252
+ self._check_reset()
253
+ path = self.resolve()
254
+ transcripts = related_transcripts(self.harness, path)
255
+ usage = self.reader.read_many(transcripts)
256
+ if usage is None:
257
+ targets = {"cache_read": 0, "cache_write": 0, "fresh_input": 0, "output": 0}
258
+ cost, model = 0.0, ""
259
+ else:
260
+ if self.reset_pending:
261
+ # Snapshot what is already recorded so counting restarts from here.
262
+ self.baseline = (usage.cache_read, usage.cache_write,
263
+ usage.fresh_input, usage.output)
264
+ self.write_5m_baseline = usage.write_5m
265
+ self.write_1h_baseline = usage.write_1h
266
+ self.call_cost_baseline = self.call_cost
267
+ self.reset_pending = False
268
+ targets = {
269
+ "cache_read": usage.cache_read - self.baseline[0],
270
+ "cache_write": usage.cache_write - self.baseline[1],
271
+ "fresh_input": usage.fresh_input - self.baseline[2],
272
+ "output": usage.output - self.baseline[3],
273
+ }
274
+ # Cost reflects what is displayed, not the session lifetime total, so it drops to
275
+ # zero on reset along with the counts above it.
276
+ # Costs are accumulated per call because GovCloud GPT rates change when that call's
277
+ # input context exceeds 272K. Pricing cumulative session totals would select the
278
+ # expensive tier merely because several smaller calls added up past the threshold.
279
+ cost = self.call_cost - self.call_cost_baseline
280
+ model = usage.model
281
+
282
+ # A transcript exists from session start but carries no usage until a turn completes, so
283
+ # counting begins at the first usage block rather than inheriting what was on disk.
284
+ waiting = sum(targets.values()) == 0
285
+
286
+ now = time.monotonic()
287
+ rolled = self.counter.advance(targets, now - self.last)
288
+ self.last = now
289
+
290
+ delta = None
291
+ if self.delta_path:
292
+ try:
293
+ with open(self.delta_path) as handle:
294
+ delta = handle.read().strip() or None
295
+ except OSError:
296
+ delta = None
297
+
298
+ lines = hud.render(self.label, model, rolled, cost, delta, waiting=waiting)
299
+ # Right-align to the pane. Erasing runs from the readout's own column to the right
300
+ # edge, so a value that shrinks leaves no digits behind while the harness output to
301
+ # the left of the block stays untouched.
302
+ size = shutil.get_terminal_size()
303
+ column = max(1, size.columns - hud.WIDTH - 1)
304
+
305
+ # The full-screen renderer repaints its entire frame while a turn streams. That can
306
+ # erase this overlay even when the HUD data has not changed, so the caller must repaint
307
+ # it continuously rather than treating an identical frame as already visible.
308
+ # DECSC/DECRC (ESC 7 / ESC 8) rather than ESC[s / ESC[u: the harness below is scrolling
309
+ # its own region continuously, and only the DEC pair restores the cursor relative to it.
310
+ # Reassert the scroll region first so the readout's rows stay excluded from that scroll
311
+ # even after the harness resets it.
312
+ out = ["\0337", f"\033[{hud.ROWS + 1};{size.lines}r"]
313
+ for index, line in enumerate(lines):
314
+ out.append(f"\033[{index + 1};{column}H\033[K{line}")
315
+ out.append("\0338")
316
+ sys.stdout.write("".join(out))
317
+ sys.stdout.flush()
318
+
319
+ # Publish this pane's numbers so the other pane can show a comparison.
320
+ if usage is not None:
321
+ state = os.environ.get("PENNY_COMPARE_STATE")
322
+ if state:
323
+ try:
324
+ with open(os.path.join(state, f"{self.title.lower()}.txt"), "w") as handle:
325
+ handle.write(f"{cost}\n{sum(targets.values())}\n{targets['cache_write']}\n")
326
+ except OSError:
327
+ pass
328
+
329
+ def loop(self):
330
+ while not self.stop.wait(1 / 30):
331
+ try:
332
+ self.paint()
333
+ except Exception:
334
+ # A painting error must not take down the harness, but silently swallowing it
335
+ # leaves the readout stuck looking like it is merely waiting. Report it once.
336
+ if not self.reported_error:
337
+ self.reported_error = True
338
+ traceback.print_exc(file=sys.stderr)
339
+
340
+
341
+ def main():
342
+ parser = argparse.ArgumentParser()
343
+ parser.add_argument("variant", choices=sorted(VARIANTS))
344
+ parser.add_argument("--delta-file", default=None)
345
+ parser.add_argument("--model", default=None,
346
+ help="start both panes on one model so the comparison is like for like")
347
+ parser.add_argument("--label", default=None,
348
+ help="name shown on this pane's readout (default: NATIVE or PENNY)")
349
+ parser.add_argument("--side", choices=("left", "right", "single"),
350
+ help="stable pane position used by the final report")
351
+ parser.add_argument("--pricing-profile", choices=("standard", "govcloud"),
352
+ default="standard")
353
+ parser.add_argument("--native-proxy", action="store_true",
354
+ help=f"route the NATIVE pane through {NATIVE_PROXY_URL_ENV} / "
355
+ f"{NATIVE_PROXY_TOKEN_ENV} instead of going direct")
356
+ args = parser.parse_args()
357
+
358
+ harness, command, title = VARIANTS[args.variant]
359
+ session_id = None
360
+ if harness == "claude-code":
361
+ session_id = str(uuid.uuid4())
362
+ command = command + ["--session-id", session_id]
363
+ if args.model:
364
+ # Passed to the harness rather than set in the environment, so Penny's own model
365
+ # configuration is not overridden on the penny side.
366
+ command = command + ["--model", args.model]
367
+
368
+ rows = shutil.get_terminal_size().lines
369
+ # Reserve the top rows for the readout and confine scrolling to what is left.
370
+ sys.stdout.write(f"\033[2J\033[{hud.ROWS + 1};{rows}r\033[{hud.ROWS + 1};1H")
371
+ sys.stdout.flush()
372
+
373
+ # Codex and OpenCode have no caller-assigned session id, so they identify their transcript by the provider
374
+ # recorded in the session header instead.
375
+ is_penny = command[0] == "penny"
376
+ readout = Readout(harness, title, args.delta_file,
377
+ # OpenCode owns provider/model selection inside its own TUI. The
378
+ # penny-opencode spelling is only a launch convenience, not a reliable
379
+ # provider identity like Codex's session header is.
380
+ provider=is_penny if harness == "codex" else None,
381
+ label=args.label,
382
+ session_id=session_id,
383
+ side=args.side,
384
+ variant=args.variant,
385
+ pricing_profile=args.pricing_profile)
386
+ thread = threading.Thread(target=readout.loop, daemon=True)
387
+ thread.start()
388
+
389
+ # The alt-screen renderer owns a full-screen buffer rather than appending to the main
390
+ # screen, so a pane that is not the active one still repaints as output arrives. The
391
+ # classic renderer leaves the inactive side blank until its turn ends, which reads as
392
+ # that harness being broken rather than merely unfocused.
393
+ env = dict(os.environ)
394
+ env.setdefault("CLAUDE_CODE_NO_FLICKER", "1")
395
+
396
+ if title == "NATIVE":
397
+ env = {k: v for k, v in env.items() if k not in NATIVE_STRIP_ENV}
398
+ if args.native_proxy:
399
+ proxy_url = os.environ.get(NATIVE_PROXY_URL_ENV)
400
+ proxy_token = os.environ.get(NATIVE_PROXY_TOKEN_ENV)
401
+ if proxy_url and proxy_token:
402
+ env["ANTHROPIC_BASE_URL"] = proxy_url
403
+ env["ANTHROPIC_AUTH_TOKEN"] = proxy_token
404
+ else:
405
+ # Missing config must not take the pane down: a pane that exits here leaves
406
+ # a one-sided comparison with no visible reason. Say so and run direct.
407
+ missing = " and ".join(
408
+ name for name, value in ((NATIVE_PROXY_URL_ENV, proxy_url),
409
+ (NATIVE_PROXY_TOKEN_ENV, proxy_token))
410
+ if not value
411
+ )
412
+ sys.stderr.write(
413
+ f"penny-compare: --native-proxy needs {missing}; "
414
+ f"running the NATIVE pane direct instead\n",
415
+ )
416
+
417
+ try:
418
+ subprocess.call(command, env=env)
419
+ except FileNotFoundError:
420
+ sys.stderr.write(f"penny-compare: {command[0]} not found on PATH\n")
421
+ return 1
422
+ finally:
423
+ readout.stop.set()
424
+ # Release the scroll region so the pane is left usable.
425
+ sys.stdout.write("\033[r\033[2J\033[H")
426
+ sys.stdout.flush()
427
+
428
+
429
+ if __name__ == "__main__":
430
+ sys.exit(main() or 0)