loopview 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. loopview/__init__.py +3 -0
  2. loopview/app.py +252 -0
  3. loopview/cli.py +128 -0
  4. loopview/cost/__init__.py +1 -0
  5. loopview/cost/pricing.json +62 -0
  6. loopview/cost/pricing.py +74 -0
  7. loopview/cost/split.py +228 -0
  8. loopview/demo_data/flagship.otlp.jsonl +13 -0
  9. loopview/devtools/__init__.py +1 -0
  10. loopview/devtools/dump_normalized.py +211 -0
  11. loopview/ingest/__init__.py +1 -0
  12. loopview/ingest/otlp.py +305 -0
  13. loopview/ingest/raw.py +47 -0
  14. loopview/live.py +154 -0
  15. loopview/normalize/__init__.py +1 -0
  16. loopview/normalize/adapters/__init__.py +14 -0
  17. loopview/normalize/adapters/base.py +87 -0
  18. loopview/normalize/adapters/gen_ai.py +252 -0
  19. loopview/normalize/adapters/generic.py +22 -0
  20. loopview/normalize/adapters/openinference.py +367 -0
  21. loopview/normalize/derived_tools.py +90 -0
  22. loopview/normalize/loop_nodes.py +193 -0
  23. loopview/normalize/normalizer.py +326 -0
  24. loopview/normalize/schema.py +161 -0
  25. loopview/normalize/transitions.py +106 -0
  26. loopview/py.typed +0 -0
  27. loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
  28. loopview/static/assets/index-CEGppBgk.css +1 -0
  29. loopview/static/assets/index-DzC96LiQ.js +21 -0
  30. loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
  31. loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
  32. loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
  33. loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
  34. loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
  35. loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
  36. loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
  37. loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
  38. loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
  39. loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
  40. loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
  41. loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
  42. loopview/static/index.html +13 -0
  43. loopview/store/__init__.py +1 -0
  44. loopview/store/capture.py +112 -0
  45. loopview/store/memory.py +165 -0
  46. loopview/tools/__init__.py +1 -0
  47. loopview/tools/report.py +347 -0
  48. loopview-0.1.0.dist-info/METADATA +72 -0
  49. loopview-0.1.0.dist-info/RECORD +51 -0
  50. loopview-0.1.0.dist-info/WHEEL +4 -0
  51. loopview-0.1.0.dist-info/entry_points.txt +3 -0
loopview/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """loopview: watch AI agents run as a live graph, from any OpenTelemetry trace."""
2
+
3
+ __version__ = "0.1.0"
loopview/app.py ADDED
@@ -0,0 +1,252 @@
1
+ """The FastAPI application.
2
+
3
+ One process serves everything on one port:
4
+ - the OTLP/HTTP trace receiver at POST /v1/traces,
5
+ - the JSON API and the live event stream under /api,
6
+ - the browser UI (static files built from ui/).
7
+
8
+ The UI only ever receives the normalized schema (normalize/schema.py), never
9
+ raw spans; /api/runs/{id}/spans exists for debugging.
10
+ """
11
+
12
+ import inspect
13
+ import json
14
+ import time
15
+ from collections.abc import AsyncIterator
16
+ from contextlib import asynccontextmanager
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ from fastapi import FastAPI, HTTPException, Request, Response
21
+ from fastapi.responses import HTMLResponse, StreamingResponse
22
+ from fastapi.staticfiles import StaticFiles
23
+
24
+ from loopview import __version__
25
+ from loopview.cost.pricing import Pricing
26
+ from loopview.ingest.otlp import (
27
+ JSON,
28
+ MAX_BODY_BYTES,
29
+ BodyTooLarge,
30
+ OtlpDecodeError,
31
+ UnsupportedContentType,
32
+ decode_body,
33
+ decompress,
34
+ encode_success_response,
35
+ spans_to_otlp_json,
36
+ )
37
+ from loopview.ingest.raw import RawSpan
38
+ from loopview.live import LiveHub, running
39
+ from loopview.normalize.schema import NormalizedRun, RunInfo
40
+ from loopview.store.capture import CapturedRequest, CaptureWriter, from_line, load_capture, to_line
41
+ from loopview.store.memory import SessionSummary, TraceStore
42
+ from loopview.tools.report import tools_report
43
+
44
+ # The UI build (vite) writes its output here. It is generated, not committed.
45
+ STATIC_DIR = Path(__file__).parent / "static"
46
+ # The flagship demo run, recorded from examples/demo (a copy of fixtures/flagship.otlp.jsonl).
47
+ DEMO_RECORDING = Path(__file__).parent / "demo_data" / "flagship.otlp.jsonl"
48
+
49
+ # Shown when someone runs the server from a source checkout without building the UI.
50
+ _MISSING_UI_PAGE = """<!doctype html>
51
+ <html><body style="font-family:sans-serif;background:#0a0a0b;color:#e5e5e5;padding:40px">
52
+ <h1>loopview is running</h1>
53
+ <p>The UI has not been built. Run <code>npm run build</code> in <code>ui/</code>.</p>
54
+ </body></html>"""
55
+
56
+
57
+ # Recent FastAPI versions trace themselves and, on startup, export to the OTLP
58
+ # endpoint in OTEL_* variables: the very ones users set to point their agents at
59
+ # loopview. loopview would then trace every request it receives and send the spans
60
+ # to itself, a loop without end. Older versions have no such option.
61
+ _NO_SELF_TELEMETRY = (
62
+ {"telemetry": {"tracing": False, "metrics": False, "logs": False, "auto_configure": False}}
63
+ if "telemetry" in inspect.signature(FastAPI.__init__).parameters
64
+ else {}
65
+ )
66
+
67
+
68
+ def create_app(
69
+ store: TraceStore | None = None,
70
+ capture: CaptureWriter | None = None,
71
+ static_dir: Path = STATIC_DIR,
72
+ pricing: Pricing | None = None,
73
+ ) -> FastAPI:
74
+ store = store if store is not None else TraceStore()
75
+ hub = LiveHub(store, pricing)
76
+ hub.mark(run.trace_id for run in store.runs()) # runs loaded from --persist
77
+
78
+ @asynccontextmanager
79
+ async def lifespan(_: FastAPI) -> AsyncIterator[None]:
80
+ async with running(hub):
81
+ yield
82
+
83
+ app = FastAPI(title="loopview", version=__version__, lifespan=lifespan, **_NO_SELF_TELEMETRY)
84
+ app.state.store = store
85
+ app.state.hub = hub
86
+
87
+ # --- OTLP receiver ----------------------------------------------------------
88
+
89
+ @app.post("/v1/traces")
90
+ async def receive_traces(request: Request) -> Response:
91
+ # `async def` keeps every store update on the event loop, one at a time.
92
+ content_type = request.headers.get("content-type", "")
93
+ declared_length = int(request.headers.get("content-length") or 0)
94
+ if declared_length > MAX_BODY_BYTES:
95
+ raise HTTPException(413, "request body too large")
96
+ try:
97
+ body = decompress(await request.body(), request.headers.get("content-encoding"))
98
+ spans = decode_body(body, content_type)
99
+ except UnsupportedContentType as exc:
100
+ raise HTTPException(415, str(exc)) from exc
101
+ except BodyTooLarge as exc:
102
+ raise HTTPException(413, str(exc)) from exc
103
+ except OtlpDecodeError as exc:
104
+ # 400 tells the exporter not to retry.
105
+ raise HTTPException(400, str(exc)) from exc
106
+
107
+ received_at_ns = time.time_ns()
108
+ hub.mark(store.add_spans(spans, received_at_ns=received_at_ns))
109
+ if capture is not None:
110
+ capture.write(CapturedRequest(received_at_ns, content_type, body))
111
+
112
+ payload, media_type = encode_success_response(content_type)
113
+ return Response(content=payload, media_type=media_type)
114
+
115
+ @app.post("/v1/loopview/span-starts")
116
+ async def receive_span_starts(request: Request) -> Response:
117
+ """Start reports from loopview-sdk: OTLP-encoded spans that have just
118
+ started (no end time). Standard exporters only send ended spans, so this
119
+ is what lets a step show as running the moment it begins."""
120
+ content_type = request.headers.get("content-type", "")
121
+ try:
122
+ body = decompress(await request.body(), request.headers.get("content-encoding"))
123
+ spans = decode_body(body, content_type)
124
+ except UnsupportedContentType as exc:
125
+ raise HTTPException(415, str(exc)) from exc
126
+ except OtlpDecodeError as exc:
127
+ raise HTTPException(400, str(exc)) from exc
128
+ hub.mark(store.add_started_spans(spans))
129
+ payload, media_type = encode_success_response(content_type)
130
+ return Response(content=payload, media_type=media_type)
131
+
132
+ # --- normalized runs and live events ------------------------------------------
133
+
134
+ @app.get("/api/health")
135
+ def health() -> dict[str, str]:
136
+ return {"status": "ok", "version": __version__}
137
+
138
+ @app.get("/api/runs")
139
+ def list_runs() -> list[RunInfo]:
140
+ return hub.run_infos()
141
+
142
+ @app.get("/api/runs/{trace_id}")
143
+ def get_run(trace_id: str) -> NormalizedRun:
144
+ normalized = hub.get(trace_id)
145
+ if normalized is None:
146
+ raise HTTPException(404, "run not found")
147
+ return normalized
148
+
149
+ @app.get("/api/sessions")
150
+ def list_sessions() -> list[SessionSummary]:
151
+ return store.sessions()
152
+
153
+ @app.get("/api/tools")
154
+ def tools(session: str | None = None, runs: str | None = None) -> dict[str, Any]:
155
+ """The Tools tab: how tools behave across runs. All runs by default, or one
156
+ session's (`?session=id`), or a list (`?runs=id1,id2`). Recomputed on
157
+ every request, which is cheap at a few hundred runs."""
158
+ if runs:
159
+ trace_ids = [t for t in runs.split(",") if t]
160
+ elif session:
161
+ match = [s for s in store.sessions() if s.session_id == session]
162
+ if not match:
163
+ raise HTTPException(404, "session not found")
164
+ trace_ids = match[0].trace_ids
165
+ else:
166
+ trace_ids = [run.trace_id for run in store.runs()]
167
+ normalized = [n for n in (hub.get(t) for t in trace_ids) if n is not None]
168
+ if runs and not normalized:
169
+ raise HTTPException(404, "none of these runs were found")
170
+ return tools_report(normalized)
171
+
172
+ @app.get("/api/events")
173
+ async def events() -> StreamingResponse:
174
+ return StreamingResponse(
175
+ hub.events(),
176
+ media_type="text/event-stream",
177
+ headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"},
178
+ )
179
+
180
+ # --- import, export, demo -----------------------------------------------------
181
+
182
+ @app.get("/api/runs/{trace_id}/export")
183
+ def export_run(trace_id: str) -> Response:
184
+ """The run as a capture file: one OTLP/JSON request per original arrival
185
+ batch, so importing it replays the same timing."""
186
+ run = store.get_run(trace_id)
187
+ if run is None:
188
+ raise HTTPException(404, "run not found")
189
+ batches: dict[int, list[RawSpan]] = {}
190
+ for span_id, span in run.spans.items():
191
+ batches.setdefault(run.received_ns[span_id], []).append(span)
192
+ lines = [
193
+ to_line(CapturedRequest(received, JSON, json.dumps(spans_to_otlp_json(spans)).encode()))
194
+ for received, spans in sorted(batches.items())
195
+ ]
196
+ return Response(
197
+ content="\n".join(lines) + "\n",
198
+ media_type="application/x-ndjson",
199
+ headers={"Content-Disposition": f'attachment; filename="run-{trace_id[:8]}.jsonl"'},
200
+ )
201
+
202
+ @app.post("/api/import")
203
+ async def import_runs(request: Request) -> dict[str, list[str]]:
204
+ changed: list[str] = []
205
+ try:
206
+ for line in (await request.body()).decode("utf-8").splitlines():
207
+ if not line.strip():
208
+ continue
209
+ captured = from_line(line)
210
+ spans = decode_body(captured.body, captured.content_type)
211
+ changed += store.add_spans(spans, received_at_ns=captured.received_at_ns)
212
+ except (ValueError, KeyError, TypeError) as exc:
213
+ raise HTTPException(400, f"not a loopview JSONL file: {exc}") from exc
214
+ hub.mark(changed)
215
+ return {"trace_ids": sorted(set(changed))}
216
+
217
+ @app.post("/api/demo")
218
+ def load_demo() -> dict[str, list[str]]:
219
+ return {"trace_ids": load_demo_into(store, hub)}
220
+
221
+ @app.get("/api/runs/{trace_id}/spans")
222
+ def run_spans(trace_id: str) -> list[RawSpan]:
223
+ """Raw spans of one run, for debugging only. The UI never uses this."""
224
+ run = store.get_run(trace_id)
225
+ if run is None:
226
+ raise HTTPException(404, "run not found")
227
+ return sorted(run.spans.values(), key=lambda s: s.start_time_unix_nano)
228
+
229
+ # --- UI ---------------------------------------------------------------------
230
+
231
+ if (static_dir / "index.html").exists():
232
+ # html=True makes "/" serve index.html. Mounted last so /api and /v1 win.
233
+ app.mount("/", StaticFiles(directory=static_dir, html=True), name="ui")
234
+ else:
235
+
236
+ @app.get("/", response_class=HTMLResponse)
237
+ def missing_ui() -> str:
238
+ return _MISSING_UI_PAGE
239
+
240
+ return app
241
+
242
+
243
+ def load_demo_into(store: TraceStore, hub: LiveHub | None = None) -> list[str]:
244
+ """Load the bundled demo recording. Loading it twice is harmless: spans are
245
+ keyed by id, so the same run is simply updated."""
246
+ demo = TraceStore()
247
+ load_capture(demo, DEMO_RECORDING)
248
+ trace_ids = [run.trace_id for run in demo.runs()]
249
+ load_capture(store, DEMO_RECORDING)
250
+ if hub is not None:
251
+ hub.mark(trace_ids)
252
+ return trace_ids
loopview/cli.py ADDED
@@ -0,0 +1,128 @@
1
+ """Command line entry point: `loopview`.
2
+
3
+ argparse (standard library) instead of click/typer: we only need a few flags,
4
+ so an extra dependency is not worth it.
5
+ """
6
+
7
+ import argparse
8
+ import logging
9
+ import threading
10
+ import webbrowser
11
+ from pathlib import Path
12
+ from types import FrameType
13
+
14
+ import uvicorn
15
+
16
+ from loopview import __version__
17
+ from loopview.app import create_app, load_demo_into
18
+ from loopview.cost.pricing import merged_pricing
19
+ from loopview.live import LiveHub
20
+ from loopview.store.capture import CaptureWriter, load_capture
21
+ from loopview.store.memory import DEFAULT_MAX_RUNS, TraceStore
22
+
23
+ # 4318 is the standard OTLP/HTTP port, so exporters work with only an endpoint change.
24
+ DEFAULT_PORT = 4318
25
+
26
+
27
+ def main(argv: list[str] | None = None) -> None:
28
+ parser = argparse.ArgumentParser(
29
+ prog="loopview", description="Watch your AI agents run as a live graph."
30
+ )
31
+ parser.add_argument(
32
+ "command",
33
+ nargs="?",
34
+ choices=["demo"],
35
+ help="demo: start with a recorded multi-agent run loaded and open the browser",
36
+ )
37
+ parser.add_argument(
38
+ "--host", default="127.0.0.1", help="interface to bind (default: localhost only)"
39
+ )
40
+ parser.add_argument(
41
+ "--port", type=int, default=DEFAULT_PORT, help=f"port (default: {DEFAULT_PORT})"
42
+ )
43
+ parser.add_argument(
44
+ "--persist",
45
+ type=Path,
46
+ metavar="FILE",
47
+ help="save every received request to this JSONL file and reload it on start",
48
+ )
49
+ parser.add_argument(
50
+ "--max-runs",
51
+ type=int,
52
+ default=DEFAULT_MAX_RUNS,
53
+ help=f"how many recent runs to keep in memory (default: {DEFAULT_MAX_RUNS})",
54
+ )
55
+ parser.add_argument(
56
+ "--prices",
57
+ type=Path,
58
+ metavar="FILE",
59
+ help="JSON file of model prices, replacing the bundled ones (see cost/pricing.json)",
60
+ )
61
+ parser.add_argument("--no-browser", action="store_true", help="don't open the browser")
62
+ parser.add_argument("--version", action="version", version=f"loopview {__version__}")
63
+ args = parser.parse_args(argv)
64
+
65
+ logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
66
+ logging.getLogger("asyncio").addFilter(_IgnoreWindowsConnectionReset())
67
+ store = TraceStore(max_runs=args.max_runs)
68
+ capture = None
69
+ if args.persist is not None:
70
+ if args.persist.exists():
71
+ loaded = load_capture(store, args.persist)
72
+ print(f"loaded {loaded} requests from {args.persist}")
73
+ capture = CaptureWriter(args.persist)
74
+ if args.command == "demo":
75
+ load_demo_into(store)
76
+
77
+ print(f"loopview {__version__}")
78
+ print(f" UI: http://{args.host}:{args.port}")
79
+ print(f" OTLP endpoint: http://{args.host}:{args.port}/v1/traces")
80
+ if capture is not None:
81
+ print(f" persisting to: {args.persist}")
82
+ # Flush now: when stdout is a pipe, Python buffers it and the banner would
83
+ # only appear when the server stops.
84
+ print(flush=True)
85
+ if args.command == "demo" and not args.no_browser:
86
+ # Give the server a moment to start listening before the browser asks.
87
+ url = f"http://{args.host}:{args.port}"
88
+ threading.Timer(1.0, webbrowser.open, args=[url]).start()
89
+ app = create_app(store=store, capture=capture, pricing=merged_pricing(args.prices))
90
+ server = _Server(
91
+ uvicorn.Config(app, host=args.host, port=args.port, log_level="warning"), app.state.hub
92
+ )
93
+ try:
94
+ server.run()
95
+ except KeyboardInterrupt:
96
+ # uvicorn re-raises Ctrl+C once it has shut down cleanly; that's the
97
+ # normal way out, not an error.
98
+ print("loopview stopped")
99
+ finally:
100
+ if capture is not None:
101
+ capture.close()
102
+
103
+
104
+ class _Server(uvicorn.Server):
105
+ """uvicorn's server, which also ends the live streams as soon as Ctrl+C is
106
+ pressed. Otherwise it would wait for the browser's open event stream, which
107
+ never closes on its own, and then cancel it with a traceback."""
108
+
109
+ def __init__(self, config: uvicorn.Config, hub: LiveHub) -> None:
110
+ super().__init__(config)
111
+ self.hub = hub
112
+
113
+ def handle_exit(self, sig: int, frame: FrameType | None) -> None:
114
+ self.hub.close()
115
+ super().handle_exit(sig, frame)
116
+
117
+
118
+ class _IgnoreWindowsConnectionReset(logging.Filter):
119
+ """On Windows, asyncio logs a ConnectionResetError when a browser closes a
120
+ connection abruptly (a known, harmless Python issue). Drop that one record."""
121
+
122
+ def filter(self, record: logging.LogRecord) -> bool:
123
+ error = record.exc_info[1] if record.exc_info else None
124
+ return not isinstance(error, ConnectionResetError)
125
+
126
+
127
+ if __name__ == "__main__":
128
+ main()
@@ -0,0 +1 @@
1
+ """Cost: where the tokens of each model call came from, and what they cost."""
@@ -0,0 +1,62 @@
1
+ {
2
+ "_about": [
3
+ "Prices per million tokens, in USD. Edit this file, or pass your own with `loopview --prices FILE` (entries there replace these).",
4
+ "A model matches the longest key it starts with, so dated IDs like claude-haiku-4-5-20251001 match claude-haiku-4-5.",
5
+ "cache_write is the 5-minute cache write price; 1-hour cache writes cost more and are not told apart in traces.",
6
+ "tool_prompt_tokens: input tokens the provider adds when tools are passed (tool_choice auto or none); shown as its own segment.",
7
+ "OpenAI: Standard tier. OpenAI caches automatically and charges nothing to write the cache, so cache_write is the input price; models without a cached price get cache_read = input.",
8
+ "Not modelled: OpenAI's long-context rates (input doubles above 272K tokens on newer models), Batch/Flex/Priority tiers, regional pricing.",
9
+ "A model missing here is shown in tokens only, never with a guessed price."
10
+ ],
11
+ "models": {
12
+ "claude-fable-5-1": {"input": 10, "cache_write": 12.5, "cache_read": 0.25, "output": 50, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
13
+ "claude-fable-5": {"input": 10, "cache_write": 12.5, "cache_read": 1, "output": 50, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
14
+ "claude-opus-5-5": {"input": 4, "cache_write": 5, "cache_read": 0.2, "output": 20, "tool_prompt_tokens": 286, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
15
+ "claude-opus-5": {"input": 5, "cache_write": 6.25, "cache_read": 0.5, "output": 25, "tool_prompt_tokens": 286, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
16
+ "claude-opus-4-8": {"input": 5, "cache_write": 6.25, "cache_read": 0.5, "output": 25, "tool_prompt_tokens": 290, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
17
+ "claude-opus-4-7": {"input": 5, "cache_write": 6.25, "cache_read": 0.5, "output": 25, "tool_prompt_tokens": 675, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
18
+ "claude-opus-4-6": {"input": 5, "cache_write": 6.25, "cache_read": 0.5, "output": 25, "tool_prompt_tokens": 497, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
19
+ "claude-opus-4-5": {"input": 5, "cache_write": 6.25, "cache_read": 0.5, "output": 25, "tool_prompt_tokens": 496, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
20
+ "claude-sonnet-5-5": {"input": 2, "cache_write": 2.5, "cache_read": 0.2, "output": 10, "tool_prompt_tokens": 286, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
21
+ "claude-sonnet-5": {"input": 2, "cache_write": 2.5, "cache_read": 0.2, "output": 10, "tool_prompt_tokens": 354, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
22
+ "claude-sonnet-4-6": {"input": 3, "cache_write": 3.75, "cache_read": 0.3, "output": 15, "tool_prompt_tokens": 497, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
23
+ "claude-sonnet-4-5": {"input": 3, "cache_write": 3.75, "cache_read": 0.3, "output": 15, "tool_prompt_tokens": 496, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
24
+ "claude-haiku-4-5": {"input": 1, "cache_write": 1.25, "cache_read": 0.1, "output": 5, "tool_prompt_tokens": 496, "source": "https://platform.claude.com/docs/en/about-claude/pricing", "checked": "2026-10-02"},
25
+ "gpt-6-astra": {"input": 10.0, "cache_write": 10.0, "cache_read": 1.0, "output": 50.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
26
+ "gpt-6.1-sol": {"input": 2.0, "cache_write": 2.0, "cache_read": 0.1, "output": 10.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
27
+ "gpt-6-sol": {"input": 2.0, "cache_write": 2.0, "cache_read": 0.2, "output": 10.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
28
+ "gpt-6-luna": {"input": 0.1, "cache_write": 0.1, "cache_read": 0.01, "output": 0.5, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
29
+ "gpt-5.6-sol": {"input": 4.0, "cache_write": 4.0, "cache_read": 0.4, "output": 20.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
30
+ "gpt-5.6-terra": {"input": 2.0, "cache_write": 2.0, "cache_read": 0.2, "output": 12.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
31
+ "gpt-5.6-luna": {"input": 0.2, "cache_write": 0.2, "cache_read": 0.02, "output": 1.2, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
32
+ "gpt-5.5": {"input": 5.0, "cache_write": 5.0, "cache_read": 0.5, "output": 30.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
33
+ "gpt-5.5-pro": {"input": 30.0, "cache_write": 30.0, "cache_read": 30.0, "output": 180.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
34
+ "gpt-5.4": {"input": 2.5, "cache_write": 2.5, "cache_read": 0.25, "output": 15.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
35
+ "gpt-5.4-mini": {"input": 0.75, "cache_write": 0.75, "cache_read": 0.075, "output": 4.5, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
36
+ "gpt-5.4-nano": {"input": 0.2, "cache_write": 0.2, "cache_read": 0.02, "output": 1.25, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
37
+ "gpt-5.4-pro": {"input": 30.0, "cache_write": 30.0, "cache_read": 30.0, "output": 180.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
38
+ "gpt-5.2": {"input": 1.75, "cache_write": 1.75, "cache_read": 0.175, "output": 14.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
39
+ "gpt-5.2-pro": {"input": 21.0, "cache_write": 21.0, "cache_read": 21.0, "output": 168.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
40
+ "gpt-5.1": {"input": 1.25, "cache_write": 1.25, "cache_read": 0.125, "output": 10.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
41
+ "gpt-5": {"input": 1.25, "cache_write": 1.25, "cache_read": 0.125, "output": 10.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
42
+ "gpt-5-mini": {"input": 0.25, "cache_write": 0.25, "cache_read": 0.025, "output": 2.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
43
+ "gpt-5-nano": {"input": 0.05, "cache_write": 0.05, "cache_read": 0.005, "output": 0.4, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
44
+ "gpt-5-pro": {"input": 15.0, "cache_write": 15.0, "cache_read": 15.0, "output": 120.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
45
+ "gpt-4.1": {"input": 2.0, "cache_write": 2.0, "cache_read": 0.5, "output": 8.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
46
+ "gpt-4.1-mini": {"input": 0.4, "cache_write": 0.4, "cache_read": 0.1, "output": 1.6, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
47
+ "gpt-4.1-nano": {"input": 0.1, "cache_write": 0.1, "cache_read": 0.025, "output": 0.4, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
48
+ "gpt-4o": {"input": 2.5, "cache_write": 2.5, "cache_read": 1.25, "output": 10.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
49
+ "gpt-4o-2024-05-13": {"input": 5.0, "cache_write": 5.0, "cache_read": 5.0, "output": 15.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
50
+ "gpt-4o-mini": {"input": 0.15, "cache_write": 0.15, "cache_read": 0.075, "output": 0.6, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
51
+ "o1": {"input": 15.0, "cache_write": 15.0, "cache_read": 7.5, "output": 60.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
52
+ "o1-pro": {"input": 150.0, "cache_write": 150.0, "cache_read": 150.0, "output": 600.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
53
+ "o3": {"input": 2.0, "cache_write": 2.0, "cache_read": 0.5, "output": 8.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
54
+ "o3-pro": {"input": 20.0, "cache_write": 20.0, "cache_read": 20.0, "output": 80.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
55
+ "o3-mini": {"input": 1.1, "cache_write": 1.1, "cache_read": 0.55, "output": 4.4, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
56
+ "o4-mini": {"input": 1.1, "cache_write": 1.1, "cache_read": 0.275, "output": 4.4, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
57
+ "gpt-4-turbo": {"input": 10.0, "cache_write": 10.0, "cache_read": 10.0, "output": 30.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
58
+ "gpt-4-0613": {"input": 30.0, "cache_write": 30.0, "cache_read": 30.0, "output": 60.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
59
+ "gpt-3.5-turbo": {"input": 0.5, "cache_write": 0.5, "cache_read": 0.5, "output": 1.5, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"},
60
+ "gpt-3.5-turbo-1106": {"input": 1.0, "cache_write": 1.0, "cache_read": 1.0, "output": 2.0, "source": "https://developers.openai.com/api/docs/pricing", "checked": "2026-10-03"}
61
+ }
62
+ }
@@ -0,0 +1,74 @@
1
+ """Model prices, from a small JSON file the user can edit.
2
+
3
+ JSON rather than YAML: Python and browsers read it with no extra dependency.
4
+ JSON has no comments, so the file explains itself in an "_about" list, and each
5
+ price carries its source and the date it was checked.
6
+ """
7
+
8
+ import json
9
+ import re
10
+ from functools import cache
11
+ from pathlib import Path
12
+
13
+ from pydantic import BaseModel
14
+
15
+ DEFAULT_PRICES = Path(__file__).parent / "pricing.json"
16
+
17
+
18
+ class ModelPrice(BaseModel):
19
+ """USD per million tokens."""
20
+
21
+ input: float
22
+ output: float
23
+ cache_read: float
24
+ cache_write: float
25
+ tool_prompt_tokens: int | None = None
26
+ source: str
27
+ checked: str
28
+
29
+
30
+ class Pricing(BaseModel):
31
+ models: dict[str, ModelPrice]
32
+
33
+ def for_model(self, model: str | None) -> tuple[str, ModelPrice] | None:
34
+ """The longest key the model ID equals or starts with (then a dash), so a
35
+ dated ID like claude-haiku-4-5-20251001 matches claude-haiku-4-5, and
36
+ claude-sonnet-5-5 doesn't fall back to claude-sonnet-5."""
37
+ if not model:
38
+ return None
39
+ for name in _model_names(model.lower()):
40
+ matches = [k for k in self.models if name == k or name.startswith(k + "-")]
41
+ if matches:
42
+ key = max(matches, key=len)
43
+ return key, self.models[key]
44
+ return None
45
+
46
+
47
+ # Routers and clouds prefix the provider's model ID: LiteLLM "anthropic/...",
48
+ # Bedrock "us.anthropic.claude-...-v1:0", Vertex "claude-...@20251001".
49
+ _PROVIDER_PREFIX = re.compile(r"^(?:[a-z]{2,4}\.)?(?:anthropic|openai|meta|mistral)\.")
50
+
51
+
52
+ def _model_names(model: str) -> list[str]:
53
+ """The ID as given, then without a router's or cloud's prefix."""
54
+ bare = model.rsplit("/", 1)[-1].replace("@", "-")
55
+ bare = _PROVIDER_PREFIX.sub("", bare)
56
+ return [model] if bare == model else [model, bare]
57
+
58
+
59
+ def load_pricing(path: Path) -> Pricing:
60
+ data = json.loads(path.read_text(encoding="utf-8"))
61
+ return Pricing(models=data["models"])
62
+
63
+
64
+ @cache
65
+ def default_pricing() -> Pricing:
66
+ return load_pricing(DEFAULT_PRICES)
67
+
68
+
69
+ def merged_pricing(override: Path | None) -> Pricing:
70
+ """The bundled prices, with the user's file replacing or adding models."""
71
+ base = default_pricing()
72
+ if override is None:
73
+ return base
74
+ return Pricing(models={**base.models, **load_pricing(override).models})