claude-finops 0.7.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -47
- package/config/pricing.json +166 -24
- package/config/settings.json +11 -5
- package/finops/actions.py +1 -1
- package/finops/agents.py +2 -1
- package/finops/analytics.py +721 -611
- package/finops/api.py +179 -55
- package/finops/diagnose.py +17 -20
- package/finops/etl.py +169 -57
- package/finops/integrate.py +22 -57
- package/finops/paths.py +12 -0
- package/finops/plan_history.py +80 -0
- package/finops/pricing.py +57 -6
- package/finops/procs.py +7 -5
- package/finops/report.py +29 -23
- package/finops/segments.py +149 -0
- package/package.json +1 -1
- package/run.cmd +3 -2
- package/run.py +26 -59
- package/web/app.js +297 -333
- package/web/charts.js +60 -16
- package/finops/advisor.py +0 -188
- package/finops/trial.py +0 -169
package/finops/analytics.py
CHANGED
|
@@ -7,16 +7,19 @@ Every number returned is tagged with a `basis`:
|
|
|
7
7
|
recommendation - suggested action, never a booked saving
|
|
8
8
|
"""
|
|
9
9
|
import json
|
|
10
|
-
import math
|
|
11
10
|
import os
|
|
12
11
|
import sqlite3
|
|
13
12
|
import statistics
|
|
14
|
-
|
|
13
|
+
import sys
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
from collections import defaultdict
|
|
15
17
|
from datetime import date, datetime, timedelta, timezone
|
|
16
18
|
|
|
17
19
|
from .pricing import Pricing
|
|
20
|
+
from .segments import is_compaction
|
|
18
21
|
|
|
19
|
-
from .paths import
|
|
22
|
+
from .paths import DB_PATH, SETTINGS_PATH, LOCAL_SETTINGS_PATH
|
|
20
23
|
|
|
21
24
|
UNAVAILABLE = "Unavailable from connected Claude data"
|
|
22
25
|
|
|
@@ -84,36 +87,131 @@ def detect_account():
|
|
|
84
87
|
return {k: v for k, v in out.items() if v}
|
|
85
88
|
|
|
86
89
|
|
|
90
|
+
def _is_num(v):
|
|
91
|
+
return isinstance(v, (int, float)) and not isinstance(v, bool)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _validate_settings(cur, defaults):
|
|
95
|
+
"""Defensive coercion of settings.local.json's numeric leaves.
|
|
96
|
+
|
|
97
|
+
Every value under budgets/limits must be a number, null, or (for
|
|
98
|
+
per_project_usd/per_model_usd) a dict of numbers; alert_thresholds_pct must be
|
|
99
|
+
a list of numbers 0..1000. Anything else is dropped and the shipped default
|
|
100
|
+
(from settings.json) is used instead, with a warning.
|
|
101
|
+
"""
|
|
102
|
+
bad = []
|
|
103
|
+
for section in ("budgets", "limits"):
|
|
104
|
+
want = defaults.get(section, {})
|
|
105
|
+
have = cur.get(section)
|
|
106
|
+
if not isinstance(have, dict):
|
|
107
|
+
bad.append(section)
|
|
108
|
+
cur[section] = want
|
|
109
|
+
continue
|
|
110
|
+
fixed = dict(have)
|
|
111
|
+
for k, v in list(have.items()):
|
|
112
|
+
if k.startswith("_"):
|
|
113
|
+
continue
|
|
114
|
+
if k in ("per_project_usd", "per_model_usd"):
|
|
115
|
+
if not isinstance(v, dict) or not all(_is_num(x) for x in v.values()):
|
|
116
|
+
bad.append(f"{section}.{k}")
|
|
117
|
+
fixed[k] = want.get(k, {})
|
|
118
|
+
elif not (v is None or _is_num(v)):
|
|
119
|
+
bad.append(f"{section}.{k}")
|
|
120
|
+
fixed[k] = want.get(k)
|
|
121
|
+
cur[section] = fixed
|
|
122
|
+
pct = cur.get("alert_thresholds_pct")
|
|
123
|
+
if not (isinstance(pct, list) and all(_is_num(x) and 0 <= x <= 1000 for x in pct)):
|
|
124
|
+
if pct is not None:
|
|
125
|
+
bad.append("alert_thresholds_pct")
|
|
126
|
+
cur["alert_thresholds_pct"] = defaults.get("alert_thresholds_pct", [])
|
|
127
|
+
for path in bad:
|
|
128
|
+
print(f"finops: settings.local.json has an invalid '{path}'; using the shipped default",
|
|
129
|
+
file=sys.stderr)
|
|
130
|
+
return cur
|
|
131
|
+
|
|
132
|
+
|
|
87
133
|
def load_settings():
|
|
88
134
|
"""Shared defaults (settings.json) + this machine's overrides (settings.local.json)."""
|
|
89
135
|
with open(SETTINGS_PATH) as fh:
|
|
90
|
-
|
|
136
|
+
base = json.load(fh)
|
|
91
137
|
# Detected identity first, so a configured settings.json still wins below.
|
|
92
138
|
detected = detect_account()
|
|
93
|
-
acct =
|
|
139
|
+
acct = base.setdefault("account", {})
|
|
94
140
|
for k, v in detected.items():
|
|
95
141
|
if not acct.get(k):
|
|
96
142
|
acct[k] = v
|
|
143
|
+
cur = json.loads(json.dumps(base)) # deep copy: base stays the fallback default
|
|
97
144
|
if os.path.exists(LOCAL_SETTINGS_PATH):
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
145
|
+
try:
|
|
146
|
+
with open(LOCAL_SETTINGS_PATH) as fh:
|
|
147
|
+
local = json.load(fh)
|
|
148
|
+
except (OSError, ValueError) as exc:
|
|
149
|
+
print(f"finops: settings.local.json unreadable ({exc}); using shipped defaults",
|
|
150
|
+
file=sys.stderr)
|
|
151
|
+
local = {}
|
|
152
|
+
if isinstance(local, dict):
|
|
153
|
+
_merge(cur, local)
|
|
154
|
+
else:
|
|
155
|
+
print("finops: settings.local.json is not an object; using shipped defaults",
|
|
156
|
+
file=sys.stderr)
|
|
157
|
+
return _validate_settings(cur, base)
|
|
101
158
|
|
|
102
159
|
|
|
103
160
|
def _d(s):
|
|
104
161
|
return datetime.strptime(s, "%Y-%m-%d").date()
|
|
105
162
|
|
|
106
163
|
|
|
164
|
+
def _cumsum(values):
|
|
165
|
+
total = 0.0
|
|
166
|
+
for v in values:
|
|
167
|
+
total += v or 0.0
|
|
168
|
+
yield total
|
|
169
|
+
|
|
170
|
+
|
|
107
171
|
class Analytics:
|
|
108
172
|
def __init__(self, db_path=DB_PATH):
|
|
109
|
-
|
|
110
|
-
|
|
173
|
+
# One connection per thread. The HTTP server is threaded, and a single sqlite
|
|
174
|
+
# connection shared across threads fails under concurrent use with "bad
|
|
175
|
+
# parameter or other API misuse" — which is exactly what a page firing several
|
|
176
|
+
# requests at once produces. Every connection is tracked so close() can release
|
|
177
|
+
# them all before the warehouse file is swapped on sync.
|
|
178
|
+
self._db_path = db_path
|
|
179
|
+
self._local = threading.local()
|
|
180
|
+
self._conns = []
|
|
181
|
+
self._conns_lock = threading.Lock()
|
|
111
182
|
self.pricing = Pricing()
|
|
112
183
|
self.settings = load_settings()
|
|
113
184
|
self.meta = {r["key"]: r["value"] for r in self.db.execute("SELECT * FROM meta")}
|
|
114
185
|
row = self.db.execute("SELECT MIN(day) a, MAX(day) b FROM requests WHERE day<>''").fetchone()
|
|
115
186
|
self.first_day, self.last_day = row["a"], row["b"]
|
|
116
187
|
|
|
188
|
+
_today = None # tests set this; production uses the clock
|
|
189
|
+
|
|
190
|
+
def today(self):
|
|
191
|
+
return self._today or datetime.now(timezone.utc).date()
|
|
192
|
+
|
|
193
|
+
@property
|
|
194
|
+
def db(self):
|
|
195
|
+
c = getattr(self._local, "conn", None)
|
|
196
|
+
if c is None:
|
|
197
|
+
c = sqlite3.connect(self._db_path)
|
|
198
|
+
c.row_factory = sqlite3.Row
|
|
199
|
+
self._local.conn = c
|
|
200
|
+
with self._conns_lock:
|
|
201
|
+
self._conns.append(c)
|
|
202
|
+
return c
|
|
203
|
+
|
|
204
|
+
def close(self):
|
|
205
|
+
"""Close every thread's connection, so the warehouse file can be replaced."""
|
|
206
|
+
with self._conns_lock:
|
|
207
|
+
conns, self._conns = self._conns, []
|
|
208
|
+
for c in conns:
|
|
209
|
+
try:
|
|
210
|
+
c.close()
|
|
211
|
+
except Exception:
|
|
212
|
+
pass
|
|
213
|
+
self._local = threading.local()
|
|
214
|
+
|
|
117
215
|
def q(self, sql, params=()):
|
|
118
216
|
return [dict(r) for r in self.db.execute(sql, params)]
|
|
119
217
|
|
|
@@ -136,7 +234,10 @@ class Analytics:
|
|
|
136
234
|
cl.append("r.model IN (%s)" % ",".join("?" * len(f["models"]))); p += f["models"]
|
|
137
235
|
if f.get("projects"):
|
|
138
236
|
cl.append("r.project_id IN (%s)" % ",".join("?" * len(f["projects"])))
|
|
139
|
-
|
|
237
|
+
try:
|
|
238
|
+
p += [int(x) for x in f["projects"]]
|
|
239
|
+
except (TypeError, ValueError):
|
|
240
|
+
raise ValueError("projects must be integers")
|
|
140
241
|
if f.get("sessions"):
|
|
141
242
|
cl.append("r.session_id IN (%s)" % ",".join("?" * len(f["sessions"]))); p += f["sessions"]
|
|
142
243
|
if f.get("categories"):
|
|
@@ -152,10 +253,36 @@ class Analytics:
|
|
|
152
253
|
cl.append("r.billable_tokens >= ?"); p.append(int(f["min_tokens"]))
|
|
153
254
|
return (" AND ".join(cl) if cl else "1=1"), p
|
|
154
255
|
|
|
256
|
+
def daily_series(self, f, days=None, end=None):
|
|
257
|
+
"""Per-day cost/token series with idle days present as zeros.
|
|
258
|
+
|
|
259
|
+
A plain GROUP BY day only returns days you actually worked, so a mean taken
|
|
260
|
+
over it is a per-ACTIVE-day rate. Every projection here multiplies that rate
|
|
261
|
+
by calendar days remaining, so the zero days have to be filled in or the
|
|
262
|
+
forecast is inflated by exactly the share of days you were idle.
|
|
263
|
+
"""
|
|
264
|
+
w, p = self.where(f)
|
|
265
|
+
rows = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost, SUM(r.billable_tokens) tokens,
|
|
266
|
+
COUNT(*) requests FROM requests r WHERE {w} AND r.day <> ''
|
|
267
|
+
GROUP BY 1 ORDER BY 1""", p)
|
|
268
|
+
if not rows:
|
|
269
|
+
return []
|
|
270
|
+
by_day = {r["day"]: r for r in rows}
|
|
271
|
+
last = _d(end) if end else max(_d(rows[-1]["day"]), self.today())
|
|
272
|
+
first = _d(rows[0]["day"])
|
|
273
|
+
if days:
|
|
274
|
+
first = max(first, last - timedelta(days=days - 1))
|
|
275
|
+
out, cur = [], first
|
|
276
|
+
while cur <= last:
|
|
277
|
+
k = cur.isoformat()
|
|
278
|
+
out.append(by_day.get(k) or {"day": k, "cost": 0.0, "tokens": 0, "requests": 0})
|
|
279
|
+
cur += timedelta(days=1)
|
|
280
|
+
return out
|
|
281
|
+
|
|
155
282
|
# ---------------- billing period ----------------
|
|
156
283
|
def billing_period(self, today=None):
|
|
157
284
|
bp = self.settings["billing_period"]
|
|
158
|
-
today = today or
|
|
285
|
+
today = today or self.today()
|
|
159
286
|
anchor = int(bp.get("anchor_day", 1))
|
|
160
287
|
if today.day >= anchor:
|
|
161
288
|
start = today.replace(day=min(anchor, 28))
|
|
@@ -208,6 +335,7 @@ class Analytics:
|
|
|
208
335
|
f" FROM requests r WHERE {w} AND {extra}", p + ep)
|
|
209
336
|
|
|
210
337
|
tot["cost_today"] = spend("r.day = ?", [today])
|
|
338
|
+
tot["cost_yesterday"] = spend("r.day = ?", [(_d(today) - timedelta(days=1)).isoformat()])
|
|
211
339
|
tot["cost_week"] = spend("r.day >= ?", [wk])
|
|
212
340
|
tot["cost_period"] = spend("r.day >= ? AND r.day <= ?", [bp["start"], bp["end"]])
|
|
213
341
|
tot["billing_period"] = bp
|
|
@@ -233,14 +361,13 @@ class Analytics:
|
|
|
233
361
|
elapsed = max(bp["elapsed_days"], 1)
|
|
234
362
|
daily_avg = used_cost / elapsed
|
|
235
363
|
|
|
236
|
-
# Trailing rates are measured over the last N days
|
|
237
|
-
#
|
|
238
|
-
#
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
last7, last14 = recent[-7:], recent[-14:]
|
|
364
|
+
# Trailing rates are measured over the last N CALENDAR days, not only the
|
|
365
|
+
# slice inside the billing period — early in a period that slice is too short
|
|
366
|
+
# to be a rate. Idle days count as zero, because these rates get multiplied by
|
|
367
|
+
# calendar days remaining. This keeps burn and forecast on one methodology.
|
|
368
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
369
|
+
last7 = self.daily_series(f, days=7, end=yesterday)
|
|
370
|
+
last14 = self.daily_series(f, days=14, end=yesterday)
|
|
244
371
|
avg7 = (sum(r["cost"] for r in last7) / len(last7)) if last7 else 0.0
|
|
245
372
|
tok_avg7 = (sum(r["tokens"] for r in last7) / len(last7)) if last7 else 0.0
|
|
246
373
|
# the projection rate matches Analytics.forecast()'s "expected" scenario
|
|
@@ -261,7 +388,8 @@ class Analytics:
|
|
|
261
388
|
"projected_period_cost": projected,
|
|
262
389
|
"projected_period_tokens": used_tokens + tok_burn * bp["remaining_days"],
|
|
263
390
|
"forecast_basis": "forecast",
|
|
264
|
-
"forecast_note": ("Projection uses the %d-day mean daily spend,
|
|
391
|
+
"forecast_note": ("Projection uses the %d-calendar-day mean daily spend, idle days "
|
|
392
|
+
"included as zero — the same rate as the "
|
|
265
393
|
"Forecast view's expected scenario." % len(last14)),
|
|
266
394
|
"series": rows,
|
|
267
395
|
"allowances": {},
|
|
@@ -277,7 +405,7 @@ class Analytics:
|
|
|
277
405
|
continue
|
|
278
406
|
remaining = allowance - used_val
|
|
279
407
|
pct = 100.0 * used_val / allowance
|
|
280
|
-
days_left = (remaining / rate) if rate > 0 else None
|
|
408
|
+
days_left = (remaining / rate) if rate > 0 and remaining > 0 else None
|
|
281
409
|
proj = used_val + rate * bp["remaining_days"]
|
|
282
410
|
out["allowances"][label] = {
|
|
283
411
|
"configured": True, "allowance": allowance, "used": used_val,
|
|
@@ -286,6 +414,7 @@ class Analytics:
|
|
|
286
414
|
"days_until_limit": (round(days_left, 1) if days_left is not None else None),
|
|
287
415
|
"limit_date": ((_d(bp["today"]) + timedelta(days=days_left)).isoformat()
|
|
288
416
|
if days_left is not None and days_left < 3650 else None),
|
|
417
|
+
"exceeded": remaining <= 0,
|
|
289
418
|
"projected_end_of_period": proj,
|
|
290
419
|
"projected_overage_pct": round(100.0 * (proj - allowance) / allowance, 1),
|
|
291
420
|
"status": self._status(pct),
|
|
@@ -316,6 +445,30 @@ class Analytics:
|
|
|
316
445
|
AVG(r.context_tokens) avg_context
|
|
317
446
|
FROM requests r WHERE {w} AND r.day <> '' GROUP BY 1 ORDER BY 1""", p)
|
|
318
447
|
|
|
448
|
+
def heatmap(self, f=None):
|
|
449
|
+
"""Spend by weekday x hour, in this machine's local time (Monday first).
|
|
450
|
+
|
|
451
|
+
Transcripts stamp UTC; bucketing on that would put a 10am IST session at
|
|
452
|
+
4am. SQLite's 'localtime' modifier applies the local offset, half-hour
|
|
453
|
+
zones included. Date filters still apply to UTC days like every view.
|
|
454
|
+
"""
|
|
455
|
+
w, p = self.where(f)
|
|
456
|
+
rows = self.q(f"""
|
|
457
|
+
SELECT CAST(strftime('%w', r.ts, 'localtime') AS INTEGER) wd,
|
|
458
|
+
CAST(strftime('%H', r.ts, 'localtime') AS INTEGER) hr,
|
|
459
|
+
COALESCE(SUM(r.est_cost_usd),0) cost, COALESCE(SUM(r.billable_tokens),0) tokens,
|
|
460
|
+
COUNT(*) requests
|
|
461
|
+
FROM requests r WHERE {w} AND r.ts <> '' GROUP BY 1, 2""", p)
|
|
462
|
+
grid = {(d, h): {"dow": d, "hour": h, "cost": 0.0, "tokens": 0, "requests": 0}
|
|
463
|
+
for d in range(7) for h in range(24)}
|
|
464
|
+
for r in rows:
|
|
465
|
+
if r["wd"] is None or r["hr"] is None:
|
|
466
|
+
continue
|
|
467
|
+
c = grid[((r["wd"] + 6) % 7, r["hr"])]
|
|
468
|
+
c["cost"], c["tokens"], c["requests"] = r["cost"], r["tokens"], r["requests"]
|
|
469
|
+
return {"cells": [grid[(d, h)] for d in range(7) for h in range(24)],
|
|
470
|
+
"tz": time.strftime("%Z"), "cost_basis": "estimated"}
|
|
471
|
+
|
|
319
472
|
# ---------------- models ----------------
|
|
320
473
|
def models(self, f=None):
|
|
321
474
|
w, p = self.where(f)
|
|
@@ -332,6 +485,24 @@ class Analytics:
|
|
|
332
485
|
FROM requests r WHERE {w} GROUP BY r.model ORDER BY cost DESC""", p)
|
|
333
486
|
tc = sum(r["cost"] for r in rows) or 1
|
|
334
487
|
tt = sum(r["tokens"] for r in rows) or 1
|
|
488
|
+
# Utilisation is measured against the window that actually served each request —
|
|
489
|
+
# a long-context variant has a bigger one — and requests that ran over a window
|
|
490
|
+
# this price table cannot explain are counted, not averaged into a figure above
|
|
491
|
+
# 100%, which is what dividing by the base model's window used to produce.
|
|
492
|
+
util = {}
|
|
493
|
+
for v in self.q(f"""SELECT r.model, r.priced_as, r.unpriced_long_context u,
|
|
494
|
+
COUNT(*) n, AVG(r.context_tokens) avg_ctx
|
|
495
|
+
FROM requests r WHERE {w}
|
|
496
|
+
GROUP BY r.model, r.priced_as, r.unpriced_long_context""", p):
|
|
497
|
+
u = util.setdefault(v["model"], {"num": 0.0, "den": 0, "over": 0, "long": 0})
|
|
498
|
+
win = self.pricing.context_window(v["priced_as"] or v["model"])
|
|
499
|
+
if v["u"] or not win:
|
|
500
|
+
u["over"] += v["n"]
|
|
501
|
+
continue
|
|
502
|
+
u["num"] += v["n"] * 100.0 * (v["avg_ctx"] or 0) / win
|
|
503
|
+
u["den"] += v["n"]
|
|
504
|
+
if v["priced_as"] and v["priced_as"].endswith("[1m]"):
|
|
505
|
+
u["long"] += v["n"]
|
|
335
506
|
for r in rows:
|
|
336
507
|
r["display_name"] = self.pricing.display_name(r["model"])
|
|
337
508
|
r["tier"] = self.pricing.tier(r["model"])
|
|
@@ -343,8 +514,10 @@ class Analytics:
|
|
|
343
514
|
r["output_per_input"] = (r["output_tokens"] / (r["input_tokens"] + r["cache_read_tokens"]
|
|
344
515
|
+ r["cache_write_tokens"])) if r["tokens"] else 0
|
|
345
516
|
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
346
|
-
|
|
347
|
-
|
|
517
|
+
u = util.get(r["model"], {})
|
|
518
|
+
r["utilization_pct"] = round(u["num"] / u["den"], 1) if u.get("den") else None
|
|
519
|
+
r["over_window_requests"] = u.get("over", 0)
|
|
520
|
+
r["long_context_requests"] = u.get("long", 0)
|
|
348
521
|
priced = [r for r in rows if r["tokens"] and r["tier"] != "none"]
|
|
349
522
|
superlatives = {}
|
|
350
523
|
if priced:
|
|
@@ -381,7 +554,7 @@ class Analytics:
|
|
|
381
554
|
r["budget_used_pct"] = round(100.0 * r["cost"] / b, 1) if b else None
|
|
382
555
|
return rows
|
|
383
556
|
|
|
384
|
-
def sessions(self, f=None, limit=500, order="cost"):
|
|
557
|
+
def sessions(self, f=None, limit=500, order="cost", offset=0):
|
|
385
558
|
w, p = self.where(f)
|
|
386
559
|
ob = {"cost": "cost DESC", "tokens": "tokens DESC", "duration": "s.duration_s DESC",
|
|
387
560
|
"recent": "s.started_at DESC", "prompts": "prompts DESC"}.get(order, "cost DESC")
|
|
@@ -398,17 +571,32 @@ class Analytics:
|
|
|
398
571
|
GROUP_CONCAT(DISTINCT r.model) models
|
|
399
572
|
FROM requests r JOIN sessions s ON s.id = r.session_id
|
|
400
573
|
JOIN projects pr ON pr.id = s.project_id
|
|
401
|
-
WHERE {w} GROUP BY s.id ORDER BY {ob} LIMIT ?""", p + [limit])
|
|
574
|
+
WHERE {w} GROUP BY s.id ORDER BY {ob} LIMIT ? OFFSET ?""", p + [limit, offset])
|
|
402
575
|
for r in rows:
|
|
403
576
|
r["cost_per_prompt"] = r["cost"] / r["prompts"] if r["prompts"] else None
|
|
404
577
|
r["tokens_per_prompt"] = r["tokens"] / r["prompts"] if r["prompts"] else None
|
|
405
578
|
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
406
|
-
r["output_ratio"] = r["output_tokens"] / r["tokens"] if r["tokens"] else 0
|
|
407
|
-
r["
|
|
408
|
-
|
|
409
|
-
if (r["cache_read_tokens"] + r["cache_write_tokens"]) else None)
|
|
579
|
+
r["output_ratio"] = (r["output_tokens"] or 0) / r["tokens"] if r["tokens"] else 0
|
|
580
|
+
cr, cw = r["cache_read_tokens"] or 0, r["cache_write_tokens"] or 0
|
|
581
|
+
r["cache_hit_ratio"] = (cr / (cr + cw)) if (cr + cw) else None
|
|
410
582
|
return rows
|
|
411
583
|
|
|
584
|
+
def sessions_total(self, f=None):
|
|
585
|
+
w, p = self.where(f)
|
|
586
|
+
return self.one(f"SELECT COUNT(DISTINCT s.id) n FROM requests r "
|
|
587
|
+
f"JOIN sessions s ON s.id = r.session_id "
|
|
588
|
+
f"JOIN projects pr ON pr.id = s.project_id WHERE {w}", p)["n"]
|
|
589
|
+
|
|
590
|
+
def prompts_total(self, f=None, search=None):
|
|
591
|
+
w, p = self.where(f)
|
|
592
|
+
extra, ep = "", []
|
|
593
|
+
if search:
|
|
594
|
+
extra = (" AND r.prompt_id IN (SELECT id FROM prompts pr WHERE pr.text LIKE ? "
|
|
595
|
+
"OR pr.session_id LIKE ? OR pr.category LIKE ?)")
|
|
596
|
+
ep = [f"%{search}%"] * 3
|
|
597
|
+
return self.one(f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r "
|
|
598
|
+
f"WHERE {w}{extra} AND r.prompt_id IS NOT NULL", p + ep)["n"]
|
|
599
|
+
|
|
412
600
|
def prompts(self, f=None, limit=300, offset=0, order="cost", search=None):
|
|
413
601
|
w, p = self.where(f)
|
|
414
602
|
ob = {"cost": "pcost DESC", "tokens": "ptokens DESC", "recent": "pr.ts DESC",
|
|
@@ -490,7 +678,6 @@ class Analytics:
|
|
|
490
678
|
p["category_evidence"] = json.loads(p.get("category_evidence") or "[]")
|
|
491
679
|
except Exception:
|
|
492
680
|
p["category_evidence"] = []
|
|
493
|
-
p["advisor"] = self.prompt_advisor(p)
|
|
494
681
|
return p
|
|
495
682
|
|
|
496
683
|
def session_detail(self, sid):
|
|
@@ -512,6 +699,11 @@ class Analytics:
|
|
|
512
699
|
s["files"] = self.q(
|
|
513
700
|
"SELECT path, GROUP_CONCAT(DISTINCT op) ops, COUNT(*) n FROM files_touched"
|
|
514
701
|
" WHERE session_id=? GROUP BY path ORDER BY n DESC LIMIT 100", (sid,))
|
|
702
|
+
# Cowork transcripts live under the desktop app's own config dir, where
|
|
703
|
+
# `claude --resume` would not find them.
|
|
704
|
+
cowork = "local-agent-mode-sessions" in (s.get("source_file") or "")
|
|
705
|
+
s["resume"] = (f"claude --resume {sid}"
|
|
706
|
+
if (s.get("agent") or "claude") == "claude" and not cowork else None)
|
|
515
707
|
return s
|
|
516
708
|
|
|
517
709
|
# ---------------- categories ----------------
|
|
@@ -543,6 +735,252 @@ class Analytics:
|
|
|
543
735
|
}
|
|
544
736
|
|
|
545
737
|
# ---------------- efficiency ----------------
|
|
738
|
+
def hygiene(self, f=None, top=12, trajectory_points=80):
|
|
739
|
+
"""Context and session hygiene: what it cost to keep re-sending a large prefix.
|
|
740
|
+
|
|
741
|
+
Everything here is observed. For each session: the context size of every
|
|
742
|
+
request in order, the request at which it first crossed each configured
|
|
743
|
+
threshold, and what was spent from that point on. Across the range: the share
|
|
744
|
+
of spend in requests above each threshold. No compaction is simulated and no
|
|
745
|
+
saving is estimated — how much a fresh session would have saved depends on what
|
|
746
|
+
the work still needed, which the transcript does not say.
|
|
747
|
+
|
|
748
|
+
Subagent (sidechain) turns are excluded: they run against their own prefix, so
|
|
749
|
+
mixing them into the parent session's trajectory would misstate both.
|
|
750
|
+
"""
|
|
751
|
+
cfg = self.settings.get("hygiene", {})
|
|
752
|
+
thresholds = sorted(int(x) for x in cfg.get("context_thresholds", [100000, 150000]))
|
|
753
|
+
w, p = self.where(f)
|
|
754
|
+
rows = self.q(f"""SELECT r.session_id, r.ts, r.context_tokens ctx, r.est_cost_usd cost
|
|
755
|
+
FROM requests r WHERE {w} AND r.is_sidechain = 0 AND r.ts <> ''
|
|
756
|
+
ORDER BY r.session_id, r.ts""", p)
|
|
757
|
+
compact_rows = self.q("SELECT session_id, ts FROM prompts WHERE source='slash:/compact'"
|
|
758
|
+
" AND ts <> '' ORDER BY session_id, ts")
|
|
759
|
+
compacts_by_session = defaultdict(list)
|
|
760
|
+
for cr in compact_rows:
|
|
761
|
+
compacts_by_session[cr["session_id"]].append(cr["ts"])
|
|
762
|
+
|
|
763
|
+
total = sum(r["cost"] or 0 for r in rows)
|
|
764
|
+
above = {t: {"requests": 0, "cost_usd": 0.0, "sessions": 0, "cost_after_first_cross_usd": 0.0}
|
|
765
|
+
for t in thresholds}
|
|
766
|
+
sessions = {}
|
|
767
|
+
for r in rows:
|
|
768
|
+
cost, ctx = (r["cost"] or 0.0), (r["ctx"] or 0)
|
|
769
|
+
s = sessions.setdefault(r["session_id"], {
|
|
770
|
+
"session_id": r["session_id"], "requests": 0, "cost_usd": 0.0,
|
|
771
|
+
"max_context": 0, "first_cross": {t: None for t in thresholds},
|
|
772
|
+
"first_cross_idx": {t: None for t in thresholds},
|
|
773
|
+
"cost_after": {t: 0.0 for t in thresholds}, "traj": [], "compactions": 0,
|
|
774
|
+
"ever_crossed": {t: False for t in thresholds}, "_next_compact_idx": 0})
|
|
775
|
+
prev_ctx = s["traj"][-1][0] if s["traj"] else 0
|
|
776
|
+
compact_ts = compacts_by_session.get(r["session_id"], [])
|
|
777
|
+
crossed_compact = (s["_next_compact_idx"] < len(compact_ts)
|
|
778
|
+
and r["ts"] > compact_ts[s["_next_compact_idx"]])
|
|
779
|
+
if crossed_compact:
|
|
780
|
+
s["_next_compact_idx"] += 1
|
|
781
|
+
if is_compaction(prev_ctx, ctx, thresholds[0]) or crossed_compact:
|
|
782
|
+
s["compactions"] += 1
|
|
783
|
+
s["first_cross"] = {t: None for t in thresholds}
|
|
784
|
+
idx = s["requests"]
|
|
785
|
+
s["requests"] += 1
|
|
786
|
+
s["cost_usd"] += cost
|
|
787
|
+
s["max_context"] = max(s["max_context"], ctx)
|
|
788
|
+
s["traj"].append((ctx, cost))
|
|
789
|
+
for t in thresholds:
|
|
790
|
+
if ctx >= t:
|
|
791
|
+
above[t]["requests"] += 1
|
|
792
|
+
above[t]["cost_usd"] += cost
|
|
793
|
+
if s["first_cross"][t] is None:
|
|
794
|
+
s["first_cross"][t] = idx
|
|
795
|
+
s["ever_crossed"][t] = True
|
|
796
|
+
if s["first_cross_idx"][t] is None:
|
|
797
|
+
s["first_cross_idx"][t] = idx # the ORIGINAL crossing; never reset
|
|
798
|
+
if s["first_cross"][t] is not None:
|
|
799
|
+
s["cost_after"][t] += cost
|
|
800
|
+
for s in sessions.values():
|
|
801
|
+
for t in thresholds:
|
|
802
|
+
if s["ever_crossed"][t]:
|
|
803
|
+
above[t]["sessions"] += 1
|
|
804
|
+
above[t]["cost_after_first_cross_usd"] += s["cost_after"][t]
|
|
805
|
+
|
|
806
|
+
rank_t = thresholds[-1]
|
|
807
|
+
ranked = sorted(sessions.values(), key=lambda s: -s["cost_after"][rank_t])[:top]
|
|
808
|
+
ids = [s["session_id"] for s in ranked]
|
|
809
|
+
meta = {}
|
|
810
|
+
if ids:
|
|
811
|
+
ph = ",".join("?" * len(ids))
|
|
812
|
+
meta = {m["id"]: m for m in self.q(f"""SELECT s.id, s.title, pj.name project
|
|
813
|
+
FROM sessions s JOIN projects pj ON pj.id=s.project_id
|
|
814
|
+
WHERE s.id IN ({ph})""", ids)}
|
|
815
|
+
|
|
816
|
+
def downsample(traj):
|
|
817
|
+
n = len(traj)
|
|
818
|
+
if n <= trajectory_points:
|
|
819
|
+
return traj
|
|
820
|
+
step = n / trajectory_points
|
|
821
|
+
return [traj[int(i * step)] for i in range(trajectory_points)]
|
|
822
|
+
|
|
823
|
+
out_sessions = []
|
|
824
|
+
for s in ranked:
|
|
825
|
+
m = meta.get(s["session_id"], {})
|
|
826
|
+
traj = downsample(s["traj"])
|
|
827
|
+
out_sessions.append({
|
|
828
|
+
"session_id": s["session_id"], "title": m.get("title"), "project": m.get("project"),
|
|
829
|
+
"requests": s["requests"], "cost_usd": s["cost_usd"], "max_context": s["max_context"],
|
|
830
|
+
"compactions": s["compactions"],
|
|
831
|
+
"first_cross": {str(t): s["first_cross"][t] for t in thresholds},
|
|
832
|
+
"first_cross_idx": {str(t): s["first_cross_idx"][t] for t in thresholds},
|
|
833
|
+
"ever_crossed": {str(t): s["ever_crossed"][t] for t in thresholds},
|
|
834
|
+
"cost_after": {str(t): s["cost_after"][t] for t in thresholds},
|
|
835
|
+
"cost_after_pct": {str(t): (round(100.0 * s["cost_after"][t] / s["cost_usd"], 1)
|
|
836
|
+
if s["cost_usd"] else 0.0) for t in thresholds},
|
|
837
|
+
"context_trajectory": [c for c, _ in traj],
|
|
838
|
+
"cumulative_cost": [round(x, 4) for x in _cumsum(cost for _, cost in traj)],
|
|
839
|
+
"basis": "actual",
|
|
840
|
+
})
|
|
841
|
+
return {
|
|
842
|
+
"thresholds": thresholds,
|
|
843
|
+
"rank_threshold": rank_t,
|
|
844
|
+
"requests": len(rows), "sessions": len(sessions), "cost_usd": total,
|
|
845
|
+
"above": {str(t): {
|
|
846
|
+
**v,
|
|
847
|
+
"share_pct": round(100.0 * v["cost_usd"] / total, 1) if total else 0.0,
|
|
848
|
+
"share_after_first_cross_pct": (round(100.0 * v["cost_after_first_cross_usd"] / total, 1)
|
|
849
|
+
if total else 0.0),
|
|
850
|
+
} for t, v in above.items()},
|
|
851
|
+
"sessions_ranked": out_sessions,
|
|
852
|
+
"excluded": "subagent turns (own prefix)",
|
|
853
|
+
"undetectable": ["/clear"],
|
|
854
|
+
"note": ("Observed shares of spend. Nothing here estimates what compaction or a "
|
|
855
|
+
"fresh session would have saved. Auto-compaction is not recorded; it is "
|
|
856
|
+
"detected as the context dropping by more than half. A typed /compact is "
|
|
857
|
+
"recorded and also counts."),
|
|
858
|
+
"basis": "actual",
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
def ttl_replay(self, f=None):
|
|
862
|
+
"""5m vs 1h cache TTL over real segments — arithmetic, no behavioural assumption.
|
|
863
|
+
|
|
864
|
+
Gated on reconciliation: if replaying the TTL you actually used cannot reproduce
|
|
865
|
+
the cost that was logged, the counterfactual is not trustworthy either and no
|
|
866
|
+
number is returned.
|
|
867
|
+
"""
|
|
868
|
+
from .segments import replay, split_segments
|
|
869
|
+
w, p = self.where(f)
|
|
870
|
+
turns = self.q(f"""SELECT r.session_id, r.ts, r.model, r.priced_as, r.is_sidechain, r.agent_id,
|
|
871
|
+
r.input_tokens, r.output_tokens, r.cache_read_tokens,
|
|
872
|
+
r.cache_write_5m, r.cache_write_1h, r.est_cost_usd, r.context_tokens ctx
|
|
873
|
+
FROM requests r WHERE {w} AND r.agent='claude' AND r.ts <> ''
|
|
874
|
+
ORDER BY r.session_id, r.ts""", p)
|
|
875
|
+
segs = split_segments(turns)
|
|
876
|
+
out = replay(segs, self.pricing)
|
|
877
|
+
out["turns"] = len(turns)
|
|
878
|
+
out["undetectable_boundaries"] = ["/clear"]
|
|
879
|
+
out["note"] = ("Segments break at session start, subagent start, model change and "
|
|
880
|
+
"compaction. Auto-compaction is not recorded; it is detected as the "
|
|
881
|
+
"context dropping by more than half. A typed /compact is recorded and "
|
|
882
|
+
"also counts.")
|
|
883
|
+
return out
|
|
884
|
+
|
|
885
|
+
def long_context_pricing(self, f=None):
|
|
886
|
+
"""Requests whose context exceeded the model's standard window.
|
|
887
|
+
|
|
888
|
+
These could only have been served by the long-context variant, which bills at a
|
|
889
|
+
premium. Where a `[1m]` price list exists they are already repriced; where it
|
|
890
|
+
does not, they are billed at the standard rate and the estimate is LOW — that is
|
|
891
|
+
a gap in config/pricing.json, not in the data, so it is reported rather than
|
|
892
|
+
guessed at.
|
|
893
|
+
"""
|
|
894
|
+
w, p = self.where(f)
|
|
895
|
+
rows = self.q(f"""SELECT r.model, r.priced_as, r.unpriced_long_context u,
|
|
896
|
+
COUNT(*) n, SUM(r.est_cost_usd) cost, MAX(r.context_tokens) mx
|
|
897
|
+
FROM requests r WHERE {w} AND r.context_tokens > 0
|
|
898
|
+
AND (r.priced_as LIKE '%[1m]' OR r.unpriced_long_context = 1)
|
|
899
|
+
GROUP BY r.model, r.priced_as, r.unpriced_long_context""", p)
|
|
900
|
+
repriced = [r for r in rows if not r["u"]]
|
|
901
|
+
unpriced = [r for r in rows if r["u"]]
|
|
902
|
+
|
|
903
|
+
# rows priced against no known model at all (model_known=0) are a distinct gap:
|
|
904
|
+
# not "over the standard window", but "no rate for this model in pricing.json".
|
|
905
|
+
unknown_rows = self.q(f"""SELECT r.model, COUNT(*) n, SUM(r.est_cost_usd) cost
|
|
906
|
+
FROM requests r WHERE {w} AND r.model_known = 0
|
|
907
|
+
AND r.model LIKE 'claude%'
|
|
908
|
+
GROUP BY r.model""", p)
|
|
909
|
+
unknown_requests = sum(r["n"] for r in unknown_rows)
|
|
910
|
+
unknown_models = sorted({r["model"] for r in unknown_rows})
|
|
911
|
+
|
|
912
|
+
return {
|
|
913
|
+
"repriced": repriced,
|
|
914
|
+
"unpriced": unpriced,
|
|
915
|
+
"repriced_requests": sum(r["n"] for r in repriced),
|
|
916
|
+
"unpriced_requests": sum(r["n"] for r in unpriced),
|
|
917
|
+
"unpriced_cost_usd": sum(r["cost"] or 0 for r in unpriced),
|
|
918
|
+
"unpriced_models": sorted({r["model"] for r in unpriced}),
|
|
919
|
+
"unknown_model_requests": unknown_requests,
|
|
920
|
+
"unknown_model_cost_usd": sum(r["cost"] or 0 for r in unknown_rows),
|
|
921
|
+
"unknown_models": unknown_models,
|
|
922
|
+
"message": (
|
|
923
|
+
"%d requests exceeded their model's standard context window with no "
|
|
924
|
+
"long-context price configured, so their cost is understated. Add a "
|
|
925
|
+
"\"<model>[1m]\" entry to config/pricing.json for: %s."
|
|
926
|
+
% (sum(r["n"] for r in unpriced), ", ".join(sorted({r["model"] for r in unpriced})))
|
|
927
|
+
if unpriced else ""),
|
|
928
|
+
"unknown_message": (
|
|
929
|
+
"%d requests used a claude-* model with no entry in pricing.json at "
|
|
930
|
+
"all, so their cost is a fallback guess. Add pricing.json entries for: %s."
|
|
931
|
+
% (unknown_requests, ", ".join(unknown_models))
|
|
932
|
+
if unknown_rows else ""),
|
|
933
|
+
"basis": "estimated",
|
|
934
|
+
}
|
|
935
|
+
|
|
936
|
+
def cache_cost_split(self, f=None):
|
|
937
|
+
"""Cache read vs write split by estimated cost, not just by token count.
|
|
938
|
+
|
|
939
|
+
A read is billed at a fraction of the input rate and a write at a premium, so
|
|
940
|
+
the token split and the dollar split are different numbers — writes are a small
|
|
941
|
+
share of cache tokens and a much larger share of cache spend. Reporting only the
|
|
942
|
+
token ratio overstates how healthy caching is, which is why the scorecard grades
|
|
943
|
+
this dimension on cost. Prices are re-derived per model here rather than read off
|
|
944
|
+
requests.est_cost_usd, which is a single blended figure per request.
|
|
945
|
+
"""
|
|
946
|
+
w, p = self.where(f)
|
|
947
|
+
rows = self.q(f"""SELECT model,
|
|
948
|
+
SUM(cache_read_tokens) cr,
|
|
949
|
+
SUM(cache_write_5m) w5, SUM(cache_write_1h) w1
|
|
950
|
+
FROM requests r WHERE {w} GROUP BY model""", p)
|
|
951
|
+
read_tok = w5_tok = w1_tok = 0
|
|
952
|
+
read_cost = w5_cost = w1_cost = 0.0
|
|
953
|
+
for r in rows:
|
|
954
|
+
cr, w5, w1 = (r["cr"] or 0), (r["w5"] or 0), (r["w1"] or 0)
|
|
955
|
+
read_tok += cr
|
|
956
|
+
w5_tok += w5
|
|
957
|
+
w1_tok += w1
|
|
958
|
+
read_cost += self.pricing.estimate(r["model"], cache_read=cr)
|
|
959
|
+
w5_cost += self.pricing.estimate(r["model"], cache_write_5m=w5)
|
|
960
|
+
w1_cost += self.pricing.estimate(r["model"], cache_write_1h=w1)
|
|
961
|
+
|
|
962
|
+
write_tok = w5_tok + w1_tok
|
|
963
|
+
write_cost = w5_cost + w1_cost
|
|
964
|
+
tok_total = read_tok + write_tok
|
|
965
|
+
cost_total = read_cost + write_cost
|
|
966
|
+
per_read = (read_cost / read_tok) if read_tok else 0
|
|
967
|
+
per_write = (write_cost / write_tok) if write_tok else 0
|
|
968
|
+
return {
|
|
969
|
+
"read_tokens": read_tok, "write_tokens": write_tok,
|
|
970
|
+
"write_5m_tokens": w5_tok, "write_1h_tokens": w1_tok,
|
|
971
|
+
"read_cost_usd": read_cost, "write_cost_usd": write_cost,
|
|
972
|
+
"write_5m_cost_usd": w5_cost, "write_1h_cost_usd": w1_cost,
|
|
973
|
+
"read_token_share": (read_tok / tok_total) if tok_total else None,
|
|
974
|
+
"read_cost_share": (read_cost / cost_total) if cost_total else None,
|
|
975
|
+
"write_cost_share": (write_cost / cost_total) if cost_total else None,
|
|
976
|
+
# how much more a write token costs than a read token, same workload
|
|
977
|
+
"write_vs_read_multiple": (per_write / per_read) if per_read else None,
|
|
978
|
+
# 1h writes cost more per token than 5m writes; whether that premium is worth
|
|
979
|
+
# paying depends on the real inter-turn gaps, which ttl_replay() prices
|
|
980
|
+
"write_1h_token_share": (w1_tok / write_tok) if write_tok else None,
|
|
981
|
+
"basis": "estimated",
|
|
982
|
+
}
|
|
983
|
+
|
|
546
984
|
def efficiency(self, f=None):
|
|
547
985
|
w, p = self.where(f)
|
|
548
986
|
t = self.one(f"""SELECT SUM(input_tokens) i, SUM(output_tokens) o,
|
|
@@ -554,6 +992,29 @@ class Analytics:
|
|
|
554
992
|
tot = t["tot"] or 1
|
|
555
993
|
prompt_side = (t["i"] or 0) + (t["cr"] or 0) + (t["cw"] or 0)
|
|
556
994
|
cache_total = (t["cr"] or 0) + (t["cw"] or 0)
|
|
995
|
+
cache_cost = self.cache_cost_split(f)
|
|
996
|
+
|
|
997
|
+
# Break-even margin: how much of cache spend came back as read discount, net of
|
|
998
|
+
# write premium. +1 = all discount, -1 = all premium, computed per model since
|
|
999
|
+
# rates differ.
|
|
1000
|
+
bm_rows = self.q(f"""SELECT COALESCE(priced_as, model) priced_as, SUM(cache_read_tokens) cr,
|
|
1001
|
+
SUM(cache_write_5m) w5, SUM(cache_write_1h) w1
|
|
1002
|
+
FROM requests r WHERE {w} GROUP BY COALESCE(priced_as, model)""", p)
|
|
1003
|
+
discount = premium = cache_cost_total = 0.0
|
|
1004
|
+
for r in bm_rows:
|
|
1005
|
+
model = r["priced_as"]
|
|
1006
|
+
reads, w5, w1 = (r["cr"] or 0), (r["w5"] or 0), (r["w1"] or 0)
|
|
1007
|
+
rt = self.pricing.rates(model) or {}
|
|
1008
|
+
inp = float(rt.get("input") or 0)
|
|
1009
|
+
read_rate = float(rt.get("cache_read") or 0)
|
|
1010
|
+
w5_rate = float(rt.get("cache_write_5m") or 0)
|
|
1011
|
+
w1_rate = float(rt.get("cache_write_1h") or 0)
|
|
1012
|
+
discount += reads * (inp - read_rate) / 1e6
|
|
1013
|
+
premium += (w5 * (w5_rate - inp) + w1 * (w1_rate - inp)) / 1e6
|
|
1014
|
+
cache_cost_total += (reads * read_rate + w5 * w5_rate + w1 * w1_rate) / 1e6
|
|
1015
|
+
breakeven_margin = (max(-1.0, min(1.0, (discount - premium) / cache_cost_total))
|
|
1016
|
+
if cache_cost_total else None)
|
|
1017
|
+
|
|
557
1018
|
sess = self.sessions(f, limit=100000, order="cost")
|
|
558
1019
|
scored = [s for s in sess if s["tokens"] and s["prompts"]]
|
|
559
1020
|
for s in scored:
|
|
@@ -564,6 +1025,7 @@ class Analytics:
|
|
|
564
1025
|
"output_per_input": ((t["o"] or 0) / prompt_side) if prompt_side else 0,
|
|
565
1026
|
"thinking_share_of_output": ((t["think"] or 0) / (t["o"] or 1)),
|
|
566
1027
|
"cache_hit_ratio": ((t["cr"] or 0) / cache_total) if cache_total else None,
|
|
1028
|
+
"cache_read_cost_share": cache_cost["read_cost_share"],
|
|
567
1029
|
"tokens_per_request": tot / (t["n"] or 1),
|
|
568
1030
|
"avg_context_tokens": t["avgctx"],
|
|
569
1031
|
"cost_per_1k_output": (1000.0 * (t["cost"] or 0) / (t["o"] or 1)),
|
|
@@ -571,10 +1033,14 @@ class Analytics:
|
|
|
571
1033
|
max(self.one(f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r WHERE {w}", p)["n"], 1)),
|
|
572
1034
|
"cache": {
|
|
573
1035
|
"reads": t["cr"], "writes": t["cw"],
|
|
1036
|
+
"cost_split": cache_cost,
|
|
1037
|
+
"breakeven_margin": breakeven_margin,
|
|
574
1038
|
"cost_with_cache": t["cost"], "cost_without_cache": t["cost_nc"],
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
1039
|
+
# Named for what it is. This used to be "estimated_savings_usd", and the
|
|
1040
|
+
# UI called it a saving; it is the gap to a run that never happened.
|
|
1041
|
+
"uncached_counterfactual_delta_usd": (t["cost_nc"] or 0) - (t["cost"] or 0),
|
|
1042
|
+
"uncached_counterfactual_pct": (round(100.0 * ((t["cost_nc"] or 0) - (t["cost"] or 0))
|
|
1043
|
+
/ (t["cost_nc"] or 1), 1)),
|
|
578
1044
|
"basis": "estimated",
|
|
579
1045
|
},
|
|
580
1046
|
"low_efficiency_sessions": scored[:10],
|
|
@@ -662,23 +1128,22 @@ class Analytics:
|
|
|
662
1128
|
ORDER BY cost DESC LIMIT 15""", p + [rules["long_prompt_chars"]])
|
|
663
1129
|
budget_chars = rules["long_prompt_chars"]
|
|
664
1130
|
for r in rows:
|
|
665
|
-
|
|
666
|
-
resent = over_tokens * max(r["requests"], 1)
|
|
667
|
-
r["excess"] = (r["cost"] * resent / r["tokens"]) if r["tokens"] else 0
|
|
1131
|
+
r["excess"] = 0.0
|
|
668
1132
|
if rows:
|
|
669
1133
|
add("high", "long_prompts",
|
|
670
1134
|
f"{len(rows)} very long prompts (>{budget_chars:,} chars)",
|
|
671
1135
|
"Long pasted prompts inflate the cached prefix re-sent on every following turn.",
|
|
672
1136
|
rows, "Move large pasted context into a file and reference it, or summarize first.",
|
|
673
1137
|
"prompt_id",
|
|
674
|
-
|
|
1138
|
+
"none claimed — flagged for review only")
|
|
675
1139
|
|
|
676
|
-
# 2. duplicate prompts — excess is the cost of the repeats, not the first ask
|
|
677
|
-
|
|
1140
|
+
# 2. duplicate prompts — excess is the cost of the repeats, not the first ask,
|
|
1141
|
+
# counted only within a single session so cross-session coincidences don't count.
|
|
1142
|
+
dups = self.q(f"""SELECT pr.norm_hash, pr.session_id, COUNT(*) n, substr(MIN(pr.text),1,160) preview,
|
|
678
1143
|
SUM(pr.est_cost_usd) cost, SUM(pr.billable_tokens) tokens,
|
|
679
1144
|
MIN(pr.est_cost_usd) first_cost, GROUP_CONCAT(pr.id) prompt_ids
|
|
680
1145
|
FROM prompts pr WHERE {pfilter} AND pr.char_len > 25
|
|
681
|
-
GROUP BY pr.norm_hash HAVING n > 1
|
|
1146
|
+
GROUP BY pr.norm_hash, pr.session_id HAVING n > 1
|
|
682
1147
|
ORDER BY cost DESC LIMIT 15""", p)
|
|
683
1148
|
for d in dups:
|
|
684
1149
|
d["prompt_id"] = int(d["prompt_ids"].split(",")[0])
|
|
@@ -692,11 +1157,18 @@ class Analytics:
|
|
|
692
1157
|
|
|
693
1158
|
# 3. low-yield sessions — excess is what the session cost ABOVE what the same
|
|
694
1159
|
# output would have cost at your own median session efficiency.
|
|
1160
|
+
# The baseline is drawn from the SAME population that is eligible to be
|
|
1161
|
+
# flagged — sessions above huge_session_tokens, inside the current filter.
|
|
1162
|
+
# Grading big sessions against the median of all sessions punishes them for
|
|
1163
|
+
# something inherent to long agentic work: output ratio falls as a session
|
|
1164
|
+
# grows, so a small-session median flags most large sessions by construction.
|
|
695
1165
|
ratios = [r["x"] for r in self.q(
|
|
696
|
-
"SELECT CAST(output_tokens AS REAL)/billable_tokens x FROM sessions
|
|
697
|
-
|
|
1166
|
+
f"""SELECT CAST(s.output_tokens AS REAL)/s.billable_tokens x FROM sessions s
|
|
1167
|
+
WHERE {sfilter} AND s.billable_tokens > ?""",
|
|
1168
|
+
p + [rules["huge_session_tokens"]])]
|
|
698
1169
|
median_ratio = statistics.median(ratios) if ratios else 0.0
|
|
699
1170
|
cutoff = median_ratio * rules["low_output_ratio_vs_median"]
|
|
1171
|
+
baseline_n = len(ratios)
|
|
700
1172
|
low = self.q(f"""SELECT s.id session_id, s.title, s.billable_tokens tokens,
|
|
701
1173
|
s.output_tokens out_tokens, s.est_cost_usd cost, s.request_count requests,
|
|
702
1174
|
(CAST(s.output_tokens AS REAL)/MAX(s.billable_tokens,1)) output_ratio,
|
|
@@ -706,16 +1178,15 @@ class Analytics:
|
|
|
706
1178
|
ORDER BY cost DESC LIMIT 15""",
|
|
707
1179
|
p + [rules["huge_session_tokens"], cutoff])
|
|
708
1180
|
for r in low:
|
|
709
|
-
|
|
710
|
-
# baseline cost scales by (actual ratio / median ratio)
|
|
711
|
-
r["excess"] = r["cost"] * (1 - (r["output_ratio"] / median_ratio)) if median_ratio else 0
|
|
1181
|
+
r["excess"] = 0.0
|
|
712
1182
|
if low:
|
|
713
|
-
add("
|
|
1183
|
+
add("medium", "low_yield_sessions",
|
|
714
1184
|
f"{len(low)} large sessions yielded under {cutoff*100:.2f}% output tokens",
|
|
715
|
-
f"
|
|
716
|
-
f"These ran well below
|
|
1185
|
+
f"Among your {baseline_n} comparably large sessions the median turns "
|
|
1186
|
+
f"{median_ratio*100:.2f}% of billable tokens into output. These ran well below "
|
|
1187
|
+
f"that while consuming heavy context.",
|
|
717
1188
|
low, "Start a fresh session or /compact once a thread stops producing new output.",
|
|
718
|
-
"session_id", "
|
|
1189
|
+
"session_id", "none claimed — the same output at another ratio is a counterfactual")
|
|
719
1190
|
|
|
720
1191
|
# 4. frontier model on small tasks — excess is computed against the cheaper tier.
|
|
721
1192
|
frontier = [m for m, v in self.pricing.models.items()
|
|
@@ -733,59 +1204,70 @@ class Analytics:
|
|
|
733
1204
|
small = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.category,
|
|
734
1205
|
pr.session_id, pr.est_cost_usd cost, pr.output_tokens out_tokens,
|
|
735
1206
|
pr.billable_tokens tokens, pr.models, pr.input_tokens,
|
|
736
|
-
pr.cache_read_tokens, pr.cache_write_tokens
|
|
1207
|
+
pr.cache_read_tokens, pr.cache_write_tokens,
|
|
1208
|
+
(SELECT COALESCE(SUM(rq.cache_write_5m),0) FROM requests rq
|
|
1209
|
+
WHERE rq.prompt_id = pr.id) c5,
|
|
1210
|
+
(SELECT COALESCE(SUM(rq.cache_write_1h),0) FROM requests rq
|
|
1211
|
+
WHERE rq.prompt_id = pr.id) c1
|
|
737
1212
|
FROM prompts pr WHERE {pfilter}
|
|
738
|
-
AND pr.output_tokens < ? AND pr.est_cost_usd > 0
|
|
1213
|
+
AND pr.output_tokens < ? AND pr.tool_calls = 0 AND pr.est_cost_usd > 0
|
|
739
1214
|
AND EXISTS (SELECT 1 FROM requests r2 WHERE r2.prompt_id=pr.id
|
|
740
1215
|
AND r2.model IN ({ph}))
|
|
741
1216
|
ORDER BY cost DESC LIMIT 15""",
|
|
742
1217
|
p + [rules["simple_task_output_tokens"]] + frontier)
|
|
743
1218
|
for r in small:
|
|
744
|
-
|
|
745
|
-
r["cache_read_tokens"], r["cache_write_tokens"], 0)
|
|
746
|
-
if cheaper else r["cost"])
|
|
747
|
-
r["excess"] = max(r["cost"] - alt, 0)
|
|
1219
|
+
r["excess"] = 0.0
|
|
748
1220
|
if small:
|
|
749
1221
|
add("medium", "frontier_on_small_tasks",
|
|
750
1222
|
f"{len(small)} frontier-model prompts produced under "
|
|
751
1223
|
f"{rules['simple_task_output_tokens']} output tokens",
|
|
752
1224
|
"Short, simple turns running on the most expensive model tier.",
|
|
753
1225
|
small, "Route short lookups and confirmations to a cheaper model tier.",
|
|
754
|
-
"prompt_id",
|
|
755
|
-
f"difference against the same tokens priced at {self.pricing.display_name(cheaper)}"
|
|
756
|
-
if cheaper else "n/a")
|
|
1226
|
+
"prompt_id", "none claimed")
|
|
757
1227
|
|
|
758
1228
|
# 5. tool loops — excess is the share of the loop beyond the threshold.
|
|
1229
|
+
loop_calls = rules.get("tool_loop_calls", 40)
|
|
759
1230
|
loops = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.session_id,
|
|
760
1231
|
pr.tool_calls tools, pr.est_cost_usd cost, pr.billable_tokens tokens
|
|
761
|
-
FROM prompts pr WHERE {pfilter} AND pr.tool_calls >
|
|
762
|
-
ORDER BY cost DESC LIMIT 15""", p)
|
|
1232
|
+
FROM prompts pr WHERE {pfilter} AND pr.tool_calls > ?
|
|
1233
|
+
ORDER BY cost DESC LIMIT 15""", p + [loop_calls])
|
|
763
1234
|
for r in loops:
|
|
764
|
-
r["excess"] =
|
|
1235
|
+
r["excess"] = 0.0
|
|
765
1236
|
if loops:
|
|
766
|
-
add("medium", "tool_loops", f"{len(loops)} prompts triggered
|
|
1237
|
+
add("medium", "tool_loops", f"{len(loops)} prompts triggered {loop_calls}+ tool calls",
|
|
767
1238
|
"Long agentic loops re-send the whole conversation each step, so cost grows super-linearly.",
|
|
768
1239
|
loops, "Split the task, or give more precise instructions up front.", "prompt_id",
|
|
769
|
-
"
|
|
1240
|
+
"none claimed")
|
|
770
1241
|
|
|
771
|
-
# 6. poor cache reuse — excess is the write premium over
|
|
1242
|
+
# 6. poor cache reuse — excess is the break-even: the cache-write premium over
|
|
1243
|
+
# plain input pricing, minus the discount actually earned on the reads.
|
|
772
1244
|
poor = self.q(f"""SELECT s.id session_id, s.title, proj.name project,
|
|
773
1245
|
s.cache_read_tokens reads, s.cache_write_tokens writes,
|
|
1246
|
+
(SELECT COALESCE(SUM(r.cache_write_5m),0) FROM requests r WHERE r.session_id=s.id) w5,
|
|
1247
|
+
(SELECT COALESCE(SUM(r.cache_write_1h),0) FROM requests r WHERE r.session_id=s.id) w1,
|
|
774
1248
|
s.est_cost_usd cost, s.request_count requests, s.models
|
|
775
1249
|
FROM sessions s JOIN projects proj ON proj.id=s.project_id
|
|
776
|
-
WHERE {sfilter} AND s.cache_write_tokens >
|
|
777
|
-
|
|
778
|
-
|
|
1250
|
+
WHERE {sfilter} AND s.cache_write_tokens > ?
|
|
1251
|
+
ORDER BY cost DESC""", p + [rules.get("poor_cache_min_writes", 500_000)])
|
|
1252
|
+
flagged = []
|
|
779
1253
|
for r in poor:
|
|
780
1254
|
m = (r["models"] or "").split(",")[0]
|
|
781
1255
|
rt = self.pricing.rates(m)
|
|
782
|
-
|
|
783
|
-
|
|
1256
|
+
g = lambda k: float(rt.get(k, 0.0))
|
|
1257
|
+
# break-even: premium paid on writes minus discount earned on reads
|
|
1258
|
+
net = (r["w5"] * (g("cache_write_5m") - g("input"))
|
|
1259
|
+
+ r["w1"] * (g("cache_write_1h") - g("input"))
|
|
1260
|
+
- r["reads"] * (g("input") - g("cache_read"))) / 1_000_000.0
|
|
1261
|
+
if net > 0:
|
|
1262
|
+
r["excess"] = net
|
|
1263
|
+
flagged.append(r)
|
|
1264
|
+
poor = flagged[:15]
|
|
784
1265
|
if poor:
|
|
785
1266
|
add("medium", "poor_cache_reuse", f"{len(poor)} sessions wrote cache they barely reused",
|
|
786
1267
|
"Cache writes cost more than plain input; they only pay off when read back repeatedly.",
|
|
787
1268
|
poor, "Keep related work in one continuous session so the cached prefix is reused.",
|
|
788
|
-
"session_id", "
|
|
1269
|
+
"session_id", "cache-write premium minus the read discount actually earned, "
|
|
1270
|
+
"at this model's rates")
|
|
789
1271
|
|
|
790
1272
|
# 7. long-lived sparse sessions — informational, no excess claimed.
|
|
791
1273
|
idle = self.q(f"""SELECT s.id session_id, s.title, proj.name project, s.duration_s,
|
|
@@ -863,436 +1345,120 @@ class Analytics:
|
|
|
863
1345
|
"touched — money worth reviewing. Estimated excess is how much more that "
|
|
864
1346
|
"work cost than a reasonable baseline, and is the actual waste figure. "
|
|
865
1347
|
"Both are estimates.",
|
|
1348
|
+
"excess_note": "Estimated excess is claimed only where the baseline is measured: the cost of "
|
|
1349
|
+
"repeating an identical prompt in the same session, and cache writes that were "
|
|
1350
|
+
"never read back enough to pay for themselves. Everything else is exposed spend "
|
|
1351
|
+
"to review, not waste.",
|
|
866
1352
|
"basis": "estimated"}
|
|
867
1353
|
|
|
868
1354
|
# ---------------- recommendations ----------------
|
|
869
|
-
#
|
|
870
|
-
#
|
|
871
|
-
#
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
"
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
TIER_RANK = {"economy": 0, "balanced": 1, "frontier": 2}
|
|
883
|
-
|
|
884
|
-
# How each vendor's agent is told to change model. Used by model_switch() and by the
|
|
885
|
-
# model_downgrade recommendations, which must name the agent they keep you inside.
|
|
886
|
-
SWITCH_HOW_BY_AGENT = {
|
|
887
|
-
"anthropic": {"agent": "Claude Code", "session": "/model <name>",
|
|
888
|
-
"project": '"model": "<name>" in <repo>/.claude/settings.json'},
|
|
889
|
-
"openai": {"agent": "Codex", "session": "/model in Codex, or codex -m <name>",
|
|
890
|
-
"project": 'model = "<name>" in ~/.codex/config.toml (or a profile)'},
|
|
891
|
-
"google": {"agent": "Gemini CLI", "session": "/model in Gemini CLI, or gemini -m <name>",
|
|
892
|
-
"project": '"model": {"name": "<name>"} in <repo>/.gemini/settings.json'},
|
|
893
|
-
}
|
|
894
|
-
|
|
895
|
-
def _cheapest(self, tier, provider=None):
|
|
896
|
-
"""Cheapest model in a tier — from the same provider, since an agent can only
|
|
897
|
-
switch between its own vendor's models (Codex can't run Haiku)."""
|
|
898
|
-
ms = [m for m, v in self.pricing.models.items() if v.get("tier") == tier
|
|
899
|
-
and (provider is None or v.get("provider", "anthropic") == provider)]
|
|
900
|
-
return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
|
|
901
|
-
|
|
902
|
-
# ---------------- evidence: what the cheaper model actually did ----------------
|
|
903
|
-
# model_switch() reprices your tokens on a cheaper model, which assumes the cheaper
|
|
904
|
-
# model would have done the same work in the same number of turns. Often it would
|
|
905
|
-
# not: a weaker model can take five times the turns on the same task, and the
|
|
906
|
-
# repricing then promises a saving that never arrives.
|
|
907
|
-
#
|
|
908
|
-
# Where you have already run more than one model on the same kind of work, we do not
|
|
909
|
-
# have to assume anything. This compares what each model actually cost per prompt on
|
|
910
|
-
# that category, and how much work it took to get there.
|
|
911
|
-
|
|
912
|
-
MIN_PROMPTS = 8 # below this a per-category average is noise, not evidence
|
|
913
|
-
SAVING_FLOOR_PCT = 20 # smaller gaps are inside the noise of what you happened to ask
|
|
914
|
-
TURN_TOLERANCE = 1.35 # more turns than this and the cheaper model was grinding
|
|
915
|
-
REPEAT_TOLERANCE = 12.0 # percentage points of extra re-asking we will accept
|
|
916
|
-
|
|
917
|
-
def model_evidence(self, f=None):
|
|
918
|
-
"""Back-test a model switch against your own history.
|
|
919
|
-
|
|
920
|
-
For every category where you ran more than one model, report what each one
|
|
921
|
-
actually cost per prompt and what it took: turns, tool calls, and how often you
|
|
922
|
-
had to ask the same thing again. A candidate is only recommended when it was
|
|
923
|
-
genuinely cheaper per prompt *and* did not need materially more work to get
|
|
924
|
-
there — which is the part a repricing cannot see.
|
|
1355
|
+
# How close to the ceiling counts as "near". 90% was the cut the old reprice used
|
|
1356
|
+
# to decide a request could not move to a smaller window; it is kept as the
|
|
1357
|
+
# observation threshold because that is where re-read cost visibly concentrates.
|
|
1358
|
+
NEAR_WINDOW_PCT = 0.9
|
|
1359
|
+
|
|
1360
|
+
def context_window_fit(self, f=None):
|
|
1361
|
+
"""How much spend ran near the ceiling of the context window actually in use.
|
|
1362
|
+
|
|
1363
|
+
This is the one piece of the old model-switch reprice worth keeping — the
|
|
1364
|
+
window check — turned from a what-if into an observation. Nothing is
|
|
1365
|
+
repriced and no alternative model is assumed. A request is "near" when its
|
|
1366
|
+
prompt side is at least NEAR_WINDOW_PCT of its model's window, and "over"
|
|
1367
|
+
when it exceeds it, which can only mean the long-context variant served it.
|
|
925
1368
|
"""
|
|
926
1369
|
w, p = self.where(f)
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
clean = ("pr.models IS NOT NULL AND pr.models NOT LIKE '%,%' "
|
|
934
|
-
"AND pr.models != '<synthetic>' AND pr.est_cost_usd > 0")
|
|
935
|
-
rows = self.q(f"""SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
936
|
-
pr.agent agent, COUNT(*) prompts, SUM(pr.est_cost_usd) cost,
|
|
937
|
-
AVG(pr.est_cost_usd) cost_per_prompt,
|
|
938
|
-
AVG(pr.request_count) turns, AVG(pr.tool_calls) tools,
|
|
939
|
-
AVG(pr.output_tokens) out_tokens, AVG(pr.max_context_tokens) ctx
|
|
940
|
-
FROM prompts pr WHERE {scope} AND {clean}
|
|
941
|
-
GROUP BY 1, 2, 3 HAVING prompts >= ?""", p + [self.MIN_PROMPTS])
|
|
942
|
-
|
|
943
|
-
repeats = {(r["category"], r["model"]): r["pct"] for r in self.q(f"""
|
|
944
|
-
SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
945
|
-
ROUND(100.0 * SUM(CASE WHEN dup.n > 1 THEN 1 ELSE 0 END) / COUNT(*), 1) pct
|
|
946
|
-
FROM prompts pr
|
|
947
|
-
LEFT JOIN (SELECT norm_hash, COUNT(*) n FROM prompts GROUP BY norm_hash) dup
|
|
948
|
-
ON dup.norm_hash = pr.norm_hash
|
|
949
|
-
WHERE {scope} AND {clean}
|
|
950
|
-
GROUP BY 1, 2""", p)}
|
|
951
|
-
|
|
952
|
-
by_cat = defaultdict(list)
|
|
953
|
-
for r in rows:
|
|
954
|
-
r["repeat_pct"] = repeats.get((r["category"], r["model"]), 0.0) or 0.0
|
|
955
|
-
r["name"] = self.pricing.display_name(r["model"])
|
|
956
|
-
r["tier"] = self.pricing.tier(r["model"])
|
|
957
|
-
by_cat[r["category"]].append(r)
|
|
958
|
-
|
|
959
|
-
out, total_save = [], 0.0
|
|
960
|
-
for cat, models in by_cat.items():
|
|
961
|
-
if len(models) < 2:
|
|
962
|
-
continue
|
|
963
|
-
# The incumbent is what you spend the most on here — that is the bill a
|
|
964
|
-
# switch would actually change.
|
|
965
|
-
cur = max(models, key=lambda m: m["cost"])
|
|
966
|
-
rule = self.SWITCH_RULES.get(cat, ("balanced", "low"))[0]
|
|
967
|
-
cands = []
|
|
968
|
-
for m in models:
|
|
969
|
-
if m["model"] == cur["model"] or m["cost_per_prompt"] >= cur["cost_per_prompt"]:
|
|
970
|
-
continue
|
|
971
|
-
# Only models the same agent can run. Telling a Claude Code user to use a
|
|
972
|
-
# GPT model is not a setting change, it is a different tool, and the
|
|
973
|
-
# comparison would be between two different ways of working.
|
|
974
|
-
if m["agent"] != cur["agent"]:
|
|
975
|
-
continue
|
|
976
|
-
save_pct = 100.0 * (1 - m["cost_per_prompt"] / cur["cost_per_prompt"])
|
|
977
|
-
turn_ratio = (m["turns"] / cur["turns"]) if cur["turns"] else 1.0
|
|
978
|
-
repeat_delta = m["repeat_pct"] - cur["repeat_pct"]
|
|
979
|
-
# Observed, not repriced: what your own prompts cost on each side.
|
|
980
|
-
save = (cur["cost_per_prompt"] - m["cost_per_prompt"]) * cur["prompts"]
|
|
981
|
-
if save_pct < self.SAVING_FLOOR_PCT:
|
|
982
|
-
verdict, why = "marginal", (
|
|
983
|
-
f"Only {save_pct:.0f}% cheaper per prompt — inside the noise of what "
|
|
984
|
-
f"you happened to ask each model.")
|
|
985
|
-
elif turn_ratio > self.TURN_TOLERANCE:
|
|
986
|
-
verdict, why = "risky", (
|
|
987
|
-
f"Cost {save_pct:.0f}% less per prompt but took {turn_ratio:.1f}x the "
|
|
988
|
-
f"turns ({m['turns']:.0f} vs {cur['turns']:.0f}). It got there by "
|
|
989
|
-
f"grinding, and that is the cost the headline number misses.")
|
|
990
|
-
elif repeat_delta > self.REPEAT_TOLERANCE:
|
|
991
|
-
verdict, why = "risky", (
|
|
992
|
-
f"{save_pct:.0f}% cheaper per prompt, but you re-asked "
|
|
993
|
-
f"{m['repeat_pct']:.0f}% of these prompts against "
|
|
994
|
-
f"{cur['repeat_pct']:.0f}% on {cur['name']} — rework you paid for twice.")
|
|
995
|
-
elif rule == "keep":
|
|
996
|
-
verdict, why = "caution", (
|
|
997
|
-
f"{save_pct:.0f}% cheaper per prompt and no more turns, but {cat.replace('_',' ')} "
|
|
998
|
-
f"is reasoning-heavy work where a miss is expensive in ways this data "
|
|
999
|
-
f"cannot show. Worth a trial, not a default.")
|
|
1000
|
-
else:
|
|
1001
|
-
verdict, why = "supported", (
|
|
1002
|
-
f"{save_pct:.0f}% cheaper per prompt on {m['prompts']} of your own "
|
|
1003
|
-
f"{cat.replace('_',' ')} prompts, in {turn_ratio:.1f}x the turns "
|
|
1004
|
-
f"({m['turns']:.0f} vs {cur['turns']:.0f}) with "
|
|
1005
|
-
f"{'less' if repeat_delta <= 0 else 'similar'} re-asking. "
|
|
1006
|
-
f"This is measured, not modelled.")
|
|
1007
|
-
cands.append({
|
|
1008
|
-
"model": m["model"], "name": m["name"], "tier": m["tier"],
|
|
1009
|
-
"prompts": m["prompts"], "cost_per_prompt": m["cost_per_prompt"],
|
|
1010
|
-
"turns": m["turns"], "tools": m["tools"], "repeat_pct": m["repeat_pct"],
|
|
1011
|
-
"savings_pct": round(save_pct, 1), "turn_ratio": round(turn_ratio, 2),
|
|
1012
|
-
"repeat_delta": round(repeat_delta, 1),
|
|
1013
|
-
"estimated_savings_usd": round(save, 2), "verdict": verdict, "why": why})
|
|
1014
|
-
if not cands:
|
|
1015
|
-
continue
|
|
1016
|
-
cands.sort(key=lambda c: (c["verdict"] != "supported", -c["estimated_savings_usd"]))
|
|
1017
|
-
best = cands[0] if cands[0]["verdict"] == "supported" else None
|
|
1018
|
-
if best:
|
|
1019
|
-
total_save += best["estimated_savings_usd"]
|
|
1020
|
-
out.append({
|
|
1021
|
-
"category": cat,
|
|
1022
|
-
"agent": cur["agent"],
|
|
1023
|
-
"current": {"model": cur["model"], "name": cur["name"], "prompts": cur["prompts"],
|
|
1024
|
-
"cost": cur["cost"], "cost_per_prompt": cur["cost_per_prompt"],
|
|
1025
|
-
"turns": cur["turns"], "tools": cur["tools"],
|
|
1026
|
-
"repeat_pct": cur["repeat_pct"]},
|
|
1027
|
-
"candidates": cands,
|
|
1028
|
-
"recommended": best["model"] if best else None,
|
|
1029
|
-
"recommended_name": best["name"] if best else None,
|
|
1030
|
-
"estimated_savings_usd": best["estimated_savings_usd"] if best else 0.0,
|
|
1031
|
-
"verdict": best["verdict"] if best else cands[0]["verdict"],
|
|
1032
|
-
"why": best["why"] if best else cands[0]["why"],
|
|
1033
|
-
"rule": rule,
|
|
1034
|
-
})
|
|
1035
|
-
out.sort(key=lambda c: -c["estimated_savings_usd"])
|
|
1036
|
-
return {
|
|
1037
|
-
"categories": out,
|
|
1038
|
-
"estimated_savings_usd": round(total_save, 2),
|
|
1039
|
-
"min_prompts": self.MIN_PROMPTS,
|
|
1040
|
-
"basis": "actual",
|
|
1041
|
-
"method": (f"Compares what each model actually cost per prompt on the same category "
|
|
1042
|
-
f"of work, using only categories where you ran both with at least "
|
|
1043
|
-
f"{self.MIN_PROMPTS} prompts each. Turns and re-asked prompts are shown "
|
|
1044
|
-
f"because a cheaper model that needs more of both is not cheaper. "
|
|
1045
|
-
f"Nothing here is repriced or modelled."),
|
|
1046
|
-
"caveat": ("Your prompts were not randomly assigned to models, so a category can "
|
|
1047
|
-
"differ in difficulty between them. Treat this as strong evidence for a "
|
|
1048
|
-
"trial, not proof."),
|
|
1049
|
-
}
|
|
1050
|
-
|
|
1051
|
-
def model_switch(self, f=None):
|
|
1052
|
-
"""Per-request what-if: reprice each request on the model its work needs.
|
|
1053
|
-
|
|
1054
|
-
Token counts are held constant (actual); costs on both sides are estimated at the
|
|
1055
|
-
configured prices. Requests whose context exceeds the target's window stay put.
|
|
1056
|
-
"""
|
|
1057
|
-
w, p = self.where(f)
|
|
1058
|
-
rows = self.q(f"""SELECT r.model, r.is_sidechain side, r.agent_type,
|
|
1059
|
-
COALESCE(pr.category,'other') category, r.context_tokens ctx,
|
|
1060
|
-
pj.id project_id, pj.name project,
|
|
1061
|
-
r.prompt_id, r.session_id, r.input_tokens i, r.output_tokens o,
|
|
1062
|
-
r.cache_read_tokens cr, r.cache_write_5m c5, r.cache_write_1h c1, r.est_cost_usd cost
|
|
1063
|
-
FROM requests r LEFT JOIN prompts pr ON pr.id=r.prompt_id
|
|
1064
|
-
JOIN projects pj ON pj.id=r.project_id WHERE {w}""", p)
|
|
1065
|
-
target_of = {}
|
|
1066
|
-
agents, proj_prov = set(), {}
|
|
1067
|
-
groups, projects = {}, defaultdict(lambda: defaultdict(float))
|
|
1068
|
-
total = blocked = 0.0
|
|
1069
|
-
blocked_n = 0
|
|
1370
|
+
rows = self.q(f"""SELECT r.model, r.priced_as, r.context_tokens ctx, r.est_cost_usd cost
|
|
1371
|
+
FROM requests r WHERE {w}""", p)
|
|
1372
|
+
per = {}
|
|
1373
|
+
unknown = {"requests": 0, "cost_usd": 0.0}
|
|
1374
|
+
total_n = total_cost = near_cost = over_cost = 0.0
|
|
1375
|
+
near_n = over_n = 0
|
|
1070
1376
|
for r in rows:
|
|
1071
1377
|
cost = r["cost"] or 0.0
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
want, conf = ("economy", "high") if is_explore else ("balanced", "medium")
|
|
1079
|
-
scope = f"Subagent: {r['agent_type'] or 'general'}"
|
|
1080
|
-
else:
|
|
1081
|
-
want, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
1082
|
-
scope = f"Prompts: {r['category'].replace('_', ' ')}"
|
|
1083
|
-
pj = projects[(r["project_id"], r["project"])]
|
|
1084
|
-
pj["cost"] += cost
|
|
1085
|
-
if tier == "frontier":
|
|
1086
|
-
pj["frontier_cost"] += cost
|
|
1087
|
-
proj_prov[(r["project_id"], r["project"])] = \
|
|
1088
|
-
self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
1089
|
-
if want == "keep" or self.TIER_RANK[want] >= self.TIER_RANK[tier]:
|
|
1090
|
-
if want == "keep" and tier == "frontier":
|
|
1091
|
-
pj["keep_cost"] += cost
|
|
1092
|
-
continue
|
|
1093
|
-
prov = self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
1094
|
-
if (want, prov) not in target_of:
|
|
1095
|
-
target_of[(want, prov)] = self._cheapest(want, prov)
|
|
1096
|
-
tgt = target_of[(want, prov)]
|
|
1097
|
-
agents.add(prov)
|
|
1098
|
-
if not tgt:
|
|
1099
|
-
continue
|
|
1100
|
-
win = self.pricing.context_window(tgt) or 0
|
|
1101
|
-
if win and (r["ctx"] or 0) > win * 0.9:
|
|
1102
|
-
blocked += cost
|
|
1103
|
-
blocked_n += 1
|
|
1104
|
-
continue
|
|
1105
|
-
alt = self.pricing.estimate(tgt, r["i"] or 0, r["o"] or 0, r["cr"] or 0,
|
|
1106
|
-
r["c5"] or 0, r["c1"] or 0)
|
|
1107
|
-
if alt >= cost:
|
|
1108
|
-
continue
|
|
1109
|
-
key = (scope, r["model"], tgt)
|
|
1110
|
-
g = groups.setdefault(key, {"scope": scope, "current_model": r["model"],
|
|
1111
|
-
"recommended_model": tgt, "confidence": conf,
|
|
1112
|
-
"requests": 0, "prompts": set(), "sessions": set(),
|
|
1113
|
-
"cost": 0.0, "alt": 0.0})
|
|
1114
|
-
g["requests"] += 1
|
|
1115
|
-
g["prompts"].add(r["prompt_id"])
|
|
1116
|
-
g["sessions"].add(r["session_id"])
|
|
1117
|
-
g["cost"] += cost
|
|
1118
|
-
g["alt"] += alt
|
|
1119
|
-
pj["savings"] += cost - alt
|
|
1120
|
-
|
|
1121
|
-
out = []
|
|
1122
|
-
for g in groups.values():
|
|
1123
|
-
save = g["cost"] - g["alt"]
|
|
1124
|
-
if save < 0.5:
|
|
1125
|
-
continue
|
|
1126
|
-
out.append({**g, "prompts": len(g["prompts"] - {None}), "sessions": len(g["sessions"]),
|
|
1127
|
-
"current_name": self.pricing.display_name(g["current_model"]),
|
|
1128
|
-
"recommended_name": self.pricing.display_name(g["recommended_model"]),
|
|
1129
|
-
"estimated_savings_usd": save,
|
|
1130
|
-
"estimated_savings_pct": round(100 * save / g["cost"], 1) if g["cost"] else 0})
|
|
1131
|
-
out.sort(key=lambda x: -x["estimated_savings_usd"])
|
|
1132
|
-
|
|
1133
|
-
by_conf = defaultdict(float)
|
|
1134
|
-
for g in out:
|
|
1135
|
-
by_conf[g["confidence"]] += g["estimated_savings_usd"]
|
|
1136
|
-
|
|
1137
|
-
proj = []
|
|
1138
|
-
for (pid, name), v in projects.items():
|
|
1139
|
-
if v["frontier_cost"] < 1:
|
|
1378
|
+
total_n += 1
|
|
1379
|
+
total_cost += cost
|
|
1380
|
+
win = self.pricing.context_window(r["priced_as"] or r["model"])
|
|
1381
|
+
if not win:
|
|
1382
|
+
unknown["requests"] += 1
|
|
1383
|
+
unknown["cost_usd"] += cost
|
|
1140
1384
|
continue
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1385
|
+
m = per.setdefault(r["model"], {
|
|
1386
|
+
"model": r["model"], "display_name": self.pricing.display_name(r["model"]),
|
|
1387
|
+
"context_window": win, "requests": 0, "cost_usd": 0.0,
|
|
1388
|
+
"near_requests": 0, "near_cost_usd": 0.0,
|
|
1389
|
+
"over_requests": 0, "over_cost_usd": 0.0})
|
|
1390
|
+
m["requests"] += 1
|
|
1391
|
+
m["cost_usd"] += cost
|
|
1392
|
+
ctx = r["ctx"] or 0
|
|
1393
|
+
if ctx > win:
|
|
1394
|
+
m["over_requests"] += 1; m["over_cost_usd"] += cost
|
|
1395
|
+
over_n += 1; over_cost += cost
|
|
1396
|
+
elif ctx >= win * self.NEAR_WINDOW_PCT:
|
|
1397
|
+
m["near_requests"] += 1; m["near_cost_usd"] += cost
|
|
1398
|
+
near_n += 1; near_cost += cost
|
|
1399
|
+
out = sorted(per.values(), key=lambda m: -(m["near_cost_usd"] + m["over_cost_usd"]))
|
|
1400
|
+
for m in out:
|
|
1401
|
+
m["near_or_over_cost_pct"] = (round(100.0 * (m["near_cost_usd"] + m["over_cost_usd"])
|
|
1402
|
+
/ m["cost_usd"], 1) if m["cost_usd"] else 0.0)
|
|
1157
1403
|
return {
|
|
1158
|
-
"
|
|
1159
|
-
"
|
|
1160
|
-
"
|
|
1161
|
-
"
|
|
1162
|
-
"
|
|
1163
|
-
"
|
|
1164
|
-
|
|
1165
|
-
"
|
|
1166
|
-
"
|
|
1167
|
-
"project": '"model": "<name>" in <repo>/.claude/settings.json',
|
|
1168
|
-
"subagent": "model: haiku (or sonnet) in the agent's frontmatter in .claude/agents/"},
|
|
1169
|
-
"how_by_agent": self.SWITCH_HOW_BY_AGENT,
|
|
1170
|
-
"providers": sorted(agents),
|
|
1171
|
-
"caveat": "Same token counts repriced on the cheaper model. Output quality and any "
|
|
1172
|
-
"extra turns a cheaper model might need are not modelled. Try it on a "
|
|
1173
|
-
"sample of work before switching everything.",
|
|
1174
|
-
"basis": "recommendation",
|
|
1404
|
+
"threshold_pct": int(self.NEAR_WINDOW_PCT * 100),
|
|
1405
|
+
"models": out,
|
|
1406
|
+
"requests": int(total_n), "cost_usd": total_cost,
|
|
1407
|
+
"near_requests": near_n, "near_cost_usd": near_cost,
|
|
1408
|
+
"over_requests": over_n, "over_cost_usd": over_cost,
|
|
1409
|
+
"near_or_over_cost_pct": (round(100.0 * (near_cost + over_cost) / total_cost, 1)
|
|
1410
|
+
if total_cost else 0.0),
|
|
1411
|
+
"unknown_window": unknown,
|
|
1412
|
+
"basis": "actual",
|
|
1175
1413
|
}
|
|
1176
1414
|
|
|
1177
1415
|
def recommendations(self, f=None):
|
|
1178
1416
|
w, p = self.where(f)
|
|
1179
1417
|
recs = []
|
|
1180
1418
|
|
|
1181
|
-
# Frontier work a cheaper model could have done. Candidates are always from the
|
|
1182
|
-
# same vendor: an agent can only switch within its own family (Codex can't run
|
|
1183
|
-
# Haiku), so mixed frontier spend is split per vendor before anything is compared.
|
|
1184
|
-
# Each recommendation offers the ladder — one step down (balanced) and the floor
|
|
1185
|
-
# (economy) — priced separately, because that trade-off is the user's to make.
|
|
1186
|
-
frontier_by_provider = defaultdict(list)
|
|
1187
|
-
for m, v in self.pricing.models.items():
|
|
1188
|
-
if v.get("tier") == "frontier":
|
|
1189
|
-
frontier_by_provider[v.get("provider", "anthropic")].append(m)
|
|
1190
|
-
|
|
1191
|
-
for prov, models in sorted(frontier_by_provider.items()):
|
|
1192
|
-
cands = []
|
|
1193
|
-
for tier in ("balanced", "economy"):
|
|
1194
|
-
c = self._cheapest(tier, prov)
|
|
1195
|
-
if c and c not in cands:
|
|
1196
|
-
cands.append(c)
|
|
1197
|
-
if not cands:
|
|
1198
|
-
continue # this vendor exposes nothing cheaper to move to
|
|
1199
|
-
ph = ",".join("?" * len(models))
|
|
1200
|
-
rows = self.q(f"""SELECT pr.category, COUNT(DISTINCT pr.id) prompts,
|
|
1201
|
-
GROUP_CONCAT(DISTINCT r.model) mods,
|
|
1202
|
-
SUM(r.est_cost_usd) cost, SUM(r.input_tokens) i,
|
|
1203
|
-
SUM(r.output_tokens) o, SUM(r.cache_read_tokens) cr,
|
|
1204
|
-
SUM(r.cache_write_5m) c5, SUM(r.cache_write_1h) c1
|
|
1205
|
-
FROM prompts pr JOIN requests r ON r.prompt_id=pr.id
|
|
1206
|
-
WHERE {w} AND r.model IN ({ph})
|
|
1207
|
-
GROUP BY pr.category HAVING prompts >= 3 AND cost > 0.5
|
|
1208
|
-
ORDER BY cost DESC""", p + models)
|
|
1209
|
-
for r in rows:
|
|
1210
|
-
# The same rules the Model switch dashboard applies, so the two pages can
|
|
1211
|
-
# never contradict each other: work the rules say to keep on a frontier
|
|
1212
|
-
# model is not offered a downgrade at all.
|
|
1213
|
-
target, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
1214
|
-
if target == "keep":
|
|
1215
|
-
continue
|
|
1216
|
-
alts = []
|
|
1217
|
-
for m in cands:
|
|
1218
|
-
alt = self.pricing.estimate(m, r["i"], r["o"], r["cr"], r["c5"], r["c1"])
|
|
1219
|
-
if alt >= r["cost"] * 0.9:
|
|
1220
|
-
continue # too close to the current cost to be worth the quality risk
|
|
1221
|
-
alts.append({
|
|
1222
|
-
"model": m, "name": self.pricing.display_name(m),
|
|
1223
|
-
"tier": self.pricing.tier(m),
|
|
1224
|
-
"estimated_cost_usd": alt,
|
|
1225
|
-
"estimated_savings_usd": r["cost"] - alt,
|
|
1226
|
-
"estimated_savings_pct": round(100.0 * (r["cost"] - alt) / r["cost"], 1),
|
|
1227
|
-
})
|
|
1228
|
-
if not alts:
|
|
1229
|
-
continue
|
|
1230
|
-
# Safest step first: balanced before economy, so the ladder reads as
|
|
1231
|
-
# increasing saving and increasing risk.
|
|
1232
|
-
alts.sort(key=lambda a: -self.TIER_RANK.get(a["tier"], 0))
|
|
1233
|
-
for a in alts:
|
|
1234
|
-
a["suggested"] = a["tier"] == target
|
|
1235
|
-
# The headline is the tier the rules actually recommend for this kind of
|
|
1236
|
-
# work, not simply the smallest step; the rest stay on offer below it.
|
|
1237
|
-
head = next((a for a in alts if a["suggested"]), alts[0])
|
|
1238
|
-
agent = self.SWITCH_HOW_BY_AGENT.get(prov, {}).get("agent", prov)
|
|
1239
|
-
recs.append({
|
|
1240
|
-
"type": "model_downgrade",
|
|
1241
|
-
"confidence": conf or "low",
|
|
1242
|
-
"title": f"Consider a cheaper {agent} model for '{r['category']}' work",
|
|
1243
|
-
"current_model": ", ".join(self.pricing.display_name(m)
|
|
1244
|
-
for m in (r["mods"] or "").split(",") if m),
|
|
1245
|
-
"recommended_model": head["name"],
|
|
1246
|
-
"provider": prov,
|
|
1247
|
-
"agent": agent,
|
|
1248
|
-
"alternatives": alts,
|
|
1249
|
-
"scope": f"{r['prompts']} prompts categorized as {r['category']}",
|
|
1250
|
-
"actual_cost_usd": r["cost"],
|
|
1251
|
-
"estimated_alternative_cost_usd": head["estimated_cost_usd"],
|
|
1252
|
-
"estimated_savings_usd": head["estimated_savings_usd"],
|
|
1253
|
-
"estimated_savings_pct": head["estimated_savings_pct"],
|
|
1254
|
-
"caveat": f"Both options stay inside {agent}, so this is a setting change, not "
|
|
1255
|
-
"a change of agent. Assumes identical token usage on the cheaper "
|
|
1256
|
-
"model. Output quality is not modelled — validate on a sample "
|
|
1257
|
-
"before switching.",
|
|
1258
|
-
"basis": "recommendation",
|
|
1259
|
-
})
|
|
1260
|
-
# Biggest opportunity first, now that several vendors can each contribute one.
|
|
1261
|
-
recs.sort(key=lambda r: -r["estimated_savings_usd"])
|
|
1262
1419
|
|
|
1420
|
+
# What follows are observations, not priced savings. The cache item used to
|
|
1421
|
+
# carry the no-cache counterfactual as a "saving" (tens of thousands of dollars
|
|
1422
|
+
# on a bill a fraction of that) and the context item multiplied its spend by a
|
|
1423
|
+
# guessed 20%. Neither number was something the method could support, so
|
|
1424
|
+
# neither is shown; what is observable is.
|
|
1263
1425
|
eff = self.efficiency(f)
|
|
1264
1426
|
c = eff["cache"]
|
|
1265
|
-
|
|
1427
|
+
split = c.get("cost_split") or {}
|
|
1428
|
+
if c["reads"] and split.get("read_cost_share"):
|
|
1266
1429
|
recs.append({
|
|
1267
|
-
"type": "cache_working", "confidence": "
|
|
1268
|
-
"title": "Prompt caching is
|
|
1430
|
+
"type": "cache_working", "confidence": "observed",
|
|
1431
|
+
"title": "Prompt caching is doing its job — keep sessions long-lived",
|
|
1432
|
+
"detail": (f"Reads are {split['read_cost_share']*100:.0f}% of your cache cost "
|
|
1433
|
+
f"({eff['cache_hit_ratio']*100:.0f}% of cache tokens). Restarting "
|
|
1434
|
+
f"sessions throws that prefix away and pays to write it again."),
|
|
1269
1435
|
"actual_cost_usd": c["cost_with_cache"],
|
|
1270
|
-
"estimated_alternative_cost_usd":
|
|
1271
|
-
"estimated_savings_usd":
|
|
1272
|
-
"
|
|
1273
|
-
|
|
1274
|
-
"basis": "
|
|
1436
|
+
"estimated_alternative_cost_usd": None,
|
|
1437
|
+
"estimated_savings_usd": None, "estimated_savings_pct": None,
|
|
1438
|
+
"caveat": "No saving is claimed: what an uncached run would have cost is a "
|
|
1439
|
+
"counterfactual, not money you avoided.",
|
|
1440
|
+
"basis": "actual",
|
|
1275
1441
|
})
|
|
1276
1442
|
|
|
1277
1443
|
ctx = self.context_analysis(f)
|
|
1278
1444
|
if ctx["large_context_cost_pct"] > 15:
|
|
1279
1445
|
recs.append({
|
|
1280
|
-
"type": "context_reduction", "confidence": "
|
|
1446
|
+
"type": "context_reduction", "confidence": "observed",
|
|
1281
1447
|
"title": f"{ctx['large_context_cost_pct']}% of spend comes from >"
|
|
1282
1448
|
f"{ctx['threshold']//1000}K-context requests",
|
|
1449
|
+
"detail": (f"{ctx['large_context_requests']:,} requests re-sent a large prefix "
|
|
1450
|
+
f"on every turn. /compact or a fresh session resets it; how much "
|
|
1451
|
+
f"that would have saved depends on what the work needed, and is "
|
|
1452
|
+
f"not estimated here."),
|
|
1283
1453
|
"scope": f"{ctx['large_context_requests']:,} requests",
|
|
1284
1454
|
"actual_cost_usd": ctx["large_context_cost"],
|
|
1285
|
-
"
|
|
1286
|
-
"estimated_savings_pct":
|
|
1287
|
-
"caveat": "
|
|
1288
|
-
|
|
1289
|
-
"basis": "recommendation",
|
|
1455
|
+
"estimated_alternative_cost_usd": None,
|
|
1456
|
+
"estimated_savings_usd": None, "estimated_savings_pct": None,
|
|
1457
|
+
"caveat": "Observed share of spend. No reduction is assumed.",
|
|
1458
|
+
"basis": "actual",
|
|
1290
1459
|
})
|
|
1291
|
-
recs.sort(key=lambda r: -(r.get("
|
|
1292
|
-
return {"recommendations": recs,
|
|
1293
|
-
"total_estimated_savings_usd": sum(r.get("estimated_savings_usd") or 0
|
|
1294
|
-
for r in recs if r["type"] != "cache_working"),
|
|
1295
|
-
"basis": "recommendation"}
|
|
1460
|
+
recs.sort(key=lambda r: -(r.get("actual_cost_usd") or 0))
|
|
1461
|
+
return {"recommendations": recs, "basis": "actual"}
|
|
1296
1462
|
|
|
1297
1463
|
# ---------------- forecast ----------------
|
|
1298
1464
|
def forecast(self, f=None):
|
|
@@ -1302,42 +1468,46 @@ class Analytics:
|
|
|
1302
1468
|
FROM requests r WHERE {w} AND r.day <> '' GROUP BY 1 ORDER BY 1""", p)
|
|
1303
1469
|
if not rows:
|
|
1304
1470
|
return {"available": False, "message": "No usage in the selected range."}
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1471
|
+
# calendar days, idle days as zero, excluding today (still partial) — the rate
|
|
1472
|
+
# below is multiplied by calendar days remaining, so a per-active-day mean
|
|
1473
|
+
# would overstate every scenario and a partial today would understate it
|
|
1474
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
1475
|
+
recent = [r for r in self.daily_series(f, days=14, end=yesterday)] # complete days only
|
|
1476
|
+
priced = [r["cost"] for r in recent]
|
|
1477
|
+
sample_days = sum(1 for c in priced if c > 0)
|
|
1478
|
+
mean = statistics.fmean(priced) if priced else 0.0
|
|
1479
|
+
sd = statistics.pstdev(priced) if len(priced) > 1 else 0.0
|
|
1309
1480
|
in_period = [r for r in rows if bp["start"] <= r["day"] <= bp["end"]]
|
|
1310
1481
|
used = sum(r["cost"] for r in in_period)
|
|
1311
1482
|
used_tok = sum(r["tokens"] for r in in_period)
|
|
1312
1483
|
left = bp["remaining_days"]
|
|
1484
|
+
insufficient = sample_days < 7
|
|
1313
1485
|
|
|
1314
|
-
def band(rate):
|
|
1315
|
-
|
|
1486
|
+
def band(rate, spread=0.0):
|
|
1487
|
+
# spend on different days is treated as independent, so the spread of a
|
|
1488
|
+
# sum over `left` days grows with sqrt(left), not left
|
|
1489
|
+
return {"daily_rate": rate,
|
|
1490
|
+
"end_of_period_cost": used + rate * left + spread * (left ** 0.5)}
|
|
1316
1491
|
|
|
1317
|
-
scenarios = {
|
|
1318
|
-
|
|
1319
|
-
"
|
|
1320
|
-
"high"
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
today_rows = [r for r in rows if r["day"] == bp["today"]]
|
|
1324
|
-
hours = max(datetime.now(timezone.utc).hour, 1)
|
|
1325
|
-
eod = (today_rows[0]["cost"] / hours * 24) if today_rows else mean
|
|
1492
|
+
scenarios = {"expected": band(mean)}
|
|
1493
|
+
if not insufficient:
|
|
1494
|
+
scenarios["conservative"] = band(mean, -sd)
|
|
1495
|
+
scenarios["high"] = band(mean, sd)
|
|
1496
|
+
scenarios["conservative"]["end_of_period_cost"] = max(
|
|
1497
|
+
scenarios["conservative"]["end_of_period_cost"], used)
|
|
1326
1498
|
|
|
1327
|
-
|
|
1328
|
-
wk_used = sum(r["cost"] for r in rows if r["day"] >= wk_start)
|
|
1329
|
-
wk_left = 6 - _d(bp["today"]).weekday()
|
|
1499
|
+
tok_mean = statistics.fmean([r["tokens"] for r in recent]) if recent else 0.0
|
|
1330
1500
|
|
|
1331
1501
|
out = {
|
|
1332
1502
|
"available": True,
|
|
1333
|
-
"method": "14
|
|
1334
|
-
|
|
1503
|
+
"method": ("mean of the last 14 complete calendar days (idle days as zero); "
|
|
1504
|
+
"bands are ±1 sd × sqrt(days remaining)"),
|
|
1505
|
+
"sample_days": sample_days,
|
|
1506
|
+
"insufficient_history": insufficient,
|
|
1335
1507
|
"daily_mean": mean, "daily_stdev": sd,
|
|
1336
1508
|
"period_used": used, "period_used_tokens": used_tok,
|
|
1337
1509
|
"remaining_days": left,
|
|
1338
1510
|
"scenarios": scenarios,
|
|
1339
|
-
"end_of_day_cost": eod,
|
|
1340
|
-
"end_of_week_cost": wk_used + mean * max(wk_left, 0),
|
|
1341
1511
|
"end_of_period_tokens": used_tok + tok_mean * left,
|
|
1342
1512
|
"estimated_monthly_cost": used + mean * left,
|
|
1343
1513
|
"basis": "forecast",
|
|
@@ -1409,34 +1579,55 @@ class Analytics:
|
|
|
1409
1579
|
w, p = self.where(f)
|
|
1410
1580
|
cfg = self.settings["anomaly"]
|
|
1411
1581
|
found = []
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
if len(
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1582
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
1583
|
+
series = [d for d in self.daily_series(dict(f or {}, agents=["claude"]), end=yesterday)]
|
|
1584
|
+
priced = [d for d in series if d["cost"] > 0]
|
|
1585
|
+
if len(priced) >= 14:
|
|
1586
|
+
# Median/MAD, not mean/stdev: unpriced $0 days from agents without pricing
|
|
1587
|
+
# data (e.g. Cursor) would otherwise pollute the mean/sd baseline and
|
|
1588
|
+
# either mask real spikes or manufacture fake ones. MAD is scaled by
|
|
1589
|
+
# 1.4826 so it estimates the same thing a standard deviation would under
|
|
1590
|
+
# a normal distribution, without a few extreme days inflating it the way
|
|
1591
|
+
# a real stdev would.
|
|
1592
|
+
vals = [d["cost"] for d in priced]
|
|
1593
|
+
med = statistics.median(vals)
|
|
1594
|
+
mad = statistics.median(abs(v - med) for v in vals) * 1.4826 or 1e-9
|
|
1595
|
+
for d in priced:
|
|
1596
|
+
score = (d["cost"] - med) / mad
|
|
1597
|
+
ratio = d["cost"] / med if med else 0
|
|
1598
|
+
if score >= cfg.get("daily_robust_z", 3.5) and ratio >= cfg["daily_ratio"]:
|
|
1422
1599
|
found.append({
|
|
1423
1600
|
"severity": "high", "type": "daily_spike", "date": d["day"],
|
|
1424
1601
|
"title": f"{d['day']} spend was {ratio:.1f}x your daily average",
|
|
1425
|
-
"detail": f"${d['cost']:,.2f} vs a ${
|
|
1426
|
-
"metric_value": d["cost"], "baseline":
|
|
1602
|
+
"detail": f"${d['cost']:,.2f} vs a ${med:,.2f} median priced day (robust z={score:.1f}).",
|
|
1603
|
+
"metric_value": d["cost"], "baseline": med, "ratio": round(ratio, 2),
|
|
1427
1604
|
"drilldown": {"filter": {"start": d["day"], "end": d["day"]}},
|
|
1428
1605
|
"basis": "estimated",
|
|
1429
1606
|
})
|
|
1430
1607
|
sess = self.sessions(f, limit=100000, order="cost")
|
|
1431
1608
|
if len(sess) >= 5:
|
|
1432
|
-
|
|
1433
|
-
mean
|
|
1434
|
-
|
|
1609
|
+
# Median, not mean: session token counts are heavily right-skewed, and a
|
|
1610
|
+
# mean lets the outliers inflate the very baseline they are measured
|
|
1611
|
+
# against — which understates how far out they really are. Candidates are
|
|
1612
|
+
# ranked by tokens too, since that is the metric being tested; ordering by
|
|
1613
|
+
# cost hid token-heavy work on cheap models.
|
|
1614
|
+
# Sessions with no token data (Cursor transcripts don't always carry it)
|
|
1615
|
+
# are not comparable and would drag the baseline down.
|
|
1616
|
+
vals = [s["tokens"] for s in sess if (s["tokens"] or 0) > 0]
|
|
1617
|
+
mean = (statistics.median(vals) if vals else 0) or 1
|
|
1618
|
+
# Session sizes are heavy-tailed enough that any fixed multiple of the
|
|
1619
|
+
# baseline still matches a fifth of them, so the threshold alone cannot
|
|
1620
|
+
# keep this list short. Take the most extreme few and leave room for the
|
|
1621
|
+
# other anomaly types, which have much smaller ratios and would otherwise
|
|
1622
|
+
# be sorted off the end of the list.
|
|
1623
|
+
outliers = 0
|
|
1624
|
+
for s in sorted(sess, key=lambda x: -(x["tokens"] or 0))[:40]:
|
|
1435
1625
|
ratio = s["tokens"] / mean
|
|
1436
|
-
if ratio >= cfg["session_ratio"]:
|
|
1626
|
+
if ratio >= cfg["session_ratio"] and outliers < cfg.get("max_session_outliers", 5):
|
|
1627
|
+
outliers += 1
|
|
1437
1628
|
found.append({
|
|
1438
|
-
"severity": "
|
|
1439
|
-
"title": f"
|
|
1629
|
+
"severity": "low", "type": "session_outlier",
|
|
1630
|
+
"title": f"Among your largest sessions: {ratio:.1f}x the median",
|
|
1440
1631
|
"detail": f"{s['title'] or s['session_id'][:8]} — {s['tokens']:,} tokens, "
|
|
1441
1632
|
f"${s['cost']:,.2f} in {s['project']}.",
|
|
1442
1633
|
"metric_value": s["tokens"], "baseline": mean, "ratio": round(ratio, 2),
|
|
@@ -1445,7 +1636,7 @@ class Analytics:
|
|
|
1445
1636
|
})
|
|
1446
1637
|
# week-over-week model shift
|
|
1447
1638
|
if self.last_day:
|
|
1448
|
-
end =
|
|
1639
|
+
end = self.today() - timedelta(days=1) # anchor on yesterday, not last_day
|
|
1449
1640
|
cur_s = (end - timedelta(days=6)).isoformat()
|
|
1450
1641
|
prev_s, prev_e = (end - timedelta(days=13)).isoformat(), (end - timedelta(days=7)).isoformat()
|
|
1451
1642
|
for m in self.q(f"SELECT DISTINCT r.model FROM requests r WHERE {w}", p):
|
|
@@ -1471,46 +1662,27 @@ class Analytics:
|
|
|
1471
1662
|
# ---------------- scorecard ----------------
|
|
1472
1663
|
def scorecard(self, f=None):
|
|
1473
1664
|
eff = self.efficiency(f)
|
|
1474
|
-
ctx = self.context_analysis(f)
|
|
1475
1665
|
wst = self.waste(f)
|
|
1476
1666
|
bud = self.budgets(f)
|
|
1477
|
-
mdl = self.models(f)
|
|
1478
1667
|
dims = []
|
|
1479
1668
|
|
|
1480
1669
|
def dim(name, score, detail, weight=1.0):
|
|
1481
1670
|
dims.append({"name": name, "score": max(0, min(100, round(score))),
|
|
1482
1671
|
"detail": detail, "weight": weight})
|
|
1483
1672
|
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1673
|
+
hy = self.hygiene(f)
|
|
1674
|
+
thr = max(int(t) for t in hy["above"])
|
|
1675
|
+
share = hy["above"][str(thr)]["share_pct"]
|
|
1676
|
+
dim("Context share", 100 - share,
|
|
1677
|
+
f"{share:.0f}% of spend ran above {thr//1000}K context.", 1.0)
|
|
1678
|
+
|
|
1679
|
+
margin = eff["cache"].get("breakeven_margin")
|
|
1680
|
+
if margin is None:
|
|
1681
|
+
dim("Cache break-even", 50, "No cache activity in range", 1.0)
|
|
1487
1682
|
else:
|
|
1488
|
-
dim("Cache
|
|
1489
|
-
f"{
|
|
1490
|
-
|
|
1491
|
-
sc_cfg = self.settings.get("scorecard", {})
|
|
1492
|
-
target = sc_cfg.get("target_output_ratio", 0.0088)
|
|
1493
|
-
outr = eff["output_ratio"]
|
|
1494
|
-
dim("Token efficiency", min(outr / target, 1.0) * 100,
|
|
1495
|
-
f"Output is {outr*100:.2f}% of billable tokens against a "
|
|
1496
|
-
f"{target*100:.2f}% reference.", 1.2)
|
|
1497
|
-
|
|
1498
|
-
big_pct = ctx["large_context_cost_pct"]
|
|
1499
|
-
dim("Context efficiency", 100 - big_pct,
|
|
1500
|
-
f"{big_pct}% of spend came from requests above "
|
|
1501
|
-
f"{ctx['threshold']//1000}K context.", 1.0)
|
|
1502
|
-
|
|
1503
|
-
excess = wst["excess_pct"]
|
|
1504
|
-
dim("Waste control", 100 - min(excess, 100),
|
|
1505
|
-
f"{excess}% of spend is estimated excess over a reasonable baseline "
|
|
1506
|
-
f"({wst['exposed_pct']}% of spend sits in items a rule touched).", 1.3)
|
|
1507
|
-
|
|
1508
|
-
priced = [r for r in mdl["rows"] if r["tier"] != "none" and r["cost"]]
|
|
1509
|
-
frontier_pct = (100.0 * sum(r["cost"] for r in priced if r["tier"] == "frontier")
|
|
1510
|
-
/ (sum(r["cost"] for r in priced) or 1))
|
|
1511
|
-
allow = sc_cfg.get("frontier_cost_share_allowance_pct", 40)
|
|
1512
|
-
dim("Model selection", 100 - max(frontier_pct - allow, 0) * 1.5,
|
|
1513
|
-
f"{frontier_pct:.0f}% of spend is on frontier-tier models.", 1.1)
|
|
1683
|
+
dim("Cache break-even", 50 + margin * 50,
|
|
1684
|
+
f"Caching returned {margin*100:.0f}% of its cost as read discount net of "
|
|
1685
|
+
"write premium.", 1.0)
|
|
1514
1686
|
|
|
1515
1687
|
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1516
1688
|
if ml and ml.get("configured"):
|
|
@@ -1519,12 +1691,7 @@ class Analytics:
|
|
|
1519
1691
|
f"Forecast is {fp:.0f}% of the configured monthly budget.", 1.3)
|
|
1520
1692
|
else:
|
|
1521
1693
|
dim("Budget adherence", 50,
|
|
1522
|
-
"No monthly budget configured — set one in config/settings.json to be
|
|
1523
|
-
|
|
1524
|
-
cpo = eff["cost_per_1k_output"]
|
|
1525
|
-
cpo_target = sc_cfg.get("target_cost_per_1k_output_usd", 0.30)
|
|
1526
|
-
dim("Cost efficiency", 100 - min(cpo / (cpo_target * 2) * 100, 100),
|
|
1527
|
-
f"${cpo:.3f} estimated per 1K output tokens.", 1.0)
|
|
1694
|
+
"No monthly budget configured — set one in config/settings.json to be measured.", 0.4)
|
|
1528
1695
|
|
|
1529
1696
|
tw = sum(d["weight"] for d in dims)
|
|
1530
1697
|
total = round(sum(d["score"] * d["weight"] for d in dims) / tw)
|
|
@@ -1532,8 +1699,7 @@ class Analytics:
|
|
|
1532
1699
|
weak = sorted(dims, key=lambda d: d["score"])[:3]
|
|
1533
1700
|
top = wst["findings"][0] if wst["findings"] else None
|
|
1534
1701
|
return {
|
|
1535
|
-
"score": total,
|
|
1536
|
-
else "C" if total >= 55 else "D" if total >= 40 else "F"),
|
|
1702
|
+
"score": total,
|
|
1537
1703
|
"dimensions": dims,
|
|
1538
1704
|
"what_is_good": [f"{d['name']}: {d['detail']}" for d in strong if d["score"] >= 60],
|
|
1539
1705
|
"needs_attention": [f"{d['name']}: {d['detail']}" for d in weak if d["score"] < 70],
|
|
@@ -1570,16 +1736,9 @@ class Analytics:
|
|
|
1570
1736
|
for r in recs["recommendations"][:2]:
|
|
1571
1737
|
if r["type"] == "cache_working":
|
|
1572
1738
|
continue
|
|
1573
|
-
# Name the models. A saving is meaningless without the swap it assumes, and
|
|
1574
|
-
# the options are what the reader actually has to choose between.
|
|
1575
|
-
opts = " or ".join(f"{a['name']} (~${a['estimated_savings_usd']:,.0f}, "
|
|
1576
|
-
f"{a['estimated_savings_pct']}%)" for a in r.get("alternatives", []))
|
|
1577
|
-
swap = f"{r['current_model']} → {opts}. " if opts else ""
|
|
1578
1739
|
actions.append({"priority": 3, "kind": "recommendation", "text": r["title"],
|
|
1579
|
-
"detail":
|
|
1580
|
-
|
|
1581
|
-
f"{r['caveat']}",
|
|
1582
|
-
"basis": "recommendation"})
|
|
1740
|
+
"detail": r.get("detail") or r.get("caveat") or "",
|
|
1741
|
+
"basis": r.get("basis", "recommendation")})
|
|
1583
1742
|
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1584
1743
|
if ml and ml.get("configured") and ml.get("forecast_pct"):
|
|
1585
1744
|
if ml["forecast_pct"] >= 90:
|
|
@@ -1599,64 +1758,15 @@ class Analytics:
|
|
|
1599
1758
|
"detail": "; ".join(f"\"{x['preview'][:60]}…\" (${x['pcost']:,.2f})" for x in pr),
|
|
1600
1759
|
"basis": "estimated"})
|
|
1601
1760
|
actions.sort(key=lambda a: a["priority"])
|
|
1602
|
-
|
|
1761
|
+
# No "savings opportunity" range: the old one was a guessed 20% of large-context
|
|
1762
|
+
# spend, then 0.6x of that for a low end. Neither factor came from the data.
|
|
1603
1763
|
return {
|
|
1604
1764
|
"question": "What should I do today?",
|
|
1605
1765
|
"actions": actions[:6],
|
|
1606
|
-
"estimated_savings_range_usd": [round(savings * 0.6, 2), round(savings, 2)],
|
|
1607
1766
|
"generated_from": "Live dashboard data for the current filter selection.",
|
|
1608
1767
|
"basis": "mixed: see per-item basis",
|
|
1609
1768
|
}
|
|
1610
1769
|
|
|
1611
|
-
# ---------------- prompt-level advisor ----------------
|
|
1612
|
-
def prompt_advisor(self, p):
|
|
1613
|
-
"""Deterministic, evidence-based analysis of one prompt. Estimates only."""
|
|
1614
|
-
reasons, suggestions = [], []
|
|
1615
|
-
chars = p.get("char_len") or 0
|
|
1616
|
-
ctx = p.get("max_context_tokens") or 0
|
|
1617
|
-
tools = p.get("tool_calls") or 0
|
|
1618
|
-
out = p.get("output_tokens") or 0
|
|
1619
|
-
tot = p.get("billable_tokens") or 0
|
|
1620
|
-
reduction = 0.0
|
|
1621
|
-
if chars > 4000:
|
|
1622
|
-
reasons.append(f"The prompt itself is {chars:,} characters, which is cached and "
|
|
1623
|
-
f"re-sent on every follow-up turn.")
|
|
1624
|
-
suggestions.append("Move long pasted content into a file and reference the path.")
|
|
1625
|
-
reduction += 0.10
|
|
1626
|
-
if ctx > 150000:
|
|
1627
|
-
reasons.append(f"It ran with up to {ctx:,} context tokens per request.")
|
|
1628
|
-
suggestions.append("Run /compact or start a fresh session before a task this large.")
|
|
1629
|
-
reduction += 0.25
|
|
1630
|
-
if tools > 40:
|
|
1631
|
-
reasons.append(f"It triggered {tools} tool calls; each one re-sends the conversation.")
|
|
1632
|
-
suggestions.append("Split into smaller, explicitly scoped sub-tasks.")
|
|
1633
|
-
reduction += 0.15
|
|
1634
|
-
if tot and out / tot < 0.01:
|
|
1635
|
-
reasons.append(f"Only {100*out/tot:.2f}% of the tokens were output — most of the "
|
|
1636
|
-
f"cost was re-reading context.")
|
|
1637
|
-
suggestions.append("Narrow the files and history in scope before asking.")
|
|
1638
|
-
reduction += 0.10
|
|
1639
|
-
models = (p.get("models") or "")
|
|
1640
|
-
if "opus" in models and out < 400:
|
|
1641
|
-
reasons.append("A frontier-tier model produced a short answer.")
|
|
1642
|
-
suggestions.append("Route short turns to a cheaper model tier.")
|
|
1643
|
-
reduction += 0.20
|
|
1644
|
-
if not reasons:
|
|
1645
|
-
return {"available": False,
|
|
1646
|
-
"message": "No cost-driver pattern detected for this prompt."}
|
|
1647
|
-
reduction = min(reduction, 0.6)
|
|
1648
|
-
return {
|
|
1649
|
-
"available": True,
|
|
1650
|
-
"why_expensive": reasons,
|
|
1651
|
-
"suggestions": suggestions,
|
|
1652
|
-
"estimated_token_reduction_pct": round(reduction * 100),
|
|
1653
|
-
"estimated_cost_reduction_pct": round(reduction * 100 * 0.85),
|
|
1654
|
-
"estimated_cost_reduction_usd": round((p.get("est_cost_usd") or 0) * reduction * 0.85, 2),
|
|
1655
|
-
"disclaimer": "ESTIMATE from structural heuristics. Not a measured saving and not a "
|
|
1656
|
-
"guarantee of equivalent output quality.",
|
|
1657
|
-
"basis": "recommendation",
|
|
1658
|
-
}
|
|
1659
|
-
|
|
1660
1770
|
# ---------------- claude code / developer ----------------
|
|
1661
1771
|
def developer(self, f=None):
|
|
1662
1772
|
w, p = self.where(f)
|