claude-finops 0.7.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -47
- package/config/pricing.json +62 -42
- package/config/settings.json +11 -5
- package/finops/actions.py +1 -1
- package/finops/analytics.py +690 -611
- package/finops/api.py +174 -55
- package/finops/diagnose.py +17 -20
- package/finops/etl.py +149 -52
- package/finops/integrate.py +22 -57
- package/finops/pricing.py +57 -6
- package/finops/procs.py +7 -5
- package/finops/report.py +29 -23
- package/finops/segments.py +149 -0
- package/package.json +1 -1
- package/run.cmd +3 -2
- package/run.py +26 -59
- package/web/app.js +253 -329
- package/web/charts.js +26 -15
- package/finops/advisor.py +0 -188
- package/finops/trial.py +0 -169
package/finops/analytics.py
CHANGED
|
@@ -7,16 +7,18 @@ Every number returned is tagged with a `basis`:
|
|
|
7
7
|
recommendation - suggested action, never a booked saving
|
|
8
8
|
"""
|
|
9
9
|
import json
|
|
10
|
-
import math
|
|
11
10
|
import os
|
|
12
11
|
import sqlite3
|
|
13
12
|
import statistics
|
|
14
|
-
|
|
13
|
+
import sys
|
|
14
|
+
import threading
|
|
15
|
+
from collections import defaultdict
|
|
15
16
|
from datetime import date, datetime, timedelta, timezone
|
|
16
17
|
|
|
17
18
|
from .pricing import Pricing
|
|
19
|
+
from .segments import is_compaction
|
|
18
20
|
|
|
19
|
-
from .paths import
|
|
21
|
+
from .paths import DB_PATH, SETTINGS_PATH, LOCAL_SETTINGS_PATH
|
|
20
22
|
|
|
21
23
|
UNAVAILABLE = "Unavailable from connected Claude data"
|
|
22
24
|
|
|
@@ -84,36 +86,131 @@ def detect_account():
|
|
|
84
86
|
return {k: v for k, v in out.items() if v}
|
|
85
87
|
|
|
86
88
|
|
|
89
|
+
def _is_num(v):
|
|
90
|
+
return isinstance(v, (int, float)) and not isinstance(v, bool)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _validate_settings(cur, defaults):
|
|
94
|
+
"""Defensive coercion of settings.local.json's numeric leaves.
|
|
95
|
+
|
|
96
|
+
Every value under budgets/limits must be a number, null, or (for
|
|
97
|
+
per_project_usd/per_model_usd) a dict of numbers; alert_thresholds_pct must be
|
|
98
|
+
a list of numbers 0..1000. Anything else is dropped and the shipped default
|
|
99
|
+
(from settings.json) is used instead, with a warning.
|
|
100
|
+
"""
|
|
101
|
+
bad = []
|
|
102
|
+
for section in ("budgets", "limits"):
|
|
103
|
+
want = defaults.get(section, {})
|
|
104
|
+
have = cur.get(section)
|
|
105
|
+
if not isinstance(have, dict):
|
|
106
|
+
bad.append(section)
|
|
107
|
+
cur[section] = want
|
|
108
|
+
continue
|
|
109
|
+
fixed = dict(have)
|
|
110
|
+
for k, v in list(have.items()):
|
|
111
|
+
if k.startswith("_"):
|
|
112
|
+
continue
|
|
113
|
+
if k in ("per_project_usd", "per_model_usd"):
|
|
114
|
+
if not isinstance(v, dict) or not all(_is_num(x) for x in v.values()):
|
|
115
|
+
bad.append(f"{section}.{k}")
|
|
116
|
+
fixed[k] = want.get(k, {})
|
|
117
|
+
elif not (v is None or _is_num(v)):
|
|
118
|
+
bad.append(f"{section}.{k}")
|
|
119
|
+
fixed[k] = want.get(k)
|
|
120
|
+
cur[section] = fixed
|
|
121
|
+
pct = cur.get("alert_thresholds_pct")
|
|
122
|
+
if not (isinstance(pct, list) and all(_is_num(x) and 0 <= x <= 1000 for x in pct)):
|
|
123
|
+
if pct is not None:
|
|
124
|
+
bad.append("alert_thresholds_pct")
|
|
125
|
+
cur["alert_thresholds_pct"] = defaults.get("alert_thresholds_pct", [])
|
|
126
|
+
for path in bad:
|
|
127
|
+
print(f"finops: settings.local.json has an invalid '{path}'; using the shipped default",
|
|
128
|
+
file=sys.stderr)
|
|
129
|
+
return cur
|
|
130
|
+
|
|
131
|
+
|
|
87
132
|
def load_settings():
|
|
88
133
|
"""Shared defaults (settings.json) + this machine's overrides (settings.local.json)."""
|
|
89
134
|
with open(SETTINGS_PATH) as fh:
|
|
90
|
-
|
|
135
|
+
base = json.load(fh)
|
|
91
136
|
# Detected identity first, so a configured settings.json still wins below.
|
|
92
137
|
detected = detect_account()
|
|
93
|
-
acct =
|
|
138
|
+
acct = base.setdefault("account", {})
|
|
94
139
|
for k, v in detected.items():
|
|
95
140
|
if not acct.get(k):
|
|
96
141
|
acct[k] = v
|
|
142
|
+
cur = json.loads(json.dumps(base)) # deep copy: base stays the fallback default
|
|
97
143
|
if os.path.exists(LOCAL_SETTINGS_PATH):
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
144
|
+
try:
|
|
145
|
+
with open(LOCAL_SETTINGS_PATH) as fh:
|
|
146
|
+
local = json.load(fh)
|
|
147
|
+
except (OSError, ValueError) as exc:
|
|
148
|
+
print(f"finops: settings.local.json unreadable ({exc}); using shipped defaults",
|
|
149
|
+
file=sys.stderr)
|
|
150
|
+
local = {}
|
|
151
|
+
if isinstance(local, dict):
|
|
152
|
+
_merge(cur, local)
|
|
153
|
+
else:
|
|
154
|
+
print("finops: settings.local.json is not an object; using shipped defaults",
|
|
155
|
+
file=sys.stderr)
|
|
156
|
+
return _validate_settings(cur, base)
|
|
101
157
|
|
|
102
158
|
|
|
103
159
|
def _d(s):
|
|
104
160
|
return datetime.strptime(s, "%Y-%m-%d").date()
|
|
105
161
|
|
|
106
162
|
|
|
163
|
+
def _cumsum(values):
|
|
164
|
+
total = 0.0
|
|
165
|
+
for v in values:
|
|
166
|
+
total += v or 0.0
|
|
167
|
+
yield total
|
|
168
|
+
|
|
169
|
+
|
|
107
170
|
class Analytics:
|
|
108
171
|
def __init__(self, db_path=DB_PATH):
|
|
109
|
-
|
|
110
|
-
|
|
172
|
+
# One connection per thread. The HTTP server is threaded, and a single sqlite
|
|
173
|
+
# connection shared across threads fails under concurrent use with "bad
|
|
174
|
+
# parameter or other API misuse" — which is exactly what a page firing several
|
|
175
|
+
# requests at once produces. Every connection is tracked so close() can release
|
|
176
|
+
# them all before the warehouse file is swapped on sync.
|
|
177
|
+
self._db_path = db_path
|
|
178
|
+
self._local = threading.local()
|
|
179
|
+
self._conns = []
|
|
180
|
+
self._conns_lock = threading.Lock()
|
|
111
181
|
self.pricing = Pricing()
|
|
112
182
|
self.settings = load_settings()
|
|
113
183
|
self.meta = {r["key"]: r["value"] for r in self.db.execute("SELECT * FROM meta")}
|
|
114
184
|
row = self.db.execute("SELECT MIN(day) a, MAX(day) b FROM requests WHERE day<>''").fetchone()
|
|
115
185
|
self.first_day, self.last_day = row["a"], row["b"]
|
|
116
186
|
|
|
187
|
+
_today = None # tests set this; production uses the clock
|
|
188
|
+
|
|
189
|
+
def today(self):
|
|
190
|
+
return self._today or datetime.now(timezone.utc).date()
|
|
191
|
+
|
|
192
|
+
@property
|
|
193
|
+
def db(self):
|
|
194
|
+
c = getattr(self._local, "conn", None)
|
|
195
|
+
if c is None:
|
|
196
|
+
c = sqlite3.connect(self._db_path)
|
|
197
|
+
c.row_factory = sqlite3.Row
|
|
198
|
+
self._local.conn = c
|
|
199
|
+
with self._conns_lock:
|
|
200
|
+
self._conns.append(c)
|
|
201
|
+
return c
|
|
202
|
+
|
|
203
|
+
def close(self):
|
|
204
|
+
"""Close every thread's connection, so the warehouse file can be replaced."""
|
|
205
|
+
with self._conns_lock:
|
|
206
|
+
conns, self._conns = self._conns, []
|
|
207
|
+
for c in conns:
|
|
208
|
+
try:
|
|
209
|
+
c.close()
|
|
210
|
+
except Exception:
|
|
211
|
+
pass
|
|
212
|
+
self._local = threading.local()
|
|
213
|
+
|
|
117
214
|
def q(self, sql, params=()):
|
|
118
215
|
return [dict(r) for r in self.db.execute(sql, params)]
|
|
119
216
|
|
|
@@ -136,7 +233,10 @@ class Analytics:
|
|
|
136
233
|
cl.append("r.model IN (%s)" % ",".join("?" * len(f["models"]))); p += f["models"]
|
|
137
234
|
if f.get("projects"):
|
|
138
235
|
cl.append("r.project_id IN (%s)" % ",".join("?" * len(f["projects"])))
|
|
139
|
-
|
|
236
|
+
try:
|
|
237
|
+
p += [int(x) for x in f["projects"]]
|
|
238
|
+
except (TypeError, ValueError):
|
|
239
|
+
raise ValueError("projects must be integers")
|
|
140
240
|
if f.get("sessions"):
|
|
141
241
|
cl.append("r.session_id IN (%s)" % ",".join("?" * len(f["sessions"]))); p += f["sessions"]
|
|
142
242
|
if f.get("categories"):
|
|
@@ -152,10 +252,36 @@ class Analytics:
|
|
|
152
252
|
cl.append("r.billable_tokens >= ?"); p.append(int(f["min_tokens"]))
|
|
153
253
|
return (" AND ".join(cl) if cl else "1=1"), p
|
|
154
254
|
|
|
255
|
+
def daily_series(self, f, days=None, end=None):
|
|
256
|
+
"""Per-day cost/token series with idle days present as zeros.
|
|
257
|
+
|
|
258
|
+
A plain GROUP BY day only returns days you actually worked, so a mean taken
|
|
259
|
+
over it is a per-ACTIVE-day rate. Every projection here multiplies that rate
|
|
260
|
+
by calendar days remaining, so the zero days have to be filled in or the
|
|
261
|
+
forecast is inflated by exactly the share of days you were idle.
|
|
262
|
+
"""
|
|
263
|
+
w, p = self.where(f)
|
|
264
|
+
rows = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost, SUM(r.billable_tokens) tokens,
|
|
265
|
+
COUNT(*) requests FROM requests r WHERE {w} AND r.day <> ''
|
|
266
|
+
GROUP BY 1 ORDER BY 1""", p)
|
|
267
|
+
if not rows:
|
|
268
|
+
return []
|
|
269
|
+
by_day = {r["day"]: r for r in rows}
|
|
270
|
+
last = _d(end) if end else max(_d(rows[-1]["day"]), self.today())
|
|
271
|
+
first = _d(rows[0]["day"])
|
|
272
|
+
if days:
|
|
273
|
+
first = max(first, last - timedelta(days=days - 1))
|
|
274
|
+
out, cur = [], first
|
|
275
|
+
while cur <= last:
|
|
276
|
+
k = cur.isoformat()
|
|
277
|
+
out.append(by_day.get(k) or {"day": k, "cost": 0.0, "tokens": 0, "requests": 0})
|
|
278
|
+
cur += timedelta(days=1)
|
|
279
|
+
return out
|
|
280
|
+
|
|
155
281
|
# ---------------- billing period ----------------
|
|
156
282
|
def billing_period(self, today=None):
|
|
157
283
|
bp = self.settings["billing_period"]
|
|
158
|
-
today = today or
|
|
284
|
+
today = today or self.today()
|
|
159
285
|
anchor = int(bp.get("anchor_day", 1))
|
|
160
286
|
if today.day >= anchor:
|
|
161
287
|
start = today.replace(day=min(anchor, 28))
|
|
@@ -233,14 +359,13 @@ class Analytics:
|
|
|
233
359
|
elapsed = max(bp["elapsed_days"], 1)
|
|
234
360
|
daily_avg = used_cost / elapsed
|
|
235
361
|
|
|
236
|
-
# Trailing rates are measured over the last N days
|
|
237
|
-
#
|
|
238
|
-
#
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
last7, last14 = recent[-7:], recent[-14:]
|
|
362
|
+
# Trailing rates are measured over the last N CALENDAR days, not only the
|
|
363
|
+
# slice inside the billing period — early in a period that slice is too short
|
|
364
|
+
# to be a rate. Idle days count as zero, because these rates get multiplied by
|
|
365
|
+
# calendar days remaining. This keeps burn and forecast on one methodology.
|
|
366
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
367
|
+
last7 = self.daily_series(f, days=7, end=yesterday)
|
|
368
|
+
last14 = self.daily_series(f, days=14, end=yesterday)
|
|
244
369
|
avg7 = (sum(r["cost"] for r in last7) / len(last7)) if last7 else 0.0
|
|
245
370
|
tok_avg7 = (sum(r["tokens"] for r in last7) / len(last7)) if last7 else 0.0
|
|
246
371
|
# the projection rate matches Analytics.forecast()'s "expected" scenario
|
|
@@ -261,7 +386,8 @@ class Analytics:
|
|
|
261
386
|
"projected_period_cost": projected,
|
|
262
387
|
"projected_period_tokens": used_tokens + tok_burn * bp["remaining_days"],
|
|
263
388
|
"forecast_basis": "forecast",
|
|
264
|
-
"forecast_note": ("Projection uses the %d-day mean daily spend,
|
|
389
|
+
"forecast_note": ("Projection uses the %d-calendar-day mean daily spend, idle days "
|
|
390
|
+
"included as zero — the same rate as the "
|
|
265
391
|
"Forecast view's expected scenario." % len(last14)),
|
|
266
392
|
"series": rows,
|
|
267
393
|
"allowances": {},
|
|
@@ -277,7 +403,7 @@ class Analytics:
|
|
|
277
403
|
continue
|
|
278
404
|
remaining = allowance - used_val
|
|
279
405
|
pct = 100.0 * used_val / allowance
|
|
280
|
-
days_left = (remaining / rate) if rate > 0 else None
|
|
406
|
+
days_left = (remaining / rate) if rate > 0 and remaining > 0 else None
|
|
281
407
|
proj = used_val + rate * bp["remaining_days"]
|
|
282
408
|
out["allowances"][label] = {
|
|
283
409
|
"configured": True, "allowance": allowance, "used": used_val,
|
|
@@ -286,6 +412,7 @@ class Analytics:
|
|
|
286
412
|
"days_until_limit": (round(days_left, 1) if days_left is not None else None),
|
|
287
413
|
"limit_date": ((_d(bp["today"]) + timedelta(days=days_left)).isoformat()
|
|
288
414
|
if days_left is not None and days_left < 3650 else None),
|
|
415
|
+
"exceeded": remaining <= 0,
|
|
289
416
|
"projected_end_of_period": proj,
|
|
290
417
|
"projected_overage_pct": round(100.0 * (proj - allowance) / allowance, 1),
|
|
291
418
|
"status": self._status(pct),
|
|
@@ -332,6 +459,24 @@ class Analytics:
|
|
|
332
459
|
FROM requests r WHERE {w} GROUP BY r.model ORDER BY cost DESC""", p)
|
|
333
460
|
tc = sum(r["cost"] for r in rows) or 1
|
|
334
461
|
tt = sum(r["tokens"] for r in rows) or 1
|
|
462
|
+
# Utilisation is measured against the window that actually served each request —
|
|
463
|
+
# a long-context variant has a bigger one — and requests that ran over a window
|
|
464
|
+
# this price table cannot explain are counted, not averaged into a figure above
|
|
465
|
+
# 100%, which is what dividing by the base model's window used to produce.
|
|
466
|
+
util = {}
|
|
467
|
+
for v in self.q(f"""SELECT r.model, r.priced_as, r.unpriced_long_context u,
|
|
468
|
+
COUNT(*) n, AVG(r.context_tokens) avg_ctx
|
|
469
|
+
FROM requests r WHERE {w}
|
|
470
|
+
GROUP BY r.model, r.priced_as, r.unpriced_long_context""", p):
|
|
471
|
+
u = util.setdefault(v["model"], {"num": 0.0, "den": 0, "over": 0, "long": 0})
|
|
472
|
+
win = self.pricing.context_window(v["priced_as"] or v["model"])
|
|
473
|
+
if v["u"] or not win:
|
|
474
|
+
u["over"] += v["n"]
|
|
475
|
+
continue
|
|
476
|
+
u["num"] += v["n"] * 100.0 * (v["avg_ctx"] or 0) / win
|
|
477
|
+
u["den"] += v["n"]
|
|
478
|
+
if v["priced_as"] and v["priced_as"].endswith("[1m]"):
|
|
479
|
+
u["long"] += v["n"]
|
|
335
480
|
for r in rows:
|
|
336
481
|
r["display_name"] = self.pricing.display_name(r["model"])
|
|
337
482
|
r["tier"] = self.pricing.tier(r["model"])
|
|
@@ -343,8 +488,10 @@ class Analytics:
|
|
|
343
488
|
r["output_per_input"] = (r["output_tokens"] / (r["input_tokens"] + r["cache_read_tokens"]
|
|
344
489
|
+ r["cache_write_tokens"])) if r["tokens"] else 0
|
|
345
490
|
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
346
|
-
|
|
347
|
-
|
|
491
|
+
u = util.get(r["model"], {})
|
|
492
|
+
r["utilization_pct"] = round(u["num"] / u["den"], 1) if u.get("den") else None
|
|
493
|
+
r["over_window_requests"] = u.get("over", 0)
|
|
494
|
+
r["long_context_requests"] = u.get("long", 0)
|
|
348
495
|
priced = [r for r in rows if r["tokens"] and r["tier"] != "none"]
|
|
349
496
|
superlatives = {}
|
|
350
497
|
if priced:
|
|
@@ -381,7 +528,7 @@ class Analytics:
|
|
|
381
528
|
r["budget_used_pct"] = round(100.0 * r["cost"] / b, 1) if b else None
|
|
382
529
|
return rows
|
|
383
530
|
|
|
384
|
-
def sessions(self, f=None, limit=500, order="cost"):
|
|
531
|
+
def sessions(self, f=None, limit=500, order="cost", offset=0):
|
|
385
532
|
w, p = self.where(f)
|
|
386
533
|
ob = {"cost": "cost DESC", "tokens": "tokens DESC", "duration": "s.duration_s DESC",
|
|
387
534
|
"recent": "s.started_at DESC", "prompts": "prompts DESC"}.get(order, "cost DESC")
|
|
@@ -398,17 +545,32 @@ class Analytics:
|
|
|
398
545
|
GROUP_CONCAT(DISTINCT r.model) models
|
|
399
546
|
FROM requests r JOIN sessions s ON s.id = r.session_id
|
|
400
547
|
JOIN projects pr ON pr.id = s.project_id
|
|
401
|
-
WHERE {w} GROUP BY s.id ORDER BY {ob} LIMIT ?""", p + [limit])
|
|
548
|
+
WHERE {w} GROUP BY s.id ORDER BY {ob} LIMIT ? OFFSET ?""", p + [limit, offset])
|
|
402
549
|
for r in rows:
|
|
403
550
|
r["cost_per_prompt"] = r["cost"] / r["prompts"] if r["prompts"] else None
|
|
404
551
|
r["tokens_per_prompt"] = r["tokens"] / r["prompts"] if r["prompts"] else None
|
|
405
552
|
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
406
|
-
r["output_ratio"] = r["output_tokens"] / r["tokens"] if r["tokens"] else 0
|
|
407
|
-
r["
|
|
408
|
-
|
|
409
|
-
if (r["cache_read_tokens"] + r["cache_write_tokens"]) else None)
|
|
553
|
+
r["output_ratio"] = (r["output_tokens"] or 0) / r["tokens"] if r["tokens"] else 0
|
|
554
|
+
cr, cw = r["cache_read_tokens"] or 0, r["cache_write_tokens"] or 0
|
|
555
|
+
r["cache_hit_ratio"] = (cr / (cr + cw)) if (cr + cw) else None
|
|
410
556
|
return rows
|
|
411
557
|
|
|
558
|
+
def sessions_total(self, f=None):
|
|
559
|
+
w, p = self.where(f)
|
|
560
|
+
return self.one(f"SELECT COUNT(DISTINCT s.id) n FROM requests r "
|
|
561
|
+
f"JOIN sessions s ON s.id = r.session_id "
|
|
562
|
+
f"JOIN projects pr ON pr.id = s.project_id WHERE {w}", p)["n"]
|
|
563
|
+
|
|
564
|
+
def prompts_total(self, f=None, search=None):
|
|
565
|
+
w, p = self.where(f)
|
|
566
|
+
extra, ep = "", []
|
|
567
|
+
if search:
|
|
568
|
+
extra = (" AND r.prompt_id IN (SELECT id FROM prompts pr WHERE pr.text LIKE ? "
|
|
569
|
+
"OR pr.session_id LIKE ? OR pr.category LIKE ?)")
|
|
570
|
+
ep = [f"%{search}%"] * 3
|
|
571
|
+
return self.one(f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r "
|
|
572
|
+
f"WHERE {w}{extra} AND r.prompt_id IS NOT NULL", p + ep)["n"]
|
|
573
|
+
|
|
412
574
|
def prompts(self, f=None, limit=300, offset=0, order="cost", search=None):
|
|
413
575
|
w, p = self.where(f)
|
|
414
576
|
ob = {"cost": "pcost DESC", "tokens": "ptokens DESC", "recent": "pr.ts DESC",
|
|
@@ -490,7 +652,6 @@ class Analytics:
|
|
|
490
652
|
p["category_evidence"] = json.loads(p.get("category_evidence") or "[]")
|
|
491
653
|
except Exception:
|
|
492
654
|
p["category_evidence"] = []
|
|
493
|
-
p["advisor"] = self.prompt_advisor(p)
|
|
494
655
|
return p
|
|
495
656
|
|
|
496
657
|
def session_detail(self, sid):
|
|
@@ -543,6 +704,252 @@ class Analytics:
|
|
|
543
704
|
}
|
|
544
705
|
|
|
545
706
|
# ---------------- efficiency ----------------
|
|
707
|
+
def hygiene(self, f=None, top=12, trajectory_points=80):
|
|
708
|
+
"""Context and session hygiene: what it cost to keep re-sending a large prefix.
|
|
709
|
+
|
|
710
|
+
Everything here is observed. For each session: the context size of every
|
|
711
|
+
request in order, the request at which it first crossed each configured
|
|
712
|
+
threshold, and what was spent from that point on. Across the range: the share
|
|
713
|
+
of spend in requests above each threshold. No compaction is simulated and no
|
|
714
|
+
saving is estimated — how much a fresh session would have saved depends on what
|
|
715
|
+
the work still needed, which the transcript does not say.
|
|
716
|
+
|
|
717
|
+
Subagent (sidechain) turns are excluded: they run against their own prefix, so
|
|
718
|
+
mixing them into the parent session's trajectory would misstate both.
|
|
719
|
+
"""
|
|
720
|
+
cfg = self.settings.get("hygiene", {})
|
|
721
|
+
thresholds = sorted(int(x) for x in cfg.get("context_thresholds", [100000, 150000]))
|
|
722
|
+
w, p = self.where(f)
|
|
723
|
+
rows = self.q(f"""SELECT r.session_id, r.ts, r.context_tokens ctx, r.est_cost_usd cost
|
|
724
|
+
FROM requests r WHERE {w} AND r.is_sidechain = 0 AND r.ts <> ''
|
|
725
|
+
ORDER BY r.session_id, r.ts""", p)
|
|
726
|
+
compact_rows = self.q("SELECT session_id, ts FROM prompts WHERE source='slash:/compact'"
|
|
727
|
+
" AND ts <> '' ORDER BY session_id, ts")
|
|
728
|
+
compacts_by_session = defaultdict(list)
|
|
729
|
+
for cr in compact_rows:
|
|
730
|
+
compacts_by_session[cr["session_id"]].append(cr["ts"])
|
|
731
|
+
|
|
732
|
+
total = sum(r["cost"] or 0 for r in rows)
|
|
733
|
+
above = {t: {"requests": 0, "cost_usd": 0.0, "sessions": 0, "cost_after_first_cross_usd": 0.0}
|
|
734
|
+
for t in thresholds}
|
|
735
|
+
sessions = {}
|
|
736
|
+
for r in rows:
|
|
737
|
+
cost, ctx = (r["cost"] or 0.0), (r["ctx"] or 0)
|
|
738
|
+
s = sessions.setdefault(r["session_id"], {
|
|
739
|
+
"session_id": r["session_id"], "requests": 0, "cost_usd": 0.0,
|
|
740
|
+
"max_context": 0, "first_cross": {t: None for t in thresholds},
|
|
741
|
+
"first_cross_idx": {t: None for t in thresholds},
|
|
742
|
+
"cost_after": {t: 0.0 for t in thresholds}, "traj": [], "compactions": 0,
|
|
743
|
+
"ever_crossed": {t: False for t in thresholds}, "_next_compact_idx": 0})
|
|
744
|
+
prev_ctx = s["traj"][-1][0] if s["traj"] else 0
|
|
745
|
+
compact_ts = compacts_by_session.get(r["session_id"], [])
|
|
746
|
+
crossed_compact = (s["_next_compact_idx"] < len(compact_ts)
|
|
747
|
+
and r["ts"] > compact_ts[s["_next_compact_idx"]])
|
|
748
|
+
if crossed_compact:
|
|
749
|
+
s["_next_compact_idx"] += 1
|
|
750
|
+
if is_compaction(prev_ctx, ctx, thresholds[0]) or crossed_compact:
|
|
751
|
+
s["compactions"] += 1
|
|
752
|
+
s["first_cross"] = {t: None for t in thresholds}
|
|
753
|
+
idx = s["requests"]
|
|
754
|
+
s["requests"] += 1
|
|
755
|
+
s["cost_usd"] += cost
|
|
756
|
+
s["max_context"] = max(s["max_context"], ctx)
|
|
757
|
+
s["traj"].append((ctx, cost))
|
|
758
|
+
for t in thresholds:
|
|
759
|
+
if ctx >= t:
|
|
760
|
+
above[t]["requests"] += 1
|
|
761
|
+
above[t]["cost_usd"] += cost
|
|
762
|
+
if s["first_cross"][t] is None:
|
|
763
|
+
s["first_cross"][t] = idx
|
|
764
|
+
s["ever_crossed"][t] = True
|
|
765
|
+
if s["first_cross_idx"][t] is None:
|
|
766
|
+
s["first_cross_idx"][t] = idx # the ORIGINAL crossing; never reset
|
|
767
|
+
if s["first_cross"][t] is not None:
|
|
768
|
+
s["cost_after"][t] += cost
|
|
769
|
+
for s in sessions.values():
|
|
770
|
+
for t in thresholds:
|
|
771
|
+
if s["ever_crossed"][t]:
|
|
772
|
+
above[t]["sessions"] += 1
|
|
773
|
+
above[t]["cost_after_first_cross_usd"] += s["cost_after"][t]
|
|
774
|
+
|
|
775
|
+
rank_t = thresholds[-1]
|
|
776
|
+
ranked = sorted(sessions.values(), key=lambda s: -s["cost_after"][rank_t])[:top]
|
|
777
|
+
ids = [s["session_id"] for s in ranked]
|
|
778
|
+
meta = {}
|
|
779
|
+
if ids:
|
|
780
|
+
ph = ",".join("?" * len(ids))
|
|
781
|
+
meta = {m["id"]: m for m in self.q(f"""SELECT s.id, s.title, pj.name project
|
|
782
|
+
FROM sessions s JOIN projects pj ON pj.id=s.project_id
|
|
783
|
+
WHERE s.id IN ({ph})""", ids)}
|
|
784
|
+
|
|
785
|
+
def downsample(traj):
|
|
786
|
+
n = len(traj)
|
|
787
|
+
if n <= trajectory_points:
|
|
788
|
+
return traj
|
|
789
|
+
step = n / trajectory_points
|
|
790
|
+
return [traj[int(i * step)] for i in range(trajectory_points)]
|
|
791
|
+
|
|
792
|
+
out_sessions = []
|
|
793
|
+
for s in ranked:
|
|
794
|
+
m = meta.get(s["session_id"], {})
|
|
795
|
+
traj = downsample(s["traj"])
|
|
796
|
+
out_sessions.append({
|
|
797
|
+
"session_id": s["session_id"], "title": m.get("title"), "project": m.get("project"),
|
|
798
|
+
"requests": s["requests"], "cost_usd": s["cost_usd"], "max_context": s["max_context"],
|
|
799
|
+
"compactions": s["compactions"],
|
|
800
|
+
"first_cross": {str(t): s["first_cross"][t] for t in thresholds},
|
|
801
|
+
"first_cross_idx": {str(t): s["first_cross_idx"][t] for t in thresholds},
|
|
802
|
+
"ever_crossed": {str(t): s["ever_crossed"][t] for t in thresholds},
|
|
803
|
+
"cost_after": {str(t): s["cost_after"][t] for t in thresholds},
|
|
804
|
+
"cost_after_pct": {str(t): (round(100.0 * s["cost_after"][t] / s["cost_usd"], 1)
|
|
805
|
+
if s["cost_usd"] else 0.0) for t in thresholds},
|
|
806
|
+
"context_trajectory": [c for c, _ in traj],
|
|
807
|
+
"cumulative_cost": [round(x, 4) for x in _cumsum(cost for _, cost in traj)],
|
|
808
|
+
"basis": "actual",
|
|
809
|
+
})
|
|
810
|
+
return {
|
|
811
|
+
"thresholds": thresholds,
|
|
812
|
+
"rank_threshold": rank_t,
|
|
813
|
+
"requests": len(rows), "sessions": len(sessions), "cost_usd": total,
|
|
814
|
+
"above": {str(t): {
|
|
815
|
+
**v,
|
|
816
|
+
"share_pct": round(100.0 * v["cost_usd"] / total, 1) if total else 0.0,
|
|
817
|
+
"share_after_first_cross_pct": (round(100.0 * v["cost_after_first_cross_usd"] / total, 1)
|
|
818
|
+
if total else 0.0),
|
|
819
|
+
} for t, v in above.items()},
|
|
820
|
+
"sessions_ranked": out_sessions,
|
|
821
|
+
"excluded": "subagent turns (own prefix)",
|
|
822
|
+
"undetectable": ["/clear"],
|
|
823
|
+
"note": ("Observed shares of spend. Nothing here estimates what compaction or a "
|
|
824
|
+
"fresh session would have saved. Auto-compaction is not recorded; it is "
|
|
825
|
+
"detected as the context dropping by more than half. A typed /compact is "
|
|
826
|
+
"recorded and also counts."),
|
|
827
|
+
"basis": "actual",
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
def ttl_replay(self, f=None):
|
|
831
|
+
"""5m vs 1h cache TTL over real segments — arithmetic, no behavioural assumption.
|
|
832
|
+
|
|
833
|
+
Gated on reconciliation: if replaying the TTL you actually used cannot reproduce
|
|
834
|
+
the cost that was logged, the counterfactual is not trustworthy either and no
|
|
835
|
+
number is returned.
|
|
836
|
+
"""
|
|
837
|
+
from .segments import replay, split_segments
|
|
838
|
+
w, p = self.where(f)
|
|
839
|
+
turns = self.q(f"""SELECT r.session_id, r.ts, r.model, r.priced_as, r.is_sidechain, r.agent_id,
|
|
840
|
+
r.input_tokens, r.output_tokens, r.cache_read_tokens,
|
|
841
|
+
r.cache_write_5m, r.cache_write_1h, r.est_cost_usd, r.context_tokens ctx
|
|
842
|
+
FROM requests r WHERE {w} AND r.agent='claude' AND r.ts <> ''
|
|
843
|
+
ORDER BY r.session_id, r.ts""", p)
|
|
844
|
+
segs = split_segments(turns)
|
|
845
|
+
out = replay(segs, self.pricing)
|
|
846
|
+
out["turns"] = len(turns)
|
|
847
|
+
out["undetectable_boundaries"] = ["/clear"]
|
|
848
|
+
out["note"] = ("Segments break at session start, subagent start, model change and "
|
|
849
|
+
"compaction. Auto-compaction is not recorded; it is detected as the "
|
|
850
|
+
"context dropping by more than half. A typed /compact is recorded and "
|
|
851
|
+
"also counts.")
|
|
852
|
+
return out
|
|
853
|
+
|
|
854
|
+
def long_context_pricing(self, f=None):
|
|
855
|
+
"""Requests whose context exceeded the model's standard window.
|
|
856
|
+
|
|
857
|
+
These could only have been served by the long-context variant, which bills at a
|
|
858
|
+
premium. Where a `[1m]` price list exists they are already repriced; where it
|
|
859
|
+
does not, they are billed at the standard rate and the estimate is LOW — that is
|
|
860
|
+
a gap in config/pricing.json, not in the data, so it is reported rather than
|
|
861
|
+
guessed at.
|
|
862
|
+
"""
|
|
863
|
+
w, p = self.where(f)
|
|
864
|
+
rows = self.q(f"""SELECT r.model, r.priced_as, r.unpriced_long_context u,
|
|
865
|
+
COUNT(*) n, SUM(r.est_cost_usd) cost, MAX(r.context_tokens) mx
|
|
866
|
+
FROM requests r WHERE {w} AND r.context_tokens > 0
|
|
867
|
+
AND (r.priced_as LIKE '%[1m]' OR r.unpriced_long_context = 1)
|
|
868
|
+
GROUP BY r.model, r.priced_as, r.unpriced_long_context""", p)
|
|
869
|
+
repriced = [r for r in rows if not r["u"]]
|
|
870
|
+
unpriced = [r for r in rows if r["u"]]
|
|
871
|
+
|
|
872
|
+
# rows priced against no known model at all (model_known=0) are a distinct gap:
|
|
873
|
+
# not "over the standard window", but "no rate for this model in pricing.json".
|
|
874
|
+
unknown_rows = self.q(f"""SELECT r.model, COUNT(*) n, SUM(r.est_cost_usd) cost
|
|
875
|
+
FROM requests r WHERE {w} AND r.model_known = 0
|
|
876
|
+
AND r.model LIKE 'claude%'
|
|
877
|
+
GROUP BY r.model""", p)
|
|
878
|
+
unknown_requests = sum(r["n"] for r in unknown_rows)
|
|
879
|
+
unknown_models = sorted({r["model"] for r in unknown_rows})
|
|
880
|
+
|
|
881
|
+
return {
|
|
882
|
+
"repriced": repriced,
|
|
883
|
+
"unpriced": unpriced,
|
|
884
|
+
"repriced_requests": sum(r["n"] for r in repriced),
|
|
885
|
+
"unpriced_requests": sum(r["n"] for r in unpriced),
|
|
886
|
+
"unpriced_cost_usd": sum(r["cost"] or 0 for r in unpriced),
|
|
887
|
+
"unpriced_models": sorted({r["model"] for r in unpriced}),
|
|
888
|
+
"unknown_model_requests": unknown_requests,
|
|
889
|
+
"unknown_model_cost_usd": sum(r["cost"] or 0 for r in unknown_rows),
|
|
890
|
+
"unknown_models": unknown_models,
|
|
891
|
+
"message": (
|
|
892
|
+
"%d requests exceeded their model's standard context window with no "
|
|
893
|
+
"long-context price configured, so their cost is understated. Add a "
|
|
894
|
+
"\"<model>[1m]\" entry to config/pricing.json for: %s."
|
|
895
|
+
% (sum(r["n"] for r in unpriced), ", ".join(sorted({r["model"] for r in unpriced})))
|
|
896
|
+
if unpriced else ""),
|
|
897
|
+
"unknown_message": (
|
|
898
|
+
"%d requests used a claude-* model with no entry in pricing.json at "
|
|
899
|
+
"all, so their cost is a fallback guess. Add pricing.json entries for: %s."
|
|
900
|
+
% (unknown_requests, ", ".join(unknown_models))
|
|
901
|
+
if unknown_rows else ""),
|
|
902
|
+
"basis": "estimated",
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
def cache_cost_split(self, f=None):
|
|
906
|
+
"""Cache read vs write split by estimated cost, not just by token count.
|
|
907
|
+
|
|
908
|
+
A read is billed at a fraction of the input rate and a write at a premium, so
|
|
909
|
+
the token split and the dollar split are different numbers — writes are a small
|
|
910
|
+
share of cache tokens and a much larger share of cache spend. Reporting only the
|
|
911
|
+
token ratio overstates how healthy caching is, which is why the scorecard grades
|
|
912
|
+
this dimension on cost. Prices are re-derived per model here rather than read off
|
|
913
|
+
requests.est_cost_usd, which is a single blended figure per request.
|
|
914
|
+
"""
|
|
915
|
+
w, p = self.where(f)
|
|
916
|
+
rows = self.q(f"""SELECT model,
|
|
917
|
+
SUM(cache_read_tokens) cr,
|
|
918
|
+
SUM(cache_write_5m) w5, SUM(cache_write_1h) w1
|
|
919
|
+
FROM requests r WHERE {w} GROUP BY model""", p)
|
|
920
|
+
read_tok = w5_tok = w1_tok = 0
|
|
921
|
+
read_cost = w5_cost = w1_cost = 0.0
|
|
922
|
+
for r in rows:
|
|
923
|
+
cr, w5, w1 = (r["cr"] or 0), (r["w5"] or 0), (r["w1"] or 0)
|
|
924
|
+
read_tok += cr
|
|
925
|
+
w5_tok += w5
|
|
926
|
+
w1_tok += w1
|
|
927
|
+
read_cost += self.pricing.estimate(r["model"], cache_read=cr)
|
|
928
|
+
w5_cost += self.pricing.estimate(r["model"], cache_write_5m=w5)
|
|
929
|
+
w1_cost += self.pricing.estimate(r["model"], cache_write_1h=w1)
|
|
930
|
+
|
|
931
|
+
write_tok = w5_tok + w1_tok
|
|
932
|
+
write_cost = w5_cost + w1_cost
|
|
933
|
+
tok_total = read_tok + write_tok
|
|
934
|
+
cost_total = read_cost + write_cost
|
|
935
|
+
per_read = (read_cost / read_tok) if read_tok else 0
|
|
936
|
+
per_write = (write_cost / write_tok) if write_tok else 0
|
|
937
|
+
return {
|
|
938
|
+
"read_tokens": read_tok, "write_tokens": write_tok,
|
|
939
|
+
"write_5m_tokens": w5_tok, "write_1h_tokens": w1_tok,
|
|
940
|
+
"read_cost_usd": read_cost, "write_cost_usd": write_cost,
|
|
941
|
+
"write_5m_cost_usd": w5_cost, "write_1h_cost_usd": w1_cost,
|
|
942
|
+
"read_token_share": (read_tok / tok_total) if tok_total else None,
|
|
943
|
+
"read_cost_share": (read_cost / cost_total) if cost_total else None,
|
|
944
|
+
"write_cost_share": (write_cost / cost_total) if cost_total else None,
|
|
945
|
+
# how much more a write token costs than a read token, same workload
|
|
946
|
+
"write_vs_read_multiple": (per_write / per_read) if per_read else None,
|
|
947
|
+
# 1h writes cost more per token than 5m writes; whether that premium is worth
|
|
948
|
+
# paying depends on the real inter-turn gaps, which ttl_replay() prices
|
|
949
|
+
"write_1h_token_share": (w1_tok / write_tok) if write_tok else None,
|
|
950
|
+
"basis": "estimated",
|
|
951
|
+
}
|
|
952
|
+
|
|
546
953
|
def efficiency(self, f=None):
|
|
547
954
|
w, p = self.where(f)
|
|
548
955
|
t = self.one(f"""SELECT SUM(input_tokens) i, SUM(output_tokens) o,
|
|
@@ -554,6 +961,29 @@ class Analytics:
|
|
|
554
961
|
tot = t["tot"] or 1
|
|
555
962
|
prompt_side = (t["i"] or 0) + (t["cr"] or 0) + (t["cw"] or 0)
|
|
556
963
|
cache_total = (t["cr"] or 0) + (t["cw"] or 0)
|
|
964
|
+
cache_cost = self.cache_cost_split(f)
|
|
965
|
+
|
|
966
|
+
# Break-even margin: how much of cache spend came back as read discount, net of
|
|
967
|
+
# write premium. +1 = all discount, -1 = all premium, computed per model since
|
|
968
|
+
# rates differ.
|
|
969
|
+
bm_rows = self.q(f"""SELECT COALESCE(priced_as, model) priced_as, SUM(cache_read_tokens) cr,
|
|
970
|
+
SUM(cache_write_5m) w5, SUM(cache_write_1h) w1
|
|
971
|
+
FROM requests r WHERE {w} GROUP BY COALESCE(priced_as, model)""", p)
|
|
972
|
+
discount = premium = cache_cost_total = 0.0
|
|
973
|
+
for r in bm_rows:
|
|
974
|
+
model = r["priced_as"]
|
|
975
|
+
reads, w5, w1 = (r["cr"] or 0), (r["w5"] or 0), (r["w1"] or 0)
|
|
976
|
+
rt = self.pricing.rates(model) or {}
|
|
977
|
+
inp = float(rt.get("input") or 0)
|
|
978
|
+
read_rate = float(rt.get("cache_read") or 0)
|
|
979
|
+
w5_rate = float(rt.get("cache_write_5m") or 0)
|
|
980
|
+
w1_rate = float(rt.get("cache_write_1h") or 0)
|
|
981
|
+
discount += reads * (inp - read_rate) / 1e6
|
|
982
|
+
premium += (w5 * (w5_rate - inp) + w1 * (w1_rate - inp)) / 1e6
|
|
983
|
+
cache_cost_total += (reads * read_rate + w5 * w5_rate + w1 * w1_rate) / 1e6
|
|
984
|
+
breakeven_margin = (max(-1.0, min(1.0, (discount - premium) / cache_cost_total))
|
|
985
|
+
if cache_cost_total else None)
|
|
986
|
+
|
|
557
987
|
sess = self.sessions(f, limit=100000, order="cost")
|
|
558
988
|
scored = [s for s in sess if s["tokens"] and s["prompts"]]
|
|
559
989
|
for s in scored:
|
|
@@ -564,6 +994,7 @@ class Analytics:
|
|
|
564
994
|
"output_per_input": ((t["o"] or 0) / prompt_side) if prompt_side else 0,
|
|
565
995
|
"thinking_share_of_output": ((t["think"] or 0) / (t["o"] or 1)),
|
|
566
996
|
"cache_hit_ratio": ((t["cr"] or 0) / cache_total) if cache_total else None,
|
|
997
|
+
"cache_read_cost_share": cache_cost["read_cost_share"],
|
|
567
998
|
"tokens_per_request": tot / (t["n"] or 1),
|
|
568
999
|
"avg_context_tokens": t["avgctx"],
|
|
569
1000
|
"cost_per_1k_output": (1000.0 * (t["cost"] or 0) / (t["o"] or 1)),
|
|
@@ -571,10 +1002,14 @@ class Analytics:
|
|
|
571
1002
|
max(self.one(f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r WHERE {w}", p)["n"], 1)),
|
|
572
1003
|
"cache": {
|
|
573
1004
|
"reads": t["cr"], "writes": t["cw"],
|
|
1005
|
+
"cost_split": cache_cost,
|
|
1006
|
+
"breakeven_margin": breakeven_margin,
|
|
574
1007
|
"cost_with_cache": t["cost"], "cost_without_cache": t["cost_nc"],
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
1008
|
+
# Named for what it is. This used to be "estimated_savings_usd", and the
|
|
1009
|
+
# UI called it a saving; it is the gap to a run that never happened.
|
|
1010
|
+
"uncached_counterfactual_delta_usd": (t["cost_nc"] or 0) - (t["cost"] or 0),
|
|
1011
|
+
"uncached_counterfactual_pct": (round(100.0 * ((t["cost_nc"] or 0) - (t["cost"] or 0))
|
|
1012
|
+
/ (t["cost_nc"] or 1), 1)),
|
|
578
1013
|
"basis": "estimated",
|
|
579
1014
|
},
|
|
580
1015
|
"low_efficiency_sessions": scored[:10],
|
|
@@ -662,23 +1097,22 @@ class Analytics:
|
|
|
662
1097
|
ORDER BY cost DESC LIMIT 15""", p + [rules["long_prompt_chars"]])
|
|
663
1098
|
budget_chars = rules["long_prompt_chars"]
|
|
664
1099
|
for r in rows:
|
|
665
|
-
|
|
666
|
-
resent = over_tokens * max(r["requests"], 1)
|
|
667
|
-
r["excess"] = (r["cost"] * resent / r["tokens"]) if r["tokens"] else 0
|
|
1100
|
+
r["excess"] = 0.0
|
|
668
1101
|
if rows:
|
|
669
1102
|
add("high", "long_prompts",
|
|
670
1103
|
f"{len(rows)} very long prompts (>{budget_chars:,} chars)",
|
|
671
1104
|
"Long pasted prompts inflate the cached prefix re-sent on every following turn.",
|
|
672
1105
|
rows, "Move large pasted context into a file and reference it, or summarize first.",
|
|
673
1106
|
"prompt_id",
|
|
674
|
-
|
|
1107
|
+
"none claimed — flagged for review only")
|
|
675
1108
|
|
|
676
|
-
# 2. duplicate prompts — excess is the cost of the repeats, not the first ask
|
|
677
|
-
|
|
1109
|
+
# 2. duplicate prompts — excess is the cost of the repeats, not the first ask,
|
|
1110
|
+
# counted only within a single session so cross-session coincidences don't count.
|
|
1111
|
+
dups = self.q(f"""SELECT pr.norm_hash, pr.session_id, COUNT(*) n, substr(MIN(pr.text),1,160) preview,
|
|
678
1112
|
SUM(pr.est_cost_usd) cost, SUM(pr.billable_tokens) tokens,
|
|
679
1113
|
MIN(pr.est_cost_usd) first_cost, GROUP_CONCAT(pr.id) prompt_ids
|
|
680
1114
|
FROM prompts pr WHERE {pfilter} AND pr.char_len > 25
|
|
681
|
-
GROUP BY pr.norm_hash HAVING n > 1
|
|
1115
|
+
GROUP BY pr.norm_hash, pr.session_id HAVING n > 1
|
|
682
1116
|
ORDER BY cost DESC LIMIT 15""", p)
|
|
683
1117
|
for d in dups:
|
|
684
1118
|
d["prompt_id"] = int(d["prompt_ids"].split(",")[0])
|
|
@@ -692,11 +1126,18 @@ class Analytics:
|
|
|
692
1126
|
|
|
693
1127
|
# 3. low-yield sessions — excess is what the session cost ABOVE what the same
|
|
694
1128
|
# output would have cost at your own median session efficiency.
|
|
1129
|
+
# The baseline is drawn from the SAME population that is eligible to be
|
|
1130
|
+
# flagged — sessions above huge_session_tokens, inside the current filter.
|
|
1131
|
+
# Grading big sessions against the median of all sessions punishes them for
|
|
1132
|
+
# something inherent to long agentic work: output ratio falls as a session
|
|
1133
|
+
# grows, so a small-session median flags most large sessions by construction.
|
|
695
1134
|
ratios = [r["x"] for r in self.q(
|
|
696
|
-
"SELECT CAST(output_tokens AS REAL)/billable_tokens x FROM sessions
|
|
697
|
-
|
|
1135
|
+
f"""SELECT CAST(s.output_tokens AS REAL)/s.billable_tokens x FROM sessions s
|
|
1136
|
+
WHERE {sfilter} AND s.billable_tokens > ?""",
|
|
1137
|
+
p + [rules["huge_session_tokens"]])]
|
|
698
1138
|
median_ratio = statistics.median(ratios) if ratios else 0.0
|
|
699
1139
|
cutoff = median_ratio * rules["low_output_ratio_vs_median"]
|
|
1140
|
+
baseline_n = len(ratios)
|
|
700
1141
|
low = self.q(f"""SELECT s.id session_id, s.title, s.billable_tokens tokens,
|
|
701
1142
|
s.output_tokens out_tokens, s.est_cost_usd cost, s.request_count requests,
|
|
702
1143
|
(CAST(s.output_tokens AS REAL)/MAX(s.billable_tokens,1)) output_ratio,
|
|
@@ -706,16 +1147,15 @@ class Analytics:
|
|
|
706
1147
|
ORDER BY cost DESC LIMIT 15""",
|
|
707
1148
|
p + [rules["huge_session_tokens"], cutoff])
|
|
708
1149
|
for r in low:
|
|
709
|
-
|
|
710
|
-
# baseline cost scales by (actual ratio / median ratio)
|
|
711
|
-
r["excess"] = r["cost"] * (1 - (r["output_ratio"] / median_ratio)) if median_ratio else 0
|
|
1150
|
+
r["excess"] = 0.0
|
|
712
1151
|
if low:
|
|
713
|
-
add("
|
|
1152
|
+
add("medium", "low_yield_sessions",
|
|
714
1153
|
f"{len(low)} large sessions yielded under {cutoff*100:.2f}% output tokens",
|
|
715
|
-
f"
|
|
716
|
-
f"These ran well below
|
|
1154
|
+
f"Among your {baseline_n} comparably large sessions the median turns "
|
|
1155
|
+
f"{median_ratio*100:.2f}% of billable tokens into output. These ran well below "
|
|
1156
|
+
f"that while consuming heavy context.",
|
|
717
1157
|
low, "Start a fresh session or /compact once a thread stops producing new output.",
|
|
718
|
-
"session_id", "
|
|
1158
|
+
"session_id", "none claimed — the same output at another ratio is a counterfactual")
|
|
719
1159
|
|
|
720
1160
|
# 4. frontier model on small tasks — excess is computed against the cheaper tier.
|
|
721
1161
|
frontier = [m for m, v in self.pricing.models.items()
|
|
@@ -733,59 +1173,70 @@ class Analytics:
|
|
|
733
1173
|
small = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.category,
|
|
734
1174
|
pr.session_id, pr.est_cost_usd cost, pr.output_tokens out_tokens,
|
|
735
1175
|
pr.billable_tokens tokens, pr.models, pr.input_tokens,
|
|
736
|
-
pr.cache_read_tokens, pr.cache_write_tokens
|
|
1176
|
+
pr.cache_read_tokens, pr.cache_write_tokens,
|
|
1177
|
+
(SELECT COALESCE(SUM(rq.cache_write_5m),0) FROM requests rq
|
|
1178
|
+
WHERE rq.prompt_id = pr.id) c5,
|
|
1179
|
+
(SELECT COALESCE(SUM(rq.cache_write_1h),0) FROM requests rq
|
|
1180
|
+
WHERE rq.prompt_id = pr.id) c1
|
|
737
1181
|
FROM prompts pr WHERE {pfilter}
|
|
738
|
-
AND pr.output_tokens < ? AND pr.est_cost_usd > 0
|
|
1182
|
+
AND pr.output_tokens < ? AND pr.tool_calls = 0 AND pr.est_cost_usd > 0
|
|
739
1183
|
AND EXISTS (SELECT 1 FROM requests r2 WHERE r2.prompt_id=pr.id
|
|
740
1184
|
AND r2.model IN ({ph}))
|
|
741
1185
|
ORDER BY cost DESC LIMIT 15""",
|
|
742
1186
|
p + [rules["simple_task_output_tokens"]] + frontier)
|
|
743
1187
|
for r in small:
|
|
744
|
-
|
|
745
|
-
r["cache_read_tokens"], r["cache_write_tokens"], 0)
|
|
746
|
-
if cheaper else r["cost"])
|
|
747
|
-
r["excess"] = max(r["cost"] - alt, 0)
|
|
1188
|
+
r["excess"] = 0.0
|
|
748
1189
|
if small:
|
|
749
1190
|
add("medium", "frontier_on_small_tasks",
|
|
750
1191
|
f"{len(small)} frontier-model prompts produced under "
|
|
751
1192
|
f"{rules['simple_task_output_tokens']} output tokens",
|
|
752
1193
|
"Short, simple turns running on the most expensive model tier.",
|
|
753
1194
|
small, "Route short lookups and confirmations to a cheaper model tier.",
|
|
754
|
-
"prompt_id",
|
|
755
|
-
f"difference against the same tokens priced at {self.pricing.display_name(cheaper)}"
|
|
756
|
-
if cheaper else "n/a")
|
|
1195
|
+
"prompt_id", "none claimed")
|
|
757
1196
|
|
|
758
1197
|
# 5. tool loops — excess is the share of the loop beyond the threshold.
|
|
1198
|
+
loop_calls = rules.get("tool_loop_calls", 40)
|
|
759
1199
|
loops = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.session_id,
|
|
760
1200
|
pr.tool_calls tools, pr.est_cost_usd cost, pr.billable_tokens tokens
|
|
761
|
-
FROM prompts pr WHERE {pfilter} AND pr.tool_calls >
|
|
762
|
-
ORDER BY cost DESC LIMIT 15""", p)
|
|
1201
|
+
FROM prompts pr WHERE {pfilter} AND pr.tool_calls > ?
|
|
1202
|
+
ORDER BY cost DESC LIMIT 15""", p + [loop_calls])
|
|
763
1203
|
for r in loops:
|
|
764
|
-
r["excess"] =
|
|
1204
|
+
r["excess"] = 0.0
|
|
765
1205
|
if loops:
|
|
766
|
-
add("medium", "tool_loops", f"{len(loops)} prompts triggered
|
|
1206
|
+
add("medium", "tool_loops", f"{len(loops)} prompts triggered {loop_calls}+ tool calls",
|
|
767
1207
|
"Long agentic loops re-send the whole conversation each step, so cost grows super-linearly.",
|
|
768
1208
|
loops, "Split the task, or give more precise instructions up front.", "prompt_id",
|
|
769
|
-
"
|
|
1209
|
+
"none claimed")
|
|
770
1210
|
|
|
771
|
-
# 6. poor cache reuse — excess is the write premium over
|
|
1211
|
+
# 6. poor cache reuse — excess is the break-even: the cache-write premium over
|
|
1212
|
+
# plain input pricing, minus the discount actually earned on the reads.
|
|
772
1213
|
poor = self.q(f"""SELECT s.id session_id, s.title, proj.name project,
|
|
773
1214
|
s.cache_read_tokens reads, s.cache_write_tokens writes,
|
|
1215
|
+
(SELECT COALESCE(SUM(r.cache_write_5m),0) FROM requests r WHERE r.session_id=s.id) w5,
|
|
1216
|
+
(SELECT COALESCE(SUM(r.cache_write_1h),0) FROM requests r WHERE r.session_id=s.id) w1,
|
|
774
1217
|
s.est_cost_usd cost, s.request_count requests, s.models
|
|
775
1218
|
FROM sessions s JOIN projects proj ON proj.id=s.project_id
|
|
776
|
-
WHERE {sfilter} AND s.cache_write_tokens >
|
|
777
|
-
|
|
778
|
-
|
|
1219
|
+
WHERE {sfilter} AND s.cache_write_tokens > ?
|
|
1220
|
+
ORDER BY cost DESC""", p + [rules.get("poor_cache_min_writes", 500_000)])
|
|
1221
|
+
flagged = []
|
|
779
1222
|
for r in poor:
|
|
780
1223
|
m = (r["models"] or "").split(",")[0]
|
|
781
1224
|
rt = self.pricing.rates(m)
|
|
782
|
-
|
|
783
|
-
|
|
1225
|
+
g = lambda k: float(rt.get(k, 0.0))
|
|
1226
|
+
# break-even: premium paid on writes minus discount earned on reads
|
|
1227
|
+
net = (r["w5"] * (g("cache_write_5m") - g("input"))
|
|
1228
|
+
+ r["w1"] * (g("cache_write_1h") - g("input"))
|
|
1229
|
+
- r["reads"] * (g("input") - g("cache_read"))) / 1_000_000.0
|
|
1230
|
+
if net > 0:
|
|
1231
|
+
r["excess"] = net
|
|
1232
|
+
flagged.append(r)
|
|
1233
|
+
poor = flagged[:15]
|
|
784
1234
|
if poor:
|
|
785
1235
|
add("medium", "poor_cache_reuse", f"{len(poor)} sessions wrote cache they barely reused",
|
|
786
1236
|
"Cache writes cost more than plain input; they only pay off when read back repeatedly.",
|
|
787
1237
|
poor, "Keep related work in one continuous session so the cached prefix is reused.",
|
|
788
|
-
"session_id", "
|
|
1238
|
+
"session_id", "cache-write premium minus the read discount actually earned, "
|
|
1239
|
+
"at this model's rates")
|
|
789
1240
|
|
|
790
1241
|
# 7. long-lived sparse sessions — informational, no excess claimed.
|
|
791
1242
|
idle = self.q(f"""SELECT s.id session_id, s.title, proj.name project, s.duration_s,
|
|
@@ -863,436 +1314,120 @@ class Analytics:
|
|
|
863
1314
|
"touched — money worth reviewing. Estimated excess is how much more that "
|
|
864
1315
|
"work cost than a reasonable baseline, and is the actual waste figure. "
|
|
865
1316
|
"Both are estimates.",
|
|
1317
|
+
"excess_note": "Estimated excess is claimed only where the baseline is measured: the cost of "
|
|
1318
|
+
"repeating an identical prompt in the same session, and cache writes that were "
|
|
1319
|
+
"never read back enough to pay for themselves. Everything else is exposed spend "
|
|
1320
|
+
"to review, not waste.",
|
|
866
1321
|
"basis": "estimated"}
|
|
867
1322
|
|
|
868
1323
|
# ---------------- recommendations ----------------
|
|
869
|
-
#
|
|
870
|
-
#
|
|
871
|
-
#
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
"
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
TIER_RANK = {"economy": 0, "balanced": 1, "frontier": 2}
|
|
883
|
-
|
|
884
|
-
# How each vendor's agent is told to change model. Used by model_switch() and by the
|
|
885
|
-
# model_downgrade recommendations, which must name the agent they keep you inside.
|
|
886
|
-
SWITCH_HOW_BY_AGENT = {
|
|
887
|
-
"anthropic": {"agent": "Claude Code", "session": "/model <name>",
|
|
888
|
-
"project": '"model": "<name>" in <repo>/.claude/settings.json'},
|
|
889
|
-
"openai": {"agent": "Codex", "session": "/model in Codex, or codex -m <name>",
|
|
890
|
-
"project": 'model = "<name>" in ~/.codex/config.toml (or a profile)'},
|
|
891
|
-
"google": {"agent": "Gemini CLI", "session": "/model in Gemini CLI, or gemini -m <name>",
|
|
892
|
-
"project": '"model": {"name": "<name>"} in <repo>/.gemini/settings.json'},
|
|
893
|
-
}
|
|
894
|
-
|
|
895
|
-
def _cheapest(self, tier, provider=None):
|
|
896
|
-
"""Cheapest model in a tier — from the same provider, since an agent can only
|
|
897
|
-
switch between its own vendor's models (Codex can't run Haiku)."""
|
|
898
|
-
ms = [m for m, v in self.pricing.models.items() if v.get("tier") == tier
|
|
899
|
-
and (provider is None or v.get("provider", "anthropic") == provider)]
|
|
900
|
-
return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
|
|
901
|
-
|
|
902
|
-
# ---------------- evidence: what the cheaper model actually did ----------------
|
|
903
|
-
# model_switch() reprices your tokens on a cheaper model, which assumes the cheaper
|
|
904
|
-
# model would have done the same work in the same number of turns. Often it would
|
|
905
|
-
# not: a weaker model can take five times the turns on the same task, and the
|
|
906
|
-
# repricing then promises a saving that never arrives.
|
|
907
|
-
#
|
|
908
|
-
# Where you have already run more than one model on the same kind of work, we do not
|
|
909
|
-
# have to assume anything. This compares what each model actually cost per prompt on
|
|
910
|
-
# that category, and how much work it took to get there.
|
|
911
|
-
|
|
912
|
-
MIN_PROMPTS = 8 # below this a per-category average is noise, not evidence
|
|
913
|
-
SAVING_FLOOR_PCT = 20 # smaller gaps are inside the noise of what you happened to ask
|
|
914
|
-
TURN_TOLERANCE = 1.35 # more turns than this and the cheaper model was grinding
|
|
915
|
-
REPEAT_TOLERANCE = 12.0 # percentage points of extra re-asking we will accept
|
|
916
|
-
|
|
917
|
-
def model_evidence(self, f=None):
|
|
918
|
-
"""Back-test a model switch against your own history.
|
|
919
|
-
|
|
920
|
-
For every category where you ran more than one model, report what each one
|
|
921
|
-
actually cost per prompt and what it took: turns, tool calls, and how often you
|
|
922
|
-
had to ask the same thing again. A candidate is only recommended when it was
|
|
923
|
-
genuinely cheaper per prompt *and* did not need materially more work to get
|
|
924
|
-
there — which is the part a repricing cannot see.
|
|
1324
|
+
# How close to the ceiling counts as "near". 90% was the cut the old reprice used
|
|
1325
|
+
# to decide a request could not move to a smaller window; it is kept as the
|
|
1326
|
+
# observation threshold because that is where re-read cost visibly concentrates.
|
|
1327
|
+
NEAR_WINDOW_PCT = 0.9
|
|
1328
|
+
|
|
1329
|
+
def context_window_fit(self, f=None):
|
|
1330
|
+
"""How much spend ran near the ceiling of the context window actually in use.
|
|
1331
|
+
|
|
1332
|
+
This is the one piece of the old model-switch reprice worth keeping — the
|
|
1333
|
+
window check — turned from a what-if into an observation. Nothing is
|
|
1334
|
+
repriced and no alternative model is assumed. A request is "near" when its
|
|
1335
|
+
prompt side is at least NEAR_WINDOW_PCT of its model's window, and "over"
|
|
1336
|
+
when it exceeds it, which can only mean the long-context variant served it.
|
|
925
1337
|
"""
|
|
926
1338
|
w, p = self.where(f)
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
clean = ("pr.models IS NOT NULL AND pr.models NOT LIKE '%,%' "
|
|
934
|
-
"AND pr.models != '<synthetic>' AND pr.est_cost_usd > 0")
|
|
935
|
-
rows = self.q(f"""SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
936
|
-
pr.agent agent, COUNT(*) prompts, SUM(pr.est_cost_usd) cost,
|
|
937
|
-
AVG(pr.est_cost_usd) cost_per_prompt,
|
|
938
|
-
AVG(pr.request_count) turns, AVG(pr.tool_calls) tools,
|
|
939
|
-
AVG(pr.output_tokens) out_tokens, AVG(pr.max_context_tokens) ctx
|
|
940
|
-
FROM prompts pr WHERE {scope} AND {clean}
|
|
941
|
-
GROUP BY 1, 2, 3 HAVING prompts >= ?""", p + [self.MIN_PROMPTS])
|
|
942
|
-
|
|
943
|
-
repeats = {(r["category"], r["model"]): r["pct"] for r in self.q(f"""
|
|
944
|
-
SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
945
|
-
ROUND(100.0 * SUM(CASE WHEN dup.n > 1 THEN 1 ELSE 0 END) / COUNT(*), 1) pct
|
|
946
|
-
FROM prompts pr
|
|
947
|
-
LEFT JOIN (SELECT norm_hash, COUNT(*) n FROM prompts GROUP BY norm_hash) dup
|
|
948
|
-
ON dup.norm_hash = pr.norm_hash
|
|
949
|
-
WHERE {scope} AND {clean}
|
|
950
|
-
GROUP BY 1, 2""", p)}
|
|
951
|
-
|
|
952
|
-
by_cat = defaultdict(list)
|
|
953
|
-
for r in rows:
|
|
954
|
-
r["repeat_pct"] = repeats.get((r["category"], r["model"]), 0.0) or 0.0
|
|
955
|
-
r["name"] = self.pricing.display_name(r["model"])
|
|
956
|
-
r["tier"] = self.pricing.tier(r["model"])
|
|
957
|
-
by_cat[r["category"]].append(r)
|
|
958
|
-
|
|
959
|
-
out, total_save = [], 0.0
|
|
960
|
-
for cat, models in by_cat.items():
|
|
961
|
-
if len(models) < 2:
|
|
962
|
-
continue
|
|
963
|
-
# The incumbent is what you spend the most on here — that is the bill a
|
|
964
|
-
# switch would actually change.
|
|
965
|
-
cur = max(models, key=lambda m: m["cost"])
|
|
966
|
-
rule = self.SWITCH_RULES.get(cat, ("balanced", "low"))[0]
|
|
967
|
-
cands = []
|
|
968
|
-
for m in models:
|
|
969
|
-
if m["model"] == cur["model"] or m["cost_per_prompt"] >= cur["cost_per_prompt"]:
|
|
970
|
-
continue
|
|
971
|
-
# Only models the same agent can run. Telling a Claude Code user to use a
|
|
972
|
-
# GPT model is not a setting change, it is a different tool, and the
|
|
973
|
-
# comparison would be between two different ways of working.
|
|
974
|
-
if m["agent"] != cur["agent"]:
|
|
975
|
-
continue
|
|
976
|
-
save_pct = 100.0 * (1 - m["cost_per_prompt"] / cur["cost_per_prompt"])
|
|
977
|
-
turn_ratio = (m["turns"] / cur["turns"]) if cur["turns"] else 1.0
|
|
978
|
-
repeat_delta = m["repeat_pct"] - cur["repeat_pct"]
|
|
979
|
-
# Observed, not repriced: what your own prompts cost on each side.
|
|
980
|
-
save = (cur["cost_per_prompt"] - m["cost_per_prompt"]) * cur["prompts"]
|
|
981
|
-
if save_pct < self.SAVING_FLOOR_PCT:
|
|
982
|
-
verdict, why = "marginal", (
|
|
983
|
-
f"Only {save_pct:.0f}% cheaper per prompt — inside the noise of what "
|
|
984
|
-
f"you happened to ask each model.")
|
|
985
|
-
elif turn_ratio > self.TURN_TOLERANCE:
|
|
986
|
-
verdict, why = "risky", (
|
|
987
|
-
f"Cost {save_pct:.0f}% less per prompt but took {turn_ratio:.1f}x the "
|
|
988
|
-
f"turns ({m['turns']:.0f} vs {cur['turns']:.0f}). It got there by "
|
|
989
|
-
f"grinding, and that is the cost the headline number misses.")
|
|
990
|
-
elif repeat_delta > self.REPEAT_TOLERANCE:
|
|
991
|
-
verdict, why = "risky", (
|
|
992
|
-
f"{save_pct:.0f}% cheaper per prompt, but you re-asked "
|
|
993
|
-
f"{m['repeat_pct']:.0f}% of these prompts against "
|
|
994
|
-
f"{cur['repeat_pct']:.0f}% on {cur['name']} — rework you paid for twice.")
|
|
995
|
-
elif rule == "keep":
|
|
996
|
-
verdict, why = "caution", (
|
|
997
|
-
f"{save_pct:.0f}% cheaper per prompt and no more turns, but {cat.replace('_',' ')} "
|
|
998
|
-
f"is reasoning-heavy work where a miss is expensive in ways this data "
|
|
999
|
-
f"cannot show. Worth a trial, not a default.")
|
|
1000
|
-
else:
|
|
1001
|
-
verdict, why = "supported", (
|
|
1002
|
-
f"{save_pct:.0f}% cheaper per prompt on {m['prompts']} of your own "
|
|
1003
|
-
f"{cat.replace('_',' ')} prompts, in {turn_ratio:.1f}x the turns "
|
|
1004
|
-
f"({m['turns']:.0f} vs {cur['turns']:.0f}) with "
|
|
1005
|
-
f"{'less' if repeat_delta <= 0 else 'similar'} re-asking. "
|
|
1006
|
-
f"This is measured, not modelled.")
|
|
1007
|
-
cands.append({
|
|
1008
|
-
"model": m["model"], "name": m["name"], "tier": m["tier"],
|
|
1009
|
-
"prompts": m["prompts"], "cost_per_prompt": m["cost_per_prompt"],
|
|
1010
|
-
"turns": m["turns"], "tools": m["tools"], "repeat_pct": m["repeat_pct"],
|
|
1011
|
-
"savings_pct": round(save_pct, 1), "turn_ratio": round(turn_ratio, 2),
|
|
1012
|
-
"repeat_delta": round(repeat_delta, 1),
|
|
1013
|
-
"estimated_savings_usd": round(save, 2), "verdict": verdict, "why": why})
|
|
1014
|
-
if not cands:
|
|
1015
|
-
continue
|
|
1016
|
-
cands.sort(key=lambda c: (c["verdict"] != "supported", -c["estimated_savings_usd"]))
|
|
1017
|
-
best = cands[0] if cands[0]["verdict"] == "supported" else None
|
|
1018
|
-
if best:
|
|
1019
|
-
total_save += best["estimated_savings_usd"]
|
|
1020
|
-
out.append({
|
|
1021
|
-
"category": cat,
|
|
1022
|
-
"agent": cur["agent"],
|
|
1023
|
-
"current": {"model": cur["model"], "name": cur["name"], "prompts": cur["prompts"],
|
|
1024
|
-
"cost": cur["cost"], "cost_per_prompt": cur["cost_per_prompt"],
|
|
1025
|
-
"turns": cur["turns"], "tools": cur["tools"],
|
|
1026
|
-
"repeat_pct": cur["repeat_pct"]},
|
|
1027
|
-
"candidates": cands,
|
|
1028
|
-
"recommended": best["model"] if best else None,
|
|
1029
|
-
"recommended_name": best["name"] if best else None,
|
|
1030
|
-
"estimated_savings_usd": best["estimated_savings_usd"] if best else 0.0,
|
|
1031
|
-
"verdict": best["verdict"] if best else cands[0]["verdict"],
|
|
1032
|
-
"why": best["why"] if best else cands[0]["why"],
|
|
1033
|
-
"rule": rule,
|
|
1034
|
-
})
|
|
1035
|
-
out.sort(key=lambda c: -c["estimated_savings_usd"])
|
|
1036
|
-
return {
|
|
1037
|
-
"categories": out,
|
|
1038
|
-
"estimated_savings_usd": round(total_save, 2),
|
|
1039
|
-
"min_prompts": self.MIN_PROMPTS,
|
|
1040
|
-
"basis": "actual",
|
|
1041
|
-
"method": (f"Compares what each model actually cost per prompt on the same category "
|
|
1042
|
-
f"of work, using only categories where you ran both with at least "
|
|
1043
|
-
f"{self.MIN_PROMPTS} prompts each. Turns and re-asked prompts are shown "
|
|
1044
|
-
f"because a cheaper model that needs more of both is not cheaper. "
|
|
1045
|
-
f"Nothing here is repriced or modelled."),
|
|
1046
|
-
"caveat": ("Your prompts were not randomly assigned to models, so a category can "
|
|
1047
|
-
"differ in difficulty between them. Treat this as strong evidence for a "
|
|
1048
|
-
"trial, not proof."),
|
|
1049
|
-
}
|
|
1050
|
-
|
|
1051
|
-
def model_switch(self, f=None):
|
|
1052
|
-
"""Per-request what-if: reprice each request on the model its work needs.
|
|
1053
|
-
|
|
1054
|
-
Token counts are held constant (actual); costs on both sides are estimated at the
|
|
1055
|
-
configured prices. Requests whose context exceeds the target's window stay put.
|
|
1056
|
-
"""
|
|
1057
|
-
w, p = self.where(f)
|
|
1058
|
-
rows = self.q(f"""SELECT r.model, r.is_sidechain side, r.agent_type,
|
|
1059
|
-
COALESCE(pr.category,'other') category, r.context_tokens ctx,
|
|
1060
|
-
pj.id project_id, pj.name project,
|
|
1061
|
-
r.prompt_id, r.session_id, r.input_tokens i, r.output_tokens o,
|
|
1062
|
-
r.cache_read_tokens cr, r.cache_write_5m c5, r.cache_write_1h c1, r.est_cost_usd cost
|
|
1063
|
-
FROM requests r LEFT JOIN prompts pr ON pr.id=r.prompt_id
|
|
1064
|
-
JOIN projects pj ON pj.id=r.project_id WHERE {w}""", p)
|
|
1065
|
-
target_of = {}
|
|
1066
|
-
agents, proj_prov = set(), {}
|
|
1067
|
-
groups, projects = {}, defaultdict(lambda: defaultdict(float))
|
|
1068
|
-
total = blocked = 0.0
|
|
1069
|
-
blocked_n = 0
|
|
1339
|
+
rows = self.q(f"""SELECT r.model, r.priced_as, r.context_tokens ctx, r.est_cost_usd cost
|
|
1340
|
+
FROM requests r WHERE {w}""", p)
|
|
1341
|
+
per = {}
|
|
1342
|
+
unknown = {"requests": 0, "cost_usd": 0.0}
|
|
1343
|
+
total_n = total_cost = near_cost = over_cost = 0.0
|
|
1344
|
+
near_n = over_n = 0
|
|
1070
1345
|
for r in rows:
|
|
1071
1346
|
cost = r["cost"] or 0.0
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
want, conf = ("economy", "high") if is_explore else ("balanced", "medium")
|
|
1079
|
-
scope = f"Subagent: {r['agent_type'] or 'general'}"
|
|
1080
|
-
else:
|
|
1081
|
-
want, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
1082
|
-
scope = f"Prompts: {r['category'].replace('_', ' ')}"
|
|
1083
|
-
pj = projects[(r["project_id"], r["project"])]
|
|
1084
|
-
pj["cost"] += cost
|
|
1085
|
-
if tier == "frontier":
|
|
1086
|
-
pj["frontier_cost"] += cost
|
|
1087
|
-
proj_prov[(r["project_id"], r["project"])] = \
|
|
1088
|
-
self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
1089
|
-
if want == "keep" or self.TIER_RANK[want] >= self.TIER_RANK[tier]:
|
|
1090
|
-
if want == "keep" and tier == "frontier":
|
|
1091
|
-
pj["keep_cost"] += cost
|
|
1092
|
-
continue
|
|
1093
|
-
prov = self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
1094
|
-
if (want, prov) not in target_of:
|
|
1095
|
-
target_of[(want, prov)] = self._cheapest(want, prov)
|
|
1096
|
-
tgt = target_of[(want, prov)]
|
|
1097
|
-
agents.add(prov)
|
|
1098
|
-
if not tgt:
|
|
1099
|
-
continue
|
|
1100
|
-
win = self.pricing.context_window(tgt) or 0
|
|
1101
|
-
if win and (r["ctx"] or 0) > win * 0.9:
|
|
1102
|
-
blocked += cost
|
|
1103
|
-
blocked_n += 1
|
|
1347
|
+
total_n += 1
|
|
1348
|
+
total_cost += cost
|
|
1349
|
+
win = self.pricing.context_window(r["priced_as"] or r["model"])
|
|
1350
|
+
if not win:
|
|
1351
|
+
unknown["requests"] += 1
|
|
1352
|
+
unknown["cost_usd"] += cost
|
|
1104
1353
|
continue
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
save = g["cost"] - g["alt"]
|
|
1124
|
-
if save < 0.5:
|
|
1125
|
-
continue
|
|
1126
|
-
out.append({**g, "prompts": len(g["prompts"] - {None}), "sessions": len(g["sessions"]),
|
|
1127
|
-
"current_name": self.pricing.display_name(g["current_model"]),
|
|
1128
|
-
"recommended_name": self.pricing.display_name(g["recommended_model"]),
|
|
1129
|
-
"estimated_savings_usd": save,
|
|
1130
|
-
"estimated_savings_pct": round(100 * save / g["cost"], 1) if g["cost"] else 0})
|
|
1131
|
-
out.sort(key=lambda x: -x["estimated_savings_usd"])
|
|
1132
|
-
|
|
1133
|
-
by_conf = defaultdict(float)
|
|
1134
|
-
for g in out:
|
|
1135
|
-
by_conf[g["confidence"]] += g["estimated_savings_usd"]
|
|
1136
|
-
|
|
1137
|
-
proj = []
|
|
1138
|
-
for (pid, name), v in projects.items():
|
|
1139
|
-
if v["frontier_cost"] < 1:
|
|
1140
|
-
continue
|
|
1141
|
-
balanced = self._cheapest("balanced", proj_prov.get((pid, name), "anthropic"))
|
|
1142
|
-
keep_pct = round(100 * v["keep_cost"] / v["frontier_cost"], 1)
|
|
1143
|
-
default = "keep" if keep_pct >= 50 else "switch"
|
|
1144
|
-
proj.append({"project": name, "project_id": pid, "cost": v["cost"],
|
|
1145
|
-
"frontier_cost": v["frontier_cost"], "keep_pct": keep_pct,
|
|
1146
|
-
"estimated_savings_usd": v["savings"],
|
|
1147
|
-
"suggested_default": (self.pricing.display_name(balanced)
|
|
1148
|
-
if default == "switch" and balanced else "Keep current"),
|
|
1149
|
-
"why": (f"{keep_pct}% of frontier spend here is debugging/architecture/"
|
|
1150
|
-
f"planning/review, which benefits from the top model."
|
|
1151
|
-
if default == "keep" else
|
|
1152
|
-
f"Only {keep_pct}% of frontier spend here is reasoning-heavy work. "
|
|
1153
|
-
f"Make {self.pricing.display_name(balanced)} the default and "
|
|
1154
|
-
f"switch up with /model only for hard problems.")})
|
|
1155
|
-
proj.sort(key=lambda x: -x["estimated_savings_usd"])
|
|
1156
|
-
|
|
1354
|
+
m = per.setdefault(r["model"], {
|
|
1355
|
+
"model": r["model"], "display_name": self.pricing.display_name(r["model"]),
|
|
1356
|
+
"context_window": win, "requests": 0, "cost_usd": 0.0,
|
|
1357
|
+
"near_requests": 0, "near_cost_usd": 0.0,
|
|
1358
|
+
"over_requests": 0, "over_cost_usd": 0.0})
|
|
1359
|
+
m["requests"] += 1
|
|
1360
|
+
m["cost_usd"] += cost
|
|
1361
|
+
ctx = r["ctx"] or 0
|
|
1362
|
+
if ctx > win:
|
|
1363
|
+
m["over_requests"] += 1; m["over_cost_usd"] += cost
|
|
1364
|
+
over_n += 1; over_cost += cost
|
|
1365
|
+
elif ctx >= win * self.NEAR_WINDOW_PCT:
|
|
1366
|
+
m["near_requests"] += 1; m["near_cost_usd"] += cost
|
|
1367
|
+
near_n += 1; near_cost += cost
|
|
1368
|
+
out = sorted(per.values(), key=lambda m: -(m["near_cost_usd"] + m["over_cost_usd"]))
|
|
1369
|
+
for m in out:
|
|
1370
|
+
m["near_or_over_cost_pct"] = (round(100.0 * (m["near_cost_usd"] + m["over_cost_usd"])
|
|
1371
|
+
/ m["cost_usd"], 1) if m["cost_usd"] else 0.0)
|
|
1157
1372
|
return {
|
|
1158
|
-
"
|
|
1159
|
-
"
|
|
1160
|
-
"
|
|
1161
|
-
"
|
|
1162
|
-
"
|
|
1163
|
-
"
|
|
1164
|
-
|
|
1165
|
-
"
|
|
1166
|
-
"
|
|
1167
|
-
"project": '"model": "<name>" in <repo>/.claude/settings.json',
|
|
1168
|
-
"subagent": "model: haiku (or sonnet) in the agent's frontmatter in .claude/agents/"},
|
|
1169
|
-
"how_by_agent": self.SWITCH_HOW_BY_AGENT,
|
|
1170
|
-
"providers": sorted(agents),
|
|
1171
|
-
"caveat": "Same token counts repriced on the cheaper model. Output quality and any "
|
|
1172
|
-
"extra turns a cheaper model might need are not modelled. Try it on a "
|
|
1173
|
-
"sample of work before switching everything.",
|
|
1174
|
-
"basis": "recommendation",
|
|
1373
|
+
"threshold_pct": int(self.NEAR_WINDOW_PCT * 100),
|
|
1374
|
+
"models": out,
|
|
1375
|
+
"requests": int(total_n), "cost_usd": total_cost,
|
|
1376
|
+
"near_requests": near_n, "near_cost_usd": near_cost,
|
|
1377
|
+
"over_requests": over_n, "over_cost_usd": over_cost,
|
|
1378
|
+
"near_or_over_cost_pct": (round(100.0 * (near_cost + over_cost) / total_cost, 1)
|
|
1379
|
+
if total_cost else 0.0),
|
|
1380
|
+
"unknown_window": unknown,
|
|
1381
|
+
"basis": "actual",
|
|
1175
1382
|
}
|
|
1176
1383
|
|
|
1177
1384
|
def recommendations(self, f=None):
|
|
1178
1385
|
w, p = self.where(f)
|
|
1179
1386
|
recs = []
|
|
1180
1387
|
|
|
1181
|
-
# Frontier work a cheaper model could have done. Candidates are always from the
|
|
1182
|
-
# same vendor: an agent can only switch within its own family (Codex can't run
|
|
1183
|
-
# Haiku), so mixed frontier spend is split per vendor before anything is compared.
|
|
1184
|
-
# Each recommendation offers the ladder — one step down (balanced) and the floor
|
|
1185
|
-
# (economy) — priced separately, because that trade-off is the user's to make.
|
|
1186
|
-
frontier_by_provider = defaultdict(list)
|
|
1187
|
-
for m, v in self.pricing.models.items():
|
|
1188
|
-
if v.get("tier") == "frontier":
|
|
1189
|
-
frontier_by_provider[v.get("provider", "anthropic")].append(m)
|
|
1190
|
-
|
|
1191
|
-
for prov, models in sorted(frontier_by_provider.items()):
|
|
1192
|
-
cands = []
|
|
1193
|
-
for tier in ("balanced", "economy"):
|
|
1194
|
-
c = self._cheapest(tier, prov)
|
|
1195
|
-
if c and c not in cands:
|
|
1196
|
-
cands.append(c)
|
|
1197
|
-
if not cands:
|
|
1198
|
-
continue # this vendor exposes nothing cheaper to move to
|
|
1199
|
-
ph = ",".join("?" * len(models))
|
|
1200
|
-
rows = self.q(f"""SELECT pr.category, COUNT(DISTINCT pr.id) prompts,
|
|
1201
|
-
GROUP_CONCAT(DISTINCT r.model) mods,
|
|
1202
|
-
SUM(r.est_cost_usd) cost, SUM(r.input_tokens) i,
|
|
1203
|
-
SUM(r.output_tokens) o, SUM(r.cache_read_tokens) cr,
|
|
1204
|
-
SUM(r.cache_write_5m) c5, SUM(r.cache_write_1h) c1
|
|
1205
|
-
FROM prompts pr JOIN requests r ON r.prompt_id=pr.id
|
|
1206
|
-
WHERE {w} AND r.model IN ({ph})
|
|
1207
|
-
GROUP BY pr.category HAVING prompts >= 3 AND cost > 0.5
|
|
1208
|
-
ORDER BY cost DESC""", p + models)
|
|
1209
|
-
for r in rows:
|
|
1210
|
-
# The same rules the Model switch dashboard applies, so the two pages can
|
|
1211
|
-
# never contradict each other: work the rules say to keep on a frontier
|
|
1212
|
-
# model is not offered a downgrade at all.
|
|
1213
|
-
target, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
1214
|
-
if target == "keep":
|
|
1215
|
-
continue
|
|
1216
|
-
alts = []
|
|
1217
|
-
for m in cands:
|
|
1218
|
-
alt = self.pricing.estimate(m, r["i"], r["o"], r["cr"], r["c5"], r["c1"])
|
|
1219
|
-
if alt >= r["cost"] * 0.9:
|
|
1220
|
-
continue # too close to the current cost to be worth the quality risk
|
|
1221
|
-
alts.append({
|
|
1222
|
-
"model": m, "name": self.pricing.display_name(m),
|
|
1223
|
-
"tier": self.pricing.tier(m),
|
|
1224
|
-
"estimated_cost_usd": alt,
|
|
1225
|
-
"estimated_savings_usd": r["cost"] - alt,
|
|
1226
|
-
"estimated_savings_pct": round(100.0 * (r["cost"] - alt) / r["cost"], 1),
|
|
1227
|
-
})
|
|
1228
|
-
if not alts:
|
|
1229
|
-
continue
|
|
1230
|
-
# Safest step first: balanced before economy, so the ladder reads as
|
|
1231
|
-
# increasing saving and increasing risk.
|
|
1232
|
-
alts.sort(key=lambda a: -self.TIER_RANK.get(a["tier"], 0))
|
|
1233
|
-
for a in alts:
|
|
1234
|
-
a["suggested"] = a["tier"] == target
|
|
1235
|
-
# The headline is the tier the rules actually recommend for this kind of
|
|
1236
|
-
# work, not simply the smallest step; the rest stay on offer below it.
|
|
1237
|
-
head = next((a for a in alts if a["suggested"]), alts[0])
|
|
1238
|
-
agent = self.SWITCH_HOW_BY_AGENT.get(prov, {}).get("agent", prov)
|
|
1239
|
-
recs.append({
|
|
1240
|
-
"type": "model_downgrade",
|
|
1241
|
-
"confidence": conf or "low",
|
|
1242
|
-
"title": f"Consider a cheaper {agent} model for '{r['category']}' work",
|
|
1243
|
-
"current_model": ", ".join(self.pricing.display_name(m)
|
|
1244
|
-
for m in (r["mods"] or "").split(",") if m),
|
|
1245
|
-
"recommended_model": head["name"],
|
|
1246
|
-
"provider": prov,
|
|
1247
|
-
"agent": agent,
|
|
1248
|
-
"alternatives": alts,
|
|
1249
|
-
"scope": f"{r['prompts']} prompts categorized as {r['category']}",
|
|
1250
|
-
"actual_cost_usd": r["cost"],
|
|
1251
|
-
"estimated_alternative_cost_usd": head["estimated_cost_usd"],
|
|
1252
|
-
"estimated_savings_usd": head["estimated_savings_usd"],
|
|
1253
|
-
"estimated_savings_pct": head["estimated_savings_pct"],
|
|
1254
|
-
"caveat": f"Both options stay inside {agent}, so this is a setting change, not "
|
|
1255
|
-
"a change of agent. Assumes identical token usage on the cheaper "
|
|
1256
|
-
"model. Output quality is not modelled — validate on a sample "
|
|
1257
|
-
"before switching.",
|
|
1258
|
-
"basis": "recommendation",
|
|
1259
|
-
})
|
|
1260
|
-
# Biggest opportunity first, now that several vendors can each contribute one.
|
|
1261
|
-
recs.sort(key=lambda r: -r["estimated_savings_usd"])
|
|
1262
1388
|
|
|
1389
|
+
# What follows are observations, not priced savings. The cache item used to
|
|
1390
|
+
# carry the no-cache counterfactual as a "saving" (tens of thousands of dollars
|
|
1391
|
+
# on a bill a fraction of that) and the context item multiplied its spend by a
|
|
1392
|
+
# guessed 20%. Neither number was something the method could support, so
|
|
1393
|
+
# neither is shown; what is observable is.
|
|
1263
1394
|
eff = self.efficiency(f)
|
|
1264
1395
|
c = eff["cache"]
|
|
1265
|
-
|
|
1396
|
+
split = c.get("cost_split") or {}
|
|
1397
|
+
if c["reads"] and split.get("read_cost_share"):
|
|
1266
1398
|
recs.append({
|
|
1267
|
-
"type": "cache_working", "confidence": "
|
|
1268
|
-
"title": "Prompt caching is
|
|
1399
|
+
"type": "cache_working", "confidence": "observed",
|
|
1400
|
+
"title": "Prompt caching is doing its job — keep sessions long-lived",
|
|
1401
|
+
"detail": (f"Reads are {split['read_cost_share']*100:.0f}% of your cache cost "
|
|
1402
|
+
f"({eff['cache_hit_ratio']*100:.0f}% of cache tokens). Restarting "
|
|
1403
|
+
f"sessions throws that prefix away and pays to write it again."),
|
|
1269
1404
|
"actual_cost_usd": c["cost_with_cache"],
|
|
1270
|
-
"estimated_alternative_cost_usd":
|
|
1271
|
-
"estimated_savings_usd":
|
|
1272
|
-
"
|
|
1273
|
-
|
|
1274
|
-
"basis": "
|
|
1405
|
+
"estimated_alternative_cost_usd": None,
|
|
1406
|
+
"estimated_savings_usd": None, "estimated_savings_pct": None,
|
|
1407
|
+
"caveat": "No saving is claimed: what an uncached run would have cost is a "
|
|
1408
|
+
"counterfactual, not money you avoided.",
|
|
1409
|
+
"basis": "actual",
|
|
1275
1410
|
})
|
|
1276
1411
|
|
|
1277
1412
|
ctx = self.context_analysis(f)
|
|
1278
1413
|
if ctx["large_context_cost_pct"] > 15:
|
|
1279
1414
|
recs.append({
|
|
1280
|
-
"type": "context_reduction", "confidence": "
|
|
1415
|
+
"type": "context_reduction", "confidence": "observed",
|
|
1281
1416
|
"title": f"{ctx['large_context_cost_pct']}% of spend comes from >"
|
|
1282
1417
|
f"{ctx['threshold']//1000}K-context requests",
|
|
1418
|
+
"detail": (f"{ctx['large_context_requests']:,} requests re-sent a large prefix "
|
|
1419
|
+
f"on every turn. /compact or a fresh session resets it; how much "
|
|
1420
|
+
f"that would have saved depends on what the work needed, and is "
|
|
1421
|
+
f"not estimated here."),
|
|
1283
1422
|
"scope": f"{ctx['large_context_requests']:,} requests",
|
|
1284
1423
|
"actual_cost_usd": ctx["large_context_cost"],
|
|
1285
|
-
"
|
|
1286
|
-
"estimated_savings_pct":
|
|
1287
|
-
"caveat": "
|
|
1288
|
-
|
|
1289
|
-
"basis": "recommendation",
|
|
1424
|
+
"estimated_alternative_cost_usd": None,
|
|
1425
|
+
"estimated_savings_usd": None, "estimated_savings_pct": None,
|
|
1426
|
+
"caveat": "Observed share of spend. No reduction is assumed.",
|
|
1427
|
+
"basis": "actual",
|
|
1290
1428
|
})
|
|
1291
|
-
recs.sort(key=lambda r: -(r.get("
|
|
1292
|
-
return {"recommendations": recs,
|
|
1293
|
-
"total_estimated_savings_usd": sum(r.get("estimated_savings_usd") or 0
|
|
1294
|
-
for r in recs if r["type"] != "cache_working"),
|
|
1295
|
-
"basis": "recommendation"}
|
|
1429
|
+
recs.sort(key=lambda r: -(r.get("actual_cost_usd") or 0))
|
|
1430
|
+
return {"recommendations": recs, "basis": "actual"}
|
|
1296
1431
|
|
|
1297
1432
|
# ---------------- forecast ----------------
|
|
1298
1433
|
def forecast(self, f=None):
|
|
@@ -1302,42 +1437,46 @@ class Analytics:
|
|
|
1302
1437
|
FROM requests r WHERE {w} AND r.day <> '' GROUP BY 1 ORDER BY 1""", p)
|
|
1303
1438
|
if not rows:
|
|
1304
1439
|
return {"available": False, "message": "No usage in the selected range."}
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1440
|
+
# calendar days, idle days as zero, excluding today (still partial) — the rate
|
|
1441
|
+
# below is multiplied by calendar days remaining, so a per-active-day mean
|
|
1442
|
+
# would overstate every scenario and a partial today would understate it
|
|
1443
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
1444
|
+
recent = [r for r in self.daily_series(f, days=14, end=yesterday)] # complete days only
|
|
1445
|
+
priced = [r["cost"] for r in recent]
|
|
1446
|
+
sample_days = sum(1 for c in priced if c > 0)
|
|
1447
|
+
mean = statistics.fmean(priced) if priced else 0.0
|
|
1448
|
+
sd = statistics.pstdev(priced) if len(priced) > 1 else 0.0
|
|
1309
1449
|
in_period = [r for r in rows if bp["start"] <= r["day"] <= bp["end"]]
|
|
1310
1450
|
used = sum(r["cost"] for r in in_period)
|
|
1311
1451
|
used_tok = sum(r["tokens"] for r in in_period)
|
|
1312
1452
|
left = bp["remaining_days"]
|
|
1453
|
+
insufficient = sample_days < 7
|
|
1313
1454
|
|
|
1314
|
-
def band(rate):
|
|
1315
|
-
|
|
1455
|
+
def band(rate, spread=0.0):
|
|
1456
|
+
# spend on different days is treated as independent, so the spread of a
|
|
1457
|
+
# sum over `left` days grows with sqrt(left), not left
|
|
1458
|
+
return {"daily_rate": rate,
|
|
1459
|
+
"end_of_period_cost": used + rate * left + spread * (left ** 0.5)}
|
|
1316
1460
|
|
|
1317
|
-
scenarios = {
|
|
1318
|
-
|
|
1319
|
-
"
|
|
1320
|
-
"high"
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
today_rows = [r for r in rows if r["day"] == bp["today"]]
|
|
1324
|
-
hours = max(datetime.now(timezone.utc).hour, 1)
|
|
1325
|
-
eod = (today_rows[0]["cost"] / hours * 24) if today_rows else mean
|
|
1461
|
+
scenarios = {"expected": band(mean)}
|
|
1462
|
+
if not insufficient:
|
|
1463
|
+
scenarios["conservative"] = band(mean, -sd)
|
|
1464
|
+
scenarios["high"] = band(mean, sd)
|
|
1465
|
+
scenarios["conservative"]["end_of_period_cost"] = max(
|
|
1466
|
+
scenarios["conservative"]["end_of_period_cost"], used)
|
|
1326
1467
|
|
|
1327
|
-
|
|
1328
|
-
wk_used = sum(r["cost"] for r in rows if r["day"] >= wk_start)
|
|
1329
|
-
wk_left = 6 - _d(bp["today"]).weekday()
|
|
1468
|
+
tok_mean = statistics.fmean([r["tokens"] for r in recent]) if recent else 0.0
|
|
1330
1469
|
|
|
1331
1470
|
out = {
|
|
1332
1471
|
"available": True,
|
|
1333
|
-
"method": "14
|
|
1334
|
-
|
|
1472
|
+
"method": ("mean of the last 14 complete calendar days (idle days as zero); "
|
|
1473
|
+
"bands are ±1 sd × sqrt(days remaining)"),
|
|
1474
|
+
"sample_days": sample_days,
|
|
1475
|
+
"insufficient_history": insufficient,
|
|
1335
1476
|
"daily_mean": mean, "daily_stdev": sd,
|
|
1336
1477
|
"period_used": used, "period_used_tokens": used_tok,
|
|
1337
1478
|
"remaining_days": left,
|
|
1338
1479
|
"scenarios": scenarios,
|
|
1339
|
-
"end_of_day_cost": eod,
|
|
1340
|
-
"end_of_week_cost": wk_used + mean * max(wk_left, 0),
|
|
1341
1480
|
"end_of_period_tokens": used_tok + tok_mean * left,
|
|
1342
1481
|
"estimated_monthly_cost": used + mean * left,
|
|
1343
1482
|
"basis": "forecast",
|
|
@@ -1409,34 +1548,55 @@ class Analytics:
|
|
|
1409
1548
|
w, p = self.where(f)
|
|
1410
1549
|
cfg = self.settings["anomaly"]
|
|
1411
1550
|
found = []
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
if len(
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1551
|
+
yesterday = (self.today() - timedelta(days=1)).isoformat()
|
|
1552
|
+
series = [d for d in self.daily_series(dict(f or {}, agents=["claude"]), end=yesterday)]
|
|
1553
|
+
priced = [d for d in series if d["cost"] > 0]
|
|
1554
|
+
if len(priced) >= 14:
|
|
1555
|
+
# Median/MAD, not mean/stdev: unpriced $0 days from agents without pricing
|
|
1556
|
+
# data (e.g. Cursor) would otherwise pollute the mean/sd baseline and
|
|
1557
|
+
# either mask real spikes or manufacture fake ones. MAD is scaled by
|
|
1558
|
+
# 1.4826 so it estimates the same thing a standard deviation would under
|
|
1559
|
+
# a normal distribution, without a few extreme days inflating it the way
|
|
1560
|
+
# a real stdev would.
|
|
1561
|
+
vals = [d["cost"] for d in priced]
|
|
1562
|
+
med = statistics.median(vals)
|
|
1563
|
+
mad = statistics.median(abs(v - med) for v in vals) * 1.4826 or 1e-9
|
|
1564
|
+
for d in priced:
|
|
1565
|
+
score = (d["cost"] - med) / mad
|
|
1566
|
+
ratio = d["cost"] / med if med else 0
|
|
1567
|
+
if score >= cfg.get("daily_robust_z", 3.5) and ratio >= cfg["daily_ratio"]:
|
|
1422
1568
|
found.append({
|
|
1423
1569
|
"severity": "high", "type": "daily_spike", "date": d["day"],
|
|
1424
1570
|
"title": f"{d['day']} spend was {ratio:.1f}x your daily average",
|
|
1425
|
-
"detail": f"${d['cost']:,.2f} vs a ${
|
|
1426
|
-
"metric_value": d["cost"], "baseline":
|
|
1571
|
+
"detail": f"${d['cost']:,.2f} vs a ${med:,.2f} median priced day (robust z={score:.1f}).",
|
|
1572
|
+
"metric_value": d["cost"], "baseline": med, "ratio": round(ratio, 2),
|
|
1427
1573
|
"drilldown": {"filter": {"start": d["day"], "end": d["day"]}},
|
|
1428
1574
|
"basis": "estimated",
|
|
1429
1575
|
})
|
|
1430
1576
|
sess = self.sessions(f, limit=100000, order="cost")
|
|
1431
1577
|
if len(sess) >= 5:
|
|
1432
|
-
|
|
1433
|
-
mean
|
|
1434
|
-
|
|
1578
|
+
# Median, not mean: session token counts are heavily right-skewed, and a
|
|
1579
|
+
# mean lets the outliers inflate the very baseline they are measured
|
|
1580
|
+
# against — which understates how far out they really are. Candidates are
|
|
1581
|
+
# ranked by tokens too, since that is the metric being tested; ordering by
|
|
1582
|
+
# cost hid token-heavy work on cheap models.
|
|
1583
|
+
# Sessions with no token data (Cursor transcripts don't always carry it)
|
|
1584
|
+
# are not comparable and would drag the baseline down.
|
|
1585
|
+
vals = [s["tokens"] for s in sess if (s["tokens"] or 0) > 0]
|
|
1586
|
+
mean = (statistics.median(vals) if vals else 0) or 1
|
|
1587
|
+
# Session sizes are heavy-tailed enough that any fixed multiple of the
|
|
1588
|
+
# baseline still matches a fifth of them, so the threshold alone cannot
|
|
1589
|
+
# keep this list short. Take the most extreme few and leave room for the
|
|
1590
|
+
# other anomaly types, which have much smaller ratios and would otherwise
|
|
1591
|
+
# be sorted off the end of the list.
|
|
1592
|
+
outliers = 0
|
|
1593
|
+
for s in sorted(sess, key=lambda x: -(x["tokens"] or 0))[:40]:
|
|
1435
1594
|
ratio = s["tokens"] / mean
|
|
1436
|
-
if ratio >= cfg["session_ratio"]:
|
|
1595
|
+
if ratio >= cfg["session_ratio"] and outliers < cfg.get("max_session_outliers", 5):
|
|
1596
|
+
outliers += 1
|
|
1437
1597
|
found.append({
|
|
1438
|
-
"severity": "
|
|
1439
|
-
"title": f"
|
|
1598
|
+
"severity": "low", "type": "session_outlier",
|
|
1599
|
+
"title": f"Among your largest sessions: {ratio:.1f}x the median",
|
|
1440
1600
|
"detail": f"{s['title'] or s['session_id'][:8]} — {s['tokens']:,} tokens, "
|
|
1441
1601
|
f"${s['cost']:,.2f} in {s['project']}.",
|
|
1442
1602
|
"metric_value": s["tokens"], "baseline": mean, "ratio": round(ratio, 2),
|
|
@@ -1445,7 +1605,7 @@ class Analytics:
|
|
|
1445
1605
|
})
|
|
1446
1606
|
# week-over-week model shift
|
|
1447
1607
|
if self.last_day:
|
|
1448
|
-
end =
|
|
1608
|
+
end = self.today() - timedelta(days=1) # anchor on yesterday, not last_day
|
|
1449
1609
|
cur_s = (end - timedelta(days=6)).isoformat()
|
|
1450
1610
|
prev_s, prev_e = (end - timedelta(days=13)).isoformat(), (end - timedelta(days=7)).isoformat()
|
|
1451
1611
|
for m in self.q(f"SELECT DISTINCT r.model FROM requests r WHERE {w}", p):
|
|
@@ -1471,46 +1631,27 @@ class Analytics:
|
|
|
1471
1631
|
# ---------------- scorecard ----------------
|
|
1472
1632
|
def scorecard(self, f=None):
|
|
1473
1633
|
eff = self.efficiency(f)
|
|
1474
|
-
ctx = self.context_analysis(f)
|
|
1475
1634
|
wst = self.waste(f)
|
|
1476
1635
|
bud = self.budgets(f)
|
|
1477
|
-
mdl = self.models(f)
|
|
1478
1636
|
dims = []
|
|
1479
1637
|
|
|
1480
1638
|
def dim(name, score, detail, weight=1.0):
|
|
1481
1639
|
dims.append({"name": name, "score": max(0, min(100, round(score))),
|
|
1482
1640
|
"detail": detail, "weight": weight})
|
|
1483
1641
|
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1642
|
+
hy = self.hygiene(f)
|
|
1643
|
+
thr = max(int(t) for t in hy["above"])
|
|
1644
|
+
share = hy["above"][str(thr)]["share_pct"]
|
|
1645
|
+
dim("Context share", 100 - share,
|
|
1646
|
+
f"{share:.0f}% of spend ran above {thr//1000}K context.", 1.0)
|
|
1647
|
+
|
|
1648
|
+
margin = eff["cache"].get("breakeven_margin")
|
|
1649
|
+
if margin is None:
|
|
1650
|
+
dim("Cache break-even", 50, "No cache activity in range", 1.0)
|
|
1487
1651
|
else:
|
|
1488
|
-
dim("Cache
|
|
1489
|
-
f"{
|
|
1490
|
-
|
|
1491
|
-
sc_cfg = self.settings.get("scorecard", {})
|
|
1492
|
-
target = sc_cfg.get("target_output_ratio", 0.0088)
|
|
1493
|
-
outr = eff["output_ratio"]
|
|
1494
|
-
dim("Token efficiency", min(outr / target, 1.0) * 100,
|
|
1495
|
-
f"Output is {outr*100:.2f}% of billable tokens against a "
|
|
1496
|
-
f"{target*100:.2f}% reference.", 1.2)
|
|
1497
|
-
|
|
1498
|
-
big_pct = ctx["large_context_cost_pct"]
|
|
1499
|
-
dim("Context efficiency", 100 - big_pct,
|
|
1500
|
-
f"{big_pct}% of spend came from requests above "
|
|
1501
|
-
f"{ctx['threshold']//1000}K context.", 1.0)
|
|
1502
|
-
|
|
1503
|
-
excess = wst["excess_pct"]
|
|
1504
|
-
dim("Waste control", 100 - min(excess, 100),
|
|
1505
|
-
f"{excess}% of spend is estimated excess over a reasonable baseline "
|
|
1506
|
-
f"({wst['exposed_pct']}% of spend sits in items a rule touched).", 1.3)
|
|
1507
|
-
|
|
1508
|
-
priced = [r for r in mdl["rows"] if r["tier"] != "none" and r["cost"]]
|
|
1509
|
-
frontier_pct = (100.0 * sum(r["cost"] for r in priced if r["tier"] == "frontier")
|
|
1510
|
-
/ (sum(r["cost"] for r in priced) or 1))
|
|
1511
|
-
allow = sc_cfg.get("frontier_cost_share_allowance_pct", 40)
|
|
1512
|
-
dim("Model selection", 100 - max(frontier_pct - allow, 0) * 1.5,
|
|
1513
|
-
f"{frontier_pct:.0f}% of spend is on frontier-tier models.", 1.1)
|
|
1652
|
+
dim("Cache break-even", 50 + margin * 50,
|
|
1653
|
+
f"Caching returned {margin*100:.0f}% of its cost as read discount net of "
|
|
1654
|
+
"write premium.", 1.0)
|
|
1514
1655
|
|
|
1515
1656
|
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1516
1657
|
if ml and ml.get("configured"):
|
|
@@ -1519,12 +1660,7 @@ class Analytics:
|
|
|
1519
1660
|
f"Forecast is {fp:.0f}% of the configured monthly budget.", 1.3)
|
|
1520
1661
|
else:
|
|
1521
1662
|
dim("Budget adherence", 50,
|
|
1522
|
-
"No monthly budget configured — set one in config/settings.json to be
|
|
1523
|
-
|
|
1524
|
-
cpo = eff["cost_per_1k_output"]
|
|
1525
|
-
cpo_target = sc_cfg.get("target_cost_per_1k_output_usd", 0.30)
|
|
1526
|
-
dim("Cost efficiency", 100 - min(cpo / (cpo_target * 2) * 100, 100),
|
|
1527
|
-
f"${cpo:.3f} estimated per 1K output tokens.", 1.0)
|
|
1663
|
+
"No monthly budget configured — set one in config/settings.json to be measured.", 0.4)
|
|
1528
1664
|
|
|
1529
1665
|
tw = sum(d["weight"] for d in dims)
|
|
1530
1666
|
total = round(sum(d["score"] * d["weight"] for d in dims) / tw)
|
|
@@ -1532,8 +1668,7 @@ class Analytics:
|
|
|
1532
1668
|
weak = sorted(dims, key=lambda d: d["score"])[:3]
|
|
1533
1669
|
top = wst["findings"][0] if wst["findings"] else None
|
|
1534
1670
|
return {
|
|
1535
|
-
"score": total,
|
|
1536
|
-
else "C" if total >= 55 else "D" if total >= 40 else "F"),
|
|
1671
|
+
"score": total,
|
|
1537
1672
|
"dimensions": dims,
|
|
1538
1673
|
"what_is_good": [f"{d['name']}: {d['detail']}" for d in strong if d["score"] >= 60],
|
|
1539
1674
|
"needs_attention": [f"{d['name']}: {d['detail']}" for d in weak if d["score"] < 70],
|
|
@@ -1570,16 +1705,9 @@ class Analytics:
|
|
|
1570
1705
|
for r in recs["recommendations"][:2]:
|
|
1571
1706
|
if r["type"] == "cache_working":
|
|
1572
1707
|
continue
|
|
1573
|
-
# Name the models. A saving is meaningless without the swap it assumes, and
|
|
1574
|
-
# the options are what the reader actually has to choose between.
|
|
1575
|
-
opts = " or ".join(f"{a['name']} (~${a['estimated_savings_usd']:,.0f}, "
|
|
1576
|
-
f"{a['estimated_savings_pct']}%)" for a in r.get("alternatives", []))
|
|
1577
|
-
swap = f"{r['current_model']} → {opts}. " if opts else ""
|
|
1578
1708
|
actions.append({"priority": 3, "kind": "recommendation", "text": r["title"],
|
|
1579
|
-
"detail":
|
|
1580
|
-
|
|
1581
|
-
f"{r['caveat']}",
|
|
1582
|
-
"basis": "recommendation"})
|
|
1709
|
+
"detail": r.get("detail") or r.get("caveat") or "",
|
|
1710
|
+
"basis": r.get("basis", "recommendation")})
|
|
1583
1711
|
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1584
1712
|
if ml and ml.get("configured") and ml.get("forecast_pct"):
|
|
1585
1713
|
if ml["forecast_pct"] >= 90:
|
|
@@ -1599,64 +1727,15 @@ class Analytics:
|
|
|
1599
1727
|
"detail": "; ".join(f"\"{x['preview'][:60]}…\" (${x['pcost']:,.2f})" for x in pr),
|
|
1600
1728
|
"basis": "estimated"})
|
|
1601
1729
|
actions.sort(key=lambda a: a["priority"])
|
|
1602
|
-
|
|
1730
|
+
# No "savings opportunity" range: the old one was a guessed 20% of large-context
|
|
1731
|
+
# spend, then 0.6x of that for a low end. Neither factor came from the data.
|
|
1603
1732
|
return {
|
|
1604
1733
|
"question": "What should I do today?",
|
|
1605
1734
|
"actions": actions[:6],
|
|
1606
|
-
"estimated_savings_range_usd": [round(savings * 0.6, 2), round(savings, 2)],
|
|
1607
1735
|
"generated_from": "Live dashboard data for the current filter selection.",
|
|
1608
1736
|
"basis": "mixed: see per-item basis",
|
|
1609
1737
|
}
|
|
1610
1738
|
|
|
1611
|
-
# ---------------- prompt-level advisor ----------------
|
|
1612
|
-
def prompt_advisor(self, p):
|
|
1613
|
-
"""Deterministic, evidence-based analysis of one prompt. Estimates only."""
|
|
1614
|
-
reasons, suggestions = [], []
|
|
1615
|
-
chars = p.get("char_len") or 0
|
|
1616
|
-
ctx = p.get("max_context_tokens") or 0
|
|
1617
|
-
tools = p.get("tool_calls") or 0
|
|
1618
|
-
out = p.get("output_tokens") or 0
|
|
1619
|
-
tot = p.get("billable_tokens") or 0
|
|
1620
|
-
reduction = 0.0
|
|
1621
|
-
if chars > 4000:
|
|
1622
|
-
reasons.append(f"The prompt itself is {chars:,} characters, which is cached and "
|
|
1623
|
-
f"re-sent on every follow-up turn.")
|
|
1624
|
-
suggestions.append("Move long pasted content into a file and reference the path.")
|
|
1625
|
-
reduction += 0.10
|
|
1626
|
-
if ctx > 150000:
|
|
1627
|
-
reasons.append(f"It ran with up to {ctx:,} context tokens per request.")
|
|
1628
|
-
suggestions.append("Run /compact or start a fresh session before a task this large.")
|
|
1629
|
-
reduction += 0.25
|
|
1630
|
-
if tools > 40:
|
|
1631
|
-
reasons.append(f"It triggered {tools} tool calls; each one re-sends the conversation.")
|
|
1632
|
-
suggestions.append("Split into smaller, explicitly scoped sub-tasks.")
|
|
1633
|
-
reduction += 0.15
|
|
1634
|
-
if tot and out / tot < 0.01:
|
|
1635
|
-
reasons.append(f"Only {100*out/tot:.2f}% of the tokens were output — most of the "
|
|
1636
|
-
f"cost was re-reading context.")
|
|
1637
|
-
suggestions.append("Narrow the files and history in scope before asking.")
|
|
1638
|
-
reduction += 0.10
|
|
1639
|
-
models = (p.get("models") or "")
|
|
1640
|
-
if "opus" in models and out < 400:
|
|
1641
|
-
reasons.append("A frontier-tier model produced a short answer.")
|
|
1642
|
-
suggestions.append("Route short turns to a cheaper model tier.")
|
|
1643
|
-
reduction += 0.20
|
|
1644
|
-
if not reasons:
|
|
1645
|
-
return {"available": False,
|
|
1646
|
-
"message": "No cost-driver pattern detected for this prompt."}
|
|
1647
|
-
reduction = min(reduction, 0.6)
|
|
1648
|
-
return {
|
|
1649
|
-
"available": True,
|
|
1650
|
-
"why_expensive": reasons,
|
|
1651
|
-
"suggestions": suggestions,
|
|
1652
|
-
"estimated_token_reduction_pct": round(reduction * 100),
|
|
1653
|
-
"estimated_cost_reduction_pct": round(reduction * 100 * 0.85),
|
|
1654
|
-
"estimated_cost_reduction_usd": round((p.get("est_cost_usd") or 0) * reduction * 0.85, 2),
|
|
1655
|
-
"disclaimer": "ESTIMATE from structural heuristics. Not a measured saving and not a "
|
|
1656
|
-
"guarantee of equivalent output quality.",
|
|
1657
|
-
"basis": "recommendation",
|
|
1658
|
-
}
|
|
1659
|
-
|
|
1660
1739
|
# ---------------- claude code / developer ----------------
|
|
1661
1740
|
def developer(self, f=None):
|
|
1662
1741
|
w, p = self.where(f)
|