claude-finops 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +361 -0
- package/bin/claude-finops.js +74 -0
- package/config/free_models.json +49 -0
- package/config/model_compare.json +13 -0
- package/config/pricing.json +232 -0
- package/config/settings.json +57 -0
- package/finops/__init__.py +0 -0
- package/finops/actions.py +671 -0
- package/finops/agents.py +296 -0
- package/finops/analytics.py +1579 -0
- package/finops/api.py +426 -0
- package/finops/classify.py +48 -0
- package/finops/cloud.py +336 -0
- package/finops/diagnose.py +1002 -0
- package/finops/etl.py +533 -0
- package/finops/paths.py +111 -0
- package/finops/playbook.py +229 -0
- package/finops/pricing.py +74 -0
- package/finops/procs.py +499 -0
- package/finops/report.py +169 -0
- package/package.json +52 -0
- package/run.cmd +4 -0
- package/run.py +210 -0
- package/run.sh +3 -0
- package/web/app.js +2923 -0
- package/web/charts.js +340 -0
- package/web/index.html +14 -0
- package/web/styles.css +387 -0
|
@@ -0,0 +1,1579 @@
|
|
|
1
|
+
"""FinOps analytics over the normalized warehouse.
|
|
2
|
+
|
|
3
|
+
Every number returned is tagged with a `basis`:
|
|
4
|
+
actual - read directly from the transcripts
|
|
5
|
+
estimated - derived from token counts x configurable pricing
|
|
6
|
+
forecast - projected from historical usage
|
|
7
|
+
recommendation - suggested action, never a booked saving
|
|
8
|
+
"""
|
|
9
|
+
import json
|
|
10
|
+
import math
|
|
11
|
+
import os
|
|
12
|
+
import sqlite3
|
|
13
|
+
import statistics
|
|
14
|
+
from collections import Counter, defaultdict
|
|
15
|
+
from datetime import date, datetime, timedelta, timezone
|
|
16
|
+
|
|
17
|
+
from .pricing import Pricing
|
|
18
|
+
|
|
19
|
+
from .paths import ROOT, DB_PATH, SETTINGS_PATH, LOCAL_SETTINGS_PATH
|
|
20
|
+
|
|
21
|
+
UNAVAILABLE = "Unavailable from connected Claude data"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _merge(base, over):
|
|
25
|
+
for k, v in over.items():
|
|
26
|
+
if isinstance(v, dict) and isinstance(base.get(k), dict):
|
|
27
|
+
_merge(base[k], v)
|
|
28
|
+
else:
|
|
29
|
+
base[k] = v
|
|
30
|
+
return base
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def detect_account():
|
|
34
|
+
"""The signed-in Claude Code account, read from ~/.claude.json (actual, not guessed)."""
|
|
35
|
+
try:
|
|
36
|
+
with open(os.path.expanduser("~/.claude.json")) as fh:
|
|
37
|
+
acct = json.load(fh).get("oauthAccount") or {}
|
|
38
|
+
except (OSError, ValueError):
|
|
39
|
+
return {}
|
|
40
|
+
return {"label": acct.get("emailAddress")} if acct.get("emailAddress") else {}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def load_settings():
|
|
44
|
+
"""Shared defaults (settings.json) + this machine's overrides (settings.local.json)."""
|
|
45
|
+
with open(SETTINGS_PATH) as fh:
|
|
46
|
+
cur = json.load(fh)
|
|
47
|
+
if not cur.get("account", {}).get("label"):
|
|
48
|
+
cur.setdefault("account", {}).update(detect_account())
|
|
49
|
+
if os.path.exists(LOCAL_SETTINGS_PATH):
|
|
50
|
+
with open(LOCAL_SETTINGS_PATH) as fh:
|
|
51
|
+
_merge(cur, json.load(fh))
|
|
52
|
+
return cur
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _d(s):
|
|
56
|
+
return datetime.strptime(s, "%Y-%m-%d").date()
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Analytics:
|
|
60
|
+
def __init__(self, db_path=DB_PATH):
|
|
61
|
+
self.db = sqlite3.connect(db_path, check_same_thread=False)
|
|
62
|
+
self.db.row_factory = sqlite3.Row
|
|
63
|
+
self.pricing = Pricing()
|
|
64
|
+
self.settings = load_settings()
|
|
65
|
+
self.meta = {r["key"]: r["value"] for r in self.db.execute("SELECT * FROM meta")}
|
|
66
|
+
row = self.db.execute("SELECT MIN(day) a, MAX(day) b FROM requests WHERE day<>''").fetchone()
|
|
67
|
+
self.first_day, self.last_day = row["a"], row["b"]
|
|
68
|
+
|
|
69
|
+
def q(self, sql, params=()):
|
|
70
|
+
return [dict(r) for r in self.db.execute(sql, params)]
|
|
71
|
+
|
|
72
|
+
def one(self, sql, params=()):
|
|
73
|
+
r = self.db.execute(sql, params).fetchone()
|
|
74
|
+
return dict(r) if r else {}
|
|
75
|
+
|
|
76
|
+
# ---------------- filters ----------------
|
|
77
|
+
def where(self, f):
|
|
78
|
+
"""Build a SQL WHERE fragment from the global filter object."""
|
|
79
|
+
cl, p = [], []
|
|
80
|
+
f = f or {}
|
|
81
|
+
if f.get("start"):
|
|
82
|
+
cl.append("r.day >= ?"); p.append(f["start"])
|
|
83
|
+
if f.get("end"):
|
|
84
|
+
cl.append("r.day <= ?"); p.append(f["end"])
|
|
85
|
+
if f.get("agents"):
|
|
86
|
+
cl.append("r.agent IN (%s)" % ",".join("?" * len(f["agents"]))); p += f["agents"]
|
|
87
|
+
if f.get("models"):
|
|
88
|
+
cl.append("r.model IN (%s)" % ",".join("?" * len(f["models"]))); p += f["models"]
|
|
89
|
+
if f.get("projects"):
|
|
90
|
+
cl.append("r.project_id IN (%s)" % ",".join("?" * len(f["projects"])))
|
|
91
|
+
p += [int(x) for x in f["projects"]]
|
|
92
|
+
if f.get("sessions"):
|
|
93
|
+
cl.append("r.session_id IN (%s)" % ",".join("?" * len(f["sessions"]))); p += f["sessions"]
|
|
94
|
+
if f.get("categories"):
|
|
95
|
+
cl.append("r.prompt_id IN (SELECT id FROM prompts WHERE category IN (%s))"
|
|
96
|
+
% ",".join("?" * len(f["categories"]))); p += f["categories"]
|
|
97
|
+
if not f.get("include_sandbox", True):
|
|
98
|
+
cl.append("r.project_id IN (SELECT id FROM projects WHERE is_sandbox=0)")
|
|
99
|
+
if f.get("min_cost") not in (None, ""):
|
|
100
|
+
cl.append("r.est_cost_usd >= ?"); p.append(float(f["min_cost"]))
|
|
101
|
+
if f.get("max_cost") not in (None, ""):
|
|
102
|
+
cl.append("r.est_cost_usd <= ?"); p.append(float(f["max_cost"]))
|
|
103
|
+
if f.get("min_tokens") not in (None, ""):
|
|
104
|
+
cl.append("r.billable_tokens >= ?"); p.append(int(f["min_tokens"]))
|
|
105
|
+
return (" AND ".join(cl) if cl else "1=1"), p
|
|
106
|
+
|
|
107
|
+
# ---------------- billing period ----------------
|
|
108
|
+
def billing_period(self, today=None):
|
|
109
|
+
bp = self.settings["billing_period"]
|
|
110
|
+
today = today or (_d(self.last_day) if self.last_day else date.today())
|
|
111
|
+
anchor = int(bp.get("anchor_day", 1))
|
|
112
|
+
if today.day >= anchor:
|
|
113
|
+
start = today.replace(day=min(anchor, 28))
|
|
114
|
+
else:
|
|
115
|
+
prev = today.replace(day=1) - timedelta(days=1)
|
|
116
|
+
start = prev.replace(day=min(anchor, 28))
|
|
117
|
+
nxt = (start.replace(day=28) + timedelta(days=8)).replace(day=min(anchor, 28))
|
|
118
|
+
end = nxt - timedelta(days=1)
|
|
119
|
+
total = (end - start).days + 1
|
|
120
|
+
elapsed = (today - start).days + 1
|
|
121
|
+
return {
|
|
122
|
+
"start": start.isoformat(), "end": end.isoformat(), "today": today.isoformat(),
|
|
123
|
+
"total_days": total, "elapsed_days": elapsed,
|
|
124
|
+
"remaining_days": max(total - elapsed, 0),
|
|
125
|
+
"pct_elapsed": round(100.0 * elapsed / total, 1),
|
|
126
|
+
"basis": "actual",
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
# ---------------- overview ----------------
|
|
130
|
+
def overview(self, f=None):
|
|
131
|
+
w, p = self.where(f)
|
|
132
|
+
tot = self.one(f"""
|
|
133
|
+
SELECT COUNT(*) requests, COUNT(DISTINCT r.session_id) sessions,
|
|
134
|
+
COUNT(DISTINCT r.project_id) projects, COUNT(DISTINCT r.day) active_days,
|
|
135
|
+
COALESCE(SUM(r.input_tokens),0) input_tokens,
|
|
136
|
+
COALESCE(SUM(r.output_tokens),0) output_tokens,
|
|
137
|
+
COALESCE(SUM(r.thinking_tokens),0) thinking_tokens,
|
|
138
|
+
COALESCE(SUM(r.cache_read_tokens),0) cache_read_tokens,
|
|
139
|
+
COALESCE(SUM(r.cache_write_tokens),0) cache_write_tokens,
|
|
140
|
+
COALESCE(SUM(r.billable_tokens),0) billable_tokens,
|
|
141
|
+
COALESCE(SUM(r.est_cost_usd),0) est_cost_usd,
|
|
142
|
+
COALESCE(SUM(r.est_cost_no_cache_usd),0) est_cost_no_cache_usd,
|
|
143
|
+
AVG(r.latency_ms) avg_latency_ms,
|
|
144
|
+
COALESCE(MAX(r.context_tokens),0) max_context,
|
|
145
|
+
AVG(r.context_tokens) avg_context
|
|
146
|
+
FROM requests r WHERE {w}""", p)
|
|
147
|
+
tot["prompts"] = self.one(
|
|
148
|
+
f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r WHERE {w} AND r.prompt_id IS NOT NULL",
|
|
149
|
+
p).get("n", 0)
|
|
150
|
+
tot["tool_calls"] = self.one(
|
|
151
|
+
f"SELECT COALESCE(SUM(r.tool_call_count),0) n FROM requests r WHERE {w}", p).get("n", 0)
|
|
152
|
+
|
|
153
|
+
bp = self.billing_period()
|
|
154
|
+
today = bp["today"]
|
|
155
|
+
wk = (_d(today) - timedelta(days=6)).isoformat()
|
|
156
|
+
|
|
157
|
+
def spend(extra, ep):
|
|
158
|
+
return self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c,"
|
|
159
|
+
f" COALESCE(SUM(r.billable_tokens),0) t, COUNT(*) n"
|
|
160
|
+
f" FROM requests r WHERE {w} AND {extra}", p + ep)
|
|
161
|
+
|
|
162
|
+
tot["cost_today"] = spend("r.day = ?", [today])
|
|
163
|
+
tot["cost_week"] = spend("r.day >= ?", [wk])
|
|
164
|
+
tot["cost_period"] = spend("r.day >= ? AND r.day <= ?", [bp["start"], bp["end"]])
|
|
165
|
+
tot["billing_period"] = bp
|
|
166
|
+
|
|
167
|
+
days = max(tot["active_days"] or 1, 1)
|
|
168
|
+
tot["avg_cost_per_active_day"] = tot["est_cost_usd"] / days
|
|
169
|
+
tot["avg_tokens_per_request"] = (tot["billable_tokens"] / tot["requests"]) if tot["requests"] else 0
|
|
170
|
+
tot["cost_basis"] = "estimated"
|
|
171
|
+
tot["date_range"] = {"first": self.first_day, "last": self.last_day}
|
|
172
|
+
return tot
|
|
173
|
+
|
|
174
|
+
# ---------------- burn rate & limits ----------------
|
|
175
|
+
def burn(self, f=None):
|
|
176
|
+
w, p = self.where(f)
|
|
177
|
+
bp = self.billing_period()
|
|
178
|
+
rows = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost, SUM(r.billable_tokens) tokens,
|
|
179
|
+
COUNT(*) requests FROM requests r
|
|
180
|
+
WHERE {w} AND r.day >= ? AND r.day <= ? GROUP BY 1 ORDER BY 1""",
|
|
181
|
+
p + [bp["start"], bp["end"]])
|
|
182
|
+
used_cost = sum(r["cost"] for r in rows)
|
|
183
|
+
used_tokens = sum(r["tokens"] for r in rows)
|
|
184
|
+
used_req = sum(r["requests"] for r in rows)
|
|
185
|
+
elapsed = max(bp["elapsed_days"], 1)
|
|
186
|
+
daily_avg = used_cost / elapsed
|
|
187
|
+
|
|
188
|
+
# Trailing rates are measured over the last N days of ACTUAL activity, not
|
|
189
|
+
# only the slice inside the billing period — early in a period that slice is
|
|
190
|
+
# too short to be a rate. This keeps burn and forecast on one methodology.
|
|
191
|
+
recent = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost,
|
|
192
|
+
SUM(r.billable_tokens) tokens, COUNT(*) requests
|
|
193
|
+
FROM requests r WHERE {w} AND r.day <> ''
|
|
194
|
+
GROUP BY 1 ORDER BY 1""", p)
|
|
195
|
+
last7, last14 = recent[-7:], recent[-14:]
|
|
196
|
+
avg7 = (sum(r["cost"] for r in last7) / len(last7)) if last7 else 0.0
|
|
197
|
+
tok_avg7 = (sum(r["tokens"] for r in last7) / len(last7)) if last7 else 0.0
|
|
198
|
+
# the projection rate matches Analytics.forecast()'s "expected" scenario
|
|
199
|
+
burn = (sum(r["cost"] for r in last14) / len(last14)) if last14 else daily_avg
|
|
200
|
+
tok_burn = (sum(r["tokens"] for r in last14) / len(last14)) if last14 else 0.0
|
|
201
|
+
projected = used_cost + burn * bp["remaining_days"]
|
|
202
|
+
tok_daily = used_tokens / elapsed
|
|
203
|
+
|
|
204
|
+
lim = self.settings["limits"]
|
|
205
|
+
out = {
|
|
206
|
+
"period": bp,
|
|
207
|
+
"used": {"cost_usd": used_cost, "tokens": used_tokens, "requests": used_req,
|
|
208
|
+
"basis": "estimated"},
|
|
209
|
+
"daily_avg_cost": daily_avg, "avg7_cost": avg7,
|
|
210
|
+
"daily_avg_tokens": tok_daily, "avg7_tokens": tok_avg7,
|
|
211
|
+
"burn_rate_cost_per_day": burn,
|
|
212
|
+
"burn_rate_window_days": len(last14),
|
|
213
|
+
"projected_period_cost": projected,
|
|
214
|
+
"projected_period_tokens": used_tokens + tok_burn * bp["remaining_days"],
|
|
215
|
+
"forecast_basis": "forecast",
|
|
216
|
+
"forecast_note": ("Projection uses the %d-day mean daily spend, the same rate as the "
|
|
217
|
+
"Forecast view's expected scenario." % len(last14)),
|
|
218
|
+
"series": rows,
|
|
219
|
+
"allowances": {},
|
|
220
|
+
}
|
|
221
|
+
for key, used_val, rate, label in (
|
|
222
|
+
("monthly_cost_allowance_usd", used_cost, burn, "cost"),
|
|
223
|
+
("monthly_token_allowance", used_tokens, (tok_burn or tok_daily), "tokens"),
|
|
224
|
+
("monthly_request_allowance", used_req, used_req / elapsed, "requests"),
|
|
225
|
+
):
|
|
226
|
+
allowance = lim.get(key)
|
|
227
|
+
if not allowance:
|
|
228
|
+
out["allowances"][label] = {"configured": False, "message": UNAVAILABLE}
|
|
229
|
+
continue
|
|
230
|
+
remaining = allowance - used_val
|
|
231
|
+
pct = 100.0 * used_val / allowance
|
|
232
|
+
days_left = (remaining / rate) if rate > 0 else None
|
|
233
|
+
proj = used_val + rate * bp["remaining_days"]
|
|
234
|
+
out["allowances"][label] = {
|
|
235
|
+
"configured": True, "allowance": allowance, "used": used_val,
|
|
236
|
+
"remaining": remaining, "used_pct": round(pct, 1),
|
|
237
|
+
"remaining_pct": round(max(100 - pct, 0), 1),
|
|
238
|
+
"days_until_limit": (round(days_left, 1) if days_left is not None else None),
|
|
239
|
+
"limit_date": ((_d(bp["today"]) + timedelta(days=days_left)).isoformat()
|
|
240
|
+
if days_left is not None and days_left < 3650 else None),
|
|
241
|
+
"projected_end_of_period": proj,
|
|
242
|
+
"projected_overage_pct": round(100.0 * (proj - allowance) / allowance, 1),
|
|
243
|
+
"status": self._status(pct),
|
|
244
|
+
"basis": "estimated+forecast",
|
|
245
|
+
}
|
|
246
|
+
out["remaining_credits_usd"] = lim.get("remaining_credits_usd") or UNAVAILABLE
|
|
247
|
+
return out
|
|
248
|
+
|
|
249
|
+
@staticmethod
|
|
250
|
+
def _status(pct):
|
|
251
|
+
if pct >= 100: return "critical"
|
|
252
|
+
if pct >= 90: return "approaching"
|
|
253
|
+
if pct >= 75: return "high"
|
|
254
|
+
return "healthy"
|
|
255
|
+
|
|
256
|
+
# ---------------- timeline ----------------
|
|
257
|
+
def timeline(self, f=None, grain="day"):
|
|
258
|
+
w, p = self.where(f)
|
|
259
|
+
col = "r.day" if grain == "day" else "substr(r.ts,1,13)"
|
|
260
|
+
return self.q(f"""
|
|
261
|
+
SELECT {col} bucket, COUNT(*) requests,
|
|
262
|
+
COUNT(DISTINCT r.session_id) sessions,
|
|
263
|
+
COUNT(DISTINCT r.prompt_id) prompts,
|
|
264
|
+
SUM(r.input_tokens) input_tokens, SUM(r.output_tokens) output_tokens,
|
|
265
|
+
SUM(r.cache_read_tokens) cache_read_tokens,
|
|
266
|
+
SUM(r.cache_write_tokens) cache_write_tokens,
|
|
267
|
+
SUM(r.billable_tokens) tokens, SUM(r.est_cost_usd) cost,
|
|
268
|
+
AVG(r.context_tokens) avg_context
|
|
269
|
+
FROM requests r WHERE {w} AND r.day <> '' GROUP BY 1 ORDER BY 1""", p)
|
|
270
|
+
|
|
271
|
+
# ---------------- models ----------------
|
|
272
|
+
def models(self, f=None):
|
|
273
|
+
w, p = self.where(f)
|
|
274
|
+
rows = self.q(f"""
|
|
275
|
+
SELECT r.model, r.model_known, COUNT(*) requests,
|
|
276
|
+
COUNT(DISTINCT r.session_id) sessions,
|
|
277
|
+
SUM(r.input_tokens) input_tokens, SUM(r.output_tokens) output_tokens,
|
|
278
|
+
SUM(r.thinking_tokens) thinking_tokens,
|
|
279
|
+
SUM(r.cache_read_tokens) cache_read_tokens,
|
|
280
|
+
SUM(r.cache_write_tokens) cache_write_tokens,
|
|
281
|
+
SUM(r.billable_tokens) tokens, SUM(r.est_cost_usd) cost,
|
|
282
|
+
AVG(r.latency_ms) avg_latency_ms, AVG(r.context_tokens) avg_context,
|
|
283
|
+
MAX(r.context_tokens) max_context
|
|
284
|
+
FROM requests r WHERE {w} GROUP BY r.model ORDER BY cost DESC""", p)
|
|
285
|
+
tc = sum(r["cost"] for r in rows) or 1
|
|
286
|
+
tt = sum(r["tokens"] for r in rows) or 1
|
|
287
|
+
for r in rows:
|
|
288
|
+
r["display_name"] = self.pricing.display_name(r["model"])
|
|
289
|
+
r["tier"] = self.pricing.tier(r["model"])
|
|
290
|
+
r["context_window"] = self.pricing.context_window(r["model"])
|
|
291
|
+
r["pricing_known"] = bool(r["model_known"])
|
|
292
|
+
r["cost_pct"] = round(100.0 * r["cost"] / tc, 1)
|
|
293
|
+
r["token_pct"] = round(100.0 * r["tokens"] / tt, 1)
|
|
294
|
+
r["cost_per_1k_output"] = (1000.0 * r["cost"] / r["output_tokens"]) if r["output_tokens"] else None
|
|
295
|
+
r["output_per_input"] = (r["output_tokens"] / (r["input_tokens"] + r["cache_read_tokens"]
|
|
296
|
+
+ r["cache_write_tokens"])) if r["tokens"] else 0
|
|
297
|
+
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
298
|
+
r["utilization_pct"] = (round(100.0 * (r["avg_context"] or 0) / r["context_window"], 1)
|
|
299
|
+
if r["context_window"] else None)
|
|
300
|
+
priced = [r for r in rows if r["tokens"] and r["tier"] != "none"]
|
|
301
|
+
superlatives = {}
|
|
302
|
+
if priced:
|
|
303
|
+
superlatives = {
|
|
304
|
+
"most_expensive": max(priced, key=lambda r: r["cost"])["model"],
|
|
305
|
+
"most_used": max(priced, key=lambda r: r["requests"])["model"],
|
|
306
|
+
"most_token_efficient": max(priced, key=lambda r: r["output_per_input"])["model"],
|
|
307
|
+
"best_cost_per_output": min(
|
|
308
|
+
[r for r in priced if r["cost_per_1k_output"]],
|
|
309
|
+
key=lambda r: r["cost_per_1k_output"], default={}).get("model"),
|
|
310
|
+
}
|
|
311
|
+
return {"rows": rows, "superlatives": superlatives, "basis": "estimated"}
|
|
312
|
+
|
|
313
|
+
# ---------------- projects / sessions / prompts ----------------
|
|
314
|
+
def projects(self, f=None):
|
|
315
|
+
w, p = self.where(f)
|
|
316
|
+
rows = self.q(f"""
|
|
317
|
+
SELECT pr.id project_id, pr.name, pr.path, pr.slug, pr.is_sandbox,
|
|
318
|
+
COUNT(DISTINCT r.session_id) sessions,
|
|
319
|
+
COUNT(DISTINCT r.prompt_id) prompts, COUNT(*) requests,
|
|
320
|
+
SUM(r.billable_tokens) tokens, SUM(r.output_tokens) output_tokens,
|
|
321
|
+
SUM(r.cache_read_tokens) cache_read_tokens,
|
|
322
|
+
SUM(r.est_cost_usd) cost, GROUP_CONCAT(DISTINCT r.model) models
|
|
323
|
+
FROM requests r JOIN projects pr ON pr.id = r.project_id
|
|
324
|
+
WHERE {w} GROUP BY pr.id ORDER BY cost DESC""", p)
|
|
325
|
+
for r in rows:
|
|
326
|
+
r["avg_cost_per_session"] = r["cost"] / r["sessions"] if r["sessions"] else 0
|
|
327
|
+
r["avg_cost_per_prompt"] = r["cost"] / r["prompts"] if r["prompts"] else None
|
|
328
|
+
r["files_touched"] = self.one(
|
|
329
|
+
"SELECT COUNT(DISTINCT path) n FROM files_touched WHERE project_id=?",
|
|
330
|
+
(r["project_id"],))["n"]
|
|
331
|
+
b = self.settings["budgets"].get("per_project_usd", {}).get(r["name"])
|
|
332
|
+
r["budget_usd"] = b
|
|
333
|
+
r["budget_used_pct"] = round(100.0 * r["cost"] / b, 1) if b else None
|
|
334
|
+
return rows
|
|
335
|
+
|
|
336
|
+
def sessions(self, f=None, limit=500, order="cost"):
|
|
337
|
+
w, p = self.where(f)
|
|
338
|
+
ob = {"cost": "cost DESC", "tokens": "tokens DESC", "duration": "s.duration_s DESC",
|
|
339
|
+
"recent": "s.started_at DESC", "prompts": "prompts DESC"}.get(order, "cost DESC")
|
|
340
|
+
rows = self.q(f"""
|
|
341
|
+
SELECT s.id session_id, s.title, s.git_branch, s.cli_version, s.started_at,
|
|
342
|
+
s.ended_at, s.duration_s, s.files_touched, pr.name project, pr.id project_id,
|
|
343
|
+
COUNT(DISTINCT r.prompt_id) prompts, COUNT(*) requests,
|
|
344
|
+
SUM(r.tool_call_count) tool_calls,
|
|
345
|
+
SUM(r.input_tokens) input_tokens, SUM(r.output_tokens) output_tokens,
|
|
346
|
+
SUM(r.cache_read_tokens) cache_read_tokens,
|
|
347
|
+
SUM(r.cache_write_tokens) cache_write_tokens,
|
|
348
|
+
SUM(r.billable_tokens) tokens, SUM(r.est_cost_usd) cost,
|
|
349
|
+
MAX(r.context_tokens) max_context, AVG(r.context_tokens) avg_context,
|
|
350
|
+
GROUP_CONCAT(DISTINCT r.model) models
|
|
351
|
+
FROM requests r JOIN sessions s ON s.id = r.session_id
|
|
352
|
+
JOIN projects pr ON pr.id = s.project_id
|
|
353
|
+
WHERE {w} GROUP BY s.id ORDER BY {ob} LIMIT ?""", p + [limit])
|
|
354
|
+
for r in rows:
|
|
355
|
+
r["cost_per_prompt"] = r["cost"] / r["prompts"] if r["prompts"] else None
|
|
356
|
+
r["tokens_per_prompt"] = r["tokens"] / r["prompts"] if r["prompts"] else None
|
|
357
|
+
r["tokens_per_request"] = r["tokens"] / r["requests"] if r["requests"] else 0
|
|
358
|
+
r["output_ratio"] = r["output_tokens"] / r["tokens"] if r["tokens"] else 0
|
|
359
|
+
r["cache_hit_ratio"] = (r["cache_read_tokens"] /
|
|
360
|
+
(r["cache_read_tokens"] + r["cache_write_tokens"])
|
|
361
|
+
if (r["cache_read_tokens"] + r["cache_write_tokens"]) else None)
|
|
362
|
+
return rows
|
|
363
|
+
|
|
364
|
+
def prompts(self, f=None, limit=300, offset=0, order="cost", search=None):
|
|
365
|
+
w, p = self.where(f)
|
|
366
|
+
ob = {"cost": "pcost DESC", "tokens": "ptokens DESC", "recent": "pr.ts DESC",
|
|
367
|
+
"cheapest": "pcost ASC", "efficiency": "efficiency DESC",
|
|
368
|
+
"length": "pr.char_len DESC"}.get(order, "pcost DESC")
|
|
369
|
+
extra, ep = "", []
|
|
370
|
+
if search:
|
|
371
|
+
extra = (" AND r.prompt_id IN (SELECT id FROM prompts pr WHERE pr.text LIKE ? "
|
|
372
|
+
"OR pr.session_id LIKE ? OR pr.category LIKE ?)")
|
|
373
|
+
ep = [f"%{search}%"] * 3
|
|
374
|
+
# Aggregate requests per prompt first, rank, and only then join the (large) prompt
|
|
375
|
+
# text for the page being returned. Grouping with the text attached was ~2s a call.
|
|
376
|
+
if order in ("recent", "length"):
|
|
377
|
+
inner_ob, inner_lim, inner_p = "", "", []
|
|
378
|
+
else:
|
|
379
|
+
inner_ob, inner_lim, inner_p = f"ORDER BY {ob}", "LIMIT ? OFFSET ?", [limit, offset]
|
|
380
|
+
rows = self.q(f"""
|
|
381
|
+
WITH agg AS (
|
|
382
|
+
SELECT r.prompt_id,
|
|
383
|
+
COUNT(r.id) requests, SUM(r.input_tokens) input_tokens,
|
|
384
|
+
SUM(r.output_tokens) output_tokens, SUM(r.cache_read_tokens) cache_read_tokens,
|
|
385
|
+
SUM(r.cache_write_tokens) cache_write_tokens,
|
|
386
|
+
SUM(r.billable_tokens) ptokens, SUM(r.est_cost_usd) pcost,
|
|
387
|
+
SUM(r.tool_call_count) tool_calls, AVG(r.latency_ms) latency_ms,
|
|
388
|
+
MAX(r.context_tokens) max_context,
|
|
389
|
+
GROUP_CONCAT(DISTINCT r.model) models,
|
|
390
|
+
(CAST(SUM(r.output_tokens) AS REAL) / MAX(SUM(r.billable_tokens),1)) efficiency
|
|
391
|
+
FROM requests r WHERE {w}{extra} AND r.prompt_id IS NOT NULL
|
|
392
|
+
GROUP BY r.prompt_id {inner_ob} {inner_lim})
|
|
393
|
+
SELECT pr.id prompt_id, pr.uuid, pr.ts, pr.day, pr.text, pr.char_len, pr.word_len,
|
|
394
|
+
pr.category, pr.category_confidence, pr.category_evidence, pr.source,
|
|
395
|
+
pr.session_id, proj.name project, proj.id project_id, s.title session_title,
|
|
396
|
+
agg.requests, agg.input_tokens, agg.output_tokens, agg.cache_read_tokens,
|
|
397
|
+
agg.cache_write_tokens, agg.ptokens, agg.pcost, agg.tool_calls, agg.latency_ms,
|
|
398
|
+
agg.max_context, agg.models, agg.efficiency
|
|
399
|
+
FROM agg JOIN prompts pr ON pr.id = agg.prompt_id
|
|
400
|
+
JOIN projects proj ON proj.id = pr.project_id
|
|
401
|
+
LEFT JOIN sessions s ON s.id = pr.session_id
|
|
402
|
+
ORDER BY {ob} {"" if inner_lim else "LIMIT ? OFFSET ?"}""",
|
|
403
|
+
p + ep + inner_p + ([] if inner_lim else [limit, offset]))
|
|
404
|
+
keep_text = bool((f or {}).get("_full_text"))
|
|
405
|
+
for r in rows:
|
|
406
|
+
r["preview"] = (r["text"] or "")[:220]
|
|
407
|
+
if not keep_text:
|
|
408
|
+
# the list endpoint ships previews only; /api/prompt/<id> carries full text
|
|
409
|
+
del r["text"]
|
|
410
|
+
r["cost_per_1k_output"] = (1000.0 * r["pcost"] / r["output_tokens"]
|
|
411
|
+
if r["output_tokens"] else None)
|
|
412
|
+
r["files_touched"] = self.one(
|
|
413
|
+
"SELECT COUNT(DISTINCT path) n FROM files_touched WHERE prompt_id=?",
|
|
414
|
+
(r["prompt_id"],))["n"]
|
|
415
|
+
try:
|
|
416
|
+
r["category_evidence"] = json.loads(r["category_evidence"] or "[]")
|
|
417
|
+
except Exception:
|
|
418
|
+
r["category_evidence"] = []
|
|
419
|
+
return rows
|
|
420
|
+
|
|
421
|
+
def prompt_detail(self, pid):
|
|
422
|
+
p = self.one("""SELECT pr.*, proj.name project, s.title session_title, s.git_branch
|
|
423
|
+
FROM prompts pr JOIN projects proj ON proj.id=pr.project_id
|
|
424
|
+
LEFT JOIN sessions s ON s.id=pr.session_id WHERE pr.id=?""", (pid,))
|
|
425
|
+
if not p:
|
|
426
|
+
return {"error": "not found"}
|
|
427
|
+
p["requests"] = self.q(
|
|
428
|
+
"SELECT ts, model, effort, input_tokens, output_tokens, thinking_tokens,"
|
|
429
|
+
" cache_read_tokens, cache_write_tokens, billable_tokens, context_tokens,"
|
|
430
|
+
" est_cost_usd, latency_ms, stop_reason, tool_call_count"
|
|
431
|
+
" FROM requests WHERE prompt_id=? ORDER BY ts", (pid,))
|
|
432
|
+
# NB: keep the scalar prompts.tool_calls count intact — the log goes under its own key
|
|
433
|
+
p["tool_call_log"] = self.q(
|
|
434
|
+
"SELECT name, target, ts FROM tool_calls WHERE prompt_id=? ORDER BY ts", (pid,))
|
|
435
|
+
p["tool_summary"] = self.q(
|
|
436
|
+
"SELECT name, COUNT(*) n FROM tool_calls WHERE prompt_id=? GROUP BY 1 ORDER BY 2 DESC",
|
|
437
|
+
(pid,))
|
|
438
|
+
p["files"] = self.q(
|
|
439
|
+
"SELECT path, op, COUNT(*) n FROM files_touched WHERE prompt_id=? GROUP BY path, op",
|
|
440
|
+
(pid,))
|
|
441
|
+
try:
|
|
442
|
+
p["category_evidence"] = json.loads(p.get("category_evidence") or "[]")
|
|
443
|
+
except Exception:
|
|
444
|
+
p["category_evidence"] = []
|
|
445
|
+
p["advisor"] = self.prompt_advisor(p)
|
|
446
|
+
return p
|
|
447
|
+
|
|
448
|
+
def session_detail(self, sid):
|
|
449
|
+
s = self.one("""SELECT s.*, pr.name project FROM sessions s
|
|
450
|
+
JOIN projects pr ON pr.id=s.project_id WHERE s.id=?""", (sid,))
|
|
451
|
+
if not s:
|
|
452
|
+
return {"error": "not found"}
|
|
453
|
+
s["prompts"] = self.q("""
|
|
454
|
+
SELECT pr.id prompt_id, pr.ts, pr.category, substr(pr.text,1,220) preview,
|
|
455
|
+
pr.char_len, pr.billable_tokens ptokens, pr.est_cost_usd pcost,
|
|
456
|
+
pr.tool_calls, pr.models, pr.max_context_tokens max_context
|
|
457
|
+
FROM prompts pr WHERE pr.session_id=? ORDER BY pr.ts""", (sid,))
|
|
458
|
+
s["timeline"] = self.q(
|
|
459
|
+
"SELECT ts, model, billable_tokens, context_tokens, output_tokens, est_cost_usd"
|
|
460
|
+
" FROM requests WHERE session_id=? ORDER BY ts", (sid,))
|
|
461
|
+
s["tools"] = self.q(
|
|
462
|
+
"SELECT name, COUNT(*) n FROM tool_calls WHERE session_id=? GROUP BY 1 ORDER BY 2 DESC",
|
|
463
|
+
(sid,))
|
|
464
|
+
s["files"] = self.q(
|
|
465
|
+
"SELECT path, GROUP_CONCAT(DISTINCT op) ops, COUNT(*) n FROM files_touched"
|
|
466
|
+
" WHERE session_id=? GROUP BY path ORDER BY n DESC LIMIT 100", (sid,))
|
|
467
|
+
return s
|
|
468
|
+
|
|
469
|
+
# ---------------- categories ----------------
|
|
470
|
+
def categories(self, f=None):
|
|
471
|
+
w, p = self.where(f)
|
|
472
|
+
rows = self.q(f"""
|
|
473
|
+
SELECT pr.category, COUNT(DISTINCT pr.id) prompts, COUNT(r.id) requests,
|
|
474
|
+
SUM(r.billable_tokens) tokens, SUM(r.output_tokens) output_tokens,
|
|
475
|
+
SUM(r.est_cost_usd) cost, AVG(pr.category_confidence) confidence,
|
|
476
|
+
AVG(pr.char_len) avg_prompt_chars
|
|
477
|
+
FROM prompts pr JOIN requests r ON r.prompt_id = pr.id
|
|
478
|
+
WHERE {w} GROUP BY pr.category ORDER BY cost DESC""", p)
|
|
479
|
+
tc = sum(r["cost"] for r in rows) or 1
|
|
480
|
+
for r in rows:
|
|
481
|
+
r["cost_pct"] = round(100.0 * r["cost"] / tc, 1)
|
|
482
|
+
r["cost_per_prompt"] = r["cost"] / r["prompts"] if r["prompts"] else 0
|
|
483
|
+
return {"rows": rows, "basis": "estimated",
|
|
484
|
+
"note": "Categories are heuristic keyword classifications of your prompt text."}
|
|
485
|
+
|
|
486
|
+
# ---------------- leaderboards ----------------
|
|
487
|
+
def leaderboards(self, f=None, n=20):
|
|
488
|
+
return {
|
|
489
|
+
"most_expensive": self.prompts(f, limit=n, order="cost"),
|
|
490
|
+
"most_token_heavy": self.prompts(f, limit=n, order="tokens"),
|
|
491
|
+
"cheapest": self.prompts(f, limit=n, order="cheapest"),
|
|
492
|
+
"most_efficient": self.prompts(f, limit=n, order="efficiency"),
|
|
493
|
+
"longest_sessions": self.sessions(f, limit=n, order="duration"),
|
|
494
|
+
"basis": "estimated",
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
# ---------------- efficiency ----------------
|
|
498
|
+
def efficiency(self, f=None):
|
|
499
|
+
w, p = self.where(f)
|
|
500
|
+
t = self.one(f"""SELECT SUM(input_tokens) i, SUM(output_tokens) o,
|
|
501
|
+
SUM(cache_read_tokens) cr, SUM(cache_write_tokens) cw,
|
|
502
|
+
SUM(billable_tokens) tot, SUM(thinking_tokens) think,
|
|
503
|
+
COUNT(*) n, SUM(est_cost_usd) cost,
|
|
504
|
+
SUM(est_cost_no_cache_usd) cost_nc, AVG(context_tokens) avgctx
|
|
505
|
+
FROM requests r WHERE {w}""", p)
|
|
506
|
+
tot = t["tot"] or 1
|
|
507
|
+
prompt_side = (t["i"] or 0) + (t["cr"] or 0) + (t["cw"] or 0)
|
|
508
|
+
cache_total = (t["cr"] or 0) + (t["cw"] or 0)
|
|
509
|
+
sess = self.sessions(f, limit=100000, order="cost")
|
|
510
|
+
scored = [s for s in sess if s["tokens"] and s["prompts"]]
|
|
511
|
+
for s in scored:
|
|
512
|
+
s["efficiency_score"] = round(100.0 * s["output_ratio"] * 10, 1)
|
|
513
|
+
scored.sort(key=lambda s: s["output_ratio"])
|
|
514
|
+
return {
|
|
515
|
+
"output_ratio": (t["o"] or 0) / tot,
|
|
516
|
+
"output_per_input": ((t["o"] or 0) / prompt_side) if prompt_side else 0,
|
|
517
|
+
"thinking_share_of_output": ((t["think"] or 0) / (t["o"] or 1)),
|
|
518
|
+
"cache_hit_ratio": ((t["cr"] or 0) / cache_total) if cache_total else None,
|
|
519
|
+
"tokens_per_request": tot / (t["n"] or 1),
|
|
520
|
+
"avg_context_tokens": t["avgctx"],
|
|
521
|
+
"cost_per_1k_output": (1000.0 * (t["cost"] or 0) / (t["o"] or 1)),
|
|
522
|
+
"cost_per_prompt": ((t["cost"] or 0) /
|
|
523
|
+
max(self.one(f"SELECT COUNT(DISTINCT r.prompt_id) n FROM requests r WHERE {w}", p)["n"], 1)),
|
|
524
|
+
"cache": {
|
|
525
|
+
"reads": t["cr"], "writes": t["cw"],
|
|
526
|
+
"cost_with_cache": t["cost"], "cost_without_cache": t["cost_nc"],
|
|
527
|
+
"estimated_savings_usd": (t["cost_nc"] or 0) - (t["cost"] or 0),
|
|
528
|
+
"savings_pct": (round(100.0 * ((t["cost_nc"] or 0) - (t["cost"] or 0))
|
|
529
|
+
/ (t["cost_nc"] or 1), 1)),
|
|
530
|
+
"basis": "estimated",
|
|
531
|
+
},
|
|
532
|
+
"low_efficiency_sessions": scored[:10],
|
|
533
|
+
"high_efficiency_sessions": scored[-10:][::-1],
|
|
534
|
+
"basis": "estimated",
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
def context_analysis(self, f=None):
|
|
538
|
+
w, p = self.where(f)
|
|
539
|
+
buckets = self.q(f"""
|
|
540
|
+
SELECT CASE
|
|
541
|
+
WHEN r.context_tokens < 25000 THEN '0-25K'
|
|
542
|
+
WHEN r.context_tokens < 50000 THEN '25-50K'
|
|
543
|
+
WHEN r.context_tokens < 100000 THEN '50-100K'
|
|
544
|
+
WHEN r.context_tokens < 150000 THEN '100-150K'
|
|
545
|
+
WHEN r.context_tokens < 200000 THEN '150-200K'
|
|
546
|
+
ELSE '200K+' END bucket,
|
|
547
|
+
COUNT(*) requests, SUM(r.billable_tokens) tokens, SUM(r.est_cost_usd) cost
|
|
548
|
+
FROM requests r WHERE {w} GROUP BY 1""", p)
|
|
549
|
+
order = ['0-25K', '25-50K', '50-100K', '100-150K', '150-200K', '200K+']
|
|
550
|
+
buckets.sort(key=lambda b: order.index(b["bucket"]) if b["bucket"] in order else 99)
|
|
551
|
+
total_cost = sum(b["cost"] for b in buckets) or 1
|
|
552
|
+
for b in buckets:
|
|
553
|
+
b["cost_pct"] = round(100.0 * b["cost"] / total_cost, 1)
|
|
554
|
+
thr = self.settings["waste_rules"]["large_context_tokens"]
|
|
555
|
+
big = self.one(f"SELECT COUNT(*) n, COALESCE(SUM(r.est_cost_usd),0) c FROM requests r"
|
|
556
|
+
f" WHERE {w} AND r.context_tokens >= ?", p + [thr])
|
|
557
|
+
heavy_sessions = self.q(f"""
|
|
558
|
+
SELECT s.id session_id, s.title, pr.name project, MAX(r.context_tokens) max_context,
|
|
559
|
+
AVG(r.context_tokens) avg_context, COUNT(*) requests, SUM(r.est_cost_usd) cost
|
|
560
|
+
FROM requests r JOIN sessions s ON s.id=r.session_id
|
|
561
|
+
JOIN projects pr ON pr.id=s.project_id
|
|
562
|
+
WHERE {w} GROUP BY s.id HAVING MAX(r.context_tokens) >= ?
|
|
563
|
+
ORDER BY cost DESC LIMIT 20""", p + [thr])
|
|
564
|
+
agg = self.one(f"SELECT AVG(r.context_tokens) a, MAX(r.context_tokens) m FROM requests r WHERE {w}", p)
|
|
565
|
+
windows = [v.get("context_window") for v in self.pricing.models.values() if v.get("context_window")]
|
|
566
|
+
return {
|
|
567
|
+
"buckets": buckets, "avg_context": agg["a"], "max_context": agg["m"],
|
|
568
|
+
"threshold": thr,
|
|
569
|
+
"large_context_requests": big["n"], "large_context_cost": big["c"],
|
|
570
|
+
"large_context_cost_pct": round(100.0 * big["c"] / total_cost, 1),
|
|
571
|
+
"heavy_sessions": heavy_sessions,
|
|
572
|
+
"typical_context_window": max(windows) if windows else None,
|
|
573
|
+
"basis": "actual token counts, estimated cost",
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
# ---------------- waste ----------------
|
|
577
|
+
def waste(self, f=None):
|
|
578
|
+
"""Waste rules over the pre-rolled prompt/session tables.
|
|
579
|
+
|
|
580
|
+
Two different numbers per finding, and the distinction matters:
|
|
581
|
+
|
|
582
|
+
* ``est_cost_usd`` — the *exposed* spend: what the flagged items cost in
|
|
583
|
+
total. It is the money worth reviewing, not the money wasted.
|
|
584
|
+
* ``est_excess_usd`` — the *estimated excess*: how much more the flagged
|
|
585
|
+
items cost than a reasonable baseline for the same work. This is
|
|
586
|
+
bounded per rule and is the honest "waste" figure.
|
|
587
|
+
|
|
588
|
+
Attributing a flagged session's whole cost to waste would be wrong — those
|
|
589
|
+
sessions did real work — and on a skewed spend distribution it saturates at
|
|
590
|
+
~100%, which makes it useless for grading. The excess estimate does not.
|
|
591
|
+
"""
|
|
592
|
+
w, p = self.where(f)
|
|
593
|
+
rules = self.settings["waste_rules"]
|
|
594
|
+
pfilter = f"pr.id IN (SELECT DISTINCT r.prompt_id FROM requests r WHERE {w})"
|
|
595
|
+
sfilter = f"s.id IN (SELECT DISTINCT r.session_id FROM requests r WHERE {w})"
|
|
596
|
+
findings = []
|
|
597
|
+
|
|
598
|
+
def add(sev, kind, title, detail, evidence, action, key, excess_basis):
|
|
599
|
+
cost = sum(e.get("cost") or 0 for e in evidence)
|
|
600
|
+
excess = sum(min(max(e.get("excess") or 0, 0), e.get("cost") or 0) for e in evidence)
|
|
601
|
+
findings.append({
|
|
602
|
+
"severity": sev, "kind": kind, "title": title, "detail": detail,
|
|
603
|
+
"est_cost_usd": cost, "est_excess_usd": excess,
|
|
604
|
+
"excess_basis": excess_basis, "evidence": evidence,
|
|
605
|
+
"affected": {key: [e[key] for e in evidence if e.get(key)]},
|
|
606
|
+
"recommended_action": action, "basis": "estimated"})
|
|
607
|
+
|
|
608
|
+
# 1. very long prompts — excess is the share of spend attributable to
|
|
609
|
+
# re-sending the oversized prompt text on every turn of the same request.
|
|
610
|
+
rows = self.q(f"""SELECT pr.id prompt_id, pr.char_len, substr(pr.text,1,160) preview,
|
|
611
|
+
pr.session_id, pr.est_cost_usd cost, pr.billable_tokens tokens,
|
|
612
|
+
pr.request_count requests
|
|
613
|
+
FROM prompts pr WHERE {pfilter} AND pr.char_len >= ?
|
|
614
|
+
ORDER BY cost DESC LIMIT 15""", p + [rules["long_prompt_chars"]])
|
|
615
|
+
budget_chars = rules["long_prompt_chars"]
|
|
616
|
+
for r in rows:
|
|
617
|
+
over_tokens = max(r["char_len"] - budget_chars, 0) / 4.0 # ~4 chars/token
|
|
618
|
+
resent = over_tokens * max(r["requests"], 1)
|
|
619
|
+
r["excess"] = (r["cost"] * resent / r["tokens"]) if r["tokens"] else 0
|
|
620
|
+
if rows:
|
|
621
|
+
add("high", "long_prompts",
|
|
622
|
+
f"{len(rows)} very long prompts (>{budget_chars:,} chars)",
|
|
623
|
+
"Long pasted prompts inflate the cached prefix re-sent on every following turn.",
|
|
624
|
+
rows, "Move large pasted context into a file and reference it, or summarize first.",
|
|
625
|
+
"prompt_id",
|
|
626
|
+
f"share of spend from prompt text beyond {budget_chars:,} characters, re-sent per request")
|
|
627
|
+
|
|
628
|
+
# 2. duplicate prompts — excess is the cost of the repeats, not the first ask.
|
|
629
|
+
dups = self.q(f"""SELECT pr.norm_hash, COUNT(*) n, substr(MIN(pr.text),1,160) preview,
|
|
630
|
+
SUM(pr.est_cost_usd) cost, SUM(pr.billable_tokens) tokens,
|
|
631
|
+
MIN(pr.est_cost_usd) first_cost, GROUP_CONCAT(pr.id) prompt_ids
|
|
632
|
+
FROM prompts pr WHERE {pfilter} AND pr.char_len > 25
|
|
633
|
+
GROUP BY pr.norm_hash HAVING n > 1
|
|
634
|
+
ORDER BY cost DESC LIMIT 15""", p)
|
|
635
|
+
for d in dups:
|
|
636
|
+
d["prompt_id"] = int(d["prompt_ids"].split(",")[0])
|
|
637
|
+
d["affected_ids"] = [int(x) for x in d["prompt_ids"].split(",")]
|
|
638
|
+
d["excess"] = max(d["cost"] - d["first_cost"], 0) # repeats only
|
|
639
|
+
if dups:
|
|
640
|
+
add("high", "duplicate_prompts", f"{len(dups)} prompts repeated more than once",
|
|
641
|
+
"The same request was sent again, re-paying for context each time.",
|
|
642
|
+
dups, "Reuse the earlier answer, or capture the recurring request as a slash command.",
|
|
643
|
+
"prompt_id", "cost of the repeat occurrences, excluding the first ask")
|
|
644
|
+
|
|
645
|
+
# 3. low-yield sessions — excess is what the session cost ABOVE what the same
|
|
646
|
+
# output would have cost at your own median session efficiency.
|
|
647
|
+
ratios = [r["x"] for r in self.q(
|
|
648
|
+
"SELECT CAST(output_tokens AS REAL)/billable_tokens x FROM sessions"
|
|
649
|
+
" WHERE billable_tokens > 100000")]
|
|
650
|
+
median_ratio = statistics.median(ratios) if ratios else 0.0
|
|
651
|
+
cutoff = median_ratio * rules["low_output_ratio_vs_median"]
|
|
652
|
+
low = self.q(f"""SELECT s.id session_id, s.title, s.billable_tokens tokens,
|
|
653
|
+
s.output_tokens out_tokens, s.est_cost_usd cost, s.request_count requests,
|
|
654
|
+
(CAST(s.output_tokens AS REAL)/MAX(s.billable_tokens,1)) output_ratio,
|
|
655
|
+
proj.name project FROM sessions s JOIN projects proj ON proj.id=s.project_id
|
|
656
|
+
WHERE {sfilter} AND s.billable_tokens > ?
|
|
657
|
+
AND (CAST(s.output_tokens AS REAL)/MAX(s.billable_tokens,1)) < ?
|
|
658
|
+
ORDER BY cost DESC LIMIT 15""",
|
|
659
|
+
p + [rules["huge_session_tokens"], cutoff])
|
|
660
|
+
for r in low:
|
|
661
|
+
# at the median ratio the same output needs out/median tokens, so the
|
|
662
|
+
# baseline cost scales by (actual ratio / median ratio)
|
|
663
|
+
r["excess"] = r["cost"] * (1 - (r["output_ratio"] / median_ratio)) if median_ratio else 0
|
|
664
|
+
if low:
|
|
665
|
+
add("high", "low_yield_sessions",
|
|
666
|
+
f"{len(low)} large sessions yielded under {cutoff*100:.2f}% output tokens",
|
|
667
|
+
f"Your median session turns {median_ratio*100:.2f}% of billable tokens into output. "
|
|
668
|
+
f"These ran well below that while consuming heavy context.",
|
|
669
|
+
low, "Start a fresh session or /compact once a thread stops producing new output.",
|
|
670
|
+
"session_id", "spend above what the same output would cost at your median session efficiency")
|
|
671
|
+
|
|
672
|
+
# 4. frontier model on small tasks — excess is computed against the cheaper tier.
|
|
673
|
+
frontier = [m for m, v in self.pricing.models.items()
|
|
674
|
+
if v.get("tier") in rules["frontier_tiers"]]
|
|
675
|
+
tiers = defaultdict(list)
|
|
676
|
+
for m, v in self.pricing.models.items():
|
|
677
|
+
tiers[v.get("tier")].append(m)
|
|
678
|
+
cheaper = None
|
|
679
|
+
for t in ("balanced", "economy"):
|
|
680
|
+
if tiers.get(t):
|
|
681
|
+
cheaper = min(tiers[t], key=lambda m: self.pricing.rates(m).get("output", 1e9))
|
|
682
|
+
break
|
|
683
|
+
if frontier:
|
|
684
|
+
ph = ",".join("?" * len(frontier))
|
|
685
|
+
small = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.category,
|
|
686
|
+
pr.session_id, pr.est_cost_usd cost, pr.output_tokens out_tokens,
|
|
687
|
+
pr.billable_tokens tokens, pr.models, pr.input_tokens,
|
|
688
|
+
pr.cache_read_tokens, pr.cache_write_tokens
|
|
689
|
+
FROM prompts pr WHERE {pfilter}
|
|
690
|
+
AND pr.output_tokens < ? AND pr.est_cost_usd > 0
|
|
691
|
+
AND EXISTS (SELECT 1 FROM requests r2 WHERE r2.prompt_id=pr.id
|
|
692
|
+
AND r2.model IN ({ph}))
|
|
693
|
+
ORDER BY cost DESC LIMIT 15""",
|
|
694
|
+
p + [rules["simple_task_output_tokens"]] + frontier)
|
|
695
|
+
for r in small:
|
|
696
|
+
alt = (self.pricing.estimate(cheaper, r["input_tokens"], r["out_tokens"],
|
|
697
|
+
r["cache_read_tokens"], r["cache_write_tokens"], 0)
|
|
698
|
+
if cheaper else r["cost"])
|
|
699
|
+
r["excess"] = max(r["cost"] - alt, 0)
|
|
700
|
+
if small:
|
|
701
|
+
add("medium", "frontier_on_small_tasks",
|
|
702
|
+
f"{len(small)} frontier-model prompts produced under "
|
|
703
|
+
f"{rules['simple_task_output_tokens']} output tokens",
|
|
704
|
+
"Short, simple turns running on the most expensive model tier.",
|
|
705
|
+
small, "Route short lookups and confirmations to a cheaper model tier.",
|
|
706
|
+
"prompt_id",
|
|
707
|
+
f"difference against the same tokens priced at {self.pricing.display_name(cheaper)}"
|
|
708
|
+
if cheaper else "n/a")
|
|
709
|
+
|
|
710
|
+
# 5. tool loops — excess is the share of the loop beyond the threshold.
|
|
711
|
+
loops = self.q(f"""SELECT pr.id prompt_id, substr(pr.text,1,160) preview, pr.session_id,
|
|
712
|
+
pr.tool_calls tools, pr.est_cost_usd cost, pr.billable_tokens tokens
|
|
713
|
+
FROM prompts pr WHERE {pfilter} AND pr.tool_calls > 40
|
|
714
|
+
ORDER BY cost DESC LIMIT 15""", p)
|
|
715
|
+
for r in loops:
|
|
716
|
+
r["excess"] = r["cost"] * max(r["tools"] - 40, 0) / max(r["tools"], 1)
|
|
717
|
+
if loops:
|
|
718
|
+
add("medium", "tool_loops", f"{len(loops)} prompts triggered 40+ tool calls",
|
|
719
|
+
"Long agentic loops re-send the whole conversation each step, so cost grows super-linearly.",
|
|
720
|
+
loops, "Split the task, or give more precise instructions up front.", "prompt_id",
|
|
721
|
+
"share of the loop beyond the first 40 tool calls")
|
|
722
|
+
|
|
723
|
+
# 6. poor cache reuse — excess is the write premium over plain input pricing.
|
|
724
|
+
poor = self.q(f"""SELECT s.id session_id, s.title, proj.name project,
|
|
725
|
+
s.cache_read_tokens reads, s.cache_write_tokens writes,
|
|
726
|
+
s.est_cost_usd cost, s.request_count requests, s.models
|
|
727
|
+
FROM sessions s JOIN projects proj ON proj.id=s.project_id
|
|
728
|
+
WHERE {sfilter} AND s.cache_write_tokens > 500000
|
|
729
|
+
AND s.cache_read_tokens < s.cache_write_tokens * 3
|
|
730
|
+
ORDER BY cost DESC LIMIT 15""", p)
|
|
731
|
+
for r in poor:
|
|
732
|
+
m = (r["models"] or "").split(",")[0]
|
|
733
|
+
rt = self.pricing.rates(m)
|
|
734
|
+
premium = float(rt.get("cache_write_5m", 0)) - float(rt.get("input", 0))
|
|
735
|
+
r["excess"] = max(r["writes"] * premium / 1_000_000.0, 0)
|
|
736
|
+
if poor:
|
|
737
|
+
add("medium", "poor_cache_reuse", f"{len(poor)} sessions wrote cache they barely reused",
|
|
738
|
+
"Cache writes cost more than plain input; they only pay off when read back repeatedly.",
|
|
739
|
+
poor, "Keep related work in one continuous session so the cached prefix is reused.",
|
|
740
|
+
"session_id", "the cache-write premium over plain input pricing on those writes")
|
|
741
|
+
|
|
742
|
+
# 7. long-lived sparse sessions — informational, no excess claimed.
|
|
743
|
+
idle = self.q(f"""SELECT s.id session_id, s.title, proj.name project, s.duration_s,
|
|
744
|
+
s.request_count requests, s.est_cost_usd cost
|
|
745
|
+
FROM sessions s JOIN projects proj ON proj.id=s.project_id
|
|
746
|
+
WHERE {sfilter} AND s.duration_s > ? AND s.request_count < 30
|
|
747
|
+
ORDER BY s.duration_s DESC LIMIT 10""",
|
|
748
|
+
p + [rules["idle_gap_minutes"] * 60 * 4])
|
|
749
|
+
for r in idle:
|
|
750
|
+
r["excess"] = 0.0
|
|
751
|
+
if idle:
|
|
752
|
+
add("low", "stale_sessions", f"{len(idle)} long-lived sessions with sparse activity",
|
|
753
|
+
"Resuming a stale session re-sends aged context that is often no longer relevant.",
|
|
754
|
+
idle, "Start a fresh session for a new task rather than resuming an old one.",
|
|
755
|
+
"session_id", "none claimed — flagged for review only")
|
|
756
|
+
|
|
757
|
+
findings.sort(key=lambda x: ({"high": 0, "medium": 1, "low": 2}[x["severity"]],
|
|
758
|
+
-x["est_excess_usd"]))
|
|
759
|
+
|
|
760
|
+
pids, sids = set(), set()
|
|
761
|
+
for fnd in findings:
|
|
762
|
+
pids.update(fnd["affected"].get("prompt_id", []))
|
|
763
|
+
sids.update(fnd["affected"].get("session_id", []))
|
|
764
|
+
for e in fnd["evidence"]:
|
|
765
|
+
pids.update(e.get("affected_ids", []))
|
|
766
|
+
exposed = 0.0
|
|
767
|
+
if pids:
|
|
768
|
+
exposed += self.one("SELECT COALESCE(SUM(est_cost_usd),0) c FROM prompts WHERE id IN (%s)"
|
|
769
|
+
% ",".join("?" * len(pids)), tuple(pids))["c"]
|
|
770
|
+
if sids:
|
|
771
|
+
exposed += self.one(
|
|
772
|
+
"SELECT COALESCE(SUM(est_cost_usd),0) c FROM requests WHERE session_id IN (%s)"
|
|
773
|
+
% ",".join("?" * len(sids)) +
|
|
774
|
+
(" AND (prompt_id IS NULL OR prompt_id NOT IN (%s))" % ",".join("?" * len(pids))
|
|
775
|
+
if pids else ""),
|
|
776
|
+
tuple(sids) + tuple(pids))["c"]
|
|
777
|
+
|
|
778
|
+
total = self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c FROM requests r WHERE {w}", p)["c"]
|
|
779
|
+
exposed = min(exposed, total)
|
|
780
|
+
|
|
781
|
+
# De-duplicate excess by entity rather than summing across rules: an item caught
|
|
782
|
+
# by two rules is one item. Per entity take the largest excess any rule claimed,
|
|
783
|
+
# then drop prompt-level excess for prompts that sit inside an already-counted
|
|
784
|
+
# session, since the session estimate already covers them.
|
|
785
|
+
p_excess, s_excess = {}, {}
|
|
786
|
+
for fnd in findings:
|
|
787
|
+
for e in fnd["evidence"]:
|
|
788
|
+
ex = min(max(e.get("excess") or 0, 0), e.get("cost") or 0)
|
|
789
|
+
if e.get("prompt_id"):
|
|
790
|
+
for pid_ in (e.get("affected_ids") or [e["prompt_id"]]):
|
|
791
|
+
p_excess[pid_] = max(p_excess.get(pid_, 0.0), ex / max(
|
|
792
|
+
len(e.get("affected_ids") or [1]), 1))
|
|
793
|
+
elif e.get("session_id"):
|
|
794
|
+
s_excess[e["session_id"]] = max(s_excess.get(e["session_id"], 0.0), ex)
|
|
795
|
+
counted_sessions = set(s_excess)
|
|
796
|
+
outside = 0.0
|
|
797
|
+
if p_excess:
|
|
798
|
+
owner = {r["id"]: r["session_id"] for r in self.q(
|
|
799
|
+
"SELECT id, session_id FROM prompts WHERE id IN (%s)"
|
|
800
|
+
% ",".join("?" * len(p_excess)), tuple(p_excess))}
|
|
801
|
+
outside = sum(v for pid_, v in p_excess.items()
|
|
802
|
+
if owner.get(pid_) not in counted_sessions)
|
|
803
|
+
excess = min(sum(s_excess.values()) + outside, exposed)
|
|
804
|
+
return {"findings": findings,
|
|
805
|
+
"estimated_excess_usd": excess,
|
|
806
|
+
"excess_pct": round(100.0 * excess / total, 1) if total else 0,
|
|
807
|
+
"exposed_cost_usd": exposed,
|
|
808
|
+
"exposed_pct": round(100.0 * exposed / total, 1) if total else 0,
|
|
809
|
+
# kept for compatibility with older callers
|
|
810
|
+
"flagged_cost_usd": exposed,
|
|
811
|
+
"flagged_pct": round(100.0 * exposed / total, 1) if total else 0,
|
|
812
|
+
"total_cost_usd": total,
|
|
813
|
+
"affected_prompts": len(pids), "affected_sessions": len(sids),
|
|
814
|
+
"note": "Exposed spend is the de-duplicated total cost of everything a rule "
|
|
815
|
+
"touched — money worth reviewing. Estimated excess is how much more that "
|
|
816
|
+
"work cost than a reasonable baseline, and is the actual waste figure. "
|
|
817
|
+
"Both are estimates.",
|
|
818
|
+
"basis": "estimated"}
|
|
819
|
+
|
|
820
|
+
# ---------------- recommendations ----------------
|
|
821
|
+
# ---------------- model switch advisor ----------------
|
|
822
|
+
# Which work needs the frontier model and which does not. "keep" = reasoning-heavy work
|
|
823
|
+
# where a cheaper model is a quality risk; the rest is routed down a tier.
|
|
824
|
+
SWITCH_RULES = {
|
|
825
|
+
"casual": ("economy", "high"), "other": ("economy", "medium"),
|
|
826
|
+
"documentation": ("balanced", "high"), "writing": ("balanced", "high"),
|
|
827
|
+
"learning": ("balanced", "high"), "testing": ("balanced", "medium"),
|
|
828
|
+
"research": ("balanced", "medium"), "data_analysis": ("balanced", "medium"),
|
|
829
|
+
"automation": ("balanced", "medium"), "coding": ("balanced", "low"),
|
|
830
|
+
"refactoring": ("balanced", "low"),
|
|
831
|
+
"debugging": ("keep", None), "architecture": ("keep", None),
|
|
832
|
+
"planning": ("keep", None), "code_review": ("keep", None),
|
|
833
|
+
}
|
|
834
|
+
TIER_RANK = {"economy": 0, "balanced": 1, "frontier": 2}
|
|
835
|
+
|
|
836
|
+
# How each vendor's agent is told to change model. Used by model_switch() and by the
|
|
837
|
+
# model_downgrade recommendations, which must name the agent they keep you inside.
|
|
838
|
+
SWITCH_HOW_BY_AGENT = {
|
|
839
|
+
"anthropic": {"agent": "Claude Code", "session": "/model <name>",
|
|
840
|
+
"project": '"model": "<name>" in <repo>/.claude/settings.json'},
|
|
841
|
+
"openai": {"agent": "Codex", "session": "/model in Codex, or codex -m <name>",
|
|
842
|
+
"project": 'model = "<name>" in ~/.codex/config.toml (or a profile)'},
|
|
843
|
+
"google": {"agent": "Gemini CLI", "session": "/model in Gemini CLI, or gemini -m <name>",
|
|
844
|
+
"project": '"model": {"name": "<name>"} in <repo>/.gemini/settings.json'},
|
|
845
|
+
}
|
|
846
|
+
|
|
847
|
+
def _cheapest(self, tier, provider=None):
|
|
848
|
+
"""Cheapest model in a tier — from the same provider, since an agent can only
|
|
849
|
+
switch between its own vendor's models (Codex can't run Haiku)."""
|
|
850
|
+
ms = [m for m, v in self.pricing.models.items() if v.get("tier") == tier
|
|
851
|
+
and (provider is None or v.get("provider", "anthropic") == provider)]
|
|
852
|
+
return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
|
|
853
|
+
|
|
854
|
+
def model_switch(self, f=None):
|
|
855
|
+
"""Per-request what-if: reprice each request on the model its work needs.
|
|
856
|
+
|
|
857
|
+
Token counts are held constant (actual); costs on both sides are estimated at the
|
|
858
|
+
configured prices. Requests whose context exceeds the target's window stay put.
|
|
859
|
+
"""
|
|
860
|
+
w, p = self.where(f)
|
|
861
|
+
rows = self.q(f"""SELECT r.model, r.is_sidechain side, r.agent_type,
|
|
862
|
+
COALESCE(pr.category,'other') category, r.context_tokens ctx,
|
|
863
|
+
pj.id project_id, pj.name project,
|
|
864
|
+
r.prompt_id, r.session_id, r.input_tokens i, r.output_tokens o,
|
|
865
|
+
r.cache_read_tokens cr, r.cache_write_5m c5, r.cache_write_1h c1, r.est_cost_usd cost
|
|
866
|
+
FROM requests r LEFT JOIN prompts pr ON pr.id=r.prompt_id
|
|
867
|
+
JOIN projects pj ON pj.id=r.project_id WHERE {w}""", p)
|
|
868
|
+
target_of = {}
|
|
869
|
+
agents, proj_prov = set(), {}
|
|
870
|
+
groups, projects = {}, defaultdict(lambda: defaultdict(float))
|
|
871
|
+
total = blocked = 0.0
|
|
872
|
+
blocked_n = 0
|
|
873
|
+
for r in rows:
|
|
874
|
+
cost = r["cost"] or 0.0
|
|
875
|
+
total += cost
|
|
876
|
+
tier = self.pricing.tier(r["model"])
|
|
877
|
+
if tier not in self.TIER_RANK:
|
|
878
|
+
continue
|
|
879
|
+
if r["side"]:
|
|
880
|
+
is_explore = (r["agent_type"] or "").lower() == "explore"
|
|
881
|
+
want, conf = ("economy", "high") if is_explore else ("balanced", "medium")
|
|
882
|
+
scope = f"Subagent: {r['agent_type'] or 'general'}"
|
|
883
|
+
else:
|
|
884
|
+
want, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
885
|
+
scope = f"Prompts: {r['category'].replace('_', ' ')}"
|
|
886
|
+
pj = projects[(r["project_id"], r["project"])]
|
|
887
|
+
pj["cost"] += cost
|
|
888
|
+
if tier == "frontier":
|
|
889
|
+
pj["frontier_cost"] += cost
|
|
890
|
+
proj_prov[(r["project_id"], r["project"])] = \
|
|
891
|
+
self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
892
|
+
if want == "keep" or self.TIER_RANK[want] >= self.TIER_RANK[tier]:
|
|
893
|
+
if want == "keep" and tier == "frontier":
|
|
894
|
+
pj["keep_cost"] += cost
|
|
895
|
+
continue
|
|
896
|
+
prov = self.pricing.rates(r["model"]).get("provider", "anthropic")
|
|
897
|
+
if (want, prov) not in target_of:
|
|
898
|
+
target_of[(want, prov)] = self._cheapest(want, prov)
|
|
899
|
+
tgt = target_of[(want, prov)]
|
|
900
|
+
agents.add(prov)
|
|
901
|
+
if not tgt:
|
|
902
|
+
continue
|
|
903
|
+
win = self.pricing.context_window(tgt) or 0
|
|
904
|
+
if win and (r["ctx"] or 0) > win * 0.9:
|
|
905
|
+
blocked += cost
|
|
906
|
+
blocked_n += 1
|
|
907
|
+
continue
|
|
908
|
+
alt = self.pricing.estimate(tgt, r["i"] or 0, r["o"] or 0, r["cr"] or 0,
|
|
909
|
+
r["c5"] or 0, r["c1"] or 0)
|
|
910
|
+
if alt >= cost:
|
|
911
|
+
continue
|
|
912
|
+
key = (scope, r["model"], tgt)
|
|
913
|
+
g = groups.setdefault(key, {"scope": scope, "current_model": r["model"],
|
|
914
|
+
"recommended_model": tgt, "confidence": conf,
|
|
915
|
+
"requests": 0, "prompts": set(), "sessions": set(),
|
|
916
|
+
"cost": 0.0, "alt": 0.0})
|
|
917
|
+
g["requests"] += 1
|
|
918
|
+
g["prompts"].add(r["prompt_id"])
|
|
919
|
+
g["sessions"].add(r["session_id"])
|
|
920
|
+
g["cost"] += cost
|
|
921
|
+
g["alt"] += alt
|
|
922
|
+
pj["savings"] += cost - alt
|
|
923
|
+
|
|
924
|
+
out = []
|
|
925
|
+
for g in groups.values():
|
|
926
|
+
save = g["cost"] - g["alt"]
|
|
927
|
+
if save < 0.5:
|
|
928
|
+
continue
|
|
929
|
+
out.append({**g, "prompts": len(g["prompts"] - {None}), "sessions": len(g["sessions"]),
|
|
930
|
+
"current_name": self.pricing.display_name(g["current_model"]),
|
|
931
|
+
"recommended_name": self.pricing.display_name(g["recommended_model"]),
|
|
932
|
+
"estimated_savings_usd": save,
|
|
933
|
+
"estimated_savings_pct": round(100 * save / g["cost"], 1) if g["cost"] else 0})
|
|
934
|
+
out.sort(key=lambda x: -x["estimated_savings_usd"])
|
|
935
|
+
|
|
936
|
+
by_conf = defaultdict(float)
|
|
937
|
+
for g in out:
|
|
938
|
+
by_conf[g["confidence"]] += g["estimated_savings_usd"]
|
|
939
|
+
|
|
940
|
+
proj = []
|
|
941
|
+
for (pid, name), v in projects.items():
|
|
942
|
+
if v["frontier_cost"] < 1:
|
|
943
|
+
continue
|
|
944
|
+
balanced = self._cheapest("balanced", proj_prov.get((pid, name), "anthropic"))
|
|
945
|
+
keep_pct = round(100 * v["keep_cost"] / v["frontier_cost"], 1)
|
|
946
|
+
default = "keep" if keep_pct >= 50 else "switch"
|
|
947
|
+
proj.append({"project": name, "project_id": pid, "cost": v["cost"],
|
|
948
|
+
"frontier_cost": v["frontier_cost"], "keep_pct": keep_pct,
|
|
949
|
+
"estimated_savings_usd": v["savings"],
|
|
950
|
+
"suggested_default": (self.pricing.display_name(balanced)
|
|
951
|
+
if default == "switch" and balanced else "Keep current"),
|
|
952
|
+
"why": (f"{keep_pct}% of frontier spend here is debugging/architecture/"
|
|
953
|
+
f"planning/review, which benefits from the top model."
|
|
954
|
+
if default == "keep" else
|
|
955
|
+
f"Only {keep_pct}% of frontier spend here is reasoning-heavy work. "
|
|
956
|
+
f"Make {self.pricing.display_name(balanced)} the default and "
|
|
957
|
+
f"switch up with /model only for hard problems.")})
|
|
958
|
+
proj.sort(key=lambda x: -x["estimated_savings_usd"])
|
|
959
|
+
|
|
960
|
+
return {
|
|
961
|
+
"total_cost_usd": total,
|
|
962
|
+
"switches": out,
|
|
963
|
+
"projects": proj,
|
|
964
|
+
"savings_by_confidence": dict(by_conf),
|
|
965
|
+
"estimated_savings_usd": sum(by_conf.values()),
|
|
966
|
+
"safe_savings_usd": by_conf.get("high", 0) + by_conf.get("medium", 0),
|
|
967
|
+
"blocked_by_context_usd": blocked, "blocked_by_context_requests": blocked_n,
|
|
968
|
+
"rules": {k: v[0] for k, v in self.SWITCH_RULES.items()},
|
|
969
|
+
"how": {"session": "/model <name> in Claude Code",
|
|
970
|
+
"project": '"model": "<name>" in <repo>/.claude/settings.json',
|
|
971
|
+
"subagent": "model: haiku (or sonnet) in the agent's frontmatter in .claude/agents/"},
|
|
972
|
+
"how_by_agent": self.SWITCH_HOW_BY_AGENT,
|
|
973
|
+
"providers": sorted(agents),
|
|
974
|
+
"caveat": "Same token counts repriced on the cheaper model. Output quality and any "
|
|
975
|
+
"extra turns a cheaper model might need are not modelled. Try it on a "
|
|
976
|
+
"sample of work before switching everything.",
|
|
977
|
+
"basis": "recommendation",
|
|
978
|
+
}
|
|
979
|
+
|
|
980
|
+
def recommendations(self, f=None):
|
|
981
|
+
w, p = self.where(f)
|
|
982
|
+
recs = []
|
|
983
|
+
|
|
984
|
+
# Frontier work a cheaper model could have done. Candidates are always from the
|
|
985
|
+
# same vendor: an agent can only switch within its own family (Codex can't run
|
|
986
|
+
# Haiku), so mixed frontier spend is split per vendor before anything is compared.
|
|
987
|
+
# Each recommendation offers the ladder — one step down (balanced) and the floor
|
|
988
|
+
# (economy) — priced separately, because that trade-off is the user's to make.
|
|
989
|
+
frontier_by_provider = defaultdict(list)
|
|
990
|
+
for m, v in self.pricing.models.items():
|
|
991
|
+
if v.get("tier") == "frontier":
|
|
992
|
+
frontier_by_provider[v.get("provider", "anthropic")].append(m)
|
|
993
|
+
|
|
994
|
+
for prov, models in sorted(frontier_by_provider.items()):
|
|
995
|
+
cands = []
|
|
996
|
+
for tier in ("balanced", "economy"):
|
|
997
|
+
c = self._cheapest(tier, prov)
|
|
998
|
+
if c and c not in cands:
|
|
999
|
+
cands.append(c)
|
|
1000
|
+
if not cands:
|
|
1001
|
+
continue # this vendor exposes nothing cheaper to move to
|
|
1002
|
+
ph = ",".join("?" * len(models))
|
|
1003
|
+
rows = self.q(f"""SELECT pr.category, COUNT(DISTINCT pr.id) prompts,
|
|
1004
|
+
GROUP_CONCAT(DISTINCT r.model) mods,
|
|
1005
|
+
SUM(r.est_cost_usd) cost, SUM(r.input_tokens) i,
|
|
1006
|
+
SUM(r.output_tokens) o, SUM(r.cache_read_tokens) cr,
|
|
1007
|
+
SUM(r.cache_write_5m) c5, SUM(r.cache_write_1h) c1
|
|
1008
|
+
FROM prompts pr JOIN requests r ON r.prompt_id=pr.id
|
|
1009
|
+
WHERE {w} AND r.model IN ({ph})
|
|
1010
|
+
GROUP BY pr.category HAVING prompts >= 3 AND cost > 0.5
|
|
1011
|
+
ORDER BY cost DESC""", p + models)
|
|
1012
|
+
for r in rows:
|
|
1013
|
+
# The same rules the Model switch dashboard applies, so the two pages can
|
|
1014
|
+
# never contradict each other: work the rules say to keep on a frontier
|
|
1015
|
+
# model is not offered a downgrade at all.
|
|
1016
|
+
target, conf = self.SWITCH_RULES.get(r["category"], ("balanced", "low"))
|
|
1017
|
+
if target == "keep":
|
|
1018
|
+
continue
|
|
1019
|
+
alts = []
|
|
1020
|
+
for m in cands:
|
|
1021
|
+
alt = self.pricing.estimate(m, r["i"], r["o"], r["cr"], r["c5"], r["c1"])
|
|
1022
|
+
if alt >= r["cost"] * 0.9:
|
|
1023
|
+
continue # too close to the current cost to be worth the quality risk
|
|
1024
|
+
alts.append({
|
|
1025
|
+
"model": m, "name": self.pricing.display_name(m),
|
|
1026
|
+
"tier": self.pricing.tier(m),
|
|
1027
|
+
"estimated_cost_usd": alt,
|
|
1028
|
+
"estimated_savings_usd": r["cost"] - alt,
|
|
1029
|
+
"estimated_savings_pct": round(100.0 * (r["cost"] - alt) / r["cost"], 1),
|
|
1030
|
+
})
|
|
1031
|
+
if not alts:
|
|
1032
|
+
continue
|
|
1033
|
+
# Safest step first: balanced before economy, so the ladder reads as
|
|
1034
|
+
# increasing saving and increasing risk.
|
|
1035
|
+
alts.sort(key=lambda a: -self.TIER_RANK.get(a["tier"], 0))
|
|
1036
|
+
for a in alts:
|
|
1037
|
+
a["suggested"] = a["tier"] == target
|
|
1038
|
+
# The headline is the tier the rules actually recommend for this kind of
|
|
1039
|
+
# work, not simply the smallest step; the rest stay on offer below it.
|
|
1040
|
+
head = next((a for a in alts if a["suggested"]), alts[0])
|
|
1041
|
+
agent = self.SWITCH_HOW_BY_AGENT.get(prov, {}).get("agent", prov)
|
|
1042
|
+
recs.append({
|
|
1043
|
+
"type": "model_downgrade",
|
|
1044
|
+
"confidence": conf or "low",
|
|
1045
|
+
"title": f"Consider a cheaper {agent} model for '{r['category']}' work",
|
|
1046
|
+
"current_model": ", ".join(self.pricing.display_name(m)
|
|
1047
|
+
for m in (r["mods"] or "").split(",") if m),
|
|
1048
|
+
"recommended_model": head["name"],
|
|
1049
|
+
"provider": prov,
|
|
1050
|
+
"agent": agent,
|
|
1051
|
+
"alternatives": alts,
|
|
1052
|
+
"scope": f"{r['prompts']} prompts categorized as {r['category']}",
|
|
1053
|
+
"actual_cost_usd": r["cost"],
|
|
1054
|
+
"estimated_alternative_cost_usd": head["estimated_cost_usd"],
|
|
1055
|
+
"estimated_savings_usd": head["estimated_savings_usd"],
|
|
1056
|
+
"estimated_savings_pct": head["estimated_savings_pct"],
|
|
1057
|
+
"caveat": f"Both options stay inside {agent}, so this is a setting change, not "
|
|
1058
|
+
"a change of agent. Assumes identical token usage on the cheaper "
|
|
1059
|
+
"model. Output quality is not modelled — validate on a sample "
|
|
1060
|
+
"before switching.",
|
|
1061
|
+
"basis": "recommendation",
|
|
1062
|
+
})
|
|
1063
|
+
# Biggest opportunity first, now that several vendors can each contribute one.
|
|
1064
|
+
recs.sort(key=lambda r: -r["estimated_savings_usd"])
|
|
1065
|
+
|
|
1066
|
+
eff = self.efficiency(f)
|
|
1067
|
+
c = eff["cache"]
|
|
1068
|
+
if c["reads"] and c["estimated_savings_usd"] > 0:
|
|
1069
|
+
recs.append({
|
|
1070
|
+
"type": "cache_working", "confidence": "high",
|
|
1071
|
+
"title": "Prompt caching is already saving money — keep sessions long-lived",
|
|
1072
|
+
"actual_cost_usd": c["cost_with_cache"],
|
|
1073
|
+
"estimated_alternative_cost_usd": c["cost_without_cache"],
|
|
1074
|
+
"estimated_savings_usd": c["estimated_savings_usd"],
|
|
1075
|
+
"estimated_savings_pct": c["savings_pct"],
|
|
1076
|
+
"caveat": "Savings vs a hypothetical no-cache baseline at configured list prices.",
|
|
1077
|
+
"basis": "recommendation",
|
|
1078
|
+
})
|
|
1079
|
+
|
|
1080
|
+
ctx = self.context_analysis(f)
|
|
1081
|
+
if ctx["large_context_cost_pct"] > 15:
|
|
1082
|
+
recs.append({
|
|
1083
|
+
"type": "context_reduction", "confidence": "medium",
|
|
1084
|
+
"title": f"{ctx['large_context_cost_pct']}% of spend comes from >"
|
|
1085
|
+
f"{ctx['threshold']//1000}K-context requests",
|
|
1086
|
+
"scope": f"{ctx['large_context_requests']:,} requests",
|
|
1087
|
+
"actual_cost_usd": ctx["large_context_cost"],
|
|
1088
|
+
"estimated_savings_usd": ctx["large_context_cost"] * 0.2,
|
|
1089
|
+
"estimated_savings_pct": 20.0,
|
|
1090
|
+
"caveat": "Assumes a 20% context reduction is achievable via /compact and tighter "
|
|
1091
|
+
"file scoping. Not a measured saving.",
|
|
1092
|
+
"basis": "recommendation",
|
|
1093
|
+
})
|
|
1094
|
+
recs.sort(key=lambda r: -(r.get("estimated_savings_usd") or 0))
|
|
1095
|
+
return {"recommendations": recs,
|
|
1096
|
+
"total_estimated_savings_usd": sum(r.get("estimated_savings_usd") or 0
|
|
1097
|
+
for r in recs if r["type"] != "cache_working"),
|
|
1098
|
+
"basis": "recommendation"}
|
|
1099
|
+
|
|
1100
|
+
# ---------------- forecast ----------------
|
|
1101
|
+
def forecast(self, f=None):
|
|
1102
|
+
w, p = self.where(f)
|
|
1103
|
+
bp = self.billing_period()
|
|
1104
|
+
rows = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost, SUM(r.billable_tokens) tokens
|
|
1105
|
+
FROM requests r WHERE {w} AND r.day <> '' GROUP BY 1 ORDER BY 1""", p)
|
|
1106
|
+
if not rows:
|
|
1107
|
+
return {"available": False, "message": "No usage in the selected range."}
|
|
1108
|
+
recent = rows[-14:]
|
|
1109
|
+
costs = [r["cost"] for r in recent]
|
|
1110
|
+
mean = statistics.fmean(costs)
|
|
1111
|
+
sd = statistics.pstdev(costs) if len(costs) > 1 else 0.0
|
|
1112
|
+
in_period = [r for r in rows if bp["start"] <= r["day"] <= bp["end"]]
|
|
1113
|
+
used = sum(r["cost"] for r in in_period)
|
|
1114
|
+
used_tok = sum(r["tokens"] for r in in_period)
|
|
1115
|
+
left = bp["remaining_days"]
|
|
1116
|
+
|
|
1117
|
+
def band(rate):
|
|
1118
|
+
return {"daily_rate": rate, "end_of_period_cost": used + rate * left}
|
|
1119
|
+
|
|
1120
|
+
scenarios = {
|
|
1121
|
+
"conservative": band(max(mean - sd, 0)),
|
|
1122
|
+
"expected": band(mean),
|
|
1123
|
+
"high": band(mean + sd),
|
|
1124
|
+
}
|
|
1125
|
+
tok_mean = statistics.fmean([r["tokens"] for r in recent])
|
|
1126
|
+
today_rows = [r for r in rows if r["day"] == bp["today"]]
|
|
1127
|
+
hours = max(datetime.now(timezone.utc).hour, 1)
|
|
1128
|
+
eod = (today_rows[0]["cost"] / hours * 24) if today_rows else mean
|
|
1129
|
+
|
|
1130
|
+
wk_start = (_d(bp["today"]) - timedelta(days=_d(bp["today"]).weekday())).isoformat()
|
|
1131
|
+
wk_used = sum(r["cost"] for r in rows if r["day"] >= wk_start)
|
|
1132
|
+
wk_left = 6 - _d(bp["today"]).weekday()
|
|
1133
|
+
|
|
1134
|
+
out = {
|
|
1135
|
+
"available": True,
|
|
1136
|
+
"method": "14-day mean daily spend with ±1 standard deviation bands",
|
|
1137
|
+
"sample_days": len(recent),
|
|
1138
|
+
"daily_mean": mean, "daily_stdev": sd,
|
|
1139
|
+
"period_used": used, "period_used_tokens": used_tok,
|
|
1140
|
+
"remaining_days": left,
|
|
1141
|
+
"scenarios": scenarios,
|
|
1142
|
+
"end_of_day_cost": eod,
|
|
1143
|
+
"end_of_week_cost": wk_used + mean * max(wk_left, 0),
|
|
1144
|
+
"end_of_period_tokens": used_tok + tok_mean * left,
|
|
1145
|
+
"estimated_monthly_cost": used + mean * left,
|
|
1146
|
+
"basis": "forecast",
|
|
1147
|
+
}
|
|
1148
|
+
allowance = self.settings["limits"].get("monthly_cost_allowance_usd")
|
|
1149
|
+
if allowance:
|
|
1150
|
+
rem = allowance - used
|
|
1151
|
+
out["limit_exhaustion_date"] = (
|
|
1152
|
+
(_d(bp["today"]) + timedelta(days=rem / mean)).isoformat()
|
|
1153
|
+
if mean > 0 and rem > 0 else bp["today"])
|
|
1154
|
+
out["will_exceed"] = scenarios["expected"]["end_of_period_cost"] > allowance
|
|
1155
|
+
else:
|
|
1156
|
+
out["limit_exhaustion_date"] = UNAVAILABLE
|
|
1157
|
+
out["will_exceed"] = None
|
|
1158
|
+
return out
|
|
1159
|
+
|
|
1160
|
+
# ---------------- budgets ----------------
|
|
1161
|
+
def budgets(self, f=None):
|
|
1162
|
+
b = self.settings["budgets"]
|
|
1163
|
+
bp = self.billing_period()
|
|
1164
|
+
fc = self.forecast(f)
|
|
1165
|
+
w, p = self.where(f)
|
|
1166
|
+
period = self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c,"
|
|
1167
|
+
f" COALESCE(SUM(r.billable_tokens),0) t FROM requests r"
|
|
1168
|
+
f" WHERE {w} AND r.day>=? AND r.day<=?", p + [bp["start"], bp["end"]])
|
|
1169
|
+
today = self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c FROM requests r"
|
|
1170
|
+
f" WHERE {w} AND r.day=?", p + [bp["today"]])
|
|
1171
|
+
out = {"thresholds_pct": self.settings["alert_thresholds_pct"], "lines": [],
|
|
1172
|
+
"basis": "estimated vs configured budget"}
|
|
1173
|
+
|
|
1174
|
+
def line(name, budget, actual, forecast_val, unit="USD"):
|
|
1175
|
+
if not budget:
|
|
1176
|
+
return {"name": name, "configured": False, "message": "No budget configured",
|
|
1177
|
+
"actual": actual, "unit": unit}
|
|
1178
|
+
pct = 100.0 * actual / budget
|
|
1179
|
+
fpct = 100.0 * forecast_val / budget if forecast_val is not None else None
|
|
1180
|
+
breached = [t for t in out["thresholds_pct"] if pct >= t]
|
|
1181
|
+
fbreach = [t for t in out["thresholds_pct"]
|
|
1182
|
+
if fpct is not None and fpct >= t and t not in breached]
|
|
1183
|
+
return {"name": name, "configured": True, "budget": budget, "actual": actual,
|
|
1184
|
+
"forecast": forecast_val,
|
|
1185
|
+
"variance": (forecast_val - budget) if forecast_val is not None else None,
|
|
1186
|
+
"used_pct": round(pct, 1),
|
|
1187
|
+
"forecast_pct": round(fpct, 1) if fpct is not None else None,
|
|
1188
|
+
"status": self._status(fpct if fpct is not None else pct),
|
|
1189
|
+
"thresholds_breached": breached, "thresholds_forecast_breach": fbreach,
|
|
1190
|
+
"unit": unit}
|
|
1191
|
+
|
|
1192
|
+
fc_cost = fc.get("scenarios", {}).get("expected", {}).get("end_of_period_cost")
|
|
1193
|
+
out["lines"].append(line("Monthly spend", b.get("monthly_usd"), period["c"], fc_cost))
|
|
1194
|
+
out["lines"].append(line("Daily spend", b.get("daily_usd"), today["c"], today["c"]))
|
|
1195
|
+
out["lines"].append(line("Monthly tokens", b.get("monthly_tokens"), period["t"],
|
|
1196
|
+
fc.get("end_of_period_tokens"), unit="tokens"))
|
|
1197
|
+
for proj, bud in (b.get("per_project_usd") or {}).items():
|
|
1198
|
+
act = self.one("""SELECT COALESCE(SUM(r.est_cost_usd),0) c FROM requests r
|
|
1199
|
+
JOIN projects pr ON pr.id=r.project_id
|
|
1200
|
+
WHERE pr.name=? AND r.day>=? AND r.day<=?""",
|
|
1201
|
+
(proj, bp["start"], bp["end"]))["c"]
|
|
1202
|
+
out["lines"].append(line(f"Project: {proj}", bud, act, None))
|
|
1203
|
+
for model, bud in (b.get("per_model_usd") or {}).items():
|
|
1204
|
+
act = self.one("""SELECT COALESCE(SUM(est_cost_usd),0) c FROM requests
|
|
1205
|
+
WHERE model=? AND day>=? AND day<=?""",
|
|
1206
|
+
(model, bp["start"], bp["end"]))["c"]
|
|
1207
|
+
out["lines"].append(line(f"Model: {self.pricing.display_name(model)}", bud, act, None))
|
|
1208
|
+
return out
|
|
1209
|
+
|
|
1210
|
+
# ---------------- anomalies ----------------
|
|
1211
|
+
def anomalies(self, f=None):
|
|
1212
|
+
w, p = self.where(f)
|
|
1213
|
+
cfg = self.settings["anomaly"]
|
|
1214
|
+
found = []
|
|
1215
|
+
days = self.q(f"""SELECT r.day, SUM(r.est_cost_usd) cost, SUM(r.billable_tokens) tokens,
|
|
1216
|
+
COUNT(*) requests FROM requests r WHERE {w} AND r.day<>''
|
|
1217
|
+
GROUP BY 1 ORDER BY 1""", p)
|
|
1218
|
+
if len(days) >= 5:
|
|
1219
|
+
vals = [d["cost"] for d in days]
|
|
1220
|
+
mean, sd = statistics.fmean(vals), (statistics.pstdev(vals) or 1e-9)
|
|
1221
|
+
for d in days:
|
|
1222
|
+
z = (d["cost"] - mean) / sd
|
|
1223
|
+
ratio = d["cost"] / mean if mean else 0
|
|
1224
|
+
if z >= cfg["daily_zscore"] and ratio >= cfg["daily_ratio"]:
|
|
1225
|
+
found.append({
|
|
1226
|
+
"severity": "high", "type": "daily_spike", "date": d["day"],
|
|
1227
|
+
"title": f"{d['day']} spend was {ratio:.1f}x your daily average",
|
|
1228
|
+
"detail": f"${d['cost']:,.2f} vs a ${mean:,.2f} daily mean (z={z:.1f}).",
|
|
1229
|
+
"metric_value": d["cost"], "baseline": mean, "ratio": round(ratio, 2),
|
|
1230
|
+
"drilldown": {"filter": {"start": d["day"], "end": d["day"]}},
|
|
1231
|
+
"basis": "estimated",
|
|
1232
|
+
})
|
|
1233
|
+
sess = self.sessions(f, limit=100000, order="cost")
|
|
1234
|
+
if len(sess) >= 5:
|
|
1235
|
+
vals = [s["tokens"] for s in sess]
|
|
1236
|
+
mean = statistics.fmean(vals) or 1
|
|
1237
|
+
for s in sess[:40]:
|
|
1238
|
+
ratio = s["tokens"] / mean
|
|
1239
|
+
if ratio >= cfg["session_ratio"]:
|
|
1240
|
+
found.append({
|
|
1241
|
+
"severity": "medium", "type": "session_outlier",
|
|
1242
|
+
"title": f"Session consumed {ratio:.1f}x the average session tokens",
|
|
1243
|
+
"detail": f"{s['title'] or s['session_id'][:8]} — {s['tokens']:,} tokens, "
|
|
1244
|
+
f"${s['cost']:,.2f} in {s['project']}.",
|
|
1245
|
+
"metric_value": s["tokens"], "baseline": mean, "ratio": round(ratio, 2),
|
|
1246
|
+
"drilldown": {"session_id": s["session_id"]},
|
|
1247
|
+
"basis": "estimated",
|
|
1248
|
+
})
|
|
1249
|
+
# week-over-week model shift
|
|
1250
|
+
if self.last_day:
|
|
1251
|
+
end = _d(self.last_day)
|
|
1252
|
+
cur_s = (end - timedelta(days=6)).isoformat()
|
|
1253
|
+
prev_s, prev_e = (end - timedelta(days=13)).isoformat(), (end - timedelta(days=7)).isoformat()
|
|
1254
|
+
for m in self.q(f"SELECT DISTINCT r.model FROM requests r WHERE {w}", p):
|
|
1255
|
+
model = m["model"]
|
|
1256
|
+
a = self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c FROM requests r WHERE {w}"
|
|
1257
|
+
f" AND r.model=? AND r.day>=?", p + [model, cur_s])["c"]
|
|
1258
|
+
bq = self.one(f"SELECT COALESCE(SUM(r.est_cost_usd),0) c FROM requests r WHERE {w}"
|
|
1259
|
+
f" AND r.model=? AND r.day>=? AND r.day<=?",
|
|
1260
|
+
p + [model, prev_s, prev_e])["c"]
|
|
1261
|
+
if bq > 1 and a > bq * 1.4:
|
|
1262
|
+
found.append({
|
|
1263
|
+
"severity": "medium", "type": "model_shift",
|
|
1264
|
+
"title": f"{self.pricing.display_name(model)} spend rose "
|
|
1265
|
+
f"{100*(a-bq)/bq:.0f}% week over week",
|
|
1266
|
+
"detail": f"${bq:,.2f} → ${a:,.2f}.",
|
|
1267
|
+
"metric_value": a, "baseline": bq, "ratio": round(a / bq, 2),
|
|
1268
|
+
"drilldown": {"filter": {"models": [model], "start": cur_s}},
|
|
1269
|
+
"basis": "estimated",
|
|
1270
|
+
})
|
|
1271
|
+
found.sort(key=lambda x: -x.get("ratio", 0))
|
|
1272
|
+
return {"anomalies": found[:25], "basis": "estimated"}
|
|
1273
|
+
|
|
1274
|
+
# ---------------- scorecard ----------------
|
|
1275
|
+
def scorecard(self, f=None):
|
|
1276
|
+
eff = self.efficiency(f)
|
|
1277
|
+
ctx = self.context_analysis(f)
|
|
1278
|
+
wst = self.waste(f)
|
|
1279
|
+
bud = self.budgets(f)
|
|
1280
|
+
mdl = self.models(f)
|
|
1281
|
+
dims = []
|
|
1282
|
+
|
|
1283
|
+
def dim(name, score, detail, weight=1.0):
|
|
1284
|
+
dims.append({"name": name, "score": max(0, min(100, round(score))),
|
|
1285
|
+
"detail": detail, "weight": weight})
|
|
1286
|
+
|
|
1287
|
+
chr_ = eff["cache_hit_ratio"]
|
|
1288
|
+
if chr_ is None:
|
|
1289
|
+
dim("Cache efficiency", 50, "No cache activity in range", 1.0)
|
|
1290
|
+
else:
|
|
1291
|
+
dim("Cache efficiency", chr_ * 100,
|
|
1292
|
+
f"{chr_*100:.1f}% of cache tokens were reads (reuse) rather than writes.", 1.2)
|
|
1293
|
+
|
|
1294
|
+
sc_cfg = self.settings.get("scorecard", {})
|
|
1295
|
+
target = sc_cfg.get("target_output_ratio", 0.0088)
|
|
1296
|
+
outr = eff["output_ratio"]
|
|
1297
|
+
dim("Token efficiency", min(outr / target, 1.0) * 100,
|
|
1298
|
+
f"Output is {outr*100:.2f}% of billable tokens against a "
|
|
1299
|
+
f"{target*100:.2f}% reference.", 1.2)
|
|
1300
|
+
|
|
1301
|
+
big_pct = ctx["large_context_cost_pct"]
|
|
1302
|
+
dim("Context efficiency", 100 - big_pct,
|
|
1303
|
+
f"{big_pct}% of spend came from requests above "
|
|
1304
|
+
f"{ctx['threshold']//1000}K context.", 1.0)
|
|
1305
|
+
|
|
1306
|
+
excess = wst["excess_pct"]
|
|
1307
|
+
dim("Waste control", 100 - min(excess, 100),
|
|
1308
|
+
f"{excess}% of spend is estimated excess over a reasonable baseline "
|
|
1309
|
+
f"({wst['exposed_pct']}% of spend sits in items a rule touched).", 1.3)
|
|
1310
|
+
|
|
1311
|
+
priced = [r for r in mdl["rows"] if r["tier"] != "none" and r["cost"]]
|
|
1312
|
+
frontier_pct = (100.0 * sum(r["cost"] for r in priced if r["tier"] == "frontier")
|
|
1313
|
+
/ (sum(r["cost"] for r in priced) or 1))
|
|
1314
|
+
allow = sc_cfg.get("frontier_cost_share_allowance_pct", 40)
|
|
1315
|
+
dim("Model selection", 100 - max(frontier_pct - allow, 0) * 1.5,
|
|
1316
|
+
f"{frontier_pct:.0f}% of spend is on frontier-tier models.", 1.1)
|
|
1317
|
+
|
|
1318
|
+
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1319
|
+
if ml and ml.get("configured"):
|
|
1320
|
+
fp = ml.get("forecast_pct") or ml["used_pct"]
|
|
1321
|
+
dim("Budget adherence", 100 - max(fp - 100, 0) * 2 - max(fp - 85, 0),
|
|
1322
|
+
f"Forecast is {fp:.0f}% of the configured monthly budget.", 1.3)
|
|
1323
|
+
else:
|
|
1324
|
+
dim("Budget adherence", 50,
|
|
1325
|
+
"No monthly budget configured — set one in config/settings.json to be graded.", 0.4)
|
|
1326
|
+
|
|
1327
|
+
cpo = eff["cost_per_1k_output"]
|
|
1328
|
+
cpo_target = sc_cfg.get("target_cost_per_1k_output_usd", 0.30)
|
|
1329
|
+
dim("Cost efficiency", 100 - min(cpo / (cpo_target * 2) * 100, 100),
|
|
1330
|
+
f"${cpo:.3f} estimated per 1K output tokens.", 1.0)
|
|
1331
|
+
|
|
1332
|
+
tw = sum(d["weight"] for d in dims)
|
|
1333
|
+
total = round(sum(d["score"] * d["weight"] for d in dims) / tw)
|
|
1334
|
+
strong = sorted(dims, key=lambda d: -d["score"])[:3]
|
|
1335
|
+
weak = sorted(dims, key=lambda d: d["score"])[:3]
|
|
1336
|
+
top = wst["findings"][0] if wst["findings"] else None
|
|
1337
|
+
return {
|
|
1338
|
+
"score": total, "grade": ("A" if total >= 85 else "B" if total >= 70
|
|
1339
|
+
else "C" if total >= 55 else "D" if total >= 40 else "F"),
|
|
1340
|
+
"dimensions": dims,
|
|
1341
|
+
"what_is_good": [f"{d['name']}: {d['detail']}" for d in strong if d["score"] >= 60],
|
|
1342
|
+
"needs_attention": [f"{d['name']}: {d['detail']}" for d in weak if d["score"] < 70],
|
|
1343
|
+
"biggest_opportunity": (
|
|
1344
|
+
{"title": top["title"], "detail": top["detail"],
|
|
1345
|
+
"exposed_cost_usd": top["est_cost_usd"],
|
|
1346
|
+
"estimated_excess_usd": top["est_excess_usd"],
|
|
1347
|
+
"action": top["recommended_action"]}
|
|
1348
|
+
if top else None),
|
|
1349
|
+
"basis": "estimated",
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
# ---------------- advisor ----------------
|
|
1353
|
+
def advisor(self, f=None):
|
|
1354
|
+
actions = []
|
|
1355
|
+
wst = self.waste(f)
|
|
1356
|
+
recs = self.recommendations(f)
|
|
1357
|
+
anos = self.anomalies(f)
|
|
1358
|
+
fc = self.forecast(f)
|
|
1359
|
+
bud = self.budgets(f)
|
|
1360
|
+
ov = self.overview(f)
|
|
1361
|
+
|
|
1362
|
+
for a in anos["anomalies"][:2]:
|
|
1363
|
+
actions.append({"priority": 1, "kind": "anomaly", "text": a["title"],
|
|
1364
|
+
"detail": a["detail"], "drilldown": a.get("drilldown"),
|
|
1365
|
+
"basis": "estimated"})
|
|
1366
|
+
for wf in wst["findings"][:2]:
|
|
1367
|
+
actions.append({"priority": 2, "kind": "waste",
|
|
1368
|
+
"text": wf["title"],
|
|
1369
|
+
"detail": f"{wf['detail']} ~${wf['est_excess_usd']:,.2f} estimated "
|
|
1370
|
+
f"excess across ${wf['est_cost_usd']:,.2f} of exposed spend. "
|
|
1371
|
+
f"{wf['recommended_action']}",
|
|
1372
|
+
"basis": "estimated"})
|
|
1373
|
+
for r in recs["recommendations"][:2]:
|
|
1374
|
+
if r["type"] == "cache_working":
|
|
1375
|
+
continue
|
|
1376
|
+
# Name the models. A saving is meaningless without the swap it assumes, and
|
|
1377
|
+
# the options are what the reader actually has to choose between.
|
|
1378
|
+
opts = " or ".join(f"{a['name']} (~${a['estimated_savings_usd']:,.0f}, "
|
|
1379
|
+
f"{a['estimated_savings_pct']}%)" for a in r.get("alternatives", []))
|
|
1380
|
+
swap = f"{r['current_model']} → {opts}. " if opts else ""
|
|
1381
|
+
actions.append({"priority": 3, "kind": "recommendation", "text": r["title"],
|
|
1382
|
+
"detail": f"{swap}Estimated saving ~${r['estimated_savings_usd']:,.2f} "
|
|
1383
|
+
f"({r['estimated_savings_pct']}%) on the suggested option. "
|
|
1384
|
+
f"{r['caveat']}",
|
|
1385
|
+
"basis": "recommendation"})
|
|
1386
|
+
ml = next((l for l in bud["lines"] if l["name"] == "Monthly spend"), None)
|
|
1387
|
+
if ml and ml.get("configured") and ml.get("forecast_pct"):
|
|
1388
|
+
if ml["forecast_pct"] >= 90:
|
|
1389
|
+
actions.append({
|
|
1390
|
+
"priority": 1, "kind": "budget",
|
|
1391
|
+
"text": f"Forecast to finish the period at {ml['forecast_pct']:.0f}% of budget",
|
|
1392
|
+
"detail": f"${ml['actual']:,.2f} spent, ${ml['forecast']:,.2f} forecast against "
|
|
1393
|
+
f"a ${ml['budget']:,.2f} budget.", "basis": "forecast"})
|
|
1394
|
+
# concentration
|
|
1395
|
+
pr = self.prompts(f, limit=3, order="cost")
|
|
1396
|
+
if pr and ov["est_cost_usd"]:
|
|
1397
|
+
share = 100.0 * sum(x["pcost"] for x in pr) / ov["est_cost_usd"]
|
|
1398
|
+
if share >= 8:
|
|
1399
|
+
actions.append({
|
|
1400
|
+
"priority": 2, "kind": "concentration",
|
|
1401
|
+
"text": f"3 prompts account for {share:.0f}% of spend in range",
|
|
1402
|
+
"detail": "; ".join(f"\"{x['preview'][:60]}…\" (${x['pcost']:,.2f})" for x in pr),
|
|
1403
|
+
"basis": "estimated"})
|
|
1404
|
+
actions.sort(key=lambda a: a["priority"])
|
|
1405
|
+
savings = recs["total_estimated_savings_usd"]
|
|
1406
|
+
return {
|
|
1407
|
+
"question": "What should I do today?",
|
|
1408
|
+
"actions": actions[:6],
|
|
1409
|
+
"estimated_savings_range_usd": [round(savings * 0.6, 2), round(savings, 2)],
|
|
1410
|
+
"generated_from": "Live dashboard data for the current filter selection.",
|
|
1411
|
+
"basis": "mixed: see per-item basis",
|
|
1412
|
+
}
|
|
1413
|
+
|
|
1414
|
+
# ---------------- prompt-level advisor ----------------
|
|
1415
|
+
def prompt_advisor(self, p):
|
|
1416
|
+
"""Deterministic, evidence-based analysis of one prompt. Estimates only."""
|
|
1417
|
+
reasons, suggestions = [], []
|
|
1418
|
+
chars = p.get("char_len") or 0
|
|
1419
|
+
ctx = p.get("max_context_tokens") or 0
|
|
1420
|
+
tools = p.get("tool_calls") or 0
|
|
1421
|
+
out = p.get("output_tokens") or 0
|
|
1422
|
+
tot = p.get("billable_tokens") or 0
|
|
1423
|
+
reduction = 0.0
|
|
1424
|
+
if chars > 4000:
|
|
1425
|
+
reasons.append(f"The prompt itself is {chars:,} characters, which is cached and "
|
|
1426
|
+
f"re-sent on every follow-up turn.")
|
|
1427
|
+
suggestions.append("Move long pasted content into a file and reference the path.")
|
|
1428
|
+
reduction += 0.10
|
|
1429
|
+
if ctx > 150000:
|
|
1430
|
+
reasons.append(f"It ran with up to {ctx:,} context tokens per request.")
|
|
1431
|
+
suggestions.append("Run /compact or start a fresh session before a task this large.")
|
|
1432
|
+
reduction += 0.25
|
|
1433
|
+
if tools > 40:
|
|
1434
|
+
reasons.append(f"It triggered {tools} tool calls; each one re-sends the conversation.")
|
|
1435
|
+
suggestions.append("Split into smaller, explicitly scoped sub-tasks.")
|
|
1436
|
+
reduction += 0.15
|
|
1437
|
+
if tot and out / tot < 0.01:
|
|
1438
|
+
reasons.append(f"Only {100*out/tot:.2f}% of the tokens were output — most of the "
|
|
1439
|
+
f"cost was re-reading context.")
|
|
1440
|
+
suggestions.append("Narrow the files and history in scope before asking.")
|
|
1441
|
+
reduction += 0.10
|
|
1442
|
+
models = (p.get("models") or "")
|
|
1443
|
+
if "opus" in models and out < 400:
|
|
1444
|
+
reasons.append("A frontier-tier model produced a short answer.")
|
|
1445
|
+
suggestions.append("Route short turns to a cheaper model tier.")
|
|
1446
|
+
reduction += 0.20
|
|
1447
|
+
if not reasons:
|
|
1448
|
+
return {"available": False,
|
|
1449
|
+
"message": "No cost-driver pattern detected for this prompt."}
|
|
1450
|
+
reduction = min(reduction, 0.6)
|
|
1451
|
+
return {
|
|
1452
|
+
"available": True,
|
|
1453
|
+
"why_expensive": reasons,
|
|
1454
|
+
"suggestions": suggestions,
|
|
1455
|
+
"estimated_token_reduction_pct": round(reduction * 100),
|
|
1456
|
+
"estimated_cost_reduction_pct": round(reduction * 100 * 0.85),
|
|
1457
|
+
"estimated_cost_reduction_usd": round((p.get("est_cost_usd") or 0) * reduction * 0.85, 2),
|
|
1458
|
+
"disclaimer": "ESTIMATE from structural heuristics. Not a measured saving and not a "
|
|
1459
|
+
"guarantee of equivalent output quality.",
|
|
1460
|
+
"basis": "recommendation",
|
|
1461
|
+
}
|
|
1462
|
+
|
|
1463
|
+
# ---------------- claude code / developer ----------------
|
|
1464
|
+
def developer(self, f=None):
|
|
1465
|
+
w, p = self.where(f)
|
|
1466
|
+
tools = self.q(f"""SELECT t.name, COUNT(*) calls, COUNT(DISTINCT t.session_id) sessions
|
|
1467
|
+
FROM tool_calls t JOIN requests r ON r.id = t.request_pk
|
|
1468
|
+
WHERE {w} GROUP BY 1 ORDER BY 2 DESC""", p)
|
|
1469
|
+
files = self.q(f"""SELECT ft.path, COUNT(*) touches, COUNT(DISTINCT ft.session_id) sessions,
|
|
1470
|
+
GROUP_CONCAT(DISTINCT ft.op) ops
|
|
1471
|
+
FROM files_touched ft WHERE ft.session_id IN
|
|
1472
|
+
(SELECT DISTINCT r.session_id FROM requests r WHERE {w})
|
|
1473
|
+
GROUP BY 1 ORDER BY 2 DESC LIMIT 40""", p)
|
|
1474
|
+
branches = self.q(f"""SELECT COALESCE(s.git_branch,'(none)') branch,
|
|
1475
|
+
COUNT(DISTINCT s.id) sessions, SUM(r.est_cost_usd) cost,
|
|
1476
|
+
SUM(r.billable_tokens) tokens
|
|
1477
|
+
FROM requests r JOIN sessions s ON s.id=r.session_id
|
|
1478
|
+
WHERE {w} GROUP BY 1 ORDER BY cost DESC LIMIT 25""", p)
|
|
1479
|
+
repos = [r for r in self.projects(f) if not r["is_sandbox"]]
|
|
1480
|
+
return {
|
|
1481
|
+
"tools": tools, "files": files, "branches": branches, "repositories": repos,
|
|
1482
|
+
"cost_per_repository": [{"repository": r["name"], "cost": r["cost"],
|
|
1483
|
+
"sessions": r["sessions"], "prompts": r["prompts"],
|
|
1484
|
+
"files_touched": r["files_touched"],
|
|
1485
|
+
"cost_per_session": r["avg_cost_per_session"]}
|
|
1486
|
+
for r in repos],
|
|
1487
|
+
"unavailable": {
|
|
1488
|
+
"lines_changed": UNAVAILABLE + " (transcripts record file paths, not diff size)",
|
|
1489
|
+
"commits": UNAVAILABLE, "pull_requests": UNAVAILABLE,
|
|
1490
|
+
"bugs_fixed": UNAVAILABLE,
|
|
1491
|
+
"cost_per_pr": UNAVAILABLE + " — no PR linkage in Claude Code transcripts",
|
|
1492
|
+
},
|
|
1493
|
+
"basis": "actual tool activity, estimated cost",
|
|
1494
|
+
}
|
|
1495
|
+
|
|
1496
|
+
# ---------------- global search ----------------
|
|
1497
|
+
def search(self, term, limit=40):
|
|
1498
|
+
like = f"%{term}%"
|
|
1499
|
+
return {
|
|
1500
|
+
"term": term,
|
|
1501
|
+
"prompts": self.q(
|
|
1502
|
+
"SELECT id prompt_id, ts, category, session_id, substr(text,1,240) preview,"
|
|
1503
|
+
" est_cost_usd, billable_tokens FROM prompts WHERE text LIKE ?"
|
|
1504
|
+
" ORDER BY est_cost_usd DESC LIMIT ?", (like, limit)),
|
|
1505
|
+
"sessions": self.q(
|
|
1506
|
+
"SELECT s.id session_id, s.title, s.git_branch, pr.name project, s.started_at,"
|
|
1507
|
+
" s.est_cost_usd, s.billable_tokens FROM sessions s"
|
|
1508
|
+
" JOIN projects pr ON pr.id=s.project_id"
|
|
1509
|
+
" WHERE s.id LIKE ? OR s.title LIKE ? OR s.git_branch LIKE ?"
|
|
1510
|
+
" ORDER BY s.est_cost_usd DESC LIMIT ?", (like, like, like, limit)),
|
|
1511
|
+
"projects": self.q(
|
|
1512
|
+
"SELECT id project_id, name, path, slug FROM projects"
|
|
1513
|
+
" WHERE name LIKE ? OR path LIKE ? LIMIT ?", (like, like, limit)),
|
|
1514
|
+
"models": self.q(
|
|
1515
|
+
"SELECT model, COUNT(*) requests, SUM(est_cost_usd) cost FROM requests"
|
|
1516
|
+
" WHERE model LIKE ? GROUP BY 1", (like,)),
|
|
1517
|
+
"tools": self.q(
|
|
1518
|
+
"SELECT name, target, COUNT(*) n FROM tool_calls"
|
|
1519
|
+
" WHERE name LIKE ? OR target LIKE ? GROUP BY name, target"
|
|
1520
|
+
" ORDER BY n DESC LIMIT ?", (like, like, limit)),
|
|
1521
|
+
"days": self.q(
|
|
1522
|
+
"SELECT day, COUNT(*) requests, SUM(est_cost_usd) cost FROM requests"
|
|
1523
|
+
" WHERE day LIKE ? GROUP BY 1 ORDER BY 1", (like,)),
|
|
1524
|
+
}
|
|
1525
|
+
|
|
1526
|
+
# ---------------- filter option lists ----------------
|
|
1527
|
+
def agents(self):
|
|
1528
|
+
"""Every agent detected on this machine, with what its local data can show."""
|
|
1529
|
+
from .agents import AGENTS, detect
|
|
1530
|
+
found = detect()
|
|
1531
|
+
used = {r["agent"]: r for r in self.q("""SELECT agent, COUNT(*) requests,
|
|
1532
|
+
COUNT(DISTINCT session_id) sessions, SUM(billable_tokens) tokens,
|
|
1533
|
+
SUM(est_cost_usd) cost, MIN(day) first, MAX(day) last FROM requests GROUP BY agent""")}
|
|
1534
|
+
out = []
|
|
1535
|
+
for k, a in AGENTS.items():
|
|
1536
|
+
u = used.get(k) or {}
|
|
1537
|
+
if not found.get(k) and not u:
|
|
1538
|
+
continue
|
|
1539
|
+
out.append({"id": k, "name": a["name"], "data": a["data"], "note": a["note"],
|
|
1540
|
+
"requests": u.get("requests") or 0, "sessions": u.get("sessions") or 0,
|
|
1541
|
+
"tokens": u.get("tokens") or 0, "cost": u.get("cost") or 0,
|
|
1542
|
+
"first": u.get("first"), "last": u.get("last")})
|
|
1543
|
+
return out
|
|
1544
|
+
|
|
1545
|
+
def by_agent(self, f=None):
|
|
1546
|
+
"""Side-by-side totals per agent for the multi-agent view."""
|
|
1547
|
+
f = dict(f or {})
|
|
1548
|
+
w, p = self.where(f)
|
|
1549
|
+
rows = self.q(f"""SELECT r.agent, COUNT(*) requests, COUNT(DISTINCT r.session_id) sessions,
|
|
1550
|
+
COUNT(DISTINCT r.prompt_id) prompts, SUM(r.billable_tokens) tokens,
|
|
1551
|
+
SUM(r.input_tokens) input_tokens, SUM(r.output_tokens) output_tokens,
|
|
1552
|
+
SUM(r.cache_read_tokens) cache_read_tokens, SUM(r.est_cost_usd) cost,
|
|
1553
|
+
SUM(r.tool_call_count) tool_calls, COUNT(DISTINCT r.day) active_days,
|
|
1554
|
+
GROUP_CONCAT(DISTINCT r.model) models
|
|
1555
|
+
FROM requests r WHERE {w} GROUP BY r.agent ORDER BY cost DESC, requests DESC""", p)
|
|
1556
|
+
daily = self.q(f"""SELECT r.day, r.agent, SUM(r.billable_tokens) tokens, SUM(r.est_cost_usd) cost,
|
|
1557
|
+
COUNT(*) requests FROM requests r WHERE {w} GROUP BY 1, 2 ORDER BY 1""", p)
|
|
1558
|
+
from .agents import AGENTS
|
|
1559
|
+
for r in rows:
|
|
1560
|
+
a = AGENTS.get(r["agent"], {})
|
|
1561
|
+
r["name"], r["data"], r["note"] = a.get("name", r["agent"]), a.get("data"), a.get("note")
|
|
1562
|
+
return {"agents": rows, "daily": daily, "basis": "estimated"}
|
|
1563
|
+
|
|
1564
|
+
def options(self):
|
|
1565
|
+
return {
|
|
1566
|
+
"agents": self.agents(),
|
|
1567
|
+
"models": self.q("SELECT model, agent, COUNT(*) n FROM requests GROUP BY 1 ORDER BY 3 DESC"),
|
|
1568
|
+
"projects": self.q("SELECT pr.id project_id, pr.name, pr.is_sandbox, pr.agent, COUNT(r.id) n"
|
|
1569
|
+
" FROM projects pr LEFT JOIN requests r ON r.project_id=pr.id"
|
|
1570
|
+
" GROUP BY pr.id HAVING n>0 ORDER BY n DESC"),
|
|
1571
|
+
"categories": self.q("SELECT category, COUNT(*) n FROM prompts GROUP BY 1 ORDER BY 2 DESC"),
|
|
1572
|
+
"date_range": {"first": self.first_day, "last": self.last_day},
|
|
1573
|
+
"billing_period": self.billing_period(),
|
|
1574
|
+
"meta": self.meta,
|
|
1575
|
+
"pricing": {"updated": self.pricing.updated, "source": self.pricing.source,
|
|
1576
|
+
"models": self.pricing.models},
|
|
1577
|
+
"settings": self.settings,
|
|
1578
|
+
"unavailable_label": UNAVAILABLE,
|
|
1579
|
+
}
|