claude-finops 0.4.3 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -886,6 +886,155 @@ class Analytics:
886
886
  and (provider is None or v.get("provider", "anthropic") == provider)]
887
887
  return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
888
888
 
889
+ # ---------------- evidence: what the cheaper model actually did ----------------
890
+ # model_switch() reprices your tokens on a cheaper model, which assumes the cheaper
891
+ # model would have done the same work in the same number of turns. Often it would
892
+ # not: a weaker model can take five times the turns on the same task, and the
893
+ # repricing then promises a saving that never arrives.
894
+ #
895
+ # Where you have already run more than one model on the same kind of work, we do not
896
+ # have to assume anything. This compares what each model actually cost per prompt on
897
+ # that category, and how much work it took to get there.
898
+
899
+ MIN_PROMPTS = 8 # below this a per-category average is noise, not evidence
900
+ SAVING_FLOOR_PCT = 20 # smaller gaps are inside the noise of what you happened to ask
901
+ TURN_TOLERANCE = 1.35 # more turns than this and the cheaper model was grinding
902
+ REPEAT_TOLERANCE = 12.0 # percentage points of extra re-asking we will accept
903
+
904
+ def model_evidence(self, f=None):
905
+ """Back-test a model switch against your own history.
906
+
907
+ For every category where you ran more than one model, report what each one
908
+ actually cost per prompt and what it took: turns, tool calls, and how often you
909
+ had to ask the same thing again. A candidate is only recommended when it was
910
+ genuinely cheaper per prompt *and* did not need materially more work to get
911
+ there — which is the part a repricing cannot see.
912
+ """
913
+ w, p = self.where(f)
914
+ # The filter is request-scoped, so select the prompts it touches as a subquery.
915
+ # Joining requests directly would repeat each prompt once per request and quietly
916
+ # multiply both the counts and every average by the turn count.
917
+ scope = f"pr.id IN (SELECT r.prompt_id FROM requests r WHERE {w})"
918
+ # Single-model prompts only: a prompt answered by two models cannot be
919
+ # attributed to either, and mixed rows would blur the comparison.
920
+ clean = ("pr.models IS NOT NULL AND pr.models NOT LIKE '%,%' "
921
+ "AND pr.models != '<synthetic>' AND pr.est_cost_usd > 0")
922
+ rows = self.q(f"""SELECT COALESCE(pr.category,'other') category, pr.models model,
923
+ pr.agent agent, COUNT(*) prompts, SUM(pr.est_cost_usd) cost,
924
+ AVG(pr.est_cost_usd) cost_per_prompt,
925
+ AVG(pr.request_count) turns, AVG(pr.tool_calls) tools,
926
+ AVG(pr.output_tokens) out_tokens, AVG(pr.max_context_tokens) ctx
927
+ FROM prompts pr WHERE {scope} AND {clean}
928
+ GROUP BY 1, 2, 3 HAVING prompts >= ?""", p + [self.MIN_PROMPTS])
929
+
930
+ repeats = {(r["category"], r["model"]): r["pct"] for r in self.q(f"""
931
+ SELECT COALESCE(pr.category,'other') category, pr.models model,
932
+ ROUND(100.0 * SUM(CASE WHEN dup.n > 1 THEN 1 ELSE 0 END) / COUNT(*), 1) pct
933
+ FROM prompts pr
934
+ LEFT JOIN (SELECT norm_hash, COUNT(*) n FROM prompts GROUP BY norm_hash) dup
935
+ ON dup.norm_hash = pr.norm_hash
936
+ WHERE {scope} AND {clean}
937
+ GROUP BY 1, 2""", p)}
938
+
939
+ by_cat = defaultdict(list)
940
+ for r in rows:
941
+ r["repeat_pct"] = repeats.get((r["category"], r["model"]), 0.0) or 0.0
942
+ r["name"] = self.pricing.display_name(r["model"])
943
+ r["tier"] = self.pricing.tier(r["model"])
944
+ by_cat[r["category"]].append(r)
945
+
946
+ out, total_save = [], 0.0
947
+ for cat, models in by_cat.items():
948
+ if len(models) < 2:
949
+ continue
950
+ # The incumbent is what you spend the most on here — that is the bill a
951
+ # switch would actually change.
952
+ cur = max(models, key=lambda m: m["cost"])
953
+ rule = self.SWITCH_RULES.get(cat, ("balanced", "low"))[0]
954
+ cands = []
955
+ for m in models:
956
+ if m["model"] == cur["model"] or m["cost_per_prompt"] >= cur["cost_per_prompt"]:
957
+ continue
958
+ # Only models the same agent can run. Telling a Claude Code user to use a
959
+ # GPT model is not a setting change, it is a different tool, and the
960
+ # comparison would be between two different ways of working.
961
+ if m["agent"] != cur["agent"]:
962
+ continue
963
+ save_pct = 100.0 * (1 - m["cost_per_prompt"] / cur["cost_per_prompt"])
964
+ turn_ratio = (m["turns"] / cur["turns"]) if cur["turns"] else 1.0
965
+ repeat_delta = m["repeat_pct"] - cur["repeat_pct"]
966
+ # Observed, not repriced: what your own prompts cost on each side.
967
+ save = (cur["cost_per_prompt"] - m["cost_per_prompt"]) * cur["prompts"]
968
+ if save_pct < self.SAVING_FLOOR_PCT:
969
+ verdict, why = "marginal", (
970
+ f"Only {save_pct:.0f}% cheaper per prompt — inside the noise of what "
971
+ f"you happened to ask each model.")
972
+ elif turn_ratio > self.TURN_TOLERANCE:
973
+ verdict, why = "risky", (
974
+ f"Cost {save_pct:.0f}% less per prompt but took {turn_ratio:.1f}x the "
975
+ f"turns ({m['turns']:.0f} vs {cur['turns']:.0f}). It got there by "
976
+ f"grinding, and that is the cost the headline number misses.")
977
+ elif repeat_delta > self.REPEAT_TOLERANCE:
978
+ verdict, why = "risky", (
979
+ f"{save_pct:.0f}% cheaper per prompt, but you re-asked "
980
+ f"{m['repeat_pct']:.0f}% of these prompts against "
981
+ f"{cur['repeat_pct']:.0f}% on {cur['name']} — rework you paid for twice.")
982
+ elif rule == "keep":
983
+ verdict, why = "caution", (
984
+ f"{save_pct:.0f}% cheaper per prompt and no more turns, but {cat.replace('_',' ')} "
985
+ f"is reasoning-heavy work where a miss is expensive in ways this data "
986
+ f"cannot show. Worth a trial, not a default.")
987
+ else:
988
+ verdict, why = "supported", (
989
+ f"{save_pct:.0f}% cheaper per prompt on {m['prompts']} of your own "
990
+ f"{cat.replace('_',' ')} prompts, in {turn_ratio:.1f}x the turns "
991
+ f"({m['turns']:.0f} vs {cur['turns']:.0f}) with "
992
+ f"{'less' if repeat_delta <= 0 else 'similar'} re-asking. "
993
+ f"This is measured, not modelled.")
994
+ cands.append({
995
+ "model": m["model"], "name": m["name"], "tier": m["tier"],
996
+ "prompts": m["prompts"], "cost_per_prompt": m["cost_per_prompt"],
997
+ "turns": m["turns"], "tools": m["tools"], "repeat_pct": m["repeat_pct"],
998
+ "savings_pct": round(save_pct, 1), "turn_ratio": round(turn_ratio, 2),
999
+ "repeat_delta": round(repeat_delta, 1),
1000
+ "estimated_savings_usd": round(save, 2), "verdict": verdict, "why": why})
1001
+ if not cands:
1002
+ continue
1003
+ cands.sort(key=lambda c: (c["verdict"] != "supported", -c["estimated_savings_usd"]))
1004
+ best = cands[0] if cands[0]["verdict"] == "supported" else None
1005
+ if best:
1006
+ total_save += best["estimated_savings_usd"]
1007
+ out.append({
1008
+ "category": cat,
1009
+ "agent": cur["agent"],
1010
+ "current": {"model": cur["model"], "name": cur["name"], "prompts": cur["prompts"],
1011
+ "cost": cur["cost"], "cost_per_prompt": cur["cost_per_prompt"],
1012
+ "turns": cur["turns"], "tools": cur["tools"],
1013
+ "repeat_pct": cur["repeat_pct"]},
1014
+ "candidates": cands,
1015
+ "recommended": best["model"] if best else None,
1016
+ "recommended_name": best["name"] if best else None,
1017
+ "estimated_savings_usd": best["estimated_savings_usd"] if best else 0.0,
1018
+ "verdict": best["verdict"] if best else cands[0]["verdict"],
1019
+ "why": best["why"] if best else cands[0]["why"],
1020
+ "rule": rule,
1021
+ })
1022
+ out.sort(key=lambda c: -c["estimated_savings_usd"])
1023
+ return {
1024
+ "categories": out,
1025
+ "estimated_savings_usd": round(total_save, 2),
1026
+ "min_prompts": self.MIN_PROMPTS,
1027
+ "basis": "actual",
1028
+ "method": (f"Compares what each model actually cost per prompt on the same category "
1029
+ f"of work, using only categories where you ran both with at least "
1030
+ f"{self.MIN_PROMPTS} prompts each. Turns and re-asked prompts are shown "
1031
+ f"because a cheaper model that needs more of both is not cheaper. "
1032
+ f"Nothing here is repriced or modelled."),
1033
+ "caveat": ("Your prompts were not randomly assigned to models, so a category can "
1034
+ "differ in difficulty between them. Treat this as strong evidence for a "
1035
+ "trial, not proof."),
1036
+ }
1037
+
889
1038
  def model_switch(self, f=None):
890
1039
  """Per-request what-if: reprice each request on the model its work needs.
891
1040
 
package/finops/api.py CHANGED
@@ -292,6 +292,8 @@ class Handler(BaseHTTPRequestHandler):
292
292
  return self.send_json(a.context_analysis(f))
293
293
  if route == "waste":
294
294
  return self.send_json(a.waste(f))
295
+ if route == "model_evidence":
296
+ return self.send_json(a.model_evidence(f))
295
297
  if route == "model_switch":
296
298
  return self.send_json(a.model_switch(f))
297
299
  if route == "recommendations":
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-finops",
3
- "version": "0.4.3",
3
+ "version": "0.5.0",
4
4
  "description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
5
5
  "bin": {
6
6
  "claude-finops": "bin/claude-finops.js"
package/run.py CHANGED
@@ -17,6 +17,7 @@ import os
17
17
  import signal
18
18
  import subprocess
19
19
  import sys
20
+ import time
20
21
  import zipfile
21
22
 
22
23
  ROOT = os.path.dirname(os.path.abspath(__file__))
@@ -44,7 +45,86 @@ def _alive(pid):
44
45
  return False
45
46
 
46
47
 
47
- def stop():
48
+ def _free_port(start):
49
+ """The first free port at or above start+1, for the suggestion we print."""
50
+ import socket
51
+ for p in range(start + 1, start + 40):
52
+ with socket.socket() as sk:
53
+ try:
54
+ sk.bind(("127.0.0.1", p))
55
+ return p
56
+ except OSError:
57
+ continue
58
+ return start + 1
59
+
60
+
61
+ def _port_owner(port):
62
+ """PID listening on 127.0.0.1:<port>, or None.
63
+
64
+ The pidfile is not enough on its own: a dashboard started from a different
65
+ copy (a global npm install alongside a checkout) writes its own, and one
66
+ that was killed hard leaves a stale file behind. Asking the OS who actually
67
+ holds the port is the only answer that is always true.
68
+ """
69
+ try:
70
+ if IS_WIN:
71
+ out = subprocess.run(["netstat", "-ano", "-p", "TCP"],
72
+ capture_output=True, text=True, timeout=10).stdout
73
+ for line in out.splitlines():
74
+ f = line.split()
75
+ if len(f) >= 5 and f[1].endswith(f":{port}") and f[3] == "LISTENING":
76
+ return int(f[4])
77
+ return None
78
+ out = subprocess.run(["lsof", "-tnP", f"-iTCP:{port}", "-sTCP:LISTEN"],
79
+ capture_output=True, text=True, timeout=10).stdout
80
+ return int(out.split()[0]) if out.split() else None
81
+ except (OSError, ValueError, subprocess.SubprocessError):
82
+ return None
83
+
84
+
85
+ def _is_ours(pid):
86
+ """True only if that PID is a claude-finops server.
87
+
88
+ Whatever is on the port may be someone else's service. We stop our own
89
+ dashboard without asking; anything else we report and leave alone.
90
+ """
91
+ try:
92
+ if IS_WIN:
93
+ out = subprocess.run(
94
+ ["wmic", "process", "where", f"ProcessId={pid}", "get", "CommandLine"],
95
+ capture_output=True, text=True, timeout=10).stdout
96
+ else:
97
+ out = subprocess.run(["ps", "-o", "command=", "-p", str(pid)],
98
+ capture_output=True, text=True, timeout=10).stdout
99
+ except (OSError, subprocess.SubprocessError):
100
+ return False
101
+ return "finops.api" in out or "claude-finops" in out
102
+
103
+
104
+ def _kill(pid):
105
+ if IS_WIN:
106
+ subprocess.run(["taskkill", "/PID", str(pid), "/F"], capture_output=True)
107
+ else:
108
+ os.kill(pid, signal.SIGTERM)
109
+
110
+
111
+ def stop(port=None):
112
+ """Stop a running dashboard. Returns True if we stopped one."""
113
+ if stop_pidfile():
114
+ return True
115
+ # No usable pidfile: fall back to whoever owns the port, but only if it is
116
+ # ours to stop.
117
+ pid = _port_owner(port or os.environ.get("PORT", "8787"))
118
+ if pid and _is_ours(pid):
119
+ try:
120
+ _kill(pid)
121
+ return True
122
+ except OSError:
123
+ return False
124
+ return False
125
+
126
+
127
+ def stop_pidfile():
48
128
  for pidfile in (PIDFILE, LEGACY_PIDFILE): # LEGACY_: a server started before the move
49
129
  try:
50
130
  with open(pidfile) as fh:
@@ -55,10 +135,7 @@ def stop():
55
135
  else:
56
136
  return False
57
137
  if _alive(pid):
58
- if IS_WIN:
59
- subprocess.run(["taskkill", "/PID", str(pid), "/F"], capture_output=True)
60
- else:
61
- os.kill(pid, signal.SIGTERM)
138
+ _kill(pid)
62
139
  os.remove(pidfile)
63
140
  return True
64
141
  os.remove(pidfile)
@@ -237,13 +314,35 @@ def main():
237
314
  print(f"This is claude-finops {_version() or 'unknown'}; "
238
315
  f"run --help to see what it supports.")
239
316
  sys.exit(2)
240
- stop() # replace a previous detached server
241
317
  port = os.environ.get("PORT", "8787")
318
+ # Decide from who actually holds *this* port, not from the pidfile alone:
319
+ # the pidfile may describe a dashboard serving some other port entirely.
320
+ owner = _port_owner(port)
321
+ if owner is None or _is_ours(owner):
322
+ if owner is not None:
323
+ # flush=True: execv below replaces this process without flushing,
324
+ # so an unflushed line would simply never reach your terminal.
325
+ print(f"Restarting the dashboard already running on {port} …", flush=True)
326
+ stop(port) # replace it, whoever started it
327
+ for _ in range(20): # give the socket a moment to clear
328
+ if _port_owner(port) is None:
329
+ break
330
+ time.sleep(0.25)
331
+ # Still occupied means it is not ours: say so instead of dying in a traceback.
332
+ if (busy := _port_owner(port)) is not None:
333
+ free = _free_port(int(port))
334
+ print(f"Port {port} is already in use by PID {busy}, and it is not a "
335
+ f"claude-finops dashboard.")
336
+ print(f"Start on another port: PORT={free} claude-finops")
337
+ print(f"Or find out what is holding it: "
338
+ + (f"netstat -ano -p TCP | findstr :{port}" if IS_WIN
339
+ else f"lsof -iTCP:{port} -sTCP:LISTEN"))
340
+ sys.exit(1)
242
341
  source = os.environ.get("CLAUDE_PROJECTS", os.path.join(os.path.expanduser("~"), ".claude", "projects"))
243
342
  if "--rebuild" in args or not os.path.exists(DB_PATH):
244
343
  if not os.path.isdir(source):
245
344
  sys.exit(f"No Claude Code transcripts at {source}. Use Claude Code once, or set CLAUDE_PROJECTS.")
246
- print(f"Building warehouse from {source} …")
345
+ print(f"Building warehouse from {source} …", flush=True)
247
346
  subprocess.run(_python() + ["-m", "finops.etl", source], check=True)
248
347
  rest = [a for a in args if a != "--rebuild"]
249
348
  if IS_WIN:
package/web/app.js CHANGED
@@ -1322,12 +1322,53 @@ VIEWS.context = async (page) => {
1322
1322
  };
1323
1323
 
1324
1324
  /* ---------- model switch ---------- */
1325
+ // Verdicts from the back-test. Wording matters here: "supported" means your own
1326
+ // history backs the switch, not that we modelled it.
1327
+ const VERDICT = {
1328
+ supported: ['🟢', 'Backed by your data', 'healthy'],
1329
+ caution: ['🟠', 'Trial first', 'approaching'],
1330
+ risky: ['🔴', 'Cost more work', 'critical'],
1331
+ marginal: ['⚪', 'Too close to call', 'high'],
1332
+ };
1333
+
1334
+ function evidenceBody(ev) {
1335
+ if (!ev || !ev.categories?.length) {
1336
+ return `<div class="empty">No category yet has ${ev?.min_prompts || 8}+ prompts on two
1337
+ different models of the same agent, so there is nothing to compare. Run a cheaper model on
1338
+ a handful of real tasks and this fills in.</div>`;
1339
+ }
1340
+ const rows = [];
1341
+ ev.categories.forEach(c => c.candidates.forEach((x, i) => rows.push({c, x, first: i === 0})));
1342
+ return table([
1343
+ {h: 'Work', f: r => r.first ? `<b>${esc(r.c.category.replace('_', ' '))}</b>` : ''},
1344
+ {h: 'You use now', f: r => r.first
1345
+ ? `${esc(r.c.current.name)} <span class="note">${fmtUSD(r.c.current.cost_per_prompt)}/prompt ·
1346
+ ${Math.round(r.c.current.turns)} turns</span>` : ''},
1347
+ {h: 'Instead of', f: r => `<b>${esc(r.x.name)}</b>`},
1348
+ {h: '$ / prompt', num: 1, f: r => fmtUSD(r.x.cost_per_prompt)},
1349
+ {h: 'Turns', num: 1, f: r => `${Math.round(r.x.turns)} <span class="note">(${r.x.turn_ratio}×)</span>`},
1350
+ {h: 'Re-asked', num: 1, f: r => `${fmtPct(r.x.repeat_pct)}<span class="note">${
1351
+ r.x.repeat_delta > 0 ? ' +' + r.x.repeat_delta : ''}</span>`},
1352
+ {h: 'On', num: 1, f: r => `${fmtInt(r.x.prompts)} prompts`},
1353
+ {h: 'Verdict', f: r => `<span title="${esc(r.x.why)}">${
1354
+ statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
1355
+ {h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
1356
+ ? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
1357
+ ], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
1358
+ }
1359
+
1325
1360
  VIEWS.modelswitch = async (page) => {
1326
- const m = await api('model_switch');
1361
+ const [m, ev] = await Promise.all([
1362
+ api('model_switch'),
1363
+ api('model_evidence').catch(() => null),
1364
+ ]);
1327
1365
  const sc = m.savings_by_confidence || {};
1328
1366
  const conf = c => `<span class="badge rec">${esc(c)} confidence</span>`;
1329
1367
  page.innerHTML = `
1330
1368
  <div class="grid g4">
1369
+ ${kpi('Backed by your own runs', fmtUSD(ev?.estimated_savings_usd || 0),
1370
+ `${ev?.categories?.filter(c => c.recommended).length || 0} categories where a cheaper model
1371
+ already did the same work for less`, {badge: BADGE.actual})}
1331
1372
  ${kpi('Potential savings', fmtUSD(m.estimated_savings_usd),
1332
1373
  `${fmtPct(m.total_cost_usd ? 100 * m.estimated_savings_usd / m.total_cost_usd : 0)} of ${fmtUSD(m.total_cost_usd)} in range`,
1333
1374
  {badge: BADGE.recommendation})}
@@ -1336,7 +1377,15 @@ VIEWS.modelswitch = async (page) => {
1336
1377
  ${kpi('Stays on current model', fmtUSD(m.blocked_by_context_usd),
1337
1378
  `${fmtInt(m.blocked_by_context_requests)} requests too big for the cheaper model's context`, {badge: BADGE.estimated})}
1338
1379
  </div>
1339
- ${card('Switch these', `<div id="ms-sw"></div>`, {badge: BADGE.recommendation, flush: 1,
1380
+ ${card('What actually happened when you used a cheaper model',
1381
+ `<div id="ms-ev">${evidenceBody(ev)}</div>`, {badge: BADGE.actual, flush: 1,
1382
+ hint: 'Measured from your own prompts — no repricing, no assumptions about tokens',
1383
+ footer: ev ? esc(ev.caveat) : ''})}
1384
+ <div class="note">The table above is history; the one below is a model. Where they disagree,
1385
+ believe the history: repricing assumes the cheaper model would finish in the same number of
1386
+ turns, and your data shows that is often where the saving goes.</div>
1387
+ ${card('Switch these (repriced, not measured)', `<div id="ms-sw"></div>`,
1388
+ {badge: BADGE.recommendation, flush: 1,
1340
1389
  hint: 'Each request repriced on the model its work needs, same tokens',
1341
1390
  footer: esc(m.caveat)})}
1342
1391
  ${card('Default model per project', `<div id="ms-pj"></div>`, {badge: BADGE.recommendation, flush: 1,