claude-finops 0.4.4 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -886,6 +886,155 @@ class Analytics:
886
886
  and (provider is None or v.get("provider", "anthropic") == provider)]
887
887
  return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
888
888
 
889
+ # ---------------- evidence: what the cheaper model actually did ----------------
890
+ # model_switch() reprices your tokens on a cheaper model, which assumes the cheaper
891
+ # model would have done the same work in the same number of turns. Often it would
892
+ # not: a weaker model can take five times the turns on the same task, and the
893
+ # repricing then promises a saving that never arrives.
894
+ #
895
+ # Where you have already run more than one model on the same kind of work, we do not
896
+ # have to assume anything. This compares what each model actually cost per prompt on
897
+ # that category, and how much work it took to get there.
898
+
899
+ MIN_PROMPTS = 8 # below this a per-category average is noise, not evidence
900
+ SAVING_FLOOR_PCT = 20 # smaller gaps are inside the noise of what you happened to ask
901
+ TURN_TOLERANCE = 1.35 # more turns than this and the cheaper model was grinding
902
+ REPEAT_TOLERANCE = 12.0 # percentage points of extra re-asking we will accept
903
+
904
+ def model_evidence(self, f=None):
905
+ """Back-test a model switch against your own history.
906
+
907
+ For every category where you ran more than one model, report what each one
908
+ actually cost per prompt and what it took: turns, tool calls, and how often you
909
+ had to ask the same thing again. A candidate is only recommended when it was
910
+ genuinely cheaper per prompt *and* did not need materially more work to get
911
+ there — which is the part a repricing cannot see.
912
+ """
913
+ w, p = self.where(f)
914
+ # The filter is request-scoped, so select the prompts it touches as a subquery.
915
+ # Joining requests directly would repeat each prompt once per request and quietly
916
+ # multiply both the counts and every average by the turn count.
917
+ scope = f"pr.id IN (SELECT r.prompt_id FROM requests r WHERE {w})"
918
+ # Single-model prompts only: a prompt answered by two models cannot be
919
+ # attributed to either, and mixed rows would blur the comparison.
920
+ clean = ("pr.models IS NOT NULL AND pr.models NOT LIKE '%,%' "
921
+ "AND pr.models != '<synthetic>' AND pr.est_cost_usd > 0")
922
+ rows = self.q(f"""SELECT COALESCE(pr.category,'other') category, pr.models model,
923
+ pr.agent agent, COUNT(*) prompts, SUM(pr.est_cost_usd) cost,
924
+ AVG(pr.est_cost_usd) cost_per_prompt,
925
+ AVG(pr.request_count) turns, AVG(pr.tool_calls) tools,
926
+ AVG(pr.output_tokens) out_tokens, AVG(pr.max_context_tokens) ctx
927
+ FROM prompts pr WHERE {scope} AND {clean}
928
+ GROUP BY 1, 2, 3 HAVING prompts >= ?""", p + [self.MIN_PROMPTS])
929
+
930
+ repeats = {(r["category"], r["model"]): r["pct"] for r in self.q(f"""
931
+ SELECT COALESCE(pr.category,'other') category, pr.models model,
932
+ ROUND(100.0 * SUM(CASE WHEN dup.n > 1 THEN 1 ELSE 0 END) / COUNT(*), 1) pct
933
+ FROM prompts pr
934
+ LEFT JOIN (SELECT norm_hash, COUNT(*) n FROM prompts GROUP BY norm_hash) dup
935
+ ON dup.norm_hash = pr.norm_hash
936
+ WHERE {scope} AND {clean}
937
+ GROUP BY 1, 2""", p)}
938
+
939
+ by_cat = defaultdict(list)
940
+ for r in rows:
941
+ r["repeat_pct"] = repeats.get((r["category"], r["model"]), 0.0) or 0.0
942
+ r["name"] = self.pricing.display_name(r["model"])
943
+ r["tier"] = self.pricing.tier(r["model"])
944
+ by_cat[r["category"]].append(r)
945
+
946
+ out, total_save = [], 0.0
947
+ for cat, models in by_cat.items():
948
+ if len(models) < 2:
949
+ continue
950
+ # The incumbent is what you spend the most on here — that is the bill a
951
+ # switch would actually change.
952
+ cur = max(models, key=lambda m: m["cost"])
953
+ rule = self.SWITCH_RULES.get(cat, ("balanced", "low"))[0]
954
+ cands = []
955
+ for m in models:
956
+ if m["model"] == cur["model"] or m["cost_per_prompt"] >= cur["cost_per_prompt"]:
957
+ continue
958
+ # Only models the same agent can run. Telling a Claude Code user to use a
959
+ # GPT model is not a setting change, it is a different tool, and the
960
+ # comparison would be between two different ways of working.
961
+ if m["agent"] != cur["agent"]:
962
+ continue
963
+ save_pct = 100.0 * (1 - m["cost_per_prompt"] / cur["cost_per_prompt"])
964
+ turn_ratio = (m["turns"] / cur["turns"]) if cur["turns"] else 1.0
965
+ repeat_delta = m["repeat_pct"] - cur["repeat_pct"]
966
+ # Observed, not repriced: what your own prompts cost on each side.
967
+ save = (cur["cost_per_prompt"] - m["cost_per_prompt"]) * cur["prompts"]
968
+ if save_pct < self.SAVING_FLOOR_PCT:
969
+ verdict, why = "marginal", (
970
+ f"Only {save_pct:.0f}% cheaper per prompt — inside the noise of what "
971
+ f"you happened to ask each model.")
972
+ elif turn_ratio > self.TURN_TOLERANCE:
973
+ verdict, why = "risky", (
974
+ f"Cost {save_pct:.0f}% less per prompt but took {turn_ratio:.1f}x the "
975
+ f"turns ({m['turns']:.0f} vs {cur['turns']:.0f}). It got there by "
976
+ f"grinding, and that is the cost the headline number misses.")
977
+ elif repeat_delta > self.REPEAT_TOLERANCE:
978
+ verdict, why = "risky", (
979
+ f"{save_pct:.0f}% cheaper per prompt, but you re-asked "
980
+ f"{m['repeat_pct']:.0f}% of these prompts against "
981
+ f"{cur['repeat_pct']:.0f}% on {cur['name']} — rework you paid for twice.")
982
+ elif rule == "keep":
983
+ verdict, why = "caution", (
984
+ f"{save_pct:.0f}% cheaper per prompt and no more turns, but {cat.replace('_',' ')} "
985
+ f"is reasoning-heavy work where a miss is expensive in ways this data "
986
+ f"cannot show. Worth a trial, not a default.")
987
+ else:
988
+ verdict, why = "supported", (
989
+ f"{save_pct:.0f}% cheaper per prompt on {m['prompts']} of your own "
990
+ f"{cat.replace('_',' ')} prompts, in {turn_ratio:.1f}x the turns "
991
+ f"({m['turns']:.0f} vs {cur['turns']:.0f}) with "
992
+ f"{'less' if repeat_delta <= 0 else 'similar'} re-asking. "
993
+ f"This is measured, not modelled.")
994
+ cands.append({
995
+ "model": m["model"], "name": m["name"], "tier": m["tier"],
996
+ "prompts": m["prompts"], "cost_per_prompt": m["cost_per_prompt"],
997
+ "turns": m["turns"], "tools": m["tools"], "repeat_pct": m["repeat_pct"],
998
+ "savings_pct": round(save_pct, 1), "turn_ratio": round(turn_ratio, 2),
999
+ "repeat_delta": round(repeat_delta, 1),
1000
+ "estimated_savings_usd": round(save, 2), "verdict": verdict, "why": why})
1001
+ if not cands:
1002
+ continue
1003
+ cands.sort(key=lambda c: (c["verdict"] != "supported", -c["estimated_savings_usd"]))
1004
+ best = cands[0] if cands[0]["verdict"] == "supported" else None
1005
+ if best:
1006
+ total_save += best["estimated_savings_usd"]
1007
+ out.append({
1008
+ "category": cat,
1009
+ "agent": cur["agent"],
1010
+ "current": {"model": cur["model"], "name": cur["name"], "prompts": cur["prompts"],
1011
+ "cost": cur["cost"], "cost_per_prompt": cur["cost_per_prompt"],
1012
+ "turns": cur["turns"], "tools": cur["tools"],
1013
+ "repeat_pct": cur["repeat_pct"]},
1014
+ "candidates": cands,
1015
+ "recommended": best["model"] if best else None,
1016
+ "recommended_name": best["name"] if best else None,
1017
+ "estimated_savings_usd": best["estimated_savings_usd"] if best else 0.0,
1018
+ "verdict": best["verdict"] if best else cands[0]["verdict"],
1019
+ "why": best["why"] if best else cands[0]["why"],
1020
+ "rule": rule,
1021
+ })
1022
+ out.sort(key=lambda c: -c["estimated_savings_usd"])
1023
+ return {
1024
+ "categories": out,
1025
+ "estimated_savings_usd": round(total_save, 2),
1026
+ "min_prompts": self.MIN_PROMPTS,
1027
+ "basis": "actual",
1028
+ "method": (f"Compares what each model actually cost per prompt on the same category "
1029
+ f"of work, using only categories where you ran both with at least "
1030
+ f"{self.MIN_PROMPTS} prompts each. Turns and re-asked prompts are shown "
1031
+ f"because a cheaper model that needs more of both is not cheaper. "
1032
+ f"Nothing here is repriced or modelled."),
1033
+ "caveat": ("Your prompts were not randomly assigned to models, so a category can "
1034
+ "differ in difficulty between them. Treat this as strong evidence for a "
1035
+ "trial, not proof."),
1036
+ }
1037
+
889
1038
  def model_switch(self, f=None):
890
1039
  """Per-request what-if: reprice each request on the model its work needs.
891
1040
 
package/finops/api.py CHANGED
@@ -292,6 +292,8 @@ class Handler(BaseHTTPRequestHandler):
292
292
  return self.send_json(a.context_analysis(f))
293
293
  if route == "waste":
294
294
  return self.send_json(a.waste(f))
295
+ if route == "model_evidence":
296
+ return self.send_json(a.model_evidence(f))
295
297
  if route == "model_switch":
296
298
  return self.send_json(a.model_switch(f))
297
299
  if route == "recommendations":
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-finops",
3
- "version": "0.4.4",
3
+ "version": "0.5.0",
4
4
  "description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
5
5
  "bin": {
6
6
  "claude-finops": "bin/claude-finops.js"
package/web/app.js CHANGED
@@ -1322,12 +1322,53 @@ VIEWS.context = async (page) => {
1322
1322
  };
1323
1323
 
1324
1324
  /* ---------- model switch ---------- */
1325
+ // Verdicts from the back-test. Wording matters here: "supported" means your own
1326
+ // history backs the switch, not that we modelled it.
1327
+ const VERDICT = {
1328
+ supported: ['🟢', 'Backed by your data', 'healthy'],
1329
+ caution: ['🟠', 'Trial first', 'approaching'],
1330
+ risky: ['🔴', 'Cost more work', 'critical'],
1331
+ marginal: ['⚪', 'Too close to call', 'high'],
1332
+ };
1333
+
1334
+ function evidenceBody(ev) {
1335
+ if (!ev || !ev.categories?.length) {
1336
+ return `<div class="empty">No category yet has ${ev?.min_prompts || 8}+ prompts on two
1337
+ different models of the same agent, so there is nothing to compare. Run a cheaper model on
1338
+ a handful of real tasks and this fills in.</div>`;
1339
+ }
1340
+ const rows = [];
1341
+ ev.categories.forEach(c => c.candidates.forEach((x, i) => rows.push({c, x, first: i === 0})));
1342
+ return table([
1343
+ {h: 'Work', f: r => r.first ? `<b>${esc(r.c.category.replace('_', ' '))}</b>` : ''},
1344
+ {h: 'You use now', f: r => r.first
1345
+ ? `${esc(r.c.current.name)} <span class="note">${fmtUSD(r.c.current.cost_per_prompt)}/prompt ·
1346
+ ${Math.round(r.c.current.turns)} turns</span>` : ''},
1347
+ {h: 'Instead of', f: r => `<b>${esc(r.x.name)}</b>`},
1348
+ {h: '$ / prompt', num: 1, f: r => fmtUSD(r.x.cost_per_prompt)},
1349
+ {h: 'Turns', num: 1, f: r => `${Math.round(r.x.turns)} <span class="note">(${r.x.turn_ratio}×)</span>`},
1350
+ {h: 'Re-asked', num: 1, f: r => `${fmtPct(r.x.repeat_pct)}<span class="note">${
1351
+ r.x.repeat_delta > 0 ? ' +' + r.x.repeat_delta : ''}</span>`},
1352
+ {h: 'On', num: 1, f: r => `${fmtInt(r.x.prompts)} prompts`},
1353
+ {h: 'Verdict', f: r => `<span title="${esc(r.x.why)}">${
1354
+ statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
1355
+ {h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
1356
+ ? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
1357
+ ], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
1358
+ }
1359
+
1325
1360
  VIEWS.modelswitch = async (page) => {
1326
- const m = await api('model_switch');
1361
+ const [m, ev] = await Promise.all([
1362
+ api('model_switch'),
1363
+ api('model_evidence').catch(() => null),
1364
+ ]);
1327
1365
  const sc = m.savings_by_confidence || {};
1328
1366
  const conf = c => `<span class="badge rec">${esc(c)} confidence</span>`;
1329
1367
  page.innerHTML = `
1330
1368
  <div class="grid g4">
1369
+ ${kpi('Backed by your own runs', fmtUSD(ev?.estimated_savings_usd || 0),
1370
+ `${ev?.categories?.filter(c => c.recommended).length || 0} categories where a cheaper model
1371
+ already did the same work for less`, {badge: BADGE.actual})}
1331
1372
  ${kpi('Potential savings', fmtUSD(m.estimated_savings_usd),
1332
1373
  `${fmtPct(m.total_cost_usd ? 100 * m.estimated_savings_usd / m.total_cost_usd : 0)} of ${fmtUSD(m.total_cost_usd)} in range`,
1333
1374
  {badge: BADGE.recommendation})}
@@ -1336,7 +1377,15 @@ VIEWS.modelswitch = async (page) => {
1336
1377
  ${kpi('Stays on current model', fmtUSD(m.blocked_by_context_usd),
1337
1378
  `${fmtInt(m.blocked_by_context_requests)} requests too big for the cheaper model's context`, {badge: BADGE.estimated})}
1338
1379
  </div>
1339
- ${card('Switch these', `<div id="ms-sw"></div>`, {badge: BADGE.recommendation, flush: 1,
1380
+ ${card('What actually happened when you used a cheaper model',
1381
+ `<div id="ms-ev">${evidenceBody(ev)}</div>`, {badge: BADGE.actual, flush: 1,
1382
+ hint: 'Measured from your own prompts — no repricing, no assumptions about tokens',
1383
+ footer: ev ? esc(ev.caveat) : ''})}
1384
+ <div class="note">The table above is history; the one below is a model. Where they disagree,
1385
+ believe the history: repricing assumes the cheaper model would finish in the same number of
1386
+ turns, and your data shows that is often where the saving goes.</div>
1387
+ ${card('Switch these (repriced, not measured)', `<div id="ms-sw"></div>`,
1388
+ {badge: BADGE.recommendation, flush: 1,
1340
1389
  hint: 'Each request repriced on the model its work needs, same tokens',
1341
1390
  footer: esc(m.caveat)})}
1342
1391
  ${card('Default model per project', `<div id="ms-pj"></div>`, {badge: BADGE.recommendation, flush: 1,