claude-finops 0.4.3 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/finops/analytics.py +149 -0
- package/finops/api.py +2 -0
- package/package.json +1 -1
- package/run.py +106 -7
- package/web/app.js +51 -2
package/finops/analytics.py
CHANGED
|
@@ -886,6 +886,155 @@ class Analytics:
|
|
|
886
886
|
and (provider is None or v.get("provider", "anthropic") == provider)]
|
|
887
887
|
return min(ms, key=lambda m: self.pricing.rates(m).get("output", 1e9)) if ms else None
|
|
888
888
|
|
|
889
|
+
# ---------------- evidence: what the cheaper model actually did ----------------
|
|
890
|
+
# model_switch() reprices your tokens on a cheaper model, which assumes the cheaper
|
|
891
|
+
# model would have done the same work in the same number of turns. Often it would
|
|
892
|
+
# not: a weaker model can take five times the turns on the same task, and the
|
|
893
|
+
# repricing then promises a saving that never arrives.
|
|
894
|
+
#
|
|
895
|
+
# Where you have already run more than one model on the same kind of work, we do not
|
|
896
|
+
# have to assume anything. This compares what each model actually cost per prompt on
|
|
897
|
+
# that category, and how much work it took to get there.
|
|
898
|
+
|
|
899
|
+
MIN_PROMPTS = 8 # below this a per-category average is noise, not evidence
|
|
900
|
+
SAVING_FLOOR_PCT = 20 # smaller gaps are inside the noise of what you happened to ask
|
|
901
|
+
TURN_TOLERANCE = 1.35 # more turns than this and the cheaper model was grinding
|
|
902
|
+
REPEAT_TOLERANCE = 12.0 # percentage points of extra re-asking we will accept
|
|
903
|
+
|
|
904
|
+
def model_evidence(self, f=None):
|
|
905
|
+
"""Back-test a model switch against your own history.
|
|
906
|
+
|
|
907
|
+
For every category where you ran more than one model, report what each one
|
|
908
|
+
actually cost per prompt and what it took: turns, tool calls, and how often you
|
|
909
|
+
had to ask the same thing again. A candidate is only recommended when it was
|
|
910
|
+
genuinely cheaper per prompt *and* did not need materially more work to get
|
|
911
|
+
there — which is the part a repricing cannot see.
|
|
912
|
+
"""
|
|
913
|
+
w, p = self.where(f)
|
|
914
|
+
# The filter is request-scoped, so select the prompts it touches as a subquery.
|
|
915
|
+
# Joining requests directly would repeat each prompt once per request and quietly
|
|
916
|
+
# multiply both the counts and every average by the turn count.
|
|
917
|
+
scope = f"pr.id IN (SELECT r.prompt_id FROM requests r WHERE {w})"
|
|
918
|
+
# Single-model prompts only: a prompt answered by two models cannot be
|
|
919
|
+
# attributed to either, and mixed rows would blur the comparison.
|
|
920
|
+
clean = ("pr.models IS NOT NULL AND pr.models NOT LIKE '%,%' "
|
|
921
|
+
"AND pr.models != '<synthetic>' AND pr.est_cost_usd > 0")
|
|
922
|
+
rows = self.q(f"""SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
923
|
+
pr.agent agent, COUNT(*) prompts, SUM(pr.est_cost_usd) cost,
|
|
924
|
+
AVG(pr.est_cost_usd) cost_per_prompt,
|
|
925
|
+
AVG(pr.request_count) turns, AVG(pr.tool_calls) tools,
|
|
926
|
+
AVG(pr.output_tokens) out_tokens, AVG(pr.max_context_tokens) ctx
|
|
927
|
+
FROM prompts pr WHERE {scope} AND {clean}
|
|
928
|
+
GROUP BY 1, 2, 3 HAVING prompts >= ?""", p + [self.MIN_PROMPTS])
|
|
929
|
+
|
|
930
|
+
repeats = {(r["category"], r["model"]): r["pct"] for r in self.q(f"""
|
|
931
|
+
SELECT COALESCE(pr.category,'other') category, pr.models model,
|
|
932
|
+
ROUND(100.0 * SUM(CASE WHEN dup.n > 1 THEN 1 ELSE 0 END) / COUNT(*), 1) pct
|
|
933
|
+
FROM prompts pr
|
|
934
|
+
LEFT JOIN (SELECT norm_hash, COUNT(*) n FROM prompts GROUP BY norm_hash) dup
|
|
935
|
+
ON dup.norm_hash = pr.norm_hash
|
|
936
|
+
WHERE {scope} AND {clean}
|
|
937
|
+
GROUP BY 1, 2""", p)}
|
|
938
|
+
|
|
939
|
+
by_cat = defaultdict(list)
|
|
940
|
+
for r in rows:
|
|
941
|
+
r["repeat_pct"] = repeats.get((r["category"], r["model"]), 0.0) or 0.0
|
|
942
|
+
r["name"] = self.pricing.display_name(r["model"])
|
|
943
|
+
r["tier"] = self.pricing.tier(r["model"])
|
|
944
|
+
by_cat[r["category"]].append(r)
|
|
945
|
+
|
|
946
|
+
out, total_save = [], 0.0
|
|
947
|
+
for cat, models in by_cat.items():
|
|
948
|
+
if len(models) < 2:
|
|
949
|
+
continue
|
|
950
|
+
# The incumbent is what you spend the most on here — that is the bill a
|
|
951
|
+
# switch would actually change.
|
|
952
|
+
cur = max(models, key=lambda m: m["cost"])
|
|
953
|
+
rule = self.SWITCH_RULES.get(cat, ("balanced", "low"))[0]
|
|
954
|
+
cands = []
|
|
955
|
+
for m in models:
|
|
956
|
+
if m["model"] == cur["model"] or m["cost_per_prompt"] >= cur["cost_per_prompt"]:
|
|
957
|
+
continue
|
|
958
|
+
# Only models the same agent can run. Telling a Claude Code user to use a
|
|
959
|
+
# GPT model is not a setting change, it is a different tool, and the
|
|
960
|
+
# comparison would be between two different ways of working.
|
|
961
|
+
if m["agent"] != cur["agent"]:
|
|
962
|
+
continue
|
|
963
|
+
save_pct = 100.0 * (1 - m["cost_per_prompt"] / cur["cost_per_prompt"])
|
|
964
|
+
turn_ratio = (m["turns"] / cur["turns"]) if cur["turns"] else 1.0
|
|
965
|
+
repeat_delta = m["repeat_pct"] - cur["repeat_pct"]
|
|
966
|
+
# Observed, not repriced: what your own prompts cost on each side.
|
|
967
|
+
save = (cur["cost_per_prompt"] - m["cost_per_prompt"]) * cur["prompts"]
|
|
968
|
+
if save_pct < self.SAVING_FLOOR_PCT:
|
|
969
|
+
verdict, why = "marginal", (
|
|
970
|
+
f"Only {save_pct:.0f}% cheaper per prompt — inside the noise of what "
|
|
971
|
+
f"you happened to ask each model.")
|
|
972
|
+
elif turn_ratio > self.TURN_TOLERANCE:
|
|
973
|
+
verdict, why = "risky", (
|
|
974
|
+
f"Cost {save_pct:.0f}% less per prompt but took {turn_ratio:.1f}x the "
|
|
975
|
+
f"turns ({m['turns']:.0f} vs {cur['turns']:.0f}). It got there by "
|
|
976
|
+
f"grinding, and that is the cost the headline number misses.")
|
|
977
|
+
elif repeat_delta > self.REPEAT_TOLERANCE:
|
|
978
|
+
verdict, why = "risky", (
|
|
979
|
+
f"{save_pct:.0f}% cheaper per prompt, but you re-asked "
|
|
980
|
+
f"{m['repeat_pct']:.0f}% of these prompts against "
|
|
981
|
+
f"{cur['repeat_pct']:.0f}% on {cur['name']} — rework you paid for twice.")
|
|
982
|
+
elif rule == "keep":
|
|
983
|
+
verdict, why = "caution", (
|
|
984
|
+
f"{save_pct:.0f}% cheaper per prompt and no more turns, but {cat.replace('_',' ')} "
|
|
985
|
+
f"is reasoning-heavy work where a miss is expensive in ways this data "
|
|
986
|
+
f"cannot show. Worth a trial, not a default.")
|
|
987
|
+
else:
|
|
988
|
+
verdict, why = "supported", (
|
|
989
|
+
f"{save_pct:.0f}% cheaper per prompt on {m['prompts']} of your own "
|
|
990
|
+
f"{cat.replace('_',' ')} prompts, in {turn_ratio:.1f}x the turns "
|
|
991
|
+
f"({m['turns']:.0f} vs {cur['turns']:.0f}) with "
|
|
992
|
+
f"{'less' if repeat_delta <= 0 else 'similar'} re-asking. "
|
|
993
|
+
f"This is measured, not modelled.")
|
|
994
|
+
cands.append({
|
|
995
|
+
"model": m["model"], "name": m["name"], "tier": m["tier"],
|
|
996
|
+
"prompts": m["prompts"], "cost_per_prompt": m["cost_per_prompt"],
|
|
997
|
+
"turns": m["turns"], "tools": m["tools"], "repeat_pct": m["repeat_pct"],
|
|
998
|
+
"savings_pct": round(save_pct, 1), "turn_ratio": round(turn_ratio, 2),
|
|
999
|
+
"repeat_delta": round(repeat_delta, 1),
|
|
1000
|
+
"estimated_savings_usd": round(save, 2), "verdict": verdict, "why": why})
|
|
1001
|
+
if not cands:
|
|
1002
|
+
continue
|
|
1003
|
+
cands.sort(key=lambda c: (c["verdict"] != "supported", -c["estimated_savings_usd"]))
|
|
1004
|
+
best = cands[0] if cands[0]["verdict"] == "supported" else None
|
|
1005
|
+
if best:
|
|
1006
|
+
total_save += best["estimated_savings_usd"]
|
|
1007
|
+
out.append({
|
|
1008
|
+
"category": cat,
|
|
1009
|
+
"agent": cur["agent"],
|
|
1010
|
+
"current": {"model": cur["model"], "name": cur["name"], "prompts": cur["prompts"],
|
|
1011
|
+
"cost": cur["cost"], "cost_per_prompt": cur["cost_per_prompt"],
|
|
1012
|
+
"turns": cur["turns"], "tools": cur["tools"],
|
|
1013
|
+
"repeat_pct": cur["repeat_pct"]},
|
|
1014
|
+
"candidates": cands,
|
|
1015
|
+
"recommended": best["model"] if best else None,
|
|
1016
|
+
"recommended_name": best["name"] if best else None,
|
|
1017
|
+
"estimated_savings_usd": best["estimated_savings_usd"] if best else 0.0,
|
|
1018
|
+
"verdict": best["verdict"] if best else cands[0]["verdict"],
|
|
1019
|
+
"why": best["why"] if best else cands[0]["why"],
|
|
1020
|
+
"rule": rule,
|
|
1021
|
+
})
|
|
1022
|
+
out.sort(key=lambda c: -c["estimated_savings_usd"])
|
|
1023
|
+
return {
|
|
1024
|
+
"categories": out,
|
|
1025
|
+
"estimated_savings_usd": round(total_save, 2),
|
|
1026
|
+
"min_prompts": self.MIN_PROMPTS,
|
|
1027
|
+
"basis": "actual",
|
|
1028
|
+
"method": (f"Compares what each model actually cost per prompt on the same category "
|
|
1029
|
+
f"of work, using only categories where you ran both with at least "
|
|
1030
|
+
f"{self.MIN_PROMPTS} prompts each. Turns and re-asked prompts are shown "
|
|
1031
|
+
f"because a cheaper model that needs more of both is not cheaper. "
|
|
1032
|
+
f"Nothing here is repriced or modelled."),
|
|
1033
|
+
"caveat": ("Your prompts were not randomly assigned to models, so a category can "
|
|
1034
|
+
"differ in difficulty between them. Treat this as strong evidence for a "
|
|
1035
|
+
"trial, not proof."),
|
|
1036
|
+
}
|
|
1037
|
+
|
|
889
1038
|
def model_switch(self, f=None):
|
|
890
1039
|
"""Per-request what-if: reprice each request on the model its work needs.
|
|
891
1040
|
|
package/finops/api.py
CHANGED
|
@@ -292,6 +292,8 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
292
292
|
return self.send_json(a.context_analysis(f))
|
|
293
293
|
if route == "waste":
|
|
294
294
|
return self.send_json(a.waste(f))
|
|
295
|
+
if route == "model_evidence":
|
|
296
|
+
return self.send_json(a.model_evidence(f))
|
|
295
297
|
if route == "model_switch":
|
|
296
298
|
return self.send_json(a.model_switch(f))
|
|
297
299
|
if route == "recommendations":
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-finops",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"claude-finops": "bin/claude-finops.js"
|
package/run.py
CHANGED
|
@@ -17,6 +17,7 @@ import os
|
|
|
17
17
|
import signal
|
|
18
18
|
import subprocess
|
|
19
19
|
import sys
|
|
20
|
+
import time
|
|
20
21
|
import zipfile
|
|
21
22
|
|
|
22
23
|
ROOT = os.path.dirname(os.path.abspath(__file__))
|
|
@@ -44,7 +45,86 @@ def _alive(pid):
|
|
|
44
45
|
return False
|
|
45
46
|
|
|
46
47
|
|
|
47
|
-
def
|
|
48
|
+
def _free_port(start):
|
|
49
|
+
"""The first free port at or above start+1, for the suggestion we print."""
|
|
50
|
+
import socket
|
|
51
|
+
for p in range(start + 1, start + 40):
|
|
52
|
+
with socket.socket() as sk:
|
|
53
|
+
try:
|
|
54
|
+
sk.bind(("127.0.0.1", p))
|
|
55
|
+
return p
|
|
56
|
+
except OSError:
|
|
57
|
+
continue
|
|
58
|
+
return start + 1
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _port_owner(port):
|
|
62
|
+
"""PID listening on 127.0.0.1:<port>, or None.
|
|
63
|
+
|
|
64
|
+
The pidfile is not enough on its own: a dashboard started from a different
|
|
65
|
+
copy (a global npm install alongside a checkout) writes its own, and one
|
|
66
|
+
that was killed hard leaves a stale file behind. Asking the OS who actually
|
|
67
|
+
holds the port is the only answer that is always true.
|
|
68
|
+
"""
|
|
69
|
+
try:
|
|
70
|
+
if IS_WIN:
|
|
71
|
+
out = subprocess.run(["netstat", "-ano", "-p", "TCP"],
|
|
72
|
+
capture_output=True, text=True, timeout=10).stdout
|
|
73
|
+
for line in out.splitlines():
|
|
74
|
+
f = line.split()
|
|
75
|
+
if len(f) >= 5 and f[1].endswith(f":{port}") and f[3] == "LISTENING":
|
|
76
|
+
return int(f[4])
|
|
77
|
+
return None
|
|
78
|
+
out = subprocess.run(["lsof", "-tnP", f"-iTCP:{port}", "-sTCP:LISTEN"],
|
|
79
|
+
capture_output=True, text=True, timeout=10).stdout
|
|
80
|
+
return int(out.split()[0]) if out.split() else None
|
|
81
|
+
except (OSError, ValueError, subprocess.SubprocessError):
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _is_ours(pid):
|
|
86
|
+
"""True only if that PID is a claude-finops server.
|
|
87
|
+
|
|
88
|
+
Whatever is on the port may be someone else's service. We stop our own
|
|
89
|
+
dashboard without asking; anything else we report and leave alone.
|
|
90
|
+
"""
|
|
91
|
+
try:
|
|
92
|
+
if IS_WIN:
|
|
93
|
+
out = subprocess.run(
|
|
94
|
+
["wmic", "process", "where", f"ProcessId={pid}", "get", "CommandLine"],
|
|
95
|
+
capture_output=True, text=True, timeout=10).stdout
|
|
96
|
+
else:
|
|
97
|
+
out = subprocess.run(["ps", "-o", "command=", "-p", str(pid)],
|
|
98
|
+
capture_output=True, text=True, timeout=10).stdout
|
|
99
|
+
except (OSError, subprocess.SubprocessError):
|
|
100
|
+
return False
|
|
101
|
+
return "finops.api" in out or "claude-finops" in out
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _kill(pid):
|
|
105
|
+
if IS_WIN:
|
|
106
|
+
subprocess.run(["taskkill", "/PID", str(pid), "/F"], capture_output=True)
|
|
107
|
+
else:
|
|
108
|
+
os.kill(pid, signal.SIGTERM)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def stop(port=None):
|
|
112
|
+
"""Stop a running dashboard. Returns True if we stopped one."""
|
|
113
|
+
if stop_pidfile():
|
|
114
|
+
return True
|
|
115
|
+
# No usable pidfile: fall back to whoever owns the port, but only if it is
|
|
116
|
+
# ours to stop.
|
|
117
|
+
pid = _port_owner(port or os.environ.get("PORT", "8787"))
|
|
118
|
+
if pid and _is_ours(pid):
|
|
119
|
+
try:
|
|
120
|
+
_kill(pid)
|
|
121
|
+
return True
|
|
122
|
+
except OSError:
|
|
123
|
+
return False
|
|
124
|
+
return False
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def stop_pidfile():
|
|
48
128
|
for pidfile in (PIDFILE, LEGACY_PIDFILE): # LEGACY_: a server started before the move
|
|
49
129
|
try:
|
|
50
130
|
with open(pidfile) as fh:
|
|
@@ -55,10 +135,7 @@ def stop():
|
|
|
55
135
|
else:
|
|
56
136
|
return False
|
|
57
137
|
if _alive(pid):
|
|
58
|
-
|
|
59
|
-
subprocess.run(["taskkill", "/PID", str(pid), "/F"], capture_output=True)
|
|
60
|
-
else:
|
|
61
|
-
os.kill(pid, signal.SIGTERM)
|
|
138
|
+
_kill(pid)
|
|
62
139
|
os.remove(pidfile)
|
|
63
140
|
return True
|
|
64
141
|
os.remove(pidfile)
|
|
@@ -237,13 +314,35 @@ def main():
|
|
|
237
314
|
print(f"This is claude-finops {_version() or 'unknown'}; "
|
|
238
315
|
f"run --help to see what it supports.")
|
|
239
316
|
sys.exit(2)
|
|
240
|
-
stop() # replace a previous detached server
|
|
241
317
|
port = os.environ.get("PORT", "8787")
|
|
318
|
+
# Decide from who actually holds *this* port, not from the pidfile alone:
|
|
319
|
+
# the pidfile may describe a dashboard serving some other port entirely.
|
|
320
|
+
owner = _port_owner(port)
|
|
321
|
+
if owner is None or _is_ours(owner):
|
|
322
|
+
if owner is not None:
|
|
323
|
+
# flush=True: execv below replaces this process without flushing,
|
|
324
|
+
# so an unflushed line would simply never reach your terminal.
|
|
325
|
+
print(f"Restarting the dashboard already running on {port} …", flush=True)
|
|
326
|
+
stop(port) # replace it, whoever started it
|
|
327
|
+
for _ in range(20): # give the socket a moment to clear
|
|
328
|
+
if _port_owner(port) is None:
|
|
329
|
+
break
|
|
330
|
+
time.sleep(0.25)
|
|
331
|
+
# Still occupied means it is not ours: say so instead of dying in a traceback.
|
|
332
|
+
if (busy := _port_owner(port)) is not None:
|
|
333
|
+
free = _free_port(int(port))
|
|
334
|
+
print(f"Port {port} is already in use by PID {busy}, and it is not a "
|
|
335
|
+
f"claude-finops dashboard.")
|
|
336
|
+
print(f"Start on another port: PORT={free} claude-finops")
|
|
337
|
+
print(f"Or find out what is holding it: "
|
|
338
|
+
+ (f"netstat -ano -p TCP | findstr :{port}" if IS_WIN
|
|
339
|
+
else f"lsof -iTCP:{port} -sTCP:LISTEN"))
|
|
340
|
+
sys.exit(1)
|
|
242
341
|
source = os.environ.get("CLAUDE_PROJECTS", os.path.join(os.path.expanduser("~"), ".claude", "projects"))
|
|
243
342
|
if "--rebuild" in args or not os.path.exists(DB_PATH):
|
|
244
343
|
if not os.path.isdir(source):
|
|
245
344
|
sys.exit(f"No Claude Code transcripts at {source}. Use Claude Code once, or set CLAUDE_PROJECTS.")
|
|
246
|
-
print(f"Building warehouse from {source} …")
|
|
345
|
+
print(f"Building warehouse from {source} …", flush=True)
|
|
247
346
|
subprocess.run(_python() + ["-m", "finops.etl", source], check=True)
|
|
248
347
|
rest = [a for a in args if a != "--rebuild"]
|
|
249
348
|
if IS_WIN:
|
package/web/app.js
CHANGED
|
@@ -1322,12 +1322,53 @@ VIEWS.context = async (page) => {
|
|
|
1322
1322
|
};
|
|
1323
1323
|
|
|
1324
1324
|
/* ---------- model switch ---------- */
|
|
1325
|
+
// Verdicts from the back-test. Wording matters here: "supported" means your own
|
|
1326
|
+
// history backs the switch, not that we modelled it.
|
|
1327
|
+
const VERDICT = {
|
|
1328
|
+
supported: ['🟢', 'Backed by your data', 'healthy'],
|
|
1329
|
+
caution: ['🟠', 'Trial first', 'approaching'],
|
|
1330
|
+
risky: ['🔴', 'Cost more work', 'critical'],
|
|
1331
|
+
marginal: ['⚪', 'Too close to call', 'high'],
|
|
1332
|
+
};
|
|
1333
|
+
|
|
1334
|
+
function evidenceBody(ev) {
|
|
1335
|
+
if (!ev || !ev.categories?.length) {
|
|
1336
|
+
return `<div class="empty">No category yet has ${ev?.min_prompts || 8}+ prompts on two
|
|
1337
|
+
different models of the same agent, so there is nothing to compare. Run a cheaper model on
|
|
1338
|
+
a handful of real tasks and this fills in.</div>`;
|
|
1339
|
+
}
|
|
1340
|
+
const rows = [];
|
|
1341
|
+
ev.categories.forEach(c => c.candidates.forEach((x, i) => rows.push({c, x, first: i === 0})));
|
|
1342
|
+
return table([
|
|
1343
|
+
{h: 'Work', f: r => r.first ? `<b>${esc(r.c.category.replace('_', ' '))}</b>` : ''},
|
|
1344
|
+
{h: 'You use now', f: r => r.first
|
|
1345
|
+
? `${esc(r.c.current.name)} <span class="note">${fmtUSD(r.c.current.cost_per_prompt)}/prompt ·
|
|
1346
|
+
${Math.round(r.c.current.turns)} turns</span>` : ''},
|
|
1347
|
+
{h: 'Instead of', f: r => `<b>${esc(r.x.name)}</b>`},
|
|
1348
|
+
{h: '$ / prompt', num: 1, f: r => fmtUSD(r.x.cost_per_prompt)},
|
|
1349
|
+
{h: 'Turns', num: 1, f: r => `${Math.round(r.x.turns)} <span class="note">(${r.x.turn_ratio}×)</span>`},
|
|
1350
|
+
{h: 'Re-asked', num: 1, f: r => `${fmtPct(r.x.repeat_pct)}<span class="note">${
|
|
1351
|
+
r.x.repeat_delta > 0 ? ' +' + r.x.repeat_delta : ''}</span>`},
|
|
1352
|
+
{h: 'On', num: 1, f: r => `${fmtInt(r.x.prompts)} prompts`},
|
|
1353
|
+
{h: 'Verdict', f: r => `<span title="${esc(r.x.why)}">${
|
|
1354
|
+
statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
|
|
1355
|
+
{h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
|
|
1356
|
+
? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
|
|
1357
|
+
], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
|
|
1358
|
+
}
|
|
1359
|
+
|
|
1325
1360
|
VIEWS.modelswitch = async (page) => {
|
|
1326
|
-
const m = await
|
|
1361
|
+
const [m, ev] = await Promise.all([
|
|
1362
|
+
api('model_switch'),
|
|
1363
|
+
api('model_evidence').catch(() => null),
|
|
1364
|
+
]);
|
|
1327
1365
|
const sc = m.savings_by_confidence || {};
|
|
1328
1366
|
const conf = c => `<span class="badge rec">${esc(c)} confidence</span>`;
|
|
1329
1367
|
page.innerHTML = `
|
|
1330
1368
|
<div class="grid g4">
|
|
1369
|
+
${kpi('Backed by your own runs', fmtUSD(ev?.estimated_savings_usd || 0),
|
|
1370
|
+
`${ev?.categories?.filter(c => c.recommended).length || 0} categories where a cheaper model
|
|
1371
|
+
already did the same work for less`, {badge: BADGE.actual})}
|
|
1331
1372
|
${kpi('Potential savings', fmtUSD(m.estimated_savings_usd),
|
|
1332
1373
|
`${fmtPct(m.total_cost_usd ? 100 * m.estimated_savings_usd / m.total_cost_usd : 0)} of ${fmtUSD(m.total_cost_usd)} in range`,
|
|
1333
1374
|
{badge: BADGE.recommendation})}
|
|
@@ -1336,7 +1377,15 @@ VIEWS.modelswitch = async (page) => {
|
|
|
1336
1377
|
${kpi('Stays on current model', fmtUSD(m.blocked_by_context_usd),
|
|
1337
1378
|
`${fmtInt(m.blocked_by_context_requests)} requests too big for the cheaper model's context`, {badge: BADGE.estimated})}
|
|
1338
1379
|
</div>
|
|
1339
|
-
${card('
|
|
1380
|
+
${card('What actually happened when you used a cheaper model',
|
|
1381
|
+
`<div id="ms-ev">${evidenceBody(ev)}</div>`, {badge: BADGE.actual, flush: 1,
|
|
1382
|
+
hint: 'Measured from your own prompts — no repricing, no assumptions about tokens',
|
|
1383
|
+
footer: ev ? esc(ev.caveat) : ''})}
|
|
1384
|
+
<div class="note">The table above is history; the one below is a model. Where they disagree,
|
|
1385
|
+
believe the history: repricing assumes the cheaper model would finish in the same number of
|
|
1386
|
+
turns, and your data shows that is often where the saving goes.</div>
|
|
1387
|
+
${card('Switch these (repriced, not measured)', `<div id="ms-sw"></div>`,
|
|
1388
|
+
{badge: BADGE.recommendation, flush: 1,
|
|
1340
1389
|
hint: 'Each request repriced on the model its work needs, same tokens',
|
|
1341
1390
|
footer: esc(m.caveat)})}
|
|
1342
1391
|
${card('Default model per project', `<div id="ms-pj"></div>`, {badge: BADGE.recommendation, flush: 1,
|