claude-finops 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -287,6 +287,23 @@ place for you to delete once you are happy.
287
287
 
288
288
  ---
289
289
 
290
+ ## Running the trial, not just recommending one
291
+
292
+ The evidence table ends by admitting its own limit: your prompts were never
293
+ randomly assigned to models, so a category can simply have been easier on one of
294
+ them. **Try it →** on any row closes that gap. It pulls prompts you actually
295
+ typed in that category, re-runs them headlessly on the candidate model, and
296
+ prices the result against what they cost the first time.
297
+
298
+ Nothing runs on its own: a trial spends real money and drives a real agent, so
299
+ it takes two clicks and shows the first-time bill before you commit. Runs happen
300
+ in a scratch directory, and headless Claude cannot ask for permission — so tools
301
+ that need it are denied and counted, and a task that needs your repo will look
302
+ smaller than it is. The verdict says which way it went: *confirmed*, *smaller
303
+ than advertised*, or *history overstated it*.
304
+
305
+ ---
306
+
290
307
  ## Live model advice
291
308
 
292
309
  The back-test tells you what to use next time. These tell you mid-session, while
package/finops/api.py CHANGED
@@ -113,6 +113,15 @@ class Handler(BaseHTTPRequestHandler):
113
113
  if path.startswith("/api/live/"):
114
114
  self._payload = payload
115
115
  return self.session_action(path)
116
+ if path == "/api/trial/run":
117
+ # Spends real money and drives a real agent, so it is guarded like
118
+ # the other side-effecting routes and only ever reached by a click.
119
+ if not self._same_origin():
120
+ return self.send_json({"ok": False, "error": "forbidden"}, 403)
121
+ from .trial import run
122
+ return self.send_json(run(payload.get("category"), payload.get("model"),
123
+ payload.get("prompts") or [],
124
+ cwd=payload.get("cwd") or None))
116
125
  if path.startswith("/api/do/"):
117
126
  if not self._same_origin():
118
127
  return self.send_json({"ok": False, "error": "forbidden"}, 403)
@@ -292,6 +301,11 @@ class Handler(BaseHTTPRequestHandler):
292
301
  return self.send_json(a.context_analysis(f))
293
302
  if route == "waste":
294
303
  return self.send_json(a.waste(f))
304
+ if route == "trial":
305
+ from .trial import samples, available
306
+ return self.send_json({"available": available(),
307
+ "samples": samples(qs.get("category", [""])[0],
308
+ int(qs.get("limit", ["3"])[0]))})
295
309
  if route == "model_evidence":
296
310
  return self.send_json(a.model_evidence(f))
297
311
  if route == "model_switch":
@@ -0,0 +1,169 @@
1
+ """Run the trial the evidence asks for, instead of only recommending one.
2
+
3
+ model_evidence() ends with an honest caveat: your prompts were never randomly
4
+ assigned to models, so a category can simply have been easier on one of them.
5
+ That caveat can only be closed by actually running the same prompt on both
6
+ models and looking at what came back.
7
+
8
+ So: take real prompts out of your own history, re-run them headlessly on the
9
+ candidate model, and compare what they cost against what they cost the first
10
+ time. `claude -p ... --output-format json` reports cost, turns and errors for
11
+ exactly this purpose.
12
+
13
+ Two things this deliberately does not do:
14
+
15
+ * It never runs anything on its own. A trial spends real money and drives a
16
+ real agent, so it happens on an explicit click, with the bill shown first.
17
+ * It runs in a scratch directory by default, not your repo. Headless Claude
18
+ cannot ask for permission, so tools that need it are denied and recorded —
19
+ but a trial that edits your working tree to prove a point is not a trial
20
+ anyone wants.
21
+ """
22
+ import json
23
+ import os
24
+ import shutil
25
+ import subprocess
26
+ import tempfile
27
+ import time
28
+
29
+ from .advisor import model_alias
30
+ from .paths import DB_PATH
31
+
32
+ TIMEOUT_S = 600
33
+ MAX_PROMPTS = 5
34
+ MAX_CHARS = 6000 # a prompt longer than this is usually a paste, not a task
35
+
36
+
37
+ def samples(category, limit=3):
38
+ """Real prompts from this category, with what they actually cost first time.
39
+
40
+ Picks around the middle of the cost distribution: the cheapest prompts are
41
+ usually "yes"/"continue" and the most expensive are outliers, and neither
42
+ tells you much about a model.
43
+ """
44
+ from .analytics import Analytics
45
+ a = Analytics(DB_PATH)
46
+ rows = a.q("""SELECT p.id, p.text, p.models model, p.est_cost_usd cost,
47
+ p.request_count turns, p.tool_calls tools, p.session_id, p.day
48
+ FROM prompts p
49
+ WHERE COALESCE(p.category,'other') = ? AND p.agent = 'claude'
50
+ AND p.models IS NOT NULL AND p.models NOT LIKE '%,%'
51
+ AND p.models != '<synthetic>' AND p.est_cost_usd > 0
52
+ AND LENGTH(p.text) BETWEEN 40 AND ?
53
+ -- Things you actually typed. System turns, tool results, queued
54
+ -- follow-ups and SDK traffic are not tasks anyone would re-run, and
55
+ -- a trial built from task-notification XML measures nothing.
56
+ AND p.source = 'typed'
57
+ AND p.text NOT LIKE '<%' AND p.text NOT LIKE '[Request interrupted%'
58
+ AND p.text NOT LIKE 'Caveat:%' AND p.text NOT LIKE '%<local-command%'
59
+ ORDER BY p.est_cost_usd""", (category, MAX_CHARS))
60
+ if not rows:
61
+ return []
62
+ lo = int(len(rows) * 0.35)
63
+ hi = max(lo + 1, int(len(rows) * 0.85))
64
+ mid = rows[lo:hi] or rows
65
+ step = max(1, len(mid) // limit)
66
+ picked = mid[::step][:limit]
67
+ return [{"prompt_id": r["id"], "text": r["text"], "baseline_model": r["model"],
68
+ "baseline_name": a.pricing.display_name(r["model"]),
69
+ "baseline_cost_usd": round(r["cost"], 4),
70
+ "baseline_turns": r["turns"], "baseline_tools": r["tools"],
71
+ "day": r["day"]} for r in picked]
72
+
73
+
74
+ def available():
75
+ return bool(shutil.which("claude"))
76
+
77
+
78
+ def run_one(text, model, cwd=None, timeout=TIMEOUT_S):
79
+ """One headless run. Returns what it cost and whether it got there."""
80
+ exe = shutil.which("claude")
81
+ if not exe:
82
+ return {"ok": False, "error": "The `claude` CLI is not on PATH."}
83
+ sandbox = cwd or tempfile.mkdtemp(prefix="finops-trial-")
84
+ started = time.time()
85
+ try:
86
+ p = subprocess.run([exe, "-p", text, "--model", model_alias(model),
87
+ "--output-format", "json"],
88
+ capture_output=True, text=True, timeout=timeout, cwd=sandbox)
89
+ except subprocess.TimeoutExpired:
90
+ return {"ok": False, "error": f"Gave up after {timeout}s.",
91
+ "elapsed_s": round(time.time() - started, 1)}
92
+ except OSError as e:
93
+ return {"ok": False, "error": str(e)}
94
+ try:
95
+ d = json.loads(p.stdout)
96
+ except ValueError:
97
+ return {"ok": False, "error": (p.stderr or p.stdout or "no output")[:400],
98
+ "elapsed_s": round(time.time() - started, 1)}
99
+ usage = d.get("usage") or {}
100
+ return {
101
+ "ok": not d.get("is_error"),
102
+ "cost_usd": d.get("total_cost_usd"),
103
+ "turns": d.get("num_turns"),
104
+ "elapsed_s": round(time.time() - started, 1),
105
+ "duration_api_ms": d.get("duration_api_ms"),
106
+ "output_tokens": usage.get("output_tokens"),
107
+ "input_tokens": usage.get("input_tokens"),
108
+ "denials": len(d.get("permission_denials") or []),
109
+ "stop_reason": d.get("stop_reason") or d.get("terminal_reason"),
110
+ "result": (d.get("result") or "")[:4000],
111
+ "session_id": d.get("session_id"),
112
+ "sandbox": sandbox,
113
+ }
114
+
115
+
116
+ def _verdict(runs, baseline_cost):
117
+ """What the runs actually showed, in the same language as the back-test."""
118
+ done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
119
+ if not done:
120
+ return "failed", ("Every run errored or produced nothing, so this says nothing about "
121
+ "cost. Check the errors below before reading anything into it.")
122
+ got = sum(r["cost_usd"] for r in done)
123
+ base = sum(baseline_cost)
124
+ denials = sum(r.get("denials") or 0 for r in done)
125
+ if base <= 0:
126
+ return "unclear", "No usable baseline cost for these prompts."
127
+ pct = 100.0 * (1 - got / base)
128
+ tail = (f" {denials} tool call(s) were denied because a headless run cannot ask for "
129
+ f"permission, so the real task may be larger than this." if denials else "")
130
+ if len(done) < len(runs):
131
+ tail += f" {len(runs) - len(done)} of {len(runs)} runs failed and are excluded."
132
+ if pct >= 25:
133
+ return "confirmed", (f"Re-running your own prompts cost {pct:.0f}% less on the cheaper "
134
+ f"model (${got:.2f} against ${base:.2f} the first time).{tail}")
135
+ if pct >= 0:
136
+ return "marginal", (f"Only {pct:.0f}% cheaper on a re-run (${got:.2f} against "
137
+ f"${base:.2f}). Not the saving the history suggested.{tail}")
138
+ return "contradicted", (f"The re-run cost {abs(pct):.0f}% *more* (${got:.2f} against "
139
+ f"${base:.2f}). The history overstated this switch.{tail}")
140
+
141
+
142
+ def run(category, model, prompts, cwd=None):
143
+ """Run a set of sample prompts on a candidate model and judge the result."""
144
+ if not prompts:
145
+ return {"ok": False, "error": "Nothing to run."}
146
+ prompts = prompts[:MAX_PROMPTS]
147
+ runs, baseline = [], []
148
+ for item in prompts:
149
+ text = (item.get("text") or "").strip()
150
+ if not text:
151
+ continue
152
+ r = run_one(text, model, cwd=cwd)
153
+ r["prompt"] = text[:300]
154
+ r["baseline_cost_usd"] = item.get("baseline_cost_usd") or 0
155
+ r["baseline_turns"] = item.get("baseline_turns")
156
+ runs.append(r)
157
+ baseline.append(r["baseline_cost_usd"])
158
+ verdict, why = _verdict(runs, baseline)
159
+ done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
160
+ return {
161
+ "ok": True, "category": category, "model": model, "alias": model_alias(model),
162
+ "runs": runs, "verdict": verdict, "why": why,
163
+ "measured_cost_usd": round(sum(r["cost_usd"] for r in done), 4),
164
+ "baseline_cost_usd": round(sum(baseline), 4),
165
+ "ran_at": int(time.time()),
166
+ "note": ("Measured by re-running these exact prompts headlessly. The agent had no "
167
+ "permission to use tools that ask, and worked in a scratch directory, so a "
168
+ "task that needs your repo will look smaller here than it is."),
169
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-finops",
3
- "version": "0.6.0",
3
+ "version": "0.7.0",
4
4
  "description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
5
5
  "bin": {
6
6
  "claude-finops": "bin/claude-finops.js"
package/web/app.js CHANGED
@@ -1354,6 +1354,8 @@ function evidenceBody(ev) {
1354
1354
  statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
1355
1355
  {h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
1356
1356
  ? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
1357
+ {h: '', f: r => `<button class="act ghost trial-btn" data-cat="${esc(r.c.category)}"
1358
+ data-model="${esc(r.x.model)}" data-name="${esc(r.x.name)}">Try it →</button>`},
1357
1359
  ], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
1358
1360
  }
1359
1361
 
@@ -1416,6 +1418,7 @@ VIEWS.modelswitch = async (page) => {
1416
1418
  {h: 'Could save', num: 1, f: r => fmtUSD(r.estimated_savings_usd)},
1417
1419
  {h: 'Why', f: r => esc(r.why)},
1418
1420
  ], m.projects);
1421
+ wireTrials(page);
1419
1422
  addChart(page, 'Savings by switch', el => C.barsH(el, {
1420
1423
  rows: m.switches.slice(0, 10), label: r => clip(`${r.scope} → ${r.recommended_name}`, 48),
1421
1424
  value: r => r.estimated_savings_usd,
@@ -1426,6 +1429,116 @@ VIEWS.modelswitch = async (page) => {
1426
1429
  {badge: BADGE.recommendation, hint: 'Colour = confidence (green high, blue medium, amber low)'});
1427
1430
  };
1428
1431
 
1432
+ /* ---------- trial: stop recommending, start measuring ----------
1433
+ The evidence ends at "strong evidence for a trial, not proof". This runs the
1434
+ trial: real prompts out of your own history, re-run headlessly on the
1435
+ candidate model, priced against what they cost the first time. It spends real
1436
+ money, so nothing happens without two clicks. */
1437
+ function trialPanelHTML(cat, model, name, s) {
1438
+ if (!s.available) {
1439
+ return `<div class="empty">The <code>claude</code> CLI is not on PATH, so a trial cannot be
1440
+ run from here.</div>`;
1441
+ }
1442
+ if (!s.samples.length) {
1443
+ return `<div class="empty">No prompt you actually typed in this category is short enough to
1444
+ re-run safely.</div>`;
1445
+ }
1446
+ const base = s.samples.reduce((a, x) => a + (x.baseline_cost_usd || 0), 0);
1447
+ return `
1448
+ <div class="dt">These are ${s.samples.length} prompts you really sent in
1449
+ <b>${esc(cat.replace('_', ' '))}</b>. Running them again on <b>${esc(name)}</b> costs money —
1450
+ they cost ${fmtUSD(base)} the first time, and the cheaper model should come in under that.</div>
1451
+ <div class="stack trial-samples">${s.samples.map((x, i) => `
1452
+ <div class="dt trial-s" data-i="${i}">
1453
+ <span class="note">${esc(x.day)} · ${esc(x.baseline_name)} · ${fmtUSD(x.baseline_cost_usd)} ·
1454
+ ${fmtInt(x.baseline_turns)} turns</span>
1455
+ <div class="trial-text">${esc(x.text.slice(0, 400))}${x.text.length > 400 ? '…' : ''}</div>
1456
+ </div>`).join('')}</div>
1457
+ <div class="dt note">Runs headlessly in a scratch directory. Tools that need permission are
1458
+ denied, because a headless agent cannot ask — so a task that needs your repo will look
1459
+ smaller here than it really is.</div>
1460
+ <div class="live-actions">
1461
+ <button class="act trial-run">▶ Run ${s.samples.length} prompts on ${esc(name)}</button>
1462
+ <button class="act ghost trial-copy">Copy the first prompt instead</button>
1463
+ <span class="trial-msg note"></span>
1464
+ </div>
1465
+ <div class="trial-out"></div>`;
1466
+ }
1467
+
1468
+ const TRIAL_VERDICT = {
1469
+ confirmed: ['healthy', 'Confirmed by running it'],
1470
+ marginal: ['high', 'Smaller than advertised'],
1471
+ contradicted: ['critical', 'History overstated it'],
1472
+ failed: ['critical', 'Runs failed'],
1473
+ unclear: ['high', 'Inconclusive'],
1474
+ };
1475
+
1476
+ function trialResultHTML(r) {
1477
+ const v = TRIAL_VERDICT[r.verdict] || TRIAL_VERDICT.unclear;
1478
+ return `<div class="dt"><b>${statusChip(v[0], v[1])}</b> ${esc(r.why)}</div>` + table([
1479
+ {h: 'Prompt', trunc: 1, f: x => esc(x.prompt)},
1480
+ {h: 'First time', num: 1, f: x => fmtUSD(x.baseline_cost_usd)},
1481
+ {h: 'On ' + esc(r.alias), num: 1, f: x => x.ok ? fmtUSD(x.cost_usd) : '—'},
1482
+ {h: 'Turns', num: 1, f: x => x.ok ? fmtInt(x.turns) : '—'},
1483
+ {h: 'Took', num: 1, f: x => x.ok ? x.elapsed_s + 's' : '—'},
1484
+ {h: 'Denied', num: 1, f: x => x.denials ? fmtInt(x.denials) : ''},
1485
+ {h: 'Result', f: x => x.ok ? `<span class="note">${esc((x.result || '').slice(0, 120))}</span>`
1486
+ : `<span class="status critical">${esc((x.error || 'failed').slice(0, 120))}</span>`},
1487
+ ], r.runs) + `<div class="note" style="padding:8px 14px">${esc(r.note)}</div>`;
1488
+ }
1489
+
1490
+ function wireTrials(page) {
1491
+ page.querySelectorAll('.trial-btn').forEach(b => b.onclick = async () => {
1492
+ const row = b.closest('tr');
1493
+ if (row.nextElementSibling?.classList.contains('trial-row')) {
1494
+ row.nextElementSibling.remove(); return;
1495
+ }
1496
+ const {cat, model, name} = b.dataset;
1497
+ const tr = h(`<tr class="trial-row"><td colspan="10"><div class="trial-panel">
1498
+ <div class="empty">Finding prompts you sent…</div></div></td></tr>`);
1499
+ row.after(tr);
1500
+ const host = tr.querySelector('.trial-panel');
1501
+ let s;
1502
+ try {
1503
+ s = await fetch(`/api/trial?category=${encodeURIComponent(cat)}&limit=3`).then(r => r.json());
1504
+ } catch (e) { host.innerHTML = `<div class="empty">${esc(e.message)}</div>`; return; }
1505
+ host.innerHTML = trialPanelHTML(cat, model, name, s);
1506
+ const msg = host.querySelector('.trial-msg');
1507
+ host.querySelector('.trial-copy')?.addEventListener('click', async () => {
1508
+ try { await navigator.clipboard.writeText(s.samples[0].text); msg.textContent = 'Copied.'; }
1509
+ catch { msg.textContent = 'Could not copy.'; }
1510
+ });
1511
+ host.querySelector('.trial-run')?.addEventListener('click', async ev => {
1512
+ const btn = ev.currentTarget;
1513
+ if (btn.dataset.armed !== '1') {
1514
+ btn.dataset.armed = '1';
1515
+ btn.textContent = 'Click again to spend real money';
1516
+ btn.classList.add('warn');
1517
+ return;
1518
+ }
1519
+ btn.disabled = true;
1520
+ btn.textContent = 'Running…';
1521
+ msg.textContent = 'Each prompt runs to completion; this can take a few minutes.';
1522
+ try {
1523
+ const r = await fetch('/api/trial/run', {
1524
+ method: 'POST',
1525
+ headers: {'X-FinOps-Action': '1', 'Content-Type': 'application/json'},
1526
+ body: JSON.stringify({category: cat, model, prompts: s.samples}),
1527
+ }).then(x => x.json());
1528
+ host.querySelector('.trial-out').innerHTML = r.ok
1529
+ ? trialResultHTML(r) : `<div class="empty">${esc(r.error || 'failed')}</div>`;
1530
+ msg.textContent = '';
1531
+ } catch (e) {
1532
+ msg.textContent = e.message;
1533
+ }
1534
+ btn.disabled = false;
1535
+ btn.classList.remove('warn');
1536
+ btn.textContent = '▶ Run again';
1537
+ btn.dataset.armed = '';
1538
+ });
1539
+ });
1540
+ }
1541
+
1429
1542
  /* ---------- waste ---------- */
1430
1543
  VIEWS.waste = async (page) => {
1431
1544
  const w = await api('waste');
package/web/styles.css CHANGED
@@ -425,3 +425,13 @@ table.tbl .sub { font-size:10.5px; color:var(--muted); }
425
425
  so it reads as a note with an accent edge rather than a warning. */
426
426
  .switch-tip { border-left: 2px solid var(--accent, #eb6834); padding-left: 9px; }
427
427
  .switch-tip .act { margin-left: 8px; }
428
+
429
+ /* Trial panel: an experiment you are about to pay for, so it reads as a quiet
430
+ workbench rather than another recommendation. */
431
+ .trial-row > td { padding: 0 !important; background: var(--surface-2); }
432
+ .trial-panel { padding: 12px 14px; display: grid; gap: 8px;
433
+ border-left: 2px solid var(--accent, #eb6834); }
434
+ .trial-text { font-size: 12px; margin-top: 3px; white-space: pre-wrap;
435
+ max-height: 76px; overflow: auto; color: var(--text-2); }
436
+ .trial-s { border-top: 1px solid var(--border); padding-top: 7px; }
437
+ .trial-out:empty { display: none; }