claude-finops 0.7.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -47
- package/config/pricing.json +62 -42
- package/config/settings.json +11 -5
- package/finops/actions.py +1 -1
- package/finops/analytics.py +690 -611
- package/finops/api.py +174 -55
- package/finops/diagnose.py +17 -20
- package/finops/etl.py +149 -52
- package/finops/integrate.py +22 -57
- package/finops/pricing.py +57 -6
- package/finops/procs.py +7 -5
- package/finops/report.py +29 -23
- package/finops/segments.py +149 -0
- package/package.json +1 -1
- package/run.cmd +3 -2
- package/run.py +26 -59
- package/web/app.js +253 -329
- package/web/charts.js +26 -15
- package/finops/advisor.py +0 -188
- package/finops/trial.py +0 -169
package/web/charts.js
CHANGED
|
@@ -4,6 +4,9 @@ const NS = 'http://www.w3.org/2000/svg';
|
|
|
4
4
|
export const SERIES = ['--s1','--s2','--s3','--s4','--s5','--s6','--s7','--s8'];
|
|
5
5
|
export const seriesVar = i => `var(${SERIES[i % SERIES.length]})`;
|
|
6
6
|
|
|
7
|
+
export const esc = s => String(s ?? '').replace(/[&<>"']/g, c =>
|
|
8
|
+
({'&': '&', '<': '<', '>': '>', '"': '"', "'": '''}[c]));
|
|
9
|
+
|
|
7
10
|
let tipEl = null;
|
|
8
11
|
function tip() {
|
|
9
12
|
if (!tipEl) { tipEl = document.createElement('div'); tipEl.className = 'tip';
|
|
@@ -140,8 +143,8 @@ export function timeSeries(host, opts) {
|
|
|
140
143
|
cross.setAttribute('x2', type === 'bar' ? cx(i) : px(i));
|
|
141
144
|
const body = active.map(s => `<div class="row"><span class="k">
|
|
142
145
|
<span class="swatch" style="background:${s.color || seriesVar(series.indexOf(s))}"></span>
|
|
143
|
-
${s.label}</span><span class="v">${(s.fmt || fmt)(+r[s.key] || 0)}</span></div>`).join('');
|
|
144
|
-
showTip(`<div class="t">${xLabel(r[xk])}</div>${body}`, ev.clientX, ev.clientY);
|
|
146
|
+
${esc(s.label)}</span><span class="v">${(s.fmt || fmt)(+r[s.key] || 0)}</span></div>`).join('');
|
|
147
|
+
showTip(`<div class="t">${esc(xLabel(r[xk]))}</div>${body}`, ev.clientX, ev.clientY);
|
|
145
148
|
});
|
|
146
149
|
hit.addEventListener('mouseleave', () => { hideTip(); cross.setAttribute('opacity', 0); });
|
|
147
150
|
if (onClick) hit.addEventListener('click', ev => onClick(rows[idxAt(ev)]));
|
|
@@ -170,7 +173,7 @@ export function barsH(host, opts) {
|
|
|
170
173
|
grp.appendChild(el('rect', {x: lw, y: y0 + 4, width: w, height: 11, rx: 4, fill: col}));
|
|
171
174
|
grp.appendChild(el('text', {x: W - 2, y: y0 + 13, 'text-anchor': 'end', class: 'val'}, fmt(v)));
|
|
172
175
|
grp.addEventListener('mousemove', ev => showTip(
|
|
173
|
-
`<div class="t">${label(r)}</div><div class="row"><span class="k">
|
|
176
|
+
`<div class="t">${esc(label(r))}</div><div class="row"><span class="k">
|
|
174
177
|
<span class="swatch" style="background:${col}"></span>Value</span>
|
|
175
178
|
<span class="v">${fmt(v)}</span></div>${sub ? sub(r) : ''}`, ev.clientX, ev.clientY));
|
|
176
179
|
grp.addEventListener('mouseleave', hideTip);
|
|
@@ -205,7 +208,7 @@ export function donut(host, opts) {
|
|
|
205
208
|
L${x3},${y3} A${r0},${r0} 0 ${large} 0 ${x4},${y4} Z`, fill: col,
|
|
206
209
|
style: onClick ? 'cursor:pointer' : ''});
|
|
207
210
|
p.addEventListener('mousemove', ev => showTip(
|
|
208
|
-
`<div class="t">${label(row)}</div><div class="row"><span class="k">
|
|
211
|
+
`<div class="t">${esc(label(row))}</div><div class="row"><span class="k">
|
|
209
212
|
<span class="swatch" style="background:${col}"></span>Estimated</span>
|
|
210
213
|
<span class="v">${fmt(v)}</span></div><div class="row"><span class="k">Share</span>
|
|
211
214
|
<span class="v">${(100 * v / total).toFixed(1)}%</span></div>`, ev.clientX, ev.clientY));
|
|
@@ -255,23 +258,29 @@ export function forecastFan(host, {history, scenarios, remainingDays, height = 2
|
|
|
255
258
|
let cum = 0;
|
|
256
259
|
const hist = history.map(d => ({day: d.day, v: (cum += d.cost)}));
|
|
257
260
|
const base = cum, n = hist.length, total = n + remainingDays;
|
|
261
|
+
// scenarios may only carry 'expected' (insufficient_history): iterate only the
|
|
262
|
+
// keys actually present so a caller that forgets to check that flag cannot crash.
|
|
263
|
+
const keys = ['conservative', 'expected', 'high'].filter(k => scenarios[k]);
|
|
258
264
|
const paths = {};
|
|
259
|
-
for (const k of
|
|
265
|
+
for (const k of keys) {
|
|
260
266
|
const rate = scenarios[k].daily_rate;
|
|
261
267
|
paths[k] = Array.from({length: remainingDays + 1}, (_, i) => base + rate * i);
|
|
262
268
|
}
|
|
263
|
-
const
|
|
269
|
+
const hasBand = paths.conservative && paths.high;
|
|
270
|
+
const max = nice(Math.max(base, ...(paths.high || paths.expected)) || 1);
|
|
264
271
|
const X = i => m.l + (total <= 1 ? 0 : i * (iw / (total - 1)));
|
|
265
272
|
const Y = v => m.t + ih - (v / max) * ih;
|
|
266
273
|
const svg = el('svg', {viewBox: `0 0 ${W} ${H}`, height: H});
|
|
267
274
|
const gi = el('g', {transform: `translate(${m.l},0)`});
|
|
268
275
|
axisLeft(gi, Y, max, iw, 4, fmtUSD); svg.appendChild(gi);
|
|
269
276
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
277
|
+
if (hasBand) {
|
|
278
|
+
const band = [];
|
|
279
|
+
paths.high.forEach((v, i) => band.push(`${i ? 'L' : 'M'}${X(n - 1 + i)},${Y(v)}`));
|
|
280
|
+
for (let i = paths.conservative.length - 1; i >= 0; i--)
|
|
281
|
+
band.push(`L${X(n - 1 + i)},${Y(paths.conservative[i])}`);
|
|
282
|
+
svg.appendChild(el('path', {d: band.join(' ') + ' Z', fill: 'var(--s1)', 'fill-opacity': .13}));
|
|
283
|
+
}
|
|
275
284
|
|
|
276
285
|
const line = (pts, color, dash) => el('path', {
|
|
277
286
|
d: pts.map((p, i) => (i ? 'L' : 'M') + p[0] + ',' + p[1]).join(' '), fill: 'none',
|
|
@@ -286,8 +295,10 @@ export function forecastFan(host, {history, scenarios, remainingDays, height = 2
|
|
|
286
295
|
svg.appendChild(el('circle', {cx: x, cy: y, r: 3.5, fill: c, stroke: 'var(--surface)',
|
|
287
296
|
'stroke-width': 2}));
|
|
288
297
|
});
|
|
289
|
-
|
|
290
|
-
|
|
298
|
+
if (paths.high) {
|
|
299
|
+
svg.appendChild(el('text', {x: X(total - 1), y: Y(paths.high.at(-1)) - 6, 'text-anchor': 'end',
|
|
300
|
+
class: 'val'}, 'High ' + fmtUSD(paths.high.at(-1))));
|
|
301
|
+
}
|
|
291
302
|
svg.appendChild(el('text', {x: X(total - 1), y: Y(paths.expected.at(-1)) - 6, 'text-anchor': 'end',
|
|
292
303
|
class: 'val'}, 'Expected ' + fmtUSD(paths.expected.at(-1))));
|
|
293
304
|
svg.appendChild(el('text', {x: X(0), y: H - 8}, history[0].day));
|
|
@@ -304,7 +315,7 @@ export function forecastFan(host, {history, scenarios, remainingDays, height = 2
|
|
|
304
315
|
ev.clientX, ev.clientY);
|
|
305
316
|
else {
|
|
306
317
|
const j = i - n + 1;
|
|
307
|
-
showTip(`<div class="t">Day +${j} · forecast</div>` +
|
|
318
|
+
showTip(`<div class="t">Day +${j} · forecast</div>` + keys
|
|
308
319
|
.map(k => `<div class="row"><span class="k">${k}</span>
|
|
309
320
|
<span class="v">${fmtUSD(paths[k][j])}</span></div>`).join(''), ev.clientX, ev.clientY);
|
|
310
321
|
}
|
|
@@ -333,7 +344,7 @@ export function legend(host, items, onToggle) {
|
|
|
333
344
|
items.forEach(it => {
|
|
334
345
|
const s = document.createElement('span');
|
|
335
346
|
s.className = 'it' + (it.off ? ' off' : '');
|
|
336
|
-
s.innerHTML = `<span class="swatch" style="background:${it.color}"></span>${it.label}`;
|
|
347
|
+
s.innerHTML = `<span class="swatch" style="background:${it.color}"></span>${esc(it.label)}`;
|
|
337
348
|
if (onToggle) { s.style.cursor = 'pointer'; s.onclick = () => onToggle(it.key); }
|
|
338
349
|
host.appendChild(s);
|
|
339
350
|
});
|
package/finops/advisor.py
DELETED
|
@@ -1,188 +0,0 @@
|
|
|
1
|
-
"""Live model advice: what to switch to, while the session is still running.
|
|
2
|
-
|
|
3
|
-
The back-test in analytics.model_evidence() answers "what should I have used?",
|
|
4
|
-
which is the wrong tense once a session is under way. This module answers it in
|
|
5
|
-
the present: given the prompts a session has actually sent and the model it is
|
|
6
|
-
on, say whether your own history already shows a cheaper model doing this kind
|
|
7
|
-
of work without taking more turns.
|
|
8
|
-
|
|
9
|
-
It is built to run three ways, so the advice is the same wherever you meet it:
|
|
10
|
-
|
|
11
|
-
* the dashboard's Running sessions view (per live session)
|
|
12
|
-
* a UserPromptSubmit hook (per prompt, before the turn runs)
|
|
13
|
-
* a statusline command (continuously, in a few characters)
|
|
14
|
-
|
|
15
|
-
The hook and statusline run on every prompt, so the evidence is computed once
|
|
16
|
-
and cached; a warehouse query per keystroke would be felt. Everything here fails
|
|
17
|
-
soft and silent — advice that breaks your terminal is worse than no advice.
|
|
18
|
-
"""
|
|
19
|
-
import json
|
|
20
|
-
import os
|
|
21
|
-
import time
|
|
22
|
-
|
|
23
|
-
from .classify import classify
|
|
24
|
-
from .paths import DATA_DIR, DB_PATH, ensure_dirs
|
|
25
|
-
|
|
26
|
-
CACHE = os.path.join(DATA_DIR, "advice_cache.json")
|
|
27
|
-
TTL_S = 3600
|
|
28
|
-
RECENT_PROMPTS = 12 # how much of the session counts as "what it is doing now"
|
|
29
|
-
MIN_SAVING_PCT = 25 # below this, interrupting someone is not worth it
|
|
30
|
-
|
|
31
|
-
# `/model <name>` takes a family name, not the pricing table's id.
|
|
32
|
-
ALIASES = ("opus", "sonnet", "haiku", "fable")
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def model_alias(model_id):
|
|
36
|
-
"""'claude-fable-5-1' -> 'fable'. Falls back to the full id we were given."""
|
|
37
|
-
low = str(model_id or "").lower()
|
|
38
|
-
for a in ALIASES:
|
|
39
|
-
if a in low:
|
|
40
|
-
return a
|
|
41
|
-
return model_id
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
# ---------------------------------------------------------------- evidence ----
|
|
45
|
-
|
|
46
|
-
def _build_evidence():
|
|
47
|
-
"""Pull the back-test into the small shape the live surfaces need."""
|
|
48
|
-
from .analytics import Analytics
|
|
49
|
-
a = Analytics(DB_PATH)
|
|
50
|
-
ev = a.model_evidence({})
|
|
51
|
-
out = {}
|
|
52
|
-
for c in ev["categories"]:
|
|
53
|
-
cands = [{"model": x["model"], "name": x["name"], "verdict": x["verdict"],
|
|
54
|
-
"cost_per_prompt": round(x["cost_per_prompt"], 3),
|
|
55
|
-
"savings_pct": x["savings_pct"], "turn_ratio": x["turn_ratio"],
|
|
56
|
-
"prompts": x["prompts"], "why": x["why"]}
|
|
57
|
-
for x in c["candidates"]]
|
|
58
|
-
out[c["category"]] = {"current": c["current"]["model"],
|
|
59
|
-
"current_name": c["current"]["name"],
|
|
60
|
-
"current_cost_per_prompt": round(c["current"]["cost_per_prompt"], 3),
|
|
61
|
-
"agent": c.get("agent", "claude"),
|
|
62
|
-
"candidates": cands}
|
|
63
|
-
return out
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
def evidence(force=False):
|
|
67
|
-
"""Cached evidence table, keyed by category. Never raises."""
|
|
68
|
-
try:
|
|
69
|
-
with open(CACHE) as fh:
|
|
70
|
-
c = json.load(fh)
|
|
71
|
-
if not force and time.time() - c.get("built_at", 0) < TTL_S:
|
|
72
|
-
return c.get("categories") or {}
|
|
73
|
-
except (OSError, ValueError):
|
|
74
|
-
pass
|
|
75
|
-
try:
|
|
76
|
-
cats = _build_evidence()
|
|
77
|
-
except Exception:
|
|
78
|
-
return {}
|
|
79
|
-
try:
|
|
80
|
-
ensure_dirs()
|
|
81
|
-
tmp = CACHE + ".tmp"
|
|
82
|
-
with open(tmp, "w") as fh:
|
|
83
|
-
json.dump({"built_at": int(time.time()), "categories": cats}, fh)
|
|
84
|
-
os.replace(tmp, CACHE)
|
|
85
|
-
except OSError:
|
|
86
|
-
pass
|
|
87
|
-
return cats
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
# ---------------------------------------------------------------- session ----
|
|
91
|
-
|
|
92
|
-
def recent_prompts(transcript, n=RECENT_PROMPTS):
|
|
93
|
-
"""The last n human prompts in a live transcript, newest last.
|
|
94
|
-
|
|
95
|
-
Reads the tail only: an active session's JSONL runs to tens of megabytes and
|
|
96
|
-
this is on the path of every prompt you type.
|
|
97
|
-
"""
|
|
98
|
-
try:
|
|
99
|
-
size = os.path.getsize(transcript)
|
|
100
|
-
with open(transcript, "rb") as fh:
|
|
101
|
-
fh.seek(max(0, size - 400_000))
|
|
102
|
-
lines = fh.read().decode("utf-8", "replace").splitlines()[1:]
|
|
103
|
-
except OSError:
|
|
104
|
-
return []
|
|
105
|
-
out = []
|
|
106
|
-
for line in reversed(lines):
|
|
107
|
-
if '"type":"user"' not in line and '"type": "user"' not in line:
|
|
108
|
-
continue
|
|
109
|
-
try:
|
|
110
|
-
d = json.loads(line)
|
|
111
|
-
except ValueError:
|
|
112
|
-
continue
|
|
113
|
-
if d.get("isMeta") or d.get("isSidechain"):
|
|
114
|
-
continue
|
|
115
|
-
msg = d.get("message") or {}
|
|
116
|
-
content = msg.get("content")
|
|
117
|
-
if isinstance(content, list):
|
|
118
|
-
content = " ".join(b.get("text", "") for b in content
|
|
119
|
-
if isinstance(b, dict) and b.get("type") == "text")
|
|
120
|
-
if not isinstance(content, str) or not content.strip():
|
|
121
|
-
continue
|
|
122
|
-
if content.lstrip().startswith(("<", "[Request interrupted")):
|
|
123
|
-
continue # tool results and interrupt markers are not prompts
|
|
124
|
-
out.append(content)
|
|
125
|
-
if len(out) >= n:
|
|
126
|
-
break
|
|
127
|
-
return list(reversed(out))
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
def category_of(texts):
|
|
131
|
-
"""The category this stretch of work is in, by weight of confidence."""
|
|
132
|
-
scores = {}
|
|
133
|
-
for t in texts:
|
|
134
|
-
cat, conf, _ = classify(t)
|
|
135
|
-
scores[cat] = scores.get(cat, 0) + max(conf, 0.1)
|
|
136
|
-
if not scores:
|
|
137
|
-
return "other", 0.0
|
|
138
|
-
cat = max(scores, key=scores.get)
|
|
139
|
-
return cat, round(scores[cat] / sum(scores.values()), 2)
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
def advise(model=None, transcript=None, prompt=None, ev=None):
|
|
143
|
-
"""What to say about a session that is running right now.
|
|
144
|
-
|
|
145
|
-
model the model the session is on (pricing-table id)
|
|
146
|
-
transcript path to the live JSONL, for reading what it has been doing
|
|
147
|
-
prompt the prompt about to be sent, when we are called from a hook
|
|
148
|
-
Returns None when there is nothing worth saying.
|
|
149
|
-
"""
|
|
150
|
-
ev = evidence() if ev is None else ev
|
|
151
|
-
if not ev:
|
|
152
|
-
return None
|
|
153
|
-
texts = list(recent_prompts(transcript)) if transcript else []
|
|
154
|
-
if prompt:
|
|
155
|
-
# The prompt in hand is what the next turn will cost, so it leads.
|
|
156
|
-
texts = texts[-4:] + [prompt] * 3
|
|
157
|
-
if not texts:
|
|
158
|
-
return None
|
|
159
|
-
cat, conf = category_of(texts)
|
|
160
|
-
row = ev.get(cat)
|
|
161
|
-
if not row:
|
|
162
|
-
return None
|
|
163
|
-
# Only speak when they are on the model the evidence is about. Advising a
|
|
164
|
-
# switch away from a model you already left would be noise.
|
|
165
|
-
if model and row["current"] and model_alias(model) != model_alias(row["current"]):
|
|
166
|
-
return None
|
|
167
|
-
usable = [c for c in row["candidates"]
|
|
168
|
-
if c["verdict"] in ("supported", "caution") and c["savings_pct"] >= MIN_SAVING_PCT]
|
|
169
|
-
if not usable:
|
|
170
|
-
return None
|
|
171
|
-
best = usable[0]
|
|
172
|
-
trial = best["verdict"] == "caution"
|
|
173
|
-
return {
|
|
174
|
-
"category": cat, "confidence": conf,
|
|
175
|
-
"current_model": row["current"], "current_name": row["current_name"],
|
|
176
|
-
"current_cost_per_prompt": row["current_cost_per_prompt"],
|
|
177
|
-
"model": best["model"], "name": best["name"], "alias": model_alias(best["model"]),
|
|
178
|
-
"cost_per_prompt": best["cost_per_prompt"], "savings_pct": best["savings_pct"],
|
|
179
|
-
"turn_ratio": best["turn_ratio"], "prompts": best["prompts"],
|
|
180
|
-
"verdict": best["verdict"], "trial": trial, "why": best["why"],
|
|
181
|
-
"command": f"/model {model_alias(best['model'])}",
|
|
182
|
-
"line": (f"{cat.replace('_', ' ')} on {row['current_name']} "
|
|
183
|
-
f"(${row['current_cost_per_prompt']:.2f}/prompt). Your last {best['prompts']} "
|
|
184
|
-
f"ran on {best['name']} at ${best['cost_per_prompt']:.2f} in "
|
|
185
|
-
f"{best['turn_ratio']}x the turns"
|
|
186
|
-
+ (" — worth a trial here." if trial else ".")),
|
|
187
|
-
"short": f"try {model_alias(best['model'])} (-{best['savings_pct']:.0f}%)",
|
|
188
|
-
}
|
package/finops/trial.py
DELETED
|
@@ -1,169 +0,0 @@
|
|
|
1
|
-
"""Run the trial the evidence asks for, instead of only recommending one.
|
|
2
|
-
|
|
3
|
-
model_evidence() ends with an honest caveat: your prompts were never randomly
|
|
4
|
-
assigned to models, so a category can simply have been easier on one of them.
|
|
5
|
-
That caveat can only be closed by actually running the same prompt on both
|
|
6
|
-
models and looking at what came back.
|
|
7
|
-
|
|
8
|
-
So: take real prompts out of your own history, re-run them headlessly on the
|
|
9
|
-
candidate model, and compare what they cost against what they cost the first
|
|
10
|
-
time. `claude -p ... --output-format json` reports cost, turns and errors for
|
|
11
|
-
exactly this purpose.
|
|
12
|
-
|
|
13
|
-
Two things this deliberately does not do:
|
|
14
|
-
|
|
15
|
-
* It never runs anything on its own. A trial spends real money and drives a
|
|
16
|
-
real agent, so it happens on an explicit click, with the bill shown first.
|
|
17
|
-
* It runs in a scratch directory by default, not your repo. Headless Claude
|
|
18
|
-
cannot ask for permission, so tools that need it are denied and recorded —
|
|
19
|
-
but a trial that edits your working tree to prove a point is not a trial
|
|
20
|
-
anyone wants.
|
|
21
|
-
"""
|
|
22
|
-
import json
|
|
23
|
-
import os
|
|
24
|
-
import shutil
|
|
25
|
-
import subprocess
|
|
26
|
-
import tempfile
|
|
27
|
-
import time
|
|
28
|
-
|
|
29
|
-
from .advisor import model_alias
|
|
30
|
-
from .paths import DB_PATH
|
|
31
|
-
|
|
32
|
-
TIMEOUT_S = 600
|
|
33
|
-
MAX_PROMPTS = 5
|
|
34
|
-
MAX_CHARS = 6000 # a prompt longer than this is usually a paste, not a task
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def samples(category, limit=3):
|
|
38
|
-
"""Real prompts from this category, with what they actually cost first time.
|
|
39
|
-
|
|
40
|
-
Picks around the middle of the cost distribution: the cheapest prompts are
|
|
41
|
-
usually "yes"/"continue" and the most expensive are outliers, and neither
|
|
42
|
-
tells you much about a model.
|
|
43
|
-
"""
|
|
44
|
-
from .analytics import Analytics
|
|
45
|
-
a = Analytics(DB_PATH)
|
|
46
|
-
rows = a.q("""SELECT p.id, p.text, p.models model, p.est_cost_usd cost,
|
|
47
|
-
p.request_count turns, p.tool_calls tools, p.session_id, p.day
|
|
48
|
-
FROM prompts p
|
|
49
|
-
WHERE COALESCE(p.category,'other') = ? AND p.agent = 'claude'
|
|
50
|
-
AND p.models IS NOT NULL AND p.models NOT LIKE '%,%'
|
|
51
|
-
AND p.models != '<synthetic>' AND p.est_cost_usd > 0
|
|
52
|
-
AND LENGTH(p.text) BETWEEN 40 AND ?
|
|
53
|
-
-- Things you actually typed. System turns, tool results, queued
|
|
54
|
-
-- follow-ups and SDK traffic are not tasks anyone would re-run, and
|
|
55
|
-
-- a trial built from task-notification XML measures nothing.
|
|
56
|
-
AND p.source = 'typed'
|
|
57
|
-
AND p.text NOT LIKE '<%' AND p.text NOT LIKE '[Request interrupted%'
|
|
58
|
-
AND p.text NOT LIKE 'Caveat:%' AND p.text NOT LIKE '%<local-command%'
|
|
59
|
-
ORDER BY p.est_cost_usd""", (category, MAX_CHARS))
|
|
60
|
-
if not rows:
|
|
61
|
-
return []
|
|
62
|
-
lo = int(len(rows) * 0.35)
|
|
63
|
-
hi = max(lo + 1, int(len(rows) * 0.85))
|
|
64
|
-
mid = rows[lo:hi] or rows
|
|
65
|
-
step = max(1, len(mid) // limit)
|
|
66
|
-
picked = mid[::step][:limit]
|
|
67
|
-
return [{"prompt_id": r["id"], "text": r["text"], "baseline_model": r["model"],
|
|
68
|
-
"baseline_name": a.pricing.display_name(r["model"]),
|
|
69
|
-
"baseline_cost_usd": round(r["cost"], 4),
|
|
70
|
-
"baseline_turns": r["turns"], "baseline_tools": r["tools"],
|
|
71
|
-
"day": r["day"]} for r in picked]
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
def available():
|
|
75
|
-
return bool(shutil.which("claude"))
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
def run_one(text, model, cwd=None, timeout=TIMEOUT_S):
|
|
79
|
-
"""One headless run. Returns what it cost and whether it got there."""
|
|
80
|
-
exe = shutil.which("claude")
|
|
81
|
-
if not exe:
|
|
82
|
-
return {"ok": False, "error": "The `claude` CLI is not on PATH."}
|
|
83
|
-
sandbox = cwd or tempfile.mkdtemp(prefix="finops-trial-")
|
|
84
|
-
started = time.time()
|
|
85
|
-
try:
|
|
86
|
-
p = subprocess.run([exe, "-p", text, "--model", model_alias(model),
|
|
87
|
-
"--output-format", "json"],
|
|
88
|
-
capture_output=True, text=True, timeout=timeout, cwd=sandbox)
|
|
89
|
-
except subprocess.TimeoutExpired:
|
|
90
|
-
return {"ok": False, "error": f"Gave up after {timeout}s.",
|
|
91
|
-
"elapsed_s": round(time.time() - started, 1)}
|
|
92
|
-
except OSError as e:
|
|
93
|
-
return {"ok": False, "error": str(e)}
|
|
94
|
-
try:
|
|
95
|
-
d = json.loads(p.stdout)
|
|
96
|
-
except ValueError:
|
|
97
|
-
return {"ok": False, "error": (p.stderr or p.stdout or "no output")[:400],
|
|
98
|
-
"elapsed_s": round(time.time() - started, 1)}
|
|
99
|
-
usage = d.get("usage") or {}
|
|
100
|
-
return {
|
|
101
|
-
"ok": not d.get("is_error"),
|
|
102
|
-
"cost_usd": d.get("total_cost_usd"),
|
|
103
|
-
"turns": d.get("num_turns"),
|
|
104
|
-
"elapsed_s": round(time.time() - started, 1),
|
|
105
|
-
"duration_api_ms": d.get("duration_api_ms"),
|
|
106
|
-
"output_tokens": usage.get("output_tokens"),
|
|
107
|
-
"input_tokens": usage.get("input_tokens"),
|
|
108
|
-
"denials": len(d.get("permission_denials") or []),
|
|
109
|
-
"stop_reason": d.get("stop_reason") or d.get("terminal_reason"),
|
|
110
|
-
"result": (d.get("result") or "")[:4000],
|
|
111
|
-
"session_id": d.get("session_id"),
|
|
112
|
-
"sandbox": sandbox,
|
|
113
|
-
}
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
def _verdict(runs, baseline_cost):
|
|
117
|
-
"""What the runs actually showed, in the same language as the back-test."""
|
|
118
|
-
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
119
|
-
if not done:
|
|
120
|
-
return "failed", ("Every run errored or produced nothing, so this says nothing about "
|
|
121
|
-
"cost. Check the errors below before reading anything into it.")
|
|
122
|
-
got = sum(r["cost_usd"] for r in done)
|
|
123
|
-
base = sum(baseline_cost)
|
|
124
|
-
denials = sum(r.get("denials") or 0 for r in done)
|
|
125
|
-
if base <= 0:
|
|
126
|
-
return "unclear", "No usable baseline cost for these prompts."
|
|
127
|
-
pct = 100.0 * (1 - got / base)
|
|
128
|
-
tail = (f" {denials} tool call(s) were denied because a headless run cannot ask for "
|
|
129
|
-
f"permission, so the real task may be larger than this." if denials else "")
|
|
130
|
-
if len(done) < len(runs):
|
|
131
|
-
tail += f" {len(runs) - len(done)} of {len(runs)} runs failed and are excluded."
|
|
132
|
-
if pct >= 25:
|
|
133
|
-
return "confirmed", (f"Re-running your own prompts cost {pct:.0f}% less on the cheaper "
|
|
134
|
-
f"model (${got:.2f} against ${base:.2f} the first time).{tail}")
|
|
135
|
-
if pct >= 0:
|
|
136
|
-
return "marginal", (f"Only {pct:.0f}% cheaper on a re-run (${got:.2f} against "
|
|
137
|
-
f"${base:.2f}). Not the saving the history suggested.{tail}")
|
|
138
|
-
return "contradicted", (f"The re-run cost {abs(pct):.0f}% *more* (${got:.2f} against "
|
|
139
|
-
f"${base:.2f}). The history overstated this switch.{tail}")
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
def run(category, model, prompts, cwd=None):
|
|
143
|
-
"""Run a set of sample prompts on a candidate model and judge the result."""
|
|
144
|
-
if not prompts:
|
|
145
|
-
return {"ok": False, "error": "Nothing to run."}
|
|
146
|
-
prompts = prompts[:MAX_PROMPTS]
|
|
147
|
-
runs, baseline = [], []
|
|
148
|
-
for item in prompts:
|
|
149
|
-
text = (item.get("text") or "").strip()
|
|
150
|
-
if not text:
|
|
151
|
-
continue
|
|
152
|
-
r = run_one(text, model, cwd=cwd)
|
|
153
|
-
r["prompt"] = text[:300]
|
|
154
|
-
r["baseline_cost_usd"] = item.get("baseline_cost_usd") or 0
|
|
155
|
-
r["baseline_turns"] = item.get("baseline_turns")
|
|
156
|
-
runs.append(r)
|
|
157
|
-
baseline.append(r["baseline_cost_usd"])
|
|
158
|
-
verdict, why = _verdict(runs, baseline)
|
|
159
|
-
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
160
|
-
return {
|
|
161
|
-
"ok": True, "category": category, "model": model, "alias": model_alias(model),
|
|
162
|
-
"runs": runs, "verdict": verdict, "why": why,
|
|
163
|
-
"measured_cost_usd": round(sum(r["cost_usd"] for r in done), 4),
|
|
164
|
-
"baseline_cost_usd": round(sum(baseline), 4),
|
|
165
|
-
"ran_at": int(time.time()),
|
|
166
|
-
"note": ("Measured by re-running these exact prompts headlessly. The agent had no "
|
|
167
|
-
"permission to use tools that ask, and worked in a scratch directory, so a "
|
|
168
|
-
"task that needs your repo will look smaller here than it is."),
|
|
169
|
-
}
|