claude-finops 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -0
- package/finops/api.py +14 -0
- package/finops/trial.py +169 -0
- package/package.json +1 -1
- package/web/app.js +113 -0
- package/web/styles.css +10 -0
package/README.md
CHANGED
|
@@ -287,6 +287,23 @@ place for you to delete once you are happy.
|
|
|
287
287
|
|
|
288
288
|
---
|
|
289
289
|
|
|
290
|
+
## Running the trial, not just recommending one
|
|
291
|
+
|
|
292
|
+
The evidence table ends by admitting its own limit: your prompts were never
|
|
293
|
+
randomly assigned to models, so a category can simply have been easier on one of
|
|
294
|
+
them. **Try it →** on any row closes that gap. It pulls prompts you actually
|
|
295
|
+
typed in that category, re-runs them headlessly on the candidate model, and
|
|
296
|
+
prices the result against what they cost the first time.
|
|
297
|
+
|
|
298
|
+
Nothing runs on its own: a trial spends real money and drives a real agent, so
|
|
299
|
+
it takes two clicks and shows the first-time bill before you commit. Runs happen
|
|
300
|
+
in a scratch directory, and headless Claude cannot ask for permission — so tools
|
|
301
|
+
that need it are denied and counted, and a task that needs your repo will look
|
|
302
|
+
smaller than it is. The verdict says which way it went: *confirmed*, *smaller
|
|
303
|
+
than advertised*, or *history overstated it*.
|
|
304
|
+
|
|
305
|
+
---
|
|
306
|
+
|
|
290
307
|
## Live model advice
|
|
291
308
|
|
|
292
309
|
The back-test tells you what to use next time. These tell you mid-session, while
|
package/finops/api.py
CHANGED
|
@@ -113,6 +113,15 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
113
113
|
if path.startswith("/api/live/"):
|
|
114
114
|
self._payload = payload
|
|
115
115
|
return self.session_action(path)
|
|
116
|
+
if path == "/api/trial/run":
|
|
117
|
+
# Spends real money and drives a real agent, so it is guarded like
|
|
118
|
+
# the other side-effecting routes and only ever reached by a click.
|
|
119
|
+
if not self._same_origin():
|
|
120
|
+
return self.send_json({"ok": False, "error": "forbidden"}, 403)
|
|
121
|
+
from .trial import run
|
|
122
|
+
return self.send_json(run(payload.get("category"), payload.get("model"),
|
|
123
|
+
payload.get("prompts") or [],
|
|
124
|
+
cwd=payload.get("cwd") or None))
|
|
116
125
|
if path.startswith("/api/do/"):
|
|
117
126
|
if not self._same_origin():
|
|
118
127
|
return self.send_json({"ok": False, "error": "forbidden"}, 403)
|
|
@@ -292,6 +301,11 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
292
301
|
return self.send_json(a.context_analysis(f))
|
|
293
302
|
if route == "waste":
|
|
294
303
|
return self.send_json(a.waste(f))
|
|
304
|
+
if route == "trial":
|
|
305
|
+
from .trial import samples, available
|
|
306
|
+
return self.send_json({"available": available(),
|
|
307
|
+
"samples": samples(qs.get("category", [""])[0],
|
|
308
|
+
int(qs.get("limit", ["3"])[0]))})
|
|
295
309
|
if route == "model_evidence":
|
|
296
310
|
return self.send_json(a.model_evidence(f))
|
|
297
311
|
if route == "model_switch":
|
package/finops/trial.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Run the trial the evidence asks for, instead of only recommending one.
|
|
2
|
+
|
|
3
|
+
model_evidence() ends with an honest caveat: your prompts were never randomly
|
|
4
|
+
assigned to models, so a category can simply have been easier on one of them.
|
|
5
|
+
That caveat can only be closed by actually running the same prompt on both
|
|
6
|
+
models and looking at what came back.
|
|
7
|
+
|
|
8
|
+
So: take real prompts out of your own history, re-run them headlessly on the
|
|
9
|
+
candidate model, and compare what they cost against what they cost the first
|
|
10
|
+
time. `claude -p ... --output-format json` reports cost, turns and errors for
|
|
11
|
+
exactly this purpose.
|
|
12
|
+
|
|
13
|
+
Two things this deliberately does not do:
|
|
14
|
+
|
|
15
|
+
* It never runs anything on its own. A trial spends real money and drives a
|
|
16
|
+
real agent, so it happens on an explicit click, with the bill shown first.
|
|
17
|
+
* It runs in a scratch directory by default, not your repo. Headless Claude
|
|
18
|
+
cannot ask for permission, so tools that need it are denied and recorded —
|
|
19
|
+
but a trial that edits your working tree to prove a point is not a trial
|
|
20
|
+
anyone wants.
|
|
21
|
+
"""
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import shutil
|
|
25
|
+
import subprocess
|
|
26
|
+
import tempfile
|
|
27
|
+
import time
|
|
28
|
+
|
|
29
|
+
from .advisor import model_alias
|
|
30
|
+
from .paths import DB_PATH
|
|
31
|
+
|
|
32
|
+
TIMEOUT_S = 600
|
|
33
|
+
MAX_PROMPTS = 5
|
|
34
|
+
MAX_CHARS = 6000 # a prompt longer than this is usually a paste, not a task
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def samples(category, limit=3):
|
|
38
|
+
"""Real prompts from this category, with what they actually cost first time.
|
|
39
|
+
|
|
40
|
+
Picks around the middle of the cost distribution: the cheapest prompts are
|
|
41
|
+
usually "yes"/"continue" and the most expensive are outliers, and neither
|
|
42
|
+
tells you much about a model.
|
|
43
|
+
"""
|
|
44
|
+
from .analytics import Analytics
|
|
45
|
+
a = Analytics(DB_PATH)
|
|
46
|
+
rows = a.q("""SELECT p.id, p.text, p.models model, p.est_cost_usd cost,
|
|
47
|
+
p.request_count turns, p.tool_calls tools, p.session_id, p.day
|
|
48
|
+
FROM prompts p
|
|
49
|
+
WHERE COALESCE(p.category,'other') = ? AND p.agent = 'claude'
|
|
50
|
+
AND p.models IS NOT NULL AND p.models NOT LIKE '%,%'
|
|
51
|
+
AND p.models != '<synthetic>' AND p.est_cost_usd > 0
|
|
52
|
+
AND LENGTH(p.text) BETWEEN 40 AND ?
|
|
53
|
+
-- Things you actually typed. System turns, tool results, queued
|
|
54
|
+
-- follow-ups and SDK traffic are not tasks anyone would re-run, and
|
|
55
|
+
-- a trial built from task-notification XML measures nothing.
|
|
56
|
+
AND p.source = 'typed'
|
|
57
|
+
AND p.text NOT LIKE '<%' AND p.text NOT LIKE '[Request interrupted%'
|
|
58
|
+
AND p.text NOT LIKE 'Caveat:%' AND p.text NOT LIKE '%<local-command%'
|
|
59
|
+
ORDER BY p.est_cost_usd""", (category, MAX_CHARS))
|
|
60
|
+
if not rows:
|
|
61
|
+
return []
|
|
62
|
+
lo = int(len(rows) * 0.35)
|
|
63
|
+
hi = max(lo + 1, int(len(rows) * 0.85))
|
|
64
|
+
mid = rows[lo:hi] or rows
|
|
65
|
+
step = max(1, len(mid) // limit)
|
|
66
|
+
picked = mid[::step][:limit]
|
|
67
|
+
return [{"prompt_id": r["id"], "text": r["text"], "baseline_model": r["model"],
|
|
68
|
+
"baseline_name": a.pricing.display_name(r["model"]),
|
|
69
|
+
"baseline_cost_usd": round(r["cost"], 4),
|
|
70
|
+
"baseline_turns": r["turns"], "baseline_tools": r["tools"],
|
|
71
|
+
"day": r["day"]} for r in picked]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def available():
|
|
75
|
+
return bool(shutil.which("claude"))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def run_one(text, model, cwd=None, timeout=TIMEOUT_S):
|
|
79
|
+
"""One headless run. Returns what it cost and whether it got there."""
|
|
80
|
+
exe = shutil.which("claude")
|
|
81
|
+
if not exe:
|
|
82
|
+
return {"ok": False, "error": "The `claude` CLI is not on PATH."}
|
|
83
|
+
sandbox = cwd or tempfile.mkdtemp(prefix="finops-trial-")
|
|
84
|
+
started = time.time()
|
|
85
|
+
try:
|
|
86
|
+
p = subprocess.run([exe, "-p", text, "--model", model_alias(model),
|
|
87
|
+
"--output-format", "json"],
|
|
88
|
+
capture_output=True, text=True, timeout=timeout, cwd=sandbox)
|
|
89
|
+
except subprocess.TimeoutExpired:
|
|
90
|
+
return {"ok": False, "error": f"Gave up after {timeout}s.",
|
|
91
|
+
"elapsed_s": round(time.time() - started, 1)}
|
|
92
|
+
except OSError as e:
|
|
93
|
+
return {"ok": False, "error": str(e)}
|
|
94
|
+
try:
|
|
95
|
+
d = json.loads(p.stdout)
|
|
96
|
+
except ValueError:
|
|
97
|
+
return {"ok": False, "error": (p.stderr or p.stdout or "no output")[:400],
|
|
98
|
+
"elapsed_s": round(time.time() - started, 1)}
|
|
99
|
+
usage = d.get("usage") or {}
|
|
100
|
+
return {
|
|
101
|
+
"ok": not d.get("is_error"),
|
|
102
|
+
"cost_usd": d.get("total_cost_usd"),
|
|
103
|
+
"turns": d.get("num_turns"),
|
|
104
|
+
"elapsed_s": round(time.time() - started, 1),
|
|
105
|
+
"duration_api_ms": d.get("duration_api_ms"),
|
|
106
|
+
"output_tokens": usage.get("output_tokens"),
|
|
107
|
+
"input_tokens": usage.get("input_tokens"),
|
|
108
|
+
"denials": len(d.get("permission_denials") or []),
|
|
109
|
+
"stop_reason": d.get("stop_reason") or d.get("terminal_reason"),
|
|
110
|
+
"result": (d.get("result") or "")[:4000],
|
|
111
|
+
"session_id": d.get("session_id"),
|
|
112
|
+
"sandbox": sandbox,
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _verdict(runs, baseline_cost):
|
|
117
|
+
"""What the runs actually showed, in the same language as the back-test."""
|
|
118
|
+
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
119
|
+
if not done:
|
|
120
|
+
return "failed", ("Every run errored or produced nothing, so this says nothing about "
|
|
121
|
+
"cost. Check the errors below before reading anything into it.")
|
|
122
|
+
got = sum(r["cost_usd"] for r in done)
|
|
123
|
+
base = sum(baseline_cost)
|
|
124
|
+
denials = sum(r.get("denials") or 0 for r in done)
|
|
125
|
+
if base <= 0:
|
|
126
|
+
return "unclear", "No usable baseline cost for these prompts."
|
|
127
|
+
pct = 100.0 * (1 - got / base)
|
|
128
|
+
tail = (f" {denials} tool call(s) were denied because a headless run cannot ask for "
|
|
129
|
+
f"permission, so the real task may be larger than this." if denials else "")
|
|
130
|
+
if len(done) < len(runs):
|
|
131
|
+
tail += f" {len(runs) - len(done)} of {len(runs)} runs failed and are excluded."
|
|
132
|
+
if pct >= 25:
|
|
133
|
+
return "confirmed", (f"Re-running your own prompts cost {pct:.0f}% less on the cheaper "
|
|
134
|
+
f"model (${got:.2f} against ${base:.2f} the first time).{tail}")
|
|
135
|
+
if pct >= 0:
|
|
136
|
+
return "marginal", (f"Only {pct:.0f}% cheaper on a re-run (${got:.2f} against "
|
|
137
|
+
f"${base:.2f}). Not the saving the history suggested.{tail}")
|
|
138
|
+
return "contradicted", (f"The re-run cost {abs(pct):.0f}% *more* (${got:.2f} against "
|
|
139
|
+
f"${base:.2f}). The history overstated this switch.{tail}")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def run(category, model, prompts, cwd=None):
|
|
143
|
+
"""Run a set of sample prompts on a candidate model and judge the result."""
|
|
144
|
+
if not prompts:
|
|
145
|
+
return {"ok": False, "error": "Nothing to run."}
|
|
146
|
+
prompts = prompts[:MAX_PROMPTS]
|
|
147
|
+
runs, baseline = [], []
|
|
148
|
+
for item in prompts:
|
|
149
|
+
text = (item.get("text") or "").strip()
|
|
150
|
+
if not text:
|
|
151
|
+
continue
|
|
152
|
+
r = run_one(text, model, cwd=cwd)
|
|
153
|
+
r["prompt"] = text[:300]
|
|
154
|
+
r["baseline_cost_usd"] = item.get("baseline_cost_usd") or 0
|
|
155
|
+
r["baseline_turns"] = item.get("baseline_turns")
|
|
156
|
+
runs.append(r)
|
|
157
|
+
baseline.append(r["baseline_cost_usd"])
|
|
158
|
+
verdict, why = _verdict(runs, baseline)
|
|
159
|
+
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
160
|
+
return {
|
|
161
|
+
"ok": True, "category": category, "model": model, "alias": model_alias(model),
|
|
162
|
+
"runs": runs, "verdict": verdict, "why": why,
|
|
163
|
+
"measured_cost_usd": round(sum(r["cost_usd"] for r in done), 4),
|
|
164
|
+
"baseline_cost_usd": round(sum(baseline), 4),
|
|
165
|
+
"ran_at": int(time.time()),
|
|
166
|
+
"note": ("Measured by re-running these exact prompts headlessly. The agent had no "
|
|
167
|
+
"permission to use tools that ask, and worked in a scratch directory, so a "
|
|
168
|
+
"task that needs your repo will look smaller here than it is."),
|
|
169
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-finops",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.7.0",
|
|
4
4
|
"description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"claude-finops": "bin/claude-finops.js"
|
package/web/app.js
CHANGED
|
@@ -1354,6 +1354,8 @@ function evidenceBody(ev) {
|
|
|
1354
1354
|
statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
|
|
1355
1355
|
{h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
|
|
1356
1356
|
? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
|
|
1357
|
+
{h: '', f: r => `<button class="act ghost trial-btn" data-cat="${esc(r.c.category)}"
|
|
1358
|
+
data-model="${esc(r.x.model)}" data-name="${esc(r.x.name)}">Try it →</button>`},
|
|
1357
1359
|
], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
|
|
1358
1360
|
}
|
|
1359
1361
|
|
|
@@ -1416,6 +1418,7 @@ VIEWS.modelswitch = async (page) => {
|
|
|
1416
1418
|
{h: 'Could save', num: 1, f: r => fmtUSD(r.estimated_savings_usd)},
|
|
1417
1419
|
{h: 'Why', f: r => esc(r.why)},
|
|
1418
1420
|
], m.projects);
|
|
1421
|
+
wireTrials(page);
|
|
1419
1422
|
addChart(page, 'Savings by switch', el => C.barsH(el, {
|
|
1420
1423
|
rows: m.switches.slice(0, 10), label: r => clip(`${r.scope} → ${r.recommended_name}`, 48),
|
|
1421
1424
|
value: r => r.estimated_savings_usd,
|
|
@@ -1426,6 +1429,116 @@ VIEWS.modelswitch = async (page) => {
|
|
|
1426
1429
|
{badge: BADGE.recommendation, hint: 'Colour = confidence (green high, blue medium, amber low)'});
|
|
1427
1430
|
};
|
|
1428
1431
|
|
|
1432
|
+
/* ---------- trial: stop recommending, start measuring ----------
|
|
1433
|
+
The evidence ends at "strong evidence for a trial, not proof". This runs the
|
|
1434
|
+
trial: real prompts out of your own history, re-run headlessly on the
|
|
1435
|
+
candidate model, priced against what they cost the first time. It spends real
|
|
1436
|
+
money, so nothing happens without two clicks. */
|
|
1437
|
+
function trialPanelHTML(cat, model, name, s) {
|
|
1438
|
+
if (!s.available) {
|
|
1439
|
+
return `<div class="empty">The <code>claude</code> CLI is not on PATH, so a trial cannot be
|
|
1440
|
+
run from here.</div>`;
|
|
1441
|
+
}
|
|
1442
|
+
if (!s.samples.length) {
|
|
1443
|
+
return `<div class="empty">No prompt you actually typed in this category is short enough to
|
|
1444
|
+
re-run safely.</div>`;
|
|
1445
|
+
}
|
|
1446
|
+
const base = s.samples.reduce((a, x) => a + (x.baseline_cost_usd || 0), 0);
|
|
1447
|
+
return `
|
|
1448
|
+
<div class="dt">These are ${s.samples.length} prompts you really sent in
|
|
1449
|
+
<b>${esc(cat.replace('_', ' '))}</b>. Running them again on <b>${esc(name)}</b> costs money —
|
|
1450
|
+
they cost ${fmtUSD(base)} the first time, and the cheaper model should come in under that.</div>
|
|
1451
|
+
<div class="stack trial-samples">${s.samples.map((x, i) => `
|
|
1452
|
+
<div class="dt trial-s" data-i="${i}">
|
|
1453
|
+
<span class="note">${esc(x.day)} · ${esc(x.baseline_name)} · ${fmtUSD(x.baseline_cost_usd)} ·
|
|
1454
|
+
${fmtInt(x.baseline_turns)} turns</span>
|
|
1455
|
+
<div class="trial-text">${esc(x.text.slice(0, 400))}${x.text.length > 400 ? '…' : ''}</div>
|
|
1456
|
+
</div>`).join('')}</div>
|
|
1457
|
+
<div class="dt note">Runs headlessly in a scratch directory. Tools that need permission are
|
|
1458
|
+
denied, because a headless agent cannot ask — so a task that needs your repo will look
|
|
1459
|
+
smaller here than it really is.</div>
|
|
1460
|
+
<div class="live-actions">
|
|
1461
|
+
<button class="act trial-run">▶ Run ${s.samples.length} prompts on ${esc(name)}</button>
|
|
1462
|
+
<button class="act ghost trial-copy">Copy the first prompt instead</button>
|
|
1463
|
+
<span class="trial-msg note"></span>
|
|
1464
|
+
</div>
|
|
1465
|
+
<div class="trial-out"></div>`;
|
|
1466
|
+
}
|
|
1467
|
+
|
|
1468
|
+
const TRIAL_VERDICT = {
|
|
1469
|
+
confirmed: ['healthy', 'Confirmed by running it'],
|
|
1470
|
+
marginal: ['high', 'Smaller than advertised'],
|
|
1471
|
+
contradicted: ['critical', 'History overstated it'],
|
|
1472
|
+
failed: ['critical', 'Runs failed'],
|
|
1473
|
+
unclear: ['high', 'Inconclusive'],
|
|
1474
|
+
};
|
|
1475
|
+
|
|
1476
|
+
function trialResultHTML(r) {
|
|
1477
|
+
const v = TRIAL_VERDICT[r.verdict] || TRIAL_VERDICT.unclear;
|
|
1478
|
+
return `<div class="dt"><b>${statusChip(v[0], v[1])}</b> ${esc(r.why)}</div>` + table([
|
|
1479
|
+
{h: 'Prompt', trunc: 1, f: x => esc(x.prompt)},
|
|
1480
|
+
{h: 'First time', num: 1, f: x => fmtUSD(x.baseline_cost_usd)},
|
|
1481
|
+
{h: 'On ' + esc(r.alias), num: 1, f: x => x.ok ? fmtUSD(x.cost_usd) : '—'},
|
|
1482
|
+
{h: 'Turns', num: 1, f: x => x.ok ? fmtInt(x.turns) : '—'},
|
|
1483
|
+
{h: 'Took', num: 1, f: x => x.ok ? x.elapsed_s + 's' : '—'},
|
|
1484
|
+
{h: 'Denied', num: 1, f: x => x.denials ? fmtInt(x.denials) : ''},
|
|
1485
|
+
{h: 'Result', f: x => x.ok ? `<span class="note">${esc((x.result || '').slice(0, 120))}</span>`
|
|
1486
|
+
: `<span class="status critical">${esc((x.error || 'failed').slice(0, 120))}</span>`},
|
|
1487
|
+
], r.runs) + `<div class="note" style="padding:8px 14px">${esc(r.note)}</div>`;
|
|
1488
|
+
}
|
|
1489
|
+
|
|
1490
|
+
function wireTrials(page) {
|
|
1491
|
+
page.querySelectorAll('.trial-btn').forEach(b => b.onclick = async () => {
|
|
1492
|
+
const row = b.closest('tr');
|
|
1493
|
+
if (row.nextElementSibling?.classList.contains('trial-row')) {
|
|
1494
|
+
row.nextElementSibling.remove(); return;
|
|
1495
|
+
}
|
|
1496
|
+
const {cat, model, name} = b.dataset;
|
|
1497
|
+
const tr = h(`<tr class="trial-row"><td colspan="10"><div class="trial-panel">
|
|
1498
|
+
<div class="empty">Finding prompts you sent…</div></div></td></tr>`);
|
|
1499
|
+
row.after(tr);
|
|
1500
|
+
const host = tr.querySelector('.trial-panel');
|
|
1501
|
+
let s;
|
|
1502
|
+
try {
|
|
1503
|
+
s = await fetch(`/api/trial?category=${encodeURIComponent(cat)}&limit=3`).then(r => r.json());
|
|
1504
|
+
} catch (e) { host.innerHTML = `<div class="empty">${esc(e.message)}</div>`; return; }
|
|
1505
|
+
host.innerHTML = trialPanelHTML(cat, model, name, s);
|
|
1506
|
+
const msg = host.querySelector('.trial-msg');
|
|
1507
|
+
host.querySelector('.trial-copy')?.addEventListener('click', async () => {
|
|
1508
|
+
try { await navigator.clipboard.writeText(s.samples[0].text); msg.textContent = 'Copied.'; }
|
|
1509
|
+
catch { msg.textContent = 'Could not copy.'; }
|
|
1510
|
+
});
|
|
1511
|
+
host.querySelector('.trial-run')?.addEventListener('click', async ev => {
|
|
1512
|
+
const btn = ev.currentTarget;
|
|
1513
|
+
if (btn.dataset.armed !== '1') {
|
|
1514
|
+
btn.dataset.armed = '1';
|
|
1515
|
+
btn.textContent = 'Click again to spend real money';
|
|
1516
|
+
btn.classList.add('warn');
|
|
1517
|
+
return;
|
|
1518
|
+
}
|
|
1519
|
+
btn.disabled = true;
|
|
1520
|
+
btn.textContent = 'Running…';
|
|
1521
|
+
msg.textContent = 'Each prompt runs to completion; this can take a few minutes.';
|
|
1522
|
+
try {
|
|
1523
|
+
const r = await fetch('/api/trial/run', {
|
|
1524
|
+
method: 'POST',
|
|
1525
|
+
headers: {'X-FinOps-Action': '1', 'Content-Type': 'application/json'},
|
|
1526
|
+
body: JSON.stringify({category: cat, model, prompts: s.samples}),
|
|
1527
|
+
}).then(x => x.json());
|
|
1528
|
+
host.querySelector('.trial-out').innerHTML = r.ok
|
|
1529
|
+
? trialResultHTML(r) : `<div class="empty">${esc(r.error || 'failed')}</div>`;
|
|
1530
|
+
msg.textContent = '';
|
|
1531
|
+
} catch (e) {
|
|
1532
|
+
msg.textContent = e.message;
|
|
1533
|
+
}
|
|
1534
|
+
btn.disabled = false;
|
|
1535
|
+
btn.classList.remove('warn');
|
|
1536
|
+
btn.textContent = '▶ Run again';
|
|
1537
|
+
btn.dataset.armed = '';
|
|
1538
|
+
});
|
|
1539
|
+
});
|
|
1540
|
+
}
|
|
1541
|
+
|
|
1429
1542
|
/* ---------- waste ---------- */
|
|
1430
1543
|
VIEWS.waste = async (page) => {
|
|
1431
1544
|
const w = await api('waste');
|
package/web/styles.css
CHANGED
|
@@ -425,3 +425,13 @@ table.tbl .sub { font-size:10.5px; color:var(--muted); }
|
|
|
425
425
|
so it reads as a note with an accent edge rather than a warning. */
|
|
426
426
|
.switch-tip { border-left: 2px solid var(--accent, #eb6834); padding-left: 9px; }
|
|
427
427
|
.switch-tip .act { margin-left: 8px; }
|
|
428
|
+
|
|
429
|
+
/* Trial panel: an experiment you are about to pay for, so it reads as a quiet
|
|
430
|
+
workbench rather than another recommendation. */
|
|
431
|
+
.trial-row > td { padding: 0 !important; background: var(--surface-2); }
|
|
432
|
+
.trial-panel { padding: 12px 14px; display: grid; gap: 8px;
|
|
433
|
+
border-left: 2px solid var(--accent, #eb6834); }
|
|
434
|
+
.trial-text { font-size: 12px; margin-top: 3px; white-space: pre-wrap;
|
|
435
|
+
max-height: 76px; overflow: auto; color: var(--text-2); }
|
|
436
|
+
.trial-s { border-top: 1px solid var(--border); padding-top: 7px; }
|
|
437
|
+
.trial-out:empty { display: none; }
|