claude-finops 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -0
- package/finops/advisor.py +188 -0
- package/finops/api.py +25 -0
- package/finops/integrate.py +168 -0
- package/finops/procs.py +3 -1
- package/finops/trial.py +169 -0
- package/package.json +1 -1
- package/run.py +42 -0
- package/web/app.js +117 -0
- package/web/styles.css +18 -0
package/README.md
CHANGED
|
@@ -287,6 +287,45 @@ place for you to delete once you are happy.
|
|
|
287
287
|
|
|
288
288
|
---
|
|
289
289
|
|
|
290
|
+
## Running the trial, not just recommending one
|
|
291
|
+
|
|
292
|
+
The evidence table ends by admitting its own limit: your prompts were never
|
|
293
|
+
randomly assigned to models, so a category can simply have been easier on one of
|
|
294
|
+
them. **Try it →** on any row closes that gap. It pulls prompts you actually
|
|
295
|
+
typed in that category, re-runs them headlessly on the candidate model, and
|
|
296
|
+
prices the result against what they cost the first time.
|
|
297
|
+
|
|
298
|
+
Nothing runs on its own: a trial spends real money and drives a real agent, so
|
|
299
|
+
it takes two clicks and shows the first-time bill before you commit. Runs happen
|
|
300
|
+
in a scratch directory, and headless Claude cannot ask for permission — so tools
|
|
301
|
+
that need it are denied and counted, and a task that needs your repo will look
|
|
302
|
+
smaller than it is. The verdict says which way it went: *confirmed*, *smaller
|
|
303
|
+
than advertised*, or *history overstated it*.
|
|
304
|
+
|
|
305
|
+
---
|
|
306
|
+
|
|
307
|
+
## Live model advice
|
|
308
|
+
|
|
309
|
+
The back-test tells you what to use next time. These tell you mid-session, while
|
|
310
|
+
you can still act on it:
|
|
311
|
+
|
|
312
|
+
```
|
|
313
|
+
claude-finops --advise what every running session should switch to
|
|
314
|
+
claude-finops --install-hook suggest a cheaper model as you send each prompt
|
|
315
|
+
claude-finops --install-statusline model, context pressure and advice in your statusline
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
The hook and statusline read the same evidence as the dashboard, cached for an
|
|
319
|
+
hour, and stay quiet unless your own history shows a cheaper model doing that
|
|
320
|
+
category of work without taking more turns. Neither can block or slow a prompt:
|
|
321
|
+
they fail silent and always exit 0. Undo with `--uninstall-hook` /
|
|
322
|
+
`--uninstall-statusline`; your `~/.claude/settings.json` is backed up first.
|
|
323
|
+
|
|
324
|
+
The Running sessions view shows the same line per live session, with the
|
|
325
|
+
`/model` command ready to copy.
|
|
326
|
+
|
|
327
|
+
---
|
|
328
|
+
|
|
290
329
|
## Privacy
|
|
291
330
|
|
|
292
331
|
`~/.claude-finops/data/finops.db` and the prompt/CSV exports contain **your full prompt text**. The
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""Live model advice: what to switch to, while the session is still running.
|
|
2
|
+
|
|
3
|
+
The back-test in analytics.model_evidence() answers "what should I have used?",
|
|
4
|
+
which is the wrong tense once a session is under way. This module answers it in
|
|
5
|
+
the present: given the prompts a session has actually sent and the model it is
|
|
6
|
+
on, say whether your own history already shows a cheaper model doing this kind
|
|
7
|
+
of work without taking more turns.
|
|
8
|
+
|
|
9
|
+
It is built to run three ways, so the advice is the same wherever you meet it:
|
|
10
|
+
|
|
11
|
+
* the dashboard's Running sessions view (per live session)
|
|
12
|
+
* a UserPromptSubmit hook (per prompt, before the turn runs)
|
|
13
|
+
* a statusline command (continuously, in a few characters)
|
|
14
|
+
|
|
15
|
+
The hook and statusline run on every prompt, so the evidence is computed once
|
|
16
|
+
and cached; a warehouse query per keystroke would be felt. Everything here fails
|
|
17
|
+
soft and silent — advice that breaks your terminal is worse than no advice.
|
|
18
|
+
"""
|
|
19
|
+
import json
|
|
20
|
+
import os
|
|
21
|
+
import time
|
|
22
|
+
|
|
23
|
+
from .classify import classify
|
|
24
|
+
from .paths import DATA_DIR, DB_PATH, ensure_dirs
|
|
25
|
+
|
|
26
|
+
CACHE = os.path.join(DATA_DIR, "advice_cache.json")
|
|
27
|
+
TTL_S = 3600
|
|
28
|
+
RECENT_PROMPTS = 12 # how much of the session counts as "what it is doing now"
|
|
29
|
+
MIN_SAVING_PCT = 25 # below this, interrupting someone is not worth it
|
|
30
|
+
|
|
31
|
+
# `/model <name>` takes a family name, not the pricing table's id.
|
|
32
|
+
ALIASES = ("opus", "sonnet", "haiku", "fable")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def model_alias(model_id):
|
|
36
|
+
"""'claude-fable-5-1' -> 'fable'. Falls back to the full id we were given."""
|
|
37
|
+
low = str(model_id or "").lower()
|
|
38
|
+
for a in ALIASES:
|
|
39
|
+
if a in low:
|
|
40
|
+
return a
|
|
41
|
+
return model_id
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# ---------------------------------------------------------------- evidence ----
|
|
45
|
+
|
|
46
|
+
def _build_evidence():
|
|
47
|
+
"""Pull the back-test into the small shape the live surfaces need."""
|
|
48
|
+
from .analytics import Analytics
|
|
49
|
+
a = Analytics(DB_PATH)
|
|
50
|
+
ev = a.model_evidence({})
|
|
51
|
+
out = {}
|
|
52
|
+
for c in ev["categories"]:
|
|
53
|
+
cands = [{"model": x["model"], "name": x["name"], "verdict": x["verdict"],
|
|
54
|
+
"cost_per_prompt": round(x["cost_per_prompt"], 3),
|
|
55
|
+
"savings_pct": x["savings_pct"], "turn_ratio": x["turn_ratio"],
|
|
56
|
+
"prompts": x["prompts"], "why": x["why"]}
|
|
57
|
+
for x in c["candidates"]]
|
|
58
|
+
out[c["category"]] = {"current": c["current"]["model"],
|
|
59
|
+
"current_name": c["current"]["name"],
|
|
60
|
+
"current_cost_per_prompt": round(c["current"]["cost_per_prompt"], 3),
|
|
61
|
+
"agent": c.get("agent", "claude"),
|
|
62
|
+
"candidates": cands}
|
|
63
|
+
return out
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def evidence(force=False):
|
|
67
|
+
"""Cached evidence table, keyed by category. Never raises."""
|
|
68
|
+
try:
|
|
69
|
+
with open(CACHE) as fh:
|
|
70
|
+
c = json.load(fh)
|
|
71
|
+
if not force and time.time() - c.get("built_at", 0) < TTL_S:
|
|
72
|
+
return c.get("categories") or {}
|
|
73
|
+
except (OSError, ValueError):
|
|
74
|
+
pass
|
|
75
|
+
try:
|
|
76
|
+
cats = _build_evidence()
|
|
77
|
+
except Exception:
|
|
78
|
+
return {}
|
|
79
|
+
try:
|
|
80
|
+
ensure_dirs()
|
|
81
|
+
tmp = CACHE + ".tmp"
|
|
82
|
+
with open(tmp, "w") as fh:
|
|
83
|
+
json.dump({"built_at": int(time.time()), "categories": cats}, fh)
|
|
84
|
+
os.replace(tmp, CACHE)
|
|
85
|
+
except OSError:
|
|
86
|
+
pass
|
|
87
|
+
return cats
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ---------------------------------------------------------------- session ----
|
|
91
|
+
|
|
92
|
+
def recent_prompts(transcript, n=RECENT_PROMPTS):
|
|
93
|
+
"""The last n human prompts in a live transcript, newest last.
|
|
94
|
+
|
|
95
|
+
Reads the tail only: an active session's JSONL runs to tens of megabytes and
|
|
96
|
+
this is on the path of every prompt you type.
|
|
97
|
+
"""
|
|
98
|
+
try:
|
|
99
|
+
size = os.path.getsize(transcript)
|
|
100
|
+
with open(transcript, "rb") as fh:
|
|
101
|
+
fh.seek(max(0, size - 400_000))
|
|
102
|
+
lines = fh.read().decode("utf-8", "replace").splitlines()[1:]
|
|
103
|
+
except OSError:
|
|
104
|
+
return []
|
|
105
|
+
out = []
|
|
106
|
+
for line in reversed(lines):
|
|
107
|
+
if '"type":"user"' not in line and '"type": "user"' not in line:
|
|
108
|
+
continue
|
|
109
|
+
try:
|
|
110
|
+
d = json.loads(line)
|
|
111
|
+
except ValueError:
|
|
112
|
+
continue
|
|
113
|
+
if d.get("isMeta") or d.get("isSidechain"):
|
|
114
|
+
continue
|
|
115
|
+
msg = d.get("message") or {}
|
|
116
|
+
content = msg.get("content")
|
|
117
|
+
if isinstance(content, list):
|
|
118
|
+
content = " ".join(b.get("text", "") for b in content
|
|
119
|
+
if isinstance(b, dict) and b.get("type") == "text")
|
|
120
|
+
if not isinstance(content, str) or not content.strip():
|
|
121
|
+
continue
|
|
122
|
+
if content.lstrip().startswith(("<", "[Request interrupted")):
|
|
123
|
+
continue # tool results and interrupt markers are not prompts
|
|
124
|
+
out.append(content)
|
|
125
|
+
if len(out) >= n:
|
|
126
|
+
break
|
|
127
|
+
return list(reversed(out))
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def category_of(texts):
|
|
131
|
+
"""The category this stretch of work is in, by weight of confidence."""
|
|
132
|
+
scores = {}
|
|
133
|
+
for t in texts:
|
|
134
|
+
cat, conf, _ = classify(t)
|
|
135
|
+
scores[cat] = scores.get(cat, 0) + max(conf, 0.1)
|
|
136
|
+
if not scores:
|
|
137
|
+
return "other", 0.0
|
|
138
|
+
cat = max(scores, key=scores.get)
|
|
139
|
+
return cat, round(scores[cat] / sum(scores.values()), 2)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def advise(model=None, transcript=None, prompt=None, ev=None):
|
|
143
|
+
"""What to say about a session that is running right now.
|
|
144
|
+
|
|
145
|
+
model the model the session is on (pricing-table id)
|
|
146
|
+
transcript path to the live JSONL, for reading what it has been doing
|
|
147
|
+
prompt the prompt about to be sent, when we are called from a hook
|
|
148
|
+
Returns None when there is nothing worth saying.
|
|
149
|
+
"""
|
|
150
|
+
ev = evidence() if ev is None else ev
|
|
151
|
+
if not ev:
|
|
152
|
+
return None
|
|
153
|
+
texts = list(recent_prompts(transcript)) if transcript else []
|
|
154
|
+
if prompt:
|
|
155
|
+
# The prompt in hand is what the next turn will cost, so it leads.
|
|
156
|
+
texts = texts[-4:] + [prompt] * 3
|
|
157
|
+
if not texts:
|
|
158
|
+
return None
|
|
159
|
+
cat, conf = category_of(texts)
|
|
160
|
+
row = ev.get(cat)
|
|
161
|
+
if not row:
|
|
162
|
+
return None
|
|
163
|
+
# Only speak when they are on the model the evidence is about. Advising a
|
|
164
|
+
# switch away from a model you already left would be noise.
|
|
165
|
+
if model and row["current"] and model_alias(model) != model_alias(row["current"]):
|
|
166
|
+
return None
|
|
167
|
+
usable = [c for c in row["candidates"]
|
|
168
|
+
if c["verdict"] in ("supported", "caution") and c["savings_pct"] >= MIN_SAVING_PCT]
|
|
169
|
+
if not usable:
|
|
170
|
+
return None
|
|
171
|
+
best = usable[0]
|
|
172
|
+
trial = best["verdict"] == "caution"
|
|
173
|
+
return {
|
|
174
|
+
"category": cat, "confidence": conf,
|
|
175
|
+
"current_model": row["current"], "current_name": row["current_name"],
|
|
176
|
+
"current_cost_per_prompt": row["current_cost_per_prompt"],
|
|
177
|
+
"model": best["model"], "name": best["name"], "alias": model_alias(best["model"]),
|
|
178
|
+
"cost_per_prompt": best["cost_per_prompt"], "savings_pct": best["savings_pct"],
|
|
179
|
+
"turn_ratio": best["turn_ratio"], "prompts": best["prompts"],
|
|
180
|
+
"verdict": best["verdict"], "trial": trial, "why": best["why"],
|
|
181
|
+
"command": f"/model {model_alias(best['model'])}",
|
|
182
|
+
"line": (f"{cat.replace('_', ' ')} on {row['current_name']} "
|
|
183
|
+
f"(${row['current_cost_per_prompt']:.2f}/prompt). Your last {best['prompts']} "
|
|
184
|
+
f"ran on {best['name']} at ${best['cost_per_prompt']:.2f} in "
|
|
185
|
+
f"{best['turn_ratio']}x the turns"
|
|
186
|
+
+ (" — worth a trial here." if trial else ".")),
|
|
187
|
+
"short": f"try {model_alias(best['model'])} (-{best['savings_pct']:.0f}%)",
|
|
188
|
+
}
|
package/finops/api.py
CHANGED
|
@@ -113,6 +113,15 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
113
113
|
if path.startswith("/api/live/"):
|
|
114
114
|
self._payload = payload
|
|
115
115
|
return self.session_action(path)
|
|
116
|
+
if path == "/api/trial/run":
|
|
117
|
+
# Spends real money and drives a real agent, so it is guarded like
|
|
118
|
+
# the other side-effecting routes and only ever reached by a click.
|
|
119
|
+
if not self._same_origin():
|
|
120
|
+
return self.send_json({"ok": False, "error": "forbidden"}, 403)
|
|
121
|
+
from .trial import run
|
|
122
|
+
return self.send_json(run(payload.get("category"), payload.get("model"),
|
|
123
|
+
payload.get("prompts") or [],
|
|
124
|
+
cwd=payload.get("cwd") or None))
|
|
116
125
|
if path.startswith("/api/do/"):
|
|
117
126
|
if not self._same_origin():
|
|
118
127
|
return self.send_json({"ok": False, "error": "forbidden"}, 403)
|
|
@@ -292,6 +301,11 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
292
301
|
return self.send_json(a.context_analysis(f))
|
|
293
302
|
if route == "waste":
|
|
294
303
|
return self.send_json(a.waste(f))
|
|
304
|
+
if route == "trial":
|
|
305
|
+
from .trial import samples, available
|
|
306
|
+
return self.send_json({"available": available(),
|
|
307
|
+
"samples": samples(qs.get("category", [""])[0],
|
|
308
|
+
int(qs.get("limit", ["3"])[0]))})
|
|
295
309
|
if route == "model_evidence":
|
|
296
310
|
return self.send_json(a.model_evidence(f))
|
|
297
311
|
if route == "model_switch":
|
|
@@ -323,6 +337,17 @@ class Handler(BaseHTTPRequestHandler):
|
|
|
323
337
|
others = list_agent_sessions(a.pricing, [x for x in want if x != "claude"] or None) \
|
|
324
338
|
if not want or any(x != "claude" for x in want) else []
|
|
325
339
|
rows = sorted(claude + others, key=lambda x: -(x.get("context") or 0))
|
|
340
|
+
# Live model advice, while the session can still act on it. One cached
|
|
341
|
+
# evidence read for the whole list, and never fatal to the view.
|
|
342
|
+
try:
|
|
343
|
+
from .advisor import advise, evidence
|
|
344
|
+
ev = evidence()
|
|
345
|
+
for r in rows:
|
|
346
|
+
r["advice"] = advise(model=r.get("model"), transcript=r.get("transcript"),
|
|
347
|
+
ev=ev) if ev else None
|
|
348
|
+
except Exception:
|
|
349
|
+
for r in rows:
|
|
350
|
+
r.setdefault("advice", None)
|
|
326
351
|
return self.send_json({"sessions": rows})
|
|
327
352
|
if route == "cloud":
|
|
328
353
|
from .cloud import report
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Live advice inside Claude Code itself: a prompt hook and a statusline.
|
|
2
|
+
|
|
3
|
+
The dashboard can only advise you if you are looking at it. These two run where
|
|
4
|
+
the decision is actually made — the terminal you are typing in.
|
|
5
|
+
|
|
6
|
+
claude-finops --hook reads a UserPromptSubmit payload on stdin
|
|
7
|
+
claude-finops --statusline reads a statusline payload on stdin
|
|
8
|
+
claude-finops --install-hook / --install-statusline wire them into settings
|
|
9
|
+
|
|
10
|
+
Both are on the path of every prompt, so both are built to be boring: fail
|
|
11
|
+
silent, never block, never take long. A hook that erases someone's prompt
|
|
12
|
+
because a database was locked is a far worse bug than missing advice, so the
|
|
13
|
+
hook never returns a blocking exit code — it exits 0 whatever happens.
|
|
14
|
+
"""
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import shutil
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
SETTINGS = os.path.join(os.path.expanduser("~"), ".claude", "settings.json")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _stdin_json():
|
|
24
|
+
try:
|
|
25
|
+
raw = sys.stdin.read()
|
|
26
|
+
return json.loads(raw) if raw.strip() else {}
|
|
27
|
+
except (ValueError, OSError):
|
|
28
|
+
return {}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _dig(d, *paths, default=None):
|
|
32
|
+
"""First present value among dotted paths.
|
|
33
|
+
|
|
34
|
+
The payload shapes are not identical across Claude Code versions, and a
|
|
35
|
+
statusline that breaks on upgrade is worse than one that misses a field, so
|
|
36
|
+
every read tries the spellings we know of.
|
|
37
|
+
"""
|
|
38
|
+
for path in paths:
|
|
39
|
+
cur = d
|
|
40
|
+
for part in path.split("."):
|
|
41
|
+
if not isinstance(cur, dict) or part not in cur:
|
|
42
|
+
cur = None
|
|
43
|
+
break
|
|
44
|
+
cur = cur[part]
|
|
45
|
+
if cur not in (None, ""):
|
|
46
|
+
return cur
|
|
47
|
+
return default
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _payload_common(d):
|
|
51
|
+
return (_dig(d, "model", "session.model", "model.id", "model.display_name"),
|
|
52
|
+
_dig(d, "transcript_path", "session.transcript_path", "transcriptPath"))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# -------------------------------------------------------------------- hook ----
|
|
56
|
+
|
|
57
|
+
def hook():
|
|
58
|
+
"""UserPromptSubmit: judge the prompt in hand, before it is paid for.
|
|
59
|
+
|
|
60
|
+
Advice goes to the user as `systemMessage`, not into Claude's context: it is
|
|
61
|
+
for the person deciding which model to use, and feeding it to the model
|
|
62
|
+
would just spend tokens telling it about its own price.
|
|
63
|
+
"""
|
|
64
|
+
try:
|
|
65
|
+
d = _stdin_json()
|
|
66
|
+
prompt = _dig(d, "user_prompt", "prompt", default="")
|
|
67
|
+
model, transcript = _payload_common(d)
|
|
68
|
+
from .advisor import advise
|
|
69
|
+
a = advise(model=model, transcript=transcript, prompt=prompt)
|
|
70
|
+
if a:
|
|
71
|
+
print(json.dumps({"systemMessage": f"finops: {a['line']} → {a['command']}"}))
|
|
72
|
+
except Exception:
|
|
73
|
+
pass # never let advice interfere with the prompt
|
|
74
|
+
return 0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# -------------------------------------------------------------- statusline ----
|
|
78
|
+
|
|
79
|
+
def statusline():
|
|
80
|
+
"""One line, refreshed constantly, so: what you are on, and what to try."""
|
|
81
|
+
try:
|
|
82
|
+
d = _stdin_json()
|
|
83
|
+
model, transcript = _payload_common(d)
|
|
84
|
+
pct = _dig(d, "context.percentUsed", "context.percent_used")
|
|
85
|
+
name = _dig(d, "model.display_name", "session.model", "model") or "claude"
|
|
86
|
+
bits = [str(name)]
|
|
87
|
+
if isinstance(pct, (int, float)):
|
|
88
|
+
bits.append(f"{pct:.0f}% ctx")
|
|
89
|
+
from .advisor import advise
|
|
90
|
+
a = advise(model=model, transcript=transcript)
|
|
91
|
+
if a:
|
|
92
|
+
bits.append(a["short"])
|
|
93
|
+
print(" · ".join(bits))
|
|
94
|
+
except Exception:
|
|
95
|
+
print("") # an empty statusline beats a stack trace under the prompt
|
|
96
|
+
return 0
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# --------------------------------------------------------------- installing ----
|
|
100
|
+
|
|
101
|
+
def _command():
|
|
102
|
+
"""How to invoke us from settings: the installed CLI if it is on PATH."""
|
|
103
|
+
exe = shutil.which("claude-finops")
|
|
104
|
+
if exe:
|
|
105
|
+
return exe
|
|
106
|
+
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
107
|
+
return f"{sys.executable} {os.path.join(root, 'run.py')}"
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _load_settings():
|
|
111
|
+
try:
|
|
112
|
+
with open(SETTINGS) as fh:
|
|
113
|
+
return json.load(fh)
|
|
114
|
+
except (OSError, ValueError):
|
|
115
|
+
return {}
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _save_settings(data):
|
|
119
|
+
os.makedirs(os.path.dirname(SETTINGS), exist_ok=True)
|
|
120
|
+
if os.path.exists(SETTINGS):
|
|
121
|
+
# Their settings file is not ours to lose.
|
|
122
|
+
shutil.copy2(SETTINGS, SETTINGS + ".finops-backup")
|
|
123
|
+
with open(SETTINGS, "w") as fh:
|
|
124
|
+
json.dump(data, fh, indent=2)
|
|
125
|
+
fh.write("\n")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def install_hook(remove=False):
|
|
129
|
+
cmd = f"{_command()} --hook"
|
|
130
|
+
s = _load_settings()
|
|
131
|
+
hooks = s.setdefault("hooks", {}).setdefault("UserPromptSubmit", [])
|
|
132
|
+
for group in hooks: # drop any earlier copy of ours
|
|
133
|
+
group["hooks"] = [h for h in group.get("hooks", [])
|
|
134
|
+
if "--hook" not in str(h.get("command", ""))
|
|
135
|
+
or "finops" not in str(h.get("command", ""))]
|
|
136
|
+
hooks[:] = [g for g in hooks if g.get("hooks")]
|
|
137
|
+
if not remove:
|
|
138
|
+
hooks.append({"matcher": "", "hooks": [{"type": "command", "command": cmd,
|
|
139
|
+
"timeout": 10}]})
|
|
140
|
+
if not hooks:
|
|
141
|
+
s["hooks"].pop("UserPromptSubmit", None)
|
|
142
|
+
if not s["hooks"]:
|
|
143
|
+
s.pop("hooks")
|
|
144
|
+
_save_settings(s)
|
|
145
|
+
print(("Removed" if remove else "Installed") + f" the prompt hook in {SETTINGS}")
|
|
146
|
+
if not remove:
|
|
147
|
+
print(" It suggests a cheaper model when your own history backs one, before the turn runs.")
|
|
148
|
+
print(" Start a new Claude Code session to pick it up. Undo: claude-finops --uninstall-hook")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def install_statusline(remove=False):
|
|
152
|
+
cmd = f"{_command()} --statusline"
|
|
153
|
+
s = _load_settings()
|
|
154
|
+
if remove:
|
|
155
|
+
if "finops" in str(s.get("statusLine", "")):
|
|
156
|
+
s.pop("statusLine", None)
|
|
157
|
+
else:
|
|
158
|
+
prev = s.get("statusLine")
|
|
159
|
+
if prev and "finops" not in str(prev):
|
|
160
|
+
print(f"You already have a statusLine configured:\n {prev}")
|
|
161
|
+
print("Leaving it alone. Remove it first if you want ours.")
|
|
162
|
+
return
|
|
163
|
+
s["statusLine"] = cmd
|
|
164
|
+
_save_settings(s)
|
|
165
|
+
print(("Removed" if remove else "Installed") + f" the statusline in {SETTINGS}")
|
|
166
|
+
if not remove:
|
|
167
|
+
print(" Shows the model, context pressure, and a cheaper model when one is warranted.")
|
|
168
|
+
print(" Undo: claude-finops --uninstall-statusline")
|
package/finops/procs.py
CHANGED
|
@@ -85,6 +85,7 @@ def _transcript_stats(session_id, pricing):
|
|
|
85
85
|
steps = tokens = 0
|
|
86
86
|
ctx = None
|
|
87
87
|
cost = 0.0
|
|
88
|
+
model = None # the model of the most recent assistant turn: what you are on now
|
|
88
89
|
with open(fp, errors="replace") as fh:
|
|
89
90
|
for line in fh:
|
|
90
91
|
if '"usage"' not in line:
|
|
@@ -108,7 +109,8 @@ def _transcript_stats(session_id, pricing):
|
|
|
108
109
|
if not (c5 or c1):
|
|
109
110
|
c5 = cw
|
|
110
111
|
cost += pricing.estimate(m.get("model"), inp, out, cr, c5, c1)
|
|
111
|
-
|
|
112
|
+
model = m.get("model") or model
|
|
113
|
+
return {"transcript": fp, "steps": steps, "tokens": tokens, "context": ctx, "model": model,
|
|
112
114
|
"est_cost_usd": cost, "last_write_s": time.time() - os.path.getmtime(fp)}
|
|
113
115
|
|
|
114
116
|
|
package/finops/trial.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Run the trial the evidence asks for, instead of only recommending one.
|
|
2
|
+
|
|
3
|
+
model_evidence() ends with an honest caveat: your prompts were never randomly
|
|
4
|
+
assigned to models, so a category can simply have been easier on one of them.
|
|
5
|
+
That caveat can only be closed by actually running the same prompt on both
|
|
6
|
+
models and looking at what came back.
|
|
7
|
+
|
|
8
|
+
So: take real prompts out of your own history, re-run them headlessly on the
|
|
9
|
+
candidate model, and compare what they cost against what they cost the first
|
|
10
|
+
time. `claude -p ... --output-format json` reports cost, turns and errors for
|
|
11
|
+
exactly this purpose.
|
|
12
|
+
|
|
13
|
+
Two things this deliberately does not do:
|
|
14
|
+
|
|
15
|
+
* It never runs anything on its own. A trial spends real money and drives a
|
|
16
|
+
real agent, so it happens on an explicit click, with the bill shown first.
|
|
17
|
+
* It runs in a scratch directory by default, not your repo. Headless Claude
|
|
18
|
+
cannot ask for permission, so tools that need it are denied and recorded —
|
|
19
|
+
but a trial that edits your working tree to prove a point is not a trial
|
|
20
|
+
anyone wants.
|
|
21
|
+
"""
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import shutil
|
|
25
|
+
import subprocess
|
|
26
|
+
import tempfile
|
|
27
|
+
import time
|
|
28
|
+
|
|
29
|
+
from .advisor import model_alias
|
|
30
|
+
from .paths import DB_PATH
|
|
31
|
+
|
|
32
|
+
TIMEOUT_S = 600
|
|
33
|
+
MAX_PROMPTS = 5
|
|
34
|
+
MAX_CHARS = 6000 # a prompt longer than this is usually a paste, not a task
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def samples(category, limit=3):
|
|
38
|
+
"""Real prompts from this category, with what they actually cost first time.
|
|
39
|
+
|
|
40
|
+
Picks around the middle of the cost distribution: the cheapest prompts are
|
|
41
|
+
usually "yes"/"continue" and the most expensive are outliers, and neither
|
|
42
|
+
tells you much about a model.
|
|
43
|
+
"""
|
|
44
|
+
from .analytics import Analytics
|
|
45
|
+
a = Analytics(DB_PATH)
|
|
46
|
+
rows = a.q("""SELECT p.id, p.text, p.models model, p.est_cost_usd cost,
|
|
47
|
+
p.request_count turns, p.tool_calls tools, p.session_id, p.day
|
|
48
|
+
FROM prompts p
|
|
49
|
+
WHERE COALESCE(p.category,'other') = ? AND p.agent = 'claude'
|
|
50
|
+
AND p.models IS NOT NULL AND p.models NOT LIKE '%,%'
|
|
51
|
+
AND p.models != '<synthetic>' AND p.est_cost_usd > 0
|
|
52
|
+
AND LENGTH(p.text) BETWEEN 40 AND ?
|
|
53
|
+
-- Things you actually typed. System turns, tool results, queued
|
|
54
|
+
-- follow-ups and SDK traffic are not tasks anyone would re-run, and
|
|
55
|
+
-- a trial built from task-notification XML measures nothing.
|
|
56
|
+
AND p.source = 'typed'
|
|
57
|
+
AND p.text NOT LIKE '<%' AND p.text NOT LIKE '[Request interrupted%'
|
|
58
|
+
AND p.text NOT LIKE 'Caveat:%' AND p.text NOT LIKE '%<local-command%'
|
|
59
|
+
ORDER BY p.est_cost_usd""", (category, MAX_CHARS))
|
|
60
|
+
if not rows:
|
|
61
|
+
return []
|
|
62
|
+
lo = int(len(rows) * 0.35)
|
|
63
|
+
hi = max(lo + 1, int(len(rows) * 0.85))
|
|
64
|
+
mid = rows[lo:hi] or rows
|
|
65
|
+
step = max(1, len(mid) // limit)
|
|
66
|
+
picked = mid[::step][:limit]
|
|
67
|
+
return [{"prompt_id": r["id"], "text": r["text"], "baseline_model": r["model"],
|
|
68
|
+
"baseline_name": a.pricing.display_name(r["model"]),
|
|
69
|
+
"baseline_cost_usd": round(r["cost"], 4),
|
|
70
|
+
"baseline_turns": r["turns"], "baseline_tools": r["tools"],
|
|
71
|
+
"day": r["day"]} for r in picked]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def available():
|
|
75
|
+
return bool(shutil.which("claude"))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def run_one(text, model, cwd=None, timeout=TIMEOUT_S):
|
|
79
|
+
"""One headless run. Returns what it cost and whether it got there."""
|
|
80
|
+
exe = shutil.which("claude")
|
|
81
|
+
if not exe:
|
|
82
|
+
return {"ok": False, "error": "The `claude` CLI is not on PATH."}
|
|
83
|
+
sandbox = cwd or tempfile.mkdtemp(prefix="finops-trial-")
|
|
84
|
+
started = time.time()
|
|
85
|
+
try:
|
|
86
|
+
p = subprocess.run([exe, "-p", text, "--model", model_alias(model),
|
|
87
|
+
"--output-format", "json"],
|
|
88
|
+
capture_output=True, text=True, timeout=timeout, cwd=sandbox)
|
|
89
|
+
except subprocess.TimeoutExpired:
|
|
90
|
+
return {"ok": False, "error": f"Gave up after {timeout}s.",
|
|
91
|
+
"elapsed_s": round(time.time() - started, 1)}
|
|
92
|
+
except OSError as e:
|
|
93
|
+
return {"ok": False, "error": str(e)}
|
|
94
|
+
try:
|
|
95
|
+
d = json.loads(p.stdout)
|
|
96
|
+
except ValueError:
|
|
97
|
+
return {"ok": False, "error": (p.stderr or p.stdout or "no output")[:400],
|
|
98
|
+
"elapsed_s": round(time.time() - started, 1)}
|
|
99
|
+
usage = d.get("usage") or {}
|
|
100
|
+
return {
|
|
101
|
+
"ok": not d.get("is_error"),
|
|
102
|
+
"cost_usd": d.get("total_cost_usd"),
|
|
103
|
+
"turns": d.get("num_turns"),
|
|
104
|
+
"elapsed_s": round(time.time() - started, 1),
|
|
105
|
+
"duration_api_ms": d.get("duration_api_ms"),
|
|
106
|
+
"output_tokens": usage.get("output_tokens"),
|
|
107
|
+
"input_tokens": usage.get("input_tokens"),
|
|
108
|
+
"denials": len(d.get("permission_denials") or []),
|
|
109
|
+
"stop_reason": d.get("stop_reason") or d.get("terminal_reason"),
|
|
110
|
+
"result": (d.get("result") or "")[:4000],
|
|
111
|
+
"session_id": d.get("session_id"),
|
|
112
|
+
"sandbox": sandbox,
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _verdict(runs, baseline_cost):
|
|
117
|
+
"""What the runs actually showed, in the same language as the back-test."""
|
|
118
|
+
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
119
|
+
if not done:
|
|
120
|
+
return "failed", ("Every run errored or produced nothing, so this says nothing about "
|
|
121
|
+
"cost. Check the errors below before reading anything into it.")
|
|
122
|
+
got = sum(r["cost_usd"] for r in done)
|
|
123
|
+
base = sum(baseline_cost)
|
|
124
|
+
denials = sum(r.get("denials") or 0 for r in done)
|
|
125
|
+
if base <= 0:
|
|
126
|
+
return "unclear", "No usable baseline cost for these prompts."
|
|
127
|
+
pct = 100.0 * (1 - got / base)
|
|
128
|
+
tail = (f" {denials} tool call(s) were denied because a headless run cannot ask for "
|
|
129
|
+
f"permission, so the real task may be larger than this." if denials else "")
|
|
130
|
+
if len(done) < len(runs):
|
|
131
|
+
tail += f" {len(runs) - len(done)} of {len(runs)} runs failed and are excluded."
|
|
132
|
+
if pct >= 25:
|
|
133
|
+
return "confirmed", (f"Re-running your own prompts cost {pct:.0f}% less on the cheaper "
|
|
134
|
+
f"model (${got:.2f} against ${base:.2f} the first time).{tail}")
|
|
135
|
+
if pct >= 0:
|
|
136
|
+
return "marginal", (f"Only {pct:.0f}% cheaper on a re-run (${got:.2f} against "
|
|
137
|
+
f"${base:.2f}). Not the saving the history suggested.{tail}")
|
|
138
|
+
return "contradicted", (f"The re-run cost {abs(pct):.0f}% *more* (${got:.2f} against "
|
|
139
|
+
f"${base:.2f}). The history overstated this switch.{tail}")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def run(category, model, prompts, cwd=None):
|
|
143
|
+
"""Run a set of sample prompts on a candidate model and judge the result."""
|
|
144
|
+
if not prompts:
|
|
145
|
+
return {"ok": False, "error": "Nothing to run."}
|
|
146
|
+
prompts = prompts[:MAX_PROMPTS]
|
|
147
|
+
runs, baseline = [], []
|
|
148
|
+
for item in prompts:
|
|
149
|
+
text = (item.get("text") or "").strip()
|
|
150
|
+
if not text:
|
|
151
|
+
continue
|
|
152
|
+
r = run_one(text, model, cwd=cwd)
|
|
153
|
+
r["prompt"] = text[:300]
|
|
154
|
+
r["baseline_cost_usd"] = item.get("baseline_cost_usd") or 0
|
|
155
|
+
r["baseline_turns"] = item.get("baseline_turns")
|
|
156
|
+
runs.append(r)
|
|
157
|
+
baseline.append(r["baseline_cost_usd"])
|
|
158
|
+
verdict, why = _verdict(runs, baseline)
|
|
159
|
+
done = [r for r in runs if r.get("ok") and r.get("cost_usd") is not None]
|
|
160
|
+
return {
|
|
161
|
+
"ok": True, "category": category, "model": model, "alias": model_alias(model),
|
|
162
|
+
"runs": runs, "verdict": verdict, "why": why,
|
|
163
|
+
"measured_cost_usd": round(sum(r["cost_usd"] for r in done), 4),
|
|
164
|
+
"baseline_cost_usd": round(sum(baseline), 4),
|
|
165
|
+
"ran_at": int(time.time()),
|
|
166
|
+
"note": ("Measured by re-running these exact prompts headlessly. The agent had no "
|
|
167
|
+
"permission to use tools that ask, and worked in a scratch directory, so a "
|
|
168
|
+
"task that needs your repo will look smaller here than it is."),
|
|
169
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-finops",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.7.0",
|
|
4
4
|
"description": "Local FinOps dashboard for Claude Code: what you used, what it cost, why it cost that much, and what to change. Reads your own transcripts, no API key, no data leaves the machine.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"claude-finops": "bin/claude-finops.js"
|
package/run.py
CHANGED
|
@@ -171,6 +171,9 @@ HELP = """Claude FinOps Command Center
|
|
|
171
171
|
claude-finops --keys list which provider keys are configured
|
|
172
172
|
claude-finops --share write ../claude-finops.zip (code only, never your data)
|
|
173
173
|
claude-finops --version print the installed version, and whether a newer one is out
|
|
174
|
+
claude-finops --advise what to switch to in the sessions running right now
|
|
175
|
+
claude-finops --install-hook suggest a cheaper model in Claude Code, as you send each prompt
|
|
176
|
+
claude-finops --install-statusline show model, context and advice in your statusline
|
|
174
177
|
claude-finops --help this message
|
|
175
178
|
|
|
176
179
|
Environment:
|
|
@@ -182,6 +185,28 @@ Environment:
|
|
|
182
185
|
"""
|
|
183
186
|
|
|
184
187
|
|
|
188
|
+
def advise_now():
|
|
189
|
+
"""What every session running right now should consider switching to."""
|
|
190
|
+
from finops.advisor import advise, evidence
|
|
191
|
+
from finops.procs import list_sessions
|
|
192
|
+
from finops.analytics import Analytics
|
|
193
|
+
ev = evidence()
|
|
194
|
+
if not ev:
|
|
195
|
+
return print("No model evidence yet — you need two models run on the same kind of "
|
|
196
|
+
"work before there is anything to compare.")
|
|
197
|
+
rows = list_sessions(Analytics(DB_PATH).pricing)
|
|
198
|
+
if not rows:
|
|
199
|
+
return print("No Claude Code sessions are running.")
|
|
200
|
+
for r in rows:
|
|
201
|
+
a = advise(model=r.get("model"), transcript=r.get("transcript"), ev=ev)
|
|
202
|
+
head = f" {r.get('project') or r.get('session_id') or 'session'}"
|
|
203
|
+
if a:
|
|
204
|
+
print(f"{head}: {a['line']}")
|
|
205
|
+
print(f"{' ' * len(head)} run {a['command']}")
|
|
206
|
+
else:
|
|
207
|
+
print(f"{head}: nothing to change.")
|
|
208
|
+
|
|
209
|
+
|
|
185
210
|
def _version():
|
|
186
211
|
from finops.update import _installed
|
|
187
212
|
return _installed()
|
|
@@ -292,6 +317,23 @@ def main():
|
|
|
292
317
|
return print(HELP)
|
|
293
318
|
if "--version" in args or "-v" in args or "-V" in args:
|
|
294
319
|
return version()
|
|
320
|
+
# Integration entry points. --hook and --statusline are invoked by Claude Code
|
|
321
|
+
# with a JSON payload on stdin, many times a session; they print one line and
|
|
322
|
+
# never fail loudly.
|
|
323
|
+
if "--hook" in args:
|
|
324
|
+
from finops.integrate import hook
|
|
325
|
+
return sys.exit(hook())
|
|
326
|
+
if "--statusline" in args:
|
|
327
|
+
from finops.integrate import statusline
|
|
328
|
+
return sys.exit(statusline())
|
|
329
|
+
if "--install-hook" in args or "--uninstall-hook" in args:
|
|
330
|
+
from finops.integrate import install_hook
|
|
331
|
+
return install_hook(remove="--uninstall-hook" in args)
|
|
332
|
+
if "--install-statusline" in args or "--uninstall-statusline" in args:
|
|
333
|
+
from finops.integrate import install_statusline
|
|
334
|
+
return install_statusline(remove="--uninstall-statusline" in args)
|
|
335
|
+
if "--advise" in args:
|
|
336
|
+
return advise_now()
|
|
295
337
|
if "--where" in args:
|
|
296
338
|
return where()
|
|
297
339
|
if "--keys" in args:
|
package/web/app.js
CHANGED
|
@@ -1354,6 +1354,8 @@ function evidenceBody(ev) {
|
|
|
1354
1354
|
statusChip(VERDICT[r.x.verdict][2], VERDICT[r.x.verdict][1])}</span>`},
|
|
1355
1355
|
{h: 'Would save', num: 1, f: r => r.x.verdict === 'supported'
|
|
1356
1356
|
? `<b>${fmtUSD(r.x.estimated_savings_usd)}</b>` : `<span class="note">${fmtUSD(r.x.estimated_savings_usd)}</span>`},
|
|
1357
|
+
{h: '', f: r => `<button class="act ghost trial-btn" data-cat="${esc(r.c.category)}"
|
|
1358
|
+
data-model="${esc(r.x.model)}" data-name="${esc(r.x.name)}">Try it →</button>`},
|
|
1357
1359
|
], rows) + `<div class="note" style="padding:10px 14px">${esc(ev.method)}</div>`;
|
|
1358
1360
|
}
|
|
1359
1361
|
|
|
@@ -1416,6 +1418,7 @@ VIEWS.modelswitch = async (page) => {
|
|
|
1416
1418
|
{h: 'Could save', num: 1, f: r => fmtUSD(r.estimated_savings_usd)},
|
|
1417
1419
|
{h: 'Why', f: r => esc(r.why)},
|
|
1418
1420
|
], m.projects);
|
|
1421
|
+
wireTrials(page);
|
|
1419
1422
|
addChart(page, 'Savings by switch', el => C.barsH(el, {
|
|
1420
1423
|
rows: m.switches.slice(0, 10), label: r => clip(`${r.scope} → ${r.recommended_name}`, 48),
|
|
1421
1424
|
value: r => r.estimated_savings_usd,
|
|
@@ -1426,6 +1429,116 @@ VIEWS.modelswitch = async (page) => {
|
|
|
1426
1429
|
{badge: BADGE.recommendation, hint: 'Colour = confidence (green high, blue medium, amber low)'});
|
|
1427
1430
|
};
|
|
1428
1431
|
|
|
1432
|
+
/* ---------- trial: stop recommending, start measuring ----------
|
|
1433
|
+
The evidence ends at "strong evidence for a trial, not proof". This runs the
|
|
1434
|
+
trial: real prompts out of your own history, re-run headlessly on the
|
|
1435
|
+
candidate model, priced against what they cost the first time. It spends real
|
|
1436
|
+
money, so nothing happens without two clicks. */
|
|
1437
|
+
function trialPanelHTML(cat, model, name, s) {
|
|
1438
|
+
if (!s.available) {
|
|
1439
|
+
return `<div class="empty">The <code>claude</code> CLI is not on PATH, so a trial cannot be
|
|
1440
|
+
run from here.</div>`;
|
|
1441
|
+
}
|
|
1442
|
+
if (!s.samples.length) {
|
|
1443
|
+
return `<div class="empty">No prompt you actually typed in this category is short enough to
|
|
1444
|
+
re-run safely.</div>`;
|
|
1445
|
+
}
|
|
1446
|
+
const base = s.samples.reduce((a, x) => a + (x.baseline_cost_usd || 0), 0);
|
|
1447
|
+
return `
|
|
1448
|
+
<div class="dt">These are ${s.samples.length} prompts you really sent in
|
|
1449
|
+
<b>${esc(cat.replace('_', ' '))}</b>. Running them again on <b>${esc(name)}</b> costs money —
|
|
1450
|
+
they cost ${fmtUSD(base)} the first time, and the cheaper model should come in under that.</div>
|
|
1451
|
+
<div class="stack trial-samples">${s.samples.map((x, i) => `
|
|
1452
|
+
<div class="dt trial-s" data-i="${i}">
|
|
1453
|
+
<span class="note">${esc(x.day)} · ${esc(x.baseline_name)} · ${fmtUSD(x.baseline_cost_usd)} ·
|
|
1454
|
+
${fmtInt(x.baseline_turns)} turns</span>
|
|
1455
|
+
<div class="trial-text">${esc(x.text.slice(0, 400))}${x.text.length > 400 ? '…' : ''}</div>
|
|
1456
|
+
</div>`).join('')}</div>
|
|
1457
|
+
<div class="dt note">Runs headlessly in a scratch directory. Tools that need permission are
|
|
1458
|
+
denied, because a headless agent cannot ask — so a task that needs your repo will look
|
|
1459
|
+
smaller here than it really is.</div>
|
|
1460
|
+
<div class="live-actions">
|
|
1461
|
+
<button class="act trial-run">▶ Run ${s.samples.length} prompts on ${esc(name)}</button>
|
|
1462
|
+
<button class="act ghost trial-copy">Copy the first prompt instead</button>
|
|
1463
|
+
<span class="trial-msg note"></span>
|
|
1464
|
+
</div>
|
|
1465
|
+
<div class="trial-out"></div>`;
|
|
1466
|
+
}
|
|
1467
|
+
|
|
1468
|
+
const TRIAL_VERDICT = {
|
|
1469
|
+
confirmed: ['healthy', 'Confirmed by running it'],
|
|
1470
|
+
marginal: ['high', 'Smaller than advertised'],
|
|
1471
|
+
contradicted: ['critical', 'History overstated it'],
|
|
1472
|
+
failed: ['critical', 'Runs failed'],
|
|
1473
|
+
unclear: ['high', 'Inconclusive'],
|
|
1474
|
+
};
|
|
1475
|
+
|
|
1476
|
+
function trialResultHTML(r) {
|
|
1477
|
+
const v = TRIAL_VERDICT[r.verdict] || TRIAL_VERDICT.unclear;
|
|
1478
|
+
return `<div class="dt"><b>${statusChip(v[0], v[1])}</b> ${esc(r.why)}</div>` + table([
|
|
1479
|
+
{h: 'Prompt', trunc: 1, f: x => esc(x.prompt)},
|
|
1480
|
+
{h: 'First time', num: 1, f: x => fmtUSD(x.baseline_cost_usd)},
|
|
1481
|
+
{h: 'On ' + esc(r.alias), num: 1, f: x => x.ok ? fmtUSD(x.cost_usd) : '—'},
|
|
1482
|
+
{h: 'Turns', num: 1, f: x => x.ok ? fmtInt(x.turns) : '—'},
|
|
1483
|
+
{h: 'Took', num: 1, f: x => x.ok ? x.elapsed_s + 's' : '—'},
|
|
1484
|
+
{h: 'Denied', num: 1, f: x => x.denials ? fmtInt(x.denials) : ''},
|
|
1485
|
+
{h: 'Result', f: x => x.ok ? `<span class="note">${esc((x.result || '').slice(0, 120))}</span>`
|
|
1486
|
+
: `<span class="status critical">${esc((x.error || 'failed').slice(0, 120))}</span>`},
|
|
1487
|
+
], r.runs) + `<div class="note" style="padding:8px 14px">${esc(r.note)}</div>`;
|
|
1488
|
+
}
|
|
1489
|
+
|
|
1490
|
+
function wireTrials(page) {
|
|
1491
|
+
page.querySelectorAll('.trial-btn').forEach(b => b.onclick = async () => {
|
|
1492
|
+
const row = b.closest('tr');
|
|
1493
|
+
if (row.nextElementSibling?.classList.contains('trial-row')) {
|
|
1494
|
+
row.nextElementSibling.remove(); return;
|
|
1495
|
+
}
|
|
1496
|
+
const {cat, model, name} = b.dataset;
|
|
1497
|
+
const tr = h(`<tr class="trial-row"><td colspan="10"><div class="trial-panel">
|
|
1498
|
+
<div class="empty">Finding prompts you sent…</div></div></td></tr>`);
|
|
1499
|
+
row.after(tr);
|
|
1500
|
+
const host = tr.querySelector('.trial-panel');
|
|
1501
|
+
let s;
|
|
1502
|
+
try {
|
|
1503
|
+
s = await fetch(`/api/trial?category=${encodeURIComponent(cat)}&limit=3`).then(r => r.json());
|
|
1504
|
+
} catch (e) { host.innerHTML = `<div class="empty">${esc(e.message)}</div>`; return; }
|
|
1505
|
+
host.innerHTML = trialPanelHTML(cat, model, name, s);
|
|
1506
|
+
const msg = host.querySelector('.trial-msg');
|
|
1507
|
+
host.querySelector('.trial-copy')?.addEventListener('click', async () => {
|
|
1508
|
+
try { await navigator.clipboard.writeText(s.samples[0].text); msg.textContent = 'Copied.'; }
|
|
1509
|
+
catch { msg.textContent = 'Could not copy.'; }
|
|
1510
|
+
});
|
|
1511
|
+
host.querySelector('.trial-run')?.addEventListener('click', async ev => {
|
|
1512
|
+
const btn = ev.currentTarget;
|
|
1513
|
+
if (btn.dataset.armed !== '1') {
|
|
1514
|
+
btn.dataset.armed = '1';
|
|
1515
|
+
btn.textContent = 'Click again to spend real money';
|
|
1516
|
+
btn.classList.add('warn');
|
|
1517
|
+
return;
|
|
1518
|
+
}
|
|
1519
|
+
btn.disabled = true;
|
|
1520
|
+
btn.textContent = 'Running…';
|
|
1521
|
+
msg.textContent = 'Each prompt runs to completion; this can take a few minutes.';
|
|
1522
|
+
try {
|
|
1523
|
+
const r = await fetch('/api/trial/run', {
|
|
1524
|
+
method: 'POST',
|
|
1525
|
+
headers: {'X-FinOps-Action': '1', 'Content-Type': 'application/json'},
|
|
1526
|
+
body: JSON.stringify({category: cat, model, prompts: s.samples}),
|
|
1527
|
+
}).then(x => x.json());
|
|
1528
|
+
host.querySelector('.trial-out').innerHTML = r.ok
|
|
1529
|
+
? trialResultHTML(r) : `<div class="empty">${esc(r.error || 'failed')}</div>`;
|
|
1530
|
+
msg.textContent = '';
|
|
1531
|
+
} catch (e) {
|
|
1532
|
+
msg.textContent = e.message;
|
|
1533
|
+
}
|
|
1534
|
+
btn.disabled = false;
|
|
1535
|
+
btn.classList.remove('warn');
|
|
1536
|
+
btn.textContent = '▶ Run again';
|
|
1537
|
+
btn.dataset.armed = '';
|
|
1538
|
+
});
|
|
1539
|
+
});
|
|
1540
|
+
}
|
|
1541
|
+
|
|
1429
1542
|
/* ---------- waste ---------- */
|
|
1430
1543
|
VIEWS.waste = async (page) => {
|
|
1431
1544
|
const w = await api('waste');
|
|
@@ -1644,6 +1757,10 @@ VIEWS.live = async (page) => {
|
|
|
1644
1757
|
<div class="dt">${x.status === 'busy' ? '<b>Working now</b>' : 'Idle'}${x.uptime ? ` · up ${esc(x.uptime)}` : ''} ·
|
|
1645
1758
|
last activity ${x.last_write_s == null ? '—' : dur(x.last_write_s)} ago${x.memory_mb == null ? '' : ` · ${fmtInt(x.memory_mb)} MB`} ·
|
|
1646
1759
|
<span class="note">${esc(x.cwd)}</span></div>
|
|
1760
|
+
${x.advice ? `<div class="dt switch-tip"><b>${x.advice.trial ? 'Worth trying' : 'Cheaper model'}:</b>
|
|
1761
|
+
${esc(x.advice.line)}
|
|
1762
|
+
<button class="act ghost" data-copy="${esc(x.advice.command)}"
|
|
1763
|
+
title="${esc(x.advice.why)}">Copy ${esc(x.advice.command)}</button></div>` : ''}
|
|
1647
1764
|
${x.severity !== 'ok' ? `<div class="dt"><b>Advice:</b> ${x.severity === 'high'
|
|
1648
1765
|
? 'Very large context. Use <b>Hand over</b> to continue in a fresh session, or split the remaining work into sub-sessions.'
|
|
1649
1766
|
: 'Getting heavy. Hit <b>Compact</b> at the next break, or close it if the task is done.'}</div>` : ''}
|
package/web/styles.css
CHANGED
|
@@ -367,6 +367,9 @@ table.tbl .sub { font-size:10.5px; color:var(--muted); }
|
|
|
367
367
|
.live-msg.ok { color: var(--good-ink); } .live-msg.err { color: var(--critical-ink); }
|
|
368
368
|
.item.gone { opacity: .45; }
|
|
369
369
|
|
|
370
|
+
/* An explicit display beats the hidden attribute, so the panel has to opt back
|
|
371
|
+
out or it is permanently open. */
|
|
372
|
+
.handover[hidden] { display: none; }
|
|
370
373
|
.handover { margin-top: 8px; padding: 10px; border: 1px dashed var(--border); border-radius: 8px; display: grid; gap: 8px; }
|
|
371
374
|
.handover textarea { width: 100%; box-sizing: border-box; font: inherit; font-size: 12.5px; padding: 8px; border-radius: 6px;
|
|
372
375
|
border: 1px solid var(--border); background: transparent; color: inherit; resize: vertical; }
|
|
@@ -417,3 +420,18 @@ table.tbl .sub { font-size:10.5px; color:var(--muted); }
|
|
|
417
420
|
overflow:hidden; text-overflow:ellipsis; white-space:nowrap; }
|
|
418
421
|
.who .em, .who .pl { font-size:10px; color:var(--muted); line-height:1.35;
|
|
419
422
|
overflow:hidden; text-overflow:ellipsis; white-space:nowrap; }
|
|
423
|
+
|
|
424
|
+
/* Live model advice on a running session: it is an opportunity, not a problem,
|
|
425
|
+
so it reads as a note with an accent edge rather than a warning. */
|
|
426
|
+
.switch-tip { border-left: 2px solid var(--accent, #eb6834); padding-left: 9px; }
|
|
427
|
+
.switch-tip .act { margin-left: 8px; }
|
|
428
|
+
|
|
429
|
+
/* Trial panel: an experiment you are about to pay for, so it reads as a quiet
|
|
430
|
+
workbench rather than another recommendation. */
|
|
431
|
+
.trial-row > td { padding: 0 !important; background: var(--surface-2); }
|
|
432
|
+
.trial-panel { padding: 12px 14px; display: grid; gap: 8px;
|
|
433
|
+
border-left: 2px solid var(--accent, #eb6834); }
|
|
434
|
+
.trial-text { font-size: 12px; margin-top: 3px; white-space: pre-wrap;
|
|
435
|
+
max-height: 76px; overflow: auto; color: var(--text-2); }
|
|
436
|
+
.trial-s { border-top: 1px solid var(--border); padding-top: 7px; }
|
|
437
|
+
.trial-out:empty { display: none; }
|