claude-finops 0.7.2 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -47
- package/config/pricing.json +62 -42
- package/config/settings.json +11 -5
- package/finops/actions.py +1 -1
- package/finops/analytics.py +690 -611
- package/finops/api.py +174 -55
- package/finops/diagnose.py +17 -20
- package/finops/etl.py +149 -52
- package/finops/integrate.py +22 -57
- package/finops/pricing.py +57 -6
- package/finops/procs.py +7 -5
- package/finops/report.py +29 -23
- package/finops/segments.py +149 -0
- package/package.json +1 -1
- package/run.cmd +3 -2
- package/run.py +26 -59
- package/web/app.js +253 -329
- package/web/charts.js +26 -15
- package/finops/advisor.py +0 -188
- package/finops/trial.py +0 -169
package/finops/etl.py
CHANGED
|
@@ -5,6 +5,7 @@ Entities: account -> billing_period -> project -> session -> prompt -> request
|
|
|
5
5
|
transcripts are stored as NULL and rendered as "Unavailable from connected Claude
|
|
6
6
|
data" by the UI.
|
|
7
7
|
"""
|
|
8
|
+
import hashlib
|
|
8
9
|
import json
|
|
9
10
|
import os
|
|
10
11
|
import re
|
|
@@ -18,6 +19,20 @@ from .pricing import Pricing
|
|
|
18
19
|
from .paths import ROOT, DB_PATH
|
|
19
20
|
DEFAULT_SOURCE = os.path.expanduser("~/.claude/projects")
|
|
20
21
|
|
|
22
|
+
SCHEMA_VERSION = 3 # 3: cross-session request_id dedup (resumed sessions copy history)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def needs_rebuild(db_path):
|
|
26
|
+
if not os.path.exists(db_path):
|
|
27
|
+
return True
|
|
28
|
+
try:
|
|
29
|
+
con = sqlite3.connect(db_path)
|
|
30
|
+
row = con.execute("SELECT value FROM meta WHERE key='schema_version'").fetchone()
|
|
31
|
+
con.close()
|
|
32
|
+
except sqlite3.Error:
|
|
33
|
+
return True
|
|
34
|
+
return not row or row[0] != str(SCHEMA_VERSION)
|
|
35
|
+
|
|
21
36
|
SCHEMA = """
|
|
22
37
|
PRAGMA journal_mode=WAL;
|
|
23
38
|
|
|
@@ -88,6 +103,8 @@ CREATE TABLE requests (
|
|
|
88
103
|
context_tokens INTEGER, -- input + cache read + cache write (prompt side)
|
|
89
104
|
est_cost_usd REAL,
|
|
90
105
|
est_cost_no_cache_usd REAL,
|
|
106
|
+
priced_as TEXT, -- price list used; differs from model for long context
|
|
107
|
+
unpriced_long_context INTEGER DEFAULT 0, -- over the window with no [1m] price to use
|
|
91
108
|
latency_ms REAL,
|
|
92
109
|
tool_call_count INTEGER DEFAULT 0,
|
|
93
110
|
is_sidechain INTEGER DEFAULT 0,
|
|
@@ -134,6 +151,9 @@ SLASH = re.compile(r"^\s*/([a-z0-9][\w:-]*)(?=\s|$)", re.I)
|
|
|
134
151
|
CMD_NAME = re.compile(r"<command-name>/?([\w:-]+)</command-name>")
|
|
135
152
|
WS = re.compile(r"\s+")
|
|
136
153
|
|
|
154
|
+
INJECTED_PREFIXES = ("<task-notification", "<local-command-stdout", "<local-command-caveat",
|
|
155
|
+
"<bash-input", "<bash-stdout", "<bash-stderr", "<system-reminder")
|
|
156
|
+
|
|
137
157
|
|
|
138
158
|
def _ts(s):
|
|
139
159
|
if not s:
|
|
@@ -191,6 +211,9 @@ class Loader:
|
|
|
191
211
|
self.projects = {}
|
|
192
212
|
self.agent = None
|
|
193
213
|
self.pending_skill = None
|
|
214
|
+
self.pending_results = {}
|
|
215
|
+
self.group = None
|
|
216
|
+
self.seen_request_ids = set() # dedup request_id across sessions (resumed sessions)
|
|
194
217
|
|
|
195
218
|
# ---------- infrastructure ----------
|
|
196
219
|
def build(self, verbose=True):
|
|
@@ -233,6 +256,7 @@ class Loader:
|
|
|
233
256
|
("pricing_updated", str(self.pricing.updated)),
|
|
234
257
|
("pricing_source", str(self.pricing.source)),
|
|
235
258
|
("cost_basis", "estimated"),
|
|
259
|
+
("schema_version", str(SCHEMA_VERSION)),
|
|
236
260
|
):
|
|
237
261
|
self.db.execute("INSERT INTO meta VALUES (?,?)", (k, v))
|
|
238
262
|
self.db.commit()
|
|
@@ -287,6 +311,8 @@ class Loader:
|
|
|
287
311
|
session_id = agent["parent"] if agent else os.path.splitext(os.path.basename(path))[0]
|
|
288
312
|
self.agent = agent
|
|
289
313
|
self.pending_skill = None
|
|
314
|
+
self.pending_results = {} # tool_use_id -> result chars, applied once the row exists
|
|
315
|
+
self.group = None
|
|
290
316
|
cwd = next((r.get("cwd") for r in rows if r.get("cwd")), None)
|
|
291
317
|
pid = self.project_id(slug, cwd)
|
|
292
318
|
title = next((r.get("aiTitle") for r in rows if r.get("type") == "ai-title"), None)
|
|
@@ -301,37 +327,64 @@ class Loader:
|
|
|
301
327
|
|
|
302
328
|
cur_prompt = None
|
|
303
329
|
prev_time = None
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
330
|
+
self.group = None # {"key", "lines": [...], "prev_time", "prompt_id"}
|
|
331
|
+
|
|
332
|
+
def flush():
|
|
333
|
+
if self.group:
|
|
334
|
+
self.insert_request(self.group["lines"], session_id, pid, self.group["prompt_id"],
|
|
335
|
+
self.group["prev_time"])
|
|
336
|
+
self.group = None
|
|
337
|
+
|
|
338
|
+
try:
|
|
339
|
+
for r in rows:
|
|
340
|
+
typ = r.get("type")
|
|
341
|
+
ts = r.get("timestamp")
|
|
342
|
+
t = _ts(ts)
|
|
343
|
+
|
|
344
|
+
if typ == "user":
|
|
345
|
+
msg = r.get("message") or {}
|
|
346
|
+
self.record_results(r, msg)
|
|
347
|
+
# tool results and meta lines are not human prompts, and must not
|
|
348
|
+
# split a request that is still waiting on its tool results
|
|
349
|
+
if r.get("toolUseResult") is not None or r.get("isMeta"):
|
|
350
|
+
prev_time = t or prev_time
|
|
351
|
+
continue
|
|
352
|
+
text = _text_of(msg.get("content"))
|
|
353
|
+
if text.lstrip().startswith(INJECTED_PREFIXES):
|
|
354
|
+
prev_time = t or prev_time
|
|
355
|
+
continue
|
|
356
|
+
if agent:
|
|
357
|
+
prev_time = t or prev_time
|
|
358
|
+
continue
|
|
359
|
+
if not text.strip():
|
|
360
|
+
prev_time = t or prev_time
|
|
361
|
+
continue
|
|
362
|
+
flush()
|
|
363
|
+
cur_prompt = self.insert_prompt(r, text, session_id, pid)
|
|
315
364
|
prev_time = t or prev_time
|
|
316
|
-
continue
|
|
317
|
-
text = _text_of(msg.get("content"))
|
|
318
|
-
if agent:
|
|
319
|
-
prev_time = t or prev_time
|
|
320
|
-
continue
|
|
321
|
-
if not text.strip():
|
|
322
|
-
prev_time = t or prev_time
|
|
323
|
-
continue
|
|
324
|
-
cur_prompt = self.insert_prompt(r, text, session_id, pid)
|
|
325
|
-
prev_time = t or prev_time
|
|
326
365
|
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
366
|
+
elif typ == "assistant":
|
|
367
|
+
if agent and cur_prompt is None:
|
|
368
|
+
cur_prompt = self.parent_prompt(session_id, r.get("timestamp"))
|
|
369
|
+
key = (r.get("requestId") or (r.get("message") or {}).get("id") or r.get("uuid"))
|
|
370
|
+
if self.group and self.group["key"] == key:
|
|
371
|
+
self.group["lines"].append(r)
|
|
372
|
+
else:
|
|
373
|
+
flush()
|
|
374
|
+
self.pending_skill = None # never apply a stale skill id to a new group
|
|
375
|
+
self.group = {"key": key, "lines": [r], "prev_time": prev_time,
|
|
376
|
+
"prompt_id": cur_prompt}
|
|
377
|
+
prev_time = t or prev_time
|
|
332
378
|
|
|
333
|
-
|
|
334
|
-
|
|
379
|
+
elif typ in ("attachment", "system"):
|
|
380
|
+
prev_time = t or prev_time
|
|
381
|
+
finally:
|
|
382
|
+
flush()
|
|
383
|
+
# any result whose tool_calls row was inserted by an earlier flush
|
|
384
|
+
for tool_use_id, n in self.pending_results.items():
|
|
385
|
+
self.db.execute("UPDATE tool_calls SET result_chars=? WHERE tool_use_id=?",
|
|
386
|
+
(n, tool_use_id))
|
|
387
|
+
self.pending_results = {}
|
|
335
388
|
|
|
336
389
|
def parent_prompt(self, session_id, ts):
|
|
337
390
|
row = self.db.execute("SELECT id FROM prompts WHERE session_id=? AND ts<=? "
|
|
@@ -341,19 +394,34 @@ class Loader:
|
|
|
341
394
|
def record_results(self, r, msg):
|
|
342
395
|
"""Size of tool results, and of skill bodies injected as meta messages."""
|
|
343
396
|
content = msg.get("content")
|
|
344
|
-
if r.get("isMeta")
|
|
345
|
-
self.
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
397
|
+
if r.get("isMeta"):
|
|
398
|
+
if self.group:
|
|
399
|
+
# the Skill's body can arrive before the group holding its tool_use
|
|
400
|
+
# flushes; stash it under that block's tool_use_id so insert_request
|
|
401
|
+
# applies it once the tool_calls row exists.
|
|
402
|
+
tool_use_id = None
|
|
403
|
+
for ln in self.group["lines"]:
|
|
404
|
+
for c in ((ln.get("message") or {}).get("content") or []):
|
|
405
|
+
if isinstance(c, dict) and c.get("type") == "tool_use" and c.get("name") == "Skill":
|
|
406
|
+
tool_use_id = c.get("id")
|
|
407
|
+
if tool_use_id:
|
|
408
|
+
n = len(_text_of(content))
|
|
409
|
+
self.pending_results[tool_use_id] = self.pending_results.get(tool_use_id, 0) + n
|
|
410
|
+
return
|
|
411
|
+
if getattr(self, "pending_skill", None):
|
|
412
|
+
self.db.execute("UPDATE tool_calls SET result_chars=result_chars+? WHERE id=?",
|
|
413
|
+
(len(_text_of(content)), self.pending_skill))
|
|
414
|
+
self.pending_skill = None
|
|
415
|
+
return
|
|
349
416
|
if not isinstance(content, list):
|
|
350
417
|
return
|
|
351
418
|
for c in content:
|
|
352
419
|
if isinstance(c, dict) and c.get("type") == "tool_result":
|
|
353
420
|
body = c.get("content")
|
|
354
421
|
n = len(body) if isinstance(body, str) else _result_chars(body)
|
|
355
|
-
|
|
356
|
-
|
|
422
|
+
# the tool_calls row for this id may not exist yet: its request is
|
|
423
|
+
# still an open group and is only inserted when it flushes
|
|
424
|
+
self.pending_results[c.get("tool_use_id")] = n
|
|
357
425
|
|
|
358
426
|
def insert_prompt(self, r, text, session_id, pid):
|
|
359
427
|
cat, conf, ev = classify(text)
|
|
@@ -363,19 +431,23 @@ class Loader:
|
|
|
363
431
|
src = f"slash:/{m.group(1)}"
|
|
364
432
|
cat = cat if cat != "other" else "automation"
|
|
365
433
|
ts = r.get("timestamp")
|
|
366
|
-
norm = WS.sub(" ", text.strip().lower())
|
|
434
|
+
norm = WS.sub(" ", text.strip().lower())
|
|
435
|
+
norm_hash = hashlib.sha1(norm.encode("utf-8")).hexdigest()[:16]
|
|
367
436
|
cur = self.db.execute(
|
|
368
437
|
"INSERT INTO prompts (uuid, session_id, project_id, ts, day, text, char_len,"
|
|
369
438
|
" word_len, category, category_confidence, category_evidence, source, norm_hash)"
|
|
370
439
|
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
371
440
|
(r.get("uuid"), session_id, pid, ts, (ts or "")[:10], text, len(text),
|
|
372
|
-
len(text.split()), cat, conf, json.dumps(ev), src,
|
|
441
|
+
len(text.split()), cat, conf, json.dumps(ev), src, norm_hash))
|
|
373
442
|
return cur.lastrowid
|
|
374
443
|
|
|
375
|
-
def insert_request(self,
|
|
444
|
+
def insert_request(self, lines, session_id, pid, prompt_id, prev_time):
|
|
445
|
+
first, last = lines[0], lines[-1]
|
|
446
|
+
r = last # stop_reason / usage from the final line
|
|
376
447
|
msg = r.get("message") or {}
|
|
377
448
|
u = msg.get("usage") or {}
|
|
378
449
|
model = msg.get("model") or "unknown"
|
|
450
|
+
speed = u.get("speed")
|
|
379
451
|
inp = int(u.get("input_tokens") or 0)
|
|
380
452
|
out = int(u.get("output_tokens") or 0)
|
|
381
453
|
think = int((u.get("output_tokens_details") or {}).get("thinking_tokens") or 0)
|
|
@@ -389,11 +461,33 @@ class Loader:
|
|
|
389
461
|
billable = inp + out + cr + cw
|
|
390
462
|
context = inp + cr + cw
|
|
391
463
|
|
|
392
|
-
|
|
393
|
-
|
|
464
|
+
# Price against the variant the context proves was used, not just the name in
|
|
465
|
+
# the transcript: anything above the standard window was the long-context
|
|
466
|
+
# variant and is billed at a premium.
|
|
467
|
+
priced_as, unpriced_long = self.pricing.effective_model(model, context, speed=speed)
|
|
468
|
+
cost = self.pricing.estimate(priced_as, inp, out, cr, c5, c1)
|
|
469
|
+
no_cache_part, cache_part = self.pricing.uncached_baseline(priced_as, cr, c5, c1)
|
|
394
470
|
cost_no_cache = cost - cache_part + no_cache_part
|
|
395
471
|
|
|
396
|
-
|
|
472
|
+
tools, seen = [], set()
|
|
473
|
+
for ln in lines:
|
|
474
|
+
for c in ((ln.get("message") or {}).get("content") or []):
|
|
475
|
+
if isinstance(c, dict) and c.get("type") == "tool_use" and c.get("id") not in seen:
|
|
476
|
+
seen.add(c.get("id"))
|
|
477
|
+
tools.append(c)
|
|
478
|
+
|
|
479
|
+
if billable == 0 and not tools:
|
|
480
|
+
return
|
|
481
|
+
|
|
482
|
+
request_id = first.get("requestId") or msg.get("id")
|
|
483
|
+
if request_id and request_id in self.seen_request_ids:
|
|
484
|
+
# resumed sessions copy earlier history verbatim, including request_ids
|
|
485
|
+
# already inserted from another session; keep only the first occurrence.
|
|
486
|
+
return
|
|
487
|
+
if request_id:
|
|
488
|
+
self.seen_request_ids.add(request_id)
|
|
489
|
+
|
|
490
|
+
ts = first.get("timestamp") # the request started at its first line
|
|
397
491
|
t = _ts(ts)
|
|
398
492
|
latency = None
|
|
399
493
|
if t and prev_time:
|
|
@@ -401,25 +495,24 @@ class Loader:
|
|
|
401
495
|
if 0 <= d <= 900_000: # ignore idle gaps > 15 min, they are not latency
|
|
402
496
|
latency = d
|
|
403
497
|
|
|
404
|
-
tools = [c for c in (msg.get("content") or [])
|
|
405
|
-
if isinstance(c, dict) and c.get("type") == "tool_use"]
|
|
406
|
-
|
|
407
498
|
cur = self.db.execute(
|
|
408
499
|
"INSERT INTO requests (uuid, request_id, session_id, project_id, prompt_id,"
|
|
409
500
|
" ts, day, hour, model, model_known, effort, service_tier, stop_reason,"
|
|
410
501
|
" input_tokens, output_tokens, thinking_tokens, cache_read_tokens,"
|
|
411
502
|
" cache_write_5m, cache_write_1h, cache_write_tokens, billable_tokens,"
|
|
412
|
-
" context_tokens, est_cost_usd, est_cost_no_cache_usd,
|
|
503
|
+
" context_tokens, est_cost_usd, est_cost_no_cache_usd,"
|
|
504
|
+
" priced_as, unpriced_long_context, latency_ms,"
|
|
413
505
|
" tool_call_count, is_sidechain, agent_id, agent_type, agent_desc)"
|
|
414
|
-
" VALUES (
|
|
415
|
-
(
|
|
416
|
-
(ts or "")[:10], (t.hour if t else None), model,
|
|
417
|
-
1 if self.pricing.is_known(model) else 0,
|
|
506
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
507
|
+
(first.get("uuid"), request_id, session_id, pid,
|
|
508
|
+
prompt_id, ts, (ts or "")[:10], (t.hour if t else None), model,
|
|
509
|
+
1 if self.pricing.is_known(model) else 0, first.get("effort"),
|
|
418
510
|
u.get("service_tier"), msg.get("stop_reason"), inp, out, think, cr,
|
|
419
|
-
c5, c1, cw, billable, context, cost, cost_no_cache,
|
|
420
|
-
|
|
511
|
+
c5, c1, cw, billable, context, cost, cost_no_cache,
|
|
512
|
+
priced_as, 1 if unpriced_long else 0, latency,
|
|
513
|
+
len(tools), 1 if (first.get("isSidechain") or self.agent) else 0,
|
|
421
514
|
*((self.agent["id"], self.agent["type"], self.agent["desc"]) if self.agent
|
|
422
|
-
else (None, "inline" if
|
|
515
|
+
else (None, "inline" if first.get("isSidechain") else None, None))))
|
|
423
516
|
rpk = cur.lastrowid
|
|
424
517
|
|
|
425
518
|
day = (ts or "")[:10]
|
|
@@ -449,6 +542,10 @@ class Loader:
|
|
|
449
542
|
" VALUES (?,?,?,?,?,?,?,?,?,?,?)",
|
|
450
543
|
(rpk, session_id, pid, prompt_id, ts, day, name, target, c.get("id"),
|
|
451
544
|
kind, server))
|
|
545
|
+
tool_use_id = c.get("id")
|
|
546
|
+
if tool_use_id in self.pending_results:
|
|
547
|
+
self.db.execute("UPDATE tool_calls SET result_chars=? WHERE tool_use_id=?",
|
|
548
|
+
(self.pending_results.pop(tool_use_id), tool_use_id))
|
|
452
549
|
if kind == "skill":
|
|
453
550
|
self.pending_skill = self.db.execute("SELECT last_insert_rowid()").fetchone()[0]
|
|
454
551
|
if name in FILE_TOOLS and isinstance(args, dict) and args.get("file_path"):
|
package/finops/integrate.py
CHANGED
|
@@ -1,16 +1,14 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""A statusline inside Claude Code itself, showing model and context usage.
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
3
|
+
claude-finops --statusline reads a statusline payload on stdin, prints model + context %
|
|
4
|
+
claude-finops --install-statusline wire it into settings
|
|
5
|
+
claude-finops --hook retired no-op, kept so existing installs do not error
|
|
6
|
+
claude-finops --install-hook prints that the hook is retired and does nothing else
|
|
5
7
|
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
Both are on the path of every prompt, so both are built to be boring: fail
|
|
11
|
-
silent, never block, never take long. A hook that erases someone's prompt
|
|
12
|
-
because a database was locked is a far worse bug than missing advice, so the
|
|
13
|
-
hook never returns a blocking exit code — it exits 0 whatever happens.
|
|
8
|
+
The prompt hook that used to "suggest a cheaper model" is retired: that
|
|
9
|
+
suggestion had no basis — it would have meant repricing work that never ran.
|
|
10
|
+
The statusline is on the path of every prompt, so it is built to be boring:
|
|
11
|
+
fail silent, never block, never take long.
|
|
14
12
|
"""
|
|
15
13
|
import json
|
|
16
14
|
import os
|
|
@@ -47,49 +45,28 @@ def _dig(d, *paths, default=None):
|
|
|
47
45
|
return default
|
|
48
46
|
|
|
49
47
|
|
|
50
|
-
def _payload_common(d):
|
|
51
|
-
return (_dig(d, "model", "session.model", "model.id", "model.display_name"),
|
|
52
|
-
_dig(d, "transcript_path", "session.transcript_path", "transcriptPath"))
|
|
53
|
-
|
|
54
|
-
|
|
55
48
|
# -------------------------------------------------------------------- hook ----
|
|
56
49
|
|
|
57
50
|
def hook():
|
|
58
|
-
"""UserPromptSubmit:
|
|
59
|
-
|
|
60
|
-
Advice goes to the user as `systemMessage`, not into Claude's context: it is
|
|
61
|
-
for the person deciding which model to use, and feeding it to the model
|
|
62
|
-
would just spend tokens telling it about its own price.
|
|
63
|
-
"""
|
|
51
|
+
"""UserPromptSubmit: kept as a no-op so existing installed hooks do not error."""
|
|
64
52
|
try:
|
|
65
|
-
|
|
66
|
-
prompt = _dig(d, "user_prompt", "prompt", default="")
|
|
67
|
-
model, transcript = _payload_common(d)
|
|
68
|
-
from .advisor import advise
|
|
69
|
-
a = advise(model=model, transcript=transcript, prompt=prompt)
|
|
70
|
-
if a:
|
|
71
|
-
print(json.dumps({"systemMessage": f"finops: {a['line']} → {a['command']}"}))
|
|
53
|
+
_stdin_json()
|
|
72
54
|
except Exception:
|
|
73
|
-
pass
|
|
55
|
+
pass
|
|
74
56
|
return 0
|
|
75
57
|
|
|
76
58
|
|
|
77
59
|
# -------------------------------------------------------------- statusline ----
|
|
78
60
|
|
|
79
61
|
def statusline():
|
|
80
|
-
"""One line, refreshed constantly
|
|
62
|
+
"""One line, refreshed constantly: model and context usage."""
|
|
81
63
|
try:
|
|
82
64
|
d = _stdin_json()
|
|
83
|
-
model, transcript = _payload_common(d)
|
|
84
65
|
pct = _dig(d, "context.percentUsed", "context.percent_used")
|
|
85
66
|
name = _dig(d, "model.display_name", "session.model", "model") or "claude"
|
|
86
67
|
bits = [str(name)]
|
|
87
68
|
if isinstance(pct, (int, float)):
|
|
88
69
|
bits.append(f"{pct:.0f}% ctx")
|
|
89
|
-
from .advisor import advise
|
|
90
|
-
a = advise(model=model, transcript=transcript)
|
|
91
|
-
if a:
|
|
92
|
-
bits.append(a["short"])
|
|
93
70
|
print(" · ".join(bits))
|
|
94
71
|
except Exception:
|
|
95
72
|
print("") # an empty statusline beats a stack trace under the prompt
|
|
@@ -126,26 +103,14 @@ def _save_settings(data):
|
|
|
126
103
|
|
|
127
104
|
|
|
128
105
|
def install_hook(remove=False):
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
if not remove:
|
|
138
|
-
hooks.append({"matcher": "", "hooks": [{"type": "command", "command": cmd,
|
|
139
|
-
"timeout": 10}]})
|
|
140
|
-
if not hooks:
|
|
141
|
-
s["hooks"].pop("UserPromptSubmit", None)
|
|
142
|
-
if not s["hooks"]:
|
|
143
|
-
s.pop("hooks")
|
|
144
|
-
_save_settings(s)
|
|
145
|
-
print(("Removed" if remove else "Installed") + f" the prompt hook in {SETTINGS}")
|
|
146
|
-
if not remove:
|
|
147
|
-
print(" It suggests a cheaper model when your own history backs one, before the turn runs.")
|
|
148
|
-
print(" Start a new Claude Code session to pick it up. Undo: claude-finops --uninstall-hook")
|
|
106
|
+
"""The prompt hook is retired: it never had a basis for its "cheaper model"
|
|
107
|
+
suggestion (that would mean repricing work that had not run yet). It is kept
|
|
108
|
+
as a no-op (see hook() above) so existing installs do not error, but nothing
|
|
109
|
+
new is installed here.
|
|
110
|
+
"""
|
|
111
|
+
print("The finops prompt hook has been retired — it made suggestions with no "
|
|
112
|
+
"basis in your data. Nothing was installed or changed.")
|
|
113
|
+
print("The statusline (claude-finops --install-statusline) still shows model and context %.")
|
|
149
114
|
|
|
150
115
|
|
|
151
116
|
def install_statusline(remove=False):
|
|
@@ -164,5 +129,5 @@ def install_statusline(remove=False):
|
|
|
164
129
|
_save_settings(s)
|
|
165
130
|
print(("Removed" if remove else "Installed") + f" the statusline in {SETTINGS}")
|
|
166
131
|
if not remove:
|
|
167
|
-
print(" Shows the model
|
|
132
|
+
print(" Shows the model and context usage percentage.")
|
|
168
133
|
print(" Undo: claude-finops --uninstall-statusline")
|
package/finops/pricing.py
CHANGED
|
@@ -5,6 +5,7 @@ produced here is an ESTIMATE. Callers must label it as such.
|
|
|
5
5
|
"""
|
|
6
6
|
import json
|
|
7
7
|
import os
|
|
8
|
+
import re
|
|
8
9
|
|
|
9
10
|
from .paths import ROOT, PRICING_PATH
|
|
10
11
|
|
|
@@ -13,6 +14,11 @@ M = 1_000_000.0
|
|
|
13
14
|
|
|
14
15
|
FREE = {"display_name": None, "tier": "free", "input": 0.0, "output": 0.0, "cache_read": 0.0,
|
|
15
16
|
"cache_write_5m": 0.0, "cache_write_1h": 0.0}
|
|
17
|
+
UNPRICED = dict(FREE, tier="unpriced")
|
|
18
|
+
|
|
19
|
+
# provider prefixes (bedrock/vertex regions), dated and @version suffixes, bedrock ":0"
|
|
20
|
+
_PREFIX = re.compile(r"^(?:[a-z]{2}(?:-[a-z]+)?\.)?anthropic\.")
|
|
21
|
+
_SUFFIX = re.compile(r"(?:-\d{8}|@\d{8}|-v\d+:\d+|:\d+)+$")
|
|
16
22
|
|
|
17
23
|
|
|
18
24
|
class Pricing:
|
|
@@ -20,21 +26,40 @@ class Pricing:
|
|
|
20
26
|
with open(path) as fh:
|
|
21
27
|
self.raw = json.load(fh)
|
|
22
28
|
self.models = self.raw.get("models", {})
|
|
23
|
-
self.
|
|
29
|
+
self.aliases = self.raw.get("aliases", {})
|
|
30
|
+
self.default = {} # kept for callers; unknown ids are UNPRICED now
|
|
24
31
|
self.updated = self.raw.get("updated")
|
|
25
32
|
self.source = self.raw.get("source")
|
|
26
33
|
|
|
27
|
-
def
|
|
34
|
+
def normalize(self, model):
|
|
35
|
+
"""Canonical price-table key for any id Claude Code may record."""
|
|
36
|
+
if not model:
|
|
37
|
+
return "unknown"
|
|
28
38
|
if model in self.models:
|
|
29
|
-
return
|
|
39
|
+
return model
|
|
40
|
+
m = _PREFIX.sub("", model)
|
|
41
|
+
m = _SUFFIX.sub("", m)
|
|
42
|
+
m = self.aliases.get(m, m)
|
|
43
|
+
if m in self.models:
|
|
44
|
+
return m
|
|
45
|
+
# a dated key in the table for an undated id: claude-haiku-4-5 -> ...-20251001
|
|
46
|
+
for k in self.models:
|
|
47
|
+
if k.startswith(m + "-") and re.fullmatch(r"\d{8}", k[len(m) + 1:]):
|
|
48
|
+
return k
|
|
49
|
+
return m
|
|
50
|
+
|
|
51
|
+
def rates(self, model):
|
|
52
|
+
m = self.normalize(model)
|
|
53
|
+
if m in self.models:
|
|
54
|
+
return self.models[m]
|
|
30
55
|
# An unlisted non-Claude model reached through Claude Code (e.g. a local Ollama or
|
|
31
56
|
# free OpenRouter model via a claude-qwen launcher) has no Anthropic price: $0.
|
|
32
|
-
if
|
|
57
|
+
if m and not m.startswith("claude") and m != "unknown":
|
|
33
58
|
return FREE
|
|
34
|
-
return
|
|
59
|
+
return UNPRICED
|
|
35
60
|
|
|
36
61
|
def is_known(self, model):
|
|
37
|
-
return model in self.models
|
|
62
|
+
return self.normalize(model) in self.models
|
|
38
63
|
|
|
39
64
|
def display_name(self, model):
|
|
40
65
|
return self.rates(model).get("display_name") or model
|
|
@@ -45,6 +70,32 @@ class Pricing:
|
|
|
45
70
|
def context_window(self, model):
|
|
46
71
|
return self.rates(model).get("context_window")
|
|
47
72
|
|
|
73
|
+
def effective_model(self, model, context_tokens=0, speed=None):
|
|
74
|
+
"""The price list that actually applied, given how much context was sent.
|
|
75
|
+
|
|
76
|
+
A request whose prompt side exceeds the model's standard context window cannot
|
|
77
|
+
have been served by the standard variant — it was the long-context one, which
|
|
78
|
+
is billed at a premium. The transcript records only the base model name, so
|
|
79
|
+
pricing off that name alone understates every long-context request. Fast mode
|
|
80
|
+
is billed at its own rate; a `[1m]` premium applies only when the table lists
|
|
81
|
+
one for this model.
|
|
82
|
+
|
|
83
|
+
Returns (model_id_to_price_with, unpriced_long_context). The flag is set when
|
|
84
|
+
the context clearly exceeded the window but no `[1m]` entry exists to price it
|
|
85
|
+
with, so callers can surface it rather than quietly bill it at the low rate.
|
|
86
|
+
"""
|
|
87
|
+
m = self.normalize(model)
|
|
88
|
+
if speed == "fast" and f"{m}[fast]" in self.models:
|
|
89
|
+
return f"{m}[fast]", False
|
|
90
|
+
r = self.models.get(m)
|
|
91
|
+
win = (r or {}).get("context_window") or 0
|
|
92
|
+
if not r or not win or not context_tokens or context_tokens <= win:
|
|
93
|
+
return m, False
|
|
94
|
+
alt = f"{m}[1m]"
|
|
95
|
+
if alt in self.models:
|
|
96
|
+
return alt, False
|
|
97
|
+
return m, True
|
|
98
|
+
|
|
48
99
|
def estimate(self, model, input_tokens=0, output_tokens=0, cache_read=0,
|
|
49
100
|
cache_write_5m=0, cache_write_1h=0):
|
|
50
101
|
"""Return estimated USD for one request."""
|
package/finops/procs.py
CHANGED
|
@@ -18,11 +18,13 @@ HOME = os.path.expanduser("~")
|
|
|
18
18
|
REG = os.path.join(HOME, ".claude", "sessions")
|
|
19
19
|
PROJECTS = os.path.join(HOME, ".claude", "projects")
|
|
20
20
|
|
|
21
|
-
ACTIONS = {
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
21
|
+
ACTIONS = {}
|
|
22
|
+
if os.name != "nt":
|
|
23
|
+
# On Windows, CTRL_C_EVENT signals the whole console process group, not just
|
|
24
|
+
# the target process, so there is no safe way to interrupt just one session.
|
|
25
|
+
ACTIONS["interrupt"] = (signal.SIGINT, "Interrupted the current turn (like pressing Esc/Ctrl-C).")
|
|
26
|
+
ACTIONS["close"] = (signal.SIGTERM, "Asked the session to exit. Resume it later with `claude --resume`.")
|
|
27
|
+
ACTIONS["kill"] = (getattr(signal, "SIGKILL", signal.SIGTERM), "Force-killed the process.")
|
|
26
28
|
|
|
27
29
|
|
|
28
30
|
def _ps():
|