cctally 1.82.0 → 1.83.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -0
- package/README.md +52 -74
- package/bin/_cctally_alerts.py +8 -1
- package/bin/_cctally_cache.py +963 -149
- package/bin/_cctally_config.py +43 -4
- package/bin/_cctally_core.py +933 -759
- package/bin/_cctally_dashboard.py +157 -47
- package/bin/_cctally_dashboard_cache_report.py +13 -6
- package/bin/_cctally_dashboard_conversation.py +1 -0
- package/bin/_cctally_dashboard_envelope.py +186 -8
- package/bin/_cctally_dashboard_share.py +60 -20
- package/bin/_cctally_dashboard_sources.py +427 -128
- package/bin/_cctally_db.py +605 -128
- package/bin/_cctally_doctor.py +413 -28
- package/bin/_cctally_five_hour.py +12 -5
- package/bin/_cctally_journal.py +2050 -156
- package/bin/_cctally_journal_repair.py +519 -0
- package/bin/_cctally_milestone_history.py +142 -56
- package/bin/_cctally_milestones.py +179 -111
- package/bin/_cctally_parser.py +42 -0
- package/bin/_cctally_project.py +24 -18
- package/bin/_cctally_quota.py +139 -25
- package/bin/_cctally_record.py +279 -108
- package/bin/_cctally_rederive.py +1052 -0
- package/bin/_cctally_reporting.py +58 -53
- package/bin/_cctally_setup.py +1 -0
- package/bin/_cctally_source_analytics.py +4 -1
- package/bin/_cctally_statusline.py +11 -11
- package/bin/_cctally_store.py +1039 -31
- package/bin/_cctally_sync_week.py +17 -8
- package/bin/_cctally_tui.py +421 -54
- package/bin/_cctally_update.py +133 -8
- package/bin/_cctally_weekrefs.py +14 -0
- package/bin/_lib_aggregators.py +10 -6
- package/bin/_lib_cache_report.py +101 -9
- package/bin/_lib_codex_pools.py +82 -0
- package/bin/_lib_conversation_query.py +126 -33
- package/bin/_lib_dashboard_sources.py +126 -1
- package/bin/_lib_diff_kernel.py +28 -15
- package/bin/_lib_doctor.py +342 -4
- package/bin/_lib_journal.py +924 -2
- package/bin/_lib_jsonl.py +43 -14
- package/bin/_lib_pricing.py +140 -21
- package/bin/_lib_readme_refresh.py +401 -0
- package/bin/_lib_rederive.py +395 -0
- package/bin/_lib_share.py +58 -2
- package/bin/cctally +56 -8
- package/dashboard/static/assets/{index-DJP4gEB7.js → index-3bgCMVHb.js} +52 -52
- package/dashboard/static/assets/index-D27EIHEI.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +6 -1
- package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
package/bin/_lib_jsonl.py
CHANGED
|
@@ -29,6 +29,7 @@ import sys
|
|
|
29
29
|
from dataclasses import dataclass, field
|
|
30
30
|
from typing import Any
|
|
31
31
|
|
|
32
|
+
from _lib_codex_pools import codex_model_scoped_quota_pool
|
|
32
33
|
from _lib_source_identity import canonical_identity_from_root_key
|
|
33
34
|
|
|
34
35
|
|
|
@@ -36,6 +37,21 @@ def _eprint(*args: Any) -> None:
|
|
|
36
37
|
print(*args, file=sys.stderr)
|
|
37
38
|
|
|
38
39
|
|
|
40
|
+
def _coerce_split_token(v):
|
|
41
|
+
"""#195: drift-safe int coercion for a nested cache_creation subfield.
|
|
42
|
+
Returns None when the value is unusable, which makes the split UNKNOWN and
|
|
43
|
+
falls back to flat-token pricing. Mirrors the malformed-costUSD hardening
|
|
44
|
+
(#279 S3) — a drifted subfield must never abort a whole cache sync."""
|
|
45
|
+
if v is None:
|
|
46
|
+
return 0
|
|
47
|
+
if isinstance(v, bool) or not isinstance(v, (int, float, str)):
|
|
48
|
+
return None
|
|
49
|
+
try:
|
|
50
|
+
return max(0, int(v))
|
|
51
|
+
except (TypeError, ValueError):
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
|
|
39
55
|
@dataclass
|
|
40
56
|
class UsageEntry:
|
|
41
57
|
timestamp: dt.datetime
|
|
@@ -204,6 +220,33 @@ def _classify_cost_entry(obj, path_str: str):
|
|
|
204
220
|
# loop forgets to double-check (see `sync_cache` in _cctally_cache.py).
|
|
205
221
|
return None, "synthetic"
|
|
206
222
|
|
|
223
|
+
# #195: normalize the nested cache-write TTL breakdown into flat sibling
|
|
224
|
+
# keys at this single chokepoint, so wire-shape knowledge lives in exactly
|
|
225
|
+
# one function and both cost paths carry byte-identical dict shapes.
|
|
226
|
+
# Placed AFTER the reject gates so a skipped line never pays for it, and
|
|
227
|
+
# BEFORE the timestamp parse so it does not disturb the gating ORDER
|
|
228
|
+
# contract (type -> raw-timestamp -> usage -> model -> synthetic -> parse).
|
|
229
|
+
cc = usage.get("cache_creation")
|
|
230
|
+
if isinstance(cc, dict):
|
|
231
|
+
_h = _coerce_split_token(cc.get("ephemeral_1h_input_tokens"))
|
|
232
|
+
_m = _coerce_split_token(cc.get("ephemeral_5m_input_tokens"))
|
|
233
|
+
if _h is not None and _m is not None:
|
|
234
|
+
# Shallow copy: _iter_sync_entries walks this same parsed obj for
|
|
235
|
+
# conversation_messages rows and must not see synthesized keys.
|
|
236
|
+
usage = dict(usage)
|
|
237
|
+
usage["cache_creation_1h_input_tokens"] = _h
|
|
238
|
+
usage["cache_creation_5m_input_tokens"] = _m
|
|
239
|
+
|
|
240
|
+
# #413: only a retained string can be an authoritative effective tier.
|
|
241
|
+
# SQLite cannot bind container values, and letting a malformed object/list
|
|
242
|
+
# reach the cache write would reject the entire source file. Normalize every
|
|
243
|
+
# non-string value to the same absent/standard shape at this shared parser
|
|
244
|
+
# chokepoint so cache ingest and the direct fallback remain in lockstep.
|
|
245
|
+
speed = usage.get("speed")
|
|
246
|
+
if speed is not None and not isinstance(speed, str):
|
|
247
|
+
usage = dict(usage)
|
|
248
|
+
usage.pop("speed", None)
|
|
249
|
+
|
|
207
250
|
try:
|
|
208
251
|
ts = dt.datetime.fromisoformat(ts_raw.strip().replace("Z", "+00:00"))
|
|
209
252
|
if ts.tzinfo is None:
|
|
@@ -598,20 +641,6 @@ def _codex_logical_limit_key(
|
|
|
598
641
|
return _codex_canonical_json(payload)
|
|
599
642
|
|
|
600
643
|
|
|
601
|
-
def codex_model_scoped_quota_pool(model: object) -> str | None:
|
|
602
|
-
"""Return the native model pool when Codex documents it as separate.
|
|
603
|
-
|
|
604
|
-
GPT Codex Spark runs against its own allowance and does not consume the
|
|
605
|
-
standard Codex quota. Native payloads currently reuse ``limit_id=codex``
|
|
606
|
-
and the same slot/duration as the standard pool, so the sticky rollout
|
|
607
|
-
model is the only retained discriminator.
|
|
608
|
-
"""
|
|
609
|
-
if not isinstance(model, str):
|
|
610
|
-
return None
|
|
611
|
-
normalized = model.strip().lower()
|
|
612
|
-
return normalized if "-codex-spark" in normalized else None
|
|
613
|
-
|
|
614
|
-
|
|
615
644
|
def _codex_quota_observations(
|
|
616
645
|
obj: dict[str, Any], payload: dict[str, Any], path_str: str, line_offset: int,
|
|
617
646
|
source_root_key: str | None, model: str | None,
|
package/bin/_lib_pricing.py
CHANGED
|
@@ -53,7 +53,7 @@ def _chip_for_model(name: str) -> str:
|
|
|
53
53
|
# Date the embedded pricing snapshots below were last verified against
|
|
54
54
|
# vendor sources. Bump whenever CLAUDE_MODEL_PRICING / CODEX_MODEL_PRICING
|
|
55
55
|
# is synced. Read by `pricing-check` + the release pre-flight staleness nudge.
|
|
56
|
-
PRICING_SNAPSHOT_DATE = "2026-07-
|
|
56
|
+
PRICING_SNAPSHOT_DATE = "2026-07-28"
|
|
57
57
|
PRICING_STALENESS_DAYS = 60 # release pre-flight WARNs past this age
|
|
58
58
|
|
|
59
59
|
# Canonical machine-readable pricing source (Claude values + Codex values).
|
|
@@ -99,7 +99,7 @@ PRICING_DRIFT_ALLOWLIST: list[dict] = [
|
|
|
99
99
|
|
|
100
100
|
# Anthropic API pricing snapshot:
|
|
101
101
|
# - Source: https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json
|
|
102
|
-
# - Captured: 2026-07-
|
|
102
|
+
# - Captured/verified: 2026-07-28 (see PRICING_SNAPSHOT_DATE)
|
|
103
103
|
# - Verified by maintainer against docs.claude.com/en/docs/about-claude/pricing;
|
|
104
104
|
# update in PRs touching this table.
|
|
105
105
|
# 2026-06-10: added claude-fable-5 ($10/$50 per MTok; 1M context, no
|
|
@@ -121,15 +121,29 @@ PRICING_DRIFT_ALLOWLIST: list[dict] = [
|
|
|
121
121
|
# published ID/alias table). LiteLLM has no opus-5 entry yet, so this table is
|
|
122
122
|
# simply ahead of it — `ahead_of_litellm` is never a drift finding, so no
|
|
123
123
|
# PRICING_DRIFT_ALLOWLIST entry is needed.
|
|
124
|
-
#
|
|
125
|
-
#
|
|
126
|
-
#
|
|
127
|
-
#
|
|
128
|
-
#
|
|
129
|
-
#
|
|
130
|
-
#
|
|
131
|
-
#
|
|
132
|
-
# (
|
|
124
|
+
# 2026-07-25 (#195): no VALUES changed. The snapshot date is bumped because
|
|
125
|
+
# cache WRITES are now priced by TTL — the 1-hour rate is DERIVED as
|
|
126
|
+
# `input_cost_per_token * CACHE_WRITE_1H_MULTIPLIER` (2.0) rather than stored
|
|
127
|
+
# per-model, so it is automatically correct for any model added later and
|
|
128
|
+
# cannot be silently missed the way an explicit field can be. The stored
|
|
129
|
+
# `cache_creation_input_token_cost` remains the 5-minute (1.25x) rate. The
|
|
130
|
+
# bump is also the deliberate fingerprint bust that re-arms the conversation
|
|
131
|
+
# rollup's materialized cost (`_arm_rollup_backfill_on_pricing_change`).
|
|
132
|
+
# 2026-07-28 (#413): modelled Claude fast mode from the authoritative retained
|
|
133
|
+
# `message.usage.speed` value. Current Opus 5/4.8 fast rows are $10/$50 per
|
|
134
|
+
# MTok (2x standard). Historical effective-fast Opus 4.6/4.7 rows retain
|
|
135
|
+
# their documented $30/$150 rate (6x standard). Current Opus 4.6 fallback
|
|
136
|
+
# reports `speed="standard"` and is therefore never premium-priced. Prompt
|
|
137
|
+
# cache multipliers stack on the fast base rate. This snapshot bump is the
|
|
138
|
+
# pricing-fingerprint bust for safe conversation-rollup rederivation; durable
|
|
139
|
+
# journaled milestones and weekly snapshots are not rewritten.
|
|
140
|
+
# Anthropic prices a cache WRITE by TTL: 1.25x base input for a 5-minute write,
|
|
141
|
+
# 2x for a 1-hour write; reads are 0.1x under both. Documented as applying
|
|
142
|
+
# consistently across all supported models, so the 1h rate is DERIVED from
|
|
143
|
+
# input_cost_per_token rather than stored per-model — a model added later
|
|
144
|
+
# cannot silently miss it (#195).
|
|
145
|
+
CACHE_WRITE_1H_MULTIPLIER = 2.0
|
|
146
|
+
|
|
133
147
|
CLAUDE_MODEL_PRICING: dict[str, dict[str, Any]] = {
|
|
134
148
|
"claude-3-5-haiku-20241022": {
|
|
135
149
|
"input_cost_per_token": 8e-07,
|
|
@@ -335,6 +349,35 @@ CLAUDE_MODEL_PRICING: dict[str, dict[str, Any]] = {
|
|
|
335
349
|
},
|
|
336
350
|
}
|
|
337
351
|
|
|
352
|
+
# Anthropic fast-mode pricing is genuinely model-specific. Unsupported models
|
|
353
|
+
# deliberately have no fallback multiplier: only a retained authoritative
|
|
354
|
+
# `usage.speed == "fast"` row on one of these exact model IDs is premium-priced.
|
|
355
|
+
# The 4.6/4.7 entries are historical retention rules; current new fast requests
|
|
356
|
+
# are supported only on Opus 5 and Opus 4.8.
|
|
357
|
+
CLAUDE_FAST_MULTIPLIER_OVERRIDES: dict[str, float] = {
|
|
358
|
+
"claude-opus-4-6": 6.0,
|
|
359
|
+
"claude-opus-4-6-20260205": 6.0,
|
|
360
|
+
"claude-opus-4-7": 6.0,
|
|
361
|
+
"claude-opus-4-7-20260416": 6.0,
|
|
362
|
+
"claude-opus-4-8": 2.0,
|
|
363
|
+
"claude-opus-5": 2.0,
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _strip_anthropic_model_prefix(model: str) -> str:
|
|
368
|
+
"""Return the pricing-table model ID behind supported provider aliases."""
|
|
369
|
+
for prefix in ("anthropic/", "anthropic."):
|
|
370
|
+
if model.startswith(prefix):
|
|
371
|
+
return model[len(prefix):]
|
|
372
|
+
return model
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _claude_fast_multiplier(model: str) -> float:
|
|
376
|
+
"""Fast-tier multiplier for a retained Claude model (standard = 1.0)."""
|
|
377
|
+
return CLAUDE_FAST_MULTIPLIER_OVERRIDES.get(
|
|
378
|
+
_strip_anthropic_model_prefix(model), 1.0
|
|
379
|
+
)
|
|
380
|
+
|
|
338
381
|
_unknown_model_warnings: set[str] = set()
|
|
339
382
|
|
|
340
383
|
# ---------------------------------------------------------------------------
|
|
@@ -740,18 +783,82 @@ def _resolve_model_pricing(model: str, warn: bool = True) -> dict[str, Any] | No
|
|
|
740
783
|
pricing = CLAUDE_MODEL_PRICING.get(model)
|
|
741
784
|
if pricing is not None:
|
|
742
785
|
return pricing
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
return pricing
|
|
786
|
+
stripped = _strip_anthropic_model_prefix(model)
|
|
787
|
+
if stripped != model:
|
|
788
|
+
pricing = CLAUDE_MODEL_PRICING.get(stripped)
|
|
789
|
+
if pricing is not None:
|
|
790
|
+
return pricing
|
|
749
791
|
if warn and model not in _unknown_model_warnings:
|
|
750
792
|
_unknown_model_warnings.add(model)
|
|
751
793
|
_eprint(f"[cost] unknown model, treating cost as $0: {model}")
|
|
752
794
|
return None
|
|
753
795
|
|
|
754
796
|
|
|
797
|
+
def claude_usage_dict(*, cache_1h_tokens, speed, input_tokens=0, output_tokens=0,
|
|
798
|
+
cache_creation_tokens=0, cache_read_tokens=0, **extra) -> dict:
|
|
799
|
+
"""Canonical usage dict for `_calculate_entry_cost` (#195).
|
|
800
|
+
|
|
801
|
+
`cache_1h_tokens` and `speed` are REQUIRED keywords: a site that forgets
|
|
802
|
+
either raises
|
|
803
|
+
TypeError instead of silently pricing every 1-hour cache write at the
|
|
804
|
+
5-minute rate. Pass an explicit None for a genuinely unknown or synthetic
|
|
805
|
+
split — that reads as a deliberate declaration at the call site.
|
|
806
|
+
|
|
807
|
+
None OMITS the key entirely, which is the exact sentinel
|
|
808
|
+
`_calculate_entry_cost` branches on for "price as before #195".
|
|
809
|
+
"""
|
|
810
|
+
usage = {
|
|
811
|
+
"input_tokens": input_tokens or 0,
|
|
812
|
+
"output_tokens": output_tokens or 0,
|
|
813
|
+
"cache_creation_input_tokens": cache_creation_tokens or 0,
|
|
814
|
+
"cache_read_input_tokens": cache_read_tokens or 0,
|
|
815
|
+
}
|
|
816
|
+
if cache_1h_tokens is not None:
|
|
817
|
+
usage["cache_creation_1h_input_tokens"] = int(cache_1h_tokens)
|
|
818
|
+
if speed is not None:
|
|
819
|
+
usage["speed"] = speed
|
|
820
|
+
usage.update(extra)
|
|
821
|
+
return usage
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def _cache_create_cost(pricing: dict, flat: int, h_raw, tiered) -> float:
|
|
825
|
+
"""USD for one entry's cache-CREATION tokens, priced by TTL (#195).
|
|
826
|
+
|
|
827
|
+
`flat` is the authoritative total (`cache_creation_input_tokens`); `h_raw`
|
|
828
|
+
is the 1-hour portion or None when the split is unknown. The 5-minute
|
|
829
|
+
quantity is the REMAINDER (flat - h), never the stored 5m column, so
|
|
830
|
+
priced tokens always equal `cache_create_tokens` and a breakdown that
|
|
831
|
+
disagrees with the flat total cannot bill tokens at $0.
|
|
832
|
+
|
|
833
|
+
`tiered` is the caller's `_tiered` closure, reused verbatim for the
|
|
834
|
+
h == 0 path so that path is byte-identical to pre-#195 behavior.
|
|
835
|
+
"""
|
|
836
|
+
if flat <= 0:
|
|
837
|
+
return tiered(flat, "cache_creation_input_token_cost",
|
|
838
|
+
"cache_creation_input_token_cost_above_200k_tokens")
|
|
839
|
+
h = 0 if h_raw is None else max(0, min(int(h_raw), flat))
|
|
840
|
+
if h == 0:
|
|
841
|
+
# Split unknown (pre-#195 row / no breakdown) OR genuinely all-5m.
|
|
842
|
+
# BOTH execute the pre-change expression VERBATIM. This early return
|
|
843
|
+
# is the byte-stability guarantee, not an optimization: the
|
|
844
|
+
# proportional form below is NOT float-identical to `_tiered` for a
|
|
845
|
+
# model with no above-200k cache-write rate (#195 gate P1-2).
|
|
846
|
+
return tiered(flat, "cache_creation_input_token_cost",
|
|
847
|
+
"cache_creation_input_token_cost_above_200k_tokens")
|
|
848
|
+
|
|
849
|
+
below = min(flat, TIERED_THRESHOLD)
|
|
850
|
+
above = max(0, flat - TIERED_THRESHOLD)
|
|
851
|
+
frac = h / flat
|
|
852
|
+
base = pricing.get("input_cost_per_token", 0.0)
|
|
853
|
+
r1h = base * CACHE_WRITE_1H_MULTIPLIER
|
|
854
|
+
r1h_200k = pricing.get("input_cost_per_token_above_200k_tokens", base) \
|
|
855
|
+
* CACHE_WRITE_1H_MULTIPLIER
|
|
856
|
+
r5m = pricing.get("cache_creation_input_token_cost", 0.0)
|
|
857
|
+
r5m_200k = pricing.get("cache_creation_input_token_cost_above_200k_tokens", r5m)
|
|
858
|
+
return ((below * frac) * r1h + (above * frac) * r1h_200k
|
|
859
|
+
+ (below * (1 - frac)) * r5m + (above * (1 - frac)) * r5m_200k)
|
|
860
|
+
|
|
861
|
+
|
|
755
862
|
def _calculate_entry_cost(
|
|
756
863
|
model: str,
|
|
757
864
|
usage: dict[str, Any],
|
|
@@ -789,10 +896,19 @@ def _calculate_entry_cost(
|
|
|
789
896
|
"output_cost_per_token",
|
|
790
897
|
"output_cost_per_token_above_200k_tokens",
|
|
791
898
|
)
|
|
792
|
-
|
|
899
|
+
# `flat` is passed RAW, exactly as the pre-#195 `_tiered(...)` call did.
|
|
900
|
+
# Coercing it here would be an unflagged behavior change in both
|
|
901
|
+
# directions: `int()` truncates a fractional count, and it would newly
|
|
902
|
+
# ACCEPT a numeric string that `_tiered`'s `tokens <= 0` used to reject.
|
|
903
|
+
# `_cache_create_cost` needs no coercion — every use of `flat` is
|
|
904
|
+
# arithmetic or comparison, and the `h_raw` side does its own `int()`
|
|
905
|
+
# inside the clamp (which IS load-bearing there: `min(int(h_raw), flat)`
|
|
906
|
+
# is what pins the 1h portion to whole tokens).
|
|
907
|
+
cache_create_cost = _cache_create_cost(
|
|
908
|
+
pricing,
|
|
793
909
|
usage.get("cache_creation_input_tokens", 0),
|
|
794
|
-
"
|
|
795
|
-
|
|
910
|
+
usage.get("cache_creation_1h_input_tokens"),
|
|
911
|
+
_tiered,
|
|
796
912
|
)
|
|
797
913
|
cache_read_cost = _tiered(
|
|
798
914
|
usage.get("cache_read_input_tokens", 0),
|
|
@@ -800,7 +916,10 @@ def _calculate_entry_cost(
|
|
|
800
916
|
"cache_read_input_token_cost_above_200k_tokens",
|
|
801
917
|
)
|
|
802
918
|
total = input_cost + output_cost + cache_create_cost + cache_read_cost
|
|
803
|
-
|
|
919
|
+
if usage.get("speed") == "fast":
|
|
920
|
+
multiplier = _claude_fast_multiplier(model)
|
|
921
|
+
if multiplier != 1.0:
|
|
922
|
+
return total * multiplier
|
|
804
923
|
return total
|
|
805
924
|
|
|
806
925
|
|