cctally 1.82.0 → 1.83.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/README.md +52 -74
  3. package/bin/_cctally_alerts.py +8 -1
  4. package/bin/_cctally_cache.py +963 -149
  5. package/bin/_cctally_config.py +43 -4
  6. package/bin/_cctally_core.py +933 -759
  7. package/bin/_cctally_dashboard.py +157 -47
  8. package/bin/_cctally_dashboard_cache_report.py +13 -6
  9. package/bin/_cctally_dashboard_conversation.py +1 -0
  10. package/bin/_cctally_dashboard_envelope.py +186 -8
  11. package/bin/_cctally_dashboard_share.py +60 -20
  12. package/bin/_cctally_dashboard_sources.py +427 -128
  13. package/bin/_cctally_db.py +605 -128
  14. package/bin/_cctally_doctor.py +413 -28
  15. package/bin/_cctally_five_hour.py +12 -5
  16. package/bin/_cctally_journal.py +2050 -156
  17. package/bin/_cctally_journal_repair.py +519 -0
  18. package/bin/_cctally_milestone_history.py +142 -56
  19. package/bin/_cctally_milestones.py +179 -111
  20. package/bin/_cctally_parser.py +42 -0
  21. package/bin/_cctally_project.py +24 -18
  22. package/bin/_cctally_quota.py +139 -25
  23. package/bin/_cctally_record.py +279 -108
  24. package/bin/_cctally_rederive.py +1052 -0
  25. package/bin/_cctally_reporting.py +58 -53
  26. package/bin/_cctally_setup.py +1 -0
  27. package/bin/_cctally_source_analytics.py +4 -1
  28. package/bin/_cctally_statusline.py +11 -11
  29. package/bin/_cctally_store.py +1039 -31
  30. package/bin/_cctally_sync_week.py +17 -8
  31. package/bin/_cctally_tui.py +421 -54
  32. package/bin/_cctally_update.py +133 -8
  33. package/bin/_cctally_weekrefs.py +14 -0
  34. package/bin/_lib_aggregators.py +10 -6
  35. package/bin/_lib_cache_report.py +101 -9
  36. package/bin/_lib_codex_pools.py +82 -0
  37. package/bin/_lib_conversation_query.py +126 -33
  38. package/bin/_lib_dashboard_sources.py +126 -1
  39. package/bin/_lib_diff_kernel.py +28 -15
  40. package/bin/_lib_doctor.py +342 -4
  41. package/bin/_lib_journal.py +924 -2
  42. package/bin/_lib_jsonl.py +43 -14
  43. package/bin/_lib_pricing.py +140 -21
  44. package/bin/_lib_readme_refresh.py +401 -0
  45. package/bin/_lib_rederive.py +395 -0
  46. package/bin/_lib_share.py +58 -2
  47. package/bin/cctally +56 -8
  48. package/dashboard/static/assets/{index-DJP4gEB7.js → index-3bgCMVHb.js} +52 -52
  49. package/dashboard/static/assets/index-D27EIHEI.css +1 -0
  50. package/dashboard/static/dashboard.html +2 -2
  51. package/package.json +6 -1
  52. package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
package/bin/_lib_jsonl.py CHANGED
@@ -29,6 +29,7 @@ import sys
29
29
  from dataclasses import dataclass, field
30
30
  from typing import Any
31
31
 
32
+ from _lib_codex_pools import codex_model_scoped_quota_pool
32
33
  from _lib_source_identity import canonical_identity_from_root_key
33
34
 
34
35
 
@@ -36,6 +37,21 @@ def _eprint(*args: Any) -> None:
36
37
  print(*args, file=sys.stderr)
37
38
 
38
39
 
40
+ def _coerce_split_token(v):
41
+ """#195: drift-safe int coercion for a nested cache_creation subfield.
42
+ Returns None when the value is unusable, which makes the split UNKNOWN and
43
+ falls back to flat-token pricing. Mirrors the malformed-costUSD hardening
44
+ (#279 S3) — a drifted subfield must never abort a whole cache sync."""
45
+ if v is None:
46
+ return 0
47
+ if isinstance(v, bool) or not isinstance(v, (int, float, str)):
48
+ return None
49
+ try:
50
+ return max(0, int(v))
51
+ except (TypeError, ValueError):
52
+ return None
53
+
54
+
39
55
  @dataclass
40
56
  class UsageEntry:
41
57
  timestamp: dt.datetime
@@ -204,6 +220,33 @@ def _classify_cost_entry(obj, path_str: str):
204
220
  # loop forgets to double-check (see `sync_cache` in _cctally_cache.py).
205
221
  return None, "synthetic"
206
222
 
223
+ # #195: normalize the nested cache-write TTL breakdown into flat sibling
224
+ # keys at this single chokepoint, so wire-shape knowledge lives in exactly
225
+ # one function and both cost paths carry byte-identical dict shapes.
226
+ # Placed AFTER the reject gates so a skipped line never pays for it, and
227
+ # BEFORE the timestamp parse so it does not disturb the gating ORDER
228
+ # contract (type -> raw-timestamp -> usage -> model -> synthetic -> parse).
229
+ cc = usage.get("cache_creation")
230
+ if isinstance(cc, dict):
231
+ _h = _coerce_split_token(cc.get("ephemeral_1h_input_tokens"))
232
+ _m = _coerce_split_token(cc.get("ephemeral_5m_input_tokens"))
233
+ if _h is not None and _m is not None:
234
+ # Shallow copy: _iter_sync_entries walks this same parsed obj for
235
+ # conversation_messages rows and must not see synthesized keys.
236
+ usage = dict(usage)
237
+ usage["cache_creation_1h_input_tokens"] = _h
238
+ usage["cache_creation_5m_input_tokens"] = _m
239
+
240
+ # #413: only a retained string can be an authoritative effective tier.
241
+ # SQLite cannot bind container values, and letting a malformed object/list
242
+ # reach the cache write would reject the entire source file. Normalize every
243
+ # non-string value to the same absent/standard shape at this shared parser
244
+ # chokepoint so cache ingest and the direct fallback remain in lockstep.
245
+ speed = usage.get("speed")
246
+ if speed is not None and not isinstance(speed, str):
247
+ usage = dict(usage)
248
+ usage.pop("speed", None)
249
+
207
250
  try:
208
251
  ts = dt.datetime.fromisoformat(ts_raw.strip().replace("Z", "+00:00"))
209
252
  if ts.tzinfo is None:
@@ -598,20 +641,6 @@ def _codex_logical_limit_key(
598
641
  return _codex_canonical_json(payload)
599
642
 
600
643
 
601
- def codex_model_scoped_quota_pool(model: object) -> str | None:
602
- """Return the native model pool when Codex documents it as separate.
603
-
604
- GPT Codex Spark runs against its own allowance and does not consume the
605
- standard Codex quota. Native payloads currently reuse ``limit_id=codex``
606
- and the same slot/duration as the standard pool, so the sticky rollout
607
- model is the only retained discriminator.
608
- """
609
- if not isinstance(model, str):
610
- return None
611
- normalized = model.strip().lower()
612
- return normalized if "-codex-spark" in normalized else None
613
-
614
-
615
644
  def _codex_quota_observations(
616
645
  obj: dict[str, Any], payload: dict[str, Any], path_str: str, line_offset: int,
617
646
  source_root_key: str | None, model: str | None,
@@ -53,7 +53,7 @@ def _chip_for_model(name: str) -> str:
53
53
  # Date the embedded pricing snapshots below were last verified against
54
54
  # vendor sources. Bump whenever CLAUDE_MODEL_PRICING / CODEX_MODEL_PRICING
55
55
  # is synced. Read by `pricing-check` + the release pre-flight staleness nudge.
56
- PRICING_SNAPSHOT_DATE = "2026-07-24"
56
+ PRICING_SNAPSHOT_DATE = "2026-07-28"
57
57
  PRICING_STALENESS_DAYS = 60 # release pre-flight WARNs past this age
58
58
 
59
59
  # Canonical machine-readable pricing source (Claude values + Codex values).
@@ -99,7 +99,7 @@ PRICING_DRIFT_ALLOWLIST: list[dict] = [
99
99
 
100
100
  # Anthropic API pricing snapshot:
101
101
  # - Source: https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json
102
- # - Captured: 2026-07-24 (see PRICING_SNAPSHOT_DATE)
102
+ # - Captured/verified: 2026-07-28 (see PRICING_SNAPSHOT_DATE)
103
103
  # - Verified by maintainer against docs.claude.com/en/docs/about-claude/pricing;
104
104
  # update in PRs touching this table.
105
105
  # 2026-06-10: added claude-fable-5 ($10/$50 per MTok; 1M context, no
@@ -121,15 +121,29 @@ PRICING_DRIFT_ALLOWLIST: list[dict] = [
121
121
  # published ID/alias table). LiteLLM has no opus-5 entry yet, so this table is
122
122
  # simply ahead of it — `ahead_of_litellm` is never a drift finding, so no
123
123
  # PRICING_DRIFT_ALLOWLIST entry is needed.
124
- #
125
- # Known gap — Claude *fast mode* is not modelled. Anthropic bills fast mode at a
126
- # premium ($10/$50 per MTok on Opus 5 and Opus 4.8; $30/$150 on Opus 4.7), and
127
- # Claude Code records the tier per assistant entry at `message.usage.speed`
128
- # ("standard" | "fast"), so it IS derivable from the JSONL we already ingest —
129
- # but this table is one rate per model and `_calculate_entry_cost` ignores speed,
130
- # so a fast-mode entry is priced at the standard rate (2x undercount on Opus 5).
131
- # The Codex side already carries the shape for this
132
- # (`_calculate_codex_entry_cost(..., speed=...)` + its fast-tier multiplier).
124
+ # 2026-07-25 (#195): no VALUES changed. The snapshot date is bumped because
125
+ # cache WRITES are now priced by TTL — the 1-hour rate is DERIVED as
126
+ # `input_cost_per_token * CACHE_WRITE_1H_MULTIPLIER` (2.0) rather than stored
127
+ # per-model, so it is automatically correct for any model added later and
128
+ # cannot be silently missed the way an explicit field can be. The stored
129
+ # `cache_creation_input_token_cost` remains the 5-minute (1.25x) rate. The
130
+ # bump is also the deliberate fingerprint bust that re-arms the conversation
131
+ # rollup's materialized cost (`_arm_rollup_backfill_on_pricing_change`).
132
+ # 2026-07-28 (#413): modelled Claude fast mode from the authoritative retained
133
+ # `message.usage.speed` value. Current Opus 5/4.8 fast rows are $10/$50 per
134
+ # MTok (2x standard). Historical effective-fast Opus 4.6/4.7 rows retain
135
+ # their documented $30/$150 rate (6x standard). Current Opus 4.6 fallback
136
+ # reports `speed="standard"` and is therefore never premium-priced. Prompt
137
+ # cache multipliers stack on the fast base rate. This snapshot bump is the
138
+ # pricing-fingerprint bust for safe conversation-rollup rederivation; durable
139
+ # journaled milestones and weekly snapshots are not rewritten.
140
+ # Anthropic prices a cache WRITE by TTL: 1.25x base input for a 5-minute write,
141
+ # 2x for a 1-hour write; reads are 0.1x under both. Documented as applying
142
+ # consistently across all supported models, so the 1h rate is DERIVED from
143
+ # input_cost_per_token rather than stored per-model — a model added later
144
+ # cannot silently miss it (#195).
145
+ CACHE_WRITE_1H_MULTIPLIER = 2.0
146
+
133
147
  CLAUDE_MODEL_PRICING: dict[str, dict[str, Any]] = {
134
148
  "claude-3-5-haiku-20241022": {
135
149
  "input_cost_per_token": 8e-07,
@@ -335,6 +349,35 @@ CLAUDE_MODEL_PRICING: dict[str, dict[str, Any]] = {
335
349
  },
336
350
  }
337
351
 
352
+ # Anthropic fast-mode pricing is genuinely model-specific. Unsupported models
353
+ # deliberately have no fallback multiplier: only a retained authoritative
354
+ # `usage.speed == "fast"` row on one of these exact model IDs is premium-priced.
355
+ # The 4.6/4.7 entries are historical retention rules; current new fast requests
356
+ # are supported only on Opus 5 and Opus 4.8.
357
+ CLAUDE_FAST_MULTIPLIER_OVERRIDES: dict[str, float] = {
358
+ "claude-opus-4-6": 6.0,
359
+ "claude-opus-4-6-20260205": 6.0,
360
+ "claude-opus-4-7": 6.0,
361
+ "claude-opus-4-7-20260416": 6.0,
362
+ "claude-opus-4-8": 2.0,
363
+ "claude-opus-5": 2.0,
364
+ }
365
+
366
+
367
+ def _strip_anthropic_model_prefix(model: str) -> str:
368
+ """Return the pricing-table model ID behind supported provider aliases."""
369
+ for prefix in ("anthropic/", "anthropic."):
370
+ if model.startswith(prefix):
371
+ return model[len(prefix):]
372
+ return model
373
+
374
+
375
+ def _claude_fast_multiplier(model: str) -> float:
376
+ """Fast-tier multiplier for a retained Claude model (standard = 1.0)."""
377
+ return CLAUDE_FAST_MULTIPLIER_OVERRIDES.get(
378
+ _strip_anthropic_model_prefix(model), 1.0
379
+ )
380
+
338
381
  _unknown_model_warnings: set[str] = set()
339
382
 
340
383
  # ---------------------------------------------------------------------------
@@ -740,18 +783,82 @@ def _resolve_model_pricing(model: str, warn: bool = True) -> dict[str, Any] | No
740
783
  pricing = CLAUDE_MODEL_PRICING.get(model)
741
784
  if pricing is not None:
742
785
  return pricing
743
- for prefix in ("anthropic/", "anthropic."):
744
- if model.startswith(prefix):
745
- stripped = model[len(prefix):]
746
- pricing = CLAUDE_MODEL_PRICING.get(stripped)
747
- if pricing is not None:
748
- return pricing
786
+ stripped = _strip_anthropic_model_prefix(model)
787
+ if stripped != model:
788
+ pricing = CLAUDE_MODEL_PRICING.get(stripped)
789
+ if pricing is not None:
790
+ return pricing
749
791
  if warn and model not in _unknown_model_warnings:
750
792
  _unknown_model_warnings.add(model)
751
793
  _eprint(f"[cost] unknown model, treating cost as $0: {model}")
752
794
  return None
753
795
 
754
796
 
797
+ def claude_usage_dict(*, cache_1h_tokens, speed, input_tokens=0, output_tokens=0,
798
+ cache_creation_tokens=0, cache_read_tokens=0, **extra) -> dict:
799
+ """Canonical usage dict for `_calculate_entry_cost` (#195).
800
+
801
+ `cache_1h_tokens` and `speed` are REQUIRED keywords: a site that forgets
802
+ either raises
803
+ TypeError instead of silently pricing every 1-hour cache write at the
804
+ 5-minute rate. Pass an explicit None for a genuinely unknown or synthetic
805
+ split — that reads as a deliberate declaration at the call site.
806
+
807
+ None OMITS the key entirely, which is the exact sentinel
808
+ `_calculate_entry_cost` branches on for "price as before #195".
809
+ """
810
+ usage = {
811
+ "input_tokens": input_tokens or 0,
812
+ "output_tokens": output_tokens or 0,
813
+ "cache_creation_input_tokens": cache_creation_tokens or 0,
814
+ "cache_read_input_tokens": cache_read_tokens or 0,
815
+ }
816
+ if cache_1h_tokens is not None:
817
+ usage["cache_creation_1h_input_tokens"] = int(cache_1h_tokens)
818
+ if speed is not None:
819
+ usage["speed"] = speed
820
+ usage.update(extra)
821
+ return usage
822
+
823
+
824
+ def _cache_create_cost(pricing: dict, flat: int, h_raw, tiered) -> float:
825
+ """USD for one entry's cache-CREATION tokens, priced by TTL (#195).
826
+
827
+ `flat` is the authoritative total (`cache_creation_input_tokens`); `h_raw`
828
+ is the 1-hour portion or None when the split is unknown. The 5-minute
829
+ quantity is the REMAINDER (flat - h), never the stored 5m column, so
830
+ priced tokens always equal `cache_create_tokens` and a breakdown that
831
+ disagrees with the flat total cannot bill tokens at $0.
832
+
833
+ `tiered` is the caller's `_tiered` closure, reused verbatim for the
834
+ h == 0 path so that path is byte-identical to pre-#195 behavior.
835
+ """
836
+ if flat <= 0:
837
+ return tiered(flat, "cache_creation_input_token_cost",
838
+ "cache_creation_input_token_cost_above_200k_tokens")
839
+ h = 0 if h_raw is None else max(0, min(int(h_raw), flat))
840
+ if h == 0:
841
+ # Split unknown (pre-#195 row / no breakdown) OR genuinely all-5m.
842
+ # BOTH execute the pre-change expression VERBATIM. This early return
843
+ # is the byte-stability guarantee, not an optimization: the
844
+ # proportional form below is NOT float-identical to `_tiered` for a
845
+ # model with no above-200k cache-write rate (#195 gate P1-2).
846
+ return tiered(flat, "cache_creation_input_token_cost",
847
+ "cache_creation_input_token_cost_above_200k_tokens")
848
+
849
+ below = min(flat, TIERED_THRESHOLD)
850
+ above = max(0, flat - TIERED_THRESHOLD)
851
+ frac = h / flat
852
+ base = pricing.get("input_cost_per_token", 0.0)
853
+ r1h = base * CACHE_WRITE_1H_MULTIPLIER
854
+ r1h_200k = pricing.get("input_cost_per_token_above_200k_tokens", base) \
855
+ * CACHE_WRITE_1H_MULTIPLIER
856
+ r5m = pricing.get("cache_creation_input_token_cost", 0.0)
857
+ r5m_200k = pricing.get("cache_creation_input_token_cost_above_200k_tokens", r5m)
858
+ return ((below * frac) * r1h + (above * frac) * r1h_200k
859
+ + (below * (1 - frac)) * r5m + (above * (1 - frac)) * r5m_200k)
860
+
861
+
755
862
  def _calculate_entry_cost(
756
863
  model: str,
757
864
  usage: dict[str, Any],
@@ -789,10 +896,19 @@ def _calculate_entry_cost(
789
896
  "output_cost_per_token",
790
897
  "output_cost_per_token_above_200k_tokens",
791
898
  )
792
- cache_create_cost = _tiered(
899
+ # `flat` is passed RAW, exactly as the pre-#195 `_tiered(...)` call did.
900
+ # Coercing it here would be an unflagged behavior change in both
901
+ # directions: `int()` truncates a fractional count, and it would newly
902
+ # ACCEPT a numeric string that `_tiered`'s `tokens <= 0` used to reject.
903
+ # `_cache_create_cost` needs no coercion — every use of `flat` is
904
+ # arithmetic or comparison, and the `h_raw` side does its own `int()`
905
+ # inside the clamp (which IS load-bearing there: `min(int(h_raw), flat)`
906
+ # is what pins the 1h portion to whole tokens).
907
+ cache_create_cost = _cache_create_cost(
908
+ pricing,
793
909
  usage.get("cache_creation_input_tokens", 0),
794
- "cache_creation_input_token_cost",
795
- "cache_creation_input_token_cost_above_200k_tokens",
910
+ usage.get("cache_creation_1h_input_tokens"),
911
+ _tiered,
796
912
  )
797
913
  cache_read_cost = _tiered(
798
914
  usage.get("cache_read_input_tokens", 0),
@@ -800,7 +916,10 @@ def _calculate_entry_cost(
800
916
  "cache_read_input_token_cost_above_200k_tokens",
801
917
  )
802
918
  total = input_cost + output_cost + cache_create_cost + cache_read_cost
803
-
919
+ if usage.get("speed") == "fast":
920
+ multiplier = _claude_fast_multiplier(model)
921
+ if multiplier != 1.0:
922
+ return total * multiplier
804
923
  return total
805
924
 
806
925