cctally 1.64.0 → 1.66.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +50 -0
  2. package/bin/_cctally_alerts.py +1 -1
  3. package/bin/_cctally_cache.py +262 -40
  4. package/bin/_cctally_cache_report.py +7 -5
  5. package/bin/_cctally_core.py +49 -14
  6. package/bin/_cctally_dashboard.py +827 -4911
  7. package/bin/_cctally_dashboard_cache_report.py +714 -0
  8. package/bin/_cctally_dashboard_conversation.py +771 -0
  9. package/bin/_cctally_dashboard_envelope.py +1520 -0
  10. package/bin/_cctally_dashboard_share.py +2012 -0
  11. package/bin/_cctally_db.py +127 -5
  12. package/bin/_cctally_doctor.py +104 -3
  13. package/bin/_cctally_five_hour.py +2 -1
  14. package/bin/_cctally_forecast.py +78 -135
  15. package/bin/_cctally_parser.py +919 -784
  16. package/bin/_cctally_percent_breakdown.py +1 -1
  17. package/bin/_cctally_pricing_check.py +39 -0
  18. package/bin/_cctally_project.py +8 -5
  19. package/bin/_cctally_record.py +279 -274
  20. package/bin/_cctally_reporting.py +1 -1
  21. package/bin/_cctally_sync_week.py +1 -1
  22. package/bin/_cctally_telemetry.py +3 -5
  23. package/bin/_cctally_tui.py +487 -254
  24. package/bin/_cctally_update.py +5 -0
  25. package/bin/_lib_alert_dispatch.py +1 -1
  26. package/bin/_lib_blocks.py +6 -1
  27. package/bin/_lib_conversation_query.py +349 -36
  28. package/bin/_lib_credit.py +133 -0
  29. package/bin/_lib_doctor.py +150 -1
  30. package/bin/_lib_five_hour.py +34 -0
  31. package/bin/_lib_forecast.py +144 -0
  32. package/bin/_lib_json_envelope.py +38 -0
  33. package/bin/_lib_jsonl.py +115 -73
  34. package/bin/_lib_log.py +96 -0
  35. package/bin/_lib_perf.py +180 -0
  36. package/bin/_lib_pricing.py +47 -1
  37. package/bin/_lib_pricing_check.py +25 -3
  38. package/bin/_lib_record.py +179 -0
  39. package/bin/_lib_render.py +18 -10
  40. package/bin/_lib_snapshot_cache.py +61 -0
  41. package/bin/_lib_transcript_access.py +23 -0
  42. package/bin/cctally +51 -7
  43. package/bin/cctally-dashboard +2 -2
  44. package/bin/cctally-statusline +1 -0
  45. package/bin/cctally-tui +2 -2
  46. package/dashboard/static/assets/index-CJTUCpKt.js +80 -0
  47. package/dashboard/static/assets/index-DQWNrIqu.css +1 -0
  48. package/dashboard/static/dashboard.html +2 -2
  49. package/package.json +11 -1
  50. package/dashboard/static/assets/index-0jzYm75p.css +0 -1
  51. package/dashboard/static/assets/index-D_Ylyqsf.js +0 -80
package/CHANGELOG.md CHANGED
@@ -5,6 +5,56 @@ based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [1.66.0] - 2026-07-10
9
+
10
+ ### Added
11
+ - Reporting `--json` output now carries a `schemaVersion: 1` key (rendered first) so downstream consumers can version-gate against the payload shape. It is added to every reporting surface that lacked it — `daily`/`daily --instances`/`monthly`/`weekly`/`session`, the Codex `codex-daily`/`codex-monthly`/`codex-weekly`/`codex-session` reports, `blocks`, `forecast`, `report` (`dollar-per-percent`), `project`, `range-cost`, `cache-report`, `percent-breakdown`, `sync-week`, `telemetry`, and the `budget set`/`unset` actions — including their empty and error forms. The already-versioned surfaces (`five-hour-blocks`, `five-hour-breakdown`, `budget` status, `record-credit`, `pricing-check`) are unchanged, and the shipped snake-case `schema_version` surfaces (`diff`, `doctor`, `db status`, `refresh-usage`, `setup`) keep their existing key. Existing keys are unchanged and consumers should tolerate unknown keys (#279).
12
+ - The cache sync now counts JSONL lines it can't parse (malformed) or has to drift-skip (usage/model/timestamp no longer parse), per vendor, and surfaces them on the `cache-sync` done lines, the hook-tick log line, and a new `cctally doctor` "Ingest parse health" check — so a Claude Code / Codex session-format change that would silently affect your cost numbers becomes visible instead of vanishing (#279).
13
+ - `CCTALLY_DEBUG=1` now makes `cctally` print the full Python traceback on stderr when a command crashes (default output is unchanged), and `CCTALLY_DEBUG_LOG=<path>` additionally appends the backend log to a file — so a bug is reportable without editing source (#279).
14
+ - The dashboard now logs server errors (HTTP 500) to the terminal running it, instead of swallowing them silently; routine client errors and the per-request access log stay quiet (#279).
15
+ - `cctally doctor` gains a SQLite integrity check (`PRAGMA quick_check`, run only from the CLI where it's affordable — FAILs on stats.db corruption with safe recovery guidance, WARNs on the re-derivable cache.db) and an informational lock-state check that reports whether a sync lock is currently held (#279).
16
+ - `CCTALLY_PERF_TRACE=1 cctally cache-sync` now traces both the Claude and Codex ingest under one shared `cache-sync` phase tree, with the Codex sync carrying the same flock/discover/walk timing seams as the Claude sync (#279).
17
+ - New cache migration `020_session_entries_physical_unique` adds a `UNIQUE(source_path, line_offset)` backstop to the Claude session cache (matching the Codex table) so an offset-bookkeeping regression can never silently double-count your cost data — a collision fails loudly and rolls back that file instead. It dedups any pre-existing duplicates (keeping the first-ingested row) and runs automatically; `cache.db` is re-derivable, so `cache-sync --rebuild` remains the escape hatch (#279).
18
+ - `cctally pricing-check` now flags embedded-pricing allowlist suppressions that no longer correspond to a real LiteLLM divergence (`staleSuppressions`) or that are past a new optional `expires` cutover date (`expiredSuppressions`) — two additive `--json` arrays and an exit-1 actionable finding — and the weekly pricing-freshness workflow files them on its auto-tracked issue, so a deliberate pricing override can't silently ossify past its stated cutover (#279).
19
+ - Codex pricing now covers the `gpt-5.6` model family (`gpt-5.6`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`), so cost for those models is computed from their own published rates instead of falling back to `gpt-5` pricing.
20
+
21
+ ### Changed
22
+ - `cctally project` and `cctally forecast` now exit `2` (was `1`) on their own usage/validation errors — a bad `--weeks`, `--weeks` combined with `--since`/`--until`, `--since > --until` or an unparseable date for `project`; `--json` together with `--status-line` or a bad `--targets` for `forecast` — matching the rest of the cctally-native command family (`diff`, `budget`, `five-hour-*`, `pricing-check`, `doctor`). The ccusage-parity commands (`daily`/`monthly`/`weekly`/`session`/`blocks`/`codex-*`) keep their exit `1` on bad dates for upstream parity. The exit-code convention is now documented in `docs/cli-contract.md` (#279).
23
+ - Release tooling (maintainer): the npm-publish poll now hard-fails on timeout with `--resume` guidance instead of silently reporting success; `--npm-soft-timeout` restores the old exit-0 behavior and marks the final line "published (npm pending verification)".
24
+ - Internal: the record write path (resets-at plausibility, weekly credit / reset-to-zero debounce, 5h in-place credit, the reset-aware HWM clamp + monotonic HWM files, 5h milestone ranges, and the projected-pace alert crossings) and the forecast projection now route their decision logic through pure, unit-tested kernels (`_lib_record.py`, `_lib_credit.py`, `_lib_forecast.py`, `_lib_five_hour.py`); no behavior change (#279).
25
+ - Internal: CLI-architecture consistency pass — `build_parser()` is now a loop over an ordered per-command registration table of builder functions (every subcommand's `--help` output is byte-for-byte unchanged, verified by a recursive sweep across the flat commands, the `claude`/`codex` subgroups, and the hidden commands); the four `_iso_z` UTC-Z serializers collapse to one canonical helper (doctor keeps its deliberately divergent copy); the `_eprint` / `_load_lib` / `_ensure_sibling_loaded` per-module bootstrap copies are normalized and pinned by AST drift-guard tests; and the three deviant thin bash wrappers are aligned to the shared template. No behavior change (#279).
26
+ - Internal: the dashboard server module (`_cctally_dashboard.py`, ~10.5K lines) is split into four consumer-only sibling modules (share, envelope, cache-report, conversation) and its HTTP request routing is now table-driven; every endpoint's response bytes and status codes are byte-identical, verified against a real data corpus (#279).
27
+ - `cctally report --json`'s `generatedAt` timestamp now routes through the same as-of clock as the rest of the report (unchanged in normal use — it still reflects wall-clock when no override is set), making the JSON output reproducible under a pinned clock (#279).
28
+ - Internal: test-infrastructure hardening — every migration in both registries now ships a per-migration golden with an enforced registry-completeness guard and a documented handler-idempotency contract; a new ancient-DB→head chain test covers migration interaction (the recompute gate + the FTS search-split→drop→title sequence); `pytest-timeout` is wired as an optional-by-detection dev dependency with a deliberate-hang canary and `timeout-minutes` on every CI job; the npm shim/postinstall harnesses now assert empty-output emptiness instead of skipping it; and `daily`, `monthly`, and `report` gain golden test harnesses (#279).
29
+
30
+ ### Fixed
31
+ - `cctally setup` now installs the `cctally-budget` symlink (it existed but was never linked); a drift-guard test keeps the wrapper list and the installer in sync.
32
+ - `CCTALLY_ALLOW_PROD_MIGRATION=0`, `CCTALLY_DISABLE_DEV_AUTODETECT=0`, `CCTALLY_DEBUG=0`, and `CCTALLY_DISABLE_UPDATE_CHECK=0` now mean "disabled" — previously any value, including `0`, enabled these flags.
33
+ - A corrupt `stats.db` now produces a one-line diagnosis with recovery guidance and exit 2 instead of a raw traceback.
34
+ - Piping output to a closed reader (`cctally daily | head`) now exits 0 quietly instead of `Error: [Errno 32] Broken pipe`.
35
+ - The dashboard share endpoints now cap request bodies at 64 KiB and the dashboard server sets a 60s socket timeout (slow-loris hardening); server-sent-event streams treat a send timeout as a normal client disconnect.
36
+ - The dashboard `/api/data` endpoint now returns a JSON `500 {"error": "internal error"}` on an internal error instead of dropping the connection with no response, and a server-sent-event stream that fails mid-flight is now logged and closed cleanly instead of dying silently (#279).
37
+ - `cache.db` now opens with `PRAGMA synchronous=NORMAL` (a fully re-derivable database; fewer fsyncs during ingest).
38
+ - Codex resume tracking now persists the ingest iterator's actual dedup watermark (the cumulative token counter it compares against) rather than a reconstructed sum of per-turn deltas, closing a latent double-count/skip on rollouts whose cumulative and per-turn token accounting diverge; healthy sessions are unaffected and need no re-ingest (#279).
39
+ - The direct-JSONL fallback (used when `cache.db` can't be opened) now parses with the same single implementation as the cache path so the two can't drift, and a malformed `costUSD` value in a session line no longer aborts the read — it degrades to the token-derived cost like every other cost path (#279).
40
+
41
+ ## [1.65.0] - 2026-07-09
42
+
43
+ ### Added
44
+ - Opt-in backend performance instrumentation: set `CCTALLY_PERF_TRACE=1` to print a phase-timing trace to stderr on `cache-sync` (and the hidden `tui --render-once` build), and read live backend timings + cache state from the dashboard's loopback-only `/api/debug/backend` endpoint. The trace is off by default and never changes command output — it goes to stderr only, so stdout stays byte-identical. New [docs/backend-performance.md](docs/backend-performance.md) documents the read model, hot paths, and invariants.
45
+ - Reproducible backend benchmark suite (`bin/cctally-bench` + `bin/build-bench-fixtures.py`, a dev/maintainer tool — never shipped): in-process timings for the dashboard snapshot (cold/warm/idle), cache ingest (no-op/delta), the conversation rail/search/find/payload/outline, and the warm reconcile paths, measured against a deterministic synthetic fixture, with a committed advisory baseline and an opt-in `--compare`/`--gate` (never a CI gate) that guard the recent backend performance wins from silent regression. See [bench/README.md](bench/README.md).
46
+ - The opt-in `CCTALLY_PERF_TRACE` backend trace now breaks whole-session conversation assembly into named sub-phases (read, dedup, build, correlate, fold, classify, cost, finalize) under the reader's detail/paginate phases, and the loopback `/api/debug/backend` diagnostic can now surface a conversation-open trace (list/search/outline/find/export/prompts/detail) as well as the data-snapshot build — so a slow transcript open is attributable to a specific stage on a live dashboard; off by default, no command output changes.
47
+ - `cctally-bench --assembly-scan` (dev/maintainer tool): sweeps the cost of assembling a whole conversation across a size ladder against its own isolated fixture and baseline (`bench/baselines/assembly.json`), the measurement behind the recorded decision to defer materializing pre-rendered conversation turns — assembly stays well under a 100 ms visibility budget for all but pathologically large (multi-thousand-turn) sessions, and the reader already caps payload bytes. See [docs/backend-performance.md](docs/backend-performance.md) §5.
48
+ - The conversation browse rail and cross-session search can now be filtered by **model family**: a new "Model" section in the filter popover lists the Opus / Sonnet / Haiku / Fable families that actually appear (each with a session count), and selecting one or more scopes both browsing and search to sessions that used that model — anywhere in the session, including its subagents — with removable chips and "Clear all" just like the other filters (#278).
49
+
50
+ ### Changed
51
+ - Conversation reader no longer re-processes an open conversation on every background refresh — it now recomputes only when the conversation actually grows, cutting idle CPU/battery use for anyone leaving a transcript open (#278).
52
+ - The dashboard now opens instantly on a heavy-history instance: it binds and paints the current-week and forecast panels almost immediately instead of waiting for the full ~2s data aggregation, and the heavier panels (sessions, projects, weekly, monthly, blocks, daily, trend, cache report) hydrate over the live stream showing a brief loading skeleton (respecting reduced-motion) rather than flashing a misleading "no data" / "restart the dashboard" empty state that then fills in (#278).
53
+ - A first-run or long-gap dashboard now fills in progressively as session history is ingested, instead of sitting blank and then snapping to fully-loaded in one jump; a warm returning session is unaffected, and a slow or backgrounded browser tab always converges to the latest data instead of getting stuck on a stale frame (#278).
54
+
55
+ ### Fixed
56
+ - The conversation filter popover now retries once if its filter options (the Project and Model lists) fail to load on a transient hiccup, instead of coming up empty and staying that way until it is reopened (#278).
57
+
8
58
  ## [1.64.0] - 2026-07-07
9
59
 
10
60
  ### Added
@@ -44,7 +44,6 @@ from __future__ import annotations
44
44
 
45
45
  import argparse
46
46
  import datetime as dt
47
- import importlib.util as _ilu
48
47
  import os
49
48
  import pathlib
50
49
  import shutil
@@ -58,6 +57,7 @@ def _load_lib(name: str):
58
57
  cached = sys.modules.get(name)
59
58
  if cached is not None:
60
59
  return cached
60
+ import importlib.util as _ilu
61
61
  p = pathlib.Path(__file__).resolve().parent / f"{name}.py"
62
62
  spec = _ilu.spec_from_file_location(name, p)
63
63
  mod = _ilu.module_from_spec(spec)
@@ -100,7 +100,6 @@ from __future__ import annotations
100
100
  import argparse
101
101
  import datetime as dt
102
102
  import fcntl
103
- import importlib.util as _ilu
104
103
  import json
105
104
  import os
106
105
  import pathlib
@@ -151,6 +150,7 @@ def _load_lib(name: str):
151
150
  cached = sys.modules.get(name)
152
151
  if cached is not None:
153
152
  return cached
153
+ import importlib.util as _ilu
154
154
  p = pathlib.Path(__file__).resolve().parent / f"{name}.py"
155
155
  spec = _ilu.spec_from_file_location(name, p)
156
156
  mod = _ilu.module_from_spec(spec)
@@ -176,6 +176,12 @@ _should_replace = _lib_jsonl._should_replace
176
176
  _lib_conversation = _load_lib("_lib_conversation")
177
177
  _iter_message_rows = _lib_conversation.iter_message_rows
178
178
 
179
+ # Opt-in backend phase-instrumentation collector (issue #276, Session A). Pure
180
+ # stdlib leaf; near-noop when CCTALLY_PERF_TRACE is unset (phase() returns a
181
+ # shared no-op singleton), so the sync_cache seam wraps below cost nothing on
182
+ # the default path.
183
+ _perf = _load_lib("_lib_perf")
184
+
179
185
  # Shared by the fused per-file walk AND backfill_conversation_messages so the
180
186
  # column list, placeholders, and tuple order live in ONE place — a column
181
187
  # add/reorder can't silently desync the two ingest paths (which would land
@@ -223,7 +229,7 @@ def _conv_row_tuple(m, path_str):
223
229
  )
224
230
 
225
231
 
226
- def _iter_sync_entries(fh, path_str):
232
+ def _iter_sync_entries(fh, path_str, stats: "IngestStats | None" = None):
227
233
  """Fused single-pass sync walker (#138). Yields
228
234
  ``(byte_offset, cost_or_None, msgrow_or_None, aititle_or_None)`` for each
229
235
  JSONL line from ``fh``'s current position that produces a cost entry, a
@@ -263,11 +269,25 @@ def _iter_sync_entries(fh, path_str):
263
269
  stripped = line.strip()
264
270
  if not stripped:
265
271
  continue
272
+ # #279 S2 F1: passive parse-health counters over the new-byte span.
273
+ # lines_seen counts non-blank lines (malformed included).
274
+ if stats is not None:
275
+ stats.lines_seen += 1
266
276
  try:
267
277
  obj = json.loads(stripped)
268
278
  except json.JSONDecodeError:
279
+ if stats is not None:
280
+ stats.lines_malformed += 1
269
281
  continue
270
282
  cost = _lib_jsonl.parse_cost_entry(obj, path_str)
283
+ if cost is None and stats is not None:
284
+ # Assistant-typed line rejected for a NON-deliberate reason
285
+ # (schema-drift tripwire; <synthetic>/non-assistant are normal).
286
+ reason = _lib_jsonl.assistant_skip_reason(obj)
287
+ if reason is not None:
288
+ stats.assistant_lines_skipped += 1
289
+ stats.skip_reasons[reason] = \
290
+ stats.skip_reasons.get(reason, 0) + 1
271
291
  mrow = _lib_conversation.parse_message_row(obj, offset)
272
292
  ai = _lib_conversation.parse_ai_title(obj, offset)
273
293
  if cost is not None or mrow is not None or ai is not None:
@@ -296,6 +316,77 @@ clear_conversation_messages = _cctally_db_sib.clear_conversation_messages
296
316
  _set_cache_meta = _cctally_db_sib._set_cache_meta
297
317
 
298
318
 
319
+ _PARSE_HEALTH_SCHEMA = 1
320
+
321
+
322
+ def _update_parse_health_meta(
323
+ conn: sqlite3.Connection,
324
+ key: str,
325
+ *,
326
+ lines_seen: int,
327
+ lines_malformed: int,
328
+ lines_skipped: int,
329
+ skip_reasons: dict,
330
+ rebuild: bool,
331
+ ) -> None:
332
+ """Anomaly-delta-gated rolling parse-health record (#279 S2 F1 /
333
+ Codex P1-2). Writes ONLY when (a) this sync's malformed+skipped delta
334
+ is nonzero, (b) rebuild=True (baseline reset — write fresh from that
335
+ walk's counters), or (c) the key is absent (first adoption). Healthy
336
+ steady-state syncs — including the ~1s live-tail targeted ingests —
337
+ never write, so the cumulative ``lines_seen`` denominator advances
338
+ only at these writes ("as of the last write"); doctor reasons from
339
+ counts + recency, never a precise ratio.
340
+
341
+ Caller holds the sync flock. Runs at end-of-sync, OUTSIDE every
342
+ per-file ``[before, after]`` total_changes window, so
343
+ ``stats.rows_changed`` stays byte-identical (#270); never bumps
344
+ ``mutation_seq``. Commits its own write (mirrors the walk-complete
345
+ sentinel's commit discipline). Fail-soft: a corrupt prior value is
346
+ treated as absent.
347
+ """
348
+ anomaly_delta = int(lines_malformed) + int(lines_skipped)
349
+ now_iso = dt.datetime.now(dt.timezone.utc).isoformat()
350
+ prior = None
351
+ try:
352
+ row = conn.execute(
353
+ "SELECT value FROM cache_meta WHERE key = ?", (key,)
354
+ ).fetchone()
355
+ if row and row[0]:
356
+ loaded = json.loads(row[0])
357
+ if isinstance(loaded, dict):
358
+ prior = loaded
359
+ except (sqlite3.DatabaseError, ValueError):
360
+ prior = None
361
+ if not rebuild and prior is not None and anomaly_delta == 0:
362
+ return # steady state: zero-write
363
+ if rebuild or prior is None:
364
+ prior = {"lines_seen": 0, "lines_malformed": 0, "lines_skipped": 0,
365
+ "reasons": {}, "last_anomaly_at": None, "since": now_iso}
366
+ record = {
367
+ "schema": _PARSE_HEALTH_SCHEMA,
368
+ "lines_seen": int(prior.get("lines_seen", 0) or 0) + int(lines_seen),
369
+ "lines_malformed": (int(prior.get("lines_malformed", 0) or 0)
370
+ + int(lines_malformed)),
371
+ "lines_skipped": (int(prior.get("lines_skipped", 0) or 0)
372
+ + int(lines_skipped)),
373
+ "last_anomaly_at": (now_iso if anomaly_delta > 0
374
+ else prior.get("last_anomaly_at")),
375
+ "last_write_at": now_iso,
376
+ "since": prior.get("since") or now_iso,
377
+ }
378
+ reasons = dict(prior.get("reasons") or {}) \
379
+ if isinstance(prior.get("reasons"), dict) else {}
380
+ for r, n in (skip_reasons or {}).items():
381
+ reasons[r] = int(reasons.get(r, 0) or 0) + int(n)
382
+ record["reasons"] = reasons
383
+ try:
384
+ _set_cache_meta(conn, key, json.dumps(record, sort_keys=True))
385
+ conn.commit()
386
+ except sqlite3.DatabaseError:
387
+ conn.rollback() # observability must never fail the sync
388
+
389
+
299
390
  # === BEGIN MOVED REGIONS ===
300
391
  # Path constants APP_DIR / CACHE_DB_PATH / CACHE_LOCK_PATH /
301
392
  # CACHE_LOCK_CODEX_PATH live in _cctally_core (promoted 2026-05-22, #84);
@@ -491,6 +582,16 @@ class IngestStats:
491
582
  # and are otherwise unaffected.
492
583
  files_failed: int = 0
493
584
  deferred_reason: "str | None" = None
585
+ # #279 S2 F1 parse-health counters — passive observers over the new-byte
586
+ # span this sync walked. lines_seen counts non-blank lines (malformed
587
+ # included); assistant_lines_skipped counts assistant-typed lines
588
+ # parse_cost_entry rejected for a NON-deliberate reason (schema-drift
589
+ # tripwire; `<synthetic>` and non-assistant lines are normal). Reason
590
+ # vocabulary in _lib_jsonl._classify_cost_entry.
591
+ lines_seen: int = 0
592
+ lines_malformed: int = 0
593
+ assistant_lines_skipped: int = 0
594
+ skip_reasons: dict = field(default_factory=dict)
494
595
 
495
596
  @property
496
597
  def targeted_clean(self) -> bool:
@@ -857,7 +958,10 @@ def sync_cache(
857
958
 
858
959
  lock_fh = open(_cctally_core.CACHE_LOCK_PATH, "w")
859
960
  try:
860
- if not _acquire_cache_flock(lock_fh, timeout=lock_timeout):
961
+ with _perf.phase("flock") as _p_flock:
962
+ _acquired = _acquire_cache_flock(lock_fh, timeout=lock_timeout)
963
+ _p_flock.set_meta(contended=not _acquired)
964
+ if not _acquired:
861
965
  eprint("[cache] sync already in progress; using existing cache")
862
966
  stats.lock_contended = True
863
967
  return stats
@@ -1044,6 +1148,12 @@ def sync_cache(
1044
1148
  # path, which already cleared the flag and repopulates via the normal
1045
1149
  # walk). A path-less/:memory: conn has no cache_meta only if the schema
1046
1150
  # was never applied; the try/except tolerates that.
1151
+ # #276 perf: bracket the (rare, upgrade-only) backfill/reingest region
1152
+ # as one coarse "backfills" phase. Opened via the context-manager
1153
+ # protocol rather than a ``with`` block so the ~150-line body below is
1154
+ # not reindented. Near-noop when tracing is off (_NULL_PHASE).
1155
+ _p_backfills = _perf.phase("backfills")
1156
+ _p_backfills.__enter__()
1047
1157
  if not rebuild and not targeted:
1048
1158
  try:
1049
1159
  _pending = conn.execute(
@@ -1191,12 +1301,22 @@ def sync_cache(
1191
1301
  # IGNORE). Touches ONLY conversation_file_touches, never
1192
1302
  # conversation_messages (P1-2).
1193
1303
  _consume_file_touches(conn)
1304
+ _p_backfills.__exit__(None, None, None)
1194
1305
 
1195
- if targeted:
1196
- paths = [pathlib.Path(p) for p in only_paths if pathlib.Path(p).is_file()]
1197
- else:
1198
- paths = list(_iter_claude_jsonl_files())
1199
- stats.files_total = len(paths)
1306
+ with _perf.phase("discover") as _p_disc:
1307
+ if targeted:
1308
+ # A requested path that vanished (session rotated/deleted
1309
+ # mid-live-tail) is deliberately DROPPED here without flagging
1310
+ # failure: marking files_failed would wedge the watch loop's
1311
+ # targeted_clean advance forever for a file that will never
1312
+ # return; the orphan-prune path owns its stale rows on the next
1313
+ # full sync. Pinned by tests/test_cache_accepted_behaviors.py
1314
+ # (#279 S3 F4).
1315
+ paths = [pathlib.Path(p) for p in only_paths if pathlib.Path(p).is_file()]
1316
+ else:
1317
+ paths = list(_iter_claude_jsonl_files())
1318
+ stats.files_total = len(paths)
1319
+ _p_disc.set_count(len(paths))
1200
1320
 
1201
1321
  # This SELECT does NOT open an implicit transaction (Python's
1202
1322
  # sqlite3 module only BEGINs on DML). Do NOT add any INSERT/
@@ -1406,6 +1526,12 @@ def sync_cache(
1406
1526
  # zero-write-lock read/parse region, so it adds no DML there.
1407
1527
  touched_sessions: set = set()
1408
1528
 
1529
+ # #276 perf: bracket the fused per-file ingest loop as ONE coarse
1530
+ # "walk" phase (never per-row — Section 2 rule: volume is a count, not
1531
+ # N timed phases). Opened via the context-manager protocol so the hot
1532
+ # loop body below is not reindented; counts recorded after the loop.
1533
+ _p_walk = _perf.phase("walk")
1534
+ _p_walk.__enter__()
1409
1535
  for jp in paths:
1410
1536
  path_str = str(jp)
1411
1537
  # Backfill session_id/project_path for A2 `session` subcommand.
@@ -1460,7 +1586,7 @@ def sync_cache(
1460
1586
  # walk over the identical span — the "identical span"
1461
1587
  # invariant is now structural (a single stop point) rather
1462
1588
  # than a prose-enforced ``>= final_offset`` runtime break.
1463
- for offset, cost, mrow, ai in _iter_sync_entries(fh, path_str):
1589
+ for offset, cost, mrow, ai in _iter_sync_entries(fh, path_str, stats=stats):
1464
1590
  if cost is not None:
1465
1591
  entry, msg_id, req_id = cost
1466
1592
  usage = entry.usage
@@ -1683,6 +1809,13 @@ def sync_cache(
1683
1809
 
1684
1810
  if progress is not None:
1685
1811
  progress(stats)
1812
+ _p_walk.__exit__(None, None, None)
1813
+ _p_walk.set_count(stats.files_processed)
1814
+ _p_walk.set_meta(
1815
+ skipped=stats.files_skipped_unchanged,
1816
+ failed=stats.files_failed,
1817
+ rows=stats.rows_changed,
1818
+ )
1686
1819
 
1687
1820
  # Browse-rail rollup maintenance (single post-walk recompute, under the
1688
1821
  # still-held flock, after every per-file commit and before the
@@ -1698,16 +1831,17 @@ def sync_cache(
1698
1831
  # ~1 session/tick). Both recomputes derive COUNT/MIN/MAX from the same
1699
1832
  # rows the rail's old live aggregate read, so the rollup stays
1700
1833
  # byte-identical to that aggregate.
1701
- if _conversation_sessions_backfill_pending(conn):
1702
- _recompute_conversation_sessions(conn)
1703
- conn.execute(
1704
- "DELETE FROM cache_meta "
1705
- "WHERE key='conversation_sessions_backfill_pending'"
1706
- )
1707
- conn.commit()
1708
- elif touched_sessions:
1709
- _recompute_conversation_sessions(conn, touched_sessions)
1710
- conn.commit()
1834
+ with _perf.phase("recompute.conversation_sessions"):
1835
+ if _conversation_sessions_backfill_pending(conn):
1836
+ _recompute_conversation_sessions(conn)
1837
+ conn.execute(
1838
+ "DELETE FROM cache_meta "
1839
+ "WHERE key='conversation_sessions_backfill_pending'"
1840
+ )
1841
+ conn.commit()
1842
+ elif touched_sessions:
1843
+ _recompute_conversation_sessions(conn, touched_sessions)
1844
+ conn.commit()
1711
1845
 
1712
1846
  # Walk-complete sentinel write (cctally-dev#93, D5a). Still inside the
1713
1847
  # held fcntl lock, before the finally-unlock. Only when the entire walk
@@ -1724,6 +1858,17 @@ def sync_cache(
1724
1858
  (dt.datetime.now(dt.timezone.utc).isoformat(),),
1725
1859
  )
1726
1860
  conn.commit()
1861
+ # #279 S2 F1: rolling parse-health record. Anomaly-delta-gated so
1862
+ # steady-state (incl. targeted live-tail) syncs stay zero-write;
1863
+ # targeted syncs still accumulate — they ingest real new bytes.
1864
+ _update_parse_health_meta(
1865
+ conn, "parse_health_claude",
1866
+ lines_seen=stats.lines_seen,
1867
+ lines_malformed=stats.lines_malformed,
1868
+ lines_skipped=stats.assistant_lines_skipped,
1869
+ skip_reasons=stats.skip_reasons,
1870
+ rebuild=rebuild,
1871
+ )
1727
1872
  # At-rest hardening (Plan 2, spec §5). Runs here — at the end of the
1728
1873
  # write transaction, while the cache.db.lock flock is still held (so a
1729
1874
  # concurrent writer can't be mid-checkpoint) AND after at least one
@@ -2794,6 +2939,13 @@ class CodexIngestStats:
2794
2939
  # $CODEX_HOME root set (issue #108 — a prior-root purge, not a delta).
2795
2940
  files_pruned: int = 0
2796
2941
  lock_contended: bool = False
2942
+ # #279 S2 F1 parse-health counters — folded from each file's
2943
+ # _CodexIterState after its drain. Same vocabulary as the iterator
2944
+ # (info-non-dict / no-last-token-usage / bad-timestamp / no-session-id).
2945
+ lines_seen: int = 0
2946
+ lines_malformed: int = 0
2947
+ token_events_skipped: int = 0
2948
+ skip_reasons: dict = field(default_factory=dict)
2797
2949
 
2798
2950
 
2799
2951
  def _progress_codex_stderr(stats: CodexIngestStats, *, force: bool = False) -> None:
@@ -2831,7 +2983,10 @@ def sync_codex_cache(
2831
2983
 
2832
2984
  lock_fh = open(_cctally_core.CACHE_LOCK_CODEX_PATH, "w")
2833
2985
  try:
2834
- if not _acquire_cache_flock(lock_fh, timeout=lock_timeout):
2986
+ with _perf.phase("flock") as _p_flock:
2987
+ _acquired = _acquire_cache_flock(lock_fh, timeout=lock_timeout)
2988
+ _p_flock.set_meta(contended=not _acquired)
2989
+ if not _acquired:
2835
2990
  eprint("[codex-cache] sync already in progress; using existing cache")
2836
2991
  stats.lock_contended = True
2837
2992
  return stats
@@ -2848,8 +3003,10 @@ def sync_codex_cache(
2848
3003
  roots = _cctally()._codex_session_roots()
2849
3004
  # Pure read (glob + is_file only); safe to run before the SELECT and
2850
3005
  # the per-file loop, where no cache.db write lock may be held.
2851
- paths: list[pathlib.Path] = list(_iter_codex_jsonl_paths(roots))
2852
- stats.files_total = len(paths)
3006
+ with _perf.phase("discover") as _p_disc:
3007
+ paths: list[pathlib.Path] = list(_iter_codex_jsonl_paths(roots))
3008
+ stats.files_total = len(paths)
3009
+ _p_disc.set_count(len(paths))
2853
3010
 
2854
3011
  # Scope the cache to the CURRENT root set: drop rows ingested under a
2855
3012
  # prior $CODEX_HOME (issue #108). iter_codex_entries() has NO root
@@ -2910,6 +3067,11 @@ def sync_codex_cache(
2910
3067
  )
2911
3068
  }
2912
3069
 
3070
+ # #279 S2 F4: ONE coarse `walk` phase bracketing the per-file loop
3071
+ # (count = files_processed, never per-row — §2 rule). Manual CM so
3072
+ # the loop stays flat, mirroring sync_cache's walk seam.
3073
+ _p_walk = _perf.phase("walk")
3074
+ _p_walk.__enter__()
2913
3075
  for jp in paths:
2914
3076
  path_str = str(jp)
2915
3077
  try:
@@ -2958,15 +3120,19 @@ def sync_codex_cache(
2958
3120
  # ends on a metadata-only tail would lose the terminal
2959
3121
  # session_id/model and the next resume would mis-attribute the
2960
3122
  # first post-resume token_count.
3123
+ # #279 S3 F1: the state is BOTH the seed carrier for the dedup
3124
+ # watermark and the sink the iterator stamps it into. Seeding it
3125
+ # with initial_total_tokens (the prior resume's persisted
3126
+ # cumulative) means iter_state.total_tokens holds the guard's
3127
+ # terminal watermark once the iterator drains — which we persist
3128
+ # directly, replacing the former initial+Σ(per-turn) reconstruction
3129
+ # that could diverge from the true cumulative and double-count/skip
3130
+ # on the next resume.
2961
3131
  iter_state = _CodexIterState(
2962
3132
  session_id=initial_session_id,
2963
3133
  model=initial_model,
3134
+ total_tokens=initial_total_tokens,
2964
3135
  )
2965
- # Track the cumulative `total_token_usage.total_tokens` across this
2966
- # call. The iterator only yields when the cumulative strictly
2967
- # advances by the current turn's `last_token_usage.total_tokens`,
2968
- # so summing the per-turn totals reconstructs the final cumulative.
2969
- running_total = initial_total_tokens
2970
3136
  yielded_count = 0
2971
3137
  try:
2972
3138
  with open(jp, "r", encoding="utf-8", errors="replace") as fh:
@@ -2991,9 +3157,18 @@ def sync_codex_cache(
2991
3157
  entry.reasoning_output_tokens,
2992
3158
  entry.total_tokens,
2993
3159
  ))
2994
- running_total += int(entry.total_tokens or 0)
2995
3160
  yielded_count += 1
2996
3161
  final_offset = fh.tell()
3162
+ # #279 S2 F1: fold this file's iterator counters into the
3163
+ # sync-level stats. iter_state is per-file, so += is
3164
+ # exact. Folded inside the try: an OSError mid-read drops
3165
+ # this file's partial counters AND its offset advance
3166
+ # together, so re-walked lines are never double-counted.
3167
+ stats.lines_seen += iter_state.lines_seen
3168
+ stats.lines_malformed += iter_state.lines_malformed
3169
+ stats.token_events_skipped += iter_state.token_events_skipped
3170
+ for _r, _n in iter_state.skip_reasons.items():
3171
+ stats.skip_reasons[_r] = stats.skip_reasons.get(_r, 0) + _n
2997
3172
  except OSError as exc:
2998
3173
  eprint(f"[codex-cache] could not read {jp}: {exc}")
2999
3174
  continue
@@ -3014,11 +3189,13 @@ def sync_codex_cache(
3014
3189
  else initial_model
3015
3190
  )
3016
3191
 
3017
- # Persist the running cumulative if we yielded this call. Otherwise
3018
- # preserve the prior value never overwrite with 0, which would
3019
- # re-enable double-counting on the next resume.
3192
+ # Persist the iterator's stamped cumulative watermark if we yielded
3193
+ # this call (iter_state.total_tokens == the dedup guard's terminal
3194
+ # value by construction, #279 S3 F1). Otherwise preserve the prior
3195
+ # value — never overwrite with 0, which would re-enable
3196
+ # double-counting on the next resume.
3020
3197
  new_last_total_tokens: int | None = (
3021
- running_total if yielded_count > 0 else prev_total_tokens
3198
+ iter_state.total_tokens if yielded_count > 0 else prev_total_tokens
3022
3199
  )
3023
3200
 
3024
3201
  # Python's sqlite3 module starts an implicit transaction on the
@@ -3070,6 +3247,21 @@ def sync_codex_cache(
3070
3247
 
3071
3248
  if progress is not None:
3072
3249
  progress(stats)
3250
+ _p_walk.__exit__(None, None, None)
3251
+ _p_walk.set_count(stats.files_processed)
3252
+ _p_walk.set_meta(skipped=stats.files_skipped_unchanged,
3253
+ rows=stats.rows_changed)
3254
+ # #279 S2 F1: rolling parse-health record (codex half). Same
3255
+ # anomaly-delta gate as the Claude tail; SQLite serializes the
3256
+ # cache.db write against a concurrent Claude sync.
3257
+ _update_parse_health_meta(
3258
+ conn, "parse_health_codex",
3259
+ lines_seen=stats.lines_seen,
3260
+ lines_malformed=stats.lines_malformed,
3261
+ lines_skipped=stats.token_events_skipped,
3262
+ skip_reasons=stats.skip_reasons,
3263
+ rebuild=rebuild,
3264
+ )
3073
3265
  return stats
3074
3266
  finally:
3075
3267
  try:
@@ -3322,6 +3514,11 @@ def open_cache_db() -> sqlite3.Connection:
3322
3514
 
3323
3515
  conn.execute("PRAGMA journal_mode=WAL")
3324
3516
  conn.execute("PRAGMA busy_timeout=5000")
3517
+ # Re-derivable DB under WAL: NORMAL risks at most the tail transaction on
3518
+ # power loss, and cache.db can always be rebuilt (cache-sync --rebuild).
3519
+ # Fewer ingest fsyncs than the default FULL. Matches stats.db
3520
+ # (bin/_cctally_core.py open_db). #279 S1 F8.
3521
+ conn.execute("PRAGMA synchronous=NORMAL")
3325
3522
 
3326
3523
  # Apply the shared cache.db schema (cctally-dev#93, D4): Claude tables +
3327
3524
  # indexes, the session_id / project_path column adds on session_files
@@ -3370,6 +3567,10 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
3370
3567
  default is 'all'.
3371
3568
  """
3372
3569
  source = getattr(args, "source", "all")
3570
+ # #276 perf: clear any prior tree on this thread so a leaked root can't be
3571
+ # flushed, then (below) time the Claude sync_cache call as the "sync_cache"
3572
+ # root phase and flush the tree to stderr when CCTALLY_PERF_TRACE is set.
3573
+ _perf.reset_thread()
3373
3574
  conn = open_cache_db()
3374
3575
 
3375
3576
  # --prune-orphans: fast, targeted cleanup of cache rows whose source
@@ -3418,10 +3619,18 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
3418
3619
  lt = _REBUILD_LOCK_TIMEOUT_SECONDS if args.rebuild else None
3419
3620
  contended = False
3420
3621
 
3622
+ # #279 S2 F4: one shared `cache-sync` root so a single flushed tree
3623
+ # carries BOTH vendors — with two sequential sync roots,
3624
+ # flush_stderr(current_root()) would show only the last one. Opened
3625
+ # after the --prune-orphans early returns so they can't leak a root.
3626
+ _p_root = _perf.phase("cache-sync")
3627
+ _p_root.__enter__()
3628
+
3421
3629
  if source in ("claude", "all"):
3422
- stats = sync_cache(
3423
- conn, progress=_progress_stderr, rebuild=args.rebuild, lock_timeout=lt
3424
- )
3630
+ with _perf.phase("sync_cache"):
3631
+ stats = sync_cache(
3632
+ conn, progress=_progress_stderr, rebuild=args.rebuild, lock_timeout=lt
3633
+ )
3425
3634
  _progress_stderr(stats, force=True)
3426
3635
  if stats.lock_contended and args.rebuild:
3427
3636
  eprint(
@@ -3434,13 +3643,17 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
3434
3643
  f"[cache-sync] claude done: {stats.files_processed} processed, "
3435
3644
  f"{stats.files_skipped_unchanged} skipped, "
3436
3645
  f"{stats.files_reset_truncated} reset, "
3437
- f"{stats.rows_changed} rows changed"
3646
+ f"{stats.rows_changed} rows changed, "
3647
+ f"{stats.lines_malformed} malformed, "
3648
+ f"{stats.assistant_lines_skipped} drift-skipped"
3438
3649
  )
3439
3650
 
3440
3651
  if source in ("codex", "all"):
3441
- stats = sync_codex_cache(
3442
- conn, progress=_progress_codex_stderr, rebuild=args.rebuild, lock_timeout=lt
3443
- )
3652
+ with _perf.phase("sync_codex_cache"):
3653
+ stats = sync_codex_cache(
3654
+ conn, progress=_progress_codex_stderr, rebuild=args.rebuild,
3655
+ lock_timeout=lt,
3656
+ )
3444
3657
  _progress_codex_stderr(stats, force=True)
3445
3658
  if stats.lock_contended and args.rebuild:
3446
3659
  eprint(
@@ -3453,7 +3666,16 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
3453
3666
  f"[cache-sync] codex done: {stats.files_processed} processed, "
3454
3667
  f"{stats.files_skipped_unchanged} skipped, "
3455
3668
  f"{stats.files_reset_truncated} reset, "
3456
- f"{stats.rows_changed} rows changed"
3669
+ f"{stats.rows_changed} rows changed, "
3670
+ f"{stats.lines_malformed} malformed, "
3671
+ f"{stats.token_events_skipped} drift-skipped"
3457
3672
  )
3458
3673
 
3674
+ _p_root.__exit__(None, None, None)
3675
+ # #276 perf: when tracing is enabled, flush the completed "cache-sync"
3676
+ # phase tree to stderr (stdout stays byte-identical). No-op when off.
3677
+ # As of #279 S2 F4 the root carries both the sync_cache and
3678
+ # sync_codex_cache children, so one flushed tree covers both vendors.
3679
+ if _perf.enabled():
3680
+ _perf.flush_stderr(_perf.current_root())
3459
3681
  return 1 if contended else 0
@@ -1010,7 +1010,7 @@ def _emit_cache_report_json(
1010
1010
  for row_dict in output[top_key]:
1011
1011
  row_dict["totalCost"] = row_dict["cost"]
1012
1012
  output["totals"]["totalCost"] = output["totals"]["cost"]
1013
- return json.dumps(output, indent=2)
1013
+ return json.dumps(_cctally().stamp_schema_version(output), indent=2)
1014
1014
 
1015
1015
 
1016
1016
  def _build_cache_report_title(args: argparse.Namespace, mode: str) -> str:
@@ -1141,8 +1141,9 @@ def cmd_cache_report(args: argparse.Namespace) -> int:
1141
1141
  if since == until:
1142
1142
  if args.json:
1143
1143
  print(json.dumps(
1144
- {top_key: [], "totals": None,
1145
- "generatedAt": now_utc_iso(now_utc=now_utc)},
1144
+ c.stamp_schema_version(
1145
+ {top_key: [], "totals": None,
1146
+ "generatedAt": now_utc_iso(now_utc=now_utc)}),
1146
1147
  indent=2,
1147
1148
  ))
1148
1149
  else:
@@ -1163,8 +1164,9 @@ def cmd_cache_report(args: argparse.Namespace) -> int:
1163
1164
  if not rows:
1164
1165
  if args.json:
1165
1166
  print(json.dumps(
1166
- {top_key: [], "totals": None,
1167
- "generatedAt": now_utc_iso(now_utc=now_utc)},
1167
+ c.stamp_schema_version(
1168
+ {top_key: [], "totals": None,
1169
+ "generatedAt": now_utc_iso(now_utc=now_utc)}),
1168
1170
  indent=2,
1169
1171
  ))
1170
1172
  else: