cctally 1.99.0 → 1.100.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -196,7 +196,7 @@ import threading
196
196
  import time
197
197
  import urllib.error
198
198
  import urllib.request
199
- from dataclasses import dataclass, replace
199
+ from dataclasses import dataclass
200
200
  from typing import Any, Callable
201
201
 
202
202
 
@@ -2368,10 +2368,12 @@ class _DashboardUpdateCheckThread(threading.Thread):
2368
2368
  ``envelope_precompute`` rather than re-reading the state file per
2369
2369
  envelope build, so a BARE publish of the held snapshot would surface
2370
2370
  the stale precompute. The republish therefore rebuilds
2371
- ``envelope_precompute`` (small config/state JSON I/O, no DB, no
2372
- module-cache mutation — see :meth:`_republish_with_fresh_envelope`)
2373
- on a fresh ``dataclasses.replace`` snapshot before publishing, and
2374
- persists it to the snapshot ref so the held snapshot stays current.
2371
+ ``envelope_precompute`` and the doctor summary on a fresh
2372
+ ``dataclasses.replace`` snapshot before publishing, and persists it to
2373
+ the snapshot ref so the held snapshot stays current. The doctor refresh
2374
+ matters because ``safety.update_state`` reads the file this thread just
2375
+ changed; retaining the old payload made ``/api/data`` disagree with the
2376
+ live ``/api/doctor`` endpoint until a later data rebuild.
2375
2377
  Under ``--no-sync`` this thread is the ONLY refresher of that
2376
2378
  precompute (the data-sync thread that otherwise rebuilds it is
2377
2379
  disabled).
@@ -2385,11 +2387,13 @@ class _DashboardUpdateCheckThread(threading.Thread):
2385
2387
  *,
2386
2388
  hub: "SSEHub | None" = None,
2387
2389
  snapshot_ref: "_SnapshotRef | None" = None,
2390
+ runtime_bind: "str | None" = None,
2388
2391
  ) -> None:
2389
2392
  super().__init__(name="cctally-update-check")
2390
2393
  self._stop_event = stop_event
2391
2394
  self._hub = hub
2392
2395
  self._ref = snapshot_ref
2396
+ self._runtime_bind = runtime_bind
2393
2397
 
2394
2398
  def run(self) -> None:
2395
2399
  c = _cctally()
@@ -2412,16 +2416,12 @@ class _DashboardUpdateCheckThread(threading.Thread):
2412
2416
  and self._hub is not None
2413
2417
  and self._ref is not None
2414
2418
  ):
2415
- snap = self._ref.get()
2416
- if snap is not None:
2417
- self._republish_with_fresh_envelope(snap)
2419
+ self._republish_with_fresh_envelope()
2418
2420
  config = load_config()
2419
2421
  if c._is_update_check_due(config):
2420
2422
  c._do_update_check()
2421
2423
  if self._hub is not None and self._ref is not None:
2422
- snap = self._ref.get()
2423
- if snap is not None:
2424
- self._republish_with_fresh_envelope(snap)
2424
+ self._republish_with_fresh_envelope()
2425
2425
  except Exception as e:
2426
2426
  # Log but never propagate — this thread must keep
2427
2427
  # ticking so a transient registry hiccup doesn't
@@ -2437,8 +2437,8 @@ class _DashboardUpdateCheckThread(threading.Thread):
2437
2437
  pass
2438
2438
  self._stop_event.wait(c.UPDATE_DASHBOARD_CHECK_POLL_S)
2439
2439
 
2440
- def _republish_with_fresh_envelope(self, snap) -> None:
2441
- """Republish ``snap`` with a freshly-rebuilt ``envelope_precompute``.
2440
+ def _republish_with_fresh_envelope(self) -> None:
2441
+ """Patch the latest snapshot with fresh config and doctor precomputes.
2442
2442
 
2443
2443
  Since M4 (#268) the SSE envelope reads update-state / update-suppress
2444
2444
  off ``snap.envelope_precompute`` rather than re-reading the JSON files
@@ -2450,26 +2450,32 @@ class _DashboardUpdateCheckThread(threading.Thread):
2450
2450
  precompute refresher) is disabled.
2451
2451
 
2452
2452
  ``_tui_precompute_envelope_config`` reads only the config / update-state
2453
- / update-suppress files and mutates NO module cache, so calling it from
2454
- this thread is safe (no shared-cache-mutation / Codex F7 hazard).
2455
- ``dataclasses.replace`` yields a NEW snapshot sharing ``prior``'s
2456
- immutable rows, so an SSE client serializing the previously-published
2457
- snapshot can't observe a torn value. The fresh snapshot is persisted to
2458
- the ref so the held snapshot stays current for the next reader.
2453
+ / update-suppress files and mutates NO module cache. The doctor memo is
2454
+ invalidated deliberately before recomputation: the update-state write
2455
+ changed an input to ``safety.update_state``, so the ordinary 30-second
2456
+ memo would otherwise hand this republish the stale summary again.
2457
+ ``_SnapshotRef.replace_fields`` applies only these three narrow fields
2458
+ to the latest held object under its lock. That atomic patch matters
2459
+ because a full sync can publish while the precomputes above are being
2460
+ gathered; writing a whole snapshot read before that sync would regress
2461
+ every data field and can restore a partial hydration seed.
2459
2462
  """
2460
2463
  c = _cctally()
2461
- fresh = replace(
2462
- snap,
2464
+ c._load_sibling("_lib_snapshot_cache").reset_doctor_memo()
2465
+ doctor_payload = c._cctally_tui._tui_precompute_doctor_payload(
2466
+ _now_utc(), self._runtime_bind,
2467
+ )
2468
+ fresh = self._ref.replace_fields(
2463
2469
  envelope_precompute=(
2464
2470
  c._cctally_tui._tui_precompute_envelope_config(load_config())
2465
2471
  ),
2472
+ doctor_payload=doctor_payload,
2466
2473
  # #278 §1.4.1: a version-banner refresh republishes complete data;
2467
2474
  # force the hydration latch clear so a republish that happens to
2468
2475
  # carry a prior hydrating seed/partial doesn't freeze the client's
2469
2476
  # loading skeletons.
2470
2477
  hydrating=False,
2471
2478
  )
2472
- self._ref.set(fresh)
2473
2479
  self._hub.publish(fresh)
2474
2480
 
2475
2481
 
@@ -93,7 +93,15 @@ CapabilityStatus = Literal[
93
93
  # optional `spendWindow` bounds, which a cycle-bounded card omits. No client
94
94
  # branches on this number; after an in-place `execvp` update a still-loaded old
95
95
  # client renders the new figures under old copy until it reloads.
96
- SOURCE_SCHEMA_VERSION = 9
96
+ # 9 -> 10 (#583 S3): `sources.all.data.providers` no longer carries the two
97
+ # provider data objects. They are published once, on the physical
98
+ # `sources.claude` / `sources.codex` entries, which is where every production
99
+ # consumer already falls back to. The key itself is RETAINED with both members
100
+ # null so a still-loaded v9 bundle does not throw — this is the first
101
+ # DESTRUCTIVE entry in this ledger rather than an additive one, and the null
102
+ # stub is what makes it survivable across an in-place `execvp` update. Removing
103
+ # the stub needs a separate change once no v9 bundle can be live.
104
+ SOURCE_SCHEMA_VERSION = 10
97
105
  DEFAULT_SOURCE = "claude"
98
106
  SOURCE_ORDER = ("claude", "codex", "all")
99
107
  SOURCE_FRESHNESS_DOMAINS = ("hero", "quota", "sessions")
@@ -1217,13 +1225,24 @@ def compose_all_state(
1217
1225
  if combined is None else {}),
1218
1226
  "alerts": {"rows": combined_alerts},
1219
1227
  # #556 S2 §3.5.1: the ONE public copy of the shared range and of
1220
- # both aggregate outcomes. The rows stay on the provider domains
1221
- # under `providers` below, so nothing is published twice.
1228
+ # both aggregate outcomes. The rows stay on the provider domains,
1229
+ # published on the physical `sources.claude` / `sources.codex`
1230
+ # entries since #583 S3, so nothing is published twice.
1222
1231
  "aggregates": aggregates,
1223
- "providers": {
1224
- "claude": claude.data,
1225
- "codex": codex.data,
1226
- },
1232
+ # #583 S3 §4: the provider data is published ONCE, on the physical
1233
+ # `sources.claude` / `sources.codex` entries. This mirror duplicated
1234
+ # about 1.56 MB of a 3.25 MB envelope by reference.
1235
+ #
1236
+ # The key is RETAINED with null members rather than removed.
1237
+ # `dashboard/web/src/lib/dashboardPresentation.ts` reads
1238
+ # `data?.providers.claude` — the optional chain guards `data`, not
1239
+ # `providers` — so an absent key throws a TypeError inside a render
1240
+ # for any tab still running the v9 bundle when `cctally update`
1241
+ # `execvp`s the server underneath it. Both members are declared
1242
+ # nullable in `types/envelope.ts`, so null is a legal v9 value and
1243
+ # every v9 consumer's `?? sources.<p>.data` fallback resolves.
1244
+ # Retiring the stub is filed as a residual, not done here.
1245
+ "providers": {"claude": None, "codex": None},
1227
1246
  },
1228
1247
  domain_freshness={
1229
1248
  domain: (
@@ -232,6 +232,11 @@ class DoctorState:
232
232
  # DOCTOR_WAL_WARN_BYTES (2x the WAL cap) — only when the journal_size_limit
233
233
  # + forced-checkpoint machinery has genuinely failed to contain the WAL.
234
234
  cache_db_wal_bytes: Optional[int] = None
235
+ # #583 S4 (F39): size in bytes of conversations.db-wal, gathered the same
236
+ # read-only way as cache_db_wal_bytes above (absent -> 0, OSError -> None).
237
+ # STORE_POLICY gives conversations.db the same 128 MiB journal_size_limit,
238
+ # so it warns above the SAME DOCTOR_WAL_WARN_BYTES threshold.
239
+ conversations_db_wal_bytes: Optional[int] = None
235
240
  # #496 S5b: the durable incomplete-quota-projection flag carried inside the
236
241
  # published stats generation. True = every quota-projection read is refused
237
242
  # until a reconciliation runs; False = the generation is complete; None =
@@ -2294,6 +2299,37 @@ def _check_db_wal_size(s: DoctorState) -> CheckResult:
2294
2299
  )
2295
2300
 
2296
2301
 
2302
+ def _check_db_conversations_wal_size(s: DoctorState) -> CheckResult:
2303
+ """Read-only backstop for the transcript store's WAL (#583 S4 / F39).
2304
+
2305
+ Sibling of `_check_db_wal_size`, sharing DOCTOR_WAL_WARN_BYTES because
2306
+ STORE_POLICY gives conversations.db the same 128 MiB journal_size_limit as
2307
+ cache.db — so the same "only fires when containment has genuinely failed"
2308
+ reasoning applies. Same fingerprint discipline: the exact byte count lives
2309
+ ONLY in the excluded details block and the summary is a stable string, so
2310
+ below-threshold drift cannot flip the fingerprint while an OK<->WARN
2311
+ crossing does.
2312
+ """
2313
+ wal = s.conversations_db_wal_bytes
2314
+ details = {"conversations_db_wal_bytes": wal}
2315
+ if isinstance(wal, int) and wal > DOCTOR_WAL_WARN_BYTES:
2316
+ return CheckResult(
2317
+ id="db.conversations_wal_size",
2318
+ title="conversations.db WAL size", severity="warn",
2319
+ summary="oversized — conversations.db WAL far above its cap",
2320
+ remediation=(
2321
+ "Run `cctally db checkpoint --db conversations` "
2322
+ "to drain the WAL."
2323
+ ),
2324
+ details=details,
2325
+ )
2326
+ return CheckResult(
2327
+ id="db.conversations_wal_size",
2328
+ title="conversations.db WAL size", severity="ok",
2329
+ summary="within limit", remediation=None, details=details,
2330
+ )
2331
+
2332
+
2297
2333
  # #315: conservative advisory threshold. A quarter of cache.db being free is
2298
2334
  # large enough to make an explicit, guarded VACUUM useful without nagging for
2299
2335
  # ordinary page churn. This is a ratio, so it remains page-size independent.
@@ -3449,6 +3485,7 @@ _CATEGORY_DEFINITIONS: tuple[tuple[str, str, tuple[tuple[str, str], ...]], ...]
3449
3485
  ("db.migrations.pending", "_check_db_migrations_pending"),
3450
3486
  ("db.lock_state", "_check_db_lock_state"),
3451
3487
  ("db.wal_size", "_check_db_wal_size"),
3488
+ ("db.conversations_wal_size", "_check_db_conversations_wal_size"),
3452
3489
  ("db.reclaimable", "_check_db_reclaimable"),
3453
3490
  ("db.conversations_reclaimable", "_check_db_conversations_reclaimable"),
3454
3491
  ("db.retained_artifacts", "_check_db_retained_artifacts"),
package/bin/_lib_jsonl.py CHANGED
@@ -65,12 +65,14 @@ class UsageEntry:
65
65
  # crashes loudly at construction instead).
66
66
 
67
67
 
68
- @dataclass
68
+ @dataclass(frozen=True)
69
69
  class CodexEntry:
70
70
  """One emitted Codex `token_count` event row.
71
71
 
72
72
  Mirrors the columns of codex_session_entries. `last_token_usage` fields
73
- are used (per-turn deltas), not the cumulative totals.
73
+ are used (per-turn deltas), not the cumulative totals. Instances are
74
+ immutable because dashboard folds intentionally alias one entry between
75
+ the merged parent and its account partition.
74
76
  """
75
77
  timestamp: dt.datetime
76
78
  session_id: str
package/bin/_lib_perf.py CHANGED
@@ -15,6 +15,7 @@ and fields may change without a version bump.
15
15
  """
16
16
  from __future__ import annotations
17
17
 
18
+ import contextlib
18
19
  import os
19
20
  import sys
20
21
  import threading
@@ -29,8 +30,18 @@ def enabled() -> bool:
29
30
 
30
31
 
31
32
  def set_enabled(value: bool) -> None:
32
- """Flip tracing at runtime (tests; not used by the dashboard, which reads
33
- the env at import time)."""
33
+ """Flip tracing immediately (tests, and ``apply_pending`` below).
34
+
35
+ Runtime arming goes through ``request_enabled`` + ``apply_pending`` instead
36
+ (#583 S1 §2.2), which defers the flip to a rebuild boundary. This entry
37
+ point stays for tests.
38
+
39
+ It deliberately does NOT clear a captured root arm: an already-open root is
40
+ wholly traced or wholly untraced whatever the global does, and that is the
41
+ property the per-root capture exists to give. Every caller that wants the
42
+ new value to take effect calls ``reset_thread()`` afterwards, which is what
43
+ both production root sites already do.
44
+ """
34
45
  global _ENABLED
35
46
  _ENABLED = bool(value)
36
47
 
@@ -38,6 +49,97 @@ def set_enabled(value: bool) -> None:
38
49
  _tls = threading.local()
39
50
 
40
51
 
52
+ # ── runtime arming: an atomic mailbox, applied per root (#583 S1 §2.2) ──────
53
+ # `request_enabled` records the latest desired state and `apply_pending`
54
+ # consumes it, both under one lock performing a LATEST-VALUE EXCHANGE — read
55
+ # and clear inside one critical section. An unsynchronised test-and-clear loses
56
+ # a request arriving between the two steps, which is the race
57
+ # `_SnapshotRef.take_sync_request` already documents in this repository.
58
+ _UNSET = object()
59
+ _MAILBOX_LOCK = threading.Lock()
60
+ _PENDING = _UNSET
61
+
62
+
63
+ def request_enabled(value: bool) -> None:
64
+ """Record the desired tracing state. Applies at the next boundary."""
65
+ global _PENDING
66
+ with _MAILBOX_LOCK:
67
+ _PENDING = bool(value)
68
+
69
+
70
+ def apply_pending() -> bool:
71
+ """Consume any pending request and apply it. Returns whether it changed.
72
+
73
+ Called ONLY from `_make_run_sync_now_locked._locked`, after the cache
74
+ connection closes and immediately before the authoritative build, so a
75
+ request arriving mid-ingest cannot split one ingest across two tracing
76
+ states. A2 partial builds never consume it.
77
+ """
78
+ global _PENDING
79
+ with _MAILBOX_LOCK:
80
+ value, _PENDING = _PENDING, _UNSET
81
+ if value is _UNSET:
82
+ return False
83
+ changed = bool(value) != _ENABLED
84
+ set_enabled(bool(value))
85
+ return changed
86
+
87
+
88
+ def pending_state() -> "tuple[bool, bool]":
89
+ """``(requested, applied)``. Equal once the request has taken effect."""
90
+ with _MAILBOX_LOCK:
91
+ pending = _PENDING
92
+ applied = _ENABLED
93
+ return (applied if pending is _UNSET else bool(pending), applied)
94
+
95
+
96
+ def applies_at() -> str:
97
+ """When a pending request takes effect. Names the boundary, not a clock.
98
+
99
+ Without this the operator cannot tell a request from its effect, and
100
+ ``--trace off`` appears to succeed while tracing is still applied until the
101
+ next authoritative build.
102
+ """
103
+ requested, applied = pending_state()
104
+ return "next_authoritative_build" if requested != applied else "none"
105
+
106
+
107
+ def root_armed() -> "bool | None":
108
+ """The tracing state this thread's open root captured, or None if no root
109
+ scope is open on this thread."""
110
+ return getattr(_tls, "root_armed", None)
111
+
112
+
113
+ @contextlib.contextmanager
114
+ def isolated_thread_state():
115
+ """Run a nested build against empty, private phase state (#583 S1 §2.1).
116
+
117
+ ``_make_a2_progress_cb`` calls ``build_partial()`` synchronously from
118
+ inside ``sync_cache``'s still-open ``walk`` phase, and that build reaches
119
+ ``_tui_build_snapshot_once``'s unconditional ``reset_thread()``. Each
120
+ ``Phase`` retains its original list in ``Phase._stack`` while
121
+ ``reset_thread()`` rebinds ``_tls.stack`` and clears ``_tls.root``, so
122
+ without isolation the later phases attach to a different list and the outer
123
+ phase closes into a detached or fragmented root.
124
+
125
+ Saves the EXACT stack and root object references — by identity, because
126
+ the open phases hold that same list — and restores them in ``finally``,
127
+ including when the body raises.
128
+ """
129
+ saved_stack = getattr(_tls, "stack", None)
130
+ saved_root = getattr(_tls, "root", None)
131
+ saved_armed = getattr(_tls, "root_armed", None)
132
+ _tls.stack = []
133
+ _tls.root = None
134
+ _tls.root_armed = None
135
+ try:
136
+ yield
137
+ finally:
138
+ _tls.stack = saved_stack
139
+ _tls.root = saved_root
140
+ _tls.root_armed = saved_armed
141
+
142
+
41
143
  def _stack():
42
144
  s = getattr(_tls, "stack", None)
43
145
  if s is None:
@@ -90,6 +192,9 @@ class Phase:
90
192
  stack[-1].children.append(self)
91
193
  else:
92
194
  _tls.root = self # outermost phase closed -> the build root
195
+ # The root scope ends with its root, so a later phase on this
196
+ # thread reads the global again rather than a stale capture.
197
+ _tls.root_armed = None
93
198
  return False
94
199
 
95
200
  def to_dict(self):
@@ -123,7 +228,14 @@ _NULL_PHASE = _NullPhase()
123
228
 
124
229
 
125
230
  def phase(name):
126
- if not _ENABLED:
231
+ # Per-root capture (#583 S1 §2.2). `_ENABLED` is process-global and HTTP
232
+ # handlers open their own roots on their own threads, so a flip landing
233
+ # mid-request would otherwise create traced phases beneath an untraced root
234
+ # or drop later children from a traced one. A root captures the armed state
235
+ # at its own creation and every phase under it consults that value; a
236
+ # thread with no root scope open falls back to the global.
237
+ armed = getattr(_tls, "root_armed", None)
238
+ if not (_ENABLED if armed is None else armed):
127
239
  return _NULL_PHASE
128
240
  return Phase(name, _stack())
129
241
 
@@ -133,8 +245,25 @@ def current_root():
133
245
 
134
246
 
135
247
  def reset_thread():
248
+ """Start a fresh root scope on this thread, capturing the armed state.
249
+
250
+ Every root in this tree is opened immediately after a `reset_thread()` —
251
+ `_tui_build_snapshot_once` and the dashboard's `_perf_scope` are the two
252
+ sites — so this is where a root's tracing decision is taken. The scope ends
253
+ when the outermost phase closes, or when the next `reset_thread()` recaptures.
254
+
255
+ KNOWN LIMIT, recorded rather than fixed. A DISARMED capture is not cleared
256
+ by the closing phase, because no `Phase` is created to close: a thread that
257
+ captures `False` and then never calls `reset_thread()` again keeps reading
258
+ that stale value and stays untraced. Unreachable today — the dashboard
259
+ threads per request and resets at the top of every build — but it would
260
+ bite a thread-POOLED server, where a worker outlives many requests. Fixing
261
+ it needs an explicit root scope that exists when tracing is off, not a
262
+ smarter capture point.
263
+ """
136
264
  _tls.stack = []
137
265
  _tls.root = None
266
+ _tls.root_armed = _ENABLED
138
267
 
139
268
 
140
269
  def flush_stderr(root):
@@ -221,8 +221,8 @@ def _totals(entries: Iterable[QualifiedCodexEntry]) -> TokenTotals:
221
221
  # `build_codex_project_result`, over the complete population, before any
222
222
  # subset is taken -- which is the only place it can be correct. Guarding
223
223
  # this call on "the subset holds more than one project identity" would
224
- # have kept nearly all of the cost, because the expensive callers are the
225
- # per-block totals over a population spanning every project.
224
+ # have kept roughly three-quarters of the cost, because the expensive
225
+ # callers are the per-block totals over a population spanning every project.
226
226
  values = tuple(entries)
227
227
  return TokenTotals(
228
228
  input_tokens=sum(entry.input_tokens for entry in values),