cctally 1.99.0 → 1.100.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,6 +39,7 @@ import sys
39
39
  import _cctally_core
40
40
  import _lib_changelog
41
41
  import _lib_journal_router
42
+ import _lib_perf
42
43
  from _cctally_core import _now_utc, eprint, now_utc_iso, parse_iso_datetime
43
44
  from _lib_dashboard_json import encode_dashboard_json
44
45
 
@@ -892,11 +893,140 @@ def _gather_accounts_state(now_utc: "dt.datetime") -> dict:
892
893
  return state
893
894
 
894
895
 
896
+ # ── The quota-observation input memo (#583 S5 §2.3) ─────────────────────────
897
+ # Task 4's attribution measured `doctor.quota_summary` at a 1,197.7 ms median
898
+ # against a copy of the real store (five fresh samples after one untimed
899
+ # warm-up, range 1,157.3-1,319.8) -- 69% of the doctor's own 1,731.2 ms median
900
+ # and 20% of the BUILDER. The 5,907.1 ms denominator behind that share is the
901
+ # wall time of `_tui_build_snapshot`, which the harness calls with
902
+ # `skip_sync=True` and which runs entirely inside `tick.build_span()`, so it
903
+ # contains no ingest and is not a whole tick; a whole tick is ingest plus
904
+ # builder, and the builder is 70.5% of one measured live. Against a whole tick
905
+ # the same probe is therefore 14%. It is one all-history
906
+ # `load_codex_quota_observations` call whose SQL runs a `MAX(...) OVER
907
+ # (PARTITION BY ...)` across every retained row (277,207 on that store) to
908
+ # return one row per identity (16).
909
+ #
910
+ # Only the ROWS are retained, and only while their evidence is unmoved. Every
911
+ # `now`-derived interpretation over them is recomputed on every gather:
912
+ # `quota_freshness` and therefore `latest_capture_at`, `freshness_state`,
913
+ # `age_seconds` and `stale_after_seconds`. Those move with the clock alone, and
914
+ # retaining one would freeze a displayed health verdict at whatever instant the
915
+ # population happened to be loaded. This is the whole of Preserve 14's room --
916
+ # the memo caches a raw INPUT, never the payload, so the aggregate and its
917
+ # `generated_at` are still recomputed at every `DOCTOR_MEMO_TTL_S` expiry and
918
+ # `DoctorChip`'s rendered check time is never older than the TTL.
919
+ #
920
+ # At most ONE entry is retained: a moved signal replaces rather than
921
+ # accumulates, so the bound is one row per retained quota identity (16 on that
922
+ # store), not a multiple of anything. The dict is a plain module global with no
923
+ # `_assert_owner`, because the doctor gathers on the CLI's main thread as well
924
+ # as on the dashboard's rebuild thread and an owner assertion would refuse the
925
+ # CLI. That is safe here for the same reason it is for the source build's memo:
926
+ # a value is served only under an exactly-matching key, and the key IS the
927
+ # invalidation basis, so a racing writer can only cause a redundant cold read,
928
+ # never a stale answer.
929
+ _QUOTA_OBSERVATION_MEMO: "dict[tuple, tuple]" = {}
930
+
931
+
932
+ def _codex_quota_observation_signal() -> "tuple | None":
933
+ """The exact mutation stream that can change the quota probe's rows.
934
+
935
+ Returns ``None`` for "cannot establish identity", and every caller must
936
+ treat that as a cold read rather than as a cache hit. An absent or
937
+ unreadable change ledger is not an idle change ledger.
938
+
939
+ Four legs, all of them cache.db. ``quota_window_change_log``'s high-water
940
+ sequence covers every insert, delete and semantic update of
941
+ ``quota_window_snapshots``: the three ledger triggers in
942
+ ``bin/_cctally_db.py`` (``trg_qws_ledger_ins`` / ``_del`` / ``_upd``, the
943
+ last firing on the semantic column set the loader interprets) record each
944
+ one, and the sequence is read from ``sqlite_sequence`` so that
945
+ ``prune_ledger_through``'s ``DELETE`` cannot walk it backwards.
946
+ ``codex_window_attribution_revision`` covers the attribution overlay, which
947
+ the journal bumps in the same transaction as the rows it applies. The
948
+ database path and the file's identity together cover replacement: a
949
+ restored or repaired cache.db can carry a LOWER sequence than the one
950
+ already retained, so the path alone would serve rows from the superseded
951
+ file.
952
+
953
+ NO STATS.DB LEG, which is a deliberate difference from the source build's
954
+ six-leg ``_codex_quota_reuse_identity``. That identity carries three stats
955
+ digests because the build's quota consumers render account decoration. This
956
+ probe does not: every field it publishes -- the five identity coordinates
957
+ plus the freshness quadruple -- is derived from ``quota_window_snapshots``
958
+ and the clock, and ``load_codex_quota_observations`` is a cache.db
959
+ projection reader that is given no stats connection and opens none. Adding
960
+ the stats legs would be over-conservative rather than safer, and it would
961
+ surrender the reuse on every tick that publishes a new stats generation
962
+ while the quota evidence sat still. ``test_stats_side_decoration_does_not
963
+ _change_the_doctor_quota_probe`` is the executable form of that claim: if
964
+ the probe ever grows a stats-side dependency, that test fails rather than
965
+ this comment going quietly out of date.
966
+ """
967
+ path = _cctally_core.CACHE_DB_PATH
968
+ try:
969
+ stat = path.stat()
970
+ except OSError:
971
+ return None
972
+ try:
973
+ conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
974
+ except sqlite3.Error:
975
+ return None
976
+ try:
977
+ table = conn.execute(
978
+ "SELECT 1 FROM sqlite_master WHERE type='table' "
979
+ "AND name='quota_window_change_log'"
980
+ ).fetchone()
981
+ if table is None:
982
+ return None
983
+ seq_row = conn.execute(
984
+ "SELECT seq FROM sqlite_sequence "
985
+ "WHERE name='quota_window_change_log'"
986
+ ).fetchone()
987
+ revision_row = conn.execute(
988
+ "SELECT value FROM cache_meta "
989
+ "WHERE key='codex_window_attribution_revision'"
990
+ ).fetchone()
991
+ except sqlite3.Error:
992
+ return None
993
+ finally:
994
+ conn.close()
995
+ return (
996
+ str(path),
997
+ stat.st_dev,
998
+ stat.st_ino,
999
+ 0 if seq_row is None else int(seq_row[0]),
1000
+ "" if revision_row is None else str(revision_row[0]),
1001
+ )
1002
+
1003
+
1004
+ def _load_codex_quota_observations_for_doctor(*, force_cold: bool = False):
1005
+ """The probe's raw population, reused while its dependency signal holds."""
1006
+ c = _cctally()
1007
+ signal = None if force_cold else _codex_quota_observation_signal()
1008
+ if signal is not None:
1009
+ cached = _QUOTA_OBSERVATION_MEMO.get(signal)
1010
+ if cached is not None:
1011
+ return cached
1012
+ loaded = tuple(
1013
+ c._cctally_quota.load_codex_quota_observations(
1014
+ latest_per_identity=True,
1015
+ )
1016
+ )
1017
+ if signal is not None:
1018
+ # Replace rather than accumulate: the bound is one entry.
1019
+ _QUOTA_OBSERVATION_MEMO.clear()
1020
+ _QUOTA_OBSERVATION_MEMO[signal] = loaded
1021
+ return loaded
1022
+
1023
+
895
1024
  def doctor_gather_state(
896
1025
  *,
897
1026
  now_utc: "dt.datetime | None" = None,
898
1027
  runtime_bind: "str | None" = None,
899
1028
  deep: bool = False,
1029
+ force_cold_inputs: bool = False,
900
1030
  ):
901
1031
  """Gather doctor state while excluding cache-family replacement.
902
1032
 
@@ -905,6 +1035,11 @@ def doctor_gather_state(
905
1035
  through every cache probe. A live/stale marker or a pending quarantine
906
1036
  suppresses all raw SQLite cache opens; when a cache exists but its lock is
907
1037
  absent, probes also degrade rather than racing a newly starting repair.
1038
+
1039
+ ``force_cold_inputs`` bypasses every reused raw input (#583 S5 §2.3), so a
1040
+ caller can obtain a genuinely fresh gather at a chosen instant. It exists
1041
+ for the equality gate that compares a reusing gather against a fresh one at
1042
+ an identical ``now_utc``; production callers leave it false.
908
1043
  """
909
1044
  cache_lock = None
910
1045
  cache_probe_allowed = not _cctally_core.CACHE_DB_PATH.exists()
@@ -914,40 +1049,41 @@ def doctor_gather_state(
914
1049
  "reason": None,
915
1050
  }
916
1051
  try:
917
- lock_path = _cctally_core.CACHE_LOCK_MAINTENANCE_PATH
918
- if lock_path.exists():
919
- cache_lock = open(lock_path, "r")
920
- fcntl.flock(cache_lock, fcntl.LOCK_SH)
921
- cache_probe_allowed = True
922
-
923
- c = _cctally()
924
- db_mod = c._load_sibling("_cctally_db")
925
- repair_marker = db_mod._repair_marker_path(
926
- _cctally_core.CACHE_DB_PATH
927
- )
928
- pending = db_mod._quarantine_pending_path(
929
- _cctally_core.CACHE_DB_PATH
930
- )
931
- if repair_marker.exists():
932
- live, reason = db_mod._repair_marker_is_live(repair_marker)
933
- cache_repair_marker = {
934
- "exists": True,
935
- "live": live,
936
- "reason": reason,
937
- }
938
- cache_probe_allowed = False
939
- elif pending.exists():
940
- cache_repair_marker = {
941
- "exists": True,
942
- "live": False,
943
- "reason": "interrupted quarantine is pending",
944
- }
945
- cache_probe_allowed = False
946
-
947
- if not cache_probe_allowed and cache_lock is not None:
948
- fcntl.flock(cache_lock, fcntl.LOCK_UN)
949
- cache_lock.close()
950
- cache_lock = None
1052
+ with _lib_perf.phase("doctor.gate"):
1053
+ lock_path = _cctally_core.CACHE_LOCK_MAINTENANCE_PATH
1054
+ if lock_path.exists():
1055
+ cache_lock = open(lock_path, "r")
1056
+ fcntl.flock(cache_lock, fcntl.LOCK_SH)
1057
+ cache_probe_allowed = True
1058
+
1059
+ c = _cctally()
1060
+ db_mod = c._load_sibling("_cctally_db")
1061
+ repair_marker = db_mod._repair_marker_path(
1062
+ _cctally_core.CACHE_DB_PATH
1063
+ )
1064
+ pending = db_mod._quarantine_pending_path(
1065
+ _cctally_core.CACHE_DB_PATH
1066
+ )
1067
+ if repair_marker.exists():
1068
+ live, reason = db_mod._repair_marker_is_live(repair_marker)
1069
+ cache_repair_marker = {
1070
+ "exists": True,
1071
+ "live": live,
1072
+ "reason": reason,
1073
+ }
1074
+ cache_probe_allowed = False
1075
+ elif pending.exists():
1076
+ cache_repair_marker = {
1077
+ "exists": True,
1078
+ "live": False,
1079
+ "reason": "interrupted quarantine is pending",
1080
+ }
1081
+ cache_probe_allowed = False
1082
+
1083
+ if not cache_probe_allowed and cache_lock is not None:
1084
+ fcntl.flock(cache_lock, fcntl.LOCK_UN)
1085
+ cache_lock.close()
1086
+ cache_lock = None
951
1087
 
952
1088
  import _cctally_store
953
1089
 
@@ -958,6 +1094,7 @@ def doctor_gather_state(
958
1094
  deep=deep,
959
1095
  _cache_probe_allowed=cache_probe_allowed,
960
1096
  _cache_repair_marker=cache_repair_marker,
1097
+ _force_cold_inputs=force_cold_inputs,
961
1098
  )
962
1099
  finally:
963
1100
  if cache_lock is not None:
@@ -974,6 +1111,7 @@ def _doctor_gather_state_impl(
974
1111
  deep: bool = False,
975
1112
  _cache_probe_allowed: bool,
976
1113
  _cache_repair_marker: dict,
1114
+ _force_cold_inputs: bool = False,
977
1115
  ):
978
1116
  """I/O chokepoint for `cctally doctor` (spec §7.2).
979
1117
 
@@ -992,1192 +1130,1236 @@ def _doctor_gather_state_impl(
992
1130
  if now_utc is None:
993
1131
  now_utc = _now_utc()
994
1132
 
995
- backup_sync_state = _gather_backup_sync_state(
996
- _cctally_core.APP_DIR,
997
- probe_time_machine=deep,
998
- )
999
-
1000
- # ── Install ──────────────────────────────────────────────────────
1001
- # #279 S2 F5d: guard the only two unguarded statements in the
1002
- # otherwise fail-soft gather — an exception here would kill the whole
1003
- # report. Downstream consumers already degrade on None.
1004
- try:
1005
- repo_root = c._setup_resolve_repo_root()
1006
- except Exception:
1007
- repo_root = None
1008
- try:
1009
- dst_dir = c._setup_local_bin_dir()
1010
- except Exception:
1011
- dst_dir = None
1012
- try:
1013
- symlink_state = c._setup_compute_symlink_state(repo_root, dst_dir)
1014
- except Exception:
1015
- symlink_state = None
1016
- try:
1017
- path_includes = c._setup_path_includes_local_bin()
1018
- except Exception:
1019
- path_includes = None
1020
- # Issue #119: availability-aware install checks. Precomputed here (the
1021
- # I/O layer) so the kernel stays pure — `shutil.which` and the on-disk
1022
- # legacy-link probe never run in _lib_doctor.
1023
- # * cctally_reachable_on_path — channel-agnostic "is the command on
1024
- # $PATH at all?" (brew <prefix>/bin, npm prefix, source ~/.local/bin
1025
- # all satisfy it). Lets install.path pass without a ~/.local/bin
1026
- # membership check.
1027
- # * symlinks_path_pinned — true iff cctally runs ONLY through a legacy
1028
- # ~/.local/bin link to a retired/foreign install (live retired link
1029
- # with no reachable_elsewhere fallback). Mirrors the pinned-only-path
1030
- # predicate in _setup_install so doctor + setup agree on the fix.
1031
- try:
1032
- cctally_reachable_on_path = shutil.which("cctally") is not None
1033
- except Exception:
1034
- cctally_reachable_on_path = None
1035
- try:
1036
- symlinks_path_pinned = any(
1037
- s == "wrong"
1038
- and (dst_dir / n).is_symlink()
1039
- and c._setup_symlink_is_retired(dst_dir / n, n, repo_root)
1040
- and (dst_dir / n).resolve(strict=False).exists()
1041
- for n, s in (symlink_state or [])
1133
+ with _lib_perf.phase("doctor.backup_sync"):
1134
+ backup_sync_state = _gather_backup_sync_state(
1135
+ _cctally_core.APP_DIR,
1136
+ probe_time_machine=deep,
1042
1137
  )
1043
- except Exception:
1044
- symlinks_path_pinned = False
1045
- # install_is_brew — channel knowledge for the install.path WARN
1046
- # remediation. Brew kegs own no ~/.local/bin symlinks (#119), so the
1047
- # ~/.local/bin / `cctally setup` hint is wrong for them; the kernel
1048
- # can't derive this from repo_root (no I/O), so precompute it here.
1049
- try:
1050
- install_is_brew = c._setup_is_brew_install(repo_root)
1051
- except Exception:
1052
- install_is_brew = False
1053
- try:
1054
- legacy_snippet = c._setup_detect_legacy_snippet()
1055
- except Exception:
1056
- legacy_snippet = None
1057
1138
 
1058
- # ── Hooks ────────────────────────────────────────────────────────
1059
- try:
1060
- settings = c._load_claude_settings()
1061
- except c.SetupError:
1062
- settings = None
1063
- # #311: precompute the statusLine.refreshInterval state via the setup
1064
- # I/O-layer classifier (wrapper recognition does file scans), so the pure
1065
- # doctor kernel stays I/O-free. `settings is None` (SetupError) → the
1066
- # classifier's `unavailable`, matching the check's always-OK posture.
1067
- try:
1068
- statusline_refresh_state = c._classify_statusline_refresh(settings)[0]
1069
- except Exception:
1070
- statusline_refresh_state = "unavailable"
1071
- # Below: fail-soft posture for the diagnostic — any unexpected error
1072
- # in a sub-probe degrades that field to None rather than aborting the
1073
- # whole report.
1074
- try:
1075
- hook_counts = c._setup_count_hook_entries(settings or {})
1076
- except Exception:
1077
- hook_counts = None
1078
- try:
1079
- legacy_bespoke = c._setup_detect_legacy_bespoke_hooks(settings or {})
1080
- except Exception:
1081
- legacy_bespoke = None
1082
- try:
1083
- activity = c._setup_recent_log_stats()
1084
- except Exception:
1085
- activity = None
1139
+ with _lib_perf.phase("doctor.install"):
1140
+ # ── Install ──────────────────────────────────────────────────────
1141
+ # #279 S2 F5d: guard the only two unguarded statements in the
1142
+ # otherwise fail-soft gather — an exception here would kill the whole
1143
+ # report. Downstream consumers already degrade on None.
1144
+ try:
1145
+ repo_root = c._setup_resolve_repo_root()
1146
+ except Exception:
1147
+ repo_root = None
1148
+ try:
1149
+ dst_dir = c._setup_local_bin_dir()
1150
+ except Exception:
1151
+ dst_dir = None
1152
+ try:
1153
+ symlink_state = c._setup_compute_symlink_state(repo_root, dst_dir)
1154
+ except Exception:
1155
+ symlink_state = None
1156
+ try:
1157
+ path_includes = c._setup_path_includes_local_bin()
1158
+ except Exception:
1159
+ path_includes = None
1160
+ # Issue #119: availability-aware install checks. Precomputed here (the
1161
+ # I/O layer) so the kernel stays pure — `shutil.which` and the on-disk
1162
+ # legacy-link probe never run in _lib_doctor.
1163
+ # * cctally_reachable_on_path — channel-agnostic "is the command on
1164
+ # $PATH at all?" (brew <prefix>/bin, npm prefix, source ~/.local/bin
1165
+ # all satisfy it). Lets install.path pass without a ~/.local/bin
1166
+ # membership check.
1167
+ # * symlinks_path_pinned — true iff cctally runs ONLY through a legacy
1168
+ # ~/.local/bin link to a retired/foreign install (live retired link
1169
+ # with no reachable_elsewhere fallback). Mirrors the pinned-only-path
1170
+ # predicate in _setup_install so doctor + setup agree on the fix.
1171
+ try:
1172
+ cctally_reachable_on_path = shutil.which("cctally") is not None
1173
+ except Exception:
1174
+ cctally_reachable_on_path = None
1175
+ try:
1176
+ symlinks_path_pinned = any(
1177
+ s == "wrong"
1178
+ and (dst_dir / n).is_symlink()
1179
+ and c._setup_symlink_is_retired(dst_dir / n, n, repo_root)
1180
+ and (dst_dir / n).resolve(strict=False).exists()
1181
+ for n, s in (symlink_state or [])
1182
+ )
1183
+ except Exception:
1184
+ symlinks_path_pinned = False
1185
+ # install_is_brew — channel knowledge for the install.path WARN
1186
+ # remediation. Brew kegs own no ~/.local/bin symlinks (#119), so the
1187
+ # ~/.local/bin / `cctally setup` hint is wrong for them; the kernel
1188
+ # can't derive this from repo_root (no I/O), so precompute it here.
1189
+ try:
1190
+ install_is_brew = c._setup_is_brew_install(repo_root)
1191
+ except Exception:
1192
+ install_is_brew = False
1193
+ try:
1194
+ legacy_snippet = c._setup_detect_legacy_snippet()
1195
+ except Exception:
1196
+ legacy_snippet = None
1086
1197
 
1087
- # ── Auth ─────────────────────────────────────────────────────────
1088
- try:
1089
- oauth_token_present = c._setup_oauth_token_present()
1090
- except OSError:
1091
- oauth_token_present = None
1198
+ with _lib_perf.phase("doctor.hooks"):
1199
+ # ── Hooks ────────────────────────────────────────────────────────
1200
+ try:
1201
+ settings = c._load_claude_settings()
1202
+ except c.SetupError:
1203
+ settings = None
1204
+ # #311: precompute the statusLine.refreshInterval state via the setup
1205
+ # I/O-layer classifier (wrapper recognition does file scans), so the pure
1206
+ # doctor kernel stays I/O-free. `settings is None` (SetupError) → the
1207
+ # classifier's `unavailable`, matching the check's always-OK posture.
1208
+ try:
1209
+ statusline_refresh_state = c._classify_statusline_refresh(settings)[0]
1210
+ except Exception:
1211
+ statusline_refresh_state = "unavailable"
1212
+ # Below: fail-soft posture for the diagnostic — any unexpected error
1213
+ # in a sub-probe degrades that field to None rather than aborting the
1214
+ # whole report.
1215
+ try:
1216
+ hook_counts = c._setup_count_hook_entries(settings or {})
1217
+ except Exception:
1218
+ hook_counts = None
1219
+ try:
1220
+ legacy_bespoke = c._setup_detect_legacy_bespoke_hooks(settings or {})
1221
+ except Exception:
1222
+ legacy_bespoke = None
1223
+ try:
1224
+ activity = c._setup_recent_log_stats()
1225
+ except Exception:
1226
+ activity = None
1092
1227
 
1093
- # ── DB ───────────────────────────────────────────────────────────
1094
- try:
1095
- import _cctally_store
1228
+ with _lib_perf.phase("doctor.auth"):
1229
+ # ── Auth ─────────────────────────────────────────────────────────
1230
+ try:
1231
+ oauth_token_present = c._setup_oauth_token_present()
1232
+ except OSError:
1233
+ oauth_token_present = None
1096
1234
 
1097
- interrupted = _cctally_store.stats_interrupted_rebuild_evidence(
1098
- _cctally_core.DB_PATH
1099
- )
1100
- if interrupted is not None and interrupted.get("live") is True:
1101
- stats_db_status = {
1102
- "path": str(_cctally_core.DB_PATH),
1103
- "user_version": 0,
1104
- "registry_size": len(c._STATS_MIGRATIONS),
1105
- "migrations": [],
1106
- }
1107
- else:
1108
- stats_db_status = c._db_status_for(
1109
- _cctally_core.DB_PATH,
1110
- c._STATS_MIGRATIONS,
1111
- "stats.db",
1112
- recover_interrupted_stats=False,
1113
- )
1114
- if not _cctally_core.DB_PATH.exists():
1115
- stats_db_status["_file_exists"] = False
1116
- if interrupted is not None:
1117
- stats_db_status["_interrupted_rebuild"] = interrupted
1118
- except sqlite3.Error as exc:
1119
- stats_db_status = {"path": str(_cctally_core.DB_PATH), "user_version": 0,
1120
- "registry_size": len(c._STATS_MIGRATIONS),
1121
- "migrations": [], "_open_error": str(exc)}
1122
- # stats.db is the epoch-versioned journal index (DB journal redesign §7.1):
1123
- # feed the epoch constant to the pure kernel so db.version_ahead classifies
1124
- # uv==1000 as HEALTHY (not a #145 version-ahead FAIL). registry_size stays the
1125
- # frozen legacy head (13) and serves as the legacy-range boundary.
1126
- stats_db_status["epoch"] = _cctally_core.STATS_INDEX_EPOCH
1127
- if _cache_probe_allowed:
1235
+ with _lib_perf.phase("doctor.db_status"):
1236
+ # ── DB ───────────────────────────────────────────────────────────
1128
1237
  try:
1129
- cache_db_status = c._db_status_for(
1130
- _cctally_core.CACHE_DB_PATH,
1131
- c._CACHE_MIGRATIONS,
1132
- "cache.db",
1238
+ import _cctally_store
1239
+
1240
+ interrupted = _cctally_store.stats_interrupted_rebuild_evidence(
1241
+ _cctally_core.DB_PATH
1133
1242
  )
1134
- if not _cctally_core.CACHE_DB_PATH.exists():
1135
- cache_db_status["_file_exists"] = False
1243
+ if interrupted is not None and interrupted.get("live") is True:
1244
+ stats_db_status = {
1245
+ "path": str(_cctally_core.DB_PATH),
1246
+ "user_version": 0,
1247
+ "registry_size": len(c._STATS_MIGRATIONS),
1248
+ "migrations": [],
1249
+ }
1250
+ else:
1251
+ stats_db_status = c._db_status_for(
1252
+ _cctally_core.DB_PATH,
1253
+ c._STATS_MIGRATIONS,
1254
+ "stats.db",
1255
+ recover_interrupted_stats=False,
1256
+ )
1257
+ if not _cctally_core.DB_PATH.exists():
1258
+ stats_db_status["_file_exists"] = False
1259
+ if interrupted is not None:
1260
+ stats_db_status["_interrupted_rebuild"] = interrupted
1136
1261
  except sqlite3.Error as exc:
1262
+ stats_db_status = {"path": str(_cctally_core.DB_PATH), "user_version": 0,
1263
+ "registry_size": len(c._STATS_MIGRATIONS),
1264
+ "migrations": [], "_open_error": str(exc)}
1265
+ # stats.db is the epoch-versioned journal index (DB journal redesign §7.1):
1266
+ # feed the epoch constant to the pure kernel so db.version_ahead classifies
1267
+ # uv==1000 as HEALTHY (not a #145 version-ahead FAIL). registry_size stays the
1268
+ # frozen legacy head (13) and serves as the legacy-range boundary.
1269
+ stats_db_status["epoch"] = _cctally_core.STATS_INDEX_EPOCH
1270
+ if _cache_probe_allowed:
1271
+ try:
1272
+ cache_db_status = c._db_status_for(
1273
+ _cctally_core.CACHE_DB_PATH,
1274
+ c._CACHE_MIGRATIONS,
1275
+ "cache.db",
1276
+ )
1277
+ if not _cctally_core.CACHE_DB_PATH.exists():
1278
+ cache_db_status["_file_exists"] = False
1279
+ except sqlite3.Error as exc:
1280
+ cache_db_status = {
1281
+ "path": str(_cctally_core.CACHE_DB_PATH),
1282
+ "user_version": 0,
1283
+ "registry_size": len(c._CACHE_MIGRATIONS),
1284
+ "migrations": [],
1285
+ "_open_error": str(exc),
1286
+ }
1287
+ else:
1137
1288
  cache_db_status = {
1138
1289
  "path": str(_cctally_core.CACHE_DB_PATH),
1139
1290
  "user_version": 0,
1140
1291
  "registry_size": len(c._CACHE_MIGRATIONS),
1141
1292
  "migrations": [],
1142
- "_open_error": str(exc),
1293
+ "_open_error": "cache maintenance excludes read probes",
1143
1294
  }
1144
- else:
1145
- cache_db_status = {
1146
- "path": str(_cctally_core.CACHE_DB_PATH),
1147
- "user_version": 0,
1148
- "registry_size": len(c._CACHE_MIGRATIONS),
1149
- "migrations": [],
1150
- "_open_error": "cache maintenance excludes read probes",
1151
- }
1152
- cache_repair_marker = _cache_repair_marker
1295
+ cache_repair_marker = _cache_repair_marker
1153
1296
 
1154
- # ── Data freshness ───────────────────────────────────────────────
1155
- latest_snapshot_at = None
1156
- forked_bucket_counts: dict | None = None
1157
- credited_weeks: list[dict] | None = None
1158
- try:
1159
- if _cctally_core.DB_PATH.exists():
1160
- # #386 spec section 3.1, third clause: EVERY opener of the live stats
1161
- # family participates in the replacement protocol, read-only probes
1162
- # included. The open mode stays read-WRITE deliberately — switching a
1163
- # WAL DB whose `-shm` may be absent to `mode=ro` fails
1164
- # SQLITE_CANTOPEN, which is not corruption and has been misread as
1165
- # such on this project twice. Participation, not read-only-ness, is
1166
- # what the clause requires.
1167
- import _cctally_store as _store_mod
1168
- conn = _store_mod.stats_open_guarded(_cctally_core.DB_PATH)
1169
- try:
1297
+ with _lib_perf.phase("doctor.stats_freshness"):
1298
+ # ── Data freshness ───────────────────────────────────────────────
1299
+ latest_snapshot_at = None
1300
+ forked_bucket_counts: dict | None = None
1301
+ credited_weeks: list[dict] | None = None
1302
+ try:
1303
+ if _cctally_core.DB_PATH.exists():
1304
+ # #386 spec section 3.1, third clause: EVERY opener of the live stats
1305
+ # family participates in the replacement protocol, read-only probes
1306
+ # included. The open mode stays read-WRITE deliberately — switching a
1307
+ # WAL DB whose `-shm` may be absent to `mode=ro` fails
1308
+ # SQLITE_CANTOPEN, which is not corruption and has been misread as
1309
+ # such on this project twice. Participation, not read-only-ness, is
1310
+ # what the clause requires.
1311
+ import _cctally_store as _store_mod
1312
+ conn = _store_mod.stats_open_guarded(_cctally_core.DB_PATH)
1170
1313
  try:
1171
- row = conn.execute(
1172
- "SELECT MAX(captured_at_utc) FROM weekly_usage_snapshots"
1173
- ).fetchone()
1174
- if row and row[0]:
1175
- latest_snapshot_at = parse_iso_datetime(
1176
- row[0], "weekly_usage_snapshots.captured_at_utc",
1177
- ).astimezone(dt.timezone.utc)
1178
- except sqlite3.OperationalError:
1179
- pass # table missing — treat as no snapshots yet
1180
- # Forked-bucket invariant probe. Each fork count is
1181
- # a raw SELECT against the already-open connection —
1182
- # no bonus open_db() recursion. Tables missing →
1183
- # count 0 (legacy DBs without one of these tables
1184
- # are intact by definition for that table).
1185
- forked_bucket_counts = {}
1186
- for table, key in (
1187
- ("weekly_usage_snapshots", "usage"),
1188
- ("weekly_cost_snapshots", "cost"),
1189
- ("percent_milestones", "milestones"),
1190
- ):
1191
1314
  try:
1192
1315
  row = conn.execute(
1193
- f"SELECT COUNT(*) FROM {table} "
1194
- f" WHERE week_start_at IS NOT NULL "
1195
- f" AND week_start_date != substr(week_start_at, 1, 10)"
1316
+ "SELECT MAX(captured_at_utc) FROM weekly_usage_snapshots"
1196
1317
  ).fetchone()
1197
- forked_bucket_counts[key] = (
1198
- int(row[0]) if row and row[0] else 0
1199
- )
1318
+ if row and row[0]:
1319
+ latest_snapshot_at = parse_iso_datetime(
1320
+ row[0], "weekly_usage_snapshots.captured_at_utc",
1321
+ ).astimezone(dt.timezone.utc)
1200
1322
  except sqlite3.OperationalError:
1201
- forked_bucket_counts[key] = 0
1202
- # v1.7.2 credited-week tracking. For each week with a
1203
- # past-effective ``week_reset_events`` row, gather the
1204
- # latest weekly_percent + count of post-credit milestones.
1205
- # The check warns when latest_percent >= 1.0 AND
1206
- # post_credit_milestone_count == 0.
1207
- # unixepoch() normalizes the cross-offset comparison.
1208
- try:
1209
- credit_rows = conn.execute(
1210
- """
1211
- SELECT wre.id AS event_id,
1212
- wre.new_week_end_at AS end_at,
1213
- wre.effective_reset_at_utc AS effective
1214
- FROM week_reset_events wre
1215
- WHERE unixepoch(wre.effective_reset_at_utc)
1216
- <= unixepoch(?)
1217
- """,
1218
- (now_utc_iso(),),
1219
- ).fetchall()
1220
- credited_weeks = []
1221
- for cr in credit_rows:
1222
- end_at = cr[1]
1223
- evt_id = cr[0]
1224
- latest = conn.execute(
1225
- """
1226
- SELECT week_start_date, weekly_percent
1227
- FROM weekly_usage_snapshots
1228
- WHERE week_end_at = ?
1229
- ORDER BY captured_at_utc DESC, id DESC
1230
- LIMIT 1
1231
- """,
1232
- (end_at,),
1233
- ).fetchone()
1234
- if latest is None or latest[0] is None:
1235
- continue
1236
- ws = latest[0]
1237
- lp = float(latest[1] or 0.0)
1323
+ pass # table missing — treat as no snapshots yet
1324
+ # Forked-bucket invariant probe. Each fork count is
1325
+ # a raw SELECT against the already-open connection —
1326
+ # no bonus open_db() recursion. Tables missing →
1327
+ # count 0 (legacy DBs without one of these tables
1328
+ # are intact by definition for that table).
1329
+ forked_bucket_counts = {}
1330
+ for table, key in (
1331
+ ("weekly_usage_snapshots", "usage"),
1332
+ ("weekly_cost_snapshots", "cost"),
1333
+ ("percent_milestones", "milestones"),
1334
+ ):
1238
1335
  try:
1239
- mc_row = conn.execute(
1240
- "SELECT COUNT(*) FROM percent_milestones "
1241
- "WHERE week_start_date = ? AND reset_event_id = ?",
1242
- (ws, evt_id),
1336
+ row = conn.execute(
1337
+ f"SELECT COUNT(*) FROM {table} "
1338
+ f" WHERE week_start_at IS NOT NULL "
1339
+ f" AND week_start_date != substr(week_start_at, 1, 10)"
1243
1340
  ).fetchone()
1244
- mc = int(mc_row[0]) if mc_row and mc_row[0] else 0
1341
+ forked_bucket_counts[key] = (
1342
+ int(row[0]) if row and row[0] else 0
1343
+ )
1245
1344
  except sqlite3.OperationalError:
1246
- mc = 0
1247
- credited_weeks.append({
1248
- "week_start_date": ws,
1249
- "latest_weekly_percent": lp,
1250
- "post_credit_milestone_count": mc,
1251
- "event_id": evt_id,
1252
- })
1253
- except sqlite3.OperationalError:
1254
- # week_reset_events table missing — treat as no
1255
- # credited weeks (pre-feature DB).
1256
- credited_weeks = []
1257
- finally:
1258
- conn.close()
1259
- except Exception:
1260
- pass
1261
-
1262
- cache_entries_count = None
1263
- cache_last_entry_at = None
1264
- cache_db_page_count = None
1265
- cache_db_freelist_count = None
1266
- try:
1267
- if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1268
- conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1269
- try:
1270
- try:
1271
- row = conn.execute("PRAGMA page_count").fetchone()
1272
- if row and row[0] is not None:
1273
- cache_db_page_count = int(row[0])
1274
- row = conn.execute("PRAGMA freelist_count").fetchone()
1275
- if row and row[0] is not None:
1276
- cache_db_freelist_count = int(row[0])
1277
- except sqlite3.Error:
1278
- pass
1279
- row = conn.execute(
1280
- "SELECT COUNT(*), MAX(timestamp_utc) FROM session_entries"
1281
- ).fetchone()
1282
- if row:
1283
- cache_entries_count = int(row[0]) if row[0] is not None else 0
1284
- if row[1]:
1285
- cache_last_entry_at = parse_iso_datetime(
1286
- row[1], "session_entries.timestamp_utc",
1287
- ).astimezone(dt.timezone.utc)
1288
- except sqlite3.OperationalError:
1289
- pass # table missing — treat as zero
1290
- finally:
1291
- conn.close()
1292
- except Exception:
1293
- pass
1345
+ forked_bucket_counts[key] = 0
1346
+ # v1.7.2 credited-week tracking. For each week with a
1347
+ # past-effective ``week_reset_events`` row, gather the
1348
+ # latest weekly_percent + count of post-credit milestones.
1349
+ # The check warns when latest_percent >= 1.0 AND
1350
+ # post_credit_milestone_count == 0.
1351
+ # unixepoch() normalizes the cross-offset comparison.
1352
+ try:
1353
+ credit_rows = conn.execute(
1354
+ """
1355
+ SELECT wre.id AS event_id,
1356
+ wre.new_week_end_at AS end_at,
1357
+ wre.effective_reset_at_utc AS effective
1358
+ FROM week_reset_events wre
1359
+ WHERE unixepoch(wre.effective_reset_at_utc)
1360
+ <= unixepoch(?)
1361
+ """,
1362
+ (now_utc_iso(),),
1363
+ ).fetchall()
1364
+ credited_weeks = []
1365
+ for cr in credit_rows:
1366
+ end_at = cr[1]
1367
+ evt_id = cr[0]
1368
+ latest = conn.execute(
1369
+ """
1370
+ SELECT week_start_date, weekly_percent
1371
+ FROM weekly_usage_snapshots
1372
+ WHERE week_end_at = ?
1373
+ ORDER BY captured_at_utc DESC, id DESC
1374
+ LIMIT 1
1375
+ """,
1376
+ (end_at,),
1377
+ ).fetchone()
1378
+ if latest is None or latest[0] is None:
1379
+ continue
1380
+ ws = latest[0]
1381
+ lp = float(latest[1] or 0.0)
1382
+ try:
1383
+ mc_row = conn.execute(
1384
+ "SELECT COUNT(*) FROM percent_milestones "
1385
+ "WHERE week_start_date = ? AND reset_event_id = ?",
1386
+ (ws, evt_id),
1387
+ ).fetchone()
1388
+ mc = int(mc_row[0]) if mc_row and mc_row[0] else 0
1389
+ except sqlite3.OperationalError:
1390
+ mc = 0
1391
+ credited_weeks.append({
1392
+ "week_start_date": ws,
1393
+ "latest_weekly_percent": lp,
1394
+ "post_credit_milestone_count": mc,
1395
+ "event_id": evt_id,
1396
+ })
1397
+ except sqlite3.OperationalError:
1398
+ # week_reset_events table missing — treat as no
1399
+ # credited weeks (pre-feature DB).
1400
+ credited_weeks = []
1401
+ finally:
1402
+ conn.close()
1403
+ except Exception:
1404
+ pass
1294
1405
 
1295
- # ── Statusline candidate arbitration (#318) ──────────────────────
1296
- # This inspection is deliberately independent of SQLite mutation: marker
1297
- # mtime, candidate/control files, and tombstones are all read fail-soft.
1298
- # In particular it uses the scan-only candidate helper, never the reducer
1299
- # loader that prunes expired or malformed spool files.
1300
- try:
1301
- statusline_pipeline = _gather_statusline_pipeline(c, now_utc=now_utc)
1302
- except Exception:
1303
- statusline_pipeline = None
1304
-
1305
- # Conversation-sessions rollup consistency (#217 S1 / U9). Two cheap COUNTs
1306
- # (graceful None on a missing table / unreadable DB) + an in-progress signal
1307
- # so a transient mid-sync mismatch never WARNs. The in-progress signal is a
1308
- # NON-BLOCKING conversations flock probe (a writer mid-walk holds it) OR the
1309
- # presence of any pending reingest/split/backfill cache_meta flag — doctor
1310
- # stays read-only and never blocks on the lock.
1311
- conv_sessions_rollup_count = None
1312
- conv_messages_distinct_sessions = None
1313
- conv_rollup_sync_in_progress = False
1314
- conversations_db_page_count = None
1315
- conversations_db_freelist_count = None
1316
- codex_prune_refusals: list[dict] = []
1317
- try:
1318
- if _cctally_core.CONVERSATIONS_DB_PATH.exists():
1319
- # This gather also runs inside dashboard snapshot precompute. A
1320
- # transcript writer or recovery may hold an exclusive lock, so use
1321
- # the recovery-aware read-only zero-timeout probe: conversation
1322
- # health can degrade, but it must never delay core snapshot
1323
- # freshness (#320, #415).
1324
- with _conversation_ro_guarded(timeout=0.0) as conn:
1325
- if conn is None:
1326
- raise sqlite3.OperationalError(
1327
- "conversation store maintenance in progress"
1328
- )
1329
- try:
1330
- row = conn.execute("PRAGMA page_count").fetchone()
1331
- if row and row[0] is not None:
1332
- conversations_db_page_count = int(row[0])
1333
- row = conn.execute("PRAGMA freelist_count").fetchone()
1334
- if row and row[0] is not None:
1335
- conversations_db_freelist_count = int(row[0])
1336
- except sqlite3.Error:
1337
- pass
1338
- try:
1339
- row = conn.execute(
1340
- "SELECT COUNT(*) FROM conversation_sessions"
1341
- ).fetchone()
1342
- if row is not None:
1343
- conv_sessions_rollup_count = int(row[0])
1344
- except sqlite3.OperationalError:
1345
- pass # table absent (pre-rollup) — leave None
1406
+ with _lib_perf.phase("doctor.cache_probes"):
1407
+ cache_entries_count = None
1408
+ cache_last_entry_at = None
1409
+ cache_db_page_count = None
1410
+ cache_db_freelist_count = None
1411
+ try:
1412
+ if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1413
+ conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1346
1414
  try:
1415
+ try:
1416
+ row = conn.execute("PRAGMA page_count").fetchone()
1417
+ if row and row[0] is not None:
1418
+ cache_db_page_count = int(row[0])
1419
+ row = conn.execute("PRAGMA freelist_count").fetchone()
1420
+ if row and row[0] is not None:
1421
+ cache_db_freelist_count = int(row[0])
1422
+ except sqlite3.Error:
1423
+ pass
1347
1424
  row = conn.execute(
1348
- "SELECT COUNT(DISTINCT session_id) "
1349
- "FROM conversation_messages WHERE session_id IS NOT NULL"
1425
+ "SELECT COUNT(*), MAX(timestamp_utc) FROM session_entries"
1350
1426
  ).fetchone()
1351
- if row is not None:
1352
- conv_messages_distinct_sessions = int(row[0])
1427
+ if row:
1428
+ cache_entries_count = int(row[0]) if row[0] is not None else 0
1429
+ if row[1]:
1430
+ cache_last_entry_at = parse_iso_datetime(
1431
+ row[1], "session_entries.timestamp_utc",
1432
+ ).astimezone(dt.timezone.utc)
1353
1433
  except sqlite3.OperationalError:
1354
- pass
1355
- try:
1356
- import _cctally_cache as _cc_sib
1357
- row = conn.execute(
1358
- "SELECT value FROM cache_meta WHERE key=?",
1359
- (_cc_sib.CODEX_ORPHAN_PRUNE_REFUSED_KEY,),
1360
- ).fetchone()
1361
- if row and row[0]:
1362
- record = json.loads(row[0])
1363
- if isinstance(record, dict):
1364
- codex_prune_refusals.append(record)
1365
- except (sqlite3.OperationalError, ValueError, TypeError):
1366
- pass
1367
- # Pending reingest/split/backfill flags ⇒ a full sync hasn't yet
1368
- # reconciled the rollup. Read the canonical flag set from
1369
- # _cctally_cache so it stays in lockstep with the sync consumers.
1370
- try:
1371
- import _cctally_cache as _cc_sib # lazy sibling
1372
- flags = tuple(_cc_sib._TARGETED_DECLINE_FLAGS)
1373
- placeholders = ",".join("?" for _ in flags)
1374
- pend = conn.execute(
1375
- f"SELECT 1 FROM cache_meta WHERE key IN ({placeholders}) "
1376
- "LIMIT 1", flags).fetchone()
1377
- if pend is not None:
1378
- conv_rollup_sync_in_progress = True
1379
- except Exception:
1380
- pass
1381
- # Non-blocking flock probe: if a transcript writer/reingest holds the
1382
- # conversations.db lock, the rollup may be mid-recompute → in progress. We
1383
- # acquire LOCK_EX|LOCK_NB and immediately release; failure (held) is the
1384
- # signal. Never blocks (LOCK_NB), so doctor stays read-only + prompt.
1385
- if not conv_rollup_sync_in_progress:
1386
- lock_path = _cctally_core.CONVERSATIONS_LOCK_PATH
1387
- if lock_path is not None and pathlib.Path(lock_path).exists():
1388
- import fcntl as _fcntl
1389
- lock_fh = open(str(lock_path), "w")
1390
- try:
1391
- _fcntl.flock(lock_fh, _fcntl.LOCK_EX | _fcntl.LOCK_NB)
1392
- _fcntl.flock(lock_fh, _fcntl.LOCK_UN) # acquired ⇒ quiescent
1393
- except (BlockingIOError, OSError):
1394
- conv_rollup_sync_in_progress = True # held ⇒ writer mid-flight
1434
+ pass # table missing — treat as zero
1395
1435
  finally:
1396
- lock_fh.close()
1397
- except Exception:
1398
- pass
1436
+ conn.close()
1437
+ except Exception:
1438
+ pass
1399
1439
 
1400
- claude_jsonl_present = False
1401
- try:
1402
- claude_dir = pathlib.Path.home() / ".claude" / "projects"
1403
- if claude_dir.exists():
1404
- claude_jsonl_present = next(claude_dir.glob("**/*.jsonl"), None) is not None
1405
- except Exception:
1406
- pass
1440
+ with _lib_perf.phase("doctor.statusline"):
1441
+ # ── Statusline candidate arbitration (#318) ──────────────────────
1442
+ # This inspection is deliberately independent of SQLite mutation: marker
1443
+ # mtime, candidate/control files, and tombstones are all read fail-soft.
1444
+ # In particular it uses the scan-only candidate helper, never the reducer
1445
+ # loader that prunes expired or malformed spool files.
1446
+ try:
1447
+ statusline_pipeline = _gather_statusline_pipeline(c, now_utc=now_utc)
1448
+ except Exception:
1449
+ statusline_pipeline = None
1450
+
1451
+ with _lib_perf.phase("doctor.conversations"):
1452
+ # Conversation-sessions rollup consistency (#217 S1 / U9). Two cheap COUNTs
1453
+ # (graceful None on a missing table / unreadable DB) + an in-progress signal
1454
+ # so a transient mid-sync mismatch never WARNs. The in-progress signal is a
1455
+ # NON-BLOCKING conversations flock probe (a writer mid-walk holds it) OR the
1456
+ # presence of any pending reingest/split/backfill cache_meta flag — doctor
1457
+ # stays read-only and never blocks on the lock.
1458
+ conv_sessions_rollup_count = None
1459
+ conv_messages_distinct_sessions = None
1460
+ conv_rollup_sync_in_progress = False
1461
+ conversations_db_page_count = None
1462
+ conversations_db_freelist_count = None
1463
+ codex_prune_refusals: list[dict] = []
1464
+ try:
1465
+ if _cctally_core.CONVERSATIONS_DB_PATH.exists():
1466
+ # This gather also runs inside dashboard snapshot precompute. A
1467
+ # transcript writer or recovery may hold an exclusive lock, so use
1468
+ # the recovery-aware read-only zero-timeout probe: conversation
1469
+ # health can degrade, but it must never delay core snapshot
1470
+ # freshness (#320, #415).
1471
+ with _conversation_ro_guarded(timeout=0.0) as conn:
1472
+ if conn is None:
1473
+ raise sqlite3.OperationalError(
1474
+ "conversation store maintenance in progress"
1475
+ )
1476
+ try:
1477
+ row = conn.execute("PRAGMA page_count").fetchone()
1478
+ if row and row[0] is not None:
1479
+ conversations_db_page_count = int(row[0])
1480
+ row = conn.execute("PRAGMA freelist_count").fetchone()
1481
+ if row and row[0] is not None:
1482
+ conversations_db_freelist_count = int(row[0])
1483
+ except sqlite3.Error:
1484
+ pass
1485
+ try:
1486
+ row = conn.execute(
1487
+ "SELECT COUNT(*) FROM conversation_sessions"
1488
+ ).fetchone()
1489
+ if row is not None:
1490
+ conv_sessions_rollup_count = int(row[0])
1491
+ except sqlite3.OperationalError:
1492
+ pass # table absent (pre-rollup) — leave None
1493
+ try:
1494
+ row = conn.execute(
1495
+ "SELECT COUNT(DISTINCT session_id) "
1496
+ "FROM conversation_messages WHERE session_id IS NOT NULL"
1497
+ ).fetchone()
1498
+ if row is not None:
1499
+ conv_messages_distinct_sessions = int(row[0])
1500
+ except sqlite3.OperationalError:
1501
+ pass
1502
+ try:
1503
+ import _cctally_cache as _cc_sib
1504
+ row = conn.execute(
1505
+ "SELECT value FROM cache_meta WHERE key=?",
1506
+ (_cc_sib.CODEX_ORPHAN_PRUNE_REFUSED_KEY,),
1507
+ ).fetchone()
1508
+ if row and row[0]:
1509
+ record = json.loads(row[0])
1510
+ if isinstance(record, dict):
1511
+ codex_prune_refusals.append(record)
1512
+ except (sqlite3.OperationalError, ValueError, TypeError):
1513
+ pass
1514
+ # Pending reingest/split/backfill flags ⇒ a full sync hasn't yet
1515
+ # reconciled the rollup. Read the canonical flag set from
1516
+ # _cctally_cache so it stays in lockstep with the sync consumers.
1517
+ try:
1518
+ import _cctally_cache as _cc_sib # lazy sibling
1519
+ flags = tuple(_cc_sib._TARGETED_DECLINE_FLAGS)
1520
+ placeholders = ",".join("?" for _ in flags)
1521
+ pend = conn.execute(
1522
+ f"SELECT 1 FROM cache_meta WHERE key IN ({placeholders}) "
1523
+ "LIMIT 1", flags).fetchone()
1524
+ if pend is not None:
1525
+ conv_rollup_sync_in_progress = True
1526
+ except Exception:
1527
+ pass
1528
+ # Non-blocking flock probe: if a transcript writer/reingest holds the
1529
+ # conversations.db lock, the rollup may be mid-recompute → in progress. We
1530
+ # acquire LOCK_EX|LOCK_NB and immediately release; failure (held) is the
1531
+ # signal. Never blocks (LOCK_NB), so doctor stays read-only + prompt.
1532
+ if not conv_rollup_sync_in_progress:
1533
+ lock_path = _cctally_core.CONVERSATIONS_LOCK_PATH
1534
+ if lock_path is not None and pathlib.Path(lock_path).exists():
1535
+ import fcntl as _fcntl
1536
+ lock_fh = open(str(lock_path), "w")
1537
+ try:
1538
+ _fcntl.flock(lock_fh, _fcntl.LOCK_EX | _fcntl.LOCK_NB)
1539
+ _fcntl.flock(lock_fh, _fcntl.LOCK_UN) # acquired ⇒ quiescent
1540
+ except (BlockingIOError, OSError):
1541
+ conv_rollup_sync_in_progress = True # held ⇒ writer mid-flight
1542
+ finally:
1543
+ lock_fh.close()
1544
+ except Exception:
1545
+ pass
1407
1546
 
1408
- codex_entries_count = None
1409
- codex_last_entry_at = None
1410
- codex_project_metadata_health = None
1411
- codex_project_metadata_error = None
1412
- codex_null_reset_anchors = 0
1413
- try:
1414
- if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1415
- conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1416
- try:
1417
- row = conn.execute(
1418
- "SELECT COUNT(*), MAX(timestamp_utc) FROM codex_session_entries"
1419
- ).fetchone()
1420
- if row:
1421
- codex_entries_count = int(row[0]) if row[0] is not None else 0
1422
- if row[1]:
1423
- codex_last_entry_at = parse_iso_datetime(
1424
- row[1], "codex_session_entries.timestamp_utc",
1425
- ).astimezone(dt.timezone.utc)
1547
+ with _lib_perf.phase("doctor.claude_files"):
1548
+ claude_jsonl_present = False
1549
+ try:
1550
+ claude_dir = pathlib.Path.home() / ".claude" / "projects"
1551
+ if claude_dir.exists():
1552
+ claude_jsonl_present = next(claude_dir.glob("**/*.jsonl"), None) is not None
1553
+ except Exception:
1554
+ pass
1555
+
1556
+ with _lib_perf.phase("doctor.codex_cache"):
1557
+ codex_entries_count = None
1558
+ codex_last_entry_at = None
1559
+ codex_project_metadata_health = None
1560
+ codex_project_metadata_error = None
1561
+ codex_null_reset_anchors = 0
1562
+ try:
1563
+ if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1564
+ conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1426
1565
  try:
1427
1566
  row = conn.execute(
1428
- "SELECT COUNT(*) FROM quota_window_snapshots "
1429
- "WHERE source = 'codex' "
1430
- "AND canonical_resets_at_utc IS NULL"
1567
+ "SELECT COUNT(*), MAX(timestamp_utc) FROM codex_session_entries"
1431
1568
  ).fetchone()
1432
- if row and row[0] is not None:
1433
- codex_null_reset_anchors = int(row[0])
1434
- except sqlite3.OperationalError:
1435
- # Pre-anchor cache shapes have no column to inspect. Their
1436
- # pending migration is reported by the DB checks instead.
1437
- pass
1438
- # Keep the health probe on the existing read-only cache
1439
- # connection. A failed probe is health evidence, not an
1440
- # empty corpus: the kernel renders it as a distinct FAIL.
1441
- try:
1442
- import _cctally_source_analytics
1569
+ if row:
1570
+ codex_entries_count = int(row[0]) if row[0] is not None else 0
1571
+ if row[1]:
1572
+ codex_last_entry_at = parse_iso_datetime(
1573
+ row[1], "codex_session_entries.timestamp_utc",
1574
+ ).astimezone(dt.timezone.utc)
1575
+ try:
1576
+ row = conn.execute(
1577
+ "SELECT COUNT(*) FROM quota_window_snapshots "
1578
+ "WHERE source = 'codex' "
1579
+ "AND canonical_resets_at_utc IS NULL"
1580
+ ).fetchone()
1581
+ if row and row[0] is not None:
1582
+ codex_null_reset_anchors = int(row[0])
1583
+ except sqlite3.OperationalError:
1584
+ # Pre-anchor cache shapes have no column to inspect. Their
1585
+ # pending migration is reported by the DB checks instead.
1586
+ pass
1587
+ # Keep the health probe on the existing read-only cache
1588
+ # connection. A failed probe is health evidence, not an
1589
+ # empty corpus: the kernel renders it as a distinct FAIL.
1590
+ try:
1591
+ import _cctally_source_analytics
1443
1592
 
1444
- health = _cctally_source_analytics.load_codex_project_metadata_health(
1445
- cache_conn=conn,
1446
- )
1447
- codex_project_metadata_health = {
1448
- "total_rows": health.total_rows,
1449
- "qualified_rows": health.qualified_rows,
1450
- "missing_conversation_key_rows": health.missing_conversation_key_rows,
1451
- "missing_thread_join_rows": health.missing_thread_join_rows,
1452
- }
1453
- except Exception as exc:
1593
+ health = _cctally_source_analytics.load_codex_project_metadata_health(
1594
+ cache_conn=conn,
1595
+ )
1596
+ codex_project_metadata_health = {
1597
+ "total_rows": health.total_rows,
1598
+ "qualified_rows": health.qualified_rows,
1599
+ "missing_conversation_key_rows": health.missing_conversation_key_rows,
1600
+ "missing_thread_join_rows": health.missing_thread_join_rows,
1601
+ }
1602
+ except Exception as exc:
1603
+ codex_project_metadata_error = type(exc).__name__
1604
+ except sqlite3.OperationalError as exc:
1605
+ # Pre-Codex cache shapes still produce the established Codex
1606
+ # cache result, while the new health check fails explicitly.
1454
1607
  codex_project_metadata_error = type(exc).__name__
1455
- except sqlite3.OperationalError as exc:
1456
- # Pre-Codex cache shapes still produce the established Codex
1457
- # cache result, while the new health check fails explicitly.
1458
- codex_project_metadata_error = type(exc).__name__
1459
- finally:
1460
- conn.close()
1461
- except Exception as exc:
1462
- codex_project_metadata_error = type(exc).__name__
1463
-
1464
- # Issue #109: probe every $CODEX_HOME session root (not the single
1465
- # hardcoded ~/.codex/sessions), matching the multi-root ingestion path
1466
- # from #108. _codex_session_roots() already applies the sessions/-subdir
1467
- # rule and filters to existing dirs, so a bare glob per root suffices.
1468
- codex_jsonl_present = False
1469
- try:
1470
- for codex_dir in c._codex_session_roots():
1471
- if next(codex_dir.glob("**/*.jsonl"), None) is not None:
1472
- codex_jsonl_present = True
1473
- break
1474
- except Exception:
1475
- pass
1608
+ finally:
1609
+ conn.close()
1610
+ except Exception as exc:
1611
+ codex_project_metadata_error = type(exc).__name__
1612
+
1613
+ with _lib_perf.phase("doctor.codex_files"):
1614
+ # Issue #109: probe every $CODEX_HOME session root (not the single
1615
+ # hardcoded ~/.codex/sessions), matching the multi-root ingestion path
1616
+ # from #108. _codex_session_roots() already applies the sessions/-subdir
1617
+ # rule and filters to existing dirs, so a bare glob per root suffices.
1618
+ codex_jsonl_present = False
1619
+ try:
1620
+ for codex_dir in c._codex_session_roots():
1621
+ if next(codex_dir.glob("**/*.jsonl"), None) is not None:
1622
+ codex_jsonl_present = True
1623
+ break
1624
+ except Exception:
1625
+ pass
1476
1626
 
1477
- # ── Codex quota lifecycle (#294 S2) ──────────────────────────────
1478
- # All three probes are read-only and root-qualified. The physical cache
1479
- # adapter preserves S1's per-window degradation, while setup's existing
1480
- # inspector supplies the exact owned-hook state without exposing paths.
1481
- codex_quota_windows: list[dict] = []
1482
- try:
1483
- # #566 §5.1 item 5: the population is unchanged — all history, every
1484
- # root, no row cap — but the read returns each identity's latest
1485
- # physical capture instead of every retained row, because that is the
1486
- # only thing this probe consumes. On the maintainer's store the former
1487
- # shape interpreted 266,337 rows to answer a question about 608 windows
1488
- # and cost about 2.7s of every dashboard build. Nothing here may bound
1489
- # the range: doing so would drop old and inactive-root identities and
1490
- # silently change `window_count`, the responsible identity and the
1491
- # WARN/OK verdict.
1492
- observations = (
1493
- c._cctally_quota.load_codex_quota_observations(
1494
- latest_per_identity=True,
1627
+ with _lib_perf.phase("doctor.quota_summary"):
1628
+ # ── Codex quota lifecycle (#294 S2) ──────────────────────────────
1629
+ # All three probes are read-only and root-qualified. The physical cache
1630
+ # adapter preserves S1's per-window degradation, while setup's existing
1631
+ # inspector supplies the exact owned-hook state without exposing paths.
1632
+ codex_quota_windows: list[dict] = []
1633
+ try:
1634
+ # #566 §5.1 item 5: the population is unchanged — all history, every
1635
+ # root, no row cap — but the read returns each identity's latest
1636
+ # physical capture instead of every retained row, because that is the
1637
+ # only thing this probe consumes. On the maintainer's store the former
1638
+ # shape interpreted 266,337 rows to answer a question about 608 windows
1639
+ # and cost about 2.7s of every dashboard build. Nothing here may bound
1640
+ # the range: doing so would drop old and inactive-root identities and
1641
+ # silently change `window_count`, the responsible identity and the
1642
+ # WARN/OK verdict.
1643
+ # #583 S5 §2.3: the population is reused while its dependency
1644
+ # signal holds, and every `now`-derived interpretation below is
1645
+ # recomputed regardless. A suppressed probe reads nothing and is
1646
+ # never served from the memo, because `_cache_probe_allowed` gates
1647
+ # the call rather than the load.
1648
+ observations = (
1649
+ _load_codex_quota_observations_for_doctor(
1650
+ force_cold=_force_cold_inputs,
1651
+ )
1652
+ if _cache_probe_allowed
1653
+ else ()
1495
1654
  )
1496
- if _cache_probe_allowed
1497
- else ()
1498
- )
1499
- by_identity: dict[object, list] = {}
1500
- for observation in observations:
1501
- by_identity.setdefault(observation.identity, []).append(observation)
1502
- for identity in sorted(
1503
- by_identity,
1504
- key=lambda item: (
1505
- item.source, item.source_root_key, item.logical_limit_key,
1506
- item.observed_slot, item.window_minutes,
1507
- ),
1508
- ):
1509
- freshness = c.quota_freshness(by_identity[identity], now_utc)
1510
- codex_quota_windows.append({
1511
- "identity": {
1512
- "source": identity.source,
1513
- "source_root_key": identity.source_root_key,
1514
- "logical_limit_key": identity.logical_limit_key,
1515
- "observed_slot": identity.observed_slot,
1516
- "window_minutes": identity.window_minutes,
1517
- },
1518
- "latest_capture_at": freshness.captured_at,
1519
- "freshness_state": freshness.state,
1520
- "age_seconds": freshness.age_seconds,
1521
- "stale_after_seconds": freshness.stale_after_seconds,
1522
- })
1523
- except Exception:
1524
- codex_quota_windows = []
1655
+ by_identity: dict[object, list] = {}
1656
+ for observation in observations:
1657
+ by_identity.setdefault(observation.identity, []).append(observation)
1658
+ for identity in sorted(
1659
+ by_identity,
1660
+ key=lambda item: (
1661
+ item.source, item.source_root_key, item.logical_limit_key,
1662
+ item.observed_slot, item.window_minutes,
1663
+ ),
1664
+ ):
1665
+ freshness = c.quota_freshness(by_identity[identity], now_utc)
1666
+ codex_quota_windows.append({
1667
+ "identity": {
1668
+ "source": identity.source,
1669
+ "source_root_key": identity.source_root_key,
1670
+ "logical_limit_key": identity.logical_limit_key,
1671
+ "observed_slot": identity.observed_slot,
1672
+ "window_minutes": identity.window_minutes,
1673
+ },
1674
+ "latest_capture_at": freshness.captured_at,
1675
+ "freshness_state": freshness.state,
1676
+ "age_seconds": freshness.age_seconds,
1677
+ "stale_after_seconds": freshness.stale_after_seconds,
1678
+ })
1679
+ except Exception:
1680
+ codex_quota_windows = []
1525
1681
 
1526
- codex_hook_roots: list[dict] = []
1527
- try:
1528
- codex_binary = str(c._setup_resolve_hook_target(repo_root))
1529
- hook_rows = [
1530
- c._cctally_setup._codex_hook_row(root, codex_binary)
1531
- for root in c._setup_codex_hook_roots()
1532
- ]
1533
- codex_hook_roots = [
1534
- {"source_root_key": row["source_root_key"], "state": row["state"]}
1535
- for row in sorted(hook_rows, key=lambda row: row["source_root_key"])
1536
- ]
1537
- except Exception:
1538
- codex_hook_roots = []
1682
+ with _lib_perf.phase("doctor.codex_hooks"):
1683
+ codex_hook_roots: list[dict] = []
1684
+ try:
1685
+ codex_binary = str(c._setup_resolve_hook_target(repo_root))
1686
+ hook_rows = [
1687
+ c._cctally_setup._codex_hook_row(root, codex_binary)
1688
+ for root in c._setup_codex_hook_roots()
1689
+ ]
1690
+ codex_hook_roots = [
1691
+ {"source_root_key": row["source_root_key"], "state": row["state"]}
1692
+ for row in sorted(hook_rows, key=lambda row: row["source_root_key"])
1693
+ ]
1694
+ except Exception:
1695
+ codex_hook_roots = []
1539
1696
 
1540
- try:
1541
- codex_lifecycle_activity_24h = _codex_lifecycle_activity_24h(
1542
- root_keys={row["source_root_key"] for row in codex_hook_roots},
1543
- now_utc=now_utc,
1544
- )
1545
- except Exception:
1546
- codex_lifecycle_activity_24h = {}
1697
+ with _lib_perf.phase("doctor.codex_lifecycle"):
1698
+ try:
1699
+ codex_lifecycle_activity_24h = _codex_lifecycle_activity_24h(
1700
+ root_keys={row["source_root_key"] for row in codex_hook_roots},
1701
+ now_utc=now_utc,
1702
+ )
1703
+ except Exception:
1704
+ codex_lifecycle_activity_24h = {}
1547
1705
 
1548
- try:
1549
- codex_quota_verify_activity = _codex_quota_verify_activity_24h(
1550
- now_utc=now_utc)
1551
- except Exception:
1552
- codex_quota_verify_activity = None
1553
-
1554
- # ── Parse health (#279 S2 F5a) ───────────────────────────────────
1555
- parse_health_claude = parse_health_codex = None
1556
- # #416 review B4: the durable record that a torn Codex `auth.json` halted
1557
- # ingest. Same cache_meta read, same degrade-to-None-on-anything contract.
1558
- codex_torn_deferred = None
1559
- # The byte-zero Codex replay stall signal. The marker itself is a bare "1";
1560
- # the sibling `blocked` record is the JSON one, so it is read through the
1561
- # same loop while the marker gets a plain existence probe. Key names come
1562
- # from the kernel constants, never inline literals.
1563
- codex_replay_pending = None
1564
- codex_replay_blocked = None
1565
- # public #5: the budgeted-decline record. Same JSON-dict contract as the
1566
- # blocked one, and the only signal a hook-only install produces when its
1567
- # Codex ingest is frozen behind an un-runnable replay.
1568
- codex_replay_deferred = None
1569
- # public #5 spec §5: the hook's budgeted-ingest backlog record. Absent means
1570
- # a zero backlog — a drained walk DELETES the row rather than zeroing it, so
1571
- # None and "nothing owed" are the same state by construction.
1572
- codex_ingest_backlog = None
1573
- try:
1574
- import _lib_codex_conversation as _codex_kern
1575
- _blocked_key = _codex_kern.CODEX_REPLAY_BLOCKED_KEY
1576
- _pending_key = _codex_kern.CODEX_REPLAY_FROM_ZERO_KEY
1577
- _deferred_key = _codex_kern.CODEX_REPLAY_DEFERRED_KEY
1578
- except Exception:
1579
- _blocked_key = "codex_replay_from_zero_blocked"
1580
- _pending_key = "codex_replay_from_zero_pending"
1581
- _deferred_key = "codex_replay_from_zero_deferred"
1582
- try:
1583
- if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1584
- conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1585
- try:
1586
- for _key in ("parse_health_claude", "parse_health_codex",
1587
- "codex_torn_auth_deferred", _blocked_key,
1588
- _deferred_key, "codex_ingest_backlog",
1589
- "codex_orphan_prune_refused"):
1706
+ try:
1707
+ codex_quota_verify_activity = _codex_quota_verify_activity_24h(
1708
+ now_utc=now_utc)
1709
+ except Exception:
1710
+ codex_quota_verify_activity = None
1711
+
1712
+ with _lib_perf.phase("doctor.parse_health"):
1713
+ # ── Parse health (#279 S2 F5a) ───────────────────────────────────
1714
+ parse_health_claude = parse_health_codex = None
1715
+ # #416 review B4: the durable record that a torn Codex `auth.json` halted
1716
+ # ingest. Same cache_meta read, same degrade-to-None-on-anything contract.
1717
+ codex_torn_deferred = None
1718
+ # The byte-zero Codex replay stall signal. The marker itself is a bare "1";
1719
+ # the sibling `blocked` record is the JSON one, so it is read through the
1720
+ # same loop while the marker gets a plain existence probe. Key names come
1721
+ # from the kernel constants, never inline literals.
1722
+ codex_replay_pending = None
1723
+ codex_replay_blocked = None
1724
+ # public #5: the budgeted-decline record. Same JSON-dict contract as the
1725
+ # blocked one, and the only signal a hook-only install produces when its
1726
+ # Codex ingest is frozen behind an un-runnable replay.
1727
+ codex_replay_deferred = None
1728
+ # public #5 spec §5: the hook's budgeted-ingest backlog record. Absent means
1729
+ # a zero backlog — a drained walk DELETES the row rather than zeroing it, so
1730
+ # None and "nothing owed" are the same state by construction.
1731
+ codex_ingest_backlog = None
1732
+ try:
1733
+ import _lib_codex_conversation as _codex_kern
1734
+ _blocked_key = _codex_kern.CODEX_REPLAY_BLOCKED_KEY
1735
+ _pending_key = _codex_kern.CODEX_REPLAY_FROM_ZERO_KEY
1736
+ _deferred_key = _codex_kern.CODEX_REPLAY_DEFERRED_KEY
1737
+ except Exception:
1738
+ _blocked_key = "codex_replay_from_zero_blocked"
1739
+ _pending_key = "codex_replay_from_zero_pending"
1740
+ _deferred_key = "codex_replay_from_zero_deferred"
1741
+ try:
1742
+ if _cache_probe_allowed and _cctally_core.CACHE_DB_PATH.exists():
1743
+ conn = sqlite3.connect(str(_cctally_core.CACHE_DB_PATH))
1744
+ try:
1745
+ for _key in ("parse_health_claude", "parse_health_codex",
1746
+ "codex_torn_auth_deferred", _blocked_key,
1747
+ _deferred_key, "codex_ingest_backlog",
1748
+ "codex_orphan_prune_refused"):
1749
+ try:
1750
+ row = conn.execute(
1751
+ "SELECT value FROM cache_meta WHERE key = ?",
1752
+ (_key,),
1753
+ ).fetchone()
1754
+ if row and row[0]:
1755
+ _parsed = json.loads(row[0])
1756
+ if isinstance(_parsed, dict):
1757
+ if _key == "parse_health_claude":
1758
+ parse_health_claude = _parsed
1759
+ elif _key == "parse_health_codex":
1760
+ parse_health_codex = _parsed
1761
+ elif _key == _blocked_key:
1762
+ codex_replay_blocked = _parsed
1763
+ elif _key == _deferred_key:
1764
+ codex_replay_deferred = _parsed
1765
+ elif _key == "codex_ingest_backlog":
1766
+ codex_ingest_backlog = _parsed
1767
+ elif _key == "codex_orphan_prune_refused":
1768
+ codex_prune_refusals.append(_parsed)
1769
+ else:
1770
+ codex_torn_deferred = _parsed
1771
+ except (sqlite3.OperationalError, ValueError):
1772
+ pass
1590
1773
  try:
1591
- row = conn.execute(
1592
- "SELECT value FROM cache_meta WHERE key = ?",
1593
- (_key,),
1594
- ).fetchone()
1595
- if row and row[0]:
1596
- _parsed = json.loads(row[0])
1597
- if isinstance(_parsed, dict):
1598
- if _key == "parse_health_claude":
1599
- parse_health_claude = _parsed
1600
- elif _key == "parse_health_codex":
1601
- parse_health_codex = _parsed
1602
- elif _key == _blocked_key:
1603
- codex_replay_blocked = _parsed
1604
- elif _key == _deferred_key:
1605
- codex_replay_deferred = _parsed
1606
- elif _key == "codex_ingest_backlog":
1607
- codex_ingest_backlog = _parsed
1608
- elif _key == "codex_orphan_prune_refused":
1609
- codex_prune_refusals.append(_parsed)
1610
- else:
1611
- codex_torn_deferred = _parsed
1612
- except (sqlite3.OperationalError, ValueError):
1774
+ codex_replay_pending = conn.execute(
1775
+ "SELECT 1 FROM cache_meta WHERE key = ?",
1776
+ (_pending_key,),
1777
+ ).fetchone() is not None
1778
+ except sqlite3.OperationalError:
1613
1779
  pass
1614
- try:
1615
- codex_replay_pending = conn.execute(
1616
- "SELECT 1 FROM cache_meta WHERE key = ?",
1617
- (_pending_key,),
1618
- ).fetchone() is not None
1619
- except sqlite3.OperationalError:
1620
- pass
1621
- finally:
1622
- conn.close()
1623
- except Exception:
1624
- pass
1780
+ finally:
1781
+ conn.close()
1782
+ except Exception:
1783
+ pass
1625
1784
 
1626
- # ── Integrity (deep only — #279 S2 F5b) ──────────────────────────
1627
- stats_db_quick_check = cache_db_quick_check = None
1628
- conversations_db_quick_check = None
1629
- if deep:
1630
- for _label, _path in (("stats", _cctally_core.DB_PATH),
1631
- ("cache", _cctally_core.CACHE_DB_PATH),
1632
- ("conversations",
1633
- _cctally_core.CONVERSATIONS_DB_PATH)):
1634
- _result = None
1635
- try:
1636
- if (
1637
- (
1638
- _path.exists()
1639
- or (
1640
- _label == "conversations"
1641
- and _path.with_name(
1642
- f"{_path.name}.recovery.json"
1643
- ).exists()
1644
- )
1645
- )
1646
- and (_label != "cache" or _cache_probe_allowed)
1647
- ):
1648
- # #386: the stats leg holds a read-write handle for the whole
1649
- # of a full quick_check — the longest-lived stats handle any
1650
- # diagnostic takes — so it participates in the replacement
1651
- # protocol. The cache leg keeps its own opener.
1652
- if _label == "stats":
1653
- import _cctally_store as _store_mod
1654
- _conn_ctx = contextlib.closing(
1655
- _store_mod.stats_open_guarded(_path)
1656
- )
1657
- elif _label == "cache":
1658
- _conn_ctx = contextlib.closing(
1659
- sqlite3.connect(str(_path))
1785
+ with _lib_perf.phase("doctor.integrity"):
1786
+ # ── Integrity (deep only — #279 S2 F5b) ──────────────────────────
1787
+ stats_db_quick_check = cache_db_quick_check = None
1788
+ conversations_db_quick_check = None
1789
+ if deep:
1790
+ for _label, _path in (("stats", _cctally_core.DB_PATH),
1791
+ ("cache", _cctally_core.CACHE_DB_PATH),
1792
+ ("conversations",
1793
+ _cctally_core.CONVERSATIONS_DB_PATH)):
1794
+ _result = None
1795
+ try:
1796
+ if (
1797
+ (
1798
+ _path.exists()
1799
+ or (
1800
+ _label == "conversations"
1801
+ and _path.with_name(
1802
+ f"{_path.name}.recovery.json"
1803
+ ).exists()
1804
+ )
1660
1805
  )
1661
- else:
1662
- _conn_ctx = _conversation_ro_guarded(timeout=2.0)
1663
- with _conn_ctx as _conn:
1664
- if _conn is not None:
1665
- _row = _conn.execute(
1666
- "PRAGMA quick_check(1)").fetchone()
1667
- _result = (
1668
- str(_row[0])
1669
- if _row and _row[0] is not None else None
1806
+ and (_label != "cache" or _cache_probe_allowed)
1807
+ ):
1808
+ # #386: the stats leg holds a read-write handle for the whole
1809
+ # of a full quick_check — the longest-lived stats handle any
1810
+ # diagnostic takes — so it participates in the replacement
1811
+ # protocol. The cache leg keeps its own opener.
1812
+ if _label == "stats":
1813
+ import _cctally_store as _store_mod
1814
+ _conn_ctx = contextlib.closing(
1815
+ _store_mod.stats_open_guarded(_path)
1670
1816
  )
1671
- elif (
1672
- _label == "conversations"
1673
- and _path.with_name(
1674
- f"{_path.name}.recovery.json"
1675
- ).exists()
1676
- ):
1677
- _result = "recovery in progress"
1678
- except sqlite3.DatabaseError as exc:
1679
- _result = f"open failed: {exc}"
1680
- except Exception:
1681
- _result = None
1682
- if _label == "stats":
1683
- stats_db_quick_check = _result
1684
- elif _label == "cache":
1685
- cache_db_quick_check = _result
1686
- else:
1687
- conversations_db_quick_check = _result
1688
-
1689
- # ── Lock state (#279 S2 F5c) — read-only: never create files ─────
1690
- locks_held: "dict | None" = None
1691
- try:
1692
- locks_held = {}
1693
- for _name, _lp in (
1694
- ("cache.db.lock", _cctally_core.CACHE_LOCK_PATH),
1695
- ("cache.db.codex.lock", _cctally_core.CACHE_LOCK_CODEX_PATH),
1696
- ("conversations.db.lock", _cctally_core.CONVERSATIONS_LOCK_PATH),
1697
- (
1698
- "conversations.db.codex.lock",
1699
- _cctally_core.CONVERSATIONS_LOCK_CODEX_PATH,
1700
- ),
1701
- (
1702
- "conversations.db.maintenance.lock",
1703
- _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH,
1704
- ),
1705
- ):
1706
- if not _lp.exists():
1707
- locks_held[_name] = False
1708
- continue
1709
- try:
1710
- with open(_lp, "r") as _lf:
1711
- try:
1712
- fcntl.flock(_lf, fcntl.LOCK_EX | fcntl.LOCK_NB)
1713
- fcntl.flock(_lf, fcntl.LOCK_UN)
1714
- locks_held[_name] = False
1715
- except OSError:
1716
- locks_held[_name] = True
1717
- except OSError:
1718
- locks_held[_name] = None
1719
- except Exception:
1720
- locks_held = None
1817
+ elif _label == "cache":
1818
+ _conn_ctx = contextlib.closing(
1819
+ sqlite3.connect(str(_path))
1820
+ )
1821
+ else:
1822
+ _conn_ctx = _conversation_ro_guarded(timeout=2.0)
1823
+ with _conn_ctx as _conn:
1824
+ if _conn is not None:
1825
+ _row = _conn.execute(
1826
+ "PRAGMA quick_check(1)").fetchone()
1827
+ _result = (
1828
+ str(_row[0])
1829
+ if _row and _row[0] is not None else None
1830
+ )
1831
+ elif (
1832
+ _label == "conversations"
1833
+ and _path.with_name(
1834
+ f"{_path.name}.recovery.json"
1835
+ ).exists()
1836
+ ):
1837
+ _result = "recovery in progress"
1838
+ except sqlite3.DatabaseError as exc:
1839
+ _result = f"open failed: {exc}"
1840
+ except Exception:
1841
+ _result = None
1842
+ if _label == "stats":
1843
+ stats_db_quick_check = _result
1844
+ elif _label == "cache":
1845
+ cache_db_quick_check = _result
1846
+ else:
1847
+ conversations_db_quick_check = _result
1721
1848
 
1722
- # ── cache.db WAL size (#297) — read-only backstop ────────────────
1723
- # Gathered OUTSIDE the deep/quick_check branch (above) so the WAL-size
1724
- # check runs in both shallow and deep gather modes. Best-effort getsize;
1725
- # None on OSError/race (doctor never blocks or raises), 0 when absent.
1726
- cache_db_wal_bytes: "int | None"
1727
- try:
1728
- _wal = pathlib.Path(f"{_cctally_core.CACHE_DB_PATH}-wal")
1729
- cache_db_wal_bytes = _wal.stat().st_size if _wal.exists() else 0
1730
- except OSError:
1731
- cache_db_wal_bytes = None
1732
-
1733
- # ── Safety ───────────────────────────────────────────────────────
1734
- # `dashboard.bind` is read via the same chokepoint that powers
1735
- # `cctally config get dashboard.bind` — `_config_known_value`
1736
- # normalizes hand-edited junk back to "loopback", matching the
1737
- # value cmd_dashboard would actually bind to.
1738
- #
1739
- # Raw JSON read (NOT load_config or _load_config_unlocked): both
1740
- # call `ensure_dirs()`, which creates `~/.local/share/cctally/`
1741
- # and `logs/` on a fresh HOME. Doctor is a read-only diagnostic
1742
- # (H1 invariant) — it must never mutate user state, even by
1743
- # creating an empty directory tree. Corrupt JSON yields
1744
- # `dashboard_bind_stored = "loopback"` (the same fallback the
1745
- # original try/except gave); the dedicated `config_json_valid`
1746
- # check surfaces the corruption separately.
1747
- #
1748
- # `dashboard.expose_transcripts` (Plan 2, spec §5) is read off the same raw
1749
- # JSON via the same chokepoint (defaults False; hand-edited junk → False).
1750
- # `_check_safety_dashboard_bind` only consults it when the bind is LAN, so
1751
- # a loopback report is byte-identical whether or not it's set.
1752
- dashboard_bind_stored = "loopback"
1753
- expose_transcripts = False
1754
- try:
1755
- if _cctally_core.CONFIG_PATH.exists():
1756
- raw_cfg = json.loads(_cctally_core.CONFIG_PATH.read_text(encoding="utf-8"))
1757
- if isinstance(raw_cfg, dict):
1758
- dashboard_bind_stored = (
1759
- c._config_known_value(raw_cfg, "dashboard.bind") or "loopback"
1760
- )
1761
- expose_transcripts = bool(
1762
- c._config_known_value(raw_cfg, "dashboard.expose_transcripts")
1763
- )
1764
- except (json.JSONDecodeError, OSError):
1765
- pass
1849
+ with _lib_perf.phase("doctor.locks"):
1850
+ # ── Lock state (#279 S2 F5c) — read-only: never create files ─────
1851
+ locks_held: "dict | None" = None
1852
+ try:
1853
+ locks_held = {}
1854
+ for _name, _lp in (
1855
+ ("cache.db.lock", _cctally_core.CACHE_LOCK_PATH),
1856
+ ("cache.db.codex.lock", _cctally_core.CACHE_LOCK_CODEX_PATH),
1857
+ ("conversations.db.lock", _cctally_core.CONVERSATIONS_LOCK_PATH),
1858
+ (
1859
+ "conversations.db.codex.lock",
1860
+ _cctally_core.CONVERSATIONS_LOCK_CODEX_PATH,
1861
+ ),
1862
+ (
1863
+ "conversations.db.maintenance.lock",
1864
+ _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH,
1865
+ ),
1866
+ ):
1867
+ if not _lp.exists():
1868
+ locks_held[_name] = False
1869
+ continue
1870
+ try:
1871
+ with open(_lp, "r") as _lf:
1872
+ try:
1873
+ fcntl.flock(_lf, fcntl.LOCK_EX | fcntl.LOCK_NB)
1874
+ fcntl.flock(_lf, fcntl.LOCK_UN)
1875
+ locks_held[_name] = False
1876
+ except OSError:
1877
+ locks_held[_name] = True
1878
+ except OSError:
1879
+ locks_held[_name] = None
1880
+ except Exception:
1881
+ locks_held = None
1882
+
1883
+ with _lib_perf.phase("doctor.wal"):
1884
+ # ── cache.db WAL size (#297) — read-only backstop ────────────────
1885
+ # Gathered OUTSIDE the deep/quick_check branch (above) so the WAL-size
1886
+ # check runs in both shallow and deep gather modes. Best-effort getsize;
1887
+ # None on OSError/race (doctor never blocks or raises), 0 when absent.
1888
+ cache_db_wal_bytes: "int | None"
1889
+ try:
1890
+ _wal = pathlib.Path(f"{_cctally_core.CACHE_DB_PATH}-wal")
1891
+ cache_db_wal_bytes = _wal.stat().st_size if _wal.exists() else 0
1892
+ except OSError:
1893
+ cache_db_wal_bytes = None
1766
1894
 
1767
- # ── Telemetry (anonymous install-count, spec 2026-07-07) ─────────
1768
- # Resolve the opt-out state via the pure kernel predicate — it reads env
1769
- # + config + the dev-checkout fact and NEVER mints an install_id / touches
1770
- # any marker (read-only H1 invariant). Uses the same raw config read as the
1771
- # safety block so doctor never auto-creates config.json; a missing/corrupt
1772
- # config degrades to `{}` (env/dev precedence still resolves correctly).
1773
- telemetry_enabled = True
1774
- telemetry_reason = "enabled"
1775
- try:
1776
- raw_tele_cfg: dict = {}
1777
- if _cctally_core.CONFIG_PATH.exists():
1778
- loaded = json.loads(_cctally_core.CONFIG_PATH.read_text(encoding="utf-8"))
1779
- if isinstance(loaded, dict):
1780
- raw_tele_cfg = loaded
1781
- telemetry_enabled, telemetry_reason = c.resolve_telemetry_state(raw_tele_cfg)
1782
- except Exception:
1783
- # Fail-soft: any read/parse/resolution error degrades to the enabled
1784
- # default (the check renders OK regardless — it never FAILs/WARNs).
1785
- telemetry_enabled, telemetry_reason = (True, "enabled")
1786
-
1787
- # config.json — RAW READ, never load_config(). load_config()
1788
- # auto-creates on first run AND silently falls back to defaults
1789
- # on corruption — both behaviors would hide diagnostic state
1790
- # (codex H1).
1791
- config_json_error = None
1792
- config_parsed: dict = {}
1793
- try:
1794
- if _cctally_core.CONFIG_PATH.exists():
1795
- config_parsed = json.loads(
1796
- _cctally_core.CONFIG_PATH.read_text(encoding="utf-8")
1895
+ # ── conversations.db WAL size (#583 S4 / F39) ────────────────────
1896
+ # Same shallow/deep placement and the same best-effort contract as the
1897
+ # cache leg above: 0 when the sidecar is absent, None when it cannot be
1898
+ # stat'ed.
1899
+ conversations_db_wal_bytes: "int | None"
1900
+ try:
1901
+ _cwal = pathlib.Path(f"{_cctally_core.CONVERSATIONS_DB_PATH}-wal")
1902
+ conversations_db_wal_bytes = (
1903
+ _cwal.stat().st_size if _cwal.exists() else 0
1797
1904
  )
1798
- except json.JSONDecodeError as exc:
1799
- config_json_error = f"{type(exc).__name__}: {exc}"
1800
- except OSError as exc:
1801
- config_json_error = f"OSError: {exc}"
1802
-
1803
- # Configured update (release) channel (beta-channel, spec 2026-07-21 §3):
1804
- # derived from the SAME raw read (never load_config, which auto-creates on
1805
- # first run). Fail-soft to "stable" — resolve_update_channel already
1806
- # tolerates a non-dict block / junk value.
1807
- try:
1808
- update_channel = c.resolve_update_channel(
1809
- config_parsed if isinstance(config_parsed, dict) else {}
1810
- )
1811
- except Exception:
1812
- update_channel = "stable"
1813
-
1814
- update_state = None
1815
- update_state_error = None
1816
- try:
1817
- update_state = c._load_update_state()
1818
- except Exception as exc:
1819
- update_state_error = f"{type(exc).__name__}: {exc}"
1820
-
1821
- update_suppress = None
1822
- update_suppress_error = None
1823
- try:
1824
- update_suppress = c._load_update_suppress()
1825
- except Exception as exc:
1826
- update_suppress_error = f"{type(exc).__name__}: {exc}"
1827
-
1828
- # Same predicate the update banner uses; doctor must not warn about
1829
- # updates the user has already skipped or deferred.
1830
- effective_update_available, effective_update_reason = (
1831
- c._compute_effective_update_available(update_state, update_suppress, now_utc)
1832
- )
1905
+ except OSError:
1906
+ conversations_db_wal_bytes = None
1907
+
1908
+ with _lib_perf.phase("doctor.config"):
1909
+ # ── Safety ───────────────────────────────────────────────────────
1910
+ # `dashboard.bind` is read via the same chokepoint that powers
1911
+ # `cctally config get dashboard.bind` — `_config_known_value`
1912
+ # normalizes hand-edited junk back to "loopback", matching the
1913
+ # value cmd_dashboard would actually bind to.
1914
+ #
1915
+ # Raw JSON read (NOT load_config or _load_config_unlocked): both
1916
+ # call `ensure_dirs()`, which creates `~/.local/share/cctally/`
1917
+ # and `logs/` on a fresh HOME. Doctor is a read-only diagnostic
1918
+ # (H1 invariant) — it must never mutate user state, even by
1919
+ # creating an empty directory tree. Corrupt JSON yields
1920
+ # `dashboard_bind_stored = "loopback"` (the same fallback the
1921
+ # original try/except gave); the dedicated `config_json_valid`
1922
+ # check surfaces the corruption separately.
1923
+ #
1924
+ # `dashboard.expose_transcripts` (Plan 2, spec §5) is read off the same raw
1925
+ # JSON via the same chokepoint (defaults False; hand-edited junk → False).
1926
+ # `_check_safety_dashboard_bind` only consults it when the bind is LAN, so
1927
+ # a loopback report is byte-identical whether or not it's set.
1928
+ dashboard_bind_stored = "loopback"
1929
+ expose_transcripts = False
1930
+ try:
1931
+ if _cctally_core.CONFIG_PATH.exists():
1932
+ raw_cfg = json.loads(_cctally_core.CONFIG_PATH.read_text(encoding="utf-8"))
1933
+ if isinstance(raw_cfg, dict):
1934
+ dashboard_bind_stored = (
1935
+ c._config_known_value(raw_cfg, "dashboard.bind") or "loopback"
1936
+ )
1937
+ expose_transcripts = bool(
1938
+ c._config_known_value(raw_cfg, "dashboard.expose_transcripts")
1939
+ )
1940
+ except (json.JSONDecodeError, OSError):
1941
+ pass
1833
1942
 
1834
- # ── Pricing coverage (spec §5.1) ─────────────────────────────────
1835
- # Read-only trailing-30d scan + classification via the pure-fn kernel.
1836
- # Any failure degrades to None so the check renders OK (never FAIL) and
1837
- # the rest of the report is unaffected — same posture as the cache reads
1838
- # above. `_pricing_observed_models` honors the no-mutation contract.
1839
- pricing_coverage = None
1840
- if _cache_probe_allowed:
1943
+ # ── Telemetry (anonymous install-count, spec 2026-07-07) ─────────
1944
+ # Resolve the opt-out state via the pure kernel predicate — it reads env
1945
+ # + config + the dev-checkout fact and NEVER mints an install_id / touches
1946
+ # any marker (read-only H1 invariant). Uses the same raw config read as the
1947
+ # safety block so doctor never auto-creates config.json; a missing/corrupt
1948
+ # config degrades to `{}` (env/dev precedence still resolves correctly).
1949
+ telemetry_enabled = True
1950
+ telemetry_reason = "enabled"
1951
+ try:
1952
+ raw_tele_cfg: dict = {}
1953
+ if _cctally_core.CONFIG_PATH.exists():
1954
+ loaded = json.loads(_cctally_core.CONFIG_PATH.read_text(encoding="utf-8"))
1955
+ if isinstance(loaded, dict):
1956
+ raw_tele_cfg = loaded
1957
+ telemetry_enabled, telemetry_reason = c.resolve_telemetry_state(raw_tele_cfg)
1958
+ except Exception:
1959
+ # Fail-soft: any read/parse/resolution error degrades to the enabled
1960
+ # default (the check renders OK regardless — it never FAILs/WARNs).
1961
+ telemetry_enabled, telemetry_reason = (True, "enabled")
1962
+
1963
+ # config.json — RAW READ, never load_config(). load_config()
1964
+ # auto-creates on first run AND silently falls back to defaults
1965
+ # on corruption — both behaviors would hide diagnostic state
1966
+ # (codex H1).
1967
+ config_json_error = None
1968
+ config_parsed: dict = {}
1969
+ try:
1970
+ if _cctally_core.CONFIG_PATH.exists():
1971
+ config_parsed = json.loads(
1972
+ _cctally_core.CONFIG_PATH.read_text(encoding="utf-8")
1973
+ )
1974
+ except json.JSONDecodeError as exc:
1975
+ config_json_error = f"{type(exc).__name__}: {exc}"
1976
+ except OSError as exc:
1977
+ config_json_error = f"OSError: {exc}"
1978
+
1979
+ # Configured update (release) channel (beta-channel, spec 2026-07-21 §3):
1980
+ # derived from the SAME raw read (never load_config, which auto-creates on
1981
+ # first run). Fail-soft to "stable" — resolve_update_channel already
1982
+ # tolerates a non-dict block / junk value.
1841
1983
  try:
1842
- observed = c._pricing_observed_models(now_utc)
1843
- # Detection-only: pass warn=False so finding an unpriced model here
1844
- # does NOT fire the cost-engine's unknown-model warning.
1845
- pricing_coverage = c.classify_coverage(
1846
- observed,
1847
- lambda m: c._resolve_model_pricing(m, warn=False),
1848
- c._is_codex_fallback,
1984
+ update_channel = c.resolve_update_channel(
1985
+ config_parsed if isinstance(config_parsed, dict) else {}
1849
1986
  )
1850
1987
  except Exception:
1851
- pricing_coverage = None
1988
+ update_channel = "stable"
1852
1989
 
1853
- # ── Meta ─────────────────────────────────────────────────────────
1854
- # ── Journal (DB journal redesign §9) ─────────────────────────────
1855
- # Read-only legs over the append-only journal: presence + appendability,
1856
- # torn-tail/malformed counts (deep-gated — reads whole segments), the ingest
1857
- # cursor lag vs. the high-water, and the auto-heal incident history. Every
1858
- # probe degrades to its always-OK posture on any error (the pure kernel then
1859
- # reports "no journal" / "not scanned").
1860
- import _lib_journal as _jl
1861
- journal_present = False
1862
- journal_appendable = None
1863
- journal_segment_count = 0
1864
- journal_has_bytes = False
1865
- journal_malformed_count = None
1866
- journal_torn_tail_count = None
1867
- journal_cursor_lag_bytes = None
1868
- journal_hw_segment = None
1869
- journal_cursor_segment = None
1870
- journal_conflicts = None
1871
- journal_protocol_violations = None
1872
- journal_protocol_acknowledged = None
1873
- journal_protocol_error = None
1874
- try:
1875
- jdir = _cctally_core.JOURNAL_DIR
1876
- journal_present = jdir.exists()
1877
- if journal_present:
1878
- try:
1879
- journal_appendable = os.access(str(jdir), os.W_OK)
1880
- except OSError:
1881
- journal_appendable = None
1882
- import _cctally_journal as _jr
1990
+ with _lib_perf.phase("doctor.update"):
1991
+ update_state = None
1992
+ update_state_error = None
1993
+ try:
1994
+ update_state = c._load_update_state()
1995
+ except Exception as exc:
1996
+ update_state_error = f"{type(exc).__name__}: {exc}"
1997
+
1998
+ update_suppress = None
1999
+ update_suppress_error = None
2000
+ try:
2001
+ update_suppress = c._load_update_suppress()
2002
+ except Exception as exc:
2003
+ update_suppress_error = f"{type(exc).__name__}: {exc}"
2004
+
2005
+ # Same predicate the update banner uses; doctor must not warn about
2006
+ # updates the user has already skipped or deferred.
2007
+ effective_update_available, effective_update_reason = (
2008
+ c._compute_effective_update_available(update_state, update_suppress, now_utc)
2009
+ )
2010
+
2011
+ with _lib_perf.phase("doctor.pricing_coverage"):
2012
+ # ── Pricing coverage (spec §5.1) ─────────────────────────────────
2013
+ # Read-only trailing-30d scan + classification via the pure-fn kernel.
2014
+ # Any failure degrades to None so the check renders OK (never FAIL) and
2015
+ # the rest of the report is unaffected — same posture as the cache reads
2016
+ # above. `_pricing_observed_models` honors the no-mutation contract.
2017
+ pricing_coverage = None
2018
+ if _cache_probe_allowed:
1883
2019
  try:
1884
- segs = _jr.list_segments() # canonical (segment) order
2020
+ observed = c._pricing_observed_models(now_utc)
2021
+ # Detection-only: pass warn=False so finding an unpriced model here
2022
+ # does NOT fire the cost-engine's unknown-model warning.
2023
+ pricing_coverage = c.classify_coverage(
2024
+ observed,
2025
+ lambda m: c._resolve_model_pricing(m, warn=False),
2026
+ c._is_codex_fallback,
2027
+ )
1885
2028
  except Exception:
1886
- segs = []
1887
- journal_segment_count = len(segs)
1888
- # #402: the disposable stats index persists the most recent complete
1889
- # selector result. Shallow Dashboard/TUI gathers read that bounded
1890
- # summary instead of rescanning a production-sized journal and
1891
- # therefore cannot turn known taint into a false OK.
1892
- try:
1893
- if _cctally_core.DB_PATH.exists():
1894
- pc = _stats_ro_guarded()
1895
- try:
1896
- protocol_rows = [
1897
- json.loads(str(row[0]))
1898
- for row in pc.execute(
1899
- "SELECT violation_json "
1900
- "FROM journal_protocol_violations "
1901
- "ORDER BY batch_id, kind, fingerprint"
1902
- )
1903
- ]
1904
- journal_protocol_violations = [
1905
- item for item in protocol_rows
1906
- if not item.get("auditId")
1907
- ]
1908
- journal_protocol_acknowledged = [
1909
- item for item in protocol_rows
1910
- if item.get("auditId")
1911
- ]
1912
- finally:
1913
- pc.close()
1914
- except (sqlite3.Error, ValueError, TypeError):
1915
- journal_protocol_violations = None
1916
- journal_protocol_acknowledged = None
1917
- sizes: dict = {}
1918
- for seg in segs:
2029
+ pricing_coverage = None
2030
+
2031
+ # ── Meta ─────────────────────────────────────────────────────────
2032
+ with _lib_perf.phase("doctor.journal"):
2033
+ # ── Journal (DB journal redesign §9) ─────────────────────────────
2034
+ # Read-only legs over the append-only journal: presence + appendability,
2035
+ # torn-tail/malformed counts (deep-gated — reads whole segments), the ingest
2036
+ # cursor lag vs. the high-water, and the auto-heal incident history. Every
2037
+ # probe degrades to its always-OK posture on any error (the pure kernel then
2038
+ # reports "no journal" / "not scanned").
2039
+ import _lib_journal as _jl
2040
+ journal_present = False
2041
+ journal_appendable = None
2042
+ journal_segment_count = 0
2043
+ journal_has_bytes = False
2044
+ journal_malformed_count = None
2045
+ journal_torn_tail_count = None
2046
+ journal_cursor_lag_bytes = None
2047
+ journal_hw_segment = None
2048
+ journal_cursor_segment = None
2049
+ journal_conflicts = None
2050
+ journal_protocol_violations = None
2051
+ journal_protocol_acknowledged = None
2052
+ journal_protocol_error = None
2053
+ try:
2054
+ jdir = _cctally_core.JOURNAL_DIR
2055
+ journal_present = jdir.exists()
2056
+ if journal_present:
1919
2057
  try:
1920
- sizes[seg] = (jdir / seg).stat().st_size
2058
+ journal_appendable = os.access(str(jdir), os.W_OK)
1921
2059
  except OSError:
1922
- sizes[seg] = 0
1923
- if segs:
1924
- journal_hw_segment = segs[-1]
1925
- journal_has_bytes = _jr._has_retained_journal_bytes(
1926
- sizes.values()
1927
- )
1928
- # deep-gated malformed / torn-tail scan (reads the whole journal;
1929
- # the dashboard's per-rebuild gather stays deep=False so it never
1930
- # pays this at the 10× envelope — mirrors the quick_check legs).
1931
- if deep and segs:
1932
- malformed = 0
1933
- torn = 0
1934
- decoded_records: list = []
1935
- protocol_evidence = []
1936
- prior_high_water = None
1937
- cutover_value = None
2060
+ journal_appendable = None
2061
+ import _cctally_journal as _jr
2062
+ try:
2063
+ segs = _jr.list_segments() # canonical (segment) order
2064
+ except Exception:
2065
+ segs = []
2066
+ journal_segment_count = len(segs)
2067
+ # #402: the disposable stats index persists the most recent complete
2068
+ # selector result. Shallow Dashboard/TUI gathers read that bounded
2069
+ # summary instead of rescanning a production-sized journal and
2070
+ # therefore cannot turn known taint into a false OK.
2071
+ try:
2072
+ if _cctally_core.DB_PATH.exists():
2073
+ pc = _stats_ro_guarded()
2074
+ try:
2075
+ protocol_rows = [
2076
+ json.loads(str(row[0]))
2077
+ for row in pc.execute(
2078
+ "SELECT violation_json "
2079
+ "FROM journal_protocol_violations "
2080
+ "ORDER BY batch_id, kind, fingerprint"
2081
+ )
2082
+ ]
2083
+ journal_protocol_violations = [
2084
+ item for item in protocol_rows
2085
+ if not item.get("auditId")
2086
+ ]
2087
+ journal_protocol_acknowledged = [
2088
+ item for item in protocol_rows
2089
+ if item.get("auditId")
2090
+ ]
2091
+ finally:
2092
+ pc.close()
2093
+ except (sqlite3.Error, ValueError, TypeError):
2094
+ journal_protocol_violations = None
2095
+ journal_protocol_acknowledged = None
2096
+ sizes: dict = {}
1938
2097
  for seg in segs:
1939
2098
  try:
1940
- data = (jdir / seg).read_bytes()
2099
+ sizes[seg] = (jdir / seg).stat().st_size
1941
2100
  except OSError:
1942
- continue
1943
- if not data:
1944
- continue
1945
- if not data.endswith(b"\n"):
1946
- torn += 1
1947
- # every element except the last is a complete line; the last
1948
- # is either "" (ended in \n) or the torn partial — not a
1949
- # mid-file line, so it is never counted as malformed.
1950
- offset = 0
1951
- for raw in data.split(b"\n")[:-1]:
1952
- if not raw:
1953
- prior_high_water = (seg, offset + 1)
1954
- offset += 1
2101
+ sizes[seg] = 0
2102
+ if segs:
2103
+ journal_hw_segment = segs[-1]
2104
+ journal_has_bytes = _jr._has_retained_journal_bytes(
2105
+ sizes.values()
2106
+ )
2107
+ # deep-gated malformed / torn-tail scan (reads the whole journal;
2108
+ # the dashboard's per-rebuild gather stays deep=False so it never
2109
+ # pays this at the 10× envelope — mirrors the quick_check legs).
2110
+ if deep and segs:
2111
+ malformed = 0
2112
+ torn = 0
2113
+ decoded_records: list = []
2114
+ protocol_evidence = []
2115
+ prior_high_water = None
2116
+ cutover_value = None
2117
+ for seg in segs:
2118
+ try:
2119
+ data = (jdir / seg).read_bytes()
2120
+ except OSError:
1955
2121
  continue
1956
- record = _jl.decode_line(raw)
1957
- if record is None:
1958
- malformed += 1
2122
+ if not data:
2123
+ continue
2124
+ if not data.endswith(b"\n"):
2125
+ torn += 1
2126
+ # every element except the last is a complete line; the last
2127
+ # is either "" (ended in \n) or the torn partial — not a
2128
+ # mid-file line, so it is never counted as malformed.
2129
+ offset = 0
2130
+ for raw in data.split(b"\n")[:-1]:
2131
+ if not raw:
2132
+ prior_high_water = (seg, offset + 1)
2133
+ offset += 1
2134
+ continue
2135
+ record = _jl.decode_line(raw)
2136
+ if record is None:
2137
+ malformed += 1
2138
+ prior_high_water = (
2139
+ seg,
2140
+ offset + len(raw) + 1,
2141
+ )
2142
+ offset += len(raw) + 1
2143
+ continue
2144
+ _jr._capture_protocol_prefix_evidence(
2145
+ record,
2146
+ prior_high_water,
2147
+ protocol_evidence,
2148
+ )
2149
+ # first cutover op wins, exactly as
2150
+ # `find_accounts_cutover_op` scans — captured here so the
2151
+ # conflict scan does not decode the whole journal twice.
2152
+ if (cutover_value is None
2153
+ and record.get("id") == _jr.CUTOVER_OP_ID):
2154
+ payload = record.get("payload")
2155
+ if isinstance(payload, dict):
2156
+ cutover_value = payload.get(
2157
+ "claude_legacy_account")
2158
+ # RETAIN ONLY what the selector consumes. `obs` lines are
2159
+ # ~97% of a real journal (984k of 1.02M) and
2160
+ # `resolve_effective_events` ignores them entirely —
2161
+ # keeping their dictionaries cost 4.3 GB of peak RSS for
2162
+ # an identical result (#374 review). They still consume a
2163
+ # lightweight slot because their physical sequence is
2164
+ # part of three durable violation fingerprints (#508).
2165
+ decoded_records.append(
2166
+ _lib_journal_router.selector_slot(record)
2167
+ )
1959
2168
  prior_high_water = (
1960
2169
  seg,
1961
2170
  offset + len(raw) + 1,
1962
2171
  )
1963
2172
  offset += len(raw) + 1
1964
- continue
1965
- _jr._capture_protocol_prefix_evidence(
1966
- record,
1967
- prior_high_water,
1968
- protocol_evidence,
1969
- )
1970
- # first cutover op wins, exactly as
1971
- # `find_accounts_cutover_op` scans — captured here so the
1972
- # conflict scan does not decode the whole journal twice.
1973
- if (cutover_value is None
1974
- and record.get("id") == _jr.CUTOVER_OP_ID):
1975
- payload = record.get("payload")
1976
- if isinstance(payload, dict):
1977
- cutover_value = payload.get(
1978
- "claude_legacy_account")
1979
- # RETAIN ONLY what the selector consumes. `obs` lines are
1980
- # ~97% of a real journal (984k of 1.02M) and
1981
- # `resolve_effective_events` ignores them entirely —
1982
- # keeping their dictionaries cost 4.3 GB of peak RSS for
1983
- # an identical result (#374 review). They still consume a
1984
- # lightweight slot because their physical sequence is
1985
- # part of three durable violation fingerprints (#508).
1986
- decoded_records.append(
1987
- _lib_journal_router.selector_slot(record)
2173
+ journal_malformed_count = malformed
2174
+ journal_torn_tail_count = torn
2175
+ # #374: same-revision quarantine, via the SHARED selector over
2176
+ # rebuild-equivalent input. Raw `(id, rev)` grouping would report
2177
+ # lower-revision groups a completed rev-1 batch legitimately
2178
+ # superseded, and false account conflicts that the rebuild's
2179
+ # `_normalize_legacy_account_stamp` resolves — so normalize
2180
+ # exactly as `rebuild_stats_index` does, then select.
2181
+ try:
2182
+ cutover_claude = (
2183
+ cutover_value if cutover_value is not None
2184
+ else _jr.resolve_cutover_claude_account()
1988
2185
  )
1989
- prior_high_water = (
1990
- seg,
1991
- offset + len(raw) + 1,
2186
+ for record in decoded_records:
2187
+ if record is not None:
2188
+ _jr._normalize_legacy_account_stamp(
2189
+ record, cutover_claude)
2190
+ selection = _jl.resolve_effective_events(
2191
+ decoded_records,
2192
+ protocol_prefix_evidence=protocol_evidence,
1992
2193
  )
1993
- offset += len(raw) + 1
1994
- journal_malformed_count = malformed
1995
- journal_torn_tail_count = torn
1996
- # #374: same-revision quarantine, via the SHARED selector over
1997
- # rebuild-equivalent input. Raw `(id, rev)` grouping would report
1998
- # lower-revision groups a completed rev-1 batch legitimately
1999
- # superseded, and false account conflicts that the rebuild's
2000
- # `_normalize_legacy_account_stamp` resolves — so normalize
2001
- # exactly as `rebuild_stats_index` does, then select.
2002
- try:
2003
- cutover_claude = (
2004
- cutover_value if cutover_value is not None
2005
- else _jr.resolve_cutover_claude_account()
2006
- )
2007
- for record in decoded_records:
2008
- if record is not None:
2009
- _jr._normalize_legacy_account_stamp(
2010
- record, cutover_claude)
2011
- selection = _jl.resolve_effective_events(
2012
- decoded_records,
2013
- protocol_prefix_evidence=protocol_evidence,
2014
- )
2015
- except _jl.JournalProtocolError as exc:
2016
- # Out-of-scope malformed known record: selection did not
2017
- # finish, so conflicts/tainted-batch results are unavailable.
2018
- journal_protocol_error = str(exc)
2019
- journal_conflicts = None
2020
- journal_protocol_violations = None
2021
- journal_protocol_acknowledged = None
2022
- except Exception:
2023
- journal_conflicts = None
2024
- journal_protocol_violations = None
2025
- journal_protocol_acknowledged = None
2026
- else:
2027
- journal_conflicts = [
2028
- conflict.to_dict() for conflict in selection.conflicts
2029
- ]
2030
- journal_protocol_violations = [
2031
- violation.to_dict()
2032
- for violation in selection.protocol_violations
2033
- ]
2034
- journal_protocol_acknowledged = [
2035
- violation.to_dict()
2036
- for violation in (
2037
- selection.acknowledged_protocol_violations
2038
- )
2039
- ]
2040
- # ingest cursor lag: unconsumed bytes between the stats index cursor
2041
- # and the journal high-water, in canonical (segment, offset) order.
2042
- cursor = None
2043
- try:
2044
- if _cctally_core.DB_PATH.exists():
2045
- jc = _stats_ro_guarded() # #386 opener protocol
2046
- try:
2047
- cursor_columns = {
2048
- str(row[1])
2049
- for row in jc.execute(
2050
- "PRAGMA table_info(journal_cursor)"
2194
+ except _jl.JournalProtocolError as exc:
2195
+ # Out-of-scope malformed known record: selection did not
2196
+ # finish, so conflicts/tainted-batch results are unavailable.
2197
+ journal_protocol_error = str(exc)
2198
+ journal_conflicts = None
2199
+ journal_protocol_violations = None
2200
+ journal_protocol_acknowledged = None
2201
+ except Exception:
2202
+ journal_conflicts = None
2203
+ journal_protocol_violations = None
2204
+ journal_protocol_acknowledged = None
2205
+ else:
2206
+ journal_conflicts = [
2207
+ conflict.to_dict() for conflict in selection.conflicts
2208
+ ]
2209
+ journal_protocol_violations = [
2210
+ violation.to_dict()
2211
+ for violation in selection.protocol_violations
2212
+ ]
2213
+ journal_protocol_acknowledged = [
2214
+ violation.to_dict()
2215
+ for violation in (
2216
+ selection.acknowledged_protocol_violations
2051
2217
  )
2052
- }
2053
- if {
2054
- "applied_segment", "applied_offset"
2055
- } <= cursor_columns:
2056
- crow = jc.execute(
2057
- "SELECT segment, offset, applied_segment, "
2058
- "applied_offset FROM journal_cursor "
2059
- "WHERE id = 1").fetchone()
2060
- if (
2061
- crow is not None
2062
- and crow[2] is not None
2063
- and crow[3] is not None
2064
- ):
2065
- cursor = (crow[2], int(crow[3]))
2066
- else:
2067
- legacy = jc.execute(
2068
- "SELECT segment, offset FROM journal_cursor "
2069
- "WHERE id = 1").fetchone()
2070
- if legacy is not None:
2071
- cursor = (legacy[0], int(legacy[1]))
2072
- except sqlite3.OperationalError:
2073
- pass # pre-cutover DB has no journal_cursor table
2074
- finally:
2075
- jc.close()
2076
- except sqlite3.Error:
2218
+ ]
2219
+ # ingest cursor lag: unconsumed bytes between the stats index cursor
2220
+ # and the journal high-water, in canonical (segment, offset) order.
2077
2221
  cursor = None
2078
- if cursor is not None and segs:
2079
- cseg, coff = cursor
2080
- journal_cursor_segment = cseg
2081
- order = {s: i for i, s in enumerate(segs)}
2082
- if cseg in order:
2083
- ci = order[cseg]
2084
- lag = max(0, sizes.get(segs[ci], 0) - coff)
2085
- for s in segs[ci + 1:]:
2086
- lag += sizes.get(s, 0)
2087
- journal_cursor_lag_bytes = lag
2088
- except Exception:
2089
- pass
2090
- # #496 S5b: the durable incomplete-quota-projection flag carried inside the
2091
- # published stats generation. Read-only, and independent of journal presence
2092
- # because the flag describes the INDEX rather than the journal. None means
2093
- # "no epoch-1009 index to ask" (absent file, missing table, unreadable DB),
2094
- # which the pure kernel reports as not applicable rather than as a fault.
2095
- #
2096
- # This is a THIRD read-only stats open in this function, and folding it into
2097
- # the `jc` open above was considered and rejected: that open sits inside
2098
- # `if journal_present:`, so carrying this SELECT there would make the flag
2099
- # unreadable on an install whose journal directory is absent — exactly the
2100
- # independence the paragraph above states. One extra guarded open on the
2101
- # doctor path is the cheaper of the two.
2102
- stats_quota_projection_incomplete: "bool | None" = None
2103
- try:
2104
- if _cctally_core.DB_PATH.exists():
2105
- qp = _stats_ro_guarded() # #386 opener protocol
2106
- try:
2107
- row = qp.execute(
2108
- "SELECT incomplete FROM stats_quota_projection_state "
2109
- "WHERE id = 1").fetchone()
2110
- if row is not None:
2111
- stats_quota_projection_incomplete = bool(int(row[0] or 0))
2112
- except sqlite3.OperationalError:
2113
- pass # pre-1009 index has no stats_quota_projection_state
2114
- finally:
2115
- qp.close()
2116
- except Exception:
2117
- stats_quota_projection_incomplete = None
2118
-
2119
- # Auto-heal incident history — independent of journal presence (a corruption
2120
- # incident can predate cutover). None only if BOTH dirs were unreadable.
2121
- journal_heal_incidents = None
2122
- _incidents: list = []
2123
- _incident_read_ok = False
2124
- try:
2125
- qroot = _cctally_core.APP_DIR / "quarantine"
2126
- if qroot.exists():
2127
- _incident_read_ok = True
2128
- for entry in qroot.iterdir():
2129
- if entry.is_dir():
2130
- record = _journal_heal_incident(
2131
- "quarantine", entry.name, now_utc)
2132
- # §7.2 escalates on a REPEATED damage shape, so the shape
2133
- # has to travel with the incident it belongs to — counted
2134
- # once per incident, never once per manifest read.
2135
- record["shape"] = _incident_shape_token(entry)
2136
- _incidents.append(record)
2137
- except OSError:
2138
- pass
2139
- try:
2140
- logdir = _cctally_core.LOG_DIR
2141
- if logdir.exists():
2142
- _incident_read_ok = True
2143
- for entry in logdir.iterdir():
2144
- n = entry.name
2145
- if "-corruption-forensics-" in n and n.endswith(".json"):
2146
- _incidents.append(
2147
- _journal_heal_incident("forensics", n, now_utc))
2148
- except OSError:
2149
- pass
2150
- if _incident_read_ok:
2151
- # most-recent first; unparseable ages (None) sort last.
2152
- _incidents.sort(key=lambda d: (d["age_s"] is None,
2153
- d["age_s"] if d["age_s"] is not None else 0))
2154
- journal_heal_incidents = _incidents
2155
-
2156
- # #496 S6 §7.2 / §7.3. Both are read-only and take no lock; both degrade to
2157
- # None rather than failing the gather, because `doctor` is reached from the
2158
- # TUI and the dashboard snapshot precompute as well as from the CLI.
2159
- journal_heal_detections = _gather_heal_detections(now_utc)
2160
- retained_artifacts = _gather_retained_artifacts(now_utc, deep=deep)
2161
-
2162
- # #386/#389 stats sole-writer guard log (spec §6.4). Read-only, fail-soft: an
2163
- # absent log is the NORMAL state and must read as INFO, never as a gather
2164
- # failure. Read only the bounded tail; rotation and cross-process throttling
2165
- # bound the writer side independently.
2166
- journal_writer_guard = None
2167
- try:
2168
- journal_writer_guard = _gather_writer_guard_log(now_utc)
2169
- except (OSError, Exception):
2222
+ try:
2223
+ if _cctally_core.DB_PATH.exists():
2224
+ jc = _stats_ro_guarded() # #386 opener protocol
2225
+ try:
2226
+ cursor_columns = {
2227
+ str(row[1])
2228
+ for row in jc.execute(
2229
+ "PRAGMA table_info(journal_cursor)"
2230
+ )
2231
+ }
2232
+ if {
2233
+ "applied_segment", "applied_offset"
2234
+ } <= cursor_columns:
2235
+ crow = jc.execute(
2236
+ "SELECT segment, offset, applied_segment, "
2237
+ "applied_offset FROM journal_cursor "
2238
+ "WHERE id = 1").fetchone()
2239
+ if (
2240
+ crow is not None
2241
+ and crow[2] is not None
2242
+ and crow[3] is not None
2243
+ ):
2244
+ cursor = (crow[2], int(crow[3]))
2245
+ else:
2246
+ legacy = jc.execute(
2247
+ "SELECT segment, offset FROM journal_cursor "
2248
+ "WHERE id = 1").fetchone()
2249
+ if legacy is not None:
2250
+ cursor = (legacy[0], int(legacy[1]))
2251
+ except sqlite3.OperationalError:
2252
+ pass # pre-cutover DB has no journal_cursor table
2253
+ finally:
2254
+ jc.close()
2255
+ except sqlite3.Error:
2256
+ cursor = None
2257
+ if cursor is not None and segs:
2258
+ cseg, coff = cursor
2259
+ journal_cursor_segment = cseg
2260
+ order = {s: i for i, s in enumerate(segs)}
2261
+ if cseg in order:
2262
+ ci = order[cseg]
2263
+ lag = max(0, sizes.get(segs[ci], 0) - coff)
2264
+ for s in segs[ci + 1:]:
2265
+ lag += sizes.get(s, 0)
2266
+ journal_cursor_lag_bytes = lag
2267
+ except Exception:
2268
+ pass
2269
+ with _lib_perf.phase("doctor.stats_projection"):
2270
+ # #496 S5b: the durable incomplete-quota-projection flag carried inside the
2271
+ # published stats generation. Read-only, and independent of journal presence
2272
+ # because the flag describes the INDEX rather than the journal. None means
2273
+ # "no epoch-1009 index to ask" (absent file, missing table, unreadable DB),
2274
+ # which the pure kernel reports as not applicable rather than as a fault.
2275
+ #
2276
+ # This is a THIRD read-only stats open in this function, and folding it into
2277
+ # the `jc` open above was considered and rejected: that open sits inside
2278
+ # `if journal_present:`, so carrying this SELECT there would make the flag
2279
+ # unreadable on an install whose journal directory is absent — exactly the
2280
+ # independence the paragraph above states. One extra guarded open on the
2281
+ # doctor path is the cheaper of the two.
2282
+ stats_quota_projection_incomplete: "bool | None" = None
2283
+ try:
2284
+ if _cctally_core.DB_PATH.exists():
2285
+ qp = _stats_ro_guarded() # #386 opener protocol
2286
+ try:
2287
+ row = qp.execute(
2288
+ "SELECT incomplete FROM stats_quota_projection_state "
2289
+ "WHERE id = 1").fetchone()
2290
+ if row is not None:
2291
+ stats_quota_projection_incomplete = bool(int(row[0] or 0))
2292
+ except sqlite3.OperationalError:
2293
+ pass # pre-1009 index has no stats_quota_projection_state
2294
+ finally:
2295
+ qp.close()
2296
+ except Exception:
2297
+ stats_quota_projection_incomplete = None
2298
+
2299
+ with _lib_perf.phase("doctor.heal"):
2300
+ # Auto-heal incident history — independent of journal presence (a corruption
2301
+ # incident can predate cutover). None only if BOTH dirs were unreadable.
2302
+ journal_heal_incidents = None
2303
+ _incidents: list = []
2304
+ _incident_read_ok = False
2305
+ try:
2306
+ qroot = _cctally_core.APP_DIR / "quarantine"
2307
+ if qroot.exists():
2308
+ _incident_read_ok = True
2309
+ for entry in qroot.iterdir():
2310
+ if entry.is_dir():
2311
+ record = _journal_heal_incident(
2312
+ "quarantine", entry.name, now_utc)
2313
+ # §7.2 escalates on a REPEATED damage shape, so the shape
2314
+ # has to travel with the incident it belongs to — counted
2315
+ # once per incident, never once per manifest read.
2316
+ record["shape"] = _incident_shape_token(entry)
2317
+ _incidents.append(record)
2318
+ except OSError:
2319
+ pass
2320
+ try:
2321
+ logdir = _cctally_core.LOG_DIR
2322
+ if logdir.exists():
2323
+ _incident_read_ok = True
2324
+ for entry in logdir.iterdir():
2325
+ n = entry.name
2326
+ if "-corruption-forensics-" in n and n.endswith(".json"):
2327
+ _incidents.append(
2328
+ _journal_heal_incident("forensics", n, now_utc))
2329
+ except OSError:
2330
+ pass
2331
+ if _incident_read_ok:
2332
+ # most-recent first; unparseable ages (None) sort last.
2333
+ _incidents.sort(key=lambda d: (d["age_s"] is None,
2334
+ d["age_s"] if d["age_s"] is not None else 0))
2335
+ journal_heal_incidents = _incidents
2336
+
2337
+ # #496 S6 §7.2 / §7.3. Both are read-only and take no lock; both degrade to
2338
+ # None rather than failing the gather, because `doctor` is reached from the
2339
+ # TUI and the dashboard snapshot precompute as well as from the CLI.
2340
+ journal_heal_detections = _gather_heal_detections(now_utc)
2341
+ retained_artifacts = _gather_retained_artifacts(now_utc, deep=deep)
2342
+
2343
+ # #386/#389 stats sole-writer guard log (spec §6.4). Read-only, fail-soft: an
2344
+ # absent log is the NORMAL state and must read as INFO, never as a gather
2345
+ # failure. Read only the bounded tail; rotation and cross-process throttling
2346
+ # bound the writer side independently.
2170
2347
  journal_writer_guard = None
2171
-
2172
- cctally_version_tuple = _lib_changelog._read_latest_changelog_version()
2173
- cctally_version = (
2174
- cctally_version_tuple[0] if cctally_version_tuple else "unknown"
2175
- )
2176
- accounts_state = _gather_accounts_state(now_utc)
2177
- accounts_state["codex_null_reset_anchors"] = codex_null_reset_anchors
2178
- accounts_state["codex_window_attribution"] = (
2179
- _gather_codex_window_attribution_state()
2180
- if _cache_probe_allowed else None)
2348
+ try:
2349
+ journal_writer_guard = _gather_writer_guard_log(now_utc)
2350
+ except (OSError, Exception):
2351
+ journal_writer_guard = None
2352
+
2353
+ with _lib_perf.phase("doctor.meta_accounts"):
2354
+ cctally_version_tuple = _lib_changelog._read_latest_changelog_version()
2355
+ cctally_version = (
2356
+ cctally_version_tuple[0] if cctally_version_tuple else "unknown"
2357
+ )
2358
+ accounts_state = _gather_accounts_state(now_utc)
2359
+ accounts_state["codex_null_reset_anchors"] = codex_null_reset_anchors
2360
+ accounts_state["codex_window_attribution"] = (
2361
+ _gather_codex_window_attribution_state()
2362
+ if _cache_probe_allowed else None)
2181
2363
 
2182
2364
  return _lib_doctor.DoctorState(
2183
2365
  symlink_state=symlink_state,
@@ -2251,6 +2433,7 @@ def _doctor_gather_state_impl(
2251
2433
  locks_held=locks_held,
2252
2434
  # #297: cache.db WAL size backstop (gathered outside the deep branch).
2253
2435
  cache_db_wal_bytes=cache_db_wal_bytes,
2436
+ conversations_db_wal_bytes=conversations_db_wal_bytes,
2254
2437
  # #374: quarantined same-revision groups + structural protocol violation.
2255
2438
  journal_conflicts=journal_conflicts,
2256
2439
  journal_protocol_violations=journal_protocol_violations,