cctally 1.91.0 → 1.92.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/README.md +4 -2
  3. package/bin/_cctally_cache.py +903 -74
  4. package/bin/_cctally_config.py +57 -0
  5. package/bin/_cctally_core.py +94 -14
  6. package/bin/_cctally_dashboard.py +217 -19
  7. package/bin/_cctally_dashboard_conversation.py +170 -20
  8. package/bin/_cctally_dashboard_envelope.py +2 -0
  9. package/bin/_cctally_db.py +481 -19
  10. package/bin/_cctally_doctor.py +18 -1
  11. package/bin/_cctally_journal.py +1156 -21
  12. package/bin/_cctally_journal_repair.py +6 -0
  13. package/bin/_cctally_parser.py +26 -0
  14. package/bin/_cctally_quota.py +171 -55
  15. package/bin/_cctally_record.py +13 -1
  16. package/bin/_cctally_rederive.py +4 -0
  17. package/bin/_cctally_statusline.py +6 -6
  18. package/bin/_cctally_store.py +1061 -40
  19. package/bin/_cctally_transcript.py +32 -2
  20. package/bin/_cctally_tui.py +54 -6
  21. package/bin/_lib_cache_report.py +8 -3
  22. package/bin/_lib_codex_conversation.py +851 -81
  23. package/bin/_lib_codex_conversation_query.py +2031 -96
  24. package/bin/_lib_codex_find_projection.py +517 -0
  25. package/bin/_lib_codex_harness_preamble.py +176 -0
  26. package/bin/_lib_codex_hooks.py +5 -3
  27. package/bin/_lib_codex_js_scan.py +254 -0
  28. package/bin/_lib_codex_landmarks.py +309 -0
  29. package/bin/_lib_codex_title_clean.py +116 -0
  30. package/bin/_lib_conversation_dispatch.py +168 -22
  31. package/bin/_lib_conversation_query.py +62 -2
  32. package/bin/_lib_conversation_watch.py +4 -2
  33. package/bin/_lib_doctor.py +64 -0
  34. package/bin/_lib_quota_alert_axes.py +31 -34
  35. package/bin/_lib_stats_damage.py +523 -0
  36. package/bin/_lib_stats_publish.py +243 -0
  37. package/bin/cctally +17 -3
  38. package/dashboard/static/assets/index-Dat-mza6.js +97 -0
  39. package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
  40. package/dashboard/static/dashboard.html +2 -2
  41. package/package.json +8 -1
  42. package/dashboard/static/assets/index-CILAoEja.js +0 -90
@@ -99,6 +99,7 @@ from _cctally_core import (
99
99
  # the migration gate-defer diagnostic routes through _lib_log so
100
100
  # CCTALLY_DEBUG verbosity is decided in one place.
101
101
  import _lib_log
102
+ from _lib_codex_find_projection import CODEX_FIND_PROJECTION_VERSION
102
103
 
103
104
 
104
105
  # Production cache dispatchers hold maintenance-exclusive + the global cache
@@ -275,6 +276,16 @@ class StatsDbCorruptError(sqlite3.DatabaseError):
275
276
  """
276
277
 
277
278
 
279
+ class StatsPublicationFailedError(StatsDbCorruptError):
280
+ """A replacement stats index was published and then FAILED validation.
281
+
282
+ Distinct from ``StatsDbCorruptError``'s ordinary case because replacement
283
+ already occurred, so the inherited "Not auto-recreated" wording would be
284
+ false. Subclasses it deliberately: every graceful-degrade site and the CLI
285
+ boundary's staged exit 3 keep applying unchanged (#496 S1 F1).
286
+ """
287
+
288
+
278
289
  class StatsDbMaintenanceError(sqlite3.OperationalError):
279
290
  """A guided repair owns stats.db; new cctally opens must stay out.
280
291
 
@@ -302,31 +313,84 @@ class StatsEpochMismatchError(sqlite3.DatabaseError):
302
313
  DB failure; ``main()`` maps it to a staged exit 3."""
303
314
 
304
315
 
305
- class StatsEpochRebuildDeferred(BaseException):
306
- """A readable wrong-epoch stats index is rebuilding out of process.
316
+ class StatsRebuildDeferred(BaseException):
317
+ """A stats.db rebuild this caller must not perform inline runs detached.
318
+
319
+ The shared parent of both deferral signals (#496 S3 §6), so every catch
320
+ site is widened ONCE rather than growing a second name each time a rebuild
321
+ class is detached. ``outcome`` records whether this caller spawned the
322
+ worker, observed an existing attempt, or could not spawn it.
307
323
 
308
- Ordinary callers must not read the schema-incompatible old index or pay
309
- whole-journal replay latency inline. ``outcome`` records whether this
310
- caller spawned the worker, observed an existing attempt, or could not
311
- spawn it; ``main()`` maps every case to prompt retry guidance and exit 3.
312
324
  This deliberately derives directly from ``BaseException``: reporting
313
325
  kernels contain many broad ``Exception`` / ``sqlite3.DatabaseError``
314
326
  fallbacks that turn missing optional data into ``n/a``. Swallowing this
315
327
  control signal there would publish a misleading partial report. The CLI
316
328
  boundary, dashboard, and statusline catch it explicitly; ``finally``
317
329
  cleanup still runs normally.
330
+
331
+ The two subclasses are NOT interchangeable at the degrade sites: the epoch
332
+ path describes a readable index at the wrong version and reports
333
+ ``corruption=False``, while a deferred heal describes an index that could
334
+ not be read and must report ``corruption=True``. Flattening them makes the
335
+ dashboard name the wrong fault.
318
336
  """
319
337
 
320
- def __init__(self, outcome: str) -> None:
338
+ def __init__(self, outcome: str, message: str) -> None:
321
339
  self.outcome = str(outcome)
340
+ super().__init__(message)
341
+
342
+
343
+ class StatsEpochRebuildDeferred(StatsRebuildDeferred):
344
+ """A readable wrong-epoch stats index is rebuilding out of process.
345
+
346
+ Ordinary callers must not read the schema-incompatible old index or pay
347
+ whole-journal replay latency inline; ``main()`` maps every case to prompt
348
+ retry guidance and exit 3.
349
+ """
350
+
351
+ def __init__(self, outcome: str) -> None:
352
+ outcome = str(outcome)
322
353
  message = (
323
354
  "could not start the stats.db index epoch rebuild; retry this "
324
355
  "command shortly"
325
- if self.outcome == "failed"
356
+ if outcome == "failed"
326
357
  else "stats.db index epoch rebuild is running in the background; "
327
358
  "retry shortly"
328
359
  )
329
- super().__init__(message)
360
+ super().__init__(outcome, message)
361
+
362
+
363
+ class StatsHealDeferred(StatsRebuildDeferred):
364
+ """A corrupt stats index is being rebuilt by the detached heal worker.
365
+
366
+ Replaces the inline heal's return value on the path where it used to
367
+ quarantine and rebuild while the caller waited (#496 S3 §6). The caller
368
+ degrades where it already degrades; it never blocks on the rebuild.
369
+
370
+ ``heal_id`` correlates this signal with the durable heal event the hook
371
+ recorded at detection, and ``forensics_path`` is the absolute bundle the
372
+ user was told about. Both are attributes rather than message text so a
373
+ consumer can use them without parsing.
374
+ """
375
+
376
+ def __init__(
377
+ self,
378
+ outcome: str,
379
+ *,
380
+ heal_id: "str | None" = None,
381
+ forensics_path: "str | None" = None,
382
+ ) -> None:
383
+ outcome = str(outcome)
384
+ self.heal_id = heal_id
385
+ self.forensics_path = forensics_path
386
+ message = (
387
+ "could not start the stats.db corruption rebuild; retry this "
388
+ "command shortly"
389
+ if outcome == "failed"
390
+ else "stats.db corruption rebuild is running in the background; "
391
+ "retry shortly"
392
+ )
393
+ super().__init__(outcome, message)
330
394
 
331
395
 
332
396
  _SQLITE_CORRUPTION_MESSAGES = (
@@ -778,6 +842,114 @@ def _corruption_trigger_record(
778
842
  }
779
843
 
780
844
 
845
+ def _capture_corruption_wal_evidence(
846
+ db_path: pathlib.Path, ts: str, *, db_label: str, trigger_exception,
847
+ ) -> dict:
848
+ """Copy the WAL/SHM bytes present at forensics time (#496 S1 F2).
849
+
850
+ Scoped to stats.db. Retention for `logs/stats.db-corruption-forensics-<ts>/`
851
+ is handed to S6 alongside the other new log families; nobody owns retention
852
+ for a cache.db or conversations.db evidence directory, and those WALs are
853
+ not bounded by stats.db's 16 MiB `journal_size_limit` — `cache.db-wal` has
854
+ been observed above 256 MiB under a multi-agent hook storm (#297).
855
+ Generalizing this stays available to a later session that brings its own
856
+ retention story.
857
+
858
+ This is a FORENSICS-TIME capture, not an exact detection-time one, and the
859
+ difference is real: the connection whose failure classified the corruption
860
+ is closed inside ``open_db`` before the heal hook runs, and the heal's
861
+ locked re-check probe runs before this, so a clean close of that read-write
862
+ connection could already have checkpointed the WAL if it was the last
863
+ handle. Closing that residual gap means capturing at the ``open_db``
864
+ corruption boundary, which is not this session's surface.
865
+
866
+ Field evidence bounds the cost: 41 of the 74 retained production bundles
867
+ recorded a non-empty WAL at exactly this point.
868
+
869
+ Both the heal's re-check and the bundle's own integrity probe open
870
+ read-only, and a read-only connection cannot checkpoint, so nothing between
871
+ this capture and those probes can empty the WAL.
872
+ """
873
+ empty = {"disposition": None, "path": None, "bytes": {}, "reason": None}
874
+ if db_label != "stats":
875
+ # Recorded rather than omitted, so a cache bundle still says WHY it
876
+ # carries no evidence.
877
+ return {**empty, "disposition": "skipped_not_stats"}
878
+ if trigger_exception is None or not _is_sqlite_corruption_error(
879
+ trigger_exception
880
+ ):
881
+ # A deliberate `db rebuild` on a healthy index must not accumulate
882
+ # evidence directories.
883
+ return {**empty, "disposition": "skipped_not_corruption"}
884
+ wal = pathlib.Path(str(db_path) + "-wal")
885
+ try:
886
+ wal_bytes = wal.stat().st_size
887
+ except OSError:
888
+ wal_bytes = 0
889
+ if wal_bytes <= 0:
890
+ return {**empty, "disposition": "skipped_empty"}
891
+ try:
892
+ evidence = (
893
+ _cctally_core.LOG_DIR
894
+ / f"{db_path.name}-corruption-forensics-{ts}"
895
+ )
896
+ evidence.mkdir(mode=0o700, parents=True, exist_ok=True)
897
+ # The main file is deliberately NOT copied: it is tens of megabytes,
898
+ # and the cutover's preservation retains it anyway.
899
+ _copy_db_family(
900
+ db_path, evidence / db_path.name, suffixes=("-wal", "-shm"),
901
+ )
902
+ copied: dict = {}
903
+ for suffix in ("-wal", "-shm"):
904
+ member = evidence / f"{db_path.name}{suffix}"
905
+ try:
906
+ copied[member.name] = member.stat().st_size
907
+ except OSError:
908
+ copied[member.name] = None
909
+ return {
910
+ "disposition": "captured",
911
+ "path": str(evidence),
912
+ "bytes": copied,
913
+ "reason": None,
914
+ }
915
+ except Exception as exc: # noqa: BLE001 — enrichment never breaks a heal
916
+ # Deliberately broader than OSError. An escaping exception propagates
917
+ # out of `write_corruption_forensics`, and in the cache path that
918
+ # caller's own `except Exception` then DECLINES destructive recovery —
919
+ # so a diagnostic copy failure would change what the heal does.
920
+ return {
921
+ **empty,
922
+ "disposition": "failed",
923
+ "reason": _bounded_forensics_text(
924
+ exc, _FORENSICS_EXCEPTION_MESSAGE_MAX,
925
+ ),
926
+ }
927
+
928
+
929
+ def _describe_corruption_damage(integrity_rows, probe_db_path) -> dict:
930
+ """Structured damage description, or a recorded reason it is unavailable.
931
+
932
+ A forensics bundle must never fail to write because characterization
933
+ failed, so every exception is captured rather than propagated (#496 S1 F8).
934
+ """
935
+ try:
936
+ import _lib_stats_damage
937
+
938
+ return _lib_stats_damage.describe_damage(
939
+ integrity_rows=integrity_rows, path=probe_db_path,
940
+ )
941
+ except Exception as exc:
942
+ return {
943
+ "schemaVersion": 1,
944
+ "method": "unavailable",
945
+ "findings": [],
946
+ "shapeToken": "none",
947
+ "reason": _bounded_forensics_text(
948
+ exc, _FORENSICS_EXCEPTION_MESSAGE_MAX,
949
+ ),
950
+ }
951
+
952
+
781
953
  def write_corruption_forensics(
782
954
  db_path,
783
955
  *,
@@ -834,6 +1006,12 @@ def write_corruption_forensics(
834
1006
  bundle["trigger"] = _corruption_trigger_record(
835
1007
  trigger_origin, trigger_exception,
836
1008
  )
1009
+ # Capture the WAL before the integrity probe, and before anything else can
1010
+ # disturb it (#496 S1 F2). Shares the bundle's timestamp stem so the pair
1011
+ # is obvious.
1012
+ bundle["walEvidence"] = _capture_corruption_wal_evidence(
1013
+ db_path, ts, db_label=db_label, trigger_exception=trigger_exception,
1014
+ )
837
1015
  for suffix in ("", "-wal", "-shm"):
838
1016
  p = pathlib.Path(str(db_path) + suffix)
839
1017
  try:
@@ -876,6 +1054,11 @@ def write_corruption_forensics(
876
1054
  reason = "integrity_check_unavailable"
877
1055
  bundle["probeDisposition"] = disposition.value
878
1056
  bundle["probeReason"] = reason
1057
+ # Scanned against `probe_db_path`, the same artifact `integrityCheck`
1058
+ # describes, so a `both` verdict never mixes two files.
1059
+ bundle["damage"] = _describe_corruption_damage(
1060
+ bundle["integrityCheck"], probe_db_path,
1061
+ )
879
1062
  try:
880
1063
  cp = subprocess.run(
881
1064
  ["lsof", "--", str(db_path)],
@@ -1196,7 +1379,15 @@ def cmd_db_rebuild(args: argparse.Namespace) -> int:
1196
1379
  # authorizer-armed `open_db(_target_path=...)` connection, and we
1197
1380
  # hold maintenance exclusive, which is what spec §3.1 sanctions.
1198
1381
  with _cctally_store.stats_write_scope("maintenance-rebuild"):
1199
- result = _cctally_journal.rebuild_stats_index()
1382
+ result = _cctally_journal.rebuild_stats_index(
1383
+ context=_cctally_journal.RebuildContext(
1384
+ trigger="db-rebuild",
1385
+ trigger_error=None,
1386
+ forensics_path=(
1387
+ str(forensics) if forensics is not None else None
1388
+ ),
1389
+ )
1390
+ )
1200
1391
  incident = result.quarantine_dir
1201
1392
  except Exception as exc:
1202
1393
  eprint(f"cctally: stats.db rebuild failed: {exc}")
@@ -3781,7 +3972,8 @@ def _apply_cache_schema(conn: sqlite3.Connection) -> None:
3781
3972
  ON codex_conversation_rollups(last_activity_utc DESC, conversation_key DESC);
3782
3973
 
3783
3974
  -- codex_conversation_file_touches: write-class axis for the `files`
3784
- -- search kind + outline file stats (§3.3). source_path gives explicit
3975
+ -- search kind (§3.3). The richer outline file stats stay derived from
3976
+ -- retained event payloads at read time. source_path gives explicit
3785
3977
  -- lineage so the per-file delete/truncate/prune paths scope deletions
3786
3978
  -- exactly as they do for the other Codex families.
3787
3979
  CREATE TABLE IF NOT EXISTS codex_conversation_file_touches (
@@ -4145,7 +4337,8 @@ def _apply_conversations_schema(conn: sqlite3.Connection) -> None:
4145
4337
  ).fetchone()
4146
4338
  except sqlite3.OperationalError:
4147
4339
  current = None
4148
- if current is not None and current[0] == "1":
4340
+ if current is not None and current[0] == "2":
4341
+ _apply_codex_find_projection_schema(conn)
4149
4342
  return
4150
4343
 
4151
4344
  _apply_cache_schema(conn)
@@ -4195,11 +4388,85 @@ def _apply_conversations_schema(conn: sqlite3.Connection) -> None:
4195
4388
  );
4196
4389
  """
4197
4390
  )
4391
+ # #347: transcript attribution belongs to the physical rows. These are
4392
+ # pure column additions; migration 005 owns only the one-time legacy
4393
+ # backfill policy. NULL is the stored spelling of the `unattributed`
4394
+ # sentinel, matching the accounting cache contract from #341.
4395
+ add_column_if_missing(conn, "conversation_messages", "account_key", "TEXT")
4396
+ add_column_if_missing(conn, "codex_conversation_events", "account_key", "TEXT")
4397
+ add_column_if_missing(conn, "codex_conversation_messages", "account_key", "TEXT")
4398
+ conn.execute(
4399
+ "CREATE INDEX IF NOT EXISTS idx_conv_messages_account_session "
4400
+ "ON conversation_messages(account_key, session_id, timestamp_utc, id)"
4401
+ )
4402
+ conn.execute(
4403
+ "CREATE INDEX IF NOT EXISTS idx_codex_conv_messages_account_conversation "
4404
+ "ON codex_conversation_messages(account_key, conversation_key, timestamp_utc, id)"
4405
+ )
4406
+ conn.execute(
4407
+ "CREATE INDEX IF NOT EXISTS idx_codex_events_account_conversation "
4408
+ "ON codex_conversation_events(account_key, conversation_key, line_offset)"
4409
+ )
4198
4410
  conn.execute(
4199
4411
  "INSERT INTO cache_meta(key,value) VALUES "
4200
- "('conversation_schema_version','1') "
4412
+ "('conversation_schema_version','2') "
4201
4413
  "ON CONFLICT(key) DO UPDATE SET value=excluded.value"
4202
4414
  )
4415
+ _apply_codex_find_projection_schema(conn)
4416
+
4417
+
4418
+ def _apply_codex_find_projection_schema(conn: sqlite3.Connection) -> None:
4419
+ """Create #482's disposable visible-text projection tables.
4420
+
4421
+ This helper is called even when the conversations base-schema marker is
4422
+ current: migration 004 must be able to add the derivation without bumping
4423
+ or replaying the much larger historical transcript schema.
4424
+ """
4425
+ conn.executescript(
4426
+ """
4427
+ CREATE TABLE IF NOT EXISTS codex_find_projection (
4428
+ message_id INTEGER NOT NULL,
4429
+ conversation_key TEXT NOT NULL,
4430
+ item_key TEXT NOT NULL,
4431
+ block_key TEXT NOT NULL,
4432
+ container_block_key TEXT NOT NULL,
4433
+ surface TEXT NOT NULL
4434
+ CHECK(surface IN ('body','call','output','completion')),
4435
+ render_order INTEGER NOT NULL,
4436
+ projected_text TEXT NOT NULL,
4437
+ leaves_json TEXT NOT NULL,
4438
+ disclosure_json TEXT NOT NULL,
4439
+ projection_version INTEGER NOT NULL,
4440
+ PRIMARY KEY(message_id, surface),
4441
+ FOREIGN KEY(message_id) REFERENCES codex_conversation_messages(id)
4442
+ ON DELETE CASCADE
4443
+ );
4444
+ CREATE INDEX IF NOT EXISTS idx_codex_find_projection_conversation_order
4445
+ ON codex_find_projection(
4446
+ conversation_key, render_order, message_id, surface
4447
+ );
4448
+ CREATE TRIGGER IF NOT EXISTS codex_find_projection_message_ad
4449
+ AFTER DELETE ON codex_conversation_messages BEGIN
4450
+ DELETE FROM codex_find_projection WHERE message_id=old.id;
4451
+ END;
4452
+ """
4453
+ )
4454
+ # Fresh stores are complete without a backfill. Populated upgrades are
4455
+ # armed by migration 004 under the provider flock.
4456
+ has_state = conn.execute(
4457
+ "SELECT 1 FROM cache_meta WHERE key IN "
4458
+ "('codex_find_projection_complete_version',"
4459
+ " 'codex_find_projection_backfill_pending') LIMIT 1"
4460
+ ).fetchone()
4461
+ if has_state is None and conn.execute(
4462
+ "SELECT 1 FROM codex_conversation_messages LIMIT 1"
4463
+ ).fetchone() is None:
4464
+ _set_cache_meta(
4465
+ conn,
4466
+ "codex_find_projection_complete_version",
4467
+ str(CODEX_FIND_PROJECTION_VERSION),
4468
+ )
4469
+ _set_cache_meta(conn, "codex_find_projection_generation", "0")
4203
4470
 
4204
4471
 
4205
4472
  def _fts5_available(conn: sqlite3.Connection) -> bool:
@@ -4336,6 +4603,181 @@ def _conv_003_background_mcp_result_replay(conn: sqlite3.Connection) -> None:
4336
4603
  _release_cache_db_writer_flocks(held)
4337
4604
 
4338
4605
 
4606
+ @conversations_migration("004_codex_find_projection")
4607
+ def _conv_004_codex_find_projection(conn: sqlite3.Connection) -> None:
4608
+ """Arm #482's bounded projection backfill for populated Codex stores.
4609
+
4610
+ The projection is fully derivable from retained normalized messages and
4611
+ event payloads, so the migration never replays JSONL. The normal Codex
4612
+ conversation synchronizer consumes the marker under the same provider
4613
+ flock, in bounded batches. Fresh empty stores were certified by the schema
4614
+ helper and this handler becomes an idempotent no-op.
4615
+ """
4616
+ held = _acquire_conversations_db_codex_provider_flock(
4617
+ conn, migration="conversations 004 Codex find projection")
4618
+ try:
4619
+ _apply_codex_find_projection_schema(conn)
4620
+ complete = conn.execute(
4621
+ "SELECT 1 FROM cache_meta "
4622
+ "WHERE key='codex_find_projection_complete_version' AND value=?",
4623
+ (str(CODEX_FIND_PROJECTION_VERSION),),
4624
+ ).fetchone()
4625
+ if complete is None:
4626
+ _set_cache_meta(conn, "codex_find_projection_backfill_pending", "1")
4627
+ conn.execute(
4628
+ "INSERT OR IGNORE INTO cache_meta(key,value) VALUES"
4629
+ "('codex_find_projection_backfill_cursor','0')"
4630
+ )
4631
+ _set_cache_meta(
4632
+ conn,
4633
+ "codex_find_projection_backfill_version",
4634
+ str(CODEX_FIND_PROJECTION_VERSION),
4635
+ )
4636
+ _set_cache_meta(conn, "codex_find_projection_generation", "0")
4637
+ conn.commit()
4638
+ finally:
4639
+ _release_cache_db_writer_flocks(held)
4640
+
4641
+
4642
+ @conversations_migration("005_conversation_account_dimension")
4643
+ def _conv_005_conversation_account_dimension(conn: sqlite3.Connection) -> None:
4644
+ """Backfill the #347 transcript account dimension.
4645
+
4646
+ Decision R4 from #341 is the compatibility rule: retained Claude history
4647
+ belongs to the journaled cutover account, while retained Codex history is
4648
+ genuinely unrecoverable and remains NULL/``unattributed``. The physical
4649
+ columns are added by the base-schema helper before dispatch; this handler
4650
+ owns only the data-shape transition and is idempotent.
4651
+ """
4652
+ import _cctally_journal
4653
+ import _lib_accounts
4654
+
4655
+ held = _acquire_conversations_db_claude_provider_flock(
4656
+ conn, migration="conversations 005 account dimension"
4657
+ )
4658
+ try:
4659
+ held += _acquire_conversations_db_codex_provider_flock(
4660
+ conn, migration="conversations 005 account dimension"
4661
+ )
4662
+ claude_key = _cctally_journal.find_accounts_cutover_op()
4663
+ if claude_key is None:
4664
+ pending = conn.execute(
4665
+ "SELECT 1 FROM conversation_messages "
4666
+ "WHERE account_key IS NULL LIMIT 1"
4667
+ ).fetchone()
4668
+ if pending is not None:
4669
+ raise MigrationGateNotMet(
4670
+ "accounts cutover op not yet appended; deferring Claude "
4671
+ "conversation backfill until the epoch transition records it"
4672
+ )
4673
+ claude_key = _lib_accounts.UNATTRIBUTED
4674
+ stored_key = (
4675
+ None if claude_key == _lib_accounts.UNATTRIBUTED else claude_key
4676
+ )
4677
+ conn.execute(
4678
+ "UPDATE conversation_messages SET account_key=? "
4679
+ "WHERE account_key IS NULL",
4680
+ (stored_key,),
4681
+ )
4682
+ # Explicitly preserve the pre-feature Codex decision. This UPDATE makes
4683
+ # the idempotent intent visible in the migration golden without guessing
4684
+ # from whichever auth.json is active during upgrade.
4685
+ conn.execute(
4686
+ "UPDATE codex_conversation_events SET account_key=NULL "
4687
+ "WHERE account_key IS NULL"
4688
+ )
4689
+ conn.execute(
4690
+ "UPDATE codex_conversation_messages SET account_key=NULL "
4691
+ "WHERE account_key IS NULL"
4692
+ )
4693
+ _set_cache_meta(conn, "conversation_account_dimension", "1")
4694
+ conn.commit()
4695
+ finally:
4696
+ _release_cache_db_writer_flocks(held)
4697
+
4698
+
4699
+ @conversations_migration("006_backfill_codex_file_touches")
4700
+ def _conv_006_backfill_codex_file_touches(conn: sqlite3.Connection) -> None:
4701
+ """Repair the file-search projection for retained dict-shaped patches.
4702
+
4703
+ The normalized writer historically recognized only list-shaped ``changes``;
4704
+ real Codex patch completions use an object keyed by file path. The physical
4705
+ event payloads are retained unbounded, so rebuild the derived touch rows from
4706
+ those authoritative bytes and link only events that still have a normalized
4707
+ message. ``INSERT OR IGNORE`` preserves valid legacy-list rows and makes a
4708
+ markerless retry idempotent.
4709
+ """
4710
+ import _lib_codex_conversation as conversation
4711
+
4712
+ held = _acquire_conversations_db_codex_provider_flock(
4713
+ conn, migration="conversations 006 Codex file touches")
4714
+ try:
4715
+ pending: list[tuple[int, str, str, str, str]] = []
4716
+ for message_id, conversation_key, source_path, payload_json in conn.execute(
4717
+ "SELECT m.id,m.conversation_key,m.source_path,e.payload_json "
4718
+ "FROM codex_conversation_events e "
4719
+ "JOIN codex_conversation_messages m "
4720
+ "ON m.source_path=e.source_path AND m.line_offset=e.line_offset "
4721
+ "WHERE e.event_type='patch_apply_end'"
4722
+ ).fetchall():
4723
+ try:
4724
+ decoded = json.loads(payload_json or "{}")
4725
+ except (json.JSONDecodeError, TypeError):
4726
+ continue
4727
+ payload = decoded.get("payload") if isinstance(decoded, dict) else None
4728
+ for file_path in conversation.codex_patch_file_paths(payload):
4729
+ pending.append((
4730
+ message_id, conversation_key, source_path,
4731
+ file_path, "apply_patch",
4732
+ ))
4733
+ if pending:
4734
+ conn.executemany(
4735
+ "INSERT OR IGNORE INTO codex_conversation_file_touches "
4736
+ "(message_id,conversation_key,source_path,file_path,tool) "
4737
+ "VALUES(?,?,?,?,?)",
4738
+ pending,
4739
+ )
4740
+ conn.commit()
4741
+ finally:
4742
+ _release_cache_db_writer_flocks(held)
4743
+
4744
+
4745
+ @conversations_migration("007_codex_find_projection_v2_meta")
4746
+ def _conv_007_codex_find_projection_v2_meta(conn: sqlite3.Connection) -> None:
4747
+ """Arm a bounded v2 rebuild so retained visible meta bodies become findable.
4748
+
4749
+ The projection is disposable. Preserve a markerless retry's v2 cursor, but
4750
+ restart a v1/incompletely-versioned backfill at zero so no conversation
4751
+ already passed by that older walk keeps stale rows.
4752
+ """
4753
+ held = _acquire_conversations_db_codex_provider_flock(
4754
+ conn, migration="conversations 007 Codex find projection v2 meta")
4755
+ try:
4756
+ target = str(CODEX_FIND_PROJECTION_VERSION)
4757
+ complete = conn.execute(
4758
+ "SELECT 1 FROM cache_meta "
4759
+ "WHERE key='codex_find_projection_complete_version' AND value=?",
4760
+ (target,),
4761
+ ).fetchone()
4762
+ if complete is not None:
4763
+ return
4764
+ pending_version = conn.execute(
4765
+ "SELECT value FROM cache_meta "
4766
+ "WHERE key='codex_find_projection_backfill_version'"
4767
+ ).fetchone()
4768
+ _set_cache_meta(conn, "codex_find_projection_backfill_pending", "1")
4769
+ if pending_version is None or pending_version[0] != target:
4770
+ _set_cache_meta(conn, "codex_find_projection_backfill_cursor", "0")
4771
+ _set_cache_meta(conn, "codex_find_projection_backfill_version", target)
4772
+ conn.execute(
4773
+ "DELETE FROM cache_meta "
4774
+ "WHERE key='codex_find_projection_complete_version'"
4775
+ )
4776
+ conn.commit()
4777
+ finally:
4778
+ _release_cache_db_writer_flocks(held)
4779
+
4780
+
4339
4781
  # #177 S6: the consolidated multi-column external-content FTS5 table that
4340
4782
  # replaces the old conversation_fts(text) + conversation_fts_aux(search_aux)
4341
4783
  # pair. The three column names MUST match the conversation_messages columns BY
@@ -4760,6 +5202,7 @@ def _codex_conversation_fts_full_clear(conn: sqlite3.Connection) -> None:
4760
5202
  "INSERT INTO codex_conversation_fts(codex_conversation_fts) VALUES('delete-all')")
4761
5203
  _create_codex_conversation_fts_triggers(conn)
4762
5204
  for stmt in (
5205
+ "DELETE FROM codex_find_projection",
4763
5206
  "DELETE FROM codex_conversation_file_touches",
4764
5207
  "DELETE FROM codex_conversation_rollups",
4765
5208
  ):
@@ -6198,8 +6641,7 @@ def _028_split_conversation_store(conn: sqlite3.Connection) -> None:
6198
6641
  "SELECT 1 FROM sqlite_master "
6199
6642
  "WHERE type='table' AND name='codex_conversation_events'"
6200
6643
  ).fetchone() is not None:
6201
- conn.execute(
6202
- "UPDATE quota_window_snapshots AS q SET observed_model=("
6644
+ model_lookup = (
6203
6645
  " SELECT json_extract(e.payload_json, '$.payload.model')"
6204
6646
  " FROM codex_conversation_events AS e"
6205
6647
  " WHERE e.source_path=q.source_path"
@@ -6207,9 +6649,21 @@ def _028_split_conversation_store(conn: sqlite3.Connection) -> None:
6207
6649
  " AND e.record_type IN ('turn_context','session_meta')"
6208
6650
  " AND json_valid(e.payload_json)"
6209
6651
  " AND json_type(e.payload_json, '$.payload.model')='text'"
6210
- " ORDER BY e.line_offset DESC LIMIT 1)"
6652
+ " ORDER BY e.line_offset DESC LIMIT 1"
6653
+ )
6654
+ changed = conn.execute(
6655
+ "UPDATE quota_window_snapshots AS q SET observed_model=("
6656
+ + model_lookup + ")"
6211
6657
  " WHERE q.source='codex' AND q.observed_model IS NULL"
6658
+ " AND (" + model_lookup + ") IS NOT NULL"
6212
6659
  )
6660
+ if changed.rowcount:
6661
+ # #457: migration DML changes the same physical quota
6662
+ # inputs as ordinary ingest, so invalidate every consumer
6663
+ # in this transaction. The IS NOT NULL guard keeps a
6664
+ # markerless retry byte-idempotent, including this token.
6665
+ import _cctally_cache
6666
+ _cctally_cache._bump_codex_physical_mutation_seq(conn)
6213
6667
  for drop in (
6214
6668
  "DROP TRIGGER IF EXISTS conv_fts_ai",
6215
6669
  "DROP TRIGGER IF EXISTS conv_fts_ad",
@@ -6854,8 +7308,10 @@ def _039_codex_quota_observed_model_backfill(conn: sqlite3.Connection) -> None:
6854
7308
  AND whose lookup actually resolves are considered, so a re-run over its own
6855
7309
  output writes nothing at all — not even a NULL-to-NULL update, which would
6856
7310
  fire the ledger trigger and dirty a window that never changed. Its real DML
6857
- IS ledgered, which is the mechanism working as designed: a migration that
6858
- rewrites this column no longer has to remember to announce it.
7311
+ IS ledgered, which is that mechanism working as designed: a migration that
7312
+ rewrites this column no longer has to remember a ledger-specific
7313
+ announcement. The independent physical mutation token still advances for
7314
+ the dashboard/certificate consumers that consult it before reconciliation.
6859
7315
 
6860
7316
  The migration is the ONE-TIME leg. It is not the whole guarantee: `db skip`
6861
7317
  and a fresh journal-repopulated cache both bypass it, so ``sync_codex_cache``
@@ -6863,7 +7319,13 @@ def _039_codex_quota_observed_model_backfill(conn: sqlite3.Connection) -> None:
6863
7319
 
6864
7320
  NO self-stamp — the dispatcher central-stamps on a clean return (#140).
6865
7321
  """
6866
- backfill_codex_quota_observed_model(conn)
7322
+ changed = backfill_codex_quota_observed_model(conn)
7323
+ if changed:
7324
+ # #457: the ledger invalidates the incremental projector, while this
7325
+ # shared token independently invalidates certificate/coherence and
7326
+ # dashboard/TUI snapshot consumers. Both move in this transaction.
7327
+ import _cctally_cache
7328
+ _cctally_cache._bump_codex_physical_mutation_seq(conn)
6867
7329
  conn.commit()
6868
7330
 
6869
7331
 
@@ -971,6 +971,7 @@ def _doctor_gather_state_impl(
971
971
  conv_rollup_sync_in_progress = False
972
972
  conversations_db_page_count = None
973
973
  conversations_db_freelist_count = None
974
+ codex_prune_refusals: list[dict] = []
974
975
  try:
975
976
  if _cctally_core.CONVERSATIONS_DB_PATH.exists():
976
977
  # This gather also runs inside dashboard snapshot precompute. A
@@ -1009,6 +1010,18 @@ def _doctor_gather_state_impl(
1009
1010
  conv_messages_distinct_sessions = int(row[0])
1010
1011
  except sqlite3.OperationalError:
1011
1012
  pass
1013
+ try:
1014
+ import _cctally_cache as _cc_sib
1015
+ row = conn.execute(
1016
+ "SELECT value FROM cache_meta WHERE key=?",
1017
+ (_cc_sib.CODEX_ORPHAN_PRUNE_REFUSED_KEY,),
1018
+ ).fetchone()
1019
+ if row and row[0]:
1020
+ record = json.loads(row[0])
1021
+ if isinstance(record, dict):
1022
+ codex_prune_refusals.append(record)
1023
+ except (sqlite3.OperationalError, ValueError, TypeError):
1024
+ pass
1012
1025
  # Pending reingest/split/backfill flags ⇒ a full sync hasn't yet
1013
1026
  # reconciled the rollup. Read the canonical flag set from
1014
1027
  # _cctally_cache so it stays in lockstep with the sync consumers.
@@ -1219,7 +1232,8 @@ def _doctor_gather_state_impl(
1219
1232
  try:
1220
1233
  for _key in ("parse_health_claude", "parse_health_codex",
1221
1234
  "codex_torn_auth_deferred", _blocked_key,
1222
- _deferred_key, "codex_ingest_backlog"):
1235
+ _deferred_key, "codex_ingest_backlog",
1236
+ "codex_orphan_prune_refused"):
1223
1237
  try:
1224
1238
  row = conn.execute(
1225
1239
  "SELECT value FROM cache_meta WHERE key = ?",
@@ -1238,6 +1252,8 @@ def _doctor_gather_state_impl(
1238
1252
  codex_replay_deferred = _parsed
1239
1253
  elif _key == "codex_ingest_backlog":
1240
1254
  codex_ingest_backlog = _parsed
1255
+ elif _key == "codex_orphan_prune_refused":
1256
+ codex_prune_refusals.append(_parsed)
1241
1257
  else:
1242
1258
  codex_torn_deferred = _parsed
1243
1259
  except (sqlite3.OperationalError, ValueError):
@@ -1828,6 +1844,7 @@ def _doctor_gather_state_impl(
1828
1844
  codex_replay_pending=codex_replay_pending,
1829
1845
  codex_replay_blocked=codex_replay_blocked,
1830
1846
  codex_replay_deferred=codex_replay_deferred,
1847
+ codex_prune_refusals=codex_prune_refusals or None,
1831
1848
  stats_db_quick_check=stats_db_quick_check,
1832
1849
  cache_db_quick_check=cache_db_quick_check,
1833
1850
  conversations_db_quick_check=conversations_db_quick_check,