cctally 1.108.0 → 1.110.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/CHANGELOG.md +147 -0
  2. package/README.md +6 -6
  3. package/bin/_cctally_account.py +4 -0
  4. package/bin/_cctally_alerts.py +6 -2
  5. package/bin/_cctally_cache.py +1652 -166
  6. package/bin/_cctally_config.py +3 -2
  7. package/bin/_cctally_core.py +594 -33
  8. package/bin/_cctally_dashboard.py +1585 -201
  9. package/bin/_cctally_dashboard_conversation.py +494 -48
  10. package/bin/_cctally_dashboard_envelope.py +167 -11
  11. package/bin/_cctally_dashboard_share.py +154 -36
  12. package/bin/_cctally_dashboard_sources.py +2615 -422
  13. package/bin/_cctally_db.py +678 -5
  14. package/bin/_cctally_diagnosis_sources.py +238 -26
  15. package/bin/_cctally_doctor.py +99 -18
  16. package/bin/_cctally_five_hour.py +1295 -40
  17. package/bin/_cctally_forecast.py +124 -28
  18. package/bin/_cctally_journal.py +408 -38
  19. package/bin/_cctally_milestone_history.py +150 -14
  20. package/bin/_cctally_parser.py +5 -0
  21. package/bin/_cctally_percent_breakdown.py +6 -1
  22. package/bin/_cctally_project.py +48 -76
  23. package/bin/_cctally_quota.py +39 -7
  24. package/bin/_cctally_quota_model.py +17 -2
  25. package/bin/_cctally_record.py +1647 -443
  26. package/bin/_cctally_rederive.py +899 -30
  27. package/bin/_cctally_refresh.py +2 -1
  28. package/bin/_cctally_share.py +24 -5
  29. package/bin/_cctally_source_analytics.py +1035 -78
  30. package/bin/_cctally_statusline.py +83 -19
  31. package/bin/_cctally_store.py +61 -4
  32. package/bin/_cctally_transcript.py +27 -2
  33. package/bin/_cctally_tui.py +264 -76
  34. package/bin/_cctally_weekrefs.py +28 -5
  35. package/bin/_lib_blocks.py +127 -20
  36. package/bin/_lib_cache_report.py +20 -5
  37. package/bin/_lib_codex_conversation_query.py +297 -64
  38. package/bin/_lib_codex_hooks.py +102 -0
  39. package/bin/_lib_codex_metadata.py +207 -0
  40. package/bin/_lib_conversation.py +119 -0
  41. package/bin/_lib_conversation_dispatch.py +31 -1
  42. package/bin/_lib_conversation_query.py +315 -51
  43. package/bin/_lib_conversation_retention.py +413 -25
  44. package/bin/_lib_dashboard_sources.py +267 -7
  45. package/bin/_lib_diagnosis.py +10 -0
  46. package/bin/_lib_diff_kernel.py +12 -1
  47. package/bin/_lib_doctor.py +230 -8
  48. package/bin/_lib_ingest_frontier.py +1366 -49
  49. package/bin/_lib_journal.py +19 -2
  50. package/bin/_lib_merge_gate.py +307 -0
  51. package/bin/_lib_pricing.py +588 -29
  52. package/bin/_lib_record.py +130 -12
  53. package/bin/_lib_rederive.py +903 -2
  54. package/bin/_lib_render.py +9 -3
  55. package/bin/_lib_retained_size.py +27 -2
  56. package/bin/_lib_segment_summary.py +66 -8
  57. package/bin/_lib_share_templates.py +28 -18
  58. package/bin/_lib_snapshot_cache.py +115 -1
  59. package/bin/_lib_source_retry.py +320 -0
  60. package/bin/_lib_statusline_candidates.py +400 -36
  61. package/bin/_lib_tick_stats.py +119 -0
  62. package/bin/_lib_view_models.py +256 -76
  63. package/bin/cctally +69 -0
  64. package/dashboard/static/assets/ConversationsView-BB9vUiL6.js +72 -0
  65. package/dashboard/static/assets/{DoctorModal-CWn3U4Wl.js → DoctorModal-DFY2iPDf.js} +1 -1
  66. package/dashboard/static/assets/ModalRoot-DvRKHbQE.js +1 -0
  67. package/dashboard/static/assets/ProjectsDrillPanel-DAv39uSo.js +1 -0
  68. package/dashboard/static/assets/SourceDetailModal-DSAWo2Fp.js +1 -0
  69. package/dashboard/static/assets/{UpdateModal-D3m8GG6V.js → UpdateModal-R_9dOVnh.js} +3 -3
  70. package/dashboard/static/assets/dashboardStream.shared-worker-DbG6Ef0T.js +1 -0
  71. package/dashboard/static/assets/index-COhxe3vU.css +1 -0
  72. package/dashboard/static/assets/index-DdU62sov.js +13 -0
  73. package/dashboard/static/assets/outlineNavigation-BOCLcRa6.js +9 -0
  74. package/dashboard/static/assets/useKeymap-CW7Up0Ay.js +1 -0
  75. package/dashboard/static/dashboard.html +3 -3
  76. package/package.json +4 -1
  77. package/dashboard/static/assets/ConversationsView-BOSaBRtu.js +0 -72
  78. package/dashboard/static/assets/ModalRoot-SR070V6c.js +0 -1
  79. package/dashboard/static/assets/ProjectsDrillPanel-QLI9i5mZ.js +0 -1
  80. package/dashboard/static/assets/SourceDetailModal-pv0wlFxl.js +0 -1
  81. package/dashboard/static/assets/dashboardStream.shared-worker-1XTMV3nr.js +0 -1
  82. package/dashboard/static/assets/index-BHu4mxd8.css +0 -1
  83. package/dashboard/static/assets/index-CIWsbux3.js +0 -13
  84. package/dashboard/static/assets/outlineNavigation-CJvKqmLV.js +0 -9
  85. package/dashboard/static/assets/useKeymap-ffqsS5G0.js +0 -1
@@ -136,6 +136,9 @@ import _lib_quota
136
136
  # #496 S5b §4: the journal-to-cache coverage certificate kernel. Stdlib-only,
137
137
  # so it is circular-safe for the same reason.
138
138
  import _lib_cache_coverage
139
+ # #845 §4.1: the decode helper is a stdlib-only leaf that imports nothing from
140
+ # this package, so binding it here is circular-safe for the same reason.
141
+ import _lib_codex_metadata
139
142
 
140
143
 
141
144
  # Module-level back-ref shims for the three out-of-scope JSONL/project
@@ -225,6 +228,16 @@ claude_usage_dict = _load_lib("_lib_pricing").claude_usage_dict
225
228
  # stdlib leaf as the two names above.
226
229
  parse_pricing_fingerprint = _load_lib("_lib_pricing").parse_pricing_fingerprint
227
230
 
231
+ # #728: the four-state observation and the two authorization predicates, from
232
+ # the same circular-safe stdlib leaf. The classifier is pure — the SELECT and
233
+ # its failure mode are reported to it by `_read_pricing_fingerprint_observation`
234
+ # below.
235
+ classify_pricing_fingerprint = _load_lib(
236
+ "_lib_pricing").classify_pricing_fingerprint
237
+ may_write_materialized_cost = _load_lib(
238
+ "_lib_pricing").may_write_materialized_cost
239
+ may_reset_rebuild_target = _load_lib("_lib_pricing").may_reset_rebuild_target
240
+
228
241
  # Shared by the fused per-file walk AND backfill_conversation_messages so the
229
242
  # column list, placeholders, and tuple order live in ONE place — a column
230
243
  # add/reorder can't silently desync the two ingest paths (which would land
@@ -257,6 +270,23 @@ _AI_TITLE_UPSERT_SQL = (
257
270
  "ai_title=excluded.ai_title, source_path=excluded.source_path, byte_offset=excluded.byte_offset"
258
271
  )
259
272
 
273
+ def _ai_title_upsert_sql(table: str = "conversation_ai_titles") -> str:
274
+ """The AI-title upsert, targeted at the live table or its staging twin.
275
+
276
+ A rebuild builds its replacement generation into staging and leaves the
277
+ live table alone until the publish (#752), so the ONE statement that writes
278
+ a title has to be able to name either. Parameterised through this helper
279
+ rather than by two literals, because two literals drift.
280
+ """
281
+ return (
282
+ f"INSERT INTO {table}(session_id,ai_title,source_path,byte_offset) "
283
+ "VALUES(?,?,?,?) "
284
+ "ON CONFLICT(session_id) DO UPDATE SET "
285
+ "ai_title=excluded.ai_title, source_path=excluded.source_path, "
286
+ "byte_offset=excluded.byte_offset"
287
+ )
288
+
289
+
260
290
  # ---------------------------------------------------------------------------
261
291
  # session_entries upsert (#195: extracted from the inline string in sync_cache
262
292
  # so the steady-state and re-walk variants share ONE body).
@@ -428,6 +458,7 @@ def _iter_sync_entries(
428
458
  *,
429
459
  include_cost: bool = True,
430
460
  include_conversations: bool = True,
461
+ with_raw: bool = False,
431
462
  ):
432
463
  """Fused single-pass sync walker (#138). Yields
433
464
  ``(byte_offset, cost_or_None, msgrow_or_None, aititle_or_None)`` for each
@@ -453,19 +484,39 @@ def _iter_sync_entries(
453
484
  partial mid-write tail line (no trailing newline) rewinds the handle and
454
485
  stops, so ``fh.tell()`` after the loop is the cost cursor's ``final_offset``
455
486
  and the next sync re-reads the line once the newline lands.
487
+
488
+ ``with_raw`` (#777) makes the walker yield FIVE-tuples, appending the
489
+ record's exact raw on-disk bytes with the terminator removed, and requires
490
+ ``fh`` to be a BINARY handle. It exists because a durable account stamp is
491
+ keyed by a digest of those bytes, and the decoded text this walker otherwise
492
+ produces is not them: the text path decodes with ``errors="replace"``, so a
493
+ record carrying invalid UTF-8 decodes to a different string than it was
494
+ written as, and its digest would move the day that replacement policy did.
495
+ Re-reading or re-parsing the line to recover the bytes would break the
496
+ one-parse-per-line invariant (#138), so the binary branch decodes the line
497
+ it already read and hands both halves to the same classification below.
498
+ ``fh.tell()`` on a binary handle is a true byte offset rather than the text
499
+ layer's cookie, which is the same number the cursor has always stored.
456
500
  """
501
+ newline = b"\n" if with_raw else "\n"
457
502
  while True:
458
503
  offset = fh.tell()
459
504
  line = fh.readline()
460
505
  if not line:
461
506
  return
462
- if not line.endswith("\n"):
507
+ if not line.endswith(newline):
463
508
  # Partial tail line — writer is mid-flight. Rewind so the next sync
464
509
  # re-reads this line once the newline is in place (and so fh.tell()
465
510
  # reports the cost cursor's stop, never past the partial).
466
511
  fh.seek(offset)
467
512
  return
468
- stripped = line.strip()
513
+ if with_raw:
514
+ raw_span = _lib_conversation.strip_record_terminator(line)
515
+ text = line.decode("utf-8", errors="replace")
516
+ else:
517
+ raw_span = None
518
+ text = line
519
+ stripped = text.strip()
469
520
  if not stripped:
470
521
  continue
471
522
  # #279 S2 F1: passive parse-health counters over the new-byte span.
@@ -496,7 +547,10 @@ def _iter_sync_entries(
496
547
  if include_conversations else None
497
548
  )
498
549
  if cost is not None or mrow is not None or ai is not None:
499
- yield offset, cost, mrow, ai
550
+ if with_raw:
551
+ yield offset, cost, mrow, ai, raw_span
552
+ else:
553
+ yield offset, cost, mrow, ai
500
554
 
501
555
 
502
556
  def _iter_claude_jsonl_files():
@@ -1977,8 +2031,13 @@ def _load_codex_session_files_rows(
1977
2031
  ) -> dict:
1978
2032
  """Cursor rows from ``codex_session_files`` for ONLY the given paths (spec
1979
2033
  §5.1 — the targeted preload must never load every row like the full-sync
1980
- path). Same 13-tuple value shape as ``sync_codex_cache``'s full ``existing``
1981
- map, so the per-file delta logic is byte-identical between the two modes."""
2034
+ path). Same 15-tuple value shape as ``sync_codex_cache``'s full ``existing``
2035
+ map, so the per-file delta logic is byte-identical between the two modes.
2036
+
2037
+ ``device_id``/``inode`` are the last two members and they are NOT
2038
+ diagnostics: the resume decision reads them, because a replacement that
2039
+ lands at the same size is otherwise indistinguishable from a file nothing
2040
+ touched (#769 S6)."""
1982
2041
  out: dict = {}
1983
2042
  if not paths:
1984
2043
  return out
@@ -1986,7 +2045,8 @@ def _load_codex_session_files_rows(
1986
2045
  "path, size_bytes, mtime_ns, last_byte_offset, "
1987
2046
  "last_session_id, last_model, last_total_tokens, source_root_key, "
1988
2047
  "last_native_thread_id, last_root_thread_id, last_parent_thread_id, "
1989
- "last_conversation_key, last_turn_id, ingest_complete"
2048
+ "last_conversation_key, last_turn_id, ingest_complete, "
2049
+ "device_id, inode"
1990
2050
  )
1991
2051
  for i in range(0, len(paths), 400):
1992
2052
  chunk = paths[i:i + 400]
@@ -1995,10 +2055,7 @@ def _load_codex_session_files_rows(
1995
2055
  f"SELECT {cols} FROM codex_session_files WHERE path IN ({placeholders})",
1996
2056
  chunk,
1997
2057
  ):
1998
- out[row[0]] = (
1999
- row[1], row[2], row[3], row[4], row[5], row[6], row[7],
2000
- row[8], row[9], row[10], row[11], row[12], row[13],
2001
- )
2058
+ out[row[0]] = tuple(row[1:])
2002
2059
  return out
2003
2060
 
2004
2061
 
@@ -2712,18 +2769,45 @@ def _recompute_codex_rollups(
2712
2769
  models = sorted({r.model for r in rows if r.model})
2713
2770
  models_json = json.dumps(models) if models else None
2714
2771
  source_root_key = rows[0].source_root_key
2772
+ # #845 §4.7. The two columns are read as BLOBs and decoded field by
2773
+ # field, so one undecodable byte no longer raises here and no longer
2774
+ # fails the whole recompute. The table name stays UNQUALIFIED, because
2775
+ # this writer must honor account scoping: under
2776
+ # `scope_conversations_db_to_account` it sees the empty TEMP view on
2777
+ # purpose.
2715
2778
  thread = conn.execute(
2716
- "SELECT cwd, git_json, parent_thread_id, source_root_key "
2779
+ "SELECT CAST(cwd AS BLOB) AS cwd_blob, "
2780
+ "CAST(git_json AS BLOB) AS git_json_blob, parent_thread_id, "
2781
+ "source_root_key "
2717
2782
  "FROM codex_conversation_threads WHERE conversation_key = ?",
2718
2783
  (conversation_key,),
2719
2784
  ).fetchone()
2720
2785
  cwd = git_json = parent_thread_id = None
2786
+ thread_malformed = False
2721
2787
  if thread is not None:
2722
- cwd, git_json, parent_thread_id, thread_root = thread
2788
+ cwd_blob, git_json_blob, parent_thread_id, thread_root = thread
2789
+ decoded = _lib_codex_metadata.decode_codex_project_metadata(
2790
+ cwd_blob, git_json_blob)
2791
+ cwd, git_json = decoded.cwd, decoded.git_json
2792
+ # The DIRECT-ONLY predicate: this writer resolves the thread's own
2793
+ # fields through `_codex_conversation_project_attribution`, which
2794
+ # stops at a usable `cwd`, so a valid `cwd` beside an undecodable
2795
+ # `git_json` is still attributed.
2796
+ thread_malformed = _lib_codex_metadata.codex_metadata_is_malformed(
2797
+ decoded, None)
2723
2798
  if thread_root:
2724
2799
  source_root_key = thread_root
2725
- project_key, project_label = _codex_conversation_project_attribution(
2726
- source_root_key, cwd, git_json)
2800
+ if thread_malformed:
2801
+ # `(unassigned)` is a real answer for a thread with no metadata.
2802
+ # Persisting it for a thread whose metadata could not be READ would
2803
+ # stamp a wrong answer that every later read prefers, which is the
2804
+ # failure class `docs/codex-gotchas.md` records for the materialized
2805
+ # `"(unassigned)"`. NULL is the absence every reader already
2806
+ # handles.
2807
+ project_key = project_label = None
2808
+ else:
2809
+ project_key, project_label = _codex_conversation_project_attribution(
2810
+ source_root_key, cwd, git_json)
2727
2811
  conn.execute(
2728
2812
  "INSERT INTO codex_conversation_rollups "
2729
2813
  "(conversation_key, source_root_key, parent_thread_id, item_count, "
@@ -3181,6 +3265,8 @@ def _write_codex_file_batch(
3181
3265
  file_account_decision: "tuple[int, str | None] | None" = None,
3182
3266
  anchor_resolver: "CodexResetAnchorResolver | None" = None,
3183
3267
  ingest_complete: bool = True,
3268
+ device_id: "int | None" = None,
3269
+ inode: "int | None" = None,
3184
3270
  ) -> int:
3185
3271
  """Write one fully-buffered Codex file atomically and return entry changes.
3186
3272
 
@@ -3278,19 +3364,27 @@ def _write_codex_file_batch(
3278
3364
  last_seen_utc=excluded.last_seen_utc""",
3279
3365
  [(*row, now_iso, now_iso) for row in thread_rows],
3280
3366
  )
3367
+ # #769 S6: `device_id`/`inode` are the identity of the file the caller
3368
+ # STATTED before reading, and they land in the same statement as the scan
3369
+ # target, the final offset and the completion flag. A replacement can
3370
+ # therefore never leave a new identity paired with an old offset: either
3371
+ # the whole file batch commits or none of it does.
3281
3372
  conn.execute(
3282
3373
  """INSERT OR REPLACE INTO codex_session_files
3283
3374
  (path, size_bytes, mtime_ns, last_byte_offset, last_ingested_at,
3284
3375
  last_session_id, last_model, last_total_tokens, source_root_key,
3285
3376
  last_native_thread_id, last_root_thread_id, last_parent_thread_id,
3286
- last_conversation_key, last_turn_id, account_key, ingest_complete)
3287
- VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
3377
+ last_conversation_key, last_turn_id, account_key, ingest_complete,
3378
+ device_id, inode)
3379
+ VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
3288
3380
  (
3289
3381
  path_str, size, mtime_ns, final_offset, now_iso, last_session_id,
3290
3382
  last_model, last_total_tokens, discovered.source_root_key,
3291
3383
  last_native_thread_id, last_root_thread_id, last_parent_thread_id,
3292
3384
  last_conversation_key, last_turn_id, account_key,
3293
3385
  1 if ingest_complete else 0,
3386
+ None if device_id is None else int(device_id),
3387
+ None if inode is None else int(inode),
3294
3388
  ),
3295
3389
  )
3296
3390
  if prune_roots:
@@ -3420,12 +3514,32 @@ class PruneResult:
3420
3514
  """Outcome of _prune_orphaned_cache_entries: how much of the derived Claude
3421
3515
  surface was removed for safely-orphaned source paths, plus the orphan paths
3422
3516
  left in place (residual — a gate failed, so `--rebuild` is the escape hatch)
3423
- and whether the flock was contended (nothing mutated)."""
3517
+ and whether the flock was contended (nothing mutated).
3518
+
3519
+ ``prune_refused`` / ``prune_refused_files`` mirror ``CodexIngestStats``
3520
+ (#729). The deletions still committed; what was refused is the
3521
+ ``conversation_sessions`` re-derive that should have followed them, so the
3522
+ store is left partially derived. `_prune_orphaned_cache_entries` discarded
3523
+ `_recompute_conversation_sessions`' return, and
3524
+ `_lib_ingest_frontier.provider_sync_certifiable` already tested
3525
+ ``getattr(stats, "prune_refused", False)`` — with no such attribute the
3526
+ test always read False and a refused prune certified as clean. Adding the
3527
+ field does not add a check; it makes an existing blind one start firing.
3528
+
3529
+ ``prune_refused_state`` carries the observation state that refused (#769
3530
+ S3). Until #728 a refusal had exactly one cause — this process's pricing
3531
+ table being older than the store's — and the operator messages stated it as
3532
+ a fact. MALFORMED and DEGRADED now refuse too, and neither is about which
3533
+ table is older, so the cause travels with the result rather than being
3534
+ re-derived (or guessed) at each message site."""
3424
3535
  pruned_files: int = 0
3425
3536
  pruned_entries: int = 0
3426
3537
  pruned_messages: int = 0
3427
3538
  residual_paths: "list[str]" = field(default_factory=list)
3428
3539
  contended: bool = False
3540
+ prune_refused: bool = False
3541
+ prune_refused_files: int = 0
3542
+ prune_refused_state: "str | None" = None
3429
3543
 
3430
3544
 
3431
3545
  def _progress_stderr(stats: IngestStats, *, force: bool = False) -> None:
@@ -3740,7 +3854,20 @@ def _prune_orphaned_cache_entries(conn, *, lock_timeout=None):
3740
3854
  f"DELETE FROM conversation_messages WHERE source_path IN ({ph})", safe_paths)
3741
3855
  conv.execute(
3742
3856
  f"DELETE FROM conversation_ai_titles WHERE source_path IN ({ph})", safe_paths)
3743
- _recompute_conversation_sessions(conv, list(pruned_sids))
3857
+ # #729: this return was discarded. A refused re-derive still
3858
+ # commits — the three DELETEs above are done, and the refusal ARMS
3859
+ # `conversation_sessions_backfill_pending` and latches its record
3860
+ # inside this same BEGIN, so rolling back would discard the very
3861
+ # signal that tells the next authorized process to finish the job.
3862
+ # What must not happen is reporting the outcome as clean.
3863
+ refusal_state: "list[str]" = []
3864
+ if not _recompute_conversation_sessions(
3865
+ conv, list(pruned_sids), refusal_out=refusal_state
3866
+ ):
3867
+ result.prune_refused = True
3868
+ result.prune_refused_files = len(safe_paths)
3869
+ result.prune_refused_state = (
3870
+ refusal_state[0] if refusal_state else None)
3744
3871
  conv.commit()
3745
3872
  except BaseException:
3746
3873
  conv.rollback()
@@ -5197,6 +5324,55 @@ CONVERSATION_ROLLUP_PRICING_REFUSED_KEY = (
5197
5324
  )
5198
5325
 
5199
5326
 
5327
+ def _read_pricing_fingerprint_observation(
5328
+ conn, key: str = CONVERSATION_ROLLUP_PRICING_FP_KEY,
5329
+ ):
5330
+ """Read one store's recorded pricing fingerprint as a four-state
5331
+ observation (#728). The ONE database half of the split.
5332
+
5333
+ Every writer site used to perform this SELECT itself and degrade a failed
5334
+ read to the same falsy value an unrecorded fingerprint produces, so a
5335
+ locked or schema-less store was indistinguishable from a fresh one and
5336
+ every writer authorized itself over it. Here the failure is reported as
5337
+ ``DEGRADED`` and the pure classifier in `_lib_pricing` decides what that
5338
+ means; no caller may re-derive the state from a bare value again.
5339
+
5340
+ ``error_kind`` is the coarse ``"operational_error"`` rather than the
5341
+ driver's message, because the message text is not a contract and the only
5342
+ decision that rests on it is "the read did not happen".
5343
+
5344
+ A store with no ``cache_meta`` TABLE is ABSENT, not DEGRADED, and the
5345
+ difference is decided by a structural probe rather than by matching the
5346
+ driver's message. A store that has nowhere to record a fingerprint has
5347
+ determinately recorded none — that is a fact read off ``sqlite_master``,
5348
+ not an assumption — while a store whose ``sqlite_master`` probe fails too
5349
+ is one this process genuinely could not read. The probe fails closed: an
5350
+ unanswerable probe reports DEGRADED.
5351
+ """
5352
+ try:
5353
+ row = conn.execute(
5354
+ "SELECT value FROM cache_meta WHERE key=?", (key,),
5355
+ ).fetchone()
5356
+ except sqlite3.OperationalError:
5357
+ try:
5358
+ table_present = conn.execute(
5359
+ "SELECT 1 FROM sqlite_master "
5360
+ "WHERE type='table' AND name='cache_meta'"
5361
+ ).fetchone() is not None
5362
+ except sqlite3.OperationalError:
5363
+ table_present = True
5364
+ if table_present:
5365
+ return classify_pricing_fingerprint(
5366
+ found=False, raw=None, error_kind="operational_error")
5367
+ return classify_pricing_fingerprint(
5368
+ found=False, raw=None, error_kind=None)
5369
+ return classify_pricing_fingerprint(
5370
+ found=row is not None,
5371
+ raw=(row[0] if row is not None else None),
5372
+ error_kind=None,
5373
+ )
5374
+
5375
+
5200
5376
  def _pricing_write_authorized(stored, process=None) -> bool:
5201
5377
  """Whether a process holding `process` pricing may write materialized
5202
5378
  conversation cost over a store that recorded `stored` (#705).
@@ -5221,26 +5397,102 @@ def _pricing_write_authorized(stored, process=None) -> bool:
5221
5397
  The parse itself lives in `_lib_pricing.parse_pricing_fingerprint`, because
5222
5398
  `doctor pricing.conversation_rollup_writer` reports which refusal state a
5223
5399
  store is in and must classify a value exactly as this function acts on it.
5224
- Two parses of one contract had already drifted apart once."""
5400
+ Two parses of one contract had already drifted apart once.
5401
+
5402
+ NO PRODUCTION CALLER REMAINS (#769 S3 A9). It takes a bare stored value, so
5403
+ it cannot express a read that failed, and every writer site now reads
5404
+ through `_read_pricing_fingerprint_observation` instead. It is kept as the
5405
+ VALUE-shaped surface over the same contract (#728), for tests and for any
5406
+ caller that genuinely holds a value rather than an observation; expressing
5407
+ it on top of the shared predicate rather than beside it keeps the two from
5408
+ drifting the way the two date parsers once did. Read this docstring as a
5409
+ description of a retained helper, not of a live authorization site."""
5225
5410
  process = PRICING_SNAPSHOT_DATE if process is None else process
5226
- if not stored:
5227
- return True
5228
- stored_date = parse_pricing_fingerprint(stored)
5229
- process_date = parse_pricing_fingerprint(process)
5230
- if stored_date is None or process_date is None:
5231
- return False
5232
- return stored_date <= process_date
5411
+ return may_write_materialized_cost(
5412
+ classify_pricing_fingerprint(
5413
+ found=stored is not None, raw=stored, error_kind=None),
5414
+ process,
5415
+ )
5233
5416
 
5234
5417
 
5235
- def _record_pricing_write_refusal(conn, stored) -> None:
5236
- """Latch that a process holding older pricing was refused a write (#705).
5418
+ #: One operator-facing clause per observation state that can refuse a
5419
+ #: materialized-cost write (#769 S3). Each message that reports a refusal reads
5420
+ #: its cause from here instead of restating the version-skew case, which #728
5421
+ #: made one of three. The `None` entry covers a caller that recorded no state —
5422
+ #: an older result object, or a refusal from a path that does not carry one —
5423
+ #: and must stay distinct from every real state rather than defaulting to the
5424
+ #: version-skew wording, which is what this mapping exists to stop.
5425
+ _PRICING_REFUSAL_CAUSE_PHRASES = {
5426
+ "present": (
5427
+ "this process's pricing table is older than the one the store recorded"
5428
+ ),
5429
+ "malformed": (
5430
+ "the pricing fingerprint the store recorded is not a date cctally can "
5431
+ "order against its own"
5432
+ ),
5433
+ "degraded": (
5434
+ "cctally could not read the pricing fingerprint the store recorded, so "
5435
+ "it has no evidence about what that store's cost needed protecting from"
5436
+ ),
5437
+ None: "cctally could not authorize the write",
5438
+ }
5439
+
5440
+
5441
+ def pricing_refusal_cause_phrase(state: "str | None") -> str:
5442
+ """The operator-facing cause clause for one refusing observation state."""
5443
+ return _PRICING_REFUSAL_CAUSE_PHRASES.get(
5444
+ state, _PRICING_REFUSAL_CAUSE_PHRASES[None])
5237
5445
 
5238
- Written only when the (process, store) pair DIFFERS from what is already
5239
- recorded. A dashboard ticks continuously, so rewriting per tick would take
5240
- the conversations writer lock purely to restate an unchanged fact — and it
5241
- would make the timestamp the latest refusal rather than the first, which
5242
- is the less useful of the two, because the first says how long the store
5243
- has been diverging.
5446
+
5447
+ def _pricing_refusal_timestamp() -> str:
5448
+ """The instant an episode opened, to the second, in Zulu form.
5449
+
5450
+ A named seam rather than an inline expression so a test can pin it: the
5451
+ episode rules below are entirely about WHICH instant survives, and at
5452
+ second granularity two refusals in one test are otherwise indistinguishable.
5453
+ """
5454
+ return (
5455
+ dt.datetime.now(dt.timezone.utc)
5456
+ .replace(microsecond=0).isoformat().replace("+00:00", "Z")
5457
+ )
5458
+
5459
+
5460
+ def _pricing_refusal_record_active(record) -> bool:
5461
+ """Whether a refusal record describes an OPEN episode (#729).
5462
+
5463
+ A record written before the ``active`` field existed was latched only on
5464
+ refusal and DELETED on convergence, so its mere presence means an open
5465
+ episode. Reading a missing ``active`` as False would silently stop
5466
+ reporting every refusal latched by an earlier version.
5467
+
5468
+ The non-dict guard is DEFENSIVE, not reachable from production today: the
5469
+ one caller decodes the stored JSON and short-circuits on anything that is
5470
+ not a dict before asking. It is kept because the value comes from a
5471
+ `cache_meta` row any version may have written, and a `True` returned for a
5472
+ truthy non-dict is the same fail-toward-reporting posture as the missing
5473
+ ``active`` key above (#769 S3 A9).
5474
+ """
5475
+ if not isinstance(record, dict):
5476
+ return bool(record)
5477
+ return record.get("active", True) is not False
5478
+
5479
+
5480
+ def _record_pricing_write_refusal(conn, stored) -> None:
5481
+ """Latch that a process holding older pricing was refused a write (#705),
5482
+ as a durable tombstone with EPISODE semantics (#729).
5483
+
5484
+ Not rewritten when the (process, store) pair is unchanged AND the episode
5485
+ is still open. A dashboard ticks continuously, so rewriting per tick would
5486
+ take the conversations writer lock purely to restate an unchanged fact —
5487
+ and it would make the timestamp the latest refusal rather than the first,
5488
+ which is the less useful of the two, because the first says how long the
5489
+ store has been diverging.
5490
+
5491
+ An INACTIVE-to-active transition for the same pair is the opposite case and
5492
+ starts a NEW episode with a fresh timestamp. Carrying the old timestamp
5493
+ across a completed convergence would report a fresh divergence as days old
5494
+ — the mirror image of the bug the preserved timestamp exists to avoid. A
5495
+ different pair likewise replaces the record and restamps.
5244
5496
 
5245
5497
  NEVER commits — the caller owns the transaction. _prune_orphaned_cache_entries
5246
5498
  reaches this from inside its own explicit ``BEGIN``, whose
@@ -5262,15 +5514,15 @@ def _record_pricing_write_refusal(conn, stored) -> None:
5262
5514
  existing = json.loads(row[0])
5263
5515
  except (ValueError, TypeError):
5264
5516
  existing = None # unreadable -> replace it with a readable one
5265
- if isinstance(existing, dict) and all(
5266
- existing.get(field) == value for field, value in pair.items()
5517
+ if (
5518
+ isinstance(existing, dict)
5519
+ and all(existing.get(f) == v for f, v in pair.items())
5520
+ and _pricing_refusal_record_active(existing)
5267
5521
  ):
5268
5522
  return
5269
5523
  record = dict(pair)
5270
- record["first_refused_at_utc"] = (
5271
- dt.datetime.now(dt.timezone.utc)
5272
- .replace(microsecond=0).isoformat().replace("+00:00", "Z")
5273
- )
5524
+ record["first_refused_at_utc"] = _pricing_refusal_timestamp()
5525
+ record["active"] = True
5274
5526
  try:
5275
5527
  _set_cache_meta(
5276
5528
  conn, CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,
@@ -5283,8 +5535,19 @@ def _record_pricing_write_refusal(conn, stored) -> None:
5283
5535
 
5284
5536
 
5285
5537
  def _clear_pricing_write_refusal(conn) -> None:
5286
- """Drop the refusal latch, unconditionally. NEVER commits — the caller owns
5287
- the transaction, like the record and arm helpers beside it.
5538
+ """SETTLE the refusal latch: close the episode, keep the record (#729).
5539
+ NEVER commits — the caller owns the transaction, like the record and arm
5540
+ helpers beside it.
5541
+
5542
+ It used to DELETE the row, so a store that had converged carried no
5543
+ evidence it had ever diverged and doctor could not tell "never refused"
5544
+ from "refused and recovered". The record now survives with
5545
+ ``active: false``, preserving the pair and the timestamp that says when
5546
+ that episode opened.
5547
+
5548
+ A store that never refused gets NO record: the settle is an UPDATE over an
5549
+ existing row, never an insert, or every healthy install would grow a
5550
+ tombstone for an episode that never happened.
5288
5551
 
5289
5552
  TWO callers, and the condition each satisfies before calling is the whole
5290
5553
  contract. The two pending-flag branches call it directly, atomically with
@@ -5301,9 +5564,31 @@ def _clear_pricing_write_refusal(conn) -> None:
5301
5564
  latched, while `doctor pricing.conversation_rollup_writer` promised the
5302
5565
  operator that the next tick would clear it."""
5303
5566
  try:
5304
- conn.execute(
5305
- "DELETE FROM cache_meta WHERE key=?",
5567
+ row = conn.execute(
5568
+ "SELECT value FROM cache_meta WHERE key=?",
5306
5569
  (CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,),
5570
+ ).fetchone()
5571
+ except sqlite3.OperationalError:
5572
+ return
5573
+ if not row or not row[0]:
5574
+ return
5575
+ try:
5576
+ record = json.loads(row[0])
5577
+ except (ValueError, TypeError):
5578
+ record = None
5579
+ if not isinstance(record, dict):
5580
+ # An unreadable record is still evidence the guard fired, and doctor
5581
+ # reports it under its own wording. There is no episode to settle and
5582
+ # nothing this function could preserve, so leave it exactly as it is
5583
+ # rather than replacing it with a settled record it cannot vouch for.
5584
+ return
5585
+ if record.get("active") is False:
5586
+ return
5587
+ record["active"] = False
5588
+ try:
5589
+ _set_cache_meta(
5590
+ conn, CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,
5591
+ json.dumps(record, sort_keys=True),
5307
5592
  )
5308
5593
  except sqlite3.OperationalError:
5309
5594
  pass
@@ -5366,19 +5651,34 @@ def _arm_rollup_backfill_on_pricing_change(conn) -> bool:
5366
5651
  Crash-safety is unchanged: the DURABLE backfill flag remains the recompute
5367
5652
  signal, so advancing the fingerprint here cannot strand stale cost (a crash
5368
5653
  after arming leaves the flag set -> next sync recomputes regardless of the
5369
- fingerprint). No-op when cache_meta is unavailable (path-less / degraded
5370
- conn). Caller path holds the cache.db.lock flock. Unlike the recompute
5654
+ fingerprint). Caller path holds the cache.db.lock flock. Unlike the recompute
5371
5655
  chokepoint, this helper owns its OWN transaction and commits, so the flag
5372
- and the refusal record are durable for the next process."""
5373
- try:
5374
- row = conn.execute(
5375
- "SELECT value FROM cache_meta WHERE key=?",
5376
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5377
- ).fetchone()
5378
- except sqlite3.OperationalError:
5379
- return True
5380
- stored = row[0] if row is not None else None
5381
- if not _pricing_write_authorized(stored):
5656
+ and the refusal record are durable for the next process.
5657
+
5658
+ #728: this site is the one whose ``except sqlite3.OperationalError:``
5659
+ clause returned True — "authorized" — DIRECTLY, without ever reaching the
5660
+ predicate, so a store it had failed to read authorized every writer behind
5661
+ it. It reads a four-state observation now like its two siblings, and a
5662
+ DEGRADED read refuses. Both refusal legs (arm the flag, latch the record,
5663
+ commit) are shared by the DEGRADED, MALFORMED and store-is-newer states;
5664
+ on a store whose ``cache_meta`` genuinely cannot be written, each of those
5665
+ three is a caught no-op, so a degraded connection still returns False
5666
+ without raising.
5667
+
5668
+ The docstring used to say "No-op when cache_meta is unavailable"; that
5669
+ sentence described the early ``return True`` #728 removed, and stating the
5670
+ replacement is the point of this paragraph (#769 S3 A9). An unavailable
5671
+ ``cache_meta`` is now TWO outcomes decided structurally rather than one.
5672
+ A store whose ``sqlite_master`` probe positively shows no ``cache_meta``
5673
+ table is ABSENT — it determinately recorded nothing — so this helper is
5674
+ authorized and proceeds, and `_set_cache_meta` creates the table and stamps
5675
+ the fingerprint. A read that failed for any other reason, including a probe
5676
+ that could not answer either, is DEGRADED and refuses; its three refusal
5677
+ writes are each individually caught, so the return is False rather than an
5678
+ exception."""
5679
+ obs = _read_pricing_fingerprint_observation(conn)
5680
+ stored = obs.raw
5681
+ if not may_write_materialized_cost(obs, PRICING_SNAPSHOT_DATE):
5382
5682
  _record_pricing_write_refusal(conn, stored)
5383
5683
  _arm_rollup_backfill_pending(conn)
5384
5684
  conn.commit()
@@ -5394,7 +5694,8 @@ def _arm_rollup_backfill_on_pricing_change(conn) -> bool:
5394
5694
 
5395
5695
  def _recompute_conversation_sessions(
5396
5696
  conn, session_ids=None, *, advance_render_revision: bool = True,
5397
- authorize: bool = True,
5697
+ authorize: bool = True, refusal_out: "list[str] | None" = None,
5698
+ target: str = "conversation_sessions",
5398
5699
  ) -> bool:
5399
5700
  """Recompute the ``conversation_sessions`` browse-rail rollup from
5400
5701
  ``conversation_messages``. The caller holds the cache.db.lock flock and owns
@@ -5446,21 +5747,37 @@ def _recompute_conversation_sessions(
5446
5747
  persistent-store guard does not apply to it. The clear is gated for the
5447
5748
  OPPOSITE reason — its unqualified ``cache_meta`` is not shadowed and would
5448
5749
  reach ``main``, so on that connection it is the one statement here that
5449
- does touch a persistent row."""
5750
+ does touch a persistent row.
5751
+
5752
+ ``refusal_out`` is an optional sink for the observation state that refused
5753
+ (#769 S3). The bare False return says a write was declined but not why, and
5754
+ since #728 there are three reasons — only one of which is the pricing-skew
5755
+ case every operator message used to state as a fact. A caller that reports
5756
+ the refusal to a person passes a list and reads the appended state; the
5757
+ read happens exactly once here, so the reported cause is the one that
5758
+ actually decided, not a second read that may disagree with it.
5759
+
5760
+ ``target`` names the table this writes. A rebuild builds its replacement
5761
+ generation into ``conversation_sessions_staging`` and leaves the live table
5762
+ untouched until the publish (#752), so the rollup derivation has to be able
5763
+ to name either. It stays a parameter with the live default rather than two
5764
+ copies of the derivation, because two copies of a GROUP BY that must stay
5765
+ byte-identical to the rail's live aggregate is exactly the drift this
5766
+ function's own docstring warns about. The account scoper's connection-local
5767
+ TEMP table shadows the DEFAULT name, so passing nothing preserves it."""
5450
5768
  ids = None if session_ids is None else [s for s in session_ids if s is not None]
5451
5769
  if ids == []:
5452
5770
  return True
5453
5771
  if authorize:
5454
- try:
5455
- row = conn.execute(
5456
- "SELECT value FROM cache_meta WHERE key=?",
5457
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5458
- ).fetchone()
5459
- except sqlite3.OperationalError:
5460
- row = None
5461
- stored = row[0] if row else None
5462
- if not _pricing_write_authorized(stored):
5463
- _record_pricing_write_refusal(conn, stored)
5772
+ # #728: the degrade-to-None shape this replaces reached the predicate,
5773
+ # but with a value that said "this store recorded nothing" for a SELECT
5774
+ # that had raised. The observation reports the failed read as DEGRADED
5775
+ # instead, which this predicate refuses.
5776
+ obs = _read_pricing_fingerprint_observation(conn)
5777
+ if not may_write_materialized_cost(obs, PRICING_SNAPSHOT_DATE):
5778
+ if refusal_out is not None:
5779
+ refusal_out.append(obs.state)
5780
+ _record_pricing_write_refusal(conn, obs.raw)
5464
5781
  _arm_rollup_backfill_pending(conn)
5465
5782
  # No commit: this helper documents that the CALLER owns it, and
5466
5783
  # every caller that can reach a refusal commits — the pruner inside
@@ -5470,16 +5787,18 @@ def _recompute_conversation_sessions(
5470
5787
  # Do NOT weaken any of those to "the arm helper already committed
5471
5788
  # the same two rows on this tick". That was the argument for the
5472
5789
  # two flag branches, and it holds only while the arm helper's read
5473
- # of the fingerprint and the read below AGREE. They diverge two
5474
- # ways. _arm_rollup_backfill_on_pricing_change returns True and
5475
- # writes nothing when its own SELECT raises OperationalError, and
5476
- # `database is locked` is transient, so the read below can succeed
5477
- # against a newer stored value and refuse. And another process can
5478
- # advance the fingerprint between the two reads — which
5479
- # _import_legacy_conversation_rows made more reachable, because it
5480
- # stamps CONVERSATION_ROLLUP_PRICING_FP_KEY at DB open holding only
5481
- # the shared maintenance lock, never the conversations writer flock.
5482
- # In either case this writes a genuinely new record,
5790
+ # of the fingerprint and the read below AGREE. They still diverge,
5791
+ # though #769 S3 corrects WHY: the pre-#728 reason — the arm helper
5792
+ # returning True and writing nothing when its own SELECT raised —
5793
+ # no longer exists, because a DEGRADED read now refuses there too.
5794
+ # What remains is that another process can advance the fingerprint
5795
+ # between the two reads, which _import_legacy_conversation_rows
5796
+ # made more reachable, because it stamps
5797
+ # CONVERSATION_ROLLUP_PRICING_FP_KEY at DB open holding only the
5798
+ # shared maintenance lock, never the conversations writer flock.
5799
+ # A transient `database is locked` also still splits the two reads,
5800
+ # now in the other direction: the arm helper refuses and this read
5801
+ # can succeed. In every case this writes a genuinely new record,
5483
5802
  # _arm_rollup_backfill_pending short-circuits on the already-set
5484
5803
  # flag, and nothing else commits.
5485
5804
  return False
@@ -5488,15 +5807,15 @@ def _recompute_conversation_sessions(
5488
5807
  if advance_render_revision else 0
5489
5808
  )
5490
5809
  if ids is None:
5491
- conn.execute("DELETE FROM conversation_sessions")
5810
+ conn.execute(f"DELETE FROM {target}")
5492
5811
  conn.execute(
5493
- "INSERT INTO conversation_sessions "
5812
+ f"INSERT INTO {target} "
5494
5813
  "(session_id, msg_count, started_utc, last_activity_utc) "
5495
5814
  + _CONV_SESSIONS_SELECT + " GROUP BY session_id"
5496
5815
  )
5497
- _fill_conversation_sessions_filter_columns(conn, None)
5816
+ _fill_conversation_sessions_filter_columns(conn, None, target=target)
5498
5817
  conn.execute(
5499
- "UPDATE conversation_sessions SET render_revision=?",
5818
+ f"UPDATE {target} SET render_revision=?",
5500
5819
  (render_revision,),
5501
5820
  )
5502
5821
  _clear_converged_pricing_write_refusal(conn, authorize=authorize)
@@ -5505,27 +5824,28 @@ def _recompute_conversation_sessions(
5505
5824
  chunk = ids[i:i + 400]
5506
5825
  placeholders = ",".join("?" for _ in chunk)
5507
5826
  conn.execute(
5508
- f"DELETE FROM conversation_sessions WHERE session_id IN ({placeholders})",
5827
+ f"DELETE FROM {target} WHERE session_id IN ({placeholders})",
5509
5828
  chunk,
5510
5829
  )
5511
5830
  conn.execute(
5512
- "INSERT INTO conversation_sessions "
5831
+ f"INSERT INTO {target} "
5513
5832
  "(session_id, msg_count, started_utc, last_activity_utc) "
5514
5833
  + _CONV_SESSIONS_SELECT
5515
5834
  + f" AND session_id IN ({placeholders}) GROUP BY session_id",
5516
5835
  chunk,
5517
5836
  )
5518
5837
  conn.execute(
5519
- f"UPDATE conversation_sessions SET render_revision=? "
5838
+ f"UPDATE {target} SET render_revision=? "
5520
5839
  f"WHERE session_id IN ({placeholders})",
5521
5840
  (render_revision, *chunk),
5522
5841
  )
5523
- _fill_conversation_sessions_filter_columns(conn, ids)
5842
+ _fill_conversation_sessions_filter_columns(conn, ids, target=target)
5524
5843
  _clear_converged_pricing_write_refusal(conn, authorize=authorize)
5525
5844
  return True
5526
5845
 
5527
5846
 
5528
- def _fill_conversation_sessions_filter_columns(conn, session_ids):
5847
+ def _fill_conversation_sessions_filter_columns(conn, session_ids, *,
5848
+ target="conversation_sessions"):
5529
5849
  """Fill the rollup's browse-FILTER columns (project_label / cost_usd /
5530
5850
  cache_rebuild_count, migration 015) AND the #302 DISPLAYED-enrichment columns
5531
5851
  (git_branch / models_json / title) for the given sessions, or ALL when
@@ -5553,13 +5873,13 @@ def _fill_conversation_sessions_filter_columns(conn, session_ids):
5553
5873
  No-op when any of the columns is absent (a pre-015 / pre-023 cache.db being
5554
5874
  re-derived before _apply_cache_schema adds them), so an early/partial sync
5555
5875
  never raises ``no such column``. The CALLER owns the commit (never commits)."""
5556
- cols = {r[1] for r in conn.execute("PRAGMA table_info(conversation_sessions)")}
5876
+ cols = {r[1] for r in conn.execute(f"PRAGMA table_info({target})")}
5557
5877
  if not {"cache_rebuild_count", "git_branch", "models_json", "title"} <= cols:
5558
5878
  return
5559
5879
  lq = _load_lib("_lib_conversation_query")
5560
5880
  if session_ids is None:
5561
5881
  ids = [r[0] for r in conn.execute(
5562
- "SELECT session_id FROM conversation_sessions")]
5882
+ f"SELECT session_id FROM {target}")]
5563
5883
  else:
5564
5884
  ids = [s for s in session_ids if s is not None]
5565
5885
  if not ids:
@@ -5576,7 +5896,7 @@ def _fill_conversation_sessions_filter_columns(conn, session_ids):
5576
5896
  models_json = json.dumps(m) if m else None
5577
5897
  title = first_titles.get(sid)
5578
5898
  conn.execute(
5579
- "UPDATE conversation_sessions SET project_label=?, cost_usd=?, "
5899
+ f"UPDATE {target} SET project_label=?, cost_usd=?, "
5580
5900
  "cache_rebuild_count=?, git_branch=?, models_json=?, title=? "
5581
5901
  "WHERE session_id=?",
5582
5902
  (proj, round(cost.get(sid, 0.0), 6), rebuilds, branch, models_json,
@@ -7323,7 +7643,9 @@ def sync_codex_cache(
7323
7643
  # mtime_ns is selected into `existing` for diagnostics only —
7324
7644
  # delta detection consults size alone (Codex rollout JSONLs are
7325
7645
  # append-only, so a size change is a sufficient signal and mtime
7326
- # is prone to clock-skew false-positives).
7646
+ # is prone to clock-skew false-positives). `device_id`/`inode` are
7647
+ # NOT diagnostics: they are the only evidence that separates a file
7648
+ # nothing touched from one replaced at the same size (#769 S6).
7327
7649
  if targeted:
7328
7650
  # §5.1: the cursor preload queries codex_session_files for the
7329
7651
  # REQUESTED paths only (the full-sync path loads every row; targeted
@@ -7332,15 +7654,13 @@ def sync_codex_cache(
7332
7654
  conn, [str(item.source_path) for item in files])
7333
7655
  else:
7334
7656
  existing = {
7335
- row[0]: (
7336
- row[1], row[2], row[3], row[4], row[5], row[6], row[7],
7337
- row[8], row[9], row[10], row[11], row[12], row[13],
7338
- )
7657
+ row[0]: tuple(row[1:])
7339
7658
  for row in conn.execute(
7340
7659
  "SELECT path, size_bytes, mtime_ns, last_byte_offset, "
7341
7660
  "last_session_id, last_model, last_total_tokens, source_root_key, "
7342
7661
  "last_native_thread_id, last_root_thread_id, last_parent_thread_id, "
7343
- "last_conversation_key, last_turn_id, ingest_complete "
7662
+ "last_conversation_key, last_turn_id, ingest_complete, "
7663
+ "device_id, inode "
7344
7664
  "FROM codex_session_files"
7345
7665
  )
7346
7666
  }
@@ -7520,12 +7840,29 @@ def sync_codex_cache(
7520
7840
  prev_size, _, prev_offset, prev_sid, prev_model, prev_ttot,
7521
7841
  prev_root_key, prev_native_thread_id, prev_root_thread_id,
7522
7842
  prev_parent_thread_id, prev_conversation_key, prev_turn_id,
7523
- prev_complete,
7843
+ prev_complete, prev_device, prev_inode,
7524
7844
  ) = prev
7525
7845
  prev_total_tokens = (
7526
7846
  int(prev_ttot) if prev_ttot is not None else None
7527
7847
  )
7528
7848
  requalified = prev_root_key != discovered.source_root_key
7849
+ # #769 S6: IDENTITY OUTRANKS SIZE, here as well as in the
7850
+ # frontier's `classify_recent_active_path`. Detecting the
7851
+ # replacement in the planner and then letting the walk skip the
7852
+ # file on `size == prev_size` retains the old offset AND the old
7853
+ # identity, so the planner escalates again on the next tick and
7854
+ # the whole-estate walk the escalation authorises becomes a
7855
+ # per-tick walk that repairs nothing.
7856
+ #
7857
+ # THE INODE DECIDES and the device only corroborates, because
7858
+ # `st_dev` is assigned at mount time: see
7859
+ # `_lib_ingest_frontier.source_identity_replaced`, which owns
7860
+ # this verdict for the planner and for both walks so the three
7861
+ # cannot drift apart. It also owns the two no-evidence
7862
+ # degradations — a NULL column and an unreadable stored value
7863
+ # both fall through to the size comparison below.
7864
+ replaced = _ingest_frontier.source_identity_replaced(
7865
+ prev_device, prev_inode, st.st_dev, st.st_ino)
7529
7866
  # public #5 spec §4. `ingest_complete` is 1 for every row a
7530
7867
  # pre-budget binary wrote and for every file read to its stored
7531
7868
  # target, so this branch is unreachable until a budgeted stop
@@ -7535,9 +7872,13 @@ def sync_codex_cache(
7535
7872
  # skipped on equality, which made the unread suffix permanently
7536
7873
  # invisible on any rollout that never grows again.
7537
7874
  incomplete = prev_complete is not None and not int(prev_complete)
7538
- if targeted and (requalified or size < prev_size):
7539
- # §5.1 preflight-snapshot scoped: a shrink or requalification
7540
- # landing AFTER the preflight is declined HERE, per file —
7875
+ if targeted and (requalified or replaced or size < prev_size):
7876
+ # §5.1 preflight-snapshot scoped: a shrink, a requalification
7877
+ # or a replacement landing AFTER the preflight is declined
7878
+ # HERE, per file — the whole-call preflight above compares
7879
+ # sizes only, and a replacement is what the frontier already
7880
+ # answers with a full walk, so targeted mode reaches this
7881
+ # state only when the escalation lost a race —
7541
7882
  # earlier per-file commits in this call stand, the call still
7542
7883
  # reports dirty (files_failed → not targeted_clean), so the
7543
7884
  # watch advances no cursor and emits nothing, and recovery
@@ -7546,7 +7887,10 @@ def sync_codex_cache(
7546
7887
  # — that whole-cache-affecting escalation is the full sync's.
7547
7888
  stats.files_failed += 1
7548
7889
  continue
7549
- if not requalified and incomplete and size >= prev_offset:
7890
+ if (
7891
+ not requalified and not replaced and incomplete
7892
+ and size >= prev_offset
7893
+ ):
7550
7894
  # Resume the stored scan target. Deliberately NOT a
7551
7895
  # `delta_append`: that flag is what authorizes consulting
7552
7896
  # the live `auth.json` and minting a new account range at
@@ -7568,10 +7912,16 @@ def sync_codex_cache(
7568
7912
  initial_session_id = prev_sid
7569
7913
  initial_model = prev_model
7570
7914
  initial_total_tokens = prev_total_tokens or 0
7571
- elif not requalified and not incomplete and size == prev_size:
7915
+ elif (
7916
+ not requalified and not replaced and not incomplete
7917
+ and size == prev_size
7918
+ ):
7572
7919
  stats.files_skipped_unchanged += 1
7573
7920
  continue
7574
- elif not requalified and not incomplete and size > prev_size:
7921
+ elif (
7922
+ not requalified and not replaced and not incomplete
7923
+ and size > prev_size
7924
+ ):
7575
7925
  start_offset = prev_offset
7576
7926
  delta_append = True
7577
7927
  initial_session_id = prev_sid
@@ -8026,6 +8376,11 @@ def sync_codex_cache(
8026
8376
  file_account_decision=pending_decision,
8027
8377
  anchor_resolver=anchor_resolver,
8028
8378
  ingest_complete=not stopped_short["value"],
8379
+ # The SAME pre-read stat that defined `scan_target`.
8380
+ # Re-statting here would describe a file this pass may
8381
+ # never have read (#769 S6).
8382
+ device_id=st.st_dev,
8383
+ inode=st.st_ino,
8029
8384
  )
8030
8385
  except sqlite3.DatabaseError as exc:
8031
8386
  conn.rollback()
@@ -10006,8 +10361,57 @@ def _acquire_conversation_provider_locks(
10006
10361
  raise
10007
10362
 
10008
10363
 
10364
+ #: #769 S6 / #802 — the process-level no-sync derivation policy.
10365
+ #:
10366
+ #: `--no-sync` freezes ingestion and the snapshot. It must also freeze the two
10367
+ #: POST-DISPATCH conversation derivations `_open_conversations_db_unlocked`
10368
+ #: runs after `_run_pending_migrations` — `_import_legacy_conversation_rows`
10369
+ #: and `_ensure_codex_conversation_contract` — because the second consumes
10370
+ #: `conversation_rebuild_codex_pending` and performs a full retained-event
10371
+ #: rebuild, measured at 135 seconds of startup against comparable launches of
10372
+ #: 15 and 30 seconds.
10373
+ #:
10374
+ #: The policy is PROCESS-level rather than call-site-level, and that is the
10375
+ #: whole point: startup is not the only opener. A read route falls back from
10376
+ #: the read-only opener to the full `open_conversations_db()` on a missing
10377
+ #: store, a lock, or a pending legacy bridge, and live-tail always uses the
10378
+ #: full opener. Either would run both derivations and consume the marker from
10379
+ #: an ordinary browse, so a startup-only flag would freeze nothing.
10380
+ #:
10381
+ #: It never affects the migration dispatcher. Under `--no-sync` no other
10382
+ #: process opens the store write-capable and the read-only reader refuses a
10383
+ #: store behind head rather than migrating from a request thread, so
10384
+ #: suppressing the dispatcher would leave that dashboard permanently degraded —
10385
+ #: the failure `_dashboard_startup_schema_migration` exists to prevent.
10386
+ _CONVERSATION_DERIVATIONS_SUPPRESSED = False
10387
+
10388
+
10389
+ def set_conversation_derivations_suppressed(value: bool) -> None:
10390
+ """Set the process-level policy. Called once, by `--no-sync` startup."""
10391
+ global _CONVERSATION_DERIVATIONS_SUPPRESSED
10392
+ _CONVERSATION_DERIVATIONS_SUPPRESSED = bool(value)
10393
+
10394
+
10395
+ def conversation_derivations_suppressed() -> bool:
10396
+ """Whether this process suppresses the two post-dispatch derivations."""
10397
+ return _CONVERSATION_DERIVATIONS_SUPPRESSED
10398
+
10399
+
10400
+ def _conversation_derivations_enabled(run_derivations: "bool | None") -> bool:
10401
+ """Resolve an explicit request against the process policy.
10402
+
10403
+ ``None`` means "follow the process policy", which is what every existing
10404
+ caller passes by omission, and the policy defaults to running them — so
10405
+ every existing caller is byte-unchanged.
10406
+ """
10407
+ if run_derivations is not None:
10408
+ return bool(run_derivations)
10409
+ return not conversation_derivations_suppressed()
10410
+
10411
+
10009
10412
  def _conversations_open_guarded(
10010
10413
  *, attach_cache: bool, allow_recovery_state: bool = False,
10414
+ run_derivations: "bool | None" = None,
10011
10415
  ) -> sqlite3.Connection:
10012
10416
  """Open conversations.db while excluding confirmed family replacement."""
10013
10417
  path = pathlib.Path(_cctally_core.CONVERSATIONS_DB_PATH)
@@ -10101,6 +10505,7 @@ def _conversations_open_guarded(
10101
10505
  try:
10102
10506
  conn = _open_conversations_db_unlocked(
10103
10507
  attach_cache=attach_cache,
10508
+ run_derivations=run_derivations,
10104
10509
  )
10105
10510
  if marker.exists() or pending.exists():
10106
10511
  conn.close()
@@ -10138,7 +10543,7 @@ def _harden_conversation_sidecars() -> None:
10138
10543
 
10139
10544
 
10140
10545
  def _open_conversations_db_unlocked(
10141
- *, attach_cache: bool = True,
10546
+ *, attach_cache: bool = True, run_derivations: "bool | None" = None,
10142
10547
  ) -> sqlite3.Connection:
10143
10548
  """Open the independent transcript/search store (#320).
10144
10549
 
@@ -10147,6 +10552,20 @@ def _open_conversations_db_unlocked(
10147
10552
  Codex-thread metadata. Core cache callers never take the inverse
10148
10553
  dependency, so a missing or locked transcript store cannot block quota or
10149
10554
  accounting refreshes.
10555
+
10556
+ ``run_derivations`` (#769 S6 / #802) selects whether the two POST-DISPATCH
10557
+ derivations below run: ``_import_legacy_conversation_rows`` and
10558
+ ``_ensure_codex_conversation_contract``. ``None`` follows the process-level
10559
+ policy, which defaults to running them, so every existing caller is
10560
+ byte-unchanged. ``--no-sync`` sets that policy to suppress them.
10561
+
10562
+ What ``--no-sync`` freezes and does not freeze is therefore exact: it
10563
+ freezes ingestion, the snapshot and these two derivations; it does NOT
10564
+ freeze the schema apply or the migration dispatcher, which run
10565
+ unconditionally so the store still reaches head. The accepted consequence
10566
+ is that a ``--no-sync`` dashboard over a store owing the Codex contract
10567
+ rebuild serves ``normalization_pending`` for Codex conversation reads until
10568
+ a mode allowed to do work runs.
10150
10569
  """
10151
10570
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
10152
10571
  try:
@@ -10220,14 +10639,283 @@ def _open_conversations_db_unlocked(
10220
10639
  cache.close()
10221
10640
  cache_uri = _cctally_core.CACHE_DB_PATH.resolve().as_uri() + "?mode=ro"
10222
10641
  conn.execute("ATTACH DATABASE ? AS cache_db", (cache_uri,))
10223
- _import_legacy_conversation_rows(conn)
10224
- _ensure_codex_conversation_contract(conn)
10642
+ if _conversation_derivations_enabled(run_derivations):
10643
+ _import_legacy_conversation_rows(conn)
10644
+ _ensure_codex_conversation_contract(conn)
10225
10645
  _harden_conversation_sidecars()
10226
10646
  return conn
10227
10647
 
10228
10648
 
10229
- def open_conversations_db(*, attach_cache: bool = True) -> sqlite3.Connection:
10230
- return _conversations_open_guarded(attach_cache=attach_cache)
10649
+ def open_conversations_db(
10650
+ *, attach_cache: bool = True, run_derivations: "bool | None" = None,
10651
+ ) -> sqlite3.Connection:
10652
+ return _conversations_open_guarded(
10653
+ attach_cache=attach_cache, run_derivations=run_derivations,
10654
+ )
10655
+
10656
+
10657
+ # --------------------------------------------------------------------------
10658
+ # #780 — read-only reader admission
10659
+ # --------------------------------------------------------------------------
10660
+
10661
+ #: Default busy timeout for a read route, in seconds. Route-bounded rather
10662
+ #: than the store policy's 15s: a reader that cannot be admitted quickly must
10663
+ #: fail soft and let the route render a degraded envelope, never hold a request
10664
+ #: thread for fifteen seconds behind a rebuild.
10665
+ CONVERSATION_READONLY_TIMEOUT_S = 0.25
10666
+
10667
+ #: Called with no arguments when a reader finds the store BEHIND head. The
10668
+ #: opener never migrates from a request thread, so this is how it asks the
10669
+ #: process's writer to. Left None outside the dashboard.
10670
+ SCHEMA_WAKE_HOOK = None
10671
+
10672
+
10673
+ class ConversationReaderUnavailable(sqlite3.DatabaseError):
10674
+ """A read route could not be admitted. Carries a typed ``reason``.
10675
+
10676
+ A subclass of ``sqlite3.DatabaseError`` so the existing route boundary,
10677
+ which already catches ``DatabaseError`` and ``OSError``, keeps catching it
10678
+ rather than letting a new exception class escape as a 500.
10679
+ """
10680
+
10681
+ reason = "unavailable"
10682
+
10683
+
10684
+ class MaintenanceInProgress(ConversationReaderUnavailable):
10685
+ """Maintenance holds the store; the reader declined to queue behind it."""
10686
+
10687
+ reason = "maintenance"
10688
+
10689
+
10690
+ class SchemaBehind(ConversationReaderUnavailable):
10691
+ """A store is behind head. Soft: a writer can still advance it."""
10692
+
10693
+ reason = "schema_behind"
10694
+
10695
+
10696
+ class SchemaAhead(ConversationReaderUnavailable):
10697
+ """A store is ahead of head. Closed: no recovery is attempted here."""
10698
+
10699
+ reason = "schema_ahead"
10700
+
10701
+
10702
+ class LegacyBridgePending(ConversationReaderUnavailable):
10703
+ """The legacy transcript bridge is owed and this process may not run it.
10704
+
10705
+ `_import_legacy_conversation_rows` is a WRITER, and under `dashboard
10706
+ --no-sync` the process policy suppresses it (#802), so a store whose rows
10707
+ are still in `cache.db` after an interrupted migration 028 stays that way
10708
+ for the life of the process. Serving that connection would render an empty
10709
+ conversation list and an empty browse project filter with nothing said,
10710
+ which is the silent failure this class exists to replace. Spec §9 names the
10711
+ alternative: a route that cannot inherit the derivation policy returns a
10712
+ typed degraded response instead.
10713
+
10714
+ Named for the state, like `normalization_pending` — the sibling answer the
10715
+ Codex contract rebuild already produces under the same suppression.
10716
+ """
10717
+
10718
+ reason = "legacy_bridge_pending"
10719
+
10720
+
10721
+ def probe_conversations_maintenance_free() -> None:
10722
+ """Raise `MaintenanceInProgress` if maintenance holds the flock right now.
10723
+
10724
+ The non-blocking admission `open_conversations_db_readonly` performs,
10725
+ factored out for the three request-path sites that must reach the BLOCKING
10726
+ full opener but must not queue behind a rebuild on the way (#780).
10727
+
10728
+ Those sites are the live-tail preflight and `_open_conversation_reader`'s
10729
+ two fallbacks. Each of them hands the open to `open_conversations_db`,
10730
+ whose maintenance admission is a plain `LOCK_SH` with no timeout, so a
10731
+ request arriving during a rebuild held a server thread for the rebuild's
10732
+ whole duration — 716.9 s in the measured run. §4a names the live-tail
10733
+ preflight among the routes that must fail soft, and the other two reach the
10734
+ same opener by the same route.
10735
+
10736
+ An absent lock file is NOT a refusal, for the same reason the read-only
10737
+ opener gives: it means no maintenance has ever claimed it, which is a
10738
+ first-run state rather than a busy one, and the full opener is what creates
10739
+ it. An unopenable lock file is not a refusal either — an undecidable probe
10740
+ must not be what takes a read route down; the full opener behind it will
10741
+ fail loudly on its own if the state is genuinely bad.
10742
+ """
10743
+ maintenance_path = pathlib.Path(
10744
+ _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH
10745
+ )
10746
+ if not maintenance_path.is_file():
10747
+ return
10748
+ try:
10749
+ probe_fh = open(maintenance_path, "r")
10750
+ except OSError:
10751
+ return
10752
+ try:
10753
+ try:
10754
+ fcntl.flock(probe_fh, fcntl.LOCK_SH | fcntl.LOCK_NB)
10755
+ except BlockingIOError as exc:
10756
+ raise MaintenanceInProgress(
10757
+ "conversations.db maintenance is in progress") from exc
10758
+ try:
10759
+ fcntl.flock(probe_fh, fcntl.LOCK_UN)
10760
+ except OSError:
10761
+ pass
10762
+ finally:
10763
+ probe_fh.close()
10764
+
10765
+
10766
+ def open_conversations_db_readonly(
10767
+ *, attach_cache: bool = True, timeout: "float | None" = None,
10768
+ ) -> sqlite3.Connection:
10769
+ """A conversations.db connection for a READ route (#780).
10770
+
10771
+ The route this replaces opened the full mutation-capable
10772
+ ``_conversations_open_guarded``, which executes, in order: a chmod on the
10773
+ data directory, ``PRAGMA auto_vacuum=INCREMENTAL``, ``PRAGMA
10774
+ journal_mode=WAL``, a schema-currency check that can apply the whole
10775
+ schema, a commit, the migration dispatcher, a write-capable
10776
+ ``open_cache_db()``, ``_import_legacy_conversation_rows``,
10777
+ ``_ensure_codex_conversation_contract``, a chmod on the database, and
10778
+ sidecar hardening. Every one of those is a write, so every browse became a
10779
+ writer that could lose the SQLite write lock to maintenance.
10780
+
10781
+ This opener performs NONE of them. It does not call ``apply_policy``: that
10782
+ helper emits ``PRAGMA auto_vacuum`` and ``PRAGMA journal_mode``, both
10783
+ write-capable, so routing through it would defeat the whole point. The
10784
+ busy timeout is supplied to ``sqlite3.connect`` instead of through a
10785
+ PRAGMA, and the conversations policy's row factory is the driver default.
10786
+
10787
+ ``PRAGMA query_only`` is deliberately NOT set. ``mode=ro`` already refuses
10788
+ every write to ``main``, and it still permits TEMP tables and TEMP views,
10789
+ which is what ``scope_conversations_db_to_account`` needs; ``query_only``
10790
+ breaks those TEMP writes and would take account scoping down with it
10791
+ (verified empirically on SQLite 3.53.4).
10792
+
10793
+ Admission is non-blocking against the maintenance flock, and the marker
10794
+ files are re-checked after the connection opens, so a maintenance pass that
10795
+ starts during the open is not served from a store it is about to replace.
10796
+
10797
+ Both stores are then gated by the schema-qualified tri-state probe.
10798
+ ``behind`` raises ``SchemaBehind`` after asking the process's writer to
10799
+ advance the store; ``ahead`` raises ``SchemaAhead`` and attempts no
10800
+ recovery, matching the existing version-ahead posture. The opener never
10801
+ migrates from a request thread.
10802
+ """
10803
+ path = pathlib.Path(_cctally_core.CONVERSATIONS_DB_PATH)
10804
+ marker = _cctally_db_sib._repair_marker_path(path)
10805
+ pending = _cctally_db_sib._quarantine_pending_path(path)
10806
+ recovery = _conversation_recovery_state_path()
10807
+ maintenance_path = pathlib.Path(
10808
+ _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH
10809
+ )
10810
+ busy = CONVERSATION_READONLY_TIMEOUT_S if timeout is None else timeout
10811
+ if not path.is_file():
10812
+ raise ConversationReaderUnavailable("transcript store is not present")
10813
+ if not maintenance_path.is_file():
10814
+ # NOT a maintenance refusal: an absent lock file means no maintenance
10815
+ # has ever claimed it, which is a first-run state, not a busy one.
10816
+ # `_conversations_open_guarded` creates it, so this hands the open back
10817
+ # to the full opener the same way an absent store does. Refusing here
10818
+ # instead left a fresh install permanently degraded, which is the
10819
+ # opposite of the condition the file's absence describes.
10820
+ raise ConversationReaderUnavailable(
10821
+ "the transcript maintenance lock has not been established yet")
10822
+ cache_path = pathlib.Path(_cctally_core.CACHE_DB_PATH)
10823
+ if attach_cache and not cache_path.is_file():
10824
+ raise ConversationReaderUnavailable("accounting store is not present")
10825
+
10826
+ conn: sqlite3.Connection | None = None
10827
+ maintenance_fh = open(maintenance_path, "r")
10828
+ try:
10829
+ try:
10830
+ fcntl.flock(maintenance_fh, fcntl.LOCK_SH | fcntl.LOCK_NB)
10831
+ except BlockingIOError as exc:
10832
+ raise MaintenanceInProgress(
10833
+ "conversations.db maintenance is in progress") from exc
10834
+ try:
10835
+ if marker.exists() or pending.exists() or recovery.exists():
10836
+ raise MaintenanceInProgress(
10837
+ "conversations.db maintenance is in progress")
10838
+ conn = sqlite3.connect(
10839
+ f"{path.resolve().as_uri()}?mode=ro",
10840
+ uri=True,
10841
+ timeout=max(busy, 0.0),
10842
+ )
10843
+ if _cctally_store._TRACE_HOOK is not None:
10844
+ conn.set_trace_callback(_cctally_store._TRACE_HOOK)
10845
+ conn.row_factory = None
10846
+ if marker.exists() or pending.exists() or recovery.exists():
10847
+ raise MaintenanceInProgress(
10848
+ "conversations.db maintenance started during open")
10849
+ if attach_cache:
10850
+ conn.execute(
10851
+ "ATTACH DATABASE ? AS cache_db",
10852
+ (f"{cache_path.resolve().as_uri()}?mode=ro",),
10853
+ )
10854
+ finally:
10855
+ try:
10856
+ fcntl.flock(maintenance_fh, fcntl.LOCK_UN)
10857
+ except OSError:
10858
+ pass
10859
+ _gate_reader_schema(conn, attach_cache=attach_cache)
10860
+ return conn
10861
+ except BaseException:
10862
+ if conn is not None:
10863
+ try:
10864
+ conn.close()
10865
+ except sqlite3.Error:
10866
+ pass
10867
+ raise
10868
+ finally:
10869
+ maintenance_fh.close()
10870
+
10871
+
10872
+ def conversation_legacy_bridge_pending(conn: sqlite3.Connection) -> bool:
10873
+ """Whether `_import_legacy_conversation_rows` still has work to do (#780).
10874
+
10875
+ The bridge is a WRITER, so the read-only opener cannot run it, and a route
10876
+ that quietly skipped it would serve an empty transcript surface on a store
10877
+ whose rows are still sitting in `cache.db` after an interrupted migration
10878
+ 028. Its own condition — the main table empty while the attached cache
10879
+ table is not — is a pure read, so a reader can detect the state even though
10880
+ it must not fix it, and hand the open back to the full opener. The route
10881
+ checks this again AFTER that hand-back: under `dashboard --no-sync` the
10882
+ bridge is suppressed, so the full opener returns without clearing it and
10883
+ the route raises `LegacyBridgePending` instead of serving (#802).
10884
+
10885
+ Reports False rather than raising on any error: an undecidable answer must
10886
+ not take a read route down.
10887
+ """
10888
+ for table in _LEGACY_BRIDGE_TABLES:
10889
+ try:
10890
+ if conn.execute(f"SELECT 1 FROM main.{table} LIMIT 1").fetchone():
10891
+ continue
10892
+ if conn.execute(
10893
+ f"SELECT 1 FROM cache_db.{table} LIMIT 1"
10894
+ ).fetchone():
10895
+ return True
10896
+ except sqlite3.Error:
10897
+ continue
10898
+ return False
10899
+
10900
+
10901
+ def _gate_reader_schema(conn: sqlite3.Connection, *, attach_cache: bool) -> None:
10902
+ """Refuse a reader whose stores are not at head, by direction (#780)."""
10903
+ probes = [("conversations", "main")]
10904
+ if attach_cache:
10905
+ probes.append(("cache", "cache_db"))
10906
+ for store_name, schema in probes:
10907
+ state = _cctally_store.schema_state(conn, store_name, schema=schema)
10908
+ if state == "current":
10909
+ continue
10910
+ if state == "behind":
10911
+ hook = SCHEMA_WAKE_HOOK
10912
+ if hook is not None:
10913
+ try:
10914
+ hook()
10915
+ except Exception as exc: # noqa: BLE001
10916
+ eprint(f"[conversations] schema wake-up failed ({exc})")
10917
+ raise SchemaBehind(f"{store_name} store is behind head")
10918
+ raise SchemaAhead(f"{store_name} store is ahead of head")
10231
10919
 
10232
10920
 
10233
10921
  def scope_conversations_db_to_account(
@@ -10354,23 +11042,42 @@ def scope_conversations_db_to_account(
10354
11042
  }
10355
11043
  safe_project_attribution: dict[str, tuple[str | None, str | None]] = {}
10356
11044
  for conversation_key in codex_keys:
11045
+ # #845 §4.7. The thread is re-tested EVEN WHEN a rollup is persisted:
11046
+ # a rollup materialized before the thread was corrupted carries a
11047
+ # pre-corruption attribution, and preferring it would keep publishing
11048
+ # an answer the store can no longer support. The raw table is named
11049
+ # explicitly because the unqualified name is an empty TEMP view here.
11050
+ thread = conn.execute(
11051
+ "SELECT source_root_key, CAST(cwd AS BLOB) AS cwd_blob, "
11052
+ "CAST(git_json AS BLOB) AS git_json_blob "
11053
+ "FROM cache_db.codex_conversation_threads WHERE conversation_key=?",
11054
+ (conversation_key,),
11055
+ ).fetchone()
11056
+ decoded = None
11057
+ thread_malformed = False
11058
+ if thread is not None:
11059
+ decoded = _lib_codex_metadata.decode_codex_project_metadata(
11060
+ thread[1], thread[2])
11061
+ thread_malformed = _lib_codex_metadata.codex_metadata_is_malformed(
11062
+ decoded, None)
10357
11063
  persisted = conn.execute(
10358
11064
  "SELECT project_key,project_label "
10359
11065
  "FROM main.codex_conversation_rollups WHERE conversation_key=?",
10360
11066
  (conversation_key,),
10361
11067
  ).fetchone()
10362
11068
  if persisted is not None:
10363
- safe_project_attribution[conversation_key] = persisted
10364
- continue
10365
- thread = conn.execute(
10366
- "SELECT source_root_key,cwd,git_json "
10367
- "FROM cache_db.codex_conversation_threads WHERE conversation_key=?",
10368
- (conversation_key,),
10369
- ).fetchone()
10370
- if thread is not None:
10371
11069
  safe_project_attribution[conversation_key] = (
10372
- _codex_conversation_project_attribution(*thread)
11070
+ (None, None) if thread_malformed else persisted
10373
11071
  )
11072
+ continue
11073
+ if thread is not None:
11074
+ if thread_malformed:
11075
+ safe_project_attribution[conversation_key] = (None, None)
11076
+ else:
11077
+ safe_project_attribution[conversation_key] = (
11078
+ _codex_conversation_project_attribution(
11079
+ thread[0], decoded.cwd, decoded.git_json)
11080
+ )
10374
11081
  _recompute_codex_rollups(
10375
11082
  conn, codex_keys, advance_render_revision=False,
10376
11083
  )
@@ -10901,8 +11608,8 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10901
11608
  which table that was in ``CONVERSATION_ROLLUP_PRICING_FP_KEY``, so the
10902
11609
  provenance is written down rather than merely implied.
10903
11610
 
10904
- Deriving is required, not merely tidier: this runs at DB OPEN, and
10905
- ``dashboard --no-sync`` never runs a sync, so merely arming
11611
+ Deriving is required, not merely tidier, in every mode allowed to do work:
11612
+ this runs at DB OPEN, and merely arming
10906
11613
  ``conversation_sessions_backfill_pending`` left the rollup EMPTY and
10907
11614
  non-authoritative for the life of that process. The rail itself survives
10908
11615
  that (the flag routes it to live aggregation) but
@@ -10910,6 +11617,16 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10910
11617
  with no authoritative gate, so the browse filter's project list went empty;
10911
11618
  and every rail read fell to the live branch, which is not the branch the
10912
11619
  materialized-cost contract is about.
11620
+
11621
+ ``dashboard --no-sync`` is the one mode NOT allowed to do work, and since
11622
+ #802 the process policy suppresses this bridge there along with the Codex
11623
+ contract rebuild. That does not reinstate the empty-rollup failure above,
11624
+ because the read routes stop serving over an owed bridge: the reader raises
11625
+ ``LegacyBridgePending`` and the route answers the typed degraded envelope
11626
+ naming ``legacy_bridge_pending``, the sibling of the
11627
+ ``normalization_pending`` the Codex case already returns under the same
11628
+ suppression. The state is disclosed rather than rendered as an empty
11629
+ surface, and it clears the moment a mode allowed to do work opens the store.
10913
11630
  """
10914
11631
  changed = False
10915
11632
  for table in _LEGACY_BRIDGE_TABLES:
@@ -11163,6 +11880,393 @@ def _prepare_claude_conversation_maintenance(
11163
11880
  return reingest
11164
11881
 
11165
11882
 
11883
+ # === #752 / #777: title generations, source incarnations, account stamps ====
11884
+
11885
+
11886
+ #: The digest of zero bytes. A rebuild resets each source file's committed
11887
+ #: prefix to this rather than dropping the row, so the replay starts at byte
11888
+ #: zero WITHOUT minting a new incarnation — which is what lets the durable
11889
+ #: stamps written before the rebuild still be found after it.
11890
+ _EMPTY_PREFIX_SHA256 = (
11891
+ "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
11892
+ )
11893
+
11894
+ #: The FLOOR of the rebuild free-space requirement, not the requirement.
11895
+ #:
11896
+ #: Tranche 2 sized a fixed 64 MiB against the staging tables alone, which hold
11897
+ #: only titles and one row per session — 65,536 B on the store observed during
11898
+ #: the #752 incident. Tranche 3 then measured what a whole rebuild costs the
11899
+ #: file: 9,310,744,576 B to 15,459,213,312 B, a growth of 6,148,468,736 B, of
11900
+ #: which 93% is freelist that `PRAGMA incremental_vacuum` returns afterwards.
11901
+ #: The clear does not let the replay reuse the pages it freed within the same
11902
+ #: rebuild, so the peak file size is roughly the live corpus written twice.
11903
+ #:
11904
+ #: A fixed constant therefore admits a rebuild that then runs out of space
11905
+ #: partway — precisely the state the preflight exists to prevent. The
11906
+ #: requirement is read off the store instead, by
11907
+ #: `_conversation_rebuild_free_bytes_required`, and this constant only stops a
11908
+ #: tiny or empty store from demanding nothing at all.
11909
+ _CONVERSATION_STAGING_FREE_BYTES_FLOOR = 64 * 1024 * 1024
11910
+
11911
+ #: Retained under its Tranche 2 name for the module's re-export surface. New
11912
+ #: code reads the measured requirement, never this.
11913
+ _CONVERSATION_STAGING_FREE_BYTES_REQUIRED = _CONVERSATION_STAGING_FREE_BYTES_FLOOR
11914
+
11915
+
11916
+ class _SourceIncarnation(NamedTuple):
11917
+ """One append-continuous life of a source file (#777).
11918
+
11919
+ ``mutable_digest`` is a LIVE, MUTABLE ``hashlib`` object, not a value, and
11920
+ it is named that way because the distinction is load-bearing. It arrives
11921
+ already positioned at ``start_offset``, and the ingester CONSUMES it by
11922
+ updating it in place with the bytes that pass reads, so it may be updated
11923
+ exactly once and its ``hexdigest()`` before that update is not the same
11924
+ number as after. Returning a hex string instead would force a full rehash
11925
+ of the whole file on every sync rather than only the committed prefix,
11926
+ which is why it is an object at all.
11927
+ """
11928
+ incarnation_id: str
11929
+ is_new: bool
11930
+ start_offset: int
11931
+ mutable_digest: Any
11932
+
11933
+
11934
+ def _path_is_genuinely_absent(path_str: str) -> bool:
11935
+ """Whether ``path_str`` is really gone, as opposed to merely unwalked.
11936
+
11937
+ Only ``ENOENT`` and ``ENOTDIR`` answer yes. Every other ``OSError`` —
11938
+ ``EPERM`` on a directory the walk could not enter, ``EIO`` on failing
11939
+ media, ``ENOTCONN`` on a stale network mount — leaves the question
11940
+ undecided, and an undecided answer must retain the row, because the
11941
+ consequence of a wrong "absent" is a permanent loss of attribution while
11942
+ the consequence of a wrong "present" is one retained row.
11943
+
11944
+ ``pathlib.Path.exists()`` cannot serve here: it reports False for every
11945
+ ``OSError``, which is exactly the conflation this function exists to
11946
+ avoid.
11947
+
11948
+ RESIDUAL, stated rather than left to be discovered. ``ENOENT`` is also what
11949
+ an UNMOUNTED volume produces for a path beneath a mount point whose
11950
+ directory still exists, so this reports such a path as genuinely absent
11951
+ when it is only unreachable. The walk-liveness guard upstream covers the
11952
+ common shape of that — a root whose whole tree went away is not walked, so
11953
+ its rows are never offered here — but a volume that unmounts between the
11954
+ walk and this check is not covered, and the consequence is the attribution
11955
+ loss the paragraph above describes. Distinguishing it needs a mount-table
11956
+ read, which this leaf deliberately does not do.
11957
+ """
11958
+ try:
11959
+ os.lstat(path_str)
11960
+ except (FileNotFoundError, NotADirectoryError):
11961
+ return True
11962
+ except OSError:
11963
+ return False
11964
+ return False
11965
+
11966
+
11967
+ #: `_conversation_staging_free_bytes` returns this when it could not measure.
11968
+ #: A SENTINEL rather than a number, because every number is wrong here: the
11969
+ #: floor reads as "enough" only while the requirement is also the floor, and
11970
+ #: the requirement is now the store's live size.
11971
+ CONVERSATION_FREE_SPACE_UNKNOWN = None
11972
+
11973
+
11974
+ def _conversation_staging_free_bytes() -> "int | None":
11975
+ """Free bytes on the volume holding ``conversations.db``.
11976
+
11977
+ A named seam so the preflight can be driven to refuse in a test without
11978
+ filling a real disk.
11979
+
11980
+ Returns ``CONVERSATION_FREE_SPACE_UNKNOWN`` when the volume cannot be
11981
+ measured, and the caller then SKIPS the check. This used to return the
11982
+ staging floor with a comment saying that refusing every rebuild because
11983
+ `statvfs` failed would be the worse failure — true while the comparison was
11984
+ against that same floor, and false the moment the requirement became
11985
+ `_conversation_rebuild_free_bytes_required`, which is gigabytes on any real
11986
+ store. The degraded reading was then always below the requirement, so the
11987
+ documented fail-open deferred every rebuild instead of admitting it.
11988
+ """
11989
+ try:
11990
+ usage = shutil.disk_usage(_cctally_core.CONVERSATIONS_DB_PATH.parent)
11991
+ except OSError:
11992
+ return CONVERSATION_FREE_SPACE_UNKNOWN
11993
+ return int(usage.free)
11994
+
11995
+
11996
+ def _conversation_rebuild_free_bytes_required(conn) -> int:
11997
+ """Free bytes this particular store's rebuild needs (#752, #780).
11998
+
11999
+ The live corpus is `(page_count - freelist_count) * page_size`. A rebuild
12000
+ clears and replays every message row without reusing the pages the clear
12001
+ freed inside the same pass — measured on a copy-on-write clone of the
12002
+ production store, where the file grew by 6.15 GB and 93% of the growth was
12003
+ freelist — so the live size is what the replay can add before any of it
12004
+ comes back. Demand that much, and never less than the staging floor.
12005
+
12006
+ An unreadable pragma degrades to the floor rather than raising: a preflight
12007
+ that cannot measure must not be the thing that stops a rebuild.
12008
+ """
12009
+ try:
12010
+ page_size = int(conn.execute("PRAGMA page_size").fetchone()[0])
12011
+ page_count = int(conn.execute("PRAGMA page_count").fetchone()[0])
12012
+ freelist = int(conn.execute("PRAGMA freelist_count").fetchone()[0])
12013
+ except (sqlite3.Error, TypeError, ValueError, IndexError):
12014
+ return _CONVERSATION_STAGING_FREE_BYTES_FLOOR
12015
+ live_bytes = max(0, page_count - freelist) * max(0, page_size)
12016
+ return max(_CONVERSATION_STAGING_FREE_BYTES_FLOOR, live_bytes)
12017
+
12018
+
12019
+ def _resolve_source_incarnation(conn, path_str, fh, st, prev_row, *,
12020
+ force_replay: bool = False):
12021
+ """Decide whether this file is still the file the cursor describes (#777).
12022
+
12023
+ Continuity requires ALL of: the same stored INODE; a current size at least
12024
+ the committed offset; and a matching guard digest over the whole committed
12025
+ prefix. Any of an inode change, a shrink, a size-preserving rewrite, or a
12026
+ digest mismatch mints a new incarnation and forces byte-zero treatment.
12027
+
12028
+ The device is stored as corroborating evidence and as a diagnostic, and it
12029
+ never decides (#814). `st_dev` is assigned when a volume is mounted rather
12030
+ than when a file is created, so an ordinary remount renumbers every stored
12031
+ `device_id` at once with no file changed; deciding on it minted a fresh
12032
+ incarnation whose high-water is zero, and `_resolve_record_account` then
12033
+ re-attributed that transcript's whole history to whichever account was
12034
+ active at that moment. The verdict is delegated to
12035
+ `_lib_ingest_frontier.source_identity_replaced`, which both Codex walks
12036
+ already call, so one rule serves every provider.
12037
+
12038
+ THE RESIDUAL that creates: a file on a DIFFERENT device presenting the same
12039
+ numeric inode, at least as long as the cursor, whose committed prefix
12040
+ hashes identically, is no longer distinguishable as a new physical
12041
+ incarnation. That is safe because it moves in the safe direction — it
12042
+ PRESERVES continuity where a boundary arguably existed, so no bump occurs,
12043
+ no high-water resets and no re-attribution happens. Resuming from the
12044
+ cursor is content-correct, because the committed records and their
12045
+ digest-bound stamps are byte-identical by construction, and any suffix is
12046
+ processed as new input under the existing rules. What is lost is a physical
12047
+ incarnation boundary, not data and not attribution.
12048
+
12049
+ This is strictly stronger than ``_conversation_target_risk``, which decides
12050
+ the size-preserving case on mtime and is therefore defeated by a ``touch``.
12051
+ A digest over the committed prefix cannot be restored by resetting metadata.
12052
+
12053
+ Reads through the caller's ALREADY-OPEN descriptor. The caller takes an
12054
+ ``fstat`` before and after and treats any change across that pair as a new
12055
+ incarnation, which closes the time-of-check/time-of-use gap a
12056
+ stat-then-open sequence leaves open.
12057
+
12058
+ ``force_replay`` is set by a rebuild, and it separates two questions the
12059
+ stored cursor would otherwise conflate: WHERE to start reading, and WHETHER
12060
+ this is still the same file. A rebuild replays from byte zero regardless,
12061
+ but it must still verify continuity against the cursor the previous life
12062
+ committed — otherwise a file rewritten in place would keep its incarnation
12063
+ across the rebuild and inherit stamps written for bytes that no longer
12064
+ exist. The returned ``digest`` is empty in that case, because the caller is
12065
+ about to hash the whole file from zero.
12066
+ """
12067
+ stored_incarnation = prev_row[3] if prev_row else None
12068
+ stored_device = prev_row[4] if prev_row else None
12069
+ stored_inode = prev_row[5] if prev_row else None
12070
+ stored_prefix = prev_row[6] if prev_row else None
12071
+ committed = int(prev_row[2]) if prev_row else 0
12072
+ continuous = (
12073
+ stored_incarnation is not None
12074
+ # THE INODE DECIDES; THE DEVICE ONLY CORROBORATES (#814). `st_dev` is
12075
+ # assigned when a volume is MOUNTED, not when a file is created, so an
12076
+ # ordinary remount renumbers every stored `device_id` at once with no
12077
+ # file changed. Deciding replacement on it mints a fresh incarnation
12078
+ # whose high-water is zero, and `_resolve_record_account` then gives
12079
+ # every already-attributed record to whichever account is active now.
12080
+ # `#769 S6` removed this from both Codex walks; this was the last
12081
+ # provider source-cursor site where the device still decided.
12082
+ and not _ingest_frontier.source_identity_replaced(
12083
+ stored_device, stored_inode, st.st_dev, st.st_ino,
12084
+ )
12085
+ and st.st_size >= committed
12086
+ )
12087
+ if continuous:
12088
+ digest = _lib_conversation.prefix_digest(fh, committed)
12089
+ if digest.hexdigest() == (stored_prefix or _EMPTY_PREFIX_SHA256):
12090
+ if force_replay:
12091
+ return _SourceIncarnation(
12092
+ stored_incarnation, False, 0,
12093
+ _lib_conversation.prefix_digest(fh, 0),
12094
+ )
12095
+ return _SourceIncarnation(
12096
+ stored_incarnation, False, committed, digest)
12097
+ return _SourceIncarnation(
12098
+ _lib_conversation.new_source_incarnation_id(), True, 0,
12099
+ _lib_conversation.prefix_digest(fh, 0),
12100
+ )
12101
+
12102
+
12103
+ def _stat_pair_broken(before, after) -> bool:
12104
+ """Whether the file's IDENTITY broke under the open descriptor (#777).
12105
+
12106
+ Classified by what changed across one open handle, not by whether anything
12107
+ changed (spec §2, corrected after the Tranche 2 review). The provider
12108
+ appends to an active transcript continuously, so reading any change as a
12109
+ break made an ordinary append raise ``files_failed`` — which withholds the
12110
+ conversation frontier certificate, leaves ``conversation_rebuild_claude_pending``
12111
+ set so the whole rebuild runs again, and discards the records the pass had
12112
+ already read.
12113
+
12114
+ * a different device or inode is a different file;
12115
+ * a SMALLER size is a truncation, or a replacement written in place;
12116
+ * the SAME size with a changed mtime is a size-preserving rewrite — the
12117
+ case ``_conversation_target_risk`` classifies as ``source_replaced``;
12118
+ * a LARGER size is an ordinary append, and a continuation. An mtime
12119
+ change alongside it is that same append and is not read separately.
12120
+
12121
+ An append is safe to continue from because the bytes the reader validated
12122
+ did not move: the guard digest covers exactly the range this pass read, the
12123
+ partial-tail rewind in ``_iter_sync_entries`` already ends the read at a
12124
+ record boundary, and the next sync resumes from the cursor this pass
12125
+ commits.
12126
+
12127
+ What the pair CANNOT see is a ``rename``. The descriptor keeps the old
12128
+ inode, so ``fstat`` reports the file the reader actually read. That is the
12129
+ right outcome rather than a hole: those bytes are self-consistent and are
12130
+ committed under the identity that really held them, and the next sync opens
12131
+ the new directory entry, sees a different inode and mints a fresh
12132
+ incarnation.
12133
+ """
12134
+ if before.st_dev != after.st_dev or before.st_ino != after.st_ino:
12135
+ return True
12136
+ if after.st_size < before.st_size:
12137
+ return True
12138
+ return (after.st_size == before.st_size
12139
+ and after.st_mtime_ns != before.st_mtime_ns)
12140
+
12141
+
12142
+ def _load_account_stamps(conn, path_str):
12143
+ """Every durable stamp and classified gap for one source path (#777)."""
12144
+ stamps = {
12145
+ (row[0], int(row[1]), row[2]): row[3] for row in conn.execute(
12146
+ "SELECT source_incarnation_id,byte_offset,record_sha256,account_key "
12147
+ "FROM claude_conversation_account_stamps WHERE canonical_source_path=?",
12148
+ (path_str,),
12149
+ )
12150
+ }
12151
+ gaps = {
12152
+ int(row[0]): row[1] for row in conn.execute(
12153
+ "SELECT byte_offset,account_key "
12154
+ "FROM claude_conversation_account_stamp_gaps WHERE source_path=?",
12155
+ (path_str,),
12156
+ )
12157
+ }
12158
+ return stamps, gaps
12159
+
12160
+
12161
+ def _stamp_coverage_complete(conn) -> bool:
12162
+ """Whether migration 009's stamp backfill has finished (#777).
12163
+
12164
+ Read defensively: a store whose ``cache_meta`` cannot be read reports
12165
+ INCOMPLETE, so the rebuild defers rather than replaying under a rule whose
12166
+ precondition it could not verify.
12167
+ """
12168
+ try:
12169
+ return conn.execute(
12170
+ "SELECT 1 FROM cache_meta WHERE key=?",
12171
+ ("claude_account_stamp_coverage_complete",),
12172
+ ).fetchone() is not None
12173
+ except sqlite3.OperationalError:
12174
+ return False
12175
+
12176
+
12177
+ def _resolve_record_account(stamps, gaps, incarnation_id, offset, digest,
12178
+ high_water, active_account_key):
12179
+ """Which account one replayed record belongs to (#777). Pure.
12180
+
12181
+ Returns ``(account_key, needs_fresh_stamp)``.
12182
+
12183
+ THREE cases, and the separation between the last two is what the high-water
12184
+ map exists to provide. Without it the miss rule and the new-bytes rule
12185
+ contradict each other: a missing stamp would have to mean both "historical
12186
+ and unknown" and "arriving now".
12187
+
12188
+ 1. A stamp under THIS incarnation for this offset and this digest is the
12189
+ recorded observation, and it wins. The digest is part of the key, so a
12190
+ record whose bytes changed at the same offset cannot inherit it, and the
12191
+ incarnation is part of the key, so a path reused by a different file
12192
+ cannot either.
12193
+ 2. Below the published high-water mark with no stamp, a classified GAP row
12194
+ still carries the attribution the pre-rebuild message row held — the
12195
+ same evidence a stamp holds, minus a digest for bytes that no longer
12196
+ exist. With neither, the record is genuinely historical and unknown, so
12197
+ it takes NULL: writing the currently active account there is precisely
12198
+ the silent re-attribution #777 reports.
12199
+ 3. At or beyond the high-water mark the record is first observed now, so it
12200
+ takes the freshly resolved active identity and gets its own stamp.
12201
+ """
12202
+ stamped = stamps.get((incarnation_id, offset, digest))
12203
+ if (incarnation_id, offset, digest) in stamps:
12204
+ return stamped, False
12205
+ if offset < high_water:
12206
+ if offset in gaps:
12207
+ return gaps[offset], False
12208
+ return None, False
12209
+ return active_account_key, True
12210
+
12211
+
12212
+ def _publish_title_generation(conn, _pause=None) -> None:
12213
+ """Replace the live title and rollup tables with the staged generation.
12214
+
12215
+ ONE transaction containing both DELETEs and both INSERTs, so WAL snapshot
12216
+ isolation gives every concurrent reader either the whole previous
12217
+ generation or the whole new one. That is the guarantee the rejected A/B
12218
+ slot design would have had to implement by hand with a published-slot
12219
+ pointer, reader routing and a retirement protocol.
12220
+
12221
+ Nothing here writes ``conversation_title_fts``. It is an external-content
12222
+ FTS5 index bound BY NAME to ``conversation_ai_titles``, so it follows
12223
+ through the existing ``conv_title_fts_ai`` / ``_ad`` / ``_au`` triggers
12224
+ when this rewrites the live table — and it is simply absent under
12225
+ ``fts5_unavailable`` or inside migration 018's pending window, where the
12226
+ triggers do not exist and there is nothing to drive.
12227
+
12228
+ Readers are NOT routed through a TEMP view. A TEMP view can serve neither
12229
+ ``MATCH`` nor ``rowid``, so a reader behind one could not use the title
12230
+ index at all.
12231
+
12232
+ ``_pause`` is a test seam invoked with the transaction OPEN and both
12233
+ DELETEs issued. A concurrent reader running at that moment is the direct
12234
+ evidence of atomicity, and raising from it is the direct evidence that a
12235
+ crash inside the publish rolls back to the previous generation.
12236
+ """
12237
+ conn.execute("BEGIN IMMEDIATE")
12238
+ try:
12239
+ conn.execute("DELETE FROM conversation_ai_titles")
12240
+ conn.execute("DELETE FROM conversation_sessions")
12241
+ if _pause is not None:
12242
+ _pause()
12243
+ conn.execute(
12244
+ "INSERT INTO conversation_ai_titles"
12245
+ "(session_id,ai_title,source_path,byte_offset) "
12246
+ "SELECT session_id,ai_title,source_path,byte_offset "
12247
+ "FROM conversation_ai_titles_staging"
12248
+ )
12249
+ conn.execute(
12250
+ "INSERT INTO conversation_sessions"
12251
+ "(session_id,msg_count,started_utc,last_activity_utc,project_label,"
12252
+ "cost_usd,cache_rebuild_count,git_branch,models_json,title,"
12253
+ "render_revision) "
12254
+ "SELECT session_id,msg_count,started_utc,last_activity_utc,"
12255
+ "project_label,cost_usd,cache_rebuild_count,git_branch,models_json,"
12256
+ "title,render_revision FROM conversation_sessions_staging"
12257
+ )
12258
+ conn.commit()
12259
+ except BaseException:
12260
+ conn.rollback()
12261
+ raise
12262
+ # Truncated only AFTER the publish commits, so a crash between the two
12263
+ # leaves a complete staged generation a retry can republish rather than a
12264
+ # half-emptied one.
12265
+ conn.execute("DELETE FROM conversation_ai_titles_staging")
12266
+ conn.execute("DELETE FROM conversation_sessions_staging")
12267
+ conn.commit()
12268
+
12269
+
11166
12270
  def _report_conversation_progress(
11167
12271
  progress: "Callable[[str, Any], None] | None",
11168
12272
  phase: str,
@@ -11187,6 +12291,19 @@ def sync_claude_conversations(
11187
12291
  transaction as its message/title rows. No cache.db table is written, and
11188
12292
  the core accounting cursor is neither read nor advanced.
11189
12293
  """
12294
+ # #779: FIRST executable statement, before IngestStats, before the data
12295
+ # directory is created, before the lock file is touched or opened, and
12296
+ # before any flock. This check used to sit below the rebuild branch, so a
12297
+ # call carrying both arguments committed the pending marker, ran the
12298
+ # maintenance preparation and executed the four destructive DELETEs, and
12299
+ # only then raised over an already-emptied store. Nothing downstream may
12300
+ # re-check the pair: `rebuild = rebuild or pending_rebuild` legitimately
12301
+ # turns a targeted call into an inherited global rebuild, and that case is
12302
+ # answered by the `deferred_reason="rebuild_pending"` return above it.
12303
+ if only_paths is not None and rebuild:
12304
+ raise ValueError(
12305
+ "sync_claude_conversations: only_paths is incompatible with rebuild"
12306
+ )
11190
12307
  stats = IngestStats()
11191
12308
  did_from_zero_replay = False
11192
12309
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
@@ -11240,15 +12357,17 @@ def sync_claude_conversations(
11240
12357
  # performing half of it from a stale binary is worse than not
11241
12358
  # performing it at all, and deferred_reason is how the caller
11242
12359
  # learns it did nothing.
11243
- try:
11244
- fp_row = conn.execute(
11245
- "SELECT value FROM cache_meta WHERE key=?",
11246
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
11247
- ).fetchone()
11248
- except sqlite3.OperationalError:
11249
- fp_row = None
11250
- stored_fp = fp_row[0] if fp_row else None
11251
- if not _pricing_write_authorized(stored_fp):
12360
+ # #728: `may_reset_rebuild_target`, not the write predicate. The
12361
+ # ordering is the same, but this is the path that must fail closed
12362
+ # — a refused clear costs nothing, while a clear performed on the
12363
+ # strength of a read that never happened empties a rollup this
12364
+ # process may then be refused permission to re-derive. The
12365
+ # degrade-to-None shape this replaces made a failed read look like
12366
+ # a store with nothing to protect, which is exactly the state that
12367
+ # authorizes the clear.
12368
+ fp_obs = _read_pricing_fingerprint_observation(conn)
12369
+ stored_fp = fp_obs.raw
12370
+ if not may_reset_rebuild_target(fp_obs, PRICING_SNAPSHOT_DATE):
11252
12371
  _record_pricing_write_refusal(conn, stored_fp)
11253
12372
  # COMMIT. _record_pricing_write_refusal leaves the transaction
11254
12373
  # to its caller, and this caller returns immediately, so
@@ -11275,6 +12394,40 @@ def sync_claude_conversations(
11275
12394
  conn.commit()
11276
12395
  stats.deferred_reason = "pricing_write_refused"
11277
12396
  return stats
12397
+ # #777: a rebuild replays every record and decides its account
12398
+ # from the durable stamps. Over a store whose backfill has not
12399
+ # finished, the lookup-miss rule would read "no stamp" as
12400
+ # "historical and unknown" and write NULL for records that DO have
12401
+ # a recorded attribution — the whole loss the backfill exists to
12402
+ # prevent, performed deliberately. Refuse before anything
12403
+ # destructive; migration 009 completes the backfill on the next
12404
+ # open, and the rebuild succeeds after it.
12405
+ if not _stamp_coverage_complete(conn):
12406
+ stats.deferred_reason = "stamp_backfill_pending"
12407
+ return stats
12408
+ # #752: check free space BEFORE the destructive step rather than
12409
+ # failing partway through. A rebuild that runs out of space after
12410
+ # clearing is the state that produced this issue.
12411
+ #
12412
+ # A reading of `CONVERSATION_FREE_SPACE_UNKNOWN` skips the check
12413
+ # entirely rather than comparing. That is the documented fail-open:
12414
+ # a preflight that could not measure must not be the thing that
12415
+ # stops a rebuild.
12416
+ _free_bytes = _conversation_staging_free_bytes()
12417
+ if (_free_bytes is not CONVERSATION_FREE_SPACE_UNKNOWN
12418
+ and _free_bytes
12419
+ < _conversation_rebuild_free_bytes_required(conn)):
12420
+ stats.deferred_reason = "insufficient_free_space"
12421
+ return stats
12422
+ # #780: refuse a new rebuild while the reclaim backlog is over its
12423
+ # hard ceiling. A rebuild adds a whole staged generation of churn
12424
+ # to a store that is already failing to return the space it freed,
12425
+ # so starting one makes the condition worse rather than better.
12426
+ _retention_sib = _load_lib("_lib_conversation_retention")
12427
+ if _retention_sib.reclaim_backlog_over_ceiling(
12428
+ _retention_sib.read_reclaim_pending(conn)):
12429
+ stats.deferred_reason = "reclaim_backlog_over_ceiling"
12430
+ return stats
11278
12431
  # Commit the retry marker before the destructive clear. A killed
11279
12432
  # #395 worker therefore leaves a partial transcript store visibly
11280
12433
  # pending instead of advancing it to a false-complete state.
@@ -11290,25 +12443,52 @@ def sync_claude_conversations(
11290
12443
  active_account_key=active_account_key,
11291
12444
  )
11292
12445
 
11293
- rebuild_account_stamps: dict[tuple[str, int], str | None] = {}
12446
+ # #777: the per-incarnation published high-water offset, captured at
12447
+ # rebuild START. It is what separates the two attribution rules the
12448
+ # review found contradictory otherwise: an offset BELOW it with no
12449
+ # stamp is genuinely historical and unknown, so it takes NULL and never
12450
+ # the active account; an offset AT OR BEYOND it is first observed now,
12451
+ # so it resolves the active identity freshly and creates its stamp.
12452
+ # Keyed by (path, incarnation) — PER-INCARNATION, which is the whole
12453
+ # point. A high-water mark is a claim about what a particular life of a
12454
+ # file already published; a file replaced since then has published
12455
+ # nothing, so every one of its records is first observed now and takes
12456
+ # the active identity rather than being read as historical-and-unknown.
12457
+ high_water: "dict[tuple[str, str], int]" = {}
12458
+ # #752: a rebuild writes its titles and rollup into staging and does
12459
+ # not touch the live tables until the publish, so readers keep seeing
12460
+ # the previous complete generation for the whole replay.
12461
+ title_table = "conversation_ai_titles"
12462
+ rollup_table = "conversation_sessions"
11294
12463
  if rebuild:
11295
- rebuild_account_stamps = {
11296
- (str(path), int(offset)): account_key
11297
- for path, offset, account_key in conn.execute(
11298
- "SELECT source_path,byte_offset,account_key "
11299
- "FROM conversation_messages"
12464
+ high_water = {
12465
+ (str(path), incarnation): int(offset)
12466
+ for path, incarnation, offset in conn.execute(
12467
+ "SELECT path,source_incarnation_id,last_byte_offset "
12468
+ "FROM conversation_source_files "
12469
+ "WHERE source_incarnation_id IS NOT NULL"
11300
12470
  )
11301
12471
  }
12472
+ title_table = "conversation_ai_titles_staging"
12473
+ rollup_table = "conversation_sessions_staging"
11302
12474
  clear_conversation_messages(conn)
11303
- conn.execute("DELETE FROM conversation_ai_titles")
11304
- conn.execute("DELETE FROM conversation_sessions")
11305
- conn.execute("DELETE FROM conversation_source_files")
12475
+ # Staging starts empty; a leftover generation from an interrupted
12476
+ # rebuild describes bytes this pass is about to replay.
12477
+ conn.execute("DELETE FROM conversation_ai_titles_staging")
12478
+ conn.execute("DELETE FROM conversation_sessions_staging")
12479
+ # The source-file rows are LEFT INTACT (#777). Deleting them, as
12480
+ # this used to, would take every `source_incarnation_id` with them,
12481
+ # and a stamp is keyed by the incarnation — so every stamp written
12482
+ # before the rebuild would miss and the replay that is supposed to
12483
+ # preserve the retained attribution would discard all of it.
12484
+ # Resetting the cursor instead would be almost as bad: continuity
12485
+ # would then be checked against an empty prefix, which any file
12486
+ # satisfies, so a file rewritten in place would keep its incarnation
12487
+ # and inherit stamps for bytes that no longer exist. The rows stay
12488
+ # exactly as the previous life committed them, and the replay is
12489
+ # forced from byte zero by `force_replay` instead.
11306
12490
  conn.commit()
11307
12491
 
11308
- if only_paths is not None and rebuild:
11309
- raise ValueError(
11310
- "sync_claude_conversations: only_paths is incompatible with rebuild"
11311
- )
11312
12492
  paths = (
11313
12493
  [pathlib.Path(path) for path in sorted(only_paths)
11314
12494
  if pathlib.Path(path).is_file()]
@@ -11317,10 +12497,15 @@ def sync_claude_conversations(
11317
12497
  )
11318
12498
  stats.files_total = len(paths)
11319
12499
  _report_conversation_progress(progress, "ingest", stats)
12500
+ # (size_bytes, mtime_ns, last_byte_offset, source_incarnation_id,
12501
+ # device_id, inode, committed_prefix_sha256) — the first three are the
12502
+ # cursor this loop has always read; the last four are #777's identity,
12503
+ # consumed by `_resolve_source_incarnation` at the same indices.
11320
12504
  existing = {
11321
- row[0]: (row[1], row[2], row[3])
12505
+ row[0]: tuple(row[1:])
11322
12506
  for row in conn.execute(
11323
- "SELECT path,size_bytes,mtime_ns,last_byte_offset "
12507
+ "SELECT path,size_bytes,mtime_ns,last_byte_offset,"
12508
+ "source_incarnation_id,device_id,inode,committed_prefix_sha256 "
11324
12509
  "FROM conversation_source_files"
11325
12510
  )
11326
12511
  }
@@ -11354,7 +12539,7 @@ def sync_claude_conversations(
11354
12539
  continue
11355
12540
  size, mtime_ns = st.st_size, st.st_mtime_ns
11356
12541
  prev = existing.get(path_str)
11357
- if prev is not None and size == prev[0]:
12542
+ if not rebuild and prev is not None and size == prev[0]:
11358
12543
  stats.files_skipped_unchanged += 1
11359
12544
  _report_conversation_progress(progress, "ingest", stats)
11360
12545
  continue
@@ -11362,23 +12547,60 @@ def sync_claude_conversations(
11362
12547
  if targeted and truncated:
11363
12548
  stats.deferred_reason = "truncation"
11364
12549
  return stats
11365
- start_offset = 0 if prev is None or truncated else prev[2]
12550
+ # The read start is decided by the incarnation resolver below,
12551
+ # which is the only place that knows whether the cursor still
12552
+ # describes this file. Initialised here only so the OSError path
12553
+ # below has a value.
12554
+ start_offset = 0
11366
12555
  conv_rows: list[tuple[Any, ...]] = []
11367
12556
  ai_rows: list[tuple[Any, ...]] = []
12557
+ stamp_rows: list[tuple[Any, ...]] = []
11368
12558
  final_offset = start_offset
12559
+ stamps, gaps = _load_account_stamps(conn, path_str)
11369
12560
  try:
11370
- with open(jp, "r", encoding="utf-8", errors="replace") as fh:
12561
+ # BINARY, and ONE descriptor for the whole file (#777). Binary
12562
+ # because a stamp's digest is over the record's raw bytes, which
12563
+ # the text layer's replacement decoding does not preserve. One
12564
+ # descriptor because the incarnation decision, the guard-digest
12565
+ # read and the ingest read must all describe the same open file:
12566
+ # a stat-then-open sequence leaves a window in which the file is
12567
+ # replaced between the decision and the read, and the records
12568
+ # would then be filed under the previous incarnation's identity.
12569
+ with open(jp, "rb") as fh:
12570
+ st_before = os.fstat(fh.fileno())
12571
+ incarnation = _resolve_source_incarnation(
12572
+ conn, path_str, fh, st_before, prev,
12573
+ force_replay=rebuild)
12574
+ start_offset = incarnation.start_offset
12575
+ file_high_water = high_water.get(
12576
+ (path_str, incarnation.incarnation_id), 0)
12577
+ if incarnation.is_new:
12578
+ # A new incarnation forces byte-zero treatment: the
12579
+ # bytes before the cursor are not the bytes the cursor
12580
+ # described, so resuming from it would skip real records
12581
+ # and stamp the rest against an identity that never held
12582
+ # them.
12583
+ truncated = prev is not None
11371
12584
  fh.seek(start_offset)
11372
- for _offset, _cost, mrow, ai in _iter_sync_entries(
12585
+ for _offset, _cost, mrow, ai, raw in _iter_sync_entries(
11373
12586
  fh,
11374
12587
  path_str,
11375
12588
  include_cost=False,
12589
+ with_raw=True,
11376
12590
  ):
11377
12591
  if mrow is not None:
11378
- account_key = rebuild_account_stamps.get(
11379
- (path_str, int(mrow.byte_offset)),
12592
+ offset = int(mrow.byte_offset)
12593
+ digest = _lib_conversation.record_sha256(raw)
12594
+ account_key, fresh_stamp = _resolve_record_account(
12595
+ stamps, gaps, incarnation.incarnation_id,
12596
+ offset, digest, file_high_water,
11380
12597
  active_account_key,
11381
12598
  )
12599
+ if fresh_stamp:
12600
+ stamp_rows.append(
12601
+ (path_str, incarnation.incarnation_id,
12602
+ offset, digest, account_key)
12603
+ )
11382
12604
  conv_rows.append(
11383
12605
  _conv_row_tuple(
11384
12606
  mrow, path_str, account_key,
@@ -11390,6 +12612,59 @@ def sync_claude_conversations(
11390
12612
  (ai.session_id, ai.ai_title, path_str, ai.byte_offset)
11391
12613
  )
11392
12614
  final_offset = fh.tell()
12615
+ st_after = os.fstat(fh.fileno())
12616
+ if _stat_pair_broken(st_before, st_after):
12617
+ # The file's IDENTITY broke while we were reading it —
12618
+ # a shrink, or a size-preserving rewrite. An append is
12619
+ # deliberately not this branch.
12620
+ #
12621
+ # Commit nothing. The next sync opens the file again
12622
+ # and re-decides the incarnation from scratch: the
12623
+ # committed-prefix guard mints a fresh incarnation and
12624
+ # replays from zero whenever the bytes before the
12625
+ # cursor moved, and otherwise continuity holds and it
12626
+ # resumes from the stored cursor. Either way nothing
12627
+ # from this partial read is carried forward, which is
12628
+ # what the descriptor pair exists to guarantee.
12629
+ stats.files_failed += 1
12630
+ _report_conversation_progress(progress, "ingest", stats)
12631
+ continue
12632
+ # The committed prefix digest is carried FORWARD from the
12633
+ # verified one rather than rehashed: the object is already
12634
+ # positioned at `start_offset`, so only the bytes this pass
12635
+ # ingested are added. That is what keeps the guard's cost
12636
+ # proportional to the append rather than to the file.
12637
+ fh.seek(start_offset)
12638
+ incarnation.mutable_digest.update(
12639
+ fh.read(max(0, final_offset - start_offset)))
12640
+ committed_prefix = incarnation.mutable_digest.hexdigest()
12641
+ device_id, inode = st_after.st_dev, st_after.st_ino
12642
+ # The recorded size and mtime come from the fstat taken at
12643
+ # the START of the read, not the end. They are not identity
12644
+ # — `_resolve_source_incarnation` decides that on device,
12645
+ # inode and the committed-prefix digest — they are the
12646
+ # change detector the `size == prev[0]` skip above reads.
12647
+ # Now that an append during the read is a continuation
12648
+ # rather than a break, recording the post-append size would
12649
+ # claim this pass consumed bytes it never reached, and the
12650
+ # next sync would skip the file as unchanged and lose those
12651
+ # records permanently. The pair is equal whenever nothing
12652
+ # raced, so this changes nothing outside that window.
12653
+ #
12654
+ # A RESIDUAL, stated rather than left implicit. The safest
12655
+ # recorded size is `final_offset`, because the partial-tail
12656
+ # rewind can end the read before `st_before.st_size` when
12657
+ # the file's last line is incomplete — so `size` can exceed
12658
+ # what this pass actually consumed by that tail. Narrowing
12659
+ # it to `final_offset` is not correct either: `size` is the
12660
+ # change detector for the `size == prev[0]` skip above, and
12661
+ # recording the consumed offset instead would make the next
12662
+ # sync see a size difference on an unchanged file and
12663
+ # re-read it every tick. The residual is bounded by one
12664
+ # partial record and is self-correcting, because the
12665
+ # writer finishes that line and the size moves again;
12666
+ # identity does not depend on either number.
12667
+ size, mtime_ns = st_before.st_size, st_before.st_mtime_ns
11393
12668
  except OSError as exc:
11394
12669
  eprint(f"[conversations] could not read {jp}: {exc}")
11395
12670
  stats.files_failed += 1
@@ -11416,7 +12691,7 @@ def sync_claude_conversations(
11416
12691
  (path_str,),
11417
12692
  )
11418
12693
  conn.execute(
11419
- "DELETE FROM conversation_ai_titles WHERE source_path=?",
12694
+ f"DELETE FROM {title_table} WHERE source_path=?",
11420
12695
  (path_str,),
11421
12696
  )
11422
12697
  stats.files_reset_truncated += 1
@@ -11425,21 +12700,43 @@ def sync_claude_conversations(
11425
12700
  _fill_file_touches(
11426
12701
  conn, scope=[(row[3], row[4]) for row in conv_rows]
11427
12702
  )
12703
+ if stamp_rows:
12704
+ # #777: batched into the SAME transaction as the message
12705
+ # batch, never issued per row. One stamp per message is new
12706
+ # write volume across every future record, and a per-row
12707
+ # transaction would multiply the ingest's commit cost by the
12708
+ # record count.
12709
+ conn.executemany(
12710
+ "INSERT OR IGNORE INTO claude_conversation_account_stamps"
12711
+ "(canonical_source_path,source_incarnation_id,"
12712
+ "byte_offset,record_sha256,account_key) "
12713
+ "VALUES(?,?,?,?,?)",
12714
+ stamp_rows,
12715
+ )
11428
12716
  if ai_rows:
11429
- conn.executemany(_AI_TITLE_UPSERT_SQL, ai_rows)
12717
+ conn.executemany(_ai_title_upsert_sql(title_table), ai_rows)
11430
12718
  conn.execute(
11431
12719
  "INSERT INTO conversation_source_files "
11432
- "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at) "
11433
- "VALUES(?,?,?,?,?) ON CONFLICT(path) DO UPDATE SET "
12720
+ "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at,"
12721
+ "device_id,inode,source_incarnation_id,"
12722
+ "committed_prefix_sha256) "
12723
+ "VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(path) DO UPDATE SET "
11434
12724
  "size_bytes=excluded.size_bytes,mtime_ns=excluded.mtime_ns,"
11435
12725
  "last_byte_offset=excluded.last_byte_offset,"
11436
- "last_ingested_at=excluded.last_ingested_at",
12726
+ "last_ingested_at=excluded.last_ingested_at,"
12727
+ "device_id=excluded.device_id,inode=excluded.inode,"
12728
+ "source_incarnation_id=excluded.source_incarnation_id,"
12729
+ "committed_prefix_sha256=excluded.committed_prefix_sha256",
11437
12730
  (
11438
12731
  path_str,
11439
12732
  size,
11440
12733
  mtime_ns,
11441
12734
  final_offset,
11442
12735
  dt.datetime.now(dt.timezone.utc).isoformat(),
12736
+ device_id,
12737
+ inode,
12738
+ incarnation.incarnation_id,
12739
+ committed_prefix,
11443
12740
  ),
11444
12741
  )
11445
12742
  conn.commit()
@@ -11454,9 +12751,127 @@ def sync_claude_conversations(
11454
12751
  stats.files_failed += 1
11455
12752
  _report_conversation_progress(progress, "ingest", stats)
11456
12753
 
12754
+ if rebuild and only_paths is None:
12755
+ # A rebuild used to DELETE every `conversation_source_files` row and
12756
+ # let the walk recreate the ones it found, which also removed rows
12757
+ # for paths the walk no longer discovers. #777 stopped deleting the
12758
+ # rows, because they carry the incarnation identity every durable
12759
+ # stamp is keyed by — so the stale rows have to be removed here
12760
+ # instead, by difference against the walked set. Without this a
12761
+ # rebuild over a different corpus leaves the previous corpus's
12762
+ # paths in the store, which is both wrong and a privacy leak.
12763
+ #
12764
+ # The STAMPS and GAPS for those paths go with the row. Tranche 2
12765
+ # kept them, on the grounds that a stamp is the only surviving
12766
+ # record of who the records belonged to — but no read path can
12767
+ # reach one once the source-file row is gone. `_resolve_record_account`
12768
+ # keys on `(source_incarnation_id, byte_offset, record_sha256)`;
12769
+ # with no `conversation_source_files` row `_resolve_source_incarnation`
12770
+ # sees `prev is None` and always mints a fresh incarnation, and the
12771
+ # per-incarnation high-water map is built from that same table, so
12772
+ # the gap half is unreachable for the same reason. Retaining them
12773
+ # grows two tables without bound and preserves nothing. Giving them
12774
+ # a reader instead was the alternative and is worse: a path-and-
12775
+ # offset fallback would let an unrelated file that reuses a path
12776
+ # inherit the old attribution, which is exactly the inheritance the
12777
+ # incarnation key exists to prevent.
12778
+ #
12779
+ # GUARDED BY WALK LIVENESS (spec §2, second loss path). "Absent
12780
+ # from this walk" is a weaker fact than "absent from disk", and
12781
+ # taking the first for the second makes an unavailable corpus
12782
+ # delete every stamp in the store: an unmounted volume or a wrong
12783
+ # `CLAUDE_CONFIG_DIR` produces an empty walk, every tracked path is
12784
+ # then a difference, and the one thing here that is NOT
12785
+ # re-derivable goes with it. Two conditions, both required.
12786
+ #
12787
+ # THE PRICE OF THAT GUARD, stated rather than left to be found. A
12788
+ # path that is still on disk but no longer in scope — the walk now
12789
+ # covers a different tree, because `CLAUDE_CONFIG_DIR` moved or the
12790
+ # store was copied beside the corpus it was derived from — is
12791
+ # RETAINED, so the paragraph above overstates the case: the prune
12792
+ # removes the previous corpus's paths only when that corpus is
12793
+ # actually gone. Retaining them costs orphaned source rows and
12794
+ # their stamps, since `clear_conversation_messages` has already
12795
+ # removed every message they describe. That is deliberate. Nothing
12796
+ # available here distinguishes a corpus that moved from a root that
12797
+ # this walk could not reach, and only one of the two mistakes is
12798
+ # recoverable.
12799
+ walked = {str(jp) for jp in paths}
12800
+ stale = (
12801
+ [
12802
+ row[0] for row in conn.execute(
12803
+ "SELECT path FROM conversation_source_files")
12804
+ if row[0] not in walked and _path_is_genuinely_absent(row[0])
12805
+ ]
12806
+ if walked
12807
+ else []
12808
+ )
12809
+ for start in range(0, len(stale), 400):
12810
+ chunk = stale[start:start + 400]
12811
+ placeholders = ",".join("?" for _ in chunk)
12812
+ conn.execute(
12813
+ f"DELETE FROM conversation_source_files "
12814
+ f"WHERE path IN ({placeholders})",
12815
+ chunk,
12816
+ )
12817
+ conn.execute(
12818
+ f"DELETE FROM claude_conversation_account_stamps "
12819
+ f"WHERE canonical_source_path IN ({placeholders})",
12820
+ chunk,
12821
+ )
12822
+ conn.execute(
12823
+ f"DELETE FROM claude_conversation_account_stamp_gaps "
12824
+ f"WHERE source_path IN ({placeholders})",
12825
+ chunk,
12826
+ )
12827
+ if stale:
12828
+ conn.commit()
12829
+
11457
12830
  _report_conversation_progress(progress, "rollup", stats)
11458
12831
  rollup_authorized = _arm_rollup_backfill_on_pricing_change(conn)
11459
- if _conversation_sessions_backfill_pending(conn):
12832
+ if rebuild:
12833
+ # #752: a rebuild ALWAYS derives its rollup in full, into staging,
12834
+ # regardless of the backfill flag — it has just replayed every
12835
+ # message row, so a scoped recompute would describe a fraction of
12836
+ # the store. The live rollup is still the previous complete
12837
+ # generation at this point and stays so until the publish below.
12838
+ if _recompute_conversation_sessions(conn, target=rollup_table):
12839
+ conn.execute(
12840
+ "DELETE FROM cache_meta "
12841
+ "WHERE key='conversation_sessions_backfill_pending'"
12842
+ )
12843
+ _clear_pricing_write_refusal(conn)
12844
+ conn.commit()
12845
+ # Titles that arrived on the LIVE table while staging was being
12846
+ # built are folded in before the publish, or the publish would
12847
+ # silently drop them. `INSERT OR IGNORE` keeps the replayed
12848
+ # generation authoritative for any session both hold.
12849
+ #
12850
+ # SCOPED to the sessions this replay actually produced (spec
12851
+ # §1, corrected after the Tranche 2 review). An unconditional
12852
+ # fold makes every published generation a permanent superset of
12853
+ # the previous one: the table never shrinks, titles for deleted
12854
+ # sessions and removed worktrees persist forever, and title
12855
+ # search returns hits for sessions with no messages and no
12856
+ # rollup row — because the rollup half IS re-derived and
12857
+ # correctly drops them. Retention cannot compensate, because
12858
+ # `_prune_claude` selects session ids from
12859
+ # `conversation_messages` and a session with no message rows is
12860
+ # never a prune candidate. The rollup staging table is the
12861
+ # replayed generation's own session set, so it is the gate.
12862
+ conn.execute(
12863
+ "INSERT OR IGNORE INTO conversation_ai_titles_staging"
12864
+ "(session_id,ai_title,source_path,byte_offset) "
12865
+ "SELECT session_id,ai_title,source_path,byte_offset "
12866
+ "FROM conversation_ai_titles WHERE session_id IN "
12867
+ "(SELECT session_id FROM conversation_sessions_staging)"
12868
+ )
12869
+ conn.commit()
12870
+ _publish_title_generation(conn)
12871
+ else:
12872
+ conn.commit()
12873
+ rollup_authorized = False
12874
+ elif _conversation_sessions_backfill_pending(conn):
11460
12875
  # #705: a refused process must NOT consume the flag. It leaves it
11461
12876
  # set for the next authorized process, so a newer process that armed
11462
12877
  # the backfill and died is not followed by a stale successor either
@@ -11552,7 +12967,28 @@ def sync_codex_conversations(
11552
12967
  only_paths: "set[str] | None" = None,
11553
12968
  progress: "Callable[[str, CodexIngestStats], None] | None" = None,
11554
12969
  ) -> CodexIngestStats:
11555
- """Delta-sync Codex events/search rows into conversations.db (#320)."""
12970
+ """Delta-sync Codex events/search rows into conversations.db (#320).
12971
+
12972
+ #779 (the Codex twin, folded in by the Tranche 1 review): the
12973
+ incompatible-argument check below is the FIRST executable statement, for
12974
+ exactly the reasons its Claude twin states. This function used to write the
12975
+ `conversation_rebuild_codex_pending` marker, commit, call
12976
+ `_clear_codex_conversation_store(conn)`, commit again, and only then raise
12977
+ — so a bad argument pair destroyed the entire Codex conversation store
12978
+ durably before refusing to do the work. The issue text named only the
12979
+ Claude path; this is the same defect in the twin function in the same file,
12980
+ and the remedy is the same reordering.
12981
+
12982
+ Nothing downstream may re-check the pair. `rebuild = rebuild or
12983
+ pending_rebuild or contract_rebuild or codex_replay_pending` legitimately
12984
+ turns a targeted call into inherited global work, and that case is answered
12985
+ by the `deferred_reason="rebuild_pending"` return above the merge, not by
12986
+ treating an inherited rebuild as an explicitly incompatible argument.
12987
+ """
12988
+ if only_paths is not None and rebuild:
12989
+ raise ValueError(
12990
+ "sync_codex_conversations: only_paths is incompatible with rebuild"
12991
+ )
11556
12992
  stats = CodexIngestStats()
11557
12993
  did_from_zero_replay = False
11558
12994
  rebuild_account_stamps: dict[tuple[str, int], str | None] = {}
@@ -11657,10 +13093,6 @@ def sync_codex_conversations(
11657
13093
  _clear_codex_conversation_store(conn)
11658
13094
  conn.commit()
11659
13095
 
11660
- if only_paths is not None and rebuild:
11661
- raise ValueError(
11662
- "sync_codex_conversations: only_paths is incompatible with rebuild"
11663
- )
11664
13096
  files = (
11665
13097
  _qualify_codex_targets(only_paths)
11666
13098
  if only_paths is not None
@@ -11668,13 +13100,18 @@ def sync_codex_conversations(
11668
13100
  )
11669
13101
  stats.files_total = len(files)
11670
13102
  _report_conversation_progress(progress, "ingest", stats)
13103
+ # The last two members are `device_id`/`inode` (#769 S6). They are read
13104
+ # by the per-file decision below, not merely retained: a replacement
13105
+ # that lands at the same size moves neither the size nor the cursor, so
13106
+ # identity is the only evidence that the retained offset describes a
13107
+ # file that is no longer there.
11671
13108
  existing = {
11672
13109
  row[0]: tuple(row[1:])
11673
13110
  for row in conn.execute(
11674
13111
  "SELECT path,size_bytes,mtime_ns,last_byte_offset,source_root_key,"
11675
13112
  "last_session_id,last_model,last_total_tokens,"
11676
13113
  "last_native_thread_id,last_root_thread_id,last_parent_thread_id,"
11677
- "last_conversation_key,last_turn_id "
13114
+ "last_conversation_key,last_turn_id,device_id,inode "
11678
13115
  "FROM codex_conversation_source_files"
11679
13116
  )
11680
13117
  }
@@ -11779,18 +13216,37 @@ def sync_codex_conversations(
11779
13216
  continue
11780
13217
  size, mtime_ns = st.st_size, st.st_mtime_ns
11781
13218
  prev = existing.get(path_str)
11782
- if prev is not None and size == prev[0] and prev[3] == discovered.source_root_key:
13219
+ # A stored identity that differs from the one just statted means the
13220
+ # retained offset points into a file that no longer occupies this
13221
+ # pathname. THE INODE DECIDES and the device only corroborates:
13222
+ # `_lib_ingest_frontier.source_identity_replaced` owns the verdict
13223
+ # for the planner and for both walks, states why a remount must not
13224
+ # reach it, and degrades a NULL or unreadable stored value to the
13225
+ # size comparison below rather than raising out of this loop.
13226
+ replaced = (
13227
+ prev is not None
13228
+ and _ingest_frontier.source_identity_replaced(
13229
+ prev[12], prev[13], st.st_dev, st.st_ino)
13230
+ )
13231
+ if (
13232
+ prev is not None and not replaced and size == prev[0]
13233
+ and prev[3] == discovered.source_root_key
13234
+ ):
11783
13235
  stats.files_skipped_unchanged += 1
11784
13236
  _report_conversation_progress(progress, "ingest", stats)
11785
13237
  continue
11786
13238
  reset_file = (
11787
13239
  prev is not None
11788
- and (size < prev[0] or prev[3] != discovered.source_root_key)
13240
+ and (
13241
+ size < prev[0] or replaced
13242
+ or prev[3] != discovered.source_root_key
13243
+ )
11789
13244
  )
11790
13245
  if targeted and reset_file:
11791
13246
  stats.deferred_reason = (
11792
13247
  "requalification"
11793
13248
  if prev is not None and prev[3] != discovered.source_root_key
13249
+ else "source_replaced" if replaced
11794
13250
  else "truncation"
11795
13251
  )
11796
13252
  return stats
@@ -11995,8 +13451,8 @@ def sync_codex_conversations(
11995
13451
  "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at,"
11996
13452
  "source_root_key,last_session_id,last_model,last_total_tokens,"
11997
13453
  "last_native_thread_id,last_root_thread_id,last_parent_thread_id,"
11998
- "last_conversation_key,last_turn_id) "
11999
- "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?) "
13454
+ "last_conversation_key,last_turn_id,device_id,inode) "
13455
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) "
12000
13456
  "ON CONFLICT(path) DO UPDATE SET "
12001
13457
  "size_bytes=excluded.size_bytes,mtime_ns=excluded.mtime_ns,"
12002
13458
  "last_byte_offset=excluded.last_byte_offset,"
@@ -12009,7 +13465,8 @@ def sync_codex_conversations(
12009
13465
  "last_root_thread_id=excluded.last_root_thread_id,"
12010
13466
  "last_parent_thread_id=excluded.last_parent_thread_id,"
12011
13467
  "last_conversation_key=excluded.last_conversation_key,"
12012
- "last_turn_id=excluded.last_turn_id",
13468
+ "last_turn_id=excluded.last_turn_id,"
13469
+ "device_id=excluded.device_id,inode=excluded.inode",
12013
13470
  (
12014
13471
  path_str,
12015
13472
  size,
@@ -12025,6 +13482,11 @@ def sync_codex_conversations(
12025
13482
  terminal.parent_thread_id if terminal else initial_parent,
12026
13483
  terminal.conversation_key if terminal else initial_conversation,
12027
13484
  normalized.terminal.turn_id,
13485
+ # The SAME pre-read stat that produced `size`/`mtime_ns`
13486
+ # above, committed in this one statement beside the
13487
+ # offset it describes (#769 S6).
13488
+ int(st.st_dev),
13489
+ int(st.st_ino),
12028
13490
  ),
12029
13491
  )
12030
13492
  conn.commit()
@@ -12402,6 +13864,30 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
12402
13864
  f"[cache-sync] pruned {res.pruned_files} orphaned file(s), "
12403
13865
  f"{res.pruned_entries} cost row(s), {res.pruned_messages} message(s)"
12404
13866
  )
13867
+ if res.prune_refused:
13868
+ # #729: a STAGED failure under docs/cli-contract.md — the command
13869
+ # was understood and attempted, and part of it did not complete —
13870
+ # so exit 3, not the parity-family 1 or the usage 2. Precedence:
13871
+ # flock contention above is reported first because nothing was
13872
+ # attempted at all, and this outranks the residual-path report
13873
+ # below because a residual is work deliberately not done, while
13874
+ # this is committed deletions that were not followed by an
13875
+ # authorized re-derive.
13876
+ # #769 S3: the cause travels on the result. Stating the version
13877
+ # skew unconditionally described one of three refusing states as
13878
+ # if it were the only one.
13879
+ eprint(
13880
+ f"[cache-sync] the conversation rollup re-derive was refused "
13881
+ f"for {res.prune_refused_files} file(s): "
13882
+ f"{pricing_refusal_cause_phrase(res.prune_refused_state)}. "
13883
+ f"The orphan rows are deleted and the rollup backfill is "
13884
+ f"armed; run `cctally cache-sync --prune-orphans` again once "
13885
+ f"an authorized process can complete it, and see `cctally "
13886
+ f"doctor pricing.conversation_rollup_writer` for the step "
13887
+ f"that clears this state."
13888
+ )
13889
+ conn.close()
13890
+ return 3
12405
13891
  if res.residual_paths:
12406
13892
  eprint(
12407
13893
  f"[cache-sync] {len(res.residual_paths)} orphan(s) left in place "