cctally 1.107.0 → 1.109.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/CHANGELOG.md +149 -0
  2. package/bin/_cctally_account.py +4 -0
  3. package/bin/_cctally_alerts.py +152 -7
  4. package/bin/_cctally_cache.py +1590 -153
  5. package/bin/_cctally_config.py +3 -2
  6. package/bin/_cctally_core.py +881 -28
  7. package/bin/_cctally_dashboard.py +1231 -180
  8. package/bin/_cctally_dashboard_conversation.py +263 -8
  9. package/bin/_cctally_dashboard_envelope.py +247 -16
  10. package/bin/_cctally_dashboard_share.py +154 -36
  11. package/bin/_cctally_dashboard_sources.py +2146 -347
  12. package/bin/_cctally_db.py +634 -5
  13. package/bin/_cctally_diagnosis_sources.py +57 -11
  14. package/bin/_cctally_diff.py +15 -8
  15. package/bin/_cctally_doctor.py +155 -18
  16. package/bin/_cctally_five_hour.py +1295 -40
  17. package/bin/_cctally_forecast.py +140 -32
  18. package/bin/_cctally_journal.py +659 -64
  19. package/bin/_cctally_milestone_history.py +86 -6
  20. package/bin/_cctally_parser.py +9 -5
  21. package/bin/_cctally_percent_breakdown.py +284 -13
  22. package/bin/_cctally_project.py +48 -76
  23. package/bin/_cctally_quota.py +55 -27
  24. package/bin/_cctally_quota_model.py +173 -17
  25. package/bin/_cctally_record.py +1978 -613
  26. package/bin/_cctally_rederive.py +0 -2
  27. package/bin/_cctally_refresh.py +2 -1
  28. package/bin/_cctally_setup.py +272 -110
  29. package/bin/_cctally_share.py +24 -5
  30. package/bin/_cctally_source_analytics.py +553 -30
  31. package/bin/_cctally_statusline.py +113 -20
  32. package/bin/_cctally_store.py +61 -4
  33. package/bin/_cctally_tui.py +257 -99
  34. package/bin/_cctally_weekrefs.py +343 -107
  35. package/bin/_lib_alerts_payload.py +48 -1
  36. package/bin/_lib_blocks.py +127 -20
  37. package/bin/_lib_cache_report.py +20 -5
  38. package/bin/_lib_codex_conversation_query.py +11 -1
  39. package/bin/_lib_codex_hooks.py +796 -83
  40. package/bin/_lib_conversation.py +119 -0
  41. package/bin/_lib_conversation_dispatch.py +31 -1
  42. package/bin/_lib_conversation_query.py +61 -14
  43. package/bin/_lib_conversation_retention.py +413 -25
  44. package/bin/_lib_dashboard_sources.py +267 -7
  45. package/bin/_lib_diagnosis.py +10 -0
  46. package/bin/_lib_diff_kernel.py +45 -11
  47. package/bin/_lib_doctor.py +385 -29
  48. package/bin/_lib_ingest_frontier.py +1366 -49
  49. package/bin/_lib_journal.py +81 -0
  50. package/bin/_lib_merge_gate.py +307 -0
  51. package/bin/_lib_meter_rate_change.py +132 -3
  52. package/bin/_lib_pricing.py +544 -23
  53. package/bin/_lib_pricing_check.py +5 -4
  54. package/bin/_lib_record.py +130 -12
  55. package/bin/_lib_render.py +9 -3
  56. package/bin/_lib_retained_size.py +27 -2
  57. package/bin/_lib_segment_summary.py +66 -8
  58. package/bin/_lib_share_templates.py +28 -18
  59. package/bin/_lib_snapshot_cache.py +115 -1
  60. package/bin/_lib_source_retry.py +320 -0
  61. package/bin/_lib_statusline_candidates.py +400 -36
  62. package/bin/_lib_subscription_weeks.py +140 -63
  63. package/bin/_lib_tick_stats.py +119 -0
  64. package/bin/_lib_view_models.py +251 -76
  65. package/bin/cctally +83 -3
  66. package/dashboard/static/assets/ConversationsView-TO5bpIyc.js +72 -0
  67. package/dashboard/static/assets/{DoctorModal-DcyMPwvn.js → DoctorModal-BZGXOlc2.js} +1 -1
  68. package/dashboard/static/assets/ModalRoot-C_fvugTl.js +1 -0
  69. package/dashboard/static/assets/{ProjectsDrillPanel-ecN8oCwv.js → ProjectsDrillPanel-KbO_KWa7.js} +1 -1
  70. package/dashboard/static/assets/SourceDetailModal-BrOPLVRL.js +1 -0
  71. package/dashboard/static/assets/{UpdateModal-CKlFE0ER.js → UpdateModal-CYgEFAi4.js} +3 -3
  72. package/dashboard/static/assets/dashboardStream.shared-worker-DbG6Ef0T.js +1 -0
  73. package/dashboard/static/assets/index-Cj_l7T7u.css +1 -0
  74. package/dashboard/static/assets/index-DLVwACUM.js +13 -0
  75. package/dashboard/static/assets/outlineNavigation-DXX-RjKl.js +9 -0
  76. package/dashboard/static/assets/useKeymap-DQx_A1MB.js +1 -0
  77. package/dashboard/static/dashboard.html +3 -3
  78. package/package.json +3 -1
  79. package/dashboard/static/assets/ConversationsView-BHbw2W1l.js +0 -72
  80. package/dashboard/static/assets/ModalRoot-BYV-99Rq.js +0 -1
  81. package/dashboard/static/assets/SourceDetailModal-CUwdD7_v.js +0 -1
  82. package/dashboard/static/assets/dashboardStream.shared-worker-1XTMV3nr.js +0 -1
  83. package/dashboard/static/assets/index-D8svRv_9.js +0 -13
  84. package/dashboard/static/assets/index-klO46NcU.css +0 -1
  85. package/dashboard/static/assets/outlineNavigation-CVse0Hj9.js +0 -9
  86. package/dashboard/static/assets/useKeymap-CJ-Pi17D.js +0 -1
@@ -225,6 +225,16 @@ claude_usage_dict = _load_lib("_lib_pricing").claude_usage_dict
225
225
  # stdlib leaf as the two names above.
226
226
  parse_pricing_fingerprint = _load_lib("_lib_pricing").parse_pricing_fingerprint
227
227
 
228
+ # #728: the four-state observation and the two authorization predicates, from
229
+ # the same circular-safe stdlib leaf. The classifier is pure — the SELECT and
230
+ # its failure mode are reported to it by `_read_pricing_fingerprint_observation`
231
+ # below.
232
+ classify_pricing_fingerprint = _load_lib(
233
+ "_lib_pricing").classify_pricing_fingerprint
234
+ may_write_materialized_cost = _load_lib(
235
+ "_lib_pricing").may_write_materialized_cost
236
+ may_reset_rebuild_target = _load_lib("_lib_pricing").may_reset_rebuild_target
237
+
228
238
  # Shared by the fused per-file walk AND backfill_conversation_messages so the
229
239
  # column list, placeholders, and tuple order live in ONE place — a column
230
240
  # add/reorder can't silently desync the two ingest paths (which would land
@@ -257,6 +267,23 @@ _AI_TITLE_UPSERT_SQL = (
257
267
  "ai_title=excluded.ai_title, source_path=excluded.source_path, byte_offset=excluded.byte_offset"
258
268
  )
259
269
 
270
+ def _ai_title_upsert_sql(table: str = "conversation_ai_titles") -> str:
271
+ """The AI-title upsert, targeted at the live table or its staging twin.
272
+
273
+ A rebuild builds its replacement generation into staging and leaves the
274
+ live table alone until the publish (#752), so the ONE statement that writes
275
+ a title has to be able to name either. Parameterised through this helper
276
+ rather than by two literals, because two literals drift.
277
+ """
278
+ return (
279
+ f"INSERT INTO {table}(session_id,ai_title,source_path,byte_offset) "
280
+ "VALUES(?,?,?,?) "
281
+ "ON CONFLICT(session_id) DO UPDATE SET "
282
+ "ai_title=excluded.ai_title, source_path=excluded.source_path, "
283
+ "byte_offset=excluded.byte_offset"
284
+ )
285
+
286
+
260
287
  # ---------------------------------------------------------------------------
261
288
  # session_entries upsert (#195: extracted from the inline string in sync_cache
262
289
  # so the steady-state and re-walk variants share ONE body).
@@ -428,6 +455,7 @@ def _iter_sync_entries(
428
455
  *,
429
456
  include_cost: bool = True,
430
457
  include_conversations: bool = True,
458
+ with_raw: bool = False,
431
459
  ):
432
460
  """Fused single-pass sync walker (#138). Yields
433
461
  ``(byte_offset, cost_or_None, msgrow_or_None, aititle_or_None)`` for each
@@ -453,19 +481,39 @@ def _iter_sync_entries(
453
481
  partial mid-write tail line (no trailing newline) rewinds the handle and
454
482
  stops, so ``fh.tell()`` after the loop is the cost cursor's ``final_offset``
455
483
  and the next sync re-reads the line once the newline lands.
484
+
485
+ ``with_raw`` (#777) makes the walker yield FIVE-tuples, appending the
486
+ record's exact raw on-disk bytes with the terminator removed, and requires
487
+ ``fh`` to be a BINARY handle. It exists because a durable account stamp is
488
+ keyed by a digest of those bytes, and the decoded text this walker otherwise
489
+ produces is not them: the text path decodes with ``errors="replace"``, so a
490
+ record carrying invalid UTF-8 decodes to a different string than it was
491
+ written as, and its digest would move the day that replacement policy did.
492
+ Re-reading or re-parsing the line to recover the bytes would break the
493
+ one-parse-per-line invariant (#138), so the binary branch decodes the line
494
+ it already read and hands both halves to the same classification below.
495
+ ``fh.tell()`` on a binary handle is a true byte offset rather than the text
496
+ layer's cookie, which is the same number the cursor has always stored.
456
497
  """
498
+ newline = b"\n" if with_raw else "\n"
457
499
  while True:
458
500
  offset = fh.tell()
459
501
  line = fh.readline()
460
502
  if not line:
461
503
  return
462
- if not line.endswith("\n"):
504
+ if not line.endswith(newline):
463
505
  # Partial tail line — writer is mid-flight. Rewind so the next sync
464
506
  # re-reads this line once the newline is in place (and so fh.tell()
465
507
  # reports the cost cursor's stop, never past the partial).
466
508
  fh.seek(offset)
467
509
  return
468
- stripped = line.strip()
510
+ if with_raw:
511
+ raw_span = _lib_conversation.strip_record_terminator(line)
512
+ text = line.decode("utf-8", errors="replace")
513
+ else:
514
+ raw_span = None
515
+ text = line
516
+ stripped = text.strip()
469
517
  if not stripped:
470
518
  continue
471
519
  # #279 S2 F1: passive parse-health counters over the new-byte span.
@@ -496,7 +544,10 @@ def _iter_sync_entries(
496
544
  if include_conversations else None
497
545
  )
498
546
  if cost is not None or mrow is not None or ai is not None:
499
- yield offset, cost, mrow, ai
547
+ if with_raw:
548
+ yield offset, cost, mrow, ai, raw_span
549
+ else:
550
+ yield offset, cost, mrow, ai
500
551
 
501
552
 
502
553
  def _iter_claude_jsonl_files():
@@ -1977,8 +2028,13 @@ def _load_codex_session_files_rows(
1977
2028
  ) -> dict:
1978
2029
  """Cursor rows from ``codex_session_files`` for ONLY the given paths (spec
1979
2030
  §5.1 — the targeted preload must never load every row like the full-sync
1980
- path). Same 13-tuple value shape as ``sync_codex_cache``'s full ``existing``
1981
- map, so the per-file delta logic is byte-identical between the two modes."""
2031
+ path). Same 15-tuple value shape as ``sync_codex_cache``'s full ``existing``
2032
+ map, so the per-file delta logic is byte-identical between the two modes.
2033
+
2034
+ ``device_id``/``inode`` are the last two members and they are NOT
2035
+ diagnostics: the resume decision reads them, because a replacement that
2036
+ lands at the same size is otherwise indistinguishable from a file nothing
2037
+ touched (#769 S6)."""
1982
2038
  out: dict = {}
1983
2039
  if not paths:
1984
2040
  return out
@@ -1986,7 +2042,8 @@ def _load_codex_session_files_rows(
1986
2042
  "path, size_bytes, mtime_ns, last_byte_offset, "
1987
2043
  "last_session_id, last_model, last_total_tokens, source_root_key, "
1988
2044
  "last_native_thread_id, last_root_thread_id, last_parent_thread_id, "
1989
- "last_conversation_key, last_turn_id, ingest_complete"
2045
+ "last_conversation_key, last_turn_id, ingest_complete, "
2046
+ "device_id, inode"
1990
2047
  )
1991
2048
  for i in range(0, len(paths), 400):
1992
2049
  chunk = paths[i:i + 400]
@@ -1995,10 +2052,7 @@ def _load_codex_session_files_rows(
1995
2052
  f"SELECT {cols} FROM codex_session_files WHERE path IN ({placeholders})",
1996
2053
  chunk,
1997
2054
  ):
1998
- out[row[0]] = (
1999
- row[1], row[2], row[3], row[4], row[5], row[6], row[7],
2000
- row[8], row[9], row[10], row[11], row[12], row[13],
2001
- )
2055
+ out[row[0]] = tuple(row[1:])
2002
2056
  return out
2003
2057
 
2004
2058
 
@@ -3181,6 +3235,8 @@ def _write_codex_file_batch(
3181
3235
  file_account_decision: "tuple[int, str | None] | None" = None,
3182
3236
  anchor_resolver: "CodexResetAnchorResolver | None" = None,
3183
3237
  ingest_complete: bool = True,
3238
+ device_id: "int | None" = None,
3239
+ inode: "int | None" = None,
3184
3240
  ) -> int:
3185
3241
  """Write one fully-buffered Codex file atomically and return entry changes.
3186
3242
 
@@ -3278,19 +3334,27 @@ def _write_codex_file_batch(
3278
3334
  last_seen_utc=excluded.last_seen_utc""",
3279
3335
  [(*row, now_iso, now_iso) for row in thread_rows],
3280
3336
  )
3337
+ # #769 S6: `device_id`/`inode` are the identity of the file the caller
3338
+ # STATTED before reading, and they land in the same statement as the scan
3339
+ # target, the final offset and the completion flag. A replacement can
3340
+ # therefore never leave a new identity paired with an old offset: either
3341
+ # the whole file batch commits or none of it does.
3281
3342
  conn.execute(
3282
3343
  """INSERT OR REPLACE INTO codex_session_files
3283
3344
  (path, size_bytes, mtime_ns, last_byte_offset, last_ingested_at,
3284
3345
  last_session_id, last_model, last_total_tokens, source_root_key,
3285
3346
  last_native_thread_id, last_root_thread_id, last_parent_thread_id,
3286
- last_conversation_key, last_turn_id, account_key, ingest_complete)
3287
- VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
3347
+ last_conversation_key, last_turn_id, account_key, ingest_complete,
3348
+ device_id, inode)
3349
+ VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
3288
3350
  (
3289
3351
  path_str, size, mtime_ns, final_offset, now_iso, last_session_id,
3290
3352
  last_model, last_total_tokens, discovered.source_root_key,
3291
3353
  last_native_thread_id, last_root_thread_id, last_parent_thread_id,
3292
3354
  last_conversation_key, last_turn_id, account_key,
3293
3355
  1 if ingest_complete else 0,
3356
+ None if device_id is None else int(device_id),
3357
+ None if inode is None else int(inode),
3294
3358
  ),
3295
3359
  )
3296
3360
  if prune_roots:
@@ -3420,12 +3484,32 @@ class PruneResult:
3420
3484
  """Outcome of _prune_orphaned_cache_entries: how much of the derived Claude
3421
3485
  surface was removed for safely-orphaned source paths, plus the orphan paths
3422
3486
  left in place (residual — a gate failed, so `--rebuild` is the escape hatch)
3423
- and whether the flock was contended (nothing mutated)."""
3487
+ and whether the flock was contended (nothing mutated).
3488
+
3489
+ ``prune_refused`` / ``prune_refused_files`` mirror ``CodexIngestStats``
3490
+ (#729). The deletions still committed; what was refused is the
3491
+ ``conversation_sessions`` re-derive that should have followed them, so the
3492
+ store is left partially derived. `_prune_orphaned_cache_entries` discarded
3493
+ `_recompute_conversation_sessions`' return, and
3494
+ `_lib_ingest_frontier.provider_sync_certifiable` already tested
3495
+ ``getattr(stats, "prune_refused", False)`` — with no such attribute the
3496
+ test always read False and a refused prune certified as clean. Adding the
3497
+ field does not add a check; it makes an existing blind one start firing.
3498
+
3499
+ ``prune_refused_state`` carries the observation state that refused (#769
3500
+ S3). Until #728 a refusal had exactly one cause — this process's pricing
3501
+ table being older than the store's — and the operator messages stated it as
3502
+ a fact. MALFORMED and DEGRADED now refuse too, and neither is about which
3503
+ table is older, so the cause travels with the result rather than being
3504
+ re-derived (or guessed) at each message site."""
3424
3505
  pruned_files: int = 0
3425
3506
  pruned_entries: int = 0
3426
3507
  pruned_messages: int = 0
3427
3508
  residual_paths: "list[str]" = field(default_factory=list)
3428
3509
  contended: bool = False
3510
+ prune_refused: bool = False
3511
+ prune_refused_files: int = 0
3512
+ prune_refused_state: "str | None" = None
3429
3513
 
3430
3514
 
3431
3515
  def _progress_stderr(stats: IngestStats, *, force: bool = False) -> None:
@@ -3740,7 +3824,20 @@ def _prune_orphaned_cache_entries(conn, *, lock_timeout=None):
3740
3824
  f"DELETE FROM conversation_messages WHERE source_path IN ({ph})", safe_paths)
3741
3825
  conv.execute(
3742
3826
  f"DELETE FROM conversation_ai_titles WHERE source_path IN ({ph})", safe_paths)
3743
- _recompute_conversation_sessions(conv, list(pruned_sids))
3827
+ # #729: this return was discarded. A refused re-derive still
3828
+ # commits — the three DELETEs above are done, and the refusal ARMS
3829
+ # `conversation_sessions_backfill_pending` and latches its record
3830
+ # inside this same BEGIN, so rolling back would discard the very
3831
+ # signal that tells the next authorized process to finish the job.
3832
+ # What must not happen is reporting the outcome as clean.
3833
+ refusal_state: "list[str]" = []
3834
+ if not _recompute_conversation_sessions(
3835
+ conv, list(pruned_sids), refusal_out=refusal_state
3836
+ ):
3837
+ result.prune_refused = True
3838
+ result.prune_refused_files = len(safe_paths)
3839
+ result.prune_refused_state = (
3840
+ refusal_state[0] if refusal_state else None)
3744
3841
  conv.commit()
3745
3842
  except BaseException:
3746
3843
  conv.rollback()
@@ -5197,6 +5294,55 @@ CONVERSATION_ROLLUP_PRICING_REFUSED_KEY = (
5197
5294
  )
5198
5295
 
5199
5296
 
5297
+ def _read_pricing_fingerprint_observation(
5298
+ conn, key: str = CONVERSATION_ROLLUP_PRICING_FP_KEY,
5299
+ ):
5300
+ """Read one store's recorded pricing fingerprint as a four-state
5301
+ observation (#728). The ONE database half of the split.
5302
+
5303
+ Every writer site used to perform this SELECT itself and degrade a failed
5304
+ read to the same falsy value an unrecorded fingerprint produces, so a
5305
+ locked or schema-less store was indistinguishable from a fresh one and
5306
+ every writer authorized itself over it. Here the failure is reported as
5307
+ ``DEGRADED`` and the pure classifier in `_lib_pricing` decides what that
5308
+ means; no caller may re-derive the state from a bare value again.
5309
+
5310
+ ``error_kind`` is the coarse ``"operational_error"`` rather than the
5311
+ driver's message, because the message text is not a contract and the only
5312
+ decision that rests on it is "the read did not happen".
5313
+
5314
+ A store with no ``cache_meta`` TABLE is ABSENT, not DEGRADED, and the
5315
+ difference is decided by a structural probe rather than by matching the
5316
+ driver's message. A store that has nowhere to record a fingerprint has
5317
+ determinately recorded none — that is a fact read off ``sqlite_master``,
5318
+ not an assumption — while a store whose ``sqlite_master`` probe fails too
5319
+ is one this process genuinely could not read. The probe fails closed: an
5320
+ unanswerable probe reports DEGRADED.
5321
+ """
5322
+ try:
5323
+ row = conn.execute(
5324
+ "SELECT value FROM cache_meta WHERE key=?", (key,),
5325
+ ).fetchone()
5326
+ except sqlite3.OperationalError:
5327
+ try:
5328
+ table_present = conn.execute(
5329
+ "SELECT 1 FROM sqlite_master "
5330
+ "WHERE type='table' AND name='cache_meta'"
5331
+ ).fetchone() is not None
5332
+ except sqlite3.OperationalError:
5333
+ table_present = True
5334
+ if table_present:
5335
+ return classify_pricing_fingerprint(
5336
+ found=False, raw=None, error_kind="operational_error")
5337
+ return classify_pricing_fingerprint(
5338
+ found=False, raw=None, error_kind=None)
5339
+ return classify_pricing_fingerprint(
5340
+ found=row is not None,
5341
+ raw=(row[0] if row is not None else None),
5342
+ error_kind=None,
5343
+ )
5344
+
5345
+
5200
5346
  def _pricing_write_authorized(stored, process=None) -> bool:
5201
5347
  """Whether a process holding `process` pricing may write materialized
5202
5348
  conversation cost over a store that recorded `stored` (#705).
@@ -5221,26 +5367,102 @@ def _pricing_write_authorized(stored, process=None) -> bool:
5221
5367
  The parse itself lives in `_lib_pricing.parse_pricing_fingerprint`, because
5222
5368
  `doctor pricing.conversation_rollup_writer` reports which refusal state a
5223
5369
  store is in and must classify a value exactly as this function acts on it.
5224
- Two parses of one contract had already drifted apart once."""
5370
+ Two parses of one contract had already drifted apart once.
5371
+
5372
+ NO PRODUCTION CALLER REMAINS (#769 S3 A9). It takes a bare stored value, so
5373
+ it cannot express a read that failed, and every writer site now reads
5374
+ through `_read_pricing_fingerprint_observation` instead. It is kept as the
5375
+ VALUE-shaped surface over the same contract (#728), for tests and for any
5376
+ caller that genuinely holds a value rather than an observation; expressing
5377
+ it on top of the shared predicate rather than beside it keeps the two from
5378
+ drifting the way the two date parsers once did. Read this docstring as a
5379
+ description of a retained helper, not of a live authorization site."""
5225
5380
  process = PRICING_SNAPSHOT_DATE if process is None else process
5226
- if not stored:
5227
- return True
5228
- stored_date = parse_pricing_fingerprint(stored)
5229
- process_date = parse_pricing_fingerprint(process)
5230
- if stored_date is None or process_date is None:
5231
- return False
5232
- return stored_date <= process_date
5381
+ return may_write_materialized_cost(
5382
+ classify_pricing_fingerprint(
5383
+ found=stored is not None, raw=stored, error_kind=None),
5384
+ process,
5385
+ )
5233
5386
 
5234
5387
 
5235
- def _record_pricing_write_refusal(conn, stored) -> None:
5236
- """Latch that a process holding older pricing was refused a write (#705).
5388
+ #: One operator-facing clause per observation state that can refuse a
5389
+ #: materialized-cost write (#769 S3). Each message that reports a refusal reads
5390
+ #: its cause from here instead of restating the version-skew case, which #728
5391
+ #: made one of three. The `None` entry covers a caller that recorded no state —
5392
+ #: an older result object, or a refusal from a path that does not carry one —
5393
+ #: and must stay distinct from every real state rather than defaulting to the
5394
+ #: version-skew wording, which is what this mapping exists to stop.
5395
+ _PRICING_REFUSAL_CAUSE_PHRASES = {
5396
+ "present": (
5397
+ "this process's pricing table is older than the one the store recorded"
5398
+ ),
5399
+ "malformed": (
5400
+ "the pricing fingerprint the store recorded is not a date cctally can "
5401
+ "order against its own"
5402
+ ),
5403
+ "degraded": (
5404
+ "cctally could not read the pricing fingerprint the store recorded, so "
5405
+ "it has no evidence about what that store's cost needed protecting from"
5406
+ ),
5407
+ None: "cctally could not authorize the write",
5408
+ }
5409
+
5410
+
5411
+ def pricing_refusal_cause_phrase(state: "str | None") -> str:
5412
+ """The operator-facing cause clause for one refusing observation state."""
5413
+ return _PRICING_REFUSAL_CAUSE_PHRASES.get(
5414
+ state, _PRICING_REFUSAL_CAUSE_PHRASES[None])
5415
+
5416
+
5417
+ def _pricing_refusal_timestamp() -> str:
5418
+ """The instant an episode opened, to the second, in Zulu form.
5419
+
5420
+ A named seam rather than an inline expression so a test can pin it: the
5421
+ episode rules below are entirely about WHICH instant survives, and at
5422
+ second granularity two refusals in one test are otherwise indistinguishable.
5423
+ """
5424
+ return (
5425
+ dt.datetime.now(dt.timezone.utc)
5426
+ .replace(microsecond=0).isoformat().replace("+00:00", "Z")
5427
+ )
5428
+
5429
+
5430
+ def _pricing_refusal_record_active(record) -> bool:
5431
+ """Whether a refusal record describes an OPEN episode (#729).
5237
5432
 
5238
- Written only when the (process, store) pair DIFFERS from what is already
5239
- recorded. A dashboard ticks continuously, so rewriting per tick would take
5240
- the conversations writer lock purely to restate an unchanged fact — and it
5241
- would make the timestamp the latest refusal rather than the first, which
5242
- is the less useful of the two, because the first says how long the store
5243
- has been diverging.
5433
+ A record written before the ``active`` field existed was latched only on
5434
+ refusal and DELETED on convergence, so its mere presence means an open
5435
+ episode. Reading a missing ``active`` as False would silently stop
5436
+ reporting every refusal latched by an earlier version.
5437
+
5438
+ The non-dict guard is DEFENSIVE, not reachable from production today: the
5439
+ one caller decodes the stored JSON and short-circuits on anything that is
5440
+ not a dict before asking. It is kept because the value comes from a
5441
+ `cache_meta` row any version may have written, and a `True` returned for a
5442
+ truthy non-dict is the same fail-toward-reporting posture as the missing
5443
+ ``active`` key above (#769 S3 A9).
5444
+ """
5445
+ if not isinstance(record, dict):
5446
+ return bool(record)
5447
+ return record.get("active", True) is not False
5448
+
5449
+
5450
+ def _record_pricing_write_refusal(conn, stored) -> None:
5451
+ """Latch that a process holding older pricing was refused a write (#705),
5452
+ as a durable tombstone with EPISODE semantics (#729).
5453
+
5454
+ Not rewritten when the (process, store) pair is unchanged AND the episode
5455
+ is still open. A dashboard ticks continuously, so rewriting per tick would
5456
+ take the conversations writer lock purely to restate an unchanged fact —
5457
+ and it would make the timestamp the latest refusal rather than the first,
5458
+ which is the less useful of the two, because the first says how long the
5459
+ store has been diverging.
5460
+
5461
+ An INACTIVE-to-active transition for the same pair is the opposite case and
5462
+ starts a NEW episode with a fresh timestamp. Carrying the old timestamp
5463
+ across a completed convergence would report a fresh divergence as days old
5464
+ — the mirror image of the bug the preserved timestamp exists to avoid. A
5465
+ different pair likewise replaces the record and restamps.
5244
5466
 
5245
5467
  NEVER commits — the caller owns the transaction. _prune_orphaned_cache_entries
5246
5468
  reaches this from inside its own explicit ``BEGIN``, whose
@@ -5262,15 +5484,15 @@ def _record_pricing_write_refusal(conn, stored) -> None:
5262
5484
  existing = json.loads(row[0])
5263
5485
  except (ValueError, TypeError):
5264
5486
  existing = None # unreadable -> replace it with a readable one
5265
- if isinstance(existing, dict) and all(
5266
- existing.get(field) == value for field, value in pair.items()
5487
+ if (
5488
+ isinstance(existing, dict)
5489
+ and all(existing.get(f) == v for f, v in pair.items())
5490
+ and _pricing_refusal_record_active(existing)
5267
5491
  ):
5268
5492
  return
5269
5493
  record = dict(pair)
5270
- record["first_refused_at_utc"] = (
5271
- dt.datetime.now(dt.timezone.utc)
5272
- .replace(microsecond=0).isoformat().replace("+00:00", "Z")
5273
- )
5494
+ record["first_refused_at_utc"] = _pricing_refusal_timestamp()
5495
+ record["active"] = True
5274
5496
  try:
5275
5497
  _set_cache_meta(
5276
5498
  conn, CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,
@@ -5283,8 +5505,19 @@ def _record_pricing_write_refusal(conn, stored) -> None:
5283
5505
 
5284
5506
 
5285
5507
  def _clear_pricing_write_refusal(conn) -> None:
5286
- """Drop the refusal latch, unconditionally. NEVER commits — the caller owns
5287
- the transaction, like the record and arm helpers beside it.
5508
+ """SETTLE the refusal latch: close the episode, keep the record (#729).
5509
+ NEVER commits — the caller owns the transaction, like the record and arm
5510
+ helpers beside it.
5511
+
5512
+ It used to DELETE the row, so a store that had converged carried no
5513
+ evidence it had ever diverged and doctor could not tell "never refused"
5514
+ from "refused and recovered". The record now survives with
5515
+ ``active: false``, preserving the pair and the timestamp that says when
5516
+ that episode opened.
5517
+
5518
+ A store that never refused gets NO record: the settle is an UPDATE over an
5519
+ existing row, never an insert, or every healthy install would grow a
5520
+ tombstone for an episode that never happened.
5288
5521
 
5289
5522
  TWO callers, and the condition each satisfies before calling is the whole
5290
5523
  contract. The two pending-flag branches call it directly, atomically with
@@ -5301,9 +5534,31 @@ def _clear_pricing_write_refusal(conn) -> None:
5301
5534
  latched, while `doctor pricing.conversation_rollup_writer` promised the
5302
5535
  operator that the next tick would clear it."""
5303
5536
  try:
5304
- conn.execute(
5305
- "DELETE FROM cache_meta WHERE key=?",
5537
+ row = conn.execute(
5538
+ "SELECT value FROM cache_meta WHERE key=?",
5306
5539
  (CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,),
5540
+ ).fetchone()
5541
+ except sqlite3.OperationalError:
5542
+ return
5543
+ if not row or not row[0]:
5544
+ return
5545
+ try:
5546
+ record = json.loads(row[0])
5547
+ except (ValueError, TypeError):
5548
+ record = None
5549
+ if not isinstance(record, dict):
5550
+ # An unreadable record is still evidence the guard fired, and doctor
5551
+ # reports it under its own wording. There is no episode to settle and
5552
+ # nothing this function could preserve, so leave it exactly as it is
5553
+ # rather than replacing it with a settled record it cannot vouch for.
5554
+ return
5555
+ if record.get("active") is False:
5556
+ return
5557
+ record["active"] = False
5558
+ try:
5559
+ _set_cache_meta(
5560
+ conn, CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,
5561
+ json.dumps(record, sort_keys=True),
5307
5562
  )
5308
5563
  except sqlite3.OperationalError:
5309
5564
  pass
@@ -5366,19 +5621,34 @@ def _arm_rollup_backfill_on_pricing_change(conn) -> bool:
5366
5621
  Crash-safety is unchanged: the DURABLE backfill flag remains the recompute
5367
5622
  signal, so advancing the fingerprint here cannot strand stale cost (a crash
5368
5623
  after arming leaves the flag set -> next sync recomputes regardless of the
5369
- fingerprint). No-op when cache_meta is unavailable (path-less / degraded
5370
- conn). Caller path holds the cache.db.lock flock. Unlike the recompute
5624
+ fingerprint). Caller path holds the cache.db.lock flock. Unlike the recompute
5371
5625
  chokepoint, this helper owns its OWN transaction and commits, so the flag
5372
- and the refusal record are durable for the next process."""
5373
- try:
5374
- row = conn.execute(
5375
- "SELECT value FROM cache_meta WHERE key=?",
5376
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5377
- ).fetchone()
5378
- except sqlite3.OperationalError:
5379
- return True
5380
- stored = row[0] if row is not None else None
5381
- if not _pricing_write_authorized(stored):
5626
+ and the refusal record are durable for the next process.
5627
+
5628
+ #728: this site is the one whose ``except sqlite3.OperationalError:``
5629
+ clause returned True — "authorized" — DIRECTLY, without ever reaching the
5630
+ predicate, so a store it had failed to read authorized every writer behind
5631
+ it. It reads a four-state observation now like its two siblings, and a
5632
+ DEGRADED read refuses. Both refusal legs (arm the flag, latch the record,
5633
+ commit) are shared by the DEGRADED, MALFORMED and store-is-newer states;
5634
+ on a store whose ``cache_meta`` genuinely cannot be written, each of those
5635
+ three is a caught no-op, so a degraded connection still returns False
5636
+ without raising.
5637
+
5638
+ The docstring used to say "No-op when cache_meta is unavailable"; that
5639
+ sentence described the early ``return True`` #728 removed, and stating the
5640
+ replacement is the point of this paragraph (#769 S3 A9). An unavailable
5641
+ ``cache_meta`` is now TWO outcomes decided structurally rather than one.
5642
+ A store whose ``sqlite_master`` probe positively shows no ``cache_meta``
5643
+ table is ABSENT — it determinately recorded nothing — so this helper is
5644
+ authorized and proceeds, and `_set_cache_meta` creates the table and stamps
5645
+ the fingerprint. A read that failed for any other reason, including a probe
5646
+ that could not answer either, is DEGRADED and refuses; its three refusal
5647
+ writes are each individually caught, so the return is False rather than an
5648
+ exception."""
5649
+ obs = _read_pricing_fingerprint_observation(conn)
5650
+ stored = obs.raw
5651
+ if not may_write_materialized_cost(obs, PRICING_SNAPSHOT_DATE):
5382
5652
  _record_pricing_write_refusal(conn, stored)
5383
5653
  _arm_rollup_backfill_pending(conn)
5384
5654
  conn.commit()
@@ -5394,7 +5664,8 @@ def _arm_rollup_backfill_on_pricing_change(conn) -> bool:
5394
5664
 
5395
5665
  def _recompute_conversation_sessions(
5396
5666
  conn, session_ids=None, *, advance_render_revision: bool = True,
5397
- authorize: bool = True,
5667
+ authorize: bool = True, refusal_out: "list[str] | None" = None,
5668
+ target: str = "conversation_sessions",
5398
5669
  ) -> bool:
5399
5670
  """Recompute the ``conversation_sessions`` browse-rail rollup from
5400
5671
  ``conversation_messages``. The caller holds the cache.db.lock flock and owns
@@ -5446,21 +5717,37 @@ def _recompute_conversation_sessions(
5446
5717
  persistent-store guard does not apply to it. The clear is gated for the
5447
5718
  OPPOSITE reason — its unqualified ``cache_meta`` is not shadowed and would
5448
5719
  reach ``main``, so on that connection it is the one statement here that
5449
- does touch a persistent row."""
5720
+ does touch a persistent row.
5721
+
5722
+ ``refusal_out`` is an optional sink for the observation state that refused
5723
+ (#769 S3). The bare False return says a write was declined but not why, and
5724
+ since #728 there are three reasons — only one of which is the pricing-skew
5725
+ case every operator message used to state as a fact. A caller that reports
5726
+ the refusal to a person passes a list and reads the appended state; the
5727
+ read happens exactly once here, so the reported cause is the one that
5728
+ actually decided, not a second read that may disagree with it.
5729
+
5730
+ ``target`` names the table this writes. A rebuild builds its replacement
5731
+ generation into ``conversation_sessions_staging`` and leaves the live table
5732
+ untouched until the publish (#752), so the rollup derivation has to be able
5733
+ to name either. It stays a parameter with the live default rather than two
5734
+ copies of the derivation, because two copies of a GROUP BY that must stay
5735
+ byte-identical to the rail's live aggregate is exactly the drift this
5736
+ function's own docstring warns about. The account scoper's connection-local
5737
+ TEMP table shadows the DEFAULT name, so passing nothing preserves it."""
5450
5738
  ids = None if session_ids is None else [s for s in session_ids if s is not None]
5451
5739
  if ids == []:
5452
5740
  return True
5453
5741
  if authorize:
5454
- try:
5455
- row = conn.execute(
5456
- "SELECT value FROM cache_meta WHERE key=?",
5457
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5458
- ).fetchone()
5459
- except sqlite3.OperationalError:
5460
- row = None
5461
- stored = row[0] if row else None
5462
- if not _pricing_write_authorized(stored):
5463
- _record_pricing_write_refusal(conn, stored)
5742
+ # #728: the degrade-to-None shape this replaces reached the predicate,
5743
+ # but with a value that said "this store recorded nothing" for a SELECT
5744
+ # that had raised. The observation reports the failed read as DEGRADED
5745
+ # instead, which this predicate refuses.
5746
+ obs = _read_pricing_fingerprint_observation(conn)
5747
+ if not may_write_materialized_cost(obs, PRICING_SNAPSHOT_DATE):
5748
+ if refusal_out is not None:
5749
+ refusal_out.append(obs.state)
5750
+ _record_pricing_write_refusal(conn, obs.raw)
5464
5751
  _arm_rollup_backfill_pending(conn)
5465
5752
  # No commit: this helper documents that the CALLER owns it, and
5466
5753
  # every caller that can reach a refusal commits — the pruner inside
@@ -5470,16 +5757,18 @@ def _recompute_conversation_sessions(
5470
5757
  # Do NOT weaken any of those to "the arm helper already committed
5471
5758
  # the same two rows on this tick". That was the argument for the
5472
5759
  # two flag branches, and it holds only while the arm helper's read
5473
- # of the fingerprint and the read below AGREE. They diverge two
5474
- # ways. _arm_rollup_backfill_on_pricing_change returns True and
5475
- # writes nothing when its own SELECT raises OperationalError, and
5476
- # `database is locked` is transient, so the read below can succeed
5477
- # against a newer stored value and refuse. And another process can
5478
- # advance the fingerprint between the two reads — which
5479
- # _import_legacy_conversation_rows made more reachable, because it
5480
- # stamps CONVERSATION_ROLLUP_PRICING_FP_KEY at DB open holding only
5481
- # the shared maintenance lock, never the conversations writer flock.
5482
- # In either case this writes a genuinely new record,
5760
+ # of the fingerprint and the read below AGREE. They still diverge,
5761
+ # though #769 S3 corrects WHY: the pre-#728 reason — the arm helper
5762
+ # returning True and writing nothing when its own SELECT raised —
5763
+ # no longer exists, because a DEGRADED read now refuses there too.
5764
+ # What remains is that another process can advance the fingerprint
5765
+ # between the two reads, which _import_legacy_conversation_rows
5766
+ # made more reachable, because it stamps
5767
+ # CONVERSATION_ROLLUP_PRICING_FP_KEY at DB open holding only the
5768
+ # shared maintenance lock, never the conversations writer flock.
5769
+ # A transient `database is locked` also still splits the two reads,
5770
+ # now in the other direction: the arm helper refuses and this read
5771
+ # can succeed. In every case this writes a genuinely new record,
5483
5772
  # _arm_rollup_backfill_pending short-circuits on the already-set
5484
5773
  # flag, and nothing else commits.
5485
5774
  return False
@@ -5488,15 +5777,15 @@ def _recompute_conversation_sessions(
5488
5777
  if advance_render_revision else 0
5489
5778
  )
5490
5779
  if ids is None:
5491
- conn.execute("DELETE FROM conversation_sessions")
5780
+ conn.execute(f"DELETE FROM {target}")
5492
5781
  conn.execute(
5493
- "INSERT INTO conversation_sessions "
5782
+ f"INSERT INTO {target} "
5494
5783
  "(session_id, msg_count, started_utc, last_activity_utc) "
5495
5784
  + _CONV_SESSIONS_SELECT + " GROUP BY session_id"
5496
5785
  )
5497
- _fill_conversation_sessions_filter_columns(conn, None)
5786
+ _fill_conversation_sessions_filter_columns(conn, None, target=target)
5498
5787
  conn.execute(
5499
- "UPDATE conversation_sessions SET render_revision=?",
5788
+ f"UPDATE {target} SET render_revision=?",
5500
5789
  (render_revision,),
5501
5790
  )
5502
5791
  _clear_converged_pricing_write_refusal(conn, authorize=authorize)
@@ -5505,27 +5794,28 @@ def _recompute_conversation_sessions(
5505
5794
  chunk = ids[i:i + 400]
5506
5795
  placeholders = ",".join("?" for _ in chunk)
5507
5796
  conn.execute(
5508
- f"DELETE FROM conversation_sessions WHERE session_id IN ({placeholders})",
5797
+ f"DELETE FROM {target} WHERE session_id IN ({placeholders})",
5509
5798
  chunk,
5510
5799
  )
5511
5800
  conn.execute(
5512
- "INSERT INTO conversation_sessions "
5801
+ f"INSERT INTO {target} "
5513
5802
  "(session_id, msg_count, started_utc, last_activity_utc) "
5514
5803
  + _CONV_SESSIONS_SELECT
5515
5804
  + f" AND session_id IN ({placeholders}) GROUP BY session_id",
5516
5805
  chunk,
5517
5806
  )
5518
5807
  conn.execute(
5519
- f"UPDATE conversation_sessions SET render_revision=? "
5808
+ f"UPDATE {target} SET render_revision=? "
5520
5809
  f"WHERE session_id IN ({placeholders})",
5521
5810
  (render_revision, *chunk),
5522
5811
  )
5523
- _fill_conversation_sessions_filter_columns(conn, ids)
5812
+ _fill_conversation_sessions_filter_columns(conn, ids, target=target)
5524
5813
  _clear_converged_pricing_write_refusal(conn, authorize=authorize)
5525
5814
  return True
5526
5815
 
5527
5816
 
5528
- def _fill_conversation_sessions_filter_columns(conn, session_ids):
5817
+ def _fill_conversation_sessions_filter_columns(conn, session_ids, *,
5818
+ target="conversation_sessions"):
5529
5819
  """Fill the rollup's browse-FILTER columns (project_label / cost_usd /
5530
5820
  cache_rebuild_count, migration 015) AND the #302 DISPLAYED-enrichment columns
5531
5821
  (git_branch / models_json / title) for the given sessions, or ALL when
@@ -5553,13 +5843,13 @@ def _fill_conversation_sessions_filter_columns(conn, session_ids):
5553
5843
  No-op when any of the columns is absent (a pre-015 / pre-023 cache.db being
5554
5844
  re-derived before _apply_cache_schema adds them), so an early/partial sync
5555
5845
  never raises ``no such column``. The CALLER owns the commit (never commits)."""
5556
- cols = {r[1] for r in conn.execute("PRAGMA table_info(conversation_sessions)")}
5846
+ cols = {r[1] for r in conn.execute(f"PRAGMA table_info({target})")}
5557
5847
  if not {"cache_rebuild_count", "git_branch", "models_json", "title"} <= cols:
5558
5848
  return
5559
5849
  lq = _load_lib("_lib_conversation_query")
5560
5850
  if session_ids is None:
5561
5851
  ids = [r[0] for r in conn.execute(
5562
- "SELECT session_id FROM conversation_sessions")]
5852
+ f"SELECT session_id FROM {target}")]
5563
5853
  else:
5564
5854
  ids = [s for s in session_ids if s is not None]
5565
5855
  if not ids:
@@ -5576,7 +5866,7 @@ def _fill_conversation_sessions_filter_columns(conn, session_ids):
5576
5866
  models_json = json.dumps(m) if m else None
5577
5867
  title = first_titles.get(sid)
5578
5868
  conn.execute(
5579
- "UPDATE conversation_sessions SET project_label=?, cost_usd=?, "
5869
+ f"UPDATE {target} SET project_label=?, cost_usd=?, "
5580
5870
  "cache_rebuild_count=?, git_branch=?, models_json=?, title=? "
5581
5871
  "WHERE session_id=?",
5582
5872
  (proj, round(cost.get(sid, 0.0), 6), rebuilds, branch, models_json,
@@ -7323,7 +7613,9 @@ def sync_codex_cache(
7323
7613
  # mtime_ns is selected into `existing` for diagnostics only —
7324
7614
  # delta detection consults size alone (Codex rollout JSONLs are
7325
7615
  # append-only, so a size change is a sufficient signal and mtime
7326
- # is prone to clock-skew false-positives).
7616
+ # is prone to clock-skew false-positives). `device_id`/`inode` are
7617
+ # NOT diagnostics: they are the only evidence that separates a file
7618
+ # nothing touched from one replaced at the same size (#769 S6).
7327
7619
  if targeted:
7328
7620
  # §5.1: the cursor preload queries codex_session_files for the
7329
7621
  # REQUESTED paths only (the full-sync path loads every row; targeted
@@ -7332,15 +7624,13 @@ def sync_codex_cache(
7332
7624
  conn, [str(item.source_path) for item in files])
7333
7625
  else:
7334
7626
  existing = {
7335
- row[0]: (
7336
- row[1], row[2], row[3], row[4], row[5], row[6], row[7],
7337
- row[8], row[9], row[10], row[11], row[12], row[13],
7338
- )
7627
+ row[0]: tuple(row[1:])
7339
7628
  for row in conn.execute(
7340
7629
  "SELECT path, size_bytes, mtime_ns, last_byte_offset, "
7341
7630
  "last_session_id, last_model, last_total_tokens, source_root_key, "
7342
7631
  "last_native_thread_id, last_root_thread_id, last_parent_thread_id, "
7343
- "last_conversation_key, last_turn_id, ingest_complete "
7632
+ "last_conversation_key, last_turn_id, ingest_complete, "
7633
+ "device_id, inode "
7344
7634
  "FROM codex_session_files"
7345
7635
  )
7346
7636
  }
@@ -7520,12 +7810,29 @@ def sync_codex_cache(
7520
7810
  prev_size, _, prev_offset, prev_sid, prev_model, prev_ttot,
7521
7811
  prev_root_key, prev_native_thread_id, prev_root_thread_id,
7522
7812
  prev_parent_thread_id, prev_conversation_key, prev_turn_id,
7523
- prev_complete,
7813
+ prev_complete, prev_device, prev_inode,
7524
7814
  ) = prev
7525
7815
  prev_total_tokens = (
7526
7816
  int(prev_ttot) if prev_ttot is not None else None
7527
7817
  )
7528
7818
  requalified = prev_root_key != discovered.source_root_key
7819
+ # #769 S6: IDENTITY OUTRANKS SIZE, here as well as in the
7820
+ # frontier's `classify_recent_active_path`. Detecting the
7821
+ # replacement in the planner and then letting the walk skip the
7822
+ # file on `size == prev_size` retains the old offset AND the old
7823
+ # identity, so the planner escalates again on the next tick and
7824
+ # the whole-estate walk the escalation authorises becomes a
7825
+ # per-tick walk that repairs nothing.
7826
+ #
7827
+ # THE INODE DECIDES and the device only corroborates, because
7828
+ # `st_dev` is assigned at mount time: see
7829
+ # `_lib_ingest_frontier.source_identity_replaced`, which owns
7830
+ # this verdict for the planner and for both walks so the three
7831
+ # cannot drift apart. It also owns the two no-evidence
7832
+ # degradations — a NULL column and an unreadable stored value
7833
+ # both fall through to the size comparison below.
7834
+ replaced = _ingest_frontier.source_identity_replaced(
7835
+ prev_device, prev_inode, st.st_dev, st.st_ino)
7529
7836
  # public #5 spec §4. `ingest_complete` is 1 for every row a
7530
7837
  # pre-budget binary wrote and for every file read to its stored
7531
7838
  # target, so this branch is unreachable until a budgeted stop
@@ -7535,9 +7842,13 @@ def sync_codex_cache(
7535
7842
  # skipped on equality, which made the unread suffix permanently
7536
7843
  # invisible on any rollout that never grows again.
7537
7844
  incomplete = prev_complete is not None and not int(prev_complete)
7538
- if targeted and (requalified or size < prev_size):
7539
- # §5.1 preflight-snapshot scoped: a shrink or requalification
7540
- # landing AFTER the preflight is declined HERE, per file —
7845
+ if targeted and (requalified or replaced or size < prev_size):
7846
+ # §5.1 preflight-snapshot scoped: a shrink, a requalification
7847
+ # or a replacement landing AFTER the preflight is declined
7848
+ # HERE, per file — the whole-call preflight above compares
7849
+ # sizes only, and a replacement is what the frontier already
7850
+ # answers with a full walk, so targeted mode reaches this
7851
+ # state only when the escalation lost a race —
7541
7852
  # earlier per-file commits in this call stand, the call still
7542
7853
  # reports dirty (files_failed → not targeted_clean), so the
7543
7854
  # watch advances no cursor and emits nothing, and recovery
@@ -7546,7 +7857,10 @@ def sync_codex_cache(
7546
7857
  # — that whole-cache-affecting escalation is the full sync's.
7547
7858
  stats.files_failed += 1
7548
7859
  continue
7549
- if not requalified and incomplete and size >= prev_offset:
7860
+ if (
7861
+ not requalified and not replaced and incomplete
7862
+ and size >= prev_offset
7863
+ ):
7550
7864
  # Resume the stored scan target. Deliberately NOT a
7551
7865
  # `delta_append`: that flag is what authorizes consulting
7552
7866
  # the live `auth.json` and minting a new account range at
@@ -7568,10 +7882,16 @@ def sync_codex_cache(
7568
7882
  initial_session_id = prev_sid
7569
7883
  initial_model = prev_model
7570
7884
  initial_total_tokens = prev_total_tokens or 0
7571
- elif not requalified and not incomplete and size == prev_size:
7885
+ elif (
7886
+ not requalified and not replaced and not incomplete
7887
+ and size == prev_size
7888
+ ):
7572
7889
  stats.files_skipped_unchanged += 1
7573
7890
  continue
7574
- elif not requalified and not incomplete and size > prev_size:
7891
+ elif (
7892
+ not requalified and not replaced and not incomplete
7893
+ and size > prev_size
7894
+ ):
7575
7895
  start_offset = prev_offset
7576
7896
  delta_append = True
7577
7897
  initial_session_id = prev_sid
@@ -8026,6 +8346,11 @@ def sync_codex_cache(
8026
8346
  file_account_decision=pending_decision,
8027
8347
  anchor_resolver=anchor_resolver,
8028
8348
  ingest_complete=not stopped_short["value"],
8349
+ # The SAME pre-read stat that defined `scan_target`.
8350
+ # Re-statting here would describe a file this pass may
8351
+ # never have read (#769 S6).
8352
+ device_id=st.st_dev,
8353
+ inode=st.st_ino,
8029
8354
  )
8030
8355
  except sqlite3.DatabaseError as exc:
8031
8356
  conn.rollback()
@@ -10006,8 +10331,57 @@ def _acquire_conversation_provider_locks(
10006
10331
  raise
10007
10332
 
10008
10333
 
10334
+ #: #769 S6 / #802 — the process-level no-sync derivation policy.
10335
+ #:
10336
+ #: `--no-sync` freezes ingestion and the snapshot. It must also freeze the two
10337
+ #: POST-DISPATCH conversation derivations `_open_conversations_db_unlocked`
10338
+ #: runs after `_run_pending_migrations` — `_import_legacy_conversation_rows`
10339
+ #: and `_ensure_codex_conversation_contract` — because the second consumes
10340
+ #: `conversation_rebuild_codex_pending` and performs a full retained-event
10341
+ #: rebuild, measured at 135 seconds of startup against comparable launches of
10342
+ #: 15 and 30 seconds.
10343
+ #:
10344
+ #: The policy is PROCESS-level rather than call-site-level, and that is the
10345
+ #: whole point: startup is not the only opener. A read route falls back from
10346
+ #: the read-only opener to the full `open_conversations_db()` on a missing
10347
+ #: store, a lock, or a pending legacy bridge, and live-tail always uses the
10348
+ #: full opener. Either would run both derivations and consume the marker from
10349
+ #: an ordinary browse, so a startup-only flag would freeze nothing.
10350
+ #:
10351
+ #: It never affects the migration dispatcher. Under `--no-sync` no other
10352
+ #: process opens the store write-capable and the read-only reader refuses a
10353
+ #: store behind head rather than migrating from a request thread, so
10354
+ #: suppressing the dispatcher would leave that dashboard permanently degraded —
10355
+ #: the failure `_dashboard_startup_schema_migration` exists to prevent.
10356
+ _CONVERSATION_DERIVATIONS_SUPPRESSED = False
10357
+
10358
+
10359
+ def set_conversation_derivations_suppressed(value: bool) -> None:
10360
+ """Set the process-level policy. Called once, by `--no-sync` startup."""
10361
+ global _CONVERSATION_DERIVATIONS_SUPPRESSED
10362
+ _CONVERSATION_DERIVATIONS_SUPPRESSED = bool(value)
10363
+
10364
+
10365
+ def conversation_derivations_suppressed() -> bool:
10366
+ """Whether this process suppresses the two post-dispatch derivations."""
10367
+ return _CONVERSATION_DERIVATIONS_SUPPRESSED
10368
+
10369
+
10370
+ def _conversation_derivations_enabled(run_derivations: "bool | None") -> bool:
10371
+ """Resolve an explicit request against the process policy.
10372
+
10373
+ ``None`` means "follow the process policy", which is what every existing
10374
+ caller passes by omission, and the policy defaults to running them — so
10375
+ every existing caller is byte-unchanged.
10376
+ """
10377
+ if run_derivations is not None:
10378
+ return bool(run_derivations)
10379
+ return not conversation_derivations_suppressed()
10380
+
10381
+
10009
10382
  def _conversations_open_guarded(
10010
10383
  *, attach_cache: bool, allow_recovery_state: bool = False,
10384
+ run_derivations: "bool | None" = None,
10011
10385
  ) -> sqlite3.Connection:
10012
10386
  """Open conversations.db while excluding confirmed family replacement."""
10013
10387
  path = pathlib.Path(_cctally_core.CONVERSATIONS_DB_PATH)
@@ -10101,6 +10475,7 @@ def _conversations_open_guarded(
10101
10475
  try:
10102
10476
  conn = _open_conversations_db_unlocked(
10103
10477
  attach_cache=attach_cache,
10478
+ run_derivations=run_derivations,
10104
10479
  )
10105
10480
  if marker.exists() or pending.exists():
10106
10481
  conn.close()
@@ -10138,7 +10513,7 @@ def _harden_conversation_sidecars() -> None:
10138
10513
 
10139
10514
 
10140
10515
  def _open_conversations_db_unlocked(
10141
- *, attach_cache: bool = True,
10516
+ *, attach_cache: bool = True, run_derivations: "bool | None" = None,
10142
10517
  ) -> sqlite3.Connection:
10143
10518
  """Open the independent transcript/search store (#320).
10144
10519
 
@@ -10147,6 +10522,20 @@ def _open_conversations_db_unlocked(
10147
10522
  Codex-thread metadata. Core cache callers never take the inverse
10148
10523
  dependency, so a missing or locked transcript store cannot block quota or
10149
10524
  accounting refreshes.
10525
+
10526
+ ``run_derivations`` (#769 S6 / #802) selects whether the two POST-DISPATCH
10527
+ derivations below run: ``_import_legacy_conversation_rows`` and
10528
+ ``_ensure_codex_conversation_contract``. ``None`` follows the process-level
10529
+ policy, which defaults to running them, so every existing caller is
10530
+ byte-unchanged. ``--no-sync`` sets that policy to suppress them.
10531
+
10532
+ What ``--no-sync`` freezes and does not freeze is therefore exact: it
10533
+ freezes ingestion, the snapshot and these two derivations; it does NOT
10534
+ freeze the schema apply or the migration dispatcher, which run
10535
+ unconditionally so the store still reaches head. The accepted consequence
10536
+ is that a ``--no-sync`` dashboard over a store owing the Codex contract
10537
+ rebuild serves ``normalization_pending`` for Codex conversation reads until
10538
+ a mode allowed to do work runs.
10150
10539
  """
10151
10540
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
10152
10541
  try:
@@ -10220,14 +10609,283 @@ def _open_conversations_db_unlocked(
10220
10609
  cache.close()
10221
10610
  cache_uri = _cctally_core.CACHE_DB_PATH.resolve().as_uri() + "?mode=ro"
10222
10611
  conn.execute("ATTACH DATABASE ? AS cache_db", (cache_uri,))
10223
- _import_legacy_conversation_rows(conn)
10224
- _ensure_codex_conversation_contract(conn)
10612
+ if _conversation_derivations_enabled(run_derivations):
10613
+ _import_legacy_conversation_rows(conn)
10614
+ _ensure_codex_conversation_contract(conn)
10225
10615
  _harden_conversation_sidecars()
10226
10616
  return conn
10227
10617
 
10228
10618
 
10229
- def open_conversations_db(*, attach_cache: bool = True) -> sqlite3.Connection:
10230
- return _conversations_open_guarded(attach_cache=attach_cache)
10619
+ def open_conversations_db(
10620
+ *, attach_cache: bool = True, run_derivations: "bool | None" = None,
10621
+ ) -> sqlite3.Connection:
10622
+ return _conversations_open_guarded(
10623
+ attach_cache=attach_cache, run_derivations=run_derivations,
10624
+ )
10625
+
10626
+
10627
+ # --------------------------------------------------------------------------
10628
+ # #780 — read-only reader admission
10629
+ # --------------------------------------------------------------------------
10630
+
10631
+ #: Default busy timeout for a read route, in seconds. Route-bounded rather
10632
+ #: than the store policy's 15s: a reader that cannot be admitted quickly must
10633
+ #: fail soft and let the route render a degraded envelope, never hold a request
10634
+ #: thread for fifteen seconds behind a rebuild.
10635
+ CONVERSATION_READONLY_TIMEOUT_S = 0.25
10636
+
10637
+ #: Called with no arguments when a reader finds the store BEHIND head. The
10638
+ #: opener never migrates from a request thread, so this is how it asks the
10639
+ #: process's writer to. Left None outside the dashboard.
10640
+ SCHEMA_WAKE_HOOK = None
10641
+
10642
+
10643
+ class ConversationReaderUnavailable(sqlite3.DatabaseError):
10644
+ """A read route could not be admitted. Carries a typed ``reason``.
10645
+
10646
+ A subclass of ``sqlite3.DatabaseError`` so the existing route boundary,
10647
+ which already catches ``DatabaseError`` and ``OSError``, keeps catching it
10648
+ rather than letting a new exception class escape as a 500.
10649
+ """
10650
+
10651
+ reason = "unavailable"
10652
+
10653
+
10654
+ class MaintenanceInProgress(ConversationReaderUnavailable):
10655
+ """Maintenance holds the store; the reader declined to queue behind it."""
10656
+
10657
+ reason = "maintenance"
10658
+
10659
+
10660
+ class SchemaBehind(ConversationReaderUnavailable):
10661
+ """A store is behind head. Soft: a writer can still advance it."""
10662
+
10663
+ reason = "schema_behind"
10664
+
10665
+
10666
+ class SchemaAhead(ConversationReaderUnavailable):
10667
+ """A store is ahead of head. Closed: no recovery is attempted here."""
10668
+
10669
+ reason = "schema_ahead"
10670
+
10671
+
10672
+ class LegacyBridgePending(ConversationReaderUnavailable):
10673
+ """The legacy transcript bridge is owed and this process may not run it.
10674
+
10675
+ `_import_legacy_conversation_rows` is a WRITER, and under `dashboard
10676
+ --no-sync` the process policy suppresses it (#802), so a store whose rows
10677
+ are still in `cache.db` after an interrupted migration 028 stays that way
10678
+ for the life of the process. Serving that connection would render an empty
10679
+ conversation list and an empty browse project filter with nothing said,
10680
+ which is the silent failure this class exists to replace. Spec §9 names the
10681
+ alternative: a route that cannot inherit the derivation policy returns a
10682
+ typed degraded response instead.
10683
+
10684
+ Named for the state, like `normalization_pending` — the sibling answer the
10685
+ Codex contract rebuild already produces under the same suppression.
10686
+ """
10687
+
10688
+ reason = "legacy_bridge_pending"
10689
+
10690
+
10691
+ def probe_conversations_maintenance_free() -> None:
10692
+ """Raise `MaintenanceInProgress` if maintenance holds the flock right now.
10693
+
10694
+ The non-blocking admission `open_conversations_db_readonly` performs,
10695
+ factored out for the three request-path sites that must reach the BLOCKING
10696
+ full opener but must not queue behind a rebuild on the way (#780).
10697
+
10698
+ Those sites are the live-tail preflight and `_open_conversation_reader`'s
10699
+ two fallbacks. Each of them hands the open to `open_conversations_db`,
10700
+ whose maintenance admission is a plain `LOCK_SH` with no timeout, so a
10701
+ request arriving during a rebuild held a server thread for the rebuild's
10702
+ whole duration — 716.9 s in the measured run. §4a names the live-tail
10703
+ preflight among the routes that must fail soft, and the other two reach the
10704
+ same opener by the same route.
10705
+
10706
+ An absent lock file is NOT a refusal, for the same reason the read-only
10707
+ opener gives: it means no maintenance has ever claimed it, which is a
10708
+ first-run state rather than a busy one, and the full opener is what creates
10709
+ it. An unopenable lock file is not a refusal either — an undecidable probe
10710
+ must not be what takes a read route down; the full opener behind it will
10711
+ fail loudly on its own if the state is genuinely bad.
10712
+ """
10713
+ maintenance_path = pathlib.Path(
10714
+ _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH
10715
+ )
10716
+ if not maintenance_path.is_file():
10717
+ return
10718
+ try:
10719
+ probe_fh = open(maintenance_path, "r")
10720
+ except OSError:
10721
+ return
10722
+ try:
10723
+ try:
10724
+ fcntl.flock(probe_fh, fcntl.LOCK_SH | fcntl.LOCK_NB)
10725
+ except BlockingIOError as exc:
10726
+ raise MaintenanceInProgress(
10727
+ "conversations.db maintenance is in progress") from exc
10728
+ try:
10729
+ fcntl.flock(probe_fh, fcntl.LOCK_UN)
10730
+ except OSError:
10731
+ pass
10732
+ finally:
10733
+ probe_fh.close()
10734
+
10735
+
10736
+ def open_conversations_db_readonly(
10737
+ *, attach_cache: bool = True, timeout: "float | None" = None,
10738
+ ) -> sqlite3.Connection:
10739
+ """A conversations.db connection for a READ route (#780).
10740
+
10741
+ The route this replaces opened the full mutation-capable
10742
+ ``_conversations_open_guarded``, which executes, in order: a chmod on the
10743
+ data directory, ``PRAGMA auto_vacuum=INCREMENTAL``, ``PRAGMA
10744
+ journal_mode=WAL``, a schema-currency check that can apply the whole
10745
+ schema, a commit, the migration dispatcher, a write-capable
10746
+ ``open_cache_db()``, ``_import_legacy_conversation_rows``,
10747
+ ``_ensure_codex_conversation_contract``, a chmod on the database, and
10748
+ sidecar hardening. Every one of those is a write, so every browse became a
10749
+ writer that could lose the SQLite write lock to maintenance.
10750
+
10751
+ This opener performs NONE of them. It does not call ``apply_policy``: that
10752
+ helper emits ``PRAGMA auto_vacuum`` and ``PRAGMA journal_mode``, both
10753
+ write-capable, so routing through it would defeat the whole point. The
10754
+ busy timeout is supplied to ``sqlite3.connect`` instead of through a
10755
+ PRAGMA, and the conversations policy's row factory is the driver default.
10756
+
10757
+ ``PRAGMA query_only`` is deliberately NOT set. ``mode=ro`` already refuses
10758
+ every write to ``main``, and it still permits TEMP tables and TEMP views,
10759
+ which is what ``scope_conversations_db_to_account`` needs; ``query_only``
10760
+ breaks those TEMP writes and would take account scoping down with it
10761
+ (verified empirically on SQLite 3.53.4).
10762
+
10763
+ Admission is non-blocking against the maintenance flock, and the marker
10764
+ files are re-checked after the connection opens, so a maintenance pass that
10765
+ starts during the open is not served from a store it is about to replace.
10766
+
10767
+ Both stores are then gated by the schema-qualified tri-state probe.
10768
+ ``behind`` raises ``SchemaBehind`` after asking the process's writer to
10769
+ advance the store; ``ahead`` raises ``SchemaAhead`` and attempts no
10770
+ recovery, matching the existing version-ahead posture. The opener never
10771
+ migrates from a request thread.
10772
+ """
10773
+ path = pathlib.Path(_cctally_core.CONVERSATIONS_DB_PATH)
10774
+ marker = _cctally_db_sib._repair_marker_path(path)
10775
+ pending = _cctally_db_sib._quarantine_pending_path(path)
10776
+ recovery = _conversation_recovery_state_path()
10777
+ maintenance_path = pathlib.Path(
10778
+ _cctally_core.CONVERSATIONS_LOCK_MAINTENANCE_PATH
10779
+ )
10780
+ busy = CONVERSATION_READONLY_TIMEOUT_S if timeout is None else timeout
10781
+ if not path.is_file():
10782
+ raise ConversationReaderUnavailable("transcript store is not present")
10783
+ if not maintenance_path.is_file():
10784
+ # NOT a maintenance refusal: an absent lock file means no maintenance
10785
+ # has ever claimed it, which is a first-run state, not a busy one.
10786
+ # `_conversations_open_guarded` creates it, so this hands the open back
10787
+ # to the full opener the same way an absent store does. Refusing here
10788
+ # instead left a fresh install permanently degraded, which is the
10789
+ # opposite of the condition the file's absence describes.
10790
+ raise ConversationReaderUnavailable(
10791
+ "the transcript maintenance lock has not been established yet")
10792
+ cache_path = pathlib.Path(_cctally_core.CACHE_DB_PATH)
10793
+ if attach_cache and not cache_path.is_file():
10794
+ raise ConversationReaderUnavailable("accounting store is not present")
10795
+
10796
+ conn: sqlite3.Connection | None = None
10797
+ maintenance_fh = open(maintenance_path, "r")
10798
+ try:
10799
+ try:
10800
+ fcntl.flock(maintenance_fh, fcntl.LOCK_SH | fcntl.LOCK_NB)
10801
+ except BlockingIOError as exc:
10802
+ raise MaintenanceInProgress(
10803
+ "conversations.db maintenance is in progress") from exc
10804
+ try:
10805
+ if marker.exists() or pending.exists() or recovery.exists():
10806
+ raise MaintenanceInProgress(
10807
+ "conversations.db maintenance is in progress")
10808
+ conn = sqlite3.connect(
10809
+ f"{path.resolve().as_uri()}?mode=ro",
10810
+ uri=True,
10811
+ timeout=max(busy, 0.0),
10812
+ )
10813
+ if _cctally_store._TRACE_HOOK is not None:
10814
+ conn.set_trace_callback(_cctally_store._TRACE_HOOK)
10815
+ conn.row_factory = None
10816
+ if marker.exists() or pending.exists() or recovery.exists():
10817
+ raise MaintenanceInProgress(
10818
+ "conversations.db maintenance started during open")
10819
+ if attach_cache:
10820
+ conn.execute(
10821
+ "ATTACH DATABASE ? AS cache_db",
10822
+ (f"{cache_path.resolve().as_uri()}?mode=ro",),
10823
+ )
10824
+ finally:
10825
+ try:
10826
+ fcntl.flock(maintenance_fh, fcntl.LOCK_UN)
10827
+ except OSError:
10828
+ pass
10829
+ _gate_reader_schema(conn, attach_cache=attach_cache)
10830
+ return conn
10831
+ except BaseException:
10832
+ if conn is not None:
10833
+ try:
10834
+ conn.close()
10835
+ except sqlite3.Error:
10836
+ pass
10837
+ raise
10838
+ finally:
10839
+ maintenance_fh.close()
10840
+
10841
+
10842
+ def conversation_legacy_bridge_pending(conn: sqlite3.Connection) -> bool:
10843
+ """Whether `_import_legacy_conversation_rows` still has work to do (#780).
10844
+
10845
+ The bridge is a WRITER, so the read-only opener cannot run it, and a route
10846
+ that quietly skipped it would serve an empty transcript surface on a store
10847
+ whose rows are still sitting in `cache.db` after an interrupted migration
10848
+ 028. Its own condition — the main table empty while the attached cache
10849
+ table is not — is a pure read, so a reader can detect the state even though
10850
+ it must not fix it, and hand the open back to the full opener. The route
10851
+ checks this again AFTER that hand-back: under `dashboard --no-sync` the
10852
+ bridge is suppressed, so the full opener returns without clearing it and
10853
+ the route raises `LegacyBridgePending` instead of serving (#802).
10854
+
10855
+ Reports False rather than raising on any error: an undecidable answer must
10856
+ not take a read route down.
10857
+ """
10858
+ for table in _LEGACY_BRIDGE_TABLES:
10859
+ try:
10860
+ if conn.execute(f"SELECT 1 FROM main.{table} LIMIT 1").fetchone():
10861
+ continue
10862
+ if conn.execute(
10863
+ f"SELECT 1 FROM cache_db.{table} LIMIT 1"
10864
+ ).fetchone():
10865
+ return True
10866
+ except sqlite3.Error:
10867
+ continue
10868
+ return False
10869
+
10870
+
10871
+ def _gate_reader_schema(conn: sqlite3.Connection, *, attach_cache: bool) -> None:
10872
+ """Refuse a reader whose stores are not at head, by direction (#780)."""
10873
+ probes = [("conversations", "main")]
10874
+ if attach_cache:
10875
+ probes.append(("cache", "cache_db"))
10876
+ for store_name, schema in probes:
10877
+ state = _cctally_store.schema_state(conn, store_name, schema=schema)
10878
+ if state == "current":
10879
+ continue
10880
+ if state == "behind":
10881
+ hook = SCHEMA_WAKE_HOOK
10882
+ if hook is not None:
10883
+ try:
10884
+ hook()
10885
+ except Exception as exc: # noqa: BLE001
10886
+ eprint(f"[conversations] schema wake-up failed ({exc})")
10887
+ raise SchemaBehind(f"{store_name} store is behind head")
10888
+ raise SchemaAhead(f"{store_name} store is ahead of head")
10231
10889
 
10232
10890
 
10233
10891
  def scope_conversations_db_to_account(
@@ -10901,8 +11559,8 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10901
11559
  which table that was in ``CONVERSATION_ROLLUP_PRICING_FP_KEY``, so the
10902
11560
  provenance is written down rather than merely implied.
10903
11561
 
10904
- Deriving is required, not merely tidier: this runs at DB OPEN, and
10905
- ``dashboard --no-sync`` never runs a sync, so merely arming
11562
+ Deriving is required, not merely tidier, in every mode allowed to do work:
11563
+ this runs at DB OPEN, and merely arming
10906
11564
  ``conversation_sessions_backfill_pending`` left the rollup EMPTY and
10907
11565
  non-authoritative for the life of that process. The rail itself survives
10908
11566
  that (the flag routes it to live aggregation) but
@@ -10910,6 +11568,16 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10910
11568
  with no authoritative gate, so the browse filter's project list went empty;
10911
11569
  and every rail read fell to the live branch, which is not the branch the
10912
11570
  materialized-cost contract is about.
11571
+
11572
+ ``dashboard --no-sync`` is the one mode NOT allowed to do work, and since
11573
+ #802 the process policy suppresses this bridge there along with the Codex
11574
+ contract rebuild. That does not reinstate the empty-rollup failure above,
11575
+ because the read routes stop serving over an owed bridge: the reader raises
11576
+ ``LegacyBridgePending`` and the route answers the typed degraded envelope
11577
+ naming ``legacy_bridge_pending``, the sibling of the
11578
+ ``normalization_pending`` the Codex case already returns under the same
11579
+ suppression. The state is disclosed rather than rendered as an empty
11580
+ surface, and it clears the moment a mode allowed to do work opens the store.
10913
11581
  """
10914
11582
  changed = False
10915
11583
  for table in _LEGACY_BRIDGE_TABLES:
@@ -11163,6 +11831,393 @@ def _prepare_claude_conversation_maintenance(
11163
11831
  return reingest
11164
11832
 
11165
11833
 
11834
+ # === #752 / #777: title generations, source incarnations, account stamps ====
11835
+
11836
+
11837
+ #: The digest of zero bytes. A rebuild resets each source file's committed
11838
+ #: prefix to this rather than dropping the row, so the replay starts at byte
11839
+ #: zero WITHOUT minting a new incarnation — which is what lets the durable
11840
+ #: stamps written before the rebuild still be found after it.
11841
+ _EMPTY_PREFIX_SHA256 = (
11842
+ "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
11843
+ )
11844
+
11845
+ #: The FLOOR of the rebuild free-space requirement, not the requirement.
11846
+ #:
11847
+ #: Tranche 2 sized a fixed 64 MiB against the staging tables alone, which hold
11848
+ #: only titles and one row per session — 65,536 B on the store observed during
11849
+ #: the #752 incident. Tranche 3 then measured what a whole rebuild costs the
11850
+ #: file: 9,310,744,576 B to 15,459,213,312 B, a growth of 6,148,468,736 B, of
11851
+ #: which 93% is freelist that `PRAGMA incremental_vacuum` returns afterwards.
11852
+ #: The clear does not let the replay reuse the pages it freed within the same
11853
+ #: rebuild, so the peak file size is roughly the live corpus written twice.
11854
+ #:
11855
+ #: A fixed constant therefore admits a rebuild that then runs out of space
11856
+ #: partway — precisely the state the preflight exists to prevent. The
11857
+ #: requirement is read off the store instead, by
11858
+ #: `_conversation_rebuild_free_bytes_required`, and this constant only stops a
11859
+ #: tiny or empty store from demanding nothing at all.
11860
+ _CONVERSATION_STAGING_FREE_BYTES_FLOOR = 64 * 1024 * 1024
11861
+
11862
+ #: Retained under its Tranche 2 name for the module's re-export surface. New
11863
+ #: code reads the measured requirement, never this.
11864
+ _CONVERSATION_STAGING_FREE_BYTES_REQUIRED = _CONVERSATION_STAGING_FREE_BYTES_FLOOR
11865
+
11866
+
11867
+ class _SourceIncarnation(NamedTuple):
11868
+ """One append-continuous life of a source file (#777).
11869
+
11870
+ ``mutable_digest`` is a LIVE, MUTABLE ``hashlib`` object, not a value, and
11871
+ it is named that way because the distinction is load-bearing. It arrives
11872
+ already positioned at ``start_offset``, and the ingester CONSUMES it by
11873
+ updating it in place with the bytes that pass reads, so it may be updated
11874
+ exactly once and its ``hexdigest()`` before that update is not the same
11875
+ number as after. Returning a hex string instead would force a full rehash
11876
+ of the whole file on every sync rather than only the committed prefix,
11877
+ which is why it is an object at all.
11878
+ """
11879
+ incarnation_id: str
11880
+ is_new: bool
11881
+ start_offset: int
11882
+ mutable_digest: Any
11883
+
11884
+
11885
+ def _path_is_genuinely_absent(path_str: str) -> bool:
11886
+ """Whether ``path_str`` is really gone, as opposed to merely unwalked.
11887
+
11888
+ Only ``ENOENT`` and ``ENOTDIR`` answer yes. Every other ``OSError`` —
11889
+ ``EPERM`` on a directory the walk could not enter, ``EIO`` on failing
11890
+ media, ``ENOTCONN`` on a stale network mount — leaves the question
11891
+ undecided, and an undecided answer must retain the row, because the
11892
+ consequence of a wrong "absent" is a permanent loss of attribution while
11893
+ the consequence of a wrong "present" is one retained row.
11894
+
11895
+ ``pathlib.Path.exists()`` cannot serve here: it reports False for every
11896
+ ``OSError``, which is exactly the conflation this function exists to
11897
+ avoid.
11898
+
11899
+ RESIDUAL, stated rather than left to be discovered. ``ENOENT`` is also what
11900
+ an UNMOUNTED volume produces for a path beneath a mount point whose
11901
+ directory still exists, so this reports such a path as genuinely absent
11902
+ when it is only unreachable. The walk-liveness guard upstream covers the
11903
+ common shape of that — a root whose whole tree went away is not walked, so
11904
+ its rows are never offered here — but a volume that unmounts between the
11905
+ walk and this check is not covered, and the consequence is the attribution
11906
+ loss the paragraph above describes. Distinguishing it needs a mount-table
11907
+ read, which this leaf deliberately does not do.
11908
+ """
11909
+ try:
11910
+ os.lstat(path_str)
11911
+ except (FileNotFoundError, NotADirectoryError):
11912
+ return True
11913
+ except OSError:
11914
+ return False
11915
+ return False
11916
+
11917
+
11918
+ #: `_conversation_staging_free_bytes` returns this when it could not measure.
11919
+ #: A SENTINEL rather than a number, because every number is wrong here: the
11920
+ #: floor reads as "enough" only while the requirement is also the floor, and
11921
+ #: the requirement is now the store's live size.
11922
+ CONVERSATION_FREE_SPACE_UNKNOWN = None
11923
+
11924
+
11925
+ def _conversation_staging_free_bytes() -> "int | None":
11926
+ """Free bytes on the volume holding ``conversations.db``.
11927
+
11928
+ A named seam so the preflight can be driven to refuse in a test without
11929
+ filling a real disk.
11930
+
11931
+ Returns ``CONVERSATION_FREE_SPACE_UNKNOWN`` when the volume cannot be
11932
+ measured, and the caller then SKIPS the check. This used to return the
11933
+ staging floor with a comment saying that refusing every rebuild because
11934
+ `statvfs` failed would be the worse failure — true while the comparison was
11935
+ against that same floor, and false the moment the requirement became
11936
+ `_conversation_rebuild_free_bytes_required`, which is gigabytes on any real
11937
+ store. The degraded reading was then always below the requirement, so the
11938
+ documented fail-open deferred every rebuild instead of admitting it.
11939
+ """
11940
+ try:
11941
+ usage = shutil.disk_usage(_cctally_core.CONVERSATIONS_DB_PATH.parent)
11942
+ except OSError:
11943
+ return CONVERSATION_FREE_SPACE_UNKNOWN
11944
+ return int(usage.free)
11945
+
11946
+
11947
+ def _conversation_rebuild_free_bytes_required(conn) -> int:
11948
+ """Free bytes this particular store's rebuild needs (#752, #780).
11949
+
11950
+ The live corpus is `(page_count - freelist_count) * page_size`. A rebuild
11951
+ clears and replays every message row without reusing the pages the clear
11952
+ freed inside the same pass — measured on a copy-on-write clone of the
11953
+ production store, where the file grew by 6.15 GB and 93% of the growth was
11954
+ freelist — so the live size is what the replay can add before any of it
11955
+ comes back. Demand that much, and never less than the staging floor.
11956
+
11957
+ An unreadable pragma degrades to the floor rather than raising: a preflight
11958
+ that cannot measure must not be the thing that stops a rebuild.
11959
+ """
11960
+ try:
11961
+ page_size = int(conn.execute("PRAGMA page_size").fetchone()[0])
11962
+ page_count = int(conn.execute("PRAGMA page_count").fetchone()[0])
11963
+ freelist = int(conn.execute("PRAGMA freelist_count").fetchone()[0])
11964
+ except (sqlite3.Error, TypeError, ValueError, IndexError):
11965
+ return _CONVERSATION_STAGING_FREE_BYTES_FLOOR
11966
+ live_bytes = max(0, page_count - freelist) * max(0, page_size)
11967
+ return max(_CONVERSATION_STAGING_FREE_BYTES_FLOOR, live_bytes)
11968
+
11969
+
11970
+ def _resolve_source_incarnation(conn, path_str, fh, st, prev_row, *,
11971
+ force_replay: bool = False):
11972
+ """Decide whether this file is still the file the cursor describes (#777).
11973
+
11974
+ Continuity requires ALL of: the same stored INODE; a current size at least
11975
+ the committed offset; and a matching guard digest over the whole committed
11976
+ prefix. Any of an inode change, a shrink, a size-preserving rewrite, or a
11977
+ digest mismatch mints a new incarnation and forces byte-zero treatment.
11978
+
11979
+ The device is stored as corroborating evidence and as a diagnostic, and it
11980
+ never decides (#814). `st_dev` is assigned when a volume is mounted rather
11981
+ than when a file is created, so an ordinary remount renumbers every stored
11982
+ `device_id` at once with no file changed; deciding on it minted a fresh
11983
+ incarnation whose high-water is zero, and `_resolve_record_account` then
11984
+ re-attributed that transcript's whole history to whichever account was
11985
+ active at that moment. The verdict is delegated to
11986
+ `_lib_ingest_frontier.source_identity_replaced`, which both Codex walks
11987
+ already call, so one rule serves every provider.
11988
+
11989
+ THE RESIDUAL that creates: a file on a DIFFERENT device presenting the same
11990
+ numeric inode, at least as long as the cursor, whose committed prefix
11991
+ hashes identically, is no longer distinguishable as a new physical
11992
+ incarnation. That is safe because it moves in the safe direction — it
11993
+ PRESERVES continuity where a boundary arguably existed, so no bump occurs,
11994
+ no high-water resets and no re-attribution happens. Resuming from the
11995
+ cursor is content-correct, because the committed records and their
11996
+ digest-bound stamps are byte-identical by construction, and any suffix is
11997
+ processed as new input under the existing rules. What is lost is a physical
11998
+ incarnation boundary, not data and not attribution.
11999
+
12000
+ This is strictly stronger than ``_conversation_target_risk``, which decides
12001
+ the size-preserving case on mtime and is therefore defeated by a ``touch``.
12002
+ A digest over the committed prefix cannot be restored by resetting metadata.
12003
+
12004
+ Reads through the caller's ALREADY-OPEN descriptor. The caller takes an
12005
+ ``fstat`` before and after and treats any change across that pair as a new
12006
+ incarnation, which closes the time-of-check/time-of-use gap a
12007
+ stat-then-open sequence leaves open.
12008
+
12009
+ ``force_replay`` is set by a rebuild, and it separates two questions the
12010
+ stored cursor would otherwise conflate: WHERE to start reading, and WHETHER
12011
+ this is still the same file. A rebuild replays from byte zero regardless,
12012
+ but it must still verify continuity against the cursor the previous life
12013
+ committed — otherwise a file rewritten in place would keep its incarnation
12014
+ across the rebuild and inherit stamps written for bytes that no longer
12015
+ exist. The returned ``digest`` is empty in that case, because the caller is
12016
+ about to hash the whole file from zero.
12017
+ """
12018
+ stored_incarnation = prev_row[3] if prev_row else None
12019
+ stored_device = prev_row[4] if prev_row else None
12020
+ stored_inode = prev_row[5] if prev_row else None
12021
+ stored_prefix = prev_row[6] if prev_row else None
12022
+ committed = int(prev_row[2]) if prev_row else 0
12023
+ continuous = (
12024
+ stored_incarnation is not None
12025
+ # THE INODE DECIDES; THE DEVICE ONLY CORROBORATES (#814). `st_dev` is
12026
+ # assigned when a volume is MOUNTED, not when a file is created, so an
12027
+ # ordinary remount renumbers every stored `device_id` at once with no
12028
+ # file changed. Deciding replacement on it mints a fresh incarnation
12029
+ # whose high-water is zero, and `_resolve_record_account` then gives
12030
+ # every already-attributed record to whichever account is active now.
12031
+ # `#769 S6` removed this from both Codex walks; this was the last
12032
+ # provider source-cursor site where the device still decided.
12033
+ and not _ingest_frontier.source_identity_replaced(
12034
+ stored_device, stored_inode, st.st_dev, st.st_ino,
12035
+ )
12036
+ and st.st_size >= committed
12037
+ )
12038
+ if continuous:
12039
+ digest = _lib_conversation.prefix_digest(fh, committed)
12040
+ if digest.hexdigest() == (stored_prefix or _EMPTY_PREFIX_SHA256):
12041
+ if force_replay:
12042
+ return _SourceIncarnation(
12043
+ stored_incarnation, False, 0,
12044
+ _lib_conversation.prefix_digest(fh, 0),
12045
+ )
12046
+ return _SourceIncarnation(
12047
+ stored_incarnation, False, committed, digest)
12048
+ return _SourceIncarnation(
12049
+ _lib_conversation.new_source_incarnation_id(), True, 0,
12050
+ _lib_conversation.prefix_digest(fh, 0),
12051
+ )
12052
+
12053
+
12054
+ def _stat_pair_broken(before, after) -> bool:
12055
+ """Whether the file's IDENTITY broke under the open descriptor (#777).
12056
+
12057
+ Classified by what changed across one open handle, not by whether anything
12058
+ changed (spec §2, corrected after the Tranche 2 review). The provider
12059
+ appends to an active transcript continuously, so reading any change as a
12060
+ break made an ordinary append raise ``files_failed`` — which withholds the
12061
+ conversation frontier certificate, leaves ``conversation_rebuild_claude_pending``
12062
+ set so the whole rebuild runs again, and discards the records the pass had
12063
+ already read.
12064
+
12065
+ * a different device or inode is a different file;
12066
+ * a SMALLER size is a truncation, or a replacement written in place;
12067
+ * the SAME size with a changed mtime is a size-preserving rewrite — the
12068
+ case ``_conversation_target_risk`` classifies as ``source_replaced``;
12069
+ * a LARGER size is an ordinary append, and a continuation. An mtime
12070
+ change alongside it is that same append and is not read separately.
12071
+
12072
+ An append is safe to continue from because the bytes the reader validated
12073
+ did not move: the guard digest covers exactly the range this pass read, the
12074
+ partial-tail rewind in ``_iter_sync_entries`` already ends the read at a
12075
+ record boundary, and the next sync resumes from the cursor this pass
12076
+ commits.
12077
+
12078
+ What the pair CANNOT see is a ``rename``. The descriptor keeps the old
12079
+ inode, so ``fstat`` reports the file the reader actually read. That is the
12080
+ right outcome rather than a hole: those bytes are self-consistent and are
12081
+ committed under the identity that really held them, and the next sync opens
12082
+ the new directory entry, sees a different inode and mints a fresh
12083
+ incarnation.
12084
+ """
12085
+ if before.st_dev != after.st_dev or before.st_ino != after.st_ino:
12086
+ return True
12087
+ if after.st_size < before.st_size:
12088
+ return True
12089
+ return (after.st_size == before.st_size
12090
+ and after.st_mtime_ns != before.st_mtime_ns)
12091
+
12092
+
12093
+ def _load_account_stamps(conn, path_str):
12094
+ """Every durable stamp and classified gap for one source path (#777)."""
12095
+ stamps = {
12096
+ (row[0], int(row[1]), row[2]): row[3] for row in conn.execute(
12097
+ "SELECT source_incarnation_id,byte_offset,record_sha256,account_key "
12098
+ "FROM claude_conversation_account_stamps WHERE canonical_source_path=?",
12099
+ (path_str,),
12100
+ )
12101
+ }
12102
+ gaps = {
12103
+ int(row[0]): row[1] for row in conn.execute(
12104
+ "SELECT byte_offset,account_key "
12105
+ "FROM claude_conversation_account_stamp_gaps WHERE source_path=?",
12106
+ (path_str,),
12107
+ )
12108
+ }
12109
+ return stamps, gaps
12110
+
12111
+
12112
+ def _stamp_coverage_complete(conn) -> bool:
12113
+ """Whether migration 009's stamp backfill has finished (#777).
12114
+
12115
+ Read defensively: a store whose ``cache_meta`` cannot be read reports
12116
+ INCOMPLETE, so the rebuild defers rather than replaying under a rule whose
12117
+ precondition it could not verify.
12118
+ """
12119
+ try:
12120
+ return conn.execute(
12121
+ "SELECT 1 FROM cache_meta WHERE key=?",
12122
+ ("claude_account_stamp_coverage_complete",),
12123
+ ).fetchone() is not None
12124
+ except sqlite3.OperationalError:
12125
+ return False
12126
+
12127
+
12128
+ def _resolve_record_account(stamps, gaps, incarnation_id, offset, digest,
12129
+ high_water, active_account_key):
12130
+ """Which account one replayed record belongs to (#777). Pure.
12131
+
12132
+ Returns ``(account_key, needs_fresh_stamp)``.
12133
+
12134
+ THREE cases, and the separation between the last two is what the high-water
12135
+ map exists to provide. Without it the miss rule and the new-bytes rule
12136
+ contradict each other: a missing stamp would have to mean both "historical
12137
+ and unknown" and "arriving now".
12138
+
12139
+ 1. A stamp under THIS incarnation for this offset and this digest is the
12140
+ recorded observation, and it wins. The digest is part of the key, so a
12141
+ record whose bytes changed at the same offset cannot inherit it, and the
12142
+ incarnation is part of the key, so a path reused by a different file
12143
+ cannot either.
12144
+ 2. Below the published high-water mark with no stamp, a classified GAP row
12145
+ still carries the attribution the pre-rebuild message row held — the
12146
+ same evidence a stamp holds, minus a digest for bytes that no longer
12147
+ exist. With neither, the record is genuinely historical and unknown, so
12148
+ it takes NULL: writing the currently active account there is precisely
12149
+ the silent re-attribution #777 reports.
12150
+ 3. At or beyond the high-water mark the record is first observed now, so it
12151
+ takes the freshly resolved active identity and gets its own stamp.
12152
+ """
12153
+ stamped = stamps.get((incarnation_id, offset, digest))
12154
+ if (incarnation_id, offset, digest) in stamps:
12155
+ return stamped, False
12156
+ if offset < high_water:
12157
+ if offset in gaps:
12158
+ return gaps[offset], False
12159
+ return None, False
12160
+ return active_account_key, True
12161
+
12162
+
12163
+ def _publish_title_generation(conn, _pause=None) -> None:
12164
+ """Replace the live title and rollup tables with the staged generation.
12165
+
12166
+ ONE transaction containing both DELETEs and both INSERTs, so WAL snapshot
12167
+ isolation gives every concurrent reader either the whole previous
12168
+ generation or the whole new one. That is the guarantee the rejected A/B
12169
+ slot design would have had to implement by hand with a published-slot
12170
+ pointer, reader routing and a retirement protocol.
12171
+
12172
+ Nothing here writes ``conversation_title_fts``. It is an external-content
12173
+ FTS5 index bound BY NAME to ``conversation_ai_titles``, so it follows
12174
+ through the existing ``conv_title_fts_ai`` / ``_ad`` / ``_au`` triggers
12175
+ when this rewrites the live table — and it is simply absent under
12176
+ ``fts5_unavailable`` or inside migration 018's pending window, where the
12177
+ triggers do not exist and there is nothing to drive.
12178
+
12179
+ Readers are NOT routed through a TEMP view. A TEMP view can serve neither
12180
+ ``MATCH`` nor ``rowid``, so a reader behind one could not use the title
12181
+ index at all.
12182
+
12183
+ ``_pause`` is a test seam invoked with the transaction OPEN and both
12184
+ DELETEs issued. A concurrent reader running at that moment is the direct
12185
+ evidence of atomicity, and raising from it is the direct evidence that a
12186
+ crash inside the publish rolls back to the previous generation.
12187
+ """
12188
+ conn.execute("BEGIN IMMEDIATE")
12189
+ try:
12190
+ conn.execute("DELETE FROM conversation_ai_titles")
12191
+ conn.execute("DELETE FROM conversation_sessions")
12192
+ if _pause is not None:
12193
+ _pause()
12194
+ conn.execute(
12195
+ "INSERT INTO conversation_ai_titles"
12196
+ "(session_id,ai_title,source_path,byte_offset) "
12197
+ "SELECT session_id,ai_title,source_path,byte_offset "
12198
+ "FROM conversation_ai_titles_staging"
12199
+ )
12200
+ conn.execute(
12201
+ "INSERT INTO conversation_sessions"
12202
+ "(session_id,msg_count,started_utc,last_activity_utc,project_label,"
12203
+ "cost_usd,cache_rebuild_count,git_branch,models_json,title,"
12204
+ "render_revision) "
12205
+ "SELECT session_id,msg_count,started_utc,last_activity_utc,"
12206
+ "project_label,cost_usd,cache_rebuild_count,git_branch,models_json,"
12207
+ "title,render_revision FROM conversation_sessions_staging"
12208
+ )
12209
+ conn.commit()
12210
+ except BaseException:
12211
+ conn.rollback()
12212
+ raise
12213
+ # Truncated only AFTER the publish commits, so a crash between the two
12214
+ # leaves a complete staged generation a retry can republish rather than a
12215
+ # half-emptied one.
12216
+ conn.execute("DELETE FROM conversation_ai_titles_staging")
12217
+ conn.execute("DELETE FROM conversation_sessions_staging")
12218
+ conn.commit()
12219
+
12220
+
11166
12221
  def _report_conversation_progress(
11167
12222
  progress: "Callable[[str, Any], None] | None",
11168
12223
  phase: str,
@@ -11187,6 +12242,19 @@ def sync_claude_conversations(
11187
12242
  transaction as its message/title rows. No cache.db table is written, and
11188
12243
  the core accounting cursor is neither read nor advanced.
11189
12244
  """
12245
+ # #779: FIRST executable statement, before IngestStats, before the data
12246
+ # directory is created, before the lock file is touched or opened, and
12247
+ # before any flock. This check used to sit below the rebuild branch, so a
12248
+ # call carrying both arguments committed the pending marker, ran the
12249
+ # maintenance preparation and executed the four destructive DELETEs, and
12250
+ # only then raised over an already-emptied store. Nothing downstream may
12251
+ # re-check the pair: `rebuild = rebuild or pending_rebuild` legitimately
12252
+ # turns a targeted call into an inherited global rebuild, and that case is
12253
+ # answered by the `deferred_reason="rebuild_pending"` return above it.
12254
+ if only_paths is not None and rebuild:
12255
+ raise ValueError(
12256
+ "sync_claude_conversations: only_paths is incompatible with rebuild"
12257
+ )
11190
12258
  stats = IngestStats()
11191
12259
  did_from_zero_replay = False
11192
12260
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
@@ -11240,15 +12308,17 @@ def sync_claude_conversations(
11240
12308
  # performing half of it from a stale binary is worse than not
11241
12309
  # performing it at all, and deferred_reason is how the caller
11242
12310
  # learns it did nothing.
11243
- try:
11244
- fp_row = conn.execute(
11245
- "SELECT value FROM cache_meta WHERE key=?",
11246
- (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
11247
- ).fetchone()
11248
- except sqlite3.OperationalError:
11249
- fp_row = None
11250
- stored_fp = fp_row[0] if fp_row else None
11251
- if not _pricing_write_authorized(stored_fp):
12311
+ # #728: `may_reset_rebuild_target`, not the write predicate. The
12312
+ # ordering is the same, but this is the path that must fail closed
12313
+ # — a refused clear costs nothing, while a clear performed on the
12314
+ # strength of a read that never happened empties a rollup this
12315
+ # process may then be refused permission to re-derive. The
12316
+ # degrade-to-None shape this replaces made a failed read look like
12317
+ # a store with nothing to protect, which is exactly the state that
12318
+ # authorizes the clear.
12319
+ fp_obs = _read_pricing_fingerprint_observation(conn)
12320
+ stored_fp = fp_obs.raw
12321
+ if not may_reset_rebuild_target(fp_obs, PRICING_SNAPSHOT_DATE):
11252
12322
  _record_pricing_write_refusal(conn, stored_fp)
11253
12323
  # COMMIT. _record_pricing_write_refusal leaves the transaction
11254
12324
  # to its caller, and this caller returns immediately, so
@@ -11275,6 +12345,40 @@ def sync_claude_conversations(
11275
12345
  conn.commit()
11276
12346
  stats.deferred_reason = "pricing_write_refused"
11277
12347
  return stats
12348
+ # #777: a rebuild replays every record and decides its account
12349
+ # from the durable stamps. Over a store whose backfill has not
12350
+ # finished, the lookup-miss rule would read "no stamp" as
12351
+ # "historical and unknown" and write NULL for records that DO have
12352
+ # a recorded attribution — the whole loss the backfill exists to
12353
+ # prevent, performed deliberately. Refuse before anything
12354
+ # destructive; migration 009 completes the backfill on the next
12355
+ # open, and the rebuild succeeds after it.
12356
+ if not _stamp_coverage_complete(conn):
12357
+ stats.deferred_reason = "stamp_backfill_pending"
12358
+ return stats
12359
+ # #752: check free space BEFORE the destructive step rather than
12360
+ # failing partway through. A rebuild that runs out of space after
12361
+ # clearing is the state that produced this issue.
12362
+ #
12363
+ # A reading of `CONVERSATION_FREE_SPACE_UNKNOWN` skips the check
12364
+ # entirely rather than comparing. That is the documented fail-open:
12365
+ # a preflight that could not measure must not be the thing that
12366
+ # stops a rebuild.
12367
+ _free_bytes = _conversation_staging_free_bytes()
12368
+ if (_free_bytes is not CONVERSATION_FREE_SPACE_UNKNOWN
12369
+ and _free_bytes
12370
+ < _conversation_rebuild_free_bytes_required(conn)):
12371
+ stats.deferred_reason = "insufficient_free_space"
12372
+ return stats
12373
+ # #780: refuse a new rebuild while the reclaim backlog is over its
12374
+ # hard ceiling. A rebuild adds a whole staged generation of churn
12375
+ # to a store that is already failing to return the space it freed,
12376
+ # so starting one makes the condition worse rather than better.
12377
+ _retention_sib = _load_lib("_lib_conversation_retention")
12378
+ if _retention_sib.reclaim_backlog_over_ceiling(
12379
+ _retention_sib.read_reclaim_pending(conn)):
12380
+ stats.deferred_reason = "reclaim_backlog_over_ceiling"
12381
+ return stats
11278
12382
  # Commit the retry marker before the destructive clear. A killed
11279
12383
  # #395 worker therefore leaves a partial transcript store visibly
11280
12384
  # pending instead of advancing it to a false-complete state.
@@ -11290,25 +12394,52 @@ def sync_claude_conversations(
11290
12394
  active_account_key=active_account_key,
11291
12395
  )
11292
12396
 
11293
- rebuild_account_stamps: dict[tuple[str, int], str | None] = {}
12397
+ # #777: the per-incarnation published high-water offset, captured at
12398
+ # rebuild START. It is what separates the two attribution rules the
12399
+ # review found contradictory otherwise: an offset BELOW it with no
12400
+ # stamp is genuinely historical and unknown, so it takes NULL and never
12401
+ # the active account; an offset AT OR BEYOND it is first observed now,
12402
+ # so it resolves the active identity freshly and creates its stamp.
12403
+ # Keyed by (path, incarnation) — PER-INCARNATION, which is the whole
12404
+ # point. A high-water mark is a claim about what a particular life of a
12405
+ # file already published; a file replaced since then has published
12406
+ # nothing, so every one of its records is first observed now and takes
12407
+ # the active identity rather than being read as historical-and-unknown.
12408
+ high_water: "dict[tuple[str, str], int]" = {}
12409
+ # #752: a rebuild writes its titles and rollup into staging and does
12410
+ # not touch the live tables until the publish, so readers keep seeing
12411
+ # the previous complete generation for the whole replay.
12412
+ title_table = "conversation_ai_titles"
12413
+ rollup_table = "conversation_sessions"
11294
12414
  if rebuild:
11295
- rebuild_account_stamps = {
11296
- (str(path), int(offset)): account_key
11297
- for path, offset, account_key in conn.execute(
11298
- "SELECT source_path,byte_offset,account_key "
11299
- "FROM conversation_messages"
12415
+ high_water = {
12416
+ (str(path), incarnation): int(offset)
12417
+ for path, incarnation, offset in conn.execute(
12418
+ "SELECT path,source_incarnation_id,last_byte_offset "
12419
+ "FROM conversation_source_files "
12420
+ "WHERE source_incarnation_id IS NOT NULL"
11300
12421
  )
11301
12422
  }
12423
+ title_table = "conversation_ai_titles_staging"
12424
+ rollup_table = "conversation_sessions_staging"
11302
12425
  clear_conversation_messages(conn)
11303
- conn.execute("DELETE FROM conversation_ai_titles")
11304
- conn.execute("DELETE FROM conversation_sessions")
11305
- conn.execute("DELETE FROM conversation_source_files")
12426
+ # Staging starts empty; a leftover generation from an interrupted
12427
+ # rebuild describes bytes this pass is about to replay.
12428
+ conn.execute("DELETE FROM conversation_ai_titles_staging")
12429
+ conn.execute("DELETE FROM conversation_sessions_staging")
12430
+ # The source-file rows are LEFT INTACT (#777). Deleting them, as
12431
+ # this used to, would take every `source_incarnation_id` with them,
12432
+ # and a stamp is keyed by the incarnation — so every stamp written
12433
+ # before the rebuild would miss and the replay that is supposed to
12434
+ # preserve the retained attribution would discard all of it.
12435
+ # Resetting the cursor instead would be almost as bad: continuity
12436
+ # would then be checked against an empty prefix, which any file
12437
+ # satisfies, so a file rewritten in place would keep its incarnation
12438
+ # and inherit stamps for bytes that no longer exist. The rows stay
12439
+ # exactly as the previous life committed them, and the replay is
12440
+ # forced from byte zero by `force_replay` instead.
11306
12441
  conn.commit()
11307
12442
 
11308
- if only_paths is not None and rebuild:
11309
- raise ValueError(
11310
- "sync_claude_conversations: only_paths is incompatible with rebuild"
11311
- )
11312
12443
  paths = (
11313
12444
  [pathlib.Path(path) for path in sorted(only_paths)
11314
12445
  if pathlib.Path(path).is_file()]
@@ -11317,10 +12448,15 @@ def sync_claude_conversations(
11317
12448
  )
11318
12449
  stats.files_total = len(paths)
11319
12450
  _report_conversation_progress(progress, "ingest", stats)
12451
+ # (size_bytes, mtime_ns, last_byte_offset, source_incarnation_id,
12452
+ # device_id, inode, committed_prefix_sha256) — the first three are the
12453
+ # cursor this loop has always read; the last four are #777's identity,
12454
+ # consumed by `_resolve_source_incarnation` at the same indices.
11320
12455
  existing = {
11321
- row[0]: (row[1], row[2], row[3])
12456
+ row[0]: tuple(row[1:])
11322
12457
  for row in conn.execute(
11323
- "SELECT path,size_bytes,mtime_ns,last_byte_offset "
12458
+ "SELECT path,size_bytes,mtime_ns,last_byte_offset,"
12459
+ "source_incarnation_id,device_id,inode,committed_prefix_sha256 "
11324
12460
  "FROM conversation_source_files"
11325
12461
  )
11326
12462
  }
@@ -11354,7 +12490,7 @@ def sync_claude_conversations(
11354
12490
  continue
11355
12491
  size, mtime_ns = st.st_size, st.st_mtime_ns
11356
12492
  prev = existing.get(path_str)
11357
- if prev is not None and size == prev[0]:
12493
+ if not rebuild and prev is not None and size == prev[0]:
11358
12494
  stats.files_skipped_unchanged += 1
11359
12495
  _report_conversation_progress(progress, "ingest", stats)
11360
12496
  continue
@@ -11362,23 +12498,60 @@ def sync_claude_conversations(
11362
12498
  if targeted and truncated:
11363
12499
  stats.deferred_reason = "truncation"
11364
12500
  return stats
11365
- start_offset = 0 if prev is None or truncated else prev[2]
12501
+ # The read start is decided by the incarnation resolver below,
12502
+ # which is the only place that knows whether the cursor still
12503
+ # describes this file. Initialised here only so the OSError path
12504
+ # below has a value.
12505
+ start_offset = 0
11366
12506
  conv_rows: list[tuple[Any, ...]] = []
11367
12507
  ai_rows: list[tuple[Any, ...]] = []
12508
+ stamp_rows: list[tuple[Any, ...]] = []
11368
12509
  final_offset = start_offset
12510
+ stamps, gaps = _load_account_stamps(conn, path_str)
11369
12511
  try:
11370
- with open(jp, "r", encoding="utf-8", errors="replace") as fh:
12512
+ # BINARY, and ONE descriptor for the whole file (#777). Binary
12513
+ # because a stamp's digest is over the record's raw bytes, which
12514
+ # the text layer's replacement decoding does not preserve. One
12515
+ # descriptor because the incarnation decision, the guard-digest
12516
+ # read and the ingest read must all describe the same open file:
12517
+ # a stat-then-open sequence leaves a window in which the file is
12518
+ # replaced between the decision and the read, and the records
12519
+ # would then be filed under the previous incarnation's identity.
12520
+ with open(jp, "rb") as fh:
12521
+ st_before = os.fstat(fh.fileno())
12522
+ incarnation = _resolve_source_incarnation(
12523
+ conn, path_str, fh, st_before, prev,
12524
+ force_replay=rebuild)
12525
+ start_offset = incarnation.start_offset
12526
+ file_high_water = high_water.get(
12527
+ (path_str, incarnation.incarnation_id), 0)
12528
+ if incarnation.is_new:
12529
+ # A new incarnation forces byte-zero treatment: the
12530
+ # bytes before the cursor are not the bytes the cursor
12531
+ # described, so resuming from it would skip real records
12532
+ # and stamp the rest against an identity that never held
12533
+ # them.
12534
+ truncated = prev is not None
11371
12535
  fh.seek(start_offset)
11372
- for _offset, _cost, mrow, ai in _iter_sync_entries(
12536
+ for _offset, _cost, mrow, ai, raw in _iter_sync_entries(
11373
12537
  fh,
11374
12538
  path_str,
11375
12539
  include_cost=False,
12540
+ with_raw=True,
11376
12541
  ):
11377
12542
  if mrow is not None:
11378
- account_key = rebuild_account_stamps.get(
11379
- (path_str, int(mrow.byte_offset)),
12543
+ offset = int(mrow.byte_offset)
12544
+ digest = _lib_conversation.record_sha256(raw)
12545
+ account_key, fresh_stamp = _resolve_record_account(
12546
+ stamps, gaps, incarnation.incarnation_id,
12547
+ offset, digest, file_high_water,
11380
12548
  active_account_key,
11381
12549
  )
12550
+ if fresh_stamp:
12551
+ stamp_rows.append(
12552
+ (path_str, incarnation.incarnation_id,
12553
+ offset, digest, account_key)
12554
+ )
11382
12555
  conv_rows.append(
11383
12556
  _conv_row_tuple(
11384
12557
  mrow, path_str, account_key,
@@ -11390,6 +12563,59 @@ def sync_claude_conversations(
11390
12563
  (ai.session_id, ai.ai_title, path_str, ai.byte_offset)
11391
12564
  )
11392
12565
  final_offset = fh.tell()
12566
+ st_after = os.fstat(fh.fileno())
12567
+ if _stat_pair_broken(st_before, st_after):
12568
+ # The file's IDENTITY broke while we were reading it —
12569
+ # a shrink, or a size-preserving rewrite. An append is
12570
+ # deliberately not this branch.
12571
+ #
12572
+ # Commit nothing. The next sync opens the file again
12573
+ # and re-decides the incarnation from scratch: the
12574
+ # committed-prefix guard mints a fresh incarnation and
12575
+ # replays from zero whenever the bytes before the
12576
+ # cursor moved, and otherwise continuity holds and it
12577
+ # resumes from the stored cursor. Either way nothing
12578
+ # from this partial read is carried forward, which is
12579
+ # what the descriptor pair exists to guarantee.
12580
+ stats.files_failed += 1
12581
+ _report_conversation_progress(progress, "ingest", stats)
12582
+ continue
12583
+ # The committed prefix digest is carried FORWARD from the
12584
+ # verified one rather than rehashed: the object is already
12585
+ # positioned at `start_offset`, so only the bytes this pass
12586
+ # ingested are added. That is what keeps the guard's cost
12587
+ # proportional to the append rather than to the file.
12588
+ fh.seek(start_offset)
12589
+ incarnation.mutable_digest.update(
12590
+ fh.read(max(0, final_offset - start_offset)))
12591
+ committed_prefix = incarnation.mutable_digest.hexdigest()
12592
+ device_id, inode = st_after.st_dev, st_after.st_ino
12593
+ # The recorded size and mtime come from the fstat taken at
12594
+ # the START of the read, not the end. They are not identity
12595
+ # — `_resolve_source_incarnation` decides that on device,
12596
+ # inode and the committed-prefix digest — they are the
12597
+ # change detector the `size == prev[0]` skip above reads.
12598
+ # Now that an append during the read is a continuation
12599
+ # rather than a break, recording the post-append size would
12600
+ # claim this pass consumed bytes it never reached, and the
12601
+ # next sync would skip the file as unchanged and lose those
12602
+ # records permanently. The pair is equal whenever nothing
12603
+ # raced, so this changes nothing outside that window.
12604
+ #
12605
+ # A RESIDUAL, stated rather than left implicit. The safest
12606
+ # recorded size is `final_offset`, because the partial-tail
12607
+ # rewind can end the read before `st_before.st_size` when
12608
+ # the file's last line is incomplete — so `size` can exceed
12609
+ # what this pass actually consumed by that tail. Narrowing
12610
+ # it to `final_offset` is not correct either: `size` is the
12611
+ # change detector for the `size == prev[0]` skip above, and
12612
+ # recording the consumed offset instead would make the next
12613
+ # sync see a size difference on an unchanged file and
12614
+ # re-read it every tick. The residual is bounded by one
12615
+ # partial record and is self-correcting, because the
12616
+ # writer finishes that line and the size moves again;
12617
+ # identity does not depend on either number.
12618
+ size, mtime_ns = st_before.st_size, st_before.st_mtime_ns
11393
12619
  except OSError as exc:
11394
12620
  eprint(f"[conversations] could not read {jp}: {exc}")
11395
12621
  stats.files_failed += 1
@@ -11416,7 +12642,7 @@ def sync_claude_conversations(
11416
12642
  (path_str,),
11417
12643
  )
11418
12644
  conn.execute(
11419
- "DELETE FROM conversation_ai_titles WHERE source_path=?",
12645
+ f"DELETE FROM {title_table} WHERE source_path=?",
11420
12646
  (path_str,),
11421
12647
  )
11422
12648
  stats.files_reset_truncated += 1
@@ -11425,21 +12651,43 @@ def sync_claude_conversations(
11425
12651
  _fill_file_touches(
11426
12652
  conn, scope=[(row[3], row[4]) for row in conv_rows]
11427
12653
  )
12654
+ if stamp_rows:
12655
+ # #777: batched into the SAME transaction as the message
12656
+ # batch, never issued per row. One stamp per message is new
12657
+ # write volume across every future record, and a per-row
12658
+ # transaction would multiply the ingest's commit cost by the
12659
+ # record count.
12660
+ conn.executemany(
12661
+ "INSERT OR IGNORE INTO claude_conversation_account_stamps"
12662
+ "(canonical_source_path,source_incarnation_id,"
12663
+ "byte_offset,record_sha256,account_key) "
12664
+ "VALUES(?,?,?,?,?)",
12665
+ stamp_rows,
12666
+ )
11428
12667
  if ai_rows:
11429
- conn.executemany(_AI_TITLE_UPSERT_SQL, ai_rows)
12668
+ conn.executemany(_ai_title_upsert_sql(title_table), ai_rows)
11430
12669
  conn.execute(
11431
12670
  "INSERT INTO conversation_source_files "
11432
- "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at) "
11433
- "VALUES(?,?,?,?,?) ON CONFLICT(path) DO UPDATE SET "
12671
+ "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at,"
12672
+ "device_id,inode,source_incarnation_id,"
12673
+ "committed_prefix_sha256) "
12674
+ "VALUES(?,?,?,?,?,?,?,?,?) ON CONFLICT(path) DO UPDATE SET "
11434
12675
  "size_bytes=excluded.size_bytes,mtime_ns=excluded.mtime_ns,"
11435
12676
  "last_byte_offset=excluded.last_byte_offset,"
11436
- "last_ingested_at=excluded.last_ingested_at",
12677
+ "last_ingested_at=excluded.last_ingested_at,"
12678
+ "device_id=excluded.device_id,inode=excluded.inode,"
12679
+ "source_incarnation_id=excluded.source_incarnation_id,"
12680
+ "committed_prefix_sha256=excluded.committed_prefix_sha256",
11437
12681
  (
11438
12682
  path_str,
11439
12683
  size,
11440
12684
  mtime_ns,
11441
12685
  final_offset,
11442
12686
  dt.datetime.now(dt.timezone.utc).isoformat(),
12687
+ device_id,
12688
+ inode,
12689
+ incarnation.incarnation_id,
12690
+ committed_prefix,
11443
12691
  ),
11444
12692
  )
11445
12693
  conn.commit()
@@ -11454,9 +12702,127 @@ def sync_claude_conversations(
11454
12702
  stats.files_failed += 1
11455
12703
  _report_conversation_progress(progress, "ingest", stats)
11456
12704
 
12705
+ if rebuild and only_paths is None:
12706
+ # A rebuild used to DELETE every `conversation_source_files` row and
12707
+ # let the walk recreate the ones it found, which also removed rows
12708
+ # for paths the walk no longer discovers. #777 stopped deleting the
12709
+ # rows, because they carry the incarnation identity every durable
12710
+ # stamp is keyed by — so the stale rows have to be removed here
12711
+ # instead, by difference against the walked set. Without this a
12712
+ # rebuild over a different corpus leaves the previous corpus's
12713
+ # paths in the store, which is both wrong and a privacy leak.
12714
+ #
12715
+ # The STAMPS and GAPS for those paths go with the row. Tranche 2
12716
+ # kept them, on the grounds that a stamp is the only surviving
12717
+ # record of who the records belonged to — but no read path can
12718
+ # reach one once the source-file row is gone. `_resolve_record_account`
12719
+ # keys on `(source_incarnation_id, byte_offset, record_sha256)`;
12720
+ # with no `conversation_source_files` row `_resolve_source_incarnation`
12721
+ # sees `prev is None` and always mints a fresh incarnation, and the
12722
+ # per-incarnation high-water map is built from that same table, so
12723
+ # the gap half is unreachable for the same reason. Retaining them
12724
+ # grows two tables without bound and preserves nothing. Giving them
12725
+ # a reader instead was the alternative and is worse: a path-and-
12726
+ # offset fallback would let an unrelated file that reuses a path
12727
+ # inherit the old attribution, which is exactly the inheritance the
12728
+ # incarnation key exists to prevent.
12729
+ #
12730
+ # GUARDED BY WALK LIVENESS (spec §2, second loss path). "Absent
12731
+ # from this walk" is a weaker fact than "absent from disk", and
12732
+ # taking the first for the second makes an unavailable corpus
12733
+ # delete every stamp in the store: an unmounted volume or a wrong
12734
+ # `CLAUDE_CONFIG_DIR` produces an empty walk, every tracked path is
12735
+ # then a difference, and the one thing here that is NOT
12736
+ # re-derivable goes with it. Two conditions, both required.
12737
+ #
12738
+ # THE PRICE OF THAT GUARD, stated rather than left to be found. A
12739
+ # path that is still on disk but no longer in scope — the walk now
12740
+ # covers a different tree, because `CLAUDE_CONFIG_DIR` moved or the
12741
+ # store was copied beside the corpus it was derived from — is
12742
+ # RETAINED, so the paragraph above overstates the case: the prune
12743
+ # removes the previous corpus's paths only when that corpus is
12744
+ # actually gone. Retaining them costs orphaned source rows and
12745
+ # their stamps, since `clear_conversation_messages` has already
12746
+ # removed every message they describe. That is deliberate. Nothing
12747
+ # available here distinguishes a corpus that moved from a root that
12748
+ # this walk could not reach, and only one of the two mistakes is
12749
+ # recoverable.
12750
+ walked = {str(jp) for jp in paths}
12751
+ stale = (
12752
+ [
12753
+ row[0] for row in conn.execute(
12754
+ "SELECT path FROM conversation_source_files")
12755
+ if row[0] not in walked and _path_is_genuinely_absent(row[0])
12756
+ ]
12757
+ if walked
12758
+ else []
12759
+ )
12760
+ for start in range(0, len(stale), 400):
12761
+ chunk = stale[start:start + 400]
12762
+ placeholders = ",".join("?" for _ in chunk)
12763
+ conn.execute(
12764
+ f"DELETE FROM conversation_source_files "
12765
+ f"WHERE path IN ({placeholders})",
12766
+ chunk,
12767
+ )
12768
+ conn.execute(
12769
+ f"DELETE FROM claude_conversation_account_stamps "
12770
+ f"WHERE canonical_source_path IN ({placeholders})",
12771
+ chunk,
12772
+ )
12773
+ conn.execute(
12774
+ f"DELETE FROM claude_conversation_account_stamp_gaps "
12775
+ f"WHERE source_path IN ({placeholders})",
12776
+ chunk,
12777
+ )
12778
+ if stale:
12779
+ conn.commit()
12780
+
11457
12781
  _report_conversation_progress(progress, "rollup", stats)
11458
12782
  rollup_authorized = _arm_rollup_backfill_on_pricing_change(conn)
11459
- if _conversation_sessions_backfill_pending(conn):
12783
+ if rebuild:
12784
+ # #752: a rebuild ALWAYS derives its rollup in full, into staging,
12785
+ # regardless of the backfill flag — it has just replayed every
12786
+ # message row, so a scoped recompute would describe a fraction of
12787
+ # the store. The live rollup is still the previous complete
12788
+ # generation at this point and stays so until the publish below.
12789
+ if _recompute_conversation_sessions(conn, target=rollup_table):
12790
+ conn.execute(
12791
+ "DELETE FROM cache_meta "
12792
+ "WHERE key='conversation_sessions_backfill_pending'"
12793
+ )
12794
+ _clear_pricing_write_refusal(conn)
12795
+ conn.commit()
12796
+ # Titles that arrived on the LIVE table while staging was being
12797
+ # built are folded in before the publish, or the publish would
12798
+ # silently drop them. `INSERT OR IGNORE` keeps the replayed
12799
+ # generation authoritative for any session both hold.
12800
+ #
12801
+ # SCOPED to the sessions this replay actually produced (spec
12802
+ # §1, corrected after the Tranche 2 review). An unconditional
12803
+ # fold makes every published generation a permanent superset of
12804
+ # the previous one: the table never shrinks, titles for deleted
12805
+ # sessions and removed worktrees persist forever, and title
12806
+ # search returns hits for sessions with no messages and no
12807
+ # rollup row — because the rollup half IS re-derived and
12808
+ # correctly drops them. Retention cannot compensate, because
12809
+ # `_prune_claude` selects session ids from
12810
+ # `conversation_messages` and a session with no message rows is
12811
+ # never a prune candidate. The rollup staging table is the
12812
+ # replayed generation's own session set, so it is the gate.
12813
+ conn.execute(
12814
+ "INSERT OR IGNORE INTO conversation_ai_titles_staging"
12815
+ "(session_id,ai_title,source_path,byte_offset) "
12816
+ "SELECT session_id,ai_title,source_path,byte_offset "
12817
+ "FROM conversation_ai_titles WHERE session_id IN "
12818
+ "(SELECT session_id FROM conversation_sessions_staging)"
12819
+ )
12820
+ conn.commit()
12821
+ _publish_title_generation(conn)
12822
+ else:
12823
+ conn.commit()
12824
+ rollup_authorized = False
12825
+ elif _conversation_sessions_backfill_pending(conn):
11460
12826
  # #705: a refused process must NOT consume the flag. It leaves it
11461
12827
  # set for the next authorized process, so a newer process that armed
11462
12828
  # the backfill and died is not followed by a stale successor either
@@ -11552,7 +12918,28 @@ def sync_codex_conversations(
11552
12918
  only_paths: "set[str] | None" = None,
11553
12919
  progress: "Callable[[str, CodexIngestStats], None] | None" = None,
11554
12920
  ) -> CodexIngestStats:
11555
- """Delta-sync Codex events/search rows into conversations.db (#320)."""
12921
+ """Delta-sync Codex events/search rows into conversations.db (#320).
12922
+
12923
+ #779 (the Codex twin, folded in by the Tranche 1 review): the
12924
+ incompatible-argument check below is the FIRST executable statement, for
12925
+ exactly the reasons its Claude twin states. This function used to write the
12926
+ `conversation_rebuild_codex_pending` marker, commit, call
12927
+ `_clear_codex_conversation_store(conn)`, commit again, and only then raise
12928
+ — so a bad argument pair destroyed the entire Codex conversation store
12929
+ durably before refusing to do the work. The issue text named only the
12930
+ Claude path; this is the same defect in the twin function in the same file,
12931
+ and the remedy is the same reordering.
12932
+
12933
+ Nothing downstream may re-check the pair. `rebuild = rebuild or
12934
+ pending_rebuild or contract_rebuild or codex_replay_pending` legitimately
12935
+ turns a targeted call into inherited global work, and that case is answered
12936
+ by the `deferred_reason="rebuild_pending"` return above the merge, not by
12937
+ treating an inherited rebuild as an explicitly incompatible argument.
12938
+ """
12939
+ if only_paths is not None and rebuild:
12940
+ raise ValueError(
12941
+ "sync_codex_conversations: only_paths is incompatible with rebuild"
12942
+ )
11556
12943
  stats = CodexIngestStats()
11557
12944
  did_from_zero_replay = False
11558
12945
  rebuild_account_stamps: dict[tuple[str, int], str | None] = {}
@@ -11657,10 +13044,6 @@ def sync_codex_conversations(
11657
13044
  _clear_codex_conversation_store(conn)
11658
13045
  conn.commit()
11659
13046
 
11660
- if only_paths is not None and rebuild:
11661
- raise ValueError(
11662
- "sync_codex_conversations: only_paths is incompatible with rebuild"
11663
- )
11664
13047
  files = (
11665
13048
  _qualify_codex_targets(only_paths)
11666
13049
  if only_paths is not None
@@ -11668,13 +13051,18 @@ def sync_codex_conversations(
11668
13051
  )
11669
13052
  stats.files_total = len(files)
11670
13053
  _report_conversation_progress(progress, "ingest", stats)
13054
+ # The last two members are `device_id`/`inode` (#769 S6). They are read
13055
+ # by the per-file decision below, not merely retained: a replacement
13056
+ # that lands at the same size moves neither the size nor the cursor, so
13057
+ # identity is the only evidence that the retained offset describes a
13058
+ # file that is no longer there.
11671
13059
  existing = {
11672
13060
  row[0]: tuple(row[1:])
11673
13061
  for row in conn.execute(
11674
13062
  "SELECT path,size_bytes,mtime_ns,last_byte_offset,source_root_key,"
11675
13063
  "last_session_id,last_model,last_total_tokens,"
11676
13064
  "last_native_thread_id,last_root_thread_id,last_parent_thread_id,"
11677
- "last_conversation_key,last_turn_id "
13065
+ "last_conversation_key,last_turn_id,device_id,inode "
11678
13066
  "FROM codex_conversation_source_files"
11679
13067
  )
11680
13068
  }
@@ -11779,18 +13167,37 @@ def sync_codex_conversations(
11779
13167
  continue
11780
13168
  size, mtime_ns = st.st_size, st.st_mtime_ns
11781
13169
  prev = existing.get(path_str)
11782
- if prev is not None and size == prev[0] and prev[3] == discovered.source_root_key:
13170
+ # A stored identity that differs from the one just statted means the
13171
+ # retained offset points into a file that no longer occupies this
13172
+ # pathname. THE INODE DECIDES and the device only corroborates:
13173
+ # `_lib_ingest_frontier.source_identity_replaced` owns the verdict
13174
+ # for the planner and for both walks, states why a remount must not
13175
+ # reach it, and degrades a NULL or unreadable stored value to the
13176
+ # size comparison below rather than raising out of this loop.
13177
+ replaced = (
13178
+ prev is not None
13179
+ and _ingest_frontier.source_identity_replaced(
13180
+ prev[12], prev[13], st.st_dev, st.st_ino)
13181
+ )
13182
+ if (
13183
+ prev is not None and not replaced and size == prev[0]
13184
+ and prev[3] == discovered.source_root_key
13185
+ ):
11783
13186
  stats.files_skipped_unchanged += 1
11784
13187
  _report_conversation_progress(progress, "ingest", stats)
11785
13188
  continue
11786
13189
  reset_file = (
11787
13190
  prev is not None
11788
- and (size < prev[0] or prev[3] != discovered.source_root_key)
13191
+ and (
13192
+ size < prev[0] or replaced
13193
+ or prev[3] != discovered.source_root_key
13194
+ )
11789
13195
  )
11790
13196
  if targeted and reset_file:
11791
13197
  stats.deferred_reason = (
11792
13198
  "requalification"
11793
13199
  if prev is not None and prev[3] != discovered.source_root_key
13200
+ else "source_replaced" if replaced
11794
13201
  else "truncation"
11795
13202
  )
11796
13203
  return stats
@@ -11995,8 +13402,8 @@ def sync_codex_conversations(
11995
13402
  "(path,size_bytes,mtime_ns,last_byte_offset,last_ingested_at,"
11996
13403
  "source_root_key,last_session_id,last_model,last_total_tokens,"
11997
13404
  "last_native_thread_id,last_root_thread_id,last_parent_thread_id,"
11998
- "last_conversation_key,last_turn_id) "
11999
- "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?) "
13405
+ "last_conversation_key,last_turn_id,device_id,inode) "
13406
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) "
12000
13407
  "ON CONFLICT(path) DO UPDATE SET "
12001
13408
  "size_bytes=excluded.size_bytes,mtime_ns=excluded.mtime_ns,"
12002
13409
  "last_byte_offset=excluded.last_byte_offset,"
@@ -12009,7 +13416,8 @@ def sync_codex_conversations(
12009
13416
  "last_root_thread_id=excluded.last_root_thread_id,"
12010
13417
  "last_parent_thread_id=excluded.last_parent_thread_id,"
12011
13418
  "last_conversation_key=excluded.last_conversation_key,"
12012
- "last_turn_id=excluded.last_turn_id",
13419
+ "last_turn_id=excluded.last_turn_id,"
13420
+ "device_id=excluded.device_id,inode=excluded.inode",
12013
13421
  (
12014
13422
  path_str,
12015
13423
  size,
@@ -12025,6 +13433,11 @@ def sync_codex_conversations(
12025
13433
  terminal.parent_thread_id if terminal else initial_parent,
12026
13434
  terminal.conversation_key if terminal else initial_conversation,
12027
13435
  normalized.terminal.turn_id,
13436
+ # The SAME pre-read stat that produced `size`/`mtime_ns`
13437
+ # above, committed in this one statement beside the
13438
+ # offset it describes (#769 S6).
13439
+ int(st.st_dev),
13440
+ int(st.st_ino),
12028
13441
  ),
12029
13442
  )
12030
13443
  conn.commit()
@@ -12402,6 +13815,30 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
12402
13815
  f"[cache-sync] pruned {res.pruned_files} orphaned file(s), "
12403
13816
  f"{res.pruned_entries} cost row(s), {res.pruned_messages} message(s)"
12404
13817
  )
13818
+ if res.prune_refused:
13819
+ # #729: a STAGED failure under docs/cli-contract.md — the command
13820
+ # was understood and attempted, and part of it did not complete —
13821
+ # so exit 3, not the parity-family 1 or the usage 2. Precedence:
13822
+ # flock contention above is reported first because nothing was
13823
+ # attempted at all, and this outranks the residual-path report
13824
+ # below because a residual is work deliberately not done, while
13825
+ # this is committed deletions that were not followed by an
13826
+ # authorized re-derive.
13827
+ # #769 S3: the cause travels on the result. Stating the version
13828
+ # skew unconditionally described one of three refusing states as
13829
+ # if it were the only one.
13830
+ eprint(
13831
+ f"[cache-sync] the conversation rollup re-derive was refused "
13832
+ f"for {res.prune_refused_files} file(s): "
13833
+ f"{pricing_refusal_cause_phrase(res.prune_refused_state)}. "
13834
+ f"The orphan rows are deleted and the rollup backfill is "
13835
+ f"armed; run `cctally cache-sync --prune-orphans` again once "
13836
+ f"an authorized process can complete it, and see `cctally "
13837
+ f"doctor pricing.conversation_rollup_writer` for the step "
13838
+ f"that clears this state."
13839
+ )
13840
+ conn.close()
13841
+ return 3
12405
13842
  if res.residual_paths:
12406
13843
  eprint(
12407
13844
  f"[cache-sync] {len(res.residual_paths)} orphan(s) left in place "