cctally 1.106.0 → 1.108.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/CHANGELOG.md +68 -0
  2. package/bin/_cctally_alerts.py +147 -6
  3. package/bin/_cctally_cache.py +553 -44
  4. package/bin/_cctally_core.py +316 -151
  5. package/bin/_cctally_dashboard.py +70 -73
  6. package/bin/_cctally_dashboard_envelope.py +119 -95
  7. package/bin/_cctally_dashboard_share.py +1 -7
  8. package/bin/_cctally_db.py +6 -133
  9. package/bin/_cctally_diff.py +15 -8
  10. package/bin/_cctally_doctor.py +162 -43
  11. package/bin/_cctally_five_hour.py +6 -53
  12. package/bin/_cctally_forecast.py +73 -203
  13. package/bin/_cctally_journal.py +305 -243
  14. package/bin/_cctally_milestone_history.py +58 -141
  15. package/bin/_cctally_milestones.py +0 -72
  16. package/bin/_cctally_parser.py +15 -15
  17. package/bin/_cctally_percent_breakdown.py +292 -32
  18. package/bin/_cctally_project.py +3 -5
  19. package/bin/_cctally_quota.py +16 -30
  20. package/bin/_cctally_quota_model.py +160 -27
  21. package/bin/_cctally_record.py +631 -1377
  22. package/bin/_cctally_rederive.py +0 -2
  23. package/bin/_cctally_reporting.py +1 -19
  24. package/bin/_cctally_setup.py +272 -110
  25. package/bin/_cctally_share.py +45 -32
  26. package/bin/_cctally_statusline.py +41 -16
  27. package/bin/_cctally_tui.py +56 -233
  28. package/bin/_cctally_weekrefs.py +404 -317
  29. package/bin/_lib_aggregators.py +7 -2
  30. package/bin/_lib_alerts_payload.py +48 -1
  31. package/bin/_lib_cache_report.py +5 -2
  32. package/bin/_lib_codex_hooks.py +694 -83
  33. package/bin/_lib_credit.py +2 -9
  34. package/bin/_lib_dashboard_sources.py +4 -16
  35. package/bin/_lib_diff_kernel.py +33 -10
  36. package/bin/_lib_doctor.py +361 -24
  37. package/bin/_lib_journal.py +64 -0
  38. package/bin/_lib_meter_rate_change.py +132 -3
  39. package/bin/_lib_pricing.py +66 -5
  40. package/bin/_lib_pricing_check.py +5 -4
  41. package/bin/_lib_quota_model.py +54 -44
  42. package/bin/_lib_record.py +9 -41
  43. package/bin/_lib_rederive.py +0 -8
  44. package/bin/_lib_render.py +19 -80
  45. package/bin/_lib_snapshot_cache.py +15 -198
  46. package/bin/_lib_subscription_weeks.py +209 -135
  47. package/bin/_lib_view_models.py +86 -133
  48. package/bin/cctally +14 -10
  49. package/dashboard/static/assets/ConversationsView-BOSaBRtu.js +72 -0
  50. package/dashboard/static/assets/{DoctorModal-5rvkyksy.js → DoctorModal-CWn3U4Wl.js} +1 -1
  51. package/dashboard/static/assets/ModalRoot-SR070V6c.js +1 -0
  52. package/dashboard/static/assets/{ProjectsDrillPanel-CsR1bjJu.js → ProjectsDrillPanel-QLI9i5mZ.js} +1 -1
  53. package/dashboard/static/assets/{SourceDetailModal-BnTIwt-R.js → SourceDetailModal-pv0wlFxl.js} +1 -1
  54. package/dashboard/static/assets/{UpdateModal-D_RfssnW.js → UpdateModal-D3m8GG6V.js} +3 -3
  55. package/dashboard/static/assets/index-BHu4mxd8.css +1 -0
  56. package/dashboard/static/assets/index-CIWsbux3.js +13 -0
  57. package/dashboard/static/assets/{outlineNavigation-C7J9lhGB.js → outlineNavigation-CJvKqmLV.js} +1 -1
  58. package/dashboard/static/assets/useKeymap-ffqsS5G0.js +1 -0
  59. package/dashboard/static/dashboard.html +3 -3
  60. package/package.json +1 -3
  61. package/bin/_lib_credit_identity.py +0 -159
  62. package/bin/_lib_credit_selection.py +0 -342
  63. package/dashboard/static/assets/ConversationsView-Dst33C0n.js +0 -72
  64. package/dashboard/static/assets/ModalRoot-BJH02VzE.js +0 -1
  65. package/dashboard/static/assets/index-rzXw99Hy.js +0 -13
  66. package/dashboard/static/assets/index-u3wLfjt2.css +0 -1
  67. package/dashboard/static/assets/useKeymap-B3KCc_c7.js +0 -1
@@ -220,6 +220,11 @@ PRICING_SNAPSHOT_DATE = _load_lib("_lib_pricing").PRICING_SNAPSHOT_DATE
220
220
  # from the same circular-safe stdlib leaf as PRICING_SNAPSHOT_DATE above.
221
221
  claude_usage_dict = _load_lib("_lib_pricing").claude_usage_dict
222
222
 
223
+ # #705: the ONE parse of the recorded-fingerprint contract, shared with
224
+ # `_lib_doctor`'s report of the same decision. Bound from the same circular-safe
225
+ # stdlib leaf as the two names above.
226
+ parse_pricing_fingerprint = _load_lib("_lib_pricing").parse_pricing_fingerprint
227
+
223
228
  # Shared by the fused per-file walk AND backfill_conversation_messages so the
224
229
  # column list, placeholders, and tuple order live in ONE place — a column
225
230
  # add/reorder can't silently desync the two ingest paths (which would land
@@ -4055,6 +4060,14 @@ def sync_cache(
4055
4060
  # post-walk recompute (after the per-file loop, still under the
4056
4061
  # flock) consumes the flag and rebuilds the rollup from the freshly
4057
4062
  # re-ingested messages, then drops it last (crash-safe).
4063
+ #
4064
+ # This clear carries NO #705 pricing authorization, unlike the
4065
+ # authoritative rebuild path in sync_claude_conversations, and that
4066
+ # is deliberate: _iter_sync_entries is called with
4067
+ # include_conversations=False below, so cache.db's conversation
4068
+ # tables are permanently empty (measured: 0 rows) and this DELETE
4069
+ # destroys nothing. Enabling conversations on THIS connection would
4070
+ # require the same pre-clear authorization the rebuild path uses.
4058
4071
  conn.execute("DELETE FROM conversation_sessions")
4059
4072
  _set_cache_meta(conn, "conversation_sessions_backfill_pending", "1")
4060
4073
  conn.commit()
@@ -4437,6 +4450,11 @@ def sync_cache(
4437
4450
  # same destructive txn, alongside clear_conversation_messages) and
4438
4451
  # arm the durable backfill flag. The post-walk recompute rebuilds it
4439
4452
  # from the re-ingested messages and drops the flag last (crash-safe).
4453
+ #
4454
+ # Unauthorized for the same reason as the rebuild clear above:
4455
+ # include_conversations=False keeps cache.db's conversation tables
4456
+ # permanently empty, so this DELETE destroys nothing. Enabling
4457
+ # conversations here would require the #705 pre-clear authorization.
4440
4458
  conn.execute("DELETE FROM conversation_sessions")
4441
4459
  _set_cache_meta(conn, "conversation_sessions_backfill_pending", "1")
4442
4460
  conn.commit()
@@ -4761,17 +4779,50 @@ def sync_cache(
4761
4779
  # embedded pricing snapshot changed since it was last derived.
4762
4780
  # Runs BEFORE the pending check so a mismatch arms the same
4763
4781
  # durable flag the full-recompute path already consumes below.
4764
- _arm_rollup_backfill_on_pricing_change(conn)
4782
+ rollup_authorized = _arm_rollup_backfill_on_pricing_change(conn)
4765
4783
  if _conversation_sessions_backfill_pending(conn):
4766
- _recompute_conversation_sessions(conn)
4767
- conn.execute(
4768
- "DELETE FROM cache_meta "
4769
- "WHERE key='conversation_sessions_backfill_pending'"
4770
- )
4771
- conn.commit()
4784
+ # #705: a refused process must NOT consume the flag. It leaves it
4785
+ # set for the next authorized process, so a newer process that armed
4786
+ # the backfill and died is not followed by a stale successor either
4787
+ # rewriting cost from its old table or declaring an unrecomputed
4788
+ # rollup authoritative.
4789
+ if _recompute_conversation_sessions(conn):
4790
+ conn.execute(
4791
+ "DELETE FROM cache_meta "
4792
+ "WHERE key='conversation_sessions_backfill_pending'"
4793
+ )
4794
+ _clear_pricing_write_refusal(conn)
4795
+ conn.commit()
4796
+ else:
4797
+ # #705: the refusal wrote a record and armed the flag,
4798
+ # and _recompute_conversation_sessions leaves the commit
4799
+ # to its caller. Nothing after this block reliably
4800
+ # commits: the walk-complete sentinel is skipped on an
4801
+ # unclean walk and _update_parse_health_meta returns
4802
+ # without writing in steady state, so on a walk with one
4803
+ # failed file the record died with the connection.
4804
+ conn.commit()
4805
+ rollup_authorized = False
4772
4806
  elif touched_sessions:
4773
- _recompute_conversation_sessions(conn, touched_sessions)
4807
+ if not _recompute_conversation_sessions(
4808
+ conn, touched_sessions):
4809
+ rollup_authorized = False
4774
4810
  conn.commit()
4811
+ if not rollup_authorized and stats.deferred_reason is None:
4812
+ # #705: say the refusal rather than leave it to be inferred.
4813
+ # A refused core sync was already non-certifiable, but only
4814
+ # because the same refusal armed
4815
+ # `conversation_sessions_backfill_pending` and
4816
+ # `_lib_ingest_frontier._pending_identity` reads that flag.
4817
+ # `provider_sync_certifiable` knows nothing about the flag
4818
+ # and refuses on any non-None `deferred_reason`, so the
4819
+ # guarantee lived entirely in a coupling nothing stated and
4820
+ # would disappear the day that arming changed. The `is
4821
+ # None` guard is DEFENSIVE, like its
4822
+ # `sync_claude_conversations` twin: every earlier
4823
+ # assignment in this function returns immediately, so no
4824
+ # earlier reason can reach this line today.
4825
+ stats.deferred_reason = "pricing_write_refused"
4775
4826
 
4776
4827
  # Walk-complete sentinel write (cctally-dev#93, D5a). Still inside the
4777
4828
  # held fcntl lock, before the finally-unlock. Only when the entire walk
@@ -5104,40 +5155,247 @@ def _conversation_sessions_backfill_pending(conn) -> bool:
5104
5155
  return False
5105
5156
 
5106
5157
 
5107
- def _arm_rollup_backfill_on_pricing_change(conn) -> None:
5158
+ def _arm_rollup_backfill_pending(conn) -> bool:
5159
+ """Arm the durable full-recompute flag, returning whether this call armed
5160
+ it. Idempotent by design: a refused dashboard ticks continuously, and
5161
+ rewriting a set flag every tick would take the writer lock to restate an
5162
+ unchanged fact.
5163
+
5164
+ Every REFUSED pricing write arms this (#705). The flag is the store's
5165
+ durable "this rollup is not fully derived" signal and a refusal is exactly
5166
+ that condition, so arming it is not a white lie — it is what makes
5167
+ ``_lib_conversation_query._rollup_authoritative()`` false and delivers the
5168
+ degrade to live aggregation. Without it the refusal is a functional
5169
+ REGRESSION rather than a guard: the refused process still ingests
5170
+ conversation_messages but writes no rollup row for the sessions it touched,
5171
+ the flag is clear (the newer process that advanced the fingerprint cleared
5172
+ it), so the rail reads an un-recomputed rollup as authoritative and those
5173
+ conversations vanish from it permanently — with no self-heal, because a
5174
+ later authorized scoped recompute only covers sessions ITS walk touched and
5175
+ the refused process already consumed those bytes.
5176
+
5177
+ NEVER commits — the caller owns the transaction, like
5178
+ _record_pricing_write_refusal beside it. Degrades to False when cache_meta
5179
+ is unavailable (path-less / schema-not-applied conn), like its neighbours.
5180
+ """
5181
+ try:
5182
+ if conn.execute(
5183
+ "SELECT 1 FROM cache_meta "
5184
+ "WHERE key='conversation_sessions_backfill_pending'"
5185
+ ).fetchone() is not None:
5186
+ return False
5187
+ _set_cache_meta(conn, "conversation_sessions_backfill_pending", "1")
5188
+ except sqlite3.OperationalError:
5189
+ return False
5190
+ return True
5191
+
5192
+
5193
+ #: cache_meta keys for the conversation rollup's pricing provenance (#302, #705).
5194
+ CONVERSATION_ROLLUP_PRICING_FP_KEY = "conversation_sessions_pricing_fp"
5195
+ CONVERSATION_ROLLUP_PRICING_REFUSED_KEY = (
5196
+ "conversation_sessions_pricing_write_refused"
5197
+ )
5198
+
5199
+
5200
+ def _pricing_write_authorized(stored, process=None) -> bool:
5201
+ """Whether a process holding `process` pricing may write materialized
5202
+ conversation cost over a store that recorded `stored` (#705).
5203
+
5204
+ ORDERED, not an equality. The bare `!=` this replaces let an OLD process
5205
+ treat a NEWER stored fingerprint as a mismatch exactly as a new process
5206
+ treats an older one, so it rewrote every session's cost from its stale
5207
+ table and stamped the fingerprint backwards.
5208
+
5209
+ Absent or empty is older: a fresh store has nothing to protect. Anything
5210
+ that does not parse as an ISO date — on EITHER side — is refused rather
5211
+ than assumed old. This guard exists because a writer assumed its own table
5212
+ beat the store's, and a value it cannot parse was written by a version
5213
+ whose format it does not understand, which is precisely where that
5214
+ assumption is least defensible.
5215
+
5216
+ Ordering is DAY-GRANULAR because PRICING_SNAPSHOT_DATE is a calendar date.
5217
+ Two pricing revisions inside one UTC day compare equal and the older
5218
+ process is authorized. `bin/_lib_pricing.py` records the contract that
5219
+ makes this sound: a pricing revision must always ADVANCE the date.
5220
+
5221
+ The parse itself lives in `_lib_pricing.parse_pricing_fingerprint`, because
5222
+ `doctor pricing.conversation_rollup_writer` reports which refusal state a
5223
+ store is in and must classify a value exactly as this function acts on it.
5224
+ Two parses of one contract had already drifted apart once."""
5225
+ process = PRICING_SNAPSHOT_DATE if process is None else process
5226
+ if not stored:
5227
+ return True
5228
+ stored_date = parse_pricing_fingerprint(stored)
5229
+ process_date = parse_pricing_fingerprint(process)
5230
+ if stored_date is None or process_date is None:
5231
+ return False
5232
+ return stored_date <= process_date
5233
+
5234
+
5235
+ def _record_pricing_write_refusal(conn, stored) -> None:
5236
+ """Latch that a process holding older pricing was refused a write (#705).
5237
+
5238
+ Written only when the (process, store) pair DIFFERS from what is already
5239
+ recorded. A dashboard ticks continuously, so rewriting per tick would take
5240
+ the conversations writer lock purely to restate an unchanged fact — and it
5241
+ would make the timestamp the latest refusal rather than the first, which
5242
+ is the less useful of the two, because the first says how long the store
5243
+ has been diverging.
5244
+
5245
+ NEVER commits — the caller owns the transaction. _prune_orphaned_cache_entries
5246
+ reaches this from inside its own explicit ``BEGIN``, whose
5247
+ ``except BaseException: rollback()`` must still be able to undo the three
5248
+ message/touch/title DELETEs that precede it."""
5249
+ pair = {
5250
+ "process_snapshot_date": PRICING_SNAPSHOT_DATE,
5251
+ "store_snapshot_date": stored,
5252
+ }
5253
+ try:
5254
+ row = conn.execute(
5255
+ "SELECT value FROM cache_meta WHERE key=?",
5256
+ (CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,),
5257
+ ).fetchone()
5258
+ except sqlite3.OperationalError:
5259
+ return
5260
+ if row and row[0]:
5261
+ try:
5262
+ existing = json.loads(row[0])
5263
+ except (ValueError, TypeError):
5264
+ existing = None # unreadable -> replace it with a readable one
5265
+ if isinstance(existing, dict) and all(
5266
+ existing.get(field) == value for field, value in pair.items()
5267
+ ):
5268
+ return
5269
+ record = dict(pair)
5270
+ record["first_refused_at_utc"] = (
5271
+ dt.datetime.now(dt.timezone.utc)
5272
+ .replace(microsecond=0).isoformat().replace("+00:00", "Z")
5273
+ )
5274
+ try:
5275
+ _set_cache_meta(
5276
+ conn, CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,
5277
+ json.dumps(record, sort_keys=True),
5278
+ )
5279
+ except sqlite3.OperationalError:
5280
+ # Symmetric with the SELECT above: a locked or schema-less store must
5281
+ # not raise out of a path that only latches a diagnostic.
5282
+ return
5283
+
5284
+
5285
+ def _clear_pricing_write_refusal(conn) -> None:
5286
+ """Drop the refusal latch, unconditionally. NEVER commits — the caller owns
5287
+ the transaction, like the record and arm helpers beside it.
5288
+
5289
+ TWO callers, and the condition each satisfies before calling is the whole
5290
+ contract. The two pending-flag branches call it directly, atomically with
5291
+ consuming ``conversation_sessions_backfill_pending`` — the flag is deleted
5292
+ in the same transaction one statement earlier, so the store is converged by
5293
+ the time this runs. `_clear_converged_pricing_write_refusal` calls it for
5294
+ every OTHER completed recompute, and checks the flag itself because those
5295
+ paths do not consume it.
5296
+
5297
+ The earlier contract was "the pending-flag branch and nothing else", and it
5298
+ made one record unreachable. The rebuild pre-clear refuses BEFORE anything
5299
+ destructive and therefore arms no flag (correctly — the rollup it declines
5300
+ to touch is intact), so no flag branch would ever run over the record it
5301
+ latched, while `doctor pricing.conversation_rollup_writer` promised the
5302
+ operator that the next tick would clear it."""
5303
+ try:
5304
+ conn.execute(
5305
+ "DELETE FROM cache_meta WHERE key=?",
5306
+ (CONVERSATION_ROLLUP_PRICING_REFUSED_KEY,),
5307
+ )
5308
+ except sqlite3.OperationalError:
5309
+ pass
5310
+
5311
+
5312
+ def _clear_converged_pricing_write_refusal(conn, *, authorize: bool) -> None:
5313
+ """Drop the refusal latch after an AUTHORIZED recompute completed over a
5314
+ store with no backfill pending (#705).
5315
+
5316
+ Both conditions carry weight. ``authorize`` is False only for the
5317
+ read-scope TEMP projection, and the gate is what stops a READ from deleting
5318
+ a real diagnostic. That projection shadows ``conversation_sessions`` with a
5319
+ TEMP table but defines no TEMP ``cache_meta``, so the unqualified ``DELETE``
5320
+ in ``_clear_pricing_write_refusal`` resolves past ``temp`` to ``main`` —
5321
+ which on that connection is conversations.db, the writable store whose own
5322
+ refusal record this is. It does NOT reach the ``?mode=ro`` ``cache_db``
5323
+ attachment, and reading it that way makes the gate look removable: a write
5324
+ to a read-only attachment raises ``OperationalError``, which the handler one
5325
+ frame down swallows, so the ungated clear would look harmless while actually
5326
+ deleting the diagnostic from ``main``. And a set backfill flag means the
5327
+ recompute that converges this store has not run yet, so clearing here would
5328
+ declare convergence early; that case belongs to the two flag branches, which
5329
+ clear atomically with consuming the flag.
5330
+
5331
+ NEVER commits. Every production caller of ``_recompute_conversation_sessions``
5332
+ commits the transaction this runs in: ``_prune_orphaned_cache_entries``
5333
+ inside its own ``BEGIN``, the legacy bridge at the end of its import (or
5334
+ rolls the whole import back), and both sync functions on the scoped branch
5335
+ immediately after the call."""
5336
+ if not authorize or _conversation_sessions_backfill_pending(conn):
5337
+ return
5338
+ _clear_pricing_write_refusal(conn)
5339
+
5340
+
5341
+ def _arm_rollup_backfill_on_pricing_change(conn) -> bool:
5108
5342
  """Arm the conversation_sessions full backfill when the embedded pricing
5109
5343
  snapshot changed since the rollup's stored cost was last derived (#302). The
5110
5344
  rail now reads MATERIALIZED cost off the rollup, so a pricing sync / cctally
5111
5345
  upgrade would otherwise leave untouched sessions' cost (and the cost
5112
5346
  filter/sort axis) stale until a manual `cache-sync --rebuild`. This self-heals
5113
5347
  it: compares a stored cache_meta fingerprint against the current
5114
- PRICING_SNAPSHOT_DATE and, on mismatch, arms
5115
- conversation_sessions_backfill_pending + advances the stored fingerprint (one
5116
- committed txn). The existing full-recompute-then-drop-flag-last machinery then
5117
- re-derives every session's cost + enrichment.
5348
+ PRICING_SNAPSHOT_DATE through the ORDERED _pricing_write_authorized and, when
5349
+ the store is OLDER, arms conversation_sessions_backfill_pending + advances the
5350
+ stored fingerprint (one committed txn). The existing
5351
+ full-recompute-then-drop-flag-last machinery then re-derives every session's
5352
+ cost + enrichment.
5353
+
5354
+ A stored fingerprint that is NEWER than this process's, or that does not
5355
+ parse, is REFUSED rather than treated as a mismatch (#705). The bare equality
5356
+ this replaced made an old process rewrite every session's cost from its stale
5357
+ table and stamp the fingerprint backwards; the refusal latches a cache_meta
5358
+ record instead, which `doctor pricing.conversation_rollup_writer` reports,
5359
+ AND arms the backfill flag so the rail degrades to live aggregation instead
5360
+ of reading an un-recomputed rollup as authoritative.
5361
+
5362
+ Returns whether this process is authorized to write materialized cost to
5363
+ this store, so the caller can report a refusal (`deferred_reason`) rather
5364
+ than finish as a silent success.
5118
5365
 
5119
5366
  Crash-safety is unchanged: the DURABLE backfill flag remains the recompute
5120
5367
  signal, so advancing the fingerprint here cannot strand stale cost (a crash
5121
5368
  after arming leaves the flag set -> next sync recomputes regardless of the
5122
5369
  fingerprint). No-op when cache_meta is unavailable (path-less / degraded
5123
- conn). Caller path holds the cache.db.lock flock."""
5370
+ conn). Caller path holds the cache.db.lock flock. Unlike the recompute
5371
+ chokepoint, this helper owns its OWN transaction and commits, so the flag
5372
+ and the refusal record are durable for the next process."""
5124
5373
  try:
5125
5374
  row = conn.execute(
5126
- "SELECT value FROM cache_meta "
5127
- "WHERE key='conversation_sessions_pricing_fp'"
5375
+ "SELECT value FROM cache_meta WHERE key=?",
5376
+ (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5128
5377
  ).fetchone()
5129
5378
  except sqlite3.OperationalError:
5130
- return
5131
- if row is not None and row[0] == PRICING_SNAPSHOT_DATE:
5132
- return
5379
+ return True
5380
+ stored = row[0] if row is not None else None
5381
+ if not _pricing_write_authorized(stored):
5382
+ _record_pricing_write_refusal(conn, stored)
5383
+ _arm_rollup_backfill_pending(conn)
5384
+ conn.commit()
5385
+ return False
5386
+ if stored == PRICING_SNAPSHOT_DATE:
5387
+ return True
5133
5388
  _set_cache_meta(conn, "conversation_sessions_backfill_pending", "1")
5134
- _set_cache_meta(conn, "conversation_sessions_pricing_fp", PRICING_SNAPSHOT_DATE)
5389
+ _set_cache_meta(
5390
+ conn, CONVERSATION_ROLLUP_PRICING_FP_KEY, PRICING_SNAPSHOT_DATE)
5135
5391
  conn.commit()
5392
+ return True
5136
5393
 
5137
5394
 
5138
5395
  def _recompute_conversation_sessions(
5139
5396
  conn, session_ids=None, *, advance_render_revision: bool = True,
5140
- ) -> None:
5397
+ authorize: bool = True,
5398
+ ) -> bool:
5141
5399
  """Recompute the ``conversation_sessions`` browse-rail rollup from
5142
5400
  ``conversation_messages``. The caller holds the cache.db.lock flock and owns
5143
5401
  the commit (this helper never commits).
@@ -5157,10 +5415,74 @@ def _recompute_conversation_sessions(
5157
5415
 
5158
5416
  The recomputed COUNT/MIN/MAX are byte-identical to the rail's prior live
5159
5417
  aggregate over the same rows — that is the load-bearing invariant
5160
- (assert_rollup_matches_live in the maintenance test pins it)."""
5418
+ (assert_rollup_matches_live in the maintenance test pins it).
5419
+
5420
+ This is the ORDERED-WRITE CHOKEPOINT (#705). Both branches materialize
5421
+ ``cost_usd`` from this process's pricing table, so authorization is consulted
5422
+ HERE, before either branch deletes a row, and the function returns whether it
5423
+ actually recomputed. Guarding only _arm_rollup_backfill_on_pricing_change is
5424
+ insufficient, because the steady-state SCOPED path and
5425
+ _prune_orphaned_cache_entries reach the write without passing through it;
5426
+ guarding _fill_conversation_sessions_filter_columns instead is too late,
5427
+ because rows are by then deleted and reinserted at the schema's default-zero
5428
+ cost, which is the 0.0 the issue reports. A refused process writes no rollup
5429
+ row at all and ARMS the durable backfill flag for the next authorized one,
5430
+ which is what makes the rail degrade to live aggregation rather than read a
5431
+ wrong value — or, for a session this refused process just ingested, read no
5432
+ value at all. Arming is required rather than optional: in the scenario #705
5433
+ describes the newer process COMPLETED, so it cleared the flag, and a refusal
5434
+ that merely declined to write would leave _rollup_authoritative() true over
5435
+ a rollup missing every newly ingested session.
5436
+
5437
+ A COMPLETED authorized recompute over a store with no backfill pending
5438
+ clears the refusal record, through
5439
+ `_clear_converged_pricing_write_refusal` on both successful returns. That
5440
+ is what makes a record latched by the rebuild pre-clear — the one refusal
5441
+ that arms no flag — reachable at all; without it no flag branch would ever
5442
+ run over that record and doctor warned indefinitely.
5443
+
5444
+ ``authorize=False`` is for a connection whose ``conversation_sessions`` is a
5445
+ connection-local TEMP projection: it mutates no persistent row, so the
5446
+ persistent-store guard does not apply to it. The clear is gated for the
5447
+ OPPOSITE reason — its unqualified ``cache_meta`` is not shadowed and would
5448
+ reach ``main``, so on that connection it is the one statement here that
5449
+ does touch a persistent row."""
5161
5450
  ids = None if session_ids is None else [s for s in session_ids if s is not None]
5162
5451
  if ids == []:
5163
- return
5452
+ return True
5453
+ if authorize:
5454
+ try:
5455
+ row = conn.execute(
5456
+ "SELECT value FROM cache_meta WHERE key=?",
5457
+ (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
5458
+ ).fetchone()
5459
+ except sqlite3.OperationalError:
5460
+ row = None
5461
+ stored = row[0] if row else None
5462
+ if not _pricing_write_authorized(stored):
5463
+ _record_pricing_write_refusal(conn, stored)
5464
+ _arm_rollup_backfill_pending(conn)
5465
+ # No commit: this helper documents that the CALLER owns it, and
5466
+ # every caller that can reach a refusal commits — the pruner inside
5467
+ # its own BEGIN, the legacy bridge, and both sync paths on both the
5468
+ # flag branch and the scoped branch.
5469
+ #
5470
+ # Do NOT weaken any of those to "the arm helper already committed
5471
+ # the same two rows on this tick". That was the argument for the
5472
+ # two flag branches, and it holds only while the arm helper's read
5473
+ # of the fingerprint and the read below AGREE. They diverge two
5474
+ # ways. _arm_rollup_backfill_on_pricing_change returns True and
5475
+ # writes nothing when its own SELECT raises OperationalError, and
5476
+ # `database is locked` is transient, so the read below can succeed
5477
+ # against a newer stored value and refuse. And another process can
5478
+ # advance the fingerprint between the two reads — which
5479
+ # _import_legacy_conversation_rows made more reachable, because it
5480
+ # stamps CONVERSATION_ROLLUP_PRICING_FP_KEY at DB open holding only
5481
+ # the shared maintenance lock, never the conversations writer flock.
5482
+ # In either case this writes a genuinely new record,
5483
+ # _arm_rollup_backfill_pending short-circuits on the already-set
5484
+ # flag, and nothing else commits.
5485
+ return False
5164
5486
  render_revision = (
5165
5487
  _next_conversation_render_revision(conn)
5166
5488
  if advance_render_revision else 0
@@ -5177,7 +5499,8 @@ def _recompute_conversation_sessions(
5177
5499
  "UPDATE conversation_sessions SET render_revision=?",
5178
5500
  (render_revision,),
5179
5501
  )
5180
- return
5502
+ _clear_converged_pricing_write_refusal(conn, authorize=authorize)
5503
+ return True
5181
5504
  for i in range(0, len(ids), 400):
5182
5505
  chunk = ids[i:i + 400]
5183
5506
  placeholders = ",".join("?" for _ in chunk)
@@ -5198,6 +5521,8 @@ def _recompute_conversation_sessions(
5198
5521
  (render_revision, *chunk),
5199
5522
  )
5200
5523
  _fill_conversation_sessions_filter_columns(conn, ids)
5524
+ _clear_converged_pricing_write_refusal(conn, authorize=authorize)
5525
+ return True
5201
5526
 
5202
5527
 
5203
5528
  def _fill_conversation_sessions_filter_columns(conn, session_ids):
@@ -10013,7 +10338,13 @@ def scope_conversations_db_to_account(
10013
10338
  )
10014
10339
  # These are read-scope TEMP projections. They must neither consume nor
10015
10340
  # advance the durable render frontier used by writer recomputes.
10016
- _recompute_conversation_sessions(conn, advance_render_revision=False)
10341
+ # authorize=False (#705): this connection shadows the leaves AND
10342
+ # conversation_sessions with connection-local TEMP objects, so the recompute
10343
+ # writes no persistent row. The ordered-write guard protects the persistent
10344
+ # store's materialized cost, so it does not apply here — and applying it
10345
+ # would make a read-scope projection refuse and record a spurious refusal.
10346
+ _recompute_conversation_sessions(
10347
+ conn, advance_render_revision=False, authorize=False)
10017
10348
  codex_keys = {
10018
10349
  row[0]
10019
10350
  for row in conn.execute(
@@ -10530,6 +10861,28 @@ def read_session_titles_bounded(
10530
10861
  maintenance_fh.close()
10531
10862
 
10532
10863
 
10864
+ #: Tables the pre-028 compatibility bridge copies out of the attached legacy
10865
+ #: store. ``conversation_sessions`` is deliberately ABSENT (#705): it is the
10866
+ #: browse-rail rollup, it carries materialized ``cost_usd``, and the bridge
10867
+ #: copies rows verbatim without any pricing comparison and without copying the
10868
+ #: fingerprint that would describe the cost it installs. The rollup is fully
10869
+ #: derivable from ``conversation_messages``, which the bridge does import, so
10870
+ #: the bridge DERIVES the rollup instead, through the ordered-write chokepoint,
10871
+ #: and stamps the fingerprint for the pricing table that derive used. (It armed
10872
+ #: the backfill and derived nothing in an earlier revision; that left a
10873
+ #: ``dashboard --no-sync`` reader with an empty rollup for its whole life. See
10874
+ #: _import_legacy_conversation_rows.)
10875
+ _LEGACY_BRIDGE_TABLES = (
10876
+ "conversation_messages",
10877
+ "conversation_ai_titles",
10878
+ "conversation_file_touches",
10879
+ "codex_conversation_events",
10880
+ "codex_conversation_messages",
10881
+ "codex_conversation_file_touches",
10882
+ "codex_conversation_rollups",
10883
+ )
10884
+
10885
+
10533
10886
  def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10534
10887
  """Bridge pre-028/compatibility rows into an empty conversation store.
10535
10888
 
@@ -10537,19 +10890,29 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10537
10890
  defensive bridge covers an interrupted upgrade and keeps historical test
10538
10891
  fixtures readable without making core sync depend on the transcript DB.
10539
10892
  It writes only the main conversation store; ``cache_db`` is attached RO.
10893
+
10894
+ It does NOT copy ``conversation_sessions`` (#705). That table is the
10895
+ browse-rail rollup and it carries materialized ``cost_usd``; copying it
10896
+ bypassed _recompute_conversation_sessions entirely, carried no pricing
10897
+ comparison, and copied no fingerprint describing the cost it installed. It
10898
+ DERIVES the rollup instead, through the ordered-write chokepoint, from the
10899
+ conversation_messages it just imported, so the installed cost comes from
10900
+ this process's pricing table rather than an unknown one — and it records
10901
+ which table that was in ``CONVERSATION_ROLLUP_PRICING_FP_KEY``, so the
10902
+ provenance is written down rather than merely implied.
10903
+
10904
+ Deriving is required, not merely tidier: this runs at DB OPEN, and
10905
+ ``dashboard --no-sync`` never runs a sync, so merely arming
10906
+ ``conversation_sessions_backfill_pending`` left the rollup EMPTY and
10907
+ non-authoritative for the life of that process. The rail itself survives
10908
+ that (the flag routes it to live aggregation) but
10909
+ ``list_conversation_facets`` reads the rollup's ``project_label`` directly
10910
+ with no authoritative gate, so the browse filter's project list went empty;
10911
+ and every rail read fell to the live branch, which is not the branch the
10912
+ materialized-cost contract is about.
10540
10913
  """
10541
- tables = (
10542
- "conversation_messages",
10543
- "conversation_ai_titles",
10544
- "conversation_sessions",
10545
- "conversation_file_touches",
10546
- "codex_conversation_events",
10547
- "codex_conversation_messages",
10548
- "codex_conversation_file_touches",
10549
- "codex_conversation_rollups",
10550
- )
10551
10914
  changed = False
10552
- for table in tables:
10915
+ for table in _LEGACY_BRIDGE_TABLES:
10553
10916
  try:
10554
10917
  if conn.execute(f"SELECT 1 FROM main.{table} LIMIT 1").fetchone():
10555
10918
  continue
@@ -10574,6 +10937,48 @@ def _import_legacy_conversation_rows(conn: sqlite3.Connection) -> None:
10574
10937
  except sqlite3.Error:
10575
10938
  continue
10576
10939
  if changed:
10940
+ # The rollup is DERIVED, never inherited. A full recompute here is
10941
+ # bounded by the rows just imported (the bridge runs only into an empty
10942
+ # store), and it goes through the ordered-write chokepoint, so the cost
10943
+ # it installs comes from THIS process's pricing table — the provenance
10944
+ # the copied cost lacked. Deliberately broad: a Codex-only import also
10945
+ # re-derives the Claude rollup, which costs one recompute over an empty
10946
+ # or tiny table and cannot under-derive.
10947
+ #
10948
+ # No unconditional arm. A completed full recompute IS what the flag
10949
+ # asks for, so arming after one would only book a redundant repeat —
10950
+ # and, because the bridge runs at DB OPEN, a `dashboard --no-sync`
10951
+ # reader would carry that flag for its whole life and read the rail's
10952
+ # LIVE fallback forever. A refused process is the case the flag is for,
10953
+ # and _recompute_conversation_sessions arms it itself on refusal.
10954
+ if _recompute_conversation_sessions(conn):
10955
+ # Stamp the fingerprint for the table this derive actually used.
10956
+ # Deriving cost and leaving an OLDER fingerprint recorded re-opens
10957
+ # #705 one step down the line: a store recording 2026-08-01 bridged
10958
+ # by a 2026-09-02 process would hold 2026-09-02 cost under a
10959
+ # 2026-08-01 fingerprint, and a later 2026-08-15 process would read
10960
+ # 2026-08-01 <= 2026-08-15, consider itself authorized, and rewrite
10961
+ # the rollup from its older table. Only on the AUTHORIZED branch: a
10962
+ # refused derive wrote no row, so claiming provenance for it would
10963
+ # be false and would backdate the store as well.
10964
+ try:
10965
+ _set_cache_meta(
10966
+ conn, CONVERSATION_ROLLUP_PRICING_FP_KEY,
10967
+ PRICING_SNAPSHOT_DATE)
10968
+ except sqlite3.OperationalError:
10969
+ # Fail-soft like every other statement in this bridge — this
10970
+ # runs at DB OPEN, so a degraded store must still open — but
10971
+ # fail-soft by DISCARDING, never by committing the pair apart.
10972
+ # The recompute above materialized cost from THIS process's
10973
+ # pricing table; committing it while the store keeps whatever
10974
+ # older fingerprint it already records is the exact sequence the
10975
+ # stamp exists to prevent (#705), and a later intermediate
10976
+ # process reads that older value as authorization to overwrite
10977
+ # the rollup from its own staler table. The bridge is idempotent
10978
+ # and runs at every open, so rolling the whole import back
10979
+ # leaves the store as it was and the next open retries it.
10980
+ conn.rollback()
10981
+ return
10577
10982
  conn.commit()
10578
10983
 
10579
10984
 
@@ -10825,6 +11230,51 @@ def sync_claude_conversations(
10825
11230
  return stats
10826
11231
  rebuild = rebuild or pending_rebuild
10827
11232
  if rebuild:
11233
+ # #705: authorize BEFORE anything destructive. This branch deletes
11234
+ # conversation_sessions long before the rollup path's pricing check
11235
+ # is reached, so guarding only _recompute_conversation_sessions
11236
+ # would let a process holding an older pricing table destroy
11237
+ # correct materialized cost and only THEN be refused the re-derive,
11238
+ # leaving an EMPTY rollup — strictly worse than the overwrite this
11239
+ # issue is about. A rebuild is an explicit maintenance operation;
11240
+ # performing half of it from a stale binary is worse than not
11241
+ # performing it at all, and deferred_reason is how the caller
11242
+ # learns it did nothing.
11243
+ try:
11244
+ fp_row = conn.execute(
11245
+ "SELECT value FROM cache_meta WHERE key=?",
11246
+ (CONVERSATION_ROLLUP_PRICING_FP_KEY,),
11247
+ ).fetchone()
11248
+ except sqlite3.OperationalError:
11249
+ fp_row = None
11250
+ stored_fp = fp_row[0] if fp_row else None
11251
+ if not _pricing_write_authorized(stored_fp):
11252
+ _record_pricing_write_refusal(conn, stored_fp)
11253
+ # COMMIT. _record_pricing_write_refusal leaves the transaction
11254
+ # to its caller, and this caller returns immediately, so
11255
+ # without this the record dies with the connection. The
11256
+ # function's other refusal sites are already covered — the
11257
+ # rollup block's are behind _arm_rollup_backfill_on_pricing_change,
11258
+ # which owns and commits its own transaction — and this one was
11259
+ # missed when the commit moved out of the helper.
11260
+ # `_run_transcript_rebuild_worker` runs the rebuild in a forked
11261
+ # child that closes the connection and calls os._exit(0) the
11262
+ # moment this returns, and sqlite3's implicit-transaction mode
11263
+ # discards an uncommitted write on close, so
11264
+ # `doctor pricing.conversation_rollup_writer` reported OK after
11265
+ # every refused `cache-sync --rebuild`.
11266
+ #
11267
+ # Do NOT arm conversation_sessions_backfill_pending here. This
11268
+ # is the one refusal path that runs BEFORE anything destructive,
11269
+ # so the rollup it declines to touch is intact and was derived
11270
+ # by a newer, authorized process — it is genuinely
11271
+ # authoritative. Arming would force every rail read onto live
11272
+ # aggregation to protect a rollup that needs no protection. The
11273
+ # other refusal sites arm precisely because they refuse a write
11274
+ # the store still needs.
11275
+ conn.commit()
11276
+ stats.deferred_reason = "pricing_write_refused"
11277
+ return stats
10828
11278
  # Commit the retry marker before the destructive clear. A killed
10829
11279
  # #395 worker therefore leaves a partial transcript store visibly
10830
11280
  # pending instead of advancing it to a false-complete state.
@@ -11005,17 +11455,44 @@ def sync_claude_conversations(
11005
11455
  _report_conversation_progress(progress, "ingest", stats)
11006
11456
 
11007
11457
  _report_conversation_progress(progress, "rollup", stats)
11008
- _arm_rollup_backfill_on_pricing_change(conn)
11458
+ rollup_authorized = _arm_rollup_backfill_on_pricing_change(conn)
11009
11459
  if _conversation_sessions_backfill_pending(conn):
11010
- _recompute_conversation_sessions(conn)
11011
- conn.execute(
11012
- "DELETE FROM cache_meta "
11013
- "WHERE key='conversation_sessions_backfill_pending'"
11014
- )
11015
- conn.commit()
11460
+ # #705: a refused process must NOT consume the flag. It leaves it
11461
+ # set for the next authorized process, so a newer process that armed
11462
+ # the backfill and died is not followed by a stale successor either
11463
+ # rewriting cost from its old table or declaring an unrecomputed
11464
+ # rollup authoritative.
11465
+ if _recompute_conversation_sessions(conn):
11466
+ conn.execute(
11467
+ "DELETE FROM cache_meta "
11468
+ "WHERE key='conversation_sessions_backfill_pending'"
11469
+ )
11470
+ _clear_pricing_write_refusal(conn)
11471
+ conn.commit()
11472
+ else:
11473
+ # #705: same as the sync_cache twin — commit the refusal record
11474
+ # and the armed flag here rather than relying on the trailing
11475
+ # `conversation_rebuild_claude_pending` clear, which is skipped
11476
+ # whenever a single file failed to ingest.
11477
+ conn.commit()
11478
+ rollup_authorized = False
11016
11479
  elif touched_sessions:
11017
- _recompute_conversation_sessions(conn, touched_sessions)
11480
+ if not _recompute_conversation_sessions(conn, touched_sessions):
11481
+ rollup_authorized = False
11018
11482
  conn.commit()
11483
+ if not rollup_authorized and stats.deferred_reason is None:
11484
+ # #705: the rebuild pre-clear was not the only refusal this call can
11485
+ # take, and a refusal on ANY of them leaves the rollup non-derived.
11486
+ # `deferred_reason` is the one channel the CLI ladder and
11487
+ # `_lib_ingest_frontier.conversation_sync_certifiable` both read, so
11488
+ # setting it here is what stops a refused sync being reported as a
11489
+ # success and being certified as caught up. The `is None` guard is
11490
+ # DEFENSIVE, not load-bearing today: every earlier assignment in
11491
+ # this function returns immediately, so no earlier reason can reach
11492
+ # this line. It preserves the first reason if one of those sites
11493
+ # ever stops returning — the earlier, more specific account of the
11494
+ # same incomplete result should win.
11495
+ stats.deferred_reason = "pricing_write_refused"
11019
11496
  if only_paths is None and stats.files_failed == 0:
11020
11497
  conn.execute(
11021
11498
  "DELETE FROM cache_meta "
@@ -12179,6 +12656,27 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
12179
12656
  f"core accounting/quota sync is complete. {retry}"
12180
12657
  )
12181
12658
  contended = True
12659
+ elif conv_stats.deferred_reason == "pricing_write_refused":
12660
+ # #705: `--rebuild` is the operator's recovery path, so this is
12661
+ # the worst place for a silent success. Without this branch a
12662
+ # refused rebuild fell through to the `done: 0 processed` line
12663
+ # below and exited 0.
12664
+ eprint(
12665
+ "[cache-sync] transcript rebuild incomplete: "
12666
+ f"provider={provider} store=conversations.db phase=pricing "
12667
+ "(cctally may not rewrite materialized conversation cost "
12668
+ "from a pricing table it cannot prove is at least the "
12669
+ "store's: the fingerprint recorded there is either newer "
12670
+ "than this cctally's or not an ISO date any version can "
12671
+ "order); core accounting/quota sync is complete. Upgrade "
12672
+ "or restart cctally so it carries at least the store's "
12673
+ "pricing table. When the recorded value is not a date, no "
12674
+ "upgrade or restart clears it — run `cctally doctor`, "
12675
+ "which reports that state under "
12676
+ "`pricing.conversation_rollup_writer` and names the step "
12677
+ f"that does. {retry}"
12678
+ )
12679
+ contended = True
12182
12680
  else:
12183
12681
  eprint(
12184
12682
  f"[cache-sync] {provider} transcripts done: "
@@ -12221,6 +12719,17 @@ def cmd_cache_sync(args: argparse.Namespace) -> int:
12221
12719
  "another process holds the conversations lock"
12222
12720
  )
12223
12721
  else:
12722
+ # #705: a routine sync reports success even when
12723
+ # `deferred_reason` is "pricing_write_refused", and that
12724
+ # asymmetry with `--rebuild` above is DELIBERATE. Message
12725
+ # ingestion did succeed here — only the rollup re-derive was
12726
+ # declined, and the armed backfill flag routes the rail to
12727
+ # live aggregation meanwhile, so nothing the user asked for
12728
+ # failed. `--rebuild` is the opposite case: the operator
12729
+ # asked for exactly the re-derive that was refused, so it
12730
+ # exits non-zero. `doctor
12731
+ # pricing.conversation_rollup_writer` is where the refusal
12732
+ # surfaces for a routine sync.
12224
12733
  eprint(
12225
12734
  f"[cache-sync] claude transcripts done: "
12226
12735
  f"{conv_stats.files_processed} processed, "