cctally 1.99.1 → 1.101.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +88 -0
  2. package/bin/_cctally_alerts.py +13 -2
  3. package/bin/_cctally_cache.py +3 -1
  4. package/bin/_cctally_cache_report.py +103 -6
  5. package/bin/_cctally_dashboard.py +2083 -356
  6. package/bin/_cctally_dashboard_envelope.py +115 -31
  7. package/bin/_cctally_dashboard_perf.py +433 -0
  8. package/bin/_cctally_dashboard_share.py +101 -29
  9. package/bin/_cctally_dashboard_sources.py +1072 -214
  10. package/bin/_cctally_db.py +30 -14
  11. package/bin/_cctally_diff.py +20 -0
  12. package/bin/_cctally_doctor.py +1331 -1148
  13. package/bin/_cctally_forecast.py +304 -96
  14. package/bin/_cctally_milestone_history.py +10 -2
  15. package/bin/_cctally_parser.py +70 -1
  16. package/bin/_cctally_project.py +155 -47
  17. package/bin/_cctally_quota.py +40 -21
  18. package/bin/_cctally_record.py +44 -3
  19. package/bin/_cctally_refresh.py +60 -10
  20. package/bin/_cctally_share.py +9 -2
  21. package/bin/_cctally_source_analytics.py +40 -4
  22. package/bin/_cctally_statusline.py +53 -7
  23. package/bin/_cctally_tui.py +723 -232
  24. package/bin/_cctally_update.py +28 -22
  25. package/bin/_lib_alert_scope.py +685 -0
  26. package/bin/_lib_alerts_payload.py +112 -7
  27. package/bin/_lib_cache_report.py +110 -1
  28. package/bin/_lib_codex_pools.py +20 -8
  29. package/bin/_lib_dashboard_sources.py +237 -30
  30. package/bin/_lib_doctor.py +37 -0
  31. package/bin/_lib_forecast.py +12 -4
  32. package/bin/_lib_jsonl.py +4 -2
  33. package/bin/_lib_perf.py +132 -3
  34. package/bin/_lib_pricing.py +8 -7
  35. package/bin/_lib_render.py +31 -3
  36. package/bin/_lib_share_templates.py +150 -55
  37. package/bin/_lib_snapshot_cache.py +71 -13
  38. package/bin/_lib_source_analytics.py +2 -2
  39. package/bin/_lib_source_identity.py +50 -2
  40. package/bin/_lib_subscription_weeks.py +65 -0
  41. package/bin/_lib_tick_stats.py +538 -0
  42. package/bin/cctally +29 -7
  43. package/dashboard/static/assets/dashboardStream.shared-worker-1XTMV3nr.js +1 -0
  44. package/dashboard/static/assets/index-D6Eb9KDn.js +97 -0
  45. package/dashboard/static/assets/index-i3g7g8zo.css +1 -0
  46. package/dashboard/static/dashboard.html +2 -2
  47. package/package.json +4 -1
  48. package/dashboard/static/assets/index-C5NBB2w9.js +0 -97
  49. package/dashboard/static/assets/index-hJP4wlIO.css +0 -1
@@ -191,11 +191,10 @@ What stays in bin/cctally:
191
191
  ``ns["_build_current_week_share_panel_data"]``,
192
192
  ``ns["_build_daily_share_panel_data"]``,
193
193
  ``ns["_build_monthly_share_panel_data"]``,
194
- ``ns["_build_blocks_share_panel_data"]``, ``ns["STATIC_DIR"]``,
195
- ``ns["_DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS"]``, plus
194
+ ``ns["_build_blocks_share_panel_data"]``, ``ns["STATIC_DIR"]``, plus
196
195
  ``monkeypatch.setitem`` mutations on
197
- ``_dashboard_build_weekly_periods``, ``_dashboard_build_blocks_panel``,
198
- and ``_DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS``). Forces the **eager
196
+ ``_dashboard_build_weekly_periods`` and
197
+ ``_dashboard_build_blocks_panel``). Forces the **eager
199
198
  re-export** carve-out per spec §4.8 (same precedent as Phase E
200
199
  #19/#20 + Phase F #21):
201
200
 
@@ -266,6 +265,7 @@ import copy
266
265
  import contextlib
267
266
  import dataclasses
268
267
  import datetime as dt
268
+ import gzip
269
269
  import hmac
270
270
  import io
271
271
  import json
@@ -285,6 +285,7 @@ import urllib.error
285
285
  import urllib.parse
286
286
  import urllib.request
287
287
  import webbrowser as _wb
288
+ import zlib
288
289
  from dataclasses import dataclass, field, replace
289
290
  from collections.abc import Mapping
290
291
  from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -419,7 +420,8 @@ from _lib_dashboard_settings_contract import (
419
420
  from _cctally_config import save_config, _load_config_unlocked
420
421
  from _cctally_db import _render_migration_error_banner
421
422
  from _cctally_cache import (
422
- get_entries, iter_entries, iter_entries_with_id, open_cache_db,
423
+ get_entries, iter_entries, iter_entries_with_id,
424
+ open_cache_db as _raw_open_cache_db,
423
425
  open_conversations_db, sync_cache, sync_claude_conversations,
424
426
  sync_codex_conversations,
425
427
  _prune_orphaned_cache_entries,
@@ -439,6 +441,65 @@ from _lib_snapshot_cache import (
439
441
  _max_id as _snapshot_max_id,
440
442
  _reset_sig as _snapshot_reset_sig,
441
443
  )
444
+ import _lib_tick_stats
445
+
446
+
447
+ # === F22a: count the silent Group A cache-open failures (#583 S1 §1.6) =====
448
+ # `_group_a_daily_buckets`, `_group_a_weekly_buckets` and
449
+ # `_group_a_monthly_buckets` each wrap their `open_cache_db()` in
450
+ # `try: … except Exception: return None`, where `None` means "fall back to the
451
+ # wide from-scratch fetch". Output stays byte-identical, so no golden moves and
452
+ # nothing surfaces — the failure is invisible even to `doctor`.
453
+ #
454
+ # The count is taken from OUTSIDE those three functions, for two reasons. From
455
+ # outside them a `None` return cannot distinguish a disabled cache, an open
456
+ # failure and a later fallback, so this wrapper is the only place the open
457
+ # failure is observable as itself. And session S5 owns those three helpers, so
458
+ # counting here leaves all three byte-for-byte unchanged and leaves no
459
+ # overlapping hunk to merge.
460
+
461
+ _GROUP_A_CACHE_OPENERS = (
462
+ ("_group_a_daily_buckets", "daily"),
463
+ ("_group_a_weekly_buckets", "weekly"),
464
+ ("_group_a_monthly_buckets", "monthly"),
465
+ )
466
+
467
+
468
+ def _group_a_cache_failure_kind(code):
469
+ """Which Group A bucket builder owns this code object, or None.
470
+
471
+ Matched by ``__code__`` IDENTITY, never by ``co_name``: a name match can be
472
+ satisfied by an unrelated function of the same name, and a diagnostic that
473
+ credits the wrong counter is worse than one that credits none. Resolved
474
+ against the live module globals rather than a memo so it cannot go stale.
475
+ """
476
+ for name, kind in _GROUP_A_CACHE_OPENERS:
477
+ if getattr(globals().get(name), "__code__", None) is code:
478
+ return kind
479
+ return None
480
+
481
+
482
+ def open_cache_db(*args, **kwargs):
483
+ """``_cctally_cache.open_cache_db``, plus the Group A failure count.
484
+
485
+ Behaviour-preserving. On an exception it identifies the caller, increments
486
+ the matching fixed counter, and re-raises the ORIGINAL exception unchanged.
487
+
488
+ It fails open in both directions: on no caller match, or on any failure of
489
+ the frame introspection itself, it increments nothing and still re-raises.
490
+ A diagnostic must never replace the error it was observing. Introspection
491
+ runs only on the already-exceptional path, so the steady-state cost is nil.
492
+ """
493
+ try:
494
+ return _raw_open_cache_db(*args, **kwargs)
495
+ except Exception:
496
+ try:
497
+ kind = _group_a_cache_failure_kind(sys._getframe(1).f_code)
498
+ if kind is not None:
499
+ _lib_tick_stats.note_cache_open_failure(kind)
500
+ except Exception: # noqa: BLE001 — never mask the original failure
501
+ pass
502
+ raise
442
503
 
443
504
 
444
505
  # === #279 S5: consumer-only dashboard siblings ============================
@@ -921,6 +982,8 @@ def _source_safe_claude_project_detail(
921
982
  "key": key,
922
983
  "label": label,
923
984
  "window_weeks": detail.get("window_weeks"),
985
+ "window_start_at": detail.get("window_start_at"),
986
+ "window_end_at": detail.get("window_end_at"),
924
987
  "window_cost_usd": detail.get("window_cost_usd"),
925
988
  "window_attributed_pct": detail.get("window_attributed_pct"),
926
989
  "models": detail.get("models", []),
@@ -1477,33 +1540,222 @@ def _next_deadline(t0: float, interval: float, work: float) -> float:
1477
1540
  return (t0 + work) + max(interval, work)
1478
1541
 
1479
1542
 
1543
+ def _conversation_next_deadline(
1544
+ t0: float, interval: float, work: float
1545
+ ) -> float:
1546
+ """Monotonic deadline for the next conversation sync pass (#583 S4 / F5).
1547
+
1548
+ Same algebra as `_next_deadline`, and deliberately a SEPARATE function. The
1549
+ two loops' bounds are independent regressions: sharing one helper would let
1550
+ a later change to the main loop's scheduling silently remove this thread's
1551
+ duty bound, which is the defect F5 exists to fix. A test asserts the two
1552
+ currently agree, so a divergence has to be a deliberate act.
1553
+
1554
+ work >= interval -> period = 2*work -> duty capped at 50% of one core,
1555
+ scale-independently. work < interval -> period = work + interval, which is
1556
+ the fixed-sleep cadence this replaced, so a small install sees no
1557
+ behavioural change.
1558
+ """
1559
+ return (t0 + work) + max(interval, work)
1560
+
1561
+
1562
+ def _log_sync_iteration_failure() -> None:
1563
+ """Route an escaped sync-iteration exception through the log chokepoint.
1564
+
1565
+ The traceback is the operator signal; the loop deliberately continues, so
1566
+ without this the failure would be entirely silent.
1567
+ """
1568
+ import traceback
1569
+ try:
1570
+ _lib_log.get_logger("dashboard").error(
1571
+ "sync iteration failed:\n%s", traceback.format_exc(),
1572
+ )
1573
+ except Exception: # noqa: BLE001 — logging must never kill the drainer
1574
+ pass
1575
+
1576
+
1577
+ def _make_dashboard_run_iteration(
1578
+ *, sync_lock, run_sync_now, run_sync_now_locked, skip_sync,
1579
+ monotonic=time.monotonic, heal_interval_seconds=60.0,
1580
+ ):
1581
+ """Return the dashboard sync thread's whole-iteration callable.
1582
+
1583
+ Module-level for the same reason ``_make_run_sync_now_locked`` is: the
1584
+ body is the only place the #583 S2 §4 contract "one OAuth refresh
1585
+ immediately before one rebuild, holding ``sync_lock`` across both" exists,
1586
+ and as a closure inside ``cmd_dashboard`` no test could reach it.
1587
+
1588
+ ``run_iteration`` performs one WHOLE iteration — the rebuild plus the
1589
+ orphan self-heal maintenance — so the loop's measured duration drives the
1590
+ cooldown deadline (#313 P2 / F10), not just the rebuild.
1591
+
1592
+ The refresh leg runs only for a batch carrying the refresh bit, and never
1593
+ under ``skip_sync``: ``--no-sync`` freezes the data and performs no network
1594
+ calls. Preserve 9's rule applies here rather than in the handler, so the
1595
+ periodic path cannot fire a redundant rebuild between the two steps, and
1596
+ many queued ``refresh=1`` requests collapse to one OAuth call because
1597
+ ``capture_batch`` ORs their intents.
1598
+
1599
+ ``_refresh_usage_inproc`` and ``_dashboard_self_heal_orphans`` are called
1600
+ by bare name on purpose, so a test patching either on this module (or on
1601
+ the ``cctally`` namespace the shim delegates to) reaches this body.
1602
+ """
1603
+ last_heal = [monotonic()]
1604
+
1605
+ def run_iteration(batch=None) -> dict:
1606
+ warnings: list = []
1607
+ if batch is not None and batch[1] and not skip_sync:
1608
+ with sync_lock:
1609
+ result = _refresh_usage_inproc()
1610
+ if result.status != "ok":
1611
+ warnings.append({"code": result.status})
1612
+ run_sync_now_locked(skip_sync=skip_sync)
1613
+ else:
1614
+ run_sync_now(skip_sync=skip_sync)
1615
+ # Self-heal removed-worktree orphans on a ~60s cadence (far rarer than
1616
+ # the sync tick — a deleted worktree is not urgent). Non-blocking on
1617
+ # the flock, so a contended tick just retries next cadence; gated off
1618
+ # under --no-sync.
1619
+ if (not skip_sync
1620
+ and monotonic() - last_heal[0] >= heal_interval_seconds):
1621
+ last_heal[0] = monotonic()
1622
+ _dashboard_self_heal_orphans(skip_sync=skip_sync)
1623
+ # A queued request has no HTTP response, so a deferred refresh's
1624
+ # warnings ride the settlement frame instead.
1625
+ return {"warnings": warnings}
1626
+
1627
+ return run_iteration
1628
+
1629
+
1630
+ def _make_sync_loop_collaborators(*, ref, hub) -> dict:
1631
+ """Bind a `_SnapshotRef` to an `SSEHub` for `_dashboard_sync_loop`.
1632
+
1633
+ Returns the loop's collaborator keyword arguments, which are the three
1634
+ publication points of #583 S2 spec §6.2 that the loop owns: point 2
1635
+ (``rebuilding=true`` with ``started_id`` advanced, immediately before work
1636
+ begins), point 4 (``rebuilding=false`` plus the settled fields), and the
1637
+ batchless equivalent of both. The reference re-stamps its held snapshot on
1638
+ every mutation, so ``ref.get()`` already carries the new counters.
1639
+
1640
+ Extracted to module level so a test can drive the loop through the SAME
1641
+ wiring the dashboard's sync thread uses. A test that rebuilt these three
1642
+ closures itself would assert only that its own copy publishes.
1643
+ """
1644
+ def capture_batch():
1645
+ batch = ref.capture_batch()
1646
+ hub.publish(ref.get())
1647
+ return batch
1648
+
1649
+ def settle(batch_id, status, warnings=()) -> None:
1650
+ ref.settle(batch_id, status, warnings)
1651
+ hub.publish(ref.get())
1652
+
1653
+ def mark_rebuilding(value) -> None:
1654
+ # Publish only on a real transition. A requested tick has already
1655
+ # published through capture_batch/settle, which set the same flag.
1656
+ if ref.mark_rebuilding(value):
1657
+ hub.publish(ref.get())
1658
+
1659
+ return {
1660
+ "pending_request": ref.pending_request,
1661
+ "capture_batch": capture_batch,
1662
+ "settle": settle,
1663
+ "mark_rebuilding": mark_rebuilding,
1664
+ }
1665
+
1666
+
1480
1667
  def _dashboard_sync_loop(
1481
1668
  *,
1482
1669
  stop,
1483
1670
  interval: float,
1484
1671
  run_iteration,
1485
- take_sync_request,
1672
+ take_sync_request=None,
1486
1673
  monotonic=time.monotonic,
1487
1674
  sleep=time.sleep,
1675
+ pending_request=None,
1676
+ capture_batch=None,
1677
+ settle=None,
1678
+ mark_rebuilding=None,
1488
1679
  ) -> None:
1489
- """Run the dashboard's automatic sync loop with a work-proportional cooldown.
1680
+ """Periodic rebuild loop with a #313-preserving request floor (#583 S2).
1490
1681
 
1491
1682
  ``run_iteration`` performs one whole automatic iteration (rebuild plus any
1492
1683
  orphan self-heal / retention maintenance the thread does), so its measured
1493
- duration — not just the rebuild — drives the deadline (F10). The manual
1494
- ``POST /api/sync`` refresh is a separate synchronous path under ``sync_lock``
1495
- and is unaffected; ``take_sync_request`` is the TUI force-refresh flag that
1496
- breaks the cooldown early.
1684
+ duration — not just the rebuild — drives the deadline (F10).
1685
+
1686
+ Automatic cadence is unchanged: ``_next_deadline(t0, interval, work)``. A
1687
+ queued request may start a batch earlier, but never before ``t0 + 2*work``,
1688
+ which is exactly the ``period >= 2*work`` that caps CPU duty at 50% of one
1689
+ core, scale-independently. When ``work >= interval`` the automatic deadline
1690
+ already equals ``t0 + 2*work``, so the floor coincides with it and a request
1691
+ changes nothing; the floor binds only when ``work < interval``, where it
1692
+ turns a 5.5 s wait into a 1.0 s one for a 0.5 s rebuild.
1693
+
1694
+ The floor is never later than the deadline, so a pending request is always
1695
+ serviced at or before the automatic tick it would otherwise wait for.
1696
+
1697
+ ``pending_request`` PEEKS and never consumes: a poll firing before the floor
1698
+ is met must not discard the request (spec §5.2). ``capture_batch`` claims
1699
+ every outstanding request atomically at the moment work starts, and
1700
+ ``settle`` records the batch's terminal state. ``take_sync_request`` is the
1701
+ legacy test-and-clear flag and is left injectable for callers that still
1702
+ drive the loop that way.
1703
+
1704
+ ``mark_rebuilding`` publishes the in-flight flag for EVERY iteration,
1705
+ requested or automatic, and clears it on every exit path. It is separate
1706
+ from ``capture_batch``/``settle`` on purpose: those two also move the
1707
+ settlement counters, which must not advance for a batch that never existed.
1497
1708
  """
1498
1709
  while not stop.is_set():
1710
+ batch = None
1711
+ if (pending_request is not None and capture_batch is not None
1712
+ and pending_request()):
1713
+ batch = capture_batch()
1714
+ if mark_rebuilding is not None:
1715
+ # BEFORE t0: the publish is a non-blocking queue put, and keeping
1716
+ # it outside the measured span leaves the #313 bound's algebra
1717
+ # exactly as it is. An automatic tick reaches this with no batch
1718
+ # captured, which is the whole point — `rebuilding` describes the
1719
+ # iteration, not the request that may or may not have started it.
1720
+ mark_rebuilding(True)
1499
1721
  t0 = monotonic()
1500
- run_iteration()
1501
- work = monotonic() - t0
1722
+ status, warnings = "ok", ()
1723
+ try:
1724
+ result = (run_iteration(batch=batch) if batch is not None
1725
+ else run_iteration())
1726
+ if isinstance(result, dict):
1727
+ warnings = tuple(result.get("warnings") or ())
1728
+ except Exception: # noqa: BLE001 — see below
1729
+ # An escaped exception must not kill the only drainer: an accepted
1730
+ # 202 would then never reach a terminal state, and the client would
1731
+ # hold `queued…` forever waiting for a settlement no surviving
1732
+ # thread can publish.
1733
+ status = "failed"
1734
+ _log_sync_iteration_failure()
1735
+ finally:
1736
+ # Failure time is charged to the cooldown exactly like success
1737
+ # time, so a crash loop cannot busy-spin.
1738
+ work = monotonic() - t0
1739
+ if batch is not None and settle is not None:
1740
+ settle(batch[0], status, warnings)
1741
+ if mark_rebuilding is not None:
1742
+ # Every exit path, including an escaped exception: a flag left
1743
+ # set would pin the client's chip at `syncing…` for the life of
1744
+ # the process. `settle` has already cleared it on a requested
1745
+ # tick, so this publishes nothing extra there.
1746
+ mark_rebuilding(False)
1747
+
1502
1748
  deadline = _next_deadline(t0, interval, work)
1503
- while not stop.is_set() and monotonic() < deadline:
1504
- if take_sync_request():
1749
+ floor = t0 + 2.0 * work
1750
+ while not stop.is_set():
1751
+ now = monotonic()
1752
+ ready = (pending_request is not None and capture_batch is not None
1753
+ and now >= floor and pending_request())
1754
+ if ready or now >= deadline:
1505
1755
  break
1506
- sleep(min(0.1, max(0.0, deadline - monotonic())))
1756
+ if take_sync_request is not None and take_sync_request():
1757
+ break # legacy test-and-clear path, unchanged
1758
+ sleep(min(0.1, max(0.0, deadline - now)))
1507
1759
 
1508
1760
 
1509
1761
  def _dashboard_maybe_prune_retention() -> None:
@@ -1532,6 +1784,147 @@ def _dashboard_maybe_prune_retention() -> None:
1532
1784
  pass
1533
1785
 
1534
1786
 
1787
+ def _conversation_sync_pass() -> str:
1788
+ """One WHOLE transcript-ingest pass (#583 S4 / F5).
1789
+
1790
+ Store open, both provider syncs, the retention prune and the close. The
1791
+ loop measures this callable's duration as `work`, so anything left outside
1792
+ it would be work outside the duty denominator — exactly the F5 defect.
1793
+
1794
+ Returns a status from `_lib_tick_stats.CONVERSATION_STATUSES`. The prune is
1795
+ attempted whenever the store OPENED, including after a failing sync: the
1796
+ documented contract is a throttled prune driven by this thread, not a prune
1797
+ conditional on a successful ingest.
1798
+
1799
+ The returned status describes the OPEN and the two SYNCS, and nothing else.
1800
+ `_dashboard_maybe_prune_retention`'s outcome is discarded here and that
1801
+ function ends in `except Exception: pass`, so a pass that ingested cleanly
1802
+ and then failed only in the prune reports `ok`. Widening the status to
1803
+ cover the prune would mean widening `CONVERSATION_STATUSES`, which is a
1804
+ closed set by design.
1805
+
1806
+ When the primary open FAILS the pass must not prune, and this is a
1807
+ constraint rather than a detail. `_dashboard_maybe_prune_retention` opens
1808
+ its own conversations connection and reaches the retention due/throttle
1809
+ check only after that open, so an unconditional prune would attempt a
1810
+ second open of the same unopenable store on every pass, swallow the
1811
+ failure, and repeat. The duty bound would cap CPU share while doing nothing
1812
+ about the duplicated migration, recovery and I/O pressure. The invariant is
1813
+ therefore NO SECOND OPEN ATTEMPT AFTER A FAILED OPEN — stated that narrowly
1814
+ because a successful pass opens the store twice by design, once here and
1815
+ once inside the prune.
1816
+ """
1817
+ try:
1818
+ conn = open_conversations_db()
1819
+ except (OSError, sqlite3.DatabaseError) as exc:
1820
+ eprint(f"[conversations] background sync unavailable: {exc}")
1821
+ return "store_unavailable"
1822
+ status = "ok"
1823
+ try:
1824
+ sync_claude_conversations(conn)
1825
+ sync_codex_conversations(conn)
1826
+ except (OSError, sqlite3.DatabaseError) as exc:
1827
+ eprint(f"[conversations] background sync unavailable: {exc}")
1828
+ status = "store_unavailable"
1829
+ except Exception as exc: # noqa: BLE001
1830
+ # Transcript parsing/normalization is deliberately outside the core
1831
+ # freshness loop. Keep this worker alive so a later clean tick can
1832
+ # self-heal instead of permanently stopping after one malformed
1833
+ # provider record.
1834
+ eprint(
1835
+ "[conversations] background sync failed: "
1836
+ f"{type(exc).__name__}: {exc}"
1837
+ )
1838
+ status = "error"
1839
+ finally:
1840
+ try:
1841
+ conn.close()
1842
+ except Exception: # noqa: BLE001
1843
+ pass
1844
+ _dashboard_maybe_prune_retention()
1845
+ return status
1846
+
1847
+
1848
+ def _conversation_sync_loop(
1849
+ *,
1850
+ stop,
1851
+ interval: float,
1852
+ run_iteration,
1853
+ monotonic=time.monotonic,
1854
+ thread_time_ns=None,
1855
+ wait=None,
1856
+ record=None,
1857
+ ) -> None:
1858
+ """Transcript ingest loop, bounded like the main one (#583 S4 / F5).
1859
+
1860
+ Extracted from a closure inside `cmd_dashboard` so the duty property can be
1861
+ driven by a virtual clock rather than merely asserted. `run_iteration`
1862
+ performs one WHOLE pass, so its measured duration is what drives the
1863
+ deadline.
1864
+
1865
+ The previous fixed `wait(interval)` prevented literal 100% duty for finite
1866
+ work but provided no scale-independent ceiling below it: a 30 s pass ran
1867
+ 30-on/5-off, about 86% duty, and nothing bounded that as the store grew.
1868
+ """
1869
+ if thread_time_ns is None:
1870
+ thread_time_ns = time.thread_time_ns
1871
+ if wait is None:
1872
+ wait = stop.wait
1873
+ seq = 0
1874
+ while not stop.is_set():
1875
+ t0 = monotonic()
1876
+ cpu0 = thread_time_ns()
1877
+ try:
1878
+ status = run_iteration() or "ok"
1879
+ except Exception: # noqa: BLE001 — the worker must outlive one bad pass
1880
+ _log_sync_iteration_failure()
1881
+ status = "error"
1882
+ work = max(0.0, monotonic() - t0)
1883
+ cpu_ns = max(0, thread_time_ns() - cpu0)
1884
+ seq += 1
1885
+ if record is not None:
1886
+ record(
1887
+ seq=seq,
1888
+ started_ns=int(t0 * 1e9),
1889
+ ended_ns=int((t0 + work) * 1e9),
1890
+ duration_ns=int(work * 1e9),
1891
+ cpu_ns=cpu_ns,
1892
+ # No period is passed: `period_ns` is the FORWARD interval, so
1893
+ # the recorder stamps it onto the PREVIOUS record when this
1894
+ # pass's start closes it. Pairing a pass's CPU with the
1895
+ # interval that preceded it would shift the denominator by one
1896
+ # pass and publish a share with no upper bound.
1897
+ status=status,
1898
+ )
1899
+ deadline = _conversation_next_deadline(t0, interval, work)
1900
+ remaining = deadline - monotonic()
1901
+ if remaining > 0:
1902
+ wait(remaining)
1903
+
1904
+
1905
+ def _make_conversation_sync_thread(*, stop, sync_interval, no_sync):
1906
+ """Build the bounded conversation-sync thread, or None under --no-sync.
1907
+
1908
+ #320: transcript/search ingestion runs on its own thread and SQLite file, so
1909
+ a multi-GB first rebuild or a contended conversations.db cannot delay
1910
+ `_run_sync_now`, its `last_sync_at` stamp, or core SSE publication. #583 S4
1911
+ gives that thread the main loop's 50%-duty bound and publishes each pass
1912
+ into the second `_lib_tick_stats` ring.
1913
+ """
1914
+ if no_sync:
1915
+ return None
1916
+ return threading.Thread(
1917
+ target=lambda: _conversation_sync_loop(
1918
+ stop=stop,
1919
+ interval=max(5.0, float(sync_interval)),
1920
+ run_iteration=_conversation_sync_pass,
1921
+ record=_lib_tick_stats.record_conversation_pass,
1922
+ ),
1923
+ daemon=True,
1924
+ name="dashboard-conversations-sync",
1925
+ )
1926
+
1927
+
1535
1928
  def _make_run_sync_now(*args, **kwargs):
1536
1929
  return sys.modules["cctally"]._make_run_sync_now(*args, **kwargs)
1537
1930
 
@@ -1626,6 +2019,10 @@ def _build_alert_payload_weekly(*args, **kwargs):
1626
2019
  return sys.modules["cctally"]._build_alert_payload_weekly(*args, **kwargs)
1627
2020
 
1628
2021
 
2022
+ def synthetic_preview_week_start(*args, **kwargs):
2023
+ return sys.modules["cctally"].synthetic_preview_week_start(*args, **kwargs)
2024
+
2025
+
1629
2026
  def _build_alert_payload_five_hour(*args, **kwargs):
1630
2027
  return sys.modules["cctally"]._build_alert_payload_five_hour(*args, **kwargs)
1631
2028
 
@@ -1897,10 +2294,9 @@ def __getattr__(name): # pylint: disable=invalid-name
1897
2294
  # are pure constants / read-only objects whose identity is stable across
1898
2295
  # the process lifetime; binding them once at load time keeps bare-name
1899
2296
  # reads in moved bodies working without per-call attribute lookups.
1900
- # Path constants and tunables that tests monkeypatch (STATIC_DIR,
1901
- # _DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS) are eager-re-exported FROM the
1902
- # sibling at bin/cctally so monkeypatches propagate; this block carries
1903
- # things that are NEVER patched at runtime.
2297
+ # Path constants and tunables that tests monkeypatch (STATIC_DIR) are
2298
+ # eager-re-exported FROM the sibling at bin/cctally so monkeypatches
2299
+ # propagate; this block carries things that are NEVER patched at runtime.
1904
2300
  BLOCK_DURATION = sys.modules["cctally"].BLOCK_DURATION
1905
2301
 
1906
2302
 
@@ -1958,25 +2354,177 @@ def _resolve_dashboard_bind_for_runtime(stored: str) -> str:
1958
2354
  # Pre-extract location: bin/cctally L16265.
1959
2355
 
1960
2356
  class _SnapshotRef:
1961
- """Thread-safe holder for the current DataSnapshot."""
2357
+ """Thread-safe holder for the current DataSnapshot and the sync queue.
2358
+
2359
+ #583 S2. This object is the SINGLE authority for queue and activity
2360
+ state. Builders must never construct activity counters themselves: A2
2361
+ and the final publish both replace the snapshot wholesale, so a request
2362
+ accepted mid-build would otherwise have its counter overwritten by the
2363
+ older snapshot the builder had already assembled. Every mutator below
2364
+ re-stamps the held snapshot, and ``set()`` merges the authoritative
2365
+ activity in and RETURNS the merged object so a publish site can send
2366
+ exactly what the reference now holds.
2367
+
2368
+ The legacy ``_sync_requested`` flag and ``take_sync_request()`` are the
2369
+ TUI's mechanism (bin/_cctally_tui.py:5581 and :5832) and keep their
2370
+ exact test-and-clear semantics. The dashboard uses the counters.
2371
+ """
1962
2372
 
1963
2373
  def __init__(self, initial: DataSnapshot) -> None:
1964
2374
  import threading
2375
+ import uuid
1965
2376
  self._lock = threading.Lock()
1966
- self._snap = initial
1967
2377
  self._sync_requested = False
2378
+ # Fixed 16 chars: the envelope's byte count must stay deterministic
2379
+ # for bench/baselines/envelope-oracle.json to remain comparable.
2380
+ self.server_epoch = uuid.uuid4().hex[:16]
2381
+ self._requested_id = 0
2382
+ self._requested_refresh = False
2383
+ self._started_id = 0
2384
+ self._settled_id = 0
2385
+ self._settled_status = None
2386
+ self._settled_warnings = ()
2387
+ # OWNER-SCOPED, not a single process-wide boolean. Two independent
2388
+ # rebuilders write this state — the periodic sync loop and any HTTP
2389
+ # handler thread that wins the non-blocking `sync_lock` acquire — and a
2390
+ # boolean gives neither of them a way to know who set it. A clear must
2391
+ # therefore remove only the clearing thread's own claim; `rebuilding`
2392
+ # is then true exactly while at least one rebuilder holds one.
2393
+ #
2394
+ # Owner scoping made a LEAKED CLAIM strictly worse than the boolean it
2395
+ # replaced, so do not record the opposite. Under the boolean a leaked
2396
+ # `True` was cleared by whichever rebuilder next reached `set_final`, so
2397
+ # it self-healed on the following rebuild. Every clear site here discards
2398
+ # `threading.get_ident()`, so a claim left behind by a thread that has
2399
+ # exited can be discarded by NO other thread, and `rebuilding` would stay
2400
+ # true — pinning every client's chip at `syncing…` for the life of the
2401
+ # process.
2402
+ #
2403
+ # No leak is reachable today, but the argument splits by CALLER, not by
2404
+ # add site. `mark_rebuilding` below is reached from both routes — the
2405
+ # sync loop through `_make_sync_loop_collaborators` and an HTTP handler
2406
+ # thread through `DashboardHTTPHandler.mark_rebuilding` — so reading one
2407
+ # add site answers for neither. Four callers add a claim.
2408
+ #
2409
+ # Two of the four are bracketed: `_handle_post_sync` and
2410
+ # `_handle_post_settings` each mark inside a `try` whose `finally`
2411
+ # clears, so nothing between the two can leak the claim.
2412
+ #
2413
+ # The other two are the sync loop's `capture_batch()` and its
2414
+ # `mark_rebuilding(True)`, and they are NOT bracketed. Both run before
2415
+ # `t0` and therefore before the `try:` whose `finally` clears them; the
2416
+ # loop's own comment at `mark_rebuilding(True)` gives the #313
2417
+ # duty-algebra reason for that one's placement. `_dashboard_sync_loop`'s
2418
+ # `while` has no outer handler, so a raise in that gap would kill the
2419
+ # drainer and leak the claim together. The gap is safe because nothing
2420
+ # in it raises: the `_restamp_locked()` each add performs is a
2421
+ # `dataclasses.replace` over `DataSnapshot`, a plain dataclass with no
2422
+ # `__post_init__`, no `init=False` field and no `InitVar`;
2423
+ # `SSEHub.publish` holds its own lock and swallows
2424
+ # `queue.Full`/`queue.Empty`; `ref.get()` is a lock-and-return; and what
2425
+ # remains is a clock read and two local assignments.
2426
+ #
2427
+ # A new `add` must therefore satisfy one of the two: a same-thread
2428
+ # `finally` that clears it, or a proven non-raising path to one. There
2429
+ # is no self-healing path behind either.
2430
+ self._rebuilding_owners: set[int] = set()
2431
+ self._snap = self._stamped_locked(initial)
2432
+
2433
+ def _activity_locked(self) -> dict:
2434
+ return {
2435
+ "server_epoch": self.server_epoch,
2436
+ "rebuilding": bool(self._rebuilding_owners),
2437
+ "requested_id": self._requested_id,
2438
+ "started_id": self._started_id,
2439
+ "settled_id": self._settled_id,
2440
+ "settled_status": self._settled_status,
2441
+ "settled_warnings": self._settled_warnings,
2442
+ }
2443
+
2444
+ def _stamped_locked(self, snap: DataSnapshot) -> DataSnapshot:
2445
+ import dataclasses
2446
+ return dataclasses.replace(snap, sync_activity=self._activity_locked())
2447
+
2448
+ def _restamp_locked(self) -> None:
2449
+ self._snap = self._stamped_locked(self._snap)
2450
+
2451
+ def activity(self) -> dict:
2452
+ with self._lock:
2453
+ return self._activity_locked()
1968
2454
 
1969
2455
  def get(self) -> DataSnapshot:
1970
2456
  with self._lock:
1971
2457
  return self._snap
1972
2458
 
1973
- def set(self, snap: DataSnapshot) -> None:
2459
+ def set(self, snap: DataSnapshot) -> DataSnapshot:
2460
+ """Store ``snap`` with the authoritative activity merged in.
2461
+
2462
+ Returns the merged object. Publish sites must send the RETURN value,
2463
+ not their local build result, or a request accepted mid-build has its
2464
+ counter erased by the older snapshot the builder assembled.
2465
+ """
2466
+ with self._lock:
2467
+ self._snap = self._stamped_locked(snap)
2468
+ return self._snap
2469
+
2470
+ def replace_fields(self, **changes) -> DataSnapshot:
2471
+ """Atomically apply narrow field changes to the latest snapshot.
2472
+
2473
+ Out-of-band publishers compute small derived fields independently of
2474
+ the main snapshot builder. They must not read a whole snapshot, do
2475
+ that work, then call ``set()``: a sync can publish in between and the
2476
+ stale whole-object write would erase its newer data. This method keeps
2477
+ the read/replace/write sequence under the reference lock and re-stamps
2478
+ the authoritative activity state before returning the publishable
2479
+ object.
2480
+ """
2481
+ import dataclasses
2482
+ with self._lock:
2483
+ self._snap = self._stamped_locked(
2484
+ dataclasses.replace(self._snap, **changes)
2485
+ )
2486
+ return self._snap
2487
+
2488
+ def set_final(self, snap: DataSnapshot) -> DataSnapshot:
2489
+ """Store ``snap`` as the iteration's TERMINAL state: the merge of
2490
+ ``set()`` plus dropping THIS thread's claim on ``_rebuilding_owners``,
2491
+ under one lock acquisition.
2492
+
2493
+ Without this a rebuild costs three published frames — the flag going
2494
+ up, the build's own final publish, and the flag coming back down — and
2495
+ the third exists only because the loop clears the flag after the
2496
+ rebuild has already published. Each extra frame is one shared
2497
+ ``snapshot_to_envelope`` plus one byte-ready JSON frame per variant,
2498
+ and one whole-store replacement in each connected browser, which is
2499
+ the opposite of what a dashboard-performance session is for.
2500
+
2501
+ Storing and clearing through two calls would not help: the frame
2502
+ published between them would carry ``rebuilding: true`` and the third
2503
+ frame would come back. The clear has to be part of the same store.
2504
+
2505
+ The clear drops THIS thread's claim only. Clearing outright would end
2506
+ one rebuilder's build by declaring every rebuilder idle: a handler
2507
+ thread that finished first would publish ``rebuilding: false`` over a
2508
+ periodic rebuild that had already marked itself and was still blocked on
2509
+ ``sync_lock``, and nothing would correct that until the next tick.
2510
+ """
1974
2511
  with self._lock:
1975
- self._snap = snap
2512
+ self._rebuilding_owners.discard(threading.get_ident())
2513
+ self._snap = self._stamped_locked(snap)
2514
+ return self._snap
1976
2515
 
1977
- def request_sync(self) -> None:
2516
+ def request_sync(self, refresh: bool = False) -> int:
2517
+ """Enqueue a coalescing sync request; returns its identifier.
2518
+
2519
+ ``refresh`` defaults False so the TUI's argument-less call site is
2520
+ unchanged.
2521
+ """
1978
2522
  with self._lock:
1979
2523
  self._sync_requested = True
2524
+ self._requested_id += 1
2525
+ self._requested_refresh = self._requested_refresh or bool(refresh)
2526
+ self._restamp_locked()
2527
+ return self._requested_id
1980
2528
 
1981
2529
  def take_sync_request(self) -> bool:
1982
2530
  # Atomic test-and-clear — threading.Event's is_set()/clear() pair
@@ -1985,6 +2533,261 @@ class _SnapshotRef:
1985
2533
  taken, self._sync_requested = self._sync_requested, False
1986
2534
  return taken
1987
2535
 
2536
+ def pending_request(self) -> bool:
2537
+ """Peek, never consume. A poll firing before the service floor must
2538
+ not clear the request (#583 S2 spec 5.2)."""
2539
+ with self._lock:
2540
+ return self._requested_id > self._started_id
2541
+
2542
+ def capture_batch(self) -> tuple:
2543
+ """Atomically claim every outstanding request as one batch."""
2544
+ with self._lock:
2545
+ self._started_id = self._requested_id
2546
+ refresh = self._requested_refresh
2547
+ self._requested_refresh = False
2548
+ self._rebuilding_owners.add(threading.get_ident())
2549
+ self._restamp_locked()
2550
+ return (self._started_id, refresh)
2551
+
2552
+ def mark_rebuilding(self, value: bool) -> bool:
2553
+ """Set the in-flight flag alone; return True iff it CHANGED.
2554
+
2555
+ Deliberately narrow. ``rebuilding`` must be true for the duration of
2556
+ every sync iteration, but ``capture_batch``/``settle`` run only when a
2557
+ request is pending, so an automatic tick — the normal case on a
2558
+ dashboard nobody is clicking — left the flag permanently false.
2559
+ Widening those two to batchless ticks is the wrong fix: ``settle``
2560
+ would advance ``settled_id``/``settled_status``/``settled_warnings``
2561
+ for a batch that never existed, and spec §6.2 says those three
2562
+ describe the most recently SETTLED batch.
2563
+
2564
+ The changed/unchanged return lets the loop's collaborator publish
2565
+ exactly once per transition: on a requested tick ``capture_batch`` has
2566
+ already set the flag and published, so this reports no change and adds
2567
+ no duplicate frame.
2568
+
2569
+ A mark adds or drops the CALLING thread's claim, and the transition is
2570
+ the emptiness of the owner set changing. Publish-on-transition is
2571
+ preserved exactly, because a transition is now empty-to-nonempty or
2572
+ nonempty-to-empty. The set is not a counter, so a claim is idempotent
2573
+ per thread and one thread cannot hold two nested claims.
2574
+ """
2575
+ with self._lock:
2576
+ ident = threading.get_ident()
2577
+ before = bool(self._rebuilding_owners)
2578
+ if value:
2579
+ self._rebuilding_owners.add(ident)
2580
+ else:
2581
+ self._rebuilding_owners.discard(ident)
2582
+ after = bool(self._rebuilding_owners)
2583
+ if before == after:
2584
+ return False
2585
+ self._restamp_locked()
2586
+ return True
2587
+
2588
+ def settle(self, batch_id: int, status: str, warnings=()) -> None:
2589
+ """Record a batch's terminal state.
2590
+
2591
+ ``settled_status`` and ``settled_warnings`` describe the most recently
2592
+ settled batch and are retained across subsequent automatic frames:
2593
+ clearing them on the next ordinary tick would tell a client its request
2594
+ settled while destroying the warnings explaining how, and the queued
2595
+ contract removed the HTTP response that used to carry them.
2596
+ """
2597
+ with self._lock:
2598
+ self._settled_id = max(self._settled_id, int(batch_id))
2599
+ self._settled_status = status
2600
+ self._settled_warnings = tuple(warnings)
2601
+ # This thread's claim only — `capture_batch` added it, and another
2602
+ # rebuilder's concurrent claim is not this batch's to end.
2603
+ self._rebuilding_owners.discard(threading.get_ident())
2604
+ self._restamp_locked()
2605
+
2606
+
2607
+ class _SSEDelivery:
2608
+ """One publication, projected and encoded at most once per variant.
2609
+
2610
+ #583 S3 §5. ``_serve_api_events`` used to call ``snapshot_to_envelope`` plus
2611
+ ``encode_dashboard_json`` inside its per-connection loop, so N connected
2612
+ clients projected and encoded the same data N times per tick — about
2613
+ 3.4 MB each on a production-scale store.
2614
+
2615
+ The clock is pinned HERE, once, so every client served from this delivery
2616
+ agrees on the age fields instead of differing by the fan-out latency.
2617
+ ``SSEHub.subscribe`` deliberately builds a FRESH delivery for its seed
2618
+ rather than handing out a stored one, because a client connecting between
2619
+ ticks would otherwise render an age frozen at the previous publication.
2620
+
2621
+ The cache lock is PER DELIVERY. The hub's own lock must never be held
2622
+ across a multi-megabyte projection.
2623
+
2624
+ Sharing is only sound when the projection is a function of the snapshot
2625
+ plus the variant key. ``_serve_api_events`` therefore refuses to share a
2626
+ snapshot that carries no ``envelope_precompute``: ``snapshot_to_envelope``
2627
+ then reads configuration inline and runs the real doctor gather per call,
2628
+ neither of which is keyed. It also captures ``_channel_env_fragment``'s
2629
+ preview-channel process state, which is process-global rather than
2630
+ connection-specific — stated so the "function of its key" claim is
2631
+ complete.
2632
+ """
2633
+
2634
+ __slots__ = ("snapshot", "pinned_now_utc", "pinned_monotonic",
2635
+ "_cache", "_lock")
2636
+
2637
+ def __init__(self, snapshot, pinned_now_utc, pinned_monotonic) -> None:
2638
+ self.snapshot = snapshot
2639
+ self.pinned_now_utc = pinned_now_utc
2640
+ self.pinned_monotonic = pinned_monotonic
2641
+ self._cache: dict = {}
2642
+ self._lock = threading.Lock()
2643
+
2644
+ def encoded(self, variant_key, project_fn) -> bytes:
2645
+ """Return complete SSE frame bytes for ``variant_key``, building once.
2646
+
2647
+ Double-checked: the fast path is a lock-free dict read, and the slow
2648
+ path re-checks under the lock so two threads racing on the same missing
2649
+ variant produce one projection, JSON encoding and frame assembly. The
2650
+ cached value is byte-ready so fan-out never repeats UTF-8 encoding for
2651
+ each connection. A MISS for any valid variant computes that variant —
2652
+ normalizing an INVALID privacy input to False is the CALLER's job, done
2653
+ before the key is built. Those are different situations and conflating
2654
+ them either leaks or breaks the gate.
2655
+ """
2656
+ hit = self._cache.get(variant_key)
2657
+ if hit is not None:
2658
+ return hit
2659
+ with self._lock:
2660
+ hit = self._cache.get(variant_key)
2661
+ if hit is not None:
2662
+ return hit
2663
+ built = project_fn(variant_key)
2664
+ self._cache[variant_key] = built
2665
+ return built
2666
+
2667
+
2668
+ # #583 S3 §5. A distinct slot for "no oauth_usage configuration at all", so it
2669
+ # cannot collide with an EMPTY configuration. Both used to canonicalize to `()`.
2670
+ _OAUTH_CFG_ABSENT = ("\x00cctally:oauth-usage-absent",)
2671
+
2672
+
2673
+ def _canonical_oauth_key(cfg):
2674
+ """A hashable canonical form of the oauth_usage config block.
2675
+
2676
+ #583 S3 §5. The delivery cache is keyed by a tuple, and the resolved
2677
+ ``oauth_usage`` config is a dict, which is unhashable. Sorted items give a
2678
+ stable key for two connections that resolved the same configuration, which
2679
+ is the normal case — every connection reads the same file.
2680
+
2681
+ An ABSENT configuration and an EMPTY one are different configurations and
2682
+ get different keys. Collapsing them is unreachable today, because
2683
+ ``_get_oauth_usage_config`` is defaults-filled and never returns an empty
2684
+ mapping — which is precisely why it would go unnoticed if a later change
2685
+ made it reachable, inside a key whose entire job is keeping two
2686
+ configurations apart. Callers pass a mapping or ``None``.
2687
+ """
2688
+ if cfg is None:
2689
+ return _OAUTH_CFG_ABSENT
2690
+ return tuple(sorted((str(k), repr(v)) for k, v in cfg.items()))
2691
+
2692
+
2693
+ # #583 S3 §6. The SSE keep-alive interval, as a module constant so a test can
2694
+ # drive the keep-alive path without waiting fifteen seconds. It matters that
2695
+ # the path is testable: under compression a raw `wfile.write` of the keep-alive
2696
+ # comment corrupts everything after it, and the failure is silent for one
2697
+ # interval and then permanent for that connection.
2698
+ _SSE_KEEPALIVE_SECONDS = 15
2699
+
2700
+
2701
+ def _accepts_gzip(header_value: "str | None") -> bool:
2702
+ """Whether this client accepts gzip, parsed on token boundaries.
2703
+
2704
+ #583 S3 §6. A substring test for "gzip" compresses for a client sending
2705
+ ``gzip;q=0``, which is an explicit refusal, and for an unrelated token such
2706
+ as ``notgzip`` or ``x-gzip``. A malformed quality value falls back to
2707
+ identity rather than raising, because this runs on the publish path and
2708
+ must never take a connection down.
2709
+
2710
+ ``*`` is honoured as a wildcard, but an explicit ``gzip`` entry wins over
2711
+ it in either direction: ``gzip;q=0, *`` is a refusal even though the
2712
+ wildcard would otherwise accept.
2713
+ """
2714
+ if not header_value:
2715
+ return False
2716
+ wildcard = None
2717
+ for part in header_value.split(","):
2718
+ token, _, params = part.strip().partition(";")
2719
+ token = token.strip().lower()
2720
+ if token not in ("gzip", "*"):
2721
+ continue
2722
+ q = 1.0
2723
+ malformed = False
2724
+ for param in params.split(";"):
2725
+ name, _, value = param.strip().partition("=")
2726
+ if name.strip().lower() != "q":
2727
+ continue
2728
+ try:
2729
+ q = float(value.strip())
2730
+ except ValueError:
2731
+ malformed = True
2732
+ break
2733
+ if malformed:
2734
+ return False
2735
+ if token == "gzip":
2736
+ return q > 0.0
2737
+ if wildcard is None:
2738
+ wildcard = q
2739
+ return bool(wildcard is not None and wildcard > 0.0)
2740
+
2741
+
2742
+ def _delivery_is_shareable(snapshot) -> bool:
2743
+ """Whether one projection of ``snapshot`` may be shared across clients.
2744
+
2745
+ #583 S3 §5. Sharing is sound only when the projection is a function of the
2746
+ snapshot plus the stated variant key. The gate keys on ONE field,
2747
+ ``envelope_precompute``: without it ``snapshot_to_envelope`` reads
2748
+ ``config.json`` inline (``bin/_cctally_dashboard_envelope.py:1194``), so two
2749
+ calls with the same key can differ and the result is not cacheable. Those
2750
+ snapshots are fixtures, the initial empty snapshot and positionally-
2751
+ constructed ones — never a live tick.
2752
+
2753
+ The doctor block is NOT part of this gate, and saying it was would misstate
2754
+ what the predicate reads. It is guarded separately by ``doctor_payload``
2755
+ (``:1627``), which is set independently of ``envelope_precompute`` because
2756
+ each catches its own failure, so a snapshot can carry the precompute and
2757
+ still run the real gather. That is sound to share anyway: the gather is
2758
+ process-global rather than connection-specific, and both of its inputs —
2759
+ ``now_utc`` and ``runtime_bind`` — are already fixed by the delivery's pin
2760
+ and by the variant key.
2761
+ """
2762
+ return getattr(snapshot, "envelope_precompute", None) is not None
2763
+
2764
+
2765
+ def _drain_to_newest(q, first):
2766
+ """Return the newest delivery queued on ``q``, discarding older ones.
2767
+
2768
+ #583 S3 §5. ``SSEHub`` uses a four-slot queue and ``publish`` discards only
2769
+ ONE oldest entry when full, so a client that falls behind holds a backlog
2770
+ of up to four deliveries. Each delivery pins its clock at publication, so
2771
+ replaying that backlog would render ages several publish periods stale — a
2772
+ regression against the present behaviour, where each frame is projected at
2773
+ consumption time and its ages are therefore current.
2774
+
2775
+ The fix is on the CONSUMER side deliberately: ``SSEHub.publish`` is
2776
+ governed by Preserve 4 and the A2 publication tests depend on its
2777
+ behaviour, so it is not modified. Draining here is the latest-wins
2778
+ behaviour the hub's own docstring already describes.
2779
+
2780
+ ``first`` is the item the caller already took off the queue with its own
2781
+ blocking ``get``, so the ``queue.Empty`` keep-alive path stays where it is.
2782
+ """
2783
+ import queue as _queue
2784
+ newest = first
2785
+ while True:
2786
+ try:
2787
+ newest = q.get_nowait()
2788
+ except _queue.Empty:
2789
+ return newest
2790
+
1988
2791
 
1989
2792
  class SSEHub:
1990
2793
  """Thread-safe fan-out hub for SSE clients.
@@ -1993,6 +2796,12 @@ class SSEHub:
1993
2796
  client queues so a slow browser cannot back-pressure the sync thread.
1994
2797
  Consumers call `subscribe()` to obtain a `queue.Queue`, then read
1995
2798
  with a timeout; call `unsubscribe()` on disconnect or at teardown.
2799
+
2800
+ #583 S3 §5: what the queues carry is a `_SSEDelivery` wrapping the
2801
+ published snapshot, not the snapshot itself, so one tick projects and
2802
+ encodes once per variant instead of once per connected client. The
2803
+ queueing behaviour below — size, latest-wins discard, lock discipline — is
2804
+ unchanged and is governed by Preserve 4.
1996
2805
  """
1997
2806
 
1998
2807
  def __init__(self, maxsize: int = 4) -> None:
@@ -2012,12 +2821,32 @@ class SSEHub:
2012
2821
  self._queues.append(q)
2013
2822
  if self._last is not None:
2014
2823
  # Seed the new subscriber so it renders immediately.
2824
+ # #583 S3 §5: a FRESH delivery over the same snapshot, with the
2825
+ # clock sampled NOW. Handing out `self._last` would render this
2826
+ # client's ages frozen at the previous publication, which for a
2827
+ # tab opened late in a publish period is visibly wrong.
2828
+ seed = _SSEDelivery(
2829
+ snapshot=self._last.snapshot,
2830
+ pinned_now_utc=dt.datetime.now(dt.timezone.utc),
2831
+ pinned_monotonic=time.monotonic(),
2832
+ )
2015
2833
  try:
2016
- q.put_nowait(self._last)
2834
+ q.put_nowait(seed)
2017
2835
  except _queue.Full:
2018
2836
  pass
2019
2837
  return q
2020
2838
 
2839
+ def latest(self):
2840
+ """The most recently published delivery, or None before the first.
2841
+
2842
+ #583 S3 §7: `/api/data` serves the most recently PUBLISHED state. The
2843
+ snapshot reference is mutated by four operations that publish
2844
+ separately, so reading the reference could report a `hydrating` flag no
2845
+ client was ever sent (#600).
2846
+ """
2847
+ with self._lock:
2848
+ return self._last
2849
+
2021
2850
  def unsubscribe(self, q) -> None:
2022
2851
  with self._lock:
2023
2852
  try:
@@ -2027,8 +2856,16 @@ class SSEHub:
2027
2856
 
2028
2857
  def publish(self, snapshot) -> None:
2029
2858
  import queue as _queue
2859
+ # #583 S3 §5: wrap ONCE, outside the hub lock, so every queue and
2860
+ # `_last` share one projection cache for this tick. Built before the
2861
+ # lock because construction must not run under it.
2862
+ delivery = _SSEDelivery(
2863
+ snapshot=snapshot,
2864
+ pinned_now_utc=dt.datetime.now(dt.timezone.utc),
2865
+ pinned_monotonic=time.monotonic(),
2866
+ )
2030
2867
  with self._lock:
2031
- self._last = snapshot
2868
+ self._last = delivery
2032
2869
  # Latest-wins coalescing (#278 §2.6): every published snapshot is a
2033
2870
  # COMPLETE state replacement, so a client only ever needs the
2034
2871
  # newest. On a full queue drop the STALE queued frame and enqueue
@@ -2045,14 +2882,14 @@ class SSEHub:
2045
2882
  # re-put cannot lose to it.
2046
2883
  for q in self._queues:
2047
2884
  try:
2048
- q.put_nowait(snapshot)
2885
+ q.put_nowait(delivery)
2049
2886
  except _queue.Full:
2050
2887
  try:
2051
2888
  q.get_nowait() # discard the oldest, stale frame
2052
2889
  except _queue.Empty:
2053
2890
  pass
2054
2891
  try:
2055
- q.put_nowait(snapshot)
2892
+ q.put_nowait(delivery)
2056
2893
  except _queue.Full:
2057
2894
  # Defensive: a consumer racing between our get and put
2058
2895
  # could only have removed items, so this is unreachable
@@ -2939,7 +3776,22 @@ def _dashboard_build_blocks_view(conn: "sqlite3.Connection",
2939
3776
  recorded-windows-widening trick (loads reset windows from
2940
3777
  ``[start - BLOCK_DURATION, end + BLOCK_DURATION]`` so a recorded
2941
3778
  reset just outside the visible window can still anchor blocks
2942
- inside it) and the strict-window entry filter.
3779
+ inside it) and the post-group block-overlap filter.
3780
+
3781
+ #620 S1 D8: the entry set is grouped into native blocks FIRST and the
3782
+ week filter is then applied to whole blocks, retaining every block whose
3783
+ interval overlaps ``[week_start_at, week_end_at)`` with its full native
3784
+ totals. It used to filter ENTRIES to the week before grouping, so a
3785
+ block straddling a week boundary was folded from only the part of itself
3786
+ that fell inside the week — while ``/api/block/<iso>`` fetches the
3787
+ block's own native window and applies no week clip, so the panel and its
3788
+ own drilldown reported different totals for the same ``start_at`` and
3789
+ the panel's was permanently short. Selecting a block that overlaps the
3790
+ week is the deliberate part of the contract and stays; clipping a
3791
+ selected block's contents was the defect.
3792
+
3793
+ The fetch already read one block duration on each side, so this needs no
3794
+ additional query.
2943
3795
 
2944
3796
  Returning the full ``BlocksView`` (rows + totals) lets the sync
2945
3797
  thread populate ``DataSnapshot.blocks_total_cost_usd`` /
@@ -2950,13 +3802,12 @@ def _dashboard_build_blocks_view(conn: "sqlite3.Connection",
2950
3802
  fetch_start = week_start_at - BLOCK_DURATION
2951
3803
  fetch_end = week_end_at + BLOCK_DURATION
2952
3804
  entries = get_entries(fetch_start, fetch_end, skip_sync=skip_sync)
2953
- entries = [e for e in entries if week_start_at <= e.timestamp < week_end_at]
2954
3805
 
2955
3806
  recorded_windows, block_start_overrides, canonical_intervals = (
2956
3807
  _load_recorded_five_hour_windows(fetch_start, fetch_end)
2957
3808
  )
2958
3809
  c = _cctally()
2959
- return c.build_blocks_view(
3810
+ view = c.build_blocks_view(
2960
3811
  entries,
2961
3812
  now_utc=now_utc,
2962
3813
  recorded_windows=recorded_windows,
@@ -2967,6 +3818,60 @@ def _dashboard_build_blocks_view(conn: "sqlite3.Connection",
2967
3818
  display_tz=display_tz,
2968
3819
  mode="auto",
2969
3820
  )
3821
+ return _blocks_view_overlapping_week(
3822
+ view, week_start_at=week_start_at, week_end_at=week_end_at,
3823
+ )
3824
+
3825
+
3826
+ def _blocks_view_overlapping_week(view, *, week_start_at, week_end_at):
3827
+ """Retain only the blocks whose interval overlaps
3828
+ ``[week_start_at, week_end_at)``, keeping each retained block's FULL
3829
+ native totals (#620 S1 D8).
3830
+
3831
+ Overlap is the standard half-open test ``start < week_end and end >
3832
+ week_start``: a block that merely touches a bound (its end exactly at
3833
+ ``week_start_at``, or its start exactly at ``week_end_at``) shares no
3834
+ instant with the week and is not retained.
3835
+
3836
+ Totals are re-derived from the retained non-gap blocks so the React
3837
+ panel's ``footer total == sum(visible rows)`` invariant still holds, and
3838
+ ``aggregated`` is filtered in lockstep so no consumer can read a block
3839
+ set that disagrees with ``rows``.
3840
+ """
3841
+ kept_blocks = []
3842
+ total_cost = 0.0
3843
+ total_tokens = 0
3844
+ kept_starts = set()
3845
+ for b in view.aggregated:
3846
+ start = getattr(b, "start_time", None)
3847
+ end = getattr(b, "end_time", None)
3848
+ if start is None or end is None:
3849
+ # API-anchored views carry dicts, not Blocks. This adapter only
3850
+ # ever sees the heuristic path, but degrade by retaining rather
3851
+ # than silently dropping a shape we cannot classify.
3852
+ kept_blocks.append(b)
3853
+ continue
3854
+ if not (start < week_end_at and end > week_start_at):
3855
+ continue
3856
+ kept_blocks.append(b)
3857
+ if getattr(b, "is_gap", False):
3858
+ continue
3859
+ # Plain `+=` rather than `stable_sum`, deliberately: this mirrors
3860
+ # `build_blocks_view`'s own accumulation (`bin/_lib_view_models.py`),
3861
+ # and this function exists to publish the SAME totals that view
3862
+ # publishes. A different fold here could disagree with it in the last
3863
+ # ULP, which is the divergence the function was written to remove.
3864
+ total_cost += b.cost_usd
3865
+ total_tokens += b.total_tokens
3866
+ kept_starts.add(start.astimezone(dt.timezone.utc).isoformat())
3867
+ rows = tuple(r for r in view.rows if r.start_at in kept_starts)
3868
+ return dataclasses.replace(
3869
+ view,
3870
+ rows=rows,
3871
+ aggregated=tuple(kept_blocks),
3872
+ total_cost_usd=total_cost,
3873
+ total_tokens=total_tokens,
3874
+ )
2970
3875
 
2971
3876
 
2972
3877
  def _dashboard_build_blocks_panel(conn: "sqlite3.Connection",
@@ -3321,6 +4226,158 @@ def _projects_week_start_monday_utc(ts: "dt.datetime") -> "dt.datetime":
3321
4226
  )
3322
4227
 
3323
4228
 
4229
+ class _ProjectsWeekGrid:
4230
+ """The ordered half-open subscription intervals the Projects panel
4231
+ attributes cost into (#620 S1 D1).
4232
+
4233
+ ``_projects_week_start_monday_utc`` above is the fallback used when no
4234
+ snapshot anchor is available; this is what replaces it when one IS
4235
+ available. Intervals come from ``_compute_subscription_weeks``, the same
4236
+ kernel ``cmd_project`` buckets by, so the panel and the CLI describe one
4237
+ set of weeks rather than two.
4238
+
4239
+ Intervals are half-open ``[start, end)`` and are NOT assumed to be seven
4240
+ days long: Anthropic's reset day drifts, and a drifted cycle produces a
4241
+ genuinely short week. ``end_for`` therefore returns the interval's own
4242
+ end, never ``start + 7d``.
4243
+ """
4244
+
4245
+ __slots__ = ("starts", "ends", "_end_by_start", "_start_by_date")
4246
+
4247
+ def __init__(self, bounds: "list[tuple[dt.datetime, dt.datetime]]"):
4248
+ ordered = sorted(bounds, key=lambda b: b[0])
4249
+ self.starts = [b[0] for b in ordered]
4250
+ self.ends = [b[1] for b in ordered]
4251
+ self._end_by_start = {s: e for s, e in ordered}
4252
+ # `weekly_usage_snapshots.week_start_date` is the date-only lookup
4253
+ # key a legacy row carries when it has no `week_start_at`. Later
4254
+ # intervals win a collision, matching the "last capture per week
4255
+ # wins" rule the percentage read already applies.
4256
+ self._start_by_date = {s.date(): s for s in self.starts}
4257
+
4258
+ def __bool__(self) -> bool:
4259
+ return bool(self.starts)
4260
+
4261
+ def week_for(self, ts: "dt.datetime") -> "dt.datetime | None":
4262
+ """The start of the interval containing ``ts``, or None when ``ts``
4263
+ falls outside every interval.
4264
+
4265
+ First-match-wins on the reset-day-drift overlap the clamp can leave
4266
+ behind — the same walk-back `cmd_project._week_start_for` performs,
4267
+ so an entry near a drifted boundary lands in the same week on both
4268
+ surfaces.
4269
+ """
4270
+ ts_utc = ts.astimezone(dt.timezone.utc)
4271
+ idx = bisect.bisect_right(self.starts, ts_utc) - 1
4272
+ if idx < 0:
4273
+ return None
4274
+ while idx > 0 and self.starts[idx - 1] <= ts_utc < self.ends[idx - 1]:
4275
+ idx -= 1
4276
+ if self.starts[idx] <= ts_utc < self.ends[idx]:
4277
+ return self.starts[idx]
4278
+ return None
4279
+
4280
+ def start_for_date(self, day: "dt.date") -> "dt.datetime | None":
4281
+ """The interval start whose own date is ``day``, for a legacy
4282
+ snapshot row that carries ``week_start_date`` but no
4283
+ ``week_start_at``."""
4284
+ return self._start_by_date.get(day)
4285
+
4286
+ def end_for(self, start: "dt.datetime") -> "dt.datetime":
4287
+ """The interval's real end. Falls back to ``start + 7d`` only for a
4288
+ start this grid does not know, which the padding below produces."""
4289
+ return self._end_by_start.get(start, start + dt.timedelta(days=7))
4290
+
4291
+ def window_ending_at(
4292
+ self, cw_start: "dt.datetime", weeks_back: int,
4293
+ ) -> "list[tuple[dt.datetime, dt.datetime]]":
4294
+ """The last ``weeks_back`` intervals up to and including the one
4295
+ starting at ``cw_start``, oldest first.
4296
+
4297
+ When the grid holds fewer than ``weeks_back`` intervals at or before
4298
+ ``cw_start``, the head is padded backwards in seven-day steps. The
4299
+ padding is a genuine no-anchor tail — history older than any snapshot
4300
+ — so the seven-day assumption is the right one there.
4301
+
4302
+ The walk itself lives in ``_lib_subscription_weeks`` because
4303
+ ``cmd_project`` performs the same one; a copy here is how the two
4304
+ surfaces drifted apart in the first place.
4305
+ """
4306
+ return _cctally().subscription_window_ending_at(
4307
+ list(zip(self.starts, self.ends)), cw_start, weeks_back,
4308
+ )
4309
+
4310
+
4311
+ def _projects_week_grid(
4312
+ conn: "sqlite3.Connection",
4313
+ *,
4314
+ anchor_utc: "dt.datetime",
4315
+ weeks_back: int,
4316
+ account_key: "str | None" = None,
4317
+ ) -> "_ProjectsWeekGrid | None":
4318
+ """Build the panel's subscription-week grid, or None when no snapshot
4319
+ row carries an anchor.
4320
+
4321
+ Returning None is the deliberate no-anchor path: the caller then keeps
4322
+ ``_projects_week_start_monday_utc`` throughout, which is what that
4323
+ function was written for and what every anchorless fixture already
4324
+ exercises byte-identically.
4325
+
4326
+ Cost: one grouped read of ``weekly_usage_snapshots`` plus the reset-event
4327
+ join `_compute_subscription_weeks` already performs. It adds no walk over
4328
+ ``session_entries`` and nothing per entry, so this does not re-open the
4329
+ per-tick rescan #583 owns.
4330
+ """
4331
+ acct_pred = "" if account_key is None else " AND account_key = ?"
4332
+ acct_params: tuple = () if account_key is None else (account_key,)
4333
+ try:
4334
+ row = conn.execute(
4335
+ "SELECT COUNT(*) FROM weekly_usage_snapshots "
4336
+ "WHERE week_start_at IS NOT NULL "
4337
+ " AND week_end_at IS NOT NULL "
4338
+ " AND week_start_date IS NOT NULL"
4339
+ f"{acct_pred}",
4340
+ acct_params,
4341
+ ).fetchone()
4342
+ except sqlite3.OperationalError:
4343
+ return None
4344
+ if not row or not row[0]:
4345
+ return None
4346
+
4347
+ # A generous provisional range, defined once in `_lib_subscription_weeks`
4348
+ # because `cmd_project` needs the identical range: the extrapolation
4349
+ # anchor `_compute_subscription_weeks` picks is relative to `range_start`,
4350
+ # so two callers asking the same question over different ranges can be
4351
+ # handed differently-phased intervals for the same history.
4352
+ range_start, range_end = _cctally().subscription_window_probe_range(
4353
+ anchor_utc, weeks_back,
4354
+ )
4355
+ try:
4356
+ subweeks = _cctally()._compute_subscription_weeks(
4357
+ conn, range_start, range_end, account_key=account_key,
4358
+ )
4359
+ except Exception:
4360
+ # A malformed anchor must not take the panel down; the Monday
4361
+ # fallback still renders a coherent (if approximate) window.
4362
+ return None
4363
+ bounds: "list[tuple[dt.datetime, dt.datetime]]" = []
4364
+ for sw in subweeks:
4365
+ try:
4366
+ s = parse_iso_datetime(
4367
+ sw.start_ts, "projects week.start_ts",
4368
+ ).astimezone(dt.timezone.utc)
4369
+ e = parse_iso_datetime(
4370
+ sw.end_ts, "projects week.end_ts",
4371
+ ).astimezone(dt.timezone.utc)
4372
+ except (TypeError, ValueError):
4373
+ continue
4374
+ if e > s:
4375
+ bounds.append((s, e))
4376
+ if not bounds:
4377
+ return None
4378
+ return _ProjectsWeekGrid(bounds)
4379
+
4380
+
3324
4381
  def _projects_week_label(week_start: "dt.datetime") -> str:
3325
4382
  """Render a `wk Mon DD` label for the trend chart x-axis.
3326
4383
 
@@ -3369,9 +4426,25 @@ def _projects_iter_session_entries(conn: "sqlite3.Connection",
3369
4426
  downstream. An ``EXPLAIN QUERY PLAN`` regression asserts the mutation_seq
3370
4427
  index seek (``tests/test_projects_envelope.py``).
3371
4428
  """
3372
- since_iso = since.astimezone(dt.timezone.utc).strftime(
3373
- "%Y-%m-%dT%H:%M:%SZ"
3374
- )
4429
+ # The SQL bounds are an outward-widened CANDIDATE filter; the real
4430
+ # membership test is the Python one every caller applies
4431
+ # (`_fold_projects_entry`'s interval gate, `_week_for`, or
4432
+ # `_fetch_delta_rows`' own pre-filter). #620 S1: the lower bound is
4433
+ # widened by one second because ingestion stores
4434
+ # `timestamp.astimezone(utc).isoformat()`, which keeps a `+00:00`
4435
+ # offset, while this predicate spells its bound `Z` — and SQLite
4436
+ # compares the column lexically, where `+` (0x2B) sorts BELOW `Z`
4437
+ # (0x5A). An entry stored at exactly `since`, or in the first second
4438
+ # after it, therefore sorts below the bound and is dropped before any
4439
+ # Python gate runs. That was unreachable while a week always started at
4440
+ # Monday 00:00 UTC, which carries no entries; a real subscription week
4441
+ # starts at the reset instant, and an entry lands on it routinely. The
4442
+ # widening admits at most one extra second of candidates, which the
4443
+ # Python gate then rejects, so no caller's result changes except the one
4444
+ # that was silently losing the boundary entry.
4445
+ since_iso = (
4446
+ since.astimezone(dt.timezone.utc) - dt.timedelta(seconds=1)
4447
+ ).strftime("%Y-%m-%dT%H:%M:%SZ")
3375
4448
  until_iso = until.astimezone(dt.timezone.utc).strftime(
3376
4449
  "%Y-%m-%dT%H:%M:%SZ"
3377
4450
  )
@@ -3562,6 +4635,25 @@ def _shared_range_row_to_usage_entry(row):
3562
4635
  )
3563
4636
 
3564
4637
 
4638
+ def _shared_range_row_to_priced_usage_entry(row):
4639
+ """Prepare one shared row's effective cost once for both range folds."""
4640
+ entry = _shared_range_row_to_usage_entry(row)
4641
+ if entry.model == "<synthetic>":
4642
+ return entry
4643
+ return _cctally().UsageEntry(
4644
+ timestamp=entry.timestamp,
4645
+ model=entry.model,
4646
+ usage=entry.usage,
4647
+ cost_usd=_calculate_entry_cost(
4648
+ entry.model,
4649
+ entry.usage,
4650
+ mode="auto",
4651
+ cost_usd=entry.cost_usd,
4652
+ ),
4653
+ source_path=entry.source_path,
4654
+ )
4655
+
4656
+
3565
4657
  def _fold_prepared_daily_entries(
3566
4658
  accumulators, entries, *, display_tz=None, mode: str = "auto",
3567
4659
  ):
@@ -3681,7 +4773,9 @@ def _fold_projects_entry(
3681
4773
  *,
3682
4774
  resolver_cache: dict,
3683
4775
  week_start: "dt.datetime | None",
4776
+ week_end: "dt.datetime | None" = None,
3684
4777
  prepared_daily_entries: "list | None" = None,
4778
+ priced_entry=None,
3685
4779
  ) -> "float | None":
3686
4780
  """Fold ONE ``_projects_iter_session_entries`` row onto ``mut`` (the shared
3687
4781
  per-row body, #271 §20 Codex-P1a).
@@ -3693,6 +4787,15 @@ def _fold_projects_entry(
3693
4787
  row is filtered out (``<synthetic>`` model, or its Monday-anchored week ≠
3694
4788
  ``week_start``) — the caller then skips ``week_total`` / ``tail`` advance.
3695
4789
 
4790
+ The membership gate is the half-open interval ``[week_start, week_end)``
4791
+ (#620 S1 D1). It was a ``_projects_week_start_monday_utc(ts) ==
4792
+ week_start`` equality, which is the SAME predicate whenever ``week_start``
4793
+ is a Monday 00:00 UTC and ``week_end`` is ``week_start + 7d`` — so every
4794
+ Monday-anchored caller is byte-unchanged — but the interval form also
4795
+ admits a real subscription week that neither starts on a Monday nor runs
4796
+ a full seven days. ``week_end`` defaults to ``week_start + 7d`` for a
4797
+ caller that has not been threaded through yet.
4798
+
3696
4799
  ``mut[bp]`` is the running mutable dict ``{"cost_usd": float,
3697
4800
  "sessions": set, "first_seen": dt, "last_seen": dt, "first_order": ts_iso,
3698
4801
  "first_id": int, "first_key": ProjectKey}``. The first row seen for a
@@ -3715,32 +4818,43 @@ def _fold_projects_entry(
3715
4818
  if model == "<synthetic>":
3716
4819
  return None
3717
4820
  ts = parse_iso_datetime(ts_iso, "session_entries.timestamp_utc")
3718
- if week_start is not None and _projects_week_start_monday_utc(ts) != week_start:
3719
- return None
3720
- usage = claude_usage_dict( # #195 chokepoint
3721
- input_tokens=input_tok,
3722
- output_tokens=output_tok,
3723
- cache_creation_tokens=cache_create,
3724
- cache_read_tokens=cache_read,
3725
- cache_1h_tokens=cache_1h,
3726
- speed=speed,
3727
- )
3728
- entry_cost = _calculate_entry_cost(
3729
- model,
3730
- usage,
3731
- mode="auto",
3732
- cost_usd=cost_raw,
3733
- )
4821
+ if week_start is not None:
4822
+ w_end = (
4823
+ week_end if week_end is not None
4824
+ else week_start + dt.timedelta(days=7)
4825
+ )
4826
+ if not (week_start <= ts < w_end):
4827
+ return None
4828
+ if priced_entry is None:
4829
+ usage = claude_usage_dict( # #195 chokepoint
4830
+ input_tokens=input_tok,
4831
+ output_tokens=output_tok,
4832
+ cache_creation_tokens=cache_create,
4833
+ cache_read_tokens=cache_read,
4834
+ cache_1h_tokens=cache_1h,
4835
+ speed=speed,
4836
+ )
4837
+ entry_cost = _calculate_entry_cost(
4838
+ model,
4839
+ usage,
4840
+ mode="auto",
4841
+ cost_usd=cost_raw,
4842
+ )
4843
+ else:
4844
+ usage = priced_entry.usage
4845
+ entry_cost = priced_entry.cost_usd
3734
4846
  if prepared_daily_entries is not None:
3735
4847
  # #567: preserve the canonical daily entry and aggregator while
3736
4848
  # handing off the effective cost this pass already computed.
3737
- prepared_daily_entries.append(c.UsageEntry(
3738
- timestamp=dt.datetime.fromisoformat(ts_iso),
3739
- model=model,
3740
- usage=usage,
3741
- cost_usd=entry_cost,
3742
- source_path=source_path,
3743
- ))
4849
+ prepared_daily_entries.append(
4850
+ priced_entry if priced_entry is not None else c.UsageEntry(
4851
+ timestamp=dt.datetime.fromisoformat(ts_iso),
4852
+ model=model,
4853
+ usage=usage,
4854
+ cost_usd=entry_cost,
4855
+ source_path=source_path,
4856
+ )
4857
+ )
3744
4858
  pkey = c._resolve_project_key(project_path, "git-root", resolver_cache)
3745
4859
  bp = pkey.bucket_path
3746
4860
  a = mut.get(bp)
@@ -3769,6 +4883,7 @@ def _fold_projects_entry(
3769
4883
 
3770
4884
  def fold_projects_over_range(
3771
4885
  rows, *, resolver_cache=None, prepared_daily_entries=None,
4886
+ priced_entries=None,
3772
4887
  ) -> "dict[str, dict]":
3773
4888
  """Fold an ALREADY-MATERIALISED candidate stream into per-bucket totals.
3774
4889
 
@@ -3791,13 +4906,17 @@ def fold_projects_over_range(
3791
4906
  """
3792
4907
  mut: "dict[str, dict]" = {}
3793
4908
  cache = {} if resolver_cache is None else resolver_cache
3794
- for row in rows:
4909
+ if priced_entries is not None and len(priced_entries) != len(rows):
4910
+ raise ValueError("priced shared-range entries do not match source rows")
4911
+ for index, row in enumerate(rows):
3795
4912
  _fold_projects_entry(
3796
4913
  mut,
3797
4914
  row,
3798
4915
  resolver_cache=cache,
3799
4916
  week_start=None,
3800
4917
  prepared_daily_entries=prepared_daily_entries,
4918
+ priced_entry=(
4919
+ priced_entries[index] if priced_entries is not None else None),
3801
4920
  )
3802
4921
  return mut
3803
4922
 
@@ -4069,25 +5188,35 @@ def _shared_range_cache_payload(
4069
5188
  }
4070
5189
 
4071
5190
 
4072
- def build_cached_claude_range_aggregates(
5191
+ @dataclass(frozen=True)
5192
+ class ClaudeRangeAggregateCapture:
5193
+ """Cache-owned inputs for one post-transaction Claude range fold."""
5194
+
5195
+ base: tuple
5196
+ max_entry_id: int
5197
+ entry_mutation_seq: int
5198
+ shared_end_exclusive: object
5199
+ prior: object
5200
+ delta_rows: tuple
5201
+ full_rows: tuple | None
5202
+
5203
+
5204
+ def capture_cached_claude_range_aggregates(
4073
5205
  conn,
4074
5206
  *,
4075
5207
  shared_start,
4076
5208
  shared_end_exclusive,
4077
- now_utc,
4078
5209
  display_tz,
4079
- legacy_labels,
4080
5210
  max_entry_id: "int | None" = None,
4081
5211
  entry_mutation_seq: "int | None" = None,
4082
5212
  generation: int = 0,
4083
5213
  ):
4084
- """Build or increment the one-snapshot Claude range folds (#567).
5214
+ """Capture only cache-backed inputs for the #567 append accumulator.
4085
5215
 
4086
- Pure appends are folded onto the cached raw accumulators. A shifted range
4087
- floor, backwards clock, generation or session-file identity change,
4088
- non-monotone signature, or an id-stable mutation of an already-folded row
4089
- falls back to one full ordered pass. The cache stores no public labels, so
4090
- the current legacy population is reapplied on every publication.
5216
+ This half may run under `_tui_build_source_bundle`'s pinned transaction. It
5217
+ performs no deepcopy, pricing, project fold, daily fold, payload assembly,
5218
+ or memo mutation. The returned rows are ordinary immutable SQLite tuples,
5219
+ so the caller can end the read transaction before consuming them.
4091
5220
  """
4092
5221
  if max_entry_id is None or entry_mutation_seq is None:
4093
5222
  observed_id, observed_seq = _shared_range_entry_signature(conn)
@@ -4104,7 +5233,8 @@ def build_cached_claude_range_aggregates(
4104
5233
  generation=generation,
4105
5234
  )
4106
5235
  prior = _CLAUDE_RANGE_AGGREGATE_MEMO.get("state")
4107
- state = None
5236
+ delta_rows: tuple = ()
5237
+ full_rows: tuple | None = None
4108
5238
  if isinstance(prior, dict) and prior.get("base") == base:
4109
5239
  monotone = (
4110
5240
  max_entry_id >= prior["max_entry_id"]
@@ -4120,9 +5250,6 @@ def build_cached_claude_range_aggregates(
4120
5250
  )
4121
5251
  )
4122
5252
  if monotone and not old_row_changed:
4123
- project_mut = copy.deepcopy(prior["project_mut"])
4124
- daily_accumulators = copy.deepcopy(prior["daily_accumulators"])
4125
- resolver_cache = dict(prior["resolver_cache"])
4126
5253
  delta_by_id = {}
4127
5254
  for row in _shared_range_entries_after_id(
4128
5255
  conn, prior["max_entry_id"],
@@ -4140,85 +5267,198 @@ def build_cached_claude_range_aggregates(
4140
5267
  ):
4141
5268
  if row[0] <= prior["max_entry_id"]:
4142
5269
  delta_by_id[row[0]] = row
4143
- delta_rows = sorted(
5270
+ delta_rows = tuple(sorted(
4144
5271
  delta_by_id.values(),
4145
5272
  key=lambda row: (row[1], row[0]),
4146
- )
5273
+ ))
4147
5274
  prior_tail = prior["tail"]
4148
- if prior_tail is None or all(
5275
+ if not (prior_tail is None or all(
4149
5276
  (row[1], row[0]) > prior_tail
4150
5277
  for row in delta_rows
4151
5278
  if row[2] != "<synthetic>"
4152
- ):
4153
- prepared = []
4154
- for row in delta_rows:
4155
- _fold_projects_entry(
4156
- project_mut,
4157
- row,
4158
- resolver_cache=resolver_cache,
4159
- week_start=None,
4160
- prepared_daily_entries=prepared,
4161
- )
4162
- _fold_prepared_daily_entries(
4163
- daily_accumulators,
4164
- prepared,
4165
- display_tz=display_tz,
4166
- )
4167
- tail = prior_tail
4168
- real_delta = [
4169
- row for row in delta_rows if row[2] != "<synthetic>"
4170
- ]
4171
- if real_delta:
4172
- last = real_delta[-1]
4173
- tail = (last[1], last[0])
4174
- state = {
4175
- "base": base,
4176
- "max_entry_id": max_entry_id,
4177
- "entry_mutation_seq": entry_mutation_seq,
4178
- "end_exclusive": shared_end_exclusive,
4179
- "tail": tail,
4180
- "project_mut": project_mut,
4181
- "daily_accumulators": daily_accumulators,
4182
- "resolver_cache": resolver_cache,
4183
- }
4184
- if state is None:
4185
- rows = tuple(iter_shared_range_entries(
5279
+ )):
5280
+ # An out-of-order append cannot be folded onto the accumulator.
5281
+ # Capture the cold carrier while the same snapshot is pinned;
5282
+ # discovering this after rollback would require a second read
5283
+ # generation.
5284
+ full_rows = tuple(iter_shared_range_entries(
5285
+ conn, start=shared_start,
5286
+ end_exclusive=shared_end_exclusive,
5287
+ ))
5288
+ delta_rows = ()
5289
+ else:
5290
+ full_rows = tuple(iter_shared_range_entries(
5291
+ conn, start=shared_start,
5292
+ end_exclusive=shared_end_exclusive,
5293
+ ))
5294
+ else:
5295
+ full_rows = tuple(iter_shared_range_entries(
4186
5296
  conn, start=shared_start, end_exclusive=shared_end_exclusive,
4187
5297
  ))
4188
- prepared = []
4189
- resolver_cache = {}
4190
- project_mut = fold_projects_over_range(
4191
- rows,
4192
- resolver_cache=resolver_cache,
4193
- prepared_daily_entries=prepared,
5298
+ return ClaudeRangeAggregateCapture(
5299
+ base=base,
5300
+ max_entry_id=max_entry_id,
5301
+ entry_mutation_seq=entry_mutation_seq,
5302
+ shared_end_exclusive=shared_end_exclusive,
5303
+ prior=prior,
5304
+ delta_rows=delta_rows,
5305
+ full_rows=full_rows,
5306
+ )
5307
+
5308
+
5309
+ def build_cached_claude_range_aggregates_from_capture(
5310
+ capture: ClaudeRangeAggregateCapture,
5311
+ *,
5312
+ now_utc,
5313
+ display_tz,
5314
+ legacy_labels,
5315
+ tolerate_leg_failures: bool = False,
5316
+ ):
5317
+ """Fold and publish one captured Claude range accumulator outside the pin.
5318
+
5319
+ ``tolerate_leg_failures`` is the source-bundle path's typed degradation
5320
+ seam. Project identity resolution and the daily calendar are independent
5321
+ folds over the same captured rows, so a project-only fault must not discard
5322
+ a valid daily result. The one-shot compatibility wrapper keeps the legacy
5323
+ raise-on-any-fault contract by leaving it false.
5324
+ """
5325
+ prior = capture.prior
5326
+ incremental = capture.full_rows is None and isinstance(prior, dict)
5327
+ rows = capture.delta_rows if incremental else (capture.full_rows or ())
5328
+ payload: dict[str, object] = {}
5329
+ outcomes = {
5330
+ "projects": {"state": "ok"},
5331
+ "daily": {"state": "ok"},
5332
+ }
5333
+ failures: dict[str, Exception] = {}
5334
+ project_mut = None
5335
+ resolver_cache = None
5336
+ daily_accumulators = None
5337
+
5338
+ try:
5339
+ prepared = tuple(
5340
+ _shared_range_row_to_priced_usage_entry(row) for row in rows)
5341
+ except Exception as exc:
5342
+ prepared = ()
5343
+ failures["projects"] = exc
5344
+ failures["daily"] = exc
5345
+ outcomes["projects"] = {
5346
+ "state": "failed", "code": "claude_fold_failed"}
5347
+ outcomes["daily"] = {
5348
+ "state": "failed", "code": "claude_fold_failed"}
5349
+
5350
+ try:
5351
+ if "projects" in failures:
5352
+ raise failures["projects"]
5353
+ if incremental:
5354
+ project_mut = copy.deepcopy(prior["project_mut"])
5355
+ resolver_cache = dict(prior["resolver_cache"])
5356
+ for row, priced_entry in zip(rows, prepared):
5357
+ _fold_projects_entry(
5358
+ project_mut,
5359
+ row,
5360
+ resolver_cache=resolver_cache,
5361
+ week_start=None,
5362
+ priced_entry=priced_entry,
5363
+ )
5364
+ else:
5365
+ resolver_cache = {}
5366
+ project_mut = fold_projects_over_range(
5367
+ rows,
5368
+ resolver_cache=resolver_cache,
5369
+ priced_entries=prepared,
5370
+ )
5371
+ payload["projects"] = _project_aggregate_rows_from_folded(
5372
+ project_mut, legacy_labels)
5373
+ except Exception as exc:
5374
+ failures["projects"] = exc
5375
+ outcomes["projects"] = {
5376
+ "state": "failed", "code": "claude_fold_failed"}
5377
+
5378
+ try:
5379
+ if "daily" in failures:
5380
+ raise failures["daily"]
5381
+ daily_accumulators = (
5382
+ copy.deepcopy(prior["daily_accumulators"])
5383
+ if incremental else {}
4194
5384
  )
4195
- daily_accumulators = {}
4196
5385
  _fold_prepared_daily_entries(
4197
- daily_accumulators, prepared, display_tz=display_tz,
4198
- )
5386
+ daily_accumulators, prepared, display_tz=display_tz)
5387
+ daily_buckets = _finalize_daily_accumulators(daily_accumulators)
5388
+ daily_rows = _build_daily_aggregate_rows_from_buckets(
5389
+ daily_buckets, now_utc=now_utc, display_tz=display_tz)
5390
+ c = _cctally()
5391
+ payload["daily"] = [
5392
+ c.daily_panel_row_to_wire(row) for row in daily_rows]
5393
+ except Exception as exc:
5394
+ failures["daily"] = exc
5395
+ outcomes["daily"] = {
5396
+ "state": "failed", "code": "claude_fold_failed"}
5397
+
5398
+ if incremental:
5399
+ tail = prior["tail"]
5400
+ real_delta = [
5401
+ row for row in capture.delta_rows if row[2] != "<synthetic>"
5402
+ ]
5403
+ if real_delta:
5404
+ last = real_delta[-1]
5405
+ tail = (last[1], last[0])
5406
+ else:
4199
5407
  real_rows = [row for row in rows if row[2] != "<synthetic>"]
4200
5408
  tail = None
4201
5409
  if real_rows:
4202
5410
  last = real_rows[-1]
4203
5411
  tail = (last[1], last[0])
5412
+
5413
+ if not failures:
4204
5414
  state = {
4205
- "base": base,
4206
- "max_entry_id": max_entry_id,
4207
- "entry_mutation_seq": entry_mutation_seq,
4208
- "end_exclusive": shared_end_exclusive,
5415
+ "base": capture.base,
5416
+ "max_entry_id": capture.max_entry_id,
5417
+ "entry_mutation_seq": capture.entry_mutation_seq,
5418
+ "end_exclusive": capture.shared_end_exclusive,
4209
5419
  "tail": tail,
4210
5420
  "project_mut": project_mut,
4211
5421
  "daily_accumulators": daily_accumulators,
4212
5422
  "resolver_cache": resolver_cache,
4213
5423
  }
4214
- payload = _shared_range_cache_payload(
4215
- state,
4216
- legacy_labels=legacy_labels,
5424
+ _cctally()._load_sibling("_lib_snapshot_cache")._assert_owner()
5425
+ _CLAUDE_RANGE_AGGREGATE_MEMO["state"] = state
5426
+
5427
+ if failures and not tolerate_leg_failures:
5428
+ raise next(iter(failures.values()))
5429
+ if tolerate_leg_failures:
5430
+ return payload, outcomes
5431
+ return payload
5432
+
5433
+
5434
+ def build_cached_claude_range_aggregates(
5435
+ conn,
5436
+ *,
5437
+ shared_start,
5438
+ shared_end_exclusive,
5439
+ now_utc,
5440
+ display_tz,
5441
+ legacy_labels,
5442
+ max_entry_id: "int | None" = None,
5443
+ entry_mutation_seq: "int | None" = None,
5444
+ generation: int = 0,
5445
+ ):
5446
+ """One-shot compatibility wrapper for non-pinned focused callers."""
5447
+ capture = capture_cached_claude_range_aggregates(
5448
+ conn,
5449
+ shared_start=shared_start,
5450
+ shared_end_exclusive=shared_end_exclusive,
5451
+ display_tz=display_tz,
5452
+ max_entry_id=max_entry_id,
5453
+ entry_mutation_seq=entry_mutation_seq,
5454
+ generation=generation,
5455
+ )
5456
+ return build_cached_claude_range_aggregates_from_capture(
5457
+ capture,
4217
5458
  now_utc=now_utc,
4218
5459
  display_tz=display_tz,
5460
+ legacy_labels=legacy_labels,
4219
5461
  )
4220
- _CLAUDE_RANGE_AGGREGATE_MEMO["state"] = state
4221
- return payload
4222
5462
 
4223
5463
 
4224
5464
  def _aggregate_projects_week_raw(
@@ -4249,7 +5489,8 @@ def _aggregate_projects_week_raw(
4249
5489
  conn, since=week_start, until=week_end,
4250
5490
  ):
4251
5491
  entry_cost = _fold_projects_entry(
4252
- mut, row, resolver_cache=resolver_cache, week_start=week_start,
5492
+ mut, row, resolver_cache=resolver_cache,
5493
+ week_start=week_start, week_end=week_end,
4253
5494
  )
4254
5495
  if entry_cost is None:
4255
5496
  continue
@@ -4312,7 +5553,7 @@ def _aggregate_projects_week(
4312
5553
  def _assemble_projects_via_cache(
4313
5554
  conn: "sqlite3.Connection",
4314
5555
  *,
4315
- weeks_full: "list[dt.datetime]",
5556
+ week_bounds: "list[tuple[dt.datetime, dt.datetime]]",
4316
5557
  cw_start: "dt.datetime",
4317
5558
  cw_end: "dt.datetime",
4318
5559
  cur_max_id: int,
@@ -4387,33 +5628,47 @@ def _assemble_projects_via_cache(
4387
5628
  if r[2] == "<synthetic>": # r[2] = model
4388
5629
  continue
4389
5630
  ts = parse_iso_datetime(r[1], "session_entries.timestamp_utc")
4390
- if _projects_week_start_monday_utc(ts) != cw_start:
5631
+ if not (cw_start <= ts < cw_end):
4391
5632
  continue
4392
5633
  out.append(r)
4393
5634
  out.sort(key=lambda r: (r[1], r[0])) # (ts_iso, id)
4394
5635
  return out
4395
5636
 
4396
- for w in weeks_full:
5637
+ # The cache identity below is `(start, end)`, while the spec named
5638
+ # "account, exact start, exact end". The account axis is omitted
5639
+ # deliberately, not by oversight: this panel always folds merged
5640
+ # (`_projects_week_grid` is called here with `account_key=None`), so
5641
+ # every entry in this cache was produced by the one merged read and two
5642
+ # scopes cannot collide in it. Adding a constant third component would
5643
+ # be a key that never varies. If the panel ever gains an account scope,
5644
+ # the axis has to be added at the same time — an account-scoped fold
5645
+ # served from a merged slot is a wrong answer, not a stale one.
5646
+ for w, w_end in week_bounds:
4397
5647
  if w == cw_start:
4398
5648
  week_buckets, week_total = sc.accumulate_projects_current_week(
4399
- week_key=sc.projects_env_week_key(cw_start),
5649
+ # #620 S1: the accumulator's identity is the INTERVAL. An
5650
+ # early reset that moves the current week's bounds must
5651
+ # cold-refold the slot rather than keep appending to a
5652
+ # running aggregate folded over the old window.
5653
+ week_key=sc.projects_env_week_key(cw_start, cw_end),
4400
5654
  cur_max_id=cur_max_id,
4401
5655
  cur_max_seq=cur_max_seq,
4402
5656
  fetch_all_raw=_fetch_all_raw,
4403
5657
  fetch_delta_rows=_fetch_delta_rows,
4404
5658
  finalize=_finalize_projects_mut,
4405
5659
  fold=lambda mut, row: _fold_projects_entry(
4406
- mut, row, resolver_cache=resolver_cache, week_start=cw_start,
5660
+ mut, row, resolver_cache=resolver_cache,
5661
+ week_start=cw_start, week_end=cw_end,
4407
5662
  ),
4408
5663
  )
4409
5664
  else:
4410
- week_iso = sc.projects_env_week_key(w)
5665
+ week_iso = sc.projects_env_week_key(w, w_end)
4411
5666
  hit = sc.projects_env_week_get(week_iso)
4412
5667
  if hit is not None:
4413
5668
  week_buckets, week_total = hit
4414
5669
  else:
4415
5670
  week_buckets, week_total = _aggregate_projects_week(
4416
- conn, week_start=w, week_end=w + dt.timedelta(days=7),
5671
+ conn, week_start=w, week_end=w_end,
4417
5672
  resolver_cache=resolver_cache,
4418
5673
  )
4419
5674
  sc.projects_env_week_put(week_iso, week_buckets, week_total)
@@ -4437,14 +5692,21 @@ def _build_projects_envelope(
4437
5692
  shape from spec §5.2 (no per-model breakdowns, no first/last seen
4438
5693
  per session, no per-row $/1%; just cost / attributed_pct / sessions).
4439
5694
 
4440
- Week boundaries follow ``cmd_project``'s Monday-anchored UTC
4441
- fallback (``bin/cctally:4711``); ``weekly_usage_snapshots`` rows are
4442
- matched by ``week_start_date`` (date-only) for ``attributed_pct``.
4443
-
4444
- ``current_week`` is passed through opaquely — if non-None and
4445
- carrying a ``.week_start_at`` UTC datetime, that boundary supplants
4446
- the Monday fallback for the current week's bucket. None (the
4447
- default) preserves the fallback.
5695
+ Week boundaries are the real subscription intervals (#620 S1 D1):
5696
+ ``_projects_week_grid`` derives them from ``_compute_subscription_weeks``,
5697
+ the same kernel ``cmd_project`` buckets by, so the two surfaces attribute
5698
+ the same projects over the same weeks. A ``weekly_usage_snapshots`` row
5699
+ is matched onto an interval by its ``week_start_at`` anchor, falling back
5700
+ to ``week_start_date`` for a legacy row that carries no anchor; a row
5701
+ that matches no interval contributes nothing, leaving ``attributed_pct``
5702
+ None rather than attributing over a mismatched population. Only a store
5703
+ with no anchored snapshot at all falls back to
5704
+ ``_projects_week_start_monday_utc``, which is what that function was
5705
+ written for.
5706
+
5707
+ ``current_week`` is passed through opaquely — if non-None and carrying a
5708
+ ``.week_start_at`` UTC datetime, that instant selects which interval is
5709
+ the current week. None (the default) uses ``now_utc``.
4448
5710
 
4449
5711
  Determinism: same conn + same ``now_utc`` ⇒ byte-identical JSON
4450
5712
  (R-PROJ5 invariant). Per-tick memoized on
@@ -4499,33 +5761,51 @@ def _build_projects_envelope(
4499
5761
  return cached
4500
5762
 
4501
5763
  # ---- Week-start anchor (current subscription week) ------------------
4502
- # ``TuiCurrentWeek.week_start_at`` is NOT a valid Monday lookup key
4503
- # after ``_apply_midweek_reset_override`` — it is shifted to the
4504
- # in-week reset instant (e.g. Friday 13:00 UTC) while the bucket
4505
- # aggregator below snaps every entry to its containing ISO-Monday
4506
- # via ``_week_for``. Using ``cw_key`` directly as the bucket-lookup
4507
- # key strands all current-week activity in an empty bucket and emits
4508
- # ``rows: []`` with ``total_cost_usd: 0.0``. Snap to the canonical
4509
- # Monday-UTC week anchor here so the lookup keys align — same
4510
- # invariant the weekly handling notes call out for
4511
- # ``weekly_usage_snapshots``/``percent_milestones`` cross-table
4512
- # joins. Regression: ``tests/fixtures/dashboard/reset-week/`` +
5764
+ # #620 S1 D1. The panel buckets cost into the REAL subscription
5765
+ # intervals — the ones `_compute_subscription_weeks` derives from the
5766
+ # retained reset anchors, and the ones `cmd_project` already buckets by.
5767
+ # `_projects_week_start_monday_utc` remains what its own docstring says
5768
+ # it is: the fallback for when no anchor is available. It used to be
5769
+ # applied to the anchor itself, which discarded the very thing it was
5770
+ # written to defer to, so the cost window and the quota window described
5771
+ # different intervals for every account whose reset is not exactly
5772
+ # Monday midnight UTC — in practice almost all of them.
5773
+ #
5774
+ # ``TuiCurrentWeek.week_start_at`` after ``_apply_midweek_reset_override``
5775
+ # is the in-week reset instant rather than the week's start. It is used
5776
+ # here only to locate the containing interval, never as a bucket key, so
5777
+ # a shifted value resolves to the same week the entry walk uses and no
5778
+ # activity is stranded. Regression:
5779
+ # ``tests/fixtures/dashboard/reset-week/`` +
4513
5780
  # ``test_current_week_rows_populated_after_midweek_reset``.
4514
- if cw_key is not None:
4515
- cw_start = _projects_week_start_monday_utc(cw_key)
5781
+ anchor_instant = cw_key if cw_key is not None else now_utc
5782
+ grid = _projects_week_grid(
5783
+ conn, anchor_utc=anchor_instant, weeks_back=weeks_back,
5784
+ )
5785
+ cw_start = grid.week_for(anchor_instant) if grid is not None else None
5786
+ if cw_start is None:
5787
+ # No anchor covers `now` — the genuine fallback path, byte-identical
5788
+ # to the pre-#620 behaviour for a store with no anchored snapshots.
5789
+ grid = None
5790
+ cw_start = _projects_week_start_monday_utc(anchor_instant)
5791
+ cw_end = cw_start + dt.timedelta(days=7)
5792
+ week_bounds = [
5793
+ (
5794
+ cw_start - dt.timedelta(days=7 * (weeks_back - 1 - i)),
5795
+ cw_start - dt.timedelta(days=7 * (weeks_back - 2 - i)),
5796
+ )
5797
+ for i in range(weeks_back)
5798
+ ]
4516
5799
  else:
4517
- cw_start = _projects_week_start_monday_utc(now_utc)
5800
+ cw_end = grid.end_for(cw_start)
5801
+ week_bounds = grid.window_ending_at(cw_start, weeks_back)
4518
5802
 
4519
- # Build a list of canonical Monday-anchored week starts ending with
4520
- # cw_start, oldest → newest, of length ``weeks_back``. Clamping to
5803
+ # Week starts, oldest → newest, of length ``weeks_back``. Clamping to
4521
5804
  # actual history happens after the entry walk reveals what weeks
4522
5805
  # have any activity.
4523
- weeks_full = [
4524
- cw_start - dt.timedelta(days=7 * (weeks_back - 1 - i))
4525
- for i in range(weeks_back)
4526
- ]
4527
- cw_end = cw_start + dt.timedelta(days=7)
4528
- since_dt = weeks_full[0]
5806
+ weeks_full = [b[0] for b in week_bounds]
5807
+ end_by_week = {s: e for s, e in week_bounds}
5808
+ since_dt = week_bounds[0][0]
4529
5809
  until_dt = cw_end # exclusive end; SQL is `>= since AND <= until`
4530
5810
 
4531
5811
  # ---- Bucket entries per (ProjectKey, week_start) --------------------
@@ -4540,7 +5820,7 @@ def _build_projects_envelope(
4540
5820
  # HTTP-drill): the original single full-window walk, byte-unchanged.
4541
5821
  if use_projects_env_cache:
4542
5822
  buckets, total_cost_by_week, key_by_bucket = _assemble_projects_via_cache(
4543
- conn, weeks_full=weeks_full, cw_start=cw_start, cw_end=cw_end,
5823
+ conn, week_bounds=week_bounds, cw_start=cw_start, cw_end=cw_end,
4544
5824
  cur_max_id=max_id, cur_max_seq=entry_mutation_seq,
4545
5825
  )
4546
5826
  else:
@@ -4556,8 +5836,13 @@ def _build_projects_envelope(
4556
5836
  key_by_bucket = {}
4557
5837
 
4558
5838
  def _week_for(ts: dt.datetime) -> "dt.datetime | None":
4559
- wstart = _projects_week_start_monday_utc(ts)
4560
- if wstart < weeks_full[0] or wstart > weeks_full[-1]:
5839
+ if grid is not None:
5840
+ wstart = grid.week_for(ts)
5841
+ if wstart is None:
5842
+ return None
5843
+ else:
5844
+ wstart = _projects_week_start_monday_utc(ts)
5845
+ if wstart not in end_by_week:
4561
5846
  return None
4562
5847
  return wstart
4563
5848
 
@@ -4653,7 +5938,7 @@ def _build_projects_envelope(
4653
5938
  weekly_pct_by_week: dict[dt.datetime, float] = {}
4654
5939
  try:
4655
5940
  cur = conn.execute(
4656
- "SELECT week_start_date, weekly_percent "
5941
+ "SELECT week_start_date, week_start_at, weekly_percent "
4657
5942
  "FROM weekly_usage_snapshots "
4658
5943
  "ORDER BY captured_at_utc ASC, id ASC"
4659
5944
  )
@@ -4662,19 +5947,40 @@ def _build_projects_envelope(
4662
5947
  # No weekly_usage_snapshots table — leaves attributed_pct = None
4663
5948
  # throughout (acceptable per spec §2.7).
4664
5949
  rows = []
4665
- for week_date_str, weekly_pct in rows:
5950
+ for week_date_str, week_start_at, weekly_pct in rows:
4666
5951
  try:
4667
5952
  wd = dt.date.fromisoformat(week_date_str)
4668
5953
  except (TypeError, ValueError):
4669
5954
  continue
4670
- # Snap the date to UTC Monday 00:00 (matches the bucketing key).
4671
- wstart = dt.datetime.combine(
4672
- wd, dt.time(0, 0, 0), tzinfo=dt.timezone.utc,
4673
- )
4674
- # Snap to Monday (snapshot rows that captured a non-Monday week
4675
- # boundary still align to the same canonical bucket as the entry
4676
- # walk, since the bucketing is Monday-anchored).
4677
- wstart = _projects_week_start_monday_utc(wstart)
5955
+ wstart: "dt.datetime | None" = None
5956
+ if grid is not None:
5957
+ # #620 S1 D1. Resolve the row onto the SAME interval the entry
5958
+ # walk buckets into, so the numerator, the denominator and this
5959
+ # percentage are all taken from one half-open window before the
5960
+ # multiplication below.
5961
+ if week_start_at:
5962
+ try:
5963
+ anchor = parse_iso_datetime(
5964
+ week_start_at, "weekly_usage_snapshots.week_start_at",
5965
+ ).astimezone(dt.timezone.utc)
5966
+ except (TypeError, ValueError):
5967
+ anchor = None
5968
+ if anchor is not None:
5969
+ wstart = grid.week_for(anchor)
5970
+ if wstart is None:
5971
+ # A legacy row carrying only the date-only boundary. One
5972
+ # shared interval still serves both the cost and the
5973
+ # percentage; taking them from different intervals is never
5974
+ # acceptable, so an unresolvable row contributes nothing and
5975
+ # `attributed_pct` stays None (#620 S1 A3).
5976
+ wstart = grid.start_for_date(wd)
5977
+ if wstart is None:
5978
+ continue
5979
+ else:
5980
+ # No anchor anywhere in the store: the genuine Monday fallback.
5981
+ wstart = _projects_week_start_monday_utc(dt.datetime.combine(
5982
+ wd, dt.time(0, 0, 0), tzinfo=dt.timezone.utc,
5983
+ ))
4678
5984
  if weekly_pct is not None:
4679
5985
  weekly_pct_by_week[wstart] = float(weekly_pct)
4680
5986
 
@@ -4701,12 +6007,11 @@ def _build_projects_envelope(
4701
6007
  if weeks_with_activity:
4702
6008
  # Window = inclusive [oldest_active_week, cw_start]. Always emits
4703
6009
  # cw_start (panel + trend share the same current_week column).
6010
+ # Keyed by the SAME rule as the current week (#620 S1 D1) — the
6011
+ # subscription intervals themselves, not a seven-day walk, because
6012
+ # a mixed keying inside one panel is the defect restated.
4704
6013
  oldest = min(weeks_with_activity[0], cw_start)
4705
- trend_weeks = []
4706
- w = oldest
4707
- while w <= cw_start:
4708
- trend_weeks.append(w)
4709
- w += dt.timedelta(days=7)
6014
+ trend_weeks = [w for w in weeks_full if oldest <= w <= cw_start]
4710
6015
  else:
4711
6016
  trend_weeks = [cw_start]
4712
6017
 
@@ -4898,13 +6203,74 @@ def _project_detail_for_window(
4898
6203
  if bucket_path is None:
4899
6204
  return None
4900
6205
 
4901
- # ---- Window bounds (Monday-anchored UTC fallback, like the builder) -
6206
+ # ---- Window bounds, from the envelope's own current-week anchor ----
6207
+ # That anchor is the account's real subscription week start; the
6208
+ # Monday-midnight snap is only the no-anchor fallback (#620).
4902
6209
  cw_start = parse_iso_datetime(
4903
6210
  env["current_week"]["week_start_at"],
4904
6211
  "projects.current_week.week_start_at",
6212
+ ).astimezone(dt.timezone.utc)
6213
+
6214
+ # The drill resolves the SAME interval its panel resolved, by rebuilding
6215
+ # the panel's grid rather than stepping back in seven-day multiples.
6216
+ # `_ProjectsWeekGrid` exists because a drifted reset day produces a
6217
+ # genuinely short week: on `non-monday-anchor` at `weeks_back=4` the grid
6218
+ # starts the window at 2026-03-27T09:00Z while a seven-day walk yields
6219
+ # 2026-03-26T09:00Z, and an early reset that shortens the current week
6220
+ # would likewise leave a `cw_start + 7d` end counting cost past the
6221
+ # week's real end. `window_start_at` / `window_end_at` publish these
6222
+ # bounds to the client as authoritative, so a divergence here renders a
6223
+ # window the panel never computed.
6224
+ #
6225
+ # The grid is rebuilt from the PANEL'S OWN anchor, not from `cw_start`.
6226
+ # `_projects_week_grid` derives its provisional range from an ISO-Monday
6227
+ # snap of whatever anchor it is handed, and `cw_start` is the interval
6228
+ # START while the panel anchors on `current_week.week_start_at` — which
6229
+ # after `_apply_midweek_reset_override` is the in-week reset instant and
6230
+ # can sit up to a week later. The two therefore snap to different Mondays
6231
+ # and `_compute_subscription_weeks` can pick a different extrapolation
6232
+ # anchor for each, so window equality would rest on a coincidence rather
6233
+ # than on the two surfaces asking the same question. `cw_start` remains
6234
+ # the anchor `window_ending_at` walks back from, because that walk needs
6235
+ # an interval start.
6236
+ panel_anchor = getattr(current_week, "week_start_at", None)
6237
+ if not isinstance(panel_anchor, dt.datetime):
6238
+ panel_anchor = now_utc
6239
+ detail_grid = _projects_week_grid(
6240
+ conn, anchor_utc=panel_anchor, weeks_back=weeks_back,
6241
+ )
6242
+ detail_bounds = (
6243
+ detail_grid.window_ending_at(cw_start, weeks_back)
6244
+ if detail_grid is not None else []
6245
+ )
6246
+ if detail_bounds:
6247
+ since_dt = detail_bounds[0][0]
6248
+ until_dt = detail_bounds[-1][1]
6249
+ else:
6250
+ # No anchor covers this start — the same no-anchor tail the panel
6251
+ # falls back to, where the seven-day assumption is the right one.
6252
+ since_dt = cw_start - dt.timedelta(days=7 * (weeks_back - 1))
6253
+ until_dt = cw_start + dt.timedelta(days=7)
6254
+ since_iso = since_dt.astimezone(dt.timezone.utc).strftime(
6255
+ "%Y-%m-%dT%H:%M:%SZ"
6256
+ )
6257
+ until_iso = until_dt.astimezone(dt.timezone.utc).strftime(
6258
+ "%Y-%m-%dT%H:%M:%SZ"
4905
6259
  )
4906
- since_dt = cw_start - dt.timedelta(days=7 * (weeks_back - 1))
4907
- until_dt = cw_start + dt.timedelta(days=7)
6260
+ # SQL candidate bound, NOT the published one. Ingestion stores
6261
+ # `timestamp_utc` as `…+00:00` while these bounds are spelled `…Z`, and
6262
+ # SQLite compares that column lexically with `+` (0x2B) below `Z` (0x5A).
6263
+ # So the lower bound drops an entry sitting exactly on it and the upper
6264
+ # bound admits one sitting exactly on it — an asymmetry in both
6265
+ # directions. Widening the lower bound by a second makes SQL an outward
6266
+ # candidate filter at both ends; the half-open membership test is then
6267
+ # enforced on the PARSED datetime in the entry loop, which is the only
6268
+ # place it can be stated honestly.
6269
+ since_sql_iso = (
6270
+ since_dt.astimezone(dt.timezone.utc) - dt.timedelta(seconds=1)
6271
+ ).strftime("%Y-%m-%dT%H:%M:%SZ")
6272
+ until_dt_utc = until_dt.astimezone(dt.timezone.utc)
6273
+ since_dt_utc = since_dt.astimezone(dt.timezone.utc)
4908
6274
 
4909
6275
  # ---- Build bucket → source_paths map for SQL-side scoping ----------
4910
6276
  # Walk session_files (~8k rows) once instead of session_entries
@@ -4956,6 +6322,8 @@ def _project_detail_for_window(
4956
6322
  "key": project_key,
4957
6323
  "bucket_path": bucket_path,
4958
6324
  "window_weeks": weeks_back,
6325
+ "window_start_at": since_iso,
6326
+ "window_end_at": until_iso,
4959
6327
  "window_cost_usd": 0.0,
4960
6328
  "window_attributed_pct": None,
4961
6329
  "models": [],
@@ -4979,13 +6347,6 @@ def _project_detail_for_window(
4979
6347
  [(p,) for p in bucket_source_paths],
4980
6348
  )
4981
6349
 
4982
- since_iso = since_dt.astimezone(dt.timezone.utc).strftime(
4983
- "%Y-%m-%dT%H:%M:%SZ"
4984
- )
4985
- until_iso = until_dt.astimezone(dt.timezone.utc).strftime(
4986
- "%Y-%m-%dT%H:%M:%SZ"
4987
- )
4988
-
4989
6350
  # ---- Walk session_entries (project-scoped) once -------------------
4990
6351
  # INNER JOIN to _drill_paths drops every row whose source_path
4991
6352
  # doesn't belong to this bucket. The Python-side filter that
@@ -5001,7 +6362,7 @@ def _project_detail_for_window(
5001
6362
  "LEFT JOIN session_files sf ON sf.path = e.source_path "
5002
6363
  "WHERE e.timestamp_utc >= ? AND e.timestamp_utc <= ? "
5003
6364
  "ORDER BY e.timestamp_utc ASC, e.id ASC",
5004
- (since_iso, until_iso),
6365
+ (since_sql_iso, until_iso),
5005
6366
  )
5006
6367
 
5007
6368
  # Per-model rollup: {model -> {cost_usd, sessions, in, out, cache_*}}
@@ -5024,6 +6385,14 @@ def _project_detail_for_window(
5024
6385
  # on _drill_paths already restricted the result set to entries
5025
6386
  # whose source_path belongs to this bucket.
5026
6387
  ts = parse_iso_datetime(ts_iso, "session_entries.timestamp_utc")
6388
+ # The half-open membership test. The SQL bounds above are a widened
6389
+ # candidate filter that admits a second on each side, because the
6390
+ # column's stored offset spelling and the bound's spelling do not
6391
+ # compare the way the interval means; this is where the interval is
6392
+ # actually decided.
6393
+ ts_utc = ts.astimezone(dt.timezone.utc)
6394
+ if not (since_dt_utc <= ts_utc < until_dt_utc):
6395
+ continue
5027
6396
  entry_cost = _calculate_entry_cost(
5028
6397
  model,
5029
6398
  claude_usage_dict( # #195 chokepoint
@@ -5146,6 +6515,8 @@ def _project_detail_for_window(
5146
6515
  "key": project_key,
5147
6516
  "bucket_path": bucket_path,
5148
6517
  "window_weeks": weeks_back,
6518
+ "window_start_at": since_iso,
6519
+ "window_end_at": until_iso,
5149
6520
  "window_cost_usd": window_cost,
5150
6521
  "window_attributed_pct": win_pct,
5151
6522
  "models": models_out,
@@ -5301,14 +6672,31 @@ def _channel_env_fragment() -> dict:
5301
6672
  return {}
5302
6673
 
5303
6674
 
5304
- # Bounded wait for /api/sync's lock acquisition. The periodic background
5305
- # sync thread holds sync_lock during sync_cache + snapshot build (often
5306
- # 100-1500ms under active CC sessions); a non-blocking try_acquire would
5307
- # 503 the user's click whenever it lands inside that window, silently
5308
- # dropping their refresh-usage intent. 2s is generous enough to span a
5309
- # normal periodic tick yet short enough to surface a stuck rebuild as
5310
- # 503 instead of hanging the request indefinitely.
5311
- _DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS = 2.0
6675
+
6676
+ # Upper bound on the ONE blocking `sync_lock` acquire left in the tree: a
6677
+ # manual `POST /api/sync` under `--no-sync`. Every other path either acquires
6678
+ # non-blocking or enqueues. #583 S2 removed the old bounded acquire along with
6679
+ # its 503, which left this one bare — and a bare acquire pins an HTTP handler
6680
+ # thread forever when a rebuild wedges (a `cache.db` read that never returns, a
6681
+ # hung builder), with no diagnostic at all.
6682
+ #
6683
+ # 30s rather than the old 2s because the holder in this mode is a whole
6684
+ # synchronous rebuild (measured 1.9-6.3 s), not a periodic tick, so the bound
6685
+ # has to be a wedge detector rather than a contention timeout. On expiry the
6686
+ # endpoint answers 200 with a `sync_busy` warning: that stays inside its
6687
+ # declared status vocabulary, does NOT reintroduce 503, and does not strand the
6688
+ # client the way a 202 would in a mode where nothing drains the queue.
6689
+ #
6690
+ # Recorded, not resolved: 30 s also outlives any plausible browser fetch
6691
+ # timeout, so on a real wedge the client aborts first and this handler writes
6692
+ # its 200 to a socket nobody is reading — broken-pipe noise in the dashboard's
6693
+ # terminal. That interaction is exactly why the earlier 2.0 s bound was chosen
6694
+ # against a 3.0 s client timeout
6695
+ # (docs/superpowers/specs/2026-06-13-refresh-usage-dashboard-nudge-design.md,
6696
+ # the "Timeout chosen above the server's lock-wait" bullet). Deciding between
6697
+ # the two needs the client-side timeout in view as well, which is out of scope
6698
+ # here; a later session should pick one with both numbers in front of it.
6699
+ _DASHBOARD_NO_SYNC_LOCK_TIMEOUT_SECONDS = 30.0
5312
6700
 
5313
6701
 
5314
6702
  # === DashboardHTTPHandler (the /api/* + static surface) ===================
@@ -5671,6 +7059,8 @@ _POST_ROUTES = (
5671
7059
  ("exact", "/api/share/presets/rename", "_handle_share_presets_rename_post",
5672
7060
  None, False),
5673
7061
  ("exact", "/api/share/history", "_handle_share_history_post", None, False),
7062
+ ("exact", "/api/debug/backend/trace", "_handle_post_debug_backend_trace",
7063
+ None, False),
5674
7064
  )
5675
7065
 
5676
7066
  _DELETE_ROUTES = (
@@ -5950,15 +7340,54 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
5950
7340
  )
5951
7341
  self.end_headers()
5952
7342
 
7343
+ @classmethod
7344
+ def publish_activity(cls) -> None:
7345
+ """Republish the held snapshot so a new counter reaches clients now.
7346
+
7347
+ #583 S2 spec 6.2 publication point 1: acknowledge an accepted request
7348
+ without waiting for a rebuild. The reference re-stamps its held
7349
+ snapshot on every mutation, so ``get()`` already carries the new
7350
+ ``requested_id``.
7351
+ """
7352
+ ref = getattr(cls, "snapshot_ref", None)
7353
+ hub = getattr(cls, "hub", None)
7354
+ if ref is None or hub is None:
7355
+ return
7356
+ hub.publish(ref.get())
7357
+
7358
+ @classmethod
7359
+ def mark_rebuilding(cls, value: bool) -> None:
7360
+ """Publish the in-flight flag for a HANDLER-driven rebuild.
7361
+
7362
+ #583 S2 §6.3. `rebuilding` describes the rebuild, not the thread that
7363
+ started it, and an UNCONTENDED manual refresh rebuilds synchronously
7364
+ right here — `202 queued` is only the contended branch. Without this
7365
+ pair a user clicking the sync chip ran a multi-second rebuild during
7366
+ which every other connected tab published `rebuilding: false` and could
7367
+ not tell a busy dashboard from a wedged one.
7368
+
7369
+ Publishes only on a real transition, exactly like the sync loop's
7370
+ collaborator, so the rebuild's own terminal publish (`set_final`) leaves
7371
+ the trailing mark with nothing to say.
7372
+ """
7373
+ ref = getattr(cls, "snapshot_ref", None)
7374
+ hub = getattr(cls, "hub", None)
7375
+ if ref is None or hub is None:
7376
+ return
7377
+ if ref.mark_rebuilding(value):
7378
+ hub.publish(ref.get())
7379
+
5953
7380
  def _handle_post_sync(self) -> None:
5954
7381
  """Trigger refresh-usage + snapshot rebuild on user demand.
5955
7382
 
5956
7383
  Flow:
5957
7384
  1. Origin/Host CSRF check.
5958
- 2. acquire(timeout=_DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS) -> 503
5959
- only on truly degenerate contention beyond the timeout.
5960
- 3. With lock held: under --no-sync skip refresh; otherwise call
5961
- _refresh_usage_inproc(); always call run_sync_now_locked.
7385
+ 2. Non-blocking acquire. A machine nudge, or a held lock, enqueues
7386
+ on ``_SnapshotRef`` and answers 202 with the request identifier
7387
+ and this process's server epoch.
7388
+ 3. Lock free, human click: exactly the pre-#583 path — under
7389
+ --no-sync skip refresh; otherwise call ``_refresh_usage_inproc()``;
7390
+ always call run_sync_now_locked.
5962
7391
  4. Return 204 on clean success, 200 + JSON warnings on
5963
7392
  non-ok refresh status, 500 on unexpected exception.
5964
7393
 
@@ -5967,17 +7396,36 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
5967
7396
  run_sync_now_locked assumes the caller holds sync_lock - that's the
5968
7397
  whole point of the Task 0 lock split.
5969
7398
 
5970
- Bounded wait (vs. earlier non-blocking try_acquire): the periodic
5971
- background thread holds the lock for hundreds of ms each tick, and
5972
- a non-blocking acquire would 503 any click that landed inside that
5973
- window — silently dropping the user's force-refresh intent (the
5974
- periodic thread doesn't run refresh-usage). Waiting up to ~2s lets
5975
- the click span a normal periodic tick while still 503-ing on
5976
- truly stuck contention.
7399
+ #583 S2. The bounded acquire and its 503 are gone. A click landing
7400
+ inside the periodic thread's lock-hold used to wait up to ~2s and then
7401
+ 503 on stuck contention, which silently dropped the user's intent; it
7402
+ now queues, is serviced by the sync loop under the duty floor, and its
7403
+ settlement arrives on the published frame. This endpoint's status
7404
+ vocabulary is 403 / 202 / 200 / 204 / 500. The two other 503 sites
7405
+ (quota_projection_incomplete) are untouched.
7406
+
7407
+ A MACHINE NUDGE always queues, even when the lock is free. This is
7408
+ load-bearing: cmd_record_usage fires at Claude Code's status-line
7409
+ cadence, so a nudge taking the synchronous path would rebuild at that
7410
+ frequency and reopen the #313 peg, bypassing the loop's floor. The
7411
+ nudge is identified by an explicit ``queue=1``; an older refresh-usage
7412
+ binary posting without it takes the synchronous path, which is today's
7413
+ behaviour.
5977
7414
 
5978
7415
  --no-sync mode: refresh skipped (frozen mode preserves "no network
5979
7416
  calls"), rebuild still runs with skip_sync=True (the wired
5980
- staticmethod closes over args.no_sync, so the no-arg call DTRT).
7417
+ staticmethod closes over args.no_sync, so the no-arg call DTRT). A
7418
+ manual request there acquires sync_lock BLOCKING rather than
7419
+ non-blocking, because nothing would drain a queue in that mode —
7420
+ bounded by ``_DASHBOARD_NO_SYNC_LOCK_TIMEOUT_SECONDS``, past which it
7421
+ answers 200 with a ``sync_busy`` warning rather than pinning the
7422
+ handler thread on a wedged rebuild. A
7423
+ machine nudge is REFUSED there with 204 and enqueues nothing: the
7424
+ documented contract freezes data to the startup snapshot, and a queued
7425
+ nudge serviced by a skip_sync rebuild would read newly persisted rows
7426
+ and unfreeze it. A manual refresh=1 states the skip with a
7427
+ ``refresh_skipped_no_sync`` warning instead of performing no refresh
7428
+ silently.
5981
7429
 
5982
7430
  Refresh failures DO NOT cause 500 - they surface as warnings in the
5983
7431
  200 envelope and the rebuild still runs so the snapshot stays
@@ -5985,28 +7433,87 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
5985
7433
  """
5986
7434
  if not self._check_origin_csrf():
5987
7435
  return
5988
- sync_lock = type(self).sync_lock
5989
- if not sync_lock.acquire(
5990
- timeout=sys.modules["cctally"]._DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS):
5991
- self.send_error(503, "sync in progress")
7436
+ cls = type(self)
7437
+ sync_lock = cls.sync_lock
7438
+ query = urllib.parse.parse_qs(urllib.parse.urlsplit(self.path).query)
7439
+ do_refresh = query.get("refresh", ["1"])[0] != "0"
7440
+ is_machine_nudge = query.get("queue", ["0"])[0] == "1"
7441
+
7442
+ if is_machine_nudge and cls.no_sync:
7443
+ self.send_response(204)
7444
+ self.end_headers()
5992
7445
  return
5993
- try:
5994
- do_refresh = (
5995
- urllib.parse.parse_qs(urllib.parse.urlsplit(self.path).query)
5996
- .get("refresh", ["1"])[0] != "0"
7446
+
7447
+ # Under --no-sync a manual request WAITS for the lock rather than
7448
+ # queueing. Nothing drains a queue in that mode (`sync_thread` is None),
7449
+ # and the lock is not always free there: POST /api/settings calls
7450
+ # `run_sync_now()`, which takes it blocking precisely so a config change
7451
+ # propagates in a mode whose periodic thread never runs. A click landing
7452
+ # inside that hold used to answer 202 for a batch nobody could capture,
7453
+ # leaving `requested_id > started_id` true forever. The holder in that
7454
+ # mode is always another short synchronous rebuild, so the wait is
7455
+ # bounded. The machine-nudge refusal above is unaffected.
7456
+ if cls.no_sync and not is_machine_nudge:
7457
+ if not sync_lock.acquire(
7458
+ timeout=_DASHBOARD_NO_SYNC_LOCK_TIMEOUT_SECONDS):
7459
+ # A wedged rebuild, not ordinary contention: the holder in this
7460
+ # mode is one short synchronous rebuild. Say so instead of
7461
+ # pinning this thread for the life of the process.
7462
+ self._respond_json(200, {
7463
+ "status": "ok",
7464
+ "warnings": [{"code": "sync_busy"}],
7465
+ })
7466
+ return
7467
+ acquired = True
7468
+ else:
7469
+ acquired = not is_machine_nudge and sync_lock.acquire(blocking=False)
7470
+ if not acquired:
7471
+ # `cls.no_sync` is UNREACHABLE-false here today, so the guard never
7472
+ # subtracts anything: a manual request under --no-sync took the
7473
+ # bounded blocking acquire above and either holds the lock or has
7474
+ # already answered, and a machine nudge under --no-sync answered 204
7475
+ # before either branch. The guard is kept because it states the
7476
+ # invariant the queue depends on — nothing drains a queue under
7477
+ # --no-sync, so a queued batch there must never carry an OAuth
7478
+ # intent — and a future queueing path in that mode would silently
7479
+ # violate it if this were dropped as dead code.
7480
+ request_id = cls.snapshot_ref.request_sync(
7481
+ refresh=do_refresh and not cls.no_sync
5997
7482
  )
7483
+ cls.publish_activity() # acknowledge before responding
7484
+ self._respond_json(202, {
7485
+ "status": "queued",
7486
+ "request_id": request_id,
7487
+ "server_epoch": cls.snapshot_ref.server_epoch,
7488
+ })
7489
+ return
7490
+ try:
7491
+ # The locked section is a REBUILD, and this is the path an
7492
+ # uncontended manual refresh actually takes, so it reports itself
7493
+ # like every other rebuild does.
7494
+ cls.mark_rebuilding(True)
5998
7495
  warnings: list = []
5999
- if do_refresh and not type(self).no_sync:
6000
- result = _refresh_usage_inproc()
6001
- if result.status != "ok":
6002
- warnings.append({"code": result.status})
7496
+ if do_refresh:
7497
+ if cls.no_sync:
7498
+ warnings.append({"code": "refresh_skipped_no_sync"})
7499
+ else:
7500
+ result = _refresh_usage_inproc()
7501
+ if result.status != "ok":
7502
+ warnings.append({"code": result.status})
6003
7503
  try:
6004
- type(self).run_sync_now_locked()
7504
+ cls.run_sync_now_locked()
6005
7505
  except Exception as exc:
6006
7506
  self.log_error("/api/sync rebuild failed: %r", exc)
6007
7507
  self.send_error(500, "sync failed")
6008
7508
  return
6009
7509
  finally:
7510
+ # Drops THIS thread's claim only, so the ordering against the lock
7511
+ # release is not what makes it safe: a concurrent rebuilder's claim
7512
+ # is a different set member and this call cannot touch it, whichever
7513
+ # side of the release it runs on. On the success path the rebuild's
7514
+ # terminal publish has already dropped this thread's claim and this
7515
+ # adds no frame; the exception path is what needs it.
7516
+ cls.mark_rebuilding(False)
6010
7517
  sync_lock.release()
6011
7518
 
6012
7519
  if warnings:
@@ -6177,12 +7684,44 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
6177
7684
  conn.close()
6178
7685
  except Exception: # noqa: BLE001 -- a diagnostic must not expose raw errors.
6179
7686
  cache_state = {"status": "unavailable"}
7687
+ perf = self._perf_gate()
7688
+ tick_state = _lib_tick_stats.snapshot()
7689
+ requested, applied = perf.pending_state()
6180
7690
  body = {
6181
7691
  "schemaVersion": 1,
6182
7692
  "version": _debug_tool_version(),
6183
7693
  "generated_at": (last or {}).get("generated_at"),
6184
7694
  "dataset": dataset,
6185
7695
  "phases": (last or {}).get("phases"),
7696
+ # #583 S1 §3.1 asked for the stored tree's instant beside
7697
+ # `phases`, so a GET after `--trace off` cannot present an old tree
7698
+ # as current — disabling tracing does not clear the stored tree.
7699
+ # It was already there: the top-level `generated_at` above IS the
7700
+ # tree's instant, read from the same slot. A `phases_generated_at`
7701
+ # key was added and measured byte-identical to it on a live
7702
+ # endpoint, so it is not repeated here.
7703
+ "tick": {
7704
+ "dispatch_counts": dict(tick_state.dispatch_counts),
7705
+ "cache_open_failures": dict(tick_state.cache_open_failures),
7706
+ "tick_seq": tick_state.tick_seq,
7707
+ "records": [r.as_wire() for r in tick_state.records],
7708
+ "standalone": (
7709
+ tick_state.standalone.as_wire()
7710
+ if tick_state.standalone is not None else None
7711
+ ),
7712
+ # #583 S4: the SECOND work loop's ring, published under the
7713
+ # same object and behind the same loopback gate. An empty list
7714
+ # is a reachable steady state (`--no-sync` never starts the
7715
+ # thread), so the key is always present.
7716
+ "conversation_sync": [
7717
+ r.as_wire() for r in tick_state.conversation_records
7718
+ ],
7719
+ },
7720
+ "tracing": {
7721
+ "requested": requested,
7722
+ "applied": applied,
7723
+ "applies_at": perf.applies_at(),
7724
+ },
6186
7725
  "cache_state": cache_state,
6187
7726
  "sources": sources,
6188
7727
  # Additive, and named rather than folded into `cache_state`: a
@@ -6195,6 +7734,60 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
6195
7734
  body["note"] = "tracing_disabled"
6196
7735
  self._respond_json(200, body)
6197
7736
 
7737
+ def _handle_post_debug_backend_trace(self) -> None:
7738
+ """POST ``/api/debug/backend/trace`` — arm the deep phase trace (§3.2).
7739
+
7740
+ Body ``{"enabled": true|false}``; any other shape is 400.
7741
+
7742
+ Gated in three layers, in this order. ``_require_api_auth`` runs first,
7743
+ automatically for any ``/api/*`` path in ``do_POST``, and enforces the
7744
+ bearer whenever the dashboard minted a token. ``_require_debug_backend_
7745
+ allowed`` then applies the loopback TCP peer plus the IP-literal
7746
+ ``Host``. ``_check_origin_csrf`` applies Origin/Host parity last.
7747
+
7748
+ Retaining the CSRF layer matters even though the peer is already known
7749
+ to be loopback: the loopback and anti-rebinding checks do not stop a
7750
+ malicious page aiming a simple form POST straight at
7751
+ ``http://127.0.0.1:8789``. It is also why a command-line client must
7752
+ send an ``Origin`` matching the ``Host`` it calls — `_check_origin_csrf`
7753
+ rejects a request with none. That is not a weakening, because a
7754
+ non-browser client can set arbitrary headers regardless; the check
7755
+ exists to stop a page making the BROWSER issue the request.
7756
+
7757
+ The flip itself happens at the rebuild boundary in `_lib_perf.
7758
+ apply_pending`, not here, which is why the response reports `applied`
7759
+ as it is now and `applies_at` names when the request takes effect.
7760
+ """
7761
+ if not self._require_debug_backend_allowed():
7762
+ return
7763
+ if not self._check_origin_csrf():
7764
+ return
7765
+ try:
7766
+ length = int(self.headers.get("Content-Length", "0") or "0")
7767
+ except ValueError:
7768
+ length = 0
7769
+ if length <= 0 or length > 4096:
7770
+ self._respond_json(400, {"error": "body required (<=4 KB)"})
7771
+ return
7772
+ try:
7773
+ body = json.loads(self.rfile.read(length).decode("utf-8"))
7774
+ except (ValueError, UnicodeDecodeError):
7775
+ self._respond_json(400, {"error": "malformed JSON body"})
7776
+ return
7777
+ if (not isinstance(body, dict) or set(body) != {"enabled"}
7778
+ or not isinstance(body.get("enabled"), bool)):
7779
+ self._respond_json(
7780
+ 400, {"error": 'body must be {"enabled": true|false}'})
7781
+ return
7782
+ perf = self._perf_gate()
7783
+ perf.request_enabled(body["enabled"])
7784
+ requested, applied = perf.pending_state()
7785
+ self._respond_json(200, {
7786
+ "requested": requested,
7787
+ "applied": applied,
7788
+ "applies_at": perf.applies_at(),
7789
+ })
7790
+
6198
7791
  def _handle_post_settings(self) -> None:
6199
7792
  """Persist a settings update and trigger an immediate SSE broadcast.
6200
7793
 
@@ -6933,10 +8526,18 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
6933
8526
  # under --no-sync). run_sync_now is the same path POST /api/sync
6934
8527
  # uses; under skip_sync=True it still rebuilds + publishes via
6935
8528
  # hub.publish, which is what each SSE listener pulls.
8529
+ #
8530
+ # #583 S2 §6.3: this rebuild is the same multi-second locked rebuild
8531
+ # POST /api/sync runs, so it reports itself the same way. The mark is
8532
+ # outside the acquire because `run_sync_now` takes `sync_lock` itself;
8533
+ # a wait for a rebuild already in flight is honestly in-flight too.
6936
8534
  try:
8535
+ type(self).mark_rebuilding(True)
6937
8536
  type(self).run_sync_now()
6938
8537
  except Exception as exc:
6939
8538
  eprint(f"warning: settings broadcast failed: {exc!r}")
8539
+ finally:
8540
+ type(self).mark_rebuilding(False)
6940
8541
 
6941
8542
  self._respond_json(200, out)
6942
8543
 
@@ -7037,10 +8638,17 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7037
8638
  return
7038
8639
 
7039
8640
  if axis == "weekly":
8641
+ # Mirrors the CLI `alerts test --axis weekly` branch: the preview
8642
+ # carries the reset INSTANT a real crossing carries, so it renders
8643
+ # the instant form rather than the day-granularity fallback.
8644
+ preview_week_start = synthetic_preview_week_start()
7040
8645
  payload = _build_alert_payload_weekly(
7041
8646
  threshold=threshold,
7042
8647
  crossed_at_utc=now_utc_iso(),
7043
- week_start_date=dt.date.today().isoformat(),
8648
+ week_start_date=preview_week_start.date().isoformat(),
8649
+ week_start_at=preview_week_start.isoformat().replace(
8650
+ "+00:00", "Z"
8651
+ ),
7044
8652
  cumulative_cost_usd=1.23,
7045
8653
  dollars_per_percent=0.01,
7046
8654
  )
@@ -7206,8 +8814,41 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7206
8814
  self.wfile.write(body)
7207
8815
 
7208
8816
  def _serve_api_data(self) -> None:
8817
+ # #583 S3 §6/§7. TWO phases. Preparation may answer a JSON 500 because
8818
+ # nothing has been sent yet. Commit may NOT: once `send_response` has
8819
+ # run, a second `_respond_json` writes another HTTP response onto an
8820
+ # already-committed stream. The old handler wrapped both in one `try`
8821
+ # and did exactly that on any partial write.
8822
+ # #600 / §7: the LAST PUBLISHED state, not the reference. The dashboard
8823
+ # mutates the reference through four operations — `set`,
8824
+ # `capture_batch`, `settle`, `mark_rebuilding` — and publication is a
8825
+ # separate call in every case, so reading the reference can report a
8826
+ # `hydrating` flag no client was ever sent. There is no atomic boundary
8827
+ # to read instead, and creating one would mean editing `_SnapshotRef`
8828
+ # and the sync loop, so this endpoint is DEFINED as serving the most
8829
+ # recently published state. Deliberately no silent fallback to the
8830
+ # reference: that fallback is the disagreement this fixes.
8831
+ #
8832
+ # #583 S3 §6: `hub.latest()` gets its own guard, and the 503 is answered
8833
+ # OUTSIDE the preparation `try` below. Writing the 503 inside that `try`
8834
+ # meant a failure part-way through it was caught by the same `except`
8835
+ # that answers a JSON 500, appending a SECOND HTTP response onto a
8836
+ # stream this handler had already committed — the very defect the
8837
+ # prepare/commit split exists to remove, on the one path it did not
8838
+ # cover. The duplicated 500 arm below is the price of that separation.
7209
8839
  try:
7210
- snap = self.snapshot_ref.get()
8840
+ delivery = self.hub.latest()
8841
+ except Exception as exc: # noqa: BLE001
8842
+ self.log_error("api/data failed before commit: %r", exc)
8843
+ self._respond_json(500, {"error": "internal error"})
8844
+ return
8845
+ if delivery is None:
8846
+ self._respond_json(503, {"error": "no snapshot published yet"})
8847
+ return
8848
+
8849
+ try:
8850
+ # ---- preparation ---------------------------------------------
8851
+ snap = delivery.snapshot
7211
8852
  # Resolve oauth_usage cfg out here so snapshot_to_envelope stays
7212
8853
  # pure (no per-request FS read on the dashboard hot path).
7213
8854
  # Tolerate user config typos -- fall back to defaults rather than
@@ -7243,18 +8884,34 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7243
8884
  # finding) — one predicate, two consumers, desync impossible.
7244
8885
  env["transcriptsEnabled"] = visible
7245
8886
  body = encode_dashboard_json_bytes(env, ensure_ascii=False)
8887
+ gzip_on = _accepts_gzip(self.headers.get("Accept-Encoding"))
8888
+ if gzip_on:
8889
+ body = gzip.compress(body, 6)
8890
+ except Exception as exc: # noqa: BLE001
8891
+ # #279 S5 F6.1 (spec §8): a snapshot/envelope/dumps failure used to
8892
+ # escape to _QuietThreadingHTTPServer.handle_error (stdlib traceback +
8893
+ # dropped socket, no 500). Mirror _handle_get_doctor: log + JSON 500.
8894
+ self.log_error("api/data failed before commit: %r", exc)
8895
+ self._respond_json(500, {"error": "internal error"})
8896
+ return
8897
+
8898
+ # ---- commit ------------------------------------------------------
8899
+ try:
7246
8900
  self.send_response(200)
7247
8901
  self.send_header("Content-Type", "application/json; charset=utf-8")
8902
+ if gzip_on:
8903
+ self.send_header("Content-Encoding", "gzip")
8904
+ self.send_header("Vary", "Accept-Encoding")
7248
8905
  self.send_header("Content-Length", str(len(body)))
7249
8906
  self.send_header("Cache-Control", "no-cache")
7250
8907
  self.end_headers()
7251
8908
  self.wfile.write(body)
7252
8909
  except Exception as exc: # noqa: BLE001
7253
- # #279 S5 F6.1 (spec §8): a snapshot/envelope/dumps failure used to
7254
- # escape to _QuietThreadingHTTPServer.handle_error (stdlib traceback +
7255
- # dropped socket, no 500). Mirror _handle_get_doctor: log + JSON 500.
7256
- self.log_error("api/data failed: %r", exc)
7257
- self._respond_json(500, {"error": "internal error"})
8910
+ # Headers are committed, so no status code is available. Log and
8911
+ # close; NEVER `_respond_json` here — that appends a second HTTP
8912
+ # response onto a stream the client is already reading as one.
8913
+ self.log_error("api/data failed after commit: %r", exc)
8914
+ self.close_connection = True
7258
8915
 
7259
8916
  def _handle_get_doctor(self) -> None:
7260
8917
  """`GET /api/doctor` — full kernel-serialized doctor report (spec §5.6).
@@ -7271,6 +8928,9 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7271
8928
  dashboard's default bind; no CSRF gating here (mirrors
7272
8929
  `/api/data`, `/api/session/:id`, `/api/block/:start_at`).
7273
8930
  """
8931
+ # Preparation and commit are separate. Before headers, a failure can
8932
+ # still become a JSON 500. After headers, another response would
8933
+ # corrupt the stream, so the only valid recovery is log + close.
7274
8934
  try:
7275
8935
  _ld = sys.modules["cctally"]._load_sibling("_lib_doctor")
7276
8936
  state = doctor_gather_state(
@@ -7280,6 +8940,12 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7280
8940
  body = encode_dashboard_json_bytes(
7281
8941
  _ld.serialize_json(report), ensure_ascii=False,
7282
8942
  )
8943
+ except Exception as exc: # noqa: BLE001
8944
+ self.log_error("/api/doctor failed before commit: %r", exc)
8945
+ self._respond_json(500, {"error": f"{type(exc).__name__}: {exc}"})
8946
+ return
8947
+
8948
+ try:
7283
8949
  self.send_response(200)
7284
8950
  self.send_header("Content-Type", "application/json; charset=utf-8")
7285
8951
  self.send_header("Content-Length", str(len(body)))
@@ -7287,8 +8953,8 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7287
8953
  self.end_headers()
7288
8954
  self.wfile.write(body)
7289
8955
  except Exception as exc: # noqa: BLE001
7290
- self.log_error("/api/doctor failed: %r", exc)
7291
- self._respond_json(500, {"error": f"{type(exc).__name__}: {exc}"})
8956
+ self.log_error("/api/doctor failed after commit: %r", exc)
8957
+ self.close_connection = True
7292
8958
 
7293
8959
  def _handle_get_session_detail(self, path: str) -> None:
7294
8960
  """Return TuiSessionDetail JSON for the given session id (spec §3.2).
@@ -7621,43 +9287,46 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7621
9287
  None,
7622
9288
  )
7623
9289
  if target is None:
9290
+ status = 404
7624
9291
  body = encode_dashboard_json_bytes({"error": "block not found"})
7625
- self.send_response(404)
7626
- self.send_header("Content-Type", "application/json; charset=utf-8")
7627
- self.send_header("Content-Length", str(len(body)))
7628
- self.end_headers()
7629
- self.wfile.write(body)
7630
- return
7631
- block_entries = [
7632
- e for e in entries_in_window
7633
- if target.start_time <= e.timestamp < target.end_time
7634
- ]
7635
- # Resolve display tz once per request so the block detail's
7636
- # `label` matches the snapshot envelope's blocks panel.
7637
- # Shared resolver -- same warn-once semantics as
7638
- # `_compute_display_block` and `_tui_build_snapshot`. F3:
7639
- # honor the dashboard's `--tz` override (set as a class attr
7640
- # by cmd_dashboard) so the block-detail label speaks the
7641
- # same zone the rest of the envelope speaks.
7642
- _detail_tz = _resolve_display_tz_obj(
7643
- _apply_display_tz_override(
7644
- load_config(), type(self).display_tz_pref_override
9292
+ else:
9293
+ block_entries = [
9294
+ e for e in entries_in_window
9295
+ if target.start_time <= e.timestamp < target.end_time
9296
+ ]
9297
+ # Resolve display tz once per request so the block detail's
9298
+ # `label` matches the snapshot envelope's blocks panel.
9299
+ # Shared resolver -- same warn-once semantics as
9300
+ # `_compute_display_block` and `_tui_build_snapshot`. F3:
9301
+ # honor the dashboard's `--tz` override (set as a class attr
9302
+ # by cmd_dashboard) so the block-detail label speaks the
9303
+ # same zone the rest of the envelope speaks.
9304
+ _detail_tz = _resolve_display_tz_obj(
9305
+ _apply_display_tz_override(
9306
+ load_config(), type(self).display_tz_pref_override
9307
+ )
7645
9308
  )
7646
- )
7647
- detail = _build_block_detail(
7648
- target, block_entries, display_tz=_detail_tz,
7649
- )
9309
+ detail = _build_block_detail(
9310
+ target, block_entries, display_tz=_detail_tz,
9311
+ )
9312
+ status = 200
9313
+ body = encode_dashboard_json_bytes(detail, ensure_ascii=False)
7650
9314
  except Exception as exc:
7651
- self.log_error("/api/block failed: %r", exc)
9315
+ self.log_error("/api/block failed before commit: %r", exc)
7652
9316
  self.send_error(500, "block detail failed")
7653
9317
  return
7654
- body = encode_dashboard_json_bytes(detail, ensure_ascii=False)
7655
- self.send_response(200)
7656
- self.send_header("Content-Type", "application/json; charset=utf-8")
7657
- self.send_header("Content-Length", str(len(body)))
7658
- self.send_header("Cache-Control", "no-cache")
7659
- self.end_headers()
7660
- self.wfile.write(body)
9318
+
9319
+ try:
9320
+ self.send_response(status)
9321
+ self.send_header("Content-Type", "application/json; charset=utf-8")
9322
+ self.send_header("Content-Length", str(len(body)))
9323
+ if status == 200:
9324
+ self.send_header("Cache-Control", "no-cache")
9325
+ self.end_headers()
9326
+ self.wfile.write(body)
9327
+ except Exception as exc: # noqa: BLE001
9328
+ self.log_error("/api/block failed after commit: %r", exc)
9329
+ self.close_connection = True
7661
9330
 
7662
9331
  def _send_milestones_json(self, status: int, body: dict) -> None:
7663
9332
  payload = encode_dashboard_json_bytes(body, ensure_ascii=False)
@@ -7786,7 +9455,7 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7786
9455
  # reset and never resolves as current.
7787
9456
  identity = resolve_codex_cycle_detail_identity(
7788
9457
  cache_conn, source_root_keys=roots, now_utc=now_utc,
7789
- account_key=account_key,
9458
+ account_key=account_key, stats_conn=stats_conn,
7790
9459
  )
7791
9460
  result = c.build_codex_cycle_detail(
7792
9461
  stats_conn, cache_conn, identity=identity, key=key,
@@ -7827,14 +9496,46 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7827
9496
 
7828
9497
  def _serve_api_events(self) -> None:
7829
9498
  import queue as _queue
9499
+ gzip_on = _accepts_gzip(self.headers.get("Accept-Encoding"))
7830
9500
  self.send_response(200)
7831
9501
  self.send_header("Content-Type", "text/event-stream; charset=utf-8")
7832
9502
  self.send_header("Cache-Control", "no-cache")
7833
9503
  self.send_header("Connection", "keep-alive")
7834
9504
  # Nginx/proxies: disable buffering so events flow immediately.
9505
+ # #583 S3 §6: this STAYS under compression. It addresses an
9506
+ # intermediary proxy, not our own buffering, and is orthogonal.
7835
9507
  self.send_header("X-Accel-Buffering", "no")
9508
+ self.send_header("Vary", "Accept-Encoding")
9509
+ if gzip_on:
9510
+ self.send_header("Content-Encoding", "gzip")
7836
9511
  self.end_headers()
7837
9512
 
9513
+ # #583 S3 §6. ONE stateful compressor per connection. EVERY byte after
9514
+ # the headers goes through it — updates AND the keep-alive comment. A
9515
+ # raw write of even two bytes corrupts the whole remainder of the
9516
+ # stream, and the failure is silent until the next frame.
9517
+ #
9518
+ # One compressor per CONNECTION rather than one per tick, deliberately:
9519
+ # sharing compressed bytes across clients requires each frame to be an
9520
+ # independent gzip member, and cross-browser support for incrementally
9521
+ # decoding a concatenated multi-member stream under `Content-Encoding`
9522
+ # is not established. The saving would be proportional to connected
9523
+ # clients minus one — exactly zero at one open tab. Filed as a residual.
9524
+ _comp = (zlib.compressobj(6, zlib.DEFLATED, 16 + zlib.MAX_WBITS)
9525
+ if gzip_on else None)
9526
+
9527
+ def _emit(raw: bytes) -> None:
9528
+ if _comp is None:
9529
+ self.wfile.write(raw)
9530
+ else:
9531
+ # Z_SYNC_FLUSH, not a bare compress(): zlib buffers a small
9532
+ # frame entirely, so without the flush the client receives
9533
+ # nothing at all until some later write happens to spill it.
9534
+ chunk = _comp.compress(raw) + _comp.flush(zlib.Z_SYNC_FLUSH)
9535
+ if chunk:
9536
+ self.wfile.write(chunk)
9537
+ self.wfile.flush()
9538
+
7838
9539
  # Resolve oauth_usage cfg once per SSE connection so the per-tick
7839
9540
  # envelope build stays free of FS reads. A config edit during the
7840
9541
  # connection's lifetime won't take effect until the client
@@ -7854,37 +9555,74 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7854
9555
  # ~15s after bootstrap). Mirrors `/api/data`'s per-request injection.
7855
9556
  transcripts_enabled = self._transcripts_visible_to_request()
7856
9557
 
9558
+ # #583 S3 §5. Normalize the privacy input BEFORE it becomes a cache
9559
+ # key: an unresolvable value is the RESTRICTIVE one. An ordinary cache
9560
+ # MISS for the `true` variant still computes `true` — that is a
9561
+ # different situation, and conflating the two either leaks transcript
9562
+ # content or breaks the gate.
9563
+ visible = bool(transcripts_enabled)
9564
+ # Only `visible` and the oauth config can differ between two
9565
+ # connections in this process today. The two process constants stay in
9566
+ # the key so a later change making either per-connection cannot
9567
+ # silently serve one client another client's payload.
9568
+ variant = (
9569
+ visible,
9570
+ _canonical_oauth_key(cfg_oauth),
9571
+ type(self).display_tz_pref_override,
9572
+ type(self).cctally_host,
9573
+ )
9574
+
7857
9575
  q = self.hub.subscribe()
7858
9576
  try:
7859
9577
  while True:
7860
9578
  try:
7861
- snap = q.get(timeout=15)
9579
+ delivery = q.get(timeout=_SSE_KEEPALIVE_SECONDS)
7862
9580
  except _queue.Empty:
7863
9581
  # Keep-alive. Comment lines are ignored by EventSource
7864
- # but stop idle-proxy timeouts.
7865
- self.wfile.write(b": keep-alive\n\n")
7866
- self.wfile.flush()
9582
+ # but stop idle-proxy timeouts. #583 S3 §6: through
9583
+ # `_emit`, never a raw `wfile.write` — under compression a
9584
+ # raw write here corrupts every byte after it.
9585
+ _emit(b": keep-alive\n\n")
7867
9586
  continue
7868
- env = snapshot_to_envelope(
7869
- snap,
7870
- now_utc=dt.datetime.now(dt.timezone.utc),
7871
- monotonic_now=time.monotonic(),
7872
- oauth_usage_cfg=cfg_oauth,
7873
- display_tz_pref_override=type(self).display_tz_pref_override,
7874
- runtime_bind=type(self).cctally_host,
7875
- # #264 S3: gate the in-envelope session `title` on the same
7876
- # connection-scoped predicate that drives transcriptsEnabled.
7877
- transcripts_visible=transcripts_enabled,
7878
- )
7879
- env["transcriptsEnabled"] = transcripts_enabled
7880
- msg = (
7881
- "event: update\n"
7882
- + "data: "
7883
- + encode_dashboard_json(env, ensure_ascii=False)
7884
- + "\n\n"
7885
- )
7886
- self.wfile.write(msg.encode("utf-8"))
7887
- self.wfile.flush()
9587
+ # #583 S3 §5: skip to the newest queued delivery. The queue
9588
+ # holds four and `publish` discards only one oldest, so a
9589
+ # lagging client would otherwise replay a backlog whose clocks
9590
+ # were pinned several publish periods ago. The blocking `get`
9591
+ # above keeps its own `queue.Empty` keep-alive path.
9592
+ delivery = _drain_to_newest(q, delivery)
9593
+
9594
+ def _project(_key, _d=delivery, _v=visible):
9595
+ env = snapshot_to_envelope(
9596
+ _d.snapshot,
9597
+ now_utc=_d.pinned_now_utc,
9598
+ monotonic_now=_d.pinned_monotonic,
9599
+ oauth_usage_cfg=cfg_oauth,
9600
+ display_tz_pref_override=type(self).display_tz_pref_override,
9601
+ runtime_bind=type(self).cctally_host,
9602
+ # #264 S3: gate the in-envelope session `title` on the
9603
+ # same connection-scoped predicate that drives
9604
+ # transcriptsEnabled.
9605
+ transcripts_visible=_v,
9606
+ )
9607
+ # Part of the CACHED variant, not a post-projection
9608
+ # mutation: the payload is shared across every client
9609
+ # holding this delivery, so mutating it here would apply
9610
+ # to all of them.
9611
+ env["transcriptsEnabled"] = _v
9612
+ payload = encode_dashboard_json_bytes(
9613
+ env, ensure_ascii=False,
9614
+ )
9615
+ return b"event: update\ndata: " + payload + b"\n\n"
9616
+
9617
+ if _delivery_is_shareable(delivery.snapshot):
9618
+ frame = delivery.encoded(variant, _project)
9619
+ else:
9620
+ # #583 S3 §5: this snapshot's projection is NOT a function
9621
+ # of the snapshot plus the key — `snapshot_to_envelope`
9622
+ # reads configuration inline and runs the real doctor
9623
+ # gather per call — so it must not be cached and shared.
9624
+ frame = _project(variant)
9625
+ _emit(frame)
7888
9626
  except (BrokenPipeError, ConnectionResetError,
7889
9627
  ConnectionAbortedError, socket.timeout):
7890
9628
  # #279 S1 F3: a stalled send past the handler timeout raises
@@ -7898,7 +9636,17 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7898
9636
  # traceback. Headers are already committed — no 500 is possible; the
7899
9637
  # win is routing the operator signal through the _lib_log chokepoint
7900
9638
  # (self.log_error) + a deliberate clean close via the finally below.
9639
+ #
9640
+ # #583 S3 §6: under compression this path must add NOTHING to the
9641
+ # stream. Do not append plaintext, another gzip member, or a
9642
+ # trailer to a truncated stream; do not call `Z_FINISH` here,
9643
+ # because the stateful compressor may have advanced even though its
9644
+ # output was not fully written; and do not skip the frame and carry
9645
+ # on, for the same reason. The browser's own EventSource reconnect
9646
+ # opens a fresh response with a fresh compressor, which is the
9647
+ # correct recovery.
7901
9648
  self.log_error("api/events stream failed: %r", exc)
9649
+ self.close_connection = True
7902
9650
  finally:
7903
9651
  self.hub.unsubscribe(q)
7904
9652
 
@@ -8269,7 +10017,10 @@ def _dashboard_stats_deferred_snapshot(args, *, pinned_now, exc):
8269
10017
  "fingerprint": "sha1:" + ("0" * 40),
8270
10018
  }
8271
10019
  replacements = {
8272
- "last_sync_at": _time.monotonic(),
10020
+ # #583 S2 §6.1: this frame is degraded by construction — it carries a
10021
+ # `stats-open` error and no successful build ran, so it must not stamp
10022
+ # a fresh success. There is no earlier success to preserve either.
10023
+ "last_sync_at": None,
8273
10024
  "last_sync_error": "; ".join(errors),
8274
10025
  "sync_failures": (
8275
10026
  tui.SyncFailureAttribution(
@@ -8412,7 +10163,10 @@ def _dashboard_initial_snapshot_once(
8412
10163
  current_week=cw,
8413
10164
  forecast=fc,
8414
10165
  forecast_view=fc_view,
8415
- last_sync_at=_time.monotonic(),
10166
+ # #583 S2 §6.1: a build that failed does not stamp a success. This is
10167
+ # the FIRST build, so a failure retains None rather than becoming
10168
+ # freshly successful — there is no earlier success to preserve.
10169
+ last_sync_at=(None if errors else _time.monotonic()),
8416
10170
  last_sync_error=("; ".join(errors) if errors else None),
8417
10171
  sync_failures=tuple(sync_failures),
8418
10172
  doctor_payload=doctor_payload,
@@ -8640,16 +10394,23 @@ def cmd_dashboard(args: argparse.Namespace) -> int:
8640
10394
 
8641
10395
  ref = _SnapshotRef(initial)
8642
10396
  hub = SSEHub()
8643
- hub.publish(initial) # seed for early subscribers
10397
+ # #583 S2: seed for early subscribers, published from the reference so the
10398
+ # very first frame already carries the process's `server_epoch`. A seed
10399
+ # published without it would make the client discard its (empty) outstanding
10400
+ # set on the next frame's epoch change — harmless, but it would also mean
10401
+ # the first frame a client ever sees disagrees with every later one.
10402
+ hub.publish(ref.get())
8644
10403
 
8645
10404
  # sync_lock serializes sync-work between the periodic sync thread and
8646
10405
  # the POST /api/sync handler. Held only around _tui_build_snapshot +
8647
10406
  # ref.set + hub.publish — NOT around the handler's response path.
8648
- # The handler uses acquire(timeout=…) so a click that lands inside
8649
- # the periodic thread's lock-hold waits briefly rather than 503-ing
8650
- # and silently dropping the user's force-refresh intent; only stuck
8651
- # contention beyond the timeout produces 503. The lock inside
8652
- # _run_sync_now is what actually prevents overlap.
10407
+ # The handler acquires it NON-BLOCKING (#583 S2): a click that lands
10408
+ # inside the periodic thread's lock-hold is queued on _SnapshotRef and
10409
+ # answered 202, and the sync loop services it under the duty floor. The
10410
+ # bounded acquire and its 503 are gone. The one exception is --no-sync,
10411
+ # where nothing would drain that queue, so a manual request there waits on
10412
+ # a blocking acquire instead. The lock inside _run_sync_now is what
10413
+ # actually prevents overlap.
8653
10414
  sync_lock = threading.Lock()
8654
10415
 
8655
10416
  # Build the two variants up front. The locked variant is exposed on the
@@ -8706,32 +10467,27 @@ def cmd_dashboard(args: argparse.Namespace) -> int:
8706
10467
 
8707
10468
  class _DashboardSyncThread(_c_for_subclass._TuiSyncThread):
8708
10469
  def _run(self) -> None:
8709
- last_heal = [_time.monotonic()]
8710
-
8711
- def run_iteration() -> None:
8712
- _run_sync_now(skip_sync=self._skip_sync)
8713
- # Self-heal removed-worktree orphans on a ~60s cadence (far
8714
- # rarer than the sync tick — a deleted worktree is not urgent).
8715
- # Non-blocking on the flock, so a contended tick just retries
8716
- # next cadence; gated off under --no-sync. Runs INSIDE the
8717
- # measured iteration so its cost counts toward the cooldown
8718
- # deadline (#313 P2 / F10).
8719
- if (not self._skip_sync
8720
- and _time.monotonic() - last_heal[0] >= 60.0):
8721
- last_heal[0] = _time.monotonic()
8722
- _dashboard_self_heal_orphans(skip_sync=self._skip_sync)
10470
+ run_iteration = _make_dashboard_run_iteration(
10471
+ sync_lock=sync_lock,
10472
+ run_sync_now=_run_sync_now,
10473
+ run_sync_now_locked=_run_sync_now_locked,
10474
+ skip_sync=self._skip_sync,
10475
+ monotonic=_time.monotonic,
10476
+ )
8723
10477
 
8724
10478
  # Work-proportional cooldown (F10): sleep to t0 + max(interval, work)
8725
- # so a slow rebuild cannot peg a full core. The manual POST /api/sync
8726
- # refresh runs synchronously under sync_lock, independent of this
8727
- # cooldown, so a user force-refresh is always immediate.
10479
+ # so a slow rebuild cannot peg a full core. #583 S2 adds the
10480
+ # request-driven start, floored at t0 + 2*work so a queued refresh
10481
+ # cannot drive the duty above the same #313 bound. The dashboard no
10482
+ # longer passes the legacy test-and-clear flag: under a floor a poll
10483
+ # firing early would clear the request and discard it.
8728
10484
  _dashboard_sync_loop(
8729
10485
  stop=self._stop,
8730
10486
  interval=self._interval,
8731
10487
  run_iteration=run_iteration,
8732
- take_sync_request=self._ref.take_sync_request,
8733
10488
  monotonic=_time.monotonic,
8734
10489
  sleep=_time.sleep,
10490
+ **_make_sync_loop_collaborators(ref=ref, hub=hub),
8735
10491
  )
8736
10492
 
8737
10493
  sync_thread = (
@@ -8743,43 +10499,14 @@ def cmd_dashboard(args: argparse.Namespace) -> int:
8743
10499
  if sync_thread is not None:
8744
10500
  sync_thread.start()
8745
10501
 
8746
- # #320: transcript/search ingestion runs on its own thread and SQLite file.
8747
- # A multi-GB first rebuild or a contended conversations.db therefore cannot
8748
- # delay `_run_sync_now`, its `last_sync_at` stamp, or core SSE publication.
10502
+ # The loop, the pass and this thread's construction are module-level (see
10503
+ # `_conversation_sync_loop`): the extraction is what lets the duty bound be
10504
+ # proven on a virtual clock instead of asserted.
8749
10505
  conversation_sync_stop = threading.Event()
8750
-
8751
- def _conversation_sync_loop() -> None:
8752
- interval = max(5.0, float(args.sync_interval))
8753
- while not conversation_sync_stop.is_set():
8754
- conn = None
8755
- try:
8756
- conn = open_conversations_db()
8757
- sync_claude_conversations(conn)
8758
- sync_codex_conversations(conn)
8759
- _dashboard_maybe_prune_retention()
8760
- except (OSError, sqlite3.DatabaseError) as exc:
8761
- eprint(f"[conversations] background sync unavailable: {exc}")
8762
- except Exception as exc: # noqa: BLE001
8763
- # Transcript parsing/normalization is deliberately outside the
8764
- # core freshness loop. Keep this worker alive so a later clean
8765
- # tick can self-heal instead of permanently stopping after one
8766
- # malformed provider record.
8767
- eprint(
8768
- "[conversations] background sync failed: "
8769
- f"{type(exc).__name__}: {exc}"
8770
- )
8771
- finally:
8772
- if conn is not None:
8773
- conn.close()
8774
- conversation_sync_stop.wait(interval)
8775
-
8776
- conversation_sync_thread = (
8777
- None if args.no_sync
8778
- else threading.Thread(
8779
- target=_conversation_sync_loop,
8780
- daemon=True,
8781
- name="dashboard-conversations-sync",
8782
- )
10506
+ conversation_sync_thread = _make_conversation_sync_thread(
10507
+ stop=conversation_sync_stop,
10508
+ sync_interval=args.sync_interval,
10509
+ no_sync=args.no_sync,
8783
10510
  )
8784
10511
  if conversation_sync_thread is not None:
8785
10512
  conversation_sync_thread.start()
@@ -8790,7 +10517,7 @@ def cmd_dashboard(args: argparse.Namespace) -> int:
8790
10517
  # relying on the data-sync thread's lifecycle.
8791
10518
  update_check_stop = threading.Event()
8792
10519
  update_check_thread = _DashboardUpdateCheckThread(
8793
- update_check_stop, hub=hub, snapshot_ref=ref,
10520
+ update_check_stop, hub=hub, snapshot_ref=ref, runtime_bind=args.host,
8794
10521
  )
8795
10522
  update_check_thread.start()
8796
10523