cctally 1.101.0 → 1.102.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1332,6 +1332,7 @@ def _build_claude_source_detail(
1332
1332
  recorded_windows, block_start_overrides, canonical_intervals = (
1333
1333
  _load_recorded_five_hour_windows(start_at - BLOCK_DURATION, end_at + BLOCK_DURATION)
1334
1334
  )
1335
+ entry_membership: dict[int, list[UsageEntry]] = {}
1335
1336
  blocks = _group_entries_into_blocks(
1336
1337
  entries,
1337
1338
  mode="auto",
@@ -1339,6 +1340,7 @@ def _build_claude_source_detail(
1339
1340
  block_start_overrides=block_start_overrides,
1340
1341
  canonical_intervals=canonical_intervals,
1341
1342
  now=now_utc,
1343
+ _entry_membership=entry_membership,
1342
1344
  )
1343
1345
  target = next(
1344
1346
  (block for block in blocks if not block.is_gap and block.start_time == start_at),
@@ -1346,9 +1348,7 @@ def _build_claude_source_detail(
1346
1348
  )
1347
1349
  if target is None:
1348
1350
  raise SourceResourceNotFound()
1349
- block_entries = [
1350
- entry for entry in entries if target.start_time <= entry.timestamp < target.end_time
1351
- ]
1351
+ block_entries = entry_membership[id(target)]
1352
1352
  return _source_safe_claude_block_detail(
1353
1353
  _build_block_detail(target, block_entries), key=key,
1354
1354
  )
@@ -1664,6 +1664,28 @@ def _make_sync_loop_collaborators(*, ref, hub) -> dict:
1664
1664
  }
1665
1665
 
1666
1666
 
1667
+ @contextlib.contextmanager
1668
+ def _rebuilding_claim(mark_rebuilding, *, prepare=None):
1669
+ """Bracket one owner-scoped rebuilding claim, including preparation.
1670
+
1671
+ ``prepare`` may itself publish/claim (``_SnapshotRef.capture_batch`` does),
1672
+ so it belongs inside the same structural cleanup boundary as the explicit
1673
+ mark. This is the single guard used by the periodic loop and both
1674
+ synchronous HTTP rebuild routes; a future exception between claim and work
1675
+ therefore cannot strand ``rebuilding=true`` for the process lifetime.
1676
+ """
1677
+ prepared = None
1678
+ try:
1679
+ if prepare is not None:
1680
+ prepared = prepare()
1681
+ if mark_rebuilding is not None:
1682
+ mark_rebuilding(True)
1683
+ yield prepared
1684
+ finally:
1685
+ if mark_rebuilding is not None:
1686
+ mark_rebuilding(False)
1687
+
1688
+
1667
1689
  def _dashboard_sync_loop(
1668
1690
  *,
1669
1691
  stop,
@@ -1707,43 +1729,32 @@ def _dashboard_sync_loop(
1707
1729
  settlement counters, which must not advance for a batch that never existed.
1708
1730
  """
1709
1731
  while not stop.is_set():
1710
- batch = None
1732
+ prepare = None
1711
1733
  if (pending_request is not None and capture_batch is not None
1712
1734
  and pending_request()):
1713
- batch = capture_batch()
1714
- if mark_rebuilding is not None:
1715
- # BEFORE t0: the publish is a non-blocking queue put, and keeping
1716
- # it outside the measured span leaves the #313 bound's algebra
1717
- # exactly as it is. An automatic tick reaches this with no batch
1718
- # captured, which is the whole point — `rebuilding` describes the
1719
- # iteration, not the request that may or may not have started it.
1720
- mark_rebuilding(True)
1721
- t0 = monotonic()
1722
- status, warnings = "ok", ()
1723
- try:
1724
- result = (run_iteration(batch=batch) if batch is not None
1725
- else run_iteration())
1726
- if isinstance(result, dict):
1727
- warnings = tuple(result.get("warnings") or ())
1728
- except Exception: # noqa: BLE001 — see below
1729
- # An escaped exception must not kill the only drainer: an accepted
1730
- # 202 would then never reach a terminal state, and the client would
1731
- # hold `queued…` forever waiting for a settlement no surviving
1732
- # thread can publish.
1733
- status = "failed"
1734
- _log_sync_iteration_failure()
1735
- finally:
1736
- # Failure time is charged to the cooldown exactly like success
1737
- # time, so a crash loop cannot busy-spin.
1738
- work = monotonic() - t0
1739
- if batch is not None and settle is not None:
1740
- settle(batch[0], status, warnings)
1741
- if mark_rebuilding is not None:
1742
- # Every exit path, including an escaped exception: a flag left
1743
- # set would pin the client's chip at `syncing…` for the life of
1744
- # the process. `settle` has already cleared it on a requested
1745
- # tick, so this publishes nothing extra there.
1746
- mark_rebuilding(False)
1735
+ prepare = capture_batch
1736
+ # Claim publication stays BEFORE t0, preserving #313's measured-work
1737
+ # algebra, but the shared guard begins before capture_batch because that
1738
+ # preparation also claims this thread's owner id.
1739
+ with _rebuilding_claim(mark_rebuilding, prepare=prepare) as batch:
1740
+ t0 = monotonic()
1741
+ status, warnings = "ok", ()
1742
+ try:
1743
+ result = (run_iteration(batch=batch) if batch is not None
1744
+ else run_iteration())
1745
+ if isinstance(result, dict):
1746
+ warnings = tuple(result.get("warnings") or ())
1747
+ except Exception: # noqa: BLE001 — see below
1748
+ # An escaped exception must not kill the only drainer: an
1749
+ # accepted 202 would then never reach a terminal state.
1750
+ status = "failed"
1751
+ _log_sync_iteration_failure()
1752
+ finally:
1753
+ # Failure time is charged to the cooldown exactly like success
1754
+ # time, so a crash loop cannot busy-spin.
1755
+ work = monotonic() - t0
1756
+ if batch is not None and settle is not None:
1757
+ settle(batch[0], status, warnings)
1747
1758
 
1748
1759
  deadline = _next_deadline(t0, interval, work)
1749
1760
  floor = t0 + 2.0 * work
@@ -2391,42 +2402,10 @@ class _SnapshotRef:
2391
2402
  # therefore remove only the clearing thread's own claim; `rebuilding`
2392
2403
  # is then true exactly while at least one rebuilder holds one.
2393
2404
  #
2394
- # Owner scoping made a LEAKED CLAIM strictly worse than the boolean it
2395
- # replaced, so do not record the opposite. Under the boolean a leaked
2396
- # `True` was cleared by whichever rebuilder next reached `set_final`, so
2397
- # it self-healed on the following rebuild. Every clear site here discards
2398
- # `threading.get_ident()`, so a claim left behind by a thread that has
2399
- # exited can be discarded by NO other thread, and `rebuilding` would stay
2400
- # true — pinning every client's chip at `syncing…` for the life of the
2401
- # process.
2402
- #
2403
- # No leak is reachable today, but the argument splits by CALLER, not by
2404
- # add site. `mark_rebuilding` below is reached from both routes — the
2405
- # sync loop through `_make_sync_loop_collaborators` and an HTTP handler
2406
- # thread through `DashboardHTTPHandler.mark_rebuilding` — so reading one
2407
- # add site answers for neither. Four callers add a claim.
2408
- #
2409
- # Two of the four are bracketed: `_handle_post_sync` and
2410
- # `_handle_post_settings` each mark inside a `try` whose `finally`
2411
- # clears, so nothing between the two can leak the claim.
2412
- #
2413
- # The other two are the sync loop's `capture_batch()` and its
2414
- # `mark_rebuilding(True)`, and they are NOT bracketed. Both run before
2415
- # `t0` and therefore before the `try:` whose `finally` clears them; the
2416
- # loop's own comment at `mark_rebuilding(True)` gives the #313
2417
- # duty-algebra reason for that one's placement. `_dashboard_sync_loop`'s
2418
- # `while` has no outer handler, so a raise in that gap would kill the
2419
- # drainer and leak the claim together. The gap is safe because nothing
2420
- # in it raises: the `_restamp_locked()` each add performs is a
2421
- # `dataclasses.replace` over `DataSnapshot`, a plain dataclass with no
2422
- # `__post_init__`, no `init=False` field and no `InitVar`;
2423
- # `SSEHub.publish` holds its own lock and swallows
2424
- # `queue.Full`/`queue.Empty`; `ref.get()` is a lock-and-return; and what
2425
- # remains is a clock read and two local assignments.
2426
- #
2427
- # A new `add` must therefore satisfy one of the two: a same-thread
2428
- # `finally` that clears it, or a proven non-raising path to one. There
2429
- # is no self-healing path behind either.
2405
+ # Owner scoping makes a leaked claim permanent: another thread cannot
2406
+ # discard this thread's id. All three rebuilding routes therefore use
2407
+ # `_rebuilding_claim`, whose boundary begins before any preparation
2408
+ # that can claim and ends after the terminal publish.
2430
2409
  self._rebuilding_owners: set[int] = set()
2431
2410
  self._snap = self._stamped_locked(initial)
2432
2411
 
@@ -4765,6 +4744,10 @@ class _ProjWeekBucket(NamedTuple):
4765
4744
  first_order: str
4766
4745
  first_id: int
4767
4746
  first_key: "Any"
4747
+ # Retained so selected-window totals can deduplicate a session that is
4748
+ # active in more than one subscription bucket (#634). The default keeps
4749
+ # older fixture constructors source-compatible.
4750
+ session_ids: "frozenset[str]" = frozenset()
4768
4751
 
4769
4752
 
4770
4753
  def _fold_projects_entry(
@@ -5514,6 +5497,7 @@ def _finalize_projects_mut(mut: dict) -> "dict[str, _ProjWeekBucket]":
5514
5497
  first_order=a["first_order"],
5515
5498
  first_id=a["first_id"],
5516
5499
  first_key=a["first_key"],
5500
+ session_ids=frozenset(a["sessions"]),
5517
5501
  )
5518
5502
  for bp, a in mut.items()
5519
5503
  }
@@ -5596,9 +5580,16 @@ def _assemble_projects_via_cache(
5596
5580
  for bp, wb in week_buckets.items():
5597
5581
  buckets[(bp, w)] = {
5598
5582
  "cost_usd": wb.cost_usd,
5599
- # `sessions` is only ever `len()`-d downstream; a `range` of the
5600
- # cached count reproduces that without storing the id set.
5601
- "sessions": range(wb.sessions_count),
5583
+ # Window totals union these identities so a resumed session
5584
+ # crossing a reset is counted once. The range fallback serves
5585
+ # test/legacy constructors that predate ``session_ids``.
5586
+ "sessions": (
5587
+ wb.session_ids
5588
+ if wb.session_ids
5589
+ else frozenset(
5590
+ (w, index) for index in range(wb.sessions_count)
5591
+ )
5592
+ ),
5602
5593
  "first_seen": wb.first_seen,
5603
5594
  "last_seen": wb.last_seen,
5604
5595
  }
@@ -5932,15 +5923,38 @@ def _build_projects_envelope(
5932
5923
  # (bin/cctally:1162-1168) and the doctor credited-week check
5933
5924
  # (bin/cctally:8706-8714).
5934
5925
  #
5935
- # Portable per-key-latest pattern: read rows ordered by capture-
5936
- # ascending and let later rows overwrite. The final value per key
5937
- # is the most-recent snapshot.
5926
+ # Reduce to one latest NON-NULL row per candidate date in SQLite, then let
5927
+ # Python resolve those bounded rows onto the anchored interval grid. NULL
5928
+ # rows never erased the prior known value in the former ascending fold, so
5929
+ # they are excluded before ranking. The outer capture order preserves the
5930
+ # same final-overwrite behaviour if two legacy date keys resolve to one
5931
+ # interval. The date bounds keep historical status-line ticks out of the
5932
+ # scan, while the per-date ranking makes the rows crossing into Python
5933
+ # proportional to candidate boundary dates rather than tick count. Do not
5934
+ # LIMIT before Python resolves legacy date keys: an unresolvable key must
5935
+ # not evict a valid rendered-week row (#620 S1 A3).
5938
5936
  weekly_pct_by_week: dict[dt.datetime, float] = {}
5939
5937
  try:
5940
5938
  cur = conn.execute(
5941
- "SELECT week_start_date, week_start_at, weekly_percent "
5942
- "FROM weekly_usage_snapshots "
5943
- "ORDER BY captured_at_utc ASC, id ASC"
5939
+ "WITH ranked AS ("
5940
+ " SELECT week_start_date, week_start_at, weekly_percent,"
5941
+ " captured_at_utc, id,"
5942
+ " ROW_NUMBER() OVER ("
5943
+ " PARTITION BY week_start_date"
5944
+ " ORDER BY captured_at_utc DESC, id DESC"
5945
+ " ) AS latest_rank"
5946
+ " FROM weekly_usage_snapshots"
5947
+ " WHERE week_start_date >= ? AND week_start_date < ?"
5948
+ " AND date(week_start_date) IS NOT NULL"
5949
+ " AND weekly_percent IS NOT NULL"
5950
+ ")"
5951
+ " SELECT week_start_date, week_start_at, weekly_percent"
5952
+ " FROM ranked WHERE latest_rank = 1"
5953
+ " ORDER BY captured_at_utc ASC, id ASC",
5954
+ (
5955
+ since_dt.date().isoformat(),
5956
+ cw_end.date().isoformat(),
5957
+ ),
5944
5958
  )
5945
5959
  rows = cur.fetchall()
5946
5960
  except sqlite3.OperationalError:
@@ -6087,6 +6101,19 @@ def _build_projects_envelope(
6087
6101
  sessions_per_week.append(len(b["sessions"]))
6088
6102
  first_seen_per_week.append(_iso_z(b["first_seen"]))
6089
6103
  last_seen_per_week.append(_iso_z(b["last_seen"]))
6104
+ session_counts_by_window: dict[str, int] = {}
6105
+ for window_weeks in PROJECT_WINDOW_WEEKS_CHOICES:
6106
+ selected_weeks = trend_weeks[-window_weeks:]
6107
+ # Cached real buckets contribute string identities; the
6108
+ # source-compatible fallback contributes opaque tuple identities.
6109
+ window_sessions: set[object] = set()
6110
+ for selected_week in selected_weeks:
6111
+ selected_bucket = buckets.get((bp, selected_week))
6112
+ if selected_bucket is not None:
6113
+ window_sessions.update(selected_bucket["sessions"])
6114
+ session_counts_by_window[str(window_weeks)] = len(
6115
+ window_sessions,
6116
+ )
6090
6117
  # Skip projects with zero total cost across the entire window
6091
6118
  # (the bucket-loop only enters projects that have at least one
6092
6119
  # entry, so this is mainly a safety check).
@@ -6098,6 +6125,7 @@ def _build_projects_envelope(
6098
6125
  "weekly_cost": weekly_cost,
6099
6126
  "weekly_pct": weekly_pct_arr,
6100
6127
  "sessions_per_week": sessions_per_week,
6128
+ "session_counts_by_window": session_counts_by_window,
6101
6129
  "first_seen_per_week": first_seen_per_week,
6102
6130
  "last_seen_per_week": last_seen_per_week,
6103
6131
  })
@@ -6211,8 +6239,11 @@ def _project_detail_for_window(
6211
6239
  "projects.current_week.week_start_at",
6212
6240
  ).astimezone(dt.timezone.utc)
6213
6241
 
6214
- # The drill resolves the SAME interval its panel resolved, by rebuilding
6215
- # the panel's grid rather than stepping back in seven-day multiples.
6242
+ # The drill resolves the SAME interval union its panel resolved, by
6243
+ # rebuilding the panel's grid rather than stepping back in seven-day
6244
+ # multiples. Subscription reset shifts can leave gaps between buckets;
6245
+ # the outer start/end pair is only a candidate-query bound, while
6246
+ # ``window_intervals`` is the authoritative membership contract.
6216
6247
  # `_ProjectsWeekGrid` exists because a drifted reset day produces a
6217
6248
  # genuinely short week: on `non-monday-anchor` at `weeks_back=4` the grid
6218
6249
  # starts the window at 2026-03-27T09:00Z while a seven-day walk yields
@@ -6251,6 +6282,18 @@ def _project_detail_for_window(
6251
6282
  # falls back to, where the seven-day assumption is the right one.
6252
6283
  since_dt = cw_start - dt.timedelta(days=7 * (weeks_back - 1))
6253
6284
  until_dt = cw_start + dt.timedelta(days=7)
6285
+ detail_bounds = [(since_dt, until_dt)]
6286
+ detail_bounds_utc = [
6287
+ (
6288
+ start.astimezone(dt.timezone.utc),
6289
+ end.astimezone(dt.timezone.utc),
6290
+ )
6291
+ for start, end in detail_bounds
6292
+ ]
6293
+ window_intervals = [
6294
+ {"start_at": _iso_z(start), "end_at": _iso_z(end)}
6295
+ for start, end in detail_bounds_utc
6296
+ ]
6254
6297
  since_iso = since_dt.astimezone(dt.timezone.utc).strftime(
6255
6298
  "%Y-%m-%dT%H:%M:%SZ"
6256
6299
  )
@@ -6269,9 +6312,6 @@ def _project_detail_for_window(
6269
6312
  since_sql_iso = (
6270
6313
  since_dt.astimezone(dt.timezone.utc) - dt.timedelta(seconds=1)
6271
6314
  ).strftime("%Y-%m-%dT%H:%M:%SZ")
6272
- until_dt_utc = until_dt.astimezone(dt.timezone.utc)
6273
- since_dt_utc = since_dt.astimezone(dt.timezone.utc)
6274
-
6275
6315
  # ---- Build bucket → source_paths map for SQL-side scoping ----------
6276
6316
  # Walk session_files (~8k rows) once instead of session_entries
6277
6317
  # (~150k+ rows). _resolve_project_key gets called at most ~distinct-
@@ -6324,6 +6364,7 @@ def _project_detail_for_window(
6324
6364
  "window_weeks": weeks_back,
6325
6365
  "window_start_at": since_iso,
6326
6366
  "window_end_at": until_iso,
6367
+ "window_intervals": window_intervals,
6327
6368
  "window_cost_usd": 0.0,
6328
6369
  "window_attributed_pct": None,
6329
6370
  "models": [],
@@ -6391,7 +6432,10 @@ def _project_detail_for_window(
6391
6432
  # compare the way the interval means; this is where the interval is
6392
6433
  # actually decided.
6393
6434
  ts_utc = ts.astimezone(dt.timezone.utc)
6394
- if not (since_dt_utc <= ts_utc < until_dt_utc):
6435
+ if not any(
6436
+ interval_start <= ts_utc < interval_end
6437
+ for interval_start, interval_end in detail_bounds_utc
6438
+ ):
6395
6439
  continue
6396
6440
  entry_cost = _calculate_entry_cost(
6397
6441
  model,
@@ -6517,6 +6561,7 @@ def _project_detail_for_window(
6517
6561
  "window_weeks": weeks_back,
6518
6562
  "window_start_at": since_iso,
6519
6563
  "window_end_at": until_iso,
6564
+ "window_intervals": window_intervals,
6520
6565
  "window_cost_usd": window_cost,
6521
6566
  "window_attributed_pct": win_pct,
6522
6567
  "models": models_out,
@@ -6731,6 +6776,36 @@ def _qs_str(q: dict, key: str, default: str | None) -> str | None:
6731
6776
  return vals[0] if vals else default
6732
6777
 
6733
6778
 
6779
+ def _qs_flag(q: dict, key: str) -> bool:
6780
+ """Parse a single query-string boolean.
6781
+
6782
+ A bare `?flag` and `?flag=1` are both true; `0`, `false` and `no` are
6783
+ false. Written out rather than spelled `bool(_qs_str(...))`, because that
6784
+ reads `?reveal_projects=0` as a REQUEST to reveal — the string "0" is
6785
+ truthy — which is the wrong direction for a privacy switch.
6786
+ """
6787
+ raw = _qs_str(q, key, None)
6788
+ if raw is None:
6789
+ return False
6790
+ return raw.strip().lower() not in ("0", "false", "no", "off")
6791
+
6792
+
6793
+ class _DiagnosisSelectorError(Exception):
6794
+ """A `/api/diagnosis` request whose selectors are wrong (#620 S2).
6795
+
6796
+ Distinct from `EstablishmentFailure`, whose codes are the closed
6797
+ report-establishment enum. A selector this route rejects — an unknown
6798
+ source, an account combined with `source=all` — never reached the point of
6799
+ establishing a report, so it carries its own code rather than borrowing one
6800
+ from that enum and claiming a stage it never got to.
6801
+ """
6802
+
6803
+ def __init__(self, code: str, message: str) -> None:
6804
+ super().__init__(message)
6805
+ self.code = code
6806
+ self.message = message
6807
+
6808
+
6734
6809
  # ── /api/debug/backend on-demand cache-state helpers (issue #276, Session A) ──
6735
6810
  # All read-only, cheap, and privacy-safe: they leak ONLY row counts, signature
6736
6811
  # legs (ints/tuples), pending-flag names, and the tool version — never prompt /
@@ -7011,6 +7086,11 @@ _GET_ROUTES = (
7011
7086
  ("exact", "/api/share/presets", "_handle_share_presets_get", None, False),
7012
7087
  ("exact", "/api/share/history", "_handle_share_history_get", None, False),
7013
7088
  ("exact", "/api/doctor", "_handle_get_doctor", None, False),
7089
+ # #620 S2 — the on-demand diagnosis. Exact, so it can neither shadow nor
7090
+ # be shadowed by the parameterized conversation routes below (dispatch
7091
+ # compares exact paths first, so the placement here is a convention).
7092
+ ("exact", "/api/diagnosis", "_handle_get_diagnosis",
7093
+ ("scope", "endpoint.diagnosis"), False),
7014
7094
  ("exact", "/api/debug/backend", "_handle_get_debug_backend", None, False),
7015
7095
  ("exact", "/api/conversations/facets", "_handle_get_conversations_facets",
7016
7096
  ("scope", "endpoint.conversations_facets"), False),
@@ -7488,32 +7568,22 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
7488
7568
  })
7489
7569
  return
7490
7570
  try:
7491
- # The locked section is a REBUILD, and this is the path an
7492
- # uncontended manual refresh actually takes, so it reports itself
7493
- # like every other rebuild does.
7494
- cls.mark_rebuilding(True)
7495
- warnings: list = []
7496
- if do_refresh:
7497
- if cls.no_sync:
7498
- warnings.append({"code": "refresh_skipped_no_sync"})
7499
- else:
7500
- result = _refresh_usage_inproc()
7501
- if result.status != "ok":
7502
- warnings.append({"code": result.status})
7503
- try:
7504
- cls.run_sync_now_locked()
7505
- except Exception as exc:
7506
- self.log_error("/api/sync rebuild failed: %r", exc)
7507
- self.send_error(500, "sync failed")
7508
- return
7571
+ with _rebuilding_claim(cls.mark_rebuilding):
7572
+ warnings: list = []
7573
+ if do_refresh:
7574
+ if cls.no_sync:
7575
+ warnings.append({"code": "refresh_skipped_no_sync"})
7576
+ else:
7577
+ result = _refresh_usage_inproc()
7578
+ if result.status != "ok":
7579
+ warnings.append({"code": result.status})
7580
+ try:
7581
+ cls.run_sync_now_locked()
7582
+ except Exception as exc:
7583
+ self.log_error("/api/sync rebuild failed: %r", exc)
7584
+ self.send_error(500, "sync failed")
7585
+ return
7509
7586
  finally:
7510
- # Drops THIS thread's claim only, so the ordering against the lock
7511
- # release is not what makes it safe: a concurrent rebuilder's claim
7512
- # is a different set member and this call cannot touch it, whichever
7513
- # side of the release it runs on. On the success path the rebuild's
7514
- # terminal publish has already dropped this thread's claim and this
7515
- # adds no frame; the exception path is what needs it.
7516
- cls.mark_rebuilding(False)
7517
7587
  sync_lock.release()
7518
7588
 
7519
7589
  if warnings:
@@ -8532,12 +8602,10 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
8532
8602
  # outside the acquire because `run_sync_now` takes `sync_lock` itself;
8533
8603
  # a wait for a rebuild already in flight is honestly in-flight too.
8534
8604
  try:
8535
- type(self).mark_rebuilding(True)
8536
- type(self).run_sync_now()
8605
+ with _rebuilding_claim(type(self).mark_rebuilding):
8606
+ type(self).run_sync_now()
8537
8607
  except Exception as exc:
8538
8608
  eprint(f"warning: settings broadcast failed: {exc!r}")
8539
- finally:
8540
- type(self).mark_rebuilding(False)
8541
8609
 
8542
8610
  self._respond_json(200, out)
8543
8611
 
@@ -8956,6 +9024,250 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
8956
9024
  self.log_error("/api/doctor failed after commit: %r", exc)
8957
9025
  self.close_connection = True
8958
9026
 
9027
+ # ── GET /api/diagnosis (#620 S2, spec §5) ───────────────────────────────
9028
+ #
9029
+ # The on-demand diagnosis. NOT an envelope key: an envelope key would pay
9030
+ # #607's client per-frame cost on every tick for a surface most ticks never
9031
+ # display, and would force a deliberate re-take of the byte-identity
9032
+ # baseline at bench/baselines/envelope-oracle.json.
9033
+
9034
+ def _send_diagnosis_json(self, status: int, body: dict) -> None:
9035
+ encoded = encode_dashboard_json_bytes(body, ensure_ascii=False)
9036
+ self.send_response(status)
9037
+ self.send_header("Content-Type", "application/json; charset=utf-8")
9038
+ self.send_header("Content-Length", str(len(encoded)))
9039
+ self.send_header("Cache-Control", "no-cache")
9040
+ self.end_headers()
9041
+ self.wfile.write(encoded)
9042
+
9043
+ def _diagnosis_selectors(self, query):
9044
+ """Resolve one request's selectors into a `DiagnosisScope`.
9045
+
9046
+ Raises `_DiagnosisSelectorError` for anything the caller got wrong and
9047
+ `EstablishmentFailure` for anything that could not be established. Both
9048
+ are 400 here; only `store_unavailable` and `generation_incoherent` are
9049
+ 503, because those two say the machine could not answer rather than
9050
+ that the request was wrong.
9051
+ """
9052
+ c = _cctally()
9053
+ diagnosis = c._load_sibling("_cctally_diagnosis")
9054
+ sources = c._load_sibling("_cctally_diagnosis_sources")
9055
+ kernel = c._load_sibling("_lib_diagnosis")
9056
+
9057
+ source = _qs_str(query, "source", "claude") or "claude"
9058
+ if source not in ("claude", "codex", "all"):
9059
+ raise _DiagnosisSelectorError(
9060
+ "invalid_selector",
9061
+ f"source must be claude, codex or all, not {source!r}")
9062
+ account = _qs_str(query, "account", None)
9063
+ if account is not None and source == "all":
9064
+ # Account keys are provider-scoped, so one selector cannot address
9065
+ # both providers. The CLI exits 2 on the same condition.
9066
+ raise _DiagnosisSelectorError(
9067
+ "invalid_selector",
9068
+ "account cannot be combined with source=all "
9069
+ "(account keys are provider-scoped)")
9070
+ speed = _qs_str(query, "speed", None)
9071
+ if speed in ("auto", ""):
9072
+ speed = None
9073
+ if speed is not None and speed not in ("standard", "fast"):
9074
+ raise _DiagnosisSelectorError(
9075
+ "invalid_selector",
9076
+ f"speed must be auto, standard or fast, not {speed!r}")
9077
+
9078
+ raw_tz = _qs_str(query, "tz", None)
9079
+ config = _apply_display_tz_override(
9080
+ load_config(), type(self).display_tz_pref_override,
9081
+ )
9082
+ try:
9083
+ tz_name = diagnosis.resolve_tz_name(
9084
+ argparse.Namespace(tz=raw_tz), config,
9085
+ )
9086
+ except ValueError as exc:
9087
+ raise _DiagnosisSelectorError("invalid_selector", str(exc)) from exc
9088
+
9089
+ now_utc = _command_as_of()
9090
+ start_raw = _qs_str(query, "start_at", None)
9091
+ end_raw = _qs_str(query, "end_at", None)
9092
+ if start_raw or end_raw:
9093
+ # The alert-follow path carries INSTANTS, not calendar days: a
9094
+ # five-hour block start is not a date, so the window grammar's
9095
+ # date-only form cannot reproduce the window a warning fired
9096
+ # against. Explicit bounds are how that window reaches the route.
9097
+ if not (start_raw and end_raw):
9098
+ raise kernel.EstablishmentFailure(
9099
+ kernel.EstablishmentError.RANGE_UNRESOLVED.value,
9100
+ "start_at and end_at must be given together")
9101
+ start_at = parse_iso_datetime(start_raw, "start_at")
9102
+ end_at = parse_iso_datetime(end_raw, "end_at")
9103
+ start_at = (start_at.replace(tzinfo=dt.timezone.utc)
9104
+ if start_at.tzinfo is None
9105
+ else start_at.astimezone(dt.timezone.utc))
9106
+ end_at = (end_at.replace(tzinfo=dt.timezone.utc)
9107
+ if end_at.tzinfo is None
9108
+ else end_at.astimezone(dt.timezone.utc))
9109
+ label = ""
9110
+ else:
9111
+ token = _qs_str(query, "window", "this-week") or "this-week"
9112
+ diff_kernel = c._load_sibling("_lib_diff_kernel")
9113
+ try:
9114
+ parsed = diagnosis._resolve_window(
9115
+ argparse.Namespace(window=token), now_utc, tz_name,
9116
+ )
9117
+ except diff_kernel.NoAnchorError as exc:
9118
+ # A week token this machine holds no anchor for is an
9119
+ # unresolved RANGE. It is a `RuntimeError`, so without this it
9120
+ # reached the outer handler and was served as 500 — an
9121
+ # unexplained internal error for a request the server could
9122
+ # perfectly well describe.
9123
+ raise kernel.EstablishmentFailure(
9124
+ kernel.EstablishmentError.RANGE_UNRESOLVED.value,
9125
+ str(exc)) from exc
9126
+ start_at, end_at, label = (parsed.start_utc, parsed.end_utc,
9127
+ parsed.label)
9128
+
9129
+ account_key = None
9130
+ if account is not None:
9131
+ account_key = self._resolve_diagnosis_account(
9132
+ sources, account,
9133
+ source if source in ("claude", "codex") else "claude",
9134
+ )
9135
+ # DiagnosisScope's own __post_init__ raises `range_unresolved` for a
9136
+ # naive or inverted window, so an out-of-order pair is rejected by the
9137
+ # same rule the CLI applies rather than by a second one here.
9138
+ return sources.DiagnosisScope(
9139
+ source=source,
9140
+ account_key=account_key,
9141
+ window_start=start_at,
9142
+ window_end=end_at,
9143
+ effective_speed=speed,
9144
+ display_tz=tz_name,
9145
+ label=label,
9146
+ ), now_utc, _qs_flag(query, "reveal_projects")
9147
+
9148
+ @staticmethod
9149
+ def _resolve_diagnosis_account(sources, ref: str, provider: str):
9150
+ """Resolve `?account=` over the diagnosis's own read-only open path.
9151
+
9152
+ `resolve_account_filter` reaches stats.db through the ordinary opener,
9153
+ which migrates, repairs, imports and replays. This route reads and
9154
+ never writes, so it resolves the ref over `mode=ro`, exactly as
9155
+ `cmd_explain` does — and a ref that could not be resolved for ANY
9156
+ reason, including a machine with no registry at all, is an unresolved
9157
+ account rather than an unavailable store.
9158
+ """
9159
+ import _lib_accounts as accounts
9160
+ kernel = _cctally()._load_sibling("_lib_diagnosis")
9161
+ try:
9162
+ conn = sources.open_read_only("stats")
9163
+ except kernel.EstablishmentFailure as exc:
9164
+ raise kernel.EstablishmentFailure(
9165
+ kernel.EstablishmentError.ACCOUNT_UNRESOLVED.value,
9166
+ f"account {ref!r} is ambiguous or unknown "
9167
+ f"(this machine holds no account registry)") from exc
9168
+ try:
9169
+ return accounts.resolve_account_ref(conn, ref, provider)
9170
+ except accounts.AccountRefError as exc:
9171
+ raise kernel.EstablishmentFailure(
9172
+ kernel.EstablishmentError.ACCOUNT_UNRESOLVED.value,
9173
+ f"account {ref!r} is ambiguous or unknown") from exc
9174
+ finally:
9175
+ conn.close()
9176
+
9177
+ def _handle_get_diagnosis(self) -> None:
9178
+ """`GET /api/diagnosis` — the on-demand diagnosis (#620 S2, spec §5).
9179
+
9180
+ Read-only and mutating nothing: the whole read goes through
9181
+ `_cctally_diagnosis_sources`, whose only opener is a `mode=ro` connect
9182
+ that performs no schema work, no migration, no legacy import and no
9183
+ contract repair. It runs on the request thread, outside the snapshot
9184
+ build and outside its pinned cache transaction.
9185
+
9186
+ `_require_api_auth` applies automatically, before dispatch. No CSRF
9187
+ check: `_check_origin_csrf` is opt-in for the routes that mutate, and
9188
+ this one does not.
9189
+
9190
+ Status codes: 200 for any valid coherently-generated report, including
9191
+ a healthy one, an empty one and one whose fields are withheld; 400 for
9192
+ a malformed selector and for an unresolved range or account; 503 for
9193
+ `generation_incoherent`, for `store_unavailable`, and for a report the
9194
+ kernel's `unreadable_store_is_terminal` holds over — that last one
9195
+ publishes the withheld REPORT as the body, not an empty error, for the
9196
+ same reason the CLI still prints it while exiting 3: a person needs to
9197
+ read the typed cause. 500 for an unexpected invariant failure, which is
9198
+ never converted into a healthy 200.
9199
+ """
9200
+ import urllib.parse as _urlparse
9201
+
9202
+ c = _cctally()
9203
+ kernel = c._load_sibling("_lib_diagnosis")
9204
+ diagnosis = c._load_sibling("_cctally_diagnosis")
9205
+ sources = c._load_sibling("_cctally_diagnosis_sources")
9206
+ query = _urlparse.parse_qs(_urlparse.urlparse(self.path).query)
9207
+
9208
+ # Preparation and commit are separate, as on every other JSON route:
9209
+ # before headers a failure can still become a JSON 500; after them the
9210
+ # only valid recovery is log + close.
9211
+ try:
9212
+ try:
9213
+ scope, now_utc, reveal = self._diagnosis_selectors(query)
9214
+ # Evaluated ONCE, before the plan is resolved, and threaded
9215
+ # into plan stage 1. A class denied there is settled: the
9216
+ # store it would have needed is never opened, probed or
9217
+ # digested, and the identifier this response publishes
9218
+ # describes what actually ran rather than what was intended.
9219
+ # The predicate itself is untouched (D-E) — this composes it,
9220
+ # it does not change it.
9221
+ report = sources.build_diagnosis(
9222
+ scope, measured_at=now_utc,
9223
+ transcripts_visible=self._transcripts_visible_to_request(),
9224
+ )
9225
+ except _DiagnosisSelectorError as exc:
9226
+ self._send_diagnosis_json(
9227
+ 400, {"error": exc.message, "code": exc.code})
9228
+ return
9229
+ except ValueError as exc:
9230
+ # A malformed instant from `parse_iso_datetime`.
9231
+ self._send_diagnosis_json(400, {
9232
+ "error": str(exc),
9233
+ "code": kernel.EstablishmentError.RANGE_UNRESOLVED.value,
9234
+ })
9235
+ return
9236
+ except kernel.EstablishmentFailure as exc:
9237
+ # `store_unavailable` reaches this arm only from SELECTOR
9238
+ # resolution — the week anchor, which reads stats.db through
9239
+ # the ordinary opener. It can never arrive from
9240
+ # `build_diagnosis`, which converts every store failure it
9241
+ # meets into `_unavailable_provider_result` so the withheld
9242
+ # report is published rather than lost; nor from
9243
+ # `_resolve_diagnosis_account`, which re-codes an unopenable
9244
+ # stats.db as `account_unresolved`. Both conversions are
9245
+ # deliberate, so do not read the mapping as dead code.
9246
+ status = 503 if exc.code in (
9247
+ kernel.EstablishmentError.STORE_UNAVAILABLE.value,
9248
+ kernel.EstablishmentError.GENERATION_INCOHERENT.value,
9249
+ ) else 400
9250
+ self._send_diagnosis_json(
9251
+ status, {"error": exc.message, "code": exc.code})
9252
+ return
9253
+
9254
+ scopes = {result.source: diagnosis._scope_for(sources, scope,
9255
+ result.source)
9256
+ for result in report.results}
9257
+ body = diagnosis.diagnosis_to_wire(report, scopes=scopes,
9258
+ reveal_projects=reveal)
9259
+ status = 503 if kernel.unreadable_store_is_terminal(report) else 200
9260
+ except Exception as exc: # noqa: BLE001
9261
+ self.log_error("/api/diagnosis failed before commit: %r", exc)
9262
+ self._send_diagnosis_json(500, {"error": "internal error"})
9263
+ return
9264
+
9265
+ try:
9266
+ self._send_diagnosis_json(status, body)
9267
+ except Exception as exc: # noqa: BLE001
9268
+ self.log_error("/api/diagnosis failed after commit: %r", exc)
9269
+ self.close_connection = True
9270
+
8959
9271
  def _handle_get_session_detail(self, path: str) -> None:
8960
9272
  """Return TuiSessionDetail JSON for the given session id (spec §3.2).
8961
9273
 
@@ -9274,12 +9586,14 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
9274
9586
  entries_in_window = list(get_entries(
9275
9587
  start_at, end_at, skip_sync=self.no_sync,
9276
9588
  ))
9589
+ entry_membership: dict[int, list[UsageEntry]] = {}
9277
9590
  blocks = _group_entries_into_blocks(
9278
9591
  entries_in_window, mode="auto",
9279
9592
  recorded_windows=recorded_windows,
9280
9593
  block_start_overrides=block_start_overrides,
9281
9594
  canonical_intervals=canonical_intervals,
9282
9595
  now=now_utc,
9596
+ _entry_membership=entry_membership,
9283
9597
  )
9284
9598
  target = next(
9285
9599
  (b for b in blocks
@@ -9290,10 +9604,7 @@ class DashboardHTTPHandler(BaseHTTPRequestHandler):
9290
9604
  status = 404
9291
9605
  body = encode_dashboard_json_bytes({"error": "block not found"})
9292
9606
  else:
9293
- block_entries = [
9294
- e for e in entries_in_window
9295
- if target.start_time <= e.timestamp < target.end_time
9296
- ]
9607
+ block_entries = entry_membership[id(target)]
9297
9608
  # Resolve display tz once per request so the block detail's
9298
9609
  # `label` matches the snapshot envelope's blocks panel.
9299
9610
  # Shared resolver -- same warn-once semantics as