cctally 1.99.1 → 1.101.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +88 -0
- package/bin/_cctally_alerts.py +13 -2
- package/bin/_cctally_cache.py +3 -1
- package/bin/_cctally_cache_report.py +103 -6
- package/bin/_cctally_dashboard.py +2083 -356
- package/bin/_cctally_dashboard_envelope.py +115 -31
- package/bin/_cctally_dashboard_perf.py +433 -0
- package/bin/_cctally_dashboard_share.py +101 -29
- package/bin/_cctally_dashboard_sources.py +1072 -214
- package/bin/_cctally_db.py +30 -14
- package/bin/_cctally_diff.py +20 -0
- package/bin/_cctally_doctor.py +1331 -1148
- package/bin/_cctally_forecast.py +304 -96
- package/bin/_cctally_milestone_history.py +10 -2
- package/bin/_cctally_parser.py +70 -1
- package/bin/_cctally_project.py +155 -47
- package/bin/_cctally_quota.py +40 -21
- package/bin/_cctally_record.py +44 -3
- package/bin/_cctally_refresh.py +60 -10
- package/bin/_cctally_share.py +9 -2
- package/bin/_cctally_source_analytics.py +40 -4
- package/bin/_cctally_statusline.py +53 -7
- package/bin/_cctally_tui.py +723 -232
- package/bin/_cctally_update.py +28 -22
- package/bin/_lib_alert_scope.py +685 -0
- package/bin/_lib_alerts_payload.py +112 -7
- package/bin/_lib_cache_report.py +110 -1
- package/bin/_lib_codex_pools.py +20 -8
- package/bin/_lib_dashboard_sources.py +237 -30
- package/bin/_lib_doctor.py +37 -0
- package/bin/_lib_forecast.py +12 -4
- package/bin/_lib_jsonl.py +4 -2
- package/bin/_lib_perf.py +132 -3
- package/bin/_lib_pricing.py +8 -7
- package/bin/_lib_render.py +31 -3
- package/bin/_lib_share_templates.py +150 -55
- package/bin/_lib_snapshot_cache.py +71 -13
- package/bin/_lib_source_analytics.py +2 -2
- package/bin/_lib_source_identity.py +50 -2
- package/bin/_lib_subscription_weeks.py +65 -0
- package/bin/_lib_tick_stats.py +538 -0
- package/bin/cctally +29 -7
- package/dashboard/static/assets/dashboardStream.shared-worker-1XTMV3nr.js +1 -0
- package/dashboard/static/assets/index-D6Eb9KDn.js +97 -0
- package/dashboard/static/assets/index-i3g7g8zo.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +4 -1
- package/dashboard/static/assets/index-C5NBB2w9.js +0 -97
- package/dashboard/static/assets/index-hJP4wlIO.css +0 -1
|
@@ -385,8 +385,9 @@ def _envelope_rows_weekly(
|
|
|
385
385
|
# key uniqueness invariant matters.
|
|
386
386
|
rows = conn.execute(
|
|
387
387
|
f"""
|
|
388
|
-
SELECT week_start_date,
|
|
389
|
-
alerted_at, cumulative_cost_usd,
|
|
388
|
+
SELECT week_start_date, week_start_at, percent_threshold,
|
|
389
|
+
captured_at_utc, alerted_at, cumulative_cost_usd,
|
|
390
|
+
reset_event_id, account_key
|
|
390
391
|
FROM {descriptor.milestone_table}
|
|
391
392
|
WHERE alerted_at IS NOT NULL
|
|
392
393
|
ORDER BY {_CANON_ALERTED_AT} DESC
|
|
@@ -409,6 +410,16 @@ def _envelope_rows_weekly(
|
|
|
409
410
|
**account_fields("claude", r["account_key"]),
|
|
410
411
|
"context": {
|
|
411
412
|
"week_start_date": r["week_start_date"],
|
|
413
|
+
# The subscription week's RESET INSTANT, which the milestone
|
|
414
|
+
# row retains and `AlertEntry.context` already declares. A
|
|
415
|
+
# reader given only `week_start_date` can place the window no
|
|
416
|
+
# better than UTC midnight, which is wrong by the whole
|
|
417
|
+
# reset-hour offset for every account that does not reset at
|
|
418
|
+
# midnight. Empty string on a row that predates the column,
|
|
419
|
+
# mirroring `block_start_at` on the five-hour axis below: the
|
|
420
|
+
# key stays on the wire and the reader degrades to day
|
|
421
|
+
# granularity instead of inventing a clock reading.
|
|
422
|
+
"week_start_at": r["week_start_at"] or "",
|
|
412
423
|
"cumulative_cost_usd": cumulative,
|
|
413
424
|
"dollars_per_percent": dpp,
|
|
414
425
|
# Round-3: parallel to the 5h context block below — both
|
|
@@ -906,6 +917,7 @@ def _sync_failure_envelope(
|
|
|
906
917
|
|
|
907
918
|
def attributed(
|
|
908
919
|
database: str, *, corruption: bool = False, leg: str | None = None,
|
|
920
|
+
sqlite_busy: bool = False,
|
|
909
921
|
) -> bool:
|
|
910
922
|
for item in attributions or ():
|
|
911
923
|
item_database = (
|
|
@@ -920,9 +932,15 @@ def _sync_failure_envelope(
|
|
|
920
932
|
item.get("leg") if isinstance(item, dict)
|
|
921
933
|
else getattr(item, "leg", None)
|
|
922
934
|
)
|
|
935
|
+
item_busy = (
|
|
936
|
+
item.get("sqlite_busy") if isinstance(item, dict)
|
|
937
|
+
else getattr(item, "sqlite_busy", False)
|
|
938
|
+
)
|
|
923
939
|
if item_database == database and (
|
|
924
940
|
not corruption or bool(item_corruption)
|
|
925
|
-
) and (leg is None or item_leg == leg)
|
|
941
|
+
) and (leg is None or item_leg == leg) and (
|
|
942
|
+
not sqlite_busy or bool(item_busy)
|
|
943
|
+
):
|
|
926
944
|
return True
|
|
927
945
|
return False
|
|
928
946
|
|
|
@@ -977,6 +995,34 @@ def _sync_failure_envelope(
|
|
|
977
995
|
"action": None,
|
|
978
996
|
}
|
|
979
997
|
|
|
998
|
+
# #583 S2 §7. A locked cache.db is named rather than left in the generic
|
|
999
|
+
# bucket. Typed attribution only — a blanket uncorrupted-cache branch
|
|
1000
|
+
# would be too broad, because ANY exception raised while reading cache.db
|
|
1001
|
+
# produces that attribution and every one would then be told to
|
|
1002
|
+
# checkpoint. `database="stats_or_cache"` deliberately does NOT reach here:
|
|
1003
|
+
# that value means ownership was not established, and a guess would
|
|
1004
|
+
# produce a confidently wrong remedy.
|
|
1005
|
+
# Corruption outranks busy, and the gate is required rather than implied by
|
|
1006
|
+
# branch order: `attributed(...)` scans the WHOLE attribution list, so
|
|
1007
|
+
# without it a tick carrying a corruption-shaped cache attribution AND a
|
|
1008
|
+
# separate busy one rendered "cache database busy" and dropped the
|
|
1009
|
+
# corruption message — the more urgent of the two, whose remedy is not
|
|
1010
|
+
# interchangeable with a checkpoint. The gate names the TYPED corruption
|
|
1011
|
+
# attribution only: a busy attribution still outranks the legacy raw-text
|
|
1012
|
+
# corruption legs below, which is what Preserve 10 asks for.
|
|
1013
|
+
if attributed("cache", sqlite_busy=True) and not attributed(
|
|
1014
|
+
"cache", corruption=True
|
|
1015
|
+
):
|
|
1016
|
+
return {
|
|
1017
|
+
"kind": "cache_busy",
|
|
1018
|
+
"label": "⚠ cache database busy",
|
|
1019
|
+
"detail": (
|
|
1020
|
+
"The dashboard could not complete sync because cache.db "
|
|
1021
|
+
"stayed locked."
|
|
1022
|
+
),
|
|
1023
|
+
"action": "cctally db checkpoint",
|
|
1024
|
+
}
|
|
1025
|
+
|
|
980
1026
|
text = error.casefold()
|
|
981
1027
|
if (
|
|
982
1028
|
"stale maintenance marker" in text
|
|
@@ -1018,6 +1064,43 @@ def _sync_failure_envelope(
|
|
|
1018
1064
|
}
|
|
1019
1065
|
|
|
1020
1066
|
|
|
1067
|
+
def _sync_activity_envelope(activity: "dict | None") -> dict:
|
|
1068
|
+
"""Serialize `_SnapshotRef`'s queue/activity state (#583 S2 spec 6.2).
|
|
1069
|
+
|
|
1070
|
+
Always returns a complete object: a snapshot built before the reference
|
|
1071
|
+
stamped one carries ``None``, and an absent object must read as idle
|
|
1072
|
+
rather than as a missing key the client has to special-case.
|
|
1073
|
+
|
|
1074
|
+
``server_epoch`` is a fixed-length, per-process token with no data content.
|
|
1075
|
+
It is excluded from `bin/cctally-snapshot-measure`'s stable digest and
|
|
1076
|
+
sentinelized by `bin/cctally-dashboard-test`, the same treatment
|
|
1077
|
+
``data_version`` and ``doctor`` already get. Everything else here is real
|
|
1078
|
+
published content and stays in the digest.
|
|
1079
|
+
"""
|
|
1080
|
+
act = activity or {}
|
|
1081
|
+
return {
|
|
1082
|
+
"server_epoch": act.get("server_epoch") or "",
|
|
1083
|
+
"rebuilding": bool(act.get("rebuilding", False)),
|
|
1084
|
+
"requested_id": int(act.get("requested_id", 0)),
|
|
1085
|
+
"started_id": int(act.get("started_id", 0)),
|
|
1086
|
+
"settled_id": int(act.get("settled_id", 0)),
|
|
1087
|
+
"settled_status": act.get("settled_status"),
|
|
1088
|
+
# Tuples are not JSON. Convert here rather than relying on the encoder
|
|
1089
|
+
# rendering one as an array by luck. The isinstance filter is a
|
|
1090
|
+
# fail-open guard: `dict(w)` raises on a non-mapping, and this
|
|
1091
|
+
# serializer runs on every published frame, so one malformed warning
|
|
1092
|
+
# would take down the whole envelope rather than drop one entry. The
|
|
1093
|
+
# test is `Mapping`, not `dict`, because the thing that raises is a
|
|
1094
|
+
# NON-mapping; a mapping that is merely not a `dict` converts fine, and
|
|
1095
|
+
# rejecting it would silently drop a good warning — and this field is
|
|
1096
|
+
# the only place a queued request's deferred outcome is reported.
|
|
1097
|
+
"settled_warnings": [
|
|
1098
|
+
dict(w) for w in (act.get("settled_warnings") or ())
|
|
1099
|
+
if isinstance(w, Mapping)
|
|
1100
|
+
],
|
|
1101
|
+
}
|
|
1102
|
+
|
|
1103
|
+
|
|
1021
1104
|
def snapshot_to_envelope(snap: "DataSnapshot", *,
|
|
1022
1105
|
now_utc: "dt.datetime",
|
|
1023
1106
|
monotonic_now: "float | None" = None,
|
|
@@ -1588,6 +1671,14 @@ def snapshot_to_envelope(snap: "DataSnapshot", *,
|
|
|
1588
1671
|
# assembled); False on every complete/stable snapshot. ``getattr``
|
|
1589
1672
|
# default keeps positionally-constructed fixture snapshots serializing.
|
|
1590
1673
|
"hydrating": bool(getattr(snap, "hydrating", False)),
|
|
1674
|
+
# #583 S2: queue / activity state, owned by `_SnapshotRef`. Published
|
|
1675
|
+
# unconditionally so a client can always read it; an absent or empty
|
|
1676
|
+
# dict renders as the idle object. Monotonic identifiers rather than a
|
|
1677
|
+
# boolean, because `SSEHub.publish` is latest-wins and the frame that
|
|
1678
|
+
# would prove a particular request finished may never be delivered.
|
|
1679
|
+
"sync_activity": _sync_activity_envelope(
|
|
1680
|
+
getattr(snap, "sync_activity", None)
|
|
1681
|
+
),
|
|
1591
1682
|
"generated_at": _iso_z(snap.generated_at),
|
|
1592
1683
|
# #300: the all-inputs data-version string at build time (changes iff any
|
|
1593
1684
|
# DB leg the detail endpoints read changed; flat on an idle tick). The
|
|
@@ -1858,29 +1949,24 @@ def snapshot_to_envelope(snap: "DataSnapshot", *,
|
|
|
1858
1949
|
|
|
1859
1950
|
|
|
1860
1951
|
def _claude_source_session_rows(envelope: dict) -> list:
|
|
1861
|
-
"""The SOURCE-scoped Claude session row lists
|
|
1862
|
-
|
|
1863
|
-
|
|
1952
|
+
"""The SOURCE-scoped Claude session row lists the overlay must cover.
|
|
1953
|
+
|
|
1954
|
+
Since source schema 10 (#583 S3 §4) there is exactly ONE such list, on the
|
|
1955
|
+
physical ``sources.claude`` entry. The null
|
|
1956
|
+
``sources.all.data.providers.claude`` compatibility stub is not a second
|
|
1957
|
+
row owner and must not receive request-local transcript data.
|
|
1958
|
+
"""
|
|
1864
1959
|
sources = envelope.get("sources")
|
|
1865
1960
|
if not isinstance(sources, Mapping):
|
|
1866
1961
|
return []
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
|
|
1873
|
-
|
|
1874
|
-
|
|
1875
|
-
if not isinstance(data, Mapping):
|
|
1876
|
-
continue
|
|
1877
|
-
sessions = data.get("sessions")
|
|
1878
|
-
if not isinstance(sessions, Mapping):
|
|
1879
|
-
continue
|
|
1880
|
-
rows = sessions.get("rows")
|
|
1881
|
-
if isinstance(rows, list):
|
|
1882
|
-
out.append(rows)
|
|
1883
|
-
return out
|
|
1962
|
+
data = (sources.get("claude") or {}).get("data")
|
|
1963
|
+
if not isinstance(data, Mapping):
|
|
1964
|
+
return []
|
|
1965
|
+
sessions = data.get("sessions")
|
|
1966
|
+
if not isinstance(sessions, Mapping):
|
|
1967
|
+
return []
|
|
1968
|
+
rows = sessions.get("rows")
|
|
1969
|
+
return [rows] if isinstance(rows, list) else []
|
|
1884
1970
|
|
|
1885
1971
|
|
|
1886
1972
|
def _overlay_claude_source_session_titles(
|
|
@@ -1930,6 +2016,11 @@ def _codex_source_session_rows(envelope: dict) -> list:
|
|
|
1930
2016
|
carries one ``account_scopes[*].sessions`` child per account. The client
|
|
1931
2017
|
swaps that child into view when an account chip is focused, so the private
|
|
1932
2018
|
label overlay must cover it under the same per-request transcript gate.
|
|
2019
|
+
|
|
2020
|
+
Since source schema 10 (#583 S3 §4) those lists all live on the physical
|
|
2021
|
+
``sources.codex`` entry. The null ``sources.all.data.providers.codex``
|
|
2022
|
+
compatibility stub is not a second row owner and must not receive
|
|
2023
|
+
request-local transcript data.
|
|
1933
2024
|
"""
|
|
1934
2025
|
sources = envelope.get("sources")
|
|
1935
2026
|
if not isinstance(sources, Mapping):
|
|
@@ -1957,14 +2048,7 @@ def _codex_source_session_rows(envelope: dict) -> list:
|
|
|
1957
2048
|
if isinstance(rows, list):
|
|
1958
2049
|
out.append(rows)
|
|
1959
2050
|
|
|
1960
|
-
|
|
1961
|
-
all_data = (sources.get("all") or {}).get("data")
|
|
1962
|
-
if isinstance(all_data, Mapping):
|
|
1963
|
-
providers = all_data.get("providers")
|
|
1964
|
-
if isinstance(providers, Mapping):
|
|
1965
|
-
candidates.append(providers.get("codex"))
|
|
1966
|
-
for data in candidates:
|
|
1967
|
-
append_rows(data)
|
|
2051
|
+
append_rows((sources.get("codex") or {}).get("data"))
|
|
1968
2052
|
return out
|
|
1969
2053
|
|
|
1970
2054
|
|
|
@@ -0,0 +1,433 @@
|
|
|
1
|
+
"""`cctally dashboard-perf` — read a running dashboard's tick cost (#583 S1 §3.3).
|
|
2
|
+
|
|
3
|
+
Reads the loopback diagnostic `/api/debug/backend` and renders the three
|
|
4
|
+
things an operator on a slow install needs: the publish period stated
|
|
5
|
+
SEPARATELY for the Codex-active and Codex-idle regimes, the tick-cost
|
|
6
|
+
breakdown with the ingest and builder halves named, and the dispatch mix. It
|
|
7
|
+
also arms and disarms the deep phase trace on the running process, so one
|
|
8
|
+
command covers both halves of F38 and neither answer needs a restart.
|
|
9
|
+
|
|
10
|
+
The command reaches only an IP-literal loopback host, and it is DELIBERATELY
|
|
11
|
+
STRICTER than the server about that. `_lib_transcript_access.is_loopback`
|
|
12
|
+
treats `localhost` and `::1` as loopback names before it tries to parse an IP
|
|
13
|
+
literal, so the endpoint serves `Host: localhost:8789` with 200. This command
|
|
14
|
+
still refuses a name: a name is resolved by the operating system, the client
|
|
15
|
+
cannot verify that the resolver's answer is this machine, and an unambiguous
|
|
16
|
+
target is worth more than the convenience. A hostname that is not a loopback
|
|
17
|
+
name is refused by the server too.
|
|
18
|
+
|
|
19
|
+
It also sends an `Origin` header matching the `Host` it calls, because
|
|
20
|
+
`_check_origin_csrf` rejects a request that carries none. Retaining that check
|
|
21
|
+
is deliberate — the loopback and anti-rebinding gates do not stop a malicious
|
|
22
|
+
page aiming a form POST at `http://127.0.0.1:8789` — and a non-browser client
|
|
23
|
+
can set arbitrary headers regardless, so satisfying it costs nothing.
|
|
24
|
+
|
|
25
|
+
Exit codes follow `docs/cli-contract.md`: 0 for a decoded HTTP 200, 2 for
|
|
26
|
+
argument validation including a non-loopback target, and 3 for connection,
|
|
27
|
+
HTTP, authentication, timeout or malformed-response failures — "no dashboard
|
|
28
|
+
is running" among them.
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import ipaddress
|
|
33
|
+
import json
|
|
34
|
+
import sys
|
|
35
|
+
import urllib.error
|
|
36
|
+
import urllib.request
|
|
37
|
+
|
|
38
|
+
from _lib_dashboard_json import encode_dashboard_json, encode_dashboard_json_bytes
|
|
39
|
+
|
|
40
|
+
_TIMEOUT_SECONDS = 5.0
|
|
41
|
+
_DIAGNOSTIC_PATH = "/api/debug/backend"
|
|
42
|
+
_TRACE_PATH = "/api/debug/backend/trace"
|
|
43
|
+
|
|
44
|
+
#: The two regimes the period is reported for. `not_observed` is discarded:
|
|
45
|
+
#: a tick no build's Codex decision reached says nothing about either cost
|
|
46
|
+
#: regime, and folding it into one of them would misreport that regime.
|
|
47
|
+
_REPORTED_REGIMES = ("active", "idle")
|
|
48
|
+
|
|
49
|
+
_REGIME_LABELS = {"active": "Codex-active", "idle": "Codex-idle"}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _cctally():
|
|
53
|
+
return sys.modules["cctally"]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class DashboardPerfError(Exception):
|
|
57
|
+
"""A staged failure: exit 3. Carries the message the user sees."""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def resolve_loopback_target(host: str) -> str:
|
|
61
|
+
"""Return `host` when it is an IP-literal loopback address, else raise.
|
|
62
|
+
|
|
63
|
+
A hostname is rejected even when it resolves to loopback, because the
|
|
64
|
+
endpoint's own anti-rebinding gate requires an IP-literal `Host`.
|
|
65
|
+
"""
|
|
66
|
+
try:
|
|
67
|
+
address = ipaddress.ip_address(host)
|
|
68
|
+
except ValueError:
|
|
69
|
+
raise ValueError(
|
|
70
|
+
f"--host must be an IP literal such as 127.0.0.1 or ::1 (got "
|
|
71
|
+
f"{host!r}); a hostname is refused by the endpoint's "
|
|
72
|
+
f"anti-rebinding gate"
|
|
73
|
+
) from None
|
|
74
|
+
if not address.is_loopback:
|
|
75
|
+
raise ValueError(
|
|
76
|
+
f"--host must be a loopback address (got {host!r}); "
|
|
77
|
+
f"dashboard-perf never contacts a LAN address"
|
|
78
|
+
)
|
|
79
|
+
return host
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _authority(host: str, port: int) -> str:
|
|
83
|
+
if ":" in host: # an IPv6 literal needs brackets
|
|
84
|
+
return f"[{host}]:{port}"
|
|
85
|
+
return f"{host}:{port}"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _request(host, port, path, *, token, body=None):
|
|
89
|
+
"""One loopback HTTP round trip. Raises DashboardPerfError on any failure."""
|
|
90
|
+
authority = _authority(host, port)
|
|
91
|
+
url = f"http://{authority}{path}"
|
|
92
|
+
data = None
|
|
93
|
+
headers = {"Origin": f"http://{authority}", "Accept": "application/json"}
|
|
94
|
+
if body is not None:
|
|
95
|
+
data = encode_dashboard_json_bytes(body)
|
|
96
|
+
headers["Content-Type"] = "application/json"
|
|
97
|
+
if token:
|
|
98
|
+
headers["Authorization"] = f"Bearer {token}"
|
|
99
|
+
request = urllib.request.Request(url, data=data, headers=headers,
|
|
100
|
+
method="POST" if data else "GET")
|
|
101
|
+
try:
|
|
102
|
+
with urllib.request.urlopen(request, timeout=_TIMEOUT_SECONDS) as resp:
|
|
103
|
+
raw = resp.read()
|
|
104
|
+
except urllib.error.HTTPError as exc:
|
|
105
|
+
detail = {401: "authentication required — pass --token",
|
|
106
|
+
403: "refused by the loopback gate",
|
|
107
|
+
404: "this dashboard predates dashboard-perf"}.get(
|
|
108
|
+
exc.code, "unexpected response")
|
|
109
|
+
raise DashboardPerfError(
|
|
110
|
+
f"{url} answered HTTP {exc.code}: {detail}") from None
|
|
111
|
+
except urllib.error.URLError as exc:
|
|
112
|
+
raise DashboardPerfError(
|
|
113
|
+
f"cannot reach {url}: {exc.reason}. Is a dashboard running on "
|
|
114
|
+
f"port {port}?") from None
|
|
115
|
+
except OSError as exc:
|
|
116
|
+
raise DashboardPerfError(f"cannot reach {url}: {exc}") from None
|
|
117
|
+
try:
|
|
118
|
+
return json.loads(raw.decode("utf-8"))
|
|
119
|
+
except (ValueError, UnicodeDecodeError):
|
|
120
|
+
raise DashboardPerfError(
|
|
121
|
+
f"{url} returned a malformed response") from None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ── the per-regime derivation (spec §3.3) ───────────────────────────────────
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def summarise_regime_periods(records) -> dict:
|
|
128
|
+
"""Partition the ring by `codex_regime` and describe each reported regime.
|
|
129
|
+
|
|
130
|
+
Discards `not_observed` and every record whose `period_ns` is null — the
|
|
131
|
+
first publish of a process has no predecessor, so it measures no period.
|
|
132
|
+
|
|
133
|
+
Reports the MEDIAN rather than the mean, because one startup or recovery
|
|
134
|
+
outlier otherwise dominates a 64-sample window; the observed range is
|
|
135
|
+
reported beside it so the outlier is still visible.
|
|
136
|
+
"""
|
|
137
|
+
# Imported HERE, not at module scope. `bin/cctally` re-exports this
|
|
138
|
+
# module eagerly, so a top-level import is paid by `statusline` on every
|
|
139
|
+
# Claude Code prompt and by `hook-tick` on every tool batch — measured at
|
|
140
|
+
# about 2.6 ms on `cctally --version` — for one `median` call in a command
|
|
141
|
+
# neither of them runs. The other importers in this tree
|
|
142
|
+
# (`_cctally_forecast.py`, `_lib_cache_report.py`) do the same.
|
|
143
|
+
import statistics
|
|
144
|
+
|
|
145
|
+
buckets = {regime: [] for regime in _REPORTED_REGIMES}
|
|
146
|
+
for record in records or ():
|
|
147
|
+
regime = record.get("codex_regime")
|
|
148
|
+
period = record.get("period_ns")
|
|
149
|
+
if regime in buckets and period is not None:
|
|
150
|
+
buckets[regime].append(int(period))
|
|
151
|
+
summary = {}
|
|
152
|
+
for regime, samples in buckets.items():
|
|
153
|
+
if not samples:
|
|
154
|
+
summary[regime] = {"samples": 0, "median_ns": None,
|
|
155
|
+
"min_ns": None, "max_ns": None}
|
|
156
|
+
continue
|
|
157
|
+
summary[regime] = {
|
|
158
|
+
"samples": len(samples),
|
|
159
|
+
"median_ns": int(statistics.median(samples)),
|
|
160
|
+
"min_ns": min(samples),
|
|
161
|
+
"max_ns": max(samples),
|
|
162
|
+
}
|
|
163
|
+
return summary
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def summarise_all_periods(records) -> dict:
|
|
167
|
+
"""The publish period over EVERY tick, whatever its Codex regime.
|
|
168
|
+
|
|
169
|
+
The two regime rows refine this; they do not gate it. A dispatch-`idle`
|
|
170
|
+
tick returns before `_tui_build_source_bundle`, so no build reaches the
|
|
171
|
+
Codex decision and the tick is stamped `not_observed`, which the regime
|
|
172
|
+
partition discards — correctly, because a tick that observed no decision
|
|
173
|
+
says nothing about either cost regime. On a mostly-idle install that left
|
|
174
|
+
the flagship figure reading `no samples yet` on both rows while every
|
|
175
|
+
record after the first carried a correct `period_ns`, so the operator
|
|
176
|
+
learned nothing about an install whose period was perfectly well measured.
|
|
177
|
+
"""
|
|
178
|
+
samples = [int(r["period_ns"]) for r in records or ()
|
|
179
|
+
if r.get("period_ns") is not None]
|
|
180
|
+
if not samples:
|
|
181
|
+
return {"samples": 0, "median_ns": None, "min_ns": None,
|
|
182
|
+
"max_ns": None}
|
|
183
|
+
import statistics
|
|
184
|
+
|
|
185
|
+
return {
|
|
186
|
+
"samples": len(samples),
|
|
187
|
+
"median_ns": int(statistics.median(samples)),
|
|
188
|
+
"min_ns": min(samples),
|
|
189
|
+
"max_ns": max(samples),
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _seconds(ns) -> str:
|
|
194
|
+
return "—" if ns is None else f"{ns / 1_000_000_000:.2f}s"
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _millis(ns) -> str:
|
|
198
|
+
return "—" if ns is None else f"{ns / 1_000_000:.0f}ms"
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _mean_or_none(values):
|
|
202
|
+
return int(sum(values) / len(values)) if values else None
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _median(values):
|
|
206
|
+
import statistics
|
|
207
|
+
|
|
208
|
+
return int(statistics.median(values))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _render_conversation_sync(tick: dict) -> list:
|
|
212
|
+
"""The second work loop's cost, beside the first's (#583 S4 / F5).
|
|
213
|
+
|
|
214
|
+
The share is `sum(cpu_ns) / sum(period_ns)` over the passes that carry a
|
|
215
|
+
forward period — a same-process ratio, so it is a measurement rather than a
|
|
216
|
+
machine-speed assertion. `period_ns` is `start[i+1] - start[i]`, so the
|
|
217
|
+
NEWEST retained pass has none: its successor has not started yet. That pass
|
|
218
|
+
contributes to neither sum, which is what keeps numerator and denominator
|
|
219
|
+
spanning the same window. Over the passes that do pair, the periods
|
|
220
|
+
telescope to `start[last] - start[first]` and each pass's CPU falls inside
|
|
221
|
+
its own period, so the ratio is the loop's true duty over that span.
|
|
222
|
+
|
|
223
|
+
Rows are read defensively. A running dashboard older or newer than this
|
|
224
|
+
reader can publish a different field set, and a diagnostic must degrade
|
|
225
|
+
rather than take itself down. A period of zero or less is dropped together
|
|
226
|
+
with its CPU: it cannot be a real interval, and a negative one would
|
|
227
|
+
subtract from the denominator.
|
|
228
|
+
"""
|
|
229
|
+
rows = tick.get("conversation_sync") or []
|
|
230
|
+
lines = ["", "Conversation sync loop"]
|
|
231
|
+
if not rows:
|
|
232
|
+
# The same literal the regime rows use. A loop with no samples must
|
|
233
|
+
# never render as a zero, which cannot be told apart from a measured
|
|
234
|
+
# idle loop. Under `--no-sync` the thread never starts, so this is that
|
|
235
|
+
# mode's correct and permanent reading.
|
|
236
|
+
lines.append(f" {'passes':<14} no samples yet")
|
|
237
|
+
return lines
|
|
238
|
+
durations = [int(r.get("duration_ns") or 0) for r in rows]
|
|
239
|
+
cpus = [int(r.get("cpu_ns") or 0) for r in rows]
|
|
240
|
+
paired = []
|
|
241
|
+
for record in rows:
|
|
242
|
+
cpu = record.get("cpu_ns")
|
|
243
|
+
period = record.get("period_ns")
|
|
244
|
+
if cpu is None or period is None or int(period) <= 0:
|
|
245
|
+
continue
|
|
246
|
+
paired.append((int(cpu), int(period)))
|
|
247
|
+
lines.append(
|
|
248
|
+
f" {'wall':<14} mean {_millis(_mean_or_none(durations))} "
|
|
249
|
+
f"over {len(rows)} pass(es)")
|
|
250
|
+
lines.append(f" {'thread cpu':<14} mean {_millis(_mean_or_none(cpus))}")
|
|
251
|
+
if paired:
|
|
252
|
+
periods = [p for _, p in paired]
|
|
253
|
+
lines.append(
|
|
254
|
+
f" {'period':<14} median {_seconds(_median(periods))} "
|
|
255
|
+
f"(range {_seconds(min(periods))}–{_seconds(max(periods))})")
|
|
256
|
+
share = sum(c for c, _ in paired) / sum(periods)
|
|
257
|
+
lines.append(f" {'cpu share':<14} {share * 100:.1f}% of one core")
|
|
258
|
+
else:
|
|
259
|
+
lines.append(f" {'period':<14} no samples yet")
|
|
260
|
+
statuses: dict = {}
|
|
261
|
+
for record in rows:
|
|
262
|
+
raw = record.get("status")
|
|
263
|
+
# A row with no status is malformed, not an outcome named `None`.
|
|
264
|
+
key = raw if isinstance(raw, str) and raw else "malformed"
|
|
265
|
+
statuses[key] = statuses.get(key, 0) + 1
|
|
266
|
+
detail = " · ".join(f"{k} {v}" for k, v in sorted(statuses.items()))
|
|
267
|
+
lines.append(f" {'status':<14} {detail}")
|
|
268
|
+
return lines
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def render_dashboard_perf(payload: dict) -> str:
|
|
272
|
+
"""The human report. Pure — takes the decoded diagnostic, returns text."""
|
|
273
|
+
tick = payload.get("tick") or {}
|
|
274
|
+
tracing = payload.get("tracing") or {}
|
|
275
|
+
records = tick.get("records") or []
|
|
276
|
+
lines = ["cctally dashboard-perf", ""]
|
|
277
|
+
|
|
278
|
+
lines.append("Publish period")
|
|
279
|
+
overall = summarise_all_periods(records)
|
|
280
|
+
if overall["samples"] == 0:
|
|
281
|
+
lines.append(f" {'all ticks':<14} no samples yet (0 of "
|
|
282
|
+
f"{len(records)} retained ticks qualify)")
|
|
283
|
+
else:
|
|
284
|
+
lines.append(
|
|
285
|
+
f" {'all ticks':<14} median {_seconds(overall['median_ns'])} "
|
|
286
|
+
f"over {overall['samples']} sample(s), "
|
|
287
|
+
f"range {_seconds(overall['min_ns'])}–"
|
|
288
|
+
f"{_seconds(overall['max_ns'])}")
|
|
289
|
+
summary = summarise_regime_periods(records)
|
|
290
|
+
for regime in _REPORTED_REGIMES:
|
|
291
|
+
stats = summary[regime]
|
|
292
|
+
label = _REGIME_LABELS[regime]
|
|
293
|
+
if stats["samples"] == 0:
|
|
294
|
+
# Stated, never inferred. A zero or a dash cannot distinguish
|
|
295
|
+
# "measured and fast" from "not measured", and telling those two
|
|
296
|
+
# apart is the whole point of this surface.
|
|
297
|
+
lines.append(f" {label:<14} no samples yet (0 of {len(records)} "
|
|
298
|
+
f"retained ticks qualify)")
|
|
299
|
+
continue
|
|
300
|
+
lines.append(
|
|
301
|
+
f" {label:<14} median {_seconds(stats['median_ns'])} "
|
|
302
|
+
f"over {stats['samples']} sample(s), "
|
|
303
|
+
f"range {_seconds(stats['min_ns'])}–{_seconds(stats['max_ns'])}")
|
|
304
|
+
lines.append("")
|
|
305
|
+
|
|
306
|
+
lines.append("Tick cost (exclusive halves; the remainder is orchestration)")
|
|
307
|
+
if records:
|
|
308
|
+
ingest = [r["ingest_ns"] for r in records if r.get("ingest_ran")]
|
|
309
|
+
builder = [r.get("builder_ns", 0) for r in records]
|
|
310
|
+
duration = [r.get("duration_ns", 0) for r in records]
|
|
311
|
+
lines.append(
|
|
312
|
+
f" ingest mean {_millis(_mean_or_none(ingest))} "
|
|
313
|
+
f"over {len(ingest)} tick(s) that ingested")
|
|
314
|
+
lines.append(
|
|
315
|
+
f" builder mean {_millis(_mean_or_none(builder))} "
|
|
316
|
+
f"over {len(builder)} tick(s)")
|
|
317
|
+
lines.append(
|
|
318
|
+
f" whole tick mean {_millis(_mean_or_none(duration))}")
|
|
319
|
+
# #583 S5 §2.4: the cache.db read pin, measured at its own BEGIN and
|
|
320
|
+
# ROLLBACK boundaries. Reported separately from `builder` because it
|
|
321
|
+
# is a SUBSET of builder time rather than a third exclusive half, and
|
|
322
|
+
# `.get(..., 0)` keeps an older record without the field readable.
|
|
323
|
+
pin = [r.get("cache_pin_ns", 0) or 0 for r in records]
|
|
324
|
+
lines.append(
|
|
325
|
+
f" cache pin mean {_millis(_mean_or_none(pin))} "
|
|
326
|
+
f"held inside builder")
|
|
327
|
+
newest = records[-1]
|
|
328
|
+
lines.append(
|
|
329
|
+
f" newest tick seq {newest.get('seq')} "
|
|
330
|
+
f"{newest.get('dispatch')}/{newest.get('codex_regime')}"
|
|
331
|
+
f"{'/cold' if newest.get('cold') else '/warm'} "
|
|
332
|
+
f"ingest {_millis(newest.get('ingest_ns'))} "
|
|
333
|
+
f"builder {_millis(newest.get('builder_ns'))} "
|
|
334
|
+
f"pin {_millis(newest.get('cache_pin_ns', 0) or 0)} "
|
|
335
|
+
f"total {_millis(newest.get('duration_ns'))}")
|
|
336
|
+
else:
|
|
337
|
+
lines.append(" no ticks recorded yet")
|
|
338
|
+
standalone = tick.get("standalone")
|
|
339
|
+
if standalone:
|
|
340
|
+
lines.append(
|
|
341
|
+
f" standalone builder {_millis(standalone.get('builder_ns'))} "
|
|
342
|
+
f"total {_millis(standalone.get('duration_ns'))} "
|
|
343
|
+
f"(the last build made outside a refresh tick)")
|
|
344
|
+
lines.extend(_render_conversation_sync(tick))
|
|
345
|
+
lines.append("")
|
|
346
|
+
|
|
347
|
+
counts = tick.get("dispatch_counts") or {}
|
|
348
|
+
lines.append(
|
|
349
|
+
f"Dispatch mix full {counts.get('full', 0)} · "
|
|
350
|
+
f"idle {counts.get('idle', 0)} · degraded {counts.get('degraded', 0)} "
|
|
351
|
+
f"(of {tick.get('tick_seq', 0)} completed ticks)")
|
|
352
|
+
|
|
353
|
+
failures = tick.get("cache_open_failures") or {}
|
|
354
|
+
if any(failures.values()):
|
|
355
|
+
detail = " · ".join(f"{k} {v}" for k, v in sorted(failures.items()))
|
|
356
|
+
lines.append(f"Group A cache-open failures {detail}")
|
|
357
|
+
lines.append(" These are silent: each one falls back to the wide "
|
|
358
|
+
"from-scratch fetch with byte-identical output.")
|
|
359
|
+
else:
|
|
360
|
+
lines.append("Group A cache-open failures none")
|
|
361
|
+
|
|
362
|
+
applied = tracing.get("applied")
|
|
363
|
+
requested = tracing.get("requested")
|
|
364
|
+
applies_at = tracing.get("applies_at", "none")
|
|
365
|
+
state = "on" if applied else "off"
|
|
366
|
+
lines.append(f"Phase trace applied {state} · requested "
|
|
367
|
+
f"{'on' if requested else 'off'} · applies_at {applies_at}")
|
|
368
|
+
if payload.get("phases") is not None:
|
|
369
|
+
# `generated_at` IS the stored tree's instant, from the same slot the
|
|
370
|
+
# tree comes from — which is what makes a stale tree readable as stale
|
|
371
|
+
# rather than as the last tick.
|
|
372
|
+
lines.append(f" a stored phase tree is available, generated at "
|
|
373
|
+
f"{payload.get('generated_at')}")
|
|
374
|
+
else:
|
|
375
|
+
lines.append(" no phase tree stored — arm one with "
|
|
376
|
+
"`cctally dashboard-perf --trace on`")
|
|
377
|
+
return "\n".join(lines) + "\n"
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
# ── the command ─────────────────────────────────────────────────────────────
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def cmd_dashboard_perf(args) -> int:
|
|
384
|
+
c = _cctally()
|
|
385
|
+
as_json = bool(getattr(args, "json", False))
|
|
386
|
+
try:
|
|
387
|
+
host = resolve_loopback_target(getattr(args, "host", None)
|
|
388
|
+
or "127.0.0.1")
|
|
389
|
+
except ValueError as exc:
|
|
390
|
+
print(f"dashboard-perf: {exc}", file=sys.stderr)
|
|
391
|
+
return 2
|
|
392
|
+
port = c._resolve_dashboard_port(getattr(args, "port", None))
|
|
393
|
+
if not isinstance(port, int) or not 1 <= port <= 65535:
|
|
394
|
+
# Argument validation, so exit 2. Left unchecked, a nonsensical port
|
|
395
|
+
# failed at connect time and reported itself as a transport failure,
|
|
396
|
+
# which `docs/cli-contract.md` reserves for a real one.
|
|
397
|
+
print(f"dashboard-perf: --port must be between 1 and 65535 (got "
|
|
398
|
+
f"{getattr(args, 'port', None)!r})", file=sys.stderr)
|
|
399
|
+
return 2
|
|
400
|
+
|
|
401
|
+
token = getattr(args, "token", None)
|
|
402
|
+
trace = getattr(args, "trace", None)
|
|
403
|
+
try:
|
|
404
|
+
trace_result = None
|
|
405
|
+
if trace is not None:
|
|
406
|
+
trace_result = _request(
|
|
407
|
+
host, port, _TRACE_PATH, token=token,
|
|
408
|
+
body={"enabled": trace == "on"},
|
|
409
|
+
)
|
|
410
|
+
payload = _request(host, port, _DIAGNOSTIC_PATH, token=token)
|
|
411
|
+
except DashboardPerfError as exc:
|
|
412
|
+
if as_json:
|
|
413
|
+
print(encode_dashboard_json(c.stamp_schema_version(
|
|
414
|
+
{"status": "error", "error": str(exc), "diagnostic": None})))
|
|
415
|
+
else:
|
|
416
|
+
print(f"dashboard-perf: {exc}", file=sys.stderr)
|
|
417
|
+
return 3
|
|
418
|
+
|
|
419
|
+
if as_json:
|
|
420
|
+
# `diagnostic` passes the server payload through VERBATIM and stays
|
|
421
|
+
# explicitly opaque, consistent with what `bin/_lib_perf.py` promises
|
|
422
|
+
# about phase names. The stamped wrapper is the stable part.
|
|
423
|
+
print(encode_dashboard_json(c.stamp_schema_version(
|
|
424
|
+
{"status": "ok", "diagnostic": payload})))
|
|
425
|
+
return 0
|
|
426
|
+
|
|
427
|
+
if trace_result is not None:
|
|
428
|
+
print(f"dashboard-perf: trace {trace} requested "
|
|
429
|
+
f"(requested={trace_result.get('requested')}, "
|
|
430
|
+
f"applied={trace_result.get('applied')}, "
|
|
431
|
+
f"applies_at={trace_result.get('applies_at')})")
|
|
432
|
+
sys.stdout.write(render_dashboard_perf(payload))
|
|
433
|
+
return 0
|