cctally 1.99.1 → 1.100.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/bin/_cctally_dashboard.py +1370 -190
- package/bin/_cctally_dashboard_envelope.py +98 -4
- package/bin/_cctally_dashboard_perf.py +433 -0
- package/bin/_cctally_dashboard_sources.py +441 -54
- package/bin/_cctally_db.py +30 -14
- package/bin/_cctally_doctor.py +1331 -1148
- package/bin/_cctally_parser.py +70 -1
- package/bin/_cctally_quota.py +40 -21
- package/bin/_cctally_record.py +37 -2
- package/bin/_cctally_refresh.py +60 -10
- package/bin/_cctally_statusline.py +53 -7
- package/bin/_cctally_tui.py +328 -28
- package/bin/_cctally_update.py +28 -22
- package/bin/_lib_dashboard_sources.py +26 -7
- package/bin/_lib_doctor.py +37 -0
- package/bin/_lib_jsonl.py +4 -2
- package/bin/_lib_perf.py +132 -3
- package/bin/_lib_source_analytics.py +2 -2
- package/bin/_lib_tick_stats.py +538 -0
- package/bin/cctally +19 -7
- package/dashboard/static/assets/index-B5YfQEtn.css +1 -0
- package/dashboard/static/assets/index-Bt59nMMO.js +97 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +3 -1
- package/dashboard/static/assets/index-C5NBB2w9.js +0 -97
- package/dashboard/static/assets/index-hJP4wlIO.css +0 -1
|
@@ -0,0 +1,538 @@
|
|
|
1
|
+
"""Always-on record of what a dashboard tick cost (issue #583, S1 spec §1).
|
|
2
|
+
|
|
3
|
+
Stdlib-only leaf module. It imports nothing from cctally, so
|
|
4
|
+
``bin/_cctally_tui.py``, ``bin/_cctally_dashboard.py``,
|
|
5
|
+
``bin/cctally-snapshot-measure`` and the ``dashboard-perf`` reader can all
|
|
6
|
+
reach it without a back-import into the dashboard module and without putting
|
|
7
|
+
ownership of the record inside ``bin/_lib_snapshot_cache.py``.
|
|
8
|
+
|
|
9
|
+
This is deliberately NOT an extension of ``bin/_lib_perf.py``. That module is
|
|
10
|
+
an opt-in diagnostic with a thread-local phase stack, unstable names and a
|
|
11
|
+
lifecycle that begins and ends inside one build. This one is always on, keeps
|
|
12
|
+
cross-thread counters, and survives across builds.
|
|
13
|
+
|
|
14
|
+
Concurrency (spec §1.1). Every update takes the single module-level lock,
|
|
15
|
+
constructs the COMPLETE replacement ``StatsSnapshot`` under it, and rebinds
|
|
16
|
+
``_STATE`` once. Readers return the current ``_STATE`` and take no lock. The
|
|
17
|
+
bare-rebind discipline of ``_lib_perf._LAST_BACKEND_PERF`` is not reused here:
|
|
18
|
+
that slot is written by whole replacement, so GIL atomicity is enough, while a
|
|
19
|
+
counter update is read-modify-write and would lose increments under the same
|
|
20
|
+
pattern.
|
|
21
|
+
|
|
22
|
+
Bounds: at most ``RING_CAPACITY`` retained records and at most
|
|
23
|
+
``MEMORY_BUDGET_BYTES`` of owned state. Both are asserted by
|
|
24
|
+
``tests/test_tick_stats.py``.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import dataclasses
|
|
29
|
+
import sys
|
|
30
|
+
import threading
|
|
31
|
+
import time
|
|
32
|
+
import types
|
|
33
|
+
|
|
34
|
+
RING_CAPACITY = 64
|
|
35
|
+
MEMORY_BUDGET_BYTES = 65536
|
|
36
|
+
|
|
37
|
+
#: `published_at` is the only variable-length field in a record, so it is the
|
|
38
|
+
#: only one that can put the whole state over `MEMORY_BUDGET_BYTES`. Measured:
|
|
39
|
+
#: the stored field set is 27,189 bytes over a full ring, and 64 distinct 4 KiB
|
|
40
|
+
#: strings would be 283,039 — 4.3x over budget. All three production callers
|
|
41
|
+
#: pass a UTC ISO-8601 instant (32 characters), so this is not reachable today;
|
|
42
|
+
#: the cap makes the budget hold by CONSTRUCTION rather than by convention.
|
|
43
|
+
PUBLISHED_AT_MAX_CHARS = 64
|
|
44
|
+
|
|
45
|
+
#: The three mutually exclusive dispatch outcomes of one outer refresh (§1.4).
|
|
46
|
+
DISPATCH_KINDS = ("idle", "full", "degraded")
|
|
47
|
+
#: The three Group A bucket builders whose cache opens can fail silently (§1.6).
|
|
48
|
+
CACHE_OPEN_FAILURE_KINDS = ("daily", "weekly", "monthly")
|
|
49
|
+
#: The realised Codex source-leg regime, aggregated over the refresh (§1.5).
|
|
50
|
+
CODEX_REGIMES = ("active", "idle", "not_observed")
|
|
51
|
+
#: What the tick published. Metadata, never a dispatch category (§1.4).
|
|
52
|
+
PUBLICATIONS = ("final", "partial", "seed", "degraded")
|
|
53
|
+
|
|
54
|
+
#: The conversation sync loop's three outcomes (#583 S4 §6). A CLOSED set: the
|
|
55
|
+
#: recorder normalizes to these, so no error text, path, or other
|
|
56
|
+
#: caller-supplied string can reach module state or the debug endpoint. The
|
|
57
|
+
#: validation lives in the recorder rather than in the dataclass because a
|
|
58
|
+
#: frozen field typed `str` enforces nothing on its own.
|
|
59
|
+
CONVERSATION_STATUSES = ("ok", "store_unavailable", "error")
|
|
60
|
+
|
|
61
|
+
_INGEST = "ingest"
|
|
62
|
+
_BUILD = "build"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
66
|
+
class TickRecord:
|
|
67
|
+
"""One completed tick. Fixed-size scalars and enum strings only — these
|
|
68
|
+
field names are also the wire names on ``/api/debug/backend`` (§3.1)."""
|
|
69
|
+
|
|
70
|
+
seq: int
|
|
71
|
+
started_ns: int
|
|
72
|
+
ended_ns: int
|
|
73
|
+
duration_ns: int
|
|
74
|
+
ingest_ran: bool
|
|
75
|
+
ingest_ns: int
|
|
76
|
+
builder_ns: int
|
|
77
|
+
dispatch: str
|
|
78
|
+
codex_regime: str
|
|
79
|
+
publication: str
|
|
80
|
+
cold: bool
|
|
81
|
+
published_ns: int
|
|
82
|
+
published_at: str
|
|
83
|
+
period_ns: "int | None"
|
|
84
|
+
cache_pin_ns: int
|
|
85
|
+
|
|
86
|
+
def as_wire(self) -> dict:
|
|
87
|
+
return {f.name: getattr(self, f.name) for f in dataclasses.fields(self)}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
91
|
+
class ConversationSyncRecord:
|
|
92
|
+
"""One completed conversation sync pass (#583 S4).
|
|
93
|
+
|
|
94
|
+
Fixed-size scalars and one validated enum string; these field names are
|
|
95
|
+
also the wire names on ``/api/debug/backend``. A SECOND ring rather than a
|
|
96
|
+
mixed record kind: mixing conversation passes into ``records`` would evict
|
|
97
|
+
main-tick samples and corrupt every aggregate computed over them.
|
|
98
|
+
|
|
99
|
+
``period_ns`` is the FORWARD interval, ``start[i+1] - start[i]``: the gap
|
|
100
|
+
from this pass's own start to the next pass's start. It is therefore
|
|
101
|
+
``None`` on the newest record until the following pass records, which
|
|
102
|
+
stamps it. Pairing a pass's CPU with the interval that PRECEDED it instead
|
|
103
|
+
shifts the denominator by one pass, and that ratio has no upper bound — a
|
|
104
|
+
long pass following short ones is charged against a short interval and the
|
|
105
|
+
published share exceeds 100%, and exceeds the 50% ceiling the loop's duty
|
|
106
|
+
bound guarantees.
|
|
107
|
+
|
|
108
|
+
``TickRecord.period_ns`` is the same name for the OPPOSITE convention: it
|
|
109
|
+
is the BACKWARD publish interval, so the FIRST tick record carries ``None``
|
|
110
|
+
while here the NEWEST conversation record does. Both ride one
|
|
111
|
+
``/api/debug/backend`` response, so do not transplant a summarizer between
|
|
112
|
+
the two rings without re-deriving which end is null.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
seq: int
|
|
116
|
+
started_ns: int
|
|
117
|
+
ended_ns: int
|
|
118
|
+
duration_ns: int
|
|
119
|
+
cpu_ns: int
|
|
120
|
+
period_ns: "int | None"
|
|
121
|
+
status: str
|
|
122
|
+
|
|
123
|
+
def as_wire(self) -> dict:
|
|
124
|
+
return {f.name: getattr(self, f.name) for f in dataclasses.fields(self)}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
128
|
+
class StatsSnapshot:
|
|
129
|
+
"""An immutable whole-state read. Rebound as one object, never mutated."""
|
|
130
|
+
|
|
131
|
+
dispatch_counts: "types.MappingProxyType[str, int]"
|
|
132
|
+
cache_open_failures: "types.MappingProxyType[str, int]"
|
|
133
|
+
tick_seq: int
|
|
134
|
+
records: "tuple[TickRecord, ...]"
|
|
135
|
+
standalone: "TickRecord | None"
|
|
136
|
+
conversation_records: "tuple[ConversationSyncRecord, ...]" = ()
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _frozen_counts(kinds, source=None) -> "types.MappingProxyType[str, int]":
|
|
140
|
+
base = {k: (source[k] if source else 0) for k in kinds}
|
|
141
|
+
return types.MappingProxyType(base)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
_EMPTY = StatsSnapshot(
|
|
145
|
+
dispatch_counts=_frozen_counts(DISPATCH_KINDS),
|
|
146
|
+
cache_open_failures=_frozen_counts(CACHE_OPEN_FAILURE_KINDS),
|
|
147
|
+
tick_seq=0,
|
|
148
|
+
records=(),
|
|
149
|
+
standalone=None,
|
|
150
|
+
conversation_records=(),
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
_LOCK = threading.Lock()
|
|
154
|
+
_STATE = _EMPTY
|
|
155
|
+
_tls = threading.local()
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def snapshot() -> StatsSnapshot:
|
|
159
|
+
"""The current whole state. Lock-free: ``_STATE`` is only ever rebound."""
|
|
160
|
+
return _STATE
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def reset_for_tests() -> None:
|
|
164
|
+
global _STATE
|
|
165
|
+
with _LOCK:
|
|
166
|
+
_STATE = _EMPTY
|
|
167
|
+
_tls.tick = None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def current() -> "TickContext | None":
|
|
171
|
+
"""The tick context open on THIS thread, or None.
|
|
172
|
+
|
|
173
|
+
``_tui_build_snapshot`` consults this to decide whether to open a
|
|
174
|
+
standalone context: inside a dashboard refresh it must not, or an A2
|
|
175
|
+
partial build would be recorded as a second tick (§1.2).
|
|
176
|
+
"""
|
|
177
|
+
return getattr(_tls, "tick", None)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def note_cache_open_failure(kind: str) -> None:
|
|
181
|
+
"""Count one silent Group A cache-open failure (§1.6)."""
|
|
182
|
+
if kind not in CACHE_OPEN_FAILURE_KINDS:
|
|
183
|
+
raise ValueError(f"unknown cache-open failure kind: {kind!r}")
|
|
184
|
+
global _STATE
|
|
185
|
+
with _LOCK:
|
|
186
|
+
prior = _STATE
|
|
187
|
+
counts = dict(prior.cache_open_failures)
|
|
188
|
+
counts[kind] += 1
|
|
189
|
+
_STATE = dataclasses.replace(
|
|
190
|
+
prior,
|
|
191
|
+
cache_open_failures=types.MappingProxyType(counts),
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class _Span:
|
|
196
|
+
"""One measured region inside a tick. Nesting-aware in both directions."""
|
|
197
|
+
|
|
198
|
+
__slots__ = ("_tick", "_kind", "_start", "child_ingest", "child_build")
|
|
199
|
+
|
|
200
|
+
def __init__(self, tick: "TickContext", kind: str):
|
|
201
|
+
self._tick = tick
|
|
202
|
+
self._kind = kind
|
|
203
|
+
self._start = 0
|
|
204
|
+
self.child_ingest = 0
|
|
205
|
+
self.child_build = 0
|
|
206
|
+
|
|
207
|
+
def __enter__(self):
|
|
208
|
+
self._start = self._tick._now()
|
|
209
|
+
self._tick._spans.append(self)
|
|
210
|
+
return self
|
|
211
|
+
|
|
212
|
+
def __exit__(self, *exc):
|
|
213
|
+
elapsed = self._tick._now() - self._start
|
|
214
|
+
spans = self._tick._spans
|
|
215
|
+
# Identity-aware unwind, so a span whose __exit__ was skipped cannot
|
|
216
|
+
# strand this one on the stack.
|
|
217
|
+
if self in spans:
|
|
218
|
+
while spans and spans[-1] is not self:
|
|
219
|
+
spans.pop()
|
|
220
|
+
spans.pop()
|
|
221
|
+
own = elapsed - self.child_ingest - self.child_build
|
|
222
|
+
if own < 0:
|
|
223
|
+
own = 0
|
|
224
|
+
self._tick._add(self._kind, own)
|
|
225
|
+
up_ingest = self.child_ingest + (own if self._kind == _INGEST else 0)
|
|
226
|
+
up_build = self.child_build + (own if self._kind == _BUILD else 0)
|
|
227
|
+
if spans:
|
|
228
|
+
spans[-1].child_ingest += up_ingest
|
|
229
|
+
spans[-1].child_build += up_build
|
|
230
|
+
return False
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
class TickContext:
|
|
234
|
+
"""A tick in progress. Thread-confined; only ``finish`` touches the lock."""
|
|
235
|
+
|
|
236
|
+
__slots__ = ("_now", "_standalone", "_started_ns", "_spans", "_finished",
|
|
237
|
+
"_ingest_ns", "_builder_ns", "_ingest_ran", "_dispatch",
|
|
238
|
+
"_codex_regime", "_publication", "_cold", "_cache_pin_ns")
|
|
239
|
+
|
|
240
|
+
def __init__(self, *, monotonic_ns, standalone: bool):
|
|
241
|
+
self._now = monotonic_ns
|
|
242
|
+
self._standalone = standalone
|
|
243
|
+
self._started_ns = monotonic_ns()
|
|
244
|
+
self._spans: list[_Span] = []
|
|
245
|
+
self._finished = False
|
|
246
|
+
self._ingest_ns = 0
|
|
247
|
+
self._builder_ns = 0
|
|
248
|
+
self._ingest_ran = False
|
|
249
|
+
# An unset dispatch is the degraded case by construction: a crash or a
|
|
250
|
+
# deferral publishes without ever reaching a dispatch decision, and
|
|
251
|
+
# leaving it unclassified would break `idle + full + degraded ==
|
|
252
|
+
# tick_seq`, which is what the operator reads as "how many ticks ran".
|
|
253
|
+
self._dispatch = "degraded"
|
|
254
|
+
self._codex_regime = "not_observed"
|
|
255
|
+
self._publication = "final"
|
|
256
|
+
self._cold = False
|
|
257
|
+
self._cache_pin_ns = 0
|
|
258
|
+
|
|
259
|
+
# ── measurement ──────────────────────────────────────────────────────
|
|
260
|
+
|
|
261
|
+
def ingest_span(self) -> _Span:
|
|
262
|
+
"""Measure an ingest region. Also stamps ``ingest_ran``."""
|
|
263
|
+
self._ingest_ran = True
|
|
264
|
+
return _Span(self, _INGEST)
|
|
265
|
+
|
|
266
|
+
def build_span(self) -> _Span:
|
|
267
|
+
"""Measure a builder region."""
|
|
268
|
+
return _Span(self, _BUILD)
|
|
269
|
+
|
|
270
|
+
def mark_ingest(self, ns: int) -> None:
|
|
271
|
+
"""Attribute already-measured time to ingest. Stamps ``ingest_ran``.
|
|
272
|
+
|
|
273
|
+
For a caller that measured the region itself rather than opening one
|
|
274
|
+
of the two built-in spans — no production seam does today, and both
|
|
275
|
+
are kept because the record must be writable from outside them.
|
|
276
|
+
|
|
277
|
+
A caller that reports ingest time is reporting that ingest ran, and
|
|
278
|
+
`finish` zeroes `ingest_ns` when the flag is false — so leaving the
|
|
279
|
+
flag to the span form alone would silently discard the figure.
|
|
280
|
+
"""
|
|
281
|
+
self._ingest_ran = True
|
|
282
|
+
self._add(_INGEST, int(ns))
|
|
283
|
+
|
|
284
|
+
def mark_build(self, ns: int) -> None:
|
|
285
|
+
"""Attribute already-measured time to the builder.
|
|
286
|
+
|
|
287
|
+
The `mark_ingest` note applies: this exists for a caller outside the
|
|
288
|
+
two built-in spans, not because a production seam uses it.
|
|
289
|
+
"""
|
|
290
|
+
self._add(_BUILD, int(ns))
|
|
291
|
+
|
|
292
|
+
def mark_cache_pin(self, ns: int) -> None:
|
|
293
|
+
"""Attribute a held cache.db read transaction to this tick (#583 S5).
|
|
294
|
+
|
|
295
|
+
Measured at the `BEGIN` and `ROLLBACK` boundaries in
|
|
296
|
+
`_tui_build_source_bundle`, so it is the HOLD itself rather than that
|
|
297
|
+
function's cumulative duration. The two differ: the duration also
|
|
298
|
+
counts the work before `BEGIN` and after `ROLLBACK`, which makes it an
|
|
299
|
+
upper bound on the hold and not the hold. Spec §2.4 forbids quoting a
|
|
300
|
+
hold figure sourced from the duration.
|
|
301
|
+
|
|
302
|
+
ACCUMULATED, not last-write, for the reason `set_codex_regime`
|
|
303
|
+
documents: A2 can run several builds inside one refresh, each opening
|
|
304
|
+
its own pin, and reporting only the last one would understate a
|
|
305
|
+
refresh that pinned twice. It is therefore a sum of holds within the
|
|
306
|
+
tick and not the longest single hold; a tick that pinned once, which
|
|
307
|
+
is every non-A2 tick, reports that one hold exactly.
|
|
308
|
+
|
|
309
|
+
Negative and non-integer inputs are clamped to zero rather than
|
|
310
|
+
raising, because this runs on the publish path and a diagnostic must
|
|
311
|
+
never take down the tick it is describing.
|
|
312
|
+
"""
|
|
313
|
+
try:
|
|
314
|
+
value = int(ns)
|
|
315
|
+
except (TypeError, ValueError):
|
|
316
|
+
return
|
|
317
|
+
self._cache_pin_ns += max(0, value)
|
|
318
|
+
|
|
319
|
+
def _add(self, kind: str, ns: int) -> None:
|
|
320
|
+
if kind == _INGEST:
|
|
321
|
+
self._ingest_ns += ns
|
|
322
|
+
else:
|
|
323
|
+
self._builder_ns += ns
|
|
324
|
+
|
|
325
|
+
# ── classification, aggregated over the whole refresh ────────────────
|
|
326
|
+
|
|
327
|
+
def set_dispatch(self, value: str) -> None:
|
|
328
|
+
"""`full` wins over `idle`, which wins over `degraded` (§1.4).
|
|
329
|
+
|
|
330
|
+
Aggregated, not last-write: A2 can run several builds inside one
|
|
331
|
+
refresh and they may disagree, and an expensive full build followed by
|
|
332
|
+
an idle one is a full tick.
|
|
333
|
+
"""
|
|
334
|
+
if value not in DISPATCH_KINDS:
|
|
335
|
+
raise ValueError(f"unknown dispatch: {value!r}")
|
|
336
|
+
rank = {"degraded": 0, "idle": 1, "full": 2}
|
|
337
|
+
if rank[value] > rank[self._dispatch]:
|
|
338
|
+
self._dispatch = value
|
|
339
|
+
|
|
340
|
+
def mark_degraded(self) -> None:
|
|
341
|
+
"""Force the degraded classification, overriding a completed build.
|
|
342
|
+
|
|
343
|
+
Precedence cannot express this: a refresh whose A2 partial built fine
|
|
344
|
+
and whose FINAL build then crashed produced no usable final build, so
|
|
345
|
+
it is degraded even though a build completed (§1.4).
|
|
346
|
+
"""
|
|
347
|
+
self._dispatch = "degraded"
|
|
348
|
+
self._publication = "degraded"
|
|
349
|
+
|
|
350
|
+
def set_codex_regime(self, value: str) -> None:
|
|
351
|
+
"""`active` wins over `idle`, which wins over `not_observed` (§1.5).
|
|
352
|
+
|
|
353
|
+
Last-write classification would move an expensive tick into the idle
|
|
354
|
+
population once a partial build has populated the reuse memo, which is
|
|
355
|
+
the F40 distortion this classifier exists to prevent.
|
|
356
|
+
"""
|
|
357
|
+
if value not in CODEX_REGIMES:
|
|
358
|
+
raise ValueError(f"unknown codex regime: {value!r}")
|
|
359
|
+
rank = {"not_observed": 0, "idle": 1, "active": 2}
|
|
360
|
+
if rank[value] > rank[self._codex_regime]:
|
|
361
|
+
self._codex_regime = value
|
|
362
|
+
|
|
363
|
+
def set_publication(self, value: str) -> None:
|
|
364
|
+
if value not in PUBLICATIONS:
|
|
365
|
+
raise ValueError(f"unknown publication: {value!r}")
|
|
366
|
+
self._publication = value
|
|
367
|
+
|
|
368
|
+
def set_cold(self, value: bool) -> None:
|
|
369
|
+
"""Sticky: a refresh containing one cold build is a cold tick."""
|
|
370
|
+
self._cold = self._cold or bool(value)
|
|
371
|
+
|
|
372
|
+
@property
|
|
373
|
+
def finished(self) -> bool:
|
|
374
|
+
return self._finished
|
|
375
|
+
|
|
376
|
+
# ── close ────────────────────────────────────────────────────────────
|
|
377
|
+
|
|
378
|
+
def finish(self, *, published_ns: int, published_at: str) -> None:
|
|
379
|
+
"""Freeze this tick into the ring. Idempotent; a second call is a
|
|
380
|
+
no-op, so a crash handler may close a tick the happy path already
|
|
381
|
+
closed."""
|
|
382
|
+
if self._finished:
|
|
383
|
+
return
|
|
384
|
+
self._finished = True
|
|
385
|
+
# Truncate rather than reject: this runs on the publish path, and a
|
|
386
|
+
# diagnostic must never take down the tick it is describing.
|
|
387
|
+
published_at = str(published_at)[:PUBLISHED_AT_MAX_CHARS]
|
|
388
|
+
while self._spans:
|
|
389
|
+
self._spans.pop().__exit__(None, None, None)
|
|
390
|
+
ended_ns = self._now()
|
|
391
|
+
if getattr(_tls, "tick", None) is self:
|
|
392
|
+
_tls.tick = None
|
|
393
|
+
|
|
394
|
+
global _STATE
|
|
395
|
+
with _LOCK:
|
|
396
|
+
prior = _STATE
|
|
397
|
+
if self._standalone:
|
|
398
|
+
seq = prior.tick_seq
|
|
399
|
+
period = None
|
|
400
|
+
else:
|
|
401
|
+
seq = prior.tick_seq + 1
|
|
402
|
+
period = (
|
|
403
|
+
published_ns - prior.records[-1].published_ns
|
|
404
|
+
if prior.records else None
|
|
405
|
+
)
|
|
406
|
+
record = TickRecord(
|
|
407
|
+
seq=seq,
|
|
408
|
+
started_ns=self._started_ns,
|
|
409
|
+
ended_ns=ended_ns,
|
|
410
|
+
duration_ns=max(0, ended_ns - self._started_ns),
|
|
411
|
+
ingest_ran=self._ingest_ran,
|
|
412
|
+
ingest_ns=self._ingest_ns if self._ingest_ran else 0,
|
|
413
|
+
builder_ns=self._builder_ns,
|
|
414
|
+
dispatch=self._dispatch,
|
|
415
|
+
codex_regime=self._codex_regime,
|
|
416
|
+
publication=self._publication,
|
|
417
|
+
cold=self._cold,
|
|
418
|
+
published_ns=int(published_ns),
|
|
419
|
+
published_at=published_at,
|
|
420
|
+
period_ns=period,
|
|
421
|
+
cache_pin_ns=self._cache_pin_ns,
|
|
422
|
+
)
|
|
423
|
+
if self._standalone:
|
|
424
|
+
_STATE = dataclasses.replace(prior, standalone=record)
|
|
425
|
+
return
|
|
426
|
+
counts = dict(prior.dispatch_counts)
|
|
427
|
+
counts[self._dispatch] += 1
|
|
428
|
+
_STATE = StatsSnapshot(
|
|
429
|
+
dispatch_counts=types.MappingProxyType(counts),
|
|
430
|
+
cache_open_failures=prior.cache_open_failures,
|
|
431
|
+
tick_seq=seq,
|
|
432
|
+
records=(prior.records + (record,))[-RING_CAPACITY:],
|
|
433
|
+
standalone=prior.standalone,
|
|
434
|
+
# Carried forward BY HAND: this is the one mutator that rebuilds
|
|
435
|
+
# the snapshot field by field instead of using
|
|
436
|
+
# `dataclasses.replace`, so dropping this line would let every
|
|
437
|
+
# refresh tick wipe the conversation ring in production while
|
|
438
|
+
# the whole suite stayed green. Regression:
|
|
439
|
+
# `tests/test_tick_stats.py`.
|
|
440
|
+
conversation_records=prior.conversation_records,
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def record_conversation_pass(
|
|
445
|
+
*, seq, started_ns, ended_ns, duration_ns, cpu_ns, status
|
|
446
|
+
) -> None:
|
|
447
|
+
"""Append one conversation sync pass to its ring, under the SHARED lock.
|
|
448
|
+
|
|
449
|
+
No second lock and no second state object: the module keeps one `_LOCK` and
|
|
450
|
+
one immutable whole-state replacement (#583 S4 §6).
|
|
451
|
+
|
|
452
|
+
`status` is normalized against `CONVERSATION_STATUSES` HERE. That is what
|
|
453
|
+
makes the no-leak constraint true — validating in the recorder rather than
|
|
454
|
+
trusting the dataclass is the point, because a frozen field typed `str`
|
|
455
|
+
would accept `str(exc)` or a filesystem path and publish it through the
|
|
456
|
+
debug endpoint.
|
|
457
|
+
|
|
458
|
+
This call also FINALIZES the previous record's `period_ns`, because that
|
|
459
|
+
field is the FORWARD interval `start[i+1] - start[i]` and this pass's start
|
|
460
|
+
is what closes it. The previous entry is replaced in place with
|
|
461
|
+
`dataclasses.replace` under the same lock; the record appended here carries
|
|
462
|
+
`period_ns=None` until its own successor arrives.
|
|
463
|
+
|
|
464
|
+
Storing the BACKWARD interval instead — the gap that preceded the pass —
|
|
465
|
+
would pair each pass's CPU with someone else's interval, and
|
|
466
|
+
`cpu[i] / (start[i] - start[i-1])` has no upper bound: a long pass after
|
|
467
|
+
short ones renders above 100%, well past the 50% ceiling the loop's duty
|
|
468
|
+
bound guarantees.
|
|
469
|
+
"""
|
|
470
|
+
global _STATE
|
|
471
|
+
safe_status = status if status in CONVERSATION_STATUSES else "error"
|
|
472
|
+
start = int(started_ns)
|
|
473
|
+
record = ConversationSyncRecord(
|
|
474
|
+
seq=int(seq),
|
|
475
|
+
started_ns=start,
|
|
476
|
+
ended_ns=int(ended_ns),
|
|
477
|
+
duration_ns=max(0, int(duration_ns)),
|
|
478
|
+
cpu_ns=max(0, int(cpu_ns)),
|
|
479
|
+
period_ns=None,
|
|
480
|
+
status=safe_status,
|
|
481
|
+
)
|
|
482
|
+
with _LOCK:
|
|
483
|
+
prior = _STATE
|
|
484
|
+
retained = prior.conversation_records
|
|
485
|
+
if retained:
|
|
486
|
+
previous = retained[-1]
|
|
487
|
+
retained = retained[:-1] + (
|
|
488
|
+
dataclasses.replace(
|
|
489
|
+
previous, period_ns=start - previous.started_ns,
|
|
490
|
+
),
|
|
491
|
+
)
|
|
492
|
+
_STATE = dataclasses.replace(
|
|
493
|
+
prior,
|
|
494
|
+
conversation_records=(retained + (record,))[-RING_CAPACITY:],
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def begin_tick(*, monotonic_ns=None, standalone: bool = False) -> TickContext:
|
|
499
|
+
"""Open a tick context and make it ``current()`` on this thread.
|
|
500
|
+
|
|
501
|
+
``standalone`` records a ``tui --render-once`` or
|
|
502
|
+
``cctally-snapshot-measure`` build: it fills the single standalone slot
|
|
503
|
+
instead of the ring, so it neither advances ``tick_seq`` nor lands in a
|
|
504
|
+
dispatch count (§1.2).
|
|
505
|
+
"""
|
|
506
|
+
ctx = TickContext(
|
|
507
|
+
monotonic_ns=monotonic_ns or time.monotonic_ns,
|
|
508
|
+
standalone=standalone,
|
|
509
|
+
)
|
|
510
|
+
_tls.tick = ctx
|
|
511
|
+
return ctx
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _deep_size(obj, _seen=None) -> int:
|
|
515
|
+
"""Recursive ``sys.getsizeof`` over the owned state. Test-only.
|
|
516
|
+
|
|
517
|
+
Each distinct object is counted once, so the interned enum strings the
|
|
518
|
+
records share are not multiplied by the ring length.
|
|
519
|
+
"""
|
|
520
|
+
if _seen is None:
|
|
521
|
+
_seen = set()
|
|
522
|
+
if id(obj) in _seen:
|
|
523
|
+
return 0
|
|
524
|
+
_seen.add(id(obj))
|
|
525
|
+
total = sys.getsizeof(obj)
|
|
526
|
+
if isinstance(obj, (str, bytes, int, float, bool, type(None))):
|
|
527
|
+
return total
|
|
528
|
+
if isinstance(obj, (dict, types.MappingProxyType)):
|
|
529
|
+
for key, value in obj.items():
|
|
530
|
+
total += _deep_size(key, _seen) + _deep_size(value, _seen)
|
|
531
|
+
return total
|
|
532
|
+
if isinstance(obj, (tuple, list, set, frozenset)):
|
|
533
|
+
for item in obj:
|
|
534
|
+
total += _deep_size(item, _seen)
|
|
535
|
+
return total
|
|
536
|
+
for field in getattr(type(obj), "__slots__", ()):
|
|
537
|
+
total += _deep_size(getattr(obj, field, None), _seen)
|
|
538
|
+
return total
|
package/bin/cctally
CHANGED
|
@@ -947,6 +947,7 @@ _bust_statusline_cache = _cctally_refresh._bust_statusline_cache
|
|
|
947
947
|
_freshness_label = _cctally_refresh._freshness_label
|
|
948
948
|
_cmd_refresh_usage_handle_rate_limit = _cctally_refresh._cmd_refresh_usage_handle_rate_limit
|
|
949
949
|
_refresh_usage_inproc = _cctally_refresh._refresh_usage_inproc
|
|
950
|
+
_resolve_dashboard_api_token = _cctally_refresh._resolve_dashboard_api_token
|
|
950
951
|
_nudge_dashboard_repaint = _cctally_refresh._nudge_dashboard_repaint
|
|
951
952
|
cmd_refresh_usage = _cctally_refresh.cmd_refresh_usage
|
|
952
953
|
_hook_tick_oauth_refresh = _cctally_refresh._hook_tick_oauth_refresh
|
|
@@ -1324,12 +1325,9 @@ cmd_telemetry = _cctally_telemetry.cmd_telemetry
|
|
|
1324
1325
|
# ``c = _cctally()`` accessor or module-level shim at call time so
|
|
1325
1326
|
# ``setitem(ns, "_dashboard_build_weekly_periods", spy)`` propagates
|
|
1326
1327
|
# into ``_share_apply_period_override`` / ``DashboardHTTPHandler``
|
|
1327
|
-
# request paths. Path constants (``STATIC_DIR
|
|
1328
|
-
# ``
|
|
1329
|
-
# ``
|
|
1330
|
-
# ``tests/test_dashboard_api_sync_refresh.py`` and the
|
|
1331
|
-
# ``monkeypatch.setitem(ns, "_DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS", …)``
|
|
1332
|
-
# sites propagate transparently.
|
|
1328
|
+
# request paths. Path constants (``STATIC_DIR``) stay re-exported here
|
|
1329
|
+
# so ``setitem(ns, "STATIC_DIR", tmp)`` in
|
|
1330
|
+
# ``tests/test_dashboard_api_sync_refresh.py`` propagates transparently.
|
|
1333
1331
|
_cctally_dashboard = _load_sibling("_cctally_dashboard")
|
|
1334
1332
|
_DASHBOARD_BIND_SEMANTIC_ALIASES = _cctally_dashboard._DASHBOARD_BIND_SEMANTIC_ALIASES
|
|
1335
1333
|
_validate_dashboard_bind_value = _cctally_dashboard._validate_dashboard_bind_value
|
|
@@ -1358,7 +1356,22 @@ _build_blocks_share_panel_data = _cctally_dashboard._build_blocks_share_panel_da
|
|
|
1358
1356
|
_build_sessions_share_panel_data = _cctally_dashboard._build_sessions_share_panel_data
|
|
1359
1357
|
_build_projects_share_panel_data = _cctally_dashboard._build_projects_share_panel_data
|
|
1360
1358
|
_SnapshotRef = _cctally_dashboard._SnapshotRef
|
|
1359
|
+
_resolve_dashboard_port = _cctally_dashboard._resolve_dashboard_port
|
|
1360
|
+
|
|
1361
|
+
# Eager re-export of bin/_cctally_dashboard_perf.py (#583 S1 §3.3) —
|
|
1362
|
+
# `cmd_dashboard_perf` is the parser's `set_defaults(func=c.cmd_dashboard_perf)`
|
|
1363
|
+
# target. It loads AFTER `_cctally_dashboard` because it resolves the port
|
|
1364
|
+
# default through `_resolve_dashboard_port`, re-exported just above.
|
|
1365
|
+
_cctally_dashboard_perf = _load_sibling("_cctally_dashboard_perf")
|
|
1366
|
+
cmd_dashboard_perf = _cctally_dashboard_perf.cmd_dashboard_perf
|
|
1361
1367
|
SSEHub = _cctally_dashboard.SSEHub
|
|
1368
|
+
# #583 S3 §5 — the per-tick projection cache the hub now fans out, plus the two
|
|
1369
|
+
# helpers its consumer needs. Re-exported so the tests can drive them directly
|
|
1370
|
+
# rather than only through a booted HTTP server.
|
|
1371
|
+
_SSEDelivery = _cctally_dashboard._SSEDelivery
|
|
1372
|
+
_drain_to_newest = _cctally_dashboard._drain_to_newest
|
|
1373
|
+
_accepts_gzip = _cctally_dashboard._accepts_gzip
|
|
1374
|
+
_canonical_oauth_key = _cctally_dashboard._canonical_oauth_key
|
|
1362
1375
|
STATIC_DIR = _cctally_dashboard.STATIC_DIR
|
|
1363
1376
|
_format_url = _cctally_dashboard._format_url
|
|
1364
1377
|
_dashboard_auth_url = _cctally_dashboard._dashboard_auth_url
|
|
@@ -1410,7 +1423,6 @@ _select_current_block_for_envelope = _cctally_dashboard._select_current_block_fo
|
|
|
1410
1423
|
_build_alerts_envelope_array = _cctally_dashboard._build_alerts_envelope_array
|
|
1411
1424
|
snapshot_to_envelope = _cctally_dashboard.snapshot_to_envelope
|
|
1412
1425
|
_session_detail_to_envelope = _cctally_dashboard._session_detail_to_envelope
|
|
1413
|
-
_DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS = _cctally_dashboard._DASHBOARD_SYNC_LOCK_TIMEOUT_SECONDS
|
|
1414
1426
|
DashboardHTTPHandler = _cctally_dashboard.DashboardHTTPHandler
|
|
1415
1427
|
_dashboard_self_heal_orphans = _cctally_dashboard._dashboard_self_heal_orphans
|
|
1416
1428
|
_register_faulthandler_sigusr1 = _cctally_dashboard._register_faulthandler_sigusr1
|