cctally 1.100.0 → 1.102.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -0
- package/README.md +8 -2
- package/bin/_cctally_alerts.py +13 -2
- package/bin/_cctally_cache.py +3 -1
- package/bin/_cctally_cache_report.py +103 -6
- package/bin/_cctally_dashboard.py +1140 -282
- package/bin/_cctally_dashboard_conversation.py +12 -0
- package/bin/_cctally_dashboard_envelope.py +53 -61
- package/bin/_cctally_dashboard_share.py +101 -29
- package/bin/_cctally_dashboard_sources.py +663 -192
- package/bin/_cctally_diagnosis.py +1172 -0
- package/bin/_cctally_diagnosis_sources.py +4054 -0
- package/bin/_cctally_diff.py +20 -0
- package/bin/_cctally_forecast.py +329 -111
- package/bin/_cctally_milestone_history.py +10 -2
- package/bin/_cctally_parser.py +84 -0
- package/bin/_cctally_project.py +155 -47
- package/bin/_cctally_quota.py +14 -0
- package/bin/_cctally_record.py +151 -71
- package/bin/_cctally_refresh.py +105 -93
- package/bin/_cctally_share.py +9 -2
- package/bin/_cctally_source_analytics.py +40 -4
- package/bin/_cctally_statusline.py +8 -1
- package/bin/_cctally_tui.py +425 -234
- package/bin/_lib_alert_scope.py +685 -0
- package/bin/_lib_alerts_payload.py +112 -7
- package/bin/_lib_blocks.py +12 -0
- package/bin/_lib_cache_report.py +110 -1
- package/bin/_lib_codex_conversation.py +14 -0
- package/bin/_lib_codex_conversation_query.py +22 -8
- package/bin/_lib_codex_pools.py +20 -8
- package/bin/_lib_conversation.py +6 -3
- package/bin/_lib_conversation_query.py +256 -69
- package/bin/_lib_dashboard_sources.py +212 -24
- package/bin/_lib_diagnosis.py +1261 -0
- package/bin/_lib_forecast.py +62 -4
- package/bin/_lib_perf.py +12 -0
- package/bin/_lib_pricing.py +8 -7
- package/bin/_lib_readme_refresh.py +26 -5
- package/bin/_lib_render.py +31 -3
- package/bin/_lib_share_templates.py +150 -55
- package/bin/_lib_snapshot_cache.py +71 -13
- package/bin/_lib_source_identity.py +50 -2
- package/bin/_lib_subscription_weeks.py +65 -0
- package/bin/cctally +103 -16
- package/bin/cctally-explain +5 -0
- package/dashboard/static/assets/dashboardStream.shared-worker-1XTMV3nr.js +1 -0
- package/dashboard/static/assets/index-Di2hljvB.css +1 -0
- package/dashboard/static/assets/index-XYCIWjVG.js +97 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +6 -1
- package/dashboard/static/assets/index-B5YfQEtn.css +0 -1
- package/dashboard/static/assets/index-Bt59nMMO.js +0 -97
|
@@ -0,0 +1,4054 @@
|
|
|
1
|
+
"""The only store-opening component of the diagnosis (#620 S2).
|
|
2
|
+
|
|
3
|
+
Both `cctally explain` and `GET /api/diagnosis` call `build_diagnosis`, which
|
|
4
|
+
is what makes the two surfaces incapable of drifting. Everything here is I/O
|
|
5
|
+
and shaping; every classification rule lives in the pure
|
|
6
|
+
`bin/_lib_diagnosis.py`.
|
|
7
|
+
|
|
8
|
+
Three things in this file are load-bearing and easy to lose.
|
|
9
|
+
|
|
10
|
+
The open path is genuinely read-only. The ordinary opener performs schema
|
|
11
|
+
work, migration, legacy import, contract repair and replay, any of which
|
|
12
|
+
would mutate a store the diagnosis is only reading and would defeat
|
|
13
|
+
component-local consistency. `open_read_only` uses the raw `mode=ro` URI
|
|
14
|
+
connect, the same class of path `db checkpoint` established with its
|
|
15
|
+
`mode=rw` connect that skips schema, migrations and purge.
|
|
16
|
+
|
|
17
|
+
Generation is a version vector with component-local consistency, not a claim
|
|
18
|
+
of an instantaneous cross-file cut: SQLite provides no such thing across
|
|
19
|
+
separate files, and a content digest describes bytes read rather than
|
|
20
|
+
proving that independently opened snapshots coexisted. Each component is
|
|
21
|
+
probed before and after its read; a component whose probe differs is re-read
|
|
22
|
+
once, and a second divergence yields `generation_incoherent`.
|
|
23
|
+
|
|
24
|
+
Codex 5-hour entries and windows are BOTH classified through
|
|
25
|
+
`bin/_lib_codex_pools.py`, and an entry joins only to a window of a
|
|
26
|
+
compatible logical pool. Existing block assembly at
|
|
27
|
+
`bin/_cctally_dashboard_sources.py:2366-2414` assigns by account and time
|
|
28
|
+
without a pool restriction, so an unrestricted join would let an overlapping
|
|
29
|
+
Spark window and a standard window both claim the same spend. An entry
|
|
30
|
+
matching no compatible window is a coverage gap and is never duplicated
|
|
31
|
+
across pools.
|
|
32
|
+
|
|
33
|
+
Spec: docs/superpowers/specs/2026-08-19-620-s2-on-demand-diagnosis.md §3, §5
|
|
34
|
+
"""
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import datetime as dt
|
|
38
|
+
import hashlib
|
|
39
|
+
import json
|
|
40
|
+
import re
|
|
41
|
+
import shlex
|
|
42
|
+
import sqlite3
|
|
43
|
+
import sys
|
|
44
|
+
from dataclasses import dataclass, field
|
|
45
|
+
from types import MappingProxyType
|
|
46
|
+
from typing import Any, Collection, Iterable, Mapping, Sequence
|
|
47
|
+
|
|
48
|
+
import _cctally_core
|
|
49
|
+
import _lib_accounts
|
|
50
|
+
import _lib_codex_pools
|
|
51
|
+
import _lib_diagnosis as kernel
|
|
52
|
+
from _lib_diagnosis import (
|
|
53
|
+
ClassResult, ContributorSpec, Denominator, DiagnosisWindow,
|
|
54
|
+
EstablishmentError, EstablishmentFailure, PopulationCoverage,
|
|
55
|
+
SubjectFacts, WithheldCause,
|
|
56
|
+
)
|
|
57
|
+
from _lib_fmt import stable_sum
|
|
58
|
+
from _lib_pricing import (
|
|
59
|
+
_calculate_entry_cost, _resolve_codex_pricing, claude_usage_dict,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
UTC = dt.timezone.utc
|
|
63
|
+
|
|
64
|
+
# Read-only connections still contend with a live writer's WAL, so they get a
|
|
65
|
+
# bounded wait rather than an immediate `database is locked`.
|
|
66
|
+
READ_BUSY_TIMEOUT_MS = 4000
|
|
67
|
+
|
|
68
|
+
_STORE_PATHS = {
|
|
69
|
+
"cache": "CACHE_DB_PATH",
|
|
70
|
+
"stats": "DB_PATH",
|
|
71
|
+
"conversations": "CONVERSATIONS_DB_PATH",
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
# The generation components. `conversations` is CONDITIONAL: a plan that
|
|
75
|
+
# reads no conversation bytes publishes no component for it at all, and
|
|
76
|
+
# `GenerationVector.as_dict` appends it only when it is non-null.
|
|
77
|
+
# The three every plan reads, in the order they are established.
|
|
78
|
+
_UNCONDITIONAL_COMPONENTS: tuple[str, ...] = ("stats", "cache", "configuration")
|
|
79
|
+
# Every component this module can publish. DERIVED from the unconditional
|
|
80
|
+
# three, so it cannot drift from them: it was a hand-written tuple that nothing
|
|
81
|
+
# in the repository read, which is a constant that can only ever be wrong.
|
|
82
|
+
GENERATION_COMPONENTS: tuple[str, ...] = _UNCONDITIONAL_COMPONENTS + (
|
|
83
|
+
"conversations",)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _cctally():
|
|
87
|
+
return sys.modules["cctally"]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# --- the scope ----------------------------------------------------------
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class DiagnosisScope:
|
|
94
|
+
"""Provider, immutable account key and a half-open `[start, end)` window.
|
|
95
|
+
|
|
96
|
+
The plan declared `window_start` and `window_end` as ISO strings. They
|
|
97
|
+
are timezone-aware datetimes here, because every consumer in this file
|
|
98
|
+
compares them against parsed row timestamps and a string window would
|
|
99
|
+
mean parsing the same two values at a dozen call sites. `window_start_iso`
|
|
100
|
+
and `window_end_iso` render the wire form.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
source: str
|
|
104
|
+
account_key: str | None
|
|
105
|
+
window_start: dt.datetime
|
|
106
|
+
window_end: dt.datetime
|
|
107
|
+
effective_speed: str | None = None
|
|
108
|
+
display_tz: str = "UTC"
|
|
109
|
+
label: str = ""
|
|
110
|
+
|
|
111
|
+
def __post_init__(self) -> None:
|
|
112
|
+
for value in (self.window_start, self.window_end):
|
|
113
|
+
if value.tzinfo is None or value.utcoffset() is None:
|
|
114
|
+
raise EstablishmentFailure(
|
|
115
|
+
EstablishmentError.RANGE_UNRESOLVED.value,
|
|
116
|
+
"diagnosis window bounds must be timezone-aware",
|
|
117
|
+
)
|
|
118
|
+
if self.window_end <= self.window_start:
|
|
119
|
+
raise EstablishmentFailure(
|
|
120
|
+
EstablishmentError.RANGE_UNRESOLVED.value,
|
|
121
|
+
"diagnosis window end must be after its start",
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
@property
|
|
125
|
+
def window_start_iso(self) -> str:
|
|
126
|
+
return _iso_z(self.window_start)
|
|
127
|
+
|
|
128
|
+
@property
|
|
129
|
+
def window_end_iso(self) -> str:
|
|
130
|
+
return _iso_z(self.window_end)
|
|
131
|
+
|
|
132
|
+
def preceding(self) -> "DiagnosisScope":
|
|
133
|
+
"""The immediately preceding equal-duration half-open window.
|
|
134
|
+
|
|
135
|
+
Same provider, account, timezone interpretation and effective speed,
|
|
136
|
+
which is what makes the baseline a comparison rather than a
|
|
137
|
+
coincidence.
|
|
138
|
+
"""
|
|
139
|
+
span = self.window_end - self.window_start
|
|
140
|
+
return DiagnosisScope(
|
|
141
|
+
source=self.source,
|
|
142
|
+
account_key=self.account_key,
|
|
143
|
+
window_start=self.window_start - span,
|
|
144
|
+
window_end=self.window_start,
|
|
145
|
+
effective_speed=self.effective_speed,
|
|
146
|
+
display_tz=self.display_tz,
|
|
147
|
+
label="baseline",
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
def window(self) -> DiagnosisWindow:
|
|
151
|
+
return DiagnosisWindow(
|
|
152
|
+
start_at=self.window_start_iso,
|
|
153
|
+
end_at=self.window_end_iso,
|
|
154
|
+
tz=self.display_tz,
|
|
155
|
+
label=self.label,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _iso_z(value: dt.datetime) -> str:
|
|
160
|
+
"""The wire form: an instant rendered with a trailing `Z`."""
|
|
161
|
+
return value.astimezone(UTC).isoformat().replace("+00:00", "Z")
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _iso_sql(value: dt.datetime) -> str:
|
|
165
|
+
"""The SQL bound form, which is deliberately NOT the wire form.
|
|
166
|
+
|
|
167
|
+
The two stores disagree about how they spell an instant:
|
|
168
|
+
`session_entries.timestamp_utc` is written with a `+00:00` offset, while
|
|
169
|
+
`codex_session_entries.timestamp_utc` goes through
|
|
170
|
+
`_lib_jsonl._format_codex_timestamp` and ends in `Z`. These comparisons
|
|
171
|
+
are lexical over an indexed TEXT column, and `+` (0x2B) sorts before `Z`
|
|
172
|
+
(0x5A), so binding the `+00:00` form makes `>= start` admit both
|
|
173
|
+
spellings at the lower bound and `< end` exclude both at the upper one.
|
|
174
|
+
Binding the `Z` form instead silently drops every Claude row that lands
|
|
175
|
+
exactly on the window start.
|
|
176
|
+
"""
|
|
177
|
+
return value.astimezone(UTC).isoformat()
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _parse_ts(raw: object) -> dt.datetime | None:
|
|
181
|
+
if not isinstance(raw, str) or not raw:
|
|
182
|
+
return None
|
|
183
|
+
try:
|
|
184
|
+
parsed = dt.datetime.fromisoformat(raw.replace("Z", "+00:00"))
|
|
185
|
+
except ValueError:
|
|
186
|
+
return None
|
|
187
|
+
if parsed.tzinfo is None:
|
|
188
|
+
parsed = parsed.replace(tzinfo=UTC)
|
|
189
|
+
return parsed.astimezone(UTC)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# --- the read-only open path -------------------------------------------
|
|
193
|
+
|
|
194
|
+
def _store_path(kind: str):
|
|
195
|
+
try:
|
|
196
|
+
attribute = _STORE_PATHS[kind]
|
|
197
|
+
except KeyError as exc:
|
|
198
|
+
raise EstablishmentFailure(
|
|
199
|
+
EstablishmentError.STORE_UNAVAILABLE.value, f"unknown store {kind}"
|
|
200
|
+
) from exc
|
|
201
|
+
return getattr(_cctally_core, attribute)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def open_read_only(kind: str) -> sqlite3.Connection:
|
|
205
|
+
"""Open one store read-only, performing no schema work at all.
|
|
206
|
+
|
|
207
|
+
`mode=ro` is what makes the claim structural rather than a promise: the
|
|
208
|
+
connection cannot write, so migration, legacy import, contract repair and
|
|
209
|
+
replay cannot run behind our back and change the very bytes whose digest
|
|
210
|
+
we are about to publish.
|
|
211
|
+
"""
|
|
212
|
+
path = _store_path(kind)
|
|
213
|
+
if not path.exists():
|
|
214
|
+
raise EstablishmentFailure(
|
|
215
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
216
|
+
f"{kind} store is not present",
|
|
217
|
+
)
|
|
218
|
+
try:
|
|
219
|
+
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
220
|
+
conn.execute(f"PRAGMA busy_timeout={READ_BUSY_TIMEOUT_MS}")
|
|
221
|
+
except sqlite3.Error as exc:
|
|
222
|
+
raise EstablishmentFailure(
|
|
223
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
224
|
+
f"{kind} store could not be opened read-only: {exc}",
|
|
225
|
+
) from exc
|
|
226
|
+
conn.row_factory = sqlite3.Row
|
|
227
|
+
return conn
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _execute(conn: sqlite3.Connection, sql: str,
|
|
231
|
+
params: Sequence[Any] = ()) -> list[sqlite3.Row]:
|
|
232
|
+
"""The single query chokepoint.
|
|
233
|
+
|
|
234
|
+
Every read in this module goes through it, so an account-scoping audit
|
|
235
|
+
can observe the whole SQL surface from one place rather than trusting a
|
|
236
|
+
grep over call sites.
|
|
237
|
+
"""
|
|
238
|
+
return list(conn.execute(sql, tuple(params)))
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
# --- generation ---------------------------------------------------------
|
|
242
|
+
|
|
243
|
+
@dataclass(frozen=True)
|
|
244
|
+
class GenerationVector:
|
|
245
|
+
stats: str
|
|
246
|
+
cache: str
|
|
247
|
+
configuration: str
|
|
248
|
+
# Optional because a plan that reads no conversation bytes publishes no
|
|
249
|
+
# conversations component at all. A `None` here is ABSENCE, not the value
|
|
250
|
+
# zero and not the string "None".
|
|
251
|
+
conversations: str | None = None
|
|
252
|
+
|
|
253
|
+
def as_dict(self) -> dict[str, str]:
|
|
254
|
+
out = {"stats": self.stats, "cache": self.cache,
|
|
255
|
+
"configuration": self.configuration}
|
|
256
|
+
if self.conversations is not None:
|
|
257
|
+
out["conversations"] = self.conversations
|
|
258
|
+
return out
|
|
259
|
+
|
|
260
|
+
def generation_id(self, scope: DiagnosisScope,
|
|
261
|
+
plan: "kernel.ExecutionPlan") -> str:
|
|
262
|
+
"""SHA-256 over the contract version, the normalized scope, the whole
|
|
263
|
+
plan and the component vector.
|
|
264
|
+
|
|
265
|
+
Two surfaces reading the same facts under the same plan produce the
|
|
266
|
+
same id, which is what makes the equivalence check meaningful. The
|
|
267
|
+
plan is part of the identity because a seven-class report with three
|
|
268
|
+
withheld results is a different report from a four-class one and must
|
|
269
|
+
not share an identifier with it. Every plan tokenizes; there is no
|
|
270
|
+
legacy exemption.
|
|
271
|
+
"""
|
|
272
|
+
digest = hashlib.sha256()
|
|
273
|
+
digest.update(str(kernel.DIAGNOSIS_CONTRACT_VERSION).encode())
|
|
274
|
+
for part in (
|
|
275
|
+
scope.source, scope.account_key or _lib_accounts.UNATTRIBUTED,
|
|
276
|
+
scope.window_start_iso, scope.window_end_iso,
|
|
277
|
+
scope.effective_speed or "", scope.display_tz,
|
|
278
|
+
):
|
|
279
|
+
digest.update(b"\x1f")
|
|
280
|
+
digest.update(str(part).encode())
|
|
281
|
+
digest.update(b"\x1d")
|
|
282
|
+
digest.update(plan.plan_token().encode())
|
|
283
|
+
# Iterate `as_dict()`, NOT `GENERATION_COMPONENTS` with `getattr`, or a
|
|
284
|
+
# null component would hash the literal string "None" and an absent
|
|
285
|
+
# component would be indistinguishable from a present one whose digest
|
|
286
|
+
# happened to be that text.
|
|
287
|
+
for component, value in self.as_dict().items():
|
|
288
|
+
digest.update(b"\x1e")
|
|
289
|
+
digest.update(component.encode())
|
|
290
|
+
digest.update(b"=")
|
|
291
|
+
digest.update(value.encode())
|
|
292
|
+
return digest.hexdigest()
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _digest_component_pair(current_rows: Iterable[Sequence[Any]],
|
|
296
|
+
baseline_rows: Iterable[Sequence[Any]] | None
|
|
297
|
+
) -> str:
|
|
298
|
+
"""One component's published digest, binding BOTH window subdigests.
|
|
299
|
+
|
|
300
|
+
The preceding-window bundle is established separately while only the
|
|
301
|
+
current bundle's vector is published, so a baseline value could change
|
|
302
|
+
while `generationId` stayed constant — a pre-existing S2 defect that three
|
|
303
|
+
more baselines would compound. Each component digest therefore covers its
|
|
304
|
+
current-window and its baseline-window row streams, and the two are
|
|
305
|
+
domain-separated so a row moving between them is visible.
|
|
306
|
+
"""
|
|
307
|
+
digest = hashlib.sha256()
|
|
308
|
+
digest.update(b"current=")
|
|
309
|
+
digest.update(_digest_rows(current_rows).encode())
|
|
310
|
+
digest.update(b"\x1dbaseline=")
|
|
311
|
+
digest.update(_digest_rows(baseline_rows or ()).encode())
|
|
312
|
+
return digest.hexdigest()
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _probe_component(component: str, bundle: "StoreBundle") -> str:
|
|
316
|
+
"""A cheap fingerprint of one component's current state.
|
|
317
|
+
|
|
318
|
+
`PRAGMA data_version` advances when another connection commits to the
|
|
319
|
+
database, which is exactly the event that would make a digest describe
|
|
320
|
+
two different states. The file size is folded in so a change the pragma
|
|
321
|
+
cannot see on a fresh connection is still visible.
|
|
322
|
+
"""
|
|
323
|
+
if component == "configuration":
|
|
324
|
+
path = _cctally_core.CONFIG_PATH
|
|
325
|
+
try:
|
|
326
|
+
stat = path.stat()
|
|
327
|
+
except OSError:
|
|
328
|
+
return "absent"
|
|
329
|
+
return f"{stat.st_size}:{stat.st_mtime_ns}"
|
|
330
|
+
if component == "conversations":
|
|
331
|
+
# This component's read runs the three conversation-derived
|
|
332
|
+
# evaluators, and prompt-cache churn reads `session_entries` off the
|
|
333
|
+
# CACHE connection to seed its walk — after the cache component's own
|
|
334
|
+
# probe pair has closed. A writer committing between the two moves
|
|
335
|
+
# `flaggedTurnCount`, which this digest binds, so a probe covering
|
|
336
|
+
# only `conversations.db` cannot fire `generation_incoherent` for a
|
|
337
|
+
# mutation that genuinely moved a published figure. The cost is that a
|
|
338
|
+
# cache commit during this read now costs a re-read, and a second one
|
|
339
|
+
# `generation_incoherent`; a component exempting itself from the
|
|
340
|
+
# coherence contract is worse than an occasional retry.
|
|
341
|
+
#
|
|
342
|
+
# The fold applies only where the read actually happens. On the
|
|
343
|
+
# schema-gap branch the component runs no evaluator, reads no cache
|
|
344
|
+
# bytes and digests one constant naming one of our own tables, so a
|
|
345
|
+
# `cache.db` commit cannot move its digest — and folding the cache
|
|
346
|
+
# probe there would let a hook tick force a re-read and a second one a
|
|
347
|
+
# 503 for a component nothing about `cache.db` can change.
|
|
348
|
+
#
|
|
349
|
+
# The gate itself can raise, and this probe runs OUTSIDE the component
|
|
350
|
+
# read, so it cannot be moved inside that read's backstop. It gets the
|
|
351
|
+
# equivalent protection instead: a `sqlite3` error here decides no
|
|
352
|
+
# fold, and the read below asks the same question inside its own
|
|
353
|
+
# backstop and classifies the failure there. A probe that raised would
|
|
354
|
+
# end the whole report — all seven classes — before the read ever
|
|
355
|
+
# reached that classification, which is the failure this catch exists
|
|
356
|
+
# to prevent.
|
|
357
|
+
#
|
|
358
|
+
# `sqlite3.Error` and not `Exception`: an unrecognised source raises
|
|
359
|
+
# `EstablishmentFailure`, which must still end the report rather than
|
|
360
|
+
# be absorbed into a probe string.
|
|
361
|
+
#
|
|
362
|
+
# The fallback probes the WIDER pair, which is the shape a readable
|
|
363
|
+
# store also probes. A lock that clears between the two probes would
|
|
364
|
+
# otherwise read as a divergence and cost a re-read for a store that
|
|
365
|
+
# is fine.
|
|
366
|
+
#
|
|
367
|
+
# The cost, stated so the next reader does not have to derive it: the
|
|
368
|
+
# wide pair binds `cache.db` even though the read that follows may
|
|
369
|
+
# fail before touching a cache byte, so a hook tick committing between
|
|
370
|
+
# the probes reads as a divergence for a component that read nothing
|
|
371
|
+
# there, and two of them raise `generation_incoherent`. The real-store
|
|
372
|
+
# pass recorded exactly that against a live store with a running
|
|
373
|
+
# dashboard. The wide pair is still right, because the probe cannot
|
|
374
|
+
# know in advance whether the read will fail, and the narrow
|
|
375
|
+
# alternative pays a shape flip and a retry in the common case.
|
|
376
|
+
try:
|
|
377
|
+
gap = bundle.conversations_schema_gap()
|
|
378
|
+
except sqlite3.Error:
|
|
379
|
+
return (_store_probe("conversations", bundle) + "|"
|
|
380
|
+
+ _store_probe("cache", bundle))
|
|
381
|
+
if gap is not None:
|
|
382
|
+
return _store_probe("conversations", bundle)
|
|
383
|
+
return (_store_probe("conversations", bundle) + "|"
|
|
384
|
+
+ _store_probe("cache", bundle))
|
|
385
|
+
return _store_probe(component, bundle)
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def _store_probe(component: str, bundle: "StoreBundle") -> str:
|
|
389
|
+
conn = bundle.connection(component)
|
|
390
|
+
try:
|
|
391
|
+
version = _execute(conn, "PRAGMA data_version")[0][0]
|
|
392
|
+
except (sqlite3.Error, IndexError):
|
|
393
|
+
version = "?"
|
|
394
|
+
try:
|
|
395
|
+
size = _store_path(component).stat().st_size
|
|
396
|
+
except OSError:
|
|
397
|
+
size = -1
|
|
398
|
+
return f"{version}:{size}"
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _read_component(component: str, scope: DiagnosisScope,
|
|
402
|
+
bundle: "StoreBundle") -> tuple[list, Any]:
|
|
403
|
+
"""Read one component's facts and return the rows that describe them.
|
|
404
|
+
|
|
405
|
+
The caller digests the returned stream in deterministic key order, so the
|
|
406
|
+
digest describes the facts the report is built from rather than the file's
|
|
407
|
+
bytes.
|
|
408
|
+
|
|
409
|
+
**Every component needs its OWN branch.** This dispatched `configuration`,
|
|
410
|
+
then `cache`, and fell through to `_read_stats_component` for anything
|
|
411
|
+
else, so a `conversations` component added without a branch would be
|
|
412
|
+
silently digested as stats: a plausible digest describing entirely the
|
|
413
|
+
wrong facts, with nothing failing. The fall-through is now an explicit
|
|
414
|
+
refusal instead.
|
|
415
|
+
"""
|
|
416
|
+
if component == "configuration":
|
|
417
|
+
return _digest_configuration(scope)
|
|
418
|
+
if component == "cache":
|
|
419
|
+
return _read_cache_component(scope, bundle)
|
|
420
|
+
if component == "conversations":
|
|
421
|
+
return _read_conversations_component(scope, bundle)
|
|
422
|
+
if component == "stats":
|
|
423
|
+
return _read_stats_component(scope, bundle)
|
|
424
|
+
raise EstablishmentFailure(
|
|
425
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
426
|
+
f"unknown generation component {component}",
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _digest_configuration(scope: DiagnosisScope) -> tuple[list, Any]:
|
|
431
|
+
try:
|
|
432
|
+
raw = _cctally_core.CONFIG_PATH.read_text()
|
|
433
|
+
payload = json.loads(raw)
|
|
434
|
+
except (OSError, ValueError):
|
|
435
|
+
payload = {}
|
|
436
|
+
relevant = {
|
|
437
|
+
"display.tz": scope.display_tz,
|
|
438
|
+
"source": scope.source,
|
|
439
|
+
"config": payload,
|
|
440
|
+
}
|
|
441
|
+
body = json.dumps(relevant, sort_keys=True, separators=(",", ":"))
|
|
442
|
+
return [("configuration", body)], payload
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def _digest_rows(rows: Iterable[Sequence[Any]]) -> str:
|
|
446
|
+
digest = hashlib.sha256()
|
|
447
|
+
for row in rows:
|
|
448
|
+
digest.update(b"\x1e")
|
|
449
|
+
digest.update("\x1f".join("" if v is None else str(v)
|
|
450
|
+
for v in row).encode())
|
|
451
|
+
return digest.hexdigest()
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
# --- raw accounting facts ----------------------------------------------
|
|
455
|
+
|
|
456
|
+
@dataclass(frozen=True)
|
|
457
|
+
class AccountingEntry:
|
|
458
|
+
"""One priced accounting row, provider-neutral for the fold."""
|
|
459
|
+
|
|
460
|
+
timestamp: dt.datetime
|
|
461
|
+
model: str
|
|
462
|
+
project_key: str
|
|
463
|
+
project_label: str
|
|
464
|
+
session_key: str
|
|
465
|
+
session_label: str
|
|
466
|
+
root_key: str
|
|
467
|
+
pool: str | None
|
|
468
|
+
cost_usd: float
|
|
469
|
+
input_tokens: int = 0
|
|
470
|
+
output_tokens: int = 0
|
|
471
|
+
cache_create_tokens: int = 0
|
|
472
|
+
cache_read_tokens: int = 0
|
|
473
|
+
is_fallback_pricing: bool = False
|
|
474
|
+
# Session identity and project identity are separate facts and are asked
|
|
475
|
+
# separately. One flag for both made the project class withhold as
|
|
476
|
+
# `unattributed_evidence` whenever `session_files` was still inside its
|
|
477
|
+
# documented lazy-backfill window, even though every project path had
|
|
478
|
+
# resolved.
|
|
479
|
+
project_identity_resolved: bool = True
|
|
480
|
+
session_identity_resolved: bool = True
|
|
481
|
+
# Whether this entry's model has a pricing row of its own. A Claude model
|
|
482
|
+
# the embedded table does not know contributes zero cost, which is what
|
|
483
|
+
# `pricing_unavailable` is about; a Codex model priced through the legacy
|
|
484
|
+
# fallback IS priced, and is a `is_fallback_pricing` qualification only.
|
|
485
|
+
pricing_resolved: bool = True
|
|
486
|
+
# --- the S3 join keys (#620 S3) ---------------------------------------
|
|
487
|
+
#
|
|
488
|
+
# Populated only under the S3-expanded projection. Three of them default to
|
|
489
|
+
# `None` there, which is absence; `source_path` defaults to the EMPTY
|
|
490
|
+
# STRING, because it is a path and a bucket derived from one, and no
|
|
491
|
+
# caller distinguishes "no path" from "a path we did not read". They are
|
|
492
|
+
# physical join keys and never reach a row field, an evidence value or
|
|
493
|
+
# the wire; the Claude pair joins the canonical turn, the Codex offset is
|
|
494
|
+
# what the event inference attributes a turn from, and `source_path` is
|
|
495
|
+
# what the Claude subagent bucket is derived from inside the reader.
|
|
496
|
+
source_path: str = ""
|
|
497
|
+
msg_id: str | None = None
|
|
498
|
+
req_id: str | None = None
|
|
499
|
+
line_offset: int | None = None
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
@dataclass(frozen=True)
|
|
503
|
+
class NativeBlock:
|
|
504
|
+
"""One provider-native 5-hour window.
|
|
505
|
+
|
|
506
|
+
`key` carries the root and the logical pool, not just the start instant.
|
|
507
|
+
Two windows on one root can begin in the same minute and belong to
|
|
508
|
+
different pools — that is exactly the Spark-beside-standard case the
|
|
509
|
+
pool-compatible join exists for — so a start-only key would collapse
|
|
510
|
+
them into one subject and undo the separation.
|
|
511
|
+
"""
|
|
512
|
+
|
|
513
|
+
key: str
|
|
514
|
+
label: str
|
|
515
|
+
start_at: dt.datetime
|
|
516
|
+
end_at: dt.datetime
|
|
517
|
+
root_key: str
|
|
518
|
+
pool: str | None
|
|
519
|
+
next_step: str = ""
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _native_block_key(root_key: str, pool: str | None,
|
|
523
|
+
start_at: dt.datetime) -> str:
|
|
524
|
+
"""The subject key both block builders publish, in one place.
|
|
525
|
+
|
|
526
|
+
A five-hour subject key is a COMPOSITE — root, logical pool and start
|
|
527
|
+
instant — and every test that hand-wrote a bare instant instead was
|
|
528
|
+
checking a shape production never produces. Deriving it from here is what
|
|
529
|
+
keeps such a test honest.
|
|
530
|
+
"""
|
|
531
|
+
return f"{root_key}|{pool or 'standard'}|{_iso_z(start_at)}"
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
@dataclass(frozen=True)
|
|
535
|
+
class RawFacts:
|
|
536
|
+
entries: tuple[AccountingEntry, ...] = ()
|
|
537
|
+
blocks: tuple[NativeBlock, ...] = ()
|
|
538
|
+
retained_start: dt.datetime | None = None
|
|
539
|
+
retained_end: dt.datetime | None = None
|
|
540
|
+
# The earliest instant this install was observing this provider at all,
|
|
541
|
+
# independent of the accounting rows. It is what separates a store pruned
|
|
542
|
+
# past the window start from an install that simply did not exist then.
|
|
543
|
+
store_horizon: dt.datetime | None = None
|
|
544
|
+
unavailable_cause: str | None = None
|
|
545
|
+
|
|
546
|
+
@property
|
|
547
|
+
def total_usd(self) -> float:
|
|
548
|
+
# `stable_sum` at every fold layer: `golden-json.txt` prints full float
|
|
549
|
+
# repr values, so a summation difference is a golden difference.
|
|
550
|
+
return stable_sum(e.cost_usd for e in self.entries)
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
# The "not yet asked" sentinel for the memoized schema-gap answer, because
|
|
554
|
+
# `None` is a real answer there and would make every probe re-ask.
|
|
555
|
+
_UNPROBED = object()
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
class StoreBundle:
|
|
559
|
+
"""The open read-only connections plus the facts read through them."""
|
|
560
|
+
|
|
561
|
+
def __init__(self, scope: DiagnosisScope,
|
|
562
|
+
plan: "kernel.ExecutionPlan") -> None:
|
|
563
|
+
self.scope = scope
|
|
564
|
+
self._connections: dict[str, sqlite3.Connection] = {}
|
|
565
|
+
self.facts = RawFacts()
|
|
566
|
+
self.baseline: RawFacts | None = None
|
|
567
|
+
self.vector: GenerationVector | None = None
|
|
568
|
+
# The POLICY plan this bundle reads under. A class denied in stage 1
|
|
569
|
+
# is settled, so a denied plan never opens, probes or digests
|
|
570
|
+
# `conversations.db`.
|
|
571
|
+
#
|
|
572
|
+
# REQUIRED, for the reason R9 made it required on `build_diagnosis`
|
|
573
|
+
# and `build_provider_diagnosis`: a default of "everything is visible"
|
|
574
|
+
# is the permissive answer, and a caller that forgot to thread the
|
|
575
|
+
# route's transcript gate would silently read a store the request was
|
|
576
|
+
# not authorized to read. Every production caller passes one; only
|
|
577
|
+
# tests and the benchmark ever reached the default.
|
|
578
|
+
self.plan = plan
|
|
579
|
+
# Whether an AUTHORIZED conversations open succeeded. `False` when the
|
|
580
|
+
# plan asked for the store and it was absent or unopenable; also
|
|
581
|
+
# `False`, and never consulted, when the plan asked for nothing.
|
|
582
|
+
self.conversations_available = False
|
|
583
|
+
# The digest row streams, per component, kept so the published digest
|
|
584
|
+
# can bind the baseline window's streams alongside these.
|
|
585
|
+
self.component_rows: dict[str, list] = {}
|
|
586
|
+
# The three conversation-derived evaluations, computed ONCE. Populated
|
|
587
|
+
# during establishment when the plan opened the conversations store,
|
|
588
|
+
# so the digest and the rows describe one evaluation rather than two.
|
|
589
|
+
self.s3_evaluations: dict | None = None
|
|
590
|
+
# Each S3 class's aggregate qualifying-cost share over the immediately
|
|
591
|
+
# preceding equal-duration window. Absent means the baseline could not
|
|
592
|
+
# be established, which is `baseline_insufficient`.
|
|
593
|
+
self.s3_baseline_shares: dict[str, float] = {}
|
|
594
|
+
# The Codex thread rows the fan-out predicate resolves over, read and
|
|
595
|
+
# digested by the `cache` component that owns them.
|
|
596
|
+
self.codex_threads: tuple = ()
|
|
597
|
+
# Memoized schema-gap answer. `_UNPROBED` rather than `None`, because
|
|
598
|
+
# `None` is the answer "this store can serve every statement".
|
|
599
|
+
self._conversations_gap: Any = _UNPROBED
|
|
600
|
+
|
|
601
|
+
def connection(self, kind: str) -> sqlite3.Connection:
|
|
602
|
+
conn = self._connections.get(kind)
|
|
603
|
+
if conn is None:
|
|
604
|
+
conn = open_read_only(kind)
|
|
605
|
+
self._connections[kind] = conn
|
|
606
|
+
return conn
|
|
607
|
+
|
|
608
|
+
def conversations_schema_gap(self) -> str | None:
|
|
609
|
+
"""Which of our own tables this store cannot serve, memoized.
|
|
610
|
+
|
|
611
|
+
Memoized because the probe pair and the read all ask, and because the
|
|
612
|
+
answer cannot change under them: a schema change needs a WRITABLE
|
|
613
|
+
reopen and this connection is `mode=ro` for the bundle's whole life.
|
|
614
|
+
"""
|
|
615
|
+
if self._conversations_gap is _UNPROBED:
|
|
616
|
+
self._conversations_gap = _conversations_schema_gap(
|
|
617
|
+
self.connection("conversations"), self.scope.source)
|
|
618
|
+
return self._conversations_gap
|
|
619
|
+
|
|
620
|
+
def close(self) -> None:
|
|
621
|
+
for conn in self._connections.values():
|
|
622
|
+
try:
|
|
623
|
+
conn.close()
|
|
624
|
+
except sqlite3.Error:
|
|
625
|
+
pass
|
|
626
|
+
self._connections.clear()
|
|
627
|
+
|
|
628
|
+
def __enter__(self) -> "StoreBundle":
|
|
629
|
+
return self
|
|
630
|
+
|
|
631
|
+
def __exit__(self, *exc_info) -> None:
|
|
632
|
+
self.close()
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
# --- Claude accounting --------------------------------------------------
|
|
636
|
+
|
|
637
|
+
_CLAUDE_ENTRIES_SQL = """
|
|
638
|
+
SELECT se.timestamp_utc, se.model,
|
|
639
|
+
se.input_tokens, se.output_tokens,
|
|
640
|
+
se.cache_create_tokens, se.cache_read_tokens,
|
|
641
|
+
se.cache_create_1h_tokens, se.cost_usd_raw, se.speed,
|
|
642
|
+
se.source_path, sf.session_id, sf.project_path,
|
|
643
|
+
se.account_key
|
|
644
|
+
FROM session_entries se
|
|
645
|
+
LEFT JOIN session_files sf ON sf.path = se.source_path
|
|
646
|
+
WHERE se.timestamp_utc >= ? AND se.timestamp_utc < ?
|
|
647
|
+
"""
|
|
648
|
+
|
|
649
|
+
# The S3-expanded Claude projection. A SECOND FIXED STRING, never the frozen
|
|
650
|
+
# one with columns concatenated onto it: a projection built by concatenation
|
|
651
|
+
# cannot be byte-frozen, because every edit to the S3 half rewrites the S2
|
|
652
|
+
# half too. It adds the canonical turn key `(msg_id, req_id)`, which is what
|
|
653
|
+
# lets `maxContextWindowFraction` be derived from the in-window accounting
|
|
654
|
+
# population the adapter has already loaded rather than by walking retained
|
|
655
|
+
# conversation history.
|
|
656
|
+
_CLAUDE_ENTRIES_S3_SQL = """
|
|
657
|
+
SELECT se.timestamp_utc, se.model,
|
|
658
|
+
se.input_tokens, se.output_tokens,
|
|
659
|
+
se.cache_create_tokens, se.cache_read_tokens,
|
|
660
|
+
se.cache_create_1h_tokens, se.cost_usd_raw, se.speed,
|
|
661
|
+
se.source_path, sf.session_id, sf.project_path,
|
|
662
|
+
se.account_key, se.msg_id, se.req_id
|
|
663
|
+
FROM session_entries se
|
|
664
|
+
LEFT JOIN session_files sf ON sf.path = se.source_path
|
|
665
|
+
WHERE se.timestamp_utc >= ? AND se.timestamp_utc < ?
|
|
666
|
+
"""
|
|
667
|
+
|
|
668
|
+
_CLAUDE_RETENTION_SQL = """
|
|
669
|
+
SELECT MIN(timestamp_utc), MAX(timestamp_utc) FROM session_entries
|
|
670
|
+
WHERE 1=1
|
|
671
|
+
"""
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def _cache_projection_sql(plan: "kernel.ExecutionPlan") -> str:
|
|
675
|
+
"""Which of the four fixed projections this plan runs.
|
|
676
|
+
|
|
677
|
+
Which projection runs is a property of the PLAN, so two reports over the
|
|
678
|
+
same window under different plans read different facts and publish
|
|
679
|
+
different `cache` digests. The plan consulted here is the POLICY plan,
|
|
680
|
+
decided before any store opens: the established plan can withhold a class
|
|
681
|
+
after the read, and choosing the projection from it would publish a digest
|
|
682
|
+
describing a read the plan no longer claims to have made.
|
|
683
|
+
"""
|
|
684
|
+
if plan.source == "codex":
|
|
685
|
+
return (_CODEX_ENTRIES_S3_SQL if plan.requires_s3_projection()
|
|
686
|
+
else _CODEX_ENTRIES_SQL)
|
|
687
|
+
return (_CLAUDE_ENTRIES_S3_SQL if plan.requires_s3_projection()
|
|
688
|
+
else _CLAUDE_ENTRIES_SQL)
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
def _account_predicate(column: str, account_key: str | None,
|
|
692
|
+
params: list[Any]) -> str:
|
|
693
|
+
"""Scope a read to one account.
|
|
694
|
+
|
|
695
|
+
`None` keeps the account-blind merged read. The reserved `unattributed`
|
|
696
|
+
sentinel matches BOTH the literal stamp and NULL, which is the cache
|
|
697
|
+
read-path rule (`NULL` is `unattributed`).
|
|
698
|
+
"""
|
|
699
|
+
if account_key is None:
|
|
700
|
+
return ""
|
|
701
|
+
if account_key == _lib_accounts.UNATTRIBUTED:
|
|
702
|
+
params.append(_lib_accounts.UNATTRIBUTED)
|
|
703
|
+
return f" AND ({column} IS NULL OR {column} = ?)"
|
|
704
|
+
params.append(account_key)
|
|
705
|
+
return f" AND {column} = ?"
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def _read_claude_entries(scope: DiagnosisScope, conn: sqlite3.Connection,
|
|
709
|
+
plan: "kernel.ExecutionPlan"
|
|
710
|
+
) -> tuple[list[AccountingEntry], list[tuple]]:
|
|
711
|
+
params: list[Any] = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
712
|
+
expanded = plan.requires_s3_projection()
|
|
713
|
+
sql = _cache_projection_sql(plan) + _account_predicate(
|
|
714
|
+
"se.account_key", scope.account_key, params
|
|
715
|
+
) + " ORDER BY se.timestamp_utc ASC, se.id ASC"
|
|
716
|
+
rows = _execute(conn, sql, params)
|
|
717
|
+
|
|
718
|
+
c = _cctally()
|
|
719
|
+
resolver_cache: dict[str, Any] = {}
|
|
720
|
+
entries: list[AccountingEntry] = []
|
|
721
|
+
digest_rows: list[tuple] = []
|
|
722
|
+
for row in rows:
|
|
723
|
+
timestamp = _parse_ts(row["timestamp_utc"])
|
|
724
|
+
if timestamp is None:
|
|
725
|
+
continue
|
|
726
|
+
usage = claude_usage_dict( # #195 chokepoint
|
|
727
|
+
input_tokens=int(row["input_tokens"] or 0),
|
|
728
|
+
output_tokens=int(row["output_tokens"] or 0),
|
|
729
|
+
cache_creation_tokens=int(row["cache_create_tokens"] or 0),
|
|
730
|
+
cache_read_tokens=int(row["cache_read_tokens"] or 0),
|
|
731
|
+
cache_1h_tokens=row["cache_create_1h_tokens"],
|
|
732
|
+
speed=row["speed"],
|
|
733
|
+
)
|
|
734
|
+
model = str(row["model"] or "")
|
|
735
|
+
# `mode="calculate"`, and no `cost_usd`: the diagnosis REPRICES every
|
|
736
|
+
# entry from `CLAUDE_MODEL_PRICING` (spec §3). `mode="auto"` beside a
|
|
737
|
+
# non-null stored cost returns that column verbatim and never consults
|
|
738
|
+
# the table, which would publish figures a pricing edit cannot correct
|
|
739
|
+
# and — worse — would report `isFallbackPricing` and `pricingResolved`
|
|
740
|
+
# computed from the embedded table beside dollars that did not come
|
|
741
|
+
# from it. A cost written before the #195 cache-write-TTL fix silently
|
|
742
|
+
# under-prices, and the diagnosis would inherit that silently too.
|
|
743
|
+
# `cost_usd_raw` is still read, for the population digest: the digest
|
|
744
|
+
# covers the facts read, not the facts used.
|
|
745
|
+
cost = _calculate_entry_cost(model, usage, mode="calculate")
|
|
746
|
+
project_path = row["project_path"]
|
|
747
|
+
project = c._resolve_project_key(project_path, "git-root", resolver_cache)
|
|
748
|
+
session_id = row["session_id"]
|
|
749
|
+
source_path = str(row["source_path"] or "")
|
|
750
|
+
entries.append(AccountingEntry(
|
|
751
|
+
timestamp=timestamp,
|
|
752
|
+
model=model,
|
|
753
|
+
project_key=project.bucket_path,
|
|
754
|
+
project_label=project.display_key,
|
|
755
|
+
session_key=str(session_id or source_path or "(unknown)"),
|
|
756
|
+
session_label=str(session_id or "(unknown)"),
|
|
757
|
+
root_key="claude",
|
|
758
|
+
pool=None,
|
|
759
|
+
cost_usd=cost,
|
|
760
|
+
input_tokens=int(row["input_tokens"] or 0),
|
|
761
|
+
output_tokens=int(row["output_tokens"] or 0),
|
|
762
|
+
cache_create_tokens=int(row["cache_create_tokens"] or 0),
|
|
763
|
+
cache_read_tokens=int(row["cache_read_tokens"] or 0),
|
|
764
|
+
is_fallback_pricing=_claude_pricing_is_fallback(model),
|
|
765
|
+
project_identity_resolved=not project.is_unknown,
|
|
766
|
+
session_identity_resolved=bool(session_id),
|
|
767
|
+
pricing_resolved=not _claude_pricing_is_fallback(model),
|
|
768
|
+
source_path=source_path,
|
|
769
|
+
msg_id=(row["msg_id"] if expanded else None),
|
|
770
|
+
req_id=(row["req_id"] if expanded else None),
|
|
771
|
+
))
|
|
772
|
+
digest_row = (
|
|
773
|
+
row["timestamp_utc"], model, row["input_tokens"],
|
|
774
|
+
row["output_tokens"], row["cache_create_tokens"],
|
|
775
|
+
row["cache_read_tokens"], row["cache_create_1h_tokens"],
|
|
776
|
+
row["cost_usd_raw"], row["speed"], source_path,
|
|
777
|
+
session_id, project_path, row["account_key"],
|
|
778
|
+
)
|
|
779
|
+
if expanded:
|
|
780
|
+
# The digest covers the facts READ, so the expanded projection's
|
|
781
|
+
# extra columns belong in it. Under the frozen projection the row
|
|
782
|
+
# is byte-identical to S2's.
|
|
783
|
+
digest_row = digest_row + (row["msg_id"], row["req_id"])
|
|
784
|
+
digest_rows.append(digest_row)
|
|
785
|
+
return entries, digest_rows
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
def _claude_pricing_is_fallback(model: str) -> bool:
|
|
789
|
+
"""Whether this model priced through a fallback rather than its own row.
|
|
790
|
+
|
|
791
|
+
`_calculate_entry_cost` returns only a float, so the qualification is
|
|
792
|
+
otherwise lost between pricing resolution and the row that reports it.
|
|
793
|
+
"""
|
|
794
|
+
c = _cctally()
|
|
795
|
+
pricing = getattr(c, "CLAUDE_MODEL_PRICING", None)
|
|
796
|
+
if not isinstance(pricing, Mapping):
|
|
797
|
+
return False
|
|
798
|
+
return model not in pricing
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
_CLAUDE_BLOCKS_SQL = """
|
|
802
|
+
SELECT five_hour_window_key, block_start_at, five_hour_resets_at,
|
|
803
|
+
account_key
|
|
804
|
+
FROM five_hour_blocks
|
|
805
|
+
WHERE five_hour_resets_at > ? AND block_start_at < ?
|
|
806
|
+
"""
|
|
807
|
+
|
|
808
|
+
# The install's own observation horizon for this provider, deliberately NOT
|
|
809
|
+
# bounded by the requested window: it is the evidence that separates a pruned
|
|
810
|
+
# store from a young one.
|
|
811
|
+
_CLAUDE_HORIZON_SQL = """
|
|
812
|
+
SELECT MIN(block_start_at) FROM five_hour_blocks
|
|
813
|
+
WHERE 1=1
|
|
814
|
+
"""
|
|
815
|
+
|
|
816
|
+
_CODEX_HORIZON_SQL = """
|
|
817
|
+
SELECT MIN(nominal_start_at_utc) FROM quota_window_blocks
|
|
818
|
+
WHERE source='codex' AND window_minutes=300
|
|
819
|
+
"""
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def _read_claude_blocks(scope: DiagnosisScope,
|
|
823
|
+
conn: sqlite3.Connection) -> tuple[list[NativeBlock], list[tuple]]:
|
|
824
|
+
params: list[Any] = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
825
|
+
sql = _CLAUDE_BLOCKS_SQL + _account_predicate(
|
|
826
|
+
"account_key", scope.account_key, params
|
|
827
|
+
) + " ORDER BY block_start_at ASC, five_hour_window_key ASC"
|
|
828
|
+
try:
|
|
829
|
+
rows = _execute(conn, sql, params)
|
|
830
|
+
except sqlite3.Error:
|
|
831
|
+
return [], []
|
|
832
|
+
blocks: list[NativeBlock] = []
|
|
833
|
+
digest_rows: list[tuple] = []
|
|
834
|
+
for row in rows:
|
|
835
|
+
start_at = _parse_ts(row["block_start_at"])
|
|
836
|
+
end_at = _parse_ts(row["five_hour_resets_at"])
|
|
837
|
+
if start_at is None or end_at is None or end_at <= start_at:
|
|
838
|
+
continue
|
|
839
|
+
blocks.append(NativeBlock(
|
|
840
|
+
key=_native_block_key("claude", None, start_at),
|
|
841
|
+
label=_iso_z(start_at),
|
|
842
|
+
start_at=start_at,
|
|
843
|
+
end_at=end_at,
|
|
844
|
+
root_key="claude",
|
|
845
|
+
pool=None,
|
|
846
|
+
next_step=(
|
|
847
|
+
f"cctally five-hour-breakdown --block-start {_iso_z(start_at)}"
|
|
848
|
+
),
|
|
849
|
+
))
|
|
850
|
+
digest_rows.append((
|
|
851
|
+
row["five_hour_window_key"], row["block_start_at"],
|
|
852
|
+
row["five_hour_resets_at"], row["account_key"],
|
|
853
|
+
))
|
|
854
|
+
return blocks, digest_rows
|
|
855
|
+
|
|
856
|
+
|
|
857
|
+
# --- Codex accounting ---------------------------------------------------
|
|
858
|
+
|
|
859
|
+
_CODEX_ENTRIES_SQL = """
|
|
860
|
+
SELECT entries.timestamp_utc, entries.session_id, entries.source_path,
|
|
861
|
+
entries.source_root_key, entries.conversation_key, entries.model,
|
|
862
|
+
entries.account_key, entries.input_tokens,
|
|
863
|
+
entries.cached_input_tokens, entries.output_tokens,
|
|
864
|
+
entries.reasoning_output_tokens, entries.total_tokens,
|
|
865
|
+
threads.cwd, threads.git_json
|
|
866
|
+
FROM codex_session_entries AS entries
|
|
867
|
+
LEFT JOIN codex_conversation_threads AS threads
|
|
868
|
+
ON threads.conversation_key = entries.conversation_key
|
|
869
|
+
AND threads.source_root_key = entries.source_root_key
|
|
870
|
+
WHERE entries.timestamp_utc >= ? AND entries.timestamp_utc < ?
|
|
871
|
+
"""
|
|
872
|
+
|
|
873
|
+
# The S3-expanded Codex projection, the second fixed string on this side. It
|
|
874
|
+
# adds `line_offset`, the physical key an accounting row is attributed by:
|
|
875
|
+
# `codex_session_entries` retains no `turn_id`, so the owning turn is
|
|
876
|
+
# recovered by mapping accounting offsets through the event inference, and
|
|
877
|
+
# without the offset the loaded rows recover neither the turn nor its
|
|
878
|
+
# per-turn capacity. `source_root_key` and `conversation_key` — the fan-out
|
|
879
|
+
# join keys — are already in the frozen projection.
|
|
880
|
+
_CODEX_ENTRIES_S3_SQL = """
|
|
881
|
+
SELECT entries.timestamp_utc, entries.session_id, entries.source_path,
|
|
882
|
+
entries.source_root_key, entries.conversation_key, entries.model,
|
|
883
|
+
entries.account_key, entries.input_tokens,
|
|
884
|
+
entries.cached_input_tokens, entries.output_tokens,
|
|
885
|
+
entries.reasoning_output_tokens, entries.total_tokens,
|
|
886
|
+
threads.cwd, threads.git_json, entries.line_offset
|
|
887
|
+
FROM codex_session_entries AS entries
|
|
888
|
+
LEFT JOIN codex_conversation_threads AS threads
|
|
889
|
+
ON threads.conversation_key = entries.conversation_key
|
|
890
|
+
AND threads.source_root_key = entries.source_root_key
|
|
891
|
+
WHERE entries.timestamp_utc >= ? AND entries.timestamp_utc < ?
|
|
892
|
+
"""
|
|
893
|
+
|
|
894
|
+
_CODEX_RETENTION_SQL = """
|
|
895
|
+
SELECT MIN(timestamp_utc), MAX(timestamp_utc) FROM codex_session_entries
|
|
896
|
+
WHERE 1=1
|
|
897
|
+
"""
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def _read_codex_entries(scope: DiagnosisScope, conn: sqlite3.Connection,
|
|
901
|
+
plan: "kernel.ExecutionPlan"
|
|
902
|
+
) -> tuple[list[AccountingEntry], list[tuple]]:
|
|
903
|
+
params: list[Any] = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
904
|
+
expanded = plan.requires_s3_projection()
|
|
905
|
+
# Every Codex query carries `account_key` (#341 / #373). A Codex read that
|
|
906
|
+
# forgets it merges two accounts' spend into one denominator.
|
|
907
|
+
sql = _cache_projection_sql(plan) + _account_predicate(
|
|
908
|
+
"entries.account_key", scope.account_key, params
|
|
909
|
+
) + (" ORDER BY entries.timestamp_utc ASC, entries.source_root_key ASC,"
|
|
910
|
+
" entries.conversation_key ASC, entries.id ASC")
|
|
911
|
+
rows = _execute(conn, sql, params)
|
|
912
|
+
|
|
913
|
+
c = _cctally()
|
|
914
|
+
speed = scope.effective_speed or "standard"
|
|
915
|
+
resolver_cache: dict[str, Any] = {}
|
|
916
|
+
entries: list[AccountingEntry] = []
|
|
917
|
+
digest_rows: list[tuple] = []
|
|
918
|
+
for row in rows:
|
|
919
|
+
timestamp = _parse_ts(row["timestamp_utc"])
|
|
920
|
+
if timestamp is None:
|
|
921
|
+
continue
|
|
922
|
+
model = str(row["model"] or "")
|
|
923
|
+
cost = c._calculate_codex_entry_cost(
|
|
924
|
+
model, int(row["input_tokens"] or 0),
|
|
925
|
+
int(row["cached_input_tokens"] or 0),
|
|
926
|
+
int(row["output_tokens"] or 0),
|
|
927
|
+
int(row["reasoning_output_tokens"] or 0),
|
|
928
|
+
speed=speed,
|
|
929
|
+
)
|
|
930
|
+
_pricing, is_fallback = _resolve_codex_pricing(model)
|
|
931
|
+
cwd = row["cwd"]
|
|
932
|
+
if isinstance(cwd, str) and cwd:
|
|
933
|
+
project = c._resolve_project_key(cwd, "git-root", resolver_cache)
|
|
934
|
+
project_key, project_label = project.bucket_path, project.display_key
|
|
935
|
+
project_resolved = not project.is_unknown
|
|
936
|
+
else:
|
|
937
|
+
# Absent Codex project metadata reduces identityCoverage
|
|
938
|
+
# explicitly and never basename-merges distinct git roots.
|
|
939
|
+
project_key = project_label = "(unassigned)"
|
|
940
|
+
project_resolved = False
|
|
941
|
+
conversation_key = str(row["conversation_key"] or "")
|
|
942
|
+
entries.append(AccountingEntry(
|
|
943
|
+
timestamp=timestamp,
|
|
944
|
+
model=model,
|
|
945
|
+
project_key=project_key,
|
|
946
|
+
project_label=project_label,
|
|
947
|
+
session_key=conversation_key or str(row["source_path"] or "(unknown)"),
|
|
948
|
+
session_label=conversation_key or "(unknown)",
|
|
949
|
+
root_key=str(row["source_root_key"] or ""),
|
|
950
|
+
pool=_lib_codex_pools.codex_model_scoped_quota_pool(model),
|
|
951
|
+
cost_usd=cost,
|
|
952
|
+
input_tokens=int(row["input_tokens"] or 0),
|
|
953
|
+
output_tokens=int(row["output_tokens"] or 0),
|
|
954
|
+
cache_read_tokens=int(row["cached_input_tokens"] or 0),
|
|
955
|
+
is_fallback_pricing=bool(is_fallback),
|
|
956
|
+
project_identity_resolved=project_resolved,
|
|
957
|
+
session_identity_resolved=bool(conversation_key),
|
|
958
|
+
# The Codex legacy fallback prices the entry, so the model is
|
|
959
|
+
# qualified rather than unpriceable.
|
|
960
|
+
pricing_resolved=True,
|
|
961
|
+
source_path=str(row["source_path"] or ""),
|
|
962
|
+
line_offset=(row["line_offset"] if expanded else None),
|
|
963
|
+
))
|
|
964
|
+
digest_row = (
|
|
965
|
+
row["timestamp_utc"], row["source_root_key"],
|
|
966
|
+
row["conversation_key"], model, row["account_key"],
|
|
967
|
+
row["input_tokens"], row["cached_input_tokens"],
|
|
968
|
+
row["output_tokens"], row["reasoning_output_tokens"],
|
|
969
|
+
row["cwd"], row["git_json"],
|
|
970
|
+
)
|
|
971
|
+
if expanded:
|
|
972
|
+
digest_row = digest_row + (row["source_path"], row["line_offset"])
|
|
973
|
+
digest_rows.append(digest_row)
|
|
974
|
+
return entries, digest_rows
|
|
975
|
+
|
|
976
|
+
|
|
977
|
+
_CODEX_BLOCKS_SQL = """
|
|
978
|
+
SELECT source_root_key, logical_limit_key, observed_slot, limit_name,
|
|
979
|
+
resets_at_utc, nominal_start_at_utc, orphaned_at, account_key
|
|
980
|
+
FROM quota_window_blocks
|
|
981
|
+
WHERE source='codex' AND window_minutes=300
|
|
982
|
+
AND resets_at_utc > ? AND nominal_start_at_utc < ?
|
|
983
|
+
"""
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
def _read_codex_blocks(scope: DiagnosisScope,
|
|
987
|
+
conn: sqlite3.Connection) -> tuple[list[NativeBlock], list[tuple]]:
|
|
988
|
+
# The Codex quota projection gate, before any fallback-catching SQL. This
|
|
989
|
+
# read is a CONSUMER of the published projection, so reading it while the
|
|
990
|
+
# projection is incomplete would report a partial set of native blocks as
|
|
991
|
+
# if it were the whole set — and the `except sqlite3.Error` below would
|
|
992
|
+
# render a refusal as an empty block list. The gate fails closed, and a
|
|
993
|
+
# store that cannot answer coherently is a report-establishment failure.
|
|
994
|
+
quota = _cctally()._load_sibling("_cctally_quota")
|
|
995
|
+
try:
|
|
996
|
+
quota.assert_projection_readable(conn)
|
|
997
|
+
except quota.QuotaProjectionIncomplete as exc:
|
|
998
|
+
raise EstablishmentFailure(
|
|
999
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
1000
|
+
f"the Codex quota projection is incomplete: {exc}",
|
|
1001
|
+
) from exc
|
|
1002
|
+
|
|
1003
|
+
params: list[Any] = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
1004
|
+
sql = _CODEX_BLOCKS_SQL + _account_predicate(
|
|
1005
|
+
"account_key", scope.account_key, params
|
|
1006
|
+
) + (" ORDER BY nominal_start_at_utc ASC, source_root_key ASC,"
|
|
1007
|
+
" logical_limit_key ASC, observed_slot ASC")
|
|
1008
|
+
try:
|
|
1009
|
+
rows = _execute(conn, sql, params)
|
|
1010
|
+
except sqlite3.Error:
|
|
1011
|
+
return [], []
|
|
1012
|
+
blocks: list[NativeBlock] = []
|
|
1013
|
+
digest_rows: list[tuple] = []
|
|
1014
|
+
for row in rows:
|
|
1015
|
+
if row["orphaned_at"] is not None:
|
|
1016
|
+
continue
|
|
1017
|
+
start_at = _parse_ts(row["nominal_start_at_utc"])
|
|
1018
|
+
end_at = _parse_ts(row["resets_at_utc"])
|
|
1019
|
+
if start_at is None or end_at is None or end_at <= start_at:
|
|
1020
|
+
continue
|
|
1021
|
+
pool = _window_pool(row["logical_limit_key"], row["limit_name"])
|
|
1022
|
+
root = str(row["source_root_key"] or "")
|
|
1023
|
+
blocks.append(NativeBlock(
|
|
1024
|
+
key=_native_block_key(root, pool, start_at),
|
|
1025
|
+
label=(_iso_z(start_at) if pool is None
|
|
1026
|
+
else f"{_iso_z(start_at)} ({pool})"),
|
|
1027
|
+
start_at=start_at,
|
|
1028
|
+
end_at=end_at,
|
|
1029
|
+
root_key=root,
|
|
1030
|
+
pool=pool,
|
|
1031
|
+
next_step="cctally codex quota blocks",
|
|
1032
|
+
))
|
|
1033
|
+
digest_rows.append((
|
|
1034
|
+
row["source_root_key"], row["logical_limit_key"],
|
|
1035
|
+
row["observed_slot"], row["limit_name"], row["resets_at_utc"],
|
|
1036
|
+
row["nominal_start_at_utc"], row["account_key"],
|
|
1037
|
+
))
|
|
1038
|
+
return blocks, digest_rows
|
|
1039
|
+
|
|
1040
|
+
|
|
1041
|
+
def _window_pool(logical_limit_key: object, limit_name: object) -> str | None:
|
|
1042
|
+
"""Classify a quota window's logical pool.
|
|
1043
|
+
|
|
1044
|
+
Pool classification has exactly one home. The two independent axes are
|
|
1045
|
+
`modelPool` in the interpreted key and a Spark `limit_name`; `limit_id`
|
|
1046
|
+
is deliberately not one.
|
|
1047
|
+
"""
|
|
1048
|
+
pool = _lib_codex_pools.codex_key_model_pool(logical_limit_key)
|
|
1049
|
+
if pool is not None:
|
|
1050
|
+
return pool
|
|
1051
|
+
return _lib_codex_pools.codex_model_scoped_quota_pool(limit_name)
|
|
1052
|
+
|
|
1053
|
+
|
|
1054
|
+
# --- component reads ----------------------------------------------------
|
|
1055
|
+
|
|
1056
|
+
def _read_cache_component(scope: DiagnosisScope,
|
|
1057
|
+
bundle: StoreBundle) -> tuple[list, Any]:
|
|
1058
|
+
conn = bundle.connection("cache")
|
|
1059
|
+
thread_rows: list = []
|
|
1060
|
+
try:
|
|
1061
|
+
if scope.source == "codex":
|
|
1062
|
+
entries, digest_rows = _read_codex_entries(scope, conn, bundle.plan)
|
|
1063
|
+
retention_sql = _CODEX_RETENTION_SQL
|
|
1064
|
+
if bundle.plan.requires_s3_projection():
|
|
1065
|
+
# The Codex fan-out signal lives in THIS store, so its rows
|
|
1066
|
+
# belong to THIS component's digest. Read once here and handed
|
|
1067
|
+
# to the evaluator: read only in the evaluator they would move
|
|
1068
|
+
# a published verdict while the identifier stood still, and
|
|
1069
|
+
# read in both places the digest could describe a different
|
|
1070
|
+
# state from the answer.
|
|
1071
|
+
thread_rows = _read_codex_threads(conn, entries)
|
|
1072
|
+
digest_rows = list(digest_rows) + _codex_thread_digest_rows(
|
|
1073
|
+
thread_rows)
|
|
1074
|
+
else:
|
|
1075
|
+
entries, digest_rows = _read_claude_entries(scope, conn, bundle.plan)
|
|
1076
|
+
retention_sql = _CLAUDE_RETENTION_SQL
|
|
1077
|
+
except sqlite3.Error as exc:
|
|
1078
|
+
# One provider's accounting tables can be absent or unreadable while
|
|
1079
|
+
# the other provider's are fine — an older cache.db carries no Codex
|
|
1080
|
+
# tables at all. That withholds this provider as `provider_unavailable`
|
|
1081
|
+
# rather than ending a two-provider report.
|
|
1082
|
+
# The digest describes the facts read, and none were. It names the
|
|
1083
|
+
# failure by exception TYPE rather than by message, because a SQLite
|
|
1084
|
+
# message is not stable across versions and this digest is published.
|
|
1085
|
+
return [("provider_unavailable", type(exc).__name__)], {
|
|
1086
|
+
"entries": (),
|
|
1087
|
+
"retained_start": None,
|
|
1088
|
+
"retained_end": None,
|
|
1089
|
+
"unavailable_cause": WithheldCause.PROVIDER_UNAVAILABLE.value,
|
|
1090
|
+
"codex_threads": (),
|
|
1091
|
+
}
|
|
1092
|
+
retention_params: list[Any] = []
|
|
1093
|
+
retention_sql += _account_predicate(
|
|
1094
|
+
"account_key", scope.account_key, retention_params
|
|
1095
|
+
)
|
|
1096
|
+
try:
|
|
1097
|
+
low, high = _execute(conn, retention_sql, retention_params)[0]
|
|
1098
|
+
except (sqlite3.Error, IndexError):
|
|
1099
|
+
low = high = None
|
|
1100
|
+
return list(digest_rows), {
|
|
1101
|
+
"entries": tuple(entries),
|
|
1102
|
+
"retained_start": _parse_ts(low),
|
|
1103
|
+
"retained_end": _parse_ts(high),
|
|
1104
|
+
"codex_threads": tuple(thread_rows),
|
|
1105
|
+
}
|
|
1106
|
+
|
|
1107
|
+
|
|
1108
|
+
def _read_codex_threads(conn: sqlite3.Connection,
|
|
1109
|
+
entries: Sequence[AccountingEntry]) -> list:
|
|
1110
|
+
"""The thread rows the Codex fan-out predicate resolves over.
|
|
1111
|
+
|
|
1112
|
+
Seeded from the scoped in-window accounting keys, never by walking the
|
|
1113
|
+
retained thread graph, and widened once to the candidate PARENTS the
|
|
1114
|
+
scoped children point at — because a parent with no in-window spend is
|
|
1115
|
+
still the thing a child has to resolve against.
|
|
1116
|
+
|
|
1117
|
+
DEDUPED by `conversation_key`, the table's primary key. The two reads
|
|
1118
|
+
overlap whenever a scoped conversation is also somebody's parent, and a
|
|
1119
|
+
row handed in twice presents itself as two matches under the complete
|
|
1120
|
+
identity, which is exactly the condition that makes a child unallocated.
|
|
1121
|
+
"""
|
|
1122
|
+
scoped = sorted({entry.session_key for entry in entries
|
|
1123
|
+
if entry.session_key})
|
|
1124
|
+
if not scoped:
|
|
1125
|
+
return []
|
|
1126
|
+
rows: dict[str, Any] = {}
|
|
1127
|
+
for row in _codex_threads_for(conn, _CODEX_THREADS_BY_KEY_SQL, scoped):
|
|
1128
|
+
rows.setdefault(str(row["conversation_key"]), row)
|
|
1129
|
+
parents = sorted({str(row["parent_thread_id"]) for row in rows.values()
|
|
1130
|
+
if row["parent_thread_id"]})
|
|
1131
|
+
if parents:
|
|
1132
|
+
for row in _codex_threads_for(conn, _CODEX_THREADS_BY_NATIVE_SQL,
|
|
1133
|
+
parents):
|
|
1134
|
+
rows.setdefault(str(row["conversation_key"]), row)
|
|
1135
|
+
return [rows[key] for key in sorted(rows)]
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
def _codex_thread_digest_rows(thread_rows: Sequence[Any]) -> list[tuple]:
|
|
1139
|
+
"""The facts read, in deterministic key order.
|
|
1140
|
+
|
|
1141
|
+
`conversation_key`, `native_thread_id` and `parent_thread_id` are opaque
|
|
1142
|
+
provider keys and never reach a row field, an evidence value or the wire;
|
|
1143
|
+
they are here because this digest describes the facts READ, which is a
|
|
1144
|
+
different contract from the conversations component's.
|
|
1145
|
+
"""
|
|
1146
|
+
return [
|
|
1147
|
+
(str(row["conversation_key"]), str(row["source_root_key"] or ""),
|
|
1148
|
+
str(row["native_thread_id"] or ""),
|
|
1149
|
+
str(row["root_thread_id"] or ""),
|
|
1150
|
+
str(row["parent_thread_id"] or ""),
|
|
1151
|
+
row["context_window"])
|
|
1152
|
+
for row in thread_rows
|
|
1153
|
+
]
|
|
1154
|
+
|
|
1155
|
+
|
|
1156
|
+
def _read_store_horizon(scope: DiagnosisScope,
|
|
1157
|
+
conn: sqlite3.Connection) -> tuple[dt.datetime | None,
|
|
1158
|
+
object]:
|
|
1159
|
+
"""The earliest provider-native block this install ever recorded.
|
|
1160
|
+
|
|
1161
|
+
Unbounded by the requested window on purpose. `_retention_coverage` uses
|
|
1162
|
+
it as the low bound of what the store could ever have answered for.
|
|
1163
|
+
|
|
1164
|
+
The Codex form reads `quota_window_blocks` behind its own `except`, so it
|
|
1165
|
+
gates the quota projection itself rather than relying on `_read_codex_blocks`
|
|
1166
|
+
having run first. Call order is not a property the enumeration guard can
|
|
1167
|
+
check, and both functions execute module-level SQL that the guard sees as
|
|
1168
|
+
one `<module>` site.
|
|
1169
|
+
"""
|
|
1170
|
+
if scope.source == "codex":
|
|
1171
|
+
quota = _cctally()._load_sibling("_cctally_quota")
|
|
1172
|
+
try:
|
|
1173
|
+
quota.assert_projection_readable(conn)
|
|
1174
|
+
except quota.QuotaProjectionIncomplete as exc:
|
|
1175
|
+
raise EstablishmentFailure(
|
|
1176
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
1177
|
+
f"the Codex quota projection is incomplete: {exc}",
|
|
1178
|
+
) from exc
|
|
1179
|
+
sql = (_CODEX_HORIZON_SQL if scope.source == "codex"
|
|
1180
|
+
else _CLAUDE_HORIZON_SQL)
|
|
1181
|
+
params: list[Any] = []
|
|
1182
|
+
sql += _account_predicate("account_key", scope.account_key, params)
|
|
1183
|
+
try:
|
|
1184
|
+
raw = _execute(conn, sql, params)[0][0]
|
|
1185
|
+
except (sqlite3.Error, IndexError):
|
|
1186
|
+
return None, None
|
|
1187
|
+
return _parse_ts(raw), raw
|
|
1188
|
+
|
|
1189
|
+
|
|
1190
|
+
def _read_stats_component(scope: DiagnosisScope,
|
|
1191
|
+
bundle: StoreBundle) -> tuple[list, Any]:
|
|
1192
|
+
conn = bundle.connection("stats")
|
|
1193
|
+
if scope.source == "codex":
|
|
1194
|
+
blocks, digest_rows = _read_codex_blocks(scope, conn)
|
|
1195
|
+
else:
|
|
1196
|
+
blocks, digest_rows = _read_claude_blocks(scope, conn)
|
|
1197
|
+
horizon, raw_horizon = _read_store_horizon(scope, conn)
|
|
1198
|
+
# The horizon is part of the facts the report rests on, so it belongs in
|
|
1199
|
+
# the digest: a mutation that moved it must move this component.
|
|
1200
|
+
digest_rows = list(digest_rows) + [("horizon", raw_horizon)]
|
|
1201
|
+
return digest_rows, {"blocks": tuple(blocks), "store_horizon": horizon}
|
|
1202
|
+
|
|
1203
|
+
|
|
1204
|
+
# --- Codex fan-out resolution (#620 S3 X4) ------------------------------
|
|
1205
|
+
#
|
|
1206
|
+
# New code rather than an extraction: nothing existing resolves a Codex parent
|
|
1207
|
+
# set-wise over prefetched rows, and the closest thing that does exist is the
|
|
1208
|
+
# per-child expansion this deliberately does not reuse.
|
|
1209
|
+
|
|
1210
|
+
# The ONLY two Codex origin categories this tree can attribute a meaning to.
|
|
1211
|
+
CODEX_ORIGIN_MAIN: str = "user"
|
|
1212
|
+
CODEX_ORIGIN_SUBAGENT: str = "subagent"
|
|
1213
|
+
_CODEX_ORIGIN_CATEGORIES = frozenset({CODEX_ORIGIN_MAIN, CODEX_ORIGIN_SUBAGENT})
|
|
1214
|
+
|
|
1215
|
+
|
|
1216
|
+
def codex_origin_category(value: object) -> str | None:
|
|
1217
|
+
"""The origin category of a `root_thread_id`, or `None` when ambiguous.
|
|
1218
|
+
|
|
1219
|
+
`_inferred_codex_thread_source` returns an explicit `thread_source`
|
|
1220
|
+
VERBATIM, whatever string it is, and no resolver, vocabulary or authority
|
|
1221
|
+
in this tree maps an arbitrary string onto an origin — its own docstring
|
|
1222
|
+
records `source: "vscode"` co-occurring with `thread_source: "user"`. Only
|
|
1223
|
+
the two exact literals are recognised; every other value belongs to
|
|
1224
|
+
neither population, lowers `evaluabilityCoverage` and carries the
|
|
1225
|
+
identifiable-subset qualification.
|
|
1226
|
+
"""
|
|
1227
|
+
if isinstance(value, str) and value in _CODEX_ORIGIN_CATEGORIES:
|
|
1228
|
+
return value
|
|
1229
|
+
return None
|
|
1230
|
+
|
|
1231
|
+
|
|
1232
|
+
@dataclass(frozen=True)
|
|
1233
|
+
class CodexFanout:
|
|
1234
|
+
"""Resolved parents, and the children that could not be resolved."""
|
|
1235
|
+
|
|
1236
|
+
# conversation_key -> (source_root_key, parent native_thread_id)
|
|
1237
|
+
parents: Mapping[str, tuple[str, str]]
|
|
1238
|
+
# conversation_keys that failed evaluation and lower evaluability coverage
|
|
1239
|
+
unallocated: tuple[str, ...]
|
|
1240
|
+
# conversation_keys whose origin category is neither of the two literals
|
|
1241
|
+
# this tree can attribute a meaning to. They belong to NEITHER population
|
|
1242
|
+
# and the predicate could not be decided for them, so they lower
|
|
1243
|
+
# `evaluabilityCoverage` exactly as an unresolvable parent does. Separate
|
|
1244
|
+
# from `unallocated`, because "we could not read the category" and "we
|
|
1245
|
+
# read it and could not resolve the parent" are different statements.
|
|
1246
|
+
ambiguous: tuple[str, ...] = ()
|
|
1247
|
+
|
|
1248
|
+
|
|
1249
|
+
def _thread_field(row: Any, name: str) -> Any:
|
|
1250
|
+
if isinstance(row, Mapping):
|
|
1251
|
+
return row.get(name)
|
|
1252
|
+
return row[name]
|
|
1253
|
+
|
|
1254
|
+
|
|
1255
|
+
def resolve_codex_fanout(thread_rows: Sequence[Any],
|
|
1256
|
+
entries: Sequence[str]) -> CodexFanout:
|
|
1257
|
+
"""Resolve each scoped child conversation to exactly one parent thread.
|
|
1258
|
+
|
|
1259
|
+
**"Resolvable parent" means exactly ONE non-self match under the complete
|
|
1260
|
+
identity.** The table's uniqueness is
|
|
1261
|
+
`(source_root_key, root_thread_id, native_thread_id)`, not the pair, and
|
|
1262
|
+
the existing per-child resolution queries the weaker pair and takes an
|
|
1263
|
+
arbitrary `fetchone()` — which this must not copy. Zero or several matches
|
|
1264
|
+
make the child unallocated and lower `evaluabilityCoverage`.
|
|
1265
|
+
|
|
1266
|
+
Set-wise over prefetched rows and seeded from the scoped in-window
|
|
1267
|
+
accounting keys in `entries`, never by walking the retained thread graph.
|
|
1268
|
+
"""
|
|
1269
|
+
scoped = [key for key in dict.fromkeys(entries) if key]
|
|
1270
|
+
by_native: dict[tuple[str, str], list[Any]] = {}
|
|
1271
|
+
by_key: dict[str, Any] = {}
|
|
1272
|
+
for row in thread_rows:
|
|
1273
|
+
root = str(_thread_field(row, "source_root_key") or "")
|
|
1274
|
+
native = _thread_field(row, "native_thread_id")
|
|
1275
|
+
key = _thread_field(row, "conversation_key")
|
|
1276
|
+
if native is not None:
|
|
1277
|
+
by_native.setdefault((root, str(native)), []).append(row)
|
|
1278
|
+
if key is not None:
|
|
1279
|
+
by_key.setdefault(str(key), row)
|
|
1280
|
+
|
|
1281
|
+
parents: dict[str, tuple[str, str]] = {}
|
|
1282
|
+
unallocated: list[str] = []
|
|
1283
|
+
ambiguous: list[str] = []
|
|
1284
|
+
for key in scoped:
|
|
1285
|
+
row = by_key.get(key)
|
|
1286
|
+
if row is None:
|
|
1287
|
+
ambiguous.append(key)
|
|
1288
|
+
continue
|
|
1289
|
+
category = codex_origin_category(_thread_field(row, "root_thread_id"))
|
|
1290
|
+
if category is None:
|
|
1291
|
+
# Neither of the two literals this tree can attribute a meaning
|
|
1292
|
+
# to, so the predicate could not be decided for this child.
|
|
1293
|
+
ambiguous.append(key)
|
|
1294
|
+
continue
|
|
1295
|
+
if category != CODEX_ORIGIN_SUBAGENT:
|
|
1296
|
+
# Decided, and not a fan-out member: a main-thread conversation.
|
|
1297
|
+
continue
|
|
1298
|
+
root = str(_thread_field(row, "source_root_key") or "")
|
|
1299
|
+
parent = _thread_field(row, "parent_thread_id")
|
|
1300
|
+
native = _thread_field(row, "native_thread_id")
|
|
1301
|
+
if parent is None:
|
|
1302
|
+
unallocated.append(key)
|
|
1303
|
+
continue
|
|
1304
|
+
candidates = [
|
|
1305
|
+
candidate for candidate in by_native.get((root, str(parent)), ())
|
|
1306
|
+
if str(_thread_field(candidate, "native_thread_id")) != str(native)
|
|
1307
|
+
]
|
|
1308
|
+
if len(candidates) != 1:
|
|
1309
|
+
unallocated.append(key)
|
|
1310
|
+
continue
|
|
1311
|
+
parents[key] = (root, str(parent))
|
|
1312
|
+
return CodexFanout(parents=parents, unallocated=tuple(unallocated),
|
|
1313
|
+
ambiguous=tuple(ambiguous))
|
|
1314
|
+
|
|
1315
|
+
|
|
1316
|
+
# --- conversations: the semantic projection (#620 S3 §4.4, §4.5) --------
|
|
1317
|
+
|
|
1318
|
+
def allocate_scan_budget(keys: Sequence[Any], budget: int) -> dict[Any, int]:
|
|
1319
|
+
"""Split one scan budget into equal per-subject shares, in key order.
|
|
1320
|
+
|
|
1321
|
+
A single long subject must not consume the whole budget and starve every
|
|
1322
|
+
subject after it, which an unallocated global budget allows and which
|
|
1323
|
+
would make the report depend on read order. The remainder is distributed
|
|
1324
|
+
in the same key order, so the allocation is a function of the key set
|
|
1325
|
+
alone.
|
|
1326
|
+
"""
|
|
1327
|
+
ordered = list(keys)
|
|
1328
|
+
if not ordered:
|
|
1329
|
+
return {}
|
|
1330
|
+
share, remainder = divmod(max(0, int(budget)), len(ordered))
|
|
1331
|
+
return {key: share + (1 if index < remainder else 0)
|
|
1332
|
+
for index, key in enumerate(ordered)}
|
|
1333
|
+
|
|
1334
|
+
|
|
1335
|
+
# --- the turn predicate (#620 S3 §2.1) ----------------------------------
|
|
1336
|
+
#
|
|
1337
|
+
# The canonical normalization is REUSED rather than reinterpreted:
|
|
1338
|
+
# `conversation_sessions.msg_count` groups `conversation_messages` with no
|
|
1339
|
+
# `is_sidechain` predicate while subagent files carry the parent's
|
|
1340
|
+
# `session_id`, so it is a sidechain-inclusive physical count and cannot
|
|
1341
|
+
# answer "how many human turns does this conversation have".
|
|
1342
|
+
|
|
1343
|
+
|
|
1344
|
+
def _conversation_query():
|
|
1345
|
+
return _cctally()._load_sibling("_lib_conversation_query")
|
|
1346
|
+
|
|
1347
|
+
|
|
1348
|
+
def _is_compaction_row(text: object, blocks_json: object) -> bool:
|
|
1349
|
+
"""Sentinel inference over the reconstructed body, the read-time authority.
|
|
1350
|
+
|
|
1351
|
+
Delegates to `_lib_conversation_query.is_compaction_row`, which is the ONE
|
|
1352
|
+
place this rule is written. A third restatement had already diverged from
|
|
1353
|
+
the canonical one on whether a malformed block element raises.
|
|
1354
|
+
"""
|
|
1355
|
+
return _conversation_query().is_compaction_row(text, blocks_json)
|
|
1356
|
+
|
|
1357
|
+
|
|
1358
|
+
def _row_field(row: Any, name: str, default: Any = None) -> Any:
|
|
1359
|
+
"""One named column, whether the row is a `sqlite3.Row` or a mapping.
|
|
1360
|
+
|
|
1361
|
+
The predicates are pure and unit-tested over plain dicts, while the
|
|
1362
|
+
evaluators feed them `sqlite3.Row` objects — which index by name but have
|
|
1363
|
+
no `get`.
|
|
1364
|
+
"""
|
|
1365
|
+
if isinstance(row, Mapping):
|
|
1366
|
+
return row.get(name, default)
|
|
1367
|
+
try:
|
|
1368
|
+
value = row[name]
|
|
1369
|
+
except (IndexError, KeyError):
|
|
1370
|
+
return default
|
|
1371
|
+
return default if value is None else value
|
|
1372
|
+
|
|
1373
|
+
|
|
1374
|
+
def _row_blocks(blocks_json: object) -> list:
|
|
1375
|
+
"""The block objects of a row, dropping any element that is not one."""
|
|
1376
|
+
return [b for b in _conversation_query().parse_blocks_json(blocks_json)
|
|
1377
|
+
if isinstance(b, Mapping)]
|
|
1378
|
+
|
|
1379
|
+
|
|
1380
|
+
# The 18-column physical stream `fold_claude_canonical` folds, in its order.
|
|
1381
|
+
# Named here so a caller reading a narrower projection states which columns it
|
|
1382
|
+
# is supplying and which it is defaulting.
|
|
1383
|
+
_CLAUDE_FOLD_COLUMNS = (
|
|
1384
|
+
"id", "uuid", "timestamp_utc", "entry_type", "text", "blocks_json",
|
|
1385
|
+
"model", "msg_id", "req_id", "is_sidechain", "cwd", "git_branch",
|
|
1386
|
+
"source_path", "parent_uuid", "source_tool_use_id", "stop_reason",
|
|
1387
|
+
"attribution_skill", "attribution_plugin",
|
|
1388
|
+
)
|
|
1389
|
+
|
|
1390
|
+
|
|
1391
|
+
def _claude_fold_row(row: Any) -> tuple:
|
|
1392
|
+
return tuple(_row_field(row, name) for name in _CLAUDE_FOLD_COLUMNS)
|
|
1393
|
+
|
|
1394
|
+
|
|
1395
|
+
def _claude_canonical_items(rows: Sequence[Any], *,
|
|
1396
|
+
allow_human_fallback: bool = False) -> list:
|
|
1397
|
+
"""The canonical items of ONE session, through the extracted fold.
|
|
1398
|
+
|
|
1399
|
+
The normalization is REUSED rather than restated. `fold_claude_canonical`
|
|
1400
|
+
deduplicates `(session_id, uuid)`, groups assistant fragments
|
|
1401
|
+
non-adjacently by `(msg_id, req_id)`, folds tool results and skill bodies,
|
|
1402
|
+
promotes a slash-command invocation carrying real args to a human turn and
|
|
1403
|
+
applies the meta classification — which is the whole of spec 2.1's Claude
|
|
1404
|
+
predicate. Restating any part of it here is how the deduplication went
|
|
1405
|
+
missing in the first place.
|
|
1406
|
+
|
|
1407
|
+
`rows` must be one session's rows in `(timestamp_utc, id)` order. The token
|
|
1408
|
+
and cost maps are empty because this path publishes no per-turn cost: cost
|
|
1409
|
+
comes from the accounting population the adapter has already read.
|
|
1410
|
+
"""
|
|
1411
|
+
query = _conversation_query()
|
|
1412
|
+
physical = [_claude_fold_row(row) for row in rows]
|
|
1413
|
+
folded = query.fold_claude_canonical(
|
|
1414
|
+
physical, {}, {}, allow_human_fallback=allow_human_fallback)
|
|
1415
|
+
return folded["items"]
|
|
1416
|
+
|
|
1417
|
+
|
|
1418
|
+
def _claude_item_is_human_turn(item: Mapping[str, Any]) -> bool:
|
|
1419
|
+
"""A non-empty final `kind='human'` item on the main thread.
|
|
1420
|
+
|
|
1421
|
+
`is_sidechain` and a non-null `subagent_key` both disqualify: subagent
|
|
1422
|
+
files carry the PARENT's `session_id`, which is why
|
|
1423
|
+
`conversation_sessions.msg_count` is a sidechain-inclusive physical count
|
|
1424
|
+
and cannot answer how many human turns a conversation has.
|
|
1425
|
+
"""
|
|
1426
|
+
if item.get("kind") != "human":
|
|
1427
|
+
return False
|
|
1428
|
+
if item.get("is_sidechain"):
|
|
1429
|
+
return False
|
|
1430
|
+
if item.get("subagent_key") is not None:
|
|
1431
|
+
return False
|
|
1432
|
+
return bool((item.get("text") or "").strip())
|
|
1433
|
+
|
|
1434
|
+
|
|
1435
|
+
def _claude_human_turn_items(items: Sequence[Mapping[str, Any]]) -> list:
|
|
1436
|
+
"""The main-thread human turns among already-folded canonical items."""
|
|
1437
|
+
return [item for item in items if _claude_item_is_human_turn(item)]
|
|
1438
|
+
|
|
1439
|
+
|
|
1440
|
+
@dataclass
|
|
1441
|
+
class _ClaudeAssociation:
|
|
1442
|
+
"""One qualifying human turn and the assistant turns that answered it."""
|
|
1443
|
+
|
|
1444
|
+
human: Mapping[str, Any]
|
|
1445
|
+
replies: list
|
|
1446
|
+
|
|
1447
|
+
|
|
1448
|
+
def _claude_associate_items(items: Sequence[Mapping[str, Any]]) -> list:
|
|
1449
|
+
"""Associate each main-thread assistant turn with its nearest preceding
|
|
1450
|
+
qualifying human, over ONE session's already-folded items.
|
|
1451
|
+
|
|
1452
|
+
Several assistant turns may belong to one human, and an assistant turn
|
|
1453
|
+
preceding any qualifying human is ORPHANED: it contributes no synthetic
|
|
1454
|
+
human turn and belongs to no association. Both sides come from the same
|
|
1455
|
+
canonical fold, so an assistant turn is one item per `(msg_id, req_id)`
|
|
1456
|
+
however many physical fragments it arrived in.
|
|
1457
|
+
"""
|
|
1458
|
+
associations: list = []
|
|
1459
|
+
current: int | None = None
|
|
1460
|
+
for item in items:
|
|
1461
|
+
if _claude_item_is_human_turn(item):
|
|
1462
|
+
associations.append(_ClaudeAssociation(human=item, replies=[]))
|
|
1463
|
+
current = len(associations) - 1
|
|
1464
|
+
continue
|
|
1465
|
+
if item.get("kind") != "assistant":
|
|
1466
|
+
continue
|
|
1467
|
+
if item.get("is_sidechain") or item.get("subagent_key") is not None:
|
|
1468
|
+
continue
|
|
1469
|
+
if current is None:
|
|
1470
|
+
continue # orphaned: no qualifying human yet
|
|
1471
|
+
associations[current].replies.append(item)
|
|
1472
|
+
return associations
|
|
1473
|
+
|
|
1474
|
+
|
|
1475
|
+
def _codex_origin(root_thread_id: object) -> str:
|
|
1476
|
+
"""`main`, `delegated`, or `ambiguous`.
|
|
1477
|
+
|
|
1478
|
+
`_inferred_codex_thread_source` returns any explicit `thread_source`
|
|
1479
|
+
VERBATIM, so there is no vocabulary to resolve an arbitrary string
|
|
1480
|
+
against. Only these two literals carry a meaning this tree can defend;
|
|
1481
|
+
every other value belongs to neither population, lowers
|
|
1482
|
+
`evaluabilityCoverage` and carries the identifiable-subset qualification.
|
|
1483
|
+
"""
|
|
1484
|
+
category = codex_origin_category(root_thread_id)
|
|
1485
|
+
if category == CODEX_ORIGIN_MAIN:
|
|
1486
|
+
return "main"
|
|
1487
|
+
if category == CODEX_ORIGIN_SUBAGENT:
|
|
1488
|
+
return "delegated"
|
|
1489
|
+
return "ambiguous"
|
|
1490
|
+
|
|
1491
|
+
|
|
1492
|
+
# The `codex_conversation_messages` columns a canonical row needs. `text` is
|
|
1493
|
+
# read because "a human turn is a NON-EMPTY canonical prompt" and emptiness is
|
|
1494
|
+
# a property of the body; nothing else about the body is used, and no part of
|
|
1495
|
+
# it reaches a row field, an evidence value or the digest.
|
|
1496
|
+
_CODEX_ROW_COLUMNS = (
|
|
1497
|
+
"conversation_key", "source_root_key", "source_path", "line_offset",
|
|
1498
|
+
"timestamp_utc", "turn_id", "call_id", "kind", "event_type",
|
|
1499
|
+
"record_family", "model", "text", "content_digest", "content_len",
|
|
1500
|
+
"detail_json", "search_tool", "search_thinking",
|
|
1501
|
+
)
|
|
1502
|
+
|
|
1503
|
+
|
|
1504
|
+
def _codex_normalized_rows(rows: Sequence[Any]) -> list:
|
|
1505
|
+
"""Physical `codex_conversation_messages` rows as canonical row objects."""
|
|
1506
|
+
codex_kernel = _cctally()._load_sibling("_lib_codex_conversation")
|
|
1507
|
+
built = []
|
|
1508
|
+
for row in rows:
|
|
1509
|
+
values = {name: _row_field(row, name) for name in _CODEX_ROW_COLUMNS}
|
|
1510
|
+
built.append(codex_kernel.CodexNormalizedRow(
|
|
1511
|
+
conversation_key=str(values["conversation_key"] or ""),
|
|
1512
|
+
source_root_key=str(values["source_root_key"] or ""),
|
|
1513
|
+
source_path=str(values["source_path"] or ""),
|
|
1514
|
+
line_offset=int(values["line_offset"] or 0),
|
|
1515
|
+
timestamp_utc=values["timestamp_utc"],
|
|
1516
|
+
turn_id=values["turn_id"],
|
|
1517
|
+
call_id=values["call_id"],
|
|
1518
|
+
kind=str(values["kind"] or ""),
|
|
1519
|
+
event_type=values["event_type"],
|
|
1520
|
+
record_family=str(values["record_family"] or ""),
|
|
1521
|
+
model=values["model"],
|
|
1522
|
+
text=str(values["text"] or ""),
|
|
1523
|
+
content_digest=str(values["content_digest"] or ""),
|
|
1524
|
+
content_len=int(values["content_len"] or 0),
|
|
1525
|
+
detail_json=values["detail_json"],
|
|
1526
|
+
search_tool=str(values["search_tool"] or ""),
|
|
1527
|
+
search_thinking=str(values["search_thinking"] or ""),
|
|
1528
|
+
))
|
|
1529
|
+
return built
|
|
1530
|
+
|
|
1531
|
+
|
|
1532
|
+
def _codex_human_turns(rows: Sequence[Any]) -> int:
|
|
1533
|
+
"""How many canonical human turns ONE Codex conversation holds.
|
|
1534
|
+
|
|
1535
|
+
A physical `COUNT(*)` over `kind='user'` over-counts on three independent
|
|
1536
|
+
axes, and each one silently drops a short conversation by calling it long:
|
|
1537
|
+
a turn-less user row canonicalizes to `unturned` rather than `prompt`, the
|
|
1538
|
+
`event_msg` member of a digest-exact mirror pair is a duplicate of its
|
|
1539
|
+
`response_item` partner, and an empty body is not a turn at all.
|
|
1540
|
+
|
|
1541
|
+
So the count goes through the same normalization
|
|
1542
|
+
`bin/_lib_codex_conversation.py` performs — `pair_mirrors` then
|
|
1543
|
+
`canonical_items` — restricted to `kind='user'` rows. That restriction is
|
|
1544
|
+
exact rather than approximate: mirror pairing groups by
|
|
1545
|
+
`(turn_id, kind, digest, len)` and tracks its unturned adjacency per kind,
|
|
1546
|
+
and the reasoning-containment pass touches `kind='reasoning'` only, so no
|
|
1547
|
+
pairing decision about a user row depends on a row of another kind.
|
|
1548
|
+
"""
|
|
1549
|
+
codex_kernel = _cctally()._load_sibling("_lib_codex_conversation")
|
|
1550
|
+
kept, _suppressed = codex_kernel.pair_mirrors(_codex_normalized_rows(rows))
|
|
1551
|
+
items = codex_kernel.canonical_items(kept)
|
|
1552
|
+
return sum(1 for item in items
|
|
1553
|
+
if item["klass"] == "prompt"
|
|
1554
|
+
and (item["anchor_row"].text or "").strip())
|
|
1555
|
+
|
|
1556
|
+
|
|
1557
|
+
def _median_turns(values: Sequence[int]) -> int | None:
|
|
1558
|
+
"""The LOWER median, so an even count publishes a real member value."""
|
|
1559
|
+
ordered = sorted(values)
|
|
1560
|
+
if not ordered:
|
|
1561
|
+
return None
|
|
1562
|
+
return ordered[(len(ordered) - 1) // 2]
|
|
1563
|
+
|
|
1564
|
+
|
|
1565
|
+
def _claude_context_window(model: object) -> int | None:
|
|
1566
|
+
"""This request's own model context window, or `None` when unknown.
|
|
1567
|
+
|
|
1568
|
+
Capacities come from the statusline's table, not from the pricing layer:
|
|
1569
|
+
a model can be priced and still have no published window, and the two
|
|
1570
|
+
tables answer different questions.
|
|
1571
|
+
"""
|
|
1572
|
+
statusline = _cctally()._load_sibling("_cctally_statusline")
|
|
1573
|
+
return statusline._resolve_context_window(model, lambda *_a, **_k: None)
|
|
1574
|
+
|
|
1575
|
+
|
|
1576
|
+
# --- the reads the evaluators run (#620 S3 §4.2) ------------------------
|
|
1577
|
+
#
|
|
1578
|
+
# Every statement below is SET-WISE and chunked. A per-session or per-file
|
|
1579
|
+
# statement is the N+1 spec §4.2 exists to forbid, and it is multiplied by two
|
|
1580
|
+
# bundles and up to two probe attempts.
|
|
1581
|
+
|
|
1582
|
+
# SQLite's compiled-parameter ceiling is far higher, but a bounded chunk keeps
|
|
1583
|
+
# one very wide window from building a statement no plan can reuse.
|
|
1584
|
+
_IN_CHUNK = 400
|
|
1585
|
+
|
|
1586
|
+
_CHURN_ROW_COLUMNS = ("id", "uuid", "entry_type", "text", "blocks_json",
|
|
1587
|
+
"model", "msg_id", "req_id", "source_path")
|
|
1588
|
+
|
|
1589
|
+
_CLAUDE_WINDOW_SESSIONS_SQL = """
|
|
1590
|
+
SELECT DISTINCT session_id FROM conversation_messages
|
|
1591
|
+
WHERE timestamp_utc >= ? AND timestamp_utc < ? AND session_id IS NOT NULL
|
|
1592
|
+
"""
|
|
1593
|
+
|
|
1594
|
+
# The seed prefix: the LAST `share` rows before the window, ranked descending
|
|
1595
|
+
# so SQLite applies the budget rather than the reader fetching everything and
|
|
1596
|
+
# declining to parse some of it. Ordered back to document order by `rank DESC`.
|
|
1597
|
+
_CLAUDE_SEED_PREFIX_SQL = """
|
|
1598
|
+
SELECT session_id, id, uuid, entry_type, text, blocks_json, model,
|
|
1599
|
+
msg_id, req_id, source_path, rank FROM (
|
|
1600
|
+
SELECT session_id, id, uuid, entry_type, text, blocks_json, model,
|
|
1601
|
+
msg_id, req_id, source_path,
|
|
1602
|
+
ROW_NUMBER() OVER (PARTITION BY session_id
|
|
1603
|
+
ORDER BY timestamp_utc DESC, id DESC) AS rank
|
|
1604
|
+
FROM conversation_messages
|
|
1605
|
+
WHERE session_id IN ({placeholders}) AND timestamp_utc < ?
|
|
1606
|
+
) WHERE rank <= ?
|
|
1607
|
+
ORDER BY session_id ASC, rank DESC
|
|
1608
|
+
"""
|
|
1609
|
+
|
|
1610
|
+
_CLAUDE_WINDOW_ROWS_SQL = """
|
|
1611
|
+
SELECT session_id, id, uuid, entry_type, text, blocks_json, model,
|
|
1612
|
+
msg_id, req_id, source_path
|
|
1613
|
+
FROM conversation_messages
|
|
1614
|
+
WHERE session_id IN ({placeholders})
|
|
1615
|
+
AND timestamp_utc >= ? AND timestamp_utc < ?
|
|
1616
|
+
ORDER BY session_id ASC, timestamp_utc ASC, id ASC
|
|
1617
|
+
"""
|
|
1618
|
+
|
|
1619
|
+
# The human-candidate population spans the WHOLE retained conversation,
|
|
1620
|
+
# because "short" is a property of the conversation and a window-clipped count
|
|
1621
|
+
# would call a long conversation short whenever the window caught only its
|
|
1622
|
+
# tail. `entry_type` is not decisive — command normalization can promote a
|
|
1623
|
+
# `meta` row to human — so each candidate costs a normalization, which is what
|
|
1624
|
+
# `DIAGNOSIS_CONVERSATION_NORMALIZE_BUDGET_ROWS` bounds.
|
|
1625
|
+
# The 18 fold columns plus `session_id` and the budget rank. The whole stream
|
|
1626
|
+
# is fed to `fold_claude_canonical`, which is the ONE statement of the Claude
|
|
1627
|
+
# turn normalization, so the projection carries what that fold reads rather
|
|
1628
|
+
# than a narrower set a second normalization would have needed.
|
|
1629
|
+
_CLAUDE_TURN_CANDIDATE_SQL = """
|
|
1630
|
+
SELECT session_id, id, uuid, timestamp_utc, entry_type, text, blocks_json,
|
|
1631
|
+
model, msg_id, req_id, is_sidechain, cwd, git_branch, source_path,
|
|
1632
|
+
parent_uuid, source_tool_use_id, stop_reason, attribution_skill,
|
|
1633
|
+
attribution_plugin, rank FROM (
|
|
1634
|
+
SELECT session_id, id, uuid, timestamp_utc, entry_type, text,
|
|
1635
|
+
blocks_json, model, msg_id, req_id, is_sidechain, cwd,
|
|
1636
|
+
git_branch, source_path, parent_uuid, source_tool_use_id,
|
|
1637
|
+
stop_reason, attribution_skill, attribution_plugin,
|
|
1638
|
+
ROW_NUMBER() OVER (PARTITION BY session_id
|
|
1639
|
+
ORDER BY timestamp_utc ASC, id ASC) AS rank
|
|
1640
|
+
FROM conversation_messages
|
|
1641
|
+
WHERE session_id IN ({placeholders})
|
|
1642
|
+
AND entry_type IN ('meta', 'human')
|
|
1643
|
+
AND (is_sidechain IS NULL OR is_sidechain = 0)
|
|
1644
|
+
) WHERE rank <= ?
|
|
1645
|
+
ORDER BY session_id ASC, rank ASC
|
|
1646
|
+
"""
|
|
1647
|
+
|
|
1648
|
+
# In-window assistant turn keys. The Claude association is a STORED key on
|
|
1649
|
+
# both sides, so the context-window fraction is derived from the accounting
|
|
1650
|
+
# population the adapter has already loaded, at a cost bounded by the window's
|
|
1651
|
+
# accounting size and independent of conversation length.
|
|
1652
|
+
_CLAUDE_WINDOW_ASSISTANT_SQL = """
|
|
1653
|
+
SELECT session_id, id, uuid, timestamp_utc, entry_type, text, blocks_json,
|
|
1654
|
+
model, msg_id, req_id, is_sidechain, cwd, git_branch, source_path,
|
|
1655
|
+
parent_uuid, source_tool_use_id, stop_reason, attribution_skill,
|
|
1656
|
+
attribution_plugin
|
|
1657
|
+
FROM conversation_messages
|
|
1658
|
+
WHERE session_id IN ({placeholders}) AND entry_type = 'assistant'
|
|
1659
|
+
AND msg_id IS NOT NULL
|
|
1660
|
+
AND timestamp_utc >= ? AND timestamp_utc < ?
|
|
1661
|
+
ORDER BY session_id ASC, timestamp_utc ASC, id ASC
|
|
1662
|
+
"""
|
|
1663
|
+
|
|
1664
|
+
# The fan-out path's own projection. It derives a bucket from the source path
|
|
1665
|
+
# and joins on the turn key, so it needs five columns and no body at all —
|
|
1666
|
+
# where the statement above carries the nineteen that feed the canonical fold.
|
|
1667
|
+
# `blocks_json` is the largest column in the table on a real store, and
|
|
1668
|
+
# loading every in-window assistant turn's body to read five columns is a cost
|
|
1669
|
+
# the fan-out class never asked for. The ORDER BY is the same, because the
|
|
1670
|
+
# deduplication keeps the FIRST occurrence per uuid and therefore depends on
|
|
1671
|
+
# it; `timestamp_utc` and `id` order the result without being selected.
|
|
1672
|
+
_CLAUDE_WINDOW_SUBAGENT_SQL = """
|
|
1673
|
+
SELECT session_id, uuid, source_path, msg_id, req_id
|
|
1674
|
+
FROM conversation_messages
|
|
1675
|
+
WHERE session_id IN ({placeholders}) AND entry_type = 'assistant'
|
|
1676
|
+
AND msg_id IS NOT NULL
|
|
1677
|
+
AND timestamp_utc >= ? AND timestamp_utc < ?
|
|
1678
|
+
ORDER BY session_id ASC, timestamp_utc ASC, id ASC
|
|
1679
|
+
"""
|
|
1680
|
+
|
|
1681
|
+
# The canonical prompt candidates. A physical `COUNT(*)` cannot answer how
|
|
1682
|
+
# many human turns a Codex conversation holds — spec 2.1 requires canonical
|
|
1683
|
+
# `klass='prompt'` items formed over mirror-paired rows — so the rows
|
|
1684
|
+
# themselves are read and normalized, under the same per-conversation
|
|
1685
|
+
# normalize budget the Claude predicate carries, because canonicalization
|
|
1686
|
+
# costs more than counting.
|
|
1687
|
+
_CODEX_PROMPT_CANDIDATE_SQL = """
|
|
1688
|
+
SELECT conversation_key, source_root_key, source_path, line_offset,
|
|
1689
|
+
timestamp_utc, turn_id, call_id, kind, event_type, record_family,
|
|
1690
|
+
model, text, content_digest, content_len, detail_json, search_tool,
|
|
1691
|
+
search_thinking, rank FROM (
|
|
1692
|
+
SELECT conversation_key, source_root_key, source_path, line_offset,
|
|
1693
|
+
timestamp_utc, turn_id, call_id, kind, event_type,
|
|
1694
|
+
record_family, model, text, content_digest, content_len,
|
|
1695
|
+
detail_json, search_tool, search_thinking,
|
|
1696
|
+
ROW_NUMBER() OVER (PARTITION BY conversation_key
|
|
1697
|
+
ORDER BY timestamp_utc ASC,
|
|
1698
|
+
source_path ASC,
|
|
1699
|
+
line_offset ASC) AS rank
|
|
1700
|
+
FROM codex_conversation_messages
|
|
1701
|
+
WHERE conversation_key IN ({placeholders}) AND kind = 'user'
|
|
1702
|
+
) WHERE rank <= ?
|
|
1703
|
+
ORDER BY conversation_key ASC, rank ASC
|
|
1704
|
+
"""
|
|
1705
|
+
|
|
1706
|
+
_CODEX_EVENT_COLUMNS = (
|
|
1707
|
+
"source_path", "line_offset", "source_root_key", "conversation_key",
|
|
1708
|
+
"native_thread_id", "root_thread_id", "parent_thread_id", "timestamp_utc",
|
|
1709
|
+
"record_type", "event_type", "turn_id", "call_id", "payload_json",
|
|
1710
|
+
)
|
|
1711
|
+
|
|
1712
|
+
# `infer_codex_event_turns` materializes a file's WHOLE sequence, so an
|
|
1713
|
+
# arbitrarily long pre-window history is scanned to attribute one in-window
|
|
1714
|
+
# entry. The budget is allocated per source file so no file starves another,
|
|
1715
|
+
# and a file that exhausts its share yields NO turn map at all — a truncated
|
|
1716
|
+
# read would produce a wrong map rather than a missing one.
|
|
1717
|
+
_CODEX_EVENTS_SQL = """
|
|
1718
|
+
SELECT source_path, line_offset, source_root_key, conversation_key,
|
|
1719
|
+
native_thread_id, root_thread_id, parent_thread_id, timestamp_utc,
|
|
1720
|
+
record_type, event_type, turn_id, call_id, payload_json, rank FROM (
|
|
1721
|
+
SELECT source_path, line_offset, source_root_key, conversation_key,
|
|
1722
|
+
native_thread_id, root_thread_id, parent_thread_id,
|
|
1723
|
+
timestamp_utc, record_type, event_type, turn_id, call_id,
|
|
1724
|
+
payload_json,
|
|
1725
|
+
ROW_NUMBER() OVER (PARTITION BY source_path
|
|
1726
|
+
ORDER BY line_offset ASC) AS rank
|
|
1727
|
+
FROM codex_conversation_events
|
|
1728
|
+
WHERE source_path IN ({placeholders})
|
|
1729
|
+
AND (record_type IN ('session_meta', 'turn_context')
|
|
1730
|
+
OR turn_id IS NOT NULL
|
|
1731
|
+
OR event_type = 'token_count')
|
|
1732
|
+
) WHERE rank <= ?
|
|
1733
|
+
ORDER BY source_path ASC, line_offset ASC
|
|
1734
|
+
"""
|
|
1735
|
+
|
|
1736
|
+
_CODEX_THREADS_BY_KEY_SQL = """
|
|
1737
|
+
SELECT conversation_key, source_root_key, native_thread_id,
|
|
1738
|
+
root_thread_id, parent_thread_id, context_window
|
|
1739
|
+
FROM codex_conversation_threads
|
|
1740
|
+
WHERE conversation_key IN ({placeholders})
|
|
1741
|
+
"""
|
|
1742
|
+
|
|
1743
|
+
_CODEX_THREADS_BY_NATIVE_SQL = """
|
|
1744
|
+
SELECT conversation_key, source_root_key, native_thread_id,
|
|
1745
|
+
root_thread_id, parent_thread_id, context_window
|
|
1746
|
+
FROM codex_conversation_threads
|
|
1747
|
+
WHERE native_thread_id IN ({placeholders})
|
|
1748
|
+
"""
|
|
1749
|
+
|
|
1750
|
+
# `cache_create_1h_tokens` is fetched even though the churn predicate reads
|
|
1751
|
+
# only the raw creation and read counts: the #195 rule is that every SELECT
|
|
1752
|
+
# reading `cache_create_tokens` off `session_entries` carries the split
|
|
1753
|
+
# column, because a missed site does not raise and does not move a golden — it
|
|
1754
|
+
# silently under-prices — and this map is one keystroke away from feeding a
|
|
1755
|
+
# cost fold.
|
|
1756
|
+
_CLAUDE_TURN_TOKENS_SQL = (
|
|
1757
|
+
"SELECT msg_id, req_id, cache_create_tokens, cache_read_tokens, "
|
|
1758
|
+
" cache_create_1h_tokens, speed "
|
|
1759
|
+
"FROM session_entries WHERE "
|
|
1760
|
+
)
|
|
1761
|
+
|
|
1762
|
+
|
|
1763
|
+
# The statements each provider runs against the CONVERSATIONS connection.
|
|
1764
|
+
#
|
|
1765
|
+
# This is the schema gate's whole input, and it is the statements themselves
|
|
1766
|
+
# rather than a hand-written list of the tables and columns they name. The
|
|
1767
|
+
# previous gate restated the requirement, nothing referenced it, and a
|
|
1768
|
+
# projection that grew a column would have left the gate passing, the
|
|
1769
|
+
# statement raising, and every affected user told we have a defect on every
|
|
1770
|
+
# report — which is precisely the outcome the gate exists to prevent. A
|
|
1771
|
+
# statement cannot disagree with itself.
|
|
1772
|
+
#
|
|
1773
|
+
# `test_every_claude_conversations_statement_is_declared` and its Codex twin
|
|
1774
|
+
# assert that every statement the evaluators actually run appears here, so a
|
|
1775
|
+
# statement added without being declared fails rather than going un-gated.
|
|
1776
|
+
_CONVERSATIONS_STATEMENTS: dict[str, tuple[str, ...]] = {
|
|
1777
|
+
"claude": (
|
|
1778
|
+
_CLAUDE_WINDOW_SESSIONS_SQL,
|
|
1779
|
+
_CLAUDE_SEED_PREFIX_SQL,
|
|
1780
|
+
_CLAUDE_WINDOW_ROWS_SQL,
|
|
1781
|
+
_CLAUDE_TURN_CANDIDATE_SQL,
|
|
1782
|
+
_CLAUDE_WINDOW_ASSISTANT_SQL,
|
|
1783
|
+
_CLAUDE_WINDOW_SUBAGENT_SQL,
|
|
1784
|
+
),
|
|
1785
|
+
"codex": (
|
|
1786
|
+
_CODEX_PROMPT_CANDIDATE_SQL,
|
|
1787
|
+
_CODEX_EVENTS_SQL,
|
|
1788
|
+
),
|
|
1789
|
+
}
|
|
1790
|
+
|
|
1791
|
+
# The table a statement reads, for the digest row alone. The regex skips the
|
|
1792
|
+
# `FROM (` of a windowed subquery because `(` is not an identifier character,
|
|
1793
|
+
# so it names the physical table in every statement above. A wrong label here
|
|
1794
|
+
# cannot change a verdict — only the word printed beside `component_schema_
|
|
1795
|
+
# incomplete`.
|
|
1796
|
+
_FROM_TABLE_RE = re.compile(r"\bFROM\s+([A-Za-z_][A-Za-z0-9_]*)")
|
|
1797
|
+
|
|
1798
|
+
|
|
1799
|
+
def _statement_table(sql: str) -> str:
|
|
1800
|
+
match = _FROM_TABLE_RE.search(sql)
|
|
1801
|
+
return match.group(1) if match else "conversations"
|
|
1802
|
+
|
|
1803
|
+
|
|
1804
|
+
# `no such table`, `no such column`, `has no column named` and `ambiguous
|
|
1805
|
+
# column name` all describe an object the statement NAMED. Before the schema
|
|
1806
|
+
# gate below one of these means the STORE cannot serve our reads; past it the
|
|
1807
|
+
# store demonstrably carries what our statements name, so one of these is a
|
|
1808
|
+
# defect in our own SQL and reporting it as an absent signal makes it
|
|
1809
|
+
# indistinguishable from a store that genuinely holds nothing.
|
|
1810
|
+
_MISSING_SCHEMA_OBJECT_RE = re.compile(
|
|
1811
|
+
r"no such (?:table|column|index|view|trigger)\b"
|
|
1812
|
+
r"|has no column named\b"
|
|
1813
|
+
r"|ambiguous column name\b"
|
|
1814
|
+
)
|
|
1815
|
+
|
|
1816
|
+
|
|
1817
|
+
def _conversations_schema_gap(conn: sqlite3.Connection,
|
|
1818
|
+
source: str) -> str | None:
|
|
1819
|
+
"""The first table this provider's reads need and the store cannot serve.
|
|
1820
|
+
|
|
1821
|
+
Answered by asking the STORE — every statement is COMPILED against it
|
|
1822
|
+
with `EXPLAIN`, which prepares the statement and runs none of it. That is
|
|
1823
|
+
what makes `_is_store_failure` below able to treat a missing schema object
|
|
1824
|
+
past this point as OUR defect: past this gate the store has been observed
|
|
1825
|
+
to carry every object the statements name, because it compiled them all.
|
|
1826
|
+
|
|
1827
|
+
A statement that fails to compile for any reason OTHER than a missing
|
|
1828
|
+
schema object is not a store shortfall — a window function unsupported by
|
|
1829
|
+
an older SQLite is our own environment, and `database is locked` is the
|
|
1830
|
+
store refusing to answer right now — so that exception propagates rather
|
|
1831
|
+
than being silently reported as an absent signal.
|
|
1832
|
+
|
|
1833
|
+
WHERE it propagates to is the point, and it was nowhere until R13. Both
|
|
1834
|
+
callers now handle it: `_read_conversations_component` calls this INSIDE
|
|
1835
|
+
its own `except Exception` backstop, where `_is_store_failure` splits a
|
|
1836
|
+
store-shaped cause from a defect in our own code and only the three
|
|
1837
|
+
conversation-derived classes are withheld; and `_probe_component`, which
|
|
1838
|
+
runs outside that read and so cannot be moved inside it, catches
|
|
1839
|
+
`sqlite3.Error` and falls back to the wider probe pair. Neither swallows
|
|
1840
|
+
the cause: the probe declines to decide the fold, and the read that
|
|
1841
|
+
follows it asks the same question again and classifies the answer.
|
|
1842
|
+
|
|
1843
|
+
An unrecognised source RAISES `EstablishmentFailure`, which is not a
|
|
1844
|
+
`sqlite3.Error` and therefore passes through the probe's catch and ends
|
|
1845
|
+
the report. Failing open would give a third provider exactly the behaviour
|
|
1846
|
+
this gate declined to allow: a statement raising past a gate that never
|
|
1847
|
+
checked it.
|
|
1848
|
+
"""
|
|
1849
|
+
try:
|
|
1850
|
+
statements = _CONVERSATIONS_STATEMENTS[source]
|
|
1851
|
+
except KeyError as exc:
|
|
1852
|
+
raise EstablishmentFailure(
|
|
1853
|
+
EstablishmentError.STORE_UNAVAILABLE.value,
|
|
1854
|
+
f"no conversations statements are declared for source {source}",
|
|
1855
|
+
) from exc
|
|
1856
|
+
for sql in statements:
|
|
1857
|
+
# One placeholder is enough to compile: `IN (?)` and `IN (?,?)` are the
|
|
1858
|
+
# same statement as far as the schema is concerned.
|
|
1859
|
+
compiled = sql.format(placeholders="?")
|
|
1860
|
+
try:
|
|
1861
|
+
_execute(conn, "EXPLAIN " + compiled,
|
|
1862
|
+
(None,) * compiled.count("?"))
|
|
1863
|
+
except sqlite3.OperationalError as exc:
|
|
1864
|
+
if _MISSING_SCHEMA_OBJECT_RE.search(str(exc)):
|
|
1865
|
+
return _statement_table(sql)
|
|
1866
|
+
raise
|
|
1867
|
+
return None
|
|
1868
|
+
|
|
1869
|
+
|
|
1870
|
+
# Messages a store failure never produces, because each one describes our own
|
|
1871
|
+
# SQL text or the SQLite build we are running on rather than the bytes on
|
|
1872
|
+
# disk. `ROW_NUMBER() OVER (...)` is a syntax error on SQLite before 3.25, and
|
|
1873
|
+
# reporting that as an absent signal tells the user to fix a store that is
|
|
1874
|
+
# perfectly healthy.
|
|
1875
|
+
_OUR_OWN_SQL_RE = re.compile(
|
|
1876
|
+
r"syntax error"
|
|
1877
|
+
r"|no such function\b"
|
|
1878
|
+
r"|no such collation sequence\b"
|
|
1879
|
+
r"|wrong number of arguments\b"
|
|
1880
|
+
r"|misuse of\b"
|
|
1881
|
+
)
|
|
1882
|
+
|
|
1883
|
+
|
|
1884
|
+
def _is_store_failure(exc: BaseException) -> bool:
|
|
1885
|
+
"""Whether a raised exception describes the STORE rather than our code.
|
|
1886
|
+
|
|
1887
|
+
Mapping EVERY `sqlite3.Error` to a store-shaped cause moved the F12
|
|
1888
|
+
conflation rather than removing it: a `ProgrammingError` about the number
|
|
1889
|
+
of bindings, or an `OperationalError` naming a schema object, is a defect
|
|
1890
|
+
in this module and must say so. What remains store-shaped is what a store
|
|
1891
|
+
actually produces — a malformed image, a locked or busy file, a disk I/O
|
|
1892
|
+
error, a read-only or corrupt page — none of which our code can cause.
|
|
1893
|
+
"""
|
|
1894
|
+
if not isinstance(exc, sqlite3.Error):
|
|
1895
|
+
return False
|
|
1896
|
+
if isinstance(exc, (sqlite3.ProgrammingError, sqlite3.InterfaceError)):
|
|
1897
|
+
return False
|
|
1898
|
+
if isinstance(exc, sqlite3.OperationalError):
|
|
1899
|
+
message = str(exc)
|
|
1900
|
+
if (_MISSING_SCHEMA_OBJECT_RE.search(message)
|
|
1901
|
+
or _OUR_OWN_SQL_RE.search(message)):
|
|
1902
|
+
return False
|
|
1903
|
+
return True
|
|
1904
|
+
|
|
1905
|
+
|
|
1906
|
+
def _chunks(values: Sequence[Any], size: int = _IN_CHUNK):
|
|
1907
|
+
for start in range(0, len(values), size):
|
|
1908
|
+
yield values[start:start + size]
|
|
1909
|
+
|
|
1910
|
+
|
|
1911
|
+
def _placeholders(count: int) -> str:
|
|
1912
|
+
return ",".join("?" for _ in range(count))
|
|
1913
|
+
|
|
1914
|
+
|
|
1915
|
+
def _claude_turn_token_map(conn: sqlite3.Connection,
|
|
1916
|
+
keys: Sequence[tuple], account_key: str | None
|
|
1917
|
+
) -> dict[tuple, dict]:
|
|
1918
|
+
"""`{(msg_id, req_id): {"cache_creation", "cache_read", "speed"}}`.
|
|
1919
|
+
|
|
1920
|
+
The cache-churn walk begins BEFORE the window, and those turns are state:
|
|
1921
|
+
their tokens move the running maximum and they contribute no support,
|
|
1922
|
+
coverage, observed USD or denominator. They are not in the loaded
|
|
1923
|
+
accounting population, so they are read here — chunked, never per turn.
|
|
1924
|
+
"""
|
|
1925
|
+
usage: dict[tuple, dict] = {}
|
|
1926
|
+
pairs = [(m, r) for (m, r) in dict.fromkeys(keys)
|
|
1927
|
+
if m is not None and r is not None]
|
|
1928
|
+
for chunk in _chunks(pairs):
|
|
1929
|
+
params: list[Any] = [value for pair in chunk for value in pair]
|
|
1930
|
+
condition = " OR ".join("(msg_id=? AND req_id=?)" for _ in chunk)
|
|
1931
|
+
# The account clause is appended AFTER the pair clause, so its
|
|
1932
|
+
# parameter follows the pair parameters in the same order.
|
|
1933
|
+
sql = (_CLAUDE_TURN_TOKENS_SQL + "(" + condition + ")"
|
|
1934
|
+
+ _account_predicate("account_key", account_key, params))
|
|
1935
|
+
for row in _execute(conn, sql, params):
|
|
1936
|
+
usage[(row[0], row[1])] = {
|
|
1937
|
+
"cache_creation": int(row[2] or 0),
|
|
1938
|
+
"cache_read": int(row[3] or 0),
|
|
1939
|
+
"cache_1h": row[4],
|
|
1940
|
+
"speed": row[5],
|
|
1941
|
+
}
|
|
1942
|
+
return usage
|
|
1943
|
+
|
|
1944
|
+
|
|
1945
|
+
# --- the evaluation result (#620 S3 §1.2, §1.6) -------------------------
|
|
1946
|
+
|
|
1947
|
+
@dataclass(frozen=True)
|
|
1948
|
+
class _EvidenceValue:
|
|
1949
|
+
"""One evidence figure, available with its value or withheld with a code."""
|
|
1950
|
+
|
|
1951
|
+
value: Any = None
|
|
1952
|
+
code: str | None = None
|
|
1953
|
+
qualifications: tuple[str, ...] = ()
|
|
1954
|
+
|
|
1955
|
+
|
|
1956
|
+
@dataclass(frozen=True)
|
|
1957
|
+
class _S3Evaluation:
|
|
1958
|
+
"""One conversation-derived class's decided facts over one window.
|
|
1959
|
+
|
|
1960
|
+
`established` False means the signal could not be established at all,
|
|
1961
|
+
which is `withheld / signal_unavailable` and never `provider_unavailable`.
|
|
1962
|
+
The three populations of §1.6 are distinct: CANDIDATE entries were
|
|
1963
|
+
eligible for an attempted evaluation, EVALUATED entries are those whose
|
|
1964
|
+
predicate could be decided, and QUALIFYING entries are the evaluated ones
|
|
1965
|
+
the predicate matched.
|
|
1966
|
+
"""
|
|
1967
|
+
|
|
1968
|
+
established: bool = False
|
|
1969
|
+
# Set when the evaluator RAISED. A caught-and-logged degrade that renders
|
|
1970
|
+
# as "the signal could not be established" is indistinguishable from a
|
|
1971
|
+
# store that genuinely holds nothing, so the two are separated here and
|
|
1972
|
+
# the loader publishes `calculation_failed` for this one.
|
|
1973
|
+
failure: str | None = None
|
|
1974
|
+
qualifying: tuple = ()
|
|
1975
|
+
evaluated: tuple = ()
|
|
1976
|
+
candidate_count: int = 0
|
|
1977
|
+
gap_codes: tuple[str, ...] = ()
|
|
1978
|
+
evidence: Mapping[str, _EvidenceValue] = field(default_factory=dict)
|
|
1979
|
+
qualifications: tuple[str, ...] = ()
|
|
1980
|
+
|
|
1981
|
+
@property
|
|
1982
|
+
def qualifying_usd(self) -> float:
|
|
1983
|
+
return stable_sum(entry.cost_usd for entry in self.qualifying)
|
|
1984
|
+
|
|
1985
|
+
def digest_rows(self, kind: str) -> list[tuple]:
|
|
1986
|
+
"""Exactly the derived facts the report publishes or aggregates.
|
|
1987
|
+
|
|
1988
|
+
Per-row topology, opaque provider keys, raw text, blocks payloads,
|
|
1989
|
+
content digests, filesystem paths and identities reach neither this
|
|
1990
|
+
digest nor the wire, so it is an equality oracle only for facts the
|
|
1991
|
+
response body already discloses. A compaction change still moves it,
|
|
1992
|
+
because a compaction change moves the flagged-turn count and the
|
|
1993
|
+
estimated wasted USD.
|
|
1994
|
+
"""
|
|
1995
|
+
rows: list[tuple] = [
|
|
1996
|
+
(f"{kind}.established", int(self.established)),
|
|
1997
|
+
(f"{kind}.failure", self.failure or ""),
|
|
1998
|
+
(f"{kind}.qualifying_usd", repr(self.qualifying_usd)),
|
|
1999
|
+
(f"{kind}.qualifying_entries", len(self.qualifying)),
|
|
2000
|
+
(f"{kind}.evaluated", len(self.evaluated)),
|
|
2001
|
+
(f"{kind}.candidates", self.candidate_count),
|
|
2002
|
+
(f"{kind}.gaps", ",".join(sorted(self.gap_codes))),
|
|
2003
|
+
(f"{kind}.qualifications", ",".join(sorted(self.qualifications))),
|
|
2004
|
+
]
|
|
2005
|
+
for name in sorted(self.evidence):
|
|
2006
|
+
field_value = self.evidence[name]
|
|
2007
|
+
rows.append((
|
|
2008
|
+
f"{kind}.evidence.{name}",
|
|
2009
|
+
field_value.code if field_value.code is not None
|
|
2010
|
+
else repr(field_value.value),
|
|
2011
|
+
))
|
|
2012
|
+
return rows
|
|
2013
|
+
|
|
2014
|
+
|
|
2015
|
+
# Each S3 class contributes ONE aggregate subject whose key, kind and label
|
|
2016
|
+
# are constants of the class rather than anything derived from a member.
|
|
2017
|
+
# `_display_key` aliases only project and session subjects and emits every
|
|
2018
|
+
# other key verbatim, so a member-derived key would reach the wire and the
|
|
2019
|
+
# screen unaliased.
|
|
2020
|
+
_S3_SUBJECT = {
|
|
2021
|
+
"cache_churn": (kernel.SUBJECT_CACHE_CHURN, "Turns that rebuilt their cache"),
|
|
2022
|
+
"short_high_context": (kernel.SUBJECT_SHORT_HIGH_CONTEXT,
|
|
2023
|
+
"Short conversations carrying large context"),
|
|
2024
|
+
"subagent_fanout": (kernel.SUBJECT_SUBAGENT_FANOUT,
|
|
2025
|
+
"Delegated subagent work"),
|
|
2026
|
+
}
|
|
2027
|
+
|
|
2028
|
+
|
|
2029
|
+
def _unestablished(gap_codes: Sequence[str] = ()) -> _S3Evaluation:
|
|
2030
|
+
"""A signal that could not be established, carrying WHY where it is known.
|
|
2031
|
+
|
|
2032
|
+
A class withheld because every spending session exhausted its seed share
|
|
2033
|
+
reported `signal_unavailable` with nothing beside it, so the one fact that
|
|
2034
|
+
explains the withholding was dropped exactly where the reader needs it.
|
|
2035
|
+
"""
|
|
2036
|
+
return _S3Evaluation(established=False, gap_codes=tuple(gap_codes))
|
|
2037
|
+
|
|
2038
|
+
|
|
2039
|
+
def _nothing_evaluable(candidates: Sequence[Any],
|
|
2040
|
+
evaluated: Sequence[Any]) -> bool:
|
|
2041
|
+
"""Whether a class had candidates and could decide none of them.
|
|
2042
|
+
|
|
2043
|
+
Spec 2.2, 2.3 and 2.4 state the same rule for all three conversation
|
|
2044
|
+
classes: a signal that could not be established at all is
|
|
2045
|
+
`withheld / signal_unavailable`, not a class answered over an empty
|
|
2046
|
+
evaluated population. Reporting the latter publishes
|
|
2047
|
+
`insufficient_population (support 0 units)` over a fully populated
|
|
2048
|
+
window, which is a false sentence about the store. A class with NO
|
|
2049
|
+
candidates is a different statement and keeps its support shortfall.
|
|
2050
|
+
"""
|
|
2051
|
+
return bool(candidates) and not evaluated
|
|
2052
|
+
|
|
2053
|
+
|
|
2054
|
+
def _seed_gap(exhausted: Sequence[Any]) -> tuple[str, ...]:
|
|
2055
|
+
"""The gap code a seed-budget exhaustion publishes, or nothing."""
|
|
2056
|
+
return (kernel.GAP_SCAN_BUDGET_EXHAUSTED,) if exhausted else ()
|
|
2057
|
+
|
|
2058
|
+
|
|
2059
|
+
def _fanout_gaps(unallocated: Sequence[Any],
|
|
2060
|
+
ambiguous: Sequence[Any]) -> tuple[str, ...]:
|
|
2061
|
+
"""The fan-out gap codes, stated once for both the established and the
|
|
2062
|
+
unestablished return."""
|
|
2063
|
+
codes = []
|
|
2064
|
+
if unallocated:
|
|
2065
|
+
codes.append(kernel.GAP_UNRESOLVED_SUBAGENT_ATTRIBUTION)
|
|
2066
|
+
if ambiguous:
|
|
2067
|
+
codes.append(kernel.GAP_AMBIGUOUS_ORIGIN_CATEGORY)
|
|
2068
|
+
return tuple(sorted(codes))
|
|
2069
|
+
|
|
2070
|
+
|
|
2071
|
+
# --- 2.2 prompt-cache churn, Claude only --------------------------------
|
|
2072
|
+
|
|
2073
|
+
def _evaluate_cache_churn(scope: DiagnosisScope, bundle: StoreBundle,
|
|
2074
|
+
conversations: sqlite3.Connection) -> _S3Evaluation:
|
|
2075
|
+
"""Flag every in-window turn that re-created the bulk of its cached prefix.
|
|
2076
|
+
|
|
2077
|
+
`_iter_cache_failures` is reused unchanged and gains no seed parameter:
|
|
2078
|
+
seeding is achieved by WHAT IT IS FED. For every session with potentially
|
|
2079
|
+
evaluable events in the window the stream begins at the last normalized
|
|
2080
|
+
compaction strictly before `window_start`, inclusive; where none exists it
|
|
2081
|
+
begins at the session's earliest retained event with an empty running
|
|
2082
|
+
maximum. Only flags whose assistant event falls inside the half-open
|
|
2083
|
+
window are published — the prefix is state.
|
|
2084
|
+
"""
|
|
2085
|
+
query = _conversation_query()
|
|
2086
|
+
facts = bundle.facts
|
|
2087
|
+
bounds = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
2088
|
+
sessions = sorted(
|
|
2089
|
+
str(row[0]) for row in
|
|
2090
|
+
_execute(conversations, _CLAUDE_WINDOW_SESSIONS_SQL, bounds)
|
|
2091
|
+
)
|
|
2092
|
+
if not sessions:
|
|
2093
|
+
return _unestablished()
|
|
2094
|
+
|
|
2095
|
+
allocation = allocate_scan_budget(sessions,
|
|
2096
|
+
kernel.DIAGNOSIS_SEED_SCAN_BUDGET_ROWS)
|
|
2097
|
+
largest = max(allocation.values(), default=0)
|
|
2098
|
+
prefix: dict[str, list] = {session: [] for session in sessions}
|
|
2099
|
+
overflowed: set[str] = set()
|
|
2100
|
+
for chunk in _chunks(sessions):
|
|
2101
|
+
sql = _CLAUDE_SEED_PREFIX_SQL.format(
|
|
2102
|
+
placeholders=_placeholders(len(chunk)))
|
|
2103
|
+
# `largest + 1` fetches ONE row past the widest share, so a session
|
|
2104
|
+
# holding more history than its own share is visible as exhausted
|
|
2105
|
+
# without a second query per session.
|
|
2106
|
+
for row in _execute(conversations, sql,
|
|
2107
|
+
list(chunk) + [bounds[0], largest + 1]):
|
|
2108
|
+
session = str(row["session_id"])
|
|
2109
|
+
if int(row["rank"]) > allocation.get(session, 0):
|
|
2110
|
+
overflowed.add(session)
|
|
2111
|
+
continue
|
|
2112
|
+
prefix[session].append(row)
|
|
2113
|
+
|
|
2114
|
+
window_rows: dict[str, list] = {session: [] for session in sessions}
|
|
2115
|
+
for chunk in _chunks(sessions):
|
|
2116
|
+
sql = _CLAUDE_WINDOW_ROWS_SQL.format(
|
|
2117
|
+
placeholders=_placeholders(len(chunk)))
|
|
2118
|
+
for row in _execute(conversations, sql, list(chunk) + bounds):
|
|
2119
|
+
window_rows[str(row["session_id"])].append(row)
|
|
2120
|
+
|
|
2121
|
+
is_compaction = query.is_compaction_row
|
|
2122
|
+
streams: dict[str, list] = {}
|
|
2123
|
+
exhausted: set[str] = set()
|
|
2124
|
+
for session in sessions:
|
|
2125
|
+
rows = prefix.get(session, [])
|
|
2126
|
+
seed_index = None
|
|
2127
|
+
for index, row in enumerate(rows):
|
|
2128
|
+
if (row["entry_type"] in ("meta", "human")
|
|
2129
|
+
and is_compaction(row["text"], row["blocks_json"])):
|
|
2130
|
+
seed_index = index
|
|
2131
|
+
if seed_index is not None:
|
|
2132
|
+
head = rows[seed_index:]
|
|
2133
|
+
elif session in overflowed:
|
|
2134
|
+
# The share was consumed without reaching a compaction and more
|
|
2135
|
+
# history exists, so the seed cannot be established for this
|
|
2136
|
+
# session. It leaves the evaluated population rather than being
|
|
2137
|
+
# walked from an arbitrary point with a fabricated running maximum.
|
|
2138
|
+
exhausted.add(session)
|
|
2139
|
+
continue
|
|
2140
|
+
else:
|
|
2141
|
+
head = rows
|
|
2142
|
+
streams[session] = head + window_rows.get(session, [])
|
|
2143
|
+
if not streams:
|
|
2144
|
+
return _unestablished(_seed_gap(exhausted))
|
|
2145
|
+
|
|
2146
|
+
turn_keys = [
|
|
2147
|
+
(row["msg_id"], row["req_id"])
|
|
2148
|
+
for rows in streams.values() for row in rows
|
|
2149
|
+
if row["entry_type"] == "assistant" and row["msg_id"] is not None
|
|
2150
|
+
]
|
|
2151
|
+
usage = _claude_turn_token_map(bundle.connection("cache"), turn_keys,
|
|
2152
|
+
scope.account_key)
|
|
2153
|
+
|
|
2154
|
+
flagged_keys: set[tuple] = set()
|
|
2155
|
+
flagged_sessions: set[str] = set()
|
|
2156
|
+
wasted: list[float] = []
|
|
2157
|
+
for session, rows in streams.items():
|
|
2158
|
+
in_window = {
|
|
2159
|
+
(row["msg_id"], row["req_id"])
|
|
2160
|
+
for row in window_rows.get(session, ())
|
|
2161
|
+
if row["entry_type"] == "assistant" and row["msg_id"] is not None
|
|
2162
|
+
}
|
|
2163
|
+
physical = [tuple(row[column] for column in _CHURN_ROW_COLUMNS)
|
|
2164
|
+
for row in rows]
|
|
2165
|
+
events, event_keys = query.fold_claude_cache_failure_events(
|
|
2166
|
+
physical, usage, with_sources=True)
|
|
2167
|
+
for index, _prev, lost, model, speed in query._iter_cache_failures(
|
|
2168
|
+
events):
|
|
2169
|
+
key = event_keys[index]
|
|
2170
|
+
if key is None or key not in in_window:
|
|
2171
|
+
continue
|
|
2172
|
+
flagged_keys.add(key)
|
|
2173
|
+
flagged_sessions.add(session)
|
|
2174
|
+
wasted.append(query._cache_failure_wasted_usd(
|
|
2175
|
+
model, lost, speed=speed))
|
|
2176
|
+
|
|
2177
|
+
known = set(sessions)
|
|
2178
|
+
seeded = set(streams)
|
|
2179
|
+
candidates = [entry for entry in facts.entries
|
|
2180
|
+
if entry.session_key in known]
|
|
2181
|
+
evaluated = [entry for entry in candidates if entry.session_key in seeded]
|
|
2182
|
+
if _nothing_evaluable(candidates, evaluated):
|
|
2183
|
+
# `streams` is a statement about SESSIONS and `evaluated` is built
|
|
2184
|
+
# from ENTRIES, so a window where one session supplies the transcript
|
|
2185
|
+
# rows that seed and a different session supplies the spend passes the
|
|
2186
|
+
# session-level guard with nothing evaluable behind it. The same rule
|
|
2187
|
+
# the other two classes apply is what stops it publishing
|
|
2188
|
+
# `insufficient_population (support 0 units)` over a populated window.
|
|
2189
|
+
return _unestablished(_seed_gap(exhausted))
|
|
2190
|
+
qualifying = [entry for entry in evaluated
|
|
2191
|
+
if (entry.msg_id, entry.req_id) in flagged_keys]
|
|
2192
|
+
return _S3Evaluation(
|
|
2193
|
+
established=True,
|
|
2194
|
+
qualifying=tuple(qualifying),
|
|
2195
|
+
evaluated=tuple(evaluated),
|
|
2196
|
+
candidate_count=len(candidates),
|
|
2197
|
+
gap_codes=_seed_gap(exhausted),
|
|
2198
|
+
evidence={
|
|
2199
|
+
"flaggedTurnCount": _EvidenceValue(value=len(flagged_keys)),
|
|
2200
|
+
"affectedConversationCount": _EvidenceValue(
|
|
2201
|
+
value=len(flagged_sessions)),
|
|
2202
|
+
# The counterfactual, and never the sort key: this class ranks on
|
|
2203
|
+
# the RETAINED cost of the turns it flags, like every other class.
|
|
2204
|
+
"estWastedUsd": _EvidenceValue(value=stable_sum(wasted)),
|
|
2205
|
+
},
|
|
2206
|
+
)
|
|
2207
|
+
|
|
2208
|
+
|
|
2209
|
+
# --- 2.3 short conversations carrying large context ---------------------
|
|
2210
|
+
|
|
2211
|
+
def _large_enough(fraction: float) -> bool:
|
|
2212
|
+
"""`>= 0.80`, with the repository's coverage slack.
|
|
2213
|
+
|
|
2214
|
+
A fraction is a ratio of a summed token count to a capacity, so a request
|
|
2215
|
+
genuinely at four fifths of its window can compute one bit below it. The
|
|
2216
|
+
slack is the same internal float guard the coverage thresholds use; the
|
|
2217
|
+
PUBLISHED rule is the threshold itself.
|
|
2218
|
+
"""
|
|
2219
|
+
return fraction >= (kernel.DIAGNOSIS_LARGE_CONTEXT_MIN_WINDOW_FRACTION
|
|
2220
|
+
- kernel.COVERAGE_EPSILON)
|
|
2221
|
+
|
|
2222
|
+
|
|
2223
|
+
def _short_context_evaluation(facts: RawFacts, *, known: Sequence[str],
|
|
2224
|
+
unevaluable: Mapping[str, str],
|
|
2225
|
+
qualifying: Sequence[str],
|
|
2226
|
+
turn_counts: Mapping[str, int],
|
|
2227
|
+
fractions: Mapping[str, float],
|
|
2228
|
+
qualifications: Sequence[str] = (),
|
|
2229
|
+
session_level: Mapping[str, bool] | None = None
|
|
2230
|
+
) -> _S3Evaluation:
|
|
2231
|
+
"""Shape one provider's decided conversations into the class result."""
|
|
2232
|
+
known_set = set(known)
|
|
2233
|
+
qualifying_set = set(qualifying)
|
|
2234
|
+
candidates = [entry for entry in facts.entries
|
|
2235
|
+
if entry.session_key in known_set]
|
|
2236
|
+
evaluated = [entry for entry in candidates
|
|
2237
|
+
if entry.session_key not in unevaluable]
|
|
2238
|
+
if _nothing_evaluable(candidates, evaluated):
|
|
2239
|
+
return _unestablished(sorted(set(unevaluable.values())))
|
|
2240
|
+
matched = [entry for entry in evaluated
|
|
2241
|
+
if entry.session_key in qualifying_set]
|
|
2242
|
+
member_turns = [turn_counts[key] for key in qualifying_set
|
|
2243
|
+
if key in turn_counts]
|
|
2244
|
+
members = sorted((fractions[key], key) for key in qualifying_set
|
|
2245
|
+
if key in fractions)
|
|
2246
|
+
thin = kernel.WithheldCause.INSUFFICIENT_POPULATION.value
|
|
2247
|
+
median = _median_turns(member_turns)
|
|
2248
|
+
fraction_field = _EvidenceValue(code=thin)
|
|
2249
|
+
if members:
|
|
2250
|
+
best_fraction, best_key = members[-1]
|
|
2251
|
+
# The qualification describes the conversation that produced the
|
|
2252
|
+
# PUBLISHED maximum, so a per-turn winner is never marked
|
|
2253
|
+
# session-level because some other conversation fell back.
|
|
2254
|
+
marks = ((kernel.QUALIFICATION_SESSION_LEVEL_CAPACITY,)
|
|
2255
|
+
if (session_level or {}).get(best_key) else ())
|
|
2256
|
+
fraction_field = _EvidenceValue(value=best_fraction,
|
|
2257
|
+
qualifications=marks)
|
|
2258
|
+
return _S3Evaluation(
|
|
2259
|
+
established=True,
|
|
2260
|
+
qualifying=tuple(matched),
|
|
2261
|
+
evaluated=tuple(evaluated),
|
|
2262
|
+
candidate_count=len(candidates),
|
|
2263
|
+
gap_codes=tuple(sorted(set(unevaluable.values()))),
|
|
2264
|
+
evidence={
|
|
2265
|
+
"conversationCount": _EvidenceValue(value=len(qualifying_set)),
|
|
2266
|
+
"medianHumanTurns": (_EvidenceValue(value=median)
|
|
2267
|
+
if median is not None
|
|
2268
|
+
else _EvidenceValue(code=thin)),
|
|
2269
|
+
"maxContextWindowFraction": fraction_field,
|
|
2270
|
+
},
|
|
2271
|
+
qualifications=tuple(qualifications),
|
|
2272
|
+
)
|
|
2273
|
+
|
|
2274
|
+
|
|
2275
|
+
def _evaluate_short_high_context(scope: DiagnosisScope, bundle: StoreBundle,
|
|
2276
|
+
conversations: sqlite3.Connection
|
|
2277
|
+
) -> _S3Evaluation:
|
|
2278
|
+
if scope.source == "codex":
|
|
2279
|
+
return _evaluate_codex_short_high_context(scope, bundle, conversations)
|
|
2280
|
+
return _evaluate_claude_short_high_context(scope, bundle, conversations)
|
|
2281
|
+
|
|
2282
|
+
|
|
2283
|
+
def _evaluate_claude_short_high_context(scope: DiagnosisScope,
|
|
2284
|
+
bundle: StoreBundle,
|
|
2285
|
+
conversations: sqlite3.Connection
|
|
2286
|
+
) -> _S3Evaluation:
|
|
2287
|
+
"""Conversations of one to three human turns holding a very large request.
|
|
2288
|
+
|
|
2289
|
+
Turn counts span the WHOLE retained conversation while the COST stays
|
|
2290
|
+
window-clipped, because "short" is a property of the conversation and a
|
|
2291
|
+
window-clipped count would call a long conversation short whenever the
|
|
2292
|
+
window caught only its tail.
|
|
2293
|
+
|
|
2294
|
+
`maxContextWindowFraction` reads no retained history at all on Claude: the
|
|
2295
|
+
association is a STORED key on both sides, so the fraction is derived from
|
|
2296
|
+
the in-window accounting population the adapter has already loaded, joined
|
|
2297
|
+
on `(msg_id, req_id)`.
|
|
2298
|
+
"""
|
|
2299
|
+
facts = bundle.facts
|
|
2300
|
+
bounds = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
2301
|
+
sessions = sorted(
|
|
2302
|
+
str(row[0]) for row in
|
|
2303
|
+
_execute(conversations, _CLAUDE_WINDOW_SESSIONS_SQL, bounds)
|
|
2304
|
+
)
|
|
2305
|
+
if not sessions:
|
|
2306
|
+
return _unestablished()
|
|
2307
|
+
|
|
2308
|
+
budget = kernel.DIAGNOSIS_CONVERSATION_NORMALIZE_BUDGET_ROWS
|
|
2309
|
+
candidate_rows: dict[str, list] = {session: [] for session in sessions}
|
|
2310
|
+
overflowed: set[str] = set()
|
|
2311
|
+
for chunk in _chunks(sessions):
|
|
2312
|
+
sql = _CLAUDE_TURN_CANDIDATE_SQL.format(
|
|
2313
|
+
placeholders=_placeholders(len(chunk)))
|
|
2314
|
+
for row in _execute(conversations, sql, list(chunk) + [budget + 1]):
|
|
2315
|
+
session = str(row["session_id"])
|
|
2316
|
+
if int(row["rank"]) > budget:
|
|
2317
|
+
overflowed.add(session)
|
|
2318
|
+
continue
|
|
2319
|
+
candidate_rows[session].append(row)
|
|
2320
|
+
|
|
2321
|
+
assistants: dict[str, list] = {session: [] for session in sessions}
|
|
2322
|
+
for chunk in _chunks(sessions):
|
|
2323
|
+
sql = _CLAUDE_WINDOW_ASSISTANT_SQL.format(
|
|
2324
|
+
placeholders=_placeholders(len(chunk)))
|
|
2325
|
+
for row in _execute(conversations, sql, list(chunk) + bounds):
|
|
2326
|
+
assistants[str(row["session_id"])].append(row)
|
|
2327
|
+
|
|
2328
|
+
by_turn: dict[tuple, list] = {}
|
|
2329
|
+
for entry in facts.entries:
|
|
2330
|
+
if entry.msg_id is not None:
|
|
2331
|
+
by_turn.setdefault((entry.msg_id, entry.req_id), []).append(entry)
|
|
2332
|
+
|
|
2333
|
+
maximum = kernel.DIAGNOSIS_SHORT_CONVERSATION_MAX_HUMAN_TURNS
|
|
2334
|
+
unevaluable: dict[str, str] = {}
|
|
2335
|
+
qualifying: list[str] = []
|
|
2336
|
+
turn_counts: dict[str, int] = {}
|
|
2337
|
+
fractions: dict[str, float] = {}
|
|
2338
|
+
for session in sessions:
|
|
2339
|
+
rows = candidate_rows.get(session, [])
|
|
2340
|
+
replies = assistants.get(session, [])
|
|
2341
|
+
# ONE fold per session over the whole stream. The human-turn count
|
|
2342
|
+
# spans the retained candidates and the associations need the
|
|
2343
|
+
# in-window assistants, and an assistant row creates no human item, so
|
|
2344
|
+
# folding the union answers both — where folding twice would parse
|
|
2345
|
+
# every `blocks_json` body twice.
|
|
2346
|
+
items = _claude_canonical_items(sorted(
|
|
2347
|
+
list(rows) + list(replies),
|
|
2348
|
+
key=lambda r: (str(_row_field(r, "timestamp_utc") or ""),
|
|
2349
|
+
int(_row_field(r, "id") or 0))))
|
|
2350
|
+
count = len(_claude_human_turn_items(items))
|
|
2351
|
+
if count > maximum:
|
|
2352
|
+
turn_counts[session] = count
|
|
2353
|
+
continue # decided: the conversation is long
|
|
2354
|
+
if session in overflowed:
|
|
2355
|
+
# Never short, never long: the budget ran out before the predicate
|
|
2356
|
+
# could be decided, so the conversation leaves the evaluated
|
|
2357
|
+
# population rather than being classified by a truncated count.
|
|
2358
|
+
unevaluable[session] = kernel.GAP_SCAN_BUDGET_EXHAUSTED
|
|
2359
|
+
continue
|
|
2360
|
+
turn_counts[session] = count
|
|
2361
|
+
if count < 1:
|
|
2362
|
+
continue # decided: no human turn at all
|
|
2363
|
+
associations = _claude_associate_items(items)
|
|
2364
|
+
# A canonical turn item is one item per `(msg_id, req_id)` however
|
|
2365
|
+
# many physical fragments it arrived in, and the fold strips the
|
|
2366
|
+
# internal turn key before returning. The item's anchor uuid names the
|
|
2367
|
+
# physical row it was seeded from, and every fragment of a turn shares
|
|
2368
|
+
# that turn's key, so the anchor recovers it exactly.
|
|
2369
|
+
key_of_uuid = {str(_row_field(row, "uuid")):
|
|
2370
|
+
(_row_field(row, "msg_id"), _row_field(row, "req_id"))
|
|
2371
|
+
for row in replies}
|
|
2372
|
+
keys = [key_of_uuid[uuid] for association in associations
|
|
2373
|
+
for reply in association.replies
|
|
2374
|
+
for uuid in [str(reply["anchor"]["uuid"])]
|
|
2375
|
+
if key_of_uuid.get(uuid, (None,))[0] is not None]
|
|
2376
|
+
best: float | None = None
|
|
2377
|
+
saw_request = False
|
|
2378
|
+
for key in dict.fromkeys(keys):
|
|
2379
|
+
for entry in by_turn.get(key, ()):
|
|
2380
|
+
saw_request = True
|
|
2381
|
+
capacity = _claude_context_window(entry.model)
|
|
2382
|
+
if not capacity:
|
|
2383
|
+
continue
|
|
2384
|
+
# The Claude numerator, matching the statusline's own context
|
|
2385
|
+
# segment: input plus BOTH cache legs.
|
|
2386
|
+
used = (entry.input_tokens + entry.cache_read_tokens
|
|
2387
|
+
+ entry.cache_create_tokens)
|
|
2388
|
+
fraction = used / capacity
|
|
2389
|
+
best = fraction if best is None else max(best, fraction)
|
|
2390
|
+
if saw_request and best is None:
|
|
2391
|
+
unevaluable[session] = kernel.GAP_UNKNOWN_CONTEXT_WINDOW
|
|
2392
|
+
continue
|
|
2393
|
+
# With several replies the LARGEST single request fraction is used,
|
|
2394
|
+
# never their sum: two replies at 0.45 do not make a full window.
|
|
2395
|
+
if best is not None and _large_enough(best):
|
|
2396
|
+
qualifying.append(session)
|
|
2397
|
+
fractions[session] = best
|
|
2398
|
+
return _short_context_evaluation(
|
|
2399
|
+
facts, known=sessions, unevaluable=unevaluable, qualifying=qualifying,
|
|
2400
|
+
turn_counts=turn_counts, fractions=fractions,
|
|
2401
|
+
)
|
|
2402
|
+
|
|
2403
|
+
|
|
2404
|
+
def _codex_turn_capacities(events: Sequence[Any]) -> dict[str, int]:
|
|
2405
|
+
"""`{turn_id: model_context_window}` from the retained lifecycle records.
|
|
2406
|
+
|
|
2407
|
+
Codex capacity is PER TURN, not per session:
|
|
2408
|
+
`codex_conversation_threads.context_window` is populated only from
|
|
2409
|
+
`session_meta` and cannot describe a request whose model changed
|
|
2410
|
+
mid-thread, while `turn_context.model_context_window` is retained per
|
|
2411
|
+
turn.
|
|
2412
|
+
"""
|
|
2413
|
+
capacities: dict[str, int] = {}
|
|
2414
|
+
for event in events:
|
|
2415
|
+
# Spec 2.3 names `turn_context.model_context_window`, and the record
|
|
2416
|
+
# type is what identifies one: `infer_codex_event_turns` and the
|
|
2417
|
+
# normalizer both branch on it. Accepting the key from any payload
|
|
2418
|
+
# takes a capacity from a record that never described a turn.
|
|
2419
|
+
if getattr(event, "record_type", None) != "turn_context":
|
|
2420
|
+
continue
|
|
2421
|
+
try:
|
|
2422
|
+
payload = json.loads(getattr(event, "payload_json", "") or "{}")
|
|
2423
|
+
except (ValueError, TypeError):
|
|
2424
|
+
continue
|
|
2425
|
+
body = payload.get("payload")
|
|
2426
|
+
if not isinstance(body, Mapping):
|
|
2427
|
+
continue
|
|
2428
|
+
window = body.get("model_context_window")
|
|
2429
|
+
turn_id = getattr(event, "turn_id", None) or body.get("turn_id")
|
|
2430
|
+
if not isinstance(window, int) or window <= 0 or not turn_id:
|
|
2431
|
+
continue
|
|
2432
|
+
capacities[str(turn_id)] = window
|
|
2433
|
+
return capacities
|
|
2434
|
+
|
|
2435
|
+
|
|
2436
|
+
def _codex_threads_for(cache: sqlite3.Connection, sql: str,
|
|
2437
|
+
values: Sequence[str]) -> list:
|
|
2438
|
+
rows: list = []
|
|
2439
|
+
for chunk in _chunks(list(values)):
|
|
2440
|
+
rows.extend(_execute(cache, sql.format(
|
|
2441
|
+
placeholders=_placeholders(len(chunk))), list(chunk)))
|
|
2442
|
+
return rows
|
|
2443
|
+
|
|
2444
|
+
|
|
2445
|
+
def _evaluate_codex_short_high_context(scope: DiagnosisScope,
|
|
2446
|
+
bundle: StoreBundle,
|
|
2447
|
+
conversations: sqlite3.Connection
|
|
2448
|
+
) -> _S3Evaluation:
|
|
2449
|
+
"""The Codex half, over normalized prompts and per-turn capacity.
|
|
2450
|
+
|
|
2451
|
+
The population is the MAIN-THREAD conversations, and only the exact
|
|
2452
|
+
literals `user` and `subagent` are recognisable origins — so an ambiguous
|
|
2453
|
+
origin excludes a thread from the main-thread population just as it
|
|
2454
|
+
excludes it from the delegated one. The consequence is stated rather than
|
|
2455
|
+
hidden: this class is an identifiable subset of unknown completeness on
|
|
2456
|
+
Codex, not only the fan-out class.
|
|
2457
|
+
"""
|
|
2458
|
+
facts = bundle.facts
|
|
2459
|
+
scoped = sorted({entry.session_key for entry in facts.entries
|
|
2460
|
+
if entry.session_key})
|
|
2461
|
+
if not scoped:
|
|
2462
|
+
return _unestablished()
|
|
2463
|
+
threads = {str(row["conversation_key"]): row
|
|
2464
|
+
for row in bundle.codex_threads}
|
|
2465
|
+
budget = kernel.DIAGNOSIS_CONVERSATION_NORMALIZE_BUDGET_ROWS
|
|
2466
|
+
candidate_rows: dict[str, list] = {key: [] for key in scoped}
|
|
2467
|
+
overflowed: set[str] = set()
|
|
2468
|
+
for chunk in _chunks(scoped):
|
|
2469
|
+
sql = _CODEX_PROMPT_CANDIDATE_SQL.format(
|
|
2470
|
+
placeholders=_placeholders(len(chunk)))
|
|
2471
|
+
for row in _execute(conversations, sql, list(chunk) + [budget + 1]):
|
|
2472
|
+
key = str(row["conversation_key"])
|
|
2473
|
+
if int(row["rank"]) > budget:
|
|
2474
|
+
overflowed.add(key)
|
|
2475
|
+
continue
|
|
2476
|
+
candidate_rows[key].append(row)
|
|
2477
|
+
prompt_counts: dict[str, int] = {}
|
|
2478
|
+
|
|
2479
|
+
def _prompt_count(key: str) -> int:
|
|
2480
|
+
"""Canonicalize ONE conversation's prompts, once, on first demand.
|
|
2481
|
+
|
|
2482
|
+
Counting every candidate up front normalized conversations the loop
|
|
2483
|
+
below then discarded — a delegated thread, an ambiguous origin and an
|
|
2484
|
+
overflowed budget all decide the conversation without ever reading the
|
|
2485
|
+
count. The work is bounded by the per-conversation normalize budget,
|
|
2486
|
+
so it was waste rather than a defect, but it is waste proportional to
|
|
2487
|
+
the window's whole conversation set.
|
|
2488
|
+
"""
|
|
2489
|
+
if key not in prompt_counts:
|
|
2490
|
+
prompt_counts[key] = _codex_human_turns(
|
|
2491
|
+
candidate_rows.get(key, []))
|
|
2492
|
+
return prompt_counts[key]
|
|
2493
|
+
|
|
2494
|
+
by_conversation: dict[str, list] = {}
|
|
2495
|
+
for entry in facts.entries:
|
|
2496
|
+
by_conversation.setdefault(entry.session_key, []).append(entry)
|
|
2497
|
+
paths = sorted({entry.source_path for entry in facts.entries
|
|
2498
|
+
if entry.source_path})
|
|
2499
|
+
turn_maps, capacities, starved = _codex_turn_attribution(conversations,
|
|
2500
|
+
paths)
|
|
2501
|
+
|
|
2502
|
+
maximum = kernel.DIAGNOSIS_SHORT_CONVERSATION_MAX_HUMAN_TURNS
|
|
2503
|
+
unevaluable: dict[str, str] = {}
|
|
2504
|
+
qualifying: list[str] = []
|
|
2505
|
+
turn_counts: dict[str, int] = {}
|
|
2506
|
+
fractions: dict[str, float] = {}
|
|
2507
|
+
# Per subject, not per provider: one boolean for the whole provider stamps
|
|
2508
|
+
# the session-level qualification on `maxContextWindowFraction` even when
|
|
2509
|
+
# the published maximum came from a per-turn capacity.
|
|
2510
|
+
session_level: dict[str, bool] = {}
|
|
2511
|
+
for key in scoped:
|
|
2512
|
+
thread = threads.get(key)
|
|
2513
|
+
origin = _codex_origin(
|
|
2514
|
+
thread["root_thread_id"] if thread is not None else None)
|
|
2515
|
+
if origin == "ambiguous":
|
|
2516
|
+
# Neither population. The value is not a category this tree can
|
|
2517
|
+
# attribute a meaning to, so the predicate could not be decided.
|
|
2518
|
+
# The ORIGIN is what could not be read, which is why this is not
|
|
2519
|
+
# `unknown_context_window`.
|
|
2520
|
+
unevaluable[key] = kernel.GAP_AMBIGUOUS_ORIGIN_CATEGORY
|
|
2521
|
+
continue
|
|
2522
|
+
if origin != "main":
|
|
2523
|
+
continue # decided: a delegated thread
|
|
2524
|
+
if key in overflowed:
|
|
2525
|
+
# Never short, never long: canonicalizing the prompts ran out of
|
|
2526
|
+
# its per-conversation share before the predicate could be decided.
|
|
2527
|
+
unevaluable[key] = kernel.GAP_SCAN_BUDGET_EXHAUSTED
|
|
2528
|
+
continue
|
|
2529
|
+
count = _prompt_count(key)
|
|
2530
|
+
turn_counts[key] = count
|
|
2531
|
+
if count > maximum or count < 1:
|
|
2532
|
+
continue
|
|
2533
|
+
entries = by_conversation.get(key, [])
|
|
2534
|
+
if any(entry.source_path in starved for entry in entries):
|
|
2535
|
+
unevaluable[key] = kernel.GAP_SCAN_BUDGET_EXHAUSTED
|
|
2536
|
+
continue
|
|
2537
|
+
best: float | None = None
|
|
2538
|
+
best_is_session_level = False
|
|
2539
|
+
saw_request = False
|
|
2540
|
+
for entry in entries:
|
|
2541
|
+
turn = turn_maps.get(entry.source_path, {}).get(entry.line_offset)
|
|
2542
|
+
if turn is None:
|
|
2543
|
+
continue # orphaned: no owning turn
|
|
2544
|
+
saw_request = True
|
|
2545
|
+
from_session = False
|
|
2546
|
+
capacity = capacities.get(str(turn))
|
|
2547
|
+
if not capacity and thread is not None:
|
|
2548
|
+
capacity = thread["context_window"]
|
|
2549
|
+
from_session = bool(capacity)
|
|
2550
|
+
if not capacity:
|
|
2551
|
+
continue
|
|
2552
|
+
# Codex `input_tokens` is already cache-inclusive.
|
|
2553
|
+
fraction = entry.input_tokens / capacity
|
|
2554
|
+
if best is None or fraction > best:
|
|
2555
|
+
best = fraction
|
|
2556
|
+
best_is_session_level = from_session
|
|
2557
|
+
if saw_request and best is None:
|
|
2558
|
+
unevaluable[key] = kernel.GAP_UNKNOWN_CONTEXT_WINDOW
|
|
2559
|
+
continue
|
|
2560
|
+
if best is not None and _large_enough(best):
|
|
2561
|
+
qualifying.append(key)
|
|
2562
|
+
fractions[key] = best
|
|
2563
|
+
session_level[key] = best_is_session_level
|
|
2564
|
+
return _short_context_evaluation(
|
|
2565
|
+
facts, known=scoped, unevaluable=unevaluable, qualifying=qualifying,
|
|
2566
|
+
turn_counts=turn_counts, fractions=fractions,
|
|
2567
|
+
qualifications=(kernel.QUALIFICATION_IDENTIFIABLE_SUBSET,),
|
|
2568
|
+
session_level=session_level,
|
|
2569
|
+
)
|
|
2570
|
+
|
|
2571
|
+
|
|
2572
|
+
def _codex_turn_attribution(conversations: sqlite3.Connection,
|
|
2573
|
+
paths: Sequence[str]):
|
|
2574
|
+
"""`(turn_maps, capacities, starved)` over the budgeted event inference.
|
|
2575
|
+
|
|
2576
|
+
A file that exhausts its share yields NO turn map: a truncated read would
|
|
2577
|
+
produce a WRONG map rather than a missing one, because the inference
|
|
2578
|
+
replays a file's whole lifecycle from its start.
|
|
2579
|
+
"""
|
|
2580
|
+
turn_maps: dict[str, dict] = {}
|
|
2581
|
+
capacities: dict[str, int] = {}
|
|
2582
|
+
starved: set[str] = set()
|
|
2583
|
+
if not paths:
|
|
2584
|
+
return turn_maps, capacities, starved
|
|
2585
|
+
codex_kernel = _cctally()._load_sibling("_lib_codex_conversation")
|
|
2586
|
+
physical = _cctally()._load_sibling("_lib_jsonl").CodexPhysicalEvent
|
|
2587
|
+
allocation = allocate_scan_budget(
|
|
2588
|
+
list(paths), kernel.DIAGNOSIS_CODEX_EVENT_SCAN_BUDGET_ROWS)
|
|
2589
|
+
per_file = kernel.DIAGNOSIS_CODEX_EVENT_SCAN_PER_FILE_ROWS
|
|
2590
|
+
allocation = {path: min(share, per_file)
|
|
2591
|
+
for path, share in allocation.items()}
|
|
2592
|
+
largest = max(allocation.values(), default=0)
|
|
2593
|
+
grouped: dict[str, list] = {path: [] for path in paths}
|
|
2594
|
+
for chunk in _chunks(list(paths)):
|
|
2595
|
+
sql = _CODEX_EVENTS_SQL.format(placeholders=_placeholders(len(chunk)))
|
|
2596
|
+
for row in _execute(conversations, sql, list(chunk) + [largest + 1]):
|
|
2597
|
+
grouped.setdefault(str(row["source_path"]), []).append(row)
|
|
2598
|
+
for path, rows in grouped.items():
|
|
2599
|
+
share = allocation.get(path, 0)
|
|
2600
|
+
if any(int(row["rank"]) > share for row in rows):
|
|
2601
|
+
starved.add(path)
|
|
2602
|
+
continue
|
|
2603
|
+
events = [physical(*[row[column] for column in _CODEX_EVENT_COLUMNS])
|
|
2604
|
+
for row in rows]
|
|
2605
|
+
turn_maps[path] = codex_kernel.fold_codex_event_turns(events)
|
|
2606
|
+
capacities.update(_codex_turn_capacities(events))
|
|
2607
|
+
return turn_maps, capacities, starved
|
|
2608
|
+
|
|
2609
|
+
|
|
2610
|
+
# --- 2.4 subagent fan-out ----------------------------------------------
|
|
2611
|
+
|
|
2612
|
+
def _fanout_evidence(facts: RawFacts, *, candidates: Sequence[Any],
|
|
2613
|
+
unallocated: Sequence[Any], allocated: Sequence[Any],
|
|
2614
|
+
bucket_usd: Mapping[Any, float], identified: int,
|
|
2615
|
+
subset: bool,
|
|
2616
|
+
ambiguous: Sequence[Any] = ()) -> _S3Evaluation:
|
|
2617
|
+
"""Shape one provider's resolved fan-out into the class result.
|
|
2618
|
+
|
|
2619
|
+
`observedUsd` carries ALLOCATED qualifying cost only. Unallocated cost is
|
|
2620
|
+
published as its own figure and never folded in or discarded, and it
|
|
2621
|
+
qualifies the row as partial attribution.
|
|
2622
|
+
|
|
2623
|
+
**An entry whose origin category could not be read is NOT unallocated.**
|
|
2624
|
+
An unallocated entry is known subagent spend that could not be joined
|
|
2625
|
+
exactly once to a bucket. An ambiguous one is spend this tree can say
|
|
2626
|
+
nothing about — a thread carrying an unrecognised origin, or no thread row
|
|
2627
|
+
at all, which is the shape a Codex rollout landing mid-`thread_source`
|
|
2628
|
+
rollout produces. Filing it as unallocated overstates `unallocatedUsd`
|
|
2629
|
+
and stamps partial attribution on conversations that may be main-thread.
|
|
2630
|
+
|
|
2631
|
+
Both kinds are candidates that FAILED evaluation, so both lower
|
|
2632
|
+
`evaluabilityCoverage` and both are excluded from `support_units`; only
|
|
2633
|
+
the unallocated ones contribute dollars to a published figure.
|
|
2634
|
+
"""
|
|
2635
|
+
unallocated_usd = stable_sum(entry.cost_usd for entry in unallocated)
|
|
2636
|
+
undecided = {id(entry) for entry in unallocated}
|
|
2637
|
+
undecided.update(id(entry) for entry in ambiguous)
|
|
2638
|
+
evaluated = [entry for entry in candidates if id(entry) not in undecided]
|
|
2639
|
+
if _nothing_evaluable(candidates, evaluated):
|
|
2640
|
+
return _unestablished(_fanout_gaps(unallocated, ambiguous))
|
|
2641
|
+
qualifying_usd = stable_sum(entry.cost_usd for entry in allocated)
|
|
2642
|
+
largest = max(bucket_usd.values(), default=0.0)
|
|
2643
|
+
subset_marks = ((kernel.QUALIFICATION_IDENTIFIABLE_SUBSET,) if subset
|
|
2644
|
+
else ())
|
|
2645
|
+
qualifications = list(subset_marks)
|
|
2646
|
+
if unallocated_usd > 0.0:
|
|
2647
|
+
qualifications.append(kernel.QUALIFICATION_PARTIAL_ATTRIBUTION)
|
|
2648
|
+
thin = kernel.WithheldCause.INSUFFICIENT_POPULATION.value
|
|
2649
|
+
return _S3Evaluation(
|
|
2650
|
+
established=True,
|
|
2651
|
+
qualifying=tuple(allocated),
|
|
2652
|
+
evaluated=tuple(evaluated),
|
|
2653
|
+
candidate_count=len(candidates),
|
|
2654
|
+
gap_codes=_fanout_gaps(unallocated, ambiguous),
|
|
2655
|
+
evidence={
|
|
2656
|
+
"identifiedSubagentCount": _EvidenceValue(
|
|
2657
|
+
value=identified, qualifications=subset_marks),
|
|
2658
|
+
# Divided by the CLASS's own qualifying USD, never by the report
|
|
2659
|
+
# denominator: this states how concentrated the fan-out is, not
|
|
2660
|
+
# how large it is.
|
|
2661
|
+
"largestSubagentShare": (
|
|
2662
|
+
_EvidenceValue(value=largest / qualifying_usd)
|
|
2663
|
+
if qualifying_usd > 0 else _EvidenceValue(code=thin)),
|
|
2664
|
+
"unallocatedUsd": _EvidenceValue(value=unallocated_usd),
|
|
2665
|
+
},
|
|
2666
|
+
qualifications=tuple(qualifications),
|
|
2667
|
+
)
|
|
2668
|
+
|
|
2669
|
+
|
|
2670
|
+
def _evaluate_subagent_fanout(scope: DiagnosisScope, bundle: StoreBundle,
|
|
2671
|
+
conversations: sqlite3.Connection | None
|
|
2672
|
+
) -> _S3Evaluation:
|
|
2673
|
+
if scope.source == "codex":
|
|
2674
|
+
# Codex derives this from `codex_conversation_threads` and
|
|
2675
|
+
# `codex_session_entries`, both of which live in `cache.db`, so it
|
|
2676
|
+
# needs no transcript authorization and none is claimed.
|
|
2677
|
+
return _evaluate_codex_fanout(scope, bundle)
|
|
2678
|
+
if conversations is None:
|
|
2679
|
+
return _unestablished()
|
|
2680
|
+
return _evaluate_claude_fanout(scope, bundle, conversations)
|
|
2681
|
+
|
|
2682
|
+
|
|
2683
|
+
def _evaluate_claude_fanout(scope: DiagnosisScope, bundle: StoreBundle,
|
|
2684
|
+
conversations: sqlite3.Connection
|
|
2685
|
+
) -> _S3Evaluation:
|
|
2686
|
+
"""Cost bucketed by `(session_id, subagent_key)`, grouped by parent.
|
|
2687
|
+
|
|
2688
|
+
The raw `source_path` never leaves the reader: `_subagent_key` strips the
|
|
2689
|
+
`agent-` prefix and the `.jsonl` suffix and returns the hash alone. Cost
|
|
2690
|
+
comes from the canonical accounting population the adapter has already
|
|
2691
|
+
read at read-time pricing; the outline's `subagent_costs` map is
|
|
2692
|
+
display-only and is reused for its GROUPING only, never as the cost
|
|
2693
|
+
authority.
|
|
2694
|
+
"""
|
|
2695
|
+
query = _conversation_query()
|
|
2696
|
+
facts = bundle.facts
|
|
2697
|
+
bounds = [_iso_sql(scope.window_start), _iso_sql(scope.window_end)]
|
|
2698
|
+
sessions = sorted(
|
|
2699
|
+
str(row[0]) for row in
|
|
2700
|
+
_execute(conversations, _CLAUDE_WINDOW_SESSIONS_SQL, bounds)
|
|
2701
|
+
)
|
|
2702
|
+
if not sessions:
|
|
2703
|
+
return _unestablished()
|
|
2704
|
+
assistant_rows: dict[str, list] = {}
|
|
2705
|
+
for chunk in _chunks(sessions):
|
|
2706
|
+
sql = _CLAUDE_WINDOW_SUBAGENT_SQL.format(
|
|
2707
|
+
placeholders=_placeholders(len(chunk)))
|
|
2708
|
+
for row in _execute(conversations, sql, list(chunk) + bounds):
|
|
2709
|
+
assistant_rows.setdefault(str(row["session_id"]), []).append(row)
|
|
2710
|
+
bucket_of: dict[tuple, set] = {}
|
|
2711
|
+
for session, rows in assistant_rows.items():
|
|
2712
|
+
# The same `(session_id, uuid)` deduplication the canonical fold
|
|
2713
|
+
# applies. Without it a row retained under two source paths puts two
|
|
2714
|
+
# buckets in one turn key, the key resolves to neither, and a
|
|
2715
|
+
# correctly attributable entry is published as unallocated.
|
|
2716
|
+
#
|
|
2717
|
+
# The uuid is read RAW, the way the fold reads it. `uuid` is nullable,
|
|
2718
|
+
# so coercing through `str` would collapse a null and a literal
|
|
2719
|
+
# `"None"` body onto one key here and onto two inside the fold — one
|
|
2720
|
+
# rule with two statements again.
|
|
2721
|
+
for row in query.dedupe_claude_uuid_rows(
|
|
2722
|
+
rows, uuid_of=lambda r: _row_field(r, "uuid")):
|
|
2723
|
+
subagent = query._subagent_key(row["source_path"])
|
|
2724
|
+
if subagent is None:
|
|
2725
|
+
continue
|
|
2726
|
+
bucket_of.setdefault((row["msg_id"], row["req_id"]), set()).add(
|
|
2727
|
+
(session, subagent))
|
|
2728
|
+
buckets_by_parent: dict[str, set] = {}
|
|
2729
|
+
for buckets in bucket_of.values():
|
|
2730
|
+
if len(buckets) != 1:
|
|
2731
|
+
continue # not resolvable to one bucket
|
|
2732
|
+
session, subagent = next(iter(buckets))
|
|
2733
|
+
buckets_by_parent.setdefault(session, set()).add(subagent)
|
|
2734
|
+
qualifying_parents = {
|
|
2735
|
+
session for session, buckets in buckets_by_parent.items()
|
|
2736
|
+
if len(buckets) >= kernel.DIAGNOSIS_MIN_SUBAGENT_BUCKETS
|
|
2737
|
+
}
|
|
2738
|
+
|
|
2739
|
+
known = set(sessions)
|
|
2740
|
+
candidates = [entry for entry in facts.entries
|
|
2741
|
+
if entry.session_key in known]
|
|
2742
|
+
allocated: list = []
|
|
2743
|
+
unallocated: list = []
|
|
2744
|
+
bucket_rows: dict[tuple, list] = {}
|
|
2745
|
+
for entry in candidates:
|
|
2746
|
+
buckets = bucket_of.get((entry.msg_id, entry.req_id))
|
|
2747
|
+
if buckets is not None and len(buckets) == 1:
|
|
2748
|
+
session, subagent = next(iter(buckets))
|
|
2749
|
+
if session in qualifying_parents:
|
|
2750
|
+
allocated.append(entry)
|
|
2751
|
+
bucket_rows.setdefault((session, subagent), []).append(entry)
|
|
2752
|
+
continue
|
|
2753
|
+
if query._subagent_key(entry.source_path) is not None:
|
|
2754
|
+
# A subagent-shaped accounting row that could not join exactly
|
|
2755
|
+
# once to a normalized bucket. Its dollars are neither folded in
|
|
2756
|
+
# nor discarded — they are published as `unallocatedUsd`.
|
|
2757
|
+
unallocated.append(entry)
|
|
2758
|
+
identified = sum(len(buckets_by_parent[session])
|
|
2759
|
+
for session in qualifying_parents)
|
|
2760
|
+
bucket_usd = {key: stable_sum(entry.cost_usd for entry in rows)
|
|
2761
|
+
for key, rows in bucket_rows.items()}
|
|
2762
|
+
return _fanout_evidence(
|
|
2763
|
+
facts, candidates=candidates, unallocated=unallocated,
|
|
2764
|
+
allocated=allocated, bucket_usd=bucket_usd, identified=identified,
|
|
2765
|
+
subset=False,
|
|
2766
|
+
)
|
|
2767
|
+
|
|
2768
|
+
|
|
2769
|
+
def _evaluate_codex_fanout(scope: DiagnosisScope,
|
|
2770
|
+
bundle: StoreBundle) -> _S3Evaluation:
|
|
2771
|
+
"""Delegated Codex work, grouped by the resolved parent thread.
|
|
2772
|
+
|
|
2773
|
+
The predicate is the delegation ORIGIN CATEGORY, matched from
|
|
2774
|
+
`root_thread_id`; `_is_fork` compares `parent_thread_id` against
|
|
2775
|
+
`native_thread_id` and is a separate concept. A parent resolves only on
|
|
2776
|
+
exactly one non-self match under the COMPLETE identity, because the
|
|
2777
|
+
table's uniqueness is the triple and the existing resolution queries the
|
|
2778
|
+
weaker pair.
|
|
2779
|
+
"""
|
|
2780
|
+
facts = bundle.facts
|
|
2781
|
+
scoped = sorted({entry.session_key for entry in facts.entries
|
|
2782
|
+
if entry.session_key})
|
|
2783
|
+
if not scoped:
|
|
2784
|
+
return _unestablished()
|
|
2785
|
+
# The thread rows come from the `cache` component that read and DIGESTED
|
|
2786
|
+
# them, so a changed origin category or parent pointer moves the published
|
|
2787
|
+
# identifier as well as the verdict.
|
|
2788
|
+
fanout = resolve_codex_fanout(list(bundle.codex_threads), scoped)
|
|
2789
|
+
groups: dict[tuple, set] = {}
|
|
2790
|
+
for key, group in fanout.parents.items():
|
|
2791
|
+
groups.setdefault(group, set()).add(key)
|
|
2792
|
+
qualifying_groups = {
|
|
2793
|
+
group for group, children in groups.items()
|
|
2794
|
+
if len(children) >= kernel.DIAGNOSIS_MIN_SUBAGENT_BUCKETS
|
|
2795
|
+
}
|
|
2796
|
+
qualifying_children = {
|
|
2797
|
+
key for key, group in fanout.parents.items()
|
|
2798
|
+
if group in qualifying_groups
|
|
2799
|
+
}
|
|
2800
|
+
unresolvable = set(fanout.unallocated)
|
|
2801
|
+
unreadable = set(fanout.ambiguous)
|
|
2802
|
+
|
|
2803
|
+
candidates = list(facts.entries)
|
|
2804
|
+
allocated: list = []
|
|
2805
|
+
unallocated: list = []
|
|
2806
|
+
ambiguous: list = []
|
|
2807
|
+
bucket_rows: dict[str, list] = {}
|
|
2808
|
+
for entry in candidates:
|
|
2809
|
+
key = entry.session_key
|
|
2810
|
+
if key in unreadable:
|
|
2811
|
+
# The origin category could not be read at all, so this is not
|
|
2812
|
+
# known subagent spend and its dollars are published nowhere.
|
|
2813
|
+
ambiguous.append(entry)
|
|
2814
|
+
continue
|
|
2815
|
+
if key in unresolvable:
|
|
2816
|
+
unallocated.append(entry)
|
|
2817
|
+
continue
|
|
2818
|
+
group = fanout.parents.get(key)
|
|
2819
|
+
if group is None or group not in qualifying_groups:
|
|
2820
|
+
continue
|
|
2821
|
+
if entry.root_key != group[0]:
|
|
2822
|
+
# The join is `(source_root_key, conversation_key)`, and a key
|
|
2823
|
+
# that reached a different root is not this child's spend.
|
|
2824
|
+
unallocated.append(entry)
|
|
2825
|
+
continue
|
|
2826
|
+
allocated.append(entry)
|
|
2827
|
+
bucket_rows.setdefault(key, []).append(entry)
|
|
2828
|
+
bucket_usd = {key: stable_sum(entry.cost_usd for entry in rows)
|
|
2829
|
+
for key, rows in bucket_rows.items()}
|
|
2830
|
+
return _fanout_evidence(
|
|
2831
|
+
facts, candidates=candidates, unallocated=unallocated,
|
|
2832
|
+
allocated=allocated, bucket_usd=bucket_usd,
|
|
2833
|
+
identified=len(qualifying_children), subset=True,
|
|
2834
|
+
ambiguous=ambiguous,
|
|
2835
|
+
)
|
|
2836
|
+
|
|
2837
|
+
|
|
2838
|
+
# --- dispatch and memoization -------------------------------------------
|
|
2839
|
+
|
|
2840
|
+
_S3_EVALUATORS = {
|
|
2841
|
+
"cache_churn": _evaluate_cache_churn,
|
|
2842
|
+
"short_high_context": _evaluate_short_high_context,
|
|
2843
|
+
"subagent_fanout": _evaluate_subagent_fanout,
|
|
2844
|
+
}
|
|
2845
|
+
|
|
2846
|
+
# The classes whose evaluation READS the conversations store. Everything else
|
|
2847
|
+
# an S3 class needs lives in `cache.db`, which every plan already opens.
|
|
2848
|
+
_S3_NEEDS_CONVERSATIONS = {
|
|
2849
|
+
"cache_churn": ("claude",),
|
|
2850
|
+
"short_high_context": ("claude", "codex"),
|
|
2851
|
+
"subagent_fanout": ("claude",),
|
|
2852
|
+
}
|
|
2853
|
+
|
|
2854
|
+
|
|
2855
|
+
def _evaluate_s3_classes(scope: DiagnosisScope, bundle: StoreBundle, *,
|
|
2856
|
+
conversations: sqlite3.Connection | None,
|
|
2857
|
+
only: "Collection[str] | None" = None
|
|
2858
|
+
) -> dict[str, _S3Evaluation]:
|
|
2859
|
+
"""Evaluate every S3 class the POLICY plan permits, once.
|
|
2860
|
+
|
|
2861
|
+
A class the plan withholds or marks `not_applicable` is not evaluated at
|
|
2862
|
+
all — a denied plan never reads the store it was denied. A permitted class
|
|
2863
|
+
whose store is absent is `established=False`, which the loader renders as
|
|
2864
|
+
`withheld / signal_unavailable`.
|
|
2865
|
+
|
|
2866
|
+
`only` narrows the set further, so the lazy fallback in `_s3_evaluations`
|
|
2867
|
+
re-runs nothing a memo already decided.
|
|
2868
|
+
"""
|
|
2869
|
+
results: dict[str, _S3Evaluation] = {}
|
|
2870
|
+
for kind, evaluator in _S3_EVALUATORS.items():
|
|
2871
|
+
if only is not None and kind not in only:
|
|
2872
|
+
continue
|
|
2873
|
+
decision = bundle.plan.mode_for(kind)
|
|
2874
|
+
if decision is None or decision.mode != kernel.ClassMode.MEASURE.value:
|
|
2875
|
+
continue
|
|
2876
|
+
needs = scope.source in _S3_NEEDS_CONVERSATIONS.get(kind, ())
|
|
2877
|
+
if needs and conversations is None:
|
|
2878
|
+
results[kind] = _unestablished()
|
|
2879
|
+
continue
|
|
2880
|
+
try:
|
|
2881
|
+
results[kind] = evaluator(scope, bundle, conversations)
|
|
2882
|
+
except Exception as exc:
|
|
2883
|
+
# An evaluator that raises must not render as healthy, and must
|
|
2884
|
+
# not take down the classes that never read a transcript.
|
|
2885
|
+
#
|
|
2886
|
+
# The cause splits by KIND rather than by layer, so this handler
|
|
2887
|
+
# and the component-level one state one principle through
|
|
2888
|
+
# `_is_store_failure`: a failure a STORE actually produces keeps
|
|
2889
|
+
# its store-shaped cause, and anything else — including a
|
|
2890
|
+
# `sqlite3` error naming a schema object the store was already
|
|
2891
|
+
# observed to carry — is a defect in our own code. Reporting a
|
|
2892
|
+
# defect as an absent signal makes it indistinguishable from a
|
|
2893
|
+
# store that genuinely holds nothing, and only one of those is
|
|
2894
|
+
# worth fixing.
|
|
2895
|
+
results[kind] = _S3Evaluation(
|
|
2896
|
+
established=False,
|
|
2897
|
+
failure=(None if _is_store_failure(exc)
|
|
2898
|
+
else type(exc).__name__))
|
|
2899
|
+
return results
|
|
2900
|
+
|
|
2901
|
+
|
|
2902
|
+
def _s3_evaluations(bundle: StoreBundle,
|
|
2903
|
+
scope: DiagnosisScope) -> dict[str, _S3Evaluation]:
|
|
2904
|
+
"""The memo the loaders read.
|
|
2905
|
+
|
|
2906
|
+
Populated during establishment when the plan opened the conversations
|
|
2907
|
+
store, so the digest and the rows describe ONE evaluation rather than two.
|
|
2908
|
+
A plan that opened no conversations connection — a denied Codex route, for
|
|
2909
|
+
instance — still evaluates its `cache.db`-only classes, lazily, here.
|
|
2910
|
+
|
|
2911
|
+
The fallback keys on WHICH CLASSES the memo covers rather than on whether
|
|
2912
|
+
a memo exists. A conversations read that raises leaves the memo empty
|
|
2913
|
+
rather than absent, and an empty mapping is still not `None`: a
|
|
2914
|
+
`None` test would skip the fallback and withhold Codex `subagent_fanout`,
|
|
2915
|
+
which reads only `cache.db` and which the Codex table of Section 3
|
|
2916
|
+
requires to measure whether that store is absent or unreadable. (C21
|
|
2917
|
+
governs the accounting-store overlay and denial precedence, which is a
|
|
2918
|
+
different rule.) A class the memo already covers is never re-evaluated,
|
|
2919
|
+
so the digest and the rows still describe one evaluation of everything
|
|
2920
|
+
the conversations read decided.
|
|
2921
|
+
"""
|
|
2922
|
+
memo = dict(bundle.s3_evaluations or {})
|
|
2923
|
+
expected = set()
|
|
2924
|
+
for kind in _S3_EVALUATORS:
|
|
2925
|
+
decision = bundle.plan.mode_for(kind)
|
|
2926
|
+
if (decision is not None
|
|
2927
|
+
and decision.mode == kernel.ClassMode.MEASURE.value):
|
|
2928
|
+
expected.add(kind)
|
|
2929
|
+
missing = expected - set(memo)
|
|
2930
|
+
if missing:
|
|
2931
|
+
# Guarded by MEMBERSHIP rather than by `setdefault`, so the sentence
|
|
2932
|
+
# above is literally true: a class the memo already decided is not
|
|
2933
|
+
# evaluated a second time and its result discarded.
|
|
2934
|
+
memo.update(_evaluate_s3_classes(scope, bundle, conversations=None,
|
|
2935
|
+
only=missing))
|
|
2936
|
+
bundle.s3_evaluations = memo
|
|
2937
|
+
return memo
|
|
2938
|
+
|
|
2939
|
+
|
|
2940
|
+
def _read_conversations_component(scope: DiagnosisScope,
|
|
2941
|
+
bundle: StoreBundle) -> tuple[list, Any]:
|
|
2942
|
+
"""Run the conversation-derived evaluators once, and digest what they
|
|
2943
|
+
publish — nothing else.
|
|
2944
|
+
|
|
2945
|
+
`generationId` is on the wire, so a digest computed over raw bodies,
|
|
2946
|
+
`blocks_json`, `content_digest` or payload JSON would let any caller test
|
|
2947
|
+
transcript-content equality, while locating a compaction requires parsing
|
|
2948
|
+
exactly those blocks. The governing rule is therefore narrower and
|
|
2949
|
+
attainable: the component is digested over exactly the derived facts the
|
|
2950
|
+
report itself publishes or aggregates. Under that rule the digest is an
|
|
2951
|
+
equality oracle only for facts the response body already discloses, so it
|
|
2952
|
+
adds no exposure — and a change to semantically INERT text moves neither
|
|
2953
|
+
the digest nor the body.
|
|
2954
|
+
|
|
2955
|
+
The evaluation is performed HERE rather than in the loaders, so the digest
|
|
2956
|
+
and the rows describe one evaluation rather than two.
|
|
2957
|
+
"""
|
|
2958
|
+
conn = bundle.connection("conversations")
|
|
2959
|
+
try:
|
|
2960
|
+
# `EstablishmentFailure` from an unrecognised source cannot reach
|
|
2961
|
+
# here: `_read_with_probe` always probes first, and the probe's
|
|
2962
|
+
# narrower `except sqlite3.Error` lets that refusal through to end the
|
|
2963
|
+
# report. Stated because this site's `except Exception` would
|
|
2964
|
+
# otherwise absorb it, contradicting the gate's own docstring.
|
|
2965
|
+
#
|
|
2966
|
+
# INSIDE the backstop, because the gate itself raises: it re-raises
|
|
2967
|
+
# every `OperationalError` that does not name a missing schema object,
|
|
2968
|
+
# and `database is locked` — the `SQLITE_BUSY_SNAPSHOT` shape this
|
|
2969
|
+
# store produces while `_conversation_sync_pass` commits — is exactly
|
|
2970
|
+
# that. Outside the handler that raise reached no classifier anywhere
|
|
2971
|
+
# in the adapter, in the route or in `cmd_explain`, and took down all
|
|
2972
|
+
# seven classes.
|
|
2973
|
+
incomplete = bundle.conversations_schema_gap()
|
|
2974
|
+
if incomplete is not None:
|
|
2975
|
+
# A store that predates a column these statements select is a
|
|
2976
|
+
# store this read cannot answer from, which is the Section 3
|
|
2977
|
+
# `absent or unreadable` cell. Deciding it HERE, by asking the
|
|
2978
|
+
# store, is what lets a schema-object failure past this point be
|
|
2979
|
+
# reported as our own defect rather than guessed at from an
|
|
2980
|
+
# exception message. The digest row names one of our own table
|
|
2981
|
+
# names and no store content.
|
|
2982
|
+
return [("component_schema_incomplete", incomplete)], {
|
|
2983
|
+
"store_readable": False, "evaluations": {}, "failure": None,
|
|
2984
|
+
}
|
|
2985
|
+
evaluations = _evaluate_s3_classes(scope, bundle, conversations=conn)
|
|
2986
|
+
# INSIDE the backstop. The extension is one loop over the evaluations
|
|
2987
|
+
# the call above produced, and it cannot raise today, but a raise here
|
|
2988
|
+
# would take down all seven classes — including the four accounting
|
|
2989
|
+
# ones that never read a transcript — where the whole point of this
|
|
2990
|
+
# handler is that they keep answering.
|
|
2991
|
+
rows: list[tuple] = []
|
|
2992
|
+
for kind in sorted(evaluations):
|
|
2993
|
+
if scope.source not in _S3_NEEDS_CONVERSATIONS.get(kind, ()):
|
|
2994
|
+
# This class read no conversation bytes, so its facts belong
|
|
2995
|
+
# to the `cache` component rather than to this one.
|
|
2996
|
+
continue
|
|
2997
|
+
rows.extend(evaluations[kind].digest_rows(kind))
|
|
2998
|
+
except Exception as exc:
|
|
2999
|
+
# `except Exception`, not `except sqlite3.Error`. An older
|
|
3000
|
+
# conversations.db carries neither table, and a `blocks_json` array
|
|
3001
|
+
# holding a bare string raises `AttributeError` out of body
|
|
3002
|
+
# reconstruction — which is the NORMAL path for a compaction row,
|
|
3003
|
+
# because compaction blanks `text`.
|
|
3004
|
+
#
|
|
3005
|
+
# The cause splits by KIND, exactly as the per-evaluator handler
|
|
3006
|
+
# does and through the same `_is_store_failure`: a failure a store
|
|
3007
|
+
# actually produces reports `signal_unavailable`, and anything else
|
|
3008
|
+
# is a defect in our own code and reports `calculation_failed`.
|
|
3009
|
+
# Neither is `provider_unavailable` — the accounting store is fine
|
|
3010
|
+
# and the four accounting classes still answer.
|
|
3011
|
+
store_readable = not _is_store_failure(exc)
|
|
3012
|
+
return [("component_failed", type(exc).__name__)], {
|
|
3013
|
+
"store_readable": store_readable, "evaluations": {},
|
|
3014
|
+
"failure": (None if _is_store_failure(exc)
|
|
3015
|
+
else type(exc).__name__),
|
|
3016
|
+
}
|
|
3017
|
+
return rows, {"store_readable": True, "evaluations": evaluations,
|
|
3018
|
+
"failure": None}
|
|
3019
|
+
|
|
3020
|
+
|
|
3021
|
+
def establish_generation(scope: DiagnosisScope,
|
|
3022
|
+
plan: "kernel.ExecutionPlan") -> GenerationVector:
|
|
3023
|
+
"""Establish the version vector alone, closing the stores afterwards.
|
|
3024
|
+
|
|
3025
|
+
The vector this returns binds the CURRENT window only. The published
|
|
3026
|
+
vector binds both windows and is assembled in `build_provider_diagnosis`,
|
|
3027
|
+
which is the only place that holds the baseline bundle.
|
|
3028
|
+
|
|
3029
|
+
`plan` is REQUIRED, and for the same reason `StoreBundle` requires it: the
|
|
3030
|
+
vector describes exactly the components the plan opened, so a default plan
|
|
3031
|
+
would publish a digest over stores this request never authorized.
|
|
3032
|
+
"""
|
|
3033
|
+
with StoreBundle(scope, plan) as bundle:
|
|
3034
|
+
_establish(scope, bundle)
|
|
3035
|
+
assert bundle.vector is not None
|
|
3036
|
+
return bundle.vector
|
|
3037
|
+
|
|
3038
|
+
|
|
3039
|
+
def _establish(scope: DiagnosisScope, bundle: StoreBundle) -> StoreBundle:
|
|
3040
|
+
"""Probe, read, re-probe; re-read once on a divergence, then refuse.
|
|
3041
|
+
|
|
3042
|
+
A second divergence is `generation_incoherent` rather than a third
|
|
3043
|
+
attempt, because a component that moves twice while we read it is a
|
|
3044
|
+
component under active write, and publishing a digest over it would
|
|
3045
|
+
describe a state that never existed as a whole.
|
|
3046
|
+
"""
|
|
3047
|
+
rows: dict[str, list] = {}
|
|
3048
|
+
payloads: dict[str, Any] = {}
|
|
3049
|
+
|
|
3050
|
+
def _read_with_probe(component: str) -> None:
|
|
3051
|
+
for attempt in (0, 1):
|
|
3052
|
+
before = _probe_component(component, bundle)
|
|
3053
|
+
component_rows, payload = _read_component(component, scope, bundle)
|
|
3054
|
+
after = _probe_component(component, bundle)
|
|
3055
|
+
if before == after:
|
|
3056
|
+
rows[component] = list(component_rows)
|
|
3057
|
+
payloads[component] = payload
|
|
3058
|
+
return
|
|
3059
|
+
if attempt == 1:
|
|
3060
|
+
raise EstablishmentFailure(
|
|
3061
|
+
EstablishmentError.GENERATION_INCOHERENT.value,
|
|
3062
|
+
f"the {component} component changed twice while it was read",
|
|
3063
|
+
)
|
|
3064
|
+
|
|
3065
|
+
for component in _UNCONDITIONAL_COMPONENTS:
|
|
3066
|
+
_read_with_probe(component)
|
|
3067
|
+
cache_payload = payloads.get("cache") or {}
|
|
3068
|
+
stats_payload = payloads.get("stats") or {}
|
|
3069
|
+
# The accounting facts are assembled BEFORE the conversations component,
|
|
3070
|
+
# because that component digests the facts the three conversation-derived
|
|
3071
|
+
# evaluators publish and every one of them divides transcript structure
|
|
3072
|
+
# against this accounting population.
|
|
3073
|
+
bundle.codex_threads = tuple(cache_payload.get("codex_threads") or ())
|
|
3074
|
+
bundle.facts = RawFacts(
|
|
3075
|
+
entries=cache_payload.get("entries", ()),
|
|
3076
|
+
blocks=stats_payload.get("blocks", ()),
|
|
3077
|
+
retained_start=cache_payload.get("retained_start"),
|
|
3078
|
+
retained_end=cache_payload.get("retained_end"),
|
|
3079
|
+
store_horizon=stats_payload.get("store_horizon"),
|
|
3080
|
+
unavailable_cause=cache_payload.get("unavailable_cause"),
|
|
3081
|
+
)
|
|
3082
|
+
# A class denied in plan stage 1 is SETTLED, so the store it would have
|
|
3083
|
+
# needed is never opened, probed or digested. The conversations component
|
|
3084
|
+
# is therefore established only when the plan asks for it.
|
|
3085
|
+
if bundle.plan.requires_conversations():
|
|
3086
|
+
try:
|
|
3087
|
+
bundle.connection("conversations")
|
|
3088
|
+
except EstablishmentFailure as exc:
|
|
3089
|
+
if exc.code != EstablishmentError.STORE_UNAVAILABLE.value:
|
|
3090
|
+
raise
|
|
3091
|
+
# An AUTHORIZED open that failed. Only the classes that needed the
|
|
3092
|
+
# store are withheld, as `signal_unavailable`, so this never
|
|
3093
|
+
# reaches `unreadable_store_is_terminal` and never turns an
|
|
3094
|
+
# answered report into exit 3.
|
|
3095
|
+
bundle.conversations_available = False
|
|
3096
|
+
else:
|
|
3097
|
+
_read_with_probe("conversations")
|
|
3098
|
+
payload = payloads.get("conversations") or {}
|
|
3099
|
+
# `store_readable` is a statement about the STORE, which is what
|
|
3100
|
+
# the plan's `signal_unavailable` cell describes. A `sqlite3.Error`
|
|
3101
|
+
# out of the read clears it; a defect in our own code does not,
|
|
3102
|
+
# because the store was readable and the failure is ours.
|
|
3103
|
+
bundle.conversations_available = bool(payload.get("store_readable"))
|
|
3104
|
+
# The evaluation the component just digested IS the one the
|
|
3105
|
+
# loaders publish. Re-running it there would read the store twice
|
|
3106
|
+
# and could describe a different state from the digest.
|
|
3107
|
+
bundle.s3_evaluations = payload.get("evaluations")
|
|
3108
|
+
if payload.get("failure") is not None:
|
|
3109
|
+
# Carry the component's own cause forward per class, so a
|
|
3110
|
+
# defect reports `calculation_failed` rather than being
|
|
3111
|
+
# flattened into "the signal could not be established".
|
|
3112
|
+
bundle.s3_evaluations = {
|
|
3113
|
+
kind: _S3Evaluation(established=False,
|
|
3114
|
+
failure=payload["failure"])
|
|
3115
|
+
for kind in _S3_EVALUATORS
|
|
3116
|
+
}
|
|
3117
|
+
bundle.component_rows = rows
|
|
3118
|
+
bundle.vector = GenerationVector(
|
|
3119
|
+
stats=_digest_rows(rows["stats"]),
|
|
3120
|
+
cache=_digest_rows(rows["cache"]),
|
|
3121
|
+
configuration=_digest_rows(rows["configuration"]),
|
|
3122
|
+
conversations=(_digest_rows(rows["conversations"])
|
|
3123
|
+
if "conversations" in rows else None),
|
|
3124
|
+
)
|
|
3125
|
+
return bundle
|
|
3126
|
+
|
|
3127
|
+
|
|
3128
|
+
# --- coverage -----------------------------------------------------------
|
|
3129
|
+
|
|
3130
|
+
_IDENTITY_FLAGS = {
|
|
3131
|
+
"project": "project_identity_resolved",
|
|
3132
|
+
"session": "session_identity_resolved",
|
|
3133
|
+
}
|
|
3134
|
+
|
|
3135
|
+
|
|
3136
|
+
def _identity_resolved(entry: AccountingEntry, identity_kind: str) -> bool:
|
|
3137
|
+
"""Whether THIS class's subject identity resolved for this entry.
|
|
3138
|
+
|
|
3139
|
+
`model_mix` names its subject by the model string and `five_hour_bursts`
|
|
3140
|
+
by a provider-native block an entry only reaches by matching it, so both
|
|
3141
|
+
resolve by construction over their own attributed subset.
|
|
3142
|
+
"""
|
|
3143
|
+
flag = _IDENTITY_FLAGS.get(identity_kind)
|
|
3144
|
+
if flag is None:
|
|
3145
|
+
return bool(entry.model) if identity_kind == "model" else True
|
|
3146
|
+
return bool(getattr(entry, flag))
|
|
3147
|
+
|
|
3148
|
+
|
|
3149
|
+
def _coverage_for(scope: DiagnosisScope, facts: RawFacts,
|
|
3150
|
+
*, attributed: Sequence[AccountingEntry],
|
|
3151
|
+
identity_kind: str = "model",
|
|
3152
|
+
gap_codes: Sequence[str] = ()) -> PopulationCoverage:
|
|
3153
|
+
"""Coverage for one class, measured over that class's own population.
|
|
3154
|
+
|
|
3155
|
+
`countCoverage` and `usdCoverage` are by definition ratios of the class's
|
|
3156
|
+
attributed subset to the whole provider population, and stay so.
|
|
3157
|
+
`identityCoverage` and `pricingCoverage` are properties OF the attributed
|
|
3158
|
+
subset and are computed over it — publishing the provider figure on every
|
|
3159
|
+
class made `session_concentration` report an identity figure derived from
|
|
3160
|
+
project resolution while its own subjects had resolved perfectly.
|
|
3161
|
+
`retentionCoverage` is genuinely provider-scoped: it compares the store's
|
|
3162
|
+
retained range with the requested window and has no per-class meaning.
|
|
3163
|
+
"""
|
|
3164
|
+
total_usd = facts.total_usd
|
|
3165
|
+
total_count = len(facts.entries)
|
|
3166
|
+
observed = [e.timestamp for e in facts.entries]
|
|
3167
|
+
attributed_usd = stable_sum(e.cost_usd for e in attributed)
|
|
3168
|
+
attributed_count = len(attributed)
|
|
3169
|
+
identity_resolved = sum(1 for e in attributed
|
|
3170
|
+
if _identity_resolved(e, identity_kind))
|
|
3171
|
+
priced_without_fallback = sum(
|
|
3172
|
+
1 for e in attributed if not e.is_fallback_pricing
|
|
3173
|
+
)
|
|
3174
|
+
return PopulationCoverage(
|
|
3175
|
+
requested_start=scope.window_start_iso,
|
|
3176
|
+
requested_end=scope.window_end_iso,
|
|
3177
|
+
observed_start=_iso_z(min(observed)) if observed else None,
|
|
3178
|
+
observed_end=_iso_z(max(observed)) if observed else None,
|
|
3179
|
+
count_coverage=(attributed_count / total_count) if total_count else None,
|
|
3180
|
+
usd_coverage=(attributed_usd / total_usd) if total_usd > 0 else None,
|
|
3181
|
+
identity_coverage=((identity_resolved / attributed_count)
|
|
3182
|
+
if attributed_count else None),
|
|
3183
|
+
retention_coverage=_retention_coverage(scope, facts),
|
|
3184
|
+
pricing_coverage=((priced_without_fallback / attributed_count)
|
|
3185
|
+
if attributed_count else None),
|
|
3186
|
+
support_units=attributed_count,
|
|
3187
|
+
gap_codes=tuple(gap_codes),
|
|
3188
|
+
)
|
|
3189
|
+
|
|
3190
|
+
|
|
3191
|
+
# Handed out BY REFERENCE to every S3 coverage object, so it is wrapped rather
|
|
3192
|
+
# than copied: a plain dict would let one caller mutate the map every other
|
|
3193
|
+
# coverage object is publishing.
|
|
3194
|
+
_S3_DIMENSIONS: Mapping[str, str] = MappingProxyType({
|
|
3195
|
+
"countCoverage": "evaluated entries / all in-window provider entries",
|
|
3196
|
+
"usdCoverage": "evaluated USD / all in-window provider USD",
|
|
3197
|
+
"identityCoverage": "evaluated entries whose subject identity resolved "
|
|
3198
|
+
"/ evaluated entries",
|
|
3199
|
+
"pricingCoverage": "evaluated entries priced from their own row "
|
|
3200
|
+
"/ evaluated entries",
|
|
3201
|
+
"retentionCoverage": "the fraction of the requested window the store "
|
|
3202
|
+
"could answer for",
|
|
3203
|
+
"evaluabilityCoverage": "entries whose predicate could be decided "
|
|
3204
|
+
"/ entries eligible for an attempted evaluation",
|
|
3205
|
+
"supportUnits": "evaluated entries",
|
|
3206
|
+
})
|
|
3207
|
+
|
|
3208
|
+
|
|
3209
|
+
def s3_coverage(scope: DiagnosisScope, facts: RawFacts, *,
|
|
3210
|
+
evaluated: Sequence[AccountingEntry],
|
|
3211
|
+
candidate_count: int,
|
|
3212
|
+
identity_kind: str = "model",
|
|
3213
|
+
gap_codes: Sequence[str] = ()) -> PopulationCoverage:
|
|
3214
|
+
"""Coverage for one conversation-derived class, over three populations.
|
|
3215
|
+
|
|
3216
|
+
S2's `_coverage_for` is NOT modified and keeps serving the four accounting
|
|
3217
|
+
classes, so every S2 coverage block stays byte-identical.
|
|
3218
|
+
|
|
3219
|
+
The three populations are distinct and each field names its own. The
|
|
3220
|
+
CANDIDATE population is every in-window priced entry eligible for an
|
|
3221
|
+
attempted evaluation, INCLUDING the ones that could not be decided —
|
|
3222
|
+
defining a candidate as a row the predicate "was applied to" excludes by
|
|
3223
|
+
construction the very rows the dimension exists to count. The EVALUATED
|
|
3224
|
+
population is the candidates whose predicate could be decided. The
|
|
3225
|
+
QUALIFYING population is the evaluated entries the predicate matched.
|
|
3226
|
+
|
|
3227
|
+
`support_units` counts the evaluated population and never the qualifying
|
|
3228
|
+
set, because counting qualifiers would make confidence rise precisely as
|
|
3229
|
+
the problem worsens.
|
|
3230
|
+
"""
|
|
3231
|
+
total_usd = facts.total_usd
|
|
3232
|
+
total_count = len(facts.entries)
|
|
3233
|
+
observed = [e.timestamp for e in facts.entries]
|
|
3234
|
+
evaluated_usd = stable_sum(e.cost_usd for e in evaluated)
|
|
3235
|
+
evaluated_count = len(evaluated)
|
|
3236
|
+
identity_resolved = sum(1 for e in evaluated
|
|
3237
|
+
if _identity_resolved(e, identity_kind))
|
|
3238
|
+
priced_without_fallback = sum(
|
|
3239
|
+
1 for e in evaluated if not e.is_fallback_pricing
|
|
3240
|
+
)
|
|
3241
|
+
return PopulationCoverage(
|
|
3242
|
+
requested_start=scope.window_start_iso,
|
|
3243
|
+
requested_end=scope.window_end_iso,
|
|
3244
|
+
observed_start=_iso_z(min(observed)) if observed else None,
|
|
3245
|
+
observed_end=_iso_z(max(observed)) if observed else None,
|
|
3246
|
+
count_coverage=(evaluated_count / total_count) if total_count else None,
|
|
3247
|
+
usd_coverage=(evaluated_usd / total_usd) if total_usd > 0 else None,
|
|
3248
|
+
identity_coverage=((identity_resolved / evaluated_count)
|
|
3249
|
+
if evaluated_count else None),
|
|
3250
|
+
retention_coverage=_retention_coverage(scope, facts),
|
|
3251
|
+
pricing_coverage=((priced_without_fallback / evaluated_count)
|
|
3252
|
+
if evaluated_count else None),
|
|
3253
|
+
evaluability_coverage=((evaluated_count / candidate_count)
|
|
3254
|
+
if candidate_count else None),
|
|
3255
|
+
dimensions=_S3_DIMENSIONS,
|
|
3256
|
+
support_units=evaluated_count,
|
|
3257
|
+
gap_codes=tuple(gap_codes),
|
|
3258
|
+
)
|
|
3259
|
+
|
|
3260
|
+
|
|
3261
|
+
def _preempted_coverage(scope: DiagnosisScope, facts: RawFacts,
|
|
3262
|
+
*, cause: str,
|
|
3263
|
+
gap_codes: Sequence[str] = ()) -> PopulationCoverage:
|
|
3264
|
+
"""The coverage of a class that was never evaluated.
|
|
3265
|
+
|
|
3266
|
+
A provider-wide cause is decided before any class shapes its subjects, so
|
|
3267
|
+
the class attributed nothing — and `0` is a measurement over an attributed
|
|
3268
|
+
population, not the absence of one. Publishing `supportUnits: 0` and
|
|
3269
|
+
`usdCoverage: 0.0` for a 60-entry window states that none of its dollars
|
|
3270
|
+
are covered, which is false; the truth is that this class never measured
|
|
3271
|
+
them. The unmeasurable dimensions are therefore absent, and `gap_codes`
|
|
3272
|
+
names the cause that preempted them.
|
|
3273
|
+
|
|
3274
|
+
`retentionCoverage` stays, because it is provider-scoped: it compares the
|
|
3275
|
+
store's retained range with the requested window and is measured before
|
|
3276
|
+
any class exists. The observed bounds stay for the same reason.
|
|
3277
|
+
|
|
3278
|
+
Both mechanisms that produce `provider_unavailable` build their coverage
|
|
3279
|
+
here, which is what makes them publish the same shape rather than two
|
|
3280
|
+
shapes that happen to agree on most keys.
|
|
3281
|
+
"""
|
|
3282
|
+
observed = [e.timestamp for e in facts.entries]
|
|
3283
|
+
return PopulationCoverage(
|
|
3284
|
+
requested_start=scope.window_start_iso,
|
|
3285
|
+
requested_end=scope.window_end_iso,
|
|
3286
|
+
observed_start=_iso_z(min(observed)) if observed else None,
|
|
3287
|
+
observed_end=_iso_z(max(observed)) if observed else None,
|
|
3288
|
+
retention_coverage=_retention_coverage(scope, facts),
|
|
3289
|
+
support_units=None,
|
|
3290
|
+
# The cause first, then whatever the class knows about WHY it could
|
|
3291
|
+
# not be established. A class withheld because every spending session
|
|
3292
|
+
# exhausted its seed share must still say `scan_budget_exhausted`, or
|
|
3293
|
+
# the reader is left with a cause and no reason.
|
|
3294
|
+
gap_codes=(cause,) + tuple(
|
|
3295
|
+
code for code in gap_codes if code != cause),
|
|
3296
|
+
)
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
def _retention_coverage(scope: DiagnosisScope,
|
|
3300
|
+
facts: RawFacts) -> float | None:
|
|
3301
|
+
"""The fraction of the requested window the store could answer for.
|
|
3302
|
+
|
|
3303
|
+
Retention is a statement about the LOW side of the window, and the
|
|
3304
|
+
earliest retained accounting row alone cannot make it: a fresh install two
|
|
3305
|
+
days into the subscription week has exactly the same shape as a store
|
|
3306
|
+
pruned five days back, and reading that shape as pruning withheld every
|
|
3307
|
+
class of a brand-new user's very first `cctally explain`.
|
|
3308
|
+
|
|
3309
|
+
The discriminator is the install's own observation horizon, which stats.db
|
|
3310
|
+
records independently of the accounting rows. An install that was already
|
|
3311
|
+
observing before the window and holds no accounting rows for the earlier
|
|
3312
|
+
part has been pruned. An install whose horizon begins inside the window
|
|
3313
|
+
never covered the earlier part at all, so that interval is excluded from
|
|
3314
|
+
the denominator rather than counted against the store.
|
|
3315
|
+
|
|
3316
|
+
A quiet tail is not a retention failure — no rows after the last piece of
|
|
3317
|
+
work is the ordinary state of every store — so the high side only decides
|
|
3318
|
+
whether the retained range overlaps the window at all.
|
|
3319
|
+
|
|
3320
|
+
**Without the horizon there is no discriminator, so there is no figure.**
|
|
3321
|
+
Neither table that carries it is written by the accounting path:
|
|
3322
|
+
`five_hour_blocks` is written by `record-usage` from the status-line hook,
|
|
3323
|
+
and `quota_window_blocks` needs the optional Codex hooks or a rollout
|
|
3324
|
+
ingest. An install that never wired either has no horizon at all, and
|
|
3325
|
+
falling back to the requested window start restores exactly the rule the
|
|
3326
|
+
horizon replaced — a young store reported as pruned, with a definite
|
|
3327
|
+
figure. An unmeasurable dimension is absent rather than derived from the
|
|
3328
|
+
very bound it replaced. The consequence is stated and accepted: a store
|
|
3329
|
+
with no provider-native blocks never reports `stale_evidence`, because
|
|
3330
|
+
pruning and youth genuinely cannot be told apart without that signal.
|
|
3331
|
+
"""
|
|
3332
|
+
if facts.retained_start is None or facts.retained_end is None:
|
|
3333
|
+
return None
|
|
3334
|
+
horizon = facts.store_horizon
|
|
3335
|
+
if horizon is None:
|
|
3336
|
+
return None
|
|
3337
|
+
answerable_start = scope.window_start
|
|
3338
|
+
if horizon > scope.window_start:
|
|
3339
|
+
answerable_start = min(horizon, scope.window_end)
|
|
3340
|
+
span = (scope.window_end - answerable_start).total_seconds()
|
|
3341
|
+
if span <= 0:
|
|
3342
|
+
# The install began observing at or after the window end, so there is
|
|
3343
|
+
# no interval it could have answered for and no claim to make.
|
|
3344
|
+
return None
|
|
3345
|
+
low = max(answerable_start, facts.retained_start)
|
|
3346
|
+
high = min(scope.window_end, facts.retained_end)
|
|
3347
|
+
if high <= low:
|
|
3348
|
+
return 0.0
|
|
3349
|
+
return min(1.0, (scope.window_end - low).total_seconds() / span)
|
|
3350
|
+
|
|
3351
|
+
|
|
3352
|
+
def _provider_withheld_cause(scope: DiagnosisScope,
|
|
3353
|
+
facts: RawFacts) -> str | None:
|
|
3354
|
+
"""The provider-wide withheld cause, in the published precedence order.
|
|
3355
|
+
|
|
3356
|
+
These are conditions of the whole population rather than of one class, so
|
|
3357
|
+
they are decided once and applied to every class. Deciding them per class
|
|
3358
|
+
would let a class with thin support report `insufficient_population` while
|
|
3359
|
+
the real reason was that the window predates what the store retains.
|
|
3360
|
+
"""
|
|
3361
|
+
if facts.unavailable_cause is not None:
|
|
3362
|
+
return facts.unavailable_cause
|
|
3363
|
+
retention = _retention_coverage(scope, facts)
|
|
3364
|
+
if not facts.entries:
|
|
3365
|
+
if retention is not None and retention <= 0.0:
|
|
3366
|
+
return WithheldCause.RETAINED_RANGE_MISMATCH.value
|
|
3367
|
+
return None
|
|
3368
|
+
if facts.total_usd <= 0.0:
|
|
3369
|
+
# A total of zero has two causes and they are not the same statement.
|
|
3370
|
+
# Nothing in the population could be priced is `pricing_unavailable`.
|
|
3371
|
+
# A priceable population that genuinely cost nothing is not a withheld
|
|
3372
|
+
# measurement at all: the answer is zero, and every class then withholds
|
|
3373
|
+
# on the ordinary "no dollars to divide by" rule.
|
|
3374
|
+
if not any(e.pricing_resolved for e in facts.entries):
|
|
3375
|
+
return WithheldCause.PRICING_UNAVAILABLE.value
|
|
3376
|
+
return None
|
|
3377
|
+
# The same slack `support_shortfall` applies to USD coverage against the
|
|
3378
|
+
# same constant, and the kernel's constant rather than a second copy of it.
|
|
3379
|
+
# Retention is a ratio of two independently computed spans, so a genuinely
|
|
3380
|
+
# half-retained window computes as 0.49999999999999994 and a bare `<`
|
|
3381
|
+
# withheld the WHOLE provider as `stale_evidence`.
|
|
3382
|
+
if (retention is not None
|
|
3383
|
+
and retention < kernel.WITHHOLD_MIN_COVERAGE
|
|
3384
|
+
- kernel.COVERAGE_EPSILON):
|
|
3385
|
+
return WithheldCause.STALE_EVIDENCE.value
|
|
3386
|
+
return None
|
|
3387
|
+
|
|
3388
|
+
|
|
3389
|
+
# --- per-class fact loading --------------------------------------------
|
|
3390
|
+
|
|
3391
|
+
@dataclass(frozen=True)
|
|
3392
|
+
class ClassFacts:
|
|
3393
|
+
contributor_class: str
|
|
3394
|
+
subjects: tuple[SubjectFacts, ...]
|
|
3395
|
+
population: PopulationCoverage
|
|
3396
|
+
total_priced_usd: float
|
|
3397
|
+
not_applicable: bool = False
|
|
3398
|
+
preempting_cause: str | None = None
|
|
3399
|
+
# True when this class's predicate was actually evaluated over an
|
|
3400
|
+
# adequately covered population. With no subjects that is the healthy
|
|
3401
|
+
# inverse — `no_contributor` — rather than a support shortfall.
|
|
3402
|
+
predicate_evaluated: bool = False
|
|
3403
|
+
# Class-specific evidence, keyed by camelCase wire name. Attached to every
|
|
3404
|
+
# row `classify_class` builds for this class.
|
|
3405
|
+
evidence: Mapping[str, Any] | None = None
|
|
3406
|
+
|
|
3407
|
+
|
|
3408
|
+
def _subject_from_group(key: str, label: str,
|
|
3409
|
+
entries: Sequence[AccountingEntry],
|
|
3410
|
+
baseline_share: float | None,
|
|
3411
|
+
baseline_code: str | None,
|
|
3412
|
+
next_step: str | None = None,
|
|
3413
|
+
qualifications: Sequence[str] = ()) -> SubjectFacts:
|
|
3414
|
+
return SubjectFacts(
|
|
3415
|
+
subject_key=key,
|
|
3416
|
+
subject_label=label,
|
|
3417
|
+
observed_usd=stable_sum(e.cost_usd for e in entries),
|
|
3418
|
+
priced_entry_count=len(entries),
|
|
3419
|
+
is_fallback_pricing=any(e.is_fallback_pricing for e in entries),
|
|
3420
|
+
baseline_share=baseline_share,
|
|
3421
|
+
baseline_code=baseline_code,
|
|
3422
|
+
next_step=next_step,
|
|
3423
|
+
qualifications=tuple(qualifications),
|
|
3424
|
+
)
|
|
3425
|
+
|
|
3426
|
+
|
|
3427
|
+
def _group_entries(entries: Sequence[AccountingEntry], kind: str):
|
|
3428
|
+
grouped: dict[str, list[AccountingEntry]] = {}
|
|
3429
|
+
labels: dict[str, str] = {}
|
|
3430
|
+
for entry in entries:
|
|
3431
|
+
if kind == "model":
|
|
3432
|
+
key, label = entry.model or "(unknown)", entry.model or "(unknown)"
|
|
3433
|
+
elif kind == "project":
|
|
3434
|
+
key, label = entry.project_key, entry.project_label
|
|
3435
|
+
else:
|
|
3436
|
+
key, label = entry.session_key, entry.session_label
|
|
3437
|
+
grouped.setdefault(key, []).append(entry)
|
|
3438
|
+
labels.setdefault(key, label)
|
|
3439
|
+
return grouped, labels
|
|
3440
|
+
|
|
3441
|
+
|
|
3442
|
+
def _assign_entries_to_blocks(entries: Sequence[AccountingEntry],
|
|
3443
|
+
blocks: Sequence[NativeBlock]):
|
|
3444
|
+
"""Join accounting entries to provider-native blocks within a pool.
|
|
3445
|
+
|
|
3446
|
+
An entry joins only to a window of a compatible logical pool, and to at
|
|
3447
|
+
most one window, so no entry is ever counted in two pools. An entry
|
|
3448
|
+
matching no compatible window is a coverage gap, reported through
|
|
3449
|
+
`unmatched`, and appears in no block.
|
|
3450
|
+
"""
|
|
3451
|
+
assigned: dict[str, list[AccountingEntry]] = {}
|
|
3452
|
+
unmatched: list[AccountingEntry] = []
|
|
3453
|
+
for entry in entries:
|
|
3454
|
+
match = None
|
|
3455
|
+
for block in blocks:
|
|
3456
|
+
if block.root_key != entry.root_key:
|
|
3457
|
+
continue
|
|
3458
|
+
if block.pool != entry.pool:
|
|
3459
|
+
continue
|
|
3460
|
+
if block.start_at <= entry.timestamp < block.end_at:
|
|
3461
|
+
match = block
|
|
3462
|
+
break
|
|
3463
|
+
if match is None:
|
|
3464
|
+
unmatched.append(entry)
|
|
3465
|
+
continue
|
|
3466
|
+
assigned.setdefault(match.key, []).append(entry)
|
|
3467
|
+
return assigned, unmatched
|
|
3468
|
+
|
|
3469
|
+
|
|
3470
|
+
def _baseline_lookup(bundle: StoreBundle, scope: DiagnosisScope,
|
|
3471
|
+
spec: ContributorSpec):
|
|
3472
|
+
"""The class-specific comparator over the immediately preceding window.
|
|
3473
|
+
|
|
3474
|
+
Returns `(lookup, code)`. `lookup(subject_key, pool) -> float` answers for
|
|
3475
|
+
one of this window's subjects; `code` is set only when the baseline could
|
|
3476
|
+
not be established at all, in which case `lookup` is `None`.
|
|
3477
|
+
|
|
3478
|
+
The comparators are the ones the spec names, and two of them are
|
|
3479
|
+
deliberately not per-key lookups. A 5-hour block's key embeds its start
|
|
3480
|
+
instant and a session key is minted per session, so neither can appear in
|
|
3481
|
+
both windows and a per-key lookup would miss every single time. The
|
|
3482
|
+
comparator for `five_hour_bursts` is therefore the maximum
|
|
3483
|
+
provider-native block share within the subject's own Codex pool, and for
|
|
3484
|
+
`session_concentration` it is the maximum single-session share.
|
|
3485
|
+
|
|
3486
|
+
A subject the baseline does not contain, in a baseline that WAS
|
|
3487
|
+
established, compares to zero rather than being withheld: a model,
|
|
3488
|
+
project or session that appears this week and did not exist last week is
|
|
3489
|
+
the most informative case the baseline has.
|
|
3490
|
+
"""
|
|
3491
|
+
baseline = bundle.baseline
|
|
3492
|
+
if baseline is None or not baseline.entries:
|
|
3493
|
+
return None, kernel.BaselineOutcome.BASELINE_INSUFFICIENT.value
|
|
3494
|
+
total = baseline.total_usd
|
|
3495
|
+
if total <= 0:
|
|
3496
|
+
return None, kernel.BaselineOutcome.BASELINE_INSUFFICIENT.value
|
|
3497
|
+
|
|
3498
|
+
if spec.kind == "five_hour_bursts":
|
|
3499
|
+
assigned, _unmatched = _assign_entries_to_blocks(
|
|
3500
|
+
baseline.entries, baseline.blocks
|
|
3501
|
+
)
|
|
3502
|
+
pool_of = {b.key: (b.pool or "standard") for b in baseline.blocks}
|
|
3503
|
+
best_by_pool: dict[str, float] = {}
|
|
3504
|
+
for key, rows in assigned.items():
|
|
3505
|
+
pool = pool_of.get(key, "standard")
|
|
3506
|
+
share = stable_sum(e.cost_usd for e in rows) / total
|
|
3507
|
+
if share > best_by_pool.get(pool, 0.0):
|
|
3508
|
+
best_by_pool[pool] = share
|
|
3509
|
+
|
|
3510
|
+
def _block_lookup(_key: str, pool: str | None) -> float:
|
|
3511
|
+
return best_by_pool.get(pool or "standard", 0.0)
|
|
3512
|
+
|
|
3513
|
+
return _block_lookup, None
|
|
3514
|
+
|
|
3515
|
+
if spec.kind == "session_concentration":
|
|
3516
|
+
grouped, _labels = _group_entries(baseline.entries, "session")
|
|
3517
|
+
top = max(
|
|
3518
|
+
(stable_sum(e.cost_usd for e in rows) / total
|
|
3519
|
+
for rows in grouped.values()),
|
|
3520
|
+
default=0.0,
|
|
3521
|
+
)
|
|
3522
|
+
|
|
3523
|
+
def _session_lookup(_key: str, _pool: str | None) -> float:
|
|
3524
|
+
return top
|
|
3525
|
+
|
|
3526
|
+
return _session_lookup, None
|
|
3527
|
+
|
|
3528
|
+
kind = "model" if spec.kind == "model_mix" else "project"
|
|
3529
|
+
grouped, _labels = _group_entries(baseline.entries, kind)
|
|
3530
|
+
shares = {key: stable_sum(e.cost_usd for e in rows) / total
|
|
3531
|
+
for key, rows in grouped.items()}
|
|
3532
|
+
|
|
3533
|
+
def _keyed_lookup(key: str, _pool: str | None) -> float:
|
|
3534
|
+
return shares.get(key, 0.0)
|
|
3535
|
+
|
|
3536
|
+
return _keyed_lookup, None
|
|
3537
|
+
|
|
3538
|
+
|
|
3539
|
+
def load_class_facts(bundle: StoreBundle, scope: DiagnosisScope,
|
|
3540
|
+
spec: ContributorSpec) -> ClassFacts:
|
|
3541
|
+
"""Shape one class's subjects and its coverage out of the loaded facts.
|
|
3542
|
+
|
|
3543
|
+
A loader exception is not allowed to render as healthy: it is caught here
|
|
3544
|
+
and reported as `calculation_failed`, which the overall verdict then
|
|
3545
|
+
withholds on.
|
|
3546
|
+
"""
|
|
3547
|
+
facts = bundle.facts
|
|
3548
|
+
provider_cause = _provider_withheld_cause(scope, facts)
|
|
3549
|
+
plan = kernel.establish_plan(
|
|
3550
|
+
bundle.plan,
|
|
3551
|
+
conversations_available=bundle.conversations_available,
|
|
3552
|
+
provider_cause=provider_cause,
|
|
3553
|
+
)
|
|
3554
|
+
decision = plan.mode_for(spec.kind)
|
|
3555
|
+
if decision is not None and decision.mode == kernel.ClassMode.NOT_APPLICABLE.value:
|
|
3556
|
+
# A capability statement rather than an availability one: it stays
|
|
3557
|
+
# true whatever the store did, so the provider overlay never reaches
|
|
3558
|
+
# it and `count_applicable` excludes it.
|
|
3559
|
+
return ClassFacts(
|
|
3560
|
+
spec.kind, (),
|
|
3561
|
+
_preempted_coverage(scope, facts,
|
|
3562
|
+
cause=kernel.VerdictState.NOT_APPLICABLE.value),
|
|
3563
|
+
facts.total_usd, not_applicable=True,
|
|
3564
|
+
)
|
|
3565
|
+
if decision is not None and decision.mode == kernel.ClassMode.WITHHOLD.value:
|
|
3566
|
+
return ClassFacts(
|
|
3567
|
+
spec.kind, (),
|
|
3568
|
+
_preempted_coverage(scope, facts, cause=decision.cause),
|
|
3569
|
+
facts.total_usd, preempting_cause=decision.cause,
|
|
3570
|
+
)
|
|
3571
|
+
if provider_cause is not None:
|
|
3572
|
+
return ClassFacts(
|
|
3573
|
+
spec.kind, (),
|
|
3574
|
+
_preempted_coverage(scope, facts, cause=provider_cause),
|
|
3575
|
+
facts.total_usd, preempting_cause=provider_cause,
|
|
3576
|
+
)
|
|
3577
|
+
try:
|
|
3578
|
+
if spec.kind in _S3_CLASS_KINDS:
|
|
3579
|
+
return _load_s3_class_facts(bundle, scope, spec)
|
|
3580
|
+
return _load_class_facts_inner(bundle, scope, spec)
|
|
3581
|
+
except Exception:
|
|
3582
|
+
return ClassFacts(
|
|
3583
|
+
spec.kind, (),
|
|
3584
|
+
_preempted_coverage(scope, facts,
|
|
3585
|
+
cause=WithheldCause.CALCULATION_FAILED.value),
|
|
3586
|
+
0.0, preempting_cause=WithheldCause.CALCULATION_FAILED.value,
|
|
3587
|
+
)
|
|
3588
|
+
|
|
3589
|
+
|
|
3590
|
+
def _s3_next_step(scope: DiagnosisScope) -> str:
|
|
3591
|
+
"""`cctally explain` over the same scope, with EXACT bounds.
|
|
3592
|
+
|
|
3593
|
+
Transcript-free and valid in every configuration, and it points at exactly
|
|
3594
|
+
the population the row measured — which date grammar cannot express,
|
|
3595
|
+
because a window can begin and end at any instant.
|
|
3596
|
+
"""
|
|
3597
|
+
parts = ["cctally", "explain", "--source", scope.source,
|
|
3598
|
+
"--start-at", scope.window_start_iso,
|
|
3599
|
+
"--end-at", scope.window_end_iso, "--json"]
|
|
3600
|
+
return " ".join(shlex.quote(part) for part in parts)
|
|
3601
|
+
|
|
3602
|
+
|
|
3603
|
+
def _load_s3_class_facts(bundle: StoreBundle, scope: DiagnosisScope,
|
|
3604
|
+
spec: ContributorSpec) -> ClassFacts:
|
|
3605
|
+
"""One conversation-derived class's subjects, coverage and evidence.
|
|
3606
|
+
|
|
3607
|
+
Under D-B the class contributes exactly ONE aggregate subject — the
|
|
3608
|
+
qualifying set — whose observed USD is that set's in-window retained cost.
|
|
3609
|
+
Members appear only as evidence fields, and the subject's key, kind and
|
|
3610
|
+
label are constants of the class rather than anything derived from a
|
|
3611
|
+
member.
|
|
3612
|
+
"""
|
|
3613
|
+
facts = bundle.facts
|
|
3614
|
+
evaluation = _s3_evaluations(bundle, scope).get(spec.kind)
|
|
3615
|
+
if evaluation is None or not evaluation.established:
|
|
3616
|
+
cause = (WithheldCause.CALCULATION_FAILED.value
|
|
3617
|
+
if evaluation is not None and evaluation.failure is not None
|
|
3618
|
+
else WithheldCause.SIGNAL_UNAVAILABLE.value)
|
|
3619
|
+
return ClassFacts(
|
|
3620
|
+
spec.kind, (),
|
|
3621
|
+
_preempted_coverage(
|
|
3622
|
+
scope, facts, cause=cause,
|
|
3623
|
+
gap_codes=(evaluation.gap_codes if evaluation is not None
|
|
3624
|
+
else ())),
|
|
3625
|
+
facts.total_usd, preempting_cause=cause,
|
|
3626
|
+
)
|
|
3627
|
+
population = s3_coverage(
|
|
3628
|
+
scope, facts, evaluated=evaluation.evaluated,
|
|
3629
|
+
candidate_count=evaluation.candidate_count,
|
|
3630
|
+
identity_kind=_identity_kind_for(spec),
|
|
3631
|
+
gap_codes=evaluation.gap_codes,
|
|
3632
|
+
)
|
|
3633
|
+
evidence = {
|
|
3634
|
+
name: (kernel.available(value.value, population, value.qualifications)
|
|
3635
|
+
if value.code is None
|
|
3636
|
+
else kernel.withheld(value.code, population,
|
|
3637
|
+
value.qualifications))
|
|
3638
|
+
for name, value in evaluation.evidence.items()
|
|
3639
|
+
}
|
|
3640
|
+
subjects: tuple[SubjectFacts, ...] = ()
|
|
3641
|
+
if evaluation.qualifying:
|
|
3642
|
+
subject_key, subject_label = _S3_SUBJECT[spec.kind]
|
|
3643
|
+
baseline_share = bundle.s3_baseline_shares.get(spec.kind)
|
|
3644
|
+
subjects = (SubjectFacts(
|
|
3645
|
+
subject_key=subject_key,
|
|
3646
|
+
subject_label=subject_label,
|
|
3647
|
+
observed_usd=evaluation.qualifying_usd,
|
|
3648
|
+
priced_entry_count=len(evaluation.qualifying),
|
|
3649
|
+
is_fallback_pricing=any(entry.is_fallback_pricing
|
|
3650
|
+
for entry in evaluation.qualifying),
|
|
3651
|
+
baseline_share=baseline_share,
|
|
3652
|
+
baseline_code=(None if baseline_share is not None else
|
|
3653
|
+
kernel.BaselineOutcome.BASELINE_INSUFFICIENT.value),
|
|
3654
|
+
next_step=_s3_next_step(scope),
|
|
3655
|
+
qualifications=evaluation.qualifications,
|
|
3656
|
+
),)
|
|
3657
|
+
# `predicate_evaluated` is what makes an evaluated predicate that matched
|
|
3658
|
+
# NOTHING report `no_contributor` rather than a support shortfall: it has
|
|
3659
|
+
# no subjects by RESULT, not by absence of data.
|
|
3660
|
+
return ClassFacts(spec.kind, subjects, population, facts.total_usd,
|
|
3661
|
+
predicate_evaluated=True, evidence=evidence)
|
|
3662
|
+
|
|
3663
|
+
|
|
3664
|
+
def _s3_baseline_shares(bundle: StoreBundle,
|
|
3665
|
+
scope: DiagnosisScope) -> dict[str, float]:
|
|
3666
|
+
"""Each S3 class's aggregate qualifying-cost share over ONE window.
|
|
3667
|
+
|
|
3668
|
+
Called against the preceding-window bundle, which is why the two windows
|
|
3669
|
+
receive SEPARATE scan budgets: a heavy current read cannot starve the
|
|
3670
|
+
baseline and silently turn every baseline into `baseline_insufficient`.
|
|
3671
|
+
"""
|
|
3672
|
+
total = bundle.facts.total_usd
|
|
3673
|
+
if total <= 0:
|
|
3674
|
+
return {}
|
|
3675
|
+
shares: dict[str, float] = {}
|
|
3676
|
+
for kind, evaluation in _s3_evaluations(bundle, scope).items():
|
|
3677
|
+
if not evaluation.established:
|
|
3678
|
+
continue
|
|
3679
|
+
share = evaluation.qualifying_usd / total
|
|
3680
|
+
if share <= 0 and len(evaluation.evaluated) < kernel.spec_for(
|
|
3681
|
+
kind).min_priced_entries:
|
|
3682
|
+
# Spec 2.4 permits a published zero only after an ADEQUATELY
|
|
3683
|
+
# evaluated zero-match. A baseline window holding three evaluated
|
|
3684
|
+
# entries and no match is not evidence that the class was absent
|
|
3685
|
+
# there, and publishing 0.0 for it states a comparison the store
|
|
3686
|
+
# cannot support. It is `baseline_insufficient` instead.
|
|
3687
|
+
continue
|
|
3688
|
+
shares[kind] = share
|
|
3689
|
+
return shares
|
|
3690
|
+
|
|
3691
|
+
|
|
3692
|
+
_CLASS_IDENTITY_KIND = {
|
|
3693
|
+
"model_mix": "model",
|
|
3694
|
+
"project_concentration": "project",
|
|
3695
|
+
"session_concentration": "session",
|
|
3696
|
+
"five_hour_bursts": "block",
|
|
3697
|
+
}
|
|
3698
|
+
|
|
3699
|
+
# The three conversation-derived classes. Named here rather than derived from
|
|
3700
|
+
# `spec.subject_kind`, so a future accounting class that happened to publish
|
|
3701
|
+
# an aggregate subject would not silently join them.
|
|
3702
|
+
_S3_CLASS_KINDS = frozenset({"cache_churn", "short_high_context",
|
|
3703
|
+
"subagent_fanout"})
|
|
3704
|
+
|
|
3705
|
+
|
|
3706
|
+
def _identity_kind_for(spec: ContributorSpec) -> str:
|
|
3707
|
+
return _CLASS_IDENTITY_KIND.get(spec.kind, "model")
|
|
3708
|
+
|
|
3709
|
+
|
|
3710
|
+
def _block_qualifications(block: NativeBlock,
|
|
3711
|
+
scope: DiagnosisScope) -> tuple[str, ...]:
|
|
3712
|
+
"""State an overlap rather than implying the block sits inside the window.
|
|
3713
|
+
|
|
3714
|
+
A native block is loaded when it OVERLAPS the requested window, so a
|
|
3715
|
+
reported burst subject can carry a start instant before the window the
|
|
3716
|
+
report claims to explain. Its dollars are still window-scoped — only
|
|
3717
|
+
entries inside the window are ever assigned — so the honest fix is to say
|
|
3718
|
+
the block extends past the window rather than to clip a provider-native
|
|
3719
|
+
boundary the provider did not clip.
|
|
3720
|
+
"""
|
|
3721
|
+
codes: list[str] = []
|
|
3722
|
+
if block.start_at < scope.window_start:
|
|
3723
|
+
codes.append("block_precedes_window")
|
|
3724
|
+
if block.end_at > scope.window_end:
|
|
3725
|
+
codes.append("block_exceeds_window")
|
|
3726
|
+
return tuple(codes)
|
|
3727
|
+
|
|
3728
|
+
|
|
3729
|
+
def _load_class_facts_inner(bundle: StoreBundle, scope: DiagnosisScope,
|
|
3730
|
+
spec: ContributorSpec) -> ClassFacts:
|
|
3731
|
+
facts = bundle.facts
|
|
3732
|
+
baseline_lookup, baseline_code = _baseline_lookup(bundle, scope, spec)
|
|
3733
|
+
|
|
3734
|
+
def _baseline_for(key: str, pool: str | None) -> float | None:
|
|
3735
|
+
return None if baseline_lookup is None else baseline_lookup(key, pool)
|
|
3736
|
+
|
|
3737
|
+
if spec.kind == "five_hour_bursts":
|
|
3738
|
+
assigned, unmatched = _assign_entries_to_blocks(
|
|
3739
|
+
facts.entries, facts.blocks
|
|
3740
|
+
)
|
|
3741
|
+
blocks_by_key = {b.key: b for b in facts.blocks}
|
|
3742
|
+
subjects = tuple(
|
|
3743
|
+
_subject_from_group(
|
|
3744
|
+
key,
|
|
3745
|
+
blocks_by_key[key].label if key in blocks_by_key else key,
|
|
3746
|
+
rows,
|
|
3747
|
+
_baseline_for(
|
|
3748
|
+
key,
|
|
3749
|
+
blocks_by_key[key].pool if key in blocks_by_key else None,
|
|
3750
|
+
),
|
|
3751
|
+
baseline_code,
|
|
3752
|
+
(blocks_by_key[key].next_step or None
|
|
3753
|
+
if key in blocks_by_key else None),
|
|
3754
|
+
(_block_qualifications(blocks_by_key[key], scope)
|
|
3755
|
+
if key in blocks_by_key else ()),
|
|
3756
|
+
)
|
|
3757
|
+
for key, rows in assigned.items()
|
|
3758
|
+
)
|
|
3759
|
+
gap_codes = ("unmatched_pool_window",) if unmatched else ()
|
|
3760
|
+
attributed = [e for rows in assigned.values() for e in rows]
|
|
3761
|
+
population = _coverage_for(
|
|
3762
|
+
scope, facts, attributed=attributed, identity_kind="block",
|
|
3763
|
+
gap_codes=gap_codes,
|
|
3764
|
+
)
|
|
3765
|
+
return ClassFacts(spec.kind, subjects, population, facts.total_usd)
|
|
3766
|
+
|
|
3767
|
+
kind = _identity_kind_for(spec)
|
|
3768
|
+
grouped, labels = _group_entries(facts.entries, kind)
|
|
3769
|
+
subjects = tuple(
|
|
3770
|
+
_subject_from_group(key, labels[key], rows,
|
|
3771
|
+
_baseline_for(key, None), baseline_code)
|
|
3772
|
+
for key, rows in grouped.items()
|
|
3773
|
+
)
|
|
3774
|
+
attributed = list(facts.entries)
|
|
3775
|
+
|
|
3776
|
+
gap_codes: tuple[str, ...] = ()
|
|
3777
|
+
preempting: str | None = None
|
|
3778
|
+
if kind == "project" and facts.entries:
|
|
3779
|
+
resolved = sum(1 for e in facts.entries
|
|
3780
|
+
if e.project_identity_resolved)
|
|
3781
|
+
if resolved == 0:
|
|
3782
|
+
# A wholly unattributed population is withheld rather than
|
|
3783
|
+
# reported under one synthetic bucket.
|
|
3784
|
+
preempting = WithheldCause.UNATTRIBUTED_EVIDENCE.value
|
|
3785
|
+
gap_codes = ("unresolved_project_identity",)
|
|
3786
|
+
elif resolved < len(facts.entries):
|
|
3787
|
+
gap_codes = ("unresolved_project_identity",)
|
|
3788
|
+
|
|
3789
|
+
population = _coverage_for(
|
|
3790
|
+
scope, facts, attributed=attributed, identity_kind=kind,
|
|
3791
|
+
gap_codes=gap_codes,
|
|
3792
|
+
)
|
|
3793
|
+
return ClassFacts(spec.kind, subjects, population, facts.total_usd,
|
|
3794
|
+
preempting_cause=preempting)
|
|
3795
|
+
|
|
3796
|
+
|
|
3797
|
+
# --- the assembled diagnosis -------------------------------------------
|
|
3798
|
+
|
|
3799
|
+
def _population_digest(facts: RawFacts) -> str:
|
|
3800
|
+
digest = hashlib.sha256()
|
|
3801
|
+
for entry in facts.entries:
|
|
3802
|
+
digest.update(f"{_iso_z(entry.timestamp)}|{entry.model}|"
|
|
3803
|
+
f"{entry.session_key}|{entry.cost_usd!r}".encode())
|
|
3804
|
+
return digest.hexdigest()[:32]
|
|
3805
|
+
|
|
3806
|
+
|
|
3807
|
+
def _next_step_context(scope: DiagnosisScope) -> dict[str, str]:
|
|
3808
|
+
return {
|
|
3809
|
+
"source": scope.source,
|
|
3810
|
+
# Whether the provider's target subcommands can be told to reproduce
|
|
3811
|
+
# this diagnosis's own accounting. The rule lives in the kernel beside
|
|
3812
|
+
# the templates that consume it.
|
|
3813
|
+
"mode_flag": kernel.next_step_mode_flag(scope.source),
|
|
3814
|
+
"window_start_date": scope.window_start.astimezone(UTC).date().isoformat(),
|
|
3815
|
+
"window_end_date": scope.window_end.astimezone(UTC).date().isoformat(),
|
|
3816
|
+
}
|
|
3817
|
+
|
|
3818
|
+
|
|
3819
|
+
def build_provider_diagnosis(scope: DiagnosisScope, *,
|
|
3820
|
+
transcripts_visible: bool
|
|
3821
|
+
) -> kernel.ProviderResult:
|
|
3822
|
+
"""One provider's whole answer, over its own read-only stores.
|
|
3823
|
+
|
|
3824
|
+
`transcripts_visible` is plan stage 1's authorization input. It is always
|
|
3825
|
+
true on the CLI; the route passes its own gate value, and a plan denied
|
|
3826
|
+
here never opens, probes or digests `conversations.db`.
|
|
3827
|
+
|
|
3828
|
+
It carries NO DEFAULT, deliberately. A `True` default is the shape of the
|
|
3829
|
+
R9 defect: the route called this without the argument, the default applied,
|
|
3830
|
+
and every request — including one the transcript gate would have denied —
|
|
3831
|
+
opened, probed and digested `conversations.db` and published an identifier
|
|
3832
|
+
bound to transcript-derived facts. Nothing failed, because a default cannot
|
|
3833
|
+
fail. An unthreaded call site now raises instead.
|
|
3834
|
+
"""
|
|
3835
|
+
policy = kernel.resolve_policy_plan(scope.source,
|
|
3836
|
+
transcripts_visible=transcripts_visible)
|
|
3837
|
+
with StoreBundle(scope, policy) as bundle:
|
|
3838
|
+
_establish(scope, bundle)
|
|
3839
|
+
baseline_rows: dict[str, list] = {}
|
|
3840
|
+
preceding = scope.preceding()
|
|
3841
|
+
try:
|
|
3842
|
+
with StoreBundle(preceding, policy) as baseline_bundle:
|
|
3843
|
+
_establish(preceding, baseline_bundle)
|
|
3844
|
+
bundle.baseline = baseline_bundle.facts
|
|
3845
|
+
baseline_rows = dict(baseline_bundle.component_rows)
|
|
3846
|
+
# The S3 comparator is the preceding window's aggregate
|
|
3847
|
+
# qualifying-cost share, so it is computed while that bundle
|
|
3848
|
+
# is still open — its stores close with the block.
|
|
3849
|
+
bundle.s3_baseline_shares = _s3_baseline_shares(
|
|
3850
|
+
baseline_bundle, preceding)
|
|
3851
|
+
except EstablishmentFailure:
|
|
3852
|
+
# A baseline that cannot be established withholds only the
|
|
3853
|
+
# baseline field. The observed-cost result still renders.
|
|
3854
|
+
bundle.baseline = None
|
|
3855
|
+
|
|
3856
|
+
# The provider coverage and the provider-wide cause are established
|
|
3857
|
+
# BEFORE the denominator, because the denominator is itself an
|
|
3858
|
+
# `EvidenceField` withheld under that same ladder. A definite
|
|
3859
|
+
# `$0.00 of locally retained cost` printed above four classes withheld
|
|
3860
|
+
# as `pricing_unavailable` states a figure nothing in the population
|
|
3861
|
+
# could support.
|
|
3862
|
+
provider_cause = _provider_withheld_cause(scope, bundle.facts)
|
|
3863
|
+
if bundle.facts.unavailable_cause is not None:
|
|
3864
|
+
# The store could not be read, so there is no population to
|
|
3865
|
+
# measure. This is the SAME builder the unopenable-store path
|
|
3866
|
+
# uses, so the two mechanisms that produce `provider_unavailable`
|
|
3867
|
+
# cannot publish different coverage.
|
|
3868
|
+
coverage = _preempted_coverage(
|
|
3869
|
+
scope, bundle.facts, cause=bundle.facts.unavailable_cause
|
|
3870
|
+
)
|
|
3871
|
+
else:
|
|
3872
|
+
coverage = _coverage_for(
|
|
3873
|
+
scope, bundle.facts, attributed=list(bundle.facts.entries),
|
|
3874
|
+
identity_kind="model",
|
|
3875
|
+
)
|
|
3876
|
+
denominator_usd = (
|
|
3877
|
+
kernel.withheld(provider_cause, coverage)
|
|
3878
|
+
if provider_cause is not None
|
|
3879
|
+
else kernel.available(bundle.facts.total_usd, coverage)
|
|
3880
|
+
)
|
|
3881
|
+
denominator = Denominator(
|
|
3882
|
+
identity="totalExplainedRetainedCost",
|
|
3883
|
+
source=scope.source,
|
|
3884
|
+
account_key=scope.account_key,
|
|
3885
|
+
window_start=scope.window_start_iso,
|
|
3886
|
+
window_end=scope.window_end_iso,
|
|
3887
|
+
population_digest=_population_digest(bundle.facts),
|
|
3888
|
+
usd=denominator_usd,
|
|
3889
|
+
)
|
|
3890
|
+
context = _next_step_context(scope)
|
|
3891
|
+
classes: list[ClassResult] = []
|
|
3892
|
+
for spec in kernel.CONTRIBUTOR_REGISTRY:
|
|
3893
|
+
class_facts = load_class_facts(bundle, scope, spec)
|
|
3894
|
+
classes.append(kernel.classify_class(
|
|
3895
|
+
spec,
|
|
3896
|
+
subjects=class_facts.subjects,
|
|
3897
|
+
denominator=denominator,
|
|
3898
|
+
population=class_facts.population,
|
|
3899
|
+
preempting_cause=class_facts.preempting_cause,
|
|
3900
|
+
not_applicable=class_facts.not_applicable,
|
|
3901
|
+
next_step_context=context,
|
|
3902
|
+
predicate_evaluated=class_facts.predicate_evaluated,
|
|
3903
|
+
evidence=class_facts.evidence,
|
|
3904
|
+
))
|
|
3905
|
+
assert bundle.vector is not None
|
|
3906
|
+
# Stage 2: the plan that is HASHED carries the availability outcomes,
|
|
3907
|
+
# not the intentions stage 1 resolved.
|
|
3908
|
+
established = kernel.establish_plan(
|
|
3909
|
+
policy,
|
|
3910
|
+
conversations_available=bundle.conversations_available,
|
|
3911
|
+
provider_cause=provider_cause,
|
|
3912
|
+
)
|
|
3913
|
+
# Each component digest binds BOTH windows, so a baseline that moved
|
|
3914
|
+
# while the current window stood still moves the identifier too.
|
|
3915
|
+
published = GenerationVector(
|
|
3916
|
+
stats=_digest_component_pair(bundle.component_rows.get("stats", []),
|
|
3917
|
+
baseline_rows.get("stats")),
|
|
3918
|
+
cache=_digest_component_pair(bundle.component_rows.get("cache", []),
|
|
3919
|
+
baseline_rows.get("cache")),
|
|
3920
|
+
configuration=_digest_component_pair(
|
|
3921
|
+
bundle.component_rows.get("configuration", []),
|
|
3922
|
+
baseline_rows.get("configuration")),
|
|
3923
|
+
conversations=(
|
|
3924
|
+
_digest_component_pair(bundle.component_rows["conversations"],
|
|
3925
|
+
baseline_rows.get("conversations"))
|
|
3926
|
+
if "conversations" in bundle.component_rows else None),
|
|
3927
|
+
)
|
|
3928
|
+
return kernel.build_provider_result(
|
|
3929
|
+
source=scope.source,
|
|
3930
|
+
account_key=scope.account_key,
|
|
3931
|
+
effective_speed=scope.effective_speed,
|
|
3932
|
+
denominator=denominator,
|
|
3933
|
+
classes=classes,
|
|
3934
|
+
generation=published,
|
|
3935
|
+
coverage=coverage,
|
|
3936
|
+
plan=established,
|
|
3937
|
+
)
|
|
3938
|
+
|
|
3939
|
+
|
|
3940
|
+
def build_diagnosis(scope: DiagnosisScope,
|
|
3941
|
+
*, measured_at: dt.datetime | None = None,
|
|
3942
|
+
transcripts_visible: bool) -> kernel.DiagnosisReport:
|
|
3943
|
+
"""The one entry point both the CLI and the dashboard route call.
|
|
3944
|
+
|
|
3945
|
+
Under `--source all` each provider gets its own scope, its own
|
|
3946
|
+
denominator and its own verdict. Nothing is ranked across providers and
|
|
3947
|
+
no denominator spans them.
|
|
3948
|
+
|
|
3949
|
+
`transcripts_visible` is keyword-required for the reason stated on
|
|
3950
|
+
`build_provider_diagnosis`: a default that means "visible" cannot fail,
|
|
3951
|
+
and the one call site that forgot it was the dashboard route.
|
|
3952
|
+
"""
|
|
3953
|
+
now = measured_at or dt.datetime.now(UTC)
|
|
3954
|
+
sources = (("claude", "codex") if scope.source == "all"
|
|
3955
|
+
else (scope.source,))
|
|
3956
|
+
results = []
|
|
3957
|
+
for source in sources:
|
|
3958
|
+
provider_scope = DiagnosisScope(
|
|
3959
|
+
source=source,
|
|
3960
|
+
account_key=scope.account_key,
|
|
3961
|
+
window_start=scope.window_start,
|
|
3962
|
+
window_end=scope.window_end,
|
|
3963
|
+
effective_speed=scope.effective_speed if source == "codex" else None,
|
|
3964
|
+
display_tz=scope.display_tz,
|
|
3965
|
+
label=scope.label,
|
|
3966
|
+
)
|
|
3967
|
+
try:
|
|
3968
|
+
results.append(build_provider_diagnosis(
|
|
3969
|
+
provider_scope, transcripts_visible=transcripts_visible))
|
|
3970
|
+
except EstablishmentFailure as exc:
|
|
3971
|
+
# A store that cannot be OPENED withholds its provider rather than
|
|
3972
|
+
# ending the request, whether or not a second provider was
|
|
3973
|
+
# requested. The failure is not lost: `unreadable_store_is_terminal`
|
|
3974
|
+
# reads it back off the finished report, and the CLI exits 3 while
|
|
3975
|
+
# still printing the typed cause. Ending the request here instead
|
|
3976
|
+
# gave a single-provider user an exit code and no report, and a
|
|
3977
|
+
# two-provider user a report and exit 0, from one condition.
|
|
3978
|
+
if exc.code != EstablishmentError.STORE_UNAVAILABLE.value:
|
|
3979
|
+
raise
|
|
3980
|
+
results.append(_unavailable_provider_result(
|
|
3981
|
+
provider_scope, detail=exc.message,
|
|
3982
|
+
transcripts_visible=transcripts_visible))
|
|
3983
|
+
return kernel.build_report(_iso_z(now), scope.window(), results)
|
|
3984
|
+
|
|
3985
|
+
|
|
3986
|
+
def _unavailable_provider_result(scope: DiagnosisScope, *,
|
|
3987
|
+
detail: str = "",
|
|
3988
|
+
transcripts_visible: bool
|
|
3989
|
+
) -> kernel.ProviderResult:
|
|
3990
|
+
"""One provider withheld as `provider_unavailable`, every class included.
|
|
3991
|
+
|
|
3992
|
+
Withheld is a field-level statement, so the class rows still exist and
|
|
3993
|
+
still name their cause; only the measurement is absent.
|
|
3994
|
+
|
|
3995
|
+
`detail` carries the failure's own message onto the withheld denominator's
|
|
3996
|
+
qualifications. It exists because one of the failures reaching here is
|
|
3997
|
+
`QuotaProjectionIncomplete`, which is a RETRY signal that names its
|
|
3998
|
+
remedy — `run \\`cctally cache-sync\\` to reconcile it` — and discarding
|
|
3999
|
+
the message left the user reading `withheld (provider_unavailable)` for a
|
|
4000
|
+
store that is readable and one command away from answering.
|
|
4001
|
+
"""
|
|
4002
|
+
coverage = _preempted_coverage(
|
|
4003
|
+
scope, RawFacts(), cause=WithheldCause.PROVIDER_UNAVAILABLE.value
|
|
4004
|
+
)
|
|
4005
|
+
denominator = Denominator(
|
|
4006
|
+
identity="totalExplainedRetainedCost",
|
|
4007
|
+
source=scope.source,
|
|
4008
|
+
account_key=scope.account_key,
|
|
4009
|
+
window_start=scope.window_start_iso,
|
|
4010
|
+
window_end=scope.window_end_iso,
|
|
4011
|
+
population_digest="",
|
|
4012
|
+
usd=kernel.withheld(WithheldCause.PROVIDER_UNAVAILABLE.value, coverage,
|
|
4013
|
+
(detail,) if detail else ()),
|
|
4014
|
+
)
|
|
4015
|
+
# The SAME two-stage plan the readable path resolves, with the accounting
|
|
4016
|
+
# overlay applied. It is what keeps a capability statement true here too:
|
|
4017
|
+
# Codex prompt-cache churn stays `not_applicable` rather than becoming a
|
|
4018
|
+
# seventh `provider_unavailable` row, because whether the store could be
|
|
4019
|
+
# read has no bearing on whether the provider retains a loss predicate.
|
|
4020
|
+
plan = kernel.establish_plan(
|
|
4021
|
+
kernel.resolve_policy_plan(scope.source,
|
|
4022
|
+
transcripts_visible=transcripts_visible),
|
|
4023
|
+
conversations_available=False,
|
|
4024
|
+
provider_cause=WithheldCause.PROVIDER_UNAVAILABLE.value,
|
|
4025
|
+
)
|
|
4026
|
+
# A `not_applicable` class names its own reason, exactly as the readable
|
|
4027
|
+
# path does, so a client cannot tell the two `provider_unavailable`
|
|
4028
|
+
# mechanisms apart by their coverage.
|
|
4029
|
+
not_applicable_coverage = _preempted_coverage(
|
|
4030
|
+
scope, RawFacts(), cause=kernel.VerdictState.NOT_APPLICABLE.value
|
|
4031
|
+
)
|
|
4032
|
+
classes = [
|
|
4033
|
+
ClassResult(
|
|
4034
|
+
decision.kind,
|
|
4035
|
+
kernel.VerdictState.NOT_APPLICABLE.value, None, (),
|
|
4036
|
+
not_applicable_coverage,
|
|
4037
|
+
)
|
|
4038
|
+
if decision.mode == kernel.ClassMode.NOT_APPLICABLE.value
|
|
4039
|
+
else ClassResult(
|
|
4040
|
+
decision.kind, kernel.VerdictState.WITHHELD.value,
|
|
4041
|
+
decision.cause, (), coverage,
|
|
4042
|
+
)
|
|
4043
|
+
for decision in plan.classes
|
|
4044
|
+
]
|
|
4045
|
+
return kernel.build_provider_result(
|
|
4046
|
+
source=scope.source,
|
|
4047
|
+
account_key=scope.account_key,
|
|
4048
|
+
effective_speed=scope.effective_speed,
|
|
4049
|
+
denominator=denominator,
|
|
4050
|
+
classes=classes,
|
|
4051
|
+
generation=None,
|
|
4052
|
+
coverage=coverage,
|
|
4053
|
+
plan=plan,
|
|
4054
|
+
)
|