cctally 1.103.0 → 1.104.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -0
- package/bin/_cctally_alerts.py +65 -4
- package/bin/_cctally_config.py +62 -2
- package/bin/_cctally_core.py +62 -1
- package/bin/_cctally_dashboard.py +11 -0
- package/bin/_cctally_dashboard_envelope.py +319 -14
- package/bin/_cctally_dashboard_share.py +56 -25
- package/bin/_cctally_doctor.py +63 -0
- package/bin/_cctally_forecast.py +917 -44
- package/bin/_cctally_journal.py +153 -5
- package/bin/_cctally_parser.py +66 -0
- package/bin/_cctally_project.py +535 -10
- package/bin/_cctally_quota.py +13 -0
- package/bin/_cctally_quota_calibration.py +146 -0
- package/bin/_cctally_quota_model.py +1616 -0
- package/bin/_cctally_record.py +114 -6
- package/bin/_cctally_share.py +16 -8
- package/bin/_cctally_statusline.py +34 -0
- package/bin/_cctally_tui.py +214 -45
- package/bin/_lib_dashboard_settings_contract.py +2 -0
- package/bin/_lib_doctor.py +159 -1
- package/bin/_lib_forecast.py +337 -43
- package/bin/_lib_meter_rate_change.py +294 -0
- package/bin/_lib_quota_calibration.py +311 -0
- package/bin/_lib_quota_copy.py +131 -0
- package/bin/_lib_quota_model.py +2333 -0
- package/bin/_lib_rederive.py +10 -0
- package/bin/_lib_render.py +6 -0
- package/bin/_lib_share_templates.py +37 -5
- package/bin/_lib_statusline.py +226 -2
- package/bin/_lib_view_models.py +30 -12
- package/bin/cctally +32 -0
- package/dashboard/static/assets/index-D19TO7Mg.js +97 -0
- package/dashboard/static/assets/index-klO46NcU.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +7 -1
- package/dashboard/static/assets/index-Di2hljvB.css +0 -1
- package/dashboard/static/assets/index-XYCIWjVG.js +0 -97
package/bin/_cctally_project.py
CHANGED
|
@@ -45,6 +45,338 @@ def _cctally():
|
|
|
45
45
|
return sys.modules["cctally"]
|
|
46
46
|
|
|
47
47
|
|
|
48
|
+
# ---------------------------------------------------------------------------
|
|
49
|
+
# Modelled quota attribution (#661 S2 spec §5)
|
|
50
|
+
#
|
|
51
|
+
# `Used %` used to be a project's share of the window's DOLLARS, scaled by the
|
|
52
|
+
# week's meter reading. Cache reads are far cheaper per token than output
|
|
53
|
+
# under the model's weights, so that proxy over-credits a cache-heavy project
|
|
54
|
+
# and under-credits an output-heavy Opus one: it reports a cost share under a
|
|
55
|
+
# column whose name says quota.
|
|
56
|
+
# ---------------------------------------------------------------------------
|
|
57
|
+
|
|
58
|
+
#: Why modelled quota is not published for a week or a whole run. A closed
|
|
59
|
+
#: set. `account-not-resolved` is the run-level one (§5.3); the rest are
|
|
60
|
+
#: per-week and are the reasons that week fell back to the cost share.
|
|
61
|
+
ATTRIBUTION_CAUSES = ("account-not-resolved", "calibration-absent",
|
|
62
|
+
"regime-boundary", "unsupported-composition",
|
|
63
|
+
"no-local-history")
|
|
64
|
+
|
|
65
|
+
#: The three bases a row or a run can carry.
|
|
66
|
+
ATTRIBUTION_BASES = ("modelled", "cost-share", "withheld")
|
|
67
|
+
|
|
68
|
+
#: The cause a residual states when it cannot be computed against a whole,
|
|
69
|
+
#: single-account, unfiltered population. A residual over a partial
|
|
70
|
+
#: population is not a residual.
|
|
71
|
+
RESIDUAL_MISALIGNED = "population-misaligned"
|
|
72
|
+
|
|
73
|
+
#: The other two reasons a residual is absent, which are NOT misalignment
|
|
74
|
+
#: (#661 S2 Stage C review). Spec §5.2 states the first outright — "an
|
|
75
|
+
#: absent observed side withholds on its own" — as a condition separate from
|
|
76
|
+
#: the account, window and filter alignment `RESIDUAL_MISALIGNED` names. All
|
|
77
|
+
#: three used to render as `population-misaligned`, and once the §5.2 footer
|
|
78
|
+
#: put that code in front of a terminal user as the sentence "the account,
|
|
79
|
+
#: the window or the population does not align", a run whose only defect was
|
|
80
|
+
#: a modelled week without a meter snapshot told the user something false
|
|
81
|
+
#: about their own request. This is the same defect class Stage C fixed for
|
|
82
|
+
#: `credit-baseline-absent` versus `reset-baseline-absent`: the withholding
|
|
83
|
+
#: is right and the named cause is wrong.
|
|
84
|
+
RESIDUAL_OBSERVED_ABSENT = "observed-absent"
|
|
85
|
+
RESIDUAL_NO_MODELLED_WEEKS = "no-modelled-weeks"
|
|
86
|
+
|
|
87
|
+
#: The closed set of residual causes, asserted by the tests exactly as
|
|
88
|
+
#: `MOVEMENT_WITHHELD_CAUSES` is. A member with no copy in
|
|
89
|
+
#: `_ATTRIBUTION_CAUSE_COPY` renders as its own code, which is readable but
|
|
90
|
+
#: is not the sentence the footer owes a terminal reader.
|
|
91
|
+
RESIDUAL_CAUSES = (RESIDUAL_MISALIGNED, RESIDUAL_OBSERVED_ABSENT,
|
|
92
|
+
RESIDUAL_NO_MODELLED_WEEKS)
|
|
93
|
+
|
|
94
|
+
#: Bumped from 1 because `attributedUsedPercent` and `costPerPercent` keep
|
|
95
|
+
#: their spellings and change their MEANING, from a share of the window's
|
|
96
|
+
#: dollars to modelled weekly quota. `docs/cli-contract.md` classifies a
|
|
97
|
+
#: changed value meaning as breaking. The additive `attribution` block and
|
|
98
|
+
#: the per-row `attributionBasis` would not on their own have required it.
|
|
99
|
+
PROJECT_JSON_SCHEMA_VERSION = 2
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class WeekAttribution:
|
|
103
|
+
"""How ONE subscription week's `Used %` is measured.
|
|
104
|
+
|
|
105
|
+
`units_per_point` is set only on the `modelled` basis; a cost-share week
|
|
106
|
+
carries the typed cause of its fallback instead.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
__slots__ = ("basis", "cause", "units_per_point")
|
|
110
|
+
|
|
111
|
+
def __init__(self, basis, cause, units_per_point):
|
|
112
|
+
self.basis = basis
|
|
113
|
+
self.cause = cause
|
|
114
|
+
self.units_per_point = units_per_point
|
|
115
|
+
|
|
116
|
+
def __repr__(self): # pragma: no cover
|
|
117
|
+
return (f"WeekAttribution(basis={self.basis!r}, cause={self.cause!r},"
|
|
118
|
+
f" units_per_point={self.units_per_point!r})")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _entry_quota_record(entry):
|
|
122
|
+
"""`(EntryRecord, weighted_units)` for one joined cache entry.
|
|
123
|
+
|
|
124
|
+
`units` is None when the entry contributes nothing to the general weekly
|
|
125
|
+
meter — a family that does not drain it, or a cache-write split the store
|
|
126
|
+
cannot supply. Both are exactly what `population_units` skips, so the
|
|
127
|
+
per-project parts keep summing to the population's own units rather than
|
|
128
|
+
drifting above it.
|
|
129
|
+
|
|
130
|
+
This runs on the RAW entry, before `_accumulate_entry_into_bucket`, which
|
|
131
|
+
drops the one-hour cache-write split. Weighting an aggregate bucket would
|
|
132
|
+
silently mis-price every cache-heavy project — the failure spec §5.1 names.
|
|
133
|
+
"""
|
|
134
|
+
qm = _cctally()._load_sibling("_lib_quota_model")
|
|
135
|
+
model = str(getattr(entry, "model", "") or "")
|
|
136
|
+
if qm.family_participation(qm.normalize_family(model)) != "general":
|
|
137
|
+
return None, None
|
|
138
|
+
record = qm.EntryRecord(
|
|
139
|
+
at=entry.timestamp,
|
|
140
|
+
model=model,
|
|
141
|
+
fresh=getattr(entry, "input_tokens", 0) or 0,
|
|
142
|
+
output=getattr(entry, "output_tokens", 0) or 0,
|
|
143
|
+
cache_create_total=getattr(entry, "cache_creation_tokens", 0) or 0,
|
|
144
|
+
cache_1h=getattr(entry, "cache_1h_tokens", None),
|
|
145
|
+
cache_read=getattr(entry, "cache_read_tokens", 0) or 0,
|
|
146
|
+
)
|
|
147
|
+
return record, qm.weighted_units(record)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def probe_provider_decoration(conn, provider: str) -> bool:
|
|
151
|
+
"""Whether `provider` renders account decoration, failing CLOSED.
|
|
152
|
+
|
|
153
|
+
Spec §5.3 hangs on this answer: `False` lets a merged read publish a
|
|
154
|
+
number under `Used %`, and on a genuinely decorated store that number is
|
|
155
|
+
the cost share F1 exists to remove. So only ONE condition may relax the
|
|
156
|
+
gate — the `accounts` table not being there at all, which really does
|
|
157
|
+
mean the store cannot hold more than one real account, and which several
|
|
158
|
+
hand-built fixtures are thin enough to hit.
|
|
159
|
+
|
|
160
|
+
Catching `sqlite3.Error` around the decoration query itself was much
|
|
161
|
+
wider than that: it also covers `database is locked`, `file is not a
|
|
162
|
+
database`, `disk I/O error` and `no such column`, and a WAL-contended
|
|
163
|
+
two-account store answering `False` is exactly the publication §5.3
|
|
164
|
+
forbids. Every such failure therefore answers `True`, which withholds.
|
|
165
|
+
The table probe is the `PRAGMA table_info` form `_snapshot_columns` uses.
|
|
166
|
+
"""
|
|
167
|
+
c = _cctally()
|
|
168
|
+
try:
|
|
169
|
+
present = bool(conn.execute("PRAGMA table_info(accounts)").fetchall())
|
|
170
|
+
except sqlite3.Error:
|
|
171
|
+
return True
|
|
172
|
+
if not present:
|
|
173
|
+
return False
|
|
174
|
+
try:
|
|
175
|
+
return bool(c.provider_is_decorated(conn, provider))
|
|
176
|
+
except sqlite3.Error:
|
|
177
|
+
return True
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def resolve_attribution_account_gate(*, account_key, decorated):
|
|
181
|
+
"""`(basis, cause)` when the run cannot model at all, else `(None, None)`.
|
|
182
|
+
|
|
183
|
+
Spec §5.3. `project` without `--account` passes `account_key=None`, which
|
|
184
|
+
means MERGED, and S1 publishes no valid merged calibration. On a decorated
|
|
185
|
+
multi-account install, modelled quota is therefore withheld and
|
|
186
|
+
`--account` is required: silently falling back to the cost share there
|
|
187
|
+
would keep publishing the very number F1 exists to remove, under a column
|
|
188
|
+
the acceptance criterion says reports the correct account.
|
|
189
|
+
|
|
190
|
+
At a single real account the #341 R8 gate means nothing decorates and the
|
|
191
|
+
merged path IS the account path, so this affects only genuinely
|
|
192
|
+
multi-account installs.
|
|
193
|
+
"""
|
|
194
|
+
if account_key is None and decorated:
|
|
195
|
+
return "withheld", "account-not-resolved"
|
|
196
|
+
return None, None
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def resolve_week_attribution(regime, records, *, week_start, week_end):
|
|
200
|
+
"""Whether one subscription week can be modelled, and at what rate.
|
|
201
|
+
|
|
202
|
+
A week whose entries do not all fall inside the regime's half-open
|
|
203
|
+
interval falls back for the WHOLE week. The validated reader publishes
|
|
204
|
+
only the OPEN regime, so the earlier segment has no rate at all, and
|
|
205
|
+
spec §5.1 forbids producing a partial-week mixture — modelling half a
|
|
206
|
+
week and cost-sharing the other half would publish a figure that is
|
|
207
|
+
neither.
|
|
208
|
+
|
|
209
|
+
Support is re-tested over the whole account-week population through
|
|
210
|
+
§1.1's apply adapter rather than per project, and rather than inherited
|
|
211
|
+
from the regime's stored status, which describes S1's fit population and
|
|
212
|
+
not this one.
|
|
213
|
+
"""
|
|
214
|
+
if regime is None:
|
|
215
|
+
return WeekAttribution("cost-share", "calibration-absent", None)
|
|
216
|
+
if week_start < regime.effective_from:
|
|
217
|
+
return WeekAttribution("cost-share", "regime-boundary", None)
|
|
218
|
+
if regime.effective_until is not None \
|
|
219
|
+
and week_end > regime.effective_until:
|
|
220
|
+
return WeekAttribution("cost-share", "regime-boundary", None)
|
|
221
|
+
qcg = _cctally()._load_sibling("_cctally_quota_calibration")
|
|
222
|
+
applied = qcg.apply_regime(regime, list(records))
|
|
223
|
+
if isinstance(applied, qcg.ApplyRejection):
|
|
224
|
+
return WeekAttribution("cost-share", applied.value, None)
|
|
225
|
+
return WeekAttribution("modelled", None, regime.units_per_point)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def range_covers_whole_weeks(since, until, bounds, *, now) -> bool:
|
|
229
|
+
"""Whether `[since, until]` slices no subscription week it touches.
|
|
230
|
+
|
|
231
|
+
`whole_weeks` used to be `not weeks_missing_snapshot`, which asks a
|
|
232
|
+
different question — whether every week has a snapshot. A three-day
|
|
233
|
+
`--since 2026-06-03 --until 2026-06-05` over one fully-snapshotted week
|
|
234
|
+
passed it, and §5.2's residual was then published against a meter
|
|
235
|
+
reading covering the whole week and a modelled population covering three
|
|
236
|
+
days of it. That is exactly the misalignment §5.2 names.
|
|
237
|
+
|
|
238
|
+
The snapshot condition is not lost by the replacement: the residual's
|
|
239
|
+
observed side is already `None` unless every modelled week carries a
|
|
240
|
+
snapshot, and a `None` observed side withholds on its own.
|
|
241
|
+
|
|
242
|
+
The OPEN week is covered whole by a range running to `now`, because the
|
|
243
|
+
meter reading and the local entries both stop there and neither side is
|
|
244
|
+
clipped relative to the other. A closed week needs the range to reach
|
|
245
|
+
its recorded end.
|
|
246
|
+
"""
|
|
247
|
+
if not bounds:
|
|
248
|
+
return False
|
|
249
|
+
if since > bounds[0][0]:
|
|
250
|
+
return False
|
|
251
|
+
last_end = bounds[-1][1]
|
|
252
|
+
if last_end > now:
|
|
253
|
+
return until >= now
|
|
254
|
+
# `until` is an INCLUSIVE instant and a week's end is EXCLUSIVE, so a
|
|
255
|
+
# date-only `--until 2026-06-07` arrives as 23:59:59.999999 and covers a
|
|
256
|
+
# week ending 2026-06-08T00:00:00 exactly. One microsecond is the
|
|
257
|
+
# resolution `datetime` has, not a tolerance.
|
|
258
|
+
return until >= last_end - dt.timedelta(microseconds=1)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def residual_withholding_cause(*, account_resolved, whole_weeks, filtered,
|
|
262
|
+
fallback_weeks):
|
|
263
|
+
"""None when the observed-minus-modelled residual may be stated.
|
|
264
|
+
|
|
265
|
+
Spec §5.2: only when the account, the window and the population align —
|
|
266
|
+
a single resolved account, whole subscription weeks, and no fallback or
|
|
267
|
+
filter splitting the population.
|
|
268
|
+
"""
|
|
269
|
+
if not account_resolved or not whole_weeks or filtered or fallback_weeks:
|
|
270
|
+
return RESIDUAL_MISALIGNED
|
|
271
|
+
return None
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
#: Human copy for the typed causes the footer renders. A cause with no entry
|
|
275
|
+
#: renders as its own code, which is readable and never blank.
|
|
276
|
+
_ATTRIBUTION_CAUSE_COPY: dict = {
|
|
277
|
+
"account-not-resolved":
|
|
278
|
+
"this install has more than one account and the read is merged; "
|
|
279
|
+
"pass --account to model quota",
|
|
280
|
+
"calibration-absent": "no usable quota calibration on this install",
|
|
281
|
+
"regime-boundary": "the week crosses a metering-rate boundary",
|
|
282
|
+
"unsupported-composition":
|
|
283
|
+
"the week's model mix sits outside the calibration's support",
|
|
284
|
+
"no-local-history": "no local entries to model",
|
|
285
|
+
RESIDUAL_MISALIGNED:
|
|
286
|
+
"the account, the window or the population does not align",
|
|
287
|
+
RESIDUAL_OBSERVED_ABSENT:
|
|
288
|
+
"at least one modelled week carries no meter snapshot, so there is "
|
|
289
|
+
"no observed side to subtract from",
|
|
290
|
+
RESIDUAL_NO_MODELLED_WEEKS:
|
|
291
|
+
"this window models no subscription week",
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _cause_copy(cause) -> str:
|
|
296
|
+
if cause is None:
|
|
297
|
+
return "no cause stated"
|
|
298
|
+
return _ATTRIBUTION_CAUSE_COPY.get(cause, str(cause))
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def render_attribution_footer(totals, *, basis, cause) -> "list[str]":
|
|
302
|
+
"""Spec §5.2's footer lines for the terminal `project` table.
|
|
303
|
+
|
|
304
|
+
The four quantities are named separately rather than conflated into "the
|
|
305
|
+
rows do not add up", and the residual is stated with its SIGN AS
|
|
306
|
+
MEASURED and never with a direction: §2.2 measured that difference with
|
|
307
|
+
both signs, so a footer asserting one would be false half the time. The
|
|
308
|
+
same sentence says outright that the difference does not identify or
|
|
309
|
+
estimate off-machine usage, because §2.2's probe in that direction is
|
|
310
|
+
structurally blind and §2.2 forbids the claim.
|
|
311
|
+
|
|
312
|
+
Returns a list of lines so the caller appends without an inline guard.
|
|
313
|
+
A withheld quantity states its cause and no number; it is absent rather
|
|
314
|
+
than zero.
|
|
315
|
+
"""
|
|
316
|
+
totals = totals or {}
|
|
317
|
+
lines: list[str] = []
|
|
318
|
+
if basis == "withheld":
|
|
319
|
+
lines.append(
|
|
320
|
+
f"Used %: modelled quota withheld \u2014 {_cause_copy(cause)}.")
|
|
321
|
+
else:
|
|
322
|
+
modelled = totals.get("modelledWeekPoints")
|
|
323
|
+
visible = totals.get("visibleRowPoints")
|
|
324
|
+
unmodelled = totals.get("filteredOrUnmodelledPoints")
|
|
325
|
+
if modelled is None:
|
|
326
|
+
lines.append(
|
|
327
|
+
"Modelled quota: withheld \u2014 "
|
|
328
|
+
f"{_cause_copy(cause)}. Used % is a cost share.")
|
|
329
|
+
else:
|
|
330
|
+
parts = [f"{modelled:,.2f} points across the modelled weeks"]
|
|
331
|
+
if visible is not None:
|
|
332
|
+
parts.append(f"{visible:,.2f} in the rows listed")
|
|
333
|
+
if unmodelled is not None:
|
|
334
|
+
parts.append(
|
|
335
|
+
f"{unmodelled:,.2f} filtered or unmodelled")
|
|
336
|
+
lines.append("Modelled quota: " + "; ".join(parts) + ".")
|
|
337
|
+
residual = totals.get("observedMinusModelledPoints")
|
|
338
|
+
residual_cause = totals.get("residualCause")
|
|
339
|
+
if residual is None:
|
|
340
|
+
lines.append(
|
|
341
|
+
"Observed meter minus modelled local quota: withheld \u2014 "
|
|
342
|
+
f"{_cause_copy(residual_cause)}.")
|
|
343
|
+
else:
|
|
344
|
+
lines.append(
|
|
345
|
+
"Observed meter minus modelled local quota: "
|
|
346
|
+
f"{residual:+,.2f} points, as measured. This is a difference "
|
|
347
|
+
"between two quantities, not an identification or an estimate "
|
|
348
|
+
"of usage from another machine.")
|
|
349
|
+
return lines
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def residual_absence_cause(*, observed, modelled):
|
|
353
|
+
"""Why `observed_minus_modelled` returned None, or None when it did not.
|
|
354
|
+
|
|
355
|
+
A pure classifier over the two operands, so the caller states which side
|
|
356
|
+
is missing instead of reporting every absence as misalignment. The
|
|
357
|
+
modelled side is checked first because when BOTH are absent the window
|
|
358
|
+
modelled nothing at all, and naming the observed side there would blame
|
|
359
|
+
the meter for a window that asked nothing of it.
|
|
360
|
+
"""
|
|
361
|
+
if modelled is None:
|
|
362
|
+
return RESIDUAL_NO_MODELLED_WEEKS
|
|
363
|
+
if observed is None:
|
|
364
|
+
return RESIDUAL_OBSERVED_ABSENT
|
|
365
|
+
return None
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def observed_minus_modelled(*, observed, modelled):
|
|
369
|
+
"""The meter's reading minus the modelled local points, or None.
|
|
370
|
+
|
|
371
|
+
§2.2 measured this difference with BOTH signs, so no caller may assume it
|
|
372
|
+
has one, and no surface may state it as off-machine usage. It is the
|
|
373
|
+
difference itself, named as such.
|
|
374
|
+
"""
|
|
375
|
+
if observed is None or modelled is None:
|
|
376
|
+
return None
|
|
377
|
+
return observed - modelled
|
|
378
|
+
|
|
379
|
+
|
|
48
380
|
def _load_week_snapshots(
|
|
49
381
|
since: dt.datetime, until: dt.datetime, *,
|
|
50
382
|
account_key: "str | None" = None,
|
|
@@ -282,9 +614,19 @@ def _project_json_payload(
|
|
|
282
614
|
warnings: list[str],
|
|
283
615
|
include_breakdown: bool,
|
|
284
616
|
week_snapshots: dict[dt.datetime, float],
|
|
617
|
+
attribution_basis: str = "cost-share",
|
|
618
|
+
attribution_cause: "str | None" = None,
|
|
619
|
+
attribution_totals: "dict | None" = None,
|
|
285
620
|
) -> dict:
|
|
286
621
|
"""Build the project subcommand's --json payload per spec §4.
|
|
287
622
|
|
|
623
|
+
`schemaVersion` is 2 from #661 S2. `attributedUsedPercent` and
|
|
624
|
+
`costPerPercent` keep their spellings and change their MEANING, from a
|
|
625
|
+
share of the window's dollars to modelled weekly quota, and
|
|
626
|
+
`docs/cli-contract.md` classifies a changed value meaning as breaking.
|
|
627
|
+
The `attribution` block and the per-row `attributionBasis` are the v2
|
|
628
|
+
additions that say which measure a given payload actually carries.
|
|
629
|
+
|
|
288
630
|
Accepts rows already sorted by the caller (so ordering flags apply
|
|
289
631
|
uniformly to both terminal and JSON modes). Aggregates `totals.costUsd`
|
|
290
632
|
from `rows` and `totals.usedPercent` from `week_snapshots` (sum over
|
|
@@ -325,6 +667,10 @@ def _project_json_payload(
|
|
|
325
667
|
round(row["cost_per_pct"], 4)
|
|
326
668
|
if row["cost_per_pct"] is not None else None
|
|
327
669
|
),
|
|
670
|
+
# A project can span weeks that resolved differently, so the
|
|
671
|
+
# basis is a row field and not only a payload one. A row is
|
|
672
|
+
# `modelled` only when EVERY contributing week was.
|
|
673
|
+
"attributionBasis": row.get("attribution_basis", "cost-share"),
|
|
328
674
|
}
|
|
329
675
|
if include_breakdown:
|
|
330
676
|
p["models"] = [
|
|
@@ -354,10 +700,26 @@ def _project_json_payload(
|
|
|
354
700
|
),
|
|
355
701
|
"weeklyAttributionAvailable": len(weeks_missing_snapshot) == 0,
|
|
356
702
|
},
|
|
703
|
+
"attribution": {
|
|
704
|
+
"basis": attribution_basis,
|
|
705
|
+
"cause": attribution_cause,
|
|
706
|
+
# The four quantities spec §5.2 names, which the earlier draft
|
|
707
|
+
# conflated into "the rows do not add up". `residualCause` is
|
|
708
|
+
# set whenever the residual could not be computed against a
|
|
709
|
+
# whole, single-account, unfiltered population.
|
|
710
|
+
"totals": dict(attribution_totals or {
|
|
711
|
+
"modelledWeekPoints": None,
|
|
712
|
+
"visibleRowPoints": None,
|
|
713
|
+
"filteredOrUnmodelledPoints": None,
|
|
714
|
+
"observedMinusModelledPoints": None,
|
|
715
|
+
"residualCause": RESIDUAL_MISALIGNED,
|
|
716
|
+
}),
|
|
717
|
+
},
|
|
357
718
|
"projects": projects_json,
|
|
358
719
|
"warnings": warnings,
|
|
359
720
|
}
|
|
360
|
-
return _cctally().stamp_schema_version(
|
|
721
|
+
return _cctally().stamp_schema_version(
|
|
722
|
+
payload, version=PROJECT_JSON_SCHEMA_VERSION)
|
|
361
723
|
|
|
362
724
|
|
|
363
725
|
def _project_json_output(**kwargs) -> str:
|
|
@@ -639,6 +1001,28 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
639
1001
|
command_label="project",
|
|
640
1002
|
)
|
|
641
1003
|
|
|
1004
|
+
# #661 S2 spec §5. The run-level gate first: on a decorated multi-account
|
|
1005
|
+
# install a merged read has no valid calibration to apply, so modelled
|
|
1006
|
+
# quota is withheld and `--account` is required rather than silently
|
|
1007
|
+
# falling back to the cost share.
|
|
1008
|
+
_decorated = probe_provider_decoration(conn, "claude")
|
|
1009
|
+
run_basis, run_cause = resolve_attribution_account_gate(
|
|
1010
|
+
account_key=acct_key, decorated=_decorated,
|
|
1011
|
+
)
|
|
1012
|
+
attribution_regime = None
|
|
1013
|
+
if run_basis is None:
|
|
1014
|
+
try:
|
|
1015
|
+
qcg = c._load_sibling("_cctally_quota_calibration")
|
|
1016
|
+
attribution_regime = qcg.read_calibration_file(
|
|
1017
|
+
account_key=acct_key).regime
|
|
1018
|
+
except Exception: # noqa: BLE001
|
|
1019
|
+
attribution_regime = None
|
|
1020
|
+
# Per-week weighted units, and the records the whole-week support test
|
|
1021
|
+
# runs over. Collected on EVERY entry — the denominator is the account's
|
|
1022
|
+
# whole week, not the user's slice.
|
|
1023
|
+
week_units_total: dict = {}
|
|
1024
|
+
week_records: dict = {}
|
|
1025
|
+
|
|
642
1026
|
for entry in joined_entries_all:
|
|
643
1027
|
# Skip synthetic entries (Claude Code internal markers) to match
|
|
644
1028
|
# `_aggregate_cache_by_session` / `_aggregate_claude_sessions`.
|
|
@@ -669,6 +1053,16 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
669
1053
|
total_cost_by_week.get(week_start, 0.0) + entry_cost
|
|
670
1054
|
)
|
|
671
1055
|
|
|
1056
|
+
# #661 S2 §5.1: weight BEFORE bucket aggregation, which drops the
|
|
1057
|
+
# one-hour cache-write split `weighted_units` needs. Same whole-week
|
|
1058
|
+
# scope as the cost denominator above, for the same reason.
|
|
1059
|
+
entry_record, entry_units = (
|
|
1060
|
+
_entry_quota_record(entry) if run_basis is None else (None, None))
|
|
1061
|
+
if entry_units is not None:
|
|
1062
|
+
week_units_total[week_start] = (
|
|
1063
|
+
week_units_total.get(week_start, 0.0) + entry_units)
|
|
1064
|
+
week_records.setdefault(week_start, []).append(entry_record)
|
|
1065
|
+
|
|
672
1066
|
# User-slice gate: visible rows only include entries within
|
|
673
1067
|
# [since_dt, until_dt]. Entries outside the slice still
|
|
674
1068
|
# contributed to the denominator above.
|
|
@@ -709,10 +1103,13 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
709
1103
|
"input": 0, "output": 0,
|
|
710
1104
|
"cache_write": 0, "cache_read": 0,
|
|
711
1105
|
"cost_usd": 0.0,
|
|
1106
|
+
"units": 0.0,
|
|
712
1107
|
"models": {},
|
|
713
1108
|
}
|
|
714
1109
|
buckets[bkey] = b
|
|
715
1110
|
_accumulate_entry_into_bucket(b, entry, pre_computed_cost=entry_cost)
|
|
1111
|
+
if entry_units is not None:
|
|
1112
|
+
b["units"] += entry_units
|
|
716
1113
|
|
|
717
1114
|
# The remediation moved OUT of these two sentences and into the shared
|
|
718
1115
|
# affordance line below (#620 S1 D11), so the terminal states the problem
|
|
@@ -758,6 +1155,36 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
758
1155
|
ws for ws in weeks_in_range if ws not in week_snapshots
|
|
759
1156
|
}
|
|
760
1157
|
|
|
1158
|
+
# #661 S2 §5.1: one decision per subscription week, over that week's WHOLE
|
|
1159
|
+
# account population. A week with an unsupported or out-of-regime segment
|
|
1160
|
+
# falls back for the whole week; partial-week mixtures are not produced.
|
|
1161
|
+
week_attribution: dict = {}
|
|
1162
|
+
for ws in sorted(set(week_starts) | set(week_records)):
|
|
1163
|
+
week_attribution[ws] = resolve_week_attribution(
|
|
1164
|
+
attribution_regime, week_records.get(ws, ()),
|
|
1165
|
+
week_start=ws, week_end=ws + dt.timedelta(days=7))
|
|
1166
|
+
modelled_weeks = sorted(
|
|
1167
|
+
ws for ws, attr in week_attribution.items()
|
|
1168
|
+
if attr.basis == "modelled")
|
|
1169
|
+
fallback_weeks = sorted(
|
|
1170
|
+
ws for ws, attr in week_attribution.items()
|
|
1171
|
+
if attr.basis != "modelled")
|
|
1172
|
+
# The run's basis is the weakest any contributing week reached, because a
|
|
1173
|
+
# payload that claimed `modelled` while some of its rows came from a cost
|
|
1174
|
+
# share would be answering the same question two ways.
|
|
1175
|
+
if run_basis is not None:
|
|
1176
|
+
attribution_basis, attribution_cause = run_basis, run_cause
|
|
1177
|
+
elif modelled_weeks and not fallback_weeks:
|
|
1178
|
+
attribution_basis, attribution_cause = "modelled", None
|
|
1179
|
+
elif modelled_weeks:
|
|
1180
|
+
attribution_basis = "cost-share"
|
|
1181
|
+
attribution_cause = week_attribution[fallback_weeks[0]].cause
|
|
1182
|
+
else:
|
|
1183
|
+
attribution_basis = "cost-share"
|
|
1184
|
+
attribution_cause = (
|
|
1185
|
+
week_attribution[fallback_weeks[0]].cause if fallback_weeks
|
|
1186
|
+
else "calibration-absent")
|
|
1187
|
+
|
|
761
1188
|
# Collapse (project_key, week) buckets into one row per project, summing
|
|
762
1189
|
# tokens / cost / sessions / first_seen / last_seen / models across the
|
|
763
1190
|
# weeks the project appears in.
|
|
@@ -783,6 +1210,11 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
783
1210
|
# lacked a snapshot" (→ None) and "genuine zero attribution"
|
|
784
1211
|
# (→ 0.0 after a real contribution). Spec §3.
|
|
785
1212
|
"attributed_pct": None,
|
|
1213
|
+
"units": 0.0,
|
|
1214
|
+
# `modelled` only when EVERY contributing week was; the first
|
|
1215
|
+
# fallback week degrades the row.
|
|
1216
|
+
"attribution_basis": (
|
|
1217
|
+
"withheld" if run_basis == "withheld" else "modelled"),
|
|
786
1218
|
"models": {},
|
|
787
1219
|
}
|
|
788
1220
|
project_rows[key.bucket_path] = row
|
|
@@ -796,6 +1228,7 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
796
1228
|
row["cache_write"] += b["cache_write"]
|
|
797
1229
|
row["cache_read"] += b["cache_read"]
|
|
798
1230
|
row["cost_usd"] += b["cost_usd"]
|
|
1231
|
+
row["units"] += b["units"]
|
|
799
1232
|
|
|
800
1233
|
# Merge per-model sub-buckets.
|
|
801
1234
|
for model, mb in b["models"].items():
|
|
@@ -819,17 +1252,33 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
819
1252
|
rm["cache_write"] += mb["cache_write"]
|
|
820
1253
|
rm["cache_read"] += mb["cache_read"]
|
|
821
1254
|
|
|
822
|
-
# Attribution contribution
|
|
823
|
-
#
|
|
824
|
-
# the
|
|
825
|
-
#
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
1255
|
+
# Attribution contribution. #661 S2 §5: MODELLED quota where this
|
|
1256
|
+
# week's population supports it — the bucket's own weighted units
|
|
1257
|
+
# over the regime's units-per-point, which needs no meter reading and
|
|
1258
|
+
# no cost denominator at all — and the #86 cost share otherwise.
|
|
1259
|
+
#
|
|
1260
|
+
# The cost share stays gated on a snapshot and a nonzero week total,
|
|
1261
|
+
# because a zero denominator would make the ratio meaningless.
|
|
1262
|
+
# `attributed_pct` stays `None` until the first real contribution;
|
|
1263
|
+
# subsequent contributions accumulate.
|
|
1264
|
+
attr = week_attribution.get(wstart)
|
|
1265
|
+
if run_basis == "withheld":
|
|
1266
|
+
# §5.3: no modelled quota and no cost-share stand-in either.
|
|
1267
|
+
pass
|
|
1268
|
+
elif attr is not None and attr.basis == "modelled":
|
|
830
1269
|
row["attributed_pct"] = (
|
|
831
|
-
(row["attributed_pct"] or 0.0)
|
|
1270
|
+
(row["attributed_pct"] or 0.0)
|
|
1271
|
+
+ b["units"] / attr.units_per_point
|
|
832
1272
|
)
|
|
1273
|
+
else:
|
|
1274
|
+
row["attribution_basis"] = "cost-share"
|
|
1275
|
+
week_pct = week_snapshots.get(wstart)
|
|
1276
|
+
week_total = total_cost_by_week.get(wstart, 0.0)
|
|
1277
|
+
if week_pct is not None and week_total > 0:
|
|
1278
|
+
contribution = (b["cost_usd"] / week_total) * week_pct
|
|
1279
|
+
row["attributed_pct"] = (
|
|
1280
|
+
(row["attributed_pct"] or 0.0) + contribution
|
|
1281
|
+
)
|
|
833
1282
|
|
|
834
1283
|
# Compute $/1% per project: `cost_per_pct = cost_usd / attributed_pct`
|
|
835
1284
|
# when attribution is positive; None otherwise (e.g. every contributing
|
|
@@ -842,6 +1291,74 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
842
1291
|
else:
|
|
843
1292
|
row["cost_per_pct"] = None
|
|
844
1293
|
|
|
1294
|
+
# #661 S2 §5.2. Four distinct quantities, named rather than conflated
|
|
1295
|
+
# into "the rows do not add up":
|
|
1296
|
+
# 1. modelled-week points — every local entry in the window's
|
|
1297
|
+
# MODELLED weeks, weighted and
|
|
1298
|
+
# converted
|
|
1299
|
+
# 2. visible-row points — the subset the rendered rows carry
|
|
1300
|
+
# 3. filtered or unmodelled points — exactly 1 minus 2
|
|
1301
|
+
# 4. observed minus modelled — the meter's reading minus 1
|
|
1302
|
+
#
|
|
1303
|
+
# The measured scope of 1 is the window's modelled weeks and NOT every
|
|
1304
|
+
# local entry in the window, which is why the key is spelled
|
|
1305
|
+
# `modelledWeekPoints`. A fallback week's points appear in NEITHER 1 nor
|
|
1306
|
+
# 3 — they are not comparable to a modelled quantity — and a run holding
|
|
1307
|
+
# one withholds the residual outright, so the two quantities that are
|
|
1308
|
+
# published are the ones this run can actually measure.
|
|
1309
|
+
if run_basis == "withheld" or not modelled_weeks:
|
|
1310
|
+
modelled_week_points = None
|
|
1311
|
+
visible_points = None
|
|
1312
|
+
unmodelled_points = None
|
|
1313
|
+
else:
|
|
1314
|
+
modelled_week_points = stable_sum(
|
|
1315
|
+
week_units_total.get(ws, 0.0)
|
|
1316
|
+
/ week_attribution[ws].units_per_point
|
|
1317
|
+
for ws in modelled_weeks)
|
|
1318
|
+
visible_points = stable_sum(
|
|
1319
|
+
b["units"] / week_attribution[wstart].units_per_point
|
|
1320
|
+
for (_key, wstart), b in buckets.items()
|
|
1321
|
+
if week_attribution.get(wstart) is not None
|
|
1322
|
+
and week_attribution[wstart].basis == "modelled")
|
|
1323
|
+
unmodelled_points = modelled_week_points - visible_points
|
|
1324
|
+
observed_points = (
|
|
1325
|
+
stable_sum(week_snapshots[ws] for ws in modelled_weeks
|
|
1326
|
+
if ws in week_snapshots)
|
|
1327
|
+
if modelled_weeks
|
|
1328
|
+
and all(ws in week_snapshots for ws in modelled_weeks)
|
|
1329
|
+
else None)
|
|
1330
|
+
residual_cause = residual_withholding_cause(
|
|
1331
|
+
account_resolved=(run_basis is None),
|
|
1332
|
+
whole_weeks=range_covers_whole_weeks(
|
|
1333
|
+
since_dt, until_dt, parsed_bounds, now=now),
|
|
1334
|
+
filtered=bool(project_patterns or model_patterns),
|
|
1335
|
+
fallback_weeks=len(fallback_weeks),
|
|
1336
|
+
)
|
|
1337
|
+
attribution_totals = {
|
|
1338
|
+
"modelledWeekPoints": (
|
|
1339
|
+
None if modelled_week_points is None
|
|
1340
|
+
else round(modelled_week_points, 4)),
|
|
1341
|
+
"visibleRowPoints": (
|
|
1342
|
+
None if visible_points is None else round(visible_points, 4)),
|
|
1343
|
+
"filteredOrUnmodelledPoints": (
|
|
1344
|
+
None if unmodelled_points is None
|
|
1345
|
+
else round(unmodelled_points, 4)),
|
|
1346
|
+
"observedMinusModelledPoints": None,
|
|
1347
|
+
"residualCause": residual_cause,
|
|
1348
|
+
}
|
|
1349
|
+
if residual_cause is None:
|
|
1350
|
+
residual = observed_minus_modelled(
|
|
1351
|
+
observed=observed_points, modelled=modelled_week_points)
|
|
1352
|
+
attribution_totals["observedMinusModelledPoints"] = (
|
|
1353
|
+
None if residual is None else round(residual, 4))
|
|
1354
|
+
if residual is None:
|
|
1355
|
+
# The population aligns, so the absence is one of the two the
|
|
1356
|
+
# alignment test cannot see: no modelled week at all, or a
|
|
1357
|
+
# modelled week with no meter snapshot. Naming either of them
|
|
1358
|
+
# `population-misaligned` states a false reason.
|
|
1359
|
+
attribution_totals["residualCause"] = residual_absence_cause(
|
|
1360
|
+
observed=observed_points, modelled=modelled_week_points)
|
|
1361
|
+
|
|
845
1362
|
# Collect warnings to surface in the JSON payload (terminal path emits
|
|
846
1363
|
# them inline via eprint earlier, so this list stays JSON-specific).
|
|
847
1364
|
warnings: list[str] = []
|
|
@@ -911,6 +1428,9 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
911
1428
|
warnings=warnings,
|
|
912
1429
|
include_breakdown=args.breakdown,
|
|
913
1430
|
week_snapshots=week_snapshots,
|
|
1431
|
+
attribution_basis=attribution_basis,
|
|
1432
|
+
attribution_cause=attribution_cause,
|
|
1433
|
+
attribution_totals=attribution_totals,
|
|
914
1434
|
)
|
|
915
1435
|
payload.update(c.account_json_fields(acct_key)) # #341 R8 decoration
|
|
916
1436
|
sink = getattr(args, "_source_result_sink", None)
|
|
@@ -943,5 +1463,10 @@ def cmd_project(args: argparse.Namespace) -> int:
|
|
|
943
1463
|
weeks_in_range=len(weeks_in_range),
|
|
944
1464
|
color=c._resolve_color_enabled(args),
|
|
945
1465
|
compact=args.compact,
|
|
1466
|
+
# #661 S2 §5.2. The four quantities existed only in `project --json`,
|
|
1467
|
+
# so a terminal user got no reconciliation information at all.
|
|
1468
|
+
attribution_footer=render_attribution_footer(
|
|
1469
|
+
attribution_totals, basis=attribution_basis,
|
|
1470
|
+
cause=attribution_cause),
|
|
946
1471
|
))
|
|
947
1472
|
return 0
|
package/bin/_cctally_quota.py
CHANGED
|
@@ -321,6 +321,11 @@ PROJECTION_DYNAMIC_READ_SITES: "dict[str, int]" = {
|
|
|
321
321
|
"_cctally_five_hour.py": 1,
|
|
322
322
|
"_cctally_journal.py": 18,
|
|
323
323
|
"_cctally_pricing_check.py": 1,
|
|
324
|
+
# One, in `_read_stats_component`: a single `SELECT {column} FROM
|
|
325
|
+
# {table}` shared by `week_reset_events` and `weekly_credit_floors`,
|
|
326
|
+
# the two authoritative credit tables. Neither is a projection table,
|
|
327
|
+
# so no `PROJECTION_DYNAMIC_READ_ACTIONS` classification applies.
|
|
328
|
+
"_cctally_quota_model.py": 1,
|
|
324
329
|
"_cctally_quota.py": 1,
|
|
325
330
|
"_cctally_record.py": 1,
|
|
326
331
|
"_cctally_release.py": 4,
|
|
@@ -331,6 +336,14 @@ PROJECTION_DYNAMIC_READ_SITES: "dict[str, int]" = {
|
|
|
331
336
|
"_lib_doctor.py": 1,
|
|
332
337
|
"_lib_snapshot_cache.py": 1,
|
|
333
338
|
"_lib_subscription_weeks.py": 1,
|
|
339
|
+
# NOT a read. `bin/_lib_test_estate.py` opens no database and issues no SQL;
|
|
340
|
+
# it is the #648 estate checker, and its one match is the English message
|
|
341
|
+
# `f"moved from {was!r} to {SKIPPED!r}"`, which the case-insensitive
|
|
342
|
+
# `FROM\s+{` pattern cannot tell from a dynamic target. Recorded rather than
|
|
343
|
+
# reworded, because the count is what makes a new match visible: if a real
|
|
344
|
+
# dynamic read ever arrives in that module the count moves to 2 and this
|
|
345
|
+
# guard reports it.
|
|
346
|
+
"_lib_test_estate.py": 1,
|
|
334
347
|
}
|
|
335
348
|
|
|
336
349
|
#: The dynamic-target reads that provably reach a projection family, named by
|