coding-agent-cost 0.1.1__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {coding_agent_cost-0.1.1/coding_agent_cost.egg-info → coding_agent_cost-0.2.0}/PKG-INFO +48 -8
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/README.md +47 -7
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/__init__.py +1 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/aggregate.py +50 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/cli.py +55 -5
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/facts.py +6 -3
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/rates.json +37 -8
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/readers/__init__.py +11 -0
- coding_agent_cost-0.2.0/agent_cost/readers/claude.py +667 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0/coding_agent_cost.egg-info}/PKG-INFO +48 -8
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/coding_agent_cost.egg-info/SOURCES.txt +5 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/pyproject.toml +1 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_astra_pricing.py +2 -2
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_cli.py +38 -3
- coding_agent_cost-0.2.0/tests/test_e0a_fable_5_1_and_sonnet_5_correction.py +198 -0
- coding_agent_cost-0.2.0/tests/test_e0a_review3_hardening.py +276 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_measure_v1_contract.py +14 -6
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_rates.py +14 -5
- coding_agent_cost-0.2.0/tests/test_reader_dedup.py +901 -0
- coding_agent_cost-0.2.0/tests/test_version_consistency.py +33 -0
- coding_agent_cost-0.1.1/agent_cost/readers/claude.py +0 -177
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/LICENSE +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/config.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/rates.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/readers/codex.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/agent_cost/renderers.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/coding_agent_cost.egg-info/dependency_links.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/coding_agent_cost.egg-info/entry_points.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/coding_agent_cost.egg-info/top_level.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/setup.cfg +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_aggregate.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_facts.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_reader_claude.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.0}/tests/test_reader_codex.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: coding-agent-cost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
|
|
5
5
|
Author: shiki-yusuke
|
|
6
6
|
License: MIT
|
|
@@ -204,9 +204,33 @@ scraping `report`.
|
|
|
204
204
|
agent-cost only reads data that is already on disk. It never talks to the
|
|
205
205
|
network, never calls `gh`, and never resolves branches or PRs.
|
|
206
206
|
|
|
207
|
-
- **Claude Code**: every
|
|
208
|
-
|
|
209
|
-
|
|
207
|
+
- **Claude Code**: every logical assistant message is one billing event,
|
|
208
|
+
attributed to the exact model on that event (a session that switches
|
|
209
|
+
models mid-conversation is not folded into one "primary model"). Claude
|
|
210
|
+
Code's transcript writes one JSONL line per content block of the same
|
|
211
|
+
message, and those lines are deduplicated first -- but only when a row
|
|
212
|
+
carries a full `message.id` + `requestId` pair; a row missing either
|
|
213
|
+
half is emitted on its own (never merged) and flagged
|
|
214
|
+
`source_quality: "identity_missing"` rather than assumed
|
|
215
|
+
billing-accurate. `identity_missing` facts are still priced and included
|
|
216
|
+
in rows/totals; the flag is a warning, not an exclusion or an unpriced
|
|
217
|
+
status. A message's lines don't necessarily repeat an identical `usage`
|
|
218
|
+
block: `model` and the input-side fields (input tokens, cache read,
|
|
219
|
+
cache-write TTL breakdown) stay the same across a message's lines, but
|
|
220
|
+
`output_tokens` typically grows line by line as the response streams in,
|
|
221
|
+
and intermediate lines usually lack `usage.speed` entirely (reported as
|
|
222
|
+
mode `"unknown"`), with only the final line carrying a concrete mode.
|
|
223
|
+
Within a deduplicated group, `model` or an input-side field that
|
|
224
|
+
actually differs across the group's rows, two rows disagreeing on a
|
|
225
|
+
*concrete* mode (`"normal"` vs `"fast"`), or an `output_tokens` value
|
|
226
|
+
that decreases or is non-monotonic across them, is counted in
|
|
227
|
+
`data_quality.conflicting_duplicate_groups` as a real billing
|
|
228
|
+
disagreement -- `output_tokens` growing row by row, and mode
|
|
229
|
+
`"unknown"` mixed with a single concrete mode elsewhere in the group,
|
|
230
|
+
are both the ordinary streaming pattern just described and are not
|
|
231
|
+
flagged, since the reader's own cross-check of real transcripts found
|
|
232
|
+
exactly that pattern in every observed duplicated group. See
|
|
233
|
+
`CHANGELOG.md`'s 0.2.0 entry.
|
|
210
234
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
211
235
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
212
236
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -318,6 +342,8 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
318
342
|
```json
|
|
319
343
|
{
|
|
320
344
|
"protocol_version": "measure/v1",
|
|
345
|
+
"producer_version": "0.2.0",
|
|
346
|
+
"accounting_basis": "agent-cost-raw-total/v2",
|
|
321
347
|
"generated_at": "...",
|
|
322
348
|
"window": { "since": "...", "until": null },
|
|
323
349
|
"timezone": "UTC",
|
|
@@ -334,7 +360,10 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
334
360
|
"skipped_files": 0,
|
|
335
361
|
"negative_deltas": 0,
|
|
336
362
|
"unpriced_tokens": 0,
|
|
337
|
-
"
|
|
363
|
+
"duplicate_rows_skipped": 0,
|
|
364
|
+
"conflicting_duplicate_groups": 0,
|
|
365
|
+
"missing_dedup_identity_rows": 0,
|
|
366
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
338
367
|
}
|
|
339
368
|
}
|
|
340
369
|
```
|
|
@@ -343,10 +372,21 @@ Rows are grouped by agent/model/token-kind only -- `measure` never buckets
|
|
|
343
372
|
by month, since a query is already scoped to specific sessions. `total` is
|
|
344
373
|
the union of every requested `session_id` (not a global report), so it's
|
|
345
374
|
the number to attribute to whatever unit of work those sessions represent.
|
|
375
|
+
`producer_version` is the agent-cost package version that produced this
|
|
376
|
+
payload; `accounting_basis` identifies the token-accounting semantics
|
|
377
|
+
behind the numbers (separate from `protocol_version`, which only tracks
|
|
378
|
+
the JSON shape) -- a consumer that persists historical measurements should
|
|
379
|
+
key comparability on `accounting_basis`, not `producer_version` alone,
|
|
380
|
+
since a future release can bump the latter while keeping the former.
|
|
346
381
|
`data_quality.unpriced_tokens` and `.source_quality` are scoped to the
|
|
347
|
-
requested sessions;
|
|
348
|
-
|
|
349
|
-
|
|
382
|
+
requested sessions; so are the three dedup counters
|
|
383
|
+
(`duplicate_rows_skipped`, `conflicting_duplicate_groups`,
|
|
384
|
+
`missing_dedup_identity_rows`), which are Claude-only and computed over
|
|
385
|
+
the intersection of the requested session ids and the `--since`/`--until`
|
|
386
|
+
window, never over an unrequested session's rows.
|
|
387
|
+
`.malformed_events`/`.skipped_files`/`.negative_deltas` describe the
|
|
388
|
+
health of the underlying log read within `--since`/`--until` and are not
|
|
389
|
+
attributable to one session.
|
|
350
390
|
|
|
351
391
|
## Privacy
|
|
352
392
|
|
|
@@ -178,9 +178,33 @@ scraping `report`.
|
|
|
178
178
|
agent-cost only reads data that is already on disk. It never talks to the
|
|
179
179
|
network, never calls `gh`, and never resolves branches or PRs.
|
|
180
180
|
|
|
181
|
-
- **Claude Code**: every
|
|
182
|
-
|
|
183
|
-
|
|
181
|
+
- **Claude Code**: every logical assistant message is one billing event,
|
|
182
|
+
attributed to the exact model on that event (a session that switches
|
|
183
|
+
models mid-conversation is not folded into one "primary model"). Claude
|
|
184
|
+
Code's transcript writes one JSONL line per content block of the same
|
|
185
|
+
message, and those lines are deduplicated first -- but only when a row
|
|
186
|
+
carries a full `message.id` + `requestId` pair; a row missing either
|
|
187
|
+
half is emitted on its own (never merged) and flagged
|
|
188
|
+
`source_quality: "identity_missing"` rather than assumed
|
|
189
|
+
billing-accurate. `identity_missing` facts are still priced and included
|
|
190
|
+
in rows/totals; the flag is a warning, not an exclusion or an unpriced
|
|
191
|
+
status. A message's lines don't necessarily repeat an identical `usage`
|
|
192
|
+
block: `model` and the input-side fields (input tokens, cache read,
|
|
193
|
+
cache-write TTL breakdown) stay the same across a message's lines, but
|
|
194
|
+
`output_tokens` typically grows line by line as the response streams in,
|
|
195
|
+
and intermediate lines usually lack `usage.speed` entirely (reported as
|
|
196
|
+
mode `"unknown"`), with only the final line carrying a concrete mode.
|
|
197
|
+
Within a deduplicated group, `model` or an input-side field that
|
|
198
|
+
actually differs across the group's rows, two rows disagreeing on a
|
|
199
|
+
*concrete* mode (`"normal"` vs `"fast"`), or an `output_tokens` value
|
|
200
|
+
that decreases or is non-monotonic across them, is counted in
|
|
201
|
+
`data_quality.conflicting_duplicate_groups` as a real billing
|
|
202
|
+
disagreement -- `output_tokens` growing row by row, and mode
|
|
203
|
+
`"unknown"` mixed with a single concrete mode elsewhere in the group,
|
|
204
|
+
are both the ordinary streaming pattern just described and are not
|
|
205
|
+
flagged, since the reader's own cross-check of real transcripts found
|
|
206
|
+
exactly that pattern in every observed duplicated group. See
|
|
207
|
+
`CHANGELOG.md`'s 0.2.0 entry.
|
|
184
208
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
185
209
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
186
210
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -292,6 +316,8 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
292
316
|
```json
|
|
293
317
|
{
|
|
294
318
|
"protocol_version": "measure/v1",
|
|
319
|
+
"producer_version": "0.2.0",
|
|
320
|
+
"accounting_basis": "agent-cost-raw-total/v2",
|
|
295
321
|
"generated_at": "...",
|
|
296
322
|
"window": { "since": "...", "until": null },
|
|
297
323
|
"timezone": "UTC",
|
|
@@ -308,7 +334,10 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
308
334
|
"skipped_files": 0,
|
|
309
335
|
"negative_deltas": 0,
|
|
310
336
|
"unpriced_tokens": 0,
|
|
311
|
-
"
|
|
337
|
+
"duplicate_rows_skipped": 0,
|
|
338
|
+
"conflicting_duplicate_groups": 0,
|
|
339
|
+
"missing_dedup_identity_rows": 0,
|
|
340
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
312
341
|
}
|
|
313
342
|
}
|
|
314
343
|
```
|
|
@@ -317,10 +346,21 @@ Rows are grouped by agent/model/token-kind only -- `measure` never buckets
|
|
|
317
346
|
by month, since a query is already scoped to specific sessions. `total` is
|
|
318
347
|
the union of every requested `session_id` (not a global report), so it's
|
|
319
348
|
the number to attribute to whatever unit of work those sessions represent.
|
|
349
|
+
`producer_version` is the agent-cost package version that produced this
|
|
350
|
+
payload; `accounting_basis` identifies the token-accounting semantics
|
|
351
|
+
behind the numbers (separate from `protocol_version`, which only tracks
|
|
352
|
+
the JSON shape) -- a consumer that persists historical measurements should
|
|
353
|
+
key comparability on `accounting_basis`, not `producer_version` alone,
|
|
354
|
+
since a future release can bump the latter while keeping the former.
|
|
320
355
|
`data_quality.unpriced_tokens` and `.source_quality` are scoped to the
|
|
321
|
-
requested sessions;
|
|
322
|
-
|
|
323
|
-
|
|
356
|
+
requested sessions; so are the three dedup counters
|
|
357
|
+
(`duplicate_rows_skipped`, `conflicting_duplicate_groups`,
|
|
358
|
+
`missing_dedup_identity_rows`), which are Claude-only and computed over
|
|
359
|
+
the intersection of the requested session ids and the `--since`/`--until`
|
|
360
|
+
window, never over an unrequested session's rows.
|
|
361
|
+
`.malformed_events`/`.skipped_files`/`.negative_deltas` describe the
|
|
362
|
+
health of the underlying log read within `--since`/`--until` and are not
|
|
363
|
+
attributable to one session.
|
|
324
364
|
|
|
325
365
|
## Privacy
|
|
326
366
|
|
|
@@ -84,12 +84,59 @@ def filter_facts(
|
|
|
84
84
|
yield f
|
|
85
85
|
|
|
86
86
|
|
|
87
|
+
def scope_dedup_units(
|
|
88
|
+
units: Iterable,
|
|
89
|
+
*,
|
|
90
|
+
since_utc: Optional[datetime] = None,
|
|
91
|
+
until_utc: Optional[datetime] = None,
|
|
92
|
+
session_ids: Optional[set] = None,
|
|
93
|
+
) -> Tuple[int, int, int]:
|
|
94
|
+
"""Re-scope Claude's per-group dedup diagnostics to a window and/or a
|
|
95
|
+
set of session ids, after the fact.
|
|
96
|
+
|
|
97
|
+
``units`` is duck-typed (each must have ``.occurred_at_utc``,
|
|
98
|
+
``.session_id``, ``.duplicate_rows_skipped``, ``.conflicting`` and
|
|
99
|
+
``.missing_identity``) rather than imported as
|
|
100
|
+
``readers.claude.ClaudeDedupUnit``, to avoid this module depending on
|
|
101
|
+
a specific reader. Returns
|
|
102
|
+
``(duplicate_rows_skipped, conflicting_duplicate_groups,
|
|
103
|
+
missing_dedup_identity_rows)`` -- the same three counters
|
|
104
|
+
``ClaudeParseResult``/``ReadResult`` expose as unscoped file-level
|
|
105
|
+
totals, but summed only over units that fall inside ``[since_utc,
|
|
106
|
+
until_utc)`` and, if given, whose ``session_id`` is in
|
|
107
|
+
``session_ids``. This mirrors ``filter_facts``'s half-open window
|
|
108
|
+
semantics so a caller's ``report``/``measure`` window matches exactly
|
|
109
|
+
what ``build_rows`` priced.
|
|
110
|
+
"""
|
|
111
|
+
duplicate_rows_skipped = 0
|
|
112
|
+
conflicting_duplicate_groups = 0
|
|
113
|
+
missing_dedup_identity_rows = 0
|
|
114
|
+
for unit in units:
|
|
115
|
+
if since_utc is not None and unit.occurred_at_utc < since_utc:
|
|
116
|
+
continue
|
|
117
|
+
if until_utc is not None and unit.occurred_at_utc >= until_utc:
|
|
118
|
+
continue
|
|
119
|
+
if session_ids is not None and unit.session_id not in session_ids:
|
|
120
|
+
continue
|
|
121
|
+
duplicate_rows_skipped += unit.duplicate_rows_skipped
|
|
122
|
+
if unit.conflicting:
|
|
123
|
+
conflicting_duplicate_groups += 1
|
|
124
|
+
if unit.missing_identity:
|
|
125
|
+
missing_dedup_identity_rows += 1
|
|
126
|
+
return duplicate_rows_skipped, conflicting_duplicate_groups, missing_dedup_identity_rows
|
|
127
|
+
|
|
128
|
+
|
|
87
129
|
@dataclass
|
|
88
130
|
class DataQuality:
|
|
89
131
|
malformed_events: int = 0
|
|
90
132
|
skipped_files: int = 0
|
|
91
133
|
negative_deltas: int = 0
|
|
92
134
|
unpriced_tokens: int = 0
|
|
135
|
+
# Claude-only dedup diagnostics (see readers/claude.py's
|
|
136
|
+
# parse_session_detailed docstring); always 0 for Codex facts.
|
|
137
|
+
duplicate_rows_skipped: int = 0
|
|
138
|
+
conflicting_duplicate_groups: int = 0
|
|
139
|
+
missing_dedup_identity_rows: int = 0
|
|
93
140
|
|
|
94
141
|
def to_dict(self) -> dict:
|
|
95
142
|
return {
|
|
@@ -97,6 +144,9 @@ class DataQuality:
|
|
|
97
144
|
"skipped_files": self.skipped_files,
|
|
98
145
|
"negative_deltas": self.negative_deltas,
|
|
99
146
|
"unpriced_tokens": self.unpriced_tokens,
|
|
147
|
+
"duplicate_rows_skipped": self.duplicate_rows_skipped,
|
|
148
|
+
"conflicting_duplicate_groups": self.conflicting_duplicate_groups,
|
|
149
|
+
"missing_dedup_identity_rows": self.missing_dedup_identity_rows,
|
|
100
150
|
}
|
|
101
151
|
|
|
102
152
|
|
|
@@ -11,7 +11,7 @@ from typing import Optional
|
|
|
11
11
|
from zoneinfo import ZoneInfo
|
|
12
12
|
|
|
13
13
|
from . import __version__
|
|
14
|
-
from .aggregate import DataQuality, build_rows, filter_facts, rows_totals
|
|
14
|
+
from .aggregate import DataQuality, build_rows, filter_facts, rows_totals, scope_dedup_units
|
|
15
15
|
from .config import load_config
|
|
16
16
|
from .facts import SOURCE_QUALITY_VALUES
|
|
17
17
|
from .rates import RatesValidationError, load_rates
|
|
@@ -25,6 +25,15 @@ from .renderers import render_csv, render_json, render_table
|
|
|
25
25
|
#: should check this before trusting the shape of the payload.
|
|
26
26
|
MEASURE_PROTOCOL_VERSION = "measure/v1"
|
|
27
27
|
|
|
28
|
+
#: Identifies the token-accounting semantics behind measure's numbers, not
|
|
29
|
+
#: the JSON shape (that's MEASURE_PROTOCOL_VERSION). "v2" marks the fix for
|
|
30
|
+
#: the Claude reader's row-per-content-block over-count (agent-cost 0.2.0);
|
|
31
|
+
#: "v1" numbers (agent-cost 0.1.x) are not comparable to "v2" numbers for
|
|
32
|
+
#: Claude facts. A consumer that persists historical measurements (e.g.
|
|
33
|
+
#: lane's ledger) should key on this, not on producer_version alone, since
|
|
34
|
+
#: a future producer_version could still share the same accounting_basis.
|
|
35
|
+
ACCOUNTING_BASIS = "agent-cost-raw-total/v2"
|
|
36
|
+
|
|
28
37
|
|
|
29
38
|
def _parse_window_bound(value: Optional[str], tz: ZoneInfo) -> Optional[datetime]:
|
|
30
39
|
if not value:
|
|
@@ -45,14 +54,29 @@ def _parse_window_bound(value: Optional[str], tz: ZoneInfo) -> Optional[datetime
|
|
|
45
54
|
|
|
46
55
|
|
|
47
56
|
def _collect_facts(config, *, agents: set, exclude_archived: bool):
|
|
57
|
+
"""Returns ``(facts, dq, claude_dedup_units)``.
|
|
58
|
+
|
|
59
|
+
``dq``'s ``malformed_events``/``skipped_files``/``negative_deltas`` are
|
|
60
|
+
unscoped file-level totals (unchanged, existing behavior).
|
|
61
|
+
``dq.duplicate_rows_skipped``/``conflicting_duplicate_groups``/
|
|
62
|
+
``missing_dedup_identity_rows`` are deliberately left at their
|
|
63
|
+
``DataQuality`` defaults here (0) rather than filled in from an
|
|
64
|
+
unscoped total: those three are Claude-only and must always be
|
|
65
|
+
re-scoped to the caller's actual window (``report``) or requested
|
|
66
|
+
session ids + window (``measure``) via ``claude_dedup_units`` and
|
|
67
|
+
``aggregate.scope_dedup_units`` -- an unscoped total would silently
|
|
68
|
+
misrepresent a windowed/session-scoped report or measure output.
|
|
69
|
+
"""
|
|
48
70
|
facts: list = []
|
|
49
71
|
dq = DataQuality()
|
|
72
|
+
claude_dedup_units: list = []
|
|
50
73
|
|
|
51
74
|
if "claude" in agents:
|
|
52
75
|
result = claude_reader.read_claude_facts(config.claude_projects_dir)
|
|
53
76
|
facts.extend(result.facts)
|
|
54
77
|
dq.malformed_events += result.malformed_events
|
|
55
78
|
dq.skipped_files += result.skipped_files
|
|
79
|
+
claude_dedup_units.extend(result.claude_dedup_units)
|
|
56
80
|
|
|
57
81
|
if "codex" in agents and config.codex_db_path.exists():
|
|
58
82
|
result = codex_reader.read_codex_facts(
|
|
@@ -65,7 +89,7 @@ def _collect_facts(config, *, agents: set, exclude_archived: bool):
|
|
|
65
89
|
dq.skipped_files += result.skipped_files
|
|
66
90
|
dq.negative_deltas += result.negative_deltas
|
|
67
91
|
|
|
68
|
-
return facts, dq
|
|
92
|
+
return facts, dq, claude_dedup_units
|
|
69
93
|
|
|
70
94
|
|
|
71
95
|
def cmd_report(args) -> int:
|
|
@@ -87,10 +111,20 @@ def cmd_report(args) -> int:
|
|
|
87
111
|
print(f"[error] rates catalog invalid: {exc}", file=sys.stderr)
|
|
88
112
|
return 2
|
|
89
113
|
|
|
90
|
-
facts, dq = _collect_facts(
|
|
114
|
+
facts, dq, claude_dedup_units = _collect_facts(
|
|
115
|
+
config, agents=agents, exclude_archived=args.exclude_archived
|
|
116
|
+
)
|
|
91
117
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
92
118
|
rows, agg_dq = build_rows(facts, catalog, group_by=group_by, timezone_name=args.timezone)
|
|
93
119
|
dq.unpriced_tokens = agg_dq.unpriced_tokens
|
|
120
|
+
# The three dedup counters are Claude-only and always re-scoped to this
|
|
121
|
+
# report's actual [since, until) window -- see _collect_facts's
|
|
122
|
+
# docstring for why they aren't filled in from an unscoped total.
|
|
123
|
+
(
|
|
124
|
+
dq.duplicate_rows_skipped,
|
|
125
|
+
dq.conflicting_duplicate_groups,
|
|
126
|
+
dq.missing_dedup_identity_rows,
|
|
127
|
+
) = scope_dedup_units(claude_dedup_units, since_utc=since, until_utc=until)
|
|
94
128
|
|
|
95
129
|
payload = {
|
|
96
130
|
"schema_version": "1",
|
|
@@ -118,7 +152,7 @@ def cmd_export(args) -> int:
|
|
|
118
152
|
until = _parse_window_bound(args.until, tz)
|
|
119
153
|
agents = set(args.agent.split(",")) if args.agent else {"claude", "codex"}
|
|
120
154
|
|
|
121
|
-
facts, _dq = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
155
|
+
facts, _dq, _claude_dedup_units = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
122
156
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
123
157
|
|
|
124
158
|
out = open(args.out, "w") if args.out else sys.stdout
|
|
@@ -180,7 +214,7 @@ def cmd_measure(args) -> int:
|
|
|
180
214
|
return 2
|
|
181
215
|
|
|
182
216
|
config = load_config()
|
|
183
|
-
facts, dq = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
217
|
+
facts, dq, claude_dedup_units = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
184
218
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
185
219
|
|
|
186
220
|
# measure is a per-session query, not a time-bucketed report: group by
|
|
@@ -190,6 +224,17 @@ def cmd_measure(args) -> int:
|
|
|
190
224
|
requested = set(session_ids)
|
|
191
225
|
combined_facts = [f for f in facts if f.session_id in requested]
|
|
192
226
|
|
|
227
|
+
# The three dedup counters are Claude-only and always re-scoped to
|
|
228
|
+
# exactly this measure call's requested session ids AND window -- an
|
|
229
|
+
# unrequested session's conflict/identity-missing rows must never
|
|
230
|
+
# affect a requested session's counters (see _collect_facts's
|
|
231
|
+
# docstring and cmd_report's equivalent window-only scoping above).
|
|
232
|
+
(
|
|
233
|
+
dedup_rows_skipped,
|
|
234
|
+
dedup_conflicting_groups,
|
|
235
|
+
dedup_missing_identity_rows,
|
|
236
|
+
) = scope_dedup_units(claude_dedup_units, since_utc=since, until_utc=until, session_ids=requested)
|
|
237
|
+
|
|
193
238
|
quality_counts = {v: 0 for v in SOURCE_QUALITY_VALUES}
|
|
194
239
|
for f in combined_facts:
|
|
195
240
|
quality_counts[f.source_quality] = quality_counts.get(f.source_quality, 0) + 1
|
|
@@ -208,6 +253,8 @@ def cmd_measure(args) -> int:
|
|
|
208
253
|
|
|
209
254
|
payload = {
|
|
210
255
|
"protocol_version": MEASURE_PROTOCOL_VERSION,
|
|
256
|
+
"producer_version": __version__,
|
|
257
|
+
"accounting_basis": ACCOUNTING_BASIS,
|
|
211
258
|
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
212
259
|
"window": {
|
|
213
260
|
"since": since.isoformat() if since else None,
|
|
@@ -227,6 +274,9 @@ def cmd_measure(args) -> int:
|
|
|
227
274
|
"skipped_files": dq.skipped_files,
|
|
228
275
|
"negative_deltas": dq.negative_deltas,
|
|
229
276
|
"unpriced_tokens": total_dq.unpriced_tokens,
|
|
277
|
+
"duplicate_rows_skipped": dedup_rows_skipped,
|
|
278
|
+
"conflicting_duplicate_groups": dedup_conflicting_groups,
|
|
279
|
+
"missing_dedup_identity_rows": dedup_missing_identity_rows,
|
|
230
280
|
"source_quality": quality_counts,
|
|
231
281
|
},
|
|
232
282
|
}
|
|
@@ -30,9 +30,12 @@ MODES = ("fast", "normal", "unknown")
|
|
|
30
30
|
# covers the ordinary case; readers add a more specific value only when a
|
|
31
31
|
# fact's derivation has a real, nameable caveat worth surfacing downstream
|
|
32
32
|
# (e.g. Codex's first delta in a rollout is measured against an assumed
|
|
33
|
-
# zero baseline
|
|
34
|
-
# so
|
|
35
|
-
|
|
33
|
+
# zero baseline; Claude's reader can't dedup a row that lacks a full
|
|
34
|
+
# (message.id, requestId) pair, so it emits that row individually and
|
|
35
|
+
# flags it "identity_missing" rather than silently treating it as "ok").
|
|
36
|
+
# This is never left unset -- every Fact defaults to "ok" so export never
|
|
37
|
+
# emits a null source_quality.
|
|
38
|
+
SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing")
|
|
36
39
|
|
|
37
40
|
_BRACKET_SUFFIX = re.compile(r"\[[^\]]*\]$")
|
|
38
41
|
_DATE_SUFFIX = re.compile(r"@\d{6,8}$")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schema_version": "1",
|
|
3
|
-
"catalog_version": "2026-09-
|
|
3
|
+
"catalog_version": "2026-09-09",
|
|
4
4
|
"currency": "USD",
|
|
5
5
|
"unit": "per_mtok",
|
|
6
6
|
"usd_per_credit": "0.04",
|
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
"All Claude rates are standard (non-batch, non-priority) API prices. Batch API is roughly 50% off, but agent-cost's logs have no signal for whether a given request went through Batch, so batch usage is priced at standard rates and will overstate cost for anyone using Batch.",
|
|
9
9
|
"claude-opus-5's launch date could not be confirmed from an authoritative source as of this catalog_version; its rate period's effective_from (2026-07-01) is a placeholder -- update it once confirmed.",
|
|
10
10
|
"gpt-5.6 (sol/terra/luna) credits could not be confirmed from the primary source (help.openai.com's Codex rate card returned HTTP 403 to automated fetches). The values here were cross-checked across several independent secondary sources that agreed with each other and are internally consistent (Sol's $5/$30 per-MTok API price x 25 = 125/750 credits, matching usd_per_credit=0.04); see the gpt-5.6-* sources entries below. Treat effective_from (2026-07-01) as a placeholder and re-verify against the primary rate card when it becomes reachable.",
|
|
11
|
-
"gpt-6-astra: Codex standard token-based rates only; not API pricing, legacy message metering, or actual contractual charges. Confirmed 2026-09-06 JST (2026-09-05T17:18:23Z). effective_from is this observation cutoff, NOT an official launch or price-start time; earlier events remain unpriced. No Codex long-context surcharge or cache-write charge is applied. Fast multiplier 2.5 applies to explicit priority request settings with matching owned turn/model context; explicit default uses Standard. Missing or ambiguous Astra modes remain unpriced. These settings do not confirm the backend processing tier. See docs/astra-pricing.md for compatibility limits and the separate GPT-5.6 discrepancy report."
|
|
11
|
+
"gpt-6-astra: Codex standard token-based rates only; not API pricing, legacy message metering, or actual contractual charges. Confirmed 2026-09-06 JST (2026-09-05T17:18:23Z). effective_from is this observation cutoff, NOT an official launch or price-start time; earlier events remain unpriced. No Codex long-context surcharge or cache-write charge is applied. Fast multiplier 2.5 applies to explicit priority request settings with matching owned turn/model context; explicit default uses Standard. Missing or ambiguous Astra modes remain unpriced. These settings do not confirm the backend processing tier. See docs/astra-pricing.md for compatibility limits and the separate GPT-5.6 discrepancy report.",
|
|
12
|
+
"claude-fable-5-1 added 2026-09-09: rate_id claude-fable-5-1-local-first-observed-2026-09-02. effective_from (2026-09-02T01:17:55Z) is the timestamp the raw model ID was first observed across all 54 local transcript files (none found in August), not a confirmed official launch date -- if an authoritative launch date is confirmed later, add it as a separate rate_id with effective_until on this one rather than editing this period in place. Values (cache_read $0.25/MTok = 0.025x base input) differ from claude-fable-5's cache_read ($1.00/MTok = 0.1x); this is a genuine per-model price difference confirmed against the primary source (see sources below), not an inconsistency to reconcile.",
|
|
13
|
+
"claude-sonnet-5 corrected 2026-09-09: a prior period (rate_id claude-sonnet-5-standard, effective_from 2026-09-01, values 3.0/0.30/3.75/6.0/15.0) priced a previously-announced rate increase that the pricing page's claude-sonnet-5-introductory-pricing note states did not occur (\"The previously scheduled increase to $3/$15 ... on September 1, 2026 will not occur\"). That period has been replaced by claude-sonnet-5-standard-2026-09-01, which carries the same $2/$0.20/$2.50/$4/$10 values as the preceding claude-sonnet-5-launch-promo period, since the launch price is confirmed to be the ongoing price with no change at the 2026-09-01 boundary. Any measure/v1 or report output computed against the pre-correction catalog for claude-sonnet-5 sessions occurring on or after 2026-09-01 overstated cost by 1.5x for that model. To detect this class of error going forward, cross-check each priced model's rate periods against its primary pricing page's Note text on a recurring (e.g. monthly) basis for mentions of a rate-change announcement being withdrawn or not taking effect -- a period can be internally valid (schema-wise) and still price a rate that was announced but never actually charged."
|
|
12
14
|
],
|
|
13
15
|
"sources": [
|
|
14
16
|
{
|
|
@@ -48,6 +50,16 @@
|
|
|
48
50
|
"url": "https://developers.openai.com/api/docs/models/gpt-6-astra",
|
|
49
51
|
"retrieved_at": "2026-09-05T17:18:23Z",
|
|
50
52
|
"note": "Boundary check only: API cache writes, long context and API Fast rates are NOT applied to this Codex entry."
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"url": "https://platform.claude.com/docs/en/about-claude/pricing",
|
|
56
|
+
"retrieved_at": "2026-09-09",
|
|
57
|
+
"note": "used for claude-fable-5-1 (cache_read 0.025x footnote) and the claude-sonnet-5-introductory-pricing correction; local snapshot 2026-09-09 sha256=d79ad28567196bd55dd50e2fd89341b9da9774a45c1d02fcf387078507fd15e0"
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
"url": "https://platform.claude.com/docs/en/build-with-claude/prompt-caching",
|
|
61
|
+
"retrieved_at": "2026-09-09",
|
|
62
|
+
"note": "cache write TTL multipliers (5m = 1.25x, 1h = 2x), cross-checked against the pricing page cache columns"
|
|
51
63
|
}
|
|
52
64
|
],
|
|
53
65
|
"models": [
|
|
@@ -68,6 +80,23 @@
|
|
|
68
80
|
}
|
|
69
81
|
]
|
|
70
82
|
},
|
|
83
|
+
{
|
|
84
|
+
"model_key": "claude-fable-5-1",
|
|
85
|
+
"aliases": [],
|
|
86
|
+
"fast_multiplier": "1.0",
|
|
87
|
+
"rates": [
|
|
88
|
+
{
|
|
89
|
+
"rate_id": "claude-fable-5-1-local-first-observed-2026-09-02",
|
|
90
|
+
"effective_from": "2026-09-02T01:17:55+00:00",
|
|
91
|
+
"effective_until": null,
|
|
92
|
+
"input_nocache": "10.0",
|
|
93
|
+
"cache_read": "0.25",
|
|
94
|
+
"cache_write_5m": "12.50",
|
|
95
|
+
"cache_write_1h": "20.0",
|
|
96
|
+
"output": "50.0"
|
|
97
|
+
}
|
|
98
|
+
]
|
|
99
|
+
},
|
|
71
100
|
{
|
|
72
101
|
"model_key": "claude-mythos-5",
|
|
73
102
|
"aliases": [],
|
|
@@ -186,14 +215,14 @@
|
|
|
186
215
|
"output": "10.0"
|
|
187
216
|
},
|
|
188
217
|
{
|
|
189
|
-
"rate_id": "claude-sonnet-5-standard",
|
|
218
|
+
"rate_id": "claude-sonnet-5-standard-2026-09-01",
|
|
190
219
|
"effective_from": "2026-09-01T00:00:00+00:00",
|
|
191
220
|
"effective_until": null,
|
|
192
|
-
"input_nocache": "
|
|
193
|
-
"cache_read": "0.
|
|
194
|
-
"cache_write_5m": "
|
|
195
|
-
"cache_write_1h": "
|
|
196
|
-
"output": "
|
|
221
|
+
"input_nocache": "2.0",
|
|
222
|
+
"cache_read": "0.20",
|
|
223
|
+
"cache_write_5m": "2.50",
|
|
224
|
+
"cache_write_1h": "4.0",
|
|
225
|
+
"output": "10.0"
|
|
197
226
|
}
|
|
198
227
|
]
|
|
199
228
|
},
|
|
@@ -24,3 +24,14 @@ class ReadResult:
|
|
|
24
24
|
# `data_quality` (which is reader-agnostic); surfaced for `doctor` /
|
|
25
25
|
# tests.
|
|
26
26
|
tokens_used_diffs: int = 0
|
|
27
|
+
# Claude-only dedup diagnostics (see readers/claude.py's
|
|
28
|
+
# parse_session_detailed docstring). Always 0 / empty for the Codex
|
|
29
|
+
# reader. The three ints are file-level totals, NOT scoped to any
|
|
30
|
+
# window or session; `claude_dedup_units` carries the same
|
|
31
|
+
# information per group/row (as `ClaudeDedupUnit`, duck-typed here to
|
|
32
|
+
# avoid a circular import with readers.claude) so a caller can
|
|
33
|
+
# re-scope it -- see `agent_cost.aggregate.scope_dedup_units`.
|
|
34
|
+
duplicate_rows_skipped: int = 0
|
|
35
|
+
conflicting_duplicate_groups: int = 0
|
|
36
|
+
missing_dedup_identity_rows: int = 0
|
|
37
|
+
claude_dedup_units: List = field(default_factory=list)
|