coding-agent-cost 0.1.1__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {coding_agent_cost-0.1.1/coding_agent_cost.egg-info → coding_agent_cost-0.2.1}/PKG-INFO +49 -10
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/README.md +48 -9
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/__init__.py +1 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/aggregate.py +50 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/cli.py +55 -5
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/facts.py +6 -3
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/rates.json +78 -11
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/readers/__init__.py +11 -0
- coding_agent_cost-0.2.1/agent_cost/readers/claude.py +667 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1/coding_agent_cost.egg-info}/PKG-INFO +49 -10
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/coding_agent_cost.egg-info/SOURCES.txt +6 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/pyproject.toml +1 -1
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_astra_pricing.py +2 -2
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_cli.py +38 -3
- coding_agent_cost-0.2.1/tests/test_e0a_fable_5_1_and_sonnet_5_correction.py +198 -0
- coding_agent_cost-0.2.1/tests/test_e0a_review3_hardening.py +276 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_measure_v1_contract.py +14 -6
- coding_agent_cost-0.2.1/tests/test_opus_5_5_rates.py +74 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_rates.py +15 -6
- coding_agent_cost-0.2.1/tests/test_reader_dedup.py +901 -0
- coding_agent_cost-0.2.1/tests/test_version_consistency.py +33 -0
- coding_agent_cost-0.1.1/agent_cost/readers/claude.py +0 -177
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/LICENSE +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/config.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/rates.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/readers/codex.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/agent_cost/renderers.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/coding_agent_cost.egg-info/dependency_links.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/coding_agent_cost.egg-info/entry_points.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/coding_agent_cost.egg-info/top_level.txt +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/setup.cfg +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_aggregate.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_facts.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_reader_claude.py +0 -0
- {coding_agent_cost-0.1.1 → coding_agent_cost-0.2.1}/tests/test_reader_codex.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: coding-agent-cost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
|
|
5
5
|
Author: shiki-yusuke
|
|
6
6
|
License: MIT
|
|
@@ -204,9 +204,33 @@ scraping `report`.
|
|
|
204
204
|
agent-cost only reads data that is already on disk. It never talks to the
|
|
205
205
|
network, never calls `gh`, and never resolves branches or PRs.
|
|
206
206
|
|
|
207
|
-
- **Claude Code**: every
|
|
208
|
-
|
|
209
|
-
|
|
207
|
+
- **Claude Code**: every logical assistant message is one billing event,
|
|
208
|
+
attributed to the exact model on that event (a session that switches
|
|
209
|
+
models mid-conversation is not folded into one "primary model"). Claude
|
|
210
|
+
Code's transcript writes one JSONL line per content block of the same
|
|
211
|
+
message, and those lines are deduplicated first -- but only when a row
|
|
212
|
+
carries a full `message.id` + `requestId` pair; a row missing either
|
|
213
|
+
half is emitted on its own (never merged) and flagged
|
|
214
|
+
`source_quality: "identity_missing"` rather than assumed
|
|
215
|
+
billing-accurate. `identity_missing` facts are still priced and included
|
|
216
|
+
in rows/totals; the flag is a warning, not an exclusion or an unpriced
|
|
217
|
+
status. A message's lines don't necessarily repeat an identical `usage`
|
|
218
|
+
block: `model` and the input-side fields (input tokens, cache read,
|
|
219
|
+
cache-write TTL breakdown) stay the same across a message's lines, but
|
|
220
|
+
`output_tokens` typically grows line by line as the response streams in,
|
|
221
|
+
and intermediate lines usually lack `usage.speed` entirely (reported as
|
|
222
|
+
mode `"unknown"`), with only the final line carrying a concrete mode.
|
|
223
|
+
Within a deduplicated group, `model` or an input-side field that
|
|
224
|
+
actually differs across the group's rows, two rows disagreeing on a
|
|
225
|
+
*concrete* mode (`"normal"` vs `"fast"`), or an `output_tokens` value
|
|
226
|
+
that decreases or is non-monotonic across them, is counted in
|
|
227
|
+
`data_quality.conflicting_duplicate_groups` as a real billing
|
|
228
|
+
disagreement -- `output_tokens` growing row by row, and mode
|
|
229
|
+
`"unknown"` mixed with a single concrete mode elsewhere in the group,
|
|
230
|
+
are both the ordinary streaming pattern just described and are not
|
|
231
|
+
flagged, since the reader's own cross-check of real transcripts found
|
|
232
|
+
exactly that pattern in every observed duplicated group. See
|
|
233
|
+
`CHANGELOG.md`'s 0.2.0 entry.
|
|
210
234
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
211
235
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
212
236
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -242,8 +266,7 @@ network, never calls `gh`, and never resolves branches or PRs.
|
|
|
242
266
|
in the report's `data_quality` block instead of being silently dropped or
|
|
243
267
|
clamped to zero.
|
|
244
268
|
- **Known catalog gaps**, tracked in `agent_cost/rates.json`'s `notes`:
|
|
245
|
-
`
|
|
246
|
-
source, so its rate period's `effective_from` is a placeholder. `gpt-5.6`
|
|
269
|
+
`gpt-5.6`
|
|
247
270
|
(Sol/Terra/Luna) credits could not be confirmed from the primary source
|
|
248
271
|
(`help.openai.com`'s Codex rate card returns HTTP 403 to automated
|
|
249
272
|
fetches); the values in the catalog come from several independent
|
|
@@ -318,6 +341,8 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
318
341
|
```json
|
|
319
342
|
{
|
|
320
343
|
"protocol_version": "measure/v1",
|
|
344
|
+
"producer_version": "0.2.1",
|
|
345
|
+
"accounting_basis": "agent-cost-raw-total/v2",
|
|
321
346
|
"generated_at": "...",
|
|
322
347
|
"window": { "since": "...", "until": null },
|
|
323
348
|
"timezone": "UTC",
|
|
@@ -334,7 +359,10 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
334
359
|
"skipped_files": 0,
|
|
335
360
|
"negative_deltas": 0,
|
|
336
361
|
"unpriced_tokens": 0,
|
|
337
|
-
"
|
|
362
|
+
"duplicate_rows_skipped": 0,
|
|
363
|
+
"conflicting_duplicate_groups": 0,
|
|
364
|
+
"missing_dedup_identity_rows": 0,
|
|
365
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
338
366
|
}
|
|
339
367
|
}
|
|
340
368
|
```
|
|
@@ -343,10 +371,21 @@ Rows are grouped by agent/model/token-kind only -- `measure` never buckets
|
|
|
343
371
|
by month, since a query is already scoped to specific sessions. `total` is
|
|
344
372
|
the union of every requested `session_id` (not a global report), so it's
|
|
345
373
|
the number to attribute to whatever unit of work those sessions represent.
|
|
374
|
+
`producer_version` is the agent-cost package version that produced this
|
|
375
|
+
payload; `accounting_basis` identifies the token-accounting semantics
|
|
376
|
+
behind the numbers (separate from `protocol_version`, which only tracks
|
|
377
|
+
the JSON shape) -- a consumer that persists historical measurements should
|
|
378
|
+
key comparability on `accounting_basis`, not `producer_version` alone,
|
|
379
|
+
since a future release can bump the latter while keeping the former.
|
|
346
380
|
`data_quality.unpriced_tokens` and `.source_quality` are scoped to the
|
|
347
|
-
requested sessions;
|
|
348
|
-
|
|
349
|
-
|
|
381
|
+
requested sessions; so are the three dedup counters
|
|
382
|
+
(`duplicate_rows_skipped`, `conflicting_duplicate_groups`,
|
|
383
|
+
`missing_dedup_identity_rows`), which are Claude-only and computed over
|
|
384
|
+
the intersection of the requested session ids and the `--since`/`--until`
|
|
385
|
+
window, never over an unrequested session's rows.
|
|
386
|
+
`.malformed_events`/`.skipped_files`/`.negative_deltas` describe the
|
|
387
|
+
health of the underlying log read within `--since`/`--until` and are not
|
|
388
|
+
attributable to one session.
|
|
350
389
|
|
|
351
390
|
## Privacy
|
|
352
391
|
|
|
@@ -178,9 +178,33 @@ scraping `report`.
|
|
|
178
178
|
agent-cost only reads data that is already on disk. It never talks to the
|
|
179
179
|
network, never calls `gh`, and never resolves branches or PRs.
|
|
180
180
|
|
|
181
|
-
- **Claude Code**: every
|
|
182
|
-
|
|
183
|
-
|
|
181
|
+
- **Claude Code**: every logical assistant message is one billing event,
|
|
182
|
+
attributed to the exact model on that event (a session that switches
|
|
183
|
+
models mid-conversation is not folded into one "primary model"). Claude
|
|
184
|
+
Code's transcript writes one JSONL line per content block of the same
|
|
185
|
+
message, and those lines are deduplicated first -- but only when a row
|
|
186
|
+
carries a full `message.id` + `requestId` pair; a row missing either
|
|
187
|
+
half is emitted on its own (never merged) and flagged
|
|
188
|
+
`source_quality: "identity_missing"` rather than assumed
|
|
189
|
+
billing-accurate. `identity_missing` facts are still priced and included
|
|
190
|
+
in rows/totals; the flag is a warning, not an exclusion or an unpriced
|
|
191
|
+
status. A message's lines don't necessarily repeat an identical `usage`
|
|
192
|
+
block: `model` and the input-side fields (input tokens, cache read,
|
|
193
|
+
cache-write TTL breakdown) stay the same across a message's lines, but
|
|
194
|
+
`output_tokens` typically grows line by line as the response streams in,
|
|
195
|
+
and intermediate lines usually lack `usage.speed` entirely (reported as
|
|
196
|
+
mode `"unknown"`), with only the final line carrying a concrete mode.
|
|
197
|
+
Within a deduplicated group, `model` or an input-side field that
|
|
198
|
+
actually differs across the group's rows, two rows disagreeing on a
|
|
199
|
+
*concrete* mode (`"normal"` vs `"fast"`), or an `output_tokens` value
|
|
200
|
+
that decreases or is non-monotonic across them, is counted in
|
|
201
|
+
`data_quality.conflicting_duplicate_groups` as a real billing
|
|
202
|
+
disagreement -- `output_tokens` growing row by row, and mode
|
|
203
|
+
`"unknown"` mixed with a single concrete mode elsewhere in the group,
|
|
204
|
+
are both the ordinary streaming pattern just described and are not
|
|
205
|
+
flagged, since the reader's own cross-check of real transcripts found
|
|
206
|
+
exactly that pattern in every observed duplicated group. See
|
|
207
|
+
`CHANGELOG.md`'s 0.2.0 entry.
|
|
184
208
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
185
209
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
186
210
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -216,8 +240,7 @@ network, never calls `gh`, and never resolves branches or PRs.
|
|
|
216
240
|
in the report's `data_quality` block instead of being silently dropped or
|
|
217
241
|
clamped to zero.
|
|
218
242
|
- **Known catalog gaps**, tracked in `agent_cost/rates.json`'s `notes`:
|
|
219
|
-
`
|
|
220
|
-
source, so its rate period's `effective_from` is a placeholder. `gpt-5.6`
|
|
243
|
+
`gpt-5.6`
|
|
221
244
|
(Sol/Terra/Luna) credits could not be confirmed from the primary source
|
|
222
245
|
(`help.openai.com`'s Codex rate card returns HTTP 403 to automated
|
|
223
246
|
fetches); the values in the catalog come from several independent
|
|
@@ -292,6 +315,8 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
292
315
|
```json
|
|
293
316
|
{
|
|
294
317
|
"protocol_version": "measure/v1",
|
|
318
|
+
"producer_version": "0.2.1",
|
|
319
|
+
"accounting_basis": "agent-cost-raw-total/v2",
|
|
295
320
|
"generated_at": "...",
|
|
296
321
|
"window": { "since": "...", "until": null },
|
|
297
322
|
"timezone": "UTC",
|
|
@@ -308,7 +333,10 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
308
333
|
"skipped_files": 0,
|
|
309
334
|
"negative_deltas": 0,
|
|
310
335
|
"unpriced_tokens": 0,
|
|
311
|
-
"
|
|
336
|
+
"duplicate_rows_skipped": 0,
|
|
337
|
+
"conflicting_duplicate_groups": 0,
|
|
338
|
+
"missing_dedup_identity_rows": 0,
|
|
339
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
312
340
|
}
|
|
313
341
|
}
|
|
314
342
|
```
|
|
@@ -317,10 +345,21 @@ Rows are grouped by agent/model/token-kind only -- `measure` never buckets
|
|
|
317
345
|
by month, since a query is already scoped to specific sessions. `total` is
|
|
318
346
|
the union of every requested `session_id` (not a global report), so it's
|
|
319
347
|
the number to attribute to whatever unit of work those sessions represent.
|
|
348
|
+
`producer_version` is the agent-cost package version that produced this
|
|
349
|
+
payload; `accounting_basis` identifies the token-accounting semantics
|
|
350
|
+
behind the numbers (separate from `protocol_version`, which only tracks
|
|
351
|
+
the JSON shape) -- a consumer that persists historical measurements should
|
|
352
|
+
key comparability on `accounting_basis`, not `producer_version` alone,
|
|
353
|
+
since a future release can bump the latter while keeping the former.
|
|
320
354
|
`data_quality.unpriced_tokens` and `.source_quality` are scoped to the
|
|
321
|
-
requested sessions;
|
|
322
|
-
|
|
323
|
-
|
|
355
|
+
requested sessions; so are the three dedup counters
|
|
356
|
+
(`duplicate_rows_skipped`, `conflicting_duplicate_groups`,
|
|
357
|
+
`missing_dedup_identity_rows`), which are Claude-only and computed over
|
|
358
|
+
the intersection of the requested session ids and the `--since`/`--until`
|
|
359
|
+
window, never over an unrequested session's rows.
|
|
360
|
+
`.malformed_events`/`.skipped_files`/`.negative_deltas` describe the
|
|
361
|
+
health of the underlying log read within `--since`/`--until` and are not
|
|
362
|
+
attributable to one session.
|
|
324
363
|
|
|
325
364
|
## Privacy
|
|
326
365
|
|
|
@@ -84,12 +84,59 @@ def filter_facts(
|
|
|
84
84
|
yield f
|
|
85
85
|
|
|
86
86
|
|
|
87
|
+
def scope_dedup_units(
|
|
88
|
+
units: Iterable,
|
|
89
|
+
*,
|
|
90
|
+
since_utc: Optional[datetime] = None,
|
|
91
|
+
until_utc: Optional[datetime] = None,
|
|
92
|
+
session_ids: Optional[set] = None,
|
|
93
|
+
) -> Tuple[int, int, int]:
|
|
94
|
+
"""Re-scope Claude's per-group dedup diagnostics to a window and/or a
|
|
95
|
+
set of session ids, after the fact.
|
|
96
|
+
|
|
97
|
+
``units`` is duck-typed (each must have ``.occurred_at_utc``,
|
|
98
|
+
``.session_id``, ``.duplicate_rows_skipped``, ``.conflicting`` and
|
|
99
|
+
``.missing_identity``) rather than imported as
|
|
100
|
+
``readers.claude.ClaudeDedupUnit``, to avoid this module depending on
|
|
101
|
+
a specific reader. Returns
|
|
102
|
+
``(duplicate_rows_skipped, conflicting_duplicate_groups,
|
|
103
|
+
missing_dedup_identity_rows)`` -- the same three counters
|
|
104
|
+
``ClaudeParseResult``/``ReadResult`` expose as unscoped file-level
|
|
105
|
+
totals, but summed only over units that fall inside ``[since_utc,
|
|
106
|
+
until_utc)`` and, if given, whose ``session_id`` is in
|
|
107
|
+
``session_ids``. This mirrors ``filter_facts``'s half-open window
|
|
108
|
+
semantics so a caller's ``report``/``measure`` window matches exactly
|
|
109
|
+
what ``build_rows`` priced.
|
|
110
|
+
"""
|
|
111
|
+
duplicate_rows_skipped = 0
|
|
112
|
+
conflicting_duplicate_groups = 0
|
|
113
|
+
missing_dedup_identity_rows = 0
|
|
114
|
+
for unit in units:
|
|
115
|
+
if since_utc is not None and unit.occurred_at_utc < since_utc:
|
|
116
|
+
continue
|
|
117
|
+
if until_utc is not None and unit.occurred_at_utc >= until_utc:
|
|
118
|
+
continue
|
|
119
|
+
if session_ids is not None and unit.session_id not in session_ids:
|
|
120
|
+
continue
|
|
121
|
+
duplicate_rows_skipped += unit.duplicate_rows_skipped
|
|
122
|
+
if unit.conflicting:
|
|
123
|
+
conflicting_duplicate_groups += 1
|
|
124
|
+
if unit.missing_identity:
|
|
125
|
+
missing_dedup_identity_rows += 1
|
|
126
|
+
return duplicate_rows_skipped, conflicting_duplicate_groups, missing_dedup_identity_rows
|
|
127
|
+
|
|
128
|
+
|
|
87
129
|
@dataclass
|
|
88
130
|
class DataQuality:
|
|
89
131
|
malformed_events: int = 0
|
|
90
132
|
skipped_files: int = 0
|
|
91
133
|
negative_deltas: int = 0
|
|
92
134
|
unpriced_tokens: int = 0
|
|
135
|
+
# Claude-only dedup diagnostics (see readers/claude.py's
|
|
136
|
+
# parse_session_detailed docstring); always 0 for Codex facts.
|
|
137
|
+
duplicate_rows_skipped: int = 0
|
|
138
|
+
conflicting_duplicate_groups: int = 0
|
|
139
|
+
missing_dedup_identity_rows: int = 0
|
|
93
140
|
|
|
94
141
|
def to_dict(self) -> dict:
|
|
95
142
|
return {
|
|
@@ -97,6 +144,9 @@ class DataQuality:
|
|
|
97
144
|
"skipped_files": self.skipped_files,
|
|
98
145
|
"negative_deltas": self.negative_deltas,
|
|
99
146
|
"unpriced_tokens": self.unpriced_tokens,
|
|
147
|
+
"duplicate_rows_skipped": self.duplicate_rows_skipped,
|
|
148
|
+
"conflicting_duplicate_groups": self.conflicting_duplicate_groups,
|
|
149
|
+
"missing_dedup_identity_rows": self.missing_dedup_identity_rows,
|
|
100
150
|
}
|
|
101
151
|
|
|
102
152
|
|
|
@@ -11,7 +11,7 @@ from typing import Optional
|
|
|
11
11
|
from zoneinfo import ZoneInfo
|
|
12
12
|
|
|
13
13
|
from . import __version__
|
|
14
|
-
from .aggregate import DataQuality, build_rows, filter_facts, rows_totals
|
|
14
|
+
from .aggregate import DataQuality, build_rows, filter_facts, rows_totals, scope_dedup_units
|
|
15
15
|
from .config import load_config
|
|
16
16
|
from .facts import SOURCE_QUALITY_VALUES
|
|
17
17
|
from .rates import RatesValidationError, load_rates
|
|
@@ -25,6 +25,15 @@ from .renderers import render_csv, render_json, render_table
|
|
|
25
25
|
#: should check this before trusting the shape of the payload.
|
|
26
26
|
MEASURE_PROTOCOL_VERSION = "measure/v1"
|
|
27
27
|
|
|
28
|
+
#: Identifies the token-accounting semantics behind measure's numbers, not
|
|
29
|
+
#: the JSON shape (that's MEASURE_PROTOCOL_VERSION). "v2" marks the fix for
|
|
30
|
+
#: the Claude reader's row-per-content-block over-count (agent-cost 0.2.0);
|
|
31
|
+
#: "v1" numbers (agent-cost 0.1.x) are not comparable to "v2" numbers for
|
|
32
|
+
#: Claude facts. A consumer that persists historical measurements (e.g.
|
|
33
|
+
#: lane's ledger) should key on this, not on producer_version alone, since
|
|
34
|
+
#: a future producer_version could still share the same accounting_basis.
|
|
35
|
+
ACCOUNTING_BASIS = "agent-cost-raw-total/v2"
|
|
36
|
+
|
|
28
37
|
|
|
29
38
|
def _parse_window_bound(value: Optional[str], tz: ZoneInfo) -> Optional[datetime]:
|
|
30
39
|
if not value:
|
|
@@ -45,14 +54,29 @@ def _parse_window_bound(value: Optional[str], tz: ZoneInfo) -> Optional[datetime
|
|
|
45
54
|
|
|
46
55
|
|
|
47
56
|
def _collect_facts(config, *, agents: set, exclude_archived: bool):
|
|
57
|
+
"""Returns ``(facts, dq, claude_dedup_units)``.
|
|
58
|
+
|
|
59
|
+
``dq``'s ``malformed_events``/``skipped_files``/``negative_deltas`` are
|
|
60
|
+
unscoped file-level totals (unchanged, existing behavior).
|
|
61
|
+
``dq.duplicate_rows_skipped``/``conflicting_duplicate_groups``/
|
|
62
|
+
``missing_dedup_identity_rows`` are deliberately left at their
|
|
63
|
+
``DataQuality`` defaults here (0) rather than filled in from an
|
|
64
|
+
unscoped total: those three are Claude-only and must always be
|
|
65
|
+
re-scoped to the caller's actual window (``report``) or requested
|
|
66
|
+
session ids + window (``measure``) via ``claude_dedup_units`` and
|
|
67
|
+
``aggregate.scope_dedup_units`` -- an unscoped total would silently
|
|
68
|
+
misrepresent a windowed/session-scoped report or measure output.
|
|
69
|
+
"""
|
|
48
70
|
facts: list = []
|
|
49
71
|
dq = DataQuality()
|
|
72
|
+
claude_dedup_units: list = []
|
|
50
73
|
|
|
51
74
|
if "claude" in agents:
|
|
52
75
|
result = claude_reader.read_claude_facts(config.claude_projects_dir)
|
|
53
76
|
facts.extend(result.facts)
|
|
54
77
|
dq.malformed_events += result.malformed_events
|
|
55
78
|
dq.skipped_files += result.skipped_files
|
|
79
|
+
claude_dedup_units.extend(result.claude_dedup_units)
|
|
56
80
|
|
|
57
81
|
if "codex" in agents and config.codex_db_path.exists():
|
|
58
82
|
result = codex_reader.read_codex_facts(
|
|
@@ -65,7 +89,7 @@ def _collect_facts(config, *, agents: set, exclude_archived: bool):
|
|
|
65
89
|
dq.skipped_files += result.skipped_files
|
|
66
90
|
dq.negative_deltas += result.negative_deltas
|
|
67
91
|
|
|
68
|
-
return facts, dq
|
|
92
|
+
return facts, dq, claude_dedup_units
|
|
69
93
|
|
|
70
94
|
|
|
71
95
|
def cmd_report(args) -> int:
|
|
@@ -87,10 +111,20 @@ def cmd_report(args) -> int:
|
|
|
87
111
|
print(f"[error] rates catalog invalid: {exc}", file=sys.stderr)
|
|
88
112
|
return 2
|
|
89
113
|
|
|
90
|
-
facts, dq = _collect_facts(
|
|
114
|
+
facts, dq, claude_dedup_units = _collect_facts(
|
|
115
|
+
config, agents=agents, exclude_archived=args.exclude_archived
|
|
116
|
+
)
|
|
91
117
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
92
118
|
rows, agg_dq = build_rows(facts, catalog, group_by=group_by, timezone_name=args.timezone)
|
|
93
119
|
dq.unpriced_tokens = agg_dq.unpriced_tokens
|
|
120
|
+
# The three dedup counters are Claude-only and always re-scoped to this
|
|
121
|
+
# report's actual [since, until) window -- see _collect_facts's
|
|
122
|
+
# docstring for why they aren't filled in from an unscoped total.
|
|
123
|
+
(
|
|
124
|
+
dq.duplicate_rows_skipped,
|
|
125
|
+
dq.conflicting_duplicate_groups,
|
|
126
|
+
dq.missing_dedup_identity_rows,
|
|
127
|
+
) = scope_dedup_units(claude_dedup_units, since_utc=since, until_utc=until)
|
|
94
128
|
|
|
95
129
|
payload = {
|
|
96
130
|
"schema_version": "1",
|
|
@@ -118,7 +152,7 @@ def cmd_export(args) -> int:
|
|
|
118
152
|
until = _parse_window_bound(args.until, tz)
|
|
119
153
|
agents = set(args.agent.split(",")) if args.agent else {"claude", "codex"}
|
|
120
154
|
|
|
121
|
-
facts, _dq = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
155
|
+
facts, _dq, _claude_dedup_units = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
122
156
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
123
157
|
|
|
124
158
|
out = open(args.out, "w") if args.out else sys.stdout
|
|
@@ -180,7 +214,7 @@ def cmd_measure(args) -> int:
|
|
|
180
214
|
return 2
|
|
181
215
|
|
|
182
216
|
config = load_config()
|
|
183
|
-
facts, dq = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
217
|
+
facts, dq, claude_dedup_units = _collect_facts(config, agents=agents, exclude_archived=False)
|
|
184
218
|
facts = list(filter_facts(facts, since_utc=since, until_utc=until, agents=agents))
|
|
185
219
|
|
|
186
220
|
# measure is a per-session query, not a time-bucketed report: group by
|
|
@@ -190,6 +224,17 @@ def cmd_measure(args) -> int:
|
|
|
190
224
|
requested = set(session_ids)
|
|
191
225
|
combined_facts = [f for f in facts if f.session_id in requested]
|
|
192
226
|
|
|
227
|
+
# The three dedup counters are Claude-only and always re-scoped to
|
|
228
|
+
# exactly this measure call's requested session ids AND window -- an
|
|
229
|
+
# unrequested session's conflict/identity-missing rows must never
|
|
230
|
+
# affect a requested session's counters (see _collect_facts's
|
|
231
|
+
# docstring and cmd_report's equivalent window-only scoping above).
|
|
232
|
+
(
|
|
233
|
+
dedup_rows_skipped,
|
|
234
|
+
dedup_conflicting_groups,
|
|
235
|
+
dedup_missing_identity_rows,
|
|
236
|
+
) = scope_dedup_units(claude_dedup_units, since_utc=since, until_utc=until, session_ids=requested)
|
|
237
|
+
|
|
193
238
|
quality_counts = {v: 0 for v in SOURCE_QUALITY_VALUES}
|
|
194
239
|
for f in combined_facts:
|
|
195
240
|
quality_counts[f.source_quality] = quality_counts.get(f.source_quality, 0) + 1
|
|
@@ -208,6 +253,8 @@ def cmd_measure(args) -> int:
|
|
|
208
253
|
|
|
209
254
|
payload = {
|
|
210
255
|
"protocol_version": MEASURE_PROTOCOL_VERSION,
|
|
256
|
+
"producer_version": __version__,
|
|
257
|
+
"accounting_basis": ACCOUNTING_BASIS,
|
|
211
258
|
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
212
259
|
"window": {
|
|
213
260
|
"since": since.isoformat() if since else None,
|
|
@@ -227,6 +274,9 @@ def cmd_measure(args) -> int:
|
|
|
227
274
|
"skipped_files": dq.skipped_files,
|
|
228
275
|
"negative_deltas": dq.negative_deltas,
|
|
229
276
|
"unpriced_tokens": total_dq.unpriced_tokens,
|
|
277
|
+
"duplicate_rows_skipped": dedup_rows_skipped,
|
|
278
|
+
"conflicting_duplicate_groups": dedup_conflicting_groups,
|
|
279
|
+
"missing_dedup_identity_rows": dedup_missing_identity_rows,
|
|
230
280
|
"source_quality": quality_counts,
|
|
231
281
|
},
|
|
232
282
|
}
|
|
@@ -30,9 +30,12 @@ MODES = ("fast", "normal", "unknown")
|
|
|
30
30
|
# covers the ordinary case; readers add a more specific value only when a
|
|
31
31
|
# fact's derivation has a real, nameable caveat worth surfacing downstream
|
|
32
32
|
# (e.g. Codex's first delta in a rollout is measured against an assumed
|
|
33
|
-
# zero baseline
|
|
34
|
-
# so
|
|
35
|
-
|
|
33
|
+
# zero baseline; Claude's reader can't dedup a row that lacks a full
|
|
34
|
+
# (message.id, requestId) pair, so it emits that row individually and
|
|
35
|
+
# flags it "identity_missing" rather than silently treating it as "ok").
|
|
36
|
+
# This is never left unset -- every Fact defaults to "ok" so export never
|
|
37
|
+
# emits a null source_quality.
|
|
38
|
+
SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing")
|
|
36
39
|
|
|
37
40
|
_BRACKET_SUFFIX = re.compile(r"\[[^\]]*\]$")
|
|
38
41
|
_DATE_SUFFIX = re.compile(r"@\d{6,8}$")
|
|
@@ -1,14 +1,17 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schema_version": "1",
|
|
3
|
-
"catalog_version": "2026-09-
|
|
3
|
+
"catalog_version": "2026-09-23",
|
|
4
4
|
"currency": "USD",
|
|
5
5
|
"unit": "per_mtok",
|
|
6
6
|
"usd_per_credit": "0.04",
|
|
7
7
|
"notes": [
|
|
8
8
|
"All Claude rates are standard (non-batch, non-priority) API prices. Batch API is roughly 50% off, but agent-cost's logs have no signal for whether a given request went through Batch, so batch usage is priced at standard rates and will overstate cost for anyone using Batch.",
|
|
9
|
-
"claude-opus-5
|
|
9
|
+
"claude-opus-5 effective_from corrected 2026-09-23: the 2026-07-01 placeholder (rate_id claude-opus-5-launch-date-unconfirmed) is replaced by the official launch date 2026-07-24 (anthropic.com/news/claude-opus-5, dated 'Jul 24, 2026'), rate_id claude-opus-5-launch-2026-07-24 -- the period is replaced rather than edited in place, following the claude-sonnet-5-standard -> claude-sonnet-5-standard-2026-09-01 correction precedent (rate_id is not consumed by any reader or report field; reports pin catalog_version + sha256). The source gives a calendar date only, so effective_from is 00:00:00Z of that day: a transcript row carrying this model ID cannot predate the public launch, so nothing on the launch day is mispriced by the earlier-of-day boundary. The earliest claude-opus-5 row across local transcripts is 2026-07-28T18:22:52Z, so no locally observed event moves from priced to unpriced; events between 2026-07-01 and 2026-07-23 that were priced under the placeholder are now unpriced, which is the correct treatment for a model that did not exist yet.",
|
|
10
10
|
"gpt-5.6 (sol/terra/luna) credits could not be confirmed from the primary source (help.openai.com's Codex rate card returned HTTP 403 to automated fetches). The values here were cross-checked across several independent secondary sources that agreed with each other and are internally consistent (Sol's $5/$30 per-MTok API price x 25 = 125/750 credits, matching usd_per_credit=0.04); see the gpt-5.6-* sources entries below. Treat effective_from (2026-07-01) as a placeholder and re-verify against the primary rate card when it becomes reachable.",
|
|
11
|
-
"gpt-6-astra: Codex standard token-based rates only; not API pricing, legacy message metering, or actual contractual charges. Confirmed 2026-09-06 JST (2026-09-05T17:18:23Z). effective_from is this observation cutoff, NOT an official launch or price-start time; earlier events remain unpriced. No Codex long-context surcharge or cache-write charge is applied. Fast multiplier 2.5 applies to explicit priority request settings with matching owned turn/model context; explicit default uses Standard. Missing or ambiguous Astra modes remain unpriced. These settings do not confirm the backend processing tier. See docs/astra-pricing.md for compatibility limits and the separate GPT-5.6 discrepancy report."
|
|
11
|
+
"gpt-6-astra: Codex standard token-based rates only; not API pricing, legacy message metering, or actual contractual charges. Confirmed 2026-09-06 JST (2026-09-05T17:18:23Z). effective_from is this observation cutoff, NOT an official launch or price-start time; earlier events remain unpriced. No Codex long-context surcharge or cache-write charge is applied. Fast multiplier 2.5 applies to explicit priority request settings with matching owned turn/model context; explicit default uses Standard. Missing or ambiguous Astra modes remain unpriced. These settings do not confirm the backend processing tier. See docs/astra-pricing.md for compatibility limits and the separate GPT-5.6 discrepancy report.",
|
|
12
|
+
"claude-fable-5-1 added 2026-09-09: rate_id claude-fable-5-1-local-first-observed-2026-09-02. effective_from (2026-09-02T01:17:55Z) is the timestamp the raw model ID was first observed across all 54 local transcript files (none found in August), not a confirmed official launch date -- if an authoritative launch date is confirmed later, add it as a separate rate_id with effective_until on this one rather than editing this period in place. Values (cache_read $0.25/MTok = 0.025x base input) differ from claude-fable-5's cache_read ($1.00/MTok = 0.1x); this is a genuine per-model price difference confirmed against the primary source (see sources below), not an inconsistency to reconcile.",
|
|
13
|
+
"claude-sonnet-5 corrected 2026-09-09: a prior period (rate_id claude-sonnet-5-standard, effective_from 2026-09-01, values 3.0/0.30/3.75/6.0/15.0) priced a previously-announced rate increase that the pricing page's claude-sonnet-5-introductory-pricing note states did not occur (\"The previously scheduled increase to $3/$15 ... on September 1, 2026 will not occur\"). That period has been replaced by claude-sonnet-5-standard-2026-09-01, which carries the same $2/$0.20/$2.50/$4/$10 values as the preceding claude-sonnet-5-launch-promo period, since the launch price is confirmed to be the ongoing price with no change at the 2026-09-01 boundary. Any measure/v1 or report output computed against the pre-correction catalog for claude-sonnet-5 sessions occurring on or after 2026-09-01 overstated cost by 1.5x for that model. To detect this class of error going forward, cross-check each priced model's rate periods against its primary pricing page's Note text on a recurring (e.g. monthly) basis for mentions of a rate-change announcement being withdrawn or not taking effect -- a period can be internally valid (schema-wise) and still price a rate that was announced but never actually charged.",
|
|
14
|
+
"claude-opus-5-5 added 2026-09-23: rate_id claude-opus-5-5-launch-2026-09-22. effective_from 2026-09-22T00:00:00Z is 00:00:00Z of the official public launch date (anthropic.com/claude-opus-5-5, 'September 22, 2026'); the launch time of day is not published, but a transcript row whose raw model ID is claude-opus-5-5 cannot predate the launch, so pricing from the start of that UTC day cannot price a pre-launch event. Values 4.0/0.20/5.0/8.0/20.0 per the pricing page: cache_read is 0.05x base input (pricing-page footnote 2), a genuine per-model difference from the standard 0.1x and from claude-fable-5-1's 0.025x. fast_multiplier 2.0 (fast mode $8/$40 per MTok on the pricing page). Claude Code's CHANGELOG entry for 2.1.280 states Opus 5.5 became the default model (see sources); rows whose raw model ID is claude-opus-5-5 would be unpriced without this entry. Claude Sonnet 5.5 and Claude Haiku 5.5 are announced to follow 'in the coming weeks' -- add them as separate entries when their model IDs and prices are published, never as aliases of the 5.x entries."
|
|
12
15
|
],
|
|
13
16
|
"sources": [
|
|
14
17
|
{
|
|
@@ -48,6 +51,36 @@
|
|
|
48
51
|
"url": "https://developers.openai.com/api/docs/models/gpt-6-astra",
|
|
49
52
|
"retrieved_at": "2026-09-05T17:18:23Z",
|
|
50
53
|
"note": "Boundary check only: API cache writes, long context and API Fast rates are NOT applied to this Codex entry."
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"url": "https://platform.claude.com/docs/en/about-claude/pricing",
|
|
57
|
+
"retrieved_at": "2026-09-09",
|
|
58
|
+
"note": "used for claude-fable-5-1 (cache_read 0.025x footnote) and the claude-sonnet-5-introductory-pricing correction; local snapshot 2026-09-09 sha256=d79ad28567196bd55dd50e2fd89341b9da9774a45c1d02fcf387078507fd15e0"
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"url": "https://platform.claude.com/docs/en/build-with-claude/prompt-caching",
|
|
62
|
+
"retrieved_at": "2026-09-09",
|
|
63
|
+
"note": "cache write TTL multipliers (5m = 1.25x, 1h = 2x), cross-checked against the pricing page cache columns"
|
|
64
|
+
},
|
|
65
|
+
{
|
|
66
|
+
"url": "https://platform.claude.com/docs/en/about-claude/pricing",
|
|
67
|
+
"retrieved_at": "2026-09-23",
|
|
68
|
+
"note": "used for claude-opus-5-5 (Opus 5.5 row: $4 / $5 / $8 / $0.20 / $20, footnote 2 cache_read 0.05x, fast mode $8/$40); fetched via WebFetch, no local snapshot"
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
"url": "https://www.anthropic.com/claude-opus-5-5",
|
|
72
|
+
"retrieved_at": "2026-09-23",
|
|
73
|
+
"note": "Opus 5.5 announcement: public launch date September 22, 2026, model ID claude-opus-5-5, Sonnet 5.5 / Haiku 5.5 to follow"
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
"url": "https://www.anthropic.com/news/claude-opus-5",
|
|
77
|
+
"retrieved_at": "2026-09-23",
|
|
78
|
+
"note": "Opus 5 announcement dated Jul 24, 2026; used to replace the claude-opus-5 effective_from placeholder"
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"url": "https://github.com/anthropics/claude-code/blob/main/CHANGELOG.md",
|
|
82
|
+
"retrieved_at": "2026-09-23",
|
|
83
|
+
"note": "Claude Code 2.1.280 entry: Claude Opus 5.5 became the default model (1M context, $4/$20 per MTok, $0.20/MTok cache reads); the reason claude-opus-5-5 rows are expected in transcripts from that version"
|
|
51
84
|
}
|
|
52
85
|
],
|
|
53
86
|
"models": [
|
|
@@ -68,6 +101,23 @@
|
|
|
68
101
|
}
|
|
69
102
|
]
|
|
70
103
|
},
|
|
104
|
+
{
|
|
105
|
+
"model_key": "claude-fable-5-1",
|
|
106
|
+
"aliases": [],
|
|
107
|
+
"fast_multiplier": "1.0",
|
|
108
|
+
"rates": [
|
|
109
|
+
{
|
|
110
|
+
"rate_id": "claude-fable-5-1-local-first-observed-2026-09-02",
|
|
111
|
+
"effective_from": "2026-09-02T01:17:55+00:00",
|
|
112
|
+
"effective_until": null,
|
|
113
|
+
"input_nocache": "10.0",
|
|
114
|
+
"cache_read": "0.25",
|
|
115
|
+
"cache_write_5m": "12.50",
|
|
116
|
+
"cache_write_1h": "20.0",
|
|
117
|
+
"output": "50.0"
|
|
118
|
+
}
|
|
119
|
+
]
|
|
120
|
+
},
|
|
71
121
|
{
|
|
72
122
|
"model_key": "claude-mythos-5",
|
|
73
123
|
"aliases": [],
|
|
@@ -108,8 +158,8 @@
|
|
|
108
158
|
"fast_multiplier": "2.0",
|
|
109
159
|
"rates": [
|
|
110
160
|
{
|
|
111
|
-
"rate_id": "claude-opus-5-launch-
|
|
112
|
-
"effective_from": "2026-07-
|
|
161
|
+
"rate_id": "claude-opus-5-launch-2026-07-24",
|
|
162
|
+
"effective_from": "2026-07-24T00:00:00+00:00",
|
|
113
163
|
"effective_until": null,
|
|
114
164
|
"input_nocache": "5.0",
|
|
115
165
|
"cache_read": "0.50",
|
|
@@ -119,6 +169,23 @@
|
|
|
119
169
|
}
|
|
120
170
|
]
|
|
121
171
|
},
|
|
172
|
+
{
|
|
173
|
+
"model_key": "claude-opus-5-5",
|
|
174
|
+
"aliases": [],
|
|
175
|
+
"fast_multiplier": "2.0",
|
|
176
|
+
"rates": [
|
|
177
|
+
{
|
|
178
|
+
"rate_id": "claude-opus-5-5-launch-2026-09-22",
|
|
179
|
+
"effective_from": "2026-09-22T00:00:00+00:00",
|
|
180
|
+
"effective_until": null,
|
|
181
|
+
"input_nocache": "4.0",
|
|
182
|
+
"cache_read": "0.20",
|
|
183
|
+
"cache_write_5m": "5.0",
|
|
184
|
+
"cache_write_1h": "8.0",
|
|
185
|
+
"output": "20.0"
|
|
186
|
+
}
|
|
187
|
+
]
|
|
188
|
+
},
|
|
122
189
|
{
|
|
123
190
|
"model_key": "claude-opus-4-7",
|
|
124
191
|
"aliases": [],
|
|
@@ -186,14 +253,14 @@
|
|
|
186
253
|
"output": "10.0"
|
|
187
254
|
},
|
|
188
255
|
{
|
|
189
|
-
"rate_id": "claude-sonnet-5-standard",
|
|
256
|
+
"rate_id": "claude-sonnet-5-standard-2026-09-01",
|
|
190
257
|
"effective_from": "2026-09-01T00:00:00+00:00",
|
|
191
258
|
"effective_until": null,
|
|
192
|
-
"input_nocache": "
|
|
193
|
-
"cache_read": "0.
|
|
194
|
-
"cache_write_5m": "
|
|
195
|
-
"cache_write_1h": "
|
|
196
|
-
"output": "
|
|
259
|
+
"input_nocache": "2.0",
|
|
260
|
+
"cache_read": "0.20",
|
|
261
|
+
"cache_write_5m": "2.50",
|
|
262
|
+
"cache_write_1h": "4.0",
|
|
263
|
+
"output": "10.0"
|
|
197
264
|
}
|
|
198
265
|
]
|
|
199
266
|
},
|
|
@@ -24,3 +24,14 @@ class ReadResult:
|
|
|
24
24
|
# `data_quality` (which is reader-agnostic); surfaced for `doctor` /
|
|
25
25
|
# tests.
|
|
26
26
|
tokens_used_diffs: int = 0
|
|
27
|
+
# Claude-only dedup diagnostics (see readers/claude.py's
|
|
28
|
+
# parse_session_detailed docstring). Always 0 / empty for the Codex
|
|
29
|
+
# reader. The three ints are file-level totals, NOT scoped to any
|
|
30
|
+
# window or session; `claude_dedup_units` carries the same
|
|
31
|
+
# information per group/row (as `ClaudeDedupUnit`, duck-typed here to
|
|
32
|
+
# avoid a circular import with readers.claude) so a caller can
|
|
33
|
+
# re-scope it -- see `agent_cost.aggregate.scope_dedup_units`.
|
|
34
|
+
duplicate_rows_skipped: int = 0
|
|
35
|
+
conflicting_duplicate_groups: int = 0
|
|
36
|
+
missing_dedup_identity_rows: int = 0
|
|
37
|
+
claude_dedup_units: List = field(default_factory=list)
|