coding-agent-cost 0.2.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {coding_agent_cost-0.2.2/coding_agent_cost.egg-info → coding_agent_cost-0.3.0}/PKG-INFO +14 -5
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/README.md +13 -4
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/__init__.py +1 -1
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/facts.py +6 -2
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/claude.py +25 -3
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0/coding_agent_cost.egg-info}/PKG-INFO +14 -5
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/SOURCES.txt +1 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/pyproject.toml +1 -1
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_cli.py +1 -0
- coding_agent_cost-0.3.0/tests/test_reader_output_lower_bound.py +265 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_version_consistency.py +6 -6
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/LICENSE +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/aggregate.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/cli.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/config.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/rates.json +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/rates.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/__init__.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/codex.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/renderers.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/dependency_links.txt +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/entry_points.txt +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/top_level.txt +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/setup.cfg +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_aggregate.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_astra_pricing.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_e0a_fable_5_1_and_sonnet_5_correction.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_e0a_review3_hardening.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_facts.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_measure_v1_contract.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_opus_5_5_rates.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_rates.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_claude.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_codex.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_dedup.py +0 -0
- {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_sonnet_5_5_rates.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: coding-agent-cost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
|
|
5
5
|
Author: shiki-yusuke
|
|
6
6
|
License: MIT
|
|
@@ -232,6 +232,14 @@ network, never calls `gh`, and never resolves branches or PRs.
|
|
|
232
232
|
flagged, since the reader's own cross-check of real transcripts found
|
|
233
233
|
exactly that pattern in every observed duplicated group. See
|
|
234
234
|
`CHANGELOG.md`'s 0.2.0 entry.
|
|
235
|
+
If the row a group adopts has no `stop_reason` (common in subagent
|
|
236
|
+
transcripts, where a message ending in `tool_use` may never get its final
|
|
237
|
+
line), its `output_tokens` is only the streaming head's value: the
|
|
238
|
+
group's `output` fact is flagged `source_quality: "output_lower_bound"`
|
|
239
|
+
-- the observed lower bound of that message's output tokens, not
|
|
240
|
+
guaranteed to be >= the true count. The group's input-side facts stay
|
|
241
|
+
`"ok"`, and the token amount itself is unchanged (still priced and
|
|
242
|
+
included in rows/totals); the flag is a warning only.
|
|
235
243
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
236
244
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
237
245
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -342,7 +350,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
342
350
|
```json
|
|
343
351
|
{
|
|
344
352
|
"protocol_version": "measure/v1",
|
|
345
|
-
"producer_version": "0.
|
|
353
|
+
"producer_version": "0.3.0",
|
|
346
354
|
"accounting_basis": "agent-cost-raw-total/v2",
|
|
347
355
|
"generated_at": "...",
|
|
348
356
|
"window": { "since": "...", "until": null },
|
|
@@ -363,7 +371,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
363
371
|
"duplicate_rows_skipped": 0,
|
|
364
372
|
"conflicting_duplicate_groups": 0,
|
|
365
373
|
"missing_dedup_identity_rows": 0,
|
|
366
|
-
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
374
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
|
|
367
375
|
}
|
|
368
376
|
}
|
|
369
377
|
```
|
|
@@ -395,8 +403,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
|
|
|
395
403
|
branch names -- only the fields needed to reproduce a cost estimate:
|
|
396
404
|
`occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
|
|
397
405
|
`token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
|
|
398
|
-
caveat about how that one fact was derived
|
|
399
|
-
`"first_event_delta"
|
|
406
|
+
caveat about how that one fact was derived: `"ok"`, Codex's
|
|
407
|
+
`"first_event_delta"`, or Claude's `"identity_missing"` /
|
|
408
|
+
`"output_lower_bound"` -- never null).
|
|
400
409
|
|
|
401
410
|
## License
|
|
402
411
|
|
|
@@ -206,6 +206,14 @@ network, never calls `gh`, and never resolves branches or PRs.
|
|
|
206
206
|
flagged, since the reader's own cross-check of real transcripts found
|
|
207
207
|
exactly that pattern in every observed duplicated group. See
|
|
208
208
|
`CHANGELOG.md`'s 0.2.0 entry.
|
|
209
|
+
If the row a group adopts has no `stop_reason` (common in subagent
|
|
210
|
+
transcripts, where a message ending in `tool_use` may never get its final
|
|
211
|
+
line), its `output_tokens` is only the streaming head's value: the
|
|
212
|
+
group's `output` fact is flagged `source_quality: "output_lower_bound"`
|
|
213
|
+
-- the observed lower bound of that message's output tokens, not
|
|
214
|
+
guaranteed to be >= the true count. The group's input-side facts stay
|
|
215
|
+
`"ok"`, and the token amount itself is unchanged (still priced and
|
|
216
|
+
included in rows/totals); the flag is a warning only.
|
|
209
217
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
210
218
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
211
219
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -316,7 +324,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
316
324
|
```json
|
|
317
325
|
{
|
|
318
326
|
"protocol_version": "measure/v1",
|
|
319
|
-
"producer_version": "0.
|
|
327
|
+
"producer_version": "0.3.0",
|
|
320
328
|
"accounting_basis": "agent-cost-raw-total/v2",
|
|
321
329
|
"generated_at": "...",
|
|
322
330
|
"window": { "since": "...", "until": null },
|
|
@@ -337,7 +345,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
337
345
|
"duplicate_rows_skipped": 0,
|
|
338
346
|
"conflicting_duplicate_groups": 0,
|
|
339
347
|
"missing_dedup_identity_rows": 0,
|
|
340
|
-
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
348
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
|
|
341
349
|
}
|
|
342
350
|
}
|
|
343
351
|
```
|
|
@@ -369,8 +377,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
|
|
|
369
377
|
branch names -- only the fields needed to reproduce a cost estimate:
|
|
370
378
|
`occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
|
|
371
379
|
`token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
|
|
372
|
-
caveat about how that one fact was derived
|
|
373
|
-
`"first_event_delta"
|
|
380
|
+
caveat about how that one fact was derived: `"ok"`, Codex's
|
|
381
|
+
`"first_event_delta"`, or Claude's `"identity_missing"` /
|
|
382
|
+
`"output_lower_bound"` -- never null).
|
|
374
383
|
|
|
375
384
|
## License
|
|
376
385
|
|
|
@@ -32,10 +32,14 @@ MODES = ("fast", "normal", "unknown")
|
|
|
32
32
|
# (e.g. Codex's first delta in a rollout is measured against an assumed
|
|
33
33
|
# zero baseline; Claude's reader can't dedup a row that lacks a full
|
|
34
34
|
# (message.id, requestId) pair, so it emits that row individually and
|
|
35
|
-
# flags it "identity_missing" rather than silently treating it as "ok"
|
|
35
|
+
# flags it "identity_missing" rather than silently treating it as "ok";
|
|
36
|
+
# "output_lower_bound" marks a Claude output fact whose tokens are only the
|
|
37
|
+
# observed lower bound of that message's output_tokens -- the reader's
|
|
38
|
+
# adopted row had no stop_reason, so the value is not guaranteed to be
|
|
39
|
+
# >= the true count).
|
|
36
40
|
# This is never left unset -- every Fact defaults to "ok" so export never
|
|
37
41
|
# emits a null source_quality.
|
|
38
|
-
SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing")
|
|
42
|
+
SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing", "output_lower_bound")
|
|
39
43
|
|
|
40
44
|
_BRACKET_SUFFIX = re.compile(r"\[[^\]]*\]$")
|
|
41
45
|
_DATE_SUFFIX = re.compile(r"@\d{6,8}$")
|
|
@@ -278,6 +278,14 @@ def _resolve_mode(rows: list, adopted: dict) -> str:
|
|
|
278
278
|
return adopted["mode"]
|
|
279
279
|
|
|
280
280
|
|
|
281
|
+
def _is_final(row: dict) -> bool:
|
|
282
|
+
"""Whether a row carries a real ``message.stop_reason`` -- a non-empty
|
|
283
|
+
``str``. Null, missing, ``False``, ``""`` and non-string values all
|
|
284
|
+
fail to qualify."""
|
|
285
|
+
stop_reason = row["stop_reason"]
|
|
286
|
+
return isinstance(stop_reason, str) and bool(stop_reason)
|
|
287
|
+
|
|
288
|
+
|
|
281
289
|
def _select_adopted_row(rows: list) -> dict:
|
|
282
290
|
"""Pick the row within one dedup group whose usage gets emitted.
|
|
283
291
|
|
|
@@ -302,8 +310,7 @@ def _select_adopted_row(rows: list) -> dict:
|
|
|
302
310
|
"""
|
|
303
311
|
adopted_idx = None
|
|
304
312
|
for idx in range(len(rows) - 1, -1, -1):
|
|
305
|
-
|
|
306
|
-
if isinstance(stop_reason, str) and stop_reason:
|
|
313
|
+
if _is_final(rows[idx]):
|
|
307
314
|
adopted_idx = idx
|
|
308
315
|
break
|
|
309
316
|
|
|
@@ -366,6 +373,19 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
|
|
|
366
373
|
the group has exactly one concrete mode elsewhere, that concrete mode
|
|
367
374
|
is emitted instead (see its docstring).
|
|
368
375
|
|
|
376
|
+
If the adopted row itself is not final (no non-empty-string
|
|
377
|
+
``stop_reason`` -- see ``_is_final``), the group's ``output`` fact gets
|
|
378
|
+
``source_quality="output_lower_bound"``: that row's ``output_tokens``
|
|
379
|
+
is a streaming head's value, observed to be a lower bound on the
|
|
380
|
+
message's real output count (subagent transcripts often end a
|
|
381
|
+
tool_use message without ever writing its final row). This keys off
|
|
382
|
+
the *adopted* row, not "does the group contain a final row", so a
|
|
383
|
+
final row overridden by a later, differing non-final row is flagged
|
|
384
|
+
too. The group's input-side facts stay ``"ok"`` (those fields don't
|
|
385
|
+
grow while streaming), and token amounts are emitted unchanged --
|
|
386
|
+
this only marks the caveat. Identity-missing rows never form a group
|
|
387
|
+
and keep ``"identity_missing"``.
|
|
388
|
+
|
|
369
389
|
``occurred_at_utc`` on the emitted fact is the *adopted* row's own
|
|
370
390
|
timestamp (not the group's first-seen timestamp): the adopted row is
|
|
371
391
|
what determines the actual token counts, and pricing/month-bucketing
|
|
@@ -493,6 +513,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
|
|
|
493
513
|
if marker in standalone_rows:
|
|
494
514
|
row = standalone_rows[marker]
|
|
495
515
|
source_quality = "identity_missing"
|
|
516
|
+
output_source_quality = source_quality
|
|
496
517
|
dedup_units.append(
|
|
497
518
|
ClaudeDedupUnit(
|
|
498
519
|
session_id=row["sid"],
|
|
@@ -544,6 +565,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
|
|
|
544
565
|
"tokens": adopted["tokens"],
|
|
545
566
|
}
|
|
546
567
|
source_quality = "ok"
|
|
568
|
+
output_source_quality = "ok" if _is_final(adopted) else "output_lower_bound"
|
|
547
569
|
|
|
548
570
|
model_key = normalize_model_key(row["model_raw"])
|
|
549
571
|
for kind, amount in row["tokens"].items():
|
|
@@ -557,7 +579,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
|
|
|
557
579
|
token_kind=kind,
|
|
558
580
|
tokens=amount,
|
|
559
581
|
mode=row["mode"],
|
|
560
|
-
source_quality=source_quality,
|
|
582
|
+
source_quality=output_source_quality if kind == "output" else source_quality,
|
|
561
583
|
)
|
|
562
584
|
)
|
|
563
585
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: coding-agent-cost
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
|
|
5
5
|
Author: shiki-yusuke
|
|
6
6
|
License: MIT
|
|
@@ -232,6 +232,14 @@ network, never calls `gh`, and never resolves branches or PRs.
|
|
|
232
232
|
flagged, since the reader's own cross-check of real transcripts found
|
|
233
233
|
exactly that pattern in every observed duplicated group. See
|
|
234
234
|
`CHANGELOG.md`'s 0.2.0 entry.
|
|
235
|
+
If the row a group adopts has no `stop_reason` (common in subagent
|
|
236
|
+
transcripts, where a message ending in `tool_use` may never get its final
|
|
237
|
+
line), its `output_tokens` is only the streaming head's value: the
|
|
238
|
+
group's `output` fact is flagged `source_quality: "output_lower_bound"`
|
|
239
|
+
-- the observed lower bound of that message's output tokens, not
|
|
240
|
+
guaranteed to be >= the true count. The group's input-side facts stay
|
|
241
|
+
`"ok"`, and the token amount itself is unchanged (still priced and
|
|
242
|
+
included in rows/totals); the flag is a warning only.
|
|
235
243
|
When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
|
|
236
244
|
present in the log, it's used; otherwise the cache-write tokens are priced
|
|
237
245
|
at the 5-minute rate as an explicit **lower bound** and flagged
|
|
@@ -342,7 +350,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
342
350
|
```json
|
|
343
351
|
{
|
|
344
352
|
"protocol_version": "measure/v1",
|
|
345
|
-
"producer_version": "0.
|
|
353
|
+
"producer_version": "0.3.0",
|
|
346
354
|
"accounting_basis": "agent-cost-raw-total/v2",
|
|
347
355
|
"generated_at": "...",
|
|
348
356
|
"window": { "since": "...", "until": null },
|
|
@@ -363,7 +371,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
|
|
|
363
371
|
"duplicate_rows_skipped": 0,
|
|
364
372
|
"conflicting_duplicate_groups": 0,
|
|
365
373
|
"missing_dedup_identity_rows": 0,
|
|
366
|
-
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
|
|
374
|
+
"source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
|
|
367
375
|
}
|
|
368
376
|
}
|
|
369
377
|
```
|
|
@@ -395,8 +403,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
|
|
|
395
403
|
branch names -- only the fields needed to reproduce a cost estimate:
|
|
396
404
|
`occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
|
|
397
405
|
`token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
|
|
398
|
-
caveat about how that one fact was derived
|
|
399
|
-
`"first_event_delta"
|
|
406
|
+
caveat about how that one fact was derived: `"ok"`, Codex's
|
|
407
|
+
`"first_event_delta"`, or Claude's `"identity_missing"` /
|
|
408
|
+
`"output_lower_bound"` -- never null).
|
|
400
409
|
|
|
401
410
|
## License
|
|
402
411
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "coding-agent-cost"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""Tests for ``source_quality="output_lower_bound"`` (E0-a4 S1).
|
|
2
|
+
|
|
3
|
+
Spec: when the row a dedup group adopts (``_select_adopted_row``) has no
|
|
4
|
+
valid ``message.stop_reason`` (a non-empty ``str``), that group's
|
|
5
|
+
``token_kind="output"`` fact is flagged ``"output_lower_bound"`` -- its
|
|
6
|
+
``output_tokens`` is the streaming head's value, a lower bound on the true
|
|
7
|
+
count. Input-side facts from the same group stay ``"ok"``. The decision is
|
|
8
|
+
"is the *adopted* row final", not "does the group contain a final row", so
|
|
9
|
+
a final row followed by a differing non-final row (adoption switches to
|
|
10
|
+
the last row) is flagged too. Identity-missing standalone rows keep
|
|
11
|
+
``"identity_missing"``. Token amounts, adoption and conflict counting are
|
|
12
|
+
unchanged -- only the flag is added.
|
|
13
|
+
|
|
14
|
+
All fixtures are synthetic and written to ``tmp_path``.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
|
|
19
|
+
import pytest
|
|
20
|
+
|
|
21
|
+
from agent_cost import cli
|
|
22
|
+
from agent_cost.facts import SOURCE_QUALITY_VALUES
|
|
23
|
+
from agent_cost.readers.claude import parse_session_detailed
|
|
24
|
+
|
|
25
|
+
_ABSENT = object()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _event(
|
|
29
|
+
ts,
|
|
30
|
+
output,
|
|
31
|
+
*,
|
|
32
|
+
stop_reason=_ABSENT,
|
|
33
|
+
msg_id="msg-1",
|
|
34
|
+
req_id="req-1",
|
|
35
|
+
session_id="s-olb",
|
|
36
|
+
input_tokens=10,
|
|
37
|
+
cache_read=20,
|
|
38
|
+
cache_write_5m=30,
|
|
39
|
+
):
|
|
40
|
+
message = {
|
|
41
|
+
"id": msg_id,
|
|
42
|
+
"model": "claude-sonnet-5",
|
|
43
|
+
"usage": {
|
|
44
|
+
"input_tokens": input_tokens,
|
|
45
|
+
"cache_read_input_tokens": cache_read,
|
|
46
|
+
"cache_creation_input_tokens": cache_write_5m,
|
|
47
|
+
"cache_creation": {"ephemeral_5m_input_tokens": cache_write_5m, "ephemeral_1h_input_tokens": 0},
|
|
48
|
+
"output_tokens": output,
|
|
49
|
+
},
|
|
50
|
+
}
|
|
51
|
+
if msg_id is None:
|
|
52
|
+
del message["id"]
|
|
53
|
+
if stop_reason is not _ABSENT:
|
|
54
|
+
message["stop_reason"] = stop_reason
|
|
55
|
+
event = {
|
|
56
|
+
"type": "assistant",
|
|
57
|
+
"timestamp": ts,
|
|
58
|
+
"sessionId": session_id,
|
|
59
|
+
"requestId": req_id,
|
|
60
|
+
"message": message,
|
|
61
|
+
}
|
|
62
|
+
if req_id is None:
|
|
63
|
+
del event["requestId"]
|
|
64
|
+
return event
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _write(path, events):
|
|
68
|
+
path.write_text("\n".join(json.dumps(e) for e in events) + "\n")
|
|
69
|
+
return path
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse(tmp_path, events):
|
|
73
|
+
return parse_session_detailed(_write(tmp_path / "session.jsonl", events))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _quality_by_kind(facts):
|
|
77
|
+
return {f.token_kind: f.source_quality for f in facts}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _tokens_by_kind(facts):
|
|
81
|
+
return {f.token_kind: f.tokens for f in facts}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
INPUT_KINDS = ("input_nocache", "cache_read", "cache_write_5m")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def test_output_lower_bound_is_last_source_quality_value():
|
|
88
|
+
assert SOURCE_QUALITY_VALUES[-1] == "output_lower_bound"
|
|
89
|
+
assert SOURCE_QUALITY_VALUES[:3] == ("ok", "first_event_delta", "identity_missing")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_group_without_final_row_flags_only_output(tmp_path):
|
|
93
|
+
"""AC1: no row in the group has a stop_reason -> adopted = last row,
|
|
94
|
+
its output fact is output_lower_bound, input-side facts stay ok."""
|
|
95
|
+
result = _parse(
|
|
96
|
+
tmp_path,
|
|
97
|
+
[
|
|
98
|
+
_event("2026-10-01T00:00:00Z", 1, stop_reason=None),
|
|
99
|
+
_event("2026-10-01T00:00:01Z", 1),
|
|
100
|
+
],
|
|
101
|
+
)
|
|
102
|
+
quality = _quality_by_kind(result.facts)
|
|
103
|
+
assert quality["output"] == "output_lower_bound"
|
|
104
|
+
for kind in INPUT_KINDS:
|
|
105
|
+
assert quality[kind] == "ok"
|
|
106
|
+
assert _tokens_by_kind(result.facts) == {"input_nocache": 10, "cache_read": 20, "cache_write_5m": 30, "output": 1}
|
|
107
|
+
assert result.duplicate_rows_skipped == 1
|
|
108
|
+
assert result.conflicting_duplicate_groups == 0
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def test_single_row_group_without_final_flags_output(tmp_path):
|
|
112
|
+
result = _parse(tmp_path, [_event("2026-10-01T00:00:00Z", 4)])
|
|
113
|
+
quality = _quality_by_kind(result.facts)
|
|
114
|
+
assert quality["output"] == "output_lower_bound"
|
|
115
|
+
for kind in INPUT_KINDS:
|
|
116
|
+
assert quality[kind] == "ok"
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def test_group_ending_in_final_row_is_all_ok(tmp_path):
|
|
120
|
+
"""AC2: the last row carries a stop_reason -> every fact is ok."""
|
|
121
|
+
result = _parse(
|
|
122
|
+
tmp_path,
|
|
123
|
+
[
|
|
124
|
+
_event("2026-10-01T00:00:00Z", 1),
|
|
125
|
+
_event("2026-10-01T00:00:01Z", 2),
|
|
126
|
+
_event("2026-10-01T00:00:02Z", 57, stop_reason="tool_use"),
|
|
127
|
+
],
|
|
128
|
+
)
|
|
129
|
+
assert {f.source_quality for f in result.facts} == {"ok"}
|
|
130
|
+
assert _tokens_by_kind(result.facts)["output"] == 57
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_final_row_followed_by_differing_row_flags_output(tmp_path):
|
|
134
|
+
"""AC2b: a final row followed by a different non-final row switches
|
|
135
|
+
adoption to the last row (no stop_reason) -> output_lower_bound."""
|
|
136
|
+
result = _parse(
|
|
137
|
+
tmp_path,
|
|
138
|
+
[
|
|
139
|
+
_event("2026-10-01T00:00:00Z", 5, stop_reason="end_turn"),
|
|
140
|
+
_event("2026-10-01T00:00:01Z", 8),
|
|
141
|
+
],
|
|
142
|
+
)
|
|
143
|
+
quality = _quality_by_kind(result.facts)
|
|
144
|
+
assert quality["output"] == "output_lower_bound"
|
|
145
|
+
for kind in INPUT_KINDS:
|
|
146
|
+
assert quality[kind] == "ok"
|
|
147
|
+
assert _tokens_by_kind(result.facts)["output"] == 8
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def test_final_row_followed_by_identical_row_stays_ok(tmp_path):
|
|
151
|
+
"""AC2c: a final row followed by an identical (non-final) row keeps the
|
|
152
|
+
final row adopted -> ok, even though the group's last row isn't final."""
|
|
153
|
+
result = _parse(
|
|
154
|
+
tmp_path,
|
|
155
|
+
[
|
|
156
|
+
_event("2026-10-01T00:00:00Z", 5, stop_reason="end_turn"),
|
|
157
|
+
_event("2026-10-01T00:00:01Z", 5),
|
|
158
|
+
],
|
|
159
|
+
)
|
|
160
|
+
assert {f.source_quality for f in result.facts} == {"ok"}
|
|
161
|
+
assert _tokens_by_kind(result.facts)["output"] == 5
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
@pytest.mark.parametrize("stop_reason", ["", None, 0, 1, 1.5, False, ["end_turn"], _ABSENT])
|
|
165
|
+
def test_invalid_stop_reason_only_group_flags_output(tmp_path, stop_reason):
|
|
166
|
+
"""AC2c: stop_reason "" / None / numbers / non-str only -> not final."""
|
|
167
|
+
result = _parse(
|
|
168
|
+
tmp_path,
|
|
169
|
+
[
|
|
170
|
+
_event("2026-10-01T00:00:00Z", 3, stop_reason=stop_reason),
|
|
171
|
+
_event("2026-10-01T00:00:01Z", 6, stop_reason=stop_reason),
|
|
172
|
+
],
|
|
173
|
+
)
|
|
174
|
+
quality = _quality_by_kind(result.facts)
|
|
175
|
+
assert quality["output"] == "output_lower_bound"
|
|
176
|
+
for kind in INPUT_KINDS:
|
|
177
|
+
assert quality[kind] == "ok"
|
|
178
|
+
assert _tokens_by_kind(result.facts)["output"] == 6
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def test_final_row_with_zero_output_is_ok(tmp_path):
|
|
182
|
+
"""AC2c: a final row with output 0 emits no output fact; the rest is ok."""
|
|
183
|
+
result = _parse(
|
|
184
|
+
tmp_path,
|
|
185
|
+
[
|
|
186
|
+
_event("2026-10-01T00:00:00Z", 0),
|
|
187
|
+
_event("2026-10-01T00:00:01Z", 0, stop_reason="end_turn"),
|
|
188
|
+
],
|
|
189
|
+
)
|
|
190
|
+
assert "output" not in _tokens_by_kind(result.facts)
|
|
191
|
+
assert {f.source_quality for f in result.facts} == {"ok"}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
@pytest.mark.parametrize("missing", ["msg_id", "req_id"])
|
|
195
|
+
def test_identity_missing_row_stays_identity_missing(tmp_path, missing):
|
|
196
|
+
"""AC2d: a standalone row without a full id pair is not a group, so it
|
|
197
|
+
keeps identity_missing on every fact (including output)."""
|
|
198
|
+
kwargs = {missing: None}
|
|
199
|
+
result = _parse(tmp_path, [_event("2026-10-01T00:00:00Z", 9, **kwargs)])
|
|
200
|
+
assert result.missing_dedup_identity_rows == 1
|
|
201
|
+
assert "output" in _tokens_by_kind(result.facts)
|
|
202
|
+
assert {f.source_quality for f in result.facts} == {"identity_missing"}
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def test_flag_is_per_group(tmp_path):
|
|
206
|
+
"""One group without a final row, one with: only the former's output is flagged."""
|
|
207
|
+
result = _parse(
|
|
208
|
+
tmp_path,
|
|
209
|
+
[
|
|
210
|
+
_event("2026-10-01T00:00:00Z", 2, msg_id="msg-a", req_id="req-a"),
|
|
211
|
+
_event("2026-10-01T00:00:01Z", 40, msg_id="msg-b", req_id="req-b", stop_reason="end_turn"),
|
|
212
|
+
],
|
|
213
|
+
)
|
|
214
|
+
output_facts = [f for f in result.facts if f.token_kind == "output"]
|
|
215
|
+
assert [(f.tokens, f.source_quality) for f in output_facts] == [(2, "output_lower_bound"), (40, "ok")]
|
|
216
|
+
assert all(f.source_quality == "ok" for f in result.facts if f.token_kind != "output")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
# ---------------------------------------------------------------------------
|
|
220
|
+
# CLI (AC3): measure's data_quality.source_quality always carries the key.
|
|
221
|
+
# ---------------------------------------------------------------------------
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _setup_claude_session(tmp_path, monkeypatch, events):
|
|
225
|
+
claude_home = tmp_path / "claude_home"
|
|
226
|
+
codex_home = tmp_path / "codex_home"
|
|
227
|
+
project_dir = claude_home / "projects" / "olb"
|
|
228
|
+
project_dir.mkdir(parents=True)
|
|
229
|
+
codex_home.mkdir()
|
|
230
|
+
monkeypatch.setenv("CLAUDE_HOME", str(claude_home))
|
|
231
|
+
monkeypatch.setenv("CODEX_HOME", str(codex_home))
|
|
232
|
+
monkeypatch.delenv("AGENT_COST_CONFIG", raising=False)
|
|
233
|
+
_write(project_dir / "session.jsonl", events)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _measure_source_quality(capsys):
|
|
237
|
+
rc = cli.main(["measure", "--session-id", "s-olb", "--since", "2026-09-01"])
|
|
238
|
+
assert rc == 0
|
|
239
|
+
return json.loads(capsys.readouterr().out)["data_quality"]["source_quality"]
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def test_measure_counts_output_lower_bound(tmp_path, monkeypatch, capsys):
|
|
243
|
+
_setup_claude_session(
|
|
244
|
+
tmp_path,
|
|
245
|
+
monkeypatch,
|
|
246
|
+
[
|
|
247
|
+
_event("2026-10-01T00:00:00Z", 1, msg_id="msg-a", req_id="req-a"),
|
|
248
|
+
_event("2026-10-01T00:00:01Z", 30, msg_id="msg-b", req_id="req-b", stop_reason="end_turn"),
|
|
249
|
+
],
|
|
250
|
+
)
|
|
251
|
+
sq = _measure_source_quality(capsys)
|
|
252
|
+
assert set(sq.keys()) == set(SOURCE_QUALITY_VALUES)
|
|
253
|
+
assert sq["output_lower_bound"] == 1
|
|
254
|
+
assert sq["ok"] == 7
|
|
255
|
+
assert sq["identity_missing"] == 0
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def test_measure_reports_zero_output_lower_bound_when_absent(tmp_path, monkeypatch, capsys):
|
|
259
|
+
_setup_claude_session(
|
|
260
|
+
tmp_path,
|
|
261
|
+
monkeypatch,
|
|
262
|
+
[_event("2026-10-01T00:00:00Z", 30, stop_reason="end_turn")],
|
|
263
|
+
)
|
|
264
|
+
sq = _measure_source_quality(capsys)
|
|
265
|
+
assert sq == {"ok": 4, "first_event_delta": 0, "identity_missing": 0, "output_lower_bound": 0}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
"""Spec: pyproject.toml's [project].version and agent_cost.__version__ must
|
|
2
|
-
agree, and both must read "0.
|
|
3
|
-
|
|
4
|
-
implementation would bump one file but not the other, or
|
|
5
|
-
entirely.
|
|
2
|
+
agree, and both must read "0.3.0" for this change (the output_lower_bound
|
|
3
|
+
source_quality value is an additive vocabulary change, so it bumps the minor
|
|
4
|
+
version). A broken implementation would bump one file but not the other, or
|
|
5
|
+
forget the bump entirely.
|
|
6
6
|
|
|
7
7
|
Uses a plain regex instead of tomllib/tomli: this repo is dependency-free
|
|
8
8
|
by design (pyproject.toml's own dependencies = []) and tomllib is
|
|
@@ -29,5 +29,5 @@ def test_pyproject_version_matches_package_version():
|
|
|
29
29
|
|
|
30
30
|
|
|
31
31
|
def test_version_is_0_2_0():
|
|
32
|
-
assert agent_cost.__version__ == "0.
|
|
33
|
-
assert _pyproject_version() == "0.
|
|
32
|
+
assert agent_cost.__version__ == "0.3.0"
|
|
33
|
+
assert _pyproject_version() == "0.3.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|