coding-agent-cost 0.2.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {coding_agent_cost-0.2.2/coding_agent_cost.egg-info → coding_agent_cost-0.3.0}/PKG-INFO +14 -5
  2. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/README.md +13 -4
  3. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/__init__.py +1 -1
  4. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/facts.py +6 -2
  5. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/claude.py +25 -3
  6. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0/coding_agent_cost.egg-info}/PKG-INFO +14 -5
  7. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/SOURCES.txt +1 -0
  8. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/pyproject.toml +1 -1
  9. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_cli.py +1 -0
  10. coding_agent_cost-0.3.0/tests/test_reader_output_lower_bound.py +265 -0
  11. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_version_consistency.py +6 -6
  12. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/LICENSE +0 -0
  13. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/aggregate.py +0 -0
  14. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/cli.py +0 -0
  15. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/config.py +0 -0
  16. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/rates.json +0 -0
  17. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/rates.py +0 -0
  18. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/__init__.py +0 -0
  19. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/readers/codex.py +0 -0
  20. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/agent_cost/renderers.py +0 -0
  21. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/dependency_links.txt +0 -0
  22. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/entry_points.txt +0 -0
  23. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/coding_agent_cost.egg-info/top_level.txt +0 -0
  24. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/setup.cfg +0 -0
  25. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_aggregate.py +0 -0
  26. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_astra_pricing.py +0 -0
  27. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_e0a_fable_5_1_and_sonnet_5_correction.py +0 -0
  28. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_e0a_review3_hardening.py +0 -0
  29. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_facts.py +0 -0
  30. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_measure_v1_contract.py +0 -0
  31. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_opus_5_5_rates.py +0 -0
  32. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_rates.py +0 -0
  33. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_claude.py +0 -0
  34. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_codex.py +0 -0
  35. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_reader_dedup.py +0 -0
  36. {coding_agent_cost-0.2.2 → coding_agent_cost-0.3.0}/tests/test_sonnet_5_5_rates.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: coding-agent-cost
3
- Version: 0.2.2
3
+ Version: 0.3.0
4
4
  Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
5
5
  Author: shiki-yusuke
6
6
  License: MIT
@@ -232,6 +232,14 @@ network, never calls `gh`, and never resolves branches or PRs.
232
232
  flagged, since the reader's own cross-check of real transcripts found
233
233
  exactly that pattern in every observed duplicated group. See
234
234
  `CHANGELOG.md`'s 0.2.0 entry.
235
+ If the row a group adopts has no `stop_reason` (common in subagent
236
+ transcripts, where a message ending in `tool_use` may never get its final
237
+ line), its `output_tokens` is only the streaming head's value: the
238
+ group's `output` fact is flagged `source_quality: "output_lower_bound"`
239
+ -- the observed lower bound of that message's output tokens, not
240
+ guaranteed to be >= the true count. The group's input-side facts stay
241
+ `"ok"`, and the token amount itself is unchanged (still priced and
242
+ included in rows/totals); the flag is a warning only.
235
243
  When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
236
244
  present in the log, it's used; otherwise the cache-write tokens are priced
237
245
  at the 5-minute rate as an explicit **lower bound** and flagged
@@ -342,7 +350,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
342
350
  ```json
343
351
  {
344
352
  "protocol_version": "measure/v1",
345
- "producer_version": "0.2.2",
353
+ "producer_version": "0.3.0",
346
354
  "accounting_basis": "agent-cost-raw-total/v2",
347
355
  "generated_at": "...",
348
356
  "window": { "since": "...", "until": null },
@@ -363,7 +371,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
363
371
  "duplicate_rows_skipped": 0,
364
372
  "conflicting_duplicate_groups": 0,
365
373
  "missing_dedup_identity_rows": 0,
366
- "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
374
+ "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
367
375
  }
368
376
  }
369
377
  ```
@@ -395,8 +403,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
395
403
  branch names -- only the fields needed to reproduce a cost estimate:
396
404
  `occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
397
405
  `token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
398
- caveat about how that one fact was derived, e.g. `"ok"` or Codex's
399
- `"first_event_delta"` -- never null).
406
+ caveat about how that one fact was derived: `"ok"`, Codex's
407
+ `"first_event_delta"`, or Claude's `"identity_missing"` /
408
+ `"output_lower_bound"` -- never null).
400
409
 
401
410
  ## License
402
411
 
@@ -206,6 +206,14 @@ network, never calls `gh`, and never resolves branches or PRs.
206
206
  flagged, since the reader's own cross-check of real transcripts found
207
207
  exactly that pattern in every observed duplicated group. See
208
208
  `CHANGELOG.md`'s 0.2.0 entry.
209
+ If the row a group adopts has no `stop_reason` (common in subagent
210
+ transcripts, where a message ending in `tool_use` may never get its final
211
+ line), its `output_tokens` is only the streaming head's value: the
212
+ group's `output` fact is flagged `source_quality: "output_lower_bound"`
213
+ -- the observed lower bound of that message's output tokens, not
214
+ guaranteed to be >= the true count. The group's input-side facts stay
215
+ `"ok"`, and the token amount itself is unchanged (still priced and
216
+ included in rows/totals); the flag is a warning only.
209
217
  When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
210
218
  present in the log, it's used; otherwise the cache-write tokens are priced
211
219
  at the 5-minute rate as an explicit **lower bound** and flagged
@@ -316,7 +324,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
316
324
  ```json
317
325
  {
318
326
  "protocol_version": "measure/v1",
319
- "producer_version": "0.2.2",
327
+ "producer_version": "0.3.0",
320
328
  "accounting_basis": "agent-cost-raw-total/v2",
321
329
  "generated_at": "...",
322
330
  "window": { "since": "...", "until": null },
@@ -337,7 +345,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
337
345
  "duplicate_rows_skipped": 0,
338
346
  "conflicting_duplicate_groups": 0,
339
347
  "missing_dedup_identity_rows": 0,
340
- "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
348
+ "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
341
349
  }
342
350
  }
343
351
  ```
@@ -369,8 +377,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
369
377
  branch names -- only the fields needed to reproduce a cost estimate:
370
378
  `occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
371
379
  `token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
372
- caveat about how that one fact was derived, e.g. `"ok"` or Codex's
373
- `"first_event_delta"` -- never null).
380
+ caveat about how that one fact was derived: `"ok"`, Codex's
381
+ `"first_event_delta"`, or Claude's `"identity_missing"` /
382
+ `"output_lower_bound"` -- never null).
374
383
 
375
384
  ## License
376
385
 
@@ -6,4 +6,4 @@ against a versioned rate catalog. Everything happens locally; there are no
6
6
  network calls.
7
7
  """
8
8
 
9
- __version__ = "0.2.2"
9
+ __version__ = "0.3.0"
@@ -32,10 +32,14 @@ MODES = ("fast", "normal", "unknown")
32
32
  # (e.g. Codex's first delta in a rollout is measured against an assumed
33
33
  # zero baseline; Claude's reader can't dedup a row that lacks a full
34
34
  # (message.id, requestId) pair, so it emits that row individually and
35
- # flags it "identity_missing" rather than silently treating it as "ok").
35
+ # flags it "identity_missing" rather than silently treating it as "ok";
36
+ # "output_lower_bound" marks a Claude output fact whose tokens are only the
37
+ # observed lower bound of that message's output_tokens -- the reader's
38
+ # adopted row had no stop_reason, so the value is not guaranteed to be
39
+ # >= the true count).
36
40
  # This is never left unset -- every Fact defaults to "ok" so export never
37
41
  # emits a null source_quality.
38
- SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing")
42
+ SOURCE_QUALITY_VALUES = ("ok", "first_event_delta", "identity_missing", "output_lower_bound")
39
43
 
40
44
  _BRACKET_SUFFIX = re.compile(r"\[[^\]]*\]$")
41
45
  _DATE_SUFFIX = re.compile(r"@\d{6,8}$")
@@ -278,6 +278,14 @@ def _resolve_mode(rows: list, adopted: dict) -> str:
278
278
  return adopted["mode"]
279
279
 
280
280
 
281
+ def _is_final(row: dict) -> bool:
282
+ """Whether a row carries a real ``message.stop_reason`` -- a non-empty
283
+ ``str``. Null, missing, ``False``, ``""`` and non-string values all
284
+ fail to qualify."""
285
+ stop_reason = row["stop_reason"]
286
+ return isinstance(stop_reason, str) and bool(stop_reason)
287
+
288
+
281
289
  def _select_adopted_row(rows: list) -> dict:
282
290
  """Pick the row within one dedup group whose usage gets emitted.
283
291
 
@@ -302,8 +310,7 @@ def _select_adopted_row(rows: list) -> dict:
302
310
  """
303
311
  adopted_idx = None
304
312
  for idx in range(len(rows) - 1, -1, -1):
305
- stop_reason = rows[idx]["stop_reason"]
306
- if isinstance(stop_reason, str) and stop_reason:
313
+ if _is_final(rows[idx]):
307
314
  adopted_idx = idx
308
315
  break
309
316
 
@@ -366,6 +373,19 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
366
373
  the group has exactly one concrete mode elsewhere, that concrete mode
367
374
  is emitted instead (see its docstring).
368
375
 
376
+ If the adopted row itself is not final (no non-empty-string
377
+ ``stop_reason`` -- see ``_is_final``), the group's ``output`` fact gets
378
+ ``source_quality="output_lower_bound"``: that row's ``output_tokens``
379
+ is a streaming head's value, observed to be a lower bound on the
380
+ message's real output count (subagent transcripts often end a
381
+ tool_use message without ever writing its final row). This keys off
382
+ the *adopted* row, not "does the group contain a final row", so a
383
+ final row overridden by a later, differing non-final row is flagged
384
+ too. The group's input-side facts stay ``"ok"`` (those fields don't
385
+ grow while streaming), and token amounts are emitted unchanged --
386
+ this only marks the caveat. Identity-missing rows never form a group
387
+ and keep ``"identity_missing"``.
388
+
369
389
  ``occurred_at_utc`` on the emitted fact is the *adopted* row's own
370
390
  timestamp (not the group's first-seen timestamp): the adopted row is
371
391
  what determines the actual token counts, and pricing/month-bucketing
@@ -493,6 +513,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
493
513
  if marker in standalone_rows:
494
514
  row = standalone_rows[marker]
495
515
  source_quality = "identity_missing"
516
+ output_source_quality = source_quality
496
517
  dedup_units.append(
497
518
  ClaudeDedupUnit(
498
519
  session_id=row["sid"],
@@ -544,6 +565,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
544
565
  "tokens": adopted["tokens"],
545
566
  }
546
567
  source_quality = "ok"
568
+ output_source_quality = "ok" if _is_final(adopted) else "output_lower_bound"
547
569
 
548
570
  model_key = normalize_model_key(row["model_raw"])
549
571
  for kind, amount in row["tokens"].items():
@@ -557,7 +579,7 @@ def parse_session_detailed(jsonl_path: Path) -> ClaudeParseResult:
557
579
  token_kind=kind,
558
580
  tokens=amount,
559
581
  mode=row["mode"],
560
- source_quality=source_quality,
582
+ source_quality=output_source_quality if kind == "output" else source_quality,
561
583
  )
562
584
  )
563
585
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: coding-agent-cost
3
- Version: 0.2.2
3
+ Version: 0.3.0
4
4
  Summary: Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs
5
5
  Author: shiki-yusuke
6
6
  License: MIT
@@ -232,6 +232,14 @@ network, never calls `gh`, and never resolves branches or PRs.
232
232
  flagged, since the reader's own cross-check of real transcripts found
233
233
  exactly that pattern in every observed duplicated group. See
234
234
  `CHANGELOG.md`'s 0.2.0 entry.
235
+ If the row a group adopts has no `stop_reason` (common in subagent
236
+ transcripts, where a message ending in `tool_use` may never get its final
237
+ line), its `output_tokens` is only the streaming head's value: the
238
+ group's `output` fact is flagged `source_quality: "output_lower_bound"`
239
+ -- the observed lower bound of that message's output tokens, not
240
+ guaranteed to be >= the true count. The group's input-side facts stay
241
+ `"ok"`, and the token amount itself is unchanged (still priced and
242
+ included in rows/totals); the flag is a warning only.
235
243
  When Anthropic's prompt-cache TTL breakdown (5-minute vs 1-hour writes) is
236
244
  present in the log, it's used; otherwise the cache-write tokens are priced
237
245
  at the 5-minute rate as an explicit **lower bound** and flagged
@@ -342,7 +350,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
342
350
  ```json
343
351
  {
344
352
  "protocol_version": "measure/v1",
345
- "producer_version": "0.2.2",
353
+ "producer_version": "0.3.0",
346
354
  "accounting_basis": "agent-cost-raw-total/v2",
347
355
  "generated_at": "...",
348
356
  "window": { "since": "...", "until": null },
@@ -363,7 +371,7 @@ agent-cost measure --session-id <id> [--session-id <id> ...] \
363
371
  "duplicate_rows_skipped": 0,
364
372
  "conflicting_duplicate_groups": 0,
365
373
  "missing_dedup_identity_rows": 0,
366
- "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0 }
374
+ "source_quality": { "ok": 41, "first_event_delta": 2, "identity_missing": 0, "output_lower_bound": 1 }
367
375
  }
368
376
  }
369
377
  ```
@@ -395,8 +403,9 @@ includes absolute file paths, rollout paths, prompt/message content, or git
395
403
  branch names -- only the fields needed to reproduce a cost estimate:
396
404
  `occurred_at_utc`, `agent`, `session_id`, `model_raw`, `model_key`,
397
405
  `token_kind`, `tokens`, `mode`, and `source_quality` (a fixed-vocabulary
398
- caveat about how that one fact was derived, e.g. `"ok"` or Codex's
399
- `"first_event_delta"` -- never null).
406
+ caveat about how that one fact was derived: `"ok"`, Codex's
407
+ `"first_event_delta"`, or Claude's `"identity_missing"` /
408
+ `"output_lower_bound"` -- never null).
400
409
 
401
410
  ## License
402
411
 
@@ -29,5 +29,6 @@ tests/test_rates.py
29
29
  tests/test_reader_claude.py
30
30
  tests/test_reader_codex.py
31
31
  tests/test_reader_dedup.py
32
+ tests/test_reader_output_lower_bound.py
32
33
  tests/test_sonnet_5_5_rates.py
33
34
  tests/test_version_consistency.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "coding-agent-cost"
7
- version = "0.2.2"
7
+ version = "0.3.0"
8
8
  description = "Estimate AI coding agent (Claude Code / Codex CLI) token usage and cost from local logs"
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -471,6 +471,7 @@ def test_measure_unknown_session_id_exits_zero_with_empty_result(tmp_path, monke
471
471
  "ok": 0,
472
472
  "first_event_delta": 0,
473
473
  "identity_missing": 0,
474
+ "output_lower_bound": 0,
474
475
  }
475
476
 
476
477
 
@@ -0,0 +1,265 @@
1
+ """Tests for ``source_quality="output_lower_bound"`` (E0-a4 S1).
2
+
3
+ Spec: when the row a dedup group adopts (``_select_adopted_row``) has no
4
+ valid ``message.stop_reason`` (a non-empty ``str``), that group's
5
+ ``token_kind="output"`` fact is flagged ``"output_lower_bound"`` -- its
6
+ ``output_tokens`` is the streaming head's value, a lower bound on the true
7
+ count. Input-side facts from the same group stay ``"ok"``. The decision is
8
+ "is the *adopted* row final", not "does the group contain a final row", so
9
+ a final row followed by a differing non-final row (adoption switches to
10
+ the last row) is flagged too. Identity-missing standalone rows keep
11
+ ``"identity_missing"``. Token amounts, adoption and conflict counting are
12
+ unchanged -- only the flag is added.
13
+
14
+ All fixtures are synthetic and written to ``tmp_path``.
15
+ """
16
+
17
+ import json
18
+
19
+ import pytest
20
+
21
+ from agent_cost import cli
22
+ from agent_cost.facts import SOURCE_QUALITY_VALUES
23
+ from agent_cost.readers.claude import parse_session_detailed
24
+
25
+ _ABSENT = object()
26
+
27
+
28
+ def _event(
29
+ ts,
30
+ output,
31
+ *,
32
+ stop_reason=_ABSENT,
33
+ msg_id="msg-1",
34
+ req_id="req-1",
35
+ session_id="s-olb",
36
+ input_tokens=10,
37
+ cache_read=20,
38
+ cache_write_5m=30,
39
+ ):
40
+ message = {
41
+ "id": msg_id,
42
+ "model": "claude-sonnet-5",
43
+ "usage": {
44
+ "input_tokens": input_tokens,
45
+ "cache_read_input_tokens": cache_read,
46
+ "cache_creation_input_tokens": cache_write_5m,
47
+ "cache_creation": {"ephemeral_5m_input_tokens": cache_write_5m, "ephemeral_1h_input_tokens": 0},
48
+ "output_tokens": output,
49
+ },
50
+ }
51
+ if msg_id is None:
52
+ del message["id"]
53
+ if stop_reason is not _ABSENT:
54
+ message["stop_reason"] = stop_reason
55
+ event = {
56
+ "type": "assistant",
57
+ "timestamp": ts,
58
+ "sessionId": session_id,
59
+ "requestId": req_id,
60
+ "message": message,
61
+ }
62
+ if req_id is None:
63
+ del event["requestId"]
64
+ return event
65
+
66
+
67
+ def _write(path, events):
68
+ path.write_text("\n".join(json.dumps(e) for e in events) + "\n")
69
+ return path
70
+
71
+
72
+ def _parse(tmp_path, events):
73
+ return parse_session_detailed(_write(tmp_path / "session.jsonl", events))
74
+
75
+
76
+ def _quality_by_kind(facts):
77
+ return {f.token_kind: f.source_quality for f in facts}
78
+
79
+
80
+ def _tokens_by_kind(facts):
81
+ return {f.token_kind: f.tokens for f in facts}
82
+
83
+
84
+ INPUT_KINDS = ("input_nocache", "cache_read", "cache_write_5m")
85
+
86
+
87
+ def test_output_lower_bound_is_last_source_quality_value():
88
+ assert SOURCE_QUALITY_VALUES[-1] == "output_lower_bound"
89
+ assert SOURCE_QUALITY_VALUES[:3] == ("ok", "first_event_delta", "identity_missing")
90
+
91
+
92
+ def test_group_without_final_row_flags_only_output(tmp_path):
93
+ """AC1: no row in the group has a stop_reason -> adopted = last row,
94
+ its output fact is output_lower_bound, input-side facts stay ok."""
95
+ result = _parse(
96
+ tmp_path,
97
+ [
98
+ _event("2026-10-01T00:00:00Z", 1, stop_reason=None),
99
+ _event("2026-10-01T00:00:01Z", 1),
100
+ ],
101
+ )
102
+ quality = _quality_by_kind(result.facts)
103
+ assert quality["output"] == "output_lower_bound"
104
+ for kind in INPUT_KINDS:
105
+ assert quality[kind] == "ok"
106
+ assert _tokens_by_kind(result.facts) == {"input_nocache": 10, "cache_read": 20, "cache_write_5m": 30, "output": 1}
107
+ assert result.duplicate_rows_skipped == 1
108
+ assert result.conflicting_duplicate_groups == 0
109
+
110
+
111
+ def test_single_row_group_without_final_flags_output(tmp_path):
112
+ result = _parse(tmp_path, [_event("2026-10-01T00:00:00Z", 4)])
113
+ quality = _quality_by_kind(result.facts)
114
+ assert quality["output"] == "output_lower_bound"
115
+ for kind in INPUT_KINDS:
116
+ assert quality[kind] == "ok"
117
+
118
+
119
+ def test_group_ending_in_final_row_is_all_ok(tmp_path):
120
+ """AC2: the last row carries a stop_reason -> every fact is ok."""
121
+ result = _parse(
122
+ tmp_path,
123
+ [
124
+ _event("2026-10-01T00:00:00Z", 1),
125
+ _event("2026-10-01T00:00:01Z", 2),
126
+ _event("2026-10-01T00:00:02Z", 57, stop_reason="tool_use"),
127
+ ],
128
+ )
129
+ assert {f.source_quality for f in result.facts} == {"ok"}
130
+ assert _tokens_by_kind(result.facts)["output"] == 57
131
+
132
+
133
+ def test_final_row_followed_by_differing_row_flags_output(tmp_path):
134
+ """AC2b: a final row followed by a different non-final row switches
135
+ adoption to the last row (no stop_reason) -> output_lower_bound."""
136
+ result = _parse(
137
+ tmp_path,
138
+ [
139
+ _event("2026-10-01T00:00:00Z", 5, stop_reason="end_turn"),
140
+ _event("2026-10-01T00:00:01Z", 8),
141
+ ],
142
+ )
143
+ quality = _quality_by_kind(result.facts)
144
+ assert quality["output"] == "output_lower_bound"
145
+ for kind in INPUT_KINDS:
146
+ assert quality[kind] == "ok"
147
+ assert _tokens_by_kind(result.facts)["output"] == 8
148
+
149
+
150
+ def test_final_row_followed_by_identical_row_stays_ok(tmp_path):
151
+ """AC2c: a final row followed by an identical (non-final) row keeps the
152
+ final row adopted -> ok, even though the group's last row isn't final."""
153
+ result = _parse(
154
+ tmp_path,
155
+ [
156
+ _event("2026-10-01T00:00:00Z", 5, stop_reason="end_turn"),
157
+ _event("2026-10-01T00:00:01Z", 5),
158
+ ],
159
+ )
160
+ assert {f.source_quality for f in result.facts} == {"ok"}
161
+ assert _tokens_by_kind(result.facts)["output"] == 5
162
+
163
+
164
+ @pytest.mark.parametrize("stop_reason", ["", None, 0, 1, 1.5, False, ["end_turn"], _ABSENT])
165
+ def test_invalid_stop_reason_only_group_flags_output(tmp_path, stop_reason):
166
+ """AC2c: stop_reason "" / None / numbers / non-str only -> not final."""
167
+ result = _parse(
168
+ tmp_path,
169
+ [
170
+ _event("2026-10-01T00:00:00Z", 3, stop_reason=stop_reason),
171
+ _event("2026-10-01T00:00:01Z", 6, stop_reason=stop_reason),
172
+ ],
173
+ )
174
+ quality = _quality_by_kind(result.facts)
175
+ assert quality["output"] == "output_lower_bound"
176
+ for kind in INPUT_KINDS:
177
+ assert quality[kind] == "ok"
178
+ assert _tokens_by_kind(result.facts)["output"] == 6
179
+
180
+
181
+ def test_final_row_with_zero_output_is_ok(tmp_path):
182
+ """AC2c: a final row with output 0 emits no output fact; the rest is ok."""
183
+ result = _parse(
184
+ tmp_path,
185
+ [
186
+ _event("2026-10-01T00:00:00Z", 0),
187
+ _event("2026-10-01T00:00:01Z", 0, stop_reason="end_turn"),
188
+ ],
189
+ )
190
+ assert "output" not in _tokens_by_kind(result.facts)
191
+ assert {f.source_quality for f in result.facts} == {"ok"}
192
+
193
+
194
+ @pytest.mark.parametrize("missing", ["msg_id", "req_id"])
195
+ def test_identity_missing_row_stays_identity_missing(tmp_path, missing):
196
+ """AC2d: a standalone row without a full id pair is not a group, so it
197
+ keeps identity_missing on every fact (including output)."""
198
+ kwargs = {missing: None}
199
+ result = _parse(tmp_path, [_event("2026-10-01T00:00:00Z", 9, **kwargs)])
200
+ assert result.missing_dedup_identity_rows == 1
201
+ assert "output" in _tokens_by_kind(result.facts)
202
+ assert {f.source_quality for f in result.facts} == {"identity_missing"}
203
+
204
+
205
+ def test_flag_is_per_group(tmp_path):
206
+ """One group without a final row, one with: only the former's output is flagged."""
207
+ result = _parse(
208
+ tmp_path,
209
+ [
210
+ _event("2026-10-01T00:00:00Z", 2, msg_id="msg-a", req_id="req-a"),
211
+ _event("2026-10-01T00:00:01Z", 40, msg_id="msg-b", req_id="req-b", stop_reason="end_turn"),
212
+ ],
213
+ )
214
+ output_facts = [f for f in result.facts if f.token_kind == "output"]
215
+ assert [(f.tokens, f.source_quality) for f in output_facts] == [(2, "output_lower_bound"), (40, "ok")]
216
+ assert all(f.source_quality == "ok" for f in result.facts if f.token_kind != "output")
217
+
218
+
219
+ # ---------------------------------------------------------------------------
220
+ # CLI (AC3): measure's data_quality.source_quality always carries the key.
221
+ # ---------------------------------------------------------------------------
222
+
223
+
224
+ def _setup_claude_session(tmp_path, monkeypatch, events):
225
+ claude_home = tmp_path / "claude_home"
226
+ codex_home = tmp_path / "codex_home"
227
+ project_dir = claude_home / "projects" / "olb"
228
+ project_dir.mkdir(parents=True)
229
+ codex_home.mkdir()
230
+ monkeypatch.setenv("CLAUDE_HOME", str(claude_home))
231
+ monkeypatch.setenv("CODEX_HOME", str(codex_home))
232
+ monkeypatch.delenv("AGENT_COST_CONFIG", raising=False)
233
+ _write(project_dir / "session.jsonl", events)
234
+
235
+
236
+ def _measure_source_quality(capsys):
237
+ rc = cli.main(["measure", "--session-id", "s-olb", "--since", "2026-09-01"])
238
+ assert rc == 0
239
+ return json.loads(capsys.readouterr().out)["data_quality"]["source_quality"]
240
+
241
+
242
+ def test_measure_counts_output_lower_bound(tmp_path, monkeypatch, capsys):
243
+ _setup_claude_session(
244
+ tmp_path,
245
+ monkeypatch,
246
+ [
247
+ _event("2026-10-01T00:00:00Z", 1, msg_id="msg-a", req_id="req-a"),
248
+ _event("2026-10-01T00:00:01Z", 30, msg_id="msg-b", req_id="req-b", stop_reason="end_turn"),
249
+ ],
250
+ )
251
+ sq = _measure_source_quality(capsys)
252
+ assert set(sq.keys()) == set(SOURCE_QUALITY_VALUES)
253
+ assert sq["output_lower_bound"] == 1
254
+ assert sq["ok"] == 7
255
+ assert sq["identity_missing"] == 0
256
+
257
+
258
+ def test_measure_reports_zero_output_lower_bound_when_absent(tmp_path, monkeypatch, capsys):
259
+ _setup_claude_session(
260
+ tmp_path,
261
+ monkeypatch,
262
+ [_event("2026-10-01T00:00:00Z", 30, stop_reason="end_turn")],
263
+ )
264
+ sq = _measure_source_quality(capsys)
265
+ assert sq == {"ok": 4, "first_event_delta": 0, "identity_missing": 0, "output_lower_bound": 0}
@@ -1,8 +1,8 @@
1
1
  """Spec: pyproject.toml's [project].version and agent_cost.__version__ must
2
- agree, and both must read "0.2.2" for this change (the Sonnet 5.5 catalog-only release bumps
3
- the package version, as the Opus 5.5 catalog release did for 0.2.1). A broken
4
- implementation would bump one file but not the other, or forget the bump
5
- entirely.
2
+ agree, and both must read "0.3.0" for this change (the output_lower_bound
3
+ source_quality value is an additive vocabulary change, so it bumps the minor
4
+ version). A broken implementation would bump one file but not the other, or
5
+ forget the bump entirely.
6
6
 
7
7
  Uses a plain regex instead of tomllib/tomli: this repo is dependency-free
8
8
  by design (pyproject.toml's own dependencies = []) and tomllib is
@@ -29,5 +29,5 @@ def test_pyproject_version_matches_package_version():
29
29
 
30
30
 
31
31
  def test_version_is_0_2_0():
32
- assert agent_cost.__version__ == "0.2.2"
33
- assert _pyproject_version() == "0.2.2"
32
+ assert agent_cost.__version__ == "0.3.0"
33
+ assert _pyproject_version() == "0.3.0"