cli-consumption 0.3.2__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/.gitignore +6 -0
  2. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/CHANGELOG.md +52 -1
  3. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/PKG-INFO +28 -4
  4. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/README.md +25 -3
  5. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/pyproject.toml +5 -2
  6. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/_shared.py +40 -12
  7. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/codex.py +101 -43
  8. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/continue_cli.py +40 -13
  9. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/opencode.py +4 -1
  10. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/registry.py +2 -1
  11. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/api.py +137 -17
  12. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/cli.py +414 -6
  13. cli_consumption-0.4.0/src/cli_consumption/dashboard.py +813 -0
  14. cli_consumption-0.4.0/src/cli_consumption/dashboard_react.css +2 -0
  15. cli_consumption-0.4.0/src/cli_consumption/dashboard_react.js +9 -0
  16. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/reporting.py +109 -23
  17. cli_consumption-0.4.0/src/cli_consumption/reporting_api.py +750 -0
  18. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/schema.py +138 -33
  19. cli_consumption-0.4.0/src/cli_consumption/snapshot_extraction.py +252 -0
  20. cli_consumption-0.4.0/src/cli_consumption/snapshot_files.py +255 -0
  21. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/sync.py +44 -2
  22. cli_consumption-0.3.2/src/cli_consumption/dashboard.py +0 -813
  23. cli_consumption-0.3.2/src/cli_consumption/dashboard_calculations.js +0 -543
  24. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/LICENSE +0 -0
  25. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/NOTICE +0 -0
  26. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/__init__.py +0 -0
  27. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/__main__.py +0 -0
  28. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/__init__.py +0 -0
  29. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/aider.py +0 -0
  30. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/amazon_q.py +0 -0
  31. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/amp.py +0 -0
  32. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/base.py +0 -0
  33. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/claude.py +0 -0
  34. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/cline.py +0 -0
  35. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/copilot.py +0 -0
  36. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/crush.py +0 -0
  37. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/cursor.py +0 -0
  38. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/gemini.py +0 -0
  39. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/goose.py +0 -0
  40. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/grok.py +0 -0
  41. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/kilo.py +0 -0
  42. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/kimi.py +0 -0
  43. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/mistral_vibe.py +0 -0
  44. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/openhands.py +0 -0
  45. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/pi.py +0 -0
  46. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/plandex.py +0 -0
  47. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/adapters/qwen.py +0 -0
  48. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/exporting.py +0 -0
  49. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/__init__.py +0 -0
  50. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/env.py +0 -0
  51. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/__init__.py +0 -0
  52. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/v0001_baseline.py +0 -0
  53. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/v0002_minimize_subagents.py +0 -0
  54. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/v0003_canonical_timestamps.py +0 -0
  55. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/v0004_subagent_scope_freshness.py +0 -0
  56. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/migrations/versions/v0005_sync_receipts.py +0 -0
  57. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/models.py +0 -0
  58. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/py.typed +0 -0
  59. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/qualifications.py +0 -0
  60. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/retention.py +0 -0
  61. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/storage.py +0 -0
  62. {cli_consumption-0.3.2 → cli_consumption-0.4.0}/src/cli_consumption/timestamps.py +0 -0
@@ -12,6 +12,12 @@ reports/
12
12
  *.sqlite
13
13
  *.sqlite3
14
14
  .env
15
+ node_modules/
16
+ apps/*/.next/
17
+ playwright-report/
18
+ test-results/
19
+ *.tsbuildinfo
20
+ packages/*/dist/
15
21
 
16
22
  # Machine-specific orchestration context. These files must never be committed.
17
23
  .agents/orchestrator.md
@@ -6,6 +6,55 @@ use [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.4.0] - 2026-09-01
10
+
11
+ ### Added
12
+
13
+ - Added a blocking browser CI gate that opens detailed and share-safe dashboards
14
+ directly from `file://`, exercises their controls, and rejects network activity.
15
+ - Defined the versioned, scoped, and bounded contracts for database upload, persistent
16
+ dashboard reporting, conversation pagination, and standalone web export.
17
+ - Added read-only, migration-free extraction of bounded snapshot-schema-v1 payloads
18
+ from current local SQLite collection databases, including WAL-consistent reads,
19
+ exact schema verification, provider grouping, and privacy-safe failures.
20
+ - Added `upload-db` with stable per-snapshot replay keys, required capability
21
+ negotiation, bounded retries, deterministic partial results, resumable uploads, and
22
+ strict fail-fast automation.
23
+ - Added scoped `read` and `export` bearer credentials plus strict, bounded reporting
24
+ endpoints for dashboard datasets, filter options, opaque stable pagination,
25
+ conversation detail, and self-contained HTML downloads.
26
+ - Added a locked TypeScript workspace with versioned dashboard contracts, pure shared
27
+ analytics, presentation helpers, Vitest parity coverage, ESM output, and a
28
+ reproducible network-free browser bundle included in Python wheels.
29
+ - Added an authenticated responsive Next.js dashboard with a server-only collector
30
+ credential, bounded reporting BFF, URL-safe filters, conversation pagination and
31
+ detail, shared analytics, accessibility checks, and desktop/mobile browser coverage.
32
+ - Replaced the classic offline renderer with the self-contained React/Tailwind runtime,
33
+ shared web primitives, streamed bounded dataset injection, reproducible packaged
34
+ assets, and detailed/share-safe `file://` browser coverage. The temporary
35
+ `--renderer` migration option was removed.
36
+ - Added persistent-dashboard offline downloads for the exact visible selection, with
37
+ detailed/share-safe profiles, a credential-isolating bounded BFF, private temporary
38
+ cleanup, fixed failures, and self-contained browser coverage.
39
+
40
+ ## [0.3.3] - 2026-08-31
41
+
42
+ ### Added
43
+
44
+ - Added deterministic, compressed offline snapshot files authenticated with Ed25519,
45
+ with bounded verification-before-parsing and idempotent SQLite/PostgreSQL ingestion.
46
+
47
+ ### Fixed
48
+
49
+ - Continue now applies provider-specific cache semantics, preventing Anthropic cache
50
+ reads and writes from being omitted from input totals, and reports persisted
51
+ conversation-compaction markers.
52
+ - Codex now preserves valid token accounting when provider counters are inconsistent,
53
+ keeps partial work and compaction records from referencing nonexistent turns, and
54
+ safely handles malformed metadata and timestamps.
55
+ - OpenCode now rejects timezone-naive text timestamps consistently while retaining its
56
+ OpenCode 1.18.23 message/part extraction for models, tokens, and tools.
57
+
9
58
  ## [0.3.2] - 2026-08-31
10
59
 
11
60
  ### Changed
@@ -115,7 +164,9 @@ use [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
115
164
 
116
165
  - Refreshed the provider guide for the first minor release ([#26]).
117
166
 
118
- [Unreleased]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.2...HEAD
167
+ [Unreleased]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.0...HEAD
168
+ [0.4.0]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.3...v0.4.0
169
+ [0.3.3]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.2...v0.3.3
119
170
  [0.3.2]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.1...v0.3.2
120
171
  [0.3.1]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.0...v0.3.1
121
172
  [0.3.0]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.2.1...v0.3.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cli-consumption
3
- Version: 0.3.2
3
+ Version: 0.4.0
4
4
  Summary: Analyze and consolidate AI coding CLI consumption across machines.
5
5
  Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
6
6
  Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
@@ -31,6 +31,8 @@ Requires-Dist: psycopg[binary]>=3.2; extra == 'postgres'
31
31
  Provides-Extra: server
32
32
  Requires-Dist: fastapi>=0.115; extra == 'server'
33
33
  Requires-Dist: uvicorn>=0.34; extra == 'server'
34
+ Provides-Extra: snapshots
35
+ Requires-Dist: cryptography>=45; extra == 'snapshots'
34
36
  Provides-Extra: sync
35
37
  Requires-Dist: httpx>=0.27; extra == 'sync'
36
38
  Description-Content-Type: text/markdown
@@ -75,7 +77,8 @@ the optional runtime capabilities you use:
75
77
 
76
78
  - `cli-consumption[sync]` for the sync client;
77
79
  - `cli-consumption[server]` for the collector service;
78
- - `cli-consumption[postgres]` for PostgreSQL.
80
+ - `cli-consumption[postgres]` for PostgreSQL;
81
+ - `cli-consumption[snapshots]` for signed, compressed offline snapshot files.
79
82
 
80
83
  Extras can be combined, for example `cli-consumption[server,postgres]` on a central
81
84
  collector.
@@ -94,6 +97,20 @@ uv run cli-consumption export --output reports
94
97
  Open `reports/dashboard.html` locally. It makes no network requests. Detailed CSV
95
98
  tables are generated only when `--csv` is passed.
96
99
 
100
+ Dashboard development lives in the locked npm workspace under `packages/`. It builds
101
+ provider-neutral ESM analytics, shared React presentation primitives, and deterministic
102
+ React/Tailwind browser assets that Python embeds in the wheel. The React runtime is the
103
+ only offline renderer; installing or using the Python CLI does not require Node.js.
104
+
105
+ The authenticated persistent dashboard lives in `apps/web/`. It reads the same
106
+ minimized reporting contract through a server-side Next.js BFF: the browser never
107
+ receives a collector credential or a database connection string. Its **Export
108
+ offline** action downloads the exact visible selection as a self-contained detailed or
109
+ share-safe HTML file. For a production
110
+ setup, required environment variables, reverse-proxy constraints, and startup commands
111
+ are documented in the
112
+ [deployment guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/deployment.md#run-the-persistent-dashboard).
113
+
97
114
  To collect one provider or select another database:
98
115
 
99
116
  ```bash
@@ -103,7 +120,7 @@ uv run cli-consumption collect --provider codex --database usage.sqlite
103
120
  Use `--source [LABEL=]PATH` for trusted offline copies and repeated
104
121
  `--project NAME=PATH_PREFIX` mappings for stable project labels. See the
105
122
  [usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
106
- for multi-machine collection, reporting,
123
+ for signed offline transfers, multi-machine collection, reporting,
107
124
  PostgreSQL, retention, synchronization, readiness, and automation.
108
125
 
109
126
  ## Supported CLIs
@@ -152,8 +169,10 @@ for ingestion, idempotency, migrations, report limits, and collector behavior.
152
169
  | Command | Purpose |
153
170
  | --- | --- |
154
171
  | `collect` | Collect local or copied provider data into SQL. |
172
+ | `snapshot` | Create or ingest signed, compressed offline snapshot files. |
155
173
  | `sync` | Collect and send metadata-only snapshots to a central API. |
156
- | `serve` | Run the central collection API. |
174
+ | `upload-db` | Upload validated snapshots reconstructed from a local collect database. |
175
+ | `serve` | Run the scoped central ingestion, reporting, and export API. |
157
176
  | `export` | Write the HTML dashboard and optional CSV tables. |
158
177
  | `providers` | List provider names and compatibility status. |
159
178
  | `retention` | Preview or apply deletion outside a retention window. |
@@ -184,6 +203,11 @@ uv run ruff check .
184
203
  uv run ty check
185
204
  uv run pytest --cov --cov-report=term-missing
186
205
  uv build
206
+ npm ci
207
+ npm run verify
208
+ npm audit --audit-level=high
209
+ npm run build:web
210
+ npm run test:e2e
187
211
  ```
188
212
 
189
213
  Development uses short-lived branches and squash-merged pull requests into protected
@@ -38,7 +38,8 @@ the optional runtime capabilities you use:
38
38
 
39
39
  - `cli-consumption[sync]` for the sync client;
40
40
  - `cli-consumption[server]` for the collector service;
41
- - `cli-consumption[postgres]` for PostgreSQL.
41
+ - `cli-consumption[postgres]` for PostgreSQL;
42
+ - `cli-consumption[snapshots]` for signed, compressed offline snapshot files.
42
43
 
43
44
  Extras can be combined, for example `cli-consumption[server,postgres]` on a central
44
45
  collector.
@@ -57,6 +58,20 @@ uv run cli-consumption export --output reports
57
58
  Open `reports/dashboard.html` locally. It makes no network requests. Detailed CSV
58
59
  tables are generated only when `--csv` is passed.
59
60
 
61
+ Dashboard development lives in the locked npm workspace under `packages/`. It builds
62
+ provider-neutral ESM analytics, shared React presentation primitives, and deterministic
63
+ React/Tailwind browser assets that Python embeds in the wheel. The React runtime is the
64
+ only offline renderer; installing or using the Python CLI does not require Node.js.
65
+
66
+ The authenticated persistent dashboard lives in `apps/web/`. It reads the same
67
+ minimized reporting contract through a server-side Next.js BFF: the browser never
68
+ receives a collector credential or a database connection string. Its **Export
69
+ offline** action downloads the exact visible selection as a self-contained detailed or
70
+ share-safe HTML file. For a production
71
+ setup, required environment variables, reverse-proxy constraints, and startup commands
72
+ are documented in the
73
+ [deployment guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/deployment.md#run-the-persistent-dashboard).
74
+
60
75
  To collect one provider or select another database:
61
76
 
62
77
  ```bash
@@ -66,7 +81,7 @@ uv run cli-consumption collect --provider codex --database usage.sqlite
66
81
  Use `--source [LABEL=]PATH` for trusted offline copies and repeated
67
82
  `--project NAME=PATH_PREFIX` mappings for stable project labels. See the
68
83
  [usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
69
- for multi-machine collection, reporting,
84
+ for signed offline transfers, multi-machine collection, reporting,
70
85
  PostgreSQL, retention, synchronization, readiness, and automation.
71
86
 
72
87
  ## Supported CLIs
@@ -115,8 +130,10 @@ for ingestion, idempotency, migrations, report limits, and collector behavior.
115
130
  | Command | Purpose |
116
131
  | --- | --- |
117
132
  | `collect` | Collect local or copied provider data into SQL. |
133
+ | `snapshot` | Create or ingest signed, compressed offline snapshot files. |
118
134
  | `sync` | Collect and send metadata-only snapshots to a central API. |
119
- | `serve` | Run the central collection API. |
135
+ | `upload-db` | Upload validated snapshots reconstructed from a local collect database. |
136
+ | `serve` | Run the scoped central ingestion, reporting, and export API. |
120
137
  | `export` | Write the HTML dashboard and optional CSV tables. |
121
138
  | `providers` | List provider names and compatibility status. |
122
139
  | `retention` | Preview or apply deletion outside a retention window. |
@@ -147,6 +164,11 @@ uv run ruff check .
147
164
  uv run ty check
148
165
  uv run pytest --cov --cov-report=term-missing
149
166
  uv build
167
+ npm ci
168
+ npm run verify
169
+ npm audit --audit-level=high
170
+ npm run build:web
171
+ npm run test:e2e
150
172
  ```
151
173
 
152
174
  Development uses short-lived branches and squash-merged pull requests into protected
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "cli-consumption"
7
- version = "0.3.2"
7
+ version = "0.4.0"
8
8
  description = "Analyze and consolidate AI coding CLI consumption across machines."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -34,6 +34,9 @@ dependencies = [
34
34
  postgres = ["psycopg[binary]>=3.2"]
35
35
  server = ["fastapi>=0.115", "uvicorn>=0.34"]
36
36
  sync = ["httpx>=0.27"]
37
+ snapshots = [
38
+ "cryptography>=45",
39
+ ]
37
40
 
38
41
  [project.urls]
39
42
  Homepage = "https://github.com/Guillaume-Lombardo/cli-consumption"
@@ -51,6 +54,7 @@ dev = [
51
54
  "httpx>=0.27",
52
55
  "hypothesis>=6.165.10",
53
56
  "pip-audit>=2.10.1",
57
+ "playwright>=1.55",
54
58
  "pre-commit>=4.6.2",
55
59
  "psycopg[binary]>=3.2",
56
60
  "pytest>=8.3",
@@ -107,6 +111,5 @@ select = ["E", "F", "I", "UP", "B", "SIM", "RUF", "S"]
107
111
  "tests/**" = ["S101"]
108
112
  "tests/smoke_minimal_install.py" = ["S603", "S607"]
109
113
  "tests/test_codex_adapter.py" = ["S608"]
110
- "tests/test_dashboard_calculations.py" = ["S607"]
111
114
  "tests/test_demo.py" = ["S603"]
112
115
  "tests/test_packaging.py" = ["S603", "S607"]
@@ -166,15 +166,22 @@ def tokens(
166
166
  reasoning: object = 0,
167
167
  total: object = 0,
168
168
  ) -> dict[str, int]:
169
- uncached_n = counter(uncached)
170
- cached_n = counter(cached)
171
- write_n = counter(cache_write)
172
- visible_n = counter(visible)
173
- reasoning_n = counter(reasoning)
174
- input_n = bounded_sum(uncached_n, cached_n, write_n)
175
- output_n = bounded_sum(visible_n, reasoning_n)
176
- attributed = bounded_sum(input_n, output_n)
177
- total_n = max(attributed, counter(total))
169
+ remaining = MAX_BIGINT
170
+ uncached_n = min(counter(uncached), remaining)
171
+ remaining -= uncached_n
172
+ cached_n = min(counter(cached), remaining)
173
+ remaining -= cached_n
174
+ write_n = min(counter(cache_write), remaining)
175
+ remaining -= write_n
176
+ visible_n = min(counter(visible), remaining)
177
+ remaining -= visible_n
178
+ reasoning_n = min(counter(reasoning), remaining)
179
+ remaining -= reasoning_n
180
+ input_n = uncached_n + cached_n + write_n
181
+ output_n = visible_n + reasoning_n
182
+ attributed = input_n + output_n
183
+ unattributed_n = min(max(0, counter(total) - attributed), remaining)
184
+ total_n = attributed + unattributed_n
178
185
  return {
179
186
  "input_tokens": input_n,
180
187
  "cached_input_tokens": cached_n,
@@ -184,13 +191,34 @@ def tokens(
184
191
  "total_tokens": total_n,
185
192
  "uncached_input_tokens": uncached_n,
186
193
  "visible_output_tokens": visible_n,
187
- "unattributed_tokens": max(0, total_n - attributed),
194
+ "unattributed_tokens": unattributed_n,
188
195
  }
189
196
 
190
197
 
191
198
  def add_tokens(target: dict[str, Any], value: dict[str, int]) -> None:
192
- for field, amount in value.items():
193
- target[field] = bounded_sum(int(target[field]), amount)
199
+ remaining = MAX_BIGINT
200
+ for field in (
201
+ "uncached_input_tokens",
202
+ "cached_input_tokens",
203
+ "cache_write_input_tokens",
204
+ "visible_output_tokens",
205
+ "reasoning_output_tokens",
206
+ "unattributed_tokens",
207
+ ):
208
+ amount = min(remaining, int(target[field]) + value[field])
209
+ target[field] = amount
210
+ remaining -= amount
211
+ target["input_tokens"] = (
212
+ target["uncached_input_tokens"]
213
+ + target["cached_input_tokens"]
214
+ + target["cache_write_input_tokens"]
215
+ )
216
+ target["output_tokens"] = (
217
+ target["visible_output_tokens"] + target["reasoning_output_tokens"]
218
+ )
219
+ target["total_tokens"] = (
220
+ target["input_tokens"] + target["output_tokens"] + target["unattributed_tokens"]
221
+ )
194
222
 
195
223
 
196
224
  def new_turn(
@@ -87,11 +87,14 @@ AGENT_ROLE_ALIASES = {
87
87
  SAFE_DIMENSION = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+-]*")
88
88
 
89
89
 
90
- def parse_timestamp(value: str | None) -> datetime | None:
91
- if not value:
90
+ def parse_timestamp(value: object) -> datetime | None:
91
+ if not isinstance(value, str) or not value:
92
92
  return None
93
93
  try:
94
- return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(UTC)
94
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
95
+ if parsed.tzinfo is None or parsed.utcoffset() is None:
96
+ return None
97
+ return parsed.astimezone(UTC)
95
98
  except ValueError:
96
99
  return None
97
100
 
@@ -99,7 +102,8 @@ def parse_timestamp(value: str | None) -> datetime | None:
99
102
  def infer_project(
100
103
  metadata: dict[str, Any], mappings: list[tuple[str, str]]
101
104
  ) -> tuple[str, str]:
102
- cwd = str(metadata.get("cwd") or "").rstrip("/\\")
105
+ raw_cwd = metadata.get("cwd")
106
+ cwd = raw_cwd.rstrip("/\\") if isinstance(raw_cwd, str) else ""
103
107
  normalized_cwd = cwd.replace("\\", "/")
104
108
  for name, prefix in sorted(mappings, key=lambda item: len(item[1]), reverse=True):
105
109
  normalized_prefix = prefix.replace("\\", "/").rstrip("/")
@@ -109,17 +113,18 @@ def infer_project(
109
113
  return name, "mapping"
110
114
  git = metadata.get("git")
111
115
  if isinstance(git, dict):
112
- repository = str(git.get("repository_url") or git.get("repository") or "")
116
+ raw_repository = git.get("repository_url") or git.get("repository")
117
+ repository = raw_repository if isinstance(raw_repository, str) else ""
113
118
  slug = re.split(r"[/\\:]", repository.rstrip("/\\"))[-1]
114
119
  if slug.endswith(".git"):
115
120
  slug = slug[:-4]
116
- if slug:
121
+ if _safe_dimension(slug, 255):
117
122
  return slug, "git"
118
123
  return OUTSIDE_PROJECT, "none"
119
124
 
120
125
 
121
126
  def extract_tools(payload: dict[str, Any]) -> list[tuple[str, str]]:
122
- outer_name = str(payload.get("name", "unknown"))
127
+ outer_name = _safe_dimension(payload.get("name"), 512) or "unknown"
123
128
  if outer_name != "exec":
124
129
  return [(outer_name, outer_name)]
125
130
  raw_input = payload.get("input", "")
@@ -233,9 +238,16 @@ class CodexAdapter:
233
238
  continue
234
239
  event_count += 1
235
240
  if event.get("type") == "session_meta":
236
- conversation_id = str(event.get("payload", {}).get("id", ""))
237
- conversation_id = conversation_id or path.stem
238
- candidate = (machine, path, event_count, digest.hexdigest())
241
+ payload = event.get("payload")
242
+ if not isinstance(payload, dict):
243
+ malformed += 1
244
+ continue
245
+ candidate_id = _safe_dimension(payload.get("id"), 512)
246
+ if candidate_id and not conversation_id:
247
+ conversation_id = candidate_id
248
+ content_hash = digest.hexdigest()
249
+ conversation_id = conversation_id or f"content-{content_hash}"
250
+ candidate = (machine, path, event_count, content_hash)
239
251
  previous = selected.get(conversation_id)
240
252
  if previous is None:
241
253
  selected[conversation_id] = candidate
@@ -273,7 +285,16 @@ class CodexAdapter:
273
285
  ),
274
286
  {},
275
287
  )
276
- conversation_id = str(metadata.get("id") or path.stem)
288
+ conversation_id = next(
289
+ (
290
+ candidate_id
291
+ for event in events
292
+ if event.get("type") == "session_meta"
293
+ and isinstance((payload := event.get("payload")), dict)
294
+ and (candidate_id := _safe_dimension(payload.get("id"), 512))
295
+ ),
296
+ f"content-{digest}",
297
+ )
277
298
  record_id = f"codex:{conversation_id}"
278
299
  project, project_source = infer_project(metadata, mappings)
279
300
  timestamps = [
@@ -316,9 +337,7 @@ class CodexAdapter:
316
337
  {
317
338
  "id": f"{record_id}:compaction:{compaction_sequence}",
318
339
  "conversation_id": record_id,
319
- "turn_id": (
320
- f"{record_id}:{active_turn_id}" if active_turn_id else None
321
- ),
340
+ "turn_id": turns.get(active_turn_id or "", {}).get("id"),
322
341
  "sequence": compaction_sequence,
323
342
  "timestamp": _iso(timestamp),
324
343
  }
@@ -343,7 +362,7 @@ class CodexAdapter:
343
362
  _merge_present(settings_by_turn[active_turn_id], updates)
344
363
  if event_type == "turn_context":
345
364
  active_turn_id = (
346
- str(payload.get("turn_id") or active_turn_id or "") or None
365
+ _safe_dimension(payload.get("turn_id"), 512) or active_turn_id
347
366
  )
348
367
  active_model = (
349
368
  _safe_dimension(payload.get("model"), 255) or active_model
@@ -362,7 +381,7 @@ class CodexAdapter:
362
381
  or setting_defaults["collaboration_mode"],
363
382
  }
364
383
  if event_type == "event_msg" and payload_type == "task_started":
365
- active_turn_id = str(payload.get("turn_id") or "") or None
384
+ active_turn_id = _safe_dimension(payload.get("turn_id"), 512)
366
385
  if active_turn_id:
367
386
  settings = settings_by_turn.setdefault(
368
387
  active_turn_id, dict(setting_defaults)
@@ -391,7 +410,9 @@ class CodexAdapter:
391
410
  "task_complete",
392
411
  "turn_aborted",
393
412
  }:
394
- turn_id = str(payload.get("turn_id") or active_turn_id or "")
413
+ turn_id = (
414
+ _safe_dimension(payload.get("turn_id"), 512) or active_turn_id or ""
415
+ )
395
416
  if turn_id in turns:
396
417
  turns[turn_id].update(
397
418
  ended_at=_iso(timestamp),
@@ -412,12 +433,13 @@ class CodexAdapter:
412
433
  work_sequence += 1
413
434
  started_at_ms = _integer_or_none(payload.get("started_at_ms"))
414
435
  completed_at_ms = _integer_or_none(payload.get("completed_at_ms"))
415
- turn_id = str(payload.get("turn_id") or active_turn_id or "") or None
436
+ turn_id = _safe_dimension(payload.get("turn_id") or active_turn_id, 512)
437
+ turn = turns.get(turn_id or "")
416
438
  snapshot.work_items.append(
417
439
  {
418
440
  "id": f"{record_id}:work:{work_sequence}",
419
441
  "conversation_id": record_id,
420
- "turn_id": f"{record_id}:{turn_id}" if turn_id else None,
442
+ "turn_id": turn["id"] if turn else None,
421
443
  "sequence": work_sequence,
422
444
  "kind": WORK_ITEM_KINDS.get(
423
445
  str(item.get("type") or ""), "other"
@@ -438,18 +460,12 @@ class CodexAdapter:
438
460
  if not isinstance(usage, dict):
439
461
  continue
440
462
  call_sequence += 1
441
- tokens = {
442
- field: _nonnegative_integer(usage.get(field))
443
- for field in TOKEN_FIELDS
444
- }
445
- tokens.update(_derived_tokens(tokens))
446
- for field, value in tokens.items():
447
- totals[field] += value
463
+ tokens = _usage_tokens(usage)
464
+ _accumulate_tokens(totals, tokens)
448
465
  turn = turns.get(active_turn_id or "")
449
466
  if turn:
450
467
  turn["model_calls"] += 1
451
- for field, value in tokens.items():
452
- turn[field] += value
468
+ _accumulate_tokens(turn, tokens)
453
469
  snapshot.model_calls.append(
454
470
  {
455
471
  "id": f"{record_id}:model:{call_sequence}",
@@ -548,24 +564,62 @@ class CodexAdapter:
548
564
  )
549
565
 
550
566
 
551
- def _derived_tokens(tokens: dict[str, int]) -> dict[str, int]:
567
+ def _usage_tokens(usage: dict[str, Any]) -> dict[str, int]:
568
+ raw = {field: _nonnegative_integer(usage.get(field)) for field in TOKEN_FIELDS}
569
+ cached = raw["cached_input_tokens"]
570
+ cache_write = min(raw["cache_write_input_tokens"], MAX_BIGINT - cached)
571
+ input_tokens = max(raw["input_tokens"], cached + cache_write)
572
+ uncached = input_tokens - cached - cache_write
573
+ remaining = MAX_BIGINT - input_tokens
574
+ reasoning = min(raw["reasoning_output_tokens"], remaining)
575
+ visible = min(
576
+ max(0, raw["output_tokens"] - raw["reasoning_output_tokens"]),
577
+ remaining - reasoning,
578
+ )
579
+ output_tokens = reasoning + visible
580
+ unattributed = min(
581
+ max(0, raw["total_tokens"] - input_tokens - output_tokens),
582
+ MAX_BIGINT - input_tokens - output_tokens,
583
+ )
552
584
  return {
553
- "uncached_input_tokens": max(
554
- 0,
555
- tokens["input_tokens"]
556
- - tokens["cached_input_tokens"]
557
- - tokens["cache_write_input_tokens"],
558
- ),
559
- "visible_output_tokens": max(
560
- 0, tokens["output_tokens"] - tokens["reasoning_output_tokens"]
561
- ),
562
- "unattributed_tokens": max(
563
- 0,
564
- tokens["total_tokens"] - tokens["input_tokens"] - tokens["output_tokens"],
565
- ),
585
+ "input_tokens": input_tokens,
586
+ "cached_input_tokens": cached,
587
+ "cache_write_input_tokens": cache_write,
588
+ "output_tokens": output_tokens,
589
+ "reasoning_output_tokens": reasoning,
590
+ "total_tokens": input_tokens + output_tokens + unattributed,
591
+ "uncached_input_tokens": uncached,
592
+ "visible_output_tokens": visible,
593
+ "unattributed_tokens": unattributed,
566
594
  }
567
595
 
568
596
 
597
+ def _accumulate_tokens(target: dict[str, Any], value: dict[str, int]) -> None:
598
+ remaining = MAX_BIGINT
599
+ for field in (
600
+ "uncached_input_tokens",
601
+ "cached_input_tokens",
602
+ "cache_write_input_tokens",
603
+ "visible_output_tokens",
604
+ "reasoning_output_tokens",
605
+ "unattributed_tokens",
606
+ ):
607
+ amount = min(remaining, int(target[field]) + value[field])
608
+ target[field] = amount
609
+ remaining -= amount
610
+ target["input_tokens"] = (
611
+ target["uncached_input_tokens"]
612
+ + target["cached_input_tokens"]
613
+ + target["cache_write_input_tokens"]
614
+ )
615
+ target["output_tokens"] = (
616
+ target["visible_output_tokens"] + target["reasoning_output_tokens"]
617
+ )
618
+ target["total_tokens"] = (
619
+ target["input_tokens"] + target["output_tokens"] + target["unattributed_tokens"]
620
+ )
621
+
622
+
569
623
  def _safe_dimension(value: object, maximum: int) -> str | None:
570
624
  if not isinstance(value, str):
571
625
  return None
@@ -650,5 +704,9 @@ def _integer_or_none(value: object) -> int | None:
650
704
 
651
705
 
652
706
  def _nonnegative_integer(value: object) -> int:
653
- parsed = _integer_or_none(value)
707
+ parsed = (
708
+ _integer_or_none(value)
709
+ if not isinstance(value, float) or value.is_integer()
710
+ else None
711
+ )
654
712
  return max(0, parsed or 0)