cli-consumption 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/CONTRIBUTING.md +6 -0
  2. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/PKG-INFO +35 -12
  3. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/README.md +34 -11
  4. cli_consumption-0.2.1/SECURITY.md +37 -0
  5. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/architecture.md +26 -7
  6. cli_consumption-0.2.1/docs/decisions/0002-canonical-utc-timestamps.md +94 -0
  7. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/privacy.md +14 -3
  8. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/provider-support.md +5 -2
  9. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/roadmap.md +5 -0
  10. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/pyproject.toml +1 -1
  11. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/_shared.py +40 -1
  12. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/aider.py +25 -25
  13. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/amazon_q.py +2 -0
  14. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/amp.py +2 -1
  15. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/claude.py +20 -21
  16. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/cline.py +2 -0
  17. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/codex.py +25 -24
  18. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/continue_cli.py +2 -1
  19. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/copilot.py +14 -14
  20. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/crush.py +3 -1
  21. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/cursor.py +18 -14
  22. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/gemini.py +17 -14
  23. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/goose.py +2 -0
  24. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/grok.py +5 -6
  25. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/kilo.py +2 -0
  26. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/kimi.py +2 -1
  27. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/mistral_vibe.py +18 -17
  28. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/opencode.py +2 -0
  29. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/openhands.py +3 -2
  30. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/pi.py +12 -12
  31. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/plandex.py +2 -1
  32. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/qwen.py +14 -14
  33. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/registry.py +96 -15
  34. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/cli.py +108 -15
  35. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/dashboard.py +10 -5
  36. cli_consumption-0.2.1/src/cli_consumption/migrations/versions/v0003_canonical_timestamps.py +94 -0
  37. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/models.py +70 -11
  38. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/reporting.py +15 -30
  39. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/retention.py +10 -14
  40. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/schema.py +100 -6
  41. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/storage.py +18 -5
  42. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/sync.py +25 -0
  43. cli_consumption-0.2.1/src/cli_consumption/timestamps.py +17 -0
  44. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_api.py +4 -0
  45. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cli.py +91 -3
  46. cli_consumption-0.2.1/tests/test_input_limits.py +57 -0
  47. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_migrations_and_retention.py +232 -10
  48. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_provider_registry.py +33 -0
  49. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_reporting.py +15 -7
  50. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_storage_and_exports.py +42 -1
  51. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_sync.py +44 -0
  52. cli_consumption-0.2.1/tests/test_timestamps.py +61 -0
  53. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/uv.lock +1 -1
  54. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/add-cli-adapter/SKILL.md +0 -0
  55. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/add-cli-adapter/agents/openai.yaml +0 -0
  56. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/audit-usage-privacy/SKILL.md +0 -0
  57. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/audit-usage-privacy/agents/openai.yaml +0 -0
  58. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/evolve-storage-schema/SKILL.md +0 -0
  59. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/evolve-storage-schema/agents/openai.yaml +0 -0
  60. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yeet-github/SKILL.md +0 -0
  61. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yeet-github/agents/openai.yaml +0 -0
  62. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yolo/SKILL.md +0 -0
  63. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yolo/agents/openai.yaml +0 -0
  64. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.github/workflows/ci.yml +0 -0
  65. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.github/workflows/release.yaml +0 -0
  66. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.gitignore +0 -0
  67. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.pre-commit-config.yaml +0 -0
  68. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.python-version +0 -0
  69. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/AGENTS.md +0 -0
  70. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/LICENSE +0 -0
  71. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/NOTICE +0 -0
  72. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/decisions/0001-versioned-schema-migrations.md +0 -0
  73. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/__init__.py +0 -0
  74. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/__main__.py +0 -0
  75. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/__init__.py +0 -0
  76. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/base.py +0 -0
  77. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/api.py +0 -0
  78. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/exporting.py +0 -0
  79. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/__init__.py +0 -0
  80. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/env.py +0 -0
  81. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/__init__.py +0 -0
  82. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/v0001_baseline.py +0 -0
  83. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/v0002_minimize_subagents.py +0 -0
  84. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/py.typed +0 -0
  85. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/conftest.py +0 -0
  86. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/smoke_minimal_install.py +0 -0
  87. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_aider_adapter.py +0 -0
  88. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_amazon_q_adapter.py +0 -0
  89. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_amp_adapter.py +0 -0
  90. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_claude_adapter.py +0 -0
  91. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cline_adapter.py +0 -0
  92. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_codex_adapter.py +0 -0
  93. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_continue_adapter.py +0 -0
  94. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_copilot_adapter.py +0 -0
  95. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_crush_adapter.py +0 -0
  96. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cursor_adapter.py +0 -0
  97. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_exporting.py +0 -0
  98. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_gemini_adapter.py +0 -0
  99. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_goose_adapter.py +0 -0
  100. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_grok_adapter.py +0 -0
  101. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_kilo_adapter.py +0 -0
  102. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_kimi_adapter.py +0 -0
  103. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_mistral_vibe_adapter.py +0 -0
  104. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_opencode_adapter.py +0 -0
  105. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_openhands_adapter.py +0 -0
  106. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_packaging.py +0 -0
  107. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_pi_adapter.py +0 -0
  108. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_plandex_adapter.py +0 -0
  109. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_qwen_adapter.py +0 -0
  110. {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_snapshot_contract.py +0 -0
@@ -30,6 +30,12 @@ errors, and the documented server-first upgrade sequence. Export changes must pr
30
30
  deterministic ordering, bounded-memory CSV streaming, complete selected conversation
31
31
  graphs, and spreadsheet-formula neutralization.
32
32
 
33
+ Provider readers must use the bounded file and JSONL helpers in `adapters/_shared.py`,
34
+ must not follow direct provider-file symlinks, and must stay within the shared snapshot
35
+ record budget. Every registered adapter test module must include a synthetic canary and
36
+ assert its absence from the normalized snapshot; shared privacy tests cover all later
37
+ output surfaces.
38
+
33
39
  ## Validate and review
34
40
 
35
41
  Run the quality gates documented in `AGENTS.md`, inspect the complete diff, then open a
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cli-consumption
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Analyze and consolidate AI coding CLI consumption across machines.
5
5
  Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
6
6
  Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
@@ -41,7 +41,8 @@ machines, and can send metadata-only snapshots to a central collector.
41
41
 
42
42
  It never stores prompts, responses, tool arguments, credentials, or raw provider
43
43
  events. Local token counters are usage metadata, not billing records. Read the
44
- [privacy boundary](docs/privacy.md) before sharing a database or report.
44
+ [privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md)
45
+ before sharing a database or report.
45
46
 
46
47
  ## Quick start
47
48
 
@@ -120,7 +121,7 @@ Use the provider name below with `--provider` to select one CLI explicitly.
120
121
 
121
122
  Provider formats are internal and can change without notice. The detailed extraction
122
123
  rules and qualification versions are documented in
123
- [Provider support](docs/provider-support.md).
124
+ [Provider support](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/provider-support.md).
124
125
 
125
126
  ## Collect copied data
126
127
 
@@ -142,6 +143,11 @@ Copy only the required provider data. For Codex, copy the `sessions/` directory
142
143
  never `auth.json` or other credentials. Globally identical conversation IDs are
143
144
  deduplicated, and the most complete copy wins.
144
145
 
146
+ Provider files are untrusted. Monolithic JSON files are limited to 64 MiB, JSONL files
147
+ to 256 MiB with an 8 MiB per-line limit, and a snapshot to 250,000 normalized records
148
+ while it is being built. Direct provider-file symlinks are refused. `collect --strict`
149
+ refuses to write a snapshot when malformed records were skipped.
150
+
145
151
  Map original working-directory prefixes to stable project labels with repeated
146
152
  `--project NAME=PATH_PREFIX` options. The longest matching prefix wins:
147
153
 
@@ -163,7 +169,7 @@ uv run cli-consumption collect --provider plandex \
163
169
 
164
170
  The dashboard can filter by time, provider, machine, project, and model. It reports
165
171
  activity, token composition, cache efficiency, latency and duration distributions,
166
- technical throughput, context pressure, work-item reliability, configuration cohorts,
172
+ turn rate, context pressure, work-item reliability, configuration cohorts,
167
173
  compactions, subagent delegation, and ingestion quality. Availability varies by
168
174
  provider, as summarized in the table above.
169
175
 
@@ -209,8 +215,14 @@ unversioned databases that exactly match a published schema are adopted before t
209
215
  upgrade; unknown or modified schemas are refused. Back up production databases before
210
216
  upgrading and do not run mixed application versions against one database while a
211
217
  migration is in progress. See the
212
- [migration decision](docs/decisions/0001-versioned-schema-migrations.md) for rollback
213
- and compatibility rules.
218
+ [migration decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0001-versioned-schema-migrations.md)
219
+ for rollback and compatibility rules.
220
+
221
+ Timezone-aware timestamps are normalized to fixed-width UTC strings during ingestion.
222
+ Revision `0003` rewrites legacy timestamp text in bounded batches and adds an indexed
223
+ conversation end-time path; see the
224
+ [timestamp decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0002-canonical-utc-timestamps.md)
225
+ for the exact representation and downgrade boundary.
214
226
 
215
227
  Preview retention before deleting normalized metadata:
216
228
 
@@ -243,8 +255,9 @@ uv run cli-consumption sync --provider all \
243
255
  ```
244
256
 
245
257
  The application refuses to bind beyond localhost without a token. Production
246
- deployments also need TLS and standard operational controls. See
247
- [Architecture](docs/architecture.md) for the trade-offs.
258
+ deployments also need TLS and standard operational controls. The sync client refuses
259
+ plain HTTP beyond loopback unless `--allow-insecure` is passed explicitly. See
260
+ [Architecture](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/architecture.md) for the trade-offs.
248
261
 
249
262
  Snapshots use strict schema version 1. The collector rejects request bodies larger
250
263
  than 32 MiB and snapshots containing more than 250,000 normalized records. A sync
@@ -263,7 +276,10 @@ uv run cli-consumption providers --json
263
276
  Each provider reports one of `no-data`, `detected`, `compatible`, `degraded`, or
264
277
  `unsupported-schema`. Diagnostics parse enough metadata to assess compatibility but do
265
278
  not persist it and never include paths, identifiers, record contents, counts, or parser
266
- errors in their output.
279
+ errors in their output. Schema version 2 also declares whether token counters are
280
+ additive, conversation aggregates, context snapshots, or unavailable. Dashboard token
281
+ per-turn percentiles use only additive providers rather than treating missing measures
282
+ as zero.
267
283
 
268
284
  ## Commands
269
285
 
@@ -278,6 +294,10 @@ errors in their output.
278
294
 
279
295
  Run `uv run cli-consumption COMMAND --help` for all options.
280
296
 
297
+ `collect`, `export`, and `retention` accept `--json` for deterministic
298
+ machine-readable results. `collect --strict` rejects snapshots containing malformed
299
+ provider records before opening the destination database.
300
+
281
301
  ## Development
282
302
 
283
303
  ```bash
@@ -292,9 +312,12 @@ uv build
292
312
  ```
293
313
 
294
314
  Development uses short-lived branches and squash-merged pull requests into protected
295
- `main`. Read [CONTRIBUTING.md](CONTRIBUTING.md) and [AGENTS.md](AGENTS.md) before
296
- changing the project.
315
+ `main`. Read
316
+ [CONTRIBUTING.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/CONTRIBUTING.md)
317
+ and [AGENTS.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/AGENTS.md)
318
+ before changing the project. Security issues follow the private reporting guidance in
319
+ [SECURITY.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/SECURITY.md).
297
320
 
298
321
  ## License
299
322
 
300
- Licensed under the [Apache License 2.0](LICENSE).
323
+ Licensed under the [Apache License 2.0](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/LICENSE).
@@ -6,7 +6,8 @@ machines, and can send metadata-only snapshots to a central collector.
6
6
 
7
7
  It never stores prompts, responses, tool arguments, credentials, or raw provider
8
8
  events. Local token counters are usage metadata, not billing records. Read the
9
- [privacy boundary](docs/privacy.md) before sharing a database or report.
9
+ [privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md)
10
+ before sharing a database or report.
10
11
 
11
12
  ## Quick start
12
13
 
@@ -85,7 +86,7 @@ Use the provider name below with `--provider` to select one CLI explicitly.
85
86
 
86
87
  Provider formats are internal and can change without notice. The detailed extraction
87
88
  rules and qualification versions are documented in
88
- [Provider support](docs/provider-support.md).
89
+ [Provider support](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/provider-support.md).
89
90
 
90
91
  ## Collect copied data
91
92
 
@@ -107,6 +108,11 @@ Copy only the required provider data. For Codex, copy the `sessions/` directory
107
108
  never `auth.json` or other credentials. Globally identical conversation IDs are
108
109
  deduplicated, and the most complete copy wins.
109
110
 
111
+ Provider files are untrusted. Monolithic JSON files are limited to 64 MiB, JSONL files
112
+ to 256 MiB with an 8 MiB per-line limit, and a snapshot to 250,000 normalized records
113
+ while it is being built. Direct provider-file symlinks are refused. `collect --strict`
114
+ refuses to write a snapshot when malformed records were skipped.
115
+
110
116
  Map original working-directory prefixes to stable project labels with repeated
111
117
  `--project NAME=PATH_PREFIX` options. The longest matching prefix wins:
112
118
 
@@ -128,7 +134,7 @@ uv run cli-consumption collect --provider plandex \
128
134
 
129
135
  The dashboard can filter by time, provider, machine, project, and model. It reports
130
136
  activity, token composition, cache efficiency, latency and duration distributions,
131
- technical throughput, context pressure, work-item reliability, configuration cohorts,
137
+ turn rate, context pressure, work-item reliability, configuration cohorts,
132
138
  compactions, subagent delegation, and ingestion quality. Availability varies by
133
139
  provider, as summarized in the table above.
134
140
 
@@ -174,8 +180,14 @@ unversioned databases that exactly match a published schema are adopted before t
174
180
  upgrade; unknown or modified schemas are refused. Back up production databases before
175
181
  upgrading and do not run mixed application versions against one database while a
176
182
  migration is in progress. See the
177
- [migration decision](docs/decisions/0001-versioned-schema-migrations.md) for rollback
178
- and compatibility rules.
183
+ [migration decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0001-versioned-schema-migrations.md)
184
+ for rollback and compatibility rules.
185
+
186
+ Timezone-aware timestamps are normalized to fixed-width UTC strings during ingestion.
187
+ Revision `0003` rewrites legacy timestamp text in bounded batches and adds an indexed
188
+ conversation end-time path; see the
189
+ [timestamp decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0002-canonical-utc-timestamps.md)
190
+ for the exact representation and downgrade boundary.
179
191
 
180
192
  Preview retention before deleting normalized metadata:
181
193
 
@@ -208,8 +220,9 @@ uv run cli-consumption sync --provider all \
208
220
  ```
209
221
 
210
222
  The application refuses to bind beyond localhost without a token. Production
211
- deployments also need TLS and standard operational controls. See
212
- [Architecture](docs/architecture.md) for the trade-offs.
223
+ deployments also need TLS and standard operational controls. The sync client refuses
224
+ plain HTTP beyond loopback unless `--allow-insecure` is passed explicitly. See
225
+ [Architecture](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/architecture.md) for the trade-offs.
213
226
 
214
227
  Snapshots use strict schema version 1. The collector rejects request bodies larger
215
228
  than 32 MiB and snapshots containing more than 250,000 normalized records. A sync
@@ -228,7 +241,10 @@ uv run cli-consumption providers --json
228
241
  Each provider reports one of `no-data`, `detected`, `compatible`, `degraded`, or
229
242
  `unsupported-schema`. Diagnostics parse enough metadata to assess compatibility but do
230
243
  not persist it and never include paths, identifiers, record contents, counts, or parser
231
- errors in their output.
244
+ errors in their output. Schema version 2 also declares whether token counters are
245
+ additive, conversation aggregates, context snapshots, or unavailable. Dashboard token
246
+ per-turn percentiles use only additive providers rather than treating missing measures
247
+ as zero.
232
248
 
233
249
  ## Commands
234
250
 
@@ -243,6 +259,10 @@ errors in their output.
243
259
 
244
260
  Run `uv run cli-consumption COMMAND --help` for all options.
245
261
 
262
+ `collect`, `export`, and `retention` accept `--json` for deterministic
263
+ machine-readable results. `collect --strict` rejects snapshots containing malformed
264
+ provider records before opening the destination database.
265
+
246
266
  ## Development
247
267
 
248
268
  ```bash
@@ -257,9 +277,12 @@ uv build
257
277
  ```
258
278
 
259
279
  Development uses short-lived branches and squash-merged pull requests into protected
260
- `main`. Read [CONTRIBUTING.md](CONTRIBUTING.md) and [AGENTS.md](AGENTS.md) before
261
- changing the project.
280
+ `main`. Read
281
+ [CONTRIBUTING.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/CONTRIBUTING.md)
282
+ and [AGENTS.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/AGENTS.md)
283
+ before changing the project. Security issues follow the private reporting guidance in
284
+ [SECURITY.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/SECURITY.md).
262
285
 
263
286
  ## License
264
287
 
265
- Licensed under the [Apache License 2.0](LICENSE).
288
+ Licensed under the [Apache License 2.0](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/LICENSE).
@@ -0,0 +1,37 @@
1
+ # Security policy
2
+
3
+ ## Supported versions
4
+
5
+ Security fixes are made on the latest released version and on `main`. Older releases
6
+ are not maintained as separate security branches. Upgrade to the latest release before
7
+ reporting a problem that may already have been fixed.
8
+
9
+ ## Reporting a vulnerability
10
+
11
+ Do not open a public issue for a suspected vulnerability. Email
12
+ `lombardo.guillaume@gmail.com` with the subject `CLI Consumption security report` and
13
+ include:
14
+
15
+ - the affected version and operating system;
16
+ - the provider, command, or API surface involved;
17
+ - reproduction steps using synthetic data;
18
+ - the expected and observed impact;
19
+ - any suggested mitigation.
20
+
21
+ Do not send real prompts, responses, credentials, provider databases, or other private
22
+ conversation data. Use a synthetic canary when demonstrating a disclosure.
23
+
24
+ The maintainer will acknowledge the report, assess whether it crosses the documented
25
+ privacy or security boundary, and coordinate a fix and disclosure when appropriate.
26
+
27
+ ## Security boundary
28
+
29
+ Provider files and incoming snapshots are untrusted input. Relevant reports include
30
+ content or credential disclosure, unsafe filesystem traversal, denial of service,
31
+ authentication bypass, cross-machine data corruption, and generated dashboards that
32
+ perform network requests or execute provider-controlled content.
33
+
34
+ Operational exposure caused solely by publishing a normalized database or detailed CSV
35
+ is outside the vulnerability boundary: those artifacts intentionally contain private
36
+ operational metadata. The precise allowed and prohibited fields are documented in the
37
+ [privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md).
@@ -32,7 +32,7 @@ provider files -> adapter -> metadata-only snapshot -> SQL storage -> dashboard/
32
32
  provide an offline HTML view by default, and stream deterministic portable CSV tables
33
33
  when explicitly requested.
34
34
  - `adapters.registry`: is the single source for canonical names, aliases, adapter
35
- classes, default homes, detection markers, and support state.
35
+ classes, default homes, detection markers, support state, and token semantics.
36
36
  - `cli`: exposes the same capabilities through one executable.
37
37
 
38
38
  When Codex exposes its local thread graph, the adapter also records metadata-only
@@ -64,28 +64,41 @@ or platform ingress. Read access is not exposed. The collector limits bodies to
64
64
  including chunked requests, accepts snapshot schema v1, and caps a snapshot at 250,000
65
65
  normalized records. `/api/v1/capabilities` publishes these limits and the supported
66
66
  schema range so clients can fail before uploading incompatible data.
67
+ The sync client refuses plain HTTP beyond loopback unless the operator passes the
68
+ explicit `--allow-insecure` override for a trusted network.
67
69
 
68
70
  ## Storage
69
71
 
70
72
  SQLite is the zero-configuration default. PostgreSQL is selected by passing a
71
73
  `postgresql+psycopg://` URL. Every database open upgrades through packaged Alembic
72
- migrations. An unversioned database is adopted only when its tables exactly match a
73
- published schema; an unknown, newer, or locally modified schema is refused. Migration
74
- revisions support SQLite and PostgreSQL and define a bounded downgrade, but application
75
- rollback can still require restoring a pre-upgrade backup. Operators must upgrade the
76
- server first and avoid mixed-version access during migration. The detailed policy is
77
- recorded in [ADR 0001](decisions/0001-versioned-schema-migrations.md).
74
+ migrations. An unversioned database is adopted only when its columns, SQL types,
75
+ nullability, primary and foreign keys, and indexes exactly match a published schema; an
76
+ unknown, newer, or locally modified schema is refused. Migration revisions support
77
+ SQLite and PostgreSQL and define a bounded downgrade, but application rollback can
78
+ still require restoring a pre-upgrade backup. Operators must upgrade the server first
79
+ and avoid mixed-version access during migration. The detailed policy is recorded in
80
+ [ADR 0001](decisions/0001-versioned-schema-migrations.md).
78
81
 
79
82
  Conversation records use a provider-qualified stable ID. Repeated ingestion skips an
80
83
  identical or less complete record. A more complete copy atomically replaces the
81
84
  conversation and its child records.
82
85
 
86
+ Subagent relationships have a provider-and-source-machine lifecycle. Every represented
87
+ scope is replaced atomically on ingestion, including when its latest snapshot contains
88
+ no edges, so deleted provider relationships do not remain indefinitely.
89
+
83
90
  Workflow analytics use additive child tables: `work_items`, `context_samples`,
84
91
  `turn_settings`, and `compaction_events`. Snapshot schema v1 validates every record,
85
92
  rejects unknown fields, enforces normalized labels and relationships, and exposes only
86
93
  generic validation errors. A newer client sent to an older strict API is rejected
87
94
  before ingestion, so central deployments must upgrade the server first.
88
95
 
96
+ Provider input is bounded before persistence. Monolithic JSON files are capped at
97
+ 64 MiB, JSONL files at 256 MiB with an 8 MiB per-line limit, and the complete in-memory
98
+ normalized snapshot at 250,000 records. Direct symlinks to provider files are rejected.
99
+ These limits use generic error codes and apply to local collection as well as snapshots
100
+ later sent to the API.
101
+
89
102
  Retention is an explicit two-step operation: `retention --keep-days N` reports what
90
103
  would be deleted, while `--apply` deletes old conversations (with cascading children),
91
104
  subagent relationships, and ingestion runs in one transaction.
@@ -103,3 +116,9 @@ each table, and neutralizes text that spreadsheet software could execute as a fo
103
116
  Adapters are introduced one at a time because local data formats are undocumented or
104
117
  can evolve independently. Each adapter must have synthetic fixtures, format detection,
105
118
  privacy tests, and a documented support level before it appears as supported.
119
+
120
+ Timestamp fields are canonical fixed-width UTC strings after strict snapshot
121
+ validation and schema revision `0003`. Reporting and retention bind values in the same
122
+ form and use direct indexed predicates with explicit null branches. The representation,
123
+ migration, and rollback boundary are documented in
124
+ [ADR 0002](decisions/0002-canonical-utc-timestamps.md).
@@ -0,0 +1,94 @@
1
+ # ADR 0002: Canonical UTC timestamp storage
2
+
3
+ - Status: Accepted and implemented in schema revision `0003`
4
+ - Scope: normalized SQL timestamps and time-window queries
5
+
6
+ ## Context
7
+
8
+ Snapshot schema v1 accepts timezone-aware ISO 8601 timestamps. Before revision `0003`,
9
+ SQL stored those values as text exactly as received. Reporting and retention therefore
10
+ called SQLite `datetime(...)` or cast PostgreSQL text to `timestamptz` for every
11
+ comparison. Those expressions prevented the existing timestamp indexes from serving
12
+ the common range predicates. Equivalent instants could also have different lexical
13
+ forms because offsets and fractional-second precision were not canonicalized.
14
+
15
+ The affected SQL fields are conversation and turn start/end times, model/tool/context/
16
+ compaction event timestamps, and ingestion-run timestamps. Millisecond epoch fields on
17
+ work items and subagent relationships already have a separate, explicit provider
18
+ meaning and are outside this decision.
19
+
20
+ An SQLite 3.53.1 query-plan check against the published schema confirmed that
21
+ `datetime(coalesce(...))` performs a table scan, while a direct comparison against a
22
+ canonical timestamp uses the timestamp index. This is a query-shape limitation, not a
23
+ SQLite version-specific parsing bug.
24
+
25
+ ## Decision
26
+
27
+ Keep timezone-aware ISO strings in snapshot schema v1 and in CSV exports, but make the
28
+ stored representation canonical:
29
+
30
+ ```text
31
+ YYYY-MM-DDTHH:MM:SS.ffffff+00:00
32
+ ```
33
+
34
+ At the strict snapshot boundary, parse each timestamp, convert it to UTC, and serialize
35
+ it with `datetime.isoformat(timespec="microseconds")`. Generate ingestion timestamps
36
+ through the same helper. Fixed-width UTC strings sort in instant order on both SQLite
37
+ and PostgreSQL, remain human-readable, preserve Python datetime precision, fit the
38
+ existing `VARCHAR(64)` columns, and stay compatible with snapshot schema v1.
39
+
40
+ Do not replace the fields with epoch integers or PostgreSQL-native timestamps. Epoch
41
+ columns would either lose sub-millisecond precision or introduce JavaScript-safe-range
42
+ considerations, while native types would create avoidable dialect and CSV differences.
43
+
44
+ ## Migration
45
+
46
+ 1. Add revision `0003` without changing column names or snapshot schema.
47
+ 2. Before mutation, read every non-null timestamp in bounded primary-key batches. Parse
48
+ it with the same strict helper and fail with a generic schema-compatibility error if
49
+ any published database contains an invalid or timezone-naive value.
50
+ 3. Rewrite valid values to fixed-width UTC form. This changes representation, never the
51
+ represented instant.
52
+ 4. Add an index on `conversations.ended_at`; retain the existing start/event/run
53
+ indexes.
54
+ 5. Change reporting overlap predicates from function-wrapped columns to direct text
55
+ predicates with explicit null branches. For example, “last activity at or after the
56
+ lower bound” becomes `ended_at >= :since OR (ended_at IS NULL AND started_at >=
57
+ :since)`.
58
+ 6. Change retention predicates in the same way and bind canonical UTC strings on both
59
+ dialects.
60
+ 7. Keep dashboard and CSV values as canonical ISO strings; no export column changes are
61
+ required.
62
+
63
+ The migration must run with writers stopped. An older writer could reintroduce a
64
+ non-canonical offset after revision `0003`, so the existing prohibition on mixed
65
+ application versions remains mandatory.
66
+
67
+ ## Downgrade and rollback
68
+
69
+ The downgrade removes the new `ended_at` index but leaves timestamps canonical. The
70
+ original offset spelling and fractional precision cannot be reconstructed, although
71
+ the instant is preserved exactly. Restore the pre-upgrade backup if byte-for-byte
72
+ lexical rollback is required.
73
+
74
+ ## Verification
75
+
76
+ - Property tests proving chronological and lexical ordering agree across offsets,
77
+ daylight-saving transitions, negative offsets, and microsecond boundaries.
78
+ - Snapshot/API tests proving equivalent offsets serialize identically.
79
+ - SQLite upgrade tests for valid legacy values, invalid-value fail-closed behavior,
80
+ idempotence, query results, and `EXPLAIN QUERY PLAN` index use.
81
+ - PostgreSQL migration and runtime tests for canonical backfill, direct predicates,
82
+ indexes, and transaction rollback on malformed legacy data.
83
+ - Export-window and retention regression tests around exact half-open boundaries and
84
+ null start/end combinations.
85
+ - Privacy assertions confirming rejected legacy values never appear in errors or logs.
86
+
87
+ ## Consequences
88
+
89
+ - No snapshot-version bump, public field rename, or CSV compatibility break.
90
+ - Range queries become indexable and dialect-specific timestamp casts disappear.
91
+ - Stored timestamps no longer retain the provider's original offset notation; CLI
92
+ Consumption analyzes instants, so that notation has no approved analytical purpose.
93
+ - Upgrading to revision `0003` performs a one-time bounded rewrite before the direct
94
+ query predicates are used.
@@ -52,6 +52,12 @@ records, constrain accepted fields, and never evaluate embedded content. The API
52
52
  constant-time bearer-token comparison when authentication is configured and refuses an
53
53
  unauthenticated non-local bind through the CLI.
54
54
 
55
+ Local parsing caps monolithic JSON files at 64 MiB, JSONL files at 256 MiB with an
56
+ 8 MiB per-line limit, and the normalized snapshot at 250,000 records during
57
+ construction. Direct provider file symlinks are rejected. Limit failures expose only
58
+ generic codes, never paths or record content. The sync client requires HTTPS beyond
59
+ loopback unless the operator uses the explicit `--allow-insecure` override.
60
+
55
61
  An exported database still reveals work patterns, model choices, project names, and
56
62
  activity times. Treat it as private operational data. Restrict filesystem and database
57
63
  access, use TLS for remote collection, rotate tokens, and define a retention policy
@@ -82,9 +88,14 @@ their own timestamp. CSV output neutralizes leading spreadsheet formula and cont
82
88
  prefixes with an apostrophe, but still contains detailed normalized operational data.
83
89
 
84
90
  Provider diagnostics inspect local stores transiently and emit only provider name,
85
- documented aliases, support state, and one coarse compatibility status. They do not
86
- persist snapshots or reveal paths, conversation identifiers, record counts, malformed
87
- values, or exception text.
91
+ documented aliases, support state, documented token semantics, and one coarse
92
+ compatibility status. They do not persist snapshots or reveal paths, conversation
93
+ identifiers, record counts, malformed values, or exception text.
94
+
95
+ Every registered adapter has a synthetic canary regression at the extraction boundary.
96
+ Shared contract tests then check that prohibited content is absent from snapshots, SQL,
97
+ API requests and errors, CSV, HTML, logs, and malformed-input diagnostics. This layered
98
+ contract is required for every newly registered provider.
88
99
 
89
100
  Retention removes normalized rows, not provider source files, existing exports,
90
101
  database backups, database engine logs, reverse-proxy logs, or snapshots already sent
@@ -33,8 +33,11 @@ ingests each metadata-only snapshot independently. With no explicit source it ch
33
33
  the local default homes; repeated `--source` paths are filtered by detected format.
34
34
 
35
35
  Provider metadata is maintained in one registry: canonical name, aliases, adapter,
36
- default home, detection markers, and support state. `providers --json` uses that same
37
- registry to check default local stores and emits deterministic schema-v1 JSON. Its
36
+ default home, detection markers, support state, and token semantics. `providers --json`
37
+ uses that same registry to check default local stores and emits deterministic schema-v2
38
+ JSON. Token semantics are one of `additive`, `conversation-aggregate`,
39
+ `context-snapshot`, or `unavailable`; a missing counter is never presented as a measured
40
+ zero in provider capability output. Its
38
41
  compatibility status is one of:
39
42
 
40
43
  - `no-data`: no registered detection marker was found;
@@ -14,6 +14,11 @@
14
14
  - Versioned SQLite/PostgreSQL migrations, safe legacy adoption, and retention previews
15
15
  - Central provider registry with privacy-minimized compatibility diagnostics
16
16
  - Deterministic, streamed, time-bounded exports with spreadsheet-safe CSV cells
17
+ - Exact legacy-schema adoption, bounded provider inputs, authoritative subagent-scope
18
+ replacement, and HTTPS-by-default synchronization
19
+ - Deterministic JSON results for collection, export, and retention automation
20
+ - Canonical fixed-width UTC timestamp storage with bounded legacy migration and
21
+ indexable reporting and retention predicates
17
22
 
18
23
  ## Next provider increments
19
24
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "cli-consumption"
7
- version = "0.2.0"
7
+ version = "0.2.1"
8
8
  description = "Analyze and consolidate AI coding CLI consumption across machines."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.12"
@@ -4,6 +4,7 @@ import hashlib
4
4
  import json
5
5
  import math
6
6
  import re
7
+ from collections.abc import Iterator
7
8
  from datetime import UTC, datetime
8
9
  from pathlib import Path
9
10
  from typing import Any
@@ -11,9 +12,16 @@ from typing import Any
11
12
  from cli_consumption.models import empty_tokens
12
13
 
13
14
  MAX_BIGINT = 9_223_372_036_854_775_807
15
+ MAX_PROVIDER_JSON_BYTES = 64 * 1024 * 1024
16
+ MAX_PROVIDER_JSONL_BYTES = 256 * 1024 * 1024
17
+ MAX_PROVIDER_JSONL_LINE_BYTES = 8 * 1024 * 1024
14
18
  SAFE_LABEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+@-]*")
15
19
 
16
20
 
21
+ class ProviderDataLimitError(ValueError):
22
+ """A privacy-safe failure raised when provider input exceeds a hard limit."""
23
+
24
+
17
25
  def mapping(value: object) -> dict[str, Any]:
18
26
  return value if isinstance(value, dict) else {}
19
27
 
@@ -138,4 +146,35 @@ def digest_records(records: object) -> str:
138
146
 
139
147
 
140
148
  def read_json(path: Path) -> object:
141
- return json.loads(path.read_text(encoding="utf-8"))
149
+ return json.loads(read_bounded_bytes(path))
150
+
151
+
152
+ def read_bounded_bytes(path: Path, maximum: int = MAX_PROVIDER_JSON_BYTES) -> bytes:
153
+ reject_provider_file_symlink(path)
154
+ if path.stat().st_size > maximum:
155
+ raise ProviderDataLimitError("provider_file_too_large")
156
+ with path.open("rb") as handle:
157
+ payload = handle.read(maximum + 1)
158
+ if len(payload) > maximum:
159
+ raise ProviderDataLimitError("provider_file_too_large")
160
+ return payload
161
+
162
+
163
+ def iter_bounded_jsonl_bytes(
164
+ path: Path,
165
+ maximum_line: int = MAX_PROVIDER_JSONL_LINE_BYTES,
166
+ maximum_file: int = MAX_PROVIDER_JSONL_BYTES,
167
+ ) -> Iterator[bytes]:
168
+ reject_provider_file_symlink(path)
169
+ if path.stat().st_size > maximum_file:
170
+ raise ProviderDataLimitError("provider_file_too_large")
171
+ with path.open("rb") as handle:
172
+ for line in handle:
173
+ if len(line) > maximum_line:
174
+ raise ProviderDataLimitError("provider_line_too_large")
175
+ yield line
176
+
177
+
178
+ def reject_provider_file_symlink(path: Path) -> None:
179
+ if path.is_symlink():
180
+ raise ProviderDataLimitError("provider_file_symlink_not_allowed")
@@ -9,6 +9,7 @@ from datetime import UTC, datetime
9
9
  from pathlib import Path
10
10
  from typing import Any
11
11
 
12
+ from cli_consumption.adapters._shared import iter_bounded_jsonl_bytes
12
13
  from cli_consumption.models import Snapshot, empty_tokens
13
14
 
14
15
  MAX_BIGINT = 9_223_372_036_854_775_807
@@ -185,32 +186,31 @@ def _read_sessions(
185
186
  active: list[dict[str, Any]] | None = None
186
187
  malformed = 0
187
188
  ordinals: dict[tuple[str, int], int] = {}
188
- with path.open("rb") as handle:
189
- for line in handle:
190
- if not line.strip():
191
- continue
192
- try:
193
- event = json.loads(line)
194
- except (json.JSONDecodeError, UnicodeDecodeError):
195
- malformed += 1
196
- continue
197
- if not isinstance(event, dict) or not isinstance(event.get("event"), str):
198
- malformed += 1
199
- continue
200
- if _epoch(event.get("time")) is None:
201
- malformed += 1
202
- continue
203
- if event["event"] == "launched":
204
- if active:
205
- sessions.append(_session(active, ordinals))
206
- active = [event]
207
- continue
208
- if active is None:
209
- continue
210
- active.append(event)
211
- if event["event"] == "exit":
189
+ for line in iter_bounded_jsonl_bytes(path):
190
+ if not line.strip():
191
+ continue
192
+ try:
193
+ event = json.loads(line)
194
+ except (json.JSONDecodeError, UnicodeDecodeError):
195
+ malformed += 1
196
+ continue
197
+ if not isinstance(event, dict) or not isinstance(event.get("event"), str):
198
+ malformed += 1
199
+ continue
200
+ if _epoch(event.get("time")) is None:
201
+ malformed += 1
202
+ continue
203
+ if event["event"] == "launched":
204
+ if active:
212
205
  sessions.append(_session(active, ordinals))
213
- active = None
206
+ active = [event]
207
+ continue
208
+ if active is None:
209
+ continue
210
+ active.append(event)
211
+ if event["event"] == "exit":
212
+ sessions.append(_session(active, ordinals))
213
+ active = None
214
214
  if active:
215
215
  sessions.append(_session(active, ordinals))
216
216
  return sessions, malformed
@@ -13,6 +13,7 @@ from cli_consumption.adapters._shared import (
13
13
  mapping,
14
14
  new_turn,
15
15
  project,
16
+ reject_provider_file_symlink,
16
17
  timestamp,
17
18
  )
18
19
  from cli_consumption.adapters.base import UnsupportedProviderFormat
@@ -63,6 +64,7 @@ def _read_database(path: Path) -> tuple[list[tuple[str, dict[str, Any]]], int]:
63
64
  malformed = 0
64
65
  result: list[tuple[str, dict[str, Any]]] = []
65
66
  try:
67
+ reject_provider_file_symlink(path)
66
68
  connection = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True)
67
69
  connection.row_factory = sqlite3.Row
68
70
  connection.execute("PRAGMA trusted_schema=OFF")