cli-consumption 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/CONTRIBUTING.md +6 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/PKG-INFO +35 -12
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/README.md +34 -11
- cli_consumption-0.2.1/SECURITY.md +37 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/architecture.md +26 -7
- cli_consumption-0.2.1/docs/decisions/0002-canonical-utc-timestamps.md +94 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/privacy.md +14 -3
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/provider-support.md +5 -2
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/roadmap.md +5 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/pyproject.toml +1 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/_shared.py +40 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/aider.py +25 -25
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/amazon_q.py +2 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/amp.py +2 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/claude.py +20 -21
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/cline.py +2 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/codex.py +25 -24
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/continue_cli.py +2 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/copilot.py +14 -14
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/crush.py +3 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/cursor.py +18 -14
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/gemini.py +17 -14
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/goose.py +2 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/grok.py +5 -6
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/kilo.py +2 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/kimi.py +2 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/mistral_vibe.py +18 -17
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/opencode.py +2 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/openhands.py +3 -2
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/pi.py +12 -12
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/plandex.py +2 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/qwen.py +14 -14
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/registry.py +96 -15
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/cli.py +108 -15
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/dashboard.py +10 -5
- cli_consumption-0.2.1/src/cli_consumption/migrations/versions/v0003_canonical_timestamps.py +94 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/models.py +70 -11
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/reporting.py +15 -30
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/retention.py +10 -14
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/schema.py +100 -6
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/storage.py +18 -5
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/sync.py +25 -0
- cli_consumption-0.2.1/src/cli_consumption/timestamps.py +17 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_api.py +4 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cli.py +91 -3
- cli_consumption-0.2.1/tests/test_input_limits.py +57 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_migrations_and_retention.py +232 -10
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_provider_registry.py +33 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_reporting.py +15 -7
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_storage_and_exports.py +42 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_sync.py +44 -0
- cli_consumption-0.2.1/tests/test_timestamps.py +61 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/uv.lock +1 -1
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/add-cli-adapter/SKILL.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/add-cli-adapter/agents/openai.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/audit-usage-privacy/SKILL.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/audit-usage-privacy/agents/openai.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/evolve-storage-schema/SKILL.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/evolve-storage-schema/agents/openai.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yeet-github/SKILL.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yeet-github/agents/openai.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yolo/SKILL.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.agents/skills/yolo/agents/openai.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.github/workflows/ci.yml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.github/workflows/release.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.gitignore +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.pre-commit-config.yaml +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/.python-version +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/AGENTS.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/LICENSE +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/NOTICE +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/docs/decisions/0001-versioned-schema-migrations.md +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/__init__.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/__main__.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/__init__.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/adapters/base.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/api.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/exporting.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/__init__.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/env.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/__init__.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/v0001_baseline.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/migrations/versions/v0002_minimize_subagents.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/src/cli_consumption/py.typed +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/conftest.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/smoke_minimal_install.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_aider_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_amazon_q_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_amp_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_claude_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cline_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_codex_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_continue_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_copilot_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_crush_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_cursor_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_exporting.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_gemini_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_goose_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_grok_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_kilo_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_kimi_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_mistral_vibe_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_opencode_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_openhands_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_packaging.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_pi_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_plandex_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_qwen_adapter.py +0 -0
- {cli_consumption-0.2.0 → cli_consumption-0.2.1}/tests/test_snapshot_contract.py +0 -0
|
@@ -30,6 +30,12 @@ errors, and the documented server-first upgrade sequence. Export changes must pr
|
|
|
30
30
|
deterministic ordering, bounded-memory CSV streaming, complete selected conversation
|
|
31
31
|
graphs, and spreadsheet-formula neutralization.
|
|
32
32
|
|
|
33
|
+
Provider readers must use the bounded file and JSONL helpers in `adapters/_shared.py`,
|
|
34
|
+
must not follow direct provider-file symlinks, and must stay within the shared snapshot
|
|
35
|
+
record budget. Every registered adapter test module must include a synthetic canary and
|
|
36
|
+
assert its absence from the normalized snapshot; shared privacy tests cover all later
|
|
37
|
+
output surfaces.
|
|
38
|
+
|
|
33
39
|
## Validate and review
|
|
34
40
|
|
|
35
41
|
Run the quality gates documented in `AGENTS.md`, inspect the complete diff, then open a
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cli-consumption
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Analyze and consolidate AI coding CLI consumption across machines.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
|
|
6
6
|
Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
|
|
@@ -41,7 +41,8 @@ machines, and can send metadata-only snapshots to a central collector.
|
|
|
41
41
|
|
|
42
42
|
It never stores prompts, responses, tool arguments, credentials, or raw provider
|
|
43
43
|
events. Local token counters are usage metadata, not billing records. Read the
|
|
44
|
-
[privacy boundary](docs/privacy.md)
|
|
44
|
+
[privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md)
|
|
45
|
+
before sharing a database or report.
|
|
45
46
|
|
|
46
47
|
## Quick start
|
|
47
48
|
|
|
@@ -120,7 +121,7 @@ Use the provider name below with `--provider` to select one CLI explicitly.
|
|
|
120
121
|
|
|
121
122
|
Provider formats are internal and can change without notice. The detailed extraction
|
|
122
123
|
rules and qualification versions are documented in
|
|
123
|
-
[Provider support](docs/provider-support.md).
|
|
124
|
+
[Provider support](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/provider-support.md).
|
|
124
125
|
|
|
125
126
|
## Collect copied data
|
|
126
127
|
|
|
@@ -142,6 +143,11 @@ Copy only the required provider data. For Codex, copy the `sessions/` directory
|
|
|
142
143
|
never `auth.json` or other credentials. Globally identical conversation IDs are
|
|
143
144
|
deduplicated, and the most complete copy wins.
|
|
144
145
|
|
|
146
|
+
Provider files are untrusted. Monolithic JSON files are limited to 64 MiB, JSONL files
|
|
147
|
+
to 256 MiB with an 8 MiB per-line limit, and a snapshot to 250,000 normalized records
|
|
148
|
+
while it is being built. Direct provider-file symlinks are refused. `collect --strict`
|
|
149
|
+
refuses to write a snapshot when malformed records were skipped.
|
|
150
|
+
|
|
145
151
|
Map original working-directory prefixes to stable project labels with repeated
|
|
146
152
|
`--project NAME=PATH_PREFIX` options. The longest matching prefix wins:
|
|
147
153
|
|
|
@@ -163,7 +169,7 @@ uv run cli-consumption collect --provider plandex \
|
|
|
163
169
|
|
|
164
170
|
The dashboard can filter by time, provider, machine, project, and model. It reports
|
|
165
171
|
activity, token composition, cache efficiency, latency and duration distributions,
|
|
166
|
-
|
|
172
|
+
turn rate, context pressure, work-item reliability, configuration cohorts,
|
|
167
173
|
compactions, subagent delegation, and ingestion quality. Availability varies by
|
|
168
174
|
provider, as summarized in the table above.
|
|
169
175
|
|
|
@@ -209,8 +215,14 @@ unversioned databases that exactly match a published schema are adopted before t
|
|
|
209
215
|
upgrade; unknown or modified schemas are refused. Back up production databases before
|
|
210
216
|
upgrading and do not run mixed application versions against one database while a
|
|
211
217
|
migration is in progress. See the
|
|
212
|
-
[migration decision](docs/decisions/0001-versioned-schema-migrations.md)
|
|
213
|
-
and compatibility rules.
|
|
218
|
+
[migration decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0001-versioned-schema-migrations.md)
|
|
219
|
+
for rollback and compatibility rules.
|
|
220
|
+
|
|
221
|
+
Timezone-aware timestamps are normalized to fixed-width UTC strings during ingestion.
|
|
222
|
+
Revision `0003` rewrites legacy timestamp text in bounded batches and adds an indexed
|
|
223
|
+
conversation end-time path; see the
|
|
224
|
+
[timestamp decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0002-canonical-utc-timestamps.md)
|
|
225
|
+
for the exact representation and downgrade boundary.
|
|
214
226
|
|
|
215
227
|
Preview retention before deleting normalized metadata:
|
|
216
228
|
|
|
@@ -243,8 +255,9 @@ uv run cli-consumption sync --provider all \
|
|
|
243
255
|
```
|
|
244
256
|
|
|
245
257
|
The application refuses to bind beyond localhost without a token. Production
|
|
246
|
-
deployments also need TLS and standard operational controls.
|
|
247
|
-
|
|
258
|
+
deployments also need TLS and standard operational controls. The sync client refuses
|
|
259
|
+
plain HTTP beyond loopback unless `--allow-insecure` is passed explicitly. See
|
|
260
|
+
[Architecture](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/architecture.md) for the trade-offs.
|
|
248
261
|
|
|
249
262
|
Snapshots use strict schema version 1. The collector rejects request bodies larger
|
|
250
263
|
than 32 MiB and snapshots containing more than 250,000 normalized records. A sync
|
|
@@ -263,7 +276,10 @@ uv run cli-consumption providers --json
|
|
|
263
276
|
Each provider reports one of `no-data`, `detected`, `compatible`, `degraded`, or
|
|
264
277
|
`unsupported-schema`. Diagnostics parse enough metadata to assess compatibility but do
|
|
265
278
|
not persist it and never include paths, identifiers, record contents, counts, or parser
|
|
266
|
-
errors in their output.
|
|
279
|
+
errors in their output. Schema version 2 also declares whether token counters are
|
|
280
|
+
additive, conversation aggregates, context snapshots, or unavailable. Dashboard token
|
|
281
|
+
per-turn percentiles use only additive providers rather than treating missing measures
|
|
282
|
+
as zero.
|
|
267
283
|
|
|
268
284
|
## Commands
|
|
269
285
|
|
|
@@ -278,6 +294,10 @@ errors in their output.
|
|
|
278
294
|
|
|
279
295
|
Run `uv run cli-consumption COMMAND --help` for all options.
|
|
280
296
|
|
|
297
|
+
`collect`, `export`, and `retention` accept `--json` for deterministic
|
|
298
|
+
machine-readable results. `collect --strict` rejects snapshots containing malformed
|
|
299
|
+
provider records before opening the destination database.
|
|
300
|
+
|
|
281
301
|
## Development
|
|
282
302
|
|
|
283
303
|
```bash
|
|
@@ -292,9 +312,12 @@ uv build
|
|
|
292
312
|
```
|
|
293
313
|
|
|
294
314
|
Development uses short-lived branches and squash-merged pull requests into protected
|
|
295
|
-
`main`. Read
|
|
296
|
-
|
|
315
|
+
`main`. Read
|
|
316
|
+
[CONTRIBUTING.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/CONTRIBUTING.md)
|
|
317
|
+
and [AGENTS.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/AGENTS.md)
|
|
318
|
+
before changing the project. Security issues follow the private reporting guidance in
|
|
319
|
+
[SECURITY.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/SECURITY.md).
|
|
297
320
|
|
|
298
321
|
## License
|
|
299
322
|
|
|
300
|
-
Licensed under the [Apache License 2.0](LICENSE).
|
|
323
|
+
Licensed under the [Apache License 2.0](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/LICENSE).
|
|
@@ -6,7 +6,8 @@ machines, and can send metadata-only snapshots to a central collector.
|
|
|
6
6
|
|
|
7
7
|
It never stores prompts, responses, tool arguments, credentials, or raw provider
|
|
8
8
|
events. Local token counters are usage metadata, not billing records. Read the
|
|
9
|
-
[privacy boundary](docs/privacy.md)
|
|
9
|
+
[privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md)
|
|
10
|
+
before sharing a database or report.
|
|
10
11
|
|
|
11
12
|
## Quick start
|
|
12
13
|
|
|
@@ -85,7 +86,7 @@ Use the provider name below with `--provider` to select one CLI explicitly.
|
|
|
85
86
|
|
|
86
87
|
Provider formats are internal and can change without notice. The detailed extraction
|
|
87
88
|
rules and qualification versions are documented in
|
|
88
|
-
[Provider support](docs/provider-support.md).
|
|
89
|
+
[Provider support](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/provider-support.md).
|
|
89
90
|
|
|
90
91
|
## Collect copied data
|
|
91
92
|
|
|
@@ -107,6 +108,11 @@ Copy only the required provider data. For Codex, copy the `sessions/` directory
|
|
|
107
108
|
never `auth.json` or other credentials. Globally identical conversation IDs are
|
|
108
109
|
deduplicated, and the most complete copy wins.
|
|
109
110
|
|
|
111
|
+
Provider files are untrusted. Monolithic JSON files are limited to 64 MiB, JSONL files
|
|
112
|
+
to 256 MiB with an 8 MiB per-line limit, and a snapshot to 250,000 normalized records
|
|
113
|
+
while it is being built. Direct provider-file symlinks are refused. `collect --strict`
|
|
114
|
+
refuses to write a snapshot when malformed records were skipped.
|
|
115
|
+
|
|
110
116
|
Map original working-directory prefixes to stable project labels with repeated
|
|
111
117
|
`--project NAME=PATH_PREFIX` options. The longest matching prefix wins:
|
|
112
118
|
|
|
@@ -128,7 +134,7 @@ uv run cli-consumption collect --provider plandex \
|
|
|
128
134
|
|
|
129
135
|
The dashboard can filter by time, provider, machine, project, and model. It reports
|
|
130
136
|
activity, token composition, cache efficiency, latency and duration distributions,
|
|
131
|
-
|
|
137
|
+
turn rate, context pressure, work-item reliability, configuration cohorts,
|
|
132
138
|
compactions, subagent delegation, and ingestion quality. Availability varies by
|
|
133
139
|
provider, as summarized in the table above.
|
|
134
140
|
|
|
@@ -174,8 +180,14 @@ unversioned databases that exactly match a published schema are adopted before t
|
|
|
174
180
|
upgrade; unknown or modified schemas are refused. Back up production databases before
|
|
175
181
|
upgrading and do not run mixed application versions against one database while a
|
|
176
182
|
migration is in progress. See the
|
|
177
|
-
[migration decision](docs/decisions/0001-versioned-schema-migrations.md)
|
|
178
|
-
and compatibility rules.
|
|
183
|
+
[migration decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0001-versioned-schema-migrations.md)
|
|
184
|
+
for rollback and compatibility rules.
|
|
185
|
+
|
|
186
|
+
Timezone-aware timestamps are normalized to fixed-width UTC strings during ingestion.
|
|
187
|
+
Revision `0003` rewrites legacy timestamp text in bounded batches and adds an indexed
|
|
188
|
+
conversation end-time path; see the
|
|
189
|
+
[timestamp decision](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/decisions/0002-canonical-utc-timestamps.md)
|
|
190
|
+
for the exact representation and downgrade boundary.
|
|
179
191
|
|
|
180
192
|
Preview retention before deleting normalized metadata:
|
|
181
193
|
|
|
@@ -208,8 +220,9 @@ uv run cli-consumption sync --provider all \
|
|
|
208
220
|
```
|
|
209
221
|
|
|
210
222
|
The application refuses to bind beyond localhost without a token. Production
|
|
211
|
-
deployments also need TLS and standard operational controls.
|
|
212
|
-
|
|
223
|
+
deployments also need TLS and standard operational controls. The sync client refuses
|
|
224
|
+
plain HTTP beyond loopback unless `--allow-insecure` is passed explicitly. See
|
|
225
|
+
[Architecture](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/architecture.md) for the trade-offs.
|
|
213
226
|
|
|
214
227
|
Snapshots use strict schema version 1. The collector rejects request bodies larger
|
|
215
228
|
than 32 MiB and snapshots containing more than 250,000 normalized records. A sync
|
|
@@ -228,7 +241,10 @@ uv run cli-consumption providers --json
|
|
|
228
241
|
Each provider reports one of `no-data`, `detected`, `compatible`, `degraded`, or
|
|
229
242
|
`unsupported-schema`. Diagnostics parse enough metadata to assess compatibility but do
|
|
230
243
|
not persist it and never include paths, identifiers, record contents, counts, or parser
|
|
231
|
-
errors in their output.
|
|
244
|
+
errors in their output. Schema version 2 also declares whether token counters are
|
|
245
|
+
additive, conversation aggregates, context snapshots, or unavailable. Dashboard token
|
|
246
|
+
per-turn percentiles use only additive providers rather than treating missing measures
|
|
247
|
+
as zero.
|
|
232
248
|
|
|
233
249
|
## Commands
|
|
234
250
|
|
|
@@ -243,6 +259,10 @@ errors in their output.
|
|
|
243
259
|
|
|
244
260
|
Run `uv run cli-consumption COMMAND --help` for all options.
|
|
245
261
|
|
|
262
|
+
`collect`, `export`, and `retention` accept `--json` for deterministic
|
|
263
|
+
machine-readable results. `collect --strict` rejects snapshots containing malformed
|
|
264
|
+
provider records before opening the destination database.
|
|
265
|
+
|
|
246
266
|
## Development
|
|
247
267
|
|
|
248
268
|
```bash
|
|
@@ -257,9 +277,12 @@ uv build
|
|
|
257
277
|
```
|
|
258
278
|
|
|
259
279
|
Development uses short-lived branches and squash-merged pull requests into protected
|
|
260
|
-
`main`. Read
|
|
261
|
-
|
|
280
|
+
`main`. Read
|
|
281
|
+
[CONTRIBUTING.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/CONTRIBUTING.md)
|
|
282
|
+
and [AGENTS.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/AGENTS.md)
|
|
283
|
+
before changing the project. Security issues follow the private reporting guidance in
|
|
284
|
+
[SECURITY.md](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/SECURITY.md).
|
|
262
285
|
|
|
263
286
|
## License
|
|
264
287
|
|
|
265
|
-
Licensed under the [Apache License 2.0](LICENSE).
|
|
288
|
+
Licensed under the [Apache License 2.0](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/LICENSE).
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Security policy
|
|
2
|
+
|
|
3
|
+
## Supported versions
|
|
4
|
+
|
|
5
|
+
Security fixes are made on the latest released version and on `main`. Older releases
|
|
6
|
+
are not maintained as separate security branches. Upgrade to the latest release before
|
|
7
|
+
reporting a problem that may already have been fixed.
|
|
8
|
+
|
|
9
|
+
## Reporting a vulnerability
|
|
10
|
+
|
|
11
|
+
Do not open a public issue for a suspected vulnerability. Email
|
|
12
|
+
`lombardo.guillaume@gmail.com` with the subject `CLI Consumption security report` and
|
|
13
|
+
include:
|
|
14
|
+
|
|
15
|
+
- the affected version and operating system;
|
|
16
|
+
- the provider, command, or API surface involved;
|
|
17
|
+
- reproduction steps using synthetic data;
|
|
18
|
+
- the expected and observed impact;
|
|
19
|
+
- any suggested mitigation.
|
|
20
|
+
|
|
21
|
+
Do not send real prompts, responses, credentials, provider databases, or other private
|
|
22
|
+
conversation data. Use a synthetic canary when demonstrating a disclosure.
|
|
23
|
+
|
|
24
|
+
The maintainer will acknowledge the report, assess whether it crosses the documented
|
|
25
|
+
privacy or security boundary, and coordinate a fix and disclosure when appropriate.
|
|
26
|
+
|
|
27
|
+
## Security boundary
|
|
28
|
+
|
|
29
|
+
Provider files and incoming snapshots are untrusted input. Relevant reports include
|
|
30
|
+
content or credential disclosure, unsafe filesystem traversal, denial of service,
|
|
31
|
+
authentication bypass, cross-machine data corruption, and generated dashboards that
|
|
32
|
+
perform network requests or execute provider-controlled content.
|
|
33
|
+
|
|
34
|
+
Operational exposure caused solely by publishing a normalized database or detailed CSV
|
|
35
|
+
is outside the vulnerability boundary: those artifacts intentionally contain private
|
|
36
|
+
operational metadata. The precise allowed and prohibited fields are documented in the
|
|
37
|
+
[privacy boundary](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/privacy.md).
|
|
@@ -32,7 +32,7 @@ provider files -> adapter -> metadata-only snapshot -> SQL storage -> dashboard/
|
|
|
32
32
|
provide an offline HTML view by default, and stream deterministic portable CSV tables
|
|
33
33
|
when explicitly requested.
|
|
34
34
|
- `adapters.registry`: is the single source for canonical names, aliases, adapter
|
|
35
|
-
classes, default homes, detection markers, and
|
|
35
|
+
classes, default homes, detection markers, support state, and token semantics.
|
|
36
36
|
- `cli`: exposes the same capabilities through one executable.
|
|
37
37
|
|
|
38
38
|
When Codex exposes its local thread graph, the adapter also records metadata-only
|
|
@@ -64,28 +64,41 @@ or platform ingress. Read access is not exposed. The collector limits bodies to
|
|
|
64
64
|
including chunked requests, accepts snapshot schema v1, and caps a snapshot at 250,000
|
|
65
65
|
normalized records. `/api/v1/capabilities` publishes these limits and the supported
|
|
66
66
|
schema range so clients can fail before uploading incompatible data.
|
|
67
|
+
The sync client refuses plain HTTP beyond loopback unless the operator passes the
|
|
68
|
+
explicit `--allow-insecure` override for a trusted network.
|
|
67
69
|
|
|
68
70
|
## Storage
|
|
69
71
|
|
|
70
72
|
SQLite is the zero-configuration default. PostgreSQL is selected by passing a
|
|
71
73
|
`postgresql+psycopg://` URL. Every database open upgrades through packaged Alembic
|
|
72
|
-
migrations. An unversioned database is adopted only when its
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
74
|
+
migrations. An unversioned database is adopted only when its columns, SQL types,
|
|
75
|
+
nullability, primary and foreign keys, and indexes exactly match a published schema; an
|
|
76
|
+
unknown, newer, or locally modified schema is refused. Migration revisions support
|
|
77
|
+
SQLite and PostgreSQL and define a bounded downgrade, but application rollback can
|
|
78
|
+
still require restoring a pre-upgrade backup. Operators must upgrade the server first
|
|
79
|
+
and avoid mixed-version access during migration. The detailed policy is recorded in
|
|
80
|
+
[ADR 0001](decisions/0001-versioned-schema-migrations.md).
|
|
78
81
|
|
|
79
82
|
Conversation records use a provider-qualified stable ID. Repeated ingestion skips an
|
|
80
83
|
identical or less complete record. A more complete copy atomically replaces the
|
|
81
84
|
conversation and its child records.
|
|
82
85
|
|
|
86
|
+
Subagent relationships have a provider-and-source-machine lifecycle. Every represented
|
|
87
|
+
scope is replaced atomically on ingestion, including when its latest snapshot contains
|
|
88
|
+
no edges, so deleted provider relationships do not remain indefinitely.
|
|
89
|
+
|
|
83
90
|
Workflow analytics use additive child tables: `work_items`, `context_samples`,
|
|
84
91
|
`turn_settings`, and `compaction_events`. Snapshot schema v1 validates every record,
|
|
85
92
|
rejects unknown fields, enforces normalized labels and relationships, and exposes only
|
|
86
93
|
generic validation errors. A newer client sent to an older strict API is rejected
|
|
87
94
|
before ingestion, so central deployments must upgrade the server first.
|
|
88
95
|
|
|
96
|
+
Provider input is bounded before persistence. Monolithic JSON files are capped at
|
|
97
|
+
64 MiB, JSONL files at 256 MiB with an 8 MiB per-line limit, and the complete in-memory
|
|
98
|
+
normalized snapshot at 250,000 records. Direct symlinks to provider files are rejected.
|
|
99
|
+
These limits use generic error codes and apply to local collection as well as snapshots
|
|
100
|
+
later sent to the API.
|
|
101
|
+
|
|
89
102
|
Retention is an explicit two-step operation: `retention --keep-days N` reports what
|
|
90
103
|
would be deleted, while `--apply` deletes old conversations (with cascading children),
|
|
91
104
|
subagent relationships, and ingestion runs in one transaction.
|
|
@@ -103,3 +116,9 @@ each table, and neutralizes text that spreadsheet software could execute as a fo
|
|
|
103
116
|
Adapters are introduced one at a time because local data formats are undocumented or
|
|
104
117
|
can evolve independently. Each adapter must have synthetic fixtures, format detection,
|
|
105
118
|
privacy tests, and a documented support level before it appears as supported.
|
|
119
|
+
|
|
120
|
+
Timestamp fields are canonical fixed-width UTC strings after strict snapshot
|
|
121
|
+
validation and schema revision `0003`. Reporting and retention bind values in the same
|
|
122
|
+
form and use direct indexed predicates with explicit null branches. The representation,
|
|
123
|
+
migration, and rollback boundary are documented in
|
|
124
|
+
[ADR 0002](decisions/0002-canonical-utc-timestamps.md).
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# ADR 0002: Canonical UTC timestamp storage
|
|
2
|
+
|
|
3
|
+
- Status: Accepted and implemented in schema revision `0003`
|
|
4
|
+
- Scope: normalized SQL timestamps and time-window queries
|
|
5
|
+
|
|
6
|
+
## Context
|
|
7
|
+
|
|
8
|
+
Snapshot schema v1 accepts timezone-aware ISO 8601 timestamps. Before revision `0003`,
|
|
9
|
+
SQL stored those values as text exactly as received. Reporting and retention therefore
|
|
10
|
+
called SQLite `datetime(...)` or cast PostgreSQL text to `timestamptz` for every
|
|
11
|
+
comparison. Those expressions prevented the existing timestamp indexes from serving
|
|
12
|
+
the common range predicates. Equivalent instants could also have different lexical
|
|
13
|
+
forms because offsets and fractional-second precision were not canonicalized.
|
|
14
|
+
|
|
15
|
+
The affected SQL fields are conversation and turn start/end times, model/tool/context/
|
|
16
|
+
compaction event timestamps, and ingestion-run timestamps. Millisecond epoch fields on
|
|
17
|
+
work items and subagent relationships already have a separate, explicit provider
|
|
18
|
+
meaning and are outside this decision.
|
|
19
|
+
|
|
20
|
+
An SQLite 3.53.1 query-plan check against the published schema confirmed that
|
|
21
|
+
`datetime(coalesce(...))` performs a table scan, while a direct comparison against a
|
|
22
|
+
canonical timestamp uses the timestamp index. This is a query-shape limitation, not a
|
|
23
|
+
SQLite version-specific parsing bug.
|
|
24
|
+
|
|
25
|
+
## Decision
|
|
26
|
+
|
|
27
|
+
Keep timezone-aware ISO strings in snapshot schema v1 and in CSV exports, but make the
|
|
28
|
+
stored representation canonical:
|
|
29
|
+
|
|
30
|
+
```text
|
|
31
|
+
YYYY-MM-DDTHH:MM:SS.ffffff+00:00
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
At the strict snapshot boundary, parse each timestamp, convert it to UTC, and serialize
|
|
35
|
+
it with `datetime.isoformat(timespec="microseconds")`. Generate ingestion timestamps
|
|
36
|
+
through the same helper. Fixed-width UTC strings sort in instant order on both SQLite
|
|
37
|
+
and PostgreSQL, remain human-readable, preserve Python datetime precision, fit the
|
|
38
|
+
existing `VARCHAR(64)` columns, and stay compatible with snapshot schema v1.
|
|
39
|
+
|
|
40
|
+
Do not replace the fields with epoch integers or PostgreSQL-native timestamps. Epoch
|
|
41
|
+
columns would either lose sub-millisecond precision or introduce JavaScript-safe-range
|
|
42
|
+
considerations, while native types would create avoidable dialect and CSV differences.
|
|
43
|
+
|
|
44
|
+
## Migration
|
|
45
|
+
|
|
46
|
+
1. Add revision `0003` without changing column names or snapshot schema.
|
|
47
|
+
2. Before mutation, read every non-null timestamp in bounded primary-key batches. Parse
|
|
48
|
+
it with the same strict helper and fail with a generic schema-compatibility error if
|
|
49
|
+
any published database contains an invalid or timezone-naive value.
|
|
50
|
+
3. Rewrite valid values to fixed-width UTC form. This changes representation, never the
|
|
51
|
+
represented instant.
|
|
52
|
+
4. Add an index on `conversations.ended_at`; retain the existing start/event/run
|
|
53
|
+
indexes.
|
|
54
|
+
5. Change reporting overlap predicates from function-wrapped columns to direct text
|
|
55
|
+
predicates with explicit null branches. For example, “last activity at or after the
|
|
56
|
+
lower bound” becomes `ended_at >= :since OR (ended_at IS NULL AND started_at >=
|
|
57
|
+
:since)`.
|
|
58
|
+
6. Change retention predicates in the same way and bind canonical UTC strings on both
|
|
59
|
+
dialects.
|
|
60
|
+
7. Keep dashboard and CSV values as canonical ISO strings; no export column changes are
|
|
61
|
+
required.
|
|
62
|
+
|
|
63
|
+
The migration must run with writers stopped. An older writer could reintroduce a
|
|
64
|
+
non-canonical offset after revision `0003`, so the existing prohibition on mixed
|
|
65
|
+
application versions remains mandatory.
|
|
66
|
+
|
|
67
|
+
## Downgrade and rollback
|
|
68
|
+
|
|
69
|
+
The downgrade removes the new `ended_at` index but leaves timestamps canonical. The
|
|
70
|
+
original offset spelling and fractional precision cannot be reconstructed, although
|
|
71
|
+
the instant is preserved exactly. Restore the pre-upgrade backup if byte-for-byte
|
|
72
|
+
lexical rollback is required.
|
|
73
|
+
|
|
74
|
+
## Verification
|
|
75
|
+
|
|
76
|
+
- Property tests proving chronological and lexical ordering agree across offsets,
|
|
77
|
+
daylight-saving transitions, negative offsets, and microsecond boundaries.
|
|
78
|
+
- Snapshot/API tests proving equivalent offsets serialize identically.
|
|
79
|
+
- SQLite upgrade tests for valid legacy values, invalid-value fail-closed behavior,
|
|
80
|
+
idempotence, query results, and `EXPLAIN QUERY PLAN` index use.
|
|
81
|
+
- PostgreSQL migration and runtime tests for canonical backfill, direct predicates,
|
|
82
|
+
indexes, and transaction rollback on malformed legacy data.
|
|
83
|
+
- Export-window and retention regression tests around exact half-open boundaries and
|
|
84
|
+
null start/end combinations.
|
|
85
|
+
- Privacy assertions confirming rejected legacy values never appear in errors or logs.
|
|
86
|
+
|
|
87
|
+
## Consequences
|
|
88
|
+
|
|
89
|
+
- No snapshot-version bump, public field rename, or CSV compatibility break.
|
|
90
|
+
- Range queries become indexable and dialect-specific timestamp casts disappear.
|
|
91
|
+
- Stored timestamps no longer retain the provider's original offset notation; CLI
|
|
92
|
+
Consumption analyzes instants, so that notation has no approved analytical purpose.
|
|
93
|
+
- Upgrading to revision `0003` performs a one-time bounded rewrite before the direct
|
|
94
|
+
query predicates are used.
|
|
@@ -52,6 +52,12 @@ records, constrain accepted fields, and never evaluate embedded content. The API
|
|
|
52
52
|
constant-time bearer-token comparison when authentication is configured and refuses an
|
|
53
53
|
unauthenticated non-local bind through the CLI.
|
|
54
54
|
|
|
55
|
+
Local parsing caps monolithic JSON files at 64 MiB, JSONL files at 256 MiB with an
|
|
56
|
+
8 MiB per-line limit, and the normalized snapshot at 250,000 records during
|
|
57
|
+
construction. Direct provider file symlinks are rejected. Limit failures expose only
|
|
58
|
+
generic codes, never paths or record content. The sync client requires HTTPS beyond
|
|
59
|
+
loopback unless the operator uses the explicit `--allow-insecure` override.
|
|
60
|
+
|
|
55
61
|
An exported database still reveals work patterns, model choices, project names, and
|
|
56
62
|
activity times. Treat it as private operational data. Restrict filesystem and database
|
|
57
63
|
access, use TLS for remote collection, rotate tokens, and define a retention policy
|
|
@@ -82,9 +88,14 @@ their own timestamp. CSV output neutralizes leading spreadsheet formula and cont
|
|
|
82
88
|
prefixes with an apostrophe, but still contains detailed normalized operational data.
|
|
83
89
|
|
|
84
90
|
Provider diagnostics inspect local stores transiently and emit only provider name,
|
|
85
|
-
documented aliases, support state, and one coarse
|
|
86
|
-
persist snapshots or reveal paths, conversation
|
|
87
|
-
values, or exception text.
|
|
91
|
+
documented aliases, support state, documented token semantics, and one coarse
|
|
92
|
+
compatibility status. They do not persist snapshots or reveal paths, conversation
|
|
93
|
+
identifiers, record counts, malformed values, or exception text.
|
|
94
|
+
|
|
95
|
+
Every registered adapter has a synthetic canary regression at the extraction boundary.
|
|
96
|
+
Shared contract tests then check that prohibited content is absent from snapshots, SQL,
|
|
97
|
+
API requests and errors, CSV, HTML, logs, and malformed-input diagnostics. This layered
|
|
98
|
+
contract is required for every newly registered provider.
|
|
88
99
|
|
|
89
100
|
Retention removes normalized rows, not provider source files, existing exports,
|
|
90
101
|
database backups, database engine logs, reverse-proxy logs, or snapshots already sent
|
|
@@ -33,8 +33,11 @@ ingests each metadata-only snapshot independently. With no explicit source it ch
|
|
|
33
33
|
the local default homes; repeated `--source` paths are filtered by detected format.
|
|
34
34
|
|
|
35
35
|
Provider metadata is maintained in one registry: canonical name, aliases, adapter,
|
|
36
|
-
default home, detection markers, and
|
|
37
|
-
registry to check default local stores and emits deterministic schema-
|
|
36
|
+
default home, detection markers, support state, and token semantics. `providers --json`
|
|
37
|
+
uses that same registry to check default local stores and emits deterministic schema-v2
|
|
38
|
+
JSON. Token semantics are one of `additive`, `conversation-aggregate`,
|
|
39
|
+
`context-snapshot`, or `unavailable`; a missing counter is never presented as a measured
|
|
40
|
+
zero in provider capability output. Its
|
|
38
41
|
compatibility status is one of:
|
|
39
42
|
|
|
40
43
|
- `no-data`: no registered detection marker was found;
|
|
@@ -14,6 +14,11 @@
|
|
|
14
14
|
- Versioned SQLite/PostgreSQL migrations, safe legacy adoption, and retention previews
|
|
15
15
|
- Central provider registry with privacy-minimized compatibility diagnostics
|
|
16
16
|
- Deterministic, streamed, time-bounded exports with spreadsheet-safe CSV cells
|
|
17
|
+
- Exact legacy-schema adoption, bounded provider inputs, authoritative subagent-scope
|
|
18
|
+
replacement, and HTTPS-by-default synchronization
|
|
19
|
+
- Deterministic JSON results for collection, export, and retention automation
|
|
20
|
+
- Canonical fixed-width UTC timestamp storage with bounded legacy migration and
|
|
21
|
+
indexable reporting and retention predicates
|
|
17
22
|
|
|
18
23
|
## Next provider increments
|
|
19
24
|
|
|
@@ -4,6 +4,7 @@ import hashlib
|
|
|
4
4
|
import json
|
|
5
5
|
import math
|
|
6
6
|
import re
|
|
7
|
+
from collections.abc import Iterator
|
|
7
8
|
from datetime import UTC, datetime
|
|
8
9
|
from pathlib import Path
|
|
9
10
|
from typing import Any
|
|
@@ -11,9 +12,16 @@ from typing import Any
|
|
|
11
12
|
from cli_consumption.models import empty_tokens
|
|
12
13
|
|
|
13
14
|
MAX_BIGINT = 9_223_372_036_854_775_807
|
|
15
|
+
MAX_PROVIDER_JSON_BYTES = 64 * 1024 * 1024
|
|
16
|
+
MAX_PROVIDER_JSONL_BYTES = 256 * 1024 * 1024
|
|
17
|
+
MAX_PROVIDER_JSONL_LINE_BYTES = 8 * 1024 * 1024
|
|
14
18
|
SAFE_LABEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+@-]*")
|
|
15
19
|
|
|
16
20
|
|
|
21
|
+
class ProviderDataLimitError(ValueError):
|
|
22
|
+
"""A privacy-safe failure raised when provider input exceeds a hard limit."""
|
|
23
|
+
|
|
24
|
+
|
|
17
25
|
def mapping(value: object) -> dict[str, Any]:
|
|
18
26
|
return value if isinstance(value, dict) else {}
|
|
19
27
|
|
|
@@ -138,4 +146,35 @@ def digest_records(records: object) -> str:
|
|
|
138
146
|
|
|
139
147
|
|
|
140
148
|
def read_json(path: Path) -> object:
|
|
141
|
-
return json.loads(path
|
|
149
|
+
return json.loads(read_bounded_bytes(path))
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def read_bounded_bytes(path: Path, maximum: int = MAX_PROVIDER_JSON_BYTES) -> bytes:
|
|
153
|
+
reject_provider_file_symlink(path)
|
|
154
|
+
if path.stat().st_size > maximum:
|
|
155
|
+
raise ProviderDataLimitError("provider_file_too_large")
|
|
156
|
+
with path.open("rb") as handle:
|
|
157
|
+
payload = handle.read(maximum + 1)
|
|
158
|
+
if len(payload) > maximum:
|
|
159
|
+
raise ProviderDataLimitError("provider_file_too_large")
|
|
160
|
+
return payload
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def iter_bounded_jsonl_bytes(
|
|
164
|
+
path: Path,
|
|
165
|
+
maximum_line: int = MAX_PROVIDER_JSONL_LINE_BYTES,
|
|
166
|
+
maximum_file: int = MAX_PROVIDER_JSONL_BYTES,
|
|
167
|
+
) -> Iterator[bytes]:
|
|
168
|
+
reject_provider_file_symlink(path)
|
|
169
|
+
if path.stat().st_size > maximum_file:
|
|
170
|
+
raise ProviderDataLimitError("provider_file_too_large")
|
|
171
|
+
with path.open("rb") as handle:
|
|
172
|
+
for line in handle:
|
|
173
|
+
if len(line) > maximum_line:
|
|
174
|
+
raise ProviderDataLimitError("provider_line_too_large")
|
|
175
|
+
yield line
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def reject_provider_file_symlink(path: Path) -> None:
|
|
179
|
+
if path.is_symlink():
|
|
180
|
+
raise ProviderDataLimitError("provider_file_symlink_not_allowed")
|
|
@@ -9,6 +9,7 @@ from datetime import UTC, datetime
|
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
from typing import Any
|
|
11
11
|
|
|
12
|
+
from cli_consumption.adapters._shared import iter_bounded_jsonl_bytes
|
|
12
13
|
from cli_consumption.models import Snapshot, empty_tokens
|
|
13
14
|
|
|
14
15
|
MAX_BIGINT = 9_223_372_036_854_775_807
|
|
@@ -185,32 +186,31 @@ def _read_sessions(
|
|
|
185
186
|
active: list[dict[str, Any]] | None = None
|
|
186
187
|
malformed = 0
|
|
187
188
|
ordinals: dict[tuple[str, int], int] = {}
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
if
|
|
204
|
-
if active:
|
|
205
|
-
sessions.append(_session(active, ordinals))
|
|
206
|
-
active = [event]
|
|
207
|
-
continue
|
|
208
|
-
if active is None:
|
|
209
|
-
continue
|
|
210
|
-
active.append(event)
|
|
211
|
-
if event["event"] == "exit":
|
|
189
|
+
for line in iter_bounded_jsonl_bytes(path):
|
|
190
|
+
if not line.strip():
|
|
191
|
+
continue
|
|
192
|
+
try:
|
|
193
|
+
event = json.loads(line)
|
|
194
|
+
except (json.JSONDecodeError, UnicodeDecodeError):
|
|
195
|
+
malformed += 1
|
|
196
|
+
continue
|
|
197
|
+
if not isinstance(event, dict) or not isinstance(event.get("event"), str):
|
|
198
|
+
malformed += 1
|
|
199
|
+
continue
|
|
200
|
+
if _epoch(event.get("time")) is None:
|
|
201
|
+
malformed += 1
|
|
202
|
+
continue
|
|
203
|
+
if event["event"] == "launched":
|
|
204
|
+
if active:
|
|
212
205
|
sessions.append(_session(active, ordinals))
|
|
213
|
-
|
|
206
|
+
active = [event]
|
|
207
|
+
continue
|
|
208
|
+
if active is None:
|
|
209
|
+
continue
|
|
210
|
+
active.append(event)
|
|
211
|
+
if event["event"] == "exit":
|
|
212
|
+
sessions.append(_session(active, ordinals))
|
|
213
|
+
active = None
|
|
214
214
|
if active:
|
|
215
215
|
sessions.append(_session(active, ordinals))
|
|
216
216
|
return sessions, malformed
|
|
@@ -13,6 +13,7 @@ from cli_consumption.adapters._shared import (
|
|
|
13
13
|
mapping,
|
|
14
14
|
new_turn,
|
|
15
15
|
project,
|
|
16
|
+
reject_provider_file_symlink,
|
|
16
17
|
timestamp,
|
|
17
18
|
)
|
|
18
19
|
from cli_consumption.adapters.base import UnsupportedProviderFormat
|
|
@@ -63,6 +64,7 @@ def _read_database(path: Path) -> tuple[list[tuple[str, dict[str, Any]]], int]:
|
|
|
63
64
|
malformed = 0
|
|
64
65
|
result: list[tuple[str, dict[str, Any]]] = []
|
|
65
66
|
try:
|
|
67
|
+
reject_provider_file_symlink(path)
|
|
66
68
|
connection = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True)
|
|
67
69
|
connection.row_factory = sqlite3.Row
|
|
68
70
|
connection.execute("PRAGMA trusted_schema=OFF")
|