cli-consumption 0.4.2__tar.gz → 0.4.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/CHANGELOG.md +11 -1
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/PKG-INFO +7 -2
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/README.md +6 -1
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/pyproject.toml +1 -1
- cli_consumption-0.4.3/src/cli_consumption/adapters/base.py +43 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/codex.py +144 -34
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/cli.py +311 -22
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/storage.py +32 -15
- cli_consumption-0.4.2/src/cli_consumption/adapters/base.py +0 -22
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/.gitignore +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/LICENSE +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/NOTICE +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/INTER_FONT_LICENSE.txt +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/__init__.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/__main__.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/__init__.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/_shared.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/aider.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/amazon_q.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/amp.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/claude.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/cline.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/continue_cli.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/copilot.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/crush.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/cursor.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/gemini.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/goose.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/grok.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/kilo.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/kimi.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/mistral_vibe.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/opencode.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/openhands.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/pi.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/plandex.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/qwen.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/registry.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/api.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/dashboard.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/dashboard_layouts.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/dashboard_react.css +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/dashboard_react.js +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/exporting.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/__init__.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/env.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/__init__.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0001_baseline.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0002_minimize_subagents.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0003_canonical_timestamps.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0004_subagent_scope_freshness.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0005_sync_receipts.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0006_dashboard_layouts.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/v0007_dashboard_layout_revision.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/models.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/py.typed +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/qualifications.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/reporting.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/reporting_api.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/retention.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/schema.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/snapshot_extraction.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/snapshot_files.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/sync.py +0 -0
- {cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/timestamps.py +0 -0
|
@@ -6,6 +6,15 @@ use [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.4.3] - 2026-09-05
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- Added opt-in `collect --incremental` automatic bounded batching for large Codex
|
|
14
|
+
stores, with restart-safe ingestion, strict metadata-only preflight staging,
|
|
15
|
+
deterministic duplicate convergence, privacy-safe partial results, and unchanged
|
|
16
|
+
per-file and per-line safety limits; authoritative subagent graphs remain untouched.
|
|
17
|
+
|
|
9
18
|
## [0.4.2] - 2026-09-05
|
|
10
19
|
|
|
11
20
|
### Added
|
|
@@ -185,7 +194,8 @@ use [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
|
185
194
|
|
|
186
195
|
- Refreshed the provider guide for the first minor release ([#26]).
|
|
187
196
|
|
|
188
|
-
[Unreleased]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.
|
|
197
|
+
[Unreleased]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.3...HEAD
|
|
198
|
+
[0.4.3]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.2...v0.4.3
|
|
189
199
|
[0.4.2]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.1...v0.4.2
|
|
190
200
|
[0.4.1]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.4.0...v0.4.1
|
|
191
201
|
[0.4.0]: https://github.com/Guillaume-Lombardo/cli-consumption/compare/v0.3.3...v0.4.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: cli-consumption
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.3
|
|
4
4
|
Summary: Analyze and consolidate AI coding CLI consumption across machines.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
|
|
6
6
|
Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
|
|
@@ -117,6 +117,10 @@ To collect one provider or select another database:
|
|
|
117
117
|
uv run cli-consumption collect --provider codex --database usage.sqlite
|
|
118
118
|
```
|
|
119
119
|
|
|
120
|
+
For a Codex store that exceeds aggregate collection limits, add `--incremental`.
|
|
121
|
+
The command automatically writes bounded, restart-safe batches to the same database;
|
|
122
|
+
individual file and line safety limits remain enforced.
|
|
123
|
+
|
|
120
124
|
Use `--source [LABEL=]PATH` for trusted offline copies and repeated
|
|
121
125
|
`--project NAME=PATH_PREFIX` mappings for stable project labels. See the
|
|
122
126
|
[usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
|
|
@@ -145,7 +149,8 @@ uv run cli-consumption providers --json
|
|
|
145
149
|
|
|
146
150
|
- Provider files are untrusted. Collection enforces discovery, file, line, SQLite row,
|
|
147
151
|
structured-field, and total normalized-record limits; direct provider-file symlinks
|
|
148
|
-
are refused.
|
|
152
|
+
are refused. Opt-in Codex incremental collection resets only aggregate per-batch
|
|
153
|
+
limits and keeps a separate hard batch-count ceiling.
|
|
149
154
|
- `collect --strict` and `sync --strict` refuse a batch when any provider skipped
|
|
150
155
|
malformed records. Collection distinguishes `provider_limit_exceeded`,
|
|
151
156
|
`provider_format_incompatible`, `invalid_snapshot`, and the unexpected-failure
|
|
@@ -78,6 +78,10 @@ To collect one provider or select another database:
|
|
|
78
78
|
uv run cli-consumption collect --provider codex --database usage.sqlite
|
|
79
79
|
```
|
|
80
80
|
|
|
81
|
+
For a Codex store that exceeds aggregate collection limits, add `--incremental`.
|
|
82
|
+
The command automatically writes bounded, restart-safe batches to the same database;
|
|
83
|
+
individual file and line safety limits remain enforced.
|
|
84
|
+
|
|
81
85
|
Use `--source [LABEL=]PATH` for trusted offline copies and repeated
|
|
82
86
|
`--project NAME=PATH_PREFIX` mappings for stable project labels. See the
|
|
83
87
|
[usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
|
|
@@ -106,7 +110,8 @@ uv run cli-consumption providers --json
|
|
|
106
110
|
|
|
107
111
|
- Provider files are untrusted. Collection enforces discovery, file, line, SQLite row,
|
|
108
112
|
structured-field, and total normalized-record limits; direct provider-file symlinks
|
|
109
|
-
are refused.
|
|
113
|
+
are refused. Opt-in Codex incremental collection resets only aggregate per-batch
|
|
114
|
+
limits and keeps a separate hard batch-count ceiling.
|
|
110
115
|
- `collect --strict` and `sync --strict` refuse a batch when any provider skipped
|
|
111
116
|
malformed records. Collection distinguishes `provider_limit_exceeded`,
|
|
112
117
|
`provider_format_incompatible`, `invalid_snapshot`, and the unexpected-failure
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Protocol, runtime_checkable
|
|
7
|
+
|
|
8
|
+
from cli_consumption.models import Snapshot
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True, slots=True)
|
|
12
|
+
class CollectionBatch:
|
|
13
|
+
"""One snapshot plus an optional subagent-scope authority override."""
|
|
14
|
+
|
|
15
|
+
snapshot: Snapshot
|
|
16
|
+
authoritative_subagent_scopes: frozenset[tuple[str, str]] | None = None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class UnsupportedProviderFormat(ValueError):
|
|
20
|
+
"""Raised when a detected provider store has an incompatible schema."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Adapter(Protocol):
|
|
24
|
+
"""Contract implemented by every supported AI CLI."""
|
|
25
|
+
|
|
26
|
+
name: str
|
|
27
|
+
|
|
28
|
+
def collect(
|
|
29
|
+
self,
|
|
30
|
+
sources: list[tuple[str, Path]],
|
|
31
|
+
project_mappings: list[tuple[str, str]],
|
|
32
|
+
) -> Snapshot: ...
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@runtime_checkable
|
|
36
|
+
class IncrementalAdapter(Adapter, Protocol):
|
|
37
|
+
"""Optional adapter contract for bounded, independently ingestible batches."""
|
|
38
|
+
|
|
39
|
+
def collect_incrementally(
|
|
40
|
+
self,
|
|
41
|
+
sources: list[tuple[str, Path]],
|
|
42
|
+
project_mappings: list[tuple[str, str]],
|
|
43
|
+
) -> Iterator[CollectionBatch]: ...
|
|
@@ -3,8 +3,10 @@ from __future__ import annotations
|
|
|
3
3
|
import hashlib
|
|
4
4
|
import json
|
|
5
5
|
import math
|
|
6
|
+
import os
|
|
6
7
|
import re
|
|
7
8
|
import sqlite3
|
|
9
|
+
from collections.abc import Iterator
|
|
8
10
|
from datetime import UTC, datetime
|
|
9
11
|
from pathlib import Path
|
|
10
12
|
from typing import Any
|
|
@@ -13,11 +15,18 @@ from cli_consumption.adapters._shared import (
|
|
|
13
15
|
MAX_BIGINT as MAX_BIGINT,
|
|
14
16
|
)
|
|
15
17
|
from cli_consumption.adapters._shared import (
|
|
18
|
+
ProviderDataLimitError,
|
|
16
19
|
ProviderInputBudget,
|
|
17
20
|
iter_bounded_jsonl_bytes,
|
|
18
21
|
open_provider_sqlite,
|
|
19
22
|
)
|
|
20
|
-
from cli_consumption.
|
|
23
|
+
from cli_consumption.adapters.base import CollectionBatch
|
|
24
|
+
from cli_consumption.models import (
|
|
25
|
+
TOKEN_FIELDS,
|
|
26
|
+
Snapshot,
|
|
27
|
+
SnapshotValidationError,
|
|
28
|
+
empty_tokens,
|
|
29
|
+
)
|
|
21
30
|
|
|
22
31
|
OUTSIDE_PROJECT = "outside-project"
|
|
23
32
|
TOOL_PATTERN = re.compile(r"(?:tools|collaboration)\.([A-Za-z][A-Za-z0-9_]*)\s*\(")
|
|
@@ -85,6 +94,24 @@ AGENT_ROLE_ALIASES = {
|
|
|
85
94
|
"worker": "worker",
|
|
86
95
|
}
|
|
87
96
|
SAFE_DIMENSION = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+-]*")
|
|
97
|
+
INCREMENTAL_CANDIDATES_PER_BATCH = 1_000
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _iter_session_files(root: Path) -> Iterator[Path]:
|
|
101
|
+
"""Walk a provider tree deterministically without materializing every path."""
|
|
102
|
+
pending = [root]
|
|
103
|
+
while pending:
|
|
104
|
+
directory = pending.pop()
|
|
105
|
+
with os.scandir(directory) as entries:
|
|
106
|
+
ordered = sorted(entries, key=lambda entry: entry.name)
|
|
107
|
+
child_directories: list[Path] = []
|
|
108
|
+
for entry in ordered:
|
|
109
|
+
path = Path(entry.path)
|
|
110
|
+
if entry.is_dir(follow_symlinks=False):
|
|
111
|
+
child_directories.append(path)
|
|
112
|
+
elif entry.name.endswith(".jsonl"):
|
|
113
|
+
yield path
|
|
114
|
+
pending.extend(reversed(child_directories))
|
|
88
115
|
|
|
89
116
|
|
|
90
117
|
def parse_timestamp(value: object) -> datetime | None:
|
|
@@ -166,6 +193,73 @@ class CodexAdapter:
|
|
|
166
193
|
)
|
|
167
194
|
return snapshot
|
|
168
195
|
|
|
196
|
+
def collect_incrementally(
|
|
197
|
+
self,
|
|
198
|
+
sources: list[tuple[str, Path]],
|
|
199
|
+
project_mappings: list[tuple[str, str]] | None = None,
|
|
200
|
+
) -> Iterator[CollectionBatch]:
|
|
201
|
+
"""Yield deterministic, bounded snapshots from arbitrarily many rollouts."""
|
|
202
|
+
mappings = project_mappings or []
|
|
203
|
+
for machine, codex_home in sources:
|
|
204
|
+
sessions = codex_home / "sessions"
|
|
205
|
+
if not sessions.is_dir():
|
|
206
|
+
raise ValueError("Missing Codex sessions directory")
|
|
207
|
+
|
|
208
|
+
yielded_sessions = False
|
|
209
|
+
batch: list[tuple[str, Path]] = []
|
|
210
|
+
for path in _iter_session_files(sessions):
|
|
211
|
+
batch.append((machine, path))
|
|
212
|
+
if len(batch) == INCREMENTAL_CANDIDATES_PER_BATCH:
|
|
213
|
+
yielded_sessions = True
|
|
214
|
+
yield from self._collect_incremental_batch(batch, mappings)
|
|
215
|
+
batch = []
|
|
216
|
+
if batch:
|
|
217
|
+
yielded_sessions = True
|
|
218
|
+
yield from self._collect_incremental_batch(batch, mappings)
|
|
219
|
+
|
|
220
|
+
if not yielded_sessions:
|
|
221
|
+
yield CollectionBatch(Snapshot(provider=self.name), frozenset())
|
|
222
|
+
|
|
223
|
+
def _collect_incremental_batch(
|
|
224
|
+
self,
|
|
225
|
+
candidates: list[tuple[str, Path]],
|
|
226
|
+
mappings: list[tuple[str, str]],
|
|
227
|
+
) -> Iterator[CollectionBatch]:
|
|
228
|
+
try:
|
|
229
|
+
yield CollectionBatch(
|
|
230
|
+
self._collect_candidates(candidates, mappings), frozenset()
|
|
231
|
+
)
|
|
232
|
+
except ProviderDataLimitError as error:
|
|
233
|
+
if str(error) != "provider_read_limit_exceeded" or len(candidates) == 1:
|
|
234
|
+
raise
|
|
235
|
+
midpoint = len(candidates) // 2
|
|
236
|
+
yield from self._collect_incremental_batch(candidates[:midpoint], mappings)
|
|
237
|
+
yield from self._collect_incremental_batch(candidates[midpoint:], mappings)
|
|
238
|
+
except SnapshotValidationError as error:
|
|
239
|
+
if error.code != "snapshot_too_large" or len(candidates) == 1:
|
|
240
|
+
raise
|
|
241
|
+
midpoint = len(candidates) // 2
|
|
242
|
+
yield from self._collect_incremental_batch(candidates[:midpoint], mappings)
|
|
243
|
+
yield from self._collect_incremental_batch(candidates[midpoint:], mappings)
|
|
244
|
+
|
|
245
|
+
def _collect_candidates(
|
|
246
|
+
self,
|
|
247
|
+
candidates: list[tuple[str, Path]],
|
|
248
|
+
mappings: list[tuple[str, str]],
|
|
249
|
+
) -> Snapshot:
|
|
250
|
+
budget = ProviderInputBudget()
|
|
251
|
+
selected, duplicates, malformed = self._discover_candidates(candidates, budget)
|
|
252
|
+
snapshot = Snapshot(
|
|
253
|
+
provider=self.name,
|
|
254
|
+
duplicate_conversations=duplicates,
|
|
255
|
+
malformed_records=malformed,
|
|
256
|
+
)
|
|
257
|
+
for machine, path, event_count, digest in selected:
|
|
258
|
+
self._read_rollout(
|
|
259
|
+
snapshot, machine, path, event_count, digest, mappings, budget
|
|
260
|
+
)
|
|
261
|
+
return snapshot
|
|
262
|
+
|
|
169
263
|
def _read_subagents(
|
|
170
264
|
self,
|
|
171
265
|
state_path: Path,
|
|
@@ -215,46 +309,62 @@ class CodexAdapter:
|
|
|
215
309
|
def _discover(
|
|
216
310
|
self, sources: list[tuple[str, Path]], budget: ProviderInputBudget
|
|
217
311
|
) -> tuple[list[tuple[str, Path, int, str]], int, int]:
|
|
218
|
-
|
|
219
|
-
duplicates = 0
|
|
220
|
-
malformed = 0
|
|
312
|
+
candidates: list[tuple[str, Path]] = []
|
|
221
313
|
for machine, codex_home in sources:
|
|
222
314
|
sessions = codex_home / "sessions"
|
|
223
315
|
if not sessions.is_dir():
|
|
224
316
|
raise ValueError(f"Missing Codex sessions directory: {sessions}")
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
317
|
+
candidates.extend(
|
|
318
|
+
(machine, path)
|
|
319
|
+
for path in budget.sorted_paths(sessions.rglob("*.jsonl"))
|
|
320
|
+
)
|
|
321
|
+
return self._discover_candidates(candidates, budget, charge_candidates=False)
|
|
322
|
+
|
|
323
|
+
def _discover_candidates(
|
|
324
|
+
self,
|
|
325
|
+
candidates: list[tuple[str, Path]],
|
|
326
|
+
budget: ProviderInputBudget,
|
|
327
|
+
*,
|
|
328
|
+
charge_candidates: bool = True,
|
|
329
|
+
) -> tuple[list[tuple[str, Path, int, str]], int, int]:
|
|
330
|
+
selected: dict[str, tuple[str, Path, int, str]] = {}
|
|
331
|
+
duplicates = 0
|
|
332
|
+
malformed = 0
|
|
333
|
+
for machine, path in candidates:
|
|
334
|
+
if charge_candidates:
|
|
335
|
+
budget.item()
|
|
336
|
+
event_count = 0
|
|
337
|
+
conversation_id = ""
|
|
338
|
+
digest = hashlib.sha256()
|
|
339
|
+
for raw_line in iter_bounded_jsonl_bytes(path, budget):
|
|
340
|
+
digest.update(raw_line)
|
|
341
|
+
try:
|
|
342
|
+
event = json.loads(raw_line)
|
|
343
|
+
except (json.JSONDecodeError, UnicodeDecodeError):
|
|
344
|
+
malformed += 1
|
|
345
|
+
continue
|
|
346
|
+
if not isinstance(event, dict):
|
|
347
|
+
malformed += 1
|
|
348
|
+
continue
|
|
349
|
+
event_count += 1
|
|
350
|
+
if event.get("type") == "session_meta":
|
|
351
|
+
payload = event.get("payload")
|
|
352
|
+
if not isinstance(payload, dict):
|
|
237
353
|
malformed += 1
|
|
238
354
|
continue
|
|
239
|
-
|
|
240
|
-
if
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
candidate
|
|
251
|
-
previous = selected.get(conversation_id)
|
|
252
|
-
if previous is None:
|
|
355
|
+
candidate_id = _safe_dimension(payload.get("id"), 512)
|
|
356
|
+
if candidate_id and not conversation_id:
|
|
357
|
+
conversation_id = candidate_id
|
|
358
|
+
content_hash = digest.hexdigest()
|
|
359
|
+
conversation_id = conversation_id or f"content-{content_hash}"
|
|
360
|
+
candidate = (machine, path, event_count, content_hash)
|
|
361
|
+
previous = selected.get(conversation_id)
|
|
362
|
+
if previous is None:
|
|
363
|
+
selected[conversation_id] = candidate
|
|
364
|
+
else:
|
|
365
|
+
duplicates += 1
|
|
366
|
+
if candidate[2:] > previous[2:]:
|
|
253
367
|
selected[conversation_id] = candidate
|
|
254
|
-
else:
|
|
255
|
-
duplicates += 1
|
|
256
|
-
if candidate[2:] > previous[2:]:
|
|
257
|
-
selected[conversation_id] = candidate
|
|
258
368
|
return list(selected.values()), duplicates, malformed
|
|
259
369
|
|
|
260
370
|
def _read_rollout(
|
|
@@ -3,16 +3,22 @@ from __future__ import annotations
|
|
|
3
3
|
import json
|
|
4
4
|
import os
|
|
5
5
|
import platform
|
|
6
|
+
import tempfile
|
|
7
|
+
from collections.abc import Iterator
|
|
6
8
|
from datetime import UTC, datetime, timedelta
|
|
7
9
|
from pathlib import Path
|
|
8
|
-
from typing import Annotated, Never
|
|
10
|
+
from typing import Annotated, Never, TextIO, TypedDict
|
|
9
11
|
|
|
10
12
|
import typer
|
|
11
13
|
from sqlalchemy.engine import Engine
|
|
12
14
|
|
|
13
15
|
from cli_consumption import __version__
|
|
14
16
|
from cli_consumption.adapters._shared import ProviderDataLimitError
|
|
15
|
-
from cli_consumption.adapters.base import
|
|
17
|
+
from cli_consumption.adapters.base import (
|
|
18
|
+
CollectionBatch,
|
|
19
|
+
IncrementalAdapter,
|
|
20
|
+
UnsupportedProviderFormat,
|
|
21
|
+
)
|
|
16
22
|
from cli_consumption.adapters.registry import (
|
|
17
23
|
ADAPTER_SPECS,
|
|
18
24
|
AdapterSpec,
|
|
@@ -35,8 +41,12 @@ from cli_consumption.storage import (
|
|
|
35
41
|
create_database_engine,
|
|
36
42
|
ingest_snapshot,
|
|
37
43
|
initialize_database,
|
|
44
|
+
validate_snapshot,
|
|
38
45
|
)
|
|
39
46
|
|
|
47
|
+
MAX_INCREMENTAL_BATCHES = 10_000
|
|
48
|
+
MAX_INCREMENTAL_STAGING_BYTES = 4 * 1024 * 1024 * 1024
|
|
49
|
+
|
|
40
50
|
app = typer.Typer(
|
|
41
51
|
name="cli-consumption",
|
|
42
52
|
help="Analyze AI coding CLI consumption without exporting conversation content.",
|
|
@@ -59,6 +69,31 @@ class CollectionFailure(RuntimeError):
|
|
|
59
69
|
super().__init__(code)
|
|
60
70
|
|
|
61
71
|
|
|
72
|
+
class IncrementalIngestion(TypedDict):
|
|
73
|
+
provider: str
|
|
74
|
+
batches: int
|
|
75
|
+
received: int
|
|
76
|
+
written: int
|
|
77
|
+
skipped: int
|
|
78
|
+
malformed: int
|
|
79
|
+
batch_duplicates: int
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class _BoundedStagingWriter:
|
|
83
|
+
"""Count UTF-8 metadata bytes before writing them to strict staging."""
|
|
84
|
+
|
|
85
|
+
def __init__(self, handle: TextIO, consumed: int) -> None:
|
|
86
|
+
self.handle = handle
|
|
87
|
+
self.consumed = consumed
|
|
88
|
+
|
|
89
|
+
def write(self, value: str) -> int:
|
|
90
|
+
size = len(value.encode("utf-8"))
|
|
91
|
+
if self.consumed + size > MAX_INCREMENTAL_STAGING_BYTES:
|
|
92
|
+
raise ProviderDataLimitError("provider_incremental_staging_limit_exceeded")
|
|
93
|
+
self.consumed += size
|
|
94
|
+
return self.handle.write(value)
|
|
95
|
+
|
|
96
|
+
|
|
62
97
|
def version_callback(value: bool) -> None:
|
|
63
98
|
if value:
|
|
64
99
|
typer.echo(__version__)
|
|
@@ -145,11 +180,31 @@ def collect(
|
|
|
145
180
|
help="Refuse ingestion when any malformed provider record was skipped.",
|
|
146
181
|
),
|
|
147
182
|
] = False,
|
|
183
|
+
incremental: Annotated[
|
|
184
|
+
bool,
|
|
185
|
+
typer.Option(
|
|
186
|
+
"--incremental",
|
|
187
|
+
help=(
|
|
188
|
+
"Automatically ingest supported large provider stores in bounded, "
|
|
189
|
+
"restart-safe batches."
|
|
190
|
+
),
|
|
191
|
+
),
|
|
192
|
+
] = False,
|
|
148
193
|
json_output: Annotated[
|
|
149
194
|
bool, typer.Option("--json", help="Emit a deterministic JSON result.")
|
|
150
195
|
] = False,
|
|
151
196
|
) -> None:
|
|
152
197
|
"""Collect one or more local/copied CLI data directories into SQL storage."""
|
|
198
|
+
if incremental:
|
|
199
|
+
_collect_incrementally(
|
|
200
|
+
provider,
|
|
201
|
+
source,
|
|
202
|
+
project,
|
|
203
|
+
database,
|
|
204
|
+
strict=strict,
|
|
205
|
+
json_output=json_output,
|
|
206
|
+
)
|
|
207
|
+
return
|
|
153
208
|
try:
|
|
154
209
|
snapshots = _collect_snapshots(provider, source, project)
|
|
155
210
|
except CollectionFailure as error:
|
|
@@ -742,25 +797,45 @@ def _emit_database_upload_json(
|
|
|
742
797
|
typer.echo(json.dumps(payload, sort_keys=True, separators=(",", ":")))
|
|
743
798
|
|
|
744
799
|
|
|
745
|
-
def _emit_collection_json(
|
|
800
|
+
def _emit_collection_json(
|
|
801
|
+
error: CollectionFailure,
|
|
802
|
+
*,
|
|
803
|
+
ingestions: list[IncrementalIngestion] | None = None,
|
|
804
|
+
incremental: bool = False,
|
|
805
|
+
) -> None:
|
|
746
806
|
"""Emit one deterministic collection failure without provider error details."""
|
|
807
|
+
payload: dict[str, object] = {
|
|
808
|
+
"error": {"code": error.code, "provider": error.provider},
|
|
809
|
+
"ingestions": ingestions or [],
|
|
810
|
+
}
|
|
811
|
+
if incremental:
|
|
812
|
+
payload["incremental"] = True
|
|
747
813
|
typer.echo(
|
|
748
814
|
json.dumps(
|
|
749
|
-
|
|
750
|
-
"error": {"code": error.code, "provider": error.provider},
|
|
751
|
-
"ingestions": [],
|
|
752
|
-
},
|
|
815
|
+
payload,
|
|
753
816
|
sort_keys=True,
|
|
754
817
|
separators=(",", ":"),
|
|
755
818
|
)
|
|
756
819
|
)
|
|
757
820
|
|
|
758
821
|
|
|
759
|
-
def _abort_collection(
|
|
822
|
+
def _abort_collection(
|
|
823
|
+
error: CollectionFailure,
|
|
824
|
+
*,
|
|
825
|
+
json_output: bool,
|
|
826
|
+
ingestions: list[IncrementalIngestion] | None = None,
|
|
827
|
+
incremental: bool = False,
|
|
828
|
+
) -> Never:
|
|
760
829
|
if json_output:
|
|
761
|
-
_emit_collection_json(error)
|
|
830
|
+
_emit_collection_json(error, ingestions=ingestions, incremental=incremental)
|
|
762
831
|
else:
|
|
763
|
-
|
|
832
|
+
completed = sum(item["batches"] for item in ingestions or [])
|
|
833
|
+
suffix = (
|
|
834
|
+
f" {completed} earlier incremental batch(es) were committed; rerun is safe."
|
|
835
|
+
if incremental and completed
|
|
836
|
+
else ""
|
|
837
|
+
)
|
|
838
|
+
typer.echo(error.message + suffix, err=True)
|
|
764
839
|
raise typer.Exit(code=2) from None
|
|
765
840
|
|
|
766
841
|
|
|
@@ -1020,11 +1095,176 @@ def serve(
|
|
|
1020
1095
|
engine.dispose()
|
|
1021
1096
|
|
|
1022
1097
|
|
|
1023
|
-
def
|
|
1098
|
+
def _collect_incrementally(
|
|
1024
1099
|
provider: str,
|
|
1025
1100
|
source_values: list[str] | None,
|
|
1026
1101
|
project_values: list[str] | None,
|
|
1027
|
-
|
|
1102
|
+
database: str,
|
|
1103
|
+
*,
|
|
1104
|
+
strict: bool,
|
|
1105
|
+
json_output: bool,
|
|
1106
|
+
) -> None:
|
|
1107
|
+
summaries: dict[str, IncrementalIngestion] = {}
|
|
1108
|
+
|
|
1109
|
+
def ingest(engine: Engine, batch: CollectionBatch) -> None:
|
|
1110
|
+
snapshot = batch.snapshot
|
|
1111
|
+
try:
|
|
1112
|
+
result = ingest_snapshot(
|
|
1113
|
+
engine,
|
|
1114
|
+
snapshot,
|
|
1115
|
+
authoritative_subagent_scopes=batch.authoritative_subagent_scopes,
|
|
1116
|
+
)
|
|
1117
|
+
except SnapshotValidationError as error:
|
|
1118
|
+
raise _snapshot_failure(snapshot.provider, error) from None
|
|
1119
|
+
summary = summaries.setdefault(
|
|
1120
|
+
snapshot.provider,
|
|
1121
|
+
{
|
|
1122
|
+
"provider": snapshot.provider,
|
|
1123
|
+
"batches": 0,
|
|
1124
|
+
"received": 0,
|
|
1125
|
+
"written": 0,
|
|
1126
|
+
"skipped": 0,
|
|
1127
|
+
"malformed": 0,
|
|
1128
|
+
"batch_duplicates": 0,
|
|
1129
|
+
},
|
|
1130
|
+
)
|
|
1131
|
+
summary["batches"] += 1
|
|
1132
|
+
summary["received"] += result.received
|
|
1133
|
+
summary["written"] += result.written
|
|
1134
|
+
summary["skipped"] += result.skipped
|
|
1135
|
+
summary["malformed"] += snapshot.malformed_records
|
|
1136
|
+
summary["batch_duplicates"] += snapshot.duplicate_conversations
|
|
1137
|
+
|
|
1138
|
+
failure: CollectionFailure | None = None
|
|
1139
|
+
if strict:
|
|
1140
|
+
with tempfile.TemporaryDirectory(prefix="cli-consumption-") as staging:
|
|
1141
|
+
staged: list[Path] = []
|
|
1142
|
+
staged_bytes = 0
|
|
1143
|
+
active_provider = provider
|
|
1144
|
+
try:
|
|
1145
|
+
for index, batch in enumerate(
|
|
1146
|
+
_iter_incremental_batches(provider, source_values, project_values)
|
|
1147
|
+
):
|
|
1148
|
+
snapshot = batch.snapshot
|
|
1149
|
+
active_provider = snapshot.provider
|
|
1150
|
+
if snapshot.malformed_records:
|
|
1151
|
+
raise CollectionFailure(
|
|
1152
|
+
snapshot.provider,
|
|
1153
|
+
"malformed_records",
|
|
1154
|
+
"--strict refused incremental snapshots containing "
|
|
1155
|
+
"malformed provider records.",
|
|
1156
|
+
)
|
|
1157
|
+
validated = validate_snapshot(snapshot)
|
|
1158
|
+
path = Path(staging) / f"{index:08d}.json"
|
|
1159
|
+
descriptor = os.open(
|
|
1160
|
+
path,
|
|
1161
|
+
os.O_WRONLY | os.O_CREAT | os.O_EXCL,
|
|
1162
|
+
0o600,
|
|
1163
|
+
)
|
|
1164
|
+
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
|
|
1165
|
+
writer = _BoundedStagingWriter(handle, staged_bytes)
|
|
1166
|
+
json.dump(
|
|
1167
|
+
{
|
|
1168
|
+
"snapshot": validated.to_dict(),
|
|
1169
|
+
"authoritative_subagent_scopes": (
|
|
1170
|
+
sorted(batch.authoritative_subagent_scopes)
|
|
1171
|
+
if batch.authoritative_subagent_scopes is not None
|
|
1172
|
+
else None
|
|
1173
|
+
),
|
|
1174
|
+
},
|
|
1175
|
+
writer,
|
|
1176
|
+
sort_keys=True,
|
|
1177
|
+
separators=(",", ":"),
|
|
1178
|
+
)
|
|
1179
|
+
staged_bytes = writer.consumed
|
|
1180
|
+
staged.append(path)
|
|
1181
|
+
except CollectionFailure as error:
|
|
1182
|
+
failure = error
|
|
1183
|
+
except SnapshotValidationError as error:
|
|
1184
|
+
failure = _snapshot_failure(active_provider, error)
|
|
1185
|
+
except ProviderDataLimitError:
|
|
1186
|
+
failure = CollectionFailure(
|
|
1187
|
+
active_provider,
|
|
1188
|
+
"provider_limit_exceeded",
|
|
1189
|
+
f"Provider {active_provider!r} data exceeds collection "
|
|
1190
|
+
"safety limits.",
|
|
1191
|
+
)
|
|
1192
|
+
except OSError:
|
|
1193
|
+
failure = CollectionFailure(
|
|
1194
|
+
active_provider,
|
|
1195
|
+
"provider_collection_failed",
|
|
1196
|
+
f"Provider {active_provider!r} collection failed.",
|
|
1197
|
+
)
|
|
1198
|
+
|
|
1199
|
+
if failure is None:
|
|
1200
|
+
engine = _open_database(database)
|
|
1201
|
+
try:
|
|
1202
|
+
for path in staged:
|
|
1203
|
+
with path.open(encoding="utf-8") as handle:
|
|
1204
|
+
payload = json.load(handle)
|
|
1205
|
+
raw_scopes = payload["authoritative_subagent_scopes"]
|
|
1206
|
+
ingest(
|
|
1207
|
+
engine,
|
|
1208
|
+
CollectionBatch(
|
|
1209
|
+
Snapshot.from_dict(payload["snapshot"]),
|
|
1210
|
+
(
|
|
1211
|
+
frozenset(
|
|
1212
|
+
(str(item[0]), str(item[1]))
|
|
1213
|
+
for item in raw_scopes
|
|
1214
|
+
)
|
|
1215
|
+
if raw_scopes is not None
|
|
1216
|
+
else None
|
|
1217
|
+
),
|
|
1218
|
+
),
|
|
1219
|
+
)
|
|
1220
|
+
except CollectionFailure as error:
|
|
1221
|
+
failure = error
|
|
1222
|
+
finally:
|
|
1223
|
+
engine.dispose()
|
|
1224
|
+
else:
|
|
1225
|
+
engine = _open_database(database)
|
|
1226
|
+
try:
|
|
1227
|
+
try:
|
|
1228
|
+
for batch in _iter_incremental_batches(
|
|
1229
|
+
provider, source_values, project_values
|
|
1230
|
+
):
|
|
1231
|
+
ingest(engine, batch)
|
|
1232
|
+
except CollectionFailure as error:
|
|
1233
|
+
failure = error
|
|
1234
|
+
finally:
|
|
1235
|
+
engine.dispose()
|
|
1236
|
+
|
|
1237
|
+
outcomes = list(summaries.values())
|
|
1238
|
+
if failure is not None:
|
|
1239
|
+
_abort_collection(
|
|
1240
|
+
failure,
|
|
1241
|
+
json_output=json_output,
|
|
1242
|
+
ingestions=outcomes,
|
|
1243
|
+
incremental=True,
|
|
1244
|
+
)
|
|
1245
|
+
if json_output:
|
|
1246
|
+
typer.echo(
|
|
1247
|
+
json.dumps(
|
|
1248
|
+
{"incremental": True, "ingestions": outcomes},
|
|
1249
|
+
sort_keys=True,
|
|
1250
|
+
separators=(",", ":"),
|
|
1251
|
+
)
|
|
1252
|
+
)
|
|
1253
|
+
return
|
|
1254
|
+
for summary in outcomes:
|
|
1255
|
+
typer.echo(
|
|
1256
|
+
f"Incremental ingestion {summary['provider']}: "
|
|
1257
|
+
f"{summary['batches']} batches, {summary['written']} written, "
|
|
1258
|
+
f"{summary['skipped']} unchanged, "
|
|
1259
|
+
f"{summary['malformed']} malformed skipped."
|
|
1260
|
+
)
|
|
1261
|
+
|
|
1262
|
+
|
|
1263
|
+
def _collection_inputs(
|
|
1264
|
+
provider: str,
|
|
1265
|
+
source_values: list[str] | None,
|
|
1266
|
+
project_values: list[str] | None,
|
|
1267
|
+
) -> tuple[list[tuple[AdapterSpec, list[tuple[str, Path]]]], list[tuple[str, str]]]:
|
|
1028
1268
|
spec = resolve_adapter_spec(provider) if provider != "all" else None
|
|
1029
1269
|
if provider != "all" and spec is None:
|
|
1030
1270
|
raise typer.BadParameter(
|
|
@@ -1032,11 +1272,9 @@ def _collect_snapshots(
|
|
|
1032
1272
|
)
|
|
1033
1273
|
mappings = _parse_project_mappings(project_values or [])
|
|
1034
1274
|
if spec is not None:
|
|
1035
|
-
return [
|
|
1036
|
-
_collect_adapter(spec, _parse_sources(source_values or [], spec), mappings)
|
|
1037
|
-
]
|
|
1275
|
+
return [(spec, _parse_sources(source_values or [], spec))], mappings
|
|
1038
1276
|
|
|
1039
|
-
|
|
1277
|
+
inputs: list[tuple[AdapterSpec, list[tuple[str, Path]]]] = []
|
|
1040
1278
|
if source_values:
|
|
1041
1279
|
sources = _parse_source_values(source_values)
|
|
1042
1280
|
matched_labels: set[str] = set()
|
|
@@ -1046,7 +1284,7 @@ def _collect_snapshots(
|
|
|
1046
1284
|
]
|
|
1047
1285
|
if matched:
|
|
1048
1286
|
matched_labels.update(label for label, _ in matched)
|
|
1049
|
-
|
|
1287
|
+
inputs.append((candidate, matched))
|
|
1050
1288
|
unmatched = [label for label, _ in sources if label not in matched_labels]
|
|
1051
1289
|
if unmatched:
|
|
1052
1290
|
raise typer.BadParameter(
|
|
@@ -1058,12 +1296,63 @@ def _collect_snapshots(
|
|
|
1058
1296
|
for candidate in ADAPTER_SPECS:
|
|
1059
1297
|
path = default_source_path(candidate)
|
|
1060
1298
|
if has_provider_data(candidate, path):
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
)
|
|
1064
|
-
if not snapshots:
|
|
1299
|
+
inputs.append((candidate, [(machine, path)]))
|
|
1300
|
+
if not inputs:
|
|
1065
1301
|
raise typer.BadParameter("No supported provider data detected.")
|
|
1066
|
-
return
|
|
1302
|
+
return inputs, mappings
|
|
1303
|
+
|
|
1304
|
+
|
|
1305
|
+
def _collect_snapshots(
|
|
1306
|
+
provider: str,
|
|
1307
|
+
source_values: list[str] | None,
|
|
1308
|
+
project_values: list[str] | None,
|
|
1309
|
+
) -> list[Snapshot]:
|
|
1310
|
+
inputs, mappings = _collection_inputs(provider, source_values, project_values)
|
|
1311
|
+
return [_collect_adapter(spec, sources, mappings) for spec, sources in inputs]
|
|
1312
|
+
|
|
1313
|
+
|
|
1314
|
+
def _iter_incremental_batches(
|
|
1315
|
+
provider: str,
|
|
1316
|
+
source_values: list[str] | None,
|
|
1317
|
+
project_values: list[str] | None,
|
|
1318
|
+
) -> Iterator[CollectionBatch]:
|
|
1319
|
+
inputs, mappings = _collection_inputs(provider, source_values, project_values)
|
|
1320
|
+
batches = 0
|
|
1321
|
+
for spec, sources in inputs:
|
|
1322
|
+
adapter = spec.adapter_type()
|
|
1323
|
+
try:
|
|
1324
|
+
batches_for_adapter = (
|
|
1325
|
+
adapter.collect_incrementally(sources, mappings)
|
|
1326
|
+
if isinstance(adapter, IncrementalAdapter)
|
|
1327
|
+
else iter((CollectionBatch(adapter.collect(sources, mappings)),))
|
|
1328
|
+
)
|
|
1329
|
+
for batch in batches_for_adapter:
|
|
1330
|
+
batches += 1
|
|
1331
|
+
if batches > MAX_INCREMENTAL_BATCHES:
|
|
1332
|
+
raise ProviderDataLimitError(
|
|
1333
|
+
"provider_incremental_batch_limit_exceeded"
|
|
1334
|
+
)
|
|
1335
|
+
yield batch
|
|
1336
|
+
except ProviderDataLimitError:
|
|
1337
|
+
raise CollectionFailure(
|
|
1338
|
+
spec.name,
|
|
1339
|
+
"provider_limit_exceeded",
|
|
1340
|
+
f"Provider {spec.name!r} data exceeds collection safety limits.",
|
|
1341
|
+
) from None
|
|
1342
|
+
except UnsupportedProviderFormat:
|
|
1343
|
+
raise CollectionFailure(
|
|
1344
|
+
spec.name,
|
|
1345
|
+
"provider_format_incompatible",
|
|
1346
|
+
f"Provider {spec.name!r} data format is incompatible.",
|
|
1347
|
+
) from None
|
|
1348
|
+
except SnapshotValidationError as error:
|
|
1349
|
+
raise _snapshot_failure(spec.name, error) from None
|
|
1350
|
+
except Exception:
|
|
1351
|
+
raise CollectionFailure(
|
|
1352
|
+
spec.name,
|
|
1353
|
+
"provider_collection_failed",
|
|
1354
|
+
f"Provider {spec.name!r} collection failed.",
|
|
1355
|
+
) from None
|
|
1067
1356
|
|
|
1068
1357
|
|
|
1069
1358
|
def _collect_adapter(
|
|
@@ -364,8 +364,16 @@ def ingest_snapshot(
|
|
|
364
364
|
snapshot: Snapshot,
|
|
365
365
|
*,
|
|
366
366
|
idempotency_key: str | None = None,
|
|
367
|
+
authoritative_subagent_scopes: frozenset[tuple[str, str]] | None = None,
|
|
367
368
|
) -> IngestionResult:
|
|
368
369
|
snapshot = validate_snapshot(snapshot)
|
|
370
|
+
if authoritative_subagent_scopes is not None and any(
|
|
371
|
+
provider != snapshot.provider
|
|
372
|
+
or not isinstance(source_machine, str)
|
|
373
|
+
or not 1 <= len(source_machine) <= 255
|
|
374
|
+
for provider, source_machine in authoritative_subagent_scopes
|
|
375
|
+
):
|
|
376
|
+
raise SnapshotValidationError()
|
|
369
377
|
initialize_database(engine)
|
|
370
378
|
if idempotency_key is not None:
|
|
371
379
|
idempotency_key = _canonical_idempotency_key(idempotency_key)
|
|
@@ -381,10 +389,15 @@ def ingest_snapshot(
|
|
|
381
389
|
context_by_conversation = _group(snapshot.context_samples)
|
|
382
390
|
settings_by_conversation = _group(snapshot.turn_settings)
|
|
383
391
|
compactions_by_conversation = _group(snapshot.compaction_events)
|
|
384
|
-
|
|
392
|
+
inferred_subagent_scopes = {
|
|
385
393
|
(snapshot.provider, str(record["source_machine"]))
|
|
386
394
|
for record in (*snapshot.conversations, *snapshot.subagents)
|
|
387
395
|
}
|
|
396
|
+
subagent_scopes = (
|
|
397
|
+
inferred_subagent_scopes
|
|
398
|
+
if authoritative_subagent_scopes is None
|
|
399
|
+
else set(authoritative_subagent_scopes)
|
|
400
|
+
)
|
|
388
401
|
stale_subagent_scopes: set[tuple[str, str]] = set()
|
|
389
402
|
richer_subagent_scopes: set[tuple[str, str]] = set()
|
|
390
403
|
try:
|
|
@@ -398,21 +411,20 @@ def ingest_snapshot(
|
|
|
398
411
|
conversation_id = str(record["id"])
|
|
399
412
|
existing = session.get(Conversation, conversation_id)
|
|
400
413
|
scope = (snapshot.provider, str(record["source_machine"]))
|
|
401
|
-
|
|
402
|
-
record["event_count"]
|
|
403
|
-
|
|
414
|
+
incoming_rank = (
|
|
415
|
+
int(record["event_count"]),
|
|
416
|
+
str(record["content_hash"]),
|
|
417
|
+
)
|
|
418
|
+
existing_rank = (
|
|
419
|
+
(existing.event_count, existing.content_hash)
|
|
420
|
+
if existing is not None
|
|
421
|
+
else None
|
|
422
|
+
)
|
|
423
|
+
if existing_rank is not None and existing_rank > incoming_rank:
|
|
404
424
|
stale_subagent_scopes.add(scope)
|
|
405
|
-
elif
|
|
406
|
-
record["event_count"]
|
|
407
|
-
):
|
|
425
|
+
elif existing_rank is not None and existing_rank < incoming_rank:
|
|
408
426
|
richer_subagent_scopes.add(scope)
|
|
409
|
-
if
|
|
410
|
-
existing.event_count > int(record["event_count"])
|
|
411
|
-
or (
|
|
412
|
-
existing.event_count == int(record["event_count"])
|
|
413
|
-
and existing.content_hash == record["content_hash"]
|
|
414
|
-
)
|
|
415
|
-
):
|
|
427
|
+
if existing_rank is not None and existing_rank >= incoming_rank:
|
|
416
428
|
skipped += 1
|
|
417
429
|
continue
|
|
418
430
|
session.execute(
|
|
@@ -461,9 +473,14 @@ def ingest_snapshot(
|
|
|
461
473
|
for compaction in compactions_by_conversation.get(conversation_id, []):
|
|
462
474
|
session.add(CompactionEvent(**compaction))
|
|
463
475
|
written += 1
|
|
464
|
-
|
|
476
|
+
inferred_authoritative_scopes = initial_subagent_scopes | (
|
|
465
477
|
richer_subagent_scopes - stale_subagent_scopes
|
|
466
478
|
)
|
|
479
|
+
authoritative_scopes = (
|
|
480
|
+
inferred_authoritative_scopes
|
|
481
|
+
if authoritative_subagent_scopes is None
|
|
482
|
+
else authoritative_subagent_scopes
|
|
483
|
+
)
|
|
467
484
|
for provider, source_machine in authoritative_scopes:
|
|
468
485
|
session.execute(
|
|
469
486
|
delete(Subagent).where(
|
|
@@ -1,22 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
from pathlib import Path
|
|
4
|
-
from typing import Protocol
|
|
5
|
-
|
|
6
|
-
from cli_consumption.models import Snapshot
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
class UnsupportedProviderFormat(ValueError):
|
|
10
|
-
"""Raised when a detected provider store has an incompatible schema."""
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
class Adapter(Protocol):
|
|
14
|
-
"""Contract implemented by every supported AI CLI."""
|
|
15
|
-
|
|
16
|
-
name: str
|
|
17
|
-
|
|
18
|
-
def collect(
|
|
19
|
-
self,
|
|
20
|
-
sources: list[tuple[str, Path]],
|
|
21
|
-
project_mappings: list[tuple[str, str]],
|
|
22
|
-
) -> Snapshot: ...
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/continue_cli.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/adapters/mistral_vibe.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cli_consumption-0.4.2 → cli_consumption-0.4.3}/src/cli_consumption/migrations/versions/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|