cli-consumption 0.3.2__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/CHANGELOG.md +16 -0
  2. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/PKG-INFO +7 -3
  3. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/README.md +4 -2
  4. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/pyproject.toml +4 -1
  5. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/_shared.py +40 -12
  6. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/codex.py +101 -43
  7. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/continue_cli.py +40 -13
  8. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/opencode.py +4 -1
  9. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/registry.py +2 -1
  10. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/cli.py +153 -0
  11. cli_consumption-0.3.3/src/cli_consumption/snapshot_files.py +255 -0
  12. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/.gitignore +0 -0
  13. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/LICENSE +0 -0
  14. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/NOTICE +0 -0
  15. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/__init__.py +0 -0
  16. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/__main__.py +0 -0
  17. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/__init__.py +0 -0
  18. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/aider.py +0 -0
  19. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/amazon_q.py +0 -0
  20. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/amp.py +0 -0
  21. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/base.py +0 -0
  22. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/claude.py +0 -0
  23. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/cline.py +0 -0
  24. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/copilot.py +0 -0
  25. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/crush.py +0 -0
  26. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/cursor.py +0 -0
  27. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/gemini.py +0 -0
  28. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/goose.py +0 -0
  29. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/grok.py +0 -0
  30. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/kilo.py +0 -0
  31. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/kimi.py +0 -0
  32. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/mistral_vibe.py +0 -0
  33. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/openhands.py +0 -0
  34. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/pi.py +0 -0
  35. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/plandex.py +0 -0
  36. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/adapters/qwen.py +0 -0
  37. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/api.py +0 -0
  38. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/dashboard.py +0 -0
  39. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/dashboard_calculations.js +0 -0
  40. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/exporting.py +0 -0
  41. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/__init__.py +0 -0
  42. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/env.py +0 -0
  43. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/__init__.py +0 -0
  44. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/v0001_baseline.py +0 -0
  45. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/v0002_minimize_subagents.py +0 -0
  46. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/v0003_canonical_timestamps.py +0 -0
  47. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/v0004_subagent_scope_freshness.py +0 -0
  48. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/migrations/versions/v0005_sync_receipts.py +0 -0
  49. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/models.py +0 -0
  50. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/py.typed +0 -0
  51. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/qualifications.py +0 -0
  52. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/reporting.py +0 -0
  53. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/retention.py +0 -0
  54. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/schema.py +0 -0
  55. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/storage.py +0 -0
  56. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/sync.py +0 -0
  57. {cli_consumption-0.3.2 → cli_consumption-0.3.3}/src/cli_consumption/timestamps.py +0 -0
@@ -6,6 +6,22 @@ use [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ### Added
10
+
11
+ - Added deterministic, compressed offline snapshot files authenticated with Ed25519,
12
+ with bounded verification-before-parsing and idempotent SQLite/PostgreSQL ingestion.
13
+
14
+ ### Fixed
15
+
16
+ - Continue now applies provider-specific cache semantics, preventing Anthropic cache
17
+ reads and writes from being omitted from input totals, and reports persisted
18
+ conversation-compaction markers.
19
+ - Codex now preserves valid token accounting when provider counters are inconsistent,
20
+ keeps partial work and compaction records from referencing nonexistent turns, and
21
+ safely handles malformed metadata and timestamps.
22
+ - OpenCode now rejects timezone-naive text timestamps consistently while retaining its
23
+ OpenCode 1.18.23 message/part extraction for models, tokens, and tools.
24
+
9
25
  ## [0.3.2] - 2026-08-31
10
26
 
11
27
  ### Changed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cli-consumption
3
- Version: 0.3.2
3
+ Version: 0.3.3
4
4
  Summary: Analyze and consolidate AI coding CLI consumption across machines.
5
5
  Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
6
6
  Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
@@ -31,6 +31,8 @@ Requires-Dist: psycopg[binary]>=3.2; extra == 'postgres'
31
31
  Provides-Extra: server
32
32
  Requires-Dist: fastapi>=0.115; extra == 'server'
33
33
  Requires-Dist: uvicorn>=0.34; extra == 'server'
34
+ Provides-Extra: snapshots
35
+ Requires-Dist: cryptography>=45; extra == 'snapshots'
34
36
  Provides-Extra: sync
35
37
  Requires-Dist: httpx>=0.27; extra == 'sync'
36
38
  Description-Content-Type: text/markdown
@@ -75,7 +77,8 @@ the optional runtime capabilities you use:
75
77
 
76
78
  - `cli-consumption[sync]` for the sync client;
77
79
  - `cli-consumption[server]` for the collector service;
78
- - `cli-consumption[postgres]` for PostgreSQL.
80
+ - `cli-consumption[postgres]` for PostgreSQL;
81
+ - `cli-consumption[snapshots]` for signed, compressed offline snapshot files.
79
82
 
80
83
  Extras can be combined, for example `cli-consumption[server,postgres]` on a central
81
84
  collector.
@@ -103,7 +106,7 @@ uv run cli-consumption collect --provider codex --database usage.sqlite
103
106
  Use `--source [LABEL=]PATH` for trusted offline copies and repeated
104
107
  `--project NAME=PATH_PREFIX` mappings for stable project labels. See the
105
108
  [usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
106
- for multi-machine collection, reporting,
109
+ for signed offline transfers, multi-machine collection, reporting,
107
110
  PostgreSQL, retention, synchronization, readiness, and automation.
108
111
 
109
112
  ## Supported CLIs
@@ -152,6 +155,7 @@ for ingestion, idempotency, migrations, report limits, and collector behavior.
152
155
  | Command | Purpose |
153
156
  | --- | --- |
154
157
  | `collect` | Collect local or copied provider data into SQL. |
158
+ | `snapshot` | Create or ingest signed, compressed offline snapshot files. |
155
159
  | `sync` | Collect and send metadata-only snapshots to a central API. |
156
160
  | `serve` | Run the central collection API. |
157
161
  | `export` | Write the HTML dashboard and optional CSV tables. |
@@ -38,7 +38,8 @@ the optional runtime capabilities you use:
38
38
 
39
39
  - `cli-consumption[sync]` for the sync client;
40
40
  - `cli-consumption[server]` for the collector service;
41
- - `cli-consumption[postgres]` for PostgreSQL.
41
+ - `cli-consumption[postgres]` for PostgreSQL;
42
+ - `cli-consumption[snapshots]` for signed, compressed offline snapshot files.
42
43
 
43
44
  Extras can be combined, for example `cli-consumption[server,postgres]` on a central
44
45
  collector.
@@ -66,7 +67,7 @@ uv run cli-consumption collect --provider codex --database usage.sqlite
66
67
  Use `--source [LABEL=]PATH` for trusted offline copies and repeated
67
68
  `--project NAME=PATH_PREFIX` mappings for stable project labels. See the
68
69
  [usage and operations guide](https://github.com/Guillaume-Lombardo/cli-consumption/blob/main/docs/usage.md)
69
- for multi-machine collection, reporting,
70
+ for signed offline transfers, multi-machine collection, reporting,
70
71
  PostgreSQL, retention, synchronization, readiness, and automation.
71
72
 
72
73
  ## Supported CLIs
@@ -115,6 +116,7 @@ for ingestion, idempotency, migrations, report limits, and collector behavior.
115
116
  | Command | Purpose |
116
117
  | --- | --- |
117
118
  | `collect` | Collect local or copied provider data into SQL. |
119
+ | `snapshot` | Create or ingest signed, compressed offline snapshot files. |
118
120
  | `sync` | Collect and send metadata-only snapshots to a central API. |
119
121
  | `serve` | Run the central collection API. |
120
122
  | `export` | Write the HTML dashboard and optional CSV tables. |
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "cli-consumption"
7
- version = "0.3.2"
7
+ version = "0.3.3"
8
8
  description = "Analyze and consolidate AI coding CLI consumption across machines."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -34,6 +34,9 @@ dependencies = [
34
34
  postgres = ["psycopg[binary]>=3.2"]
35
35
  server = ["fastapi>=0.115", "uvicorn>=0.34"]
36
36
  sync = ["httpx>=0.27"]
37
+ snapshots = [
38
+ "cryptography>=45",
39
+ ]
37
40
 
38
41
  [project.urls]
39
42
  Homepage = "https://github.com/Guillaume-Lombardo/cli-consumption"
@@ -166,15 +166,22 @@ def tokens(
166
166
  reasoning: object = 0,
167
167
  total: object = 0,
168
168
  ) -> dict[str, int]:
169
- uncached_n = counter(uncached)
170
- cached_n = counter(cached)
171
- write_n = counter(cache_write)
172
- visible_n = counter(visible)
173
- reasoning_n = counter(reasoning)
174
- input_n = bounded_sum(uncached_n, cached_n, write_n)
175
- output_n = bounded_sum(visible_n, reasoning_n)
176
- attributed = bounded_sum(input_n, output_n)
177
- total_n = max(attributed, counter(total))
169
+ remaining = MAX_BIGINT
170
+ uncached_n = min(counter(uncached), remaining)
171
+ remaining -= uncached_n
172
+ cached_n = min(counter(cached), remaining)
173
+ remaining -= cached_n
174
+ write_n = min(counter(cache_write), remaining)
175
+ remaining -= write_n
176
+ visible_n = min(counter(visible), remaining)
177
+ remaining -= visible_n
178
+ reasoning_n = min(counter(reasoning), remaining)
179
+ remaining -= reasoning_n
180
+ input_n = uncached_n + cached_n + write_n
181
+ output_n = visible_n + reasoning_n
182
+ attributed = input_n + output_n
183
+ unattributed_n = min(max(0, counter(total) - attributed), remaining)
184
+ total_n = attributed + unattributed_n
178
185
  return {
179
186
  "input_tokens": input_n,
180
187
  "cached_input_tokens": cached_n,
@@ -184,13 +191,34 @@ def tokens(
184
191
  "total_tokens": total_n,
185
192
  "uncached_input_tokens": uncached_n,
186
193
  "visible_output_tokens": visible_n,
187
- "unattributed_tokens": max(0, total_n - attributed),
194
+ "unattributed_tokens": unattributed_n,
188
195
  }
189
196
 
190
197
 
191
198
  def add_tokens(target: dict[str, Any], value: dict[str, int]) -> None:
192
- for field, amount in value.items():
193
- target[field] = bounded_sum(int(target[field]), amount)
199
+ remaining = MAX_BIGINT
200
+ for field in (
201
+ "uncached_input_tokens",
202
+ "cached_input_tokens",
203
+ "cache_write_input_tokens",
204
+ "visible_output_tokens",
205
+ "reasoning_output_tokens",
206
+ "unattributed_tokens",
207
+ ):
208
+ amount = min(remaining, int(target[field]) + value[field])
209
+ target[field] = amount
210
+ remaining -= amount
211
+ target["input_tokens"] = (
212
+ target["uncached_input_tokens"]
213
+ + target["cached_input_tokens"]
214
+ + target["cache_write_input_tokens"]
215
+ )
216
+ target["output_tokens"] = (
217
+ target["visible_output_tokens"] + target["reasoning_output_tokens"]
218
+ )
219
+ target["total_tokens"] = (
220
+ target["input_tokens"] + target["output_tokens"] + target["unattributed_tokens"]
221
+ )
194
222
 
195
223
 
196
224
  def new_turn(
@@ -87,11 +87,14 @@ AGENT_ROLE_ALIASES = {
87
87
  SAFE_DIMENSION = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+-]*")
88
88
 
89
89
 
90
- def parse_timestamp(value: str | None) -> datetime | None:
91
- if not value:
90
+ def parse_timestamp(value: object) -> datetime | None:
91
+ if not isinstance(value, str) or not value:
92
92
  return None
93
93
  try:
94
- return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(UTC)
94
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
95
+ if parsed.tzinfo is None or parsed.utcoffset() is None:
96
+ return None
97
+ return parsed.astimezone(UTC)
95
98
  except ValueError:
96
99
  return None
97
100
 
@@ -99,7 +102,8 @@ def parse_timestamp(value: str | None) -> datetime | None:
99
102
  def infer_project(
100
103
  metadata: dict[str, Any], mappings: list[tuple[str, str]]
101
104
  ) -> tuple[str, str]:
102
- cwd = str(metadata.get("cwd") or "").rstrip("/\\")
105
+ raw_cwd = metadata.get("cwd")
106
+ cwd = raw_cwd.rstrip("/\\") if isinstance(raw_cwd, str) else ""
103
107
  normalized_cwd = cwd.replace("\\", "/")
104
108
  for name, prefix in sorted(mappings, key=lambda item: len(item[1]), reverse=True):
105
109
  normalized_prefix = prefix.replace("\\", "/").rstrip("/")
@@ -109,17 +113,18 @@ def infer_project(
109
113
  return name, "mapping"
110
114
  git = metadata.get("git")
111
115
  if isinstance(git, dict):
112
- repository = str(git.get("repository_url") or git.get("repository") or "")
116
+ raw_repository = git.get("repository_url") or git.get("repository")
117
+ repository = raw_repository if isinstance(raw_repository, str) else ""
113
118
  slug = re.split(r"[/\\:]", repository.rstrip("/\\"))[-1]
114
119
  if slug.endswith(".git"):
115
120
  slug = slug[:-4]
116
- if slug:
121
+ if _safe_dimension(slug, 255):
117
122
  return slug, "git"
118
123
  return OUTSIDE_PROJECT, "none"
119
124
 
120
125
 
121
126
  def extract_tools(payload: dict[str, Any]) -> list[tuple[str, str]]:
122
- outer_name = str(payload.get("name", "unknown"))
127
+ outer_name = _safe_dimension(payload.get("name"), 512) or "unknown"
123
128
  if outer_name != "exec":
124
129
  return [(outer_name, outer_name)]
125
130
  raw_input = payload.get("input", "")
@@ -233,9 +238,16 @@ class CodexAdapter:
233
238
  continue
234
239
  event_count += 1
235
240
  if event.get("type") == "session_meta":
236
- conversation_id = str(event.get("payload", {}).get("id", ""))
237
- conversation_id = conversation_id or path.stem
238
- candidate = (machine, path, event_count, digest.hexdigest())
241
+ payload = event.get("payload")
242
+ if not isinstance(payload, dict):
243
+ malformed += 1
244
+ continue
245
+ candidate_id = _safe_dimension(payload.get("id"), 512)
246
+ if candidate_id and not conversation_id:
247
+ conversation_id = candidate_id
248
+ content_hash = digest.hexdigest()
249
+ conversation_id = conversation_id or f"content-{content_hash}"
250
+ candidate = (machine, path, event_count, content_hash)
239
251
  previous = selected.get(conversation_id)
240
252
  if previous is None:
241
253
  selected[conversation_id] = candidate
@@ -273,7 +285,16 @@ class CodexAdapter:
273
285
  ),
274
286
  {},
275
287
  )
276
- conversation_id = str(metadata.get("id") or path.stem)
288
+ conversation_id = next(
289
+ (
290
+ candidate_id
291
+ for event in events
292
+ if event.get("type") == "session_meta"
293
+ and isinstance((payload := event.get("payload")), dict)
294
+ and (candidate_id := _safe_dimension(payload.get("id"), 512))
295
+ ),
296
+ f"content-{digest}",
297
+ )
277
298
  record_id = f"codex:{conversation_id}"
278
299
  project, project_source = infer_project(metadata, mappings)
279
300
  timestamps = [
@@ -316,9 +337,7 @@ class CodexAdapter:
316
337
  {
317
338
  "id": f"{record_id}:compaction:{compaction_sequence}",
318
339
  "conversation_id": record_id,
319
- "turn_id": (
320
- f"{record_id}:{active_turn_id}" if active_turn_id else None
321
- ),
340
+ "turn_id": turns.get(active_turn_id or "", {}).get("id"),
322
341
  "sequence": compaction_sequence,
323
342
  "timestamp": _iso(timestamp),
324
343
  }
@@ -343,7 +362,7 @@ class CodexAdapter:
343
362
  _merge_present(settings_by_turn[active_turn_id], updates)
344
363
  if event_type == "turn_context":
345
364
  active_turn_id = (
346
- str(payload.get("turn_id") or active_turn_id or "") or None
365
+ _safe_dimension(payload.get("turn_id"), 512) or active_turn_id
347
366
  )
348
367
  active_model = (
349
368
  _safe_dimension(payload.get("model"), 255) or active_model
@@ -362,7 +381,7 @@ class CodexAdapter:
362
381
  or setting_defaults["collaboration_mode"],
363
382
  }
364
383
  if event_type == "event_msg" and payload_type == "task_started":
365
- active_turn_id = str(payload.get("turn_id") or "") or None
384
+ active_turn_id = _safe_dimension(payload.get("turn_id"), 512)
366
385
  if active_turn_id:
367
386
  settings = settings_by_turn.setdefault(
368
387
  active_turn_id, dict(setting_defaults)
@@ -391,7 +410,9 @@ class CodexAdapter:
391
410
  "task_complete",
392
411
  "turn_aborted",
393
412
  }:
394
- turn_id = str(payload.get("turn_id") or active_turn_id or "")
413
+ turn_id = (
414
+ _safe_dimension(payload.get("turn_id"), 512) or active_turn_id or ""
415
+ )
395
416
  if turn_id in turns:
396
417
  turns[turn_id].update(
397
418
  ended_at=_iso(timestamp),
@@ -412,12 +433,13 @@ class CodexAdapter:
412
433
  work_sequence += 1
413
434
  started_at_ms = _integer_or_none(payload.get("started_at_ms"))
414
435
  completed_at_ms = _integer_or_none(payload.get("completed_at_ms"))
415
- turn_id = str(payload.get("turn_id") or active_turn_id or "") or None
436
+ turn_id = _safe_dimension(payload.get("turn_id") or active_turn_id, 512)
437
+ turn = turns.get(turn_id or "")
416
438
  snapshot.work_items.append(
417
439
  {
418
440
  "id": f"{record_id}:work:{work_sequence}",
419
441
  "conversation_id": record_id,
420
- "turn_id": f"{record_id}:{turn_id}" if turn_id else None,
442
+ "turn_id": turn["id"] if turn else None,
421
443
  "sequence": work_sequence,
422
444
  "kind": WORK_ITEM_KINDS.get(
423
445
  str(item.get("type") or ""), "other"
@@ -438,18 +460,12 @@ class CodexAdapter:
438
460
  if not isinstance(usage, dict):
439
461
  continue
440
462
  call_sequence += 1
441
- tokens = {
442
- field: _nonnegative_integer(usage.get(field))
443
- for field in TOKEN_FIELDS
444
- }
445
- tokens.update(_derived_tokens(tokens))
446
- for field, value in tokens.items():
447
- totals[field] += value
463
+ tokens = _usage_tokens(usage)
464
+ _accumulate_tokens(totals, tokens)
448
465
  turn = turns.get(active_turn_id or "")
449
466
  if turn:
450
467
  turn["model_calls"] += 1
451
- for field, value in tokens.items():
452
- turn[field] += value
468
+ _accumulate_tokens(turn, tokens)
453
469
  snapshot.model_calls.append(
454
470
  {
455
471
  "id": f"{record_id}:model:{call_sequence}",
@@ -548,24 +564,62 @@ class CodexAdapter:
548
564
  )
549
565
 
550
566
 
551
- def _derived_tokens(tokens: dict[str, int]) -> dict[str, int]:
567
+ def _usage_tokens(usage: dict[str, Any]) -> dict[str, int]:
568
+ raw = {field: _nonnegative_integer(usage.get(field)) for field in TOKEN_FIELDS}
569
+ cached = raw["cached_input_tokens"]
570
+ cache_write = min(raw["cache_write_input_tokens"], MAX_BIGINT - cached)
571
+ input_tokens = max(raw["input_tokens"], cached + cache_write)
572
+ uncached = input_tokens - cached - cache_write
573
+ remaining = MAX_BIGINT - input_tokens
574
+ reasoning = min(raw["reasoning_output_tokens"], remaining)
575
+ visible = min(
576
+ max(0, raw["output_tokens"] - raw["reasoning_output_tokens"]),
577
+ remaining - reasoning,
578
+ )
579
+ output_tokens = reasoning + visible
580
+ unattributed = min(
581
+ max(0, raw["total_tokens"] - input_tokens - output_tokens),
582
+ MAX_BIGINT - input_tokens - output_tokens,
583
+ )
552
584
  return {
553
- "uncached_input_tokens": max(
554
- 0,
555
- tokens["input_tokens"]
556
- - tokens["cached_input_tokens"]
557
- - tokens["cache_write_input_tokens"],
558
- ),
559
- "visible_output_tokens": max(
560
- 0, tokens["output_tokens"] - tokens["reasoning_output_tokens"]
561
- ),
562
- "unattributed_tokens": max(
563
- 0,
564
- tokens["total_tokens"] - tokens["input_tokens"] - tokens["output_tokens"],
565
- ),
585
+ "input_tokens": input_tokens,
586
+ "cached_input_tokens": cached,
587
+ "cache_write_input_tokens": cache_write,
588
+ "output_tokens": output_tokens,
589
+ "reasoning_output_tokens": reasoning,
590
+ "total_tokens": input_tokens + output_tokens + unattributed,
591
+ "uncached_input_tokens": uncached,
592
+ "visible_output_tokens": visible,
593
+ "unattributed_tokens": unattributed,
566
594
  }
567
595
 
568
596
 
597
+ def _accumulate_tokens(target: dict[str, Any], value: dict[str, int]) -> None:
598
+ remaining = MAX_BIGINT
599
+ for field in (
600
+ "uncached_input_tokens",
601
+ "cached_input_tokens",
602
+ "cache_write_input_tokens",
603
+ "visible_output_tokens",
604
+ "reasoning_output_tokens",
605
+ "unattributed_tokens",
606
+ ):
607
+ amount = min(remaining, int(target[field]) + value[field])
608
+ target[field] = amount
609
+ remaining -= amount
610
+ target["input_tokens"] = (
611
+ target["uncached_input_tokens"]
612
+ + target["cached_input_tokens"]
613
+ + target["cache_write_input_tokens"]
614
+ )
615
+ target["output_tokens"] = (
616
+ target["visible_output_tokens"] + target["reasoning_output_tokens"]
617
+ )
618
+ target["total_tokens"] = (
619
+ target["input_tokens"] + target["output_tokens"] + target["unattributed_tokens"]
620
+ )
621
+
622
+
569
623
  def _safe_dimension(value: object, maximum: int) -> str | None:
570
624
  if not isinstance(value, str):
571
625
  return None
@@ -650,5 +704,9 @@ def _integer_or_none(value: object) -> int | None:
650
704
 
651
705
 
652
706
  def _nonnegative_integer(value: object) -> int:
653
- parsed = _integer_or_none(value)
707
+ parsed = (
708
+ _integer_or_none(value)
709
+ if not isinstance(value, float) or value.is_integer()
710
+ else None
711
+ )
654
712
  return max(0, parsed or 0)
@@ -18,6 +18,9 @@ from cli_consumption.adapters._shared import (
18
18
  ProviderInputBudget,
19
19
  read_bounded_bytes,
20
20
  )
21
+ from cli_consumption.adapters._shared import (
22
+ add_tokens as _add_tokens,
23
+ )
21
24
  from cli_consumption.adapters._shared import (
22
25
  bounded_sum as _sum,
23
26
  )
@@ -96,6 +99,7 @@ class ContinueAdapter:
96
99
  active: int | None = None
97
100
  raw_calls: list[tuple[int | None, str, dict[str, int]]] = []
98
101
  tools: list[tuple[int | None, str]] = []
102
+ compactions: list[int | None] = []
99
103
  seen_tools: set[str] = set()
100
104
 
101
105
  for item in history:
@@ -126,6 +130,8 @@ class ContinueAdapter:
126
130
 
127
131
  if role != "assistant":
128
132
  continue
133
+ if isinstance(item.get("conversationSummary"), str):
134
+ compactions.append(active)
129
135
  model = _message_model(item) or session_model or "unknown"
130
136
  usage = _usage(message.get("usage"))
131
137
  raw_calls.append((active, model, usage))
@@ -152,7 +158,7 @@ class ContinueAdapter:
152
158
  totals = empty_tokens()
153
159
  models: set[str] = set()
154
160
  for sequence, (turn_index, model, usage) in enumerate(raw_calls, 1):
155
- tokens = _tokens(usage)
161
+ tokens = _tokens(usage, model)
156
162
  models.add(model)
157
163
  _add_tokens(totals, tokens)
158
164
  turn = turns[turn_index] if turn_index is not None else None
@@ -187,6 +193,18 @@ class ContinueAdapter:
187
193
  }
188
194
  )
189
195
 
196
+ for sequence, turn_index in enumerate(compactions, 1):
197
+ turn = turns[turn_index] if turn_index is not None else None
198
+ snapshot.compaction_events.append(
199
+ {
200
+ "id": f"{conversation_id}:compaction:{sequence}",
201
+ "conversation_id": conversation_id,
202
+ "turn_id": turn["id"] if turn else None,
203
+ "sequence": sequence,
204
+ "timestamp": None,
205
+ }
206
+ )
207
+
190
208
  for index, turn in enumerate(turns):
191
209
  snapshot.turns.append(turn)
192
210
  observed_models = turn_models[index]
@@ -224,7 +242,7 @@ class ContinueAdapter:
224
242
  "iterations": len(turns),
225
243
  "model_calls": len(raw_calls),
226
244
  "tool_calls": len(tools),
227
- "compactions": 0,
245
+ "compactions": len(compactions),
228
246
  "event_count": source.event_count,
229
247
  "content_hash": source.digest,
230
248
  **totals,
@@ -318,11 +336,25 @@ def _residual_usage(
318
336
  }
319
337
 
320
338
 
321
- def _tokens(usage: dict[str, int]) -> dict[str, int]:
322
- input_tokens = usage["prompt"]
323
- cached = min(input_tokens, usage["cached"])
324
- cache_write = min(max(0, input_tokens - cached), usage["cache_write"])
325
- output_tokens = usage["completion"]
339
+ def _tokens(usage: dict[str, int], model: str) -> dict[str, int]:
340
+ prompt = usage["prompt"]
341
+ provider = model.partition("/")[0].casefold() if "/" in model else ""
342
+ separate_cache = (
343
+ provider in {"anthropic", "bedrock"}
344
+ or usage["cache_write"] > 0
345
+ or usage["cached"] > prompt
346
+ )
347
+ if separate_cache:
348
+ uncached = prompt
349
+ cached = min(usage["cached"], MAX_BIGINT - uncached)
350
+ cache_write = min(usage["cache_write"], MAX_BIGINT - uncached - cached)
351
+ input_tokens = uncached + cached + cache_write
352
+ else:
353
+ input_tokens = prompt
354
+ cached = min(input_tokens, usage["cached"])
355
+ cache_write = min(max(0, input_tokens - cached), usage["cache_write"])
356
+ uncached = input_tokens - cached - cache_write
357
+ output_tokens = min(usage["completion"], MAX_BIGINT - input_tokens)
326
358
  reasoning = min(output_tokens, usage["reasoning"])
327
359
  return {
328
360
  "input_tokens": input_tokens,
@@ -331,7 +363,7 @@ def _tokens(usage: dict[str, int]) -> dict[str, int]:
331
363
  "output_tokens": output_tokens,
332
364
  "reasoning_output_tokens": reasoning,
333
365
  "total_tokens": _sum(input_tokens, output_tokens),
334
- "uncached_input_tokens": input_tokens - cached - cache_write,
366
+ "uncached_input_tokens": uncached,
335
367
  "visible_output_tokens": output_tokens - reasoning,
336
368
  "unattributed_tokens": 0,
337
369
  }
@@ -368,11 +400,6 @@ def _has_visible_content(value: object) -> bool:
368
400
  )
369
401
 
370
402
 
371
- def _add_tokens(target: dict[str, Any], values: dict[str, int]) -> None:
372
- for key in empty_tokens():
373
- target[key] = _sum(target[key], values[key])
374
-
375
-
376
403
  def _counter(value: object) -> int:
377
404
  if isinstance(value, bool):
378
405
  return 0
@@ -609,7 +609,10 @@ def _timestamp(value: object) -> datetime | None:
609
609
  if isinstance(value, int | float) and math.isfinite(value):
610
610
  return datetime.fromtimestamp(value / 1000, UTC)
611
611
  if isinstance(value, str) and value:
612
- return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(UTC)
612
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
613
+ if parsed.tzinfo is None or parsed.utcoffset() is None:
614
+ return None
615
+ return parsed.astimezone(UTC)
613
616
  except (OSError, OverflowError, ValueError):
614
617
  pass
615
618
  return None
@@ -198,7 +198,8 @@ ADAPTER_SPECS = (
198
198
  "session schema (unversioned)",
199
199
  "session JSON",
200
200
  "https://github.com/continuedev/continue/tree/5522c6f44ca0ac3528b37244818fbfa39b5af470",
201
- "No reliable message timing, context windows, compactions, or latency.",
201
+ "No reliable message timing, context windows, compaction timing, "
202
+ "or latency.",
202
203
  ),
203
204
  ),
204
205
  AdapterSpec(
@@ -38,6 +38,11 @@ app = typer.Typer(
38
38
  help="Analyze AI coding CLI consumption without exporting conversation content.",
39
39
  no_args_is_help=True,
40
40
  )
41
+ snapshot_app = typer.Typer(
42
+ help="Create and ingest signed, compressed metadata-only snapshot files.",
43
+ no_args_is_help=True,
44
+ )
45
+ app.add_typer(snapshot_app, name="snapshot")
41
46
 
42
47
 
43
48
  class CollectionFailure(RuntimeError):
@@ -193,6 +198,154 @@ def collect(
193
198
  )
194
199
 
195
200
 
201
+ @snapshot_app.command("create")
202
+ def snapshot_create(
203
+ signing_key: Annotated[Path, typer.Option(help="Ed25519 private key in PEM form.")],
204
+ output: Annotated[Path, typer.Option("--output", "-o")],
205
+ source: Annotated[
206
+ list[str] | None,
207
+ typer.Option("--source", "-s", help="[LABEL=]PROVIDER_HOME. Repeat as needed."),
208
+ ] = None,
209
+ provider: Annotated[
210
+ str, typer.Option(help="CLI provider to collect, or 'all' to auto-detect.")
211
+ ] = "codex",
212
+ project: Annotated[
213
+ list[str] | None,
214
+ typer.Option("--project", help="NAME=PATH_PREFIX project mapping."),
215
+ ] = None,
216
+ strict: Annotated[
217
+ bool,
218
+ typer.Option(
219
+ "--strict",
220
+ help="Refuse creation when any malformed provider record was skipped.",
221
+ ),
222
+ ] = False,
223
+ json_output: Annotated[
224
+ bool, typer.Option("--json", help="Emit a deterministic JSON result.")
225
+ ] = False,
226
+ ) -> None:
227
+ """Collect metadata and write one authenticated offline snapshot file."""
228
+ from cli_consumption.snapshot_files import SnapshotFileError, write_snapshot_file
229
+
230
+ try:
231
+ snapshots = _collect_snapshots(provider, source, project)
232
+ except CollectionFailure as error:
233
+ _abort_snapshot(error.code, json_output=json_output)
234
+ except Exception:
235
+ _abort_snapshot("local_collection_failed", json_output=json_output)
236
+ if strict and any(snapshot.malformed_records for snapshot in snapshots):
237
+ _abort_snapshot("malformed_records", json_output=json_output)
238
+ try:
239
+ write_snapshot_file(snapshots, output, signing_key)
240
+ except (OSError, SnapshotFileError, SnapshotValidationError) as error:
241
+ _abort_snapshot(
242
+ getattr(error, "code", "snapshot_file_invalid"),
243
+ json_output=json_output,
244
+ )
245
+ result = {
246
+ "providers": [snapshot.provider for snapshot in snapshots],
247
+ "snapshots": len(snapshots),
248
+ }
249
+ if json_output:
250
+ typer.echo(json.dumps(result, sort_keys=True, separators=(",", ":")))
251
+ else:
252
+ typer.echo(f"Wrote {len(snapshots)} signed metadata snapshots.")
253
+
254
+
255
+ @snapshot_app.command("ingest")
256
+ def snapshot_ingest(
257
+ input_path: Annotated[Path, typer.Option("--input", "-i")],
258
+ verification_key: Annotated[
259
+ Path, typer.Option(help="Trusted Ed25519 public key in PEM form.")
260
+ ],
261
+ database: Annotated[
262
+ str,
263
+ typer.Option(
264
+ "--database",
265
+ "-d",
266
+ envvar="CLI_CONSUMPTION_DATABASE",
267
+ help="SQLite path or SQLAlchemy PostgreSQL URL.",
268
+ ),
269
+ ] = "cli-consumption.sqlite",
270
+ json_output: Annotated[
271
+ bool, typer.Option("--json", help="Emit a deterministic JSON result.")
272
+ ] = False,
273
+ ) -> None:
274
+ """Verify and ingest an authenticated offline snapshot file."""
275
+ from cli_consumption.snapshot_files import SnapshotFileError, read_snapshot_file
276
+
277
+ try:
278
+ snapshots = read_snapshot_file(input_path, verification_key)
279
+ except (OSError, SnapshotFileError, SnapshotValidationError) as error:
280
+ _abort_snapshot(
281
+ getattr(error, "code", "snapshot_file_invalid"),
282
+ json_output=json_output,
283
+ )
284
+ engine = _open_database(database)
285
+ try:
286
+ results = [
287
+ (snapshot, ingest_snapshot(engine, snapshot)) for snapshot in snapshots
288
+ ]
289
+ except SnapshotValidationError:
290
+ _abort_snapshot("snapshot_file_invalid", json_output=json_output)
291
+ finally:
292
+ engine.dispose()
293
+ payload = {
294
+ "ingestions": [
295
+ {
296
+ "provider": snapshot.provider,
297
+ "run_id": result.run_id,
298
+ "received": result.received,
299
+ "written": result.written,
300
+ "skipped": result.skipped,
301
+ "malformed": snapshot.malformed_records,
302
+ "duplicates": snapshot.duplicate_conversations,
303
+ }
304
+ for snapshot, result in results
305
+ ]
306
+ }
307
+ if json_output:
308
+ typer.echo(json.dumps(payload, sort_keys=True, separators=(",", ":")))
309
+ else:
310
+ typer.echo(f"Ingested {len(results)} verified metadata snapshots.")
311
+
312
+
313
+ def _abort_snapshot(code: str, *, json_output: bool) -> Never:
314
+ safe_codes = {
315
+ "invalid_snapshot",
316
+ "local_collection_failed",
317
+ "malformed_records",
318
+ "provider_collection_failed",
319
+ "provider_format_incompatible",
320
+ "provider_limit_exceeded",
321
+ "snapshot_dependency_missing",
322
+ "snapshot_file_invalid",
323
+ "snapshot_file_too_large",
324
+ "snapshot_key_invalid",
325
+ "snapshot_payload_too_large",
326
+ "snapshot_signature_invalid",
327
+ }
328
+ bounded_code = code if code in safe_codes else "snapshot_file_invalid"
329
+ if json_output:
330
+ typer.echo(
331
+ json.dumps(
332
+ {"error": {"code": bounded_code}},
333
+ sort_keys=True,
334
+ separators=(",", ":"),
335
+ )
336
+ )
337
+ else:
338
+ if bounded_code == "snapshot_dependency_missing":
339
+ typer.echo(
340
+ "Snapshot files require optional dependencies; "
341
+ "install cli-consumption[snapshots].",
342
+ err=True,
343
+ )
344
+ else:
345
+ typer.echo(f"Snapshot operation failed ({bounded_code}).", err=True)
346
+ raise typer.Exit(code=2) from None
347
+
348
+
196
349
  @app.command()
197
350
  def sync(
198
351
  endpoint: Annotated[
@@ -0,0 +1,255 @@
1
+ from __future__ import annotations
2
+
3
+ import gzip
4
+ import json
5
+ import os
6
+ import stat
7
+ import tempfile
8
+ import zlib
9
+ from contextlib import suppress
10
+ from io import BytesIO
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from cli_consumption.models import MAX_SNAPSHOT_RECORDS, Snapshot
15
+
16
+ FILE_MAGIC = b"CLI-CONSUMPTION-SNAPSHOT-V1\n"
17
+ SIGNATURE_SIZE = 64
18
+ MAX_SIGNED_FILE_BYTES = 64 * 1024 * 1024
19
+ MAX_DECOMPRESSED_BYTES = 256 * 1024 * 1024
20
+ MAX_KEY_BYTES = 64 * 1024
21
+ MAX_SNAPSHOTS = 64
22
+
23
+ _SNAPSHOT_COLLECTIONS = (
24
+ "conversations",
25
+ "turns",
26
+ "model_calls",
27
+ "tool_calls",
28
+ "work_items",
29
+ "context_samples",
30
+ "turn_settings",
31
+ "compaction_events",
32
+ "subagents",
33
+ )
34
+
35
+
36
+ class SnapshotFileError(ValueError):
37
+ """A bounded snapshot-file failure safe for CLI output."""
38
+
39
+ def __init__(self, code: str) -> None:
40
+ self.code = code
41
+ super().__init__(code)
42
+
43
+
44
+ def write_snapshot_file(
45
+ snapshots: list[Snapshot], output: Path, signing_key: Path
46
+ ) -> None:
47
+ """Validate, compress, sign, and atomically install an offline snapshot file."""
48
+ if not snapshots or len(snapshots) > MAX_SNAPSHOTS:
49
+ raise SnapshotFileError("snapshot_file_invalid")
50
+ validated = [Snapshot.from_dict(snapshot.to_dict()) for snapshot in snapshots]
51
+ _bound_total_records(validated)
52
+ payload = json.dumps(
53
+ {
54
+ "format": "cli-consumption.snapshot",
55
+ "format_version": 1,
56
+ "snapshots": [snapshot.to_dict() for snapshot in validated],
57
+ },
58
+ sort_keys=True,
59
+ separators=(",", ":"),
60
+ ).encode("utf-8")
61
+ if len(payload) > MAX_DECOMPRESSED_BYTES:
62
+ raise SnapshotFileError("snapshot_payload_too_large")
63
+ compressed = gzip.compress(payload, compresslevel=9, mtime=0)
64
+ private_key = _load_private_key(_read_regular_file(signing_key, MAX_KEY_BYTES))
65
+ signed_content = FILE_MAGIC + compressed
66
+ signature = private_key.sign(signed_content)
67
+ if len(signature) != SIGNATURE_SIZE:
68
+ raise SnapshotFileError("snapshot_key_invalid")
69
+ contents = FILE_MAGIC + signature + compressed
70
+ if len(contents) > MAX_SIGNED_FILE_BYTES:
71
+ raise SnapshotFileError("snapshot_file_too_large")
72
+ _atomic_write(output, contents)
73
+
74
+
75
+ def read_snapshot_file(input_path: Path, verification_key: Path) -> list[Snapshot]:
76
+ """Verify an offline snapshot before bounded decompression and strict parsing."""
77
+ contents = _read_regular_file(input_path, MAX_SIGNED_FILE_BYTES)
78
+ if len(contents) <= len(FILE_MAGIC) + SIGNATURE_SIZE or not contents.startswith(
79
+ FILE_MAGIC
80
+ ):
81
+ raise SnapshotFileError("snapshot_file_invalid")
82
+ signature_start = len(FILE_MAGIC)
83
+ signature_end = signature_start + SIGNATURE_SIZE
84
+ signature = contents[signature_start:signature_end]
85
+ compressed = contents[signature_end:]
86
+ public_key = _load_public_key(_read_regular_file(verification_key, MAX_KEY_BYTES))
87
+ try:
88
+ public_key.verify(signature, FILE_MAGIC + compressed)
89
+ except Exception as error:
90
+ if error.__class__.__module__.startswith("cryptography"):
91
+ raise SnapshotFileError("snapshot_signature_invalid") from None
92
+ raise
93
+ payload = _decompress_bounded(compressed)
94
+ try:
95
+ envelope = json.loads(payload, object_pairs_hook=_unique_object)
96
+ except (UnicodeDecodeError, json.JSONDecodeError, SnapshotFileError):
97
+ raise SnapshotFileError("snapshot_file_invalid") from None
98
+ if not isinstance(envelope, dict) or set(envelope) != {
99
+ "format",
100
+ "format_version",
101
+ "snapshots",
102
+ }:
103
+ raise SnapshotFileError("snapshot_file_invalid")
104
+ if (
105
+ envelope["format"] != "cli-consumption.snapshot"
106
+ or envelope["format_version"] != 1
107
+ or isinstance(envelope["format_version"], bool)
108
+ or not isinstance(envelope["snapshots"], list)
109
+ or not 1 <= len(envelope["snapshots"]) <= MAX_SNAPSHOTS
110
+ ):
111
+ raise SnapshotFileError("snapshot_file_invalid")
112
+ try:
113
+ snapshots = [
114
+ Snapshot.from_dict(value)
115
+ for value in envelope["snapshots"]
116
+ if isinstance(value, dict)
117
+ ]
118
+ except ValueError:
119
+ raise SnapshotFileError("snapshot_file_invalid") from None
120
+ if len(snapshots) != len(envelope["snapshots"]):
121
+ raise SnapshotFileError("snapshot_file_invalid")
122
+ _bound_total_records(snapshots)
123
+ return snapshots
124
+
125
+
126
+ def _load_private_key(data: bytes) -> Any:
127
+ try:
128
+ from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
129
+ from cryptography.hazmat.primitives.serialization import load_pem_private_key
130
+ except ModuleNotFoundError:
131
+ raise SnapshotFileError("snapshot_dependency_missing") from None
132
+ try:
133
+ key = load_pem_private_key(data, password=None)
134
+ except (TypeError, ValueError):
135
+ raise SnapshotFileError("snapshot_key_invalid") from None
136
+ if not isinstance(key, Ed25519PrivateKey):
137
+ raise SnapshotFileError("snapshot_key_invalid")
138
+ return key
139
+
140
+
141
+ def _load_public_key(data: bytes) -> Any:
142
+ try:
143
+ from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PublicKey
144
+ from cryptography.hazmat.primitives.serialization import load_pem_public_key
145
+ except ModuleNotFoundError:
146
+ raise SnapshotFileError("snapshot_dependency_missing") from None
147
+ try:
148
+ key = load_pem_public_key(data)
149
+ except ValueError:
150
+ raise SnapshotFileError("snapshot_key_invalid") from None
151
+ if not isinstance(key, Ed25519PublicKey):
152
+ raise SnapshotFileError("snapshot_key_invalid")
153
+ return key
154
+
155
+
156
+ def _read_regular_file(path: Path, limit: int) -> bytes:
157
+ flags = os.O_RDONLY | getattr(os, "O_BINARY", 0)
158
+ if hasattr(os, "O_NOFOLLOW"):
159
+ flags |= os.O_NOFOLLOW
160
+ try:
161
+ descriptor = os.open(path, flags)
162
+ except OSError:
163
+ raise SnapshotFileError("snapshot_file_invalid") from None
164
+ try:
165
+ file_stat = os.fstat(descriptor)
166
+ if not stat.S_ISREG(file_stat.st_mode) or file_stat.st_size > limit:
167
+ code = (
168
+ "snapshot_file_too_large"
169
+ if file_stat.st_size > limit
170
+ else "snapshot_file_invalid"
171
+ )
172
+ raise SnapshotFileError(code)
173
+ with os.fdopen(descriptor, "rb", closefd=False) as handle:
174
+ data = handle.read(limit + 1)
175
+ if len(data) > limit:
176
+ raise SnapshotFileError("snapshot_file_too_large")
177
+ return data
178
+ finally:
179
+ os.close(descriptor)
180
+
181
+
182
+ def _decompress_bounded(compressed: bytes) -> bytes:
183
+ try:
184
+ with gzip.GzipFile(fileobj=BytesIO(compressed), mode="rb") as handle:
185
+ payload = handle.read(MAX_DECOMPRESSED_BYTES + 1)
186
+ except (EOFError, OSError, zlib.error):
187
+ raise SnapshotFileError("snapshot_file_invalid") from None
188
+ if len(payload) > MAX_DECOMPRESSED_BYTES:
189
+ raise SnapshotFileError("snapshot_payload_too_large")
190
+ return payload
191
+
192
+
193
+ def _unique_object(pairs: list[tuple[str, object]]) -> dict[str, object]:
194
+ result: dict[str, object] = {}
195
+ for key, value in pairs:
196
+ if key in result:
197
+ raise SnapshotFileError("snapshot_file_invalid")
198
+ result[key] = value
199
+ return result
200
+
201
+
202
+ def _bound_total_records(snapshots: list[Snapshot]) -> None:
203
+ total = sum(
204
+ len(getattr(snapshot, collection))
205
+ for snapshot in snapshots
206
+ for collection in _SNAPSHOT_COLLECTIONS
207
+ )
208
+ if total > MAX_SNAPSHOT_RECORDS:
209
+ raise SnapshotFileError("snapshot_payload_too_large")
210
+
211
+
212
+ def _atomic_write(output: Path, contents: bytes) -> None:
213
+ output.parent.mkdir(parents=True, exist_ok=True)
214
+ mode = 0o600
215
+ try:
216
+ current = output.stat(follow_symlinks=False)
217
+ except FileNotFoundError:
218
+ pass
219
+ else:
220
+ if stat.S_ISREG(current.st_mode):
221
+ mode = stat.S_IMODE(current.st_mode)
222
+ temporary_path: Path | None = None
223
+ try:
224
+ with tempfile.NamedTemporaryFile(
225
+ mode="wb",
226
+ prefix=f".{output.name}.",
227
+ suffix=".tmp",
228
+ dir=output.parent,
229
+ delete=False,
230
+ ) as handle:
231
+ temporary_path = Path(handle.name)
232
+ os.chmod(temporary_path, mode)
233
+ handle.write(contents)
234
+ handle.flush()
235
+ os.fsync(handle.fileno())
236
+ os.replace(temporary_path, output)
237
+ _fsync_directory(output.parent)
238
+ finally:
239
+ if temporary_path is not None:
240
+ with suppress(FileNotFoundError):
241
+ temporary_path.unlink()
242
+
243
+
244
+ def _fsync_directory(directory: Path) -> None:
245
+ flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0)
246
+ try:
247
+ descriptor = os.open(directory, flags)
248
+ except OSError:
249
+ return
250
+ try:
251
+ os.fsync(descriptor)
252
+ except OSError:
253
+ pass
254
+ finally:
255
+ os.close(descriptor)
File without changes
File without changes