cli-consumption 0.0.3__tar.gz → 0.0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/PKG-INFO +17 -3
  2. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/README.md +16 -2
  3. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/docs/architecture.md +2 -2
  4. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/docs/provider-support.md +17 -4
  5. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/pyproject.toml +1 -1
  6. cli_consumption-0.0.4/src/cli_consumption/adapters/__init__.py +5 -0
  7. cli_consumption-0.0.4/src/cli_consumption/adapters/opencode.py +463 -0
  8. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/cli.py +12 -6
  9. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_cli.py +52 -1
  10. cli_consumption-0.0.4/tests/test_opencode_adapter.py +261 -0
  11. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/uv.lock +1 -1
  12. cli_consumption-0.0.3/src/cli_consumption/adapters/__init__.py +0 -4
  13. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/add-cli-adapter/SKILL.md +0 -0
  14. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/add-cli-adapter/agents/openai.yaml +0 -0
  15. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/audit-usage-privacy/SKILL.md +0 -0
  16. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/audit-usage-privacy/agents/openai.yaml +0 -0
  17. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/evolve-storage-schema/SKILL.md +0 -0
  18. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/evolve-storage-schema/agents/openai.yaml +0 -0
  19. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/yeet-github/SKILL.md +0 -0
  20. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/yeet-github/agents/openai.yaml +0 -0
  21. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/yolo/SKILL.md +0 -0
  22. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.agents/skills/yolo/agents/openai.yaml +0 -0
  23. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.github/workflows/ci.yml +0 -0
  24. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.github/workflows/release.yaml +0 -0
  25. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.gitignore +0 -0
  26. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.pre-commit-config.yaml +0 -0
  27. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/.python-version +0 -0
  28. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/AGENTS.md +0 -0
  29. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/CONTRIBUTING.md +0 -0
  30. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/LICENSE +0 -0
  31. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/NOTICE +0 -0
  32. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/docs/privacy.md +0 -0
  33. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/docs/roadmap.md +0 -0
  34. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/__init__.py +0 -0
  35. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/__main__.py +0 -0
  36. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/adapters/base.py +0 -0
  37. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/adapters/claude.py +0 -0
  38. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/adapters/codex.py +0 -0
  39. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/api.py +0 -0
  40. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/dashboard.py +0 -0
  41. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/exporting.py +0 -0
  42. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/models.py +0 -0
  43. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/py.typed +0 -0
  44. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/storage.py +0 -0
  45. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/src/cli_consumption/sync.py +0 -0
  46. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/conftest.py +0 -0
  47. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_api.py +0 -0
  48. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_claude_adapter.py +0 -0
  49. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_codex_adapter.py +0 -0
  50. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_storage_and_exports.py +0 -0
  51. {cli_consumption-0.0.3 → cli_consumption-0.0.4}/tests/test_sync.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cli-consumption
3
- Version: 0.0.3
3
+ Version: 0.0.4
4
4
  Summary: Analyze and consolidate AI coding CLI consumption across machines.
5
5
  Project-URL: Homepage, https://github.com/Guillaume-Lombardo/cli-consumption
6
6
  Project-URL: Documentation, https://github.com/Guillaume-Lombardo/cli-consumption#readme
@@ -33,8 +33,8 @@ models, tokens, tools, conversations, and turns. It can analyze one workstation,
33
33
  consolidate copied data from several machines, or send metadata-only snapshots to a
34
34
  central collector.
35
35
 
36
- Codex and the core Claude Code local transcript format are supported. OpenCode, Kilo
37
- Code, and Pi are planned behind the same provider-neutral adapter contract.
36
+ Codex, OpenCode, and the core Claude Code local transcript format are supported. Kilo
37
+ Code and Pi are planned behind the same provider-neutral adapter contract.
38
38
 
39
39
  The collector deliberately excludes prompts, responses, tool arguments, and
40
40
  credentials. See [Privacy](docs/privacy.md) before sharing a database or export.
@@ -102,6 +102,12 @@ Select Claude Code to read `~/.claude/projects/` instead:
102
102
  uv run cli-consumption collect --provider claude --database usage.sqlite
103
103
  ```
104
104
 
105
+ Select OpenCode to read `~/.local/share/opencode/opencode.db` instead:
106
+
107
+ ```bash
108
+ uv run cli-consumption collect --provider opencode --database usage.sqlite
109
+ ```
110
+
105
111
  Use `all` to detect and collect every supported provider present on the machine:
106
112
 
107
113
  ```bash
@@ -142,6 +148,14 @@ uv run cli-consumption collect --provider claude \
142
148
  --source laptop=/data/claude/laptop
143
149
  ```
144
150
 
151
+ Copied OpenCode sources point to the data directory containing `opencode.db`:
152
+
153
+ ```bash
154
+ uv run cli-consumption collect --provider opencode \
155
+ --source desktop=/data/opencode/desktop \
156
+ --source laptop=/data/opencode/laptop
157
+ ```
158
+
145
159
  ## SQLite and PostgreSQL
146
160
 
147
161
  A file path selects SQLite. A SQLAlchemy URL selects PostgreSQL:
@@ -5,8 +5,8 @@ models, tokens, tools, conversations, and turns. It can analyze one workstation,
5
5
  consolidate copied data from several machines, or send metadata-only snapshots to a
6
6
  central collector.
7
7
 
8
- Codex and the core Claude Code local transcript format are supported. OpenCode, Kilo
9
- Code, and Pi are planned behind the same provider-neutral adapter contract.
8
+ Codex, OpenCode, and the core Claude Code local transcript format are supported. Kilo
9
+ Code and Pi are planned behind the same provider-neutral adapter contract.
10
10
 
11
11
  The collector deliberately excludes prompts, responses, tool arguments, and
12
12
  credentials. See [Privacy](docs/privacy.md) before sharing a database or export.
@@ -74,6 +74,12 @@ Select Claude Code to read `~/.claude/projects/` instead:
74
74
  uv run cli-consumption collect --provider claude --database usage.sqlite
75
75
  ```
76
76
 
77
+ Select OpenCode to read `~/.local/share/opencode/opencode.db` instead:
78
+
79
+ ```bash
80
+ uv run cli-consumption collect --provider opencode --database usage.sqlite
81
+ ```
82
+
77
83
  Use `all` to detect and collect every supported provider present on the machine:
78
84
 
79
85
  ```bash
@@ -114,6 +120,14 @@ uv run cli-consumption collect --provider claude \
114
120
  --source laptop=/data/claude/laptop
115
121
  ```
116
122
 
123
+ Copied OpenCode sources point to the data directory containing `opencode.db`:
124
+
125
+ ```bash
126
+ uv run cli-consumption collect --provider opencode \
127
+ --source desktop=/data/opencode/desktop \
128
+ --source laptop=/data/opencode/laptop
129
+ ```
130
+
117
131
  ## SQLite and PostgreSQL
118
132
 
119
133
  A file path selects SQLite. A SQLAlchemy URL selects PostgreSQL:
@@ -16,8 +16,8 @@ provider files -> adapter -> metadata-only snapshot -> SQL storage -> dashboard/
16
16
 
17
17
  - `adapters`: parse a CLI's local data into conversations, turns, model calls, tool
18
18
  calls, context-pressure samples, bounded turn settings, compactions, and content-free
19
- work-item intervals. Codex exposes the complete analytics contract; Claude Code
20
- exposes the core dimensions available in its local transcripts.
19
+ work-item intervals. Codex exposes the complete analytics contract; Claude Code and
20
+ OpenCode expose the core dimensions available in their local stores.
21
21
  - `models`: define the transport boundary shared by offline and API ingestion.
22
22
  - `storage`: owns the normalized schema, idempotent replacement rules, SQLite, and
23
23
  PostgreSQL engine creation.
@@ -4,7 +4,7 @@
4
4
  | --- | --- | --- |
5
5
  | Codex | Supported | Local rollout JSONL and optional metadata-only subagent state |
6
6
  | Claude Code | Supported (core) | Local project transcript JSONL |
7
- | OpenCode | Planned | To be verified before implementation |
7
+ | OpenCode | Supported (core) | Local SQLite v2 session store |
8
8
  | Kilo Code | Planned | To be verified before implementation |
9
9
  | Pi | Planned | To be verified before implementation |
10
10
 
@@ -36,6 +36,19 @@ is not billing data. This first increment does not collect subagent transcripts,
36
36
  context-window sizes, effort/service-tier settings, TTFT, provider-reported duration,
37
37
  or technical work-item intervals.
38
38
 
39
- Provider formats can change without notice. Unknown fields are ignored; malformed JSONL
40
- records are counted and skipped. Compatibility fixes should add a fixture for both the
41
- old and new format whenever possible.
39
+ OpenCode reads `opencode.db` from its XDG data directory (normally
40
+ `~/.local/share/opencode/`). It extracts v2 session messages, model references, token
41
+ usage, tool names, and compaction timestamps while discarding message text, reasoning,
42
+ tool inputs/results, shell commands/output, paths, titles, errors, costs, and arbitrary
43
+ metadata. Model labels combine OpenCode's provider and model identifiers.
44
+
45
+ OpenCode reports uncached input, cache reads, cache writes, visible output, and
46
+ reasoning separately. Normalized input and output totals include their respective
47
+ components. The adapter does not currently read pre-v2 JSON storage, legacy
48
+ `message`/`part` tables, child-session relationships, context-window sizes, or
49
+ provider-reported cost. The SQLite schema is internal and may change without notice;
50
+ local token events are not billing data.
51
+
52
+ Provider formats can change without notice. Unknown fields are ignored; malformed
53
+ provider records are counted and skipped. Compatibility fixes should add a fixture for
54
+ both the old and new format whenever possible.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "cli-consumption"
7
- version = "0.0.3"
7
+ version = "0.0.4"
8
8
  description = "Analyze and consolidate AI coding CLI consumption across machines."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.14"
@@ -0,0 +1,5 @@
1
+ from cli_consumption.adapters.claude import ClaudeAdapter
2
+ from cli_consumption.adapters.codex import CodexAdapter
3
+ from cli_consumption.adapters.opencode import OpenCodeAdapter
4
+
5
+ __all__ = ["ClaudeAdapter", "CodexAdapter", "OpenCodeAdapter"]
@@ -0,0 +1,463 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import math
6
+ import re
7
+ import sqlite3
8
+ from dataclasses import dataclass
9
+ from datetime import UTC, datetime
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from cli_consumption.models import Snapshot, empty_tokens
14
+
15
+ MAX_BIGINT = 9_223_372_036_854_775_807
16
+ SAFE_LABEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+-]*")
17
+
18
+
19
+ @dataclass(slots=True)
20
+ class _Message:
21
+ external_id: str
22
+ kind: str
23
+ sequence: int
24
+ created_at: datetime | None
25
+ updated_at: datetime | None
26
+ data: dict[str, Any]
27
+
28
+
29
+ @dataclass(slots=True)
30
+ class _Conversation:
31
+ machine: str
32
+ external_id: str
33
+ directory: str | None
34
+ created_at: datetime | None
35
+ updated_at: datetime | None
36
+ messages: list[_Message]
37
+ digest: str
38
+
39
+
40
+ class OpenCodeAdapter:
41
+ """Read metadata from OpenCode's SQLite v2 store without retaining content."""
42
+
43
+ name = "opencode"
44
+
45
+ def collect(
46
+ self,
47
+ sources: list[tuple[str, Path]],
48
+ project_mappings: list[tuple[str, str]] | None = None,
49
+ ) -> Snapshot:
50
+ selected: dict[str, _Conversation] = {}
51
+ duplicates = malformed = 0
52
+ for machine, home in sources:
53
+ database = home / "opencode.db"
54
+ if not database.is_file():
55
+ raise ValueError(f"Missing OpenCode database: {database}")
56
+ conversations, invalid = _read_database(database, machine)
57
+ malformed += invalid
58
+ for candidate in conversations:
59
+ previous = selected.get(candidate.external_id)
60
+ if previous is None:
61
+ selected[candidate.external_id] = candidate
62
+ continue
63
+ duplicates += 1
64
+ if _rank(candidate) > _rank(previous):
65
+ selected[candidate.external_id] = candidate
66
+
67
+ snapshot = Snapshot(
68
+ provider=self.name,
69
+ duplicate_conversations=duplicates,
70
+ malformed_records=malformed,
71
+ )
72
+ for conversation in sorted(
73
+ selected.values(), key=lambda item: item.external_id
74
+ ):
75
+ self._normalize(snapshot, conversation, project_mappings or [])
76
+ return snapshot
77
+
78
+ def _normalize(
79
+ self,
80
+ snapshot: Snapshot,
81
+ source: _Conversation,
82
+ mappings: list[tuple[str, str]],
83
+ ) -> None:
84
+ conversation_id = f"opencode:{source.external_id}"
85
+ turns: dict[str, dict[str, Any]] = {}
86
+ turn_models: dict[str, set[str]] = {}
87
+ active: str | None = None
88
+ effective_model: str | None = None
89
+ calls: list[tuple[str | None, datetime | None, str, dict[str, int]]] = []
90
+ tools: list[tuple[str | None, datetime | None, str]] = []
91
+ compactions: list[tuple[str | None, datetime | None]] = []
92
+ timestamps = [source.created_at]
93
+
94
+ for message in source.messages:
95
+ timestamp = message.created_at or _timestamp(
96
+ _mapping(message.data.get("time")).get("created")
97
+ )
98
+ completed_at = _timestamp(
99
+ _mapping(message.data.get("time")).get("completed")
100
+ )
101
+ timestamps.extend((timestamp, completed_at))
102
+
103
+ if message.kind == "user":
104
+ if active is not None:
105
+ _finish_turn(turns[active], timestamp)
106
+ active = message.external_id
107
+ turns[active] = {
108
+ "id": f"{conversation_id}:{active}",
109
+ "conversation_id": conversation_id,
110
+ "external_id": active,
111
+ "started_at": _iso(timestamp),
112
+ "ended_at": None,
113
+ "status": "in-progress",
114
+ "duration_ms": None,
115
+ "time_to_first_token_ms": None,
116
+ "model_calls": 0,
117
+ "tool_calls": 0,
118
+ **empty_tokens(),
119
+ }
120
+ turn_models[active] = set()
121
+ continue
122
+
123
+ if message.kind == "model-switched":
124
+ effective_model = _model(message.data.get("model"))
125
+ continue
126
+
127
+ if message.kind == "compaction":
128
+ compactions.append((active, timestamp))
129
+ continue
130
+
131
+ if message.kind != "assistant":
132
+ continue
133
+
134
+ model = _model(message.data.get("model")) or effective_model or "unknown"
135
+ effective_model = model if model != "unknown" else effective_model
136
+ tokens = _usage(message.data.get("tokens"))
137
+ calls.append((active, timestamp, model, tokens))
138
+ turn = turns.get(active or "")
139
+ if turn:
140
+ turn["model_calls"] += 1
141
+ turn_models[active or ""].add(model)
142
+ _add_tokens(turn, tokens)
143
+ turn["ended_at"] = _iso(completed_at or timestamp) or turn["ended_at"]
144
+ if isinstance(message.data.get("error"), dict):
145
+ turn["status"] = "aborted"
146
+ elif (
147
+ completed_at is not None or _label(message.data.get("finish"), 255)
148
+ ) and turn["status"] != "aborted":
149
+ turn["status"] = "completed"
150
+
151
+ content = message.data.get("content")
152
+ if not isinstance(content, list):
153
+ continue
154
+ for part in content:
155
+ if not isinstance(part, dict) or part.get("type") != "tool":
156
+ continue
157
+ name = _label(part.get("name"), 512)
158
+ if not name:
159
+ continue
160
+ part_time = _mapping(part.get("time"))
161
+ tools.append(
162
+ (active, _timestamp(part_time.get("created")) or timestamp, name)
163
+ )
164
+
165
+ ended_at = max(
166
+ (value for value in timestamps if value is not None), default=None
167
+ )
168
+ if active is not None:
169
+ _finish_turn(turns[active], ended_at or source.updated_at)
170
+
171
+ totals = empty_tokens()
172
+ models: set[str] = set()
173
+ for sequence, (turn_key, timestamp, model, tokens) in enumerate(calls, 1):
174
+ models.add(model)
175
+ _add_tokens(totals, tokens)
176
+ turn = turns.get(turn_key or "")
177
+ snapshot.model_calls.append(
178
+ {
179
+ "id": f"{conversation_id}:model:{sequence}",
180
+ "conversation_id": conversation_id,
181
+ "turn_id": turn["id"] if turn else None,
182
+ "sequence": sequence,
183
+ "timestamp": _iso(timestamp),
184
+ "model": model,
185
+ **tokens,
186
+ }
187
+ )
188
+
189
+ for sequence, (turn_key, timestamp, name) in enumerate(tools, 1):
190
+ turn = turns.get(turn_key or "")
191
+ if turn:
192
+ turn["tool_calls"] += 1
193
+ snapshot.tool_calls.append(
194
+ {
195
+ "id": f"{conversation_id}:tool:{sequence}",
196
+ "conversation_id": conversation_id,
197
+ "turn_id": turn["id"] if turn else None,
198
+ "sequence": sequence,
199
+ "timestamp": _iso(timestamp),
200
+ "tool_name": name,
201
+ "outer_tool_name": name,
202
+ }
203
+ )
204
+
205
+ for sequence, (turn_key, timestamp) in enumerate(compactions, 1):
206
+ turn = turns.get(turn_key or "")
207
+ snapshot.compaction_events.append(
208
+ {
209
+ "id": f"{conversation_id}:compaction:{sequence}",
210
+ "conversation_id": conversation_id,
211
+ "turn_id": turn["id"] if turn else None,
212
+ "sequence": sequence,
213
+ "timestamp": _iso(timestamp),
214
+ }
215
+ )
216
+
217
+ for key, turn in turns.items():
218
+ snapshot.turns.append(turn)
219
+ observed = turn_models[key]
220
+ snapshot.turn_settings.append(
221
+ {
222
+ "id": f"{conversation_id}:settings:{key}",
223
+ "conversation_id": conversation_id,
224
+ "turn_id": turn["id"],
225
+ "model": next(iter(observed)) if len(observed) == 1 else None,
226
+ "effort": None,
227
+ "collaboration_mode": None,
228
+ "service_tier": None,
229
+ "context_window_tokens": None,
230
+ }
231
+ )
232
+
233
+ started_at = source.created_at or min(
234
+ (value for value in timestamps if value is not None), default=None
235
+ )
236
+ ended_at = ended_at or source.updated_at
237
+ project, project_source = _project(source.directory, mappings)
238
+ snapshot.conversations.append(
239
+ {
240
+ "id": conversation_id,
241
+ "provider": self.name,
242
+ "external_id": source.external_id,
243
+ "source_machine": source.machine,
244
+ "project": project,
245
+ "project_source": project_source,
246
+ "started_at": _iso(started_at),
247
+ "ended_at": _iso(ended_at),
248
+ "duration_seconds": (
249
+ max(0.0, (ended_at - started_at).total_seconds())
250
+ if started_at and ended_at
251
+ else None
252
+ ),
253
+ "source": "local-sqlite-v2",
254
+ "models": sorted(models),
255
+ "iterations": len(turns),
256
+ "model_calls": len(calls),
257
+ "tool_calls": len(tools),
258
+ "compactions": len(compactions),
259
+ "event_count": len(source.messages),
260
+ "content_hash": source.digest,
261
+ **totals,
262
+ }
263
+ )
264
+
265
+
266
+ def _read_database(path: Path, machine: str) -> tuple[list[_Conversation], int]:
267
+ try:
268
+ connection = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True)
269
+ connection.row_factory = sqlite3.Row
270
+ connection.execute("PRAGMA trusted_schema=OFF")
271
+ connection.execute("PRAGMA query_only=ON")
272
+ session_columns = _columns(connection, "session")
273
+ message_columns = _columns(connection, "session_message")
274
+ required_session = {"id", "time_created", "time_updated"}
275
+ required_message = {
276
+ "id",
277
+ "session_id",
278
+ "type",
279
+ "seq",
280
+ "time_created",
281
+ "time_updated",
282
+ "data",
283
+ }
284
+ if (
285
+ not required_session <= session_columns
286
+ or not required_message <= message_columns
287
+ ):
288
+ raise ValueError(f"Unsupported OpenCode database schema: {path}")
289
+
290
+ directory = "directory" if "directory" in session_columns else "NULL"
291
+ rows = connection.execute(
292
+ f"SELECT id, {directory} AS directory, time_created, time_updated "
293
+ "FROM session ORDER BY id"
294
+ ).fetchall()
295
+ conversations: list[_Conversation] = []
296
+ malformed = 0
297
+ for row in rows:
298
+ external_id = _label(row["id"], 500)
299
+ if not external_id:
300
+ malformed += 1
301
+ continue
302
+ message_rows = connection.execute(
303
+ "SELECT id, type, seq, time_created, time_updated, data "
304
+ "FROM session_message WHERE session_id = ? ORDER BY seq, id",
305
+ (row["id"],),
306
+ ).fetchall()
307
+ digest = hashlib.sha256()
308
+ messages: list[_Message] = []
309
+ for message_row in message_rows:
310
+ digest.update(str(tuple(message_row)).encode())
311
+ message_id = _label(message_row["id"], 512)
312
+ kind = _label(message_row["type"], 64)
313
+ try:
314
+ data = json.loads(message_row["data"])
315
+ except (json.JSONDecodeError, TypeError, UnicodeDecodeError):
316
+ malformed += 1
317
+ continue
318
+ if not message_id or not kind or not isinstance(data, dict):
319
+ malformed += 1
320
+ continue
321
+ sequence = _counter(message_row["seq"])
322
+ messages.append(
323
+ _Message(
324
+ message_id,
325
+ kind,
326
+ sequence,
327
+ _timestamp(message_row["time_created"]),
328
+ _timestamp(message_row["time_updated"]),
329
+ data,
330
+ )
331
+ )
332
+ conversations.append(
333
+ _Conversation(
334
+ machine=machine,
335
+ external_id=external_id,
336
+ directory=row["directory"]
337
+ if isinstance(row["directory"], str)
338
+ else None,
339
+ created_at=_timestamp(row["time_created"]),
340
+ updated_at=_timestamp(row["time_updated"]),
341
+ messages=messages,
342
+ digest=digest.hexdigest(),
343
+ )
344
+ )
345
+ return conversations, malformed
346
+ except sqlite3.DatabaseError:
347
+ raise ValueError(f"Could not read OpenCode database: {path}") from None
348
+ finally:
349
+ if "connection" in locals():
350
+ connection.close()
351
+
352
+
353
+ def _columns(connection: sqlite3.Connection, table: str) -> set[str]:
354
+ return {
355
+ str(row["name"]) for row in connection.execute(f'PRAGMA table_info("{table}")')
356
+ }
357
+
358
+
359
+ def _rank(value: _Conversation) -> tuple[int, datetime, str]:
360
+ return (
361
+ len(value.messages),
362
+ value.updated_at or datetime.min.replace(tzinfo=UTC),
363
+ value.digest,
364
+ )
365
+
366
+
367
+ def _usage(value: object) -> dict[str, int]:
368
+ usage = _mapping(value)
369
+ cache = _mapping(usage.get("cache"))
370
+ uncached = _counter(usage.get("input"))
371
+ cached = _counter(cache.get("read"))
372
+ cache_write = _counter(cache.get("write"))
373
+ visible = _counter(usage.get("output"))
374
+ reasoning = _counter(usage.get("reasoning"))
375
+ input_tokens = _sum(uncached, cached, cache_write)
376
+ output_tokens = _sum(visible, reasoning)
377
+ attributed = _sum(input_tokens, output_tokens)
378
+ reported = _counter(usage.get("total"))
379
+ total = max(attributed, reported)
380
+ return {
381
+ "input_tokens": input_tokens,
382
+ "cached_input_tokens": cached,
383
+ "cache_write_input_tokens": cache_write,
384
+ "output_tokens": output_tokens,
385
+ "reasoning_output_tokens": reasoning,
386
+ "total_tokens": total,
387
+ "uncached_input_tokens": uncached,
388
+ "visible_output_tokens": visible,
389
+ "unattributed_tokens": max(0, total - attributed),
390
+ }
391
+
392
+
393
+ def _model(value: object) -> str | None:
394
+ model = _mapping(value)
395
+ provider = _label(model.get("providerID"), 127)
396
+ identifier = _label(model.get("id") or model.get("modelID"), 127)
397
+ if provider and identifier:
398
+ return f"{provider}/{identifier}"
399
+ return identifier
400
+
401
+
402
+ def _project(directory: str | None, mappings: list[tuple[str, str]]) -> tuple[str, str]:
403
+ if directory:
404
+ normalized = directory.replace("\\", "/").rstrip("/")
405
+ for name, prefix in sorted(
406
+ mappings, key=lambda item: len(item[1]), reverse=True
407
+ ):
408
+ prefix = prefix.replace("\\", "/").rstrip("/")
409
+ if normalized == prefix or normalized.startswith(prefix + "/"):
410
+ return name, "mapping"
411
+ return "outside-project", "none"
412
+
413
+
414
+ def _finish_turn(turn: dict[str, Any], fallback: datetime | None) -> None:
415
+ turn["ended_at"] = turn["ended_at"] or _iso(fallback)
416
+ start, end = _timestamp(turn["started_at"]), _timestamp(turn["ended_at"])
417
+ if start and end:
418
+ turn["duration_ms"] = max(0, int((end - start).total_seconds() * 1000))
419
+
420
+
421
+ def _mapping(value: object) -> dict[str, Any]:
422
+ return value if isinstance(value, dict) else {}
423
+
424
+
425
+ def _timestamp(value: object) -> datetime | None:
426
+ try:
427
+ if isinstance(value, bool):
428
+ return None
429
+ if isinstance(value, int | float) and math.isfinite(value):
430
+ return datetime.fromtimestamp(value / 1000, UTC)
431
+ if isinstance(value, str) and value:
432
+ return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(UTC)
433
+ except (OSError, OverflowError, ValueError):
434
+ pass
435
+ return None
436
+
437
+
438
+ def _iso(value: object) -> str | None:
439
+ return value.isoformat() if isinstance(value, datetime) else None
440
+
441
+
442
+ def _label(value: object, maximum: int) -> str | None:
443
+ if not isinstance(value, str):
444
+ return None
445
+ value = value.strip()
446
+ return value if 0 < len(value) <= maximum and SAFE_LABEL.fullmatch(value) else None
447
+
448
+
449
+ def _counter(value: object) -> int:
450
+ if isinstance(value, bool) or not isinstance(value, int | float):
451
+ return 0
452
+ if isinstance(value, float) and not math.isfinite(value):
453
+ return 0
454
+ return min(MAX_BIGINT, max(0, int(value)))
455
+
456
+
457
+ def _sum(*values: int) -> int:
458
+ return min(MAX_BIGINT, sum(values))
459
+
460
+
461
+ def _add_tokens(target: dict[str, Any], tokens: dict[str, int]) -> None:
462
+ for field, value in tokens.items():
463
+ target[field] = _sum(int(target[field]), value)
@@ -8,7 +8,7 @@ from typing import Annotated
8
8
  import typer
9
9
 
10
10
  from cli_consumption import __version__
11
- from cli_consumption.adapters import ClaudeAdapter, CodexAdapter
11
+ from cli_consumption.adapters import ClaudeAdapter, CodexAdapter, OpenCodeAdapter
12
12
  from cli_consumption.api import create_app
13
13
  from cli_consumption.dashboard import generate_dashboard
14
14
  from cli_consumption.exporting import export_csv
@@ -49,7 +49,8 @@ def providers() -> None:
49
49
  typer.echo("all auto-detect supported providers")
50
50
  typer.echo("codex supported")
51
51
  typer.echo("claude supported")
52
- for provider in ("opencode", "kilo", "pi"):
52
+ typer.echo("opencode supported")
53
+ for provider in ("kilo", "pi"):
53
54
  typer.echo(f"{provider:<8} planned")
54
55
 
55
56
 
@@ -224,6 +225,11 @@ def _collect_snapshots(
224
225
  adapters = {
225
226
  "codex": (CodexAdapter, ".codex", "sessions"),
226
227
  "claude": (ClaudeAdapter, ".claude", "projects"),
228
+ "opencode": (
229
+ OpenCodeAdapter,
230
+ ".local/share/opencode",
231
+ "opencode.db",
232
+ ),
227
233
  }
228
234
  if provider != "all" and provider not in adapters:
229
235
  raise typer.BadParameter(
@@ -243,7 +249,7 @@ def _collect_snapshots(
243
249
  sources = _parse_source_values(source_values)
244
250
  matched_labels: set[str] = set()
245
251
  for adapter, _, directory in adapters.values():
246
- matched = [source for source in sources if (source[1] / directory).is_dir()]
252
+ matched = [source for source in sources if (source[1] / directory).exists()]
247
253
  if matched:
248
254
  matched_labels.update(label for label, _ in matched)
249
255
  snapshots.append(adapter().collect(matched, mappings))
@@ -257,7 +263,7 @@ def _collect_snapshots(
257
263
  machine = platform.node()
258
264
  for adapter, home, directory in adapters.values():
259
265
  path = (Path.home() / home).resolve()
260
- if (path / directory).is_dir():
266
+ if (path / directory).exists():
261
267
  snapshots.append(adapter().collect([(machine, path)], mappings))
262
268
  if not snapshots:
263
269
  raise typer.BadParameter("No supported provider data detected.")
@@ -271,9 +277,9 @@ def _parse_sources(
271
277
  values = [f"{platform.node()}={Path.home() / home}"]
272
278
  result = _parse_source_values(values)
273
279
  for _, path in result:
274
- if not (path / directory).is_dir():
280
+ if not (path / directory).exists():
275
281
  raise typer.BadParameter(
276
- f"Missing {directory} directory: {path / directory}"
282
+ f"Missing provider data {directory}: {path / directory}"
277
283
  )
278
284
  return result
279
285
 
@@ -1,5 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import json
4
+ import sqlite3
3
5
  from pathlib import Path
4
6
 
5
7
  from click.utils import strip_ansi
@@ -22,6 +24,7 @@ def test_provider_status_is_explicit() -> None:
22
24
  assert "all auto-detect" in result.stdout
23
25
  assert "codex supported" in result.stdout
24
26
  assert "claude supported" in result.stdout
27
+ assert "opencode supported" in result.stdout
25
28
 
26
29
 
27
30
  def test_version_and_unsupported_provider_are_explicit(tmp_path: Path) -> None:
@@ -31,7 +34,7 @@ def test_version_and_unsupported_provider_are_explicit(tmp_path: Path) -> None:
31
34
 
32
35
  result = runner.invoke(
33
36
  app,
34
- ["collect", "--provider", "opencode", "--database", str(tmp_path / "db")],
37
+ ["collect", "--provider", "kilo", "--database", str(tmp_path / "db")],
35
38
  )
36
39
  assert result.exit_code == 2
37
40
  assert "not implemented yet" in result.output
@@ -62,6 +65,54 @@ def test_collects_claude_code(tmp_path: Path) -> None:
62
65
  assert "1 written" in result.stdout
63
66
 
64
67
 
68
+ def test_collects_opencode(tmp_path: Path) -> None:
69
+ home = tmp_path / "opencode"
70
+ home.mkdir()
71
+ connection = sqlite3.connect(home / "opencode.db")
72
+ connection.executescript(
73
+ """
74
+ CREATE TABLE session (
75
+ id TEXT PRIMARY KEY,
76
+ time_created INTEGER NOT NULL,
77
+ time_updated INTEGER NOT NULL
78
+ );
79
+ CREATE TABLE session_message (
80
+ id TEXT PRIMARY KEY,
81
+ session_id TEXT NOT NULL,
82
+ type TEXT NOT NULL,
83
+ seq INTEGER NOT NULL,
84
+ time_created INTEGER NOT NULL,
85
+ time_updated INTEGER NOT NULL,
86
+ data TEXT NOT NULL
87
+ );
88
+ """
89
+ )
90
+ connection.execute("INSERT INTO session VALUES ('ses_cli', 1000, 2000)")
91
+ connection.execute(
92
+ "INSERT INTO session_message VALUES (?, ?, ?, ?, ?, ?, ?)",
93
+ ("msg_cli", "ses_cli", "user", 1, 1000, 1000, json.dumps({"text": "x"})),
94
+ )
95
+ connection.commit()
96
+ connection.close()
97
+
98
+ result = runner.invoke(
99
+ app,
100
+ [
101
+ "collect",
102
+ "--provider",
103
+ "opencode",
104
+ "--source",
105
+ f"desktop={home}",
106
+ "--database",
107
+ str(tmp_path / "opencode.sqlite"),
108
+ ],
109
+ )
110
+
111
+ assert result.exit_code == 0, result.output
112
+ assert "Ingestion opencode" in result.stdout
113
+ assert "1 written" in result.stdout
114
+
115
+
65
116
  def test_collects_all_detected_providers(tmp_path: Path, rollout_factory) -> None:
66
117
  codex_home = tmp_path / "codex"
67
118
  claude_home = tmp_path / "claude"
@@ -0,0 +1,261 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import sqlite3
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ import pytest
9
+
10
+ from cli_consumption.adapters.opencode import OpenCodeAdapter
11
+ from cli_consumption.dashboard import generate_dashboard
12
+ from cli_consumption.exporting import export_csv
13
+ from cli_consumption.storage import (
14
+ TABLES,
15
+ create_database_engine,
16
+ ingest_snapshot,
17
+ read_table,
18
+ )
19
+
20
+ CANARY = "privacy canary secret"
21
+
22
+
23
+ def database(home: Path, *, extra: bool = False, malformed: bool = False) -> Path:
24
+ path = home / "opencode.db"
25
+ home.mkdir(parents=True, exist_ok=True)
26
+ connection = sqlite3.connect(path)
27
+ connection.executescript(
28
+ """
29
+ CREATE TABLE session (
30
+ id TEXT PRIMARY KEY,
31
+ directory TEXT NOT NULL,
32
+ title TEXT NOT NULL,
33
+ metadata TEXT,
34
+ time_created INTEGER NOT NULL,
35
+ time_updated INTEGER NOT NULL
36
+ );
37
+ CREATE TABLE session_message (
38
+ id TEXT PRIMARY KEY,
39
+ session_id TEXT NOT NULL,
40
+ type TEXT NOT NULL,
41
+ seq INTEGER NOT NULL,
42
+ time_created INTEGER NOT NULL,
43
+ time_updated INTEGER NOT NULL,
44
+ data TEXT NOT NULL
45
+ );
46
+ """
47
+ )
48
+ connection.execute(
49
+ "INSERT INTO session VALUES (?, ?, ?, ?, ?, ?)",
50
+ (
51
+ "ses_test",
52
+ "/srv/work/acme/service",
53
+ CANARY,
54
+ json.dumps({"secret": CANARY}),
55
+ 1_777_114_800_000,
56
+ 1_777_114_812_000 if extra else 1_777_114_811_000,
57
+ ),
58
+ )
59
+ messages: list[tuple[str, str, int, int, dict[str, Any]]] = [
60
+ (
61
+ "msg_user_1",
62
+ "user",
63
+ 1,
64
+ 1_777_114_800_000,
65
+ {"text": CANARY, "files": [{"path": CANARY}]},
66
+ ),
67
+ (
68
+ "msg_assistant_1",
69
+ "assistant",
70
+ 2,
71
+ 1_777_114_801_000,
72
+ {
73
+ "agent": "build",
74
+ "model": {"providerID": "anthropic", "id": "claude-sonnet-4-6"},
75
+ "content": [
76
+ {"type": "text", "id": "part_text", "text": CANARY},
77
+ {
78
+ "type": "tool",
79
+ "id": "part_tool",
80
+ "name": "bash",
81
+ "state": {
82
+ "status": "completed",
83
+ "input": {"command": CANARY},
84
+ "content": [{"type": "text", "text": CANARY}],
85
+ "structured": {"output": CANARY},
86
+ },
87
+ "time": {
88
+ "created": 1_777_114_801_500,
89
+ "completed": 1_777_114_802_000,
90
+ },
91
+ },
92
+ ],
93
+ "finish": "tool-calls",
94
+ "tokens": {
95
+ "input": 100,
96
+ "output": 20,
97
+ "reasoning": 5,
98
+ "cache": {"read": 40, "write": 10},
99
+ },
100
+ "time": {
101
+ "created": 1_777_114_801_000,
102
+ "completed": 1_777_114_802_000,
103
+ },
104
+ "metadata": {"secret": CANARY},
105
+ },
106
+ ),
107
+ (
108
+ "msg_compaction",
109
+ "compaction",
110
+ 3,
111
+ 1_777_114_803_000,
112
+ {"reason": "auto", "summary": CANARY, "recent": CANARY},
113
+ ),
114
+ (
115
+ "msg_user_2",
116
+ "user",
117
+ 4,
118
+ 1_777_114_810_000,
119
+ {"text": CANARY, "files": []},
120
+ ),
121
+ (
122
+ "msg_assistant_2",
123
+ "assistant",
124
+ 5,
125
+ 1_777_114_811_000,
126
+ {
127
+ "model": {"providerID": "openai", "id": "gpt-5"},
128
+ "content": [{"type": "text", "id": "part", "text": CANARY}],
129
+ "tokens": {
130
+ "input": 2,
131
+ "output": 0,
132
+ "reasoning": 1,
133
+ "cache": {"read": 0, "write": 0},
134
+ },
135
+ "error": {"type": "unknown", "message": CANARY},
136
+ "time": {
137
+ "created": 1_777_114_811_000,
138
+ "completed": 1_777_114_811_500,
139
+ },
140
+ },
141
+ ),
142
+ ]
143
+ if extra:
144
+ messages.append(
145
+ (
146
+ "msg_system",
147
+ "system",
148
+ 6,
149
+ 1_777_114_812_000,
150
+ {"text": CANARY},
151
+ )
152
+ )
153
+ connection.executemany(
154
+ "INSERT INTO session_message "
155
+ "(id, session_id, type, seq, time_created, time_updated, data) "
156
+ "VALUES (?, 'ses_test', ?, ?, ?, ?, ?)",
157
+ [
158
+ (identifier, kind, sequence, timestamp, timestamp, json.dumps(data))
159
+ for identifier, kind, sequence, timestamp, data in messages
160
+ ],
161
+ )
162
+ if malformed:
163
+ connection.execute(
164
+ "INSERT INTO session_message VALUES (?, ?, ?, ?, ?, ?, ?)",
165
+ (
166
+ "msg_bad_json",
167
+ "ses_test",
168
+ "assistant",
169
+ 99,
170
+ 1_777_114_899_000,
171
+ 1_777_114_899_000,
172
+ "not-json",
173
+ ),
174
+ )
175
+ connection.commit()
176
+ connection.close()
177
+ return path
178
+
179
+
180
+ def test_collects_v2_usage_tools_turns_and_compactions(tmp_path: Path) -> None:
181
+ home = tmp_path / "opencode"
182
+ database(home)
183
+
184
+ snapshot = OpenCodeAdapter().collect(
185
+ [("laptop", home)], [("acme", "/srv/work/acme")]
186
+ )
187
+
188
+ conversation = snapshot.conversations[0]
189
+ assert conversation["id"] == "opencode:ses_test"
190
+ assert conversation["project"] == "acme"
191
+ assert conversation["models"] == ["anthropic/claude-sonnet-4-6", "openai/gpt-5"]
192
+ assert conversation["model_calls"] == 2
193
+ assert conversation["tool_calls"] == 1
194
+ assert conversation["input_tokens"] == 152
195
+ assert conversation["uncached_input_tokens"] == 102
196
+ assert conversation["cached_input_tokens"] == 40
197
+ assert conversation["cache_write_input_tokens"] == 10
198
+ assert conversation["output_tokens"] == 26
199
+ assert conversation["reasoning_output_tokens"] == 6
200
+ assert conversation["visible_output_tokens"] == 20
201
+ assert conversation["total_tokens"] == 178
202
+ assert [turn["status"] for turn in snapshot.turns] == ["completed", "aborted"]
203
+ assert snapshot.tool_calls[0]["tool_name"] == "bash"
204
+ assert snapshot.compaction_events[0]["turn_id"] == "opencode:ses_test:msg_user_1"
205
+ assert CANARY not in str(snapshot.to_dict())
206
+
207
+
208
+ def test_deduplicates_copies_and_handles_partial_or_malformed_rows(
209
+ tmp_path: Path,
210
+ ) -> None:
211
+ first, second = tmp_path / "first", tmp_path / "second"
212
+ database(first, malformed=True)
213
+ database(second, extra=True)
214
+
215
+ snapshot = OpenCodeAdapter().collect([("desktop", first), ("laptop", second)])
216
+
217
+ assert snapshot.duplicate_conversations == 1
218
+ assert snapshot.malformed_records == 1
219
+ assert snapshot.conversations[0]["source_machine"] == "laptop"
220
+ assert snapshot.conversations[0]["event_count"] == 6
221
+ assert (
222
+ snapshot.to_dict()
223
+ == OpenCodeAdapter().collect([("desktop", first), ("laptop", second)]).to_dict()
224
+ )
225
+
226
+ with pytest.raises(ValueError, match="Missing OpenCode database"):
227
+ OpenCodeAdapter().collect([("machine", tmp_path / "missing")])
228
+
229
+
230
+ def test_rejects_old_schema_without_exposing_records(tmp_path: Path) -> None:
231
+ home = tmp_path / "old"
232
+ home.mkdir()
233
+ connection = sqlite3.connect(home / "opencode.db")
234
+ connection.execute("CREATE TABLE session (id TEXT PRIMARY KEY)")
235
+ connection.commit()
236
+ connection.close()
237
+
238
+ with pytest.raises(
239
+ ValueError, match="Unsupported OpenCode database schema"
240
+ ) as error:
241
+ OpenCodeAdapter().collect([("machine", home)])
242
+ assert CANARY not in str(error.value)
243
+
244
+
245
+ def test_privacy_canary_is_absent_from_storage_and_exports(tmp_path: Path) -> None:
246
+ home = tmp_path / "opencode"
247
+ database(home, extra=True)
248
+ snapshot = OpenCodeAdapter().collect([("machine", home)])
249
+ engine = create_database_engine(tmp_path / "usage.sqlite")
250
+ try:
251
+ ingest_snapshot(engine, snapshot)
252
+ rows = {name: read_table(engine, name) for name in TABLES}
253
+ assert CANARY not in json.dumps(rows)
254
+ output = tmp_path / "reports"
255
+ paths = export_csv(engine, output)
256
+ dashboard = output / "dashboard.html"
257
+ generate_dashboard(engine, dashboard)
258
+ assert all(CANARY not in path.read_text() for path in paths)
259
+ assert CANARY not in dashboard.read_text()
260
+ finally:
261
+ engine.dispose()
@@ -56,7 +56,7 @@ wheels = [
56
56
 
57
57
  [[package]]
58
58
  name = "cli-consumption"
59
- version = "0.0.3"
59
+ version = "0.0.4"
60
60
  source = { editable = "." }
61
61
  dependencies = [
62
62
  { name = "fastapi" },
@@ -1,4 +0,0 @@
1
- from cli_consumption.adapters.claude import ClaudeAdapter
2
- from cli_consumption.adapters.codex import CodexAdapter
3
-
4
- __all__ = ["ClaudeAdapter", "CodexAdapter"]
File without changes
File without changes