cli-consumption 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,525 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ import uuid
6
+ from dataclasses import dataclass
7
+ from datetime import UTC, datetime
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ from sqlalchemy import (
12
+ BigInteger,
13
+ Float,
14
+ ForeignKey,
15
+ Integer,
16
+ String,
17
+ Text,
18
+ create_engine,
19
+ delete,
20
+ event,
21
+ select,
22
+ )
23
+ from sqlalchemy.engine import Engine
24
+ from sqlalchemy.orm import DeclarativeBase, Mapped, Session, mapped_column
25
+ from sqlalchemy.pool import NullPool
26
+
27
+ from cli_consumption.models import Snapshot
28
+
29
+ DEFAULT_DATABASE = "sqlite:///cli-consumption.sqlite"
30
+ MAX_BIGINT = 9_223_372_036_854_775_807
31
+ NORMALIZED_LABEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:/+-]*")
32
+ WORK_ITEM_KINDS = {
33
+ "agent-coordination",
34
+ "command",
35
+ "compaction",
36
+ "dynamic-tool",
37
+ "extension",
38
+ "file-change",
39
+ "mcp-tool",
40
+ "media",
41
+ "message",
42
+ "other",
43
+ "reasoning",
44
+ "subagent-activity",
45
+ "user-message",
46
+ }
47
+ WORK_ITEM_STATUSES = {"completed", "failed", "in-progress", "unknown"}
48
+
49
+
50
+ class Base(DeclarativeBase):
51
+ pass
52
+
53
+
54
+ class Conversation(Base):
55
+ __tablename__ = "conversations"
56
+
57
+ id: Mapped[str] = mapped_column(String(512), primary_key=True)
58
+ provider: Mapped[str] = mapped_column(String(64), index=True)
59
+ external_id: Mapped[str] = mapped_column(String(512), index=True)
60
+ source_machine: Mapped[str] = mapped_column(String(255), index=True)
61
+ project: Mapped[str] = mapped_column(String(512), index=True)
62
+ project_source: Mapped[str] = mapped_column(String(32))
63
+ started_at: Mapped[str | None] = mapped_column(String(64), index=True)
64
+ ended_at: Mapped[str | None] = mapped_column(String(64))
65
+ duration_seconds: Mapped[float | None] = mapped_column(Float)
66
+ source: Mapped[str] = mapped_column(String(255))
67
+ models_json: Mapped[str] = mapped_column(Text)
68
+ iterations: Mapped[int] = mapped_column(Integer)
69
+ model_calls: Mapped[int] = mapped_column(Integer)
70
+ tool_calls: Mapped[int] = mapped_column(Integer)
71
+ compactions: Mapped[int] = mapped_column(Integer)
72
+ event_count: Mapped[int] = mapped_column(Integer)
73
+ content_hash: Mapped[str] = mapped_column(String(64))
74
+ input_tokens: Mapped[int] = mapped_column(BigInteger)
75
+ cached_input_tokens: Mapped[int] = mapped_column(BigInteger)
76
+ cache_write_input_tokens: Mapped[int] = mapped_column(BigInteger)
77
+ uncached_input_tokens: Mapped[int] = mapped_column(BigInteger)
78
+ output_tokens: Mapped[int] = mapped_column(BigInteger)
79
+ reasoning_output_tokens: Mapped[int] = mapped_column(BigInteger)
80
+ visible_output_tokens: Mapped[int] = mapped_column(BigInteger)
81
+ unattributed_tokens: Mapped[int] = mapped_column(BigInteger)
82
+ total_tokens: Mapped[int] = mapped_column(BigInteger)
83
+
84
+
85
+ class Turn(Base):
86
+ __tablename__ = "turns"
87
+
88
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
89
+ conversation_id: Mapped[str] = mapped_column(
90
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
91
+ )
92
+ external_id: Mapped[str] = mapped_column(String(512))
93
+ started_at: Mapped[str | None] = mapped_column(String(64), index=True)
94
+ ended_at: Mapped[str | None] = mapped_column(String(64))
95
+ status: Mapped[str] = mapped_column(String(32))
96
+ duration_ms: Mapped[int | None] = mapped_column(BigInteger)
97
+ time_to_first_token_ms: Mapped[int | None] = mapped_column(BigInteger)
98
+ model_calls: Mapped[int] = mapped_column(Integer)
99
+ tool_calls: Mapped[int] = mapped_column(Integer)
100
+ input_tokens: Mapped[int] = mapped_column(BigInteger)
101
+ cached_input_tokens: Mapped[int] = mapped_column(BigInteger)
102
+ cache_write_input_tokens: Mapped[int] = mapped_column(BigInteger)
103
+ uncached_input_tokens: Mapped[int] = mapped_column(BigInteger)
104
+ output_tokens: Mapped[int] = mapped_column(BigInteger)
105
+ reasoning_output_tokens: Mapped[int] = mapped_column(BigInteger)
106
+ visible_output_tokens: Mapped[int] = mapped_column(BigInteger)
107
+ unattributed_tokens: Mapped[int] = mapped_column(BigInteger)
108
+ total_tokens: Mapped[int] = mapped_column(BigInteger)
109
+
110
+
111
+ class ModelCall(Base):
112
+ __tablename__ = "model_calls"
113
+
114
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
115
+ conversation_id: Mapped[str] = mapped_column(
116
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
117
+ )
118
+ turn_id: Mapped[str | None] = mapped_column(String(1024), index=True)
119
+ sequence: Mapped[int] = mapped_column(Integer)
120
+ timestamp: Mapped[str | None] = mapped_column(String(64), index=True)
121
+ model: Mapped[str] = mapped_column(String(255), index=True)
122
+ input_tokens: Mapped[int] = mapped_column(BigInteger)
123
+ cached_input_tokens: Mapped[int] = mapped_column(BigInteger)
124
+ cache_write_input_tokens: Mapped[int] = mapped_column(BigInteger)
125
+ uncached_input_tokens: Mapped[int] = mapped_column(BigInteger)
126
+ output_tokens: Mapped[int] = mapped_column(BigInteger)
127
+ reasoning_output_tokens: Mapped[int] = mapped_column(BigInteger)
128
+ visible_output_tokens: Mapped[int] = mapped_column(BigInteger)
129
+ unattributed_tokens: Mapped[int] = mapped_column(BigInteger)
130
+ total_tokens: Mapped[int] = mapped_column(BigInteger)
131
+
132
+
133
+ class ToolCall(Base):
134
+ __tablename__ = "tool_calls"
135
+
136
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
137
+ conversation_id: Mapped[str] = mapped_column(
138
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
139
+ )
140
+ turn_id: Mapped[str | None] = mapped_column(String(1024), index=True)
141
+ sequence: Mapped[int] = mapped_column(Integer)
142
+ timestamp: Mapped[str | None] = mapped_column(String(64), index=True)
143
+ tool_name: Mapped[str] = mapped_column(String(512), index=True)
144
+ outer_tool_name: Mapped[str] = mapped_column(String(512))
145
+
146
+
147
+ class WorkItem(Base):
148
+ """A content-free provider activity interval within a conversation."""
149
+
150
+ __tablename__ = "work_items"
151
+
152
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
153
+ conversation_id: Mapped[str] = mapped_column(
154
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
155
+ )
156
+ turn_id: Mapped[str | None] = mapped_column(String(1024), index=True)
157
+ sequence: Mapped[int] = mapped_column(Integer)
158
+ kind: Mapped[str] = mapped_column(String(64), index=True)
159
+ tool_name: Mapped[str | None] = mapped_column(String(512), index=True)
160
+ started_at_ms: Mapped[int | None] = mapped_column(BigInteger)
161
+ completed_at_ms: Mapped[int | None] = mapped_column(BigInteger)
162
+ duration_ms: Mapped[int | None] = mapped_column(BigInteger)
163
+ status: Mapped[str] = mapped_column(String(32), index=True)
164
+
165
+
166
+ class ContextSample(Base):
167
+ """Context-window pressure reported for one provider model-usage event."""
168
+
169
+ __tablename__ = "context_samples"
170
+
171
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
172
+ conversation_id: Mapped[str] = mapped_column(
173
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
174
+ )
175
+ turn_id: Mapped[str | None] = mapped_column(String(1024), index=True)
176
+ sequence: Mapped[int] = mapped_column(Integer)
177
+ timestamp: Mapped[str | None] = mapped_column(String(64), index=True)
178
+ input_tokens: Mapped[int] = mapped_column(BigInteger)
179
+ context_window_tokens: Mapped[int] = mapped_column(BigInteger)
180
+
181
+
182
+ class TurnSetting(Base):
183
+ """Last effective provider configuration observed for a turn."""
184
+
185
+ __tablename__ = "turn_settings"
186
+
187
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
188
+ conversation_id: Mapped[str] = mapped_column(
189
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
190
+ )
191
+ turn_id: Mapped[str] = mapped_column(String(1024), index=True)
192
+ model: Mapped[str | None] = mapped_column(String(255), index=True)
193
+ effort: Mapped[str | None] = mapped_column(String(64), index=True)
194
+ collaboration_mode: Mapped[str | None] = mapped_column(String(64), index=True)
195
+ service_tier: Mapped[str | None] = mapped_column(String(64), index=True)
196
+ context_window_tokens: Mapped[int | None] = mapped_column(BigInteger)
197
+
198
+
199
+ class CompactionEvent(Base):
200
+ """A timestamped context compaction without replacement content or window IDs."""
201
+
202
+ __tablename__ = "compaction_events"
203
+
204
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
205
+ conversation_id: Mapped[str] = mapped_column(
206
+ ForeignKey("conversations.id", ondelete="CASCADE"), index=True
207
+ )
208
+ turn_id: Mapped[str | None] = mapped_column(String(1024), index=True)
209
+ sequence: Mapped[int] = mapped_column(Integer)
210
+ timestamp: Mapped[str | None] = mapped_column(String(64), index=True)
211
+
212
+
213
+ class Subagent(Base):
214
+ __tablename__ = "subagents"
215
+
216
+ id: Mapped[str] = mapped_column(String(1024), primary_key=True)
217
+ provider: Mapped[str] = mapped_column(String(64), index=True)
218
+ source_machine: Mapped[str] = mapped_column(String(255), index=True)
219
+ parent_thread_id: Mapped[str] = mapped_column(String(512), index=True)
220
+ child_thread_id: Mapped[str] = mapped_column(String(512), index=True)
221
+ status: Mapped[str] = mapped_column(String(64), index=True)
222
+ created_at_ms: Mapped[int | None] = mapped_column(BigInteger)
223
+ updated_at_ms: Mapped[int | None] = mapped_column(BigInteger)
224
+ agent_nickname: Mapped[str] = mapped_column(String(255))
225
+ agent_role: Mapped[str] = mapped_column(String(255))
226
+ tokens_used: Mapped[int | None] = mapped_column(BigInteger)
227
+
228
+
229
+ class IngestionRun(Base):
230
+ __tablename__ = "ingestion_runs"
231
+
232
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
233
+ provider: Mapped[str] = mapped_column(String(64), index=True)
234
+ ingested_at: Mapped[str] = mapped_column(String(64), index=True)
235
+ conversations_received: Mapped[int] = mapped_column(Integer)
236
+ conversations_written: Mapped[int] = mapped_column(Integer)
237
+ conversations_skipped: Mapped[int] = mapped_column(Integer)
238
+ malformed_records: Mapped[int] = mapped_column(Integer)
239
+ duplicate_conversations: Mapped[int] = mapped_column(Integer)
240
+
241
+
242
+ TABLES = {
243
+ "conversations": Conversation,
244
+ "turns": Turn,
245
+ "model_calls": ModelCall,
246
+ "tool_calls": ToolCall,
247
+ "work_items": WorkItem,
248
+ "context_samples": ContextSample,
249
+ "turn_settings": TurnSetting,
250
+ "compaction_events": CompactionEvent,
251
+ "subagents": Subagent,
252
+ "ingestion_runs": IngestionRun,
253
+ }
254
+
255
+
256
+ @dataclass(frozen=True, slots=True)
257
+ class IngestionResult:
258
+ run_id: str
259
+ received: int
260
+ written: int
261
+ skipped: int
262
+
263
+
264
+ def normalize_database_url(value: str | Path) -> str:
265
+ raw = str(value)
266
+ if "://" in raw:
267
+ return raw
268
+ path = Path(raw).expanduser().resolve()
269
+ path.parent.mkdir(parents=True, exist_ok=True)
270
+ return f"sqlite:///{path}"
271
+
272
+
273
+ def create_database_engine(database: str | Path) -> Engine:
274
+ url = normalize_database_url(database)
275
+ engine = (
276
+ create_engine(url, poolclass=NullPool)
277
+ if url.startswith("sqlite:")
278
+ else create_engine(url)
279
+ )
280
+ if url.startswith("sqlite:"):
281
+
282
+ @event.listens_for(engine, "connect")
283
+ def _enable_foreign_keys(dbapi_connection: Any, _: Any) -> None:
284
+ cursor = dbapi_connection.cursor()
285
+ cursor.execute("PRAGMA foreign_keys=ON")
286
+ cursor.close()
287
+
288
+ return engine
289
+
290
+
291
+ def initialize_database(engine: Engine) -> None:
292
+ Base.metadata.create_all(engine)
293
+
294
+
295
+ def ingest_snapshot(engine: Engine, snapshot: Snapshot) -> IngestionResult:
296
+ validate_snapshot(snapshot)
297
+ initialize_database(engine)
298
+ run_id = str(uuid.uuid4())
299
+ written = 0
300
+ skipped = 0
301
+ turns_by_conversation = _group(snapshot.turns)
302
+ calls_by_conversation = _group(snapshot.model_calls)
303
+ tools_by_conversation = _group(snapshot.tool_calls)
304
+ work_by_conversation = _group(snapshot.work_items)
305
+ context_by_conversation = _group(snapshot.context_samples)
306
+ settings_by_conversation = _group(snapshot.turn_settings)
307
+ compactions_by_conversation = _group(snapshot.compaction_events)
308
+ with Session(engine) as session, session.begin():
309
+ for subagent in snapshot.subagents:
310
+ session.merge(Subagent(**subagent))
311
+ for record in snapshot.conversations:
312
+ conversation_id = str(record["id"])
313
+ existing = session.get(Conversation, conversation_id)
314
+ if existing is not None and (
315
+ existing.event_count > int(record["event_count"])
316
+ or (
317
+ existing.event_count == int(record["event_count"])
318
+ and existing.content_hash == record["content_hash"]
319
+ )
320
+ ):
321
+ skipped += 1
322
+ continue
323
+ session.execute(
324
+ delete(ModelCall).where(ModelCall.conversation_id == conversation_id)
325
+ )
326
+ session.execute(
327
+ delete(ToolCall).where(ToolCall.conversation_id == conversation_id)
328
+ )
329
+ session.execute(
330
+ delete(WorkItem).where(WorkItem.conversation_id == conversation_id)
331
+ )
332
+ session.execute(
333
+ delete(ContextSample).where(
334
+ ContextSample.conversation_id == conversation_id
335
+ )
336
+ )
337
+ session.execute(
338
+ delete(TurnSetting).where(
339
+ TurnSetting.conversation_id == conversation_id
340
+ )
341
+ )
342
+ session.execute(
343
+ delete(CompactionEvent).where(
344
+ CompactionEvent.conversation_id == conversation_id
345
+ )
346
+ )
347
+ session.execute(delete(Turn).where(Turn.conversation_id == conversation_id))
348
+ session.merge(_conversation_from_record(record))
349
+ session.flush()
350
+ for turn in turns_by_conversation.get(conversation_id, []):
351
+ session.add(Turn(**turn))
352
+ for call in calls_by_conversation.get(conversation_id, []):
353
+ session.add(ModelCall(**call))
354
+ for tool in tools_by_conversation.get(conversation_id, []):
355
+ session.add(ToolCall(**tool))
356
+ for work_item in work_by_conversation.get(conversation_id, []):
357
+ session.add(WorkItem(**work_item))
358
+ for sample in context_by_conversation.get(conversation_id, []):
359
+ session.add(ContextSample(**sample))
360
+ for setting in settings_by_conversation.get(conversation_id, []):
361
+ session.add(TurnSetting(**setting))
362
+ for compaction in compactions_by_conversation.get(conversation_id, []):
363
+ session.add(CompactionEvent(**compaction))
364
+ written += 1
365
+ session.add(
366
+ IngestionRun(
367
+ id=run_id,
368
+ provider=snapshot.provider,
369
+ ingested_at=datetime.now(UTC).isoformat(),
370
+ conversations_received=len(snapshot.conversations),
371
+ conversations_written=written,
372
+ conversations_skipped=skipped,
373
+ malformed_records=snapshot.malformed_records,
374
+ duplicate_conversations=snapshot.duplicate_conversations,
375
+ )
376
+ )
377
+ return IngestionResult(run_id, len(snapshot.conversations), written, skipped)
378
+
379
+
380
+ def read_table(engine: Engine, table_name: str) -> list[dict[str, Any]]:
381
+ initialize_database(engine)
382
+ model = TABLES.get(table_name)
383
+ if model is None:
384
+ raise ValueError(f"Unknown table: {table_name}")
385
+ with Session(engine) as session:
386
+ rows = session.execute(select(model)).scalars().all()
387
+ return [
388
+ {
389
+ column.name: getattr(row, column.name)
390
+ for column in model.__table__.columns
391
+ }
392
+ for row in rows
393
+ ]
394
+
395
+
396
+ def validate_snapshot(snapshot: Snapshot) -> None:
397
+ """Reject missing or unexpected transport fields before opening a transaction."""
398
+ groups: tuple[tuple[str, list[dict[str, Any]], set[str]], ...] = (
399
+ (
400
+ "conversation",
401
+ snapshot.conversations,
402
+ (set(Conversation.__table__.columns.keys()) - {"models_json"}) | {"models"},
403
+ ),
404
+ ("turn", snapshot.turns, set(Turn.__table__.columns.keys())),
405
+ ("model call", snapshot.model_calls, set(ModelCall.__table__.columns.keys())),
406
+ ("tool call", snapshot.tool_calls, set(ToolCall.__table__.columns.keys())),
407
+ ("work item", snapshot.work_items, set(WorkItem.__table__.columns.keys())),
408
+ (
409
+ "context sample",
410
+ snapshot.context_samples,
411
+ set(ContextSample.__table__.columns.keys()),
412
+ ),
413
+ (
414
+ "turn setting",
415
+ snapshot.turn_settings,
416
+ set(TurnSetting.__table__.columns.keys()),
417
+ ),
418
+ (
419
+ "compaction event",
420
+ snapshot.compaction_events,
421
+ set(CompactionEvent.__table__.columns.keys()),
422
+ ),
423
+ ("subagent", snapshot.subagents, set(Subagent.__table__.columns.keys())),
424
+ )
425
+ for record_type, records, expected in groups:
426
+ for record in records:
427
+ actual = set(record)
428
+ if actual != expected:
429
+ missing = sorted(expected - actual)
430
+ unexpected = sorted(actual - expected)
431
+ raise ValueError(
432
+ f"Invalid {record_type} fields; missing={missing}, "
433
+ f"unexpected={unexpected}"
434
+ )
435
+ _validate_analytics_values(snapshot)
436
+
437
+
438
+ def _validate_analytics_values(snapshot: Snapshot) -> None:
439
+ for record in snapshot.work_items:
440
+ if (
441
+ record["kind"] not in WORK_ITEM_KINDS
442
+ or record["status"] not in WORK_ITEM_STATUSES
443
+ or not _optional_label(record["tool_name"], 512)
444
+ or not all(
445
+ _optional_nonnegative_integer(record[field])
446
+ for field in ("started_at_ms", "completed_at_ms", "duration_ms")
447
+ )
448
+ ):
449
+ raise ValueError("Invalid normalized work item values")
450
+ for record in snapshot.context_samples:
451
+ if (
452
+ not _optional_timestamp(record["timestamp"])
453
+ or not _nonnegative_integer(record["input_tokens"])
454
+ or not _positive_integer(record["context_window_tokens"])
455
+ ):
456
+ raise ValueError("Invalid normalized context sample values")
457
+ for record in snapshot.turn_settings:
458
+ if (
459
+ not _optional_label(record["model"], 255)
460
+ or not _optional_label(record["effort"], 64)
461
+ or not _optional_label(record["collaboration_mode"], 64)
462
+ or not _optional_label(record["service_tier"], 64)
463
+ or not _optional_positive_integer(record["context_window_tokens"])
464
+ ):
465
+ raise ValueError("Invalid normalized turn setting values")
466
+ for record in snapshot.compaction_events:
467
+ if not _optional_timestamp(record["timestamp"]):
468
+ raise ValueError("Invalid normalized compaction event values")
469
+
470
+
471
+ def _optional_label(value: object, maximum: int) -> bool:
472
+ return value is None or (
473
+ isinstance(value, str)
474
+ and len(value) <= maximum
475
+ and NORMALIZED_LABEL.fullmatch(value) is not None
476
+ )
477
+
478
+
479
+ def _optional_timestamp(value: object) -> bool:
480
+ if value is None:
481
+ return True
482
+ if not isinstance(value, str):
483
+ return False
484
+ try:
485
+ datetime.fromisoformat(value.replace("Z", "+00:00"))
486
+ except ValueError:
487
+ return False
488
+ return True
489
+
490
+
491
+ def _optional_nonnegative_integer(value: object) -> bool:
492
+ return value is None or _nonnegative_integer(value)
493
+
494
+
495
+ def _optional_positive_integer(value: object) -> bool:
496
+ return value is None or _positive_integer(value)
497
+
498
+
499
+ def _nonnegative_integer(value: object) -> bool:
500
+ return (
501
+ isinstance(value, int)
502
+ and not isinstance(value, bool)
503
+ and 0 <= value <= MAX_BIGINT
504
+ )
505
+
506
+
507
+ def _positive_integer(value: object) -> bool:
508
+ return (
509
+ isinstance(value, int)
510
+ and not isinstance(value, bool)
511
+ and 0 < value <= MAX_BIGINT
512
+ )
513
+
514
+
515
+ def _group(rows: list[dict[str, Any]]) -> dict[str, list[dict[str, Any]]]:
516
+ result: dict[str, list[dict[str, Any]]] = {}
517
+ for row in rows:
518
+ result.setdefault(str(row["conversation_id"]), []).append(row)
519
+ return result
520
+
521
+
522
+ def _conversation_from_record(record: dict[str, Any]) -> Conversation:
523
+ values = dict(record)
524
+ values["models_json"] = json.dumps(values.pop("models"), separators=(",", ":"))
525
+ return Conversation(**values)
@@ -0,0 +1,25 @@
1
+ from __future__ import annotations
2
+
3
+ import httpx
4
+
5
+ from cli_consumption.models import Snapshot
6
+
7
+
8
+ def send_snapshot(
9
+ snapshot: Snapshot,
10
+ endpoint: str,
11
+ token: str | None = None,
12
+ timeout: float = 60.0,
13
+ ) -> dict[str, int | str]:
14
+ headers = {"Authorization": f"Bearer {token}"} if token else {}
15
+ response = httpx.post(
16
+ endpoint.rstrip("/") + "/api/v1/snapshots",
17
+ json=snapshot.to_dict(),
18
+ headers=headers,
19
+ timeout=timeout,
20
+ )
21
+ response.raise_for_status()
22
+ payload = response.json()
23
+ if not isinstance(payload, dict):
24
+ raise ValueError("Collector returned an invalid response")
25
+ return payload