usagebassoon 0.1.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ # SPDX-FileCopyrightText: 2026 Israel Flores-Arbolay
2
+ # SPDX-License-Identifier: AGPL-3.0-only
3
+
4
+
5
+ def main() -> None:
6
+ print("Hello from UsageBassoon!")
@@ -0,0 +1,30 @@
1
+ # SPDX-FileCopyrightText: 2026 Israel Flores-Arbolay
2
+ # SPDX-License-Identifier: AGPL-3.0-only
3
+
4
+ """base.py — Backend connection contracts shared by DuckDB and MotherDuck."""
5
+
6
+ from __future__ import annotations
7
+
8
+ from contextlib import AbstractContextManager
9
+ from typing import Protocol
10
+
11
+ import duckdb
12
+
13
+
14
+ class DatabaseBackend(Protocol):
15
+ """A backend capable of opening a DuckDB-compatible connection."""
16
+
17
+ def connect(
18
+ self,
19
+ *,
20
+ read_only: bool = False,
21
+ ) -> AbstractContextManager[duckdb.DuckDBPyConnection]:
22
+ """Open a database connection managed by a context manager.
23
+
24
+ Args:
25
+ read_only: Open the underlying database without write access.
26
+
27
+ Returns:
28
+ A context manager yielding an active DuckDB connection.
29
+ """
30
+ ...
@@ -0,0 +1,51 @@
1
+ # SPDX-FileCopyrightText: 2026 Israel Flores-Arbolay
2
+ # SPDX-License-Identifier: AGPL-3.0-only
3
+
4
+ """duckdb_local.py — Local-file DuckDB backend."""
5
+
6
+ from __future__ import annotations
7
+
8
+ from collections.abc import Iterator
9
+ from contextlib import contextmanager
10
+ from pathlib import Path
11
+
12
+ import duckdb
13
+
14
+
15
+ class LocalDuckDBBackend:
16
+ """Open a local DuckDB database file.
17
+
18
+ Attributes:
19
+ database: Resolved path to the local DuckDB database file.
20
+ """
21
+
22
+ def __init__(self, database: str | Path) -> None:
23
+ """Initialize the backend with a database file path.
24
+
25
+ Args:
26
+ database: Path to the database file; parent directories are
27
+ created for writable connections.
28
+ """
29
+ self.database = Path(database).expanduser()
30
+
31
+ @contextmanager
32
+ def connect(
33
+ self,
34
+ *,
35
+ read_only: bool = False,
36
+ ) -> Iterator[duckdb.DuckDBPyConnection]:
37
+ """Open and close a local DuckDB connection.
38
+
39
+ Args:
40
+ read_only: Open the file read-only; the file must exist.
41
+
42
+ Yields:
43
+ An active DuckDB Python connection.
44
+ """
45
+ if not read_only:
46
+ self.database.parent.mkdir(parents=True, exist_ok=True)
47
+ connection = duckdb.connect(str(self.database), read_only=read_only)
48
+ try:
49
+ yield connection
50
+ finally:
51
+ connection.close()
@@ -0,0 +1,65 @@
1
+ # SPDX-FileCopyrightText: 2026 Israel Flores-Arbolay
2
+ # SPDX-License-Identifier: AGPL-3.0-only
3
+
4
+ """motherduck.py — MotherDuck backend using DuckDB's MotherDuck connection URI."""
5
+
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ from collections.abc import Iterator
10
+ from contextlib import contextmanager
11
+ from urllib.parse import quote
12
+
13
+ import duckdb
14
+
15
+
16
+ class MotherDuckBackend:
17
+ """Open a MotherDuck database through DuckDB's `md:` URI.
18
+
19
+ Attributes:
20
+ database: MotherDuck database name, without the `md:` prefix.
21
+ """
22
+
23
+ def __init__(self, database: str, *, token: str | None = None) -> None:
24
+ """Initialize the backend with a database name and optional token.
25
+
26
+ Args:
27
+ database: MotherDuck database name (no `md:` prefix).
28
+ token: Service-account or user token. If omitted, the backend
29
+ reads `MOTHERDUCK_TOKEN` when opening a connection.
30
+
31
+ Raises:
32
+ ValueError: If the database name is empty or already prefixed.
33
+ """
34
+ if not database or database.startswith("md:"):
35
+ raise ValueError("database must be a non-empty MotherDuck database name")
36
+ self.database = database
37
+ self._token = token
38
+
39
+ @contextmanager
40
+ def connect(
41
+ self,
42
+ *,
43
+ read_only: bool = False,
44
+ ) -> Iterator[duckdb.DuckDBPyConnection]:
45
+ """Open and close a MotherDuck connection.
46
+
47
+ Args:
48
+ read_only: Requested read-only mode; MotherDuck enforces
49
+ permissions server-side, so this is advisory.
50
+
51
+ Yields:
52
+ An active DuckDB Python connection.
53
+
54
+ Raises:
55
+ RuntimeError: If no MotherDuck token is available.
56
+ """
57
+ token = self._token or os.environ.get("MOTHERDUCK_TOKEN")
58
+ if not token:
59
+ raise RuntimeError("MOTHERDUCK_TOKEN is required for MotherDuck connections")
60
+ uri = f"md:{self.database}?motherduck_token={quote(token, safe='')}"
61
+ connection = duckdb.connect(uri, read_only=read_only)
62
+ try:
63
+ yield connection
64
+ finally:
65
+ connection.close()
usagebassoon/merge.py ADDED
@@ -0,0 +1,461 @@
1
+ # SPDX-FileCopyrightText: 2026 Israel Flores-Arbolay
2
+ # SPDX-License-Identifier: AGPL-3.0-only
3
+
4
+ """merge.py — Transactional persistence of normalized tokscale payloads into DuckDB.
5
+
6
+ Conforms to the final design doc: slim sessions (no LLM summary columns),
7
+ deterministic session labels, embedded point-in-time pricing, and user
8
+ tables (tags, notes) are never touched by merge.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ from collections.abc import Iterable
15
+ from dataclasses import dataclass
16
+ from datetime import datetime
17
+ from typing import Any
18
+ from uuid import UUID
19
+
20
+ import duckdb
21
+ from usagebassoon.parsers.graph import GraphPayload
22
+ from usagebassoon.parsers.models import ModelsPayload, ModelStatsRow
23
+ from usagebassoon.parsers.pricing import PricingRow
24
+ from usagebassoon.parsers.report import SessionRow, make_session_label
25
+ from usagebassoon.reconcile import ReconciliationResult
26
+
27
+
28
+ @dataclass(frozen=True, slots=True)
29
+ class CollectionBundle:
30
+ """All normalized data produced by one successful collector invocation.
31
+
32
+ Attributes:
33
+ run_id: Identifier for this collection run.
34
+ started_at: When the collector began invoking tokscale.
35
+ finished_at: When parsing finished; used as the merge timestamp.
36
+ host: Hostname or container id, if known.
37
+ models: Validated models payload (metrics authority).
38
+ report_rows: Validated report rows (metadata authority).
39
+ graph: Validated graph payload (daily dimension + telemetry).
40
+ pricing_by_model: Resolved rate cards keyed by model id.
41
+ raw_exports: Verbatim decoded payloads keyed by kind.
42
+ reconciliation: Cross-payload consistency results.
43
+ """
44
+
45
+ run_id: UUID
46
+ started_at: datetime
47
+ finished_at: datetime
48
+ host: str | None
49
+ models: ModelsPayload
50
+ report_rows: list[SessionRow]
51
+ graph: GraphPayload
52
+ pricing_by_model: dict[str, PricingRow]
53
+ raw_exports: dict[str, Any]
54
+ reconciliation: ReconciliationResult
55
+
56
+
57
+ class MergeError(RuntimeError):
58
+ """Raised when a collection bundle cannot be persisted atomically."""
59
+
60
+
61
+ def _json(value: Any) -> str:
62
+ """Serialize raw payloads compactly for a DuckDB JSON column.
63
+
64
+ Args:
65
+ value: Any JSON-serializable decoded payload.
66
+
67
+ Returns:
68
+ A compact UTF-8-safe JSON string.
69
+ """
70
+ return json.dumps(value, separators=(",", ":"), ensure_ascii=False)
71
+
72
+
73
+ def _insert_raw_exports(
74
+ connection: duckdb.DuckDBPyConnection,
75
+ run_id: UUID,
76
+ captured_at: datetime,
77
+ raw_exports: dict[str, Any],
78
+ ) -> None:
79
+ """Insert the immutable raw JSON record for every command payload.
80
+
81
+ Args:
82
+ connection: An active DuckDB-compatible connection.
83
+ run_id: Identifier for the owning collection run.
84
+ captured_at: Timestamp recorded for each raw payload row.
85
+ raw_exports: Decoded payloads keyed by payload kind.
86
+ """
87
+ connection.executemany(
88
+ "INSERT INTO raw_exports (run_id, kind, ingested_at, payload) "
89
+ "VALUES (?, ?, ?, CAST(? AS JSON))",
90
+ [(run_id, kind, captured_at, _json(p)) for kind, p in raw_exports.items()],
91
+ )
92
+
93
+
94
+ def _merge_sessions(
95
+ connection: duckdb.DuckDBPyConnection,
96
+ rows: Iterable[SessionRow],
97
+ captured_at: datetime,
98
+ ) -> None:
99
+ """Upsert the current metadata snapshot for every observed session.
100
+
101
+ Args:
102
+ connection: An active DuckDB-compatible connection.
103
+ rows: Validated session metadata rows from the report payload.
104
+ captured_at: Timestamp stamped onto first/last seen columns.
105
+ """
106
+ values = [
107
+ (
108
+ row.client,
109
+ row.session_id,
110
+ row.workspace,
111
+ row.workspace_label,
112
+ row.created_at,
113
+ row.last_active,
114
+ row.duration_minutes,
115
+ row.message_count,
116
+ row.cost_usd,
117
+ list(row.models_used),
118
+ make_session_label(row),
119
+ captured_at,
120
+ captured_at,
121
+ )
122
+ for row in rows
123
+ ]
124
+ if not values:
125
+ return
126
+ connection.executemany(
127
+ """
128
+ INSERT INTO sessions AS target (
129
+ client, session_id, workspace, workspace_label, created_at, last_active,
130
+ duration_minutes, message_count, cost_usd, models_used, session_label,
131
+ first_seen_at, last_seen_at
132
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
133
+ ON CONFLICT (client, session_id) DO UPDATE SET
134
+ workspace = excluded.workspace,
135
+ workspace_label = excluded.workspace_label,
136
+ created_at = excluded.created_at,
137
+ last_active = excluded.last_active,
138
+ duration_minutes = excluded.duration_minutes,
139
+ message_count = excluded.message_count,
140
+ cost_usd = excluded.cost_usd,
141
+ models_used = excluded.models_used,
142
+ session_label = excluded.session_label,
143
+ last_seen_at = excluded.last_seen_at
144
+ """,
145
+ values,
146
+ )
147
+
148
+
149
+ def _pricing_values(
150
+ model: str,
151
+ pricing_by_model: dict[str, PricingRow],
152
+ ) -> tuple[Any, ...]:
153
+ """Return nullable row-level pricing fields aligned with the insert.
154
+
155
+ Args:
156
+ model: The model id to resolve pricing for.
157
+ pricing_by_model: Rate cards captured during this run.
158
+
159
+ Returns:
160
+ An eight-tuple of pricing column values, all None if uncaptured.
161
+ """
162
+ price = pricing_by_model.get(model)
163
+ if price is None:
164
+ return (None,) * 8
165
+ return (
166
+ price.pricing.input_cost_per_token,
167
+ price.pricing.output_cost_per_token,
168
+ price.pricing.cache_read_input_token_cost,
169
+ price.pricing.cache_write_input_token_cost,
170
+ price.matched_key,
171
+ price.resolution.kind,
172
+ price.resolution.alias_applied,
173
+ price.source,
174
+ )
175
+
176
+
177
+ def _merge_session_model_stats(
178
+ connection: duckdb.DuckDBPyConnection,
179
+ rows: Iterable[ModelStatsRow],
180
+ pricing_by_model: dict[str, PricingRow],
181
+ captured_at: datetime,
182
+ ) -> None:
183
+ """Upsert cumulative session-model metrics with contemporaneous pricing.
184
+
185
+ Args:
186
+ connection: An active DuckDB-compatible connection.
187
+ rows: Validated per-(client, session, model) cumulative entries.
188
+ pricing_by_model: Rate cards captured during this run.
189
+ captured_at: Timestamp stamped onto seen/pricing columns.
190
+ """
191
+ values = [
192
+ (
193
+ row.client,
194
+ row.session_id,
195
+ row.model,
196
+ row.provider,
197
+ row.input_tokens,
198
+ row.output_tokens,
199
+ row.cache_read,
200
+ row.cache_write,
201
+ row.reasoning,
202
+ row.message_count,
203
+ row.cost_usd,
204
+ row.ms_per_1k_tokens,
205
+ row.perf_duration_ms,
206
+ row.perf_token_coverage,
207
+ *_pricing_values(row.model, pricing_by_model),
208
+ captured_at,
209
+ captured_at,
210
+ captured_at,
211
+ )
212
+ for row in rows
213
+ ]
214
+ if not values:
215
+ return
216
+ connection.executemany(
217
+ """
218
+ INSERT INTO session_model_stats AS target (
219
+ client, session_id, model, provider,
220
+ input_tokens, output_tokens, cache_read, cache_write, reasoning,
221
+ message_count, cost_usd,
222
+ ms_per_1k_tokens, perf_duration_ms, perf_token_coverage,
223
+ price_input_per_token, price_output_per_token,
224
+ price_cache_read_per_token, price_cache_write_per_token,
225
+ price_matched_key, price_match_kind, price_alias_applied, price_source,
226
+ price_captured_at, first_seen_at, last_seen_at
227
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
228
+ ON CONFLICT (client, session_id, model) DO UPDATE SET
229
+ provider = excluded.provider,
230
+ input_tokens = excluded.input_tokens,
231
+ output_tokens = excluded.output_tokens,
232
+ cache_read = excluded.cache_read,
233
+ cache_write = excluded.cache_write,
234
+ reasoning = excluded.reasoning,
235
+ message_count = excluded.message_count,
236
+ cost_usd = excluded.cost_usd,
237
+ ms_per_1k_tokens = excluded.ms_per_1k_tokens,
238
+ perf_duration_ms = excluded.perf_duration_ms,
239
+ perf_token_coverage = excluded.perf_token_coverage,
240
+ price_input_per_token = excluded.price_input_per_token,
241
+ price_output_per_token = excluded.price_output_per_token,
242
+ price_cache_read_per_token = excluded.price_cache_read_per_token,
243
+ price_cache_write_per_token = excluded.price_cache_write_per_token,
244
+ price_matched_key = excluded.price_matched_key,
245
+ price_match_kind = excluded.price_match_kind,
246
+ price_alias_applied = excluded.price_alias_applied,
247
+ price_source = excluded.price_source,
248
+ price_captured_at = excluded.price_captured_at,
249
+ last_seen_at = excluded.last_seen_at
250
+ """,
251
+ values,
252
+ )
253
+
254
+
255
+ def _merge_daily(
256
+ connection: duckdb.DuckDBPyConnection,
257
+ graph: GraphPayload,
258
+ ) -> None:
259
+ """Upsert additive graph facts by their natural keys.
260
+
261
+ Args:
262
+ connection: An active DuckDB-compatible connection.
263
+ graph: Validated graph payload with contributions and telemetry.
264
+ """
265
+ stats = [
266
+ (
267
+ c.date,
268
+ client.client,
269
+ client.model_id,
270
+ client.provider_id,
271
+ client.tokens.input,
272
+ client.tokens.output,
273
+ client.tokens.cache_read,
274
+ client.tokens.cache_write,
275
+ client.tokens.reasoning,
276
+ client.messages,
277
+ client.cost,
278
+ )
279
+ for c in graph.contributions
280
+ for client in c.clients
281
+ ]
282
+ if stats:
283
+ connection.executemany(
284
+ """
285
+ INSERT INTO daily_stats AS target (
286
+ day, client, model, provider, input_tokens, output_tokens,
287
+ cache_read, cache_write, reasoning, message_count, cost_usd
288
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
289
+ ON CONFLICT (day, client, model) DO UPDATE SET
290
+ provider = excluded.provider,
291
+ input_tokens = excluded.input_tokens,
292
+ output_tokens = excluded.output_tokens,
293
+ cache_read = excluded.cache_read,
294
+ cache_write = excluded.cache_write,
295
+ reasoning = excluded.reasoning,
296
+ message_count = excluded.message_count,
297
+ cost_usd = excluded.cost_usd
298
+ """,
299
+ stats,
300
+ )
301
+ activity = [(c.date, c.intensity, c.active_time_ms) for c in graph.contributions]
302
+ if activity:
303
+ connection.executemany(
304
+ """
305
+ INSERT INTO daily_activity AS target (day, intensity, active_time_ms)
306
+ VALUES (?, ?, ?)
307
+ ON CONFLICT (day) DO UPDATE SET
308
+ intensity = excluded.intensity,
309
+ active_time_ms = excluded.active_time_ms
310
+ """,
311
+ activity,
312
+ )
313
+
314
+
315
+ def _insert_pricing_snapshots(
316
+ connection: duckdb.DuckDBPyConnection,
317
+ pricing_by_model: dict[str, PricingRow],
318
+ captured_at: datetime,
319
+ ) -> None:
320
+ """Append rate cards observed during this collection run.
321
+
322
+ Args:
323
+ connection: An active DuckDB-compatible connection.
324
+ pricing_by_model: Rate cards captured during this run.
325
+ captured_at: Timestamp recorded for each snapshot row.
326
+ """
327
+ values = [
328
+ (
329
+ captured_at,
330
+ p.model_id,
331
+ p.source,
332
+ p.matched_key,
333
+ p.resolution.kind,
334
+ p.pricing.input_cost_per_token,
335
+ p.pricing.output_cost_per_token,
336
+ p.pricing.cache_read_input_token_cost,
337
+ p.pricing.cache_write_input_token_cost,
338
+ )
339
+ for p in pricing_by_model.values()
340
+ ]
341
+ if values:
342
+ connection.executemany(
343
+ """
344
+ INSERT INTO pricing_snapshots (
345
+ captured_at, model, source, matched_key, match_kind,
346
+ price_input_per_token, price_output_per_token,
347
+ price_cache_read_per_token, price_cache_write_per_token
348
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
349
+ ON CONFLICT DO NOTHING
350
+ """,
351
+ values,
352
+ )
353
+
354
+
355
+ def _insert_run_metrics(
356
+ connection: duckdb.DuckDBPyConnection,
357
+ run_id: UUID,
358
+ graph: GraphPayload,
359
+ ) -> None:
360
+ """Persist graph-level telemetry for the collection run.
361
+
362
+ Args:
363
+ connection: An active DuckDB-compatible connection.
364
+ run_id: Identifier for the owning collection run.
365
+ graph: Validated graph payload; provides summary + timeMetrics.
366
+ """
367
+ tm, summary = graph.time_metrics, graph.summary
368
+ connection.execute(
369
+ """
370
+ INSERT INTO run_metrics (
371
+ run_id, captured_at, total_tokens, total_cost, active_days,
372
+ total_active_time_ms, longest_continuous_ms,
373
+ max_concurrent_sessions, graph_session_count
374
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
375
+ """,
376
+ [
377
+ run_id,
378
+ graph.meta.generated_at,
379
+ summary.total_tokens,
380
+ summary.total_cost,
381
+ summary.active_days,
382
+ tm.total_active_time_ms,
383
+ tm.longest_continuous_ms,
384
+ tm.max_concurrent_sessions,
385
+ tm.session_count,
386
+ ],
387
+ )
388
+
389
+
390
+ def _record_issues(
391
+ connection: duckdb.DuckDBPyConnection,
392
+ run_id: UUID,
393
+ reconciliation: ReconciliationResult,
394
+ ) -> None:
395
+ """Persist non-fatal reconciliation warnings for the run.
396
+
397
+ Args:
398
+ connection: An active DuckDB-compatible connection.
399
+ run_id: Identifier for the owning collection run.
400
+ reconciliation: Results of the cross-payload consistency checks.
401
+ """
402
+ if not reconciliation.issues:
403
+ return
404
+ connection.executemany(
405
+ "INSERT INTO reconciliation_issues (run_id, check_name, issue_key, message) "
406
+ "VALUES (?, ?, ?, ?)",
407
+ [(run_id, i.check, i.key, i.message) for i in reconciliation.issues],
408
+ )
409
+
410
+
411
+ def merge_collection(
412
+ connection: duckdb.DuckDBPyConnection,
413
+ bundle: CollectionBundle,
414
+ ) -> None:
415
+ """Atomically persist one fully parsed collection bundle.
416
+
417
+ Owns a single SQL transaction; leaves no partial rows on failure. The
418
+ caller must have initialized the DDL.
419
+
420
+ Args:
421
+ connection: A local DuckDB or MotherDuck connection.
422
+ bundle: Parsed payloads, raw JSON, rates, and reconciliation output.
423
+
424
+ Raises:
425
+ MergeError: If the bundle could not be written atomically.
426
+ """
427
+ connection.execute("BEGIN TRANSACTION")
428
+ try:
429
+ connection.execute(
430
+ "INSERT INTO ingest_runs (run_id, started_at, finished_at, host, "
431
+ "tokscale_ver, status, rows_in, rows_merged) "
432
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
433
+ [
434
+ bundle.run_id,
435
+ bundle.started_at,
436
+ bundle.finished_at,
437
+ bundle.host,
438
+ bundle.graph.meta.version,
439
+ "ok" if bundle.reconciliation.ok else "partial",
440
+ len(bundle.models.entries)
441
+ + len(bundle.report_rows)
442
+ + sum(len(c.clients) for c in bundle.graph.contributions),
443
+ len(bundle.models.entries),
444
+ ],
445
+ )
446
+ _insert_raw_exports(connection, bundle.run_id, bundle.finished_at, bundle.raw_exports)
447
+ _merge_sessions(connection, bundle.report_rows, bundle.finished_at)
448
+ _merge_session_model_stats(
449
+ connection, bundle.models.entries, bundle.pricing_by_model, bundle.finished_at
450
+ )
451
+ _merge_daily(connection, bundle.graph)
452
+ _insert_pricing_snapshots(connection, bundle.pricing_by_model, bundle.finished_at)
453
+ _insert_run_metrics(connection, bundle.run_id, bundle.graph)
454
+ _record_issues(connection, bundle.run_id, bundle.reconciliation)
455
+ connection.execute("COMMIT")
456
+ except Exception as error:
457
+ try:
458
+ connection.execute("ROLLBACK")
459
+ except duckdb.Error:
460
+ pass
461
+ raise MergeError(f"failed to merge collection {bundle.run_id}") from error