bqlens 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bqlens/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """bqlens — find wasted BigQuery spend in one command."""
2
+
3
+ __version__ = "0.1.0"
bqlens/cli.py ADDED
@@ -0,0 +1,135 @@
1
+ """Command line interface for bqlens."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ from . import __version__
10
+ from .models import DEFAULT_PRICE_PER_TIB, ScanResult
11
+ from .report import render_html, render_json, render_terminal
12
+ from .rules import run_all
13
+
14
+
15
+ def build_parser() -> argparse.ArgumentParser:
16
+ parser = argparse.ArgumentParser(
17
+ prog="bqlens",
18
+ description="Find wasted BigQuery spend. Read-only; never touches your table data.",
19
+ )
20
+ parser.add_argument("--version", action="version", version=f"bqlens {__version__}")
21
+ sub = parser.add_subparsers(dest="command", required=True)
22
+
23
+ scan = sub.add_parser("scan", help="Scan query history for recoverable spend")
24
+ scan.add_argument("--project", help="GCP project ID to scan")
25
+ scan.add_argument(
26
+ "--region",
27
+ default="region-us",
28
+ help="BigQuery region of the JOBS view (default: region-us)",
29
+ )
30
+ scan.add_argument("--days", type=int, default=30, help="Days of history (default: 30)")
31
+ scan.add_argument(
32
+ "--price-per-tib",
33
+ type=float,
34
+ default=DEFAULT_PRICE_PER_TIB,
35
+ help=f"On-demand price per TiB scanned (default: {DEFAULT_PRICE_PER_TIB})",
36
+ )
37
+ scan.add_argument(
38
+ "--min-gib",
39
+ type=float,
40
+ default=1.0,
41
+ help="Ignore queries smaller than this many GiB (default: 1.0)",
42
+ )
43
+ scan.add_argument(
44
+ "--min-repeats",
45
+ type=int,
46
+ default=10,
47
+ help="Repeats before a query shape is flagged for materialisation (default: 10)",
48
+ )
49
+ scan.add_argument(
50
+ "--format",
51
+ choices=["terminal", "json", "html"],
52
+ default="terminal",
53
+ help="Output format (default: terminal)",
54
+ )
55
+ scan.add_argument("--out", help="Write output to this file instead of stdout")
56
+ scan.add_argument("--no-color", action="store_true", help="Disable ANSI colour")
57
+ scan.add_argument(
58
+ "--demo",
59
+ action="store_true",
60
+ help="Run against built-in synthetic history. No credentials needed.",
61
+ )
62
+ scan.add_argument(
63
+ "--fail-over",
64
+ type=float,
65
+ default=None,
66
+ metavar="USD",
67
+ help="Exit non-zero if monthly recoverable spend exceeds this. For CI.",
68
+ )
69
+ return parser
70
+
71
+
72
+ def cmd_scan(args: argparse.Namespace) -> int:
73
+ if args.demo:
74
+ from .demo import generate
75
+
76
+ jobs = generate()
77
+ project = "demo-project"
78
+ else:
79
+ if not args.project:
80
+ print("error: --project is required (or use --demo)", file=sys.stderr)
81
+ return 2
82
+ from .fetch import fetch_jobs
83
+
84
+ jobs = fetch_jobs(project=args.project, region=args.region, days=args.days)
85
+ project = args.project
86
+
87
+ if not jobs:
88
+ print("No query jobs found in that window.", file=sys.stderr)
89
+ return 0
90
+
91
+ min_bytes = int(args.min_gib * 1024**3)
92
+ findings = run_all(
93
+ jobs,
94
+ price_per_tib=args.price_per_tib,
95
+ min_bytes=min_bytes,
96
+ min_repeats=args.min_repeats,
97
+ )
98
+
99
+ result = ScanResult(
100
+ findings=findings,
101
+ total_jobs=len(jobs),
102
+ total_spend_usd=sum(j.cost(args.price_per_tib) for j in jobs),
103
+ days=args.days,
104
+ project=project,
105
+ price_per_tib=args.price_per_tib,
106
+ )
107
+
108
+ if args.format == "json":
109
+ output = render_json(result)
110
+ elif args.format == "html":
111
+ output = render_html(result)
112
+ else:
113
+ colour = sys.stdout.isatty() and not args.no_color
114
+ output = render_terminal(result, colour=colour)
115
+
116
+ if args.out:
117
+ Path(args.out).write_text(output, encoding="utf-8")
118
+ print(f"Wrote {args.out}", file=sys.stderr)
119
+ else:
120
+ print(output)
121
+
122
+ if args.fail_over is not None and result.monthly_recoverable_usd > args.fail_over:
123
+ return 1
124
+ return 0
125
+
126
+
127
+ def main(argv: list[str] | None = None) -> int:
128
+ args = build_parser().parse_args(argv)
129
+ if args.command == "scan":
130
+ return cmd_scan(args)
131
+ return 2
132
+
133
+
134
+ if __name__ == "__main__":
135
+ raise SystemExit(main())
bqlens/demo.py ADDED
@@ -0,0 +1,188 @@
1
+ """Synthetic job history so the scanner can be tried without credentials.
2
+
3
+ `bqlens scan --demo` runs the full pipeline against this. It is also what the
4
+ test suite asserts against, so the numbers below are fixtures, not decoration.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import random
10
+ from datetime import datetime, timedelta
11
+
12
+ from .models import Job
13
+
14
+ GIB = 1024**3
15
+
16
+ _USERS = [
17
+ "analytics@example.com",
18
+ "dbt-runner@example.iam.gserviceaccount.com",
19
+ "looker@example.iam.gserviceaccount.com",
20
+ "priya@example.com",
21
+ ]
22
+
23
+
24
+ def _job(
25
+ idx: int,
26
+ query: str,
27
+ gib: float,
28
+ user: str,
29
+ *,
30
+ cache_hit: bool = False,
31
+ error: str | None = None,
32
+ when: datetime | None = None,
33
+ ) -> Job:
34
+ billed = int(gib * GIB)
35
+ return Job(
36
+ job_id=f"demo_job_{idx:05d}",
37
+ user_email=user,
38
+ creation_time=when or (datetime.utcnow() - timedelta(hours=idx % 720)),
39
+ query=query,
40
+ total_bytes_billed=0 if cache_hit else billed,
41
+ total_bytes_processed=billed,
42
+ cache_hit=cache_hit,
43
+ statement_type="SELECT",
44
+ referenced_tables=["demo-project.warehouse.events"],
45
+ total_slot_ms=int(gib * 1200),
46
+ error_result=error,
47
+ )
48
+
49
+
50
+ def generate(seed: int = 7) -> list[Job]:
51
+ """A plausible 30 days of query history for a mid-sized data team."""
52
+ rng = random.Random(seed)
53
+ jobs: list[Job] = []
54
+ idx = 0
55
+
56
+ # 1. A Looker dashboard hammering the same unaggregated query all month.
57
+ dashboard_sql = """
58
+ SELECT user_id, event_name, event_timestamp, platform, country,
59
+ session_id, revenue_usd
60
+ FROM `demo-project.warehouse.events`
61
+ WHERE event_date >= '2026-08-01'
62
+ """
63
+ for _ in range(280):
64
+ idx += 1
65
+ jobs.append(_job(idx, dashboard_sql, rng.uniform(7.5, 9.0), "looker@example.iam.gserviceaccount.com"))
66
+
67
+ # 2. GA4-style wildcard scans with no _TABLE_SUFFIX filter. The classic.
68
+ wildcard_sql = """
69
+ SELECT event_name, COUNT(*) AS n
70
+ FROM `demo-project.analytics_311.events_*`
71
+ GROUP BY event_name
72
+ """
73
+ for _ in range(22):
74
+ idx += 1
75
+ jobs.append(_job(idx, wildcard_sql, rng.uniform(140, 190), "analytics@example.com"))
76
+
77
+ # 3. Analysts exploring with SELECT * LIMIT, believing LIMIT is cheap.
78
+ for _ in range(46):
79
+ idx += 1
80
+ jobs.append(
81
+ _job(
82
+ idx,
83
+ "SELECT * FROM `demo-project.warehouse.events` LIMIT 100",
84
+ rng.uniform(58, 72),
85
+ "priya@example.com",
86
+ )
87
+ )
88
+
89
+ # 4. SELECT * feeding a transform that uses four columns.
90
+ for _ in range(35):
91
+ idx += 1
92
+ jobs.append(
93
+ _job(
94
+ idx,
95
+ "SELECT * FROM `demo-project.warehouse.transactions` "
96
+ "WHERE txn_date BETWEEN '2026-08-01' AND '2026-08-31'",
97
+ rng.uniform(18, 26),
98
+ "dbt-runner@example.iam.gserviceaccount.com",
99
+ )
100
+ )
101
+
102
+ # 5. A broken scheduled query retrying all month and billing every time.
103
+ for _ in range(64):
104
+ idx += 1
105
+ jobs.append(
106
+ _job(
107
+ idx,
108
+ "SELECT customer_id, SUM(amount) FROM `demo-project.warehouse.ledger` "
109
+ "WHERE posted_date >= '2026-08-01' GROUP BY 1",
110
+ rng.uniform(11, 15),
111
+ "dbt-runner@example.iam.gserviceaccount.com",
112
+ error="Not found: Table demo-project:warehouse.ledger_v2",
113
+ )
114
+ )
115
+
116
+ # 6. An unfiltered full scan somebody schedules nightly.
117
+ for _ in range(30):
118
+ idx += 1
119
+ jobs.append(
120
+ _job(
121
+ idx,
122
+ "SELECT customer_id, lifetime_value, segment "
123
+ "FROM `demo-project.warehouse.customer_360`",
124
+ rng.uniform(34, 44),
125
+ "analytics@example.com",
126
+ )
127
+ )
128
+
129
+ # 7. A join missing its ON clause.
130
+ for _ in range(6):
131
+ idx += 1
132
+ jobs.append(
133
+ _job(
134
+ idx,
135
+ "SELECT a.user_id, b.campaign_id "
136
+ "FROM `demo-project.warehouse.users` a "
137
+ "JOIN `demo-project.warehouse.campaigns` b "
138
+ "WHERE a.country = 'IN'",
139
+ rng.uniform(25, 33),
140
+ "priya@example.com",
141
+ )
142
+ )
143
+
144
+ # 8. Plenty of ordinary, well-written, cheap queries — the healthy majority.
145
+ for _ in range(900):
146
+ idx += 1
147
+ col = rng.choice(["event_name", "platform", "country"])
148
+ jobs.append(
149
+ _job(
150
+ idx,
151
+ f"SELECT {col}, COUNT(*) FROM `demo-project.warehouse.events` "
152
+ f"WHERE event_date = '2026-09-{rng.randint(1, 17):02d}' GROUP BY 1",
153
+ rng.uniform(0.05, 0.9),
154
+ rng.choice(_USERS),
155
+ )
156
+ )
157
+
158
+ # 9. Legitimate heavy analytical work: large, well-written, distinct
159
+ # queries. This is the bulk of a healthy bill and nothing should flag
160
+ # it. Without this block the demo would claim an absurd savings rate.
161
+ _dims = ["country", "platform", "channel", "device_type", "cohort", "tier", "region"]
162
+ _metrics = ["revenue_usd", "sessions", "orders", "refunds", "margin_usd", "units"]
163
+ _tables = ["transactions", "sessions_daily", "orders", "subscriptions", "inventory_daily"]
164
+ for i in range(300):
165
+ idx += 1
166
+ dim = _dims[i % len(_dims)]
167
+ metric = _metrics[(i // 7) % len(_metrics)]
168
+ table = _tables[(i // 3) % len(_tables)]
169
+ jobs.append(
170
+ _job(
171
+ idx,
172
+ f"SELECT {dim}, DATE_TRUNC(txn_date, MONTH) AS m, SUM({metric}) AS total_{i} "
173
+ f"FROM `demo-project.warehouse.{table}` t "
174
+ f"JOIN `demo-project.warehouse.dim_customer` c ON t.customer_id = c.customer_id "
175
+ f"WHERE txn_date >= '2026-0{rng.randint(1, 8)}-01' "
176
+ f"GROUP BY 1, 2",
177
+ rng.uniform(40, 90),
178
+ rng.choice(_USERS),
179
+ )
180
+ )
181
+
182
+ # 10. Cache hits — free, and correctly ignored by every rule.
183
+ for _ in range(150):
184
+ idx += 1
185
+ jobs.append(_job(idx, dashboard_sql, 8.0, "looker@example.iam.gserviceaccount.com", cache_hit=True))
186
+
187
+ rng.shuffle(jobs)
188
+ return jobs
bqlens/fetch.py ADDED
@@ -0,0 +1,65 @@
1
+ """Read query history from INFORMATION_SCHEMA.JOBS.
2
+
3
+ Read-only. bqlens never reads a row of your actual table data — only job
4
+ metadata and the SQL text of the queries themselves.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from .models import Job
10
+
11
+ # The JOBS view is region-qualified: region-us, region-eu, region-asia-south1, ...
12
+ JOBS_SQL_REGIONAL = """
13
+ SELECT
14
+ job_id,
15
+ user_email,
16
+ creation_time,
17
+ query,
18
+ total_bytes_billed,
19
+ total_bytes_processed,
20
+ cache_hit,
21
+ statement_type,
22
+ total_slot_ms,
23
+ referenced_tables,
24
+ -- error_result is a STRUCT<reason, location, debug_info, message>, so it
25
+ -- cannot be cast to STRING directly. Serialise it, and keep NULL as NULL
26
+ -- so that `is_failed` stays false for jobs that succeeded.
27
+ IF(error_result.reason IS NULL, NULL, TO_JSON_STRING(error_result)) AS error_result
28
+ FROM `{project}`.`{region}`.INFORMATION_SCHEMA.JOBS_BY_PROJECT
29
+ WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL @days DAY)
30
+ AND job_type = 'QUERY'
31
+ AND query IS NOT NULL
32
+ ORDER BY total_bytes_billed DESC
33
+ LIMIT @max_jobs
34
+ """
35
+
36
+
37
+ def fetch_jobs(
38
+ project: str,
39
+ region: str = "region-us",
40
+ days: int = 30,
41
+ max_jobs: int = 20000,
42
+ ) -> list[Job]:
43
+ """Pull query jobs from BigQuery. Requires google-cloud-bigquery."""
44
+ try:
45
+ from google.cloud import bigquery
46
+ except ImportError as exc: # pragma: no cover - depends on optional extra
47
+ raise SystemExit(
48
+ "google-cloud-bigquery is not installed.\n"
49
+ "Install it with: pip install 'bqlens[bigquery]'\n"
50
+ "Or try the scanner first with: bqlens scan --demo"
51
+ ) from exc
52
+
53
+ client = bigquery.Client(project=project)
54
+ sql = JOBS_SQL_REGIONAL.format(project=project, region=region)
55
+ config = bigquery.QueryJobConfig(
56
+ query_parameters=[
57
+ bigquery.ScalarQueryParameter("days", "INT64", days),
58
+ bigquery.ScalarQueryParameter("max_jobs", "INT64", max_jobs),
59
+ ],
60
+ # Reading job metadata is itself billable. Cap it so the scanner can
61
+ # never be the expensive thing in someone's bill.
62
+ maximum_bytes_billed=10 * 1024**3,
63
+ )
64
+ rows = client.query(sql, job_config=config).result()
65
+ return [Job.from_row(dict(row)) for row in rows]
bqlens/models.py ADDED
@@ -0,0 +1,161 @@
1
+ """Core data structures for bqlens."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from datetime import datetime
7
+ from typing import Any
8
+
9
+ # BigQuery on-demand analysis pricing. Default is the US multi-region list
10
+ # price per TiB scanned. Override with --price-per-tib for your region or
11
+ # negotiated rate.
12
+ DEFAULT_PRICE_PER_TIB = 6.25
13
+
14
+ BYTES_PER_TIB = 1024**4
15
+
16
+
17
+ def bytes_to_usd(num_bytes: float, price_per_tib: float = DEFAULT_PRICE_PER_TIB) -> float:
18
+ """Convert bytes billed into on-demand dollars."""
19
+ if num_bytes <= 0:
20
+ return 0.0
21
+ return (num_bytes / BYTES_PER_TIB) * price_per_tib
22
+
23
+
24
+ def fmt_usd(amount: float) -> str:
25
+ """Format dollars without rendering real money as a meaningless $0.00.
26
+
27
+ A finding worth a third of a cent is still a real finding — on a bill
28
+ 100x larger it is $0.33 — so we say "<$0.01" rather than "$0.00", which
29
+ reads as a bug.
30
+ """
31
+ if amount >= 0.01:
32
+ return f"${amount:,.2f}"
33
+ if amount > 0:
34
+ return "<$0.01"
35
+ return "$0.00"
36
+
37
+
38
+ def human_bytes(num_bytes: float) -> str:
39
+ """Render a byte count in the largest sensible binary unit."""
40
+ step = 1024.0
41
+ value = float(num_bytes)
42
+ for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
43
+ if abs(value) < step or unit == "TiB":
44
+ if unit == "B":
45
+ return f"{value:.0f} B"
46
+ return f"{value:.2f} {unit}"
47
+ value /= step
48
+ return f"{value:.2f} TiB"
49
+
50
+
51
+ @dataclass
52
+ class Job:
53
+ """One BigQuery query job, normalised from INFORMATION_SCHEMA.JOBS."""
54
+
55
+ job_id: str
56
+ user_email: str
57
+ creation_time: datetime
58
+ query: str
59
+ total_bytes_billed: int
60
+ total_bytes_processed: int
61
+ cache_hit: bool = False
62
+ statement_type: str = "SELECT"
63
+ referenced_tables: list[str] = field(default_factory=list)
64
+ total_slot_ms: int = 0
65
+ destination_table: str | None = None
66
+ error_result: str | None = None
67
+
68
+ @property
69
+ def is_failed(self) -> bool:
70
+ return bool(self.error_result)
71
+
72
+ def cost(self, price_per_tib: float = DEFAULT_PRICE_PER_TIB) -> float:
73
+ return bytes_to_usd(self.total_bytes_billed, price_per_tib)
74
+
75
+ @classmethod
76
+ def from_row(cls, row: Any) -> "Job":
77
+ """Build a Job from a BigQuery result row (or any mapping-like object)."""
78
+ get = row.get if hasattr(row, "get") else lambda k, d=None: getattr(row, k, d)
79
+
80
+ refs = get("referenced_tables") or []
81
+ normalised_refs = []
82
+ for ref in refs:
83
+ if isinstance(ref, str):
84
+ normalised_refs.append(ref)
85
+ else:
86
+ # BigQuery returns STRUCT<project_id, dataset_id, table_id>
87
+ project = ref.get("project_id") if hasattr(ref, "get") else getattr(ref, "project_id", None)
88
+ dataset = ref.get("dataset_id") if hasattr(ref, "get") else getattr(ref, "dataset_id", None)
89
+ table = ref.get("table_id") if hasattr(ref, "get") else getattr(ref, "table_id", None)
90
+ normalised_refs.append(f"{project}.{dataset}.{table}")
91
+
92
+ return cls(
93
+ job_id=get("job_id") or "",
94
+ user_email=get("user_email") or "unknown",
95
+ creation_time=get("creation_time"),
96
+ query=get("query") or "",
97
+ total_bytes_billed=int(get("total_bytes_billed") or 0),
98
+ total_bytes_processed=int(get("total_bytes_processed") or 0),
99
+ cache_hit=bool(get("cache_hit")),
100
+ statement_type=get("statement_type") or "SELECT",
101
+ referenced_tables=normalised_refs,
102
+ total_slot_ms=int(get("total_slot_ms") or 0),
103
+ destination_table=get("destination_table"),
104
+ error_result=get("error_result"),
105
+ )
106
+
107
+
108
+ @dataclass
109
+ class Finding:
110
+ """One thing costing money that could stop costing money."""
111
+
112
+ rule_id: str
113
+ title: str
114
+ # Dollars per period that this finding accounts for.
115
+ wasted_usd: float
116
+ # Dollars per period we believe are *recoverable* if the fix is applied.
117
+ recoverable_usd: float
118
+ detail: str
119
+ fix: str
120
+ severity: str = "medium" # low | medium | high
121
+ job_count: int = 0
122
+ sample_job_ids: list[str] = field(default_factory=list)
123
+ owners: list[str] = field(default_factory=list)
124
+
125
+ def __post_init__(self) -> None:
126
+ if self.recoverable_usd > self.wasted_usd:
127
+ self.recoverable_usd = self.wasted_usd
128
+
129
+
130
+ @dataclass
131
+ class ScanResult:
132
+ """Everything one scan produced."""
133
+
134
+ findings: list[Finding]
135
+ total_jobs: int
136
+ total_spend_usd: float
137
+ days: int
138
+ project: str
139
+ price_per_tib: float = DEFAULT_PRICE_PER_TIB
140
+
141
+ @property
142
+ def recoverable_usd(self) -> float:
143
+ return sum(f.recoverable_usd for f in self.findings)
144
+
145
+ @property
146
+ def recoverable_pct(self) -> float:
147
+ if self.total_spend_usd <= 0:
148
+ return 0.0
149
+ return 100.0 * self.recoverable_usd / self.total_spend_usd
150
+
151
+ @property
152
+ def monthly_spend_usd(self) -> float:
153
+ if self.days <= 0:
154
+ return 0.0
155
+ return self.total_spend_usd * (30.0 / self.days)
156
+
157
+ @property
158
+ def monthly_recoverable_usd(self) -> float:
159
+ if self.days <= 0:
160
+ return 0.0
161
+ return self.recoverable_usd * (30.0 / self.days)