bqlens 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bqlens/__init__.py +3 -0
- bqlens/cli.py +135 -0
- bqlens/demo.py +188 -0
- bqlens/fetch.py +65 -0
- bqlens/models.py +161 -0
- bqlens/report.py +239 -0
- bqlens/rules.py +526 -0
- bqlens-0.1.0.dist-info/METADATA +163 -0
- bqlens-0.1.0.dist-info/RECORD +13 -0
- bqlens-0.1.0.dist-info/WHEEL +5 -0
- bqlens-0.1.0.dist-info/entry_points.txt +2 -0
- bqlens-0.1.0.dist-info/licenses/LICENSE +15 -0
- bqlens-0.1.0.dist-info/top_level.txt +1 -0
bqlens/__init__.py
ADDED
bqlens/cli.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Command line interface for bqlens."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .models import DEFAULT_PRICE_PER_TIB, ScanResult
|
|
11
|
+
from .report import render_html, render_json, render_terminal
|
|
12
|
+
from .rules import run_all
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
prog="bqlens",
|
|
18
|
+
description="Find wasted BigQuery spend. Read-only; never touches your table data.",
|
|
19
|
+
)
|
|
20
|
+
parser.add_argument("--version", action="version", version=f"bqlens {__version__}")
|
|
21
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
22
|
+
|
|
23
|
+
scan = sub.add_parser("scan", help="Scan query history for recoverable spend")
|
|
24
|
+
scan.add_argument("--project", help="GCP project ID to scan")
|
|
25
|
+
scan.add_argument(
|
|
26
|
+
"--region",
|
|
27
|
+
default="region-us",
|
|
28
|
+
help="BigQuery region of the JOBS view (default: region-us)",
|
|
29
|
+
)
|
|
30
|
+
scan.add_argument("--days", type=int, default=30, help="Days of history (default: 30)")
|
|
31
|
+
scan.add_argument(
|
|
32
|
+
"--price-per-tib",
|
|
33
|
+
type=float,
|
|
34
|
+
default=DEFAULT_PRICE_PER_TIB,
|
|
35
|
+
help=f"On-demand price per TiB scanned (default: {DEFAULT_PRICE_PER_TIB})",
|
|
36
|
+
)
|
|
37
|
+
scan.add_argument(
|
|
38
|
+
"--min-gib",
|
|
39
|
+
type=float,
|
|
40
|
+
default=1.0,
|
|
41
|
+
help="Ignore queries smaller than this many GiB (default: 1.0)",
|
|
42
|
+
)
|
|
43
|
+
scan.add_argument(
|
|
44
|
+
"--min-repeats",
|
|
45
|
+
type=int,
|
|
46
|
+
default=10,
|
|
47
|
+
help="Repeats before a query shape is flagged for materialisation (default: 10)",
|
|
48
|
+
)
|
|
49
|
+
scan.add_argument(
|
|
50
|
+
"--format",
|
|
51
|
+
choices=["terminal", "json", "html"],
|
|
52
|
+
default="terminal",
|
|
53
|
+
help="Output format (default: terminal)",
|
|
54
|
+
)
|
|
55
|
+
scan.add_argument("--out", help="Write output to this file instead of stdout")
|
|
56
|
+
scan.add_argument("--no-color", action="store_true", help="Disable ANSI colour")
|
|
57
|
+
scan.add_argument(
|
|
58
|
+
"--demo",
|
|
59
|
+
action="store_true",
|
|
60
|
+
help="Run against built-in synthetic history. No credentials needed.",
|
|
61
|
+
)
|
|
62
|
+
scan.add_argument(
|
|
63
|
+
"--fail-over",
|
|
64
|
+
type=float,
|
|
65
|
+
default=None,
|
|
66
|
+
metavar="USD",
|
|
67
|
+
help="Exit non-zero if monthly recoverable spend exceeds this. For CI.",
|
|
68
|
+
)
|
|
69
|
+
return parser
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def cmd_scan(args: argparse.Namespace) -> int:
|
|
73
|
+
if args.demo:
|
|
74
|
+
from .demo import generate
|
|
75
|
+
|
|
76
|
+
jobs = generate()
|
|
77
|
+
project = "demo-project"
|
|
78
|
+
else:
|
|
79
|
+
if not args.project:
|
|
80
|
+
print("error: --project is required (or use --demo)", file=sys.stderr)
|
|
81
|
+
return 2
|
|
82
|
+
from .fetch import fetch_jobs
|
|
83
|
+
|
|
84
|
+
jobs = fetch_jobs(project=args.project, region=args.region, days=args.days)
|
|
85
|
+
project = args.project
|
|
86
|
+
|
|
87
|
+
if not jobs:
|
|
88
|
+
print("No query jobs found in that window.", file=sys.stderr)
|
|
89
|
+
return 0
|
|
90
|
+
|
|
91
|
+
min_bytes = int(args.min_gib * 1024**3)
|
|
92
|
+
findings = run_all(
|
|
93
|
+
jobs,
|
|
94
|
+
price_per_tib=args.price_per_tib,
|
|
95
|
+
min_bytes=min_bytes,
|
|
96
|
+
min_repeats=args.min_repeats,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
result = ScanResult(
|
|
100
|
+
findings=findings,
|
|
101
|
+
total_jobs=len(jobs),
|
|
102
|
+
total_spend_usd=sum(j.cost(args.price_per_tib) for j in jobs),
|
|
103
|
+
days=args.days,
|
|
104
|
+
project=project,
|
|
105
|
+
price_per_tib=args.price_per_tib,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
if args.format == "json":
|
|
109
|
+
output = render_json(result)
|
|
110
|
+
elif args.format == "html":
|
|
111
|
+
output = render_html(result)
|
|
112
|
+
else:
|
|
113
|
+
colour = sys.stdout.isatty() and not args.no_color
|
|
114
|
+
output = render_terminal(result, colour=colour)
|
|
115
|
+
|
|
116
|
+
if args.out:
|
|
117
|
+
Path(args.out).write_text(output, encoding="utf-8")
|
|
118
|
+
print(f"Wrote {args.out}", file=sys.stderr)
|
|
119
|
+
else:
|
|
120
|
+
print(output)
|
|
121
|
+
|
|
122
|
+
if args.fail_over is not None and result.monthly_recoverable_usd > args.fail_over:
|
|
123
|
+
return 1
|
|
124
|
+
return 0
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def main(argv: list[str] | None = None) -> int:
|
|
128
|
+
args = build_parser().parse_args(argv)
|
|
129
|
+
if args.command == "scan":
|
|
130
|
+
return cmd_scan(args)
|
|
131
|
+
return 2
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
if __name__ == "__main__":
|
|
135
|
+
raise SystemExit(main())
|
bqlens/demo.py
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""Synthetic job history so the scanner can be tried without credentials.
|
|
2
|
+
|
|
3
|
+
`bqlens scan --demo` runs the full pipeline against this. It is also what the
|
|
4
|
+
test suite asserts against, so the numbers below are fixtures, not decoration.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import random
|
|
10
|
+
from datetime import datetime, timedelta
|
|
11
|
+
|
|
12
|
+
from .models import Job
|
|
13
|
+
|
|
14
|
+
GIB = 1024**3
|
|
15
|
+
|
|
16
|
+
_USERS = [
|
|
17
|
+
"analytics@example.com",
|
|
18
|
+
"dbt-runner@example.iam.gserviceaccount.com",
|
|
19
|
+
"looker@example.iam.gserviceaccount.com",
|
|
20
|
+
"priya@example.com",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _job(
|
|
25
|
+
idx: int,
|
|
26
|
+
query: str,
|
|
27
|
+
gib: float,
|
|
28
|
+
user: str,
|
|
29
|
+
*,
|
|
30
|
+
cache_hit: bool = False,
|
|
31
|
+
error: str | None = None,
|
|
32
|
+
when: datetime | None = None,
|
|
33
|
+
) -> Job:
|
|
34
|
+
billed = int(gib * GIB)
|
|
35
|
+
return Job(
|
|
36
|
+
job_id=f"demo_job_{idx:05d}",
|
|
37
|
+
user_email=user,
|
|
38
|
+
creation_time=when or (datetime.utcnow() - timedelta(hours=idx % 720)),
|
|
39
|
+
query=query,
|
|
40
|
+
total_bytes_billed=0 if cache_hit else billed,
|
|
41
|
+
total_bytes_processed=billed,
|
|
42
|
+
cache_hit=cache_hit,
|
|
43
|
+
statement_type="SELECT",
|
|
44
|
+
referenced_tables=["demo-project.warehouse.events"],
|
|
45
|
+
total_slot_ms=int(gib * 1200),
|
|
46
|
+
error_result=error,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def generate(seed: int = 7) -> list[Job]:
|
|
51
|
+
"""A plausible 30 days of query history for a mid-sized data team."""
|
|
52
|
+
rng = random.Random(seed)
|
|
53
|
+
jobs: list[Job] = []
|
|
54
|
+
idx = 0
|
|
55
|
+
|
|
56
|
+
# 1. A Looker dashboard hammering the same unaggregated query all month.
|
|
57
|
+
dashboard_sql = """
|
|
58
|
+
SELECT user_id, event_name, event_timestamp, platform, country,
|
|
59
|
+
session_id, revenue_usd
|
|
60
|
+
FROM `demo-project.warehouse.events`
|
|
61
|
+
WHERE event_date >= '2026-08-01'
|
|
62
|
+
"""
|
|
63
|
+
for _ in range(280):
|
|
64
|
+
idx += 1
|
|
65
|
+
jobs.append(_job(idx, dashboard_sql, rng.uniform(7.5, 9.0), "looker@example.iam.gserviceaccount.com"))
|
|
66
|
+
|
|
67
|
+
# 2. GA4-style wildcard scans with no _TABLE_SUFFIX filter. The classic.
|
|
68
|
+
wildcard_sql = """
|
|
69
|
+
SELECT event_name, COUNT(*) AS n
|
|
70
|
+
FROM `demo-project.analytics_311.events_*`
|
|
71
|
+
GROUP BY event_name
|
|
72
|
+
"""
|
|
73
|
+
for _ in range(22):
|
|
74
|
+
idx += 1
|
|
75
|
+
jobs.append(_job(idx, wildcard_sql, rng.uniform(140, 190), "analytics@example.com"))
|
|
76
|
+
|
|
77
|
+
# 3. Analysts exploring with SELECT * LIMIT, believing LIMIT is cheap.
|
|
78
|
+
for _ in range(46):
|
|
79
|
+
idx += 1
|
|
80
|
+
jobs.append(
|
|
81
|
+
_job(
|
|
82
|
+
idx,
|
|
83
|
+
"SELECT * FROM `demo-project.warehouse.events` LIMIT 100",
|
|
84
|
+
rng.uniform(58, 72),
|
|
85
|
+
"priya@example.com",
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# 4. SELECT * feeding a transform that uses four columns.
|
|
90
|
+
for _ in range(35):
|
|
91
|
+
idx += 1
|
|
92
|
+
jobs.append(
|
|
93
|
+
_job(
|
|
94
|
+
idx,
|
|
95
|
+
"SELECT * FROM `demo-project.warehouse.transactions` "
|
|
96
|
+
"WHERE txn_date BETWEEN '2026-08-01' AND '2026-08-31'",
|
|
97
|
+
rng.uniform(18, 26),
|
|
98
|
+
"dbt-runner@example.iam.gserviceaccount.com",
|
|
99
|
+
)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# 5. A broken scheduled query retrying all month and billing every time.
|
|
103
|
+
for _ in range(64):
|
|
104
|
+
idx += 1
|
|
105
|
+
jobs.append(
|
|
106
|
+
_job(
|
|
107
|
+
idx,
|
|
108
|
+
"SELECT customer_id, SUM(amount) FROM `demo-project.warehouse.ledger` "
|
|
109
|
+
"WHERE posted_date >= '2026-08-01' GROUP BY 1",
|
|
110
|
+
rng.uniform(11, 15),
|
|
111
|
+
"dbt-runner@example.iam.gserviceaccount.com",
|
|
112
|
+
error="Not found: Table demo-project:warehouse.ledger_v2",
|
|
113
|
+
)
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
# 6. An unfiltered full scan somebody schedules nightly.
|
|
117
|
+
for _ in range(30):
|
|
118
|
+
idx += 1
|
|
119
|
+
jobs.append(
|
|
120
|
+
_job(
|
|
121
|
+
idx,
|
|
122
|
+
"SELECT customer_id, lifetime_value, segment "
|
|
123
|
+
"FROM `demo-project.warehouse.customer_360`",
|
|
124
|
+
rng.uniform(34, 44),
|
|
125
|
+
"analytics@example.com",
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# 7. A join missing its ON clause.
|
|
130
|
+
for _ in range(6):
|
|
131
|
+
idx += 1
|
|
132
|
+
jobs.append(
|
|
133
|
+
_job(
|
|
134
|
+
idx,
|
|
135
|
+
"SELECT a.user_id, b.campaign_id "
|
|
136
|
+
"FROM `demo-project.warehouse.users` a "
|
|
137
|
+
"JOIN `demo-project.warehouse.campaigns` b "
|
|
138
|
+
"WHERE a.country = 'IN'",
|
|
139
|
+
rng.uniform(25, 33),
|
|
140
|
+
"priya@example.com",
|
|
141
|
+
)
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
# 8. Plenty of ordinary, well-written, cheap queries — the healthy majority.
|
|
145
|
+
for _ in range(900):
|
|
146
|
+
idx += 1
|
|
147
|
+
col = rng.choice(["event_name", "platform", "country"])
|
|
148
|
+
jobs.append(
|
|
149
|
+
_job(
|
|
150
|
+
idx,
|
|
151
|
+
f"SELECT {col}, COUNT(*) FROM `demo-project.warehouse.events` "
|
|
152
|
+
f"WHERE event_date = '2026-09-{rng.randint(1, 17):02d}' GROUP BY 1",
|
|
153
|
+
rng.uniform(0.05, 0.9),
|
|
154
|
+
rng.choice(_USERS),
|
|
155
|
+
)
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
# 9. Legitimate heavy analytical work: large, well-written, distinct
|
|
159
|
+
# queries. This is the bulk of a healthy bill and nothing should flag
|
|
160
|
+
# it. Without this block the demo would claim an absurd savings rate.
|
|
161
|
+
_dims = ["country", "platform", "channel", "device_type", "cohort", "tier", "region"]
|
|
162
|
+
_metrics = ["revenue_usd", "sessions", "orders", "refunds", "margin_usd", "units"]
|
|
163
|
+
_tables = ["transactions", "sessions_daily", "orders", "subscriptions", "inventory_daily"]
|
|
164
|
+
for i in range(300):
|
|
165
|
+
idx += 1
|
|
166
|
+
dim = _dims[i % len(_dims)]
|
|
167
|
+
metric = _metrics[(i // 7) % len(_metrics)]
|
|
168
|
+
table = _tables[(i // 3) % len(_tables)]
|
|
169
|
+
jobs.append(
|
|
170
|
+
_job(
|
|
171
|
+
idx,
|
|
172
|
+
f"SELECT {dim}, DATE_TRUNC(txn_date, MONTH) AS m, SUM({metric}) AS total_{i} "
|
|
173
|
+
f"FROM `demo-project.warehouse.{table}` t "
|
|
174
|
+
f"JOIN `demo-project.warehouse.dim_customer` c ON t.customer_id = c.customer_id "
|
|
175
|
+
f"WHERE txn_date >= '2026-0{rng.randint(1, 8)}-01' "
|
|
176
|
+
f"GROUP BY 1, 2",
|
|
177
|
+
rng.uniform(40, 90),
|
|
178
|
+
rng.choice(_USERS),
|
|
179
|
+
)
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
# 10. Cache hits — free, and correctly ignored by every rule.
|
|
183
|
+
for _ in range(150):
|
|
184
|
+
idx += 1
|
|
185
|
+
jobs.append(_job(idx, dashboard_sql, 8.0, "looker@example.iam.gserviceaccount.com", cache_hit=True))
|
|
186
|
+
|
|
187
|
+
rng.shuffle(jobs)
|
|
188
|
+
return jobs
|
bqlens/fetch.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Read query history from INFORMATION_SCHEMA.JOBS.
|
|
2
|
+
|
|
3
|
+
Read-only. bqlens never reads a row of your actual table data — only job
|
|
4
|
+
metadata and the SQL text of the queries themselves.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .models import Job
|
|
10
|
+
|
|
11
|
+
# The JOBS view is region-qualified: region-us, region-eu, region-asia-south1, ...
|
|
12
|
+
JOBS_SQL_REGIONAL = """
|
|
13
|
+
SELECT
|
|
14
|
+
job_id,
|
|
15
|
+
user_email,
|
|
16
|
+
creation_time,
|
|
17
|
+
query,
|
|
18
|
+
total_bytes_billed,
|
|
19
|
+
total_bytes_processed,
|
|
20
|
+
cache_hit,
|
|
21
|
+
statement_type,
|
|
22
|
+
total_slot_ms,
|
|
23
|
+
referenced_tables,
|
|
24
|
+
-- error_result is a STRUCT<reason, location, debug_info, message>, so it
|
|
25
|
+
-- cannot be cast to STRING directly. Serialise it, and keep NULL as NULL
|
|
26
|
+
-- so that `is_failed` stays false for jobs that succeeded.
|
|
27
|
+
IF(error_result.reason IS NULL, NULL, TO_JSON_STRING(error_result)) AS error_result
|
|
28
|
+
FROM `{project}`.`{region}`.INFORMATION_SCHEMA.JOBS_BY_PROJECT
|
|
29
|
+
WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL @days DAY)
|
|
30
|
+
AND job_type = 'QUERY'
|
|
31
|
+
AND query IS NOT NULL
|
|
32
|
+
ORDER BY total_bytes_billed DESC
|
|
33
|
+
LIMIT @max_jobs
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def fetch_jobs(
|
|
38
|
+
project: str,
|
|
39
|
+
region: str = "region-us",
|
|
40
|
+
days: int = 30,
|
|
41
|
+
max_jobs: int = 20000,
|
|
42
|
+
) -> list[Job]:
|
|
43
|
+
"""Pull query jobs from BigQuery. Requires google-cloud-bigquery."""
|
|
44
|
+
try:
|
|
45
|
+
from google.cloud import bigquery
|
|
46
|
+
except ImportError as exc: # pragma: no cover - depends on optional extra
|
|
47
|
+
raise SystemExit(
|
|
48
|
+
"google-cloud-bigquery is not installed.\n"
|
|
49
|
+
"Install it with: pip install 'bqlens[bigquery]'\n"
|
|
50
|
+
"Or try the scanner first with: bqlens scan --demo"
|
|
51
|
+
) from exc
|
|
52
|
+
|
|
53
|
+
client = bigquery.Client(project=project)
|
|
54
|
+
sql = JOBS_SQL_REGIONAL.format(project=project, region=region)
|
|
55
|
+
config = bigquery.QueryJobConfig(
|
|
56
|
+
query_parameters=[
|
|
57
|
+
bigquery.ScalarQueryParameter("days", "INT64", days),
|
|
58
|
+
bigquery.ScalarQueryParameter("max_jobs", "INT64", max_jobs),
|
|
59
|
+
],
|
|
60
|
+
# Reading job metadata is itself billable. Cap it so the scanner can
|
|
61
|
+
# never be the expensive thing in someone's bill.
|
|
62
|
+
maximum_bytes_billed=10 * 1024**3,
|
|
63
|
+
)
|
|
64
|
+
rows = client.query(sql, job_config=config).result()
|
|
65
|
+
return [Job.from_row(dict(row)) for row in rows]
|
bqlens/models.py
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Core data structures for bqlens."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from datetime import datetime
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
# BigQuery on-demand analysis pricing. Default is the US multi-region list
|
|
10
|
+
# price per TiB scanned. Override with --price-per-tib for your region or
|
|
11
|
+
# negotiated rate.
|
|
12
|
+
DEFAULT_PRICE_PER_TIB = 6.25
|
|
13
|
+
|
|
14
|
+
BYTES_PER_TIB = 1024**4
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def bytes_to_usd(num_bytes: float, price_per_tib: float = DEFAULT_PRICE_PER_TIB) -> float:
|
|
18
|
+
"""Convert bytes billed into on-demand dollars."""
|
|
19
|
+
if num_bytes <= 0:
|
|
20
|
+
return 0.0
|
|
21
|
+
return (num_bytes / BYTES_PER_TIB) * price_per_tib
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def fmt_usd(amount: float) -> str:
|
|
25
|
+
"""Format dollars without rendering real money as a meaningless $0.00.
|
|
26
|
+
|
|
27
|
+
A finding worth a third of a cent is still a real finding — on a bill
|
|
28
|
+
100x larger it is $0.33 — so we say "<$0.01" rather than "$0.00", which
|
|
29
|
+
reads as a bug.
|
|
30
|
+
"""
|
|
31
|
+
if amount >= 0.01:
|
|
32
|
+
return f"${amount:,.2f}"
|
|
33
|
+
if amount > 0:
|
|
34
|
+
return "<$0.01"
|
|
35
|
+
return "$0.00"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def human_bytes(num_bytes: float) -> str:
|
|
39
|
+
"""Render a byte count in the largest sensible binary unit."""
|
|
40
|
+
step = 1024.0
|
|
41
|
+
value = float(num_bytes)
|
|
42
|
+
for unit in ("B", "KiB", "MiB", "GiB", "TiB"):
|
|
43
|
+
if abs(value) < step or unit == "TiB":
|
|
44
|
+
if unit == "B":
|
|
45
|
+
return f"{value:.0f} B"
|
|
46
|
+
return f"{value:.2f} {unit}"
|
|
47
|
+
value /= step
|
|
48
|
+
return f"{value:.2f} TiB"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class Job:
|
|
53
|
+
"""One BigQuery query job, normalised from INFORMATION_SCHEMA.JOBS."""
|
|
54
|
+
|
|
55
|
+
job_id: str
|
|
56
|
+
user_email: str
|
|
57
|
+
creation_time: datetime
|
|
58
|
+
query: str
|
|
59
|
+
total_bytes_billed: int
|
|
60
|
+
total_bytes_processed: int
|
|
61
|
+
cache_hit: bool = False
|
|
62
|
+
statement_type: str = "SELECT"
|
|
63
|
+
referenced_tables: list[str] = field(default_factory=list)
|
|
64
|
+
total_slot_ms: int = 0
|
|
65
|
+
destination_table: str | None = None
|
|
66
|
+
error_result: str | None = None
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def is_failed(self) -> bool:
|
|
70
|
+
return bool(self.error_result)
|
|
71
|
+
|
|
72
|
+
def cost(self, price_per_tib: float = DEFAULT_PRICE_PER_TIB) -> float:
|
|
73
|
+
return bytes_to_usd(self.total_bytes_billed, price_per_tib)
|
|
74
|
+
|
|
75
|
+
@classmethod
|
|
76
|
+
def from_row(cls, row: Any) -> "Job":
|
|
77
|
+
"""Build a Job from a BigQuery result row (or any mapping-like object)."""
|
|
78
|
+
get = row.get if hasattr(row, "get") else lambda k, d=None: getattr(row, k, d)
|
|
79
|
+
|
|
80
|
+
refs = get("referenced_tables") or []
|
|
81
|
+
normalised_refs = []
|
|
82
|
+
for ref in refs:
|
|
83
|
+
if isinstance(ref, str):
|
|
84
|
+
normalised_refs.append(ref)
|
|
85
|
+
else:
|
|
86
|
+
# BigQuery returns STRUCT<project_id, dataset_id, table_id>
|
|
87
|
+
project = ref.get("project_id") if hasattr(ref, "get") else getattr(ref, "project_id", None)
|
|
88
|
+
dataset = ref.get("dataset_id") if hasattr(ref, "get") else getattr(ref, "dataset_id", None)
|
|
89
|
+
table = ref.get("table_id") if hasattr(ref, "get") else getattr(ref, "table_id", None)
|
|
90
|
+
normalised_refs.append(f"{project}.{dataset}.{table}")
|
|
91
|
+
|
|
92
|
+
return cls(
|
|
93
|
+
job_id=get("job_id") or "",
|
|
94
|
+
user_email=get("user_email") or "unknown",
|
|
95
|
+
creation_time=get("creation_time"),
|
|
96
|
+
query=get("query") or "",
|
|
97
|
+
total_bytes_billed=int(get("total_bytes_billed") or 0),
|
|
98
|
+
total_bytes_processed=int(get("total_bytes_processed") or 0),
|
|
99
|
+
cache_hit=bool(get("cache_hit")),
|
|
100
|
+
statement_type=get("statement_type") or "SELECT",
|
|
101
|
+
referenced_tables=normalised_refs,
|
|
102
|
+
total_slot_ms=int(get("total_slot_ms") or 0),
|
|
103
|
+
destination_table=get("destination_table"),
|
|
104
|
+
error_result=get("error_result"),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass
|
|
109
|
+
class Finding:
|
|
110
|
+
"""One thing costing money that could stop costing money."""
|
|
111
|
+
|
|
112
|
+
rule_id: str
|
|
113
|
+
title: str
|
|
114
|
+
# Dollars per period that this finding accounts for.
|
|
115
|
+
wasted_usd: float
|
|
116
|
+
# Dollars per period we believe are *recoverable* if the fix is applied.
|
|
117
|
+
recoverable_usd: float
|
|
118
|
+
detail: str
|
|
119
|
+
fix: str
|
|
120
|
+
severity: str = "medium" # low | medium | high
|
|
121
|
+
job_count: int = 0
|
|
122
|
+
sample_job_ids: list[str] = field(default_factory=list)
|
|
123
|
+
owners: list[str] = field(default_factory=list)
|
|
124
|
+
|
|
125
|
+
def __post_init__(self) -> None:
|
|
126
|
+
if self.recoverable_usd > self.wasted_usd:
|
|
127
|
+
self.recoverable_usd = self.wasted_usd
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass
|
|
131
|
+
class ScanResult:
|
|
132
|
+
"""Everything one scan produced."""
|
|
133
|
+
|
|
134
|
+
findings: list[Finding]
|
|
135
|
+
total_jobs: int
|
|
136
|
+
total_spend_usd: float
|
|
137
|
+
days: int
|
|
138
|
+
project: str
|
|
139
|
+
price_per_tib: float = DEFAULT_PRICE_PER_TIB
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def recoverable_usd(self) -> float:
|
|
143
|
+
return sum(f.recoverable_usd for f in self.findings)
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def recoverable_pct(self) -> float:
|
|
147
|
+
if self.total_spend_usd <= 0:
|
|
148
|
+
return 0.0
|
|
149
|
+
return 100.0 * self.recoverable_usd / self.total_spend_usd
|
|
150
|
+
|
|
151
|
+
@property
|
|
152
|
+
def monthly_spend_usd(self) -> float:
|
|
153
|
+
if self.days <= 0:
|
|
154
|
+
return 0.0
|
|
155
|
+
return self.total_spend_usd * (30.0 / self.days)
|
|
156
|
+
|
|
157
|
+
@property
|
|
158
|
+
def monthly_recoverable_usd(self) -> float:
|
|
159
|
+
if self.days <= 0:
|
|
160
|
+
return 0.0
|
|
161
|
+
return self.recoverable_usd * (30.0 / self.days)
|