bigquery-cleaner 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bigquery_cleaner/__init__.py +10 -0
- bigquery_cleaner/bq_client.py +401 -0
- bigquery_cleaner/cli.py +523 -0
- bigquery_cleaner/config.py +139 -0
- bigquery_cleaner/core_operations.py +315 -0
- bigquery_cleaner/queries/sql.py +66 -0
- bigquery_cleaner/utils.py +91 -0
- bigquery_cleaner-0.1.0.dist-info/METADATA +182 -0
- bigquery_cleaner-0.1.0.dist-info/RECORD +12 -0
- bigquery_cleaner-0.1.0.dist-info/WHEEL +4 -0
- bigquery_cleaner-0.1.0.dist-info/entry_points.txt +2 -0
- bigquery_cleaner-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,401 @@
|
|
|
1
|
+
"""BigQuery client helpers for dataset and table discovery."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from collections.abc import Iterable
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
|
|
10
|
+
from google.cloud import bigquery
|
|
11
|
+
from google.cloud.exceptions import NotFound
|
|
12
|
+
|
|
13
|
+
from . import logger
|
|
14
|
+
from .queries.sql import (
|
|
15
|
+
list_all_tables_across_datasets_sql,
|
|
16
|
+
list_old_tables_across_datasets_sql,
|
|
17
|
+
recent_references_across_datasets_sql,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class TableMetadata:
|
|
23
|
+
"""Detailed information about a BigQuery table."""
|
|
24
|
+
table_id: str
|
|
25
|
+
created: datetime | None = None
|
|
26
|
+
modified: datetime | None = None
|
|
27
|
+
size_bytes: int | None = None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def get_client(project_id: str | None = None) -> bigquery.Client:
|
|
31
|
+
"""Return an authenticated BigQuery client.
|
|
32
|
+
|
|
33
|
+
Uses Application Default Credentials (ADC). Optionally pin a project.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
project_id: Optional GCP project ID to pin the client to.
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
An authenticated bigquery.Client instance.
|
|
40
|
+
"""
|
|
41
|
+
return bigquery.Client(project=project_id) if project_id else bigquery.Client()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _split_dataset_ref(dataset: str, project_id: str | None) -> tuple[str, str]:
|
|
45
|
+
"""Split a dataset ref into (project, dataset), using fallback project if needed.
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
dataset: Dataset ID string (e.g., "dataset" or "project.dataset").
|
|
49
|
+
project_id: Fallback project ID if the dataset string is not project-qualified.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
A tuple of (project_id, dataset_id).
|
|
53
|
+
|
|
54
|
+
Raises:
|
|
55
|
+
ValueError: If dataset is not qualified and no project_id is provided.
|
|
56
|
+
"""
|
|
57
|
+
if "." in dataset:
|
|
58
|
+
project_id, dataset_id = dataset.split(".", 1)
|
|
59
|
+
return project_id, dataset_id
|
|
60
|
+
if not project_id:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
"Dataset must be 'project.dataset' or provide project via --project"
|
|
63
|
+
)
|
|
64
|
+
return project_id, dataset
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def list_datasets(project_id: str | None) -> list[str]:
|
|
68
|
+
"""Return dataset IDs available in the given or default project.
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
project_id: Optional GCP project ID.
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
A list of dataset IDs.
|
|
75
|
+
"""
|
|
76
|
+
client = get_client(project_id)
|
|
77
|
+
# Fetch and return all dataset IDs from the project.
|
|
78
|
+
return [dataset.dataset_id for dataset in client.list_datasets()] # type: ignore[attr-defined]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def list_tables(dataset: str, project_id: str | None) -> list[str]:
|
|
82
|
+
"""Return table IDs for the given dataset (project-qualified or not).
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
dataset: Dataset ID string (e.g., "dataset" or "project.dataset").
|
|
86
|
+
project_id: Fallback project ID if the dataset string is not project-qualified.
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
A list of table IDs in the dataset.
|
|
90
|
+
"""
|
|
91
|
+
project_id, dataset_id = _split_dataset_ref(dataset, project_id)
|
|
92
|
+
client = get_client(project_id)
|
|
93
|
+
# Fetch and return all table IDs from the specified dataset.
|
|
94
|
+
return [table.table_id for table in client.list_tables(dataset_id)] # type: ignore[attr-defined]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def normalize_datasets(
|
|
98
|
+
client: bigquery.Client,
|
|
99
|
+
datasets: list[str] | None,
|
|
100
|
+
default_project: str,
|
|
101
|
+
exclude_datasets: list[str] | None = None,
|
|
102
|
+
) -> list[tuple[str, str]]:
|
|
103
|
+
"""Normalize dataset inputs to (project, dataset) project_dataset_pairs.
|
|
104
|
+
|
|
105
|
+
If ``datasets`` is None/empty, list all datasets in the client's project.
|
|
106
|
+
Filters out any datasets present in ``exclude_datasets``.
|
|
107
|
+
|
|
108
|
+
Args:
|
|
109
|
+
client: BigQuery client instance.
|
|
110
|
+
datasets: Optional list of dataset strings to normalize.
|
|
111
|
+
default_project: Fallback project ID for non-qualified datasets.
|
|
112
|
+
exclude_datasets: Optional list of dataset strings to exclude from the result.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
A list of (project_id, dataset_id) tuples.
|
|
116
|
+
"""
|
|
117
|
+
project_dataset_pairs: list[tuple[str, str]] = []
|
|
118
|
+
if datasets and len(datasets) > 0:
|
|
119
|
+
for dataset in datasets:
|
|
120
|
+
project_id, dataset_id = _split_dataset_ref(dataset, default_project)
|
|
121
|
+
project_dataset_pairs.append((project_id, dataset_id))
|
|
122
|
+
else:
|
|
123
|
+
# Generate a list of (project, dataset) tuples for all datasets in the project.
|
|
124
|
+
project_dataset_pairs = [
|
|
125
|
+
(client.project, dataset.dataset_id) for dataset in client.list_datasets()
|
|
126
|
+
]
|
|
127
|
+
|
|
128
|
+
if exclude_datasets:
|
|
129
|
+
# Normalize exclude list for comparison
|
|
130
|
+
excluded = {
|
|
131
|
+
_split_dataset_ref(excluded_ds, default_project) for excluded_ds in exclude_datasets
|
|
132
|
+
}
|
|
133
|
+
# Filter out the datasets that are marked for exclusion.
|
|
134
|
+
project_dataset_pairs = [
|
|
135
|
+
pair for pair in project_dataset_pairs if pair not in excluded
|
|
136
|
+
]
|
|
137
|
+
|
|
138
|
+
return project_dataset_pairs
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def group_datasets_by_location(
|
|
142
|
+
client: bigquery.Client, project_dataset_pairs: Iterable[tuple[str, str]]
|
|
143
|
+
) -> defaultdict[str, list[tuple[str, str]]]:
|
|
144
|
+
"""Group (project, dataset) project_dataset_pairs by dataset location using metadata lookups.
|
|
145
|
+
|
|
146
|
+
Args:
|
|
147
|
+
client: BigQuery client instance.
|
|
148
|
+
project_dataset_pairs: Iterable of (project_id, dataset_id) tuples.
|
|
149
|
+
|
|
150
|
+
Returns:
|
|
151
|
+
A dictionary mapping location strings to lists of (project_id, dataset_id) tuples.
|
|
152
|
+
"""
|
|
153
|
+
groups: defaultdict[str, list[tuple[str, str]]] = defaultdict(list)
|
|
154
|
+
for project_id, dataset_id in project_dataset_pairs:
|
|
155
|
+
ds_obj = client.get_dataset(bigquery.DatasetReference(project_id, dataset_id))
|
|
156
|
+
groups[ds_obj.location].append((project_id, dataset_id))
|
|
157
|
+
return groups
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def get_recent_referenced_tables_by_dataset(
|
|
161
|
+
client: bigquery.Client,
|
|
162
|
+
location: str,
|
|
163
|
+
project_dataset_pairs: list[tuple[str, str]],
|
|
164
|
+
days: int,
|
|
165
|
+
) -> defaultdict[str, set[str]]:
|
|
166
|
+
"""Return recent referenced tables grouped by dataset for a location.
|
|
167
|
+
|
|
168
|
+
Falls back to empty sets on failure.
|
|
169
|
+
|
|
170
|
+
Args:
|
|
171
|
+
client: BigQuery client instance.
|
|
172
|
+
location: Dataset location (e.g., "US").
|
|
173
|
+
project_dataset_pairs: List of (project_id, dataset_id) tuples.
|
|
174
|
+
days: Lookback window in days.
|
|
175
|
+
|
|
176
|
+
Returns:
|
|
177
|
+
A dictionary mapping dataset ID to a set of table IDs referenced in queries.
|
|
178
|
+
"""
|
|
179
|
+
region_dataset = f"region-{location.lower()}"
|
|
180
|
+
project = project_dataset_pairs[0][0]
|
|
181
|
+
# Extract unique dataset IDs from the project-dataset pairs.
|
|
182
|
+
dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
|
|
183
|
+
|
|
184
|
+
query = recent_references_across_datasets_sql(project, region_dataset)
|
|
185
|
+
cfg = bigquery.QueryJobConfig(
|
|
186
|
+
query_parameters=[
|
|
187
|
+
bigquery.ScalarQueryParameter("days", "INT64", days),
|
|
188
|
+
bigquery.ScalarQueryParameter("project_id", "STRING", project),
|
|
189
|
+
bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids),
|
|
190
|
+
]
|
|
191
|
+
)
|
|
192
|
+
out: defaultdict[str, set[str]] = defaultdict(set)
|
|
193
|
+
try:
|
|
194
|
+
for row in client.query(query, job_config=cfg, location=location).result():
|
|
195
|
+
out[row["dataset_id"]].add(row["table_id"])
|
|
196
|
+
except Exception as err:
|
|
197
|
+
logger.warning("Recent references query failed for location %s: %s", location, err)
|
|
198
|
+
out = defaultdict(set)
|
|
199
|
+
raise err
|
|
200
|
+
return out
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def get_all_tables_for_location(
|
|
204
|
+
client: bigquery.Client,
|
|
205
|
+
location: str,
|
|
206
|
+
project_dataset_pairs: list[tuple[str, str]],
|
|
207
|
+
) -> defaultdict[str, dict[str, TableMetadata]]:
|
|
208
|
+
"""Return all tables grouped by dataset for a location.
|
|
209
|
+
|
|
210
|
+
Falls back to listing per dataset via API on failure.
|
|
211
|
+
|
|
212
|
+
Args:
|
|
213
|
+
client: BigQuery client instance.
|
|
214
|
+
location: Dataset location (e.g., "US").
|
|
215
|
+
project_dataset_pairs: List of (project_id, dataset_id) tuples.
|
|
216
|
+
|
|
217
|
+
Returns:
|
|
218
|
+
A dictionary mapping dataset ID to a dict of table_id -> TableMetadata.
|
|
219
|
+
"""
|
|
220
|
+
region_dataset = f"region-{location.lower()}"
|
|
221
|
+
project = project_dataset_pairs[0][0]
|
|
222
|
+
# Extract unique dataset IDs from the project-dataset pairs.
|
|
223
|
+
dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
|
|
224
|
+
|
|
225
|
+
query = list_all_tables_across_datasets_sql(project, region_dataset)
|
|
226
|
+
cfg = bigquery.QueryJobConfig(
|
|
227
|
+
query_parameters=[bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids)]
|
|
228
|
+
)
|
|
229
|
+
out: defaultdict[str, dict[str, TableMetadata]] = defaultdict(dict)
|
|
230
|
+
try:
|
|
231
|
+
for row in client.query(query, job_config=cfg, location=location).result():
|
|
232
|
+
table_id = row["table_id"]
|
|
233
|
+
ds_id = row["dataset_id"]
|
|
234
|
+
out[ds_id][table_id] = TableMetadata(
|
|
235
|
+
table_id=table_id,
|
|
236
|
+
created=row.get("creation_time"),
|
|
237
|
+
size_bytes=row.get("total_bytes"),
|
|
238
|
+
)
|
|
239
|
+
except Exception as err:
|
|
240
|
+
logger.warning("Batched table list query failed for location %s: %s. Falling back to per-dataset API calls.", location, err)
|
|
241
|
+
raise err
|
|
242
|
+
|
|
243
|
+
return out
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def get_old_modified_tables_for_location(
|
|
247
|
+
client: bigquery.Client,
|
|
248
|
+
location: str,
|
|
249
|
+
project_dataset_pairs: list[tuple[str, str]],
|
|
250
|
+
days: int,
|
|
251
|
+
) -> dict[str, list[TableMetadata]]:
|
|
252
|
+
"""Return tables modified before the lookback window for a location.
|
|
253
|
+
|
|
254
|
+
Falls back to per-dataset queries on failure.
|
|
255
|
+
|
|
256
|
+
Args:
|
|
257
|
+
client: BigQuery client instance.
|
|
258
|
+
location: Dataset location (e.g., "US").
|
|
259
|
+
project_dataset_pairs: List of (project_id, dataset_id) tuples.
|
|
260
|
+
days: Lookback window in days.
|
|
261
|
+
|
|
262
|
+
Returns:
|
|
263
|
+
Mapping: ``project.dataset`` -> [TableMetadata ...]
|
|
264
|
+
"""
|
|
265
|
+
region_dataset = f"region-{location.lower()}"
|
|
266
|
+
project = project_dataset_pairs[0][0]
|
|
267
|
+
# Extract unique dataset IDs from the project-dataset pairs.
|
|
268
|
+
dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
|
|
269
|
+
|
|
270
|
+
query = list_old_tables_across_datasets_sql(project, region_dataset)
|
|
271
|
+
cfg = bigquery.QueryJobConfig(
|
|
272
|
+
query_parameters=[
|
|
273
|
+
bigquery.ScalarQueryParameter("days", "INT64", days),
|
|
274
|
+
bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids),
|
|
275
|
+
]
|
|
276
|
+
)
|
|
277
|
+
all_by_ds: defaultdict[str, dict[str, TableMetadata]] = defaultdict(dict)
|
|
278
|
+
try:
|
|
279
|
+
for row in client.query(query, job_config=cfg, location=location).result():
|
|
280
|
+
table_id = row["table_id"]
|
|
281
|
+
ds_id = row["dataset_id"]
|
|
282
|
+
all_by_ds[ds_id][table_id] = TableMetadata(
|
|
283
|
+
table_id=table_id,
|
|
284
|
+
modified=row.get("storage_last_modified_time"),
|
|
285
|
+
size_bytes=row.get("total_bytes"),
|
|
286
|
+
)
|
|
287
|
+
except Exception as err:
|
|
288
|
+
logger.warning(
|
|
289
|
+
"Batched old-tables query failed for location %s: %s.",
|
|
290
|
+
location,
|
|
291
|
+
err,
|
|
292
|
+
)
|
|
293
|
+
raise err
|
|
294
|
+
|
|
295
|
+
# Format results
|
|
296
|
+
out: dict[str, list[TableMetadata]] = {}
|
|
297
|
+
for project_id, dataset_id in project_dataset_pairs:
|
|
298
|
+
key = f"{project_id}.{dataset_id}"
|
|
299
|
+
ds_tables = all_by_ds.get(dataset_id, {})
|
|
300
|
+
# Convert the dictionary of table metadata into a sorted list.
|
|
301
|
+
out[key] = [
|
|
302
|
+
ds_tables[table_id] for table_id in sorted(ds_tables.keys())
|
|
303
|
+
]
|
|
304
|
+
return out
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def rename_table(
|
|
308
|
+
client: bigquery.Client,
|
|
309
|
+
project_id: str,
|
|
310
|
+
dataset_id: str,
|
|
311
|
+
table_id: str,
|
|
312
|
+
new_table_id: str,
|
|
313
|
+
location: str,
|
|
314
|
+
) -> None:
|
|
315
|
+
"""Rename a BigQuery table using ALTER TABLE ... RENAME TO ...
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
client: BigQuery client instance.
|
|
319
|
+
project_id: GCP project ID.
|
|
320
|
+
dataset_id: Dataset ID.
|
|
321
|
+
table_id: Original table name.
|
|
322
|
+
new_table_id: New table name.
|
|
323
|
+
location: Dataset location.
|
|
324
|
+
"""
|
|
325
|
+
sql = f"ALTER TABLE `{project_id}.{dataset_id}.{table_id}` RENAME TO `{new_table_id}`"
|
|
326
|
+
client.query(sql, location=location).result()
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def rename_tables(
|
|
330
|
+
client: bigquery.Client,
|
|
331
|
+
statements: list[str],
|
|
332
|
+
location: str,
|
|
333
|
+
) -> None:
|
|
334
|
+
"""Execute multiple SQL statements in a single query.
|
|
335
|
+
|
|
336
|
+
Args:
|
|
337
|
+
client: BigQuery client instance.
|
|
338
|
+
statements: List of SQL statements to execute.
|
|
339
|
+
location: Dataset location.
|
|
340
|
+
"""
|
|
341
|
+
if not statements:
|
|
342
|
+
return
|
|
343
|
+
sql = ";\n".join(statements)
|
|
344
|
+
client.query(sql, location=location).result()
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def delete_tables(
|
|
348
|
+
client: bigquery.Client,
|
|
349
|
+
statements: list[str],
|
|
350
|
+
location: str,
|
|
351
|
+
) -> None:
|
|
352
|
+
"""Execute multiple DROP TABLE statements in a single query.
|
|
353
|
+
|
|
354
|
+
Args:
|
|
355
|
+
client: BigQuery client instance.
|
|
356
|
+
statements: List of SQL statements to execute.
|
|
357
|
+
location: Dataset location.
|
|
358
|
+
"""
|
|
359
|
+
if not statements:
|
|
360
|
+
return
|
|
361
|
+
sql = ";\n".join(statements)
|
|
362
|
+
client.query(sql, location=location).result()
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def delete_dataset(
|
|
366
|
+
client: bigquery.Client, project_id: str, dataset_id: str, not_found_ok: bool = True
|
|
367
|
+
) -> None:
|
|
368
|
+
"""Delete a BigQuery dataset.
|
|
369
|
+
|
|
370
|
+
Args:
|
|
371
|
+
client: BigQuery client instance.
|
|
372
|
+
project_id: GCP project ID.
|
|
373
|
+
dataset_id: Dataset ID.
|
|
374
|
+
not_found_ok: If True, do not raise error if dataset doesn't exist.
|
|
375
|
+
"""
|
|
376
|
+
dataset_ref = bigquery.DatasetReference(project_id, dataset_id)
|
|
377
|
+
client.delete_dataset(dataset_ref, delete_contents=False, not_found_ok=not_found_ok)
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def table_exists(
|
|
381
|
+
client: bigquery.Client, project_id: str, dataset_id: str, table_id: str
|
|
382
|
+
) -> bool:
|
|
383
|
+
"""Check if a table exists in BigQuery.
|
|
384
|
+
|
|
385
|
+
Args:
|
|
386
|
+
client: BigQuery client instance.
|
|
387
|
+
project_id: GCP project ID.
|
|
388
|
+
dataset_id: Dataset ID.
|
|
389
|
+
table_id: Table ID to check.
|
|
390
|
+
|
|
391
|
+
Returns:
|
|
392
|
+
True if the table exists, False otherwise.
|
|
393
|
+
"""
|
|
394
|
+
table_ref = bigquery.TableReference(
|
|
395
|
+
bigquery.DatasetReference(project_id, dataset_id), table_id
|
|
396
|
+
)
|
|
397
|
+
try:
|
|
398
|
+
client.get_table(table_ref)
|
|
399
|
+
return True
|
|
400
|
+
except NotFound:
|
|
401
|
+
return False
|