bigquery-cleaner 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,10 @@
1
+ """Package metadata for BigQuery Cleaner."""
2
+
3
+ import logging
4
+
5
+ __all__ = ["__version__", "logger"]
6
+
7
+ __version__ = "0.1.0"
8
+
9
+ # Configure logging
10
+ logger = logging.getLogger("bigquery_cleaner")
@@ -0,0 +1,401 @@
1
+ """BigQuery client helpers for dataset and table discovery."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import defaultdict
6
+ from collections.abc import Iterable
7
+ from dataclasses import dataclass
8
+ from datetime import datetime
9
+
10
+ from google.cloud import bigquery
11
+ from google.cloud.exceptions import NotFound
12
+
13
+ from . import logger
14
+ from .queries.sql import (
15
+ list_all_tables_across_datasets_sql,
16
+ list_old_tables_across_datasets_sql,
17
+ recent_references_across_datasets_sql,
18
+ )
19
+
20
+
21
+ @dataclass
22
+ class TableMetadata:
23
+ """Detailed information about a BigQuery table."""
24
+ table_id: str
25
+ created: datetime | None = None
26
+ modified: datetime | None = None
27
+ size_bytes: int | None = None
28
+
29
+
30
+ def get_client(project_id: str | None = None) -> bigquery.Client:
31
+ """Return an authenticated BigQuery client.
32
+
33
+ Uses Application Default Credentials (ADC). Optionally pin a project.
34
+
35
+ Args:
36
+ project_id: Optional GCP project ID to pin the client to.
37
+
38
+ Returns:
39
+ An authenticated bigquery.Client instance.
40
+ """
41
+ return bigquery.Client(project=project_id) if project_id else bigquery.Client()
42
+
43
+
44
+ def _split_dataset_ref(dataset: str, project_id: str | None) -> tuple[str, str]:
45
+ """Split a dataset ref into (project, dataset), using fallback project if needed.
46
+
47
+ Args:
48
+ dataset: Dataset ID string (e.g., "dataset" or "project.dataset").
49
+ project_id: Fallback project ID if the dataset string is not project-qualified.
50
+
51
+ Returns:
52
+ A tuple of (project_id, dataset_id).
53
+
54
+ Raises:
55
+ ValueError: If dataset is not qualified and no project_id is provided.
56
+ """
57
+ if "." in dataset:
58
+ project_id, dataset_id = dataset.split(".", 1)
59
+ return project_id, dataset_id
60
+ if not project_id:
61
+ raise ValueError(
62
+ "Dataset must be 'project.dataset' or provide project via --project"
63
+ )
64
+ return project_id, dataset
65
+
66
+
67
+ def list_datasets(project_id: str | None) -> list[str]:
68
+ """Return dataset IDs available in the given or default project.
69
+
70
+ Args:
71
+ project_id: Optional GCP project ID.
72
+
73
+ Returns:
74
+ A list of dataset IDs.
75
+ """
76
+ client = get_client(project_id)
77
+ # Fetch and return all dataset IDs from the project.
78
+ return [dataset.dataset_id for dataset in client.list_datasets()] # type: ignore[attr-defined]
79
+
80
+
81
+ def list_tables(dataset: str, project_id: str | None) -> list[str]:
82
+ """Return table IDs for the given dataset (project-qualified or not).
83
+
84
+ Args:
85
+ dataset: Dataset ID string (e.g., "dataset" or "project.dataset").
86
+ project_id: Fallback project ID if the dataset string is not project-qualified.
87
+
88
+ Returns:
89
+ A list of table IDs in the dataset.
90
+ """
91
+ project_id, dataset_id = _split_dataset_ref(dataset, project_id)
92
+ client = get_client(project_id)
93
+ # Fetch and return all table IDs from the specified dataset.
94
+ return [table.table_id for table in client.list_tables(dataset_id)] # type: ignore[attr-defined]
95
+
96
+
97
+ def normalize_datasets(
98
+ client: bigquery.Client,
99
+ datasets: list[str] | None,
100
+ default_project: str,
101
+ exclude_datasets: list[str] | None = None,
102
+ ) -> list[tuple[str, str]]:
103
+ """Normalize dataset inputs to (project, dataset) project_dataset_pairs.
104
+
105
+ If ``datasets`` is None/empty, list all datasets in the client's project.
106
+ Filters out any datasets present in ``exclude_datasets``.
107
+
108
+ Args:
109
+ client: BigQuery client instance.
110
+ datasets: Optional list of dataset strings to normalize.
111
+ default_project: Fallback project ID for non-qualified datasets.
112
+ exclude_datasets: Optional list of dataset strings to exclude from the result.
113
+
114
+ Returns:
115
+ A list of (project_id, dataset_id) tuples.
116
+ """
117
+ project_dataset_pairs: list[tuple[str, str]] = []
118
+ if datasets and len(datasets) > 0:
119
+ for dataset in datasets:
120
+ project_id, dataset_id = _split_dataset_ref(dataset, default_project)
121
+ project_dataset_pairs.append((project_id, dataset_id))
122
+ else:
123
+ # Generate a list of (project, dataset) tuples for all datasets in the project.
124
+ project_dataset_pairs = [
125
+ (client.project, dataset.dataset_id) for dataset in client.list_datasets()
126
+ ]
127
+
128
+ if exclude_datasets:
129
+ # Normalize exclude list for comparison
130
+ excluded = {
131
+ _split_dataset_ref(excluded_ds, default_project) for excluded_ds in exclude_datasets
132
+ }
133
+ # Filter out the datasets that are marked for exclusion.
134
+ project_dataset_pairs = [
135
+ pair for pair in project_dataset_pairs if pair not in excluded
136
+ ]
137
+
138
+ return project_dataset_pairs
139
+
140
+
141
+ def group_datasets_by_location(
142
+ client: bigquery.Client, project_dataset_pairs: Iterable[tuple[str, str]]
143
+ ) -> defaultdict[str, list[tuple[str, str]]]:
144
+ """Group (project, dataset) project_dataset_pairs by dataset location using metadata lookups.
145
+
146
+ Args:
147
+ client: BigQuery client instance.
148
+ project_dataset_pairs: Iterable of (project_id, dataset_id) tuples.
149
+
150
+ Returns:
151
+ A dictionary mapping location strings to lists of (project_id, dataset_id) tuples.
152
+ """
153
+ groups: defaultdict[str, list[tuple[str, str]]] = defaultdict(list)
154
+ for project_id, dataset_id in project_dataset_pairs:
155
+ ds_obj = client.get_dataset(bigquery.DatasetReference(project_id, dataset_id))
156
+ groups[ds_obj.location].append((project_id, dataset_id))
157
+ return groups
158
+
159
+
160
+ def get_recent_referenced_tables_by_dataset(
161
+ client: bigquery.Client,
162
+ location: str,
163
+ project_dataset_pairs: list[tuple[str, str]],
164
+ days: int,
165
+ ) -> defaultdict[str, set[str]]:
166
+ """Return recent referenced tables grouped by dataset for a location.
167
+
168
+ Falls back to empty sets on failure.
169
+
170
+ Args:
171
+ client: BigQuery client instance.
172
+ location: Dataset location (e.g., "US").
173
+ project_dataset_pairs: List of (project_id, dataset_id) tuples.
174
+ days: Lookback window in days.
175
+
176
+ Returns:
177
+ A dictionary mapping dataset ID to a set of table IDs referenced in queries.
178
+ """
179
+ region_dataset = f"region-{location.lower()}"
180
+ project = project_dataset_pairs[0][0]
181
+ # Extract unique dataset IDs from the project-dataset pairs.
182
+ dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
183
+
184
+ query = recent_references_across_datasets_sql(project, region_dataset)
185
+ cfg = bigquery.QueryJobConfig(
186
+ query_parameters=[
187
+ bigquery.ScalarQueryParameter("days", "INT64", days),
188
+ bigquery.ScalarQueryParameter("project_id", "STRING", project),
189
+ bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids),
190
+ ]
191
+ )
192
+ out: defaultdict[str, set[str]] = defaultdict(set)
193
+ try:
194
+ for row in client.query(query, job_config=cfg, location=location).result():
195
+ out[row["dataset_id"]].add(row["table_id"])
196
+ except Exception as err:
197
+ logger.warning("Recent references query failed for location %s: %s", location, err)
198
+ out = defaultdict(set)
199
+ raise err
200
+ return out
201
+
202
+
203
+ def get_all_tables_for_location(
204
+ client: bigquery.Client,
205
+ location: str,
206
+ project_dataset_pairs: list[tuple[str, str]],
207
+ ) -> defaultdict[str, dict[str, TableMetadata]]:
208
+ """Return all tables grouped by dataset for a location.
209
+
210
+ Falls back to listing per dataset via API on failure.
211
+
212
+ Args:
213
+ client: BigQuery client instance.
214
+ location: Dataset location (e.g., "US").
215
+ project_dataset_pairs: List of (project_id, dataset_id) tuples.
216
+
217
+ Returns:
218
+ A dictionary mapping dataset ID to a dict of table_id -> TableMetadata.
219
+ """
220
+ region_dataset = f"region-{location.lower()}"
221
+ project = project_dataset_pairs[0][0]
222
+ # Extract unique dataset IDs from the project-dataset pairs.
223
+ dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
224
+
225
+ query = list_all_tables_across_datasets_sql(project, region_dataset)
226
+ cfg = bigquery.QueryJobConfig(
227
+ query_parameters=[bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids)]
228
+ )
229
+ out: defaultdict[str, dict[str, TableMetadata]] = defaultdict(dict)
230
+ try:
231
+ for row in client.query(query, job_config=cfg, location=location).result():
232
+ table_id = row["table_id"]
233
+ ds_id = row["dataset_id"]
234
+ out[ds_id][table_id] = TableMetadata(
235
+ table_id=table_id,
236
+ created=row.get("creation_time"),
237
+ size_bytes=row.get("total_bytes"),
238
+ )
239
+ except Exception as err:
240
+ logger.warning("Batched table list query failed for location %s: %s. Falling back to per-dataset API calls.", location, err)
241
+ raise err
242
+
243
+ return out
244
+
245
+
246
+ def get_old_modified_tables_for_location(
247
+ client: bigquery.Client,
248
+ location: str,
249
+ project_dataset_pairs: list[tuple[str, str]],
250
+ days: int,
251
+ ) -> dict[str, list[TableMetadata]]:
252
+ """Return tables modified before the lookback window for a location.
253
+
254
+ Falls back to per-dataset queries on failure.
255
+
256
+ Args:
257
+ client: BigQuery client instance.
258
+ location: Dataset location (e.g., "US").
259
+ project_dataset_pairs: List of (project_id, dataset_id) tuples.
260
+ days: Lookback window in days.
261
+
262
+ Returns:
263
+ Mapping: ``project.dataset`` -> [TableMetadata ...]
264
+ """
265
+ region_dataset = f"region-{location.lower()}"
266
+ project = project_dataset_pairs[0][0]
267
+ # Extract unique dataset IDs from the project-dataset pairs.
268
+ dataset_ids = sorted({dataset_id for _, dataset_id in project_dataset_pairs})
269
+
270
+ query = list_old_tables_across_datasets_sql(project, region_dataset)
271
+ cfg = bigquery.QueryJobConfig(
272
+ query_parameters=[
273
+ bigquery.ScalarQueryParameter("days", "INT64", days),
274
+ bigquery.ArrayQueryParameter("dataset_ids", "STRING", dataset_ids),
275
+ ]
276
+ )
277
+ all_by_ds: defaultdict[str, dict[str, TableMetadata]] = defaultdict(dict)
278
+ try:
279
+ for row in client.query(query, job_config=cfg, location=location).result():
280
+ table_id = row["table_id"]
281
+ ds_id = row["dataset_id"]
282
+ all_by_ds[ds_id][table_id] = TableMetadata(
283
+ table_id=table_id,
284
+ modified=row.get("storage_last_modified_time"),
285
+ size_bytes=row.get("total_bytes"),
286
+ )
287
+ except Exception as err:
288
+ logger.warning(
289
+ "Batched old-tables query failed for location %s: %s.",
290
+ location,
291
+ err,
292
+ )
293
+ raise err
294
+
295
+ # Format results
296
+ out: dict[str, list[TableMetadata]] = {}
297
+ for project_id, dataset_id in project_dataset_pairs:
298
+ key = f"{project_id}.{dataset_id}"
299
+ ds_tables = all_by_ds.get(dataset_id, {})
300
+ # Convert the dictionary of table metadata into a sorted list.
301
+ out[key] = [
302
+ ds_tables[table_id] for table_id in sorted(ds_tables.keys())
303
+ ]
304
+ return out
305
+
306
+
307
+ def rename_table(
308
+ client: bigquery.Client,
309
+ project_id: str,
310
+ dataset_id: str,
311
+ table_id: str,
312
+ new_table_id: str,
313
+ location: str,
314
+ ) -> None:
315
+ """Rename a BigQuery table using ALTER TABLE ... RENAME TO ...
316
+
317
+ Args:
318
+ client: BigQuery client instance.
319
+ project_id: GCP project ID.
320
+ dataset_id: Dataset ID.
321
+ table_id: Original table name.
322
+ new_table_id: New table name.
323
+ location: Dataset location.
324
+ """
325
+ sql = f"ALTER TABLE `{project_id}.{dataset_id}.{table_id}` RENAME TO `{new_table_id}`"
326
+ client.query(sql, location=location).result()
327
+
328
+
329
+ def rename_tables(
330
+ client: bigquery.Client,
331
+ statements: list[str],
332
+ location: str,
333
+ ) -> None:
334
+ """Execute multiple SQL statements in a single query.
335
+
336
+ Args:
337
+ client: BigQuery client instance.
338
+ statements: List of SQL statements to execute.
339
+ location: Dataset location.
340
+ """
341
+ if not statements:
342
+ return
343
+ sql = ";\n".join(statements)
344
+ client.query(sql, location=location).result()
345
+
346
+
347
+ def delete_tables(
348
+ client: bigquery.Client,
349
+ statements: list[str],
350
+ location: str,
351
+ ) -> None:
352
+ """Execute multiple DROP TABLE statements in a single query.
353
+
354
+ Args:
355
+ client: BigQuery client instance.
356
+ statements: List of SQL statements to execute.
357
+ location: Dataset location.
358
+ """
359
+ if not statements:
360
+ return
361
+ sql = ";\n".join(statements)
362
+ client.query(sql, location=location).result()
363
+
364
+
365
+ def delete_dataset(
366
+ client: bigquery.Client, project_id: str, dataset_id: str, not_found_ok: bool = True
367
+ ) -> None:
368
+ """Delete a BigQuery dataset.
369
+
370
+ Args:
371
+ client: BigQuery client instance.
372
+ project_id: GCP project ID.
373
+ dataset_id: Dataset ID.
374
+ not_found_ok: If True, do not raise error if dataset doesn't exist.
375
+ """
376
+ dataset_ref = bigquery.DatasetReference(project_id, dataset_id)
377
+ client.delete_dataset(dataset_ref, delete_contents=False, not_found_ok=not_found_ok)
378
+
379
+
380
+ def table_exists(
381
+ client: bigquery.Client, project_id: str, dataset_id: str, table_id: str
382
+ ) -> bool:
383
+ """Check if a table exists in BigQuery.
384
+
385
+ Args:
386
+ client: BigQuery client instance.
387
+ project_id: GCP project ID.
388
+ dataset_id: Dataset ID.
389
+ table_id: Table ID to check.
390
+
391
+ Returns:
392
+ True if the table exists, False otherwise.
393
+ """
394
+ table_ref = bigquery.TableReference(
395
+ bigquery.DatasetReference(project_id, dataset_id), table_id
396
+ )
397
+ try:
398
+ client.get_table(table_ref)
399
+ return True
400
+ except NotFound:
401
+ return False