interloper-google-cloud 0.76.0__tar.gz → 0.78.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/PKG-INFO +2 -1
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/pyproject.toml +5 -2
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/pyproject.toml.orig +4 -3
- interloper_google_cloud-0.78.0/src/interloper_google_cloud/bigquery/destination.py +588 -0
- interloper_google_cloud-0.78.0/src/interloper_google_cloud/bigquery/types.py +90 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/gcs/destination.py +24 -31
- interloper_google_cloud-0.76.0/src/interloper_google_cloud/bigquery/destination.py +0 -892
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/README.md +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/__init__.py +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/bigquery/__init__.py +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/connection.py +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/gcs/__init__.py +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/gcs/formats.py +0 -0
- {interloper_google_cloud-0.76.0 → interloper_google_cloud-0.78.0}/src/interloper_google_cloud/serialization.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: interloper-google-cloud
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.78.0
|
|
4
4
|
Summary: Interloper Google Cloud integration: BigQuery and Cloud Storage destinations
|
|
5
5
|
Author: Guillaume Onfroy
|
|
6
6
|
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
@@ -10,6 +10,7 @@ Requires-Dist: interloper-core
|
|
|
10
10
|
Requires-Dist: interloper-pandas
|
|
11
11
|
Requires-Dist: pyarrow>=14
|
|
12
12
|
Requires-Dist: db-dtypes>=1.0
|
|
13
|
+
Requires-Dist: pandas-gbq>=0.26.1
|
|
13
14
|
Requires-Python: >=3.10
|
|
14
15
|
Description-Content-Type: text/markdown
|
|
15
16
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "interloper-google-cloud"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.78.0"
|
|
4
4
|
description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.10"
|
|
@@ -11,6 +11,7 @@ dependencies = [
|
|
|
11
11
|
"interloper-pandas",
|
|
12
12
|
"pyarrow>=14",
|
|
13
13
|
"db-dtypes>=1.0",
|
|
14
|
+
"pandas-gbq>=0.26.1",
|
|
14
15
|
]
|
|
15
16
|
|
|
16
17
|
[[project.authors]]
|
|
@@ -21,7 +22,7 @@ email = "guillaume@digitlcloud.com"
|
|
|
21
22
|
google_cloud = "interloper_google_cloud"
|
|
22
23
|
|
|
23
24
|
[build-system]
|
|
24
|
-
requires = ["uv_build>=0.
|
|
25
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
25
26
|
build-backend = "uv_build"
|
|
26
27
|
|
|
27
28
|
[tool.uv.sources.interloper-core]
|
|
@@ -61,4 +62,6 @@ convention = "google"
|
|
|
61
62
|
"D102",
|
|
62
63
|
"D103",
|
|
63
64
|
"D104",
|
|
65
|
+
"RUF069",
|
|
66
|
+
"PLW0108",
|
|
64
67
|
]
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# ###############
|
|
4
4
|
[project]
|
|
5
5
|
name = "interloper-google-cloud"
|
|
6
|
-
version = "0.
|
|
6
|
+
version = "0.78.0"
|
|
7
7
|
description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
@@ -15,13 +15,14 @@ dependencies = [
|
|
|
15
15
|
"interloper-pandas",
|
|
16
16
|
"pyarrow>=14",
|
|
17
17
|
"db-dtypes>=1.0",
|
|
18
|
+
"pandas-gbq>=0.26.1",
|
|
18
19
|
]
|
|
19
20
|
|
|
20
21
|
[project.entry-points."interloper.components"]
|
|
21
22
|
google_cloud = "interloper_google_cloud"
|
|
22
23
|
|
|
23
24
|
[build-system]
|
|
24
|
-
requires = ["uv_build>=0.
|
|
25
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
25
26
|
build-backend = "uv_build"
|
|
26
27
|
|
|
27
28
|
[tool.uv.sources]
|
|
@@ -43,4 +44,4 @@ convention = "google"
|
|
|
43
44
|
|
|
44
45
|
[tool.ruff.lint.per-file-ignores]
|
|
45
46
|
"__init__.py" = ["F401", "F403"]
|
|
46
|
-
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104"]
|
|
47
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,588 @@
|
|
|
1
|
+
"""BigQuery destination implementation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import inspect
|
|
7
|
+
import json
|
|
8
|
+
import warnings
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
from functools import cached_property
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import google.auth
|
|
14
|
+
import pandas as pd
|
|
15
|
+
from google.cloud import bigquery
|
|
16
|
+
from google.cloud.exceptions import Conflict, NotFound
|
|
17
|
+
from google.oauth2 import service_account
|
|
18
|
+
from interloper.destination import IOContext, destination
|
|
19
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
20
|
+
from interloper.errors import ConfigError, DataNotFoundError
|
|
21
|
+
from interloper.partitioning import PartitionConfig, TimeGranularity, TimePartitionConfig
|
|
22
|
+
from interloper.representation import Representation
|
|
23
|
+
from interloper.resource.fields import FetchField, InputField, SelectField
|
|
24
|
+
|
|
25
|
+
from interloper_google_cloud.bigquery.types import TIME_PARTITIONABLE, column_type, parameter_type, to_fields
|
|
26
|
+
from interloper_google_cloud.connection import GoogleCloudConnection
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@destination(
|
|
30
|
+
key="bigquery_destination",
|
|
31
|
+
name="BigQuery",
|
|
32
|
+
icon="icon:bigquery",
|
|
33
|
+
tags=["Cloud"],
|
|
34
|
+
)
|
|
35
|
+
class BigQueryDestination(DatabaseDestination):
|
|
36
|
+
"""BigQuery destination."""
|
|
37
|
+
|
|
38
|
+
connection: GoogleCloudConnection
|
|
39
|
+
|
|
40
|
+
project: str = FetchField(
|
|
41
|
+
provider="connection.projects",
|
|
42
|
+
label_key="name",
|
|
43
|
+
value_key="project_id",
|
|
44
|
+
description="Google Cloud project ID",
|
|
45
|
+
discriminator=True,
|
|
46
|
+
)
|
|
47
|
+
location: str = SelectField(
|
|
48
|
+
description="BigQuery dataset location",
|
|
49
|
+
options=[
|
|
50
|
+
{"label": "EU", "value": "EU"},
|
|
51
|
+
{"label": "US", "value": "US"},
|
|
52
|
+
],
|
|
53
|
+
)
|
|
54
|
+
default_dataset: str | None = InputField(default=None, description="Default BigQuery dataset")
|
|
55
|
+
|
|
56
|
+
@cached_property
|
|
57
|
+
def client(self) -> bigquery.Client:
|
|
58
|
+
"""The BigQuery client every write goes through.
|
|
59
|
+
|
|
60
|
+
Built from the connection's service-account key when there is one,
|
|
61
|
+
else from ambient credentials (workload identity in-cluster).
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
The client, cached per destination instance.
|
|
65
|
+
"""
|
|
66
|
+
if self.connection and self.connection.service_account_key:
|
|
67
|
+
key_info = json.loads(self.connection.service_account_key)
|
|
68
|
+
credentials = service_account.Credentials.from_service_account_info(key_info)
|
|
69
|
+
else:
|
|
70
|
+
credentials, _ = google.auth.default()
|
|
71
|
+
|
|
72
|
+
return bigquery.Client(
|
|
73
|
+
project=self.project,
|
|
74
|
+
credentials=credentials,
|
|
75
|
+
location=self.location,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# -- Helpers ---------------------------------------------------------------
|
|
79
|
+
|
|
80
|
+
def _resolve_dataset(self, dataset: str | None) -> str:
|
|
81
|
+
"""Return the BigQuery dataset to use.
|
|
82
|
+
|
|
83
|
+
Prefers *dataset* (the asset's). Falls back to
|
|
84
|
+
the destination's ``dataset`` field.
|
|
85
|
+
|
|
86
|
+
Args:
|
|
87
|
+
dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
The resolved dataset name.
|
|
91
|
+
|
|
92
|
+
Raises:
|
|
93
|
+
ConfigError: If neither the asset nor the destination names a dataset.
|
|
94
|
+
"""
|
|
95
|
+
ds = dataset or self.default_dataset
|
|
96
|
+
if ds is None:
|
|
97
|
+
raise ConfigError(
|
|
98
|
+
"BigQueryDestination requires a dataset. Either set 'dataset' on the asset "
|
|
99
|
+
"or provide 'default_dataset' on the destination."
|
|
100
|
+
)
|
|
101
|
+
return ds
|
|
102
|
+
|
|
103
|
+
def _table_ref(self, table: str, dataset: str | None) -> str:
|
|
104
|
+
"""Build a fully-qualified BigQuery table reference.
|
|
105
|
+
|
|
106
|
+
Args:
|
|
107
|
+
table: Table name.
|
|
108
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
``project.dataset.table`` string.
|
|
112
|
+
"""
|
|
113
|
+
ds = self._resolve_dataset(dataset)
|
|
114
|
+
return f"{self.project}.{ds}.{table}"
|
|
115
|
+
|
|
116
|
+
def _get_table(self, table: str, dataset: str | None) -> bigquery.Table | None:
|
|
117
|
+
"""Fetch a BigQuery table.
|
|
118
|
+
|
|
119
|
+
Args:
|
|
120
|
+
table: Table name.
|
|
121
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
122
|
+
|
|
123
|
+
Returns:
|
|
124
|
+
The table, or ``None`` if it does not exist.
|
|
125
|
+
"""
|
|
126
|
+
try:
|
|
127
|
+
return self.client.get_table(self._table_ref(table, dataset))
|
|
128
|
+
except NotFound:
|
|
129
|
+
return None
|
|
130
|
+
|
|
131
|
+
def _table_exists(self, table: str, dataset: str | None) -> bool:
|
|
132
|
+
"""Check whether a BigQuery table exists.
|
|
133
|
+
|
|
134
|
+
Args:
|
|
135
|
+
table: Table name.
|
|
136
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
137
|
+
|
|
138
|
+
Returns:
|
|
139
|
+
``True`` if the table exists, ``False`` otherwise.
|
|
140
|
+
"""
|
|
141
|
+
return self._get_table(table, dataset) is not None
|
|
142
|
+
|
|
143
|
+
def _create_table(
|
|
144
|
+
self,
|
|
145
|
+
table: str,
|
|
146
|
+
dataset: str | None,
|
|
147
|
+
bq_schema: list[bigquery.SchemaField],
|
|
148
|
+
time_partitioning: bigquery.TimePartitioning | None = None,
|
|
149
|
+
description: str | None = None,
|
|
150
|
+
) -> None:
|
|
151
|
+
"""Create a BigQuery table with an explicit schema, unless one just appeared.
|
|
152
|
+
|
|
153
|
+
Two runs writing different partitions of the same new table race on its
|
|
154
|
+
creation (a backfill, or a queue burst after downtime): whichever loses
|
|
155
|
+
gets a 409 from BigQuery, which is not an error for it, since the table
|
|
156
|
+
it wanted now exists with the same schema, so it loads into that one.
|
|
157
|
+
|
|
158
|
+
Args:
|
|
159
|
+
table: Target table name.
|
|
160
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
161
|
+
bq_schema: BigQuery field definitions.
|
|
162
|
+
time_partitioning: Time partitioning spec, if the asset is partitioned.
|
|
163
|
+
description: Table description (the asset's description).
|
|
164
|
+
"""
|
|
165
|
+
bq_table = bigquery.Table(self._table_ref(table, dataset), schema=bq_schema)
|
|
166
|
+
bq_table.time_partitioning = time_partitioning
|
|
167
|
+
bq_table.description = description
|
|
168
|
+
try:
|
|
169
|
+
self.client.create_table(bq_table)
|
|
170
|
+
except Conflict:
|
|
171
|
+
pass # Created by a concurrent run of the same asset
|
|
172
|
+
|
|
173
|
+
def _sync_table_metadata(
|
|
174
|
+
self,
|
|
175
|
+
bq_table: bigquery.Table,
|
|
176
|
+
bq_schema: list[bigquery.SchemaField] | None,
|
|
177
|
+
description: str | None,
|
|
178
|
+
) -> None:
|
|
179
|
+
"""Push field and table descriptions onto an existing table.
|
|
180
|
+
|
|
181
|
+
Keeps BigQuery metadata in sync when descriptions change on the asset
|
|
182
|
+
or its schema after the table was created. Only descriptions are
|
|
183
|
+
updated (types, modes, and partitioning are immutable here), empty
|
|
184
|
+
descriptions never clear existing ones, and no API call is made when
|
|
185
|
+
nothing changed.
|
|
186
|
+
|
|
187
|
+
Args:
|
|
188
|
+
bq_table: The existing table.
|
|
189
|
+
bq_schema: Field definitions carrying the wanted descriptions.
|
|
190
|
+
description: Wanted table description (the asset's description).
|
|
191
|
+
"""
|
|
192
|
+
update_fields = []
|
|
193
|
+
if bq_schema is not None:
|
|
194
|
+
merged, changed = _merge_field_descriptions(bq_table.schema, bq_schema)
|
|
195
|
+
if changed:
|
|
196
|
+
bq_table.schema = merged
|
|
197
|
+
update_fields.append("schema")
|
|
198
|
+
if description and bq_table.description != description:
|
|
199
|
+
bq_table.description = description
|
|
200
|
+
update_fields.append("description")
|
|
201
|
+
if update_fields:
|
|
202
|
+
self.client.update_table(bq_table, update_fields)
|
|
203
|
+
|
|
204
|
+
def _ensure_dataset(self, dataset: str | None) -> None:
|
|
205
|
+
"""Create the BigQuery dataset if it does not already exist.
|
|
206
|
+
|
|
207
|
+
Args:
|
|
208
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
209
|
+
"""
|
|
210
|
+
ds = self._resolve_dataset(dataset)
|
|
211
|
+
dataset_ref = bigquery.DatasetReference(self.project, ds)
|
|
212
|
+
try:
|
|
213
|
+
self.client.get_dataset(dataset_ref)
|
|
214
|
+
except NotFound:
|
|
215
|
+
bq_dataset = bigquery.Dataset(dataset_ref)
|
|
216
|
+
bq_dataset.location = self.client.location
|
|
217
|
+
try:
|
|
218
|
+
self.client.create_dataset(bq_dataset)
|
|
219
|
+
except Conflict:
|
|
220
|
+
pass # Created by a concurrent asset — already exists
|
|
221
|
+
|
|
222
|
+
# -- DatabaseDestination hooks ---------------------------------------------
|
|
223
|
+
|
|
224
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
225
|
+
"""Insert data into BigQuery through one Parquet load job, whatever its representation.
|
|
226
|
+
|
|
227
|
+
The data is converted to a DataFrame (``Representation.of(data).to("dataframe")``)
|
|
228
|
+
and loaded with ``load_table_from_dataframe``. The effective schema
|
|
229
|
+
from the IO context (declared on the asset, or inferred during
|
|
230
|
+
conform) drives both table DDL and the load job's schema, so the
|
|
231
|
+
table shape is deterministic and stable across partitions. Without a
|
|
232
|
+
schema, the load job creates the table from the frame's dtypes.
|
|
233
|
+
|
|
234
|
+
New tables are created with the asset's partitioning (time
|
|
235
|
+
partitioning on the partition column at the asset's granularity) and
|
|
236
|
+
carry field and table descriptions; on existing tables, descriptions
|
|
237
|
+
are kept in sync.
|
|
238
|
+
|
|
239
|
+
Args:
|
|
240
|
+
table: Target table name.
|
|
241
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
242
|
+
data: The data in its native representation.
|
|
243
|
+
context: IO context carrying the asset and effective schema.
|
|
244
|
+
"""
|
|
245
|
+
frame = Representation.of(data).to("dataframe")
|
|
246
|
+
bq_schema = to_fields(context.schema.field_specs()) if context.schema is not None else None
|
|
247
|
+
description = _asset_description(context.asset)
|
|
248
|
+
partitioning = context.asset.partitioning
|
|
249
|
+
|
|
250
|
+
bq_table = self._get_table(table, dataset)
|
|
251
|
+
time_partitioning = None
|
|
252
|
+
if bq_table is not None:
|
|
253
|
+
self._sync_table_metadata(bq_table, bq_schema, description)
|
|
254
|
+
elif bq_schema is not None:
|
|
255
|
+
self._ensure_dataset(dataset)
|
|
256
|
+
tp = _time_partitioning(partitioning, bq_schema)
|
|
257
|
+
self._create_table(table, dataset, bq_schema, time_partitioning=tp, description=description)
|
|
258
|
+
else:
|
|
259
|
+
self._ensure_dataset(dataset)
|
|
260
|
+
time_partitioning = _time_partitioning(partitioning, None)
|
|
261
|
+
|
|
262
|
+
self._load(self._table_ref(table, dataset), frame, bq_schema, time_partitioning=time_partitioning)
|
|
263
|
+
|
|
264
|
+
def _load(
|
|
265
|
+
self,
|
|
266
|
+
ref: str,
|
|
267
|
+
df: pd.DataFrame,
|
|
268
|
+
bq_schema: list[bigquery.SchemaField] | None,
|
|
269
|
+
time_partitioning: bigquery.TimePartitioning | None = None,
|
|
270
|
+
) -> None:
|
|
271
|
+
"""Load a DataFrame via a Parquet load job.
|
|
272
|
+
|
|
273
|
+
When a schema is available, columns are aligned to it: extra columns
|
|
274
|
+
are dropped (with a warning) and the load job receives explicit field
|
|
275
|
+
types, so pyarrow casts values (including ``NaN`` → ``NULL``) instead
|
|
276
|
+
of relying on dtype autodetection.
|
|
277
|
+
|
|
278
|
+
Args:
|
|
279
|
+
ref: Fully-qualified table reference.
|
|
280
|
+
df: The DataFrame to load.
|
|
281
|
+
bq_schema: BigQuery field definitions, or ``None`` to autodetect.
|
|
282
|
+
time_partitioning: Partitioning spec for the table the load job is
|
|
283
|
+
about to create; ``None`` when the table already exists.
|
|
284
|
+
"""
|
|
285
|
+
job_config = bigquery.LoadJobConfig(write_disposition=bigquery.WriteDisposition.WRITE_APPEND)
|
|
286
|
+
if time_partitioning is not None:
|
|
287
|
+
job_config.time_partitioning = time_partitioning
|
|
288
|
+
|
|
289
|
+
if bq_schema is not None:
|
|
290
|
+
schema_columns = [field.name for field in bq_schema]
|
|
291
|
+
extras = [str(c) for c in df.columns if str(c) not in schema_columns]
|
|
292
|
+
if extras:
|
|
293
|
+
warnings.warn(
|
|
294
|
+
f"Columns {extras} are not in the schema for '{ref}' and will not be written.",
|
|
295
|
+
UserWarning,
|
|
296
|
+
stacklevel=2,
|
|
297
|
+
)
|
|
298
|
+
present = [c for c in schema_columns if c in df.columns]
|
|
299
|
+
df = df[present]
|
|
300
|
+
job_config.schema = [field for field in bq_schema if field.name in present]
|
|
301
|
+
|
|
302
|
+
# The client deprecates this in favour of pandas_gbq.to_gbq(), which
|
|
303
|
+
# wraps this same call without exposing the job config we rely on.
|
|
304
|
+
with warnings.catch_warnings():
|
|
305
|
+
warnings.filterwarnings("ignore", category=PendingDeprecationWarning, module="google.cloud.bigquery")
|
|
306
|
+
job = self.client.load_table_from_dataframe(df, ref, job_config=job_config)
|
|
307
|
+
job.result()
|
|
308
|
+
|
|
309
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
310
|
+
"""Delete the rows a filter selects, or truncate the table.
|
|
311
|
+
|
|
312
|
+
A table that does not exist has nothing to delete.
|
|
313
|
+
|
|
314
|
+
Args:
|
|
315
|
+
table: Target table name.
|
|
316
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
317
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
318
|
+
"""
|
|
319
|
+
bq_table = self._get_table(table, dataset)
|
|
320
|
+
if bq_table is None:
|
|
321
|
+
return
|
|
322
|
+
ref = self._table_ref(table, dataset)
|
|
323
|
+
if where is None:
|
|
324
|
+
self.client.query(f"TRUNCATE TABLE `{ref}`").result()
|
|
325
|
+
return
|
|
326
|
+
predicate, parameters = _predicate(bq_table, where)
|
|
327
|
+
job_config = bigquery.QueryJobConfig(query_parameters=parameters)
|
|
328
|
+
self.client.query(f"DELETE FROM `{ref}` WHERE {predicate}", job_config=job_config).result()
|
|
329
|
+
|
|
330
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
|
|
331
|
+
"""Select the rows a filter selects, or every row, as a DataFrame.
|
|
332
|
+
|
|
333
|
+
The client builds the frame from the query's Arrow result, so column
|
|
334
|
+
types survive the read without a pass through Python records.
|
|
335
|
+
|
|
336
|
+
Args:
|
|
337
|
+
table: Target table name.
|
|
338
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
339
|
+
where: The rows to select; ``None`` for the whole table.
|
|
340
|
+
|
|
341
|
+
Returns:
|
|
342
|
+
The selected rows.
|
|
343
|
+
|
|
344
|
+
Raises:
|
|
345
|
+
DataNotFoundError: If the table does not exist yet.
|
|
346
|
+
"""
|
|
347
|
+
bq_table = self._get_table(table, dataset)
|
|
348
|
+
ref = self._table_ref(table, dataset)
|
|
349
|
+
if bq_table is None:
|
|
350
|
+
raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
|
|
351
|
+
if where is None:
|
|
352
|
+
return self.client.query(f"SELECT * FROM `{ref}`").result().to_dataframe()
|
|
353
|
+
predicate, parameters = _predicate(bq_table, where)
|
|
354
|
+
job_config = bigquery.QueryJobConfig(query_parameters=parameters)
|
|
355
|
+
query = self.client.query(f"SELECT * FROM `{ref}` WHERE {predicate}", job_config=job_config)
|
|
356
|
+
return query.result().to_dataframe()
|
|
357
|
+
|
|
358
|
+
# -- Introspection ---------------------------------------------------------
|
|
359
|
+
|
|
360
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
361
|
+
"""Return row counts grouped by partition column via BigQuery SQL.
|
|
362
|
+
|
|
363
|
+
Args:
|
|
364
|
+
table: Target table name.
|
|
365
|
+
dataset: The BigQuery dataset, or ``None`` for the destination's default.
|
|
366
|
+
column: Column to group by.
|
|
367
|
+
|
|
368
|
+
Returns:
|
|
369
|
+
Mapping from partition value (as string) to row count.
|
|
370
|
+
|
|
371
|
+
Raises:
|
|
372
|
+
DataNotFoundError: If the table does not exist.
|
|
373
|
+
"""
|
|
374
|
+
if not self._table_exists(table, dataset):
|
|
375
|
+
ref = self._table_ref(table, dataset)
|
|
376
|
+
raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
|
|
377
|
+
|
|
378
|
+
ref = self._table_ref(table, dataset)
|
|
379
|
+
query = f"SELECT CAST(`{column}` AS STRING) AS partition_value, COUNT(*) AS cnt FROM `{ref}` GROUP BY 1"
|
|
380
|
+
rows = self.client.query(query).result()
|
|
381
|
+
return {row["partition_value"]: row["cnt"] for row in rows}
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# -- Utility functions ---------------------------------------------------------
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _asset_description(asset: Any) -> str | None:
|
|
388
|
+
"""Return the asset's description (its class docstring), cleaned.
|
|
389
|
+
|
|
390
|
+
This mirrors how ``Component.definition()`` derives descriptions.
|
|
391
|
+
|
|
392
|
+
Args:
|
|
393
|
+
asset: The asset whose docstring is read.
|
|
394
|
+
|
|
395
|
+
Returns:
|
|
396
|
+
The cleaned docstring, or ``None`` when the asset has none.
|
|
397
|
+
"""
|
|
398
|
+
doc = type(asset).__doc__
|
|
399
|
+
return inspect.cleandoc(doc) if doc else None
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
# BigQuery time partitioning only supports DATE / DATETIME / TIMESTAMP columns.
|
|
403
|
+
_GRANULARITY_TO_BQ_TYPE = {
|
|
404
|
+
TimeGranularity.HOUR: bigquery.TimePartitioningType.HOUR,
|
|
405
|
+
TimeGranularity.DAY: bigquery.TimePartitioningType.DAY,
|
|
406
|
+
TimeGranularity.MONTH: bigquery.TimePartitioningType.MONTH,
|
|
407
|
+
TimeGranularity.YEAR: bigquery.TimePartitioningType.YEAR,
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def _time_partitioning(
|
|
412
|
+
config: PartitionConfig | None,
|
|
413
|
+
bq_schema: list[bigquery.SchemaField] | None,
|
|
414
|
+
) -> bigquery.TimePartitioning | None:
|
|
415
|
+
"""Resolve the asset's partition config into a BigQuery time partitioning spec.
|
|
416
|
+
|
|
417
|
+
Partitioning at the asset's declared granularity, when BigQuery supports
|
|
418
|
+
it: the column must be DATE / DATETIME / TIMESTAMP, and hourly requires a
|
|
419
|
+
sub-day type (BigQuery rejects HOUR partitioning on a DATE column). When
|
|
420
|
+
the column type is known (from the schema) and does not support the
|
|
421
|
+
granularity, or the config is not time-based and no schema is available,
|
|
422
|
+
returns ``None`` — the table is created unpartitioned, which is always
|
|
423
|
+
valid.
|
|
424
|
+
|
|
425
|
+
Args:
|
|
426
|
+
config: The asset's partition config, if any.
|
|
427
|
+
bq_schema: The table's field definitions, when known.
|
|
428
|
+
|
|
429
|
+
Returns:
|
|
430
|
+
A ``TimePartitioning`` at the asset's granularity, or ``None``.
|
|
431
|
+
"""
|
|
432
|
+
if config is None:
|
|
433
|
+
return None
|
|
434
|
+
|
|
435
|
+
granularity = config.granularity if isinstance(config, TimePartitionConfig) else TimeGranularity.DAY
|
|
436
|
+
if bq_schema is not None:
|
|
437
|
+
field = next((f for f in bq_schema if f.name == config.column), None)
|
|
438
|
+
unsupported = (
|
|
439
|
+
field is None
|
|
440
|
+
or field.field_type not in TIME_PARTITIONABLE
|
|
441
|
+
or (granularity is TimeGranularity.HOUR and field.field_type == "DATE")
|
|
442
|
+
)
|
|
443
|
+
if unsupported:
|
|
444
|
+
if isinstance(config, TimePartitionConfig):
|
|
445
|
+
reason = (
|
|
446
|
+
"missing from the schema"
|
|
447
|
+
if field is None
|
|
448
|
+
else f"of type {field.field_type}"
|
|
449
|
+
+ (" (hourly partitioning needs DATETIME or TIMESTAMP)" if field.field_type == "DATE" else "")
|
|
450
|
+
)
|
|
451
|
+
warnings.warn(
|
|
452
|
+
f"Partition column '{config.column}' is {reason}; "
|
|
453
|
+
"the BigQuery table will not be time-partitioned.",
|
|
454
|
+
UserWarning,
|
|
455
|
+
stacklevel=2,
|
|
456
|
+
)
|
|
457
|
+
return None
|
|
458
|
+
elif not isinstance(config, TimePartitionConfig):
|
|
459
|
+
return None
|
|
460
|
+
|
|
461
|
+
return bigquery.TimePartitioning(type_=_GRANULARITY_TO_BQ_TYPE[granularity], field=config.column)
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def _merge_field_descriptions(
|
|
465
|
+
existing: Sequence[bigquery.SchemaField],
|
|
466
|
+
desired: Sequence[bigquery.SchemaField],
|
|
467
|
+
) -> tuple[list[bigquery.SchemaField], bool]:
|
|
468
|
+
"""Overlay desired field descriptions onto an existing table schema.
|
|
469
|
+
|
|
470
|
+
Only descriptions are touched — types and modes stay as they are in the
|
|
471
|
+
table, so the result is always a valid ``update_table`` schema. Empty
|
|
472
|
+
desired descriptions never clear an existing one (e.g. set manually in
|
|
473
|
+
the BigQuery console).
|
|
474
|
+
|
|
475
|
+
Args:
|
|
476
|
+
existing: The table's current field definitions.
|
|
477
|
+
desired: Field definitions carrying the wanted descriptions.
|
|
478
|
+
|
|
479
|
+
Returns:
|
|
480
|
+
The merged field list and whether anything changed.
|
|
481
|
+
"""
|
|
482
|
+
desired_by_name = {f.name: f for f in desired}
|
|
483
|
+
merged: list[bigquery.SchemaField] = []
|
|
484
|
+
changed = False
|
|
485
|
+
for field in existing:
|
|
486
|
+
want = desired_by_name.get(field.name)
|
|
487
|
+
if want is None:
|
|
488
|
+
merged.append(field)
|
|
489
|
+
continue
|
|
490
|
+
api = field.to_api_repr()
|
|
491
|
+
if want.description and want.description != field.description:
|
|
492
|
+
api["description"] = want.description
|
|
493
|
+
changed = True
|
|
494
|
+
if field.field_type == "RECORD" and want.fields:
|
|
495
|
+
sub_fields, sub_changed = _merge_field_descriptions(field.fields, want.fields)
|
|
496
|
+
if sub_changed:
|
|
497
|
+
api["fields"] = [f.to_api_repr() for f in sub_fields]
|
|
498
|
+
changed = True
|
|
499
|
+
merged.append(bigquery.SchemaField.from_api_repr(api))
|
|
500
|
+
return merged, changed
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
# BigQuery column type -> query parameter type, for partition predicates.
|
|
504
|
+
def _predicate(bq_table: bigquery.Table, where: PartitionFilter) -> tuple[str, list[bigquery.ScalarQueryParameter]]:
|
|
505
|
+
"""Render a partition filter as a parameterised SQL predicate.
|
|
506
|
+
|
|
507
|
+
Args:
|
|
508
|
+
bq_table: The target table, for the column's type.
|
|
509
|
+
where: The filter to render.
|
|
510
|
+
|
|
511
|
+
Returns:
|
|
512
|
+
The predicate text and the typed parameters it names.
|
|
513
|
+
"""
|
|
514
|
+
column = where.column
|
|
515
|
+
if where.bounds is None:
|
|
516
|
+
return f"`{column}` = @partition_value", [_partition_param(bq_table, column, where.value)]
|
|
517
|
+
start, end = where.bounds
|
|
518
|
+
return (
|
|
519
|
+
f"`{column}` >= @partition_start AND `{column}` < @partition_end",
|
|
520
|
+
[
|
|
521
|
+
_partition_param(bq_table, column, start, name="partition_start"),
|
|
522
|
+
_partition_param(bq_table, column, end, name="partition_end"),
|
|
523
|
+
],
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _partition_param(
|
|
528
|
+
bq_table: bigquery.Table,
|
|
529
|
+
column: str,
|
|
530
|
+
value: Any,
|
|
531
|
+
name: str = "partition_value",
|
|
532
|
+
) -> bigquery.ScalarQueryParameter:
|
|
533
|
+
"""Build a partition-predicate query parameter, typed from the table.
|
|
534
|
+
|
|
535
|
+
Partition values arrive as strings (``Partition.id``) or as the period
|
|
536
|
+
bounds of a time partition (dates/datetimes), but the column may be
|
|
537
|
+
DATE / TIMESTAMP / INTEGER — BigQuery does not coerce a STRING parameter
|
|
538
|
+
in a comparison predicate, so the parameter type must match the actual
|
|
539
|
+
column type. Falls back to inferring from the value when the column is
|
|
540
|
+
not in the table schema.
|
|
541
|
+
|
|
542
|
+
Args:
|
|
543
|
+
bq_table: The target table (for column type lookup).
|
|
544
|
+
column: Partition column name.
|
|
545
|
+
value: Partition value or bound.
|
|
546
|
+
name: The query parameter's name.
|
|
547
|
+
|
|
548
|
+
Returns:
|
|
549
|
+
A scalar parameter with the matching type.
|
|
550
|
+
"""
|
|
551
|
+
field = next((f for f in bq_table.schema if f.name == column), None)
|
|
552
|
+
if field is not None:
|
|
553
|
+
param_type = parameter_type(field.field_type)
|
|
554
|
+
else:
|
|
555
|
+
param_type = column_type(type(value))
|
|
556
|
+
if param_type in ("TIMESTAMP", "DATETIME") and isinstance(value, str):
|
|
557
|
+
# The client serializes these from datetime objects; a date-only
|
|
558
|
+
# string like "2024-01-01" (a TimePartition id) fails to format.
|
|
559
|
+
try:
|
|
560
|
+
value = datetime.datetime.fromisoformat(value)
|
|
561
|
+
except ValueError:
|
|
562
|
+
pass
|
|
563
|
+
if param_type in ("TIMESTAMP", "DATETIME") and isinstance(value, datetime.date) and not isinstance(
|
|
564
|
+
value, datetime.datetime
|
|
565
|
+
):
|
|
566
|
+
# A date bound against a datetime-typed column: promote to midnight so
|
|
567
|
+
# the client serializes it in the parameter's declared type.
|
|
568
|
+
value = datetime.datetime(value.year, value.month, value.day)
|
|
569
|
+
if param_type == "DATE" and isinstance(value, datetime.datetime):
|
|
570
|
+
# A datetime bound against a DATE column loses sub-day precision by
|
|
571
|
+
# definition; the caller guards hourly-on-DATE at table creation.
|
|
572
|
+
value = value.date()
|
|
573
|
+
if param_type == "STRING" and not isinstance(value, str):
|
|
574
|
+
value = _iso_string(value)
|
|
575
|
+
return bigquery.ScalarQueryParameter(name, param_type, value)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def _iso_string(value: Any) -> str:
|
|
579
|
+
"""Render a bound as an ISO-8601 string for STRING-typed columns.
|
|
580
|
+
|
|
581
|
+
Args:
|
|
582
|
+
value: The bound to render.
|
|
583
|
+
|
|
584
|
+
Returns:
|
|
585
|
+
The ISO rendering of *value*.
|
|
586
|
+
"""
|
|
587
|
+
return value.isoformat() if isinstance(value, (datetime.date, datetime.datetime)) else str(value)
|
|
588
|
+
|