interloper-google-cloud 0.76.0__tar.gz → 0.78.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: interloper-google-cloud
3
- Version: 0.76.0
3
+ Version: 0.78.0
4
4
  Summary: Interloper Google Cloud integration: BigQuery and Cloud Storage destinations
5
5
  Author: Guillaume Onfroy
6
6
  Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
@@ -10,6 +10,7 @@ Requires-Dist: interloper-core
10
10
  Requires-Dist: interloper-pandas
11
11
  Requires-Dist: pyarrow>=14
12
12
  Requires-Dist: db-dtypes>=1.0
13
+ Requires-Dist: pandas-gbq>=0.26.1
13
14
  Requires-Python: >=3.10
14
15
  Description-Content-Type: text/markdown
15
16
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "interloper-google-cloud"
3
- version = "0.76.0"
3
+ version = "0.78.0"
4
4
  description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -11,6 +11,7 @@ dependencies = [
11
11
  "interloper-pandas",
12
12
  "pyarrow>=14",
13
13
  "db-dtypes>=1.0",
14
+ "pandas-gbq>=0.26.1",
14
15
  ]
15
16
 
16
17
  [[project.authors]]
@@ -21,7 +22,7 @@ email = "guillaume@digitlcloud.com"
21
22
  google_cloud = "interloper_google_cloud"
22
23
 
23
24
  [build-system]
24
- requires = ["uv_build>=0.11.5,<0.12"]
25
+ requires = ["uv_build>=0.12.9,<0.13"]
25
26
  build-backend = "uv_build"
26
27
 
27
28
  [tool.uv.sources.interloper-core]
@@ -61,4 +62,6 @@ convention = "google"
61
62
  "D102",
62
63
  "D103",
63
64
  "D104",
65
+ "RUF069",
66
+ "PLW0108",
64
67
  ]
@@ -3,7 +3,7 @@
3
3
  # ###############
4
4
  [project]
5
5
  name = "interloper-google-cloud"
6
- version = "0.76.0"
6
+ version = "0.78.0"
7
7
  description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
8
8
  readme = "README.md"
9
9
  authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
@@ -15,13 +15,14 @@ dependencies = [
15
15
  "interloper-pandas",
16
16
  "pyarrow>=14",
17
17
  "db-dtypes>=1.0",
18
+ "pandas-gbq>=0.26.1",
18
19
  ]
19
20
 
20
21
  [project.entry-points."interloper.components"]
21
22
  google_cloud = "interloper_google_cloud"
22
23
 
23
24
  [build-system]
24
- requires = ["uv_build>=0.11.5,<0.12"]
25
+ requires = ["uv_build>=0.12.9,<0.13"]
25
26
  build-backend = "uv_build"
26
27
 
27
28
  [tool.uv.sources]
@@ -43,4 +44,4 @@ convention = "google"
43
44
 
44
45
  [tool.ruff.lint.per-file-ignores]
45
46
  "__init__.py" = ["F401", "F403"]
46
- "tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104"]
47
+ "tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
@@ -0,0 +1,588 @@
1
+ """BigQuery destination implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ import inspect
7
+ import json
8
+ import warnings
9
+ from collections.abc import Sequence
10
+ from functools import cached_property
11
+ from typing import Any
12
+
13
+ import google.auth
14
+ import pandas as pd
15
+ from google.cloud import bigquery
16
+ from google.cloud.exceptions import Conflict, NotFound
17
+ from google.oauth2 import service_account
18
+ from interloper.destination import IOContext, destination
19
+ from interloper.destination.database import DatabaseDestination, PartitionFilter
20
+ from interloper.errors import ConfigError, DataNotFoundError
21
+ from interloper.partitioning import PartitionConfig, TimeGranularity, TimePartitionConfig
22
+ from interloper.representation import Representation
23
+ from interloper.resource.fields import FetchField, InputField, SelectField
24
+
25
+ from interloper_google_cloud.bigquery.types import TIME_PARTITIONABLE, column_type, parameter_type, to_fields
26
+ from interloper_google_cloud.connection import GoogleCloudConnection
27
+
28
+
29
+ @destination(
30
+ key="bigquery_destination",
31
+ name="BigQuery",
32
+ icon="icon:bigquery",
33
+ tags=["Cloud"],
34
+ )
35
+ class BigQueryDestination(DatabaseDestination):
36
+ """BigQuery destination."""
37
+
38
+ connection: GoogleCloudConnection
39
+
40
+ project: str = FetchField(
41
+ provider="connection.projects",
42
+ label_key="name",
43
+ value_key="project_id",
44
+ description="Google Cloud project ID",
45
+ discriminator=True,
46
+ )
47
+ location: str = SelectField(
48
+ description="BigQuery dataset location",
49
+ options=[
50
+ {"label": "EU", "value": "EU"},
51
+ {"label": "US", "value": "US"},
52
+ ],
53
+ )
54
+ default_dataset: str | None = InputField(default=None, description="Default BigQuery dataset")
55
+
56
+ @cached_property
57
+ def client(self) -> bigquery.Client:
58
+ """The BigQuery client every write goes through.
59
+
60
+ Built from the connection's service-account key when there is one,
61
+ else from ambient credentials (workload identity in-cluster).
62
+
63
+ Returns:
64
+ The client, cached per destination instance.
65
+ """
66
+ if self.connection and self.connection.service_account_key:
67
+ key_info = json.loads(self.connection.service_account_key)
68
+ credentials = service_account.Credentials.from_service_account_info(key_info)
69
+ else:
70
+ credentials, _ = google.auth.default()
71
+
72
+ return bigquery.Client(
73
+ project=self.project,
74
+ credentials=credentials,
75
+ location=self.location,
76
+ )
77
+
78
+ # -- Helpers ---------------------------------------------------------------
79
+
80
+ def _resolve_dataset(self, dataset: str | None) -> str:
81
+ """Return the BigQuery dataset to use.
82
+
83
+ Prefers *dataset* (the asset's). Falls back to
84
+ the destination's ``dataset`` field.
85
+
86
+ Args:
87
+ dataset: The asset's dataset, or ``None`` to fall back to the destination's default.
88
+
89
+ Returns:
90
+ The resolved dataset name.
91
+
92
+ Raises:
93
+ ConfigError: If neither the asset nor the destination names a dataset.
94
+ """
95
+ ds = dataset or self.default_dataset
96
+ if ds is None:
97
+ raise ConfigError(
98
+ "BigQueryDestination requires a dataset. Either set 'dataset' on the asset "
99
+ "or provide 'default_dataset' on the destination."
100
+ )
101
+ return ds
102
+
103
+ def _table_ref(self, table: str, dataset: str | None) -> str:
104
+ """Build a fully-qualified BigQuery table reference.
105
+
106
+ Args:
107
+ table: Table name.
108
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
109
+
110
+ Returns:
111
+ ``project.dataset.table`` string.
112
+ """
113
+ ds = self._resolve_dataset(dataset)
114
+ return f"{self.project}.{ds}.{table}"
115
+
116
+ def _get_table(self, table: str, dataset: str | None) -> bigquery.Table | None:
117
+ """Fetch a BigQuery table.
118
+
119
+ Args:
120
+ table: Table name.
121
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
122
+
123
+ Returns:
124
+ The table, or ``None`` if it does not exist.
125
+ """
126
+ try:
127
+ return self.client.get_table(self._table_ref(table, dataset))
128
+ except NotFound:
129
+ return None
130
+
131
+ def _table_exists(self, table: str, dataset: str | None) -> bool:
132
+ """Check whether a BigQuery table exists.
133
+
134
+ Args:
135
+ table: Table name.
136
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
137
+
138
+ Returns:
139
+ ``True`` if the table exists, ``False`` otherwise.
140
+ """
141
+ return self._get_table(table, dataset) is not None
142
+
143
+ def _create_table(
144
+ self,
145
+ table: str,
146
+ dataset: str | None,
147
+ bq_schema: list[bigquery.SchemaField],
148
+ time_partitioning: bigquery.TimePartitioning | None = None,
149
+ description: str | None = None,
150
+ ) -> None:
151
+ """Create a BigQuery table with an explicit schema, unless one just appeared.
152
+
153
+ Two runs writing different partitions of the same new table race on its
154
+ creation (a backfill, or a queue burst after downtime): whichever loses
155
+ gets a 409 from BigQuery, which is not an error for it, since the table
156
+ it wanted now exists with the same schema, so it loads into that one.
157
+
158
+ Args:
159
+ table: Target table name.
160
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
161
+ bq_schema: BigQuery field definitions.
162
+ time_partitioning: Time partitioning spec, if the asset is partitioned.
163
+ description: Table description (the asset's description).
164
+ """
165
+ bq_table = bigquery.Table(self._table_ref(table, dataset), schema=bq_schema)
166
+ bq_table.time_partitioning = time_partitioning
167
+ bq_table.description = description
168
+ try:
169
+ self.client.create_table(bq_table)
170
+ except Conflict:
171
+ pass # Created by a concurrent run of the same asset
172
+
173
+ def _sync_table_metadata(
174
+ self,
175
+ bq_table: bigquery.Table,
176
+ bq_schema: list[bigquery.SchemaField] | None,
177
+ description: str | None,
178
+ ) -> None:
179
+ """Push field and table descriptions onto an existing table.
180
+
181
+ Keeps BigQuery metadata in sync when descriptions change on the asset
182
+ or its schema after the table was created. Only descriptions are
183
+ updated (types, modes, and partitioning are immutable here), empty
184
+ descriptions never clear existing ones, and no API call is made when
185
+ nothing changed.
186
+
187
+ Args:
188
+ bq_table: The existing table.
189
+ bq_schema: Field definitions carrying the wanted descriptions.
190
+ description: Wanted table description (the asset's description).
191
+ """
192
+ update_fields = []
193
+ if bq_schema is not None:
194
+ merged, changed = _merge_field_descriptions(bq_table.schema, bq_schema)
195
+ if changed:
196
+ bq_table.schema = merged
197
+ update_fields.append("schema")
198
+ if description and bq_table.description != description:
199
+ bq_table.description = description
200
+ update_fields.append("description")
201
+ if update_fields:
202
+ self.client.update_table(bq_table, update_fields)
203
+
204
+ def _ensure_dataset(self, dataset: str | None) -> None:
205
+ """Create the BigQuery dataset if it does not already exist.
206
+
207
+ Args:
208
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
209
+ """
210
+ ds = self._resolve_dataset(dataset)
211
+ dataset_ref = bigquery.DatasetReference(self.project, ds)
212
+ try:
213
+ self.client.get_dataset(dataset_ref)
214
+ except NotFound:
215
+ bq_dataset = bigquery.Dataset(dataset_ref)
216
+ bq_dataset.location = self.client.location
217
+ try:
218
+ self.client.create_dataset(bq_dataset)
219
+ except Conflict:
220
+ pass # Created by a concurrent asset — already exists
221
+
222
+ # -- DatabaseDestination hooks ---------------------------------------------
223
+
224
+ def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
225
+ """Insert data into BigQuery through one Parquet load job, whatever its representation.
226
+
227
+ The data is converted to a DataFrame (``Representation.of(data).to("dataframe")``)
228
+ and loaded with ``load_table_from_dataframe``. The effective schema
229
+ from the IO context (declared on the asset, or inferred during
230
+ conform) drives both table DDL and the load job's schema, so the
231
+ table shape is deterministic and stable across partitions. Without a
232
+ schema, the load job creates the table from the frame's dtypes.
233
+
234
+ New tables are created with the asset's partitioning (time
235
+ partitioning on the partition column at the asset's granularity) and
236
+ carry field and table descriptions; on existing tables, descriptions
237
+ are kept in sync.
238
+
239
+ Args:
240
+ table: Target table name.
241
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
242
+ data: The data in its native representation.
243
+ context: IO context carrying the asset and effective schema.
244
+ """
245
+ frame = Representation.of(data).to("dataframe")
246
+ bq_schema = to_fields(context.schema.field_specs()) if context.schema is not None else None
247
+ description = _asset_description(context.asset)
248
+ partitioning = context.asset.partitioning
249
+
250
+ bq_table = self._get_table(table, dataset)
251
+ time_partitioning = None
252
+ if bq_table is not None:
253
+ self._sync_table_metadata(bq_table, bq_schema, description)
254
+ elif bq_schema is not None:
255
+ self._ensure_dataset(dataset)
256
+ tp = _time_partitioning(partitioning, bq_schema)
257
+ self._create_table(table, dataset, bq_schema, time_partitioning=tp, description=description)
258
+ else:
259
+ self._ensure_dataset(dataset)
260
+ time_partitioning = _time_partitioning(partitioning, None)
261
+
262
+ self._load(self._table_ref(table, dataset), frame, bq_schema, time_partitioning=time_partitioning)
263
+
264
+ def _load(
265
+ self,
266
+ ref: str,
267
+ df: pd.DataFrame,
268
+ bq_schema: list[bigquery.SchemaField] | None,
269
+ time_partitioning: bigquery.TimePartitioning | None = None,
270
+ ) -> None:
271
+ """Load a DataFrame via a Parquet load job.
272
+
273
+ When a schema is available, columns are aligned to it: extra columns
274
+ are dropped (with a warning) and the load job receives explicit field
275
+ types, so pyarrow casts values (including ``NaN`` → ``NULL``) instead
276
+ of relying on dtype autodetection.
277
+
278
+ Args:
279
+ ref: Fully-qualified table reference.
280
+ df: The DataFrame to load.
281
+ bq_schema: BigQuery field definitions, or ``None`` to autodetect.
282
+ time_partitioning: Partitioning spec for the table the load job is
283
+ about to create; ``None`` when the table already exists.
284
+ """
285
+ job_config = bigquery.LoadJobConfig(write_disposition=bigquery.WriteDisposition.WRITE_APPEND)
286
+ if time_partitioning is not None:
287
+ job_config.time_partitioning = time_partitioning
288
+
289
+ if bq_schema is not None:
290
+ schema_columns = [field.name for field in bq_schema]
291
+ extras = [str(c) for c in df.columns if str(c) not in schema_columns]
292
+ if extras:
293
+ warnings.warn(
294
+ f"Columns {extras} are not in the schema for '{ref}' and will not be written.",
295
+ UserWarning,
296
+ stacklevel=2,
297
+ )
298
+ present = [c for c in schema_columns if c in df.columns]
299
+ df = df[present]
300
+ job_config.schema = [field for field in bq_schema if field.name in present]
301
+
302
+ # The client deprecates this in favour of pandas_gbq.to_gbq(), which
303
+ # wraps this same call without exposing the job config we rely on.
304
+ with warnings.catch_warnings():
305
+ warnings.filterwarnings("ignore", category=PendingDeprecationWarning, module="google.cloud.bigquery")
306
+ job = self.client.load_table_from_dataframe(df, ref, job_config=job_config)
307
+ job.result()
308
+
309
+ def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
310
+ """Delete the rows a filter selects, or truncate the table.
311
+
312
+ A table that does not exist has nothing to delete.
313
+
314
+ Args:
315
+ table: Target table name.
316
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
317
+ where: The rows to delete; ``None`` for the whole table.
318
+ """
319
+ bq_table = self._get_table(table, dataset)
320
+ if bq_table is None:
321
+ return
322
+ ref = self._table_ref(table, dataset)
323
+ if where is None:
324
+ self.client.query(f"TRUNCATE TABLE `{ref}`").result()
325
+ return
326
+ predicate, parameters = _predicate(bq_table, where)
327
+ job_config = bigquery.QueryJobConfig(query_parameters=parameters)
328
+ self.client.query(f"DELETE FROM `{ref}` WHERE {predicate}", job_config=job_config).result()
329
+
330
+ def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> pd.DataFrame:
331
+ """Select the rows a filter selects, or every row, as a DataFrame.
332
+
333
+ The client builds the frame from the query's Arrow result, so column
334
+ types survive the read without a pass through Python records.
335
+
336
+ Args:
337
+ table: Target table name.
338
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
339
+ where: The rows to select; ``None`` for the whole table.
340
+
341
+ Returns:
342
+ The selected rows.
343
+
344
+ Raises:
345
+ DataNotFoundError: If the table does not exist yet.
346
+ """
347
+ bq_table = self._get_table(table, dataset)
348
+ ref = self._table_ref(table, dataset)
349
+ if bq_table is None:
350
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
351
+ if where is None:
352
+ return self.client.query(f"SELECT * FROM `{ref}`").result().to_dataframe()
353
+ predicate, parameters = _predicate(bq_table, where)
354
+ job_config = bigquery.QueryJobConfig(query_parameters=parameters)
355
+ query = self.client.query(f"SELECT * FROM `{ref}` WHERE {predicate}", job_config=job_config)
356
+ return query.result().to_dataframe()
357
+
358
+ # -- Introspection ---------------------------------------------------------
359
+
360
+ def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
361
+ """Return row counts grouped by partition column via BigQuery SQL.
362
+
363
+ Args:
364
+ table: Target table name.
365
+ dataset: The BigQuery dataset, or ``None`` for the destination's default.
366
+ column: Column to group by.
367
+
368
+ Returns:
369
+ Mapping from partition value (as string) to row count.
370
+
371
+ Raises:
372
+ DataNotFoundError: If the table does not exist.
373
+ """
374
+ if not self._table_exists(table, dataset):
375
+ ref = self._table_ref(table, dataset)
376
+ raise DataNotFoundError(f"Table '{ref}' does not exist. Has the asset been materialized?")
377
+
378
+ ref = self._table_ref(table, dataset)
379
+ query = f"SELECT CAST(`{column}` AS STRING) AS partition_value, COUNT(*) AS cnt FROM `{ref}` GROUP BY 1"
380
+ rows = self.client.query(query).result()
381
+ return {row["partition_value"]: row["cnt"] for row in rows}
382
+
383
+
384
+ # -- Utility functions ---------------------------------------------------------
385
+
386
+
387
+ def _asset_description(asset: Any) -> str | None:
388
+ """Return the asset's description (its class docstring), cleaned.
389
+
390
+ This mirrors how ``Component.definition()`` derives descriptions.
391
+
392
+ Args:
393
+ asset: The asset whose docstring is read.
394
+
395
+ Returns:
396
+ The cleaned docstring, or ``None`` when the asset has none.
397
+ """
398
+ doc = type(asset).__doc__
399
+ return inspect.cleandoc(doc) if doc else None
400
+
401
+
402
+ # BigQuery time partitioning only supports DATE / DATETIME / TIMESTAMP columns.
403
+ _GRANULARITY_TO_BQ_TYPE = {
404
+ TimeGranularity.HOUR: bigquery.TimePartitioningType.HOUR,
405
+ TimeGranularity.DAY: bigquery.TimePartitioningType.DAY,
406
+ TimeGranularity.MONTH: bigquery.TimePartitioningType.MONTH,
407
+ TimeGranularity.YEAR: bigquery.TimePartitioningType.YEAR,
408
+ }
409
+
410
+
411
+ def _time_partitioning(
412
+ config: PartitionConfig | None,
413
+ bq_schema: list[bigquery.SchemaField] | None,
414
+ ) -> bigquery.TimePartitioning | None:
415
+ """Resolve the asset's partition config into a BigQuery time partitioning spec.
416
+
417
+ Partitioning at the asset's declared granularity, when BigQuery supports
418
+ it: the column must be DATE / DATETIME / TIMESTAMP, and hourly requires a
419
+ sub-day type (BigQuery rejects HOUR partitioning on a DATE column). When
420
+ the column type is known (from the schema) and does not support the
421
+ granularity, or the config is not time-based and no schema is available,
422
+ returns ``None`` — the table is created unpartitioned, which is always
423
+ valid.
424
+
425
+ Args:
426
+ config: The asset's partition config, if any.
427
+ bq_schema: The table's field definitions, when known.
428
+
429
+ Returns:
430
+ A ``TimePartitioning`` at the asset's granularity, or ``None``.
431
+ """
432
+ if config is None:
433
+ return None
434
+
435
+ granularity = config.granularity if isinstance(config, TimePartitionConfig) else TimeGranularity.DAY
436
+ if bq_schema is not None:
437
+ field = next((f for f in bq_schema if f.name == config.column), None)
438
+ unsupported = (
439
+ field is None
440
+ or field.field_type not in TIME_PARTITIONABLE
441
+ or (granularity is TimeGranularity.HOUR and field.field_type == "DATE")
442
+ )
443
+ if unsupported:
444
+ if isinstance(config, TimePartitionConfig):
445
+ reason = (
446
+ "missing from the schema"
447
+ if field is None
448
+ else f"of type {field.field_type}"
449
+ + (" (hourly partitioning needs DATETIME or TIMESTAMP)" if field.field_type == "DATE" else "")
450
+ )
451
+ warnings.warn(
452
+ f"Partition column '{config.column}' is {reason}; "
453
+ "the BigQuery table will not be time-partitioned.",
454
+ UserWarning,
455
+ stacklevel=2,
456
+ )
457
+ return None
458
+ elif not isinstance(config, TimePartitionConfig):
459
+ return None
460
+
461
+ return bigquery.TimePartitioning(type_=_GRANULARITY_TO_BQ_TYPE[granularity], field=config.column)
462
+
463
+
464
+ def _merge_field_descriptions(
465
+ existing: Sequence[bigquery.SchemaField],
466
+ desired: Sequence[bigquery.SchemaField],
467
+ ) -> tuple[list[bigquery.SchemaField], bool]:
468
+ """Overlay desired field descriptions onto an existing table schema.
469
+
470
+ Only descriptions are touched — types and modes stay as they are in the
471
+ table, so the result is always a valid ``update_table`` schema. Empty
472
+ desired descriptions never clear an existing one (e.g. set manually in
473
+ the BigQuery console).
474
+
475
+ Args:
476
+ existing: The table's current field definitions.
477
+ desired: Field definitions carrying the wanted descriptions.
478
+
479
+ Returns:
480
+ The merged field list and whether anything changed.
481
+ """
482
+ desired_by_name = {f.name: f for f in desired}
483
+ merged: list[bigquery.SchemaField] = []
484
+ changed = False
485
+ for field in existing:
486
+ want = desired_by_name.get(field.name)
487
+ if want is None:
488
+ merged.append(field)
489
+ continue
490
+ api = field.to_api_repr()
491
+ if want.description and want.description != field.description:
492
+ api["description"] = want.description
493
+ changed = True
494
+ if field.field_type == "RECORD" and want.fields:
495
+ sub_fields, sub_changed = _merge_field_descriptions(field.fields, want.fields)
496
+ if sub_changed:
497
+ api["fields"] = [f.to_api_repr() for f in sub_fields]
498
+ changed = True
499
+ merged.append(bigquery.SchemaField.from_api_repr(api))
500
+ return merged, changed
501
+
502
+
503
+ # BigQuery column type -> query parameter type, for partition predicates.
504
+ def _predicate(bq_table: bigquery.Table, where: PartitionFilter) -> tuple[str, list[bigquery.ScalarQueryParameter]]:
505
+ """Render a partition filter as a parameterised SQL predicate.
506
+
507
+ Args:
508
+ bq_table: The target table, for the column's type.
509
+ where: The filter to render.
510
+
511
+ Returns:
512
+ The predicate text and the typed parameters it names.
513
+ """
514
+ column = where.column
515
+ if where.bounds is None:
516
+ return f"`{column}` = @partition_value", [_partition_param(bq_table, column, where.value)]
517
+ start, end = where.bounds
518
+ return (
519
+ f"`{column}` >= @partition_start AND `{column}` < @partition_end",
520
+ [
521
+ _partition_param(bq_table, column, start, name="partition_start"),
522
+ _partition_param(bq_table, column, end, name="partition_end"),
523
+ ],
524
+ )
525
+
526
+
527
+ def _partition_param(
528
+ bq_table: bigquery.Table,
529
+ column: str,
530
+ value: Any,
531
+ name: str = "partition_value",
532
+ ) -> bigquery.ScalarQueryParameter:
533
+ """Build a partition-predicate query parameter, typed from the table.
534
+
535
+ Partition values arrive as strings (``Partition.id``) or as the period
536
+ bounds of a time partition (dates/datetimes), but the column may be
537
+ DATE / TIMESTAMP / INTEGER — BigQuery does not coerce a STRING parameter
538
+ in a comparison predicate, so the parameter type must match the actual
539
+ column type. Falls back to inferring from the value when the column is
540
+ not in the table schema.
541
+
542
+ Args:
543
+ bq_table: The target table (for column type lookup).
544
+ column: Partition column name.
545
+ value: Partition value or bound.
546
+ name: The query parameter's name.
547
+
548
+ Returns:
549
+ A scalar parameter with the matching type.
550
+ """
551
+ field = next((f for f in bq_table.schema if f.name == column), None)
552
+ if field is not None:
553
+ param_type = parameter_type(field.field_type)
554
+ else:
555
+ param_type = column_type(type(value))
556
+ if param_type in ("TIMESTAMP", "DATETIME") and isinstance(value, str):
557
+ # The client serializes these from datetime objects; a date-only
558
+ # string like "2024-01-01" (a TimePartition id) fails to format.
559
+ try:
560
+ value = datetime.datetime.fromisoformat(value)
561
+ except ValueError:
562
+ pass
563
+ if param_type in ("TIMESTAMP", "DATETIME") and isinstance(value, datetime.date) and not isinstance(
564
+ value, datetime.datetime
565
+ ):
566
+ # A date bound against a datetime-typed column: promote to midnight so
567
+ # the client serializes it in the parameter's declared type.
568
+ value = datetime.datetime(value.year, value.month, value.day)
569
+ if param_type == "DATE" and isinstance(value, datetime.datetime):
570
+ # A datetime bound against a DATE column loses sub-day precision by
571
+ # definition; the caller guards hourly-on-DATE at table creation.
572
+ value = value.date()
573
+ if param_type == "STRING" and not isinstance(value, str):
574
+ value = _iso_string(value)
575
+ return bigquery.ScalarQueryParameter(name, param_type, value)
576
+
577
+
578
+ def _iso_string(value: Any) -> str:
579
+ """Render a bound as an ISO-8601 string for STRING-typed columns.
580
+
581
+ Args:
582
+ value: The bound to render.
583
+
584
+ Returns:
585
+ The ISO rendering of *value*.
586
+ """
587
+ return value.isoformat() if isinstance(value, (datetime.date, datetime.datetime)) else str(value)
588
+