interloper-google-cloud 0.78.0__tar.gz → 0.79.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/PKG-INFO +1 -1
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/pyproject.toml +1 -1
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/pyproject.toml.orig +1 -1
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/bigquery/destination.py +19 -2
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/README.md +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/__init__.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/bigquery/__init__.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/bigquery/types.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/connection.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/gcs/__init__.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/gcs/destination.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/gcs/formats.py +0 -0
- {interloper_google_cloud-0.78.0 → interloper_google_cloud-0.79.0}/src/interloper_google_cloud/serialization.py +0 -0
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# ###############
|
|
4
4
|
[project]
|
|
5
5
|
name = "interloper-google-cloud"
|
|
6
|
-
version = "0.
|
|
6
|
+
version = "0.79.0"
|
|
7
7
|
description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
@@ -259,7 +259,13 @@ class BigQueryDestination(DatabaseDestination):
|
|
|
259
259
|
self._ensure_dataset(dataset)
|
|
260
260
|
time_partitioning = _time_partitioning(partitioning, None)
|
|
261
261
|
|
|
262
|
-
self._load(
|
|
262
|
+
self._load(
|
|
263
|
+
self._table_ref(table, dataset),
|
|
264
|
+
frame,
|
|
265
|
+
bq_schema,
|
|
266
|
+
time_partitioning=time_partitioning,
|
|
267
|
+
allow_field_addition=context.asset.schema is not None,
|
|
268
|
+
)
|
|
263
269
|
|
|
264
270
|
def _load(
|
|
265
271
|
self,
|
|
@@ -267,13 +273,19 @@ class BigQueryDestination(DatabaseDestination):
|
|
|
267
273
|
df: pd.DataFrame,
|
|
268
274
|
bq_schema: list[bigquery.SchemaField] | None,
|
|
269
275
|
time_partitioning: bigquery.TimePartitioning | None = None,
|
|
276
|
+
*,
|
|
277
|
+
allow_field_addition: bool = False,
|
|
270
278
|
) -> None:
|
|
271
279
|
"""Load a DataFrame via a Parquet load job.
|
|
272
280
|
|
|
273
281
|
When a schema is available, columns are aligned to it: extra columns
|
|
274
282
|
are dropped (with a warning) and the load job receives explicit field
|
|
275
283
|
types, so pyarrow casts values (including ``NaN`` → ``NULL``) instead
|
|
276
|
-
of relying on dtype autodetection.
|
|
284
|
+
of relying on dtype autodetection. A load behind a declared schema may
|
|
285
|
+
add the columns that schema gained since the table was created: the
|
|
286
|
+
asset's contract changed, and conform already shaped the data to it.
|
|
287
|
+
An inferred schema never grows a table, since a column BigQuery does
|
|
288
|
+
not know is then drift in the data.
|
|
277
289
|
|
|
278
290
|
Args:
|
|
279
291
|
ref: Fully-qualified table reference.
|
|
@@ -281,10 +293,15 @@ class BigQueryDestination(DatabaseDestination):
|
|
|
281
293
|
bq_schema: BigQuery field definitions, or ``None`` to autodetect.
|
|
282
294
|
time_partitioning: Partitioning spec for the table the load job is
|
|
283
295
|
about to create; ``None`` when the table already exists.
|
|
296
|
+
allow_field_addition: Whether the asset declares the schema the
|
|
297
|
+
load carries, so new declared columns may be added to the
|
|
298
|
+
table; defaults to ``False``.
|
|
284
299
|
"""
|
|
285
300
|
job_config = bigquery.LoadJobConfig(write_disposition=bigquery.WriteDisposition.WRITE_APPEND)
|
|
286
301
|
if time_partitioning is not None:
|
|
287
302
|
job_config.time_partitioning = time_partitioning
|
|
303
|
+
if allow_field_addition and bq_schema is not None:
|
|
304
|
+
job_config.schema_update_options = [bigquery.SchemaUpdateOption.ALLOW_FIELD_ADDITION]
|
|
288
305
|
|
|
289
306
|
if bq_schema is not None:
|
|
290
307
|
schema_columns = [field.name for field in bq_schema]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|