interloper-google-cloud 0.78.0__tar.gz → 0.79.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: interloper-google-cloud
3
- Version: 0.78.0
3
+ Version: 0.79.0
4
4
  Summary: Interloper Google Cloud integration: BigQuery and Cloud Storage destinations
5
5
  Author: Guillaume Onfroy
6
6
  Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "interloper-google-cloud"
3
- version = "0.78.0"
3
+ version = "0.79.0"
4
4
  description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -3,7 +3,7 @@
3
3
  # ###############
4
4
  [project]
5
5
  name = "interloper-google-cloud"
6
- version = "0.78.0"
6
+ version = "0.79.0"
7
7
  description = "Interloper Google Cloud integration: BigQuery and Cloud Storage destinations"
8
8
  readme = "README.md"
9
9
  authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
@@ -259,7 +259,13 @@ class BigQueryDestination(DatabaseDestination):
259
259
  self._ensure_dataset(dataset)
260
260
  time_partitioning = _time_partitioning(partitioning, None)
261
261
 
262
- self._load(self._table_ref(table, dataset), frame, bq_schema, time_partitioning=time_partitioning)
262
+ self._load(
263
+ self._table_ref(table, dataset),
264
+ frame,
265
+ bq_schema,
266
+ time_partitioning=time_partitioning,
267
+ allow_field_addition=context.asset.schema is not None,
268
+ )
263
269
 
264
270
  def _load(
265
271
  self,
@@ -267,13 +273,19 @@ class BigQueryDestination(DatabaseDestination):
267
273
  df: pd.DataFrame,
268
274
  bq_schema: list[bigquery.SchemaField] | None,
269
275
  time_partitioning: bigquery.TimePartitioning | None = None,
276
+ *,
277
+ allow_field_addition: bool = False,
270
278
  ) -> None:
271
279
  """Load a DataFrame via a Parquet load job.
272
280
 
273
281
  When a schema is available, columns are aligned to it: extra columns
274
282
  are dropped (with a warning) and the load job receives explicit field
275
283
  types, so pyarrow casts values (including ``NaN`` → ``NULL``) instead
276
- of relying on dtype autodetection.
284
+ of relying on dtype autodetection. A load behind a declared schema may
285
+ add the columns that schema gained since the table was created: the
286
+ asset's contract changed, and conform already shaped the data to it.
287
+ An inferred schema never grows a table, since a column BigQuery does
288
+ not know is then drift in the data.
277
289
 
278
290
  Args:
279
291
  ref: Fully-qualified table reference.
@@ -281,10 +293,15 @@ class BigQueryDestination(DatabaseDestination):
281
293
  bq_schema: BigQuery field definitions, or ``None`` to autodetect.
282
294
  time_partitioning: Partitioning spec for the table the load job is
283
295
  about to create; ``None`` when the table already exists.
296
+ allow_field_addition: Whether the asset declares the schema the
297
+ load carries, so new declared columns may be added to the
298
+ table; defaults to ``False``.
284
299
  """
285
300
  job_config = bigquery.LoadJobConfig(write_disposition=bigquery.WriteDisposition.WRITE_APPEND)
286
301
  if time_partitioning is not None:
287
302
  job_config.time_partitioning = time_partitioning
303
+ if allow_field_addition and bq_schema is not None:
304
+ job_config.schema_update_options = [bigquery.SchemaUpdateOption.ALLOW_FIELD_ADDITION]
288
305
 
289
306
  if bq_schema is not None:
290
307
  schema_columns = [field.name for field in bq_schema]