google-cloud-bigquery 3.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. google/cloud/bigquery/__init__.py +249 -0
  2. google/cloud/bigquery/_helpers.py +1102 -0
  3. google/cloud/bigquery/_http.py +47 -0
  4. google/cloud/bigquery/_job_helpers.py +600 -0
  5. google/cloud/bigquery/_pandas_helpers.py +1181 -0
  6. google/cloud/bigquery/_pyarrow_helpers.py +147 -0
  7. google/cloud/bigquery/_tqdm_helpers.py +137 -0
  8. google/cloud/bigquery/_versions_helpers.py +264 -0
  9. google/cloud/bigquery/client.py +4406 -0
  10. google/cloud/bigquery/dataset.py +1076 -0
  11. google/cloud/bigquery/dbapi/__init__.py +87 -0
  12. google/cloud/bigquery/dbapi/_helpers.py +522 -0
  13. google/cloud/bigquery/dbapi/connection.py +128 -0
  14. google/cloud/bigquery/dbapi/cursor.py +586 -0
  15. google/cloud/bigquery/dbapi/exceptions.py +58 -0
  16. google/cloud/bigquery/dbapi/types.py +96 -0
  17. google/cloud/bigquery/encryption_configuration.py +84 -0
  18. google/cloud/bigquery/enums.py +389 -0
  19. google/cloud/bigquery/exceptions.py +35 -0
  20. google/cloud/bigquery/external_config.py +1188 -0
  21. google/cloud/bigquery/format_options.py +147 -0
  22. google/cloud/bigquery/iam.py +38 -0
  23. google/cloud/bigquery/job/__init__.py +87 -0
  24. google/cloud/bigquery/job/base.py +1116 -0
  25. google/cloud/bigquery/job/copy_.py +282 -0
  26. google/cloud/bigquery/job/extract.py +271 -0
  27. google/cloud/bigquery/job/load.py +985 -0
  28. google/cloud/bigquery/job/query.py +2498 -0
  29. google/cloud/bigquery/magics/__init__.py +20 -0
  30. google/cloud/bigquery/magics/line_arg_parser/__init__.py +34 -0
  31. google/cloud/bigquery/magics/line_arg_parser/exceptions.py +25 -0
  32. google/cloud/bigquery/magics/line_arg_parser/lexer.py +200 -0
  33. google/cloud/bigquery/magics/line_arg_parser/parser.py +484 -0
  34. google/cloud/bigquery/magics/line_arg_parser/visitors.py +159 -0
  35. google/cloud/bigquery/magics/magics.py +776 -0
  36. google/cloud/bigquery/model.py +517 -0
  37. google/cloud/bigquery/opentelemetry_tracing.py +164 -0
  38. google/cloud/bigquery/py.typed +2 -0
  39. google/cloud/bigquery/query.py +1344 -0
  40. google/cloud/bigquery/retry.py +207 -0
  41. google/cloud/bigquery/routine/__init__.py +33 -0
  42. google/cloud/bigquery/routine/routine.py +744 -0
  43. google/cloud/bigquery/schema.py +896 -0
  44. google/cloud/bigquery/standard_sql.py +389 -0
  45. google/cloud/bigquery/table.py +3594 -0
  46. google/cloud/bigquery/version.py +15 -0
  47. google/cloud/bigquery_v2/__init__.py +56 -0
  48. google/cloud/bigquery_v2/types/__init__.py +54 -0
  49. google/cloud/bigquery_v2/types/encryption_config.py +48 -0
  50. google/cloud/bigquery_v2/types/model.py +1994 -0
  51. google/cloud/bigquery_v2/types/model_reference.py +57 -0
  52. google/cloud/bigquery_v2/types/standard_sql.py +156 -0
  53. google/cloud/bigquery_v2/types/table_reference.py +80 -0
  54. google_cloud_bigquery-3.31.0.dist-info/LICENSE +202 -0
  55. google_cloud_bigquery-3.31.0.dist-info/METADATA +203 -0
  56. google_cloud_bigquery-3.31.0.dist-info/RECORD +58 -0
  57. google_cloud_bigquery-3.31.0.dist-info/WHEEL +5 -0
  58. google_cloud_bigquery-3.31.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1181 @@
1
+ # Copyright 2019 Google LLC
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Shared helper functions for connecting BigQuery and pandas.
16
+
17
+ NOTE: This module is DEPRECATED. Please make updates in the pandas-gbq package,
18
+ instead. See: go/pandas-gbq-and-bigframes-redundancy and
19
+ https://github.com/googleapis/python-bigquery-pandas/blob/main/pandas_gbq/schema/pandas_to_bigquery.py
20
+ """
21
+
22
+ import concurrent.futures
23
+ from datetime import datetime
24
+ import functools
25
+ from itertools import islice
26
+ import logging
27
+ import queue
28
+ import threading
29
+ import warnings
30
+ from typing import Any, Union, Optional, Callable, Generator, List
31
+
32
+
33
+ from google.cloud.bigquery import _pyarrow_helpers
34
+ from google.cloud.bigquery import _versions_helpers
35
+ from google.cloud.bigquery import schema
36
+
37
+
38
+ try:
39
+ import pandas # type: ignore
40
+
41
+ pandas_import_exception = None
42
+ except ImportError as exc:
43
+ pandas = None
44
+ pandas_import_exception = exc
45
+ else:
46
+ import numpy
47
+
48
+
49
+ try:
50
+ import pandas_gbq.schema.pandas_to_bigquery # type: ignore
51
+
52
+ pandas_gbq_import_exception = None
53
+ except ImportError as exc:
54
+ pandas_gbq = None
55
+ pandas_gbq_import_exception = exc
56
+
57
+
58
+ try:
59
+ import db_dtypes # type: ignore
60
+
61
+ date_dtype_name = db_dtypes.DateDtype.name
62
+ time_dtype_name = db_dtypes.TimeDtype.name
63
+ db_dtypes_import_exception = None
64
+ except ImportError as exc:
65
+ db_dtypes = None
66
+ db_dtypes_import_exception = exc
67
+ date_dtype_name = time_dtype_name = "" # Use '' rather than None because pytype
68
+
69
+ pyarrow = _versions_helpers.PYARROW_VERSIONS.try_import()
70
+
71
+ try:
72
+ # _BaseGeometry is used to detect shapely objevys in `bq_to_arrow_array`
73
+ from shapely.geometry.base import BaseGeometry as _BaseGeometry # type: ignore
74
+ except ImportError:
75
+ # No shapely, use NoneType for _BaseGeometry as a placeholder.
76
+ _BaseGeometry = type(None)
77
+ else:
78
+ # We don't have any unit test sessions that install shapely but not pandas.
79
+ if pandas is not None: # pragma: NO COVER
80
+
81
+ def _to_wkb():
82
+ from shapely import wkb # type: ignore
83
+
84
+ write = wkb.dumps
85
+ notnull = pandas.notnull
86
+
87
+ def _to_wkb(v):
88
+ return write(v) if notnull(v) else v
89
+
90
+ return _to_wkb
91
+
92
+ _to_wkb = _to_wkb()
93
+
94
+ try:
95
+ from google.cloud.bigquery_storage_v1.types import ArrowSerializationOptions
96
+ except ImportError:
97
+ _ARROW_COMPRESSION_SUPPORT = False
98
+ else:
99
+ # Having BQ Storage available implies that pyarrow >=1.0.0 is available, too.
100
+ _ARROW_COMPRESSION_SUPPORT = True
101
+
102
+ _LOGGER = logging.getLogger(__name__)
103
+
104
+ _PROGRESS_INTERVAL = 0.2 # Maximum time between download status checks, in seconds.
105
+
106
+ _MAX_QUEUE_SIZE_DEFAULT = object() # max queue size sentinel for BQ Storage downloads
107
+
108
+ _NO_PANDAS_ERROR = "Please install the 'pandas' package to use this function."
109
+ _NO_DB_TYPES_ERROR = "Please install the 'db-dtypes' package to use this function."
110
+
111
+ _PANDAS_DTYPE_TO_BQ = {
112
+ "bool": "BOOLEAN",
113
+ "datetime64[ns, UTC]": "TIMESTAMP",
114
+ "datetime64[ns]": "DATETIME",
115
+ "float32": "FLOAT",
116
+ "float64": "FLOAT",
117
+ "int8": "INTEGER",
118
+ "int16": "INTEGER",
119
+ "int32": "INTEGER",
120
+ "int64": "INTEGER",
121
+ "uint8": "INTEGER",
122
+ "uint16": "INTEGER",
123
+ "uint32": "INTEGER",
124
+ "geometry": "GEOGRAPHY",
125
+ date_dtype_name: "DATE",
126
+ time_dtype_name: "TIME",
127
+ }
128
+
129
+
130
+ class _DownloadState(object):
131
+ """Flag to indicate that a thread should exit early."""
132
+
133
+ def __init__(self):
134
+ # No need for a lock because reading/replacing a variable is defined to
135
+ # be an atomic operation in the Python language definition (enforced by
136
+ # the global interpreter lock).
137
+ self.done = False
138
+ # To assist with testing and understanding the behavior of the
139
+ # download, use this object as shared state to track how many worker
140
+ # threads have started and have gracefully shutdown.
141
+ self._started_workers_lock = threading.Lock()
142
+ self.started_workers = 0
143
+ self._finished_workers_lock = threading.Lock()
144
+ self.finished_workers = 0
145
+
146
+ def start(self):
147
+ with self._started_workers_lock:
148
+ self.started_workers += 1
149
+
150
+ def finish(self):
151
+ with self._finished_workers_lock:
152
+ self.finished_workers += 1
153
+
154
+
155
+ BQ_FIELD_TYPE_TO_ARROW_FIELD_METADATA = {
156
+ "GEOGRAPHY": {
157
+ b"ARROW:extension:name": b"google:sqlType:geography",
158
+ b"ARROW:extension:metadata": b'{"encoding": "WKT"}',
159
+ },
160
+ "DATETIME": {b"ARROW:extension:name": b"google:sqlType:datetime"},
161
+ "JSON": {b"ARROW:extension:name": b"google:sqlType:json"},
162
+ }
163
+
164
+
165
+ def bq_to_arrow_struct_data_type(field):
166
+ arrow_fields = []
167
+ for subfield in field.fields:
168
+ arrow_subfield = bq_to_arrow_field(subfield)
169
+ if arrow_subfield:
170
+ arrow_fields.append(arrow_subfield)
171
+ else:
172
+ # Could not determine a subfield type. Fallback to type
173
+ # inference.
174
+ return None
175
+ return pyarrow.struct(arrow_fields)
176
+
177
+
178
+ def bq_to_arrow_range_data_type(field):
179
+ if field is None:
180
+ raise ValueError(
181
+ "Range element type cannot be None, must be one of "
182
+ "DATE, DATETIME, or TIMESTAMP"
183
+ )
184
+ element_type = field.element_type.upper()
185
+ arrow_element_type = _pyarrow_helpers.bq_to_arrow_scalars(element_type)()
186
+ return pyarrow.struct([("start", arrow_element_type), ("end", arrow_element_type)])
187
+
188
+
189
+ def bq_to_arrow_data_type(field):
190
+ """Return the Arrow data type, corresponding to a given BigQuery column.
191
+
192
+ Returns:
193
+ None: if default Arrow type inspection should be used.
194
+ """
195
+ if field.mode is not None and field.mode.upper() == "REPEATED":
196
+ inner_type = bq_to_arrow_data_type(
197
+ schema.SchemaField(field.name, field.field_type, fields=field.fields)
198
+ )
199
+ if inner_type:
200
+ return pyarrow.list_(inner_type)
201
+ return None
202
+
203
+ field_type_upper = field.field_type.upper() if field.field_type else ""
204
+ if field_type_upper in schema._STRUCT_TYPES:
205
+ return bq_to_arrow_struct_data_type(field)
206
+
207
+ if field_type_upper == "RANGE":
208
+ return bq_to_arrow_range_data_type(field.range_element_type)
209
+
210
+ data_type_constructor = _pyarrow_helpers.bq_to_arrow_scalars(field_type_upper)
211
+ if data_type_constructor is None:
212
+ return None
213
+ return data_type_constructor()
214
+
215
+
216
+ def bq_to_arrow_field(bq_field, array_type=None):
217
+ """Return the Arrow field, corresponding to a given BigQuery column.
218
+
219
+ Returns:
220
+ None: if the Arrow type cannot be determined.
221
+ """
222
+ arrow_type = bq_to_arrow_data_type(bq_field)
223
+ if arrow_type is not None:
224
+ if array_type is not None:
225
+ arrow_type = array_type # For GEOGRAPHY, at least initially
226
+ metadata = BQ_FIELD_TYPE_TO_ARROW_FIELD_METADATA.get(
227
+ bq_field.field_type.upper() if bq_field.field_type else ""
228
+ )
229
+ return pyarrow.field(
230
+ bq_field.name,
231
+ arrow_type,
232
+ # Even if the remote schema is REQUIRED, there's a chance there's
233
+ # local NULL values. Arrow will gladly interpret these NULL values
234
+ # as non-NULL and give you an arbitrary value. See:
235
+ # https://github.com/googleapis/python-bigquery/issues/1692
236
+ nullable=False if bq_field.mode.upper() == "REPEATED" else True,
237
+ metadata=metadata,
238
+ )
239
+
240
+ warnings.warn(
241
+ "Unable to determine Arrow type for field '{}'.".format(bq_field.name)
242
+ )
243
+ return None
244
+
245
+
246
+ def bq_to_arrow_schema(bq_schema):
247
+ """Return the Arrow schema, corresponding to a given BigQuery schema.
248
+
249
+ Returns:
250
+ None: if any Arrow type cannot be determined.
251
+ """
252
+ arrow_fields = []
253
+ for bq_field in bq_schema:
254
+ arrow_field = bq_to_arrow_field(bq_field)
255
+ if arrow_field is None:
256
+ # Auto-detect the schema if there is an unknown field type.
257
+ return None
258
+ arrow_fields.append(arrow_field)
259
+ return pyarrow.schema(arrow_fields)
260
+
261
+
262
+ def default_types_mapper(
263
+ date_as_object: bool = False,
264
+ bool_dtype: Union[Any, None] = None,
265
+ int_dtype: Union[Any, None] = None,
266
+ float_dtype: Union[Any, None] = None,
267
+ string_dtype: Union[Any, None] = None,
268
+ date_dtype: Union[Any, None] = None,
269
+ datetime_dtype: Union[Any, None] = None,
270
+ time_dtype: Union[Any, None] = None,
271
+ timestamp_dtype: Union[Any, None] = None,
272
+ range_date_dtype: Union[Any, None] = None,
273
+ range_datetime_dtype: Union[Any, None] = None,
274
+ range_timestamp_dtype: Union[Any, None] = None,
275
+ ):
276
+ """Create a mapping from pyarrow types to pandas types.
277
+
278
+ This overrides the pandas defaults to use null-safe extension types where
279
+ available.
280
+
281
+ See: https://arrow.apache.org/docs/python/api/datatypes.html for a list of
282
+ data types. See:
283
+ tests/unit/test__pandas_helpers.py::test_bq_to_arrow_data_type for
284
+ BigQuery to Arrow type mapping.
285
+
286
+ Note to google-cloud-bigquery developers: If you update the default dtypes,
287
+ also update the docs at docs/usage/pandas.rst.
288
+ """
289
+
290
+ def types_mapper(arrow_data_type):
291
+ if bool_dtype is not None and pyarrow.types.is_boolean(arrow_data_type):
292
+ return bool_dtype
293
+
294
+ elif int_dtype is not None and pyarrow.types.is_integer(arrow_data_type):
295
+ return int_dtype
296
+
297
+ elif float_dtype is not None and pyarrow.types.is_floating(arrow_data_type):
298
+ return float_dtype
299
+
300
+ elif string_dtype is not None and pyarrow.types.is_string(arrow_data_type):
301
+ return string_dtype
302
+
303
+ elif (
304
+ # If date_as_object is True, we know some DATE columns are
305
+ # out-of-bounds of what is supported by pandas.
306
+ date_dtype is not None
307
+ and not date_as_object
308
+ and pyarrow.types.is_date(arrow_data_type)
309
+ ):
310
+ return date_dtype
311
+
312
+ elif (
313
+ datetime_dtype is not None
314
+ and pyarrow.types.is_timestamp(arrow_data_type)
315
+ and arrow_data_type.tz is None
316
+ ):
317
+ return datetime_dtype
318
+
319
+ elif (
320
+ timestamp_dtype is not None
321
+ and pyarrow.types.is_timestamp(arrow_data_type)
322
+ and arrow_data_type.tz is not None
323
+ ):
324
+ return timestamp_dtype
325
+
326
+ elif time_dtype is not None and pyarrow.types.is_time(arrow_data_type):
327
+ return time_dtype
328
+
329
+ elif pyarrow.types.is_struct(arrow_data_type):
330
+ if range_datetime_dtype is not None and arrow_data_type.equals(
331
+ range_datetime_dtype.pyarrow_dtype
332
+ ):
333
+ return range_datetime_dtype
334
+
335
+ elif range_date_dtype is not None and arrow_data_type.equals(
336
+ range_date_dtype.pyarrow_dtype
337
+ ):
338
+ return range_date_dtype
339
+
340
+ # TODO: this section does not have a test yet OR at least not one that is
341
+ # recognized by coverage, hence the pragma. See Issue: #2132
342
+ elif (
343
+ range_timestamp_dtype is not None
344
+ and arrow_data_type.equals( # pragma: NO COVER
345
+ range_timestamp_dtype.pyarrow_dtype
346
+ )
347
+ ):
348
+ return range_timestamp_dtype
349
+
350
+ return types_mapper
351
+
352
+
353
+ def bq_to_arrow_array(series, bq_field):
354
+ if bq_field.field_type.upper() == "GEOGRAPHY":
355
+ arrow_type = None
356
+ first = _first_valid(series)
357
+ if first is not None:
358
+ if series.dtype.name == "geometry" or isinstance(first, _BaseGeometry):
359
+ arrow_type = pyarrow.binary()
360
+ # Convert shapey geometry to WKB binary format:
361
+ series = series.apply(_to_wkb)
362
+ elif isinstance(first, bytes):
363
+ arrow_type = pyarrow.binary()
364
+ elif series.dtype.name == "geometry":
365
+ # We have a GeoSeries containing all nulls, convert it to a pandas series
366
+ series = pandas.Series(numpy.array(series))
367
+
368
+ if arrow_type is None:
369
+ arrow_type = bq_to_arrow_data_type(bq_field)
370
+ else:
371
+ arrow_type = bq_to_arrow_data_type(bq_field)
372
+
373
+ field_type_upper = bq_field.field_type.upper() if bq_field.field_type else ""
374
+
375
+ try:
376
+ if bq_field.mode.upper() == "REPEATED":
377
+ return pyarrow.ListArray.from_pandas(series, type=arrow_type)
378
+ if field_type_upper in schema._STRUCT_TYPES:
379
+ return pyarrow.StructArray.from_pandas(series, type=arrow_type)
380
+ return pyarrow.Array.from_pandas(series, type=arrow_type)
381
+ except pyarrow.ArrowTypeError:
382
+ msg = f"""Error converting Pandas column with name: "{series.name}" and datatype: "{series.dtype}" to an appropriate pyarrow datatype: Array, ListArray, or StructArray"""
383
+ _LOGGER.error(msg)
384
+ raise pyarrow.ArrowTypeError(msg)
385
+
386
+
387
+ def get_column_or_index(dataframe, name):
388
+ """Return a column or index as a pandas series."""
389
+ if name in dataframe.columns:
390
+ return dataframe[name].reset_index(drop=True)
391
+
392
+ if isinstance(dataframe.index, pandas.MultiIndex):
393
+ if name in dataframe.index.names:
394
+ return (
395
+ dataframe.index.get_level_values(name)
396
+ .to_series()
397
+ .reset_index(drop=True)
398
+ )
399
+ else:
400
+ if name == dataframe.index.name:
401
+ return dataframe.index.to_series().reset_index(drop=True)
402
+
403
+ raise ValueError("column or index '{}' not found.".format(name))
404
+
405
+
406
+ def list_columns_and_indexes(dataframe):
407
+ """Return all index and column names with dtypes.
408
+
409
+ Returns:
410
+ Sequence[Tuple[str, dtype]]:
411
+ Returns a sorted list of indexes and column names with
412
+ corresponding dtypes. If an index is missing a name or has the
413
+ same name as a column, the index is omitted.
414
+ """
415
+ column_names = frozenset(dataframe.columns)
416
+ columns_and_indexes = []
417
+ if isinstance(dataframe.index, pandas.MultiIndex):
418
+ for name in dataframe.index.names:
419
+ if name and name not in column_names:
420
+ values = dataframe.index.get_level_values(name)
421
+ columns_and_indexes.append((name, values.dtype))
422
+ else:
423
+ if dataframe.index.name and dataframe.index.name not in column_names:
424
+ columns_and_indexes.append((dataframe.index.name, dataframe.index.dtype))
425
+
426
+ columns_and_indexes += zip(dataframe.columns, dataframe.dtypes)
427
+ return columns_and_indexes
428
+
429
+
430
+ def _first_valid(series):
431
+ first_valid_index = series.first_valid_index()
432
+ if first_valid_index is not None:
433
+ return series.at[first_valid_index]
434
+
435
+
436
+ def _first_array_valid(series):
437
+ """Return the first "meaningful" element from the array series.
438
+
439
+ Here, "meaningful" means the first non-None element in one of the arrays that can
440
+ be used for type detextion.
441
+ """
442
+ first_valid_index = series.first_valid_index()
443
+ if first_valid_index is None:
444
+ return None
445
+
446
+ valid_array = series.at[first_valid_index]
447
+ valid_item = next((item for item in valid_array if not pandas.isna(item)), None)
448
+
449
+ if valid_item is not None:
450
+ return valid_item
451
+
452
+ # Valid item is None because all items in the "valid" array are invalid. Try
453
+ # to find a true valid array manually.
454
+ for array in islice(series, first_valid_index + 1, None):
455
+ try:
456
+ array_iter = iter(array)
457
+ except TypeError:
458
+ continue # Not an array, apparently, e.g. None, thus skip.
459
+ valid_item = next((item for item in array_iter if not pandas.isna(item)), None)
460
+ if valid_item is not None:
461
+ break
462
+
463
+ return valid_item
464
+
465
+
466
+ def dataframe_to_bq_schema(dataframe, bq_schema):
467
+ """Convert a pandas DataFrame schema to a BigQuery schema.
468
+
469
+ DEPRECATED: Use
470
+ pandas_gbq.schema.pandas_to_bigquery.dataframe_to_bigquery_fields(),
471
+ instead. See: go/pandas-gbq-and-bigframes-redundancy.
472
+
473
+ Args:
474
+ dataframe (pandas.DataFrame):
475
+ DataFrame for which the client determines the BigQuery schema.
476
+ bq_schema (Sequence[Union[ \
477
+ :class:`~google.cloud.bigquery.schema.SchemaField`, \
478
+ Mapping[str, Any] \
479
+ ]]):
480
+ A BigQuery schema. Use this argument to override the autodetected
481
+ type for some or all of the DataFrame columns.
482
+
483
+ Returns:
484
+ Optional[Sequence[google.cloud.bigquery.schema.SchemaField]]:
485
+ The automatically determined schema. Returns None if the type of
486
+ any column cannot be determined.
487
+ """
488
+ if pandas_gbq is None:
489
+ warnings.warn(
490
+ "Loading pandas DataFrame into BigQuery will require pandas-gbq "
491
+ "package version 0.26.1 or greater in the future. "
492
+ f"Tried to import pandas-gbq and got: {pandas_gbq_import_exception}",
493
+ category=FutureWarning,
494
+ )
495
+ else:
496
+ return pandas_gbq.schema.pandas_to_bigquery.dataframe_to_bigquery_fields(
497
+ dataframe,
498
+ override_bigquery_fields=bq_schema,
499
+ index=True,
500
+ )
501
+
502
+ if bq_schema:
503
+ bq_schema = schema._to_schema_fields(bq_schema)
504
+ bq_schema_index = {field.name: field for field in bq_schema}
505
+ bq_schema_unused = set(bq_schema_index.keys())
506
+ else:
507
+ bq_schema_index = {}
508
+ bq_schema_unused = set()
509
+
510
+ bq_schema_out = []
511
+ unknown_type_fields = []
512
+
513
+ for column, dtype in list_columns_and_indexes(dataframe):
514
+ # Use provided type from schema, if present.
515
+ bq_field = bq_schema_index.get(column)
516
+ if bq_field:
517
+ bq_schema_out.append(bq_field)
518
+ bq_schema_unused.discard(bq_field.name)
519
+ continue
520
+
521
+ # Otherwise, try to automatically determine the type based on the
522
+ # pandas dtype.
523
+ bq_type = _PANDAS_DTYPE_TO_BQ.get(dtype.name)
524
+ if bq_type is None:
525
+ sample_data = _first_valid(dataframe.reset_index()[column])
526
+ if (
527
+ isinstance(sample_data, _BaseGeometry)
528
+ and sample_data is not None # Paranoia
529
+ ):
530
+ bq_type = "GEOGRAPHY"
531
+ bq_field = schema.SchemaField(column, bq_type)
532
+ bq_schema_out.append(bq_field)
533
+
534
+ if bq_field.field_type is None:
535
+ unknown_type_fields.append(bq_field)
536
+
537
+ # Catch any schema mismatch. The developer explicitly asked to serialize a
538
+ # column, but it was not found.
539
+ if bq_schema_unused:
540
+ raise ValueError(
541
+ "bq_schema contains fields not present in dataframe: {}".format(
542
+ bq_schema_unused
543
+ )
544
+ )
545
+
546
+ # If schema detection was not successful for all columns, also try with
547
+ # pyarrow, if available.
548
+ if unknown_type_fields:
549
+ if not pyarrow:
550
+ msg = "Could not determine the type of columns: {}".format(
551
+ ", ".join(field.name for field in unknown_type_fields)
552
+ )
553
+ warnings.warn(msg)
554
+ return None # We cannot detect the schema in full.
555
+
556
+ # The augment_schema() helper itself will also issue unknown type
557
+ # warnings if detection still fails for any of the fields.
558
+ bq_schema_out = augment_schema(dataframe, bq_schema_out)
559
+
560
+ return tuple(bq_schema_out) if bq_schema_out else None
561
+
562
+
563
+ def augment_schema(dataframe, current_bq_schema):
564
+ """Try to deduce the unknown field types and return an improved schema.
565
+
566
+ This function requires ``pyarrow`` to run. If all the missing types still
567
+ cannot be detected, ``None`` is returned. If all types are already known,
568
+ a shallow copy of the given schema is returned.
569
+
570
+ Args:
571
+ dataframe (pandas.DataFrame):
572
+ DataFrame for which some of the field types are still unknown.
573
+ current_bq_schema (Sequence[google.cloud.bigquery.schema.SchemaField]):
574
+ A BigQuery schema for ``dataframe``. The types of some or all of
575
+ the fields may be ``None``.
576
+ Returns:
577
+ Optional[Sequence[google.cloud.bigquery.schema.SchemaField]]
578
+ """
579
+ # pytype: disable=attribute-error
580
+ augmented_schema = []
581
+ unknown_type_fields = []
582
+ for field in current_bq_schema:
583
+ if field.field_type is not None:
584
+ augmented_schema.append(field)
585
+ continue
586
+
587
+ arrow_table = pyarrow.array(dataframe.reset_index()[field.name])
588
+
589
+ if pyarrow.types.is_list(arrow_table.type):
590
+ # `pyarrow.ListType`
591
+ detected_mode = "REPEATED"
592
+ detected_type = _pyarrow_helpers.arrow_scalar_ids_to_bq(
593
+ arrow_table.values.type.id
594
+ )
595
+
596
+ # For timezone-naive datetimes, pyarrow assumes the UTC timezone and adds
597
+ # it to such datetimes, causing them to be recognized as TIMESTAMP type.
598
+ # We thus additionally check the actual data to see if we need to overrule
599
+ # that and choose DATETIME instead.
600
+ # Note that this should only be needed for datetime values inside a list,
601
+ # since scalar datetime values have a proper Pandas dtype that allows
602
+ # distinguishing between timezone-naive and timezone-aware values before
603
+ # even requiring the additional schema augment logic in this method.
604
+ if detected_type == "TIMESTAMP":
605
+ valid_item = _first_array_valid(dataframe[field.name])
606
+ if isinstance(valid_item, datetime) and valid_item.tzinfo is None:
607
+ detected_type = "DATETIME"
608
+ else:
609
+ detected_mode = field.mode
610
+ detected_type = _pyarrow_helpers.arrow_scalar_ids_to_bq(arrow_table.type.id)
611
+ if detected_type == "NUMERIC" and arrow_table.type.scale > 9:
612
+ detected_type = "BIGNUMERIC"
613
+
614
+ if detected_type is None:
615
+ unknown_type_fields.append(field)
616
+ continue
617
+
618
+ new_field = schema.SchemaField(
619
+ name=field.name,
620
+ field_type=detected_type,
621
+ mode=detected_mode,
622
+ description=field.description,
623
+ fields=field.fields,
624
+ )
625
+ augmented_schema.append(new_field)
626
+
627
+ if unknown_type_fields:
628
+ warnings.warn(
629
+ "Pyarrow could not determine the type of columns: {}.".format(
630
+ ", ".join(field.name for field in unknown_type_fields)
631
+ )
632
+ )
633
+ return None
634
+
635
+ return augmented_schema
636
+ # pytype: enable=attribute-error
637
+
638
+
639
+ def dataframe_to_arrow(dataframe, bq_schema):
640
+ """Convert pandas dataframe to Arrow table, using BigQuery schema.
641
+
642
+ Args:
643
+ dataframe (pandas.DataFrame):
644
+ DataFrame to convert to Arrow table.
645
+ bq_schema (Sequence[Union[ \
646
+ :class:`~google.cloud.bigquery.schema.SchemaField`, \
647
+ Mapping[str, Any] \
648
+ ]]):
649
+ Desired BigQuery schema. The number of columns must match the
650
+ number of columns in the DataFrame.
651
+
652
+ Returns:
653
+ pyarrow.Table:
654
+ Table containing dataframe data, with schema derived from
655
+ BigQuery schema.
656
+ """
657
+ column_names = set(dataframe.columns)
658
+ column_and_index_names = set(
659
+ name for name, _ in list_columns_and_indexes(dataframe)
660
+ )
661
+
662
+ bq_schema = schema._to_schema_fields(bq_schema)
663
+ bq_field_names = set(field.name for field in bq_schema)
664
+
665
+ extra_fields = bq_field_names - column_and_index_names
666
+ if extra_fields:
667
+ raise ValueError(
668
+ "bq_schema contains fields not present in dataframe: {}".format(
669
+ extra_fields
670
+ )
671
+ )
672
+
673
+ # It's okay for indexes to be missing from bq_schema, but it's not okay to
674
+ # be missing columns.
675
+ missing_fields = column_names - bq_field_names
676
+ if missing_fields:
677
+ raise ValueError(
678
+ "bq_schema is missing fields from dataframe: {}".format(missing_fields)
679
+ )
680
+
681
+ arrow_arrays = []
682
+ arrow_names = []
683
+ arrow_fields = []
684
+ for bq_field in bq_schema:
685
+ arrow_names.append(bq_field.name)
686
+ arrow_arrays.append(
687
+ bq_to_arrow_array(get_column_or_index(dataframe, bq_field.name), bq_field)
688
+ )
689
+ arrow_fields.append(bq_to_arrow_field(bq_field, arrow_arrays[-1].type))
690
+
691
+ if all((field is not None for field in arrow_fields)):
692
+ return pyarrow.Table.from_arrays(
693
+ arrow_arrays, schema=pyarrow.schema(arrow_fields)
694
+ )
695
+ return pyarrow.Table.from_arrays(arrow_arrays, names=arrow_names)
696
+
697
+
698
+ def dataframe_to_parquet(
699
+ dataframe,
700
+ bq_schema,
701
+ filepath,
702
+ parquet_compression="SNAPPY",
703
+ parquet_use_compliant_nested_type=True,
704
+ ):
705
+ """Write dataframe as a Parquet file, according to the desired BQ schema.
706
+
707
+ This function requires the :mod:`pyarrow` package. Arrow is used as an
708
+ intermediate format.
709
+
710
+ Args:
711
+ dataframe (pandas.DataFrame):
712
+ DataFrame to convert to Parquet file.
713
+ bq_schema (Sequence[Union[ \
714
+ :class:`~google.cloud.bigquery.schema.SchemaField`, \
715
+ Mapping[str, Any] \
716
+ ]]):
717
+ Desired BigQuery schema. Number of columns must match number of
718
+ columns in the DataFrame.
719
+ filepath (str):
720
+ Path to write Parquet file to.
721
+ parquet_compression (Optional[str]):
722
+ The compression codec to use by the the ``pyarrow.parquet.write_table``
723
+ serializing method. Defaults to "SNAPPY".
724
+ https://arrow.apache.org/docs/python/generated/pyarrow.parquet.write_table.html#pyarrow-parquet-write-table
725
+ parquet_use_compliant_nested_type (bool):
726
+ Whether the ``pyarrow.parquet.write_table`` serializing method should write
727
+ compliant Parquet nested type (lists). Defaults to ``True``.
728
+ https://github.com/apache/parquet-format/blob/master/LogicalTypes.md#nested-types
729
+ https://arrow.apache.org/docs/python/generated/pyarrow.parquet.write_table.html#pyarrow-parquet-write-table
730
+
731
+ This argument is ignored for ``pyarrow`` versions earlier than ``4.0.0``.
732
+ """
733
+ pyarrow = _versions_helpers.PYARROW_VERSIONS.try_import(raise_if_error=True)
734
+
735
+ import pyarrow.parquet # type: ignore
736
+
737
+ kwargs = (
738
+ {"use_compliant_nested_type": parquet_use_compliant_nested_type}
739
+ if _versions_helpers.PYARROW_VERSIONS.use_compliant_nested_type
740
+ else {}
741
+ )
742
+
743
+ bq_schema = schema._to_schema_fields(bq_schema)
744
+ arrow_table = dataframe_to_arrow(dataframe, bq_schema)
745
+ pyarrow.parquet.write_table(
746
+ arrow_table,
747
+ filepath,
748
+ compression=parquet_compression,
749
+ **kwargs,
750
+ )
751
+
752
+
753
+ def _row_iterator_page_to_arrow(page, column_names, arrow_types):
754
+ # Iterate over the page to force the API request to get the page data.
755
+ try:
756
+ next(iter(page))
757
+ except StopIteration:
758
+ pass
759
+
760
+ arrays = []
761
+ for column_index, arrow_type in enumerate(arrow_types):
762
+ arrays.append(pyarrow.array(page._columns[column_index], type=arrow_type))
763
+
764
+ if isinstance(column_names, pyarrow.Schema):
765
+ return pyarrow.RecordBatch.from_arrays(arrays, schema=column_names)
766
+ return pyarrow.RecordBatch.from_arrays(arrays, names=column_names)
767
+
768
+
769
+ def download_arrow_row_iterator(pages, bq_schema):
770
+ """Use HTTP JSON RowIterator to construct an iterable of RecordBatches.
771
+
772
+ Args:
773
+ pages (Iterator[:class:`google.api_core.page_iterator.Page`]):
774
+ An iterator over the result pages.
775
+ bq_schema (Sequence[Union[ \
776
+ :class:`~google.cloud.bigquery.schema.SchemaField`, \
777
+ Mapping[str, Any] \
778
+ ]]):
779
+ A decription of the fields in result pages.
780
+ Yields:
781
+ :class:`pyarrow.RecordBatch`
782
+ The next page of records as a ``pyarrow`` record batch.
783
+ """
784
+ bq_schema = schema._to_schema_fields(bq_schema)
785
+ column_names = bq_to_arrow_schema(bq_schema) or [field.name for field in bq_schema]
786
+ arrow_types = [bq_to_arrow_data_type(field) for field in bq_schema]
787
+
788
+ for page in pages:
789
+ yield _row_iterator_page_to_arrow(page, column_names, arrow_types)
790
+
791
+
792
+ def _row_iterator_page_to_dataframe(page, column_names, dtypes):
793
+ # Iterate over the page to force the API request to get the page data.
794
+ try:
795
+ next(iter(page))
796
+ except StopIteration:
797
+ pass
798
+
799
+ columns = {}
800
+ for column_index, column_name in enumerate(column_names):
801
+ dtype = dtypes.get(column_name)
802
+ columns[column_name] = pandas.Series(page._columns[column_index], dtype=dtype)
803
+
804
+ return pandas.DataFrame(columns, columns=column_names)
805
+
806
+
807
+ def download_dataframe_row_iterator(pages, bq_schema, dtypes):
808
+ """Use HTTP JSON RowIterator to construct a DataFrame.
809
+
810
+ Args:
811
+ pages (Iterator[:class:`google.api_core.page_iterator.Page`]):
812
+ An iterator over the result pages.
813
+ bq_schema (Sequence[Union[ \
814
+ :class:`~google.cloud.bigquery.schema.SchemaField`, \
815
+ Mapping[str, Any] \
816
+ ]]):
817
+ A decription of the fields in result pages.
818
+ dtypes(Mapping[str, numpy.dtype]):
819
+ The types of columns in result data to hint construction of the
820
+ resulting DataFrame. Not all column types have to be specified.
821
+ Yields:
822
+ :class:`pandas.DataFrame`
823
+ The next page of records as a ``pandas.DataFrame`` record batch.
824
+ """
825
+ bq_schema = schema._to_schema_fields(bq_schema)
826
+ column_names = [field.name for field in bq_schema]
827
+ for page in pages:
828
+ yield _row_iterator_page_to_dataframe(page, column_names, dtypes)
829
+
830
+
831
+ def _bqstorage_page_to_arrow(page):
832
+ return page.to_arrow()
833
+
834
+
835
+ def _bqstorage_page_to_dataframe(column_names, dtypes, page):
836
+ # page.to_dataframe() does not preserve column order in some versions
837
+ # of google-cloud-bigquery-storage. Access by column name to rearrange.
838
+ return page.to_dataframe(dtypes=dtypes)[column_names]
839
+
840
+
841
+ def _download_table_bqstorage_stream(
842
+ download_state, bqstorage_client, session, stream, worker_queue, page_to_item
843
+ ):
844
+ download_state.start()
845
+ try:
846
+ reader = bqstorage_client.read_rows(stream.name)
847
+
848
+ # Avoid deprecation warnings for passing in unnecessary read session.
849
+ # https://github.com/googleapis/python-bigquery-storage/issues/229
850
+ if _versions_helpers.BQ_STORAGE_VERSIONS.is_read_session_optional:
851
+ rowstream = reader.rows()
852
+ else:
853
+ rowstream = reader.rows(session)
854
+
855
+ for page in rowstream.pages:
856
+ item = page_to_item(page)
857
+
858
+ # Make sure we set a timeout on put() so that we give the worker
859
+ # thread opportunities to shutdown gracefully, for example if the
860
+ # parent thread shuts down or the parent generator object which
861
+ # collects rows from all workers goes out of scope. See:
862
+ # https://github.com/googleapis/python-bigquery/issues/2032
863
+ while True:
864
+ if download_state.done:
865
+ return
866
+ try:
867
+ worker_queue.put(item, timeout=_PROGRESS_INTERVAL)
868
+ break
869
+ except queue.Full:
870
+ continue
871
+ finally:
872
+ download_state.finish()
873
+
874
+
875
+ def _nowait(futures):
876
+ """Separate finished and unfinished threads, much like
877
+ :func:`concurrent.futures.wait`, but don't wait.
878
+ """
879
+ done = []
880
+ not_done = []
881
+ for future in futures:
882
+ if future.done():
883
+ done.append(future)
884
+ else:
885
+ not_done.append(future)
886
+ return done, not_done
887
+
888
+
889
+ def _download_table_bqstorage(
890
+ project_id: str,
891
+ table: Any,
892
+ bqstorage_client: Any,
893
+ preserve_order: bool = False,
894
+ selected_fields: Optional[List[Any]] = None,
895
+ page_to_item: Optional[Callable] = None,
896
+ max_queue_size: Any = _MAX_QUEUE_SIZE_DEFAULT,
897
+ max_stream_count: Optional[int] = None,
898
+ download_state: Optional[_DownloadState] = None,
899
+ ) -> Generator[Any, None, None]:
900
+ """Downloads a BigQuery table using the BigQuery Storage API.
901
+
902
+ This method uses the faster, but potentially more expensive, BigQuery
903
+ Storage API to download a table as a Pandas DataFrame. It supports
904
+ parallel downloads and optional data transformations.
905
+
906
+ Args:
907
+ project_id (str): The ID of the Google Cloud project containing
908
+ the table.
909
+ table (Any): The BigQuery table to download.
910
+ bqstorage_client (Any): An
911
+ authenticated BigQuery Storage API client.
912
+ preserve_order (bool, optional): Whether to preserve the order
913
+ of the rows as they are read from BigQuery. If True this limits
914
+ the number of streams to one and overrides `max_stream_count`.
915
+ Defaults to False.
916
+ selected_fields (Optional[List[SchemaField]]):
917
+ A list of BigQuery schema fields to select for download. If None,
918
+ all fields are downloaded. Defaults to None.
919
+ page_to_item (Optional[Callable]): An optional callable
920
+ function that takes a page of data from the BigQuery Storage API
921
+ max_stream_count (Optional[int]): The maximum number of
922
+ concurrent streams to use for downloading data. If `preserve_order`
923
+ is True, the requested streams are limited to 1 regardless of the
924
+ `max_stream_count` value. If 0 or None, then the number of
925
+ requested streams will be unbounded. Defaults to None.
926
+ download_state (Optional[_DownloadState]):
927
+ A threadsafe state object which can be used to observe the
928
+ behavior of the worker threads created by this method.
929
+
930
+ Yields:
931
+ pandas.DataFrame: Pandas DataFrames, one for each chunk of data
932
+ downloaded from BigQuery.
933
+
934
+ Raises:
935
+ ValueError: If attempting to read from a specific partition or snapshot.
936
+
937
+ Note:
938
+ This method requires the `google-cloud-bigquery-storage` library
939
+ to be installed.
940
+ """
941
+
942
+ from google.cloud import bigquery_storage
943
+
944
+ if "$" in table.table_id:
945
+ raise ValueError(
946
+ "Reading from a specific partition is not currently supported."
947
+ )
948
+ if "@" in table.table_id:
949
+ raise ValueError("Reading from a specific snapshot is not currently supported.")
950
+
951
+ requested_streams = determine_requested_streams(preserve_order, max_stream_count)
952
+
953
+ requested_session = bigquery_storage.types.stream.ReadSession(
954
+ table=table.to_bqstorage(),
955
+ data_format=bigquery_storage.types.stream.DataFormat.ARROW,
956
+ )
957
+ if selected_fields is not None:
958
+ for field in selected_fields:
959
+ requested_session.read_options.selected_fields.append(field.name)
960
+
961
+ if _ARROW_COMPRESSION_SUPPORT:
962
+ requested_session.read_options.arrow_serialization_options.buffer_compression = (
963
+ # CompressionCodec(1) -> LZ4_FRAME
964
+ ArrowSerializationOptions.CompressionCodec(1)
965
+ )
966
+
967
+ session = bqstorage_client.create_read_session(
968
+ parent="projects/{}".format(project_id),
969
+ read_session=requested_session,
970
+ max_stream_count=requested_streams,
971
+ )
972
+
973
+ _LOGGER.debug(
974
+ "Started reading table '{}.{}.{}' with BQ Storage API session '{}'.".format(
975
+ table.project, table.dataset_id, table.table_id, session.name
976
+ )
977
+ )
978
+
979
+ # Avoid reading rows from an empty table.
980
+ if not session.streams:
981
+ return
982
+
983
+ total_streams = len(session.streams)
984
+
985
+ # Use _DownloadState to notify worker threads when to quit.
986
+ # See: https://stackoverflow.com/a/29237343/101923
987
+ if download_state is None:
988
+ download_state = _DownloadState()
989
+
990
+ # Create a queue to collect frames as they are created in each thread.
991
+ #
992
+ # The queue needs to be bounded by default, because if the user code processes the
993
+ # fetched result pages too slowly, while at the same time new pages are rapidly being
994
+ # fetched from the server, the queue can grow to the point where the process runs
995
+ # out of memory.
996
+ if max_queue_size is _MAX_QUEUE_SIZE_DEFAULT:
997
+ max_queue_size = total_streams
998
+ elif max_queue_size is None:
999
+ max_queue_size = 0 # unbounded
1000
+
1001
+ worker_queue: queue.Queue[int] = queue.Queue(maxsize=max_queue_size)
1002
+
1003
+ with concurrent.futures.ThreadPoolExecutor(max_workers=total_streams) as pool:
1004
+ try:
1005
+ # Manually submit jobs and wait for download to complete rather
1006
+ # than using pool.map because pool.map continues running in the
1007
+ # background even if there is an exception on the main thread.
1008
+ # See: https://github.com/googleapis/google-cloud-python/pull/7698
1009
+ not_done = [
1010
+ pool.submit(
1011
+ _download_table_bqstorage_stream,
1012
+ download_state,
1013
+ bqstorage_client,
1014
+ session,
1015
+ stream,
1016
+ worker_queue,
1017
+ page_to_item,
1018
+ )
1019
+ for stream in session.streams
1020
+ ]
1021
+
1022
+ while not_done:
1023
+ # Don't block on the worker threads. For performance reasons,
1024
+ # we want to block on the queue's get method, instead. This
1025
+ # prevents the queue from filling up, because the main thread
1026
+ # has smaller gaps in time between calls to the queue's get
1027
+ # method. For a detailed explanation, see:
1028
+ # https://friendliness.dev/2019/06/18/python-nowait/
1029
+ done, not_done = _nowait(not_done)
1030
+ for future in done:
1031
+ # Call result() on any finished threads to raise any
1032
+ # exceptions encountered.
1033
+ future.result()
1034
+
1035
+ try:
1036
+ frame = worker_queue.get(timeout=_PROGRESS_INTERVAL)
1037
+ yield frame
1038
+ except queue.Empty: # pragma: NO COVER
1039
+ continue
1040
+
1041
+ # Return any remaining values after the workers finished.
1042
+ while True: # pragma: NO COVER
1043
+ try:
1044
+ frame = worker_queue.get_nowait()
1045
+ yield frame
1046
+ except queue.Empty: # pragma: NO COVER
1047
+ break
1048
+ finally:
1049
+ # No need for a lock because reading/replacing a variable is
1050
+ # defined to be an atomic operation in the Python language
1051
+ # definition (enforced by the global interpreter lock).
1052
+ download_state.done = True
1053
+
1054
+ # Shutdown all background threads, now that they should know to
1055
+ # exit early.
1056
+ pool.shutdown(wait=True)
1057
+
1058
+
1059
+ def download_arrow_bqstorage(
1060
+ project_id,
1061
+ table,
1062
+ bqstorage_client,
1063
+ preserve_order=False,
1064
+ selected_fields=None,
1065
+ max_queue_size=_MAX_QUEUE_SIZE_DEFAULT,
1066
+ max_stream_count=None,
1067
+ ):
1068
+ return _download_table_bqstorage(
1069
+ project_id,
1070
+ table,
1071
+ bqstorage_client,
1072
+ preserve_order=preserve_order,
1073
+ selected_fields=selected_fields,
1074
+ page_to_item=_bqstorage_page_to_arrow,
1075
+ max_queue_size=max_queue_size,
1076
+ max_stream_count=max_stream_count,
1077
+ )
1078
+
1079
+
1080
+ def download_dataframe_bqstorage(
1081
+ project_id,
1082
+ table,
1083
+ bqstorage_client,
1084
+ column_names,
1085
+ dtypes,
1086
+ preserve_order=False,
1087
+ selected_fields=None,
1088
+ max_queue_size=_MAX_QUEUE_SIZE_DEFAULT,
1089
+ max_stream_count=None,
1090
+ ):
1091
+ page_to_item = functools.partial(_bqstorage_page_to_dataframe, column_names, dtypes)
1092
+ return _download_table_bqstorage(
1093
+ project_id,
1094
+ table,
1095
+ bqstorage_client,
1096
+ preserve_order=preserve_order,
1097
+ selected_fields=selected_fields,
1098
+ page_to_item=page_to_item,
1099
+ max_queue_size=max_queue_size,
1100
+ max_stream_count=max_stream_count,
1101
+ )
1102
+
1103
+
1104
+ def dataframe_to_json_generator(dataframe):
1105
+ for row in dataframe.itertuples(index=False, name=None):
1106
+ output = {}
1107
+ for column, value in zip(dataframe.columns, row):
1108
+ # Omit NaN values.
1109
+ is_nan = pandas.isna(value)
1110
+
1111
+ # isna() can also return an array-like of bools, but the latter's boolean
1112
+ # value is ambiguous, hence an extra check. An array-like value is *not*
1113
+ # considered a NaN, however.
1114
+ if isinstance(is_nan, bool) and is_nan:
1115
+ continue
1116
+
1117
+ # Convert numpy types to corresponding Python types.
1118
+ # https://stackoverflow.com/a/60441783/101923
1119
+ if isinstance(value, numpy.bool_):
1120
+ value = bool(value)
1121
+ elif isinstance(
1122
+ value,
1123
+ (
1124
+ numpy.int64,
1125
+ numpy.int32,
1126
+ numpy.int16,
1127
+ numpy.int8,
1128
+ numpy.uint64,
1129
+ numpy.uint32,
1130
+ numpy.uint16,
1131
+ numpy.uint8,
1132
+ ),
1133
+ ):
1134
+ value = int(value)
1135
+ output[column] = value
1136
+
1137
+ yield output
1138
+
1139
+
1140
+ def verify_pandas_imports():
1141
+ if pandas is None:
1142
+ raise ValueError(_NO_PANDAS_ERROR) from pandas_import_exception
1143
+ if db_dtypes is None:
1144
+ raise ValueError(_NO_DB_TYPES_ERROR) from db_dtypes_import_exception
1145
+
1146
+
1147
+ def determine_requested_streams(
1148
+ preserve_order: bool,
1149
+ max_stream_count: Union[int, None],
1150
+ ) -> int:
1151
+ """Determines the value of requested_streams based on the values of
1152
+ `preserve_order` and `max_stream_count`.
1153
+
1154
+ Args:
1155
+ preserve_order (bool): Whether to preserve the order of streams. If True,
1156
+ this limits the number of streams to one. `preserve_order` takes
1157
+ precedence over `max_stream_count`.
1158
+ max_stream_count (Union[int, None]]): The maximum number of streams
1159
+ allowed. Must be a non-negative number or None, where None indicates
1160
+ the value is unset. NOTE: if `preserve_order` is also set, it takes
1161
+ precedence over `max_stream_count`, thus to ensure that `max_stream_count`
1162
+ is used, ensure that `preserve_order` is None.
1163
+
1164
+ Returns:
1165
+ (int) The appropriate value for requested_streams.
1166
+ """
1167
+
1168
+ if preserve_order:
1169
+ # If preserve order is set, it takes precendence.
1170
+ # Limit the requested streams to 1, to ensure that order
1171
+ # is preserved)
1172
+ return 1
1173
+
1174
+ elif max_stream_count is not None:
1175
+ # If preserve_order is not set, only then do we consider max_stream_count
1176
+ if max_stream_count <= -1:
1177
+ raise ValueError("max_stream_count must be non-negative OR None")
1178
+ return max_stream_count
1179
+
1180
+ # Default to zero requested streams (unbounded).
1181
+ return 0