google-cloud-bigquery 3.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. google/cloud/bigquery/__init__.py +249 -0
  2. google/cloud/bigquery/_helpers.py +1102 -0
  3. google/cloud/bigquery/_http.py +47 -0
  4. google/cloud/bigquery/_job_helpers.py +600 -0
  5. google/cloud/bigquery/_pandas_helpers.py +1181 -0
  6. google/cloud/bigquery/_pyarrow_helpers.py +147 -0
  7. google/cloud/bigquery/_tqdm_helpers.py +137 -0
  8. google/cloud/bigquery/_versions_helpers.py +264 -0
  9. google/cloud/bigquery/client.py +4406 -0
  10. google/cloud/bigquery/dataset.py +1076 -0
  11. google/cloud/bigquery/dbapi/__init__.py +87 -0
  12. google/cloud/bigquery/dbapi/_helpers.py +522 -0
  13. google/cloud/bigquery/dbapi/connection.py +128 -0
  14. google/cloud/bigquery/dbapi/cursor.py +586 -0
  15. google/cloud/bigquery/dbapi/exceptions.py +58 -0
  16. google/cloud/bigquery/dbapi/types.py +96 -0
  17. google/cloud/bigquery/encryption_configuration.py +84 -0
  18. google/cloud/bigquery/enums.py +389 -0
  19. google/cloud/bigquery/exceptions.py +35 -0
  20. google/cloud/bigquery/external_config.py +1188 -0
  21. google/cloud/bigquery/format_options.py +147 -0
  22. google/cloud/bigquery/iam.py +38 -0
  23. google/cloud/bigquery/job/__init__.py +87 -0
  24. google/cloud/bigquery/job/base.py +1116 -0
  25. google/cloud/bigquery/job/copy_.py +282 -0
  26. google/cloud/bigquery/job/extract.py +271 -0
  27. google/cloud/bigquery/job/load.py +985 -0
  28. google/cloud/bigquery/job/query.py +2498 -0
  29. google/cloud/bigquery/magics/__init__.py +20 -0
  30. google/cloud/bigquery/magics/line_arg_parser/__init__.py +34 -0
  31. google/cloud/bigquery/magics/line_arg_parser/exceptions.py +25 -0
  32. google/cloud/bigquery/magics/line_arg_parser/lexer.py +200 -0
  33. google/cloud/bigquery/magics/line_arg_parser/parser.py +484 -0
  34. google/cloud/bigquery/magics/line_arg_parser/visitors.py +159 -0
  35. google/cloud/bigquery/magics/magics.py +776 -0
  36. google/cloud/bigquery/model.py +517 -0
  37. google/cloud/bigquery/opentelemetry_tracing.py +164 -0
  38. google/cloud/bigquery/py.typed +2 -0
  39. google/cloud/bigquery/query.py +1344 -0
  40. google/cloud/bigquery/retry.py +207 -0
  41. google/cloud/bigquery/routine/__init__.py +33 -0
  42. google/cloud/bigquery/routine/routine.py +744 -0
  43. google/cloud/bigquery/schema.py +896 -0
  44. google/cloud/bigquery/standard_sql.py +389 -0
  45. google/cloud/bigquery/table.py +3594 -0
  46. google/cloud/bigquery/version.py +15 -0
  47. google/cloud/bigquery_v2/__init__.py +56 -0
  48. google/cloud/bigquery_v2/types/__init__.py +54 -0
  49. google/cloud/bigquery_v2/types/encryption_config.py +48 -0
  50. google/cloud/bigquery_v2/types/model.py +1994 -0
  51. google/cloud/bigquery_v2/types/model_reference.py +57 -0
  52. google/cloud/bigquery_v2/types/standard_sql.py +156 -0
  53. google/cloud/bigquery_v2/types/table_reference.py +80 -0
  54. google_cloud_bigquery-3.31.0.dist-info/LICENSE +202 -0
  55. google_cloud_bigquery-3.31.0.dist-info/METADATA +203 -0
  56. google_cloud_bigquery-3.31.0.dist-info/RECORD +58 -0
  57. google_cloud_bigquery-3.31.0.dist-info/WHEEL +5 -0
  58. google_cloud_bigquery-3.31.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,47 @@
1
+ # Copyright 2015 Google LLC
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Create / interact with Google BigQuery connections."""
16
+
17
+ from google.cloud import _http # type: ignore # pytype: disable=import-error
18
+ from google.cloud.bigquery import __version__
19
+
20
+
21
+ class Connection(_http.JSONConnection):
22
+ """A connection to Google BigQuery via the JSON REST API.
23
+
24
+ Args:
25
+ client (google.cloud.bigquery.client.Client): The client that owns the current connection.
26
+
27
+ client_info (Optional[google.api_core.client_info.ClientInfo]): Instance used to generate user agent.
28
+
29
+ api_endpoint (str): The api_endpoint to use. If None, the library will decide what endpoint to use.
30
+ """
31
+
32
+ DEFAULT_API_ENDPOINT = "https://bigquery.googleapis.com"
33
+ DEFAULT_API_MTLS_ENDPOINT = "https://bigquery.mtls.googleapis.com"
34
+
35
+ def __init__(self, client, client_info=None, api_endpoint=None):
36
+ super(Connection, self).__init__(client, client_info)
37
+ self.API_BASE_URL = api_endpoint or self.DEFAULT_API_ENDPOINT
38
+ self.API_BASE_MTLS_URL = self.DEFAULT_API_MTLS_ENDPOINT
39
+ self.ALLOW_AUTO_SWITCH_TO_MTLS_URL = api_endpoint is None
40
+ self._client_info.gapic_version = __version__
41
+ self._client_info.client_library_version = __version__
42
+
43
+ API_VERSION = "v2" # type: ignore
44
+ """The version of the API, used in building the API call's URL."""
45
+
46
+ API_URL_TEMPLATE = "{api_base_url}/bigquery/{api_version}{path}" # type: ignore
47
+ """A template for the URL of a particular API call."""
@@ -0,0 +1,600 @@
1
+ # Copyright 2021 Google LLC
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Helpers for interacting with the job REST APIs from the client.
16
+
17
+ For queries, there are three cases to consider:
18
+
19
+ 1. jobs.insert: This always returns a job resource.
20
+ 2. jobs.query, jobCreationMode=JOB_CREATION_REQUIRED:
21
+ This sometimes can return the results inline, but always includes a job ID.
22
+ 3. jobs.query, jobCreationMode=JOB_CREATION_OPTIONAL:
23
+ This sometimes doesn't create a job at all, instead returning the results.
24
+ For better debugging, an auto-generated query ID is included in the
25
+ response.
26
+
27
+ Client.query() calls either (1) or (2), depending on what the user provides
28
+ for the api_method parameter. query() always returns a QueryJob object, which
29
+ can retry the query when the query job fails for a retriable reason.
30
+
31
+ Client.query_and_wait() calls (3). This returns a RowIterator that may wrap
32
+ local results from the response or may wrap a query job containing multiple
33
+ pages of results. Even though query_and_wait() waits for the job to complete,
34
+ we still need a separate job_retry object because there are different
35
+ predicates where it is safe to generate a new query ID.
36
+ """
37
+
38
+ import copy
39
+ import functools
40
+ import os
41
+ import uuid
42
+ from typing import Any, Dict, Optional, TYPE_CHECKING, Union
43
+
44
+ import google.api_core.exceptions as core_exceptions
45
+ from google.api_core import retry as retries
46
+
47
+ from google.cloud.bigquery import job
48
+ import google.cloud.bigquery.query
49
+ from google.cloud.bigquery import table
50
+ import google.cloud.bigquery.retry
51
+ from google.cloud.bigquery.retry import POLLING_DEFAULT_VALUE
52
+
53
+ # Avoid circular imports
54
+ if TYPE_CHECKING: # pragma: NO COVER
55
+ from google.cloud.bigquery.client import Client
56
+
57
+
58
+ # The purpose of _TIMEOUT_BUFFER_MILLIS is to allow the server-side timeout to
59
+ # happen before the client-side timeout. This is not strictly necessary, as the
60
+ # client retries client-side timeouts, but the hope by making the server-side
61
+ # timeout slightly shorter is that it can save the server from some unncessary
62
+ # processing time.
63
+ #
64
+ # 250 milliseconds is chosen arbitrarily, though should be about the right
65
+ # order of magnitude for network latency and switching delays. It is about the
66
+ # amount of time for light to circumnavigate the world twice.
67
+ _TIMEOUT_BUFFER_MILLIS = 250
68
+
69
+
70
+ def make_job_id(job_id: Optional[str] = None, prefix: Optional[str] = None) -> str:
71
+ """Construct an ID for a new job.
72
+
73
+ Args:
74
+ job_id: the user-provided job ID.
75
+ prefix: the user-provided prefix for a job ID.
76
+
77
+ Returns:
78
+ str: A job ID
79
+ """
80
+ if job_id is not None:
81
+ return job_id
82
+ elif prefix is not None:
83
+ return str(prefix) + str(uuid.uuid4())
84
+ else:
85
+ return str(uuid.uuid4())
86
+
87
+
88
+ def job_config_with_defaults(
89
+ job_config: Optional[job.QueryJobConfig],
90
+ default_job_config: Optional[job.QueryJobConfig],
91
+ ) -> Optional[job.QueryJobConfig]:
92
+ """Create a copy of `job_config`, replacing unset values with those from
93
+ `default_job_config`.
94
+ """
95
+ if job_config is None:
96
+ return default_job_config
97
+
98
+ if default_job_config is None:
99
+ return job_config
100
+
101
+ # Both job_config and default_job_config are not None, so make a copy of
102
+ # job_config merged with default_job_config. Anything already explicitly
103
+ # set on job_config should not be replaced.
104
+ return job_config._fill_from_default(default_job_config)
105
+
106
+
107
+ def query_jobs_insert(
108
+ client: "Client",
109
+ query: str,
110
+ job_config: Optional[job.QueryJobConfig],
111
+ job_id: Optional[str],
112
+ job_id_prefix: Optional[str],
113
+ location: Optional[str],
114
+ project: str,
115
+ retry: Optional[retries.Retry],
116
+ timeout: Optional[float],
117
+ job_retry: Optional[retries.Retry],
118
+ ) -> job.QueryJob:
119
+ """Initiate a query using jobs.insert.
120
+
121
+ See: https://cloud.google.com/bigquery/docs/reference/rest/v2/jobs/insert
122
+ """
123
+ job_id_given = job_id is not None
124
+ job_id_save = job_id
125
+ job_config_save = job_config
126
+
127
+ def do_query():
128
+ # Make a copy now, so that original doesn't get changed by the process
129
+ # below and to facilitate retry
130
+ job_config = copy.deepcopy(job_config_save)
131
+
132
+ job_id = make_job_id(job_id_save, job_id_prefix)
133
+ job_ref = job._JobReference(job_id, project=project, location=location)
134
+ query_job = job.QueryJob(job_ref, query, client=client, job_config=job_config)
135
+
136
+ try:
137
+ query_job._begin(retry=retry, timeout=timeout)
138
+ except core_exceptions.Conflict as create_exc:
139
+ # The thought is if someone is providing their own job IDs and they get
140
+ # their job ID generation wrong, this could end up returning results for
141
+ # the wrong query. We thus only try to recover if job ID was not given.
142
+ if job_id_given:
143
+ raise create_exc
144
+
145
+ try:
146
+ # Sometimes we get a 404 after a Conflict. In this case, we
147
+ # have pretty high confidence that by retrying the 404, we'll
148
+ # (hopefully) eventually recover the job.
149
+ # https://github.com/googleapis/python-bigquery/issues/2134
150
+ #
151
+ # Allow users who want to completely disable retries to
152
+ # continue to do so by setting retry to None.
153
+ get_job_retry = retry
154
+ if retry is not None:
155
+ # TODO(tswast): Amend the user's retry object with allowing
156
+ # 404 to retry when there's a public way to do so.
157
+ # https://github.com/googleapis/python-api-core/issues/796
158
+ get_job_retry = (
159
+ google.cloud.bigquery.retry._DEFAULT_GET_JOB_CONFLICT_RETRY
160
+ )
161
+
162
+ query_job = client.get_job(
163
+ job_id,
164
+ project=project,
165
+ location=location,
166
+ retry=get_job_retry,
167
+ timeout=google.cloud.bigquery.retry.DEFAULT_GET_JOB_TIMEOUT,
168
+ )
169
+ except core_exceptions.GoogleAPIError: # (includes RetryError)
170
+ raise
171
+ else:
172
+ return query_job
173
+ else:
174
+ return query_job
175
+
176
+ # Allow users who want to completely disable retries to
177
+ # continue to do so by setting job_retry to None.
178
+ if job_retry is not None:
179
+ do_query = google.cloud.bigquery.retry._DEFAULT_QUERY_JOB_INSERT_RETRY(do_query)
180
+
181
+ future = do_query()
182
+
183
+ # The future might be in a failed state now, but if it's
184
+ # unrecoverable, we'll find out when we ask for it's result, at which
185
+ # point, we may retry.
186
+ if not job_id_given:
187
+ future._retry_do_query = do_query # in case we have to retry later
188
+ future._job_retry = job_retry
189
+
190
+ return future
191
+
192
+
193
+ def _validate_job_config(request_body: Dict[str, Any], invalid_key: str):
194
+ """Catch common mistakes, such as passing in a *JobConfig object of the
195
+ wrong type.
196
+ """
197
+ if invalid_key in request_body:
198
+ raise ValueError(f"got unexpected key {repr(invalid_key)} in job_config")
199
+
200
+
201
+ def _to_query_request(
202
+ job_config: Optional[job.QueryJobConfig] = None,
203
+ *,
204
+ query: str,
205
+ location: Optional[str] = None,
206
+ timeout: Optional[float] = None,
207
+ ) -> Dict[str, Any]:
208
+ """Transform from Job resource to QueryRequest resource.
209
+
210
+ Most of the keys in job.configuration.query are in common with
211
+ QueryRequest. If any configuration property is set that is not available in
212
+ jobs.query, it will result in a server-side error.
213
+ """
214
+ request_body = copy.copy(job_config.to_api_repr()) if job_config else {}
215
+
216
+ _validate_job_config(request_body, job.CopyJob._JOB_TYPE)
217
+ _validate_job_config(request_body, job.ExtractJob._JOB_TYPE)
218
+ _validate_job_config(request_body, job.LoadJob._JOB_TYPE)
219
+
220
+ # Move query.* properties to top-level.
221
+ query_config_resource = request_body.pop("query", {})
222
+ request_body.update(query_config_resource)
223
+
224
+ # Default to standard SQL.
225
+ request_body.setdefault("useLegacySql", False)
226
+
227
+ # Since jobs.query can return results, ensure we use the lossless timestamp
228
+ # format. See: https://github.com/googleapis/python-bigquery/issues/395
229
+ request_body.setdefault("formatOptions", {})
230
+ request_body["formatOptions"]["useInt64Timestamp"] = True # type: ignore
231
+
232
+ if timeout is not None:
233
+ # Subtract a buffer for context switching, network latency, etc.
234
+ request_body["timeoutMs"] = max(0, int(1000 * timeout) - _TIMEOUT_BUFFER_MILLIS)
235
+
236
+ if location is not None:
237
+ request_body["location"] = location
238
+
239
+ request_body["query"] = query
240
+
241
+ return request_body
242
+
243
+
244
+ def _to_query_job(
245
+ client: "Client",
246
+ query: str,
247
+ request_config: Optional[job.QueryJobConfig],
248
+ query_response: Dict[str, Any],
249
+ ) -> job.QueryJob:
250
+ job_ref_resource = query_response["jobReference"]
251
+ job_ref = job._JobReference._from_api_repr(job_ref_resource)
252
+ query_job = job.QueryJob(job_ref, query, client=client)
253
+ query_job._properties.setdefault("configuration", {})
254
+
255
+ # Not all relevant properties are in the jobs.query response. Populate some
256
+ # expected properties based on the job configuration.
257
+ if request_config is not None:
258
+ query_job._properties["configuration"].update(request_config.to_api_repr())
259
+
260
+ query_job._properties["configuration"].setdefault("query", {})
261
+ query_job._properties["configuration"]["query"]["query"] = query
262
+ query_job._properties["configuration"]["query"].setdefault("useLegacySql", False)
263
+
264
+ query_job._properties.setdefault("statistics", {})
265
+ query_job._properties["statistics"].setdefault("query", {})
266
+ query_job._properties["statistics"]["query"]["cacheHit"] = query_response.get(
267
+ "cacheHit"
268
+ )
269
+ query_job._properties["statistics"]["query"]["schema"] = query_response.get(
270
+ "schema"
271
+ )
272
+ query_job._properties["statistics"]["query"][
273
+ "totalBytesProcessed"
274
+ ] = query_response.get("totalBytesProcessed")
275
+
276
+ # Set errors if any were encountered.
277
+ query_job._properties.setdefault("status", {})
278
+ if "errors" in query_response:
279
+ # Set errors but not errorResult. If there was an error that failed
280
+ # the job, jobs.query behaves like jobs.getQueryResults and returns a
281
+ # non-success HTTP status code.
282
+ errors = query_response["errors"]
283
+ query_job._properties["status"]["errors"] = errors
284
+
285
+ # Avoid an extra call to `getQueryResults` if the query has finished.
286
+ job_complete = query_response.get("jobComplete")
287
+ if job_complete:
288
+ query_job._query_results = google.cloud.bigquery.query._QueryResults(
289
+ query_response
290
+ )
291
+
292
+ # We want job.result() to refresh the job state, so the conversion is
293
+ # always "PENDING", even if the job is finished.
294
+ query_job._properties["status"]["state"] = "PENDING"
295
+
296
+ return query_job
297
+
298
+
299
+ def _to_query_path(project: str) -> str:
300
+ return f"/projects/{project}/queries"
301
+
302
+
303
+ def query_jobs_query(
304
+ client: "Client",
305
+ query: str,
306
+ job_config: Optional[job.QueryJobConfig],
307
+ location: Optional[str],
308
+ project: str,
309
+ retry: retries.Retry,
310
+ timeout: Optional[float],
311
+ job_retry: retries.Retry,
312
+ ) -> job.QueryJob:
313
+ """Initiate a query using jobs.query with jobCreationMode=JOB_CREATION_REQUIRED.
314
+
315
+ See: https://cloud.google.com/bigquery/docs/reference/rest/v2/jobs/query
316
+ """
317
+ path = _to_query_path(project)
318
+ request_body = _to_query_request(
319
+ query=query, job_config=job_config, location=location, timeout=timeout
320
+ )
321
+
322
+ def do_query():
323
+ request_body["requestId"] = make_job_id()
324
+ span_attributes = {"path": path}
325
+ api_response = client._call_api(
326
+ retry,
327
+ span_name="BigQuery.query",
328
+ span_attributes=span_attributes,
329
+ method="POST",
330
+ path=path,
331
+ data=request_body,
332
+ timeout=timeout,
333
+ )
334
+ return _to_query_job(client, query, job_config, api_response)
335
+
336
+ future = do_query()
337
+
338
+ # The future might be in a failed state now, but if it's
339
+ # unrecoverable, we'll find out when we ask for it's result, at which
340
+ # point, we may retry.
341
+ future._retry_do_query = do_query # in case we have to retry later
342
+ future._job_retry = job_retry
343
+
344
+ return future
345
+
346
+
347
+ def query_and_wait(
348
+ client: "Client",
349
+ query: str,
350
+ *,
351
+ job_config: Optional[job.QueryJobConfig],
352
+ location: Optional[str],
353
+ project: str,
354
+ api_timeout: Optional[float] = None,
355
+ wait_timeout: Optional[Union[float, object]] = POLLING_DEFAULT_VALUE,
356
+ retry: Optional[retries.Retry],
357
+ job_retry: Optional[retries.Retry],
358
+ page_size: Optional[int] = None,
359
+ max_results: Optional[int] = None,
360
+ ) -> table.RowIterator:
361
+ """Run the query, wait for it to finish, and return the results.
362
+
363
+ While ``jobCreationMode=JOB_CREATION_OPTIONAL`` is in preview in the
364
+ ``jobs.query`` REST API, use the default ``jobCreationMode`` unless
365
+ the environment variable ``QUERY_PREVIEW_ENABLED=true``. After
366
+ ``jobCreationMode`` is GA, this method will always use
367
+ ``jobCreationMode=JOB_CREATION_OPTIONAL``. See:
368
+ https://cloud.google.com/bigquery/docs/reference/rest/v2/jobs/query
369
+
370
+ Args:
371
+ client:
372
+ BigQuery client to make API calls.
373
+ query (str):
374
+ SQL query to be executed. Defaults to the standard SQL
375
+ dialect. Use the ``job_config`` parameter to change dialects.
376
+ job_config (Optional[google.cloud.bigquery.job.QueryJobConfig]):
377
+ Extra configuration options for the job.
378
+ To override any options that were previously set in
379
+ the ``default_query_job_config`` given to the
380
+ ``Client`` constructor, manually set those options to ``None``,
381
+ or whatever value is preferred.
382
+ location (Optional[str]):
383
+ Location where to run the job. Must match the location of the
384
+ table used in the query as well as the destination table.
385
+ project (Optional[str]):
386
+ Project ID of the project of where to run the job. Defaults
387
+ to the client's project.
388
+ api_timeout (Optional[float]):
389
+ The number of seconds to wait for the underlying HTTP transport
390
+ before using ``retry``.
391
+ wait_timeout (Optional[Union[float, object]]):
392
+ The number of seconds to wait for the query to finish. If the
393
+ query doesn't finish before this timeout, the client attempts
394
+ to cancel the query. If unset, the underlying Client.get_job() API
395
+ call has timeout, but we still wait indefinitely for the job to
396
+ finish.
397
+ retry (Optional[google.api_core.retry.Retry]):
398
+ How to retry the RPC. This only applies to making RPC
399
+ calls. It isn't used to retry failed jobs. This has
400
+ a reasonable default that should only be overridden
401
+ with care.
402
+ job_retry (Optional[google.api_core.retry.Retry]):
403
+ How to retry failed jobs. The default retries
404
+ rate-limit-exceeded errors. Passing ``None`` disables
405
+ job retry. Not all jobs can be retried.
406
+ page_size (Optional[int]):
407
+ The maximum number of rows in each page of results from this
408
+ request. Non-positive values are ignored.
409
+ max_results (Optional[int]):
410
+ The maximum total number of rows from this request.
411
+
412
+ Returns:
413
+ google.cloud.bigquery.table.RowIterator:
414
+ Iterator of row data
415
+ :class:`~google.cloud.bigquery.table.Row`-s. During each
416
+ page, the iterator will have the ``total_rows`` attribute
417
+ set, which counts the total number of rows **in the result
418
+ set** (this is distinct from the total number of rows in the
419
+ current page: ``iterator.page.num_items``).
420
+
421
+ If the query is a special query that produces no results, e.g.
422
+ a DDL query, an ``_EmptyRowIterator`` instance is returned.
423
+
424
+ Raises:
425
+ TypeError:
426
+ If ``job_config`` is not an instance of
427
+ :class:`~google.cloud.bigquery.job.QueryJobConfig`
428
+ class.
429
+ """
430
+ request_body = _to_query_request(
431
+ query=query, job_config=job_config, location=location, timeout=api_timeout
432
+ )
433
+
434
+ # Some API parameters aren't supported by the jobs.query API. In these
435
+ # cases, fallback to a jobs.insert call.
436
+ if not _supported_by_jobs_query(request_body):
437
+ return _wait_or_cancel(
438
+ query_jobs_insert(
439
+ client=client,
440
+ query=query,
441
+ job_id=None,
442
+ job_id_prefix=None,
443
+ job_config=job_config,
444
+ location=location,
445
+ project=project,
446
+ retry=retry,
447
+ timeout=api_timeout,
448
+ job_retry=job_retry,
449
+ ),
450
+ api_timeout=api_timeout,
451
+ wait_timeout=wait_timeout,
452
+ retry=retry,
453
+ page_size=page_size,
454
+ max_results=max_results,
455
+ )
456
+
457
+ path = _to_query_path(project)
458
+
459
+ if page_size is not None and max_results is not None:
460
+ request_body["maxResults"] = min(page_size, max_results)
461
+ elif page_size is not None or max_results is not None:
462
+ request_body["maxResults"] = page_size or max_results
463
+
464
+ if os.getenv("QUERY_PREVIEW_ENABLED", "").casefold() == "true":
465
+ request_body["jobCreationMode"] = "JOB_CREATION_OPTIONAL"
466
+
467
+ def do_query():
468
+ request_body["requestId"] = make_job_id()
469
+ span_attributes = {"path": path}
470
+
471
+ # For easier testing, handle the retries ourselves.
472
+ if retry is not None:
473
+ response = retry(client._call_api)(
474
+ retry=None, # We're calling the retry decorator ourselves.
475
+ span_name="BigQuery.query",
476
+ span_attributes=span_attributes,
477
+ method="POST",
478
+ path=path,
479
+ data=request_body,
480
+ timeout=api_timeout,
481
+ )
482
+ else:
483
+ response = client._call_api(
484
+ retry=None,
485
+ span_name="BigQuery.query",
486
+ span_attributes=span_attributes,
487
+ method="POST",
488
+ path=path,
489
+ data=request_body,
490
+ timeout=api_timeout,
491
+ )
492
+
493
+ # Even if we run with JOB_CREATION_OPTIONAL, if there are more pages
494
+ # to fetch, there will be a job ID for jobs.getQueryResults.
495
+ query_results = google.cloud.bigquery.query._QueryResults.from_api_repr(
496
+ response
497
+ )
498
+ page_token = query_results.page_token
499
+ more_pages = page_token is not None
500
+
501
+ if more_pages or not query_results.complete:
502
+ # TODO(swast): Avoid a call to jobs.get in some cases (few
503
+ # remaining pages) by waiting for the query to finish and calling
504
+ # client._list_rows_from_query_results directly. Need to update
505
+ # RowIterator to fetch destination table via the job ID if needed.
506
+ return _wait_or_cancel(
507
+ _to_query_job(client, query, job_config, response),
508
+ api_timeout=api_timeout,
509
+ wait_timeout=wait_timeout,
510
+ retry=retry,
511
+ page_size=page_size,
512
+ max_results=max_results,
513
+ )
514
+
515
+ return table.RowIterator(
516
+ client=client,
517
+ api_request=functools.partial(client._call_api, retry, timeout=api_timeout),
518
+ path=None,
519
+ schema=query_results.schema,
520
+ max_results=max_results,
521
+ page_size=page_size,
522
+ total_rows=query_results.total_rows,
523
+ first_page_response=response,
524
+ location=query_results.location,
525
+ job_id=query_results.job_id,
526
+ query_id=query_results.query_id,
527
+ project=query_results.project,
528
+ num_dml_affected_rows=query_results.num_dml_affected_rows,
529
+ query=query,
530
+ total_bytes_processed=query_results.total_bytes_processed,
531
+ )
532
+
533
+ if job_retry is not None:
534
+ return job_retry(do_query)()
535
+ else:
536
+ return do_query()
537
+
538
+
539
+ def _supported_by_jobs_query(request_body: Dict[str, Any]) -> bool:
540
+ """True if jobs.query can be used. False if jobs.insert is needed."""
541
+ request_keys = frozenset(request_body.keys())
542
+
543
+ # Per issue: https://github.com/googleapis/python-bigquery/issues/1867
544
+ # use an allowlist here instead of a denylist because the backend API allows
545
+ # unsupported parameters without any warning or failure. Instead, keep this
546
+ # set in sync with those in QueryRequest:
547
+ # https://cloud.google.com/bigquery/docs/reference/rest/v2/jobs/query#QueryRequest
548
+ keys_allowlist = {
549
+ "kind",
550
+ "query",
551
+ "maxResults",
552
+ "defaultDataset",
553
+ "timeoutMs",
554
+ "dryRun",
555
+ "preserveNulls",
556
+ "useQueryCache",
557
+ "useLegacySql",
558
+ "parameterMode",
559
+ "queryParameters",
560
+ "location",
561
+ "formatOptions",
562
+ "connectionProperties",
563
+ "labels",
564
+ "maximumBytesBilled",
565
+ "requestId",
566
+ "createSession",
567
+ }
568
+
569
+ unsupported_keys = request_keys - keys_allowlist
570
+ return len(unsupported_keys) == 0
571
+
572
+
573
+ def _wait_or_cancel(
574
+ job: job.QueryJob,
575
+ api_timeout: Optional[float],
576
+ wait_timeout: Optional[Union[object, float]],
577
+ retry: Optional[retries.Retry],
578
+ page_size: Optional[int],
579
+ max_results: Optional[int],
580
+ ) -> table.RowIterator:
581
+ """Wait for a job to complete and return the results.
582
+
583
+ If we can't return the results within the ``wait_timeout``, try to cancel
584
+ the job.
585
+ """
586
+ try:
587
+ return job.result(
588
+ page_size=page_size,
589
+ max_results=max_results,
590
+ retry=retry,
591
+ timeout=wait_timeout,
592
+ )
593
+ except Exception:
594
+ # Attempt to cancel the job since we can't return the results.
595
+ try:
596
+ job.cancel(retry=retry, timeout=api_timeout)
597
+ except Exception:
598
+ # Don't eat the original exception if cancel fails.
599
+ pass
600
+ raise