turbine-api 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- turbine/api/__init__.py +1 -0
- turbine/api/app.py +658 -0
- turbine/api/cursor.py +141 -0
- turbine/api/extension.py +31 -0
- turbine/api/lint.py +311 -0
- turbine/api/pool.py +97 -0
- turbine/api/publication.py +287 -0
- turbine/api/query.py +397 -0
- turbine_api-1.0.0.dist-info/METADATA +27 -0
- turbine_api-1.0.0.dist-info/RECORD +12 -0
- turbine_api-1.0.0.dist-info/WHEEL +4 -0
- turbine_api-1.0.0.dist-info/entry_points.txt +2 -0
turbine/api/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Data Product API Extension: serve the published Datasets over HTTP."""
|
turbine/api/app.py
ADDED
|
@@ -0,0 +1,658 @@
|
|
|
1
|
+
"""Build the Litestar application that serves the Data Product API.
|
|
2
|
+
|
|
3
|
+
One route exists for each published Dataset::
|
|
4
|
+
|
|
5
|
+
project --> products --> API servers --> datasets --> inspection
|
|
6
|
+
--> PublishedDataset --> route handler --> JSON page
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import inspect
|
|
13
|
+
import logging
|
|
14
|
+
from dataclasses import replace
|
|
15
|
+
from http import HTTPStatus
|
|
16
|
+
from types import GenericAlias
|
|
17
|
+
from typing import Annotated, Any, Literal
|
|
18
|
+
|
|
19
|
+
import msgspec
|
|
20
|
+
import pyarrow as pa
|
|
21
|
+
from litestar import Litestar, Request, Response, get
|
|
22
|
+
from litestar.exceptions import HTTPException
|
|
23
|
+
from litestar.openapi.config import OpenAPIConfig
|
|
24
|
+
from litestar.openapi.datastructures import ResponseSpec
|
|
25
|
+
from litestar.openapi.plugins import (
|
|
26
|
+
JsonRenderPlugin,
|
|
27
|
+
RapidocRenderPlugin,
|
|
28
|
+
RedocRenderPlugin,
|
|
29
|
+
ScalarRenderPlugin,
|
|
30
|
+
StoplightRenderPlugin,
|
|
31
|
+
SwaggerRenderPlugin,
|
|
32
|
+
YamlRenderPlugin,
|
|
33
|
+
)
|
|
34
|
+
from litestar.openapi.spec import Server as OpenAPIServer
|
|
35
|
+
from litestar.params import QueryParameter
|
|
36
|
+
from litestar.plugins.problem_details import (
|
|
37
|
+
ProblemDetailsConfig,
|
|
38
|
+
ProblemDetailsException,
|
|
39
|
+
ProblemDetailsPlugin,
|
|
40
|
+
)
|
|
41
|
+
from litestar.types import Scope
|
|
42
|
+
from sqlglot import exp
|
|
43
|
+
from turbine.api.cursor import encode_cursor
|
|
44
|
+
from turbine.api.extension import MOUNT_PATH
|
|
45
|
+
from turbine.api.lint import (
|
|
46
|
+
published_columns_have_types,
|
|
47
|
+
published_dataset_has_primary_key,
|
|
48
|
+
publishes_api,
|
|
49
|
+
)
|
|
50
|
+
from turbine.api.pool import PooledConnectionOpener
|
|
51
|
+
from turbine.api.publication import PublishedColumn, PublishedDataset, dataset_path
|
|
52
|
+
from turbine.api.query import META_PARAMETERS, PageQuery, filter_parameter
|
|
53
|
+
|
|
54
|
+
from turbine.core.sdk import (
|
|
55
|
+
Finding,
|
|
56
|
+
ProductIdentity,
|
|
57
|
+
ServeContext,
|
|
58
|
+
ServeDatasetInspection,
|
|
59
|
+
ServeDatasource,
|
|
60
|
+
ServerNotSelectedError,
|
|
61
|
+
Severity,
|
|
62
|
+
TextLocation,
|
|
63
|
+
TurbineError,
|
|
64
|
+
)
|
|
65
|
+
from turbine.core.sdk.datasource import DriverFault
|
|
66
|
+
from turbine.core.sdk.lint import RuleOutcome
|
|
67
|
+
from turbine.core.sdk.nodes import (
|
|
68
|
+
Column,
|
|
69
|
+
ContributionRole,
|
|
70
|
+
DataProduct,
|
|
71
|
+
Dataset,
|
|
72
|
+
Server,
|
|
73
|
+
SlaPropertyKind,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
DEFAULT_PAGE_SIZE = 100
|
|
77
|
+
MAX_PAGE_SIZE = 100_000
|
|
78
|
+
ENCODE_CHUNK_ROWS = 65_536
|
|
79
|
+
"""Rows encoded per Arrow batch, so one page never holds every row as Python objects."""
|
|
80
|
+
PUBLICATION_RULES = (published_columns_have_types, published_dataset_has_primary_key)
|
|
81
|
+
"""The lint rules that a Dataset must pass before the Data API publishes it.
|
|
82
|
+
|
|
83
|
+
`turbine lint` and the editor show the same findings before the server starts.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class QueryProblem(msgspec.Struct):
|
|
88
|
+
"""Body of a 400 response, as problem details."""
|
|
89
|
+
|
|
90
|
+
status: Literal[400]
|
|
91
|
+
title: str
|
|
92
|
+
detail: str
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class PageLinks(msgspec.Struct):
|
|
96
|
+
"""Links section of a page response."""
|
|
97
|
+
|
|
98
|
+
next: str | None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _problem_details_from_http_exception(error: HTTPException) -> ProblemDetailsException:
|
|
102
|
+
"""Convert an HTTP error into problem details.
|
|
103
|
+
|
|
104
|
+
The conversion renames the internal key 'page_size' to 'page[size]', so
|
|
105
|
+
the response speaks the names of the public query parameters.
|
|
106
|
+
"""
|
|
107
|
+
extra = error.extra
|
|
108
|
+
if isinstance(extra, list):
|
|
109
|
+
extra = [
|
|
110
|
+
{**item, "key": "page[size]"}
|
|
111
|
+
if isinstance(item, dict) and item.get("key") == "page_size"
|
|
112
|
+
else item
|
|
113
|
+
for item in extra
|
|
114
|
+
]
|
|
115
|
+
return ProblemDetailsException(
|
|
116
|
+
status_code=error.status_code,
|
|
117
|
+
title=HTTPStatus(error.status_code).phrase,
|
|
118
|
+
detail=str(error.detail),
|
|
119
|
+
extra=extra,
|
|
120
|
+
headers=error.headers,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class ProductApiScalarRenderPlugin(ScalarRenderPlugin):
|
|
125
|
+
"""Scalar documentation plugin that knows the mount path."""
|
|
126
|
+
|
|
127
|
+
@staticmethod
|
|
128
|
+
def get_openapi_json_route(request: Request[Any, Any, Any]) -> str:
|
|
129
|
+
"""Give the URL of the OpenAPI document below the mount path."""
|
|
130
|
+
del request
|
|
131
|
+
return f"{MOUNT_PATH}/docs/openapi.json"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
async def _log_request_exception(error: Exception, _scope: Scope) -> None:
|
|
135
|
+
"""Log a server error. Client errors make no log record."""
|
|
136
|
+
if isinstance(error, HTTPException) and error.status_code < HTTPStatus.INTERNAL_SERVER_ERROR:
|
|
137
|
+
return
|
|
138
|
+
logging.getLogger(__name__).debug(
|
|
139
|
+
"Data API request failed", exc_info=(type(error), error, error.__traceback__)
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def create_data_api(context: ServeContext) -> Litestar:
|
|
144
|
+
"""Build the Data API application for the project.
|
|
145
|
+
|
|
146
|
+
Parameters
|
|
147
|
+
----------
|
|
148
|
+
context : ServeContext
|
|
149
|
+
The project, the datasource bindings, and the output channels.
|
|
150
|
+
|
|
151
|
+
Returns
|
|
152
|
+
-------
|
|
153
|
+
Litestar
|
|
154
|
+
An application with one route for each published Dataset.
|
|
155
|
+
|
|
156
|
+
Raises
|
|
157
|
+
------
|
|
158
|
+
TurbineError
|
|
159
|
+
If two Data Products use the same API path.
|
|
160
|
+
"""
|
|
161
|
+
pool = PooledConnectionOpener(context.connection_opener)
|
|
162
|
+
resources = _published_datasets(context)
|
|
163
|
+
resources_by_path: dict[str, PublishedDataset] = {}
|
|
164
|
+
for resource in resources:
|
|
165
|
+
previous = resources_by_path.get(resource.path)
|
|
166
|
+
if previous is not None:
|
|
167
|
+
raise TurbineError(
|
|
168
|
+
f"Data Products {previous.product.key} and {resource.product.key} use "
|
|
169
|
+
f"the same API path {resource.path!r}. Give the products different major versions. "
|
|
170
|
+
"Or disable publication for one version. Remove its API Server from the Data "
|
|
171
|
+
"Contract, or set api: false at the top level of the standalone Quality Spec."
|
|
172
|
+
)
|
|
173
|
+
resources_by_path[resource.path] = resource
|
|
174
|
+
handlers = [_route(resource, context, pool) for resource in resources]
|
|
175
|
+
app = Litestar(
|
|
176
|
+
route_handlers=handlers,
|
|
177
|
+
logging_config=None,
|
|
178
|
+
openapi_config=OpenAPIConfig(
|
|
179
|
+
title="Turbine Data Product API",
|
|
180
|
+
version="v1",
|
|
181
|
+
path="/docs",
|
|
182
|
+
servers=[OpenAPIServer(url=MOUNT_PATH)],
|
|
183
|
+
render_plugins=(
|
|
184
|
+
ProductApiScalarRenderPlugin(),
|
|
185
|
+
StoplightRenderPlugin(),
|
|
186
|
+
SwaggerRenderPlugin(),
|
|
187
|
+
RedocRenderPlugin(),
|
|
188
|
+
RapidocRenderPlugin(),
|
|
189
|
+
JsonRenderPlugin(),
|
|
190
|
+
YamlRenderPlugin(),
|
|
191
|
+
),
|
|
192
|
+
),
|
|
193
|
+
plugins=[
|
|
194
|
+
ProblemDetailsPlugin(
|
|
195
|
+
ProblemDetailsConfig(
|
|
196
|
+
exception_to_problem_detail_map={HTTPException: _problem_details_from_http_exception}
|
|
197
|
+
)
|
|
198
|
+
)
|
|
199
|
+
],
|
|
200
|
+
after_exception=[_log_request_exception],
|
|
201
|
+
on_shutdown=[pool.close_all],
|
|
202
|
+
)
|
|
203
|
+
base_url = f"{context.base_url}{MOUNT_PATH}"
|
|
204
|
+
context.announce(f"\nOpen the Data API documentation in a browser:\n {base_url}/docs\n")
|
|
205
|
+
for resource in resources:
|
|
206
|
+
fields = ", ".join(sorted(resource.queryable)) or "none"
|
|
207
|
+
context.announce(f"Dataset URL: {base_url}{resource.path}\n Filterable fields: {fields}")
|
|
208
|
+
if not resources:
|
|
209
|
+
context.announce(
|
|
210
|
+
"The Data API has no published Datasets.\n"
|
|
211
|
+
"Read the publication warnings. Run turbine serve --help for configuration instructions."
|
|
212
|
+
)
|
|
213
|
+
return app
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _published_datasets(context: ServeContext) -> tuple[PublishedDataset, ...]:
|
|
217
|
+
"""Collect the published datasets of all Data Products."""
|
|
218
|
+
return tuple(
|
|
219
|
+
resource
|
|
220
|
+
for product in context.project.data_products
|
|
221
|
+
for resource in _published_product(product, context)
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _published_product(product: DataProduct, context: ServeContext) -> tuple[PublishedDataset, ...]:
|
|
226
|
+
"""Collect the published datasets of one Data Product.
|
|
227
|
+
|
|
228
|
+
A product that cannot connect makes a warning and no route. The other
|
|
229
|
+
products stay available.
|
|
230
|
+
"""
|
|
231
|
+
api_servers = _api_servers(product, context.environment)
|
|
232
|
+
datasets = _api_datasets(product, api_servers)
|
|
233
|
+
if not datasets:
|
|
234
|
+
return ()
|
|
235
|
+
product = replace(product, datasets=datasets)
|
|
236
|
+
try:
|
|
237
|
+
inspection = context.inspect(product, api_servers)
|
|
238
|
+
return tuple(
|
|
239
|
+
resource
|
|
240
|
+
for dataset in inspection.datasets
|
|
241
|
+
if (resource := _validated_resource(product, dataset, inspection.datasource, context))
|
|
242
|
+
is not None
|
|
243
|
+
)
|
|
244
|
+
except ServerNotSelectedError as error:
|
|
245
|
+
# No connection was opened, so the advice about connections does not apply.
|
|
246
|
+
_report(
|
|
247
|
+
context,
|
|
248
|
+
product,
|
|
249
|
+
"turbine-api.serve.server-not-selected",
|
|
250
|
+
Severity.WARNING,
|
|
251
|
+
f"Data Product '{product.key.id}' is not published. Cause: {error}",
|
|
252
|
+
)
|
|
253
|
+
return ()
|
|
254
|
+
except Exception as error: # noqa: BLE001 - one failed product must not stop siblings.
|
|
255
|
+
_report(
|
|
256
|
+
context,
|
|
257
|
+
product,
|
|
258
|
+
"turbine-api.serve.connection-failed",
|
|
259
|
+
Severity.WARNING,
|
|
260
|
+
f"Data Product '{product.key.id}' is not published. Cause: {error}\n"
|
|
261
|
+
"Run turbine status. Check the Server bindings and Datasource connections.",
|
|
262
|
+
# A Turbine error tells the cause. Turbine did not expect any other error.
|
|
263
|
+
report_bug=not isinstance(error, TurbineError),
|
|
264
|
+
)
|
|
265
|
+
return ()
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _api_datasets(product: DataProduct, api_servers: tuple[Server, ...]) -> tuple[Dataset, ...]:
|
|
269
|
+
"""Give the Datasets that the product offers to the API."""
|
|
270
|
+
if product.role is ContributionRole.STANDALONE_QUALITY_SPEC:
|
|
271
|
+
return product.datasets if product.api_enabled is True else ()
|
|
272
|
+
if product.role is not ContributionRole.DATA_CONTRACT or not api_servers:
|
|
273
|
+
return ()
|
|
274
|
+
return product.datasets
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _api_servers(product: DataProduct, environment: str | None) -> tuple[Server, ...]:
|
|
278
|
+
"""Give the API Servers of the product for an environment."""
|
|
279
|
+
return tuple(server for server in product.metadata.servers_in(environment) if publishes_api(server))
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _validated_resource(
|
|
283
|
+
product: DataProduct,
|
|
284
|
+
inspection: ServeDatasetInspection,
|
|
285
|
+
datasource: ServeDatasource,
|
|
286
|
+
context: ServeContext,
|
|
287
|
+
) -> PublishedDataset | None:
|
|
288
|
+
"""Make a published dataset, or report why publication is not possible.
|
|
289
|
+
|
|
290
|
+
A Dataset is published only when it has a name, a database name, public
|
|
291
|
+
types, a primary key, and a schema that matches the live table.
|
|
292
|
+
|
|
293
|
+
Returns
|
|
294
|
+
-------
|
|
295
|
+
PublishedDataset or None
|
|
296
|
+
The resource, or None when the Dataset is not published.
|
|
297
|
+
"""
|
|
298
|
+
dataset = inspection.dataset
|
|
299
|
+
if not isinstance(dataset.name, str) or dataset.database_name is None:
|
|
300
|
+
return None
|
|
301
|
+
for rule in PUBLICATION_RULES:
|
|
302
|
+
outcome = rule(dataset, product)
|
|
303
|
+
if isinstance(outcome, RuleOutcome):
|
|
304
|
+
_report(
|
|
305
|
+
context, product, rule.rule_id, Severity.WARNING, outcome.message, source=dataset.source
|
|
306
|
+
)
|
|
307
|
+
return None
|
|
308
|
+
errors = tuple(
|
|
309
|
+
finding for finding in inspection.validation.findings if finding.severity is Severity.ERROR
|
|
310
|
+
)
|
|
311
|
+
if errors:
|
|
312
|
+
_report(
|
|
313
|
+
context,
|
|
314
|
+
product,
|
|
315
|
+
"turbine-api.serve.schema-incompatible",
|
|
316
|
+
Severity.WARNING,
|
|
317
|
+
f"Dataset '{dataset.name}' is not published. "
|
|
318
|
+
+ ". ".join(item.headline.rstrip(".") for item in errors)
|
|
319
|
+
+ ".\nMake the declared columns and the live table columns the same.",
|
|
320
|
+
)
|
|
321
|
+
return None
|
|
322
|
+
columns = tuple(
|
|
323
|
+
_published_column(column, inspection)
|
|
324
|
+
for column in dataset.columns
|
|
325
|
+
if isinstance(column.name, str)
|
|
326
|
+
)
|
|
327
|
+
return PublishedDataset(product, dataset, datasource, inspection.table, columns)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _published_column(column: Column, inspection: ServeDatasetInspection) -> PublishedColumn:
|
|
331
|
+
"""Make a published column with the facts that only the live column gives."""
|
|
332
|
+
live_name = (column.database_name or str(column.name)).casefold()
|
|
333
|
+
return PublishedColumn.from_column(
|
|
334
|
+
column,
|
|
335
|
+
approximate=live_name in inspection.validation.approximate_columns,
|
|
336
|
+
geometry=live_name in inspection.validation.geometry_columns,
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _route(resource: PublishedDataset, context: ServeContext, pool: PooledConnectionOpener) -> Any:
|
|
341
|
+
"""Build the GET route handler of one published dataset.
|
|
342
|
+
|
|
343
|
+
Notes
|
|
344
|
+
-----
|
|
345
|
+
The handler signature is made at runtime, because each dataset has its
|
|
346
|
+
own filter parameters and its own response schema::
|
|
347
|
+
|
|
348
|
+
request --> PageQuery --> SQL --> Arrow table --> JSON page
|
|
349
|
+
--> ETag, Link headers
|
|
350
|
+
"""
|
|
351
|
+
summary = f"List {resource.dataset.name}"
|
|
352
|
+
|
|
353
|
+
def read_dataset(
|
|
354
|
+
request: Request[Any, Any, Any], page_size: int | None = None, **_query: str | None
|
|
355
|
+
) -> Response[bytes]:
|
|
356
|
+
"""Return one page of rows of the dataset."""
|
|
357
|
+
size = DEFAULT_PAGE_SIZE if page_size is None else page_size
|
|
358
|
+
query = PageQuery.build(resource, request.query_params, size)
|
|
359
|
+
table = _query_table(resource, pool, query)
|
|
360
|
+
has_next = table.num_rows > size
|
|
361
|
+
if has_next and all(
|
|
362
|
+
table[name][size - 1].as_py() == table[name][size].as_py() for name in query.sort_fields
|
|
363
|
+
):
|
|
364
|
+
raise HTTPException(
|
|
365
|
+
status_code=409,
|
|
366
|
+
detail="The API cannot return the next page. The pagination key is not unique. "
|
|
367
|
+
"Ask the Data Product owner to check the primary key. "
|
|
368
|
+
"Ask the owner to check for duplicate rows.",
|
|
369
|
+
)
|
|
370
|
+
page = table.slice(0, size)
|
|
371
|
+
next_url = (
|
|
372
|
+
_next_page_url(request, encode_cursor(page, query.sort_fields, query.cursor_context))
|
|
373
|
+
if has_next
|
|
374
|
+
else None
|
|
375
|
+
)
|
|
376
|
+
body = _encode_page(page, resource, query.fields, next_url)
|
|
377
|
+
headers = _headers(resource, context, body, next_url=next_url)
|
|
378
|
+
etag = headers["ETag"]
|
|
379
|
+
if _etag_matches(request.headers.get("If-None-Match"), etag):
|
|
380
|
+
return Response(content=b"", status_code=304, headers=headers)
|
|
381
|
+
return Response(content=body, media_type="application/json", headers=headers)
|
|
382
|
+
|
|
383
|
+
schema_name = _schema_name(resource)
|
|
384
|
+
read_dataset.__name__ = f"read_{schema_name}"
|
|
385
|
+
return_annotation = _response_schema(resource)
|
|
386
|
+
parameters = _openapi_parameters(resource)
|
|
387
|
+
# Litestar reads the handler signature with inspect.signature, which uses `__signature__`.
|
|
388
|
+
# The function type does not declare that attribute, so write it to the function dict.
|
|
389
|
+
read_dataset.__dict__["__signature__"] = inspect.Signature(
|
|
390
|
+
parameters, return_annotation=return_annotation
|
|
391
|
+
)
|
|
392
|
+
read_dataset.__annotations__ = {parameter.name: parameter.annotation for parameter in parameters} | {
|
|
393
|
+
"return": return_annotation
|
|
394
|
+
}
|
|
395
|
+
return get(
|
|
396
|
+
resource.path,
|
|
397
|
+
summary=summary,
|
|
398
|
+
tags=[resource.product.metadata.domain or resource.product.key.id],
|
|
399
|
+
deprecated=resource.product.metadata.end_of_life,
|
|
400
|
+
responses={
|
|
401
|
+
304: ResponseSpec(data_container=None, description="Response has not changed."),
|
|
402
|
+
400: ResponseSpec(
|
|
403
|
+
data_container=QueryProblem,
|
|
404
|
+
media_type="application/problem+json",
|
|
405
|
+
description="Invalid query. Read title for the cause and correction.",
|
|
406
|
+
generate_examples=False,
|
|
407
|
+
),
|
|
408
|
+
},
|
|
409
|
+
sync_to_thread=True,
|
|
410
|
+
)(read_dataset)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _query_table(resource: PublishedDataset, pool: PooledConnectionOpener, query: PageQuery) -> pa.Table:
|
|
414
|
+
"""Run the page query and give the rows.
|
|
415
|
+
|
|
416
|
+
A driver error that the request caused becomes a 400 response. All
|
|
417
|
+
other errors stay as they are.
|
|
418
|
+
"""
|
|
419
|
+
with pool.lease(resource.datasource.name) as connection:
|
|
420
|
+
try:
|
|
421
|
+
return connection.query(
|
|
422
|
+
connection.render_sql(query.select),
|
|
423
|
+
parameters=query.parameters,
|
|
424
|
+
timeout=resource.datasource.connection_timeout,
|
|
425
|
+
)
|
|
426
|
+
except Exception as error:
|
|
427
|
+
fault = connection.classify(error)
|
|
428
|
+
if query.select.find(exp.RegexpLike) is not None and fault is DriverFault.BAD_QUERY:
|
|
429
|
+
raise HTTPException(
|
|
430
|
+
status_code=400,
|
|
431
|
+
detail="The regular expression is invalid or unsupported. "
|
|
432
|
+
"Correct the pattern after '*=' or use an equality filter.",
|
|
433
|
+
) from error
|
|
434
|
+
if query.parameters and (fault is DriverFault.BAD_QUERY or isinstance(error, OverflowError)):
|
|
435
|
+
raise HTTPException(
|
|
436
|
+
status_code=400,
|
|
437
|
+
detail="A filter value cannot be used by this dataset. "
|
|
438
|
+
"Use values that fit the field types in this dataset's API schema.",
|
|
439
|
+
) from error
|
|
440
|
+
raise
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _openapi_parameters(resource: PublishedDataset) -> list[inspect.Parameter]:
|
|
444
|
+
"""Build the handler parameters that OpenAPI documents."""
|
|
445
|
+
request = inspect.Parameter(
|
|
446
|
+
"request", inspect.Parameter.POSITIONAL_OR_KEYWORD, annotation=Request[Any, Any, Any]
|
|
447
|
+
)
|
|
448
|
+
descriptions = {
|
|
449
|
+
"fields": "Return only these comma-separated field names. Use 'none' for primary keys only.",
|
|
450
|
+
"sort": "Sort by comma-separated field names. Prefix a name with '-' for descending order.",
|
|
451
|
+
"page[after]": "Use links.next instead of constructing the next page request.",
|
|
452
|
+
"page[size]": (
|
|
453
|
+
f"Maximum rows per page: 1 through {MAX_PAGE_SIZE}. Default: {DEFAULT_PAGE_SIZE}."
|
|
454
|
+
),
|
|
455
|
+
}
|
|
456
|
+
query_names = (
|
|
457
|
+
*sorted(META_PARAMETERS),
|
|
458
|
+
*(filter_parameter(name) for name in sorted(resource.queryable)),
|
|
459
|
+
)
|
|
460
|
+
query_parameters: list[inspect.Parameter] = []
|
|
461
|
+
for index, name in enumerate(query_names):
|
|
462
|
+
description = descriptions.get(
|
|
463
|
+
name,
|
|
464
|
+
"Filter by a value, comma-separated values, or a value prefixed with !=, >, >=, <, or <=. "
|
|
465
|
+
"Repeat the parameter to give more than one condition. "
|
|
466
|
+
"For string fields only, use '*=' followed by a regular expression.",
|
|
467
|
+
)
|
|
468
|
+
annotation: Any
|
|
469
|
+
if name == "page[size]":
|
|
470
|
+
annotation = Annotated[
|
|
471
|
+
int | None,
|
|
472
|
+
QueryParameter(
|
|
473
|
+
name=name, description=description, required=False, ge=1, le=MAX_PAGE_SIZE
|
|
474
|
+
),
|
|
475
|
+
]
|
|
476
|
+
else:
|
|
477
|
+
annotation = Annotated[
|
|
478
|
+
str | None, QueryParameter(name=name, description=description, required=False)
|
|
479
|
+
]
|
|
480
|
+
parameter_name = "page_size" if name == "page[size]" else f"query_{index}"
|
|
481
|
+
query_parameters.append(
|
|
482
|
+
inspect.Parameter(
|
|
483
|
+
parameter_name, inspect.Parameter.KEYWORD_ONLY, annotation=annotation, default=None
|
|
484
|
+
)
|
|
485
|
+
)
|
|
486
|
+
return [request, *query_parameters]
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _response_schema(resource: PublishedDataset) -> type[msgspec.Struct]:
|
|
490
|
+
"""Build the response type of one dataset.
|
|
491
|
+
|
|
492
|
+
Notes
|
|
493
|
+
-----
|
|
494
|
+
The result has this shape::
|
|
495
|
+
|
|
496
|
+
<name>_row_page
|
|
497
|
+
data : list of <name>_row
|
|
498
|
+
links : PageLinks
|
|
499
|
+
"""
|
|
500
|
+
fields: list[tuple[str, Any, msgspec.UnsetType]] = []
|
|
501
|
+
renames: dict[str, str] = {}
|
|
502
|
+
for index, column in enumerate(resource.columns):
|
|
503
|
+
internal_name = f"field_{index}"
|
|
504
|
+
fields.append((internal_name, column.wire_type, msgspec.UNSET))
|
|
505
|
+
renames[internal_name] = column.name
|
|
506
|
+
row = msgspec.defstruct(
|
|
507
|
+
f"{_schema_name(resource)}_row", fields, rename=renames, module="turbine.api.generated"
|
|
508
|
+
)
|
|
509
|
+
# msgspec accepts every type expression, but its stub asks for `type`. The value
|
|
510
|
+
# `list[row]` is a GenericAlias, not a `type`. Python has no static type for a
|
|
511
|
+
# type expression, thus the value is Any.
|
|
512
|
+
rows: Any = GenericAlias(list, row)
|
|
513
|
+
page_fields: list[str | tuple[str, type] | tuple[str, type, Any]] = [
|
|
514
|
+
("data", rows),
|
|
515
|
+
("links", PageLinks),
|
|
516
|
+
]
|
|
517
|
+
return msgspec.defstruct(f"{row.__name__}_page", page_fields, module="turbine.api.generated")
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def _next_page_url(request: Request[Any, Any, Any], cursor: str) -> str:
|
|
521
|
+
"""Build the URL of the next page."""
|
|
522
|
+
query = request.query_params.copy()
|
|
523
|
+
query["page[after]"] = cursor
|
|
524
|
+
return str(request.url.with_replacements(query=query))
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _encode_page(
|
|
528
|
+
table: pa.Table, resource: PublishedDataset, selected: tuple[str, ...], next_url: str | None
|
|
529
|
+
) -> bytes:
|
|
530
|
+
"""Encode a page as a JSON body.
|
|
531
|
+
|
|
532
|
+
The rows are encoded in batches, so one page never holds all rows as
|
|
533
|
+
Python objects at the same time.
|
|
534
|
+
"""
|
|
535
|
+
encoder = msgspec.json.Encoder()
|
|
536
|
+
prepared = pa.table(
|
|
537
|
+
[resource.column(name).to_wire(table[name]) for name in selected], names=selected
|
|
538
|
+
)
|
|
539
|
+
batches = prepared.to_batches(max_chunksize=ENCODE_CHUNK_ROWS)
|
|
540
|
+
object_names = tuple(name for name in selected if resource.column(name).python_type is dict)
|
|
541
|
+
chunks = [
|
|
542
|
+
encoder.encode(_decode_object_fields(batch.to_pylist(), object_names))[1:-1]
|
|
543
|
+
for batch in batches
|
|
544
|
+
if batch.num_rows
|
|
545
|
+
]
|
|
546
|
+
return b'{"data":[' + b",".join(chunks) + b'],"links":{"next":' + encoder.encode(next_url) + b"}}"
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
def _decode_object_fields(
|
|
550
|
+
rows: list[dict[str, Any]], object_names: tuple[str, ...]
|
|
551
|
+
) -> list[dict[str, Any]]:
|
|
552
|
+
"""Replace JSON text in object fields with the decoded object."""
|
|
553
|
+
for row in rows:
|
|
554
|
+
for name in object_names:
|
|
555
|
+
value = row[name]
|
|
556
|
+
if isinstance(value, str):
|
|
557
|
+
row[name] = msgspec.json.decode(value, type=dict[str, Any])
|
|
558
|
+
return rows
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _etag_matches(header: str | None, etag: str) -> bool:
|
|
562
|
+
"""Tell if an If-None-Match header holds the ETag."""
|
|
563
|
+
if header is None:
|
|
564
|
+
return False
|
|
565
|
+
expected = etag.removeprefix("W/")
|
|
566
|
+
return any(
|
|
567
|
+
candidate == "*" or candidate.removeprefix("W/") == expected
|
|
568
|
+
for candidate in (part.strip() for part in header.split(","))
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _headers(
|
|
573
|
+
resource: PublishedDataset, context: ServeContext, body: bytes, *, next_url: str | None
|
|
574
|
+
) -> dict[str, str]:
|
|
575
|
+
"""Build the response headers of a page.
|
|
576
|
+
|
|
577
|
+
The headers carry the contract status, the cache instructions, the
|
|
578
|
+
ETag, and the links to the next page and to a successor version.
|
|
579
|
+
"""
|
|
580
|
+
product = resource.product
|
|
581
|
+
status = product.metadata.status
|
|
582
|
+
headers: dict[str, str] = {
|
|
583
|
+
"X-Contract-Id": product.key.id,
|
|
584
|
+
"X-Contract-Version": product.metadata.version or "",
|
|
585
|
+
"Cache-Control": _property(product, "cacheControl") or "no-cache",
|
|
586
|
+
}
|
|
587
|
+
links = [f'<{next_url}>; rel="next"'] if next_url is not None else []
|
|
588
|
+
if status is not None:
|
|
589
|
+
headers["X-Contract-Status"] = status.value
|
|
590
|
+
if product.metadata.end_of_life:
|
|
591
|
+
headers["Deprecation"] = "true"
|
|
592
|
+
sunset = next(
|
|
593
|
+
(
|
|
594
|
+
str(item.value)
|
|
595
|
+
for item in product.metadata.sla_properties
|
|
596
|
+
if item.property is SlaPropertyKind.END_OF_LIFE and item.value is not None
|
|
597
|
+
),
|
|
598
|
+
None,
|
|
599
|
+
) or _property(product, "sunsetDate")
|
|
600
|
+
if sunset is not None:
|
|
601
|
+
headers["Sunset"] = sunset
|
|
602
|
+
successor = _property(product, "supersededBy")
|
|
603
|
+
if successor is not None:
|
|
604
|
+
replacement = context.project.product(ProductIdentity(id=successor))
|
|
605
|
+
if replacement is not None and (path := _successor_path(resource, replacement)) is not None:
|
|
606
|
+
links.append(f'<{path}>; rel="successor-version"')
|
|
607
|
+
if links:
|
|
608
|
+
headers["Link"] = ", ".join(links)
|
|
609
|
+
digest = hashlib.sha256(
|
|
610
|
+
product.key.id.encode() + (product.metadata.version or "").encode() + body
|
|
611
|
+
).hexdigest()
|
|
612
|
+
headers["ETag"] = f'"{digest}"'
|
|
613
|
+
return headers
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def _successor_path(resource: PublishedDataset, replacement: DataProduct) -> str | None:
|
|
617
|
+
"""Give the URL of the same Dataset in the successor product."""
|
|
618
|
+
dataset = replacement.declared_dataset(resource.dataset.name)
|
|
619
|
+
if dataset is None:
|
|
620
|
+
return None
|
|
621
|
+
return f"{MOUNT_PATH}{dataset_path(replacement, dataset)}"
|
|
622
|
+
|
|
623
|
+
|
|
624
|
+
def _report(
|
|
625
|
+
context: ServeContext,
|
|
626
|
+
product: DataProduct,
|
|
627
|
+
finding_id: str,
|
|
628
|
+
severity: Severity,
|
|
629
|
+
message: str,
|
|
630
|
+
*,
|
|
631
|
+
source: TextLocation | None = None,
|
|
632
|
+
report_bug: bool = False,
|
|
633
|
+
) -> None:
|
|
634
|
+
"""Add a publication finding to the diagnostics of the context.
|
|
635
|
+
|
|
636
|
+
``report_bug`` adds the last help line that tells where to open an issue.
|
|
637
|
+
"""
|
|
638
|
+
finding = Finding(finding_id, severity, message)
|
|
639
|
+
if report_bug:
|
|
640
|
+
finding = finding.with_bug_report()
|
|
641
|
+
if source is not None:
|
|
642
|
+
finding = finding.at_location(source)
|
|
643
|
+
elif product.defining_uri is not None:
|
|
644
|
+
finding = finding.at(product.defining_uri)
|
|
645
|
+
context.diagnostics.append(finding)
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _property(product: DataProduct, name: str) -> str | None:
|
|
649
|
+
"""Read a turbine extension property of the product."""
|
|
650
|
+
return product.extension_properties.text("turbine", name)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
def _schema_name(resource: PublishedDataset) -> str:
|
|
654
|
+
"""Build a unique name for the generated schema of a dataset."""
|
|
655
|
+
product = resource.product.key.id.replace("-", "_")
|
|
656
|
+
dataset = str(resource.dataset.name).replace("-", "_")
|
|
657
|
+
route_identity = hashlib.sha256(resource.path.encode()).hexdigest()
|
|
658
|
+
return f"{product}_{dataset}_{route_identity}"
|