hotdata-framework 0.12.1__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/CHANGELOG.md +113 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/PKG-INFO +2 -2
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/README.md +1 -1
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/client.py +17 -6
- hotdata_framework-0.14.0/hotdata_framework/errors.py +140 -0
- hotdata_framework-0.14.0/hotdata_framework/managed_client.py +313 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/pyproject.toml +1 -1
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_client.py +52 -6
- hotdata_framework-0.14.0/tests/test_errors.py +125 -0
- hotdata_framework-0.14.0/tests/test_managed_client.py +969 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_retry_policy.py +10 -4
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/uv.lock +1 -1
- hotdata_framework-0.12.1/hotdata_framework/errors.py +0 -40
- hotdata_framework-0.12.1/hotdata_framework/managed_client.py +0 -241
- hotdata_framework-0.12.1/tests/test_errors.py +0 -48
- hotdata_framework-0.12.1/tests/test_managed_client.py +0 -248
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/publish.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/release.yml +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.gitignore +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/CONTRACT.md +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/RELEASING.md +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/__init__.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/databases.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_contract.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_indexes.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_version.py +0 -0
|
@@ -8,6 +8,119 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
## [0.14.0] - 2026-09-01
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- fix(managed): wait on the query run instead of downloading the result to check it.
|
|
16
|
+
|
|
17
|
+
Reading a managed table made three calls and used one. `POST /v1/query` returned
|
|
18
|
+
an inline preview of the rows, `GET /v1/results/{id}` was polled until the
|
|
19
|
+
result was `ready`, and the result was then fetched as Arrow. Only the Arrow
|
|
20
|
+
copy was used.
|
|
21
|
+
|
|
22
|
+
The readiness poll was the expensive one. `limit` on that endpoint defaults to
|
|
23
|
+
unbounded, so polling a ready result downloads the entire result body to read
|
|
24
|
+
one status field. It is also the wrong endpoint to lean on as a table grows:
|
|
25
|
+
a JSON body over the instance's per-fetch memory budget is refused with 413,
|
|
26
|
+
and one that would fit alone but not alongside concurrent JSON fetches with
|
|
27
|
+
429 — so the readiness check starts failing on exactly the largest tables.
|
|
28
|
+
|
|
29
|
+
The query is now submitted with `async`, so the server returns a run id rather
|
|
30
|
+
than a preview, and readiness comes from `GET /v1/query-runs/{id}`, which
|
|
31
|
+
carries no rows at any size. `result_id` is read off the run rather than off
|
|
32
|
+
the query reply, because a run can succeed having saved nothing and the run is
|
|
33
|
+
what reports that — and that case now raises rather than reading as an empty
|
|
34
|
+
table. `fetch_table` answered `None` for it, which `fetch_table_rows` turns
|
|
35
|
+
into `[]`, the same answer both give for a table that is not synced. A
|
|
36
|
+
read-modify-write load would have read no existing rows and written only its
|
|
37
|
+
new batch, dropping every row already there. A reply shape this client does not
|
|
38
|
+
recognise raises for the same reason, as `HotdataClient` already did — so a
|
|
39
|
+
`None` from `fetch_table` now means one thing only: the table is not synced.
|
|
40
|
+
Arrow stays the only path the data travels, so column types come from the
|
|
41
|
+
server's schema rather than being inferred from JSON.
|
|
42
|
+
|
|
43
|
+
Costs one extra round trip on a query that would have answered synchronously,
|
|
44
|
+
in exchange for not transferring the result twice.
|
|
45
|
+
|
|
46
|
+
The Arrow fetch now also waits out a result that reports itself not ready, in
|
|
47
|
+
case that ordering ever stops holding. It should be unreachable, and it is
|
|
48
|
+
cheap to keep: that endpoint answers a result which is not ready with a small
|
|
49
|
+
refusal rather than with data, which is exactly what made waiting on the JSON
|
|
50
|
+
result body expensive and waiting here not.
|
|
51
|
+
|
|
52
|
+
- fix(managed): recognise `interrupted`, and drop a run status the API never sends.
|
|
53
|
+
|
|
54
|
+
Both `ManagedDatabaseClient` and `HotdataClient` treated `failed` and
|
|
55
|
+
`cancelled` as the terminal run failures. `cancelled` is not a status this API
|
|
56
|
+
returns. `interrupted` is — a run whose server was replaced before it finished
|
|
57
|
+
— and it matched neither, so an interrupted run was polled for the full
|
|
58
|
+
five-minute timeout and then raised `TimeoutError`: a retryable condition
|
|
59
|
+
hidden behind a long wait and an error naming the wrong problem.
|
|
60
|
+
|
|
61
|
+
On `ManagedDatabaseClient` an interrupted run is now raised as transient, so
|
|
62
|
+
the surrounding retry re-submits the query. That needed `classify_sdk_error` to
|
|
63
|
+
pass an already-classified error through unchanged rather than demoting a
|
|
64
|
+
caller-raised transient error to terminal. `HotdataClient.execute_sql` now
|
|
65
|
+
fails fast on it with the run's own message.
|
|
66
|
+
|
|
67
|
+
Both polls keep enumerating the statuses that mean *finished*, and an
|
|
68
|
+
unrecognised status still waits. Calling an unknown status terminal would make
|
|
69
|
+
the omission easier to diagnose and much worse to live with: one status added
|
|
70
|
+
upstream would fail every read at once, where waiting costs a single slow call.
|
|
71
|
+
What made `interrupted` expensive was not the waiting — it was that the
|
|
72
|
+
timeout never said which status it had been waiting on. Both timeouts now name
|
|
73
|
+
it.
|
|
74
|
+
|
|
75
|
+
## [0.13.0] - 2026-08-27
|
|
76
|
+
|
|
77
|
+
### Fixed
|
|
78
|
+
|
|
79
|
+
- fix(load): retry an `append` load instead of running it at most once.
|
|
80
|
+
|
|
81
|
+
`append` was excluded from retries on the grounds that it is not idempotent:
|
|
82
|
+
if the server commits but the response is lost, a retry would duplicate rows.
|
|
83
|
+
That is not how the server behaves. It keys a receipt on `upload_id`, and a
|
|
84
|
+
re-POST of the same id replays the committed result instead of applying the
|
|
85
|
+
load again — so what makes a retry safe is re-sending the same upload, not
|
|
86
|
+
the mode. This client stages once, in `upload_parquet`, outside the retried
|
|
87
|
+
operation, so the invariant holds for every mode.
|
|
88
|
+
|
|
89
|
+
The exclusion cost real availability. The destination serialises writes per
|
|
90
|
+
table and refuses rather than queues, so concurrent writers to one table get
|
|
91
|
+
`409 RESOURCE_LOCKED` — and an append had no budget to wait it out, whatever
|
|
92
|
+
`max_retries` the caller had configured.
|
|
93
|
+
|
|
94
|
+
`HotdataClient.load_managed_table(file=...)` uploads inside the call and so
|
|
95
|
+
does not hold the invariant. It is unwrapped and unaffected.
|
|
96
|
+
|
|
97
|
+
- fix(errors): classify a 409 by its `error.code` rather than by the status alone.
|
|
98
|
+
|
|
99
|
+
`CONFLICT` is now terminal: it means the request cannot succeed as posted, so
|
|
100
|
+
the previous behaviour spent the entire retry budget arriving at the same
|
|
101
|
+
answer. `RESOURCE_LOCKED` stays transient. A 409 with no error envelope — a
|
|
102
|
+
failed query result, say — is classified as before.
|
|
103
|
+
|
|
104
|
+
- fix(retry): honour `Retry-After`, and jitter the backoff.
|
|
105
|
+
|
|
106
|
+
`Retry-After` is taken as a floor on the ramp, capped like the ramp so a bad
|
|
107
|
+
header cannot park an attempt for an hour. Jitter of up to +50% is added on
|
|
108
|
+
top and never subtracted, so a stated `Retry-After` is not undercut. Without
|
|
109
|
+
it, writers that collided on one table retry in lockstep and collide again.
|
|
110
|
+
|
|
111
|
+
This lengthens a 20-attempt budget from 285s to roughly 316-405s.
|
|
112
|
+
|
|
113
|
+
- docs: scope the "a load is not idempotent" claim in the README and in
|
|
114
|
+
`test_retry_policy` to the transport layer, which is where it is still true
|
|
115
|
+
and where those two were always talking about. Left unscoped they read as
|
|
116
|
+
repo-wide and contradict the call-layer retry above.
|
|
117
|
+
|
|
118
|
+
### Added
|
|
119
|
+
|
|
120
|
+
- `HotdataError` carries `status_code`, `code` and `retry_after_seconds`. The
|
|
121
|
+
message is flattened and truncated for readability, so it could not serve as
|
|
122
|
+
a discriminator; these can.
|
|
123
|
+
|
|
11
124
|
## [0.12.1] - 2026-08-18
|
|
12
125
|
|
|
13
126
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -38,7 +38,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
38
38
|
|
|
39
39
|
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
40
40
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
41
|
-
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a
|
|
41
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
|
|
42
42
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
43
43
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
44
44
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -10,7 +10,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
10
10
|
|
|
11
11
|
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
12
12
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
13
|
-
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a
|
|
13
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
|
|
14
14
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
15
15
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
16
16
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -77,8 +77,17 @@ VectorMetric = Literal["l2", "cosine", "dot"]
|
|
|
77
77
|
_INDEX_TYPES = frozenset(get_args(IndexType))
|
|
78
78
|
_VECTOR_METRICS = frozenset(get_args(VectorMetric))
|
|
79
79
|
|
|
80
|
-
|
|
81
|
-
|
|
80
|
+
# Query-run statuses that mean the run is over. `interrupted` belongs here --
|
|
81
|
+
# omitting it is what made an interrupted run wait out the full timeout -- and
|
|
82
|
+
# `cancelled`, listed here for a long time, is not a status this API sends.
|
|
83
|
+
#
|
|
84
|
+
# Enumerating the terminal side rather than the in-flight side is deliberate. An
|
|
85
|
+
# unrecognised status then keeps polling and costs one slow call, where treating
|
|
86
|
+
# it as terminal would fail every query the moment a status is added upstream.
|
|
87
|
+
# The timeout names the status it last saw, so a missing one is diagnosable
|
|
88
|
+
# without being dangerous.
|
|
89
|
+
_RUN_TERMINAL = frozenset({"succeeded", "failed", "interrupted"})
|
|
90
|
+
_RESULT_FAILURE = frozenset({"failed"})
|
|
82
91
|
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
83
92
|
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
84
93
|
|
|
@@ -918,7 +927,7 @@ class HotdataClient:
|
|
|
918
927
|
last = None
|
|
919
928
|
while time.monotonic() < deadline:
|
|
920
929
|
last = runs.get_query_run(query_run_id)
|
|
921
|
-
if last.status in
|
|
930
|
+
if last.status in _RUN_TERMINAL:
|
|
922
931
|
return last
|
|
923
932
|
time.sleep(interval_s)
|
|
924
933
|
raise TimeoutError(
|
|
@@ -985,9 +994,11 @@ class HotdataClient:
|
|
|
985
994
|
durable state rather than from a connection that has to stay alive. That
|
|
986
995
|
also gives a caller a handle: the job id is returned on
|
|
987
996
|
`LoadManagedTableResult`, so "did it land?" is answerable after a lost
|
|
988
|
-
response.
|
|
989
|
-
|
|
990
|
-
the
|
|
997
|
+
response. That answer is a convenience rather than a precondition for
|
|
998
|
+
retrying: re-sending the same upload_id replays the server's receipt
|
|
999
|
+
instead of applying the load a second time, which is what makes a retry
|
|
1000
|
+
safe in every mode. It stops being safe for a caller that re-stages the
|
|
1001
|
+
upload, because a fresh upload id has no receipt to replay.
|
|
991
1002
|
|
|
992
1003
|
`partially_succeeded` is terminal and carries a message, so it is raised
|
|
993
1004
|
rather than returned -- a caller asked for a table's contents to be
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
|
|
6
|
+
from hotdata.rest import ApiException
|
|
7
|
+
|
|
8
|
+
# The API explains a 409 with a machine-readable code, and the two it sends
|
|
9
|
+
# mean opposite things to a retry policy. RESOURCE_LOCKED is a refusal taken
|
|
10
|
+
# before any work: the insert that would have created the unit of work lost a
|
|
11
|
+
# unique-constraint race, so nothing was claimed and nothing was written.
|
|
12
|
+
# CONFLICT is the opposite — the request cannot succeed as posted, so retrying
|
|
13
|
+
# spends the whole budget arriving at the same answer.
|
|
14
|
+
_TERMINAL_CONFLICT_CODE = "CONFLICT"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class HotdataError(RuntimeError):
|
|
18
|
+
"""An API failure, carrying what a retry policy needs to decide.
|
|
19
|
+
|
|
20
|
+
The message cannot be the discriminator: it is flattened and truncated for
|
|
21
|
+
readability, so keying on it means substring-matching prose. ``status_code``
|
|
22
|
+
and ``code`` are the machine-readable form of the same answer, and
|
|
23
|
+
``retry_after_seconds`` is the server's own estimate of how long the
|
|
24
|
+
condition it just refused will last.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
message: str,
|
|
30
|
+
*,
|
|
31
|
+
status_code: int | None = None,
|
|
32
|
+
code: str | None = None,
|
|
33
|
+
retry_after_seconds: float | None = None,
|
|
34
|
+
) -> None:
|
|
35
|
+
super().__init__(message)
|
|
36
|
+
self.status_code = status_code
|
|
37
|
+
self.code = code
|
|
38
|
+
self.retry_after_seconds = retry_after_seconds
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class HotdataTransientError(HotdataError):
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class HotdataTerminalError(HotdataError):
|
|
46
|
+
pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _error_code(body: object) -> str | None:
|
|
50
|
+
"""The ``error.code`` an API error envelope carries, if this body is one.
|
|
51
|
+
|
|
52
|
+
Not every 409 comes from an endpoint that speaks the envelope — a failed
|
|
53
|
+
query result is reported as one and carries a result document instead — so
|
|
54
|
+
a missing code is ordinary, and callers fall back to the status.
|
|
55
|
+
"""
|
|
56
|
+
if not isinstance(body, (str, bytes, bytearray)):
|
|
57
|
+
return None
|
|
58
|
+
try:
|
|
59
|
+
parsed: object = json.loads(body)
|
|
60
|
+
except ValueError:
|
|
61
|
+
return None
|
|
62
|
+
if not isinstance(parsed, Mapping):
|
|
63
|
+
return None
|
|
64
|
+
error: object = parsed.get("error")
|
|
65
|
+
if not isinstance(error, Mapping):
|
|
66
|
+
return None
|
|
67
|
+
code: object = error.get("code")
|
|
68
|
+
return code if isinstance(code, str) else None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _retry_after_seconds(headers: object) -> float | None:
|
|
72
|
+
"""``Retry-After`` as a number of seconds, when the response states one.
|
|
73
|
+
|
|
74
|
+
Only the delta-seconds form is read. That is what the API sends, and the
|
|
75
|
+
HTTP-date form would need a comparison against a server clock we do not
|
|
76
|
+
have to be worth anything.
|
|
77
|
+
"""
|
|
78
|
+
if not isinstance(headers, Mapping):
|
|
79
|
+
return None
|
|
80
|
+
raw: object = headers.get("Retry-After")
|
|
81
|
+
if raw is None:
|
|
82
|
+
# The SDK hands us urllib3's case-insensitive mapping and the API sends
|
|
83
|
+
# the header lower-cased, so the direct hit is what normally answers.
|
|
84
|
+
# Fall back for any plain dict that reaches us instead — a missed
|
|
85
|
+
# header is silent, and silence here reads as "the server asked for
|
|
86
|
+
# nothing".
|
|
87
|
+
raw = next((v for k, v in headers.items() if str(k).lower() == "retry-after"), None)
|
|
88
|
+
if raw is None:
|
|
89
|
+
return None
|
|
90
|
+
try:
|
|
91
|
+
seconds = float(str(raw).strip())
|
|
92
|
+
except ValueError:
|
|
93
|
+
return None
|
|
94
|
+
return seconds if seconds >= 0 else None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _error_class(status_code: int, code: str | None) -> type[HotdataError]:
|
|
98
|
+
if status_code == 409 and code == _TERMINAL_CONFLICT_CODE:
|
|
99
|
+
# The request cannot succeed as posted — an upload already consumed
|
|
100
|
+
# with nothing to replay, a receipt naming a different target, an
|
|
101
|
+
# incompatible column type. Every retry reaches the same 409.
|
|
102
|
+
return HotdataTerminalError
|
|
103
|
+
if status_code in (408, 409, 425, 429):
|
|
104
|
+
return HotdataTransientError
|
|
105
|
+
if status_code == 501:
|
|
106
|
+
# Not Implemented is a permanent capability gap (e.g. the storage
|
|
107
|
+
# backend cannot issue presigned URLs) — retrying cannot succeed.
|
|
108
|
+
return HotdataTerminalError
|
|
109
|
+
if 500 <= status_code <= 599:
|
|
110
|
+
return HotdataTransientError
|
|
111
|
+
return HotdataTerminalError
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def classify_sdk_error(error: Exception) -> HotdataError:
|
|
115
|
+
if isinstance(error, HotdataError):
|
|
116
|
+
# Already classified. A caller that read transience off a typed status
|
|
117
|
+
# -- an interrupted query run, say -- knows more than this function can
|
|
118
|
+
# recover from the exception, and the fallback below would demote it to
|
|
119
|
+
# terminal and cost the retry.
|
|
120
|
+
return error
|
|
121
|
+
if isinstance(error, TimeoutError):
|
|
122
|
+
return HotdataTransientError(str(error))
|
|
123
|
+
if isinstance(error, ConnectionError):
|
|
124
|
+
return HotdataTransientError(str(error))
|
|
125
|
+
if isinstance(error, ApiException):
|
|
126
|
+
status_code = int(error.status or 0)
|
|
127
|
+
message = f"{status_code}: {error.reason or 'unknown error'}"
|
|
128
|
+
# The response body is where the API explains itself (e.g. which
|
|
129
|
+
# header is missing) — without it "400: Bad Request" is undebuggable.
|
|
130
|
+
body: object = getattr(error, "body", None)
|
|
131
|
+
if body:
|
|
132
|
+
message = f"{message} — {' '.join(str(body).split())[:500]}"
|
|
133
|
+
code = _error_code(body)
|
|
134
|
+
return _error_class(status_code, code)(
|
|
135
|
+
message,
|
|
136
|
+
status_code=status_code,
|
|
137
|
+
code=code,
|
|
138
|
+
retry_after_seconds=_retry_after_seconds(getattr(error, "headers", None)),
|
|
139
|
+
)
|
|
140
|
+
return HotdataTerminalError(str(error))
|
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
"""Retry-wrapped managed-database client shared by Hotdata adapter packages.
|
|
2
|
+
|
|
3
|
+
Both hotdata-airflow and hotdata-dlt-destination import this module so that
|
|
4
|
+
the higher-level client logic (retries, Arrow queries, table management) lives
|
|
5
|
+
in one place rather than being duplicated per adapter.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import random
|
|
11
|
+
import time
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from typing import Any, TypeVar
|
|
14
|
+
|
|
15
|
+
import pyarrow as pa
|
|
16
|
+
from hotdata.api.query_api import QueryApi
|
|
17
|
+
from hotdata.api.query_runs_api import QueryRunsApi
|
|
18
|
+
from hotdata.arrow import ResultNotReadyError
|
|
19
|
+
from hotdata.arrow import ResultsApi as ArrowResultsApi
|
|
20
|
+
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
21
|
+
from hotdata.models.query_request import QueryRequest
|
|
22
|
+
from hotdata.models.query_response import QueryResponse
|
|
23
|
+
|
|
24
|
+
from hotdata_framework.client import HotdataClient as RuntimeClient
|
|
25
|
+
from hotdata_framework.client import ManagedLoadMode
|
|
26
|
+
from hotdata_framework.databases import LoadManagedTableResult, ManagedDatabase
|
|
27
|
+
from hotdata_framework.errors import (
|
|
28
|
+
HotdataTransientError,
|
|
29
|
+
classify_sdk_error,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
T = TypeVar("T")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ManagedDatabaseClient:
|
|
36
|
+
"""Managed-database client with bounded retries over hotdata-framework.
|
|
37
|
+
|
|
38
|
+
This is the shared client used by Hotdata adapter packages (Airflow,
|
|
39
|
+
dlt, etc.). It wraps the lower-level RuntimeClient with retry logic,
|
|
40
|
+
Arrow-based result fetching, and convenience helpers for the managed
|
|
41
|
+
database lifecycle.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
_QUERY_TIMEOUT_SECONDS = 300.0
|
|
45
|
+
_POLL_INTERVAL_SECONDS = 0.4
|
|
46
|
+
_MAX_BACKOFF_SECONDS = 30.0
|
|
47
|
+
# Spread as a fraction of the wait, added on top of it. Half an interval is
|
|
48
|
+
# enough to decorrelate writers that started together without materially
|
|
49
|
+
# changing how long the budget lasts.
|
|
50
|
+
_RETRY_JITTER_FRACTION = 0.5
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
*,
|
|
55
|
+
api_key: str,
|
|
56
|
+
workspace_id: str,
|
|
57
|
+
api_base_url: str,
|
|
58
|
+
max_retries: int,
|
|
59
|
+
retry_backoff_seconds: float,
|
|
60
|
+
request_timeout: float | tuple[float, float] | None = None,
|
|
61
|
+
) -> None:
|
|
62
|
+
self._max_retries = max_retries
|
|
63
|
+
self._retry_backoff_seconds = retry_backoff_seconds
|
|
64
|
+
self._runtime = RuntimeClient(
|
|
65
|
+
api_key,
|
|
66
|
+
workspace_id,
|
|
67
|
+
host=api_base_url.rstrip("/"),
|
|
68
|
+
request_timeout=request_timeout,
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
def close(self) -> None:
|
|
72
|
+
self._runtime.close()
|
|
73
|
+
|
|
74
|
+
def ensure_managed_database(
|
|
75
|
+
self,
|
|
76
|
+
name: str,
|
|
77
|
+
*,
|
|
78
|
+
schema: str,
|
|
79
|
+
tables: list[str],
|
|
80
|
+
create_if_missing: bool,
|
|
81
|
+
) -> ManagedDatabase:
|
|
82
|
+
def operation() -> ManagedDatabase:
|
|
83
|
+
try:
|
|
84
|
+
return self._runtime.resolve_managed_database(name)
|
|
85
|
+
except KeyError:
|
|
86
|
+
if not create_if_missing:
|
|
87
|
+
raise
|
|
88
|
+
return self._runtime.create_managed_database(
|
|
89
|
+
description=name,
|
|
90
|
+
schema=schema,
|
|
91
|
+
tables=sorted(set(tables)),
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
return self._request_with_retry(operation)
|
|
95
|
+
|
|
96
|
+
def table_is_synced(self, database: str, table: str, *, schema: str) -> bool:
|
|
97
|
+
for managed_table in self._runtime.list_managed_tables(database, schema=schema):
|
|
98
|
+
if managed_table.table == table:
|
|
99
|
+
return managed_table.synced
|
|
100
|
+
return False
|
|
101
|
+
|
|
102
|
+
def fetch_table(self, *, database: str, schema: str, table: str) -> pa.Table | None:
|
|
103
|
+
def operation() -> pa.Table | None:
|
|
104
|
+
if not self.table_is_synced(database, table, schema=schema):
|
|
105
|
+
return None
|
|
106
|
+
db = self._runtime.resolve_managed_database(database)
|
|
107
|
+
sql = f'SELECT * FROM "default"."{schema}"."{table}"'
|
|
108
|
+
result_id = self._query_database_scoped(sql, database_id=db.id)
|
|
109
|
+
if result_id is None:
|
|
110
|
+
return None
|
|
111
|
+
return self._fetch_result_arrow(result_id, database_id=db.id)
|
|
112
|
+
|
|
113
|
+
return self._request_with_retry(operation)
|
|
114
|
+
|
|
115
|
+
def _fetch_result_arrow(self, result_id: str, *, database_id: str) -> pa.Table:
|
|
116
|
+
"""Fetch a ready result as Arrow, carrying the database scope.
|
|
117
|
+
|
|
118
|
+
Results of database-scoped queries are themselves database-scoped —
|
|
119
|
+
the results endpoints reject requests without the scope. The hotdata
|
|
120
|
+
0.6.0 SDK exposes (and requires) ``x_database_id`` on the Arrow
|
|
121
|
+
helper directly.
|
|
122
|
+
"""
|
|
123
|
+
arrow = ArrowResultsApi(self._runtime.api)
|
|
124
|
+
deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
|
|
125
|
+
while True:
|
|
126
|
+
try:
|
|
127
|
+
return arrow.get_result_arrow(result_id, x_database_id=database_id)
|
|
128
|
+
except ResultNotReadyError:
|
|
129
|
+
# Waiting on the run should already have made this unreachable:
|
|
130
|
+
# a run reports `succeeded` only once its result is saved and
|
|
131
|
+
# ready. Tolerating it anyway costs nothing and removes the need
|
|
132
|
+
# to take that ordering on trust. The Arrow endpoint answers a
|
|
133
|
+
# result that is not ready with a small refusal rather than with
|
|
134
|
+
# data, so waiting here is cheap in the way waiting on the JSON
|
|
135
|
+
# result body -- which is what this change removed -- is not.
|
|
136
|
+
if time.monotonic() >= deadline:
|
|
137
|
+
raise
|
|
138
|
+
time.sleep(self._POLL_INTERVAL_SECONDS)
|
|
139
|
+
|
|
140
|
+
def _query_database_scoped(self, sql: str, *, database_id: str) -> str | None:
|
|
141
|
+
raw = QueryApi(self._runtime.api).query(
|
|
142
|
+
# Asked asynchronously because this caller wants a result id, not
|
|
143
|
+
# rows. A synchronous submit always builds an inline preview of the
|
|
144
|
+
# result and sends it -- megabytes, on a path that then reads the
|
|
145
|
+
# whole result as Arrow anyway and never looks at the preview. The
|
|
146
|
+
# async reply carries a run id and nothing else, and there is no way
|
|
147
|
+
# to suppress the preview on a synchronous one.
|
|
148
|
+
#
|
|
149
|
+
# It also settles the types: the preview is JSON, which has no Arrow
|
|
150
|
+
# schema and renders non-finite floats as null, so it could not have
|
|
151
|
+
# substituted for the Arrow fetch even when it holds every row.
|
|
152
|
+
#
|
|
153
|
+
# `var_async` is the generated SDK's spelling of the wire field
|
|
154
|
+
# `async`, which is a Python keyword and so cannot be the attribute
|
|
155
|
+
# name.
|
|
156
|
+
QueryRequest(sql=sql, var_async=True),
|
|
157
|
+
x_database_id=database_id,
|
|
158
|
+
)
|
|
159
|
+
# Both reply shapes carry `query_run_id`, and the run is the readiness
|
|
160
|
+
# signal for either -- a synchronous reply (which `async_after_ms` can
|
|
161
|
+
# still produce) returns rows inline but goes on saving the full result
|
|
162
|
+
# in the background, so it is not the finish line either.
|
|
163
|
+
if isinstance(raw, (QueryResponse, AsyncQueryResponse)):
|
|
164
|
+
return self._await_query_run(raw.query_run_id, database_id=database_id)
|
|
165
|
+
# Returning nothing here would read as an empty table: `fetch_table`
|
|
166
|
+
# answers `None`, `fetch_table_rows` turns that into `[]`, and a
|
|
167
|
+
# read-modify-write load would write only its new batch over rows it
|
|
168
|
+
# believed were not there. A reply shape this client does not know is a
|
|
169
|
+
# reason to stop, not to report emptiness. `HotdataClient` raises on the
|
|
170
|
+
# same condition.
|
|
171
|
+
raise RuntimeError(f"Unexpected query response type: {type(raw)!r}")
|
|
172
|
+
|
|
173
|
+
def _await_query_run(self, query_run_id: str, *, database_id: str) -> str | None:
|
|
174
|
+
"""Wait for a query run to finish; return the result id it produced.
|
|
175
|
+
|
|
176
|
+
The run is the whole wait. A run turns `succeeded` only after its result
|
|
177
|
+
has been saved and is `ready`, so `succeeded` needs no second check
|
|
178
|
+
against the result -- and asking the result endpoint instead would mean
|
|
179
|
+
downloading the entire result to read one field, which the server
|
|
180
|
+
refuses outright (413/429) once the result is large enough.
|
|
181
|
+
|
|
182
|
+
`result_id` comes off the run rather than off the query reply because a
|
|
183
|
+
`succeeded` run reports none when every row came back inline but the
|
|
184
|
+
result could not be saved for later retrieval.
|
|
185
|
+
"""
|
|
186
|
+
runs = QueryRunsApi(self._runtime.api)
|
|
187
|
+
deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
|
|
188
|
+
last_status: str | None = None
|
|
189
|
+
while time.monotonic() < deadline:
|
|
190
|
+
# Runs (like results) of database-scoped queries are database-scoped.
|
|
191
|
+
run = runs.get_query_run(query_run_id, x_database_id=database_id)
|
|
192
|
+
last_status = run.status
|
|
193
|
+
if run.status == "succeeded":
|
|
194
|
+
if run.result_id is None:
|
|
195
|
+
# A run succeeds with no result id when its rows were
|
|
196
|
+
# returned inline but the result could not be saved for
|
|
197
|
+
# later retrieval. Returning nothing here would surface as
|
|
198
|
+
# an empty table -- `fetch_table` answers `None`, and
|
|
199
|
+
# `fetch_table_rows` turns that into `[]`, which is the same
|
|
200
|
+
# answer it gives for a table that does not exist. A
|
|
201
|
+
# read-modify-write load would then read no existing rows
|
|
202
|
+
# and write only its new batch, dropping what was there.
|
|
203
|
+
# Terminal rather than transient: re-running the query
|
|
204
|
+
# cannot save a result that was already discarded.
|
|
205
|
+
# `getattr` because this runs while building an error: if
|
|
206
|
+
# the field ever goes away, losing the explanation is a far
|
|
207
|
+
# better outcome than an AttributeError replacing the raise.
|
|
208
|
+
warning = getattr(run, "warning_message", None)
|
|
209
|
+
raise RuntimeError(
|
|
210
|
+
f"Query run {query_run_id} succeeded but its result was not "
|
|
211
|
+
f"saved, so the table cannot be read"
|
|
212
|
+
+ (f": {warning}" if warning else "")
|
|
213
|
+
)
|
|
214
|
+
return run.result_id
|
|
215
|
+
if run.status == "interrupted":
|
|
216
|
+
# Terminal, but the server lost the run rather than rejecting
|
|
217
|
+
# the query, so it is the one failure here worth re-running.
|
|
218
|
+
# Raised pre-classified: `classify_sdk_error` cannot tell this
|
|
219
|
+
# apart from an ordinary RuntimeError and would call it terminal.
|
|
220
|
+
raise HotdataTransientError(
|
|
221
|
+
run.error_message or f"Query run {query_run_id} was interrupted"
|
|
222
|
+
)
|
|
223
|
+
if run.status == "failed":
|
|
224
|
+
raise RuntimeError(run.error_message or f"Query run {query_run_id} failed")
|
|
225
|
+
# Any other status keeps polling, including one this client has never
|
|
226
|
+
# seen. Treating an unrecognised status as terminal is the cheaper
|
|
227
|
+
# failure to diagnose and by far the more expensive one to suffer: a
|
|
228
|
+
# single status added upstream would then fail every query at once,
|
|
229
|
+
# where waiting costs one slow call. What made `interrupted`
|
|
230
|
+
# expensive was not the waiting, it was that the timeout never said
|
|
231
|
+
# which status it had waited on -- so the message now carries it.
|
|
232
|
+
time.sleep(self._POLL_INTERVAL_SECONDS)
|
|
233
|
+
raise TimeoutError(
|
|
234
|
+
f"Query run {query_run_id} did not finish within "
|
|
235
|
+
f"{self._QUERY_TIMEOUT_SECONDS}s (last status: {last_status})"
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
def fetch_table_rows(self, *, database: str, schema: str, table: str) -> list[dict[str, Any]]:
|
|
239
|
+
result = self.fetch_table(database=database, schema=schema, table=table)
|
|
240
|
+
return result.to_pylist() if result is not None else []
|
|
241
|
+
|
|
242
|
+
def upload_parquet(self, path: str) -> str:
|
|
243
|
+
return self._request_with_retry(lambda: self._runtime.upload_parquet(path))
|
|
244
|
+
|
|
245
|
+
def load_managed_table(
|
|
246
|
+
self,
|
|
247
|
+
database: str,
|
|
248
|
+
table: str,
|
|
249
|
+
*,
|
|
250
|
+
schema: str,
|
|
251
|
+
upload_id: str,
|
|
252
|
+
mode: ManagedLoadMode = "replace",
|
|
253
|
+
key: list[str] | None = None,
|
|
254
|
+
) -> LoadManagedTableResult:
|
|
255
|
+
# Retryable in every mode, append included. A retry re-sends the SAME
|
|
256
|
+
# upload_id, and the server keys a receipt on it: a replay returns the
|
|
257
|
+
# committed result rather than applying the load a second time. So the
|
|
258
|
+
# invariant that makes this safe is the upload id, not the mode — a
|
|
259
|
+
# caller that re-stages the upload between attempts mints a new id,
|
|
260
|
+
# loses the receipt, and a retried append would then duplicate rows.
|
|
261
|
+
# This client stages once, in upload_parquet, outside the operation
|
|
262
|
+
# retried here. `HotdataClient.load_managed_table(file=...)` uploads
|
|
263
|
+
# inside the call and so does not hold the invariant; it is unwrapped,
|
|
264
|
+
# and retrying an append through it is the caller's to justify.
|
|
265
|
+
#
|
|
266
|
+
# `key` is the merge key for delete/update/upsert loads: when set it is
|
|
267
|
+
# matched per-load instead of a key declared at table creation. Omit it
|
|
268
|
+
# to use the table's declared key. Ignored for replace/append.
|
|
269
|
+
return self._request_with_retry(
|
|
270
|
+
lambda: self._runtime.load_managed_table(
|
|
271
|
+
database,
|
|
272
|
+
table,
|
|
273
|
+
schema=schema,
|
|
274
|
+
upload_id=upload_id,
|
|
275
|
+
mode=mode,
|
|
276
|
+
key=key,
|
|
277
|
+
)
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
def _request_with_retry(self, operation: Callable[[], T]) -> T:
|
|
281
|
+
max_attempts = self._max_retries
|
|
282
|
+
for attempt in range(1, max_attempts + 1):
|
|
283
|
+
try:
|
|
284
|
+
return operation()
|
|
285
|
+
except Exception as error:
|
|
286
|
+
mapped_error = classify_sdk_error(error.__cause__ or error)
|
|
287
|
+
if isinstance(mapped_error, HotdataTransientError) and attempt < max_attempts:
|
|
288
|
+
time.sleep(self._retry_delay(attempt, mapped_error.retry_after_seconds))
|
|
289
|
+
continue
|
|
290
|
+
raise mapped_error from error
|
|
291
|
+
raise RuntimeError("No retry attempts configured")
|
|
292
|
+
|
|
293
|
+
def _retry_delay(self, attempt: int, retry_after_seconds: float | None) -> float:
|
|
294
|
+
"""A linear ramp, floored by the server's Retry-After and spread by jitter.
|
|
295
|
+
|
|
296
|
+
Retry-After is a floor rather than a replacement: it says how long the
|
|
297
|
+
condition just refused typically lasts, while the ramp is what gives up
|
|
298
|
+
eventually, and taking the larger of the two honours both. It is capped
|
|
299
|
+
like the ramp so a hostile or mistaken header cannot park an attempt for
|
|
300
|
+
an hour.
|
|
301
|
+
|
|
302
|
+
Jitter is added on top and never subtracted, so a stated Retry-After is
|
|
303
|
+
not undercut. It matters because the callers that collide are the ones
|
|
304
|
+
that started together: writers refused by one table's lock would retry
|
|
305
|
+
in lockstep on an identical ramp and re-collide every time.
|
|
306
|
+
_MAX_BACKOFF_SECONDS caps the ramp, deliberately not the jitter above
|
|
307
|
+
it — clamping the total would flatten every late attempt onto the same
|
|
308
|
+
value and re-correlate exactly the waits that most need spreading.
|
|
309
|
+
"""
|
|
310
|
+
base = min(self._retry_backoff_seconds * attempt, self._MAX_BACKOFF_SECONDS)
|
|
311
|
+
if retry_after_seconds is not None:
|
|
312
|
+
base = max(base, min(retry_after_seconds, self._MAX_BACKOFF_SECONDS))
|
|
313
|
+
return base * (1.0 + random.random() * self._RETRY_JITTER_FRACTION)
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "hotdata-framework"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.14.0"
|
|
8
8
|
description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|