hotdata-framework 0.13.0__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/CHANGELOG.md +65 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/PKG-INFO +1 -1
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/client.py +12 -3
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/errors.py +6 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/managed_client.py +105 -65
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/pyproject.toml +1 -1
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_client.py +50 -3
- hotdata_framework-0.14.0/tests/test_managed_client.py +969 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/uv.lock +1 -1
- hotdata_framework-0.13.0/tests/test_managed_client.py +0 -430
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/workflows/publish.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.github/workflows/release.yml +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/.gitignore +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/CONTRACT.md +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/README.md +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/RELEASING.md +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/__init__.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/databases.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_contract.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_indexes.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_retry_policy.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.13.0 → hotdata_framework-0.14.0}/tests/test_version.py +0 -0
|
@@ -7,6 +7,71 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
|
|
11
|
+
## [0.14.0] - 2026-09-01
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- fix(managed): wait on the query run instead of downloading the result to check it.
|
|
16
|
+
|
|
17
|
+
Reading a managed table made three calls and used one. `POST /v1/query` returned
|
|
18
|
+
an inline preview of the rows, `GET /v1/results/{id}` was polled until the
|
|
19
|
+
result was `ready`, and the result was then fetched as Arrow. Only the Arrow
|
|
20
|
+
copy was used.
|
|
21
|
+
|
|
22
|
+
The readiness poll was the expensive one. `limit` on that endpoint defaults to
|
|
23
|
+
unbounded, so polling a ready result downloads the entire result body to read
|
|
24
|
+
one status field. It is also the wrong endpoint to lean on as a table grows:
|
|
25
|
+
a JSON body over the instance's per-fetch memory budget is refused with 413,
|
|
26
|
+
and one that would fit alone but not alongside concurrent JSON fetches with
|
|
27
|
+
429 — so the readiness check starts failing on exactly the largest tables.
|
|
28
|
+
|
|
29
|
+
The query is now submitted with `async`, so the server returns a run id rather
|
|
30
|
+
than a preview, and readiness comes from `GET /v1/query-runs/{id}`, which
|
|
31
|
+
carries no rows at any size. `result_id` is read off the run rather than off
|
|
32
|
+
the query reply, because a run can succeed having saved nothing and the run is
|
|
33
|
+
what reports that — and that case now raises rather than reading as an empty
|
|
34
|
+
table. `fetch_table` answered `None` for it, which `fetch_table_rows` turns
|
|
35
|
+
into `[]`, the same answer both give for a table that is not synced. A
|
|
36
|
+
read-modify-write load would have read no existing rows and written only its
|
|
37
|
+
new batch, dropping every row already there. A reply shape this client does not
|
|
38
|
+
recognise raises for the same reason, as `HotdataClient` already did — so a
|
|
39
|
+
`None` from `fetch_table` now means one thing only: the table is not synced.
|
|
40
|
+
Arrow stays the only path the data travels, so column types come from the
|
|
41
|
+
server's schema rather than being inferred from JSON.
|
|
42
|
+
|
|
43
|
+
Costs one extra round trip on a query that would have answered synchronously,
|
|
44
|
+
in exchange for not transferring the result twice.
|
|
45
|
+
|
|
46
|
+
The Arrow fetch now also waits out a result that reports itself not ready, in
|
|
47
|
+
case that ordering ever stops holding. It should be unreachable, and it is
|
|
48
|
+
cheap to keep: that endpoint answers a result which is not ready with a small
|
|
49
|
+
refusal rather than with data, which is exactly what made waiting on the JSON
|
|
50
|
+
result body expensive and waiting here not.
|
|
51
|
+
|
|
52
|
+
- fix(managed): recognise `interrupted`, and drop a run status the API never sends.
|
|
53
|
+
|
|
54
|
+
Both `ManagedDatabaseClient` and `HotdataClient` treated `failed` and
|
|
55
|
+
`cancelled` as the terminal run failures. `cancelled` is not a status this API
|
|
56
|
+
returns. `interrupted` is — a run whose server was replaced before it finished
|
|
57
|
+
— and it matched neither, so an interrupted run was polled for the full
|
|
58
|
+
five-minute timeout and then raised `TimeoutError`: a retryable condition
|
|
59
|
+
hidden behind a long wait and an error naming the wrong problem.
|
|
60
|
+
|
|
61
|
+
On `ManagedDatabaseClient` an interrupted run is now raised as transient, so
|
|
62
|
+
the surrounding retry re-submits the query. That needed `classify_sdk_error` to
|
|
63
|
+
pass an already-classified error through unchanged rather than demoting a
|
|
64
|
+
caller-raised transient error to terminal. `HotdataClient.execute_sql` now
|
|
65
|
+
fails fast on it with the run's own message.
|
|
66
|
+
|
|
67
|
+
Both polls keep enumerating the statuses that mean *finished*, and an
|
|
68
|
+
unrecognised status still waits. Calling an unknown status terminal would make
|
|
69
|
+
the omission easier to diagnose and much worse to live with: one status added
|
|
70
|
+
upstream would fail every read at once, where waiting costs a single slow call.
|
|
71
|
+
What made `interrupted` expensive was not the waiting — it was that the
|
|
72
|
+
timeout never said which status it had been waiting on. Both timeouts now name
|
|
73
|
+
it.
|
|
74
|
+
|
|
10
75
|
## [0.13.0] - 2026-08-27
|
|
11
76
|
|
|
12
77
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -77,8 +77,17 @@ VectorMetric = Literal["l2", "cosine", "dot"]
|
|
|
77
77
|
_INDEX_TYPES = frozenset(get_args(IndexType))
|
|
78
78
|
_VECTOR_METRICS = frozenset(get_args(VectorMetric))
|
|
79
79
|
|
|
80
|
-
|
|
81
|
-
|
|
80
|
+
# Query-run statuses that mean the run is over. `interrupted` belongs here --
|
|
81
|
+
# omitting it is what made an interrupted run wait out the full timeout -- and
|
|
82
|
+
# `cancelled`, listed here for a long time, is not a status this API sends.
|
|
83
|
+
#
|
|
84
|
+
# Enumerating the terminal side rather than the in-flight side is deliberate. An
|
|
85
|
+
# unrecognised status then keeps polling and costs one slow call, where treating
|
|
86
|
+
# it as terminal would fail every query the moment a status is added upstream.
|
|
87
|
+
# The timeout names the status it last saw, so a missing one is diagnosable
|
|
88
|
+
# without being dangerous.
|
|
89
|
+
_RUN_TERMINAL = frozenset({"succeeded", "failed", "interrupted"})
|
|
90
|
+
_RESULT_FAILURE = frozenset({"failed"})
|
|
82
91
|
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
83
92
|
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
84
93
|
|
|
@@ -918,7 +927,7 @@ class HotdataClient:
|
|
|
918
927
|
last = None
|
|
919
928
|
while time.monotonic() < deadline:
|
|
920
929
|
last = runs.get_query_run(query_run_id)
|
|
921
|
-
if last.status in
|
|
930
|
+
if last.status in _RUN_TERMINAL:
|
|
922
931
|
return last
|
|
923
932
|
time.sleep(interval_s)
|
|
924
933
|
raise TimeoutError(
|
|
@@ -112,6 +112,12 @@ def _error_class(status_code: int, code: str | None) -> type[HotdataError]:
|
|
|
112
112
|
|
|
113
113
|
|
|
114
114
|
def classify_sdk_error(error: Exception) -> HotdataError:
|
|
115
|
+
if isinstance(error, HotdataError):
|
|
116
|
+
# Already classified. A caller that read transience off a typed status
|
|
117
|
+
# -- an interrupted query run, say -- knows more than this function can
|
|
118
|
+
# recover from the exception, and the fallback below would demote it to
|
|
119
|
+
# terminal and cost the retry.
|
|
120
|
+
return error
|
|
115
121
|
if isinstance(error, TimeoutError):
|
|
116
122
|
return HotdataTransientError(str(error))
|
|
117
123
|
if isinstance(error, ConnectionError):
|
|
@@ -10,12 +10,12 @@ from __future__ import annotations
|
|
|
10
10
|
import random
|
|
11
11
|
import time
|
|
12
12
|
from collections.abc import Callable
|
|
13
|
-
from typing import Any,
|
|
13
|
+
from typing import Any, TypeVar
|
|
14
14
|
|
|
15
15
|
import pyarrow as pa
|
|
16
16
|
from hotdata.api.query_api import QueryApi
|
|
17
17
|
from hotdata.api.query_runs_api import QueryRunsApi
|
|
18
|
-
from hotdata.
|
|
18
|
+
from hotdata.arrow import ResultNotReadyError
|
|
19
19
|
from hotdata.arrow import ResultsApi as ArrowResultsApi
|
|
20
20
|
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
21
21
|
from hotdata.models.query_request import QueryRequest
|
|
@@ -32,16 +32,6 @@ from hotdata_framework.errors import (
|
|
|
32
32
|
T = TypeVar("T")
|
|
33
33
|
|
|
34
34
|
|
|
35
|
-
class _StatusResponse(Protocol):
|
|
36
|
-
"""Async resources (query runs, results) expose a status and error message."""
|
|
37
|
-
|
|
38
|
-
status: str
|
|
39
|
-
error_message: str | None
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
S = TypeVar("S", bound=_StatusResponse)
|
|
43
|
-
|
|
44
|
-
|
|
45
35
|
class ManagedDatabaseClient:
|
|
46
36
|
"""Managed-database client with bounded retries over hotdata-framework.
|
|
47
37
|
|
|
@@ -130,70 +120,120 @@ class ManagedDatabaseClient:
|
|
|
130
120
|
0.6.0 SDK exposes (and requires) ``x_database_id`` on the Arrow
|
|
131
121
|
helper directly.
|
|
132
122
|
"""
|
|
133
|
-
|
|
134
|
-
result_id, x_database_id=database_id
|
|
135
|
-
)
|
|
136
|
-
|
|
137
|
-
def _poll(
|
|
138
|
-
self,
|
|
139
|
-
fetch: Callable[[], S],
|
|
140
|
-
*,
|
|
141
|
-
is_ready: Callable[[S], bool],
|
|
142
|
-
describe: str,
|
|
143
|
-
) -> S:
|
|
144
|
-
"""Poll ``fetch`` until ``is_ready`` is satisfied, or raise on failure/timeout.
|
|
145
|
-
|
|
146
|
-
``failed``/``cancelled`` statuses raise ``RuntimeError``; exceeding
|
|
147
|
-
:attr:`_QUERY_TIMEOUT_SECONDS` raises ``TimeoutError``.
|
|
148
|
-
"""
|
|
123
|
+
arrow = ArrowResultsApi(self._runtime.api)
|
|
149
124
|
deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
|
|
150
|
-
while
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
125
|
+
while True:
|
|
126
|
+
try:
|
|
127
|
+
return arrow.get_result_arrow(result_id, x_database_id=database_id)
|
|
128
|
+
except ResultNotReadyError:
|
|
129
|
+
# Waiting on the run should already have made this unreachable:
|
|
130
|
+
# a run reports `succeeded` only once its result is saved and
|
|
131
|
+
# ready. Tolerating it anyway costs nothing and removes the need
|
|
132
|
+
# to take that ordering on trust. The Arrow endpoint answers a
|
|
133
|
+
# result that is not ready with a small refusal rather than with
|
|
134
|
+
# data, so waiting here is cheap in the way waiting on the JSON
|
|
135
|
+
# result body -- which is what this change removed -- is not.
|
|
136
|
+
if time.monotonic() >= deadline:
|
|
137
|
+
raise
|
|
138
|
+
time.sleep(self._POLL_INTERVAL_SECONDS)
|
|
158
139
|
|
|
159
140
|
def _query_database_scoped(self, sql: str, *, database_id: str) -> str | None:
|
|
160
141
|
raw = QueryApi(self._runtime.api).query(
|
|
161
|
-
|
|
142
|
+
# Asked asynchronously because this caller wants a result id, not
|
|
143
|
+
# rows. A synchronous submit always builds an inline preview of the
|
|
144
|
+
# result and sends it -- megabytes, on a path that then reads the
|
|
145
|
+
# whole result as Arrow anyway and never looks at the preview. The
|
|
146
|
+
# async reply carries a run id and nothing else, and there is no way
|
|
147
|
+
# to suppress the preview on a synchronous one.
|
|
148
|
+
#
|
|
149
|
+
# It also settles the types: the preview is JSON, which has no Arrow
|
|
150
|
+
# schema and renders non-finite floats as null, so it could not have
|
|
151
|
+
# substituted for the Arrow fetch even when it holds every row.
|
|
152
|
+
#
|
|
153
|
+
# `var_async` is the generated SDK's spelling of the wire field
|
|
154
|
+
# `async`, which is a Python keyword and so cannot be the attribute
|
|
155
|
+
# name.
|
|
156
|
+
QueryRequest(sql=sql, var_async=True),
|
|
162
157
|
x_database_id=database_id,
|
|
163
158
|
)
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
return self.
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
159
|
+
# Both reply shapes carry `query_run_id`, and the run is the readiness
|
|
160
|
+
# signal for either -- a synchronous reply (which `async_after_ms` can
|
|
161
|
+
# still produce) returns rows inline but goes on saving the full result
|
|
162
|
+
# in the background, so it is not the finish line either.
|
|
163
|
+
if isinstance(raw, (QueryResponse, AsyncQueryResponse)):
|
|
164
|
+
return self._await_query_run(raw.query_run_id, database_id=database_id)
|
|
165
|
+
# Returning nothing here would read as an empty table: `fetch_table`
|
|
166
|
+
# answers `None`, `fetch_table_rows` turns that into `[]`, and a
|
|
167
|
+
# read-modify-write load would write only its new batch over rows it
|
|
168
|
+
# believed were not there. A reply shape this client does not know is a
|
|
169
|
+
# reason to stop, not to report emptiness. `HotdataClient` raises on the
|
|
170
|
+
# same condition.
|
|
171
|
+
raise RuntimeError(f"Unexpected query response type: {type(raw)!r}")
|
|
174
172
|
|
|
175
173
|
def _await_query_run(self, query_run_id: str, *, database_id: str) -> str | None:
|
|
174
|
+
"""Wait for a query run to finish; return the result id it produced.
|
|
175
|
+
|
|
176
|
+
The run is the whole wait. A run turns `succeeded` only after its result
|
|
177
|
+
has been saved and is `ready`, so `succeeded` needs no second check
|
|
178
|
+
against the result -- and asking the result endpoint instead would mean
|
|
179
|
+
downloading the entire result to read one field, which the server
|
|
180
|
+
refuses outright (413/429) once the result is large enough.
|
|
181
|
+
|
|
182
|
+
`result_id` comes off the run rather than off the query reply because a
|
|
183
|
+
`succeeded` run reports none when every row came back inline but the
|
|
184
|
+
result could not be saved for later retrieval.
|
|
185
|
+
"""
|
|
176
186
|
runs = QueryRunsApi(self._runtime.api)
|
|
177
|
-
|
|
187
|
+
deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
|
|
188
|
+
last_status: str | None = None
|
|
189
|
+
while time.monotonic() < deadline:
|
|
178
190
|
# Runs (like results) of database-scoped queries are database-scoped.
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
191
|
+
run = runs.get_query_run(query_run_id, x_database_id=database_id)
|
|
192
|
+
last_status = run.status
|
|
193
|
+
if run.status == "succeeded":
|
|
194
|
+
if run.result_id is None:
|
|
195
|
+
# A run succeeds with no result id when its rows were
|
|
196
|
+
# returned inline but the result could not be saved for
|
|
197
|
+
# later retrieval. Returning nothing here would surface as
|
|
198
|
+
# an empty table -- `fetch_table` answers `None`, and
|
|
199
|
+
# `fetch_table_rows` turns that into `[]`, which is the same
|
|
200
|
+
# answer it gives for a table that does not exist. A
|
|
201
|
+
# read-modify-write load would then read no existing rows
|
|
202
|
+
# and write only its new batch, dropping what was there.
|
|
203
|
+
# Terminal rather than transient: re-running the query
|
|
204
|
+
# cannot save a result that was already discarded.
|
|
205
|
+
# `getattr` because this runs while building an error: if
|
|
206
|
+
# the field ever goes away, losing the explanation is a far
|
|
207
|
+
# better outcome than an AttributeError replacing the raise.
|
|
208
|
+
warning = getattr(run, "warning_message", None)
|
|
209
|
+
raise RuntimeError(
|
|
210
|
+
f"Query run {query_run_id} succeeded but its result was not "
|
|
211
|
+
f"saved, so the table cannot be read"
|
|
212
|
+
+ (f": {warning}" if warning else "")
|
|
213
|
+
)
|
|
214
|
+
return run.result_id
|
|
215
|
+
if run.status == "interrupted":
|
|
216
|
+
# Terminal, but the server lost the run rather than rejecting
|
|
217
|
+
# the query, so it is the one failure here worth re-running.
|
|
218
|
+
# Raised pre-classified: `classify_sdk_error` cannot tell this
|
|
219
|
+
# apart from an ordinary RuntimeError and would call it terminal.
|
|
220
|
+
raise HotdataTransientError(
|
|
221
|
+
run.error_message or f"Query run {query_run_id} was interrupted"
|
|
222
|
+
)
|
|
223
|
+
if run.status == "failed":
|
|
224
|
+
raise RuntimeError(run.error_message or f"Query run {query_run_id} failed")
|
|
225
|
+
# Any other status keeps polling, including one this client has never
|
|
226
|
+
# seen. Treating an unrecognised status as terminal is the cheaper
|
|
227
|
+
# failure to diagnose and by far the more expensive one to suffer: a
|
|
228
|
+
# single status added upstream would then fail every query at once,
|
|
229
|
+
# where waiting costs one slow call. What made `interrupted`
|
|
230
|
+
# expensive was not the waiting, it was that the timeout never said
|
|
231
|
+
# which status it had waited on -- so the message now carries it.
|
|
232
|
+
time.sleep(self._POLL_INTERVAL_SECONDS)
|
|
233
|
+
raise TimeoutError(
|
|
234
|
+
f"Query run {query_run_id} did not finish within "
|
|
235
|
+
f"{self._QUERY_TIMEOUT_SECONDS}s (last status: {last_status})"
|
|
195
236
|
)
|
|
196
|
-
return result_id
|
|
197
237
|
|
|
198
238
|
def fetch_table_rows(self, *, database: str, schema: str, table: str) -> list[dict[str, Any]]:
|
|
199
239
|
result = self.fetch_table(database=database, schema=schema, table=table)
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "hotdata-framework"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.14.0"
|
|
8
8
|
description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -245,20 +245,67 @@ def test_list_qualified_table_names_passes_connection_id():
|
|
|
245
245
|
assert it.call_args.kwargs["connection_id"] == "conn_a"
|
|
246
246
|
|
|
247
247
|
|
|
248
|
-
def
|
|
248
|
+
def test_wait_result_ready_raises_on_a_failed_result():
|
|
249
249
|
client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
|
|
250
250
|
|
|
251
251
|
class FakeResultsApi:
|
|
252
252
|
def get_result(self, result_id: str):
|
|
253
|
-
return SimpleNamespace(status="
|
|
253
|
+
return SimpleNamespace(status="failed", error_message="out of memory")
|
|
254
254
|
|
|
255
255
|
with (
|
|
256
256
|
patch.object(client, "_results_api", return_value=FakeResultsApi()),
|
|
257
|
-
pytest.raises(RuntimeError, match="
|
|
257
|
+
pytest.raises(RuntimeError, match="out of memory"),
|
|
258
258
|
):
|
|
259
259
|
client._wait_result_ready("res_1", timeout_s=0.1, interval_s=0)
|
|
260
260
|
|
|
261
261
|
|
|
262
|
+
def test_unknown_result_status_times_out_and_names_the_status():
|
|
263
|
+
"""An unrecognised status keeps polling rather than being called terminal.
|
|
264
|
+
|
|
265
|
+
Failing fast on an unknown status would be easier to debug, and far worse to
|
|
266
|
+
live with: one status added upstream would fail every read at once, where
|
|
267
|
+
waiting costs a single slow call. The timeout names what it waited on, which
|
|
268
|
+
is what makes the omission findable -- and what was missing when
|
|
269
|
+
`interrupted` went unrecognised.
|
|
270
|
+
"""
|
|
271
|
+
client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
|
|
272
|
+
|
|
273
|
+
class FakeResultsApi:
|
|
274
|
+
def get_result(self, result_id: str):
|
|
275
|
+
return SimpleNamespace(status="something_new", error_message=None)
|
|
276
|
+
|
|
277
|
+
with (
|
|
278
|
+
patch.object(client, "_results_api", return_value=FakeResultsApi()),
|
|
279
|
+
pytest.raises(TimeoutError, match="something_new"),
|
|
280
|
+
):
|
|
281
|
+
client._wait_result_ready("res_1", timeout_s=0.05, interval_s=0)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def test_poll_query_run_returns_promptly_on_interrupted():
|
|
285
|
+
"""`interrupted` is terminal, so the poll must stop on it.
|
|
286
|
+
|
|
287
|
+
It was absent from the terminal set, so an interrupted run was polled for the
|
|
288
|
+
full timeout and then raised `TimeoutError` -- hiding a retryable condition
|
|
289
|
+
behind a five-minute wait, behind an error naming the wrong problem.
|
|
290
|
+
"""
|
|
291
|
+
client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
|
|
292
|
+
calls: list[str] = []
|
|
293
|
+
|
|
294
|
+
class FakeQueryRunsApi:
|
|
295
|
+
def get_query_run(self, query_run_id: str):
|
|
296
|
+
calls.append(query_run_id)
|
|
297
|
+
return SimpleNamespace(
|
|
298
|
+
status="interrupted", error_message="instance lost", result_id=None
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
with patch.object(client, "_query_runs_api", return_value=FakeQueryRunsApi()):
|
|
302
|
+
run = client._poll_query_run("qrun_1", timeout_s=30.0, interval_s=0)
|
|
303
|
+
|
|
304
|
+
assert run.status == "interrupted"
|
|
305
|
+
# One request, not a timeout's worth.
|
|
306
|
+
assert len(calls) == 1
|
|
307
|
+
|
|
308
|
+
|
|
262
309
|
def test_connection_id_by_name_raises_on_duplicate_names():
|
|
263
310
|
client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
|
|
264
311
|
listing = SimpleNamespace(
|