hotdata-framework 0.12.1__tar.gz → 0.14.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/CHANGELOG.md +113 -0
  2. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/PKG-INFO +2 -2
  3. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/README.md +1 -1
  4. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/client.py +17 -6
  5. hotdata_framework-0.14.0/hotdata_framework/errors.py +140 -0
  6. hotdata_framework-0.14.0/hotdata_framework/managed_client.py +313 -0
  7. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/pyproject.toml +1 -1
  8. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_client.py +52 -6
  9. hotdata_framework-0.14.0/tests/test_errors.py +125 -0
  10. hotdata_framework-0.14.0/tests/test_managed_client.py +969 -0
  11. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_retry_policy.py +10 -4
  12. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/uv.lock +1 -1
  13. hotdata_framework-0.12.1/hotdata_framework/errors.py +0 -40
  14. hotdata_framework-0.12.1/hotdata_framework/managed_client.py +0 -241
  15. hotdata_framework-0.12.1/tests/test_errors.py +0 -48
  16. hotdata_framework-0.12.1/tests/test_managed_client.py +0 -248
  17. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/CODEOWNERS +0 -0
  18. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/dependabot.yml +0 -0
  19. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/check-release.yml +0 -0
  20. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/ci.yml +0 -0
  21. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/dependabot-automerge.yml +0 -0
  22. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/publish.yml +0 -0
  23. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.github/workflows/release.yml +0 -0
  24. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/.gitignore +0 -0
  25. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/CONTRACT.md +0 -0
  26. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/RELEASING.md +0 -0
  27. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/examples/basic_usage.py +0 -0
  28. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/__init__.py +0 -0
  29. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/databases.py +0 -0
  30. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/env.py +0 -0
  31. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/health.py +0 -0
  32. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/py.typed +0 -0
  33. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/hotdata_framework/result.py +0 -0
  34. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/check-release.py +0 -0
  35. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/extract-changelog.py +0 -0
  36. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/publish-workflow.sh +0 -0
  37. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/release.sh +0 -0
  38. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/scripts/update_changelog.py +0 -0
  39. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_contract.py +0 -0
  40. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_databases.py +0 -0
  41. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_health.py +0 -0
  42. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_indexes.py +0 -0
  43. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_request_timeout.py +0 -0
  44. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_result.py +0 -0
  45. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_update_changelog.py +0 -0
  46. {hotdata_framework-0.12.1 → hotdata_framework-0.14.0}/tests/test_version.py +0 -0
@@ -8,6 +8,119 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
8
8
  ## [Unreleased]
9
9
 
10
10
 
11
+ ## [0.14.0] - 2026-09-01
12
+
13
+ ### Fixed
14
+
15
+ - fix(managed): wait on the query run instead of downloading the result to check it.
16
+
17
+ Reading a managed table made three calls and used one. `POST /v1/query` returned
18
+ an inline preview of the rows, `GET /v1/results/{id}` was polled until the
19
+ result was `ready`, and the result was then fetched as Arrow. Only the Arrow
20
+ copy was used.
21
+
22
+ The readiness poll was the expensive one. `limit` on that endpoint defaults to
23
+ unbounded, so polling a ready result downloads the entire result body to read
24
+ one status field. It is also the wrong endpoint to lean on as a table grows:
25
+ a JSON body over the instance's per-fetch memory budget is refused with 413,
26
+ and one that would fit alone but not alongside concurrent JSON fetches with
27
+ 429 — so the readiness check starts failing on exactly the largest tables.
28
+
29
+ The query is now submitted with `async`, so the server returns a run id rather
30
+ than a preview, and readiness comes from `GET /v1/query-runs/{id}`, which
31
+ carries no rows at any size. `result_id` is read off the run rather than off
32
+ the query reply, because a run can succeed having saved nothing and the run is
33
+ what reports that — and that case now raises rather than reading as an empty
34
+ table. `fetch_table` answered `None` for it, which `fetch_table_rows` turns
35
+ into `[]`, the same answer both give for a table that is not synced. A
36
+ read-modify-write load would have read no existing rows and written only its
37
+ new batch, dropping every row already there. A reply shape this client does not
38
+ recognise raises for the same reason, as `HotdataClient` already did — so a
39
+ `None` from `fetch_table` now means one thing only: the table is not synced.
40
+ Arrow stays the only path the data travels, so column types come from the
41
+ server's schema rather than being inferred from JSON.
42
+
43
+ Costs one extra round trip on a query that would have answered synchronously,
44
+ in exchange for not transferring the result twice.
45
+
46
+ The Arrow fetch now also waits out a result that reports itself not ready, in
47
+ case that ordering ever stops holding. It should be unreachable, and it is
48
+ cheap to keep: that endpoint answers a result which is not ready with a small
49
+ refusal rather than with data, which is exactly what made waiting on the JSON
50
+ result body expensive and waiting here not.
51
+
52
+ - fix(managed): recognise `interrupted`, and drop a run status the API never sends.
53
+
54
+ Both `ManagedDatabaseClient` and `HotdataClient` treated `failed` and
55
+ `cancelled` as the terminal run failures. `cancelled` is not a status this API
56
+ returns. `interrupted` is — a run whose server was replaced before it finished
57
+ — and it matched neither, so an interrupted run was polled for the full
58
+ five-minute timeout and then raised `TimeoutError`: a retryable condition
59
+ hidden behind a long wait and an error naming the wrong problem.
60
+
61
+ On `ManagedDatabaseClient` an interrupted run is now raised as transient, so
62
+ the surrounding retry re-submits the query. That needed `classify_sdk_error` to
63
+ pass an already-classified error through unchanged rather than demoting a
64
+ caller-raised transient error to terminal. `HotdataClient.execute_sql` now
65
+ fails fast on it with the run's own message.
66
+
67
+ Both polls keep enumerating the statuses that mean *finished*, and an
68
+ unrecognised status still waits. Calling an unknown status terminal would make
69
+ the omission easier to diagnose and much worse to live with: one status added
70
+ upstream would fail every read at once, where waiting costs a single slow call.
71
+ What made `interrupted` expensive was not the waiting — it was that the
72
+ timeout never said which status it had been waiting on. Both timeouts now name
73
+ it.
74
+
75
+ ## [0.13.0] - 2026-08-27
76
+
77
+ ### Fixed
78
+
79
+ - fix(load): retry an `append` load instead of running it at most once.
80
+
81
+ `append` was excluded from retries on the grounds that it is not idempotent:
82
+ if the server commits but the response is lost, a retry would duplicate rows.
83
+ That is not how the server behaves. It keys a receipt on `upload_id`, and a
84
+ re-POST of the same id replays the committed result instead of applying the
85
+ load again — so what makes a retry safe is re-sending the same upload, not
86
+ the mode. This client stages once, in `upload_parquet`, outside the retried
87
+ operation, so the invariant holds for every mode.
88
+
89
+ The exclusion cost real availability. The destination serialises writes per
90
+ table and refuses rather than queues, so concurrent writers to one table get
91
+ `409 RESOURCE_LOCKED` — and an append had no budget to wait it out, whatever
92
+ `max_retries` the caller had configured.
93
+
94
+ `HotdataClient.load_managed_table(file=...)` uploads inside the call and so
95
+ does not hold the invariant. It is unwrapped and unaffected.
96
+
97
+ - fix(errors): classify a 409 by its `error.code` rather than by the status alone.
98
+
99
+ `CONFLICT` is now terminal: it means the request cannot succeed as posted, so
100
+ the previous behaviour spent the entire retry budget arriving at the same
101
+ answer. `RESOURCE_LOCKED` stays transient. A 409 with no error envelope — a
102
+ failed query result, say — is classified as before.
103
+
104
+ - fix(retry): honour `Retry-After`, and jitter the backoff.
105
+
106
+ `Retry-After` is taken as a floor on the ramp, capped like the ramp so a bad
107
+ header cannot park an attempt for an hour. Jitter of up to +50% is added on
108
+ top and never subtracted, so a stated `Retry-After` is not undercut. Without
109
+ it, writers that collided on one table retry in lockstep and collide again.
110
+
111
+ This lengthens a 20-attempt budget from 285s to roughly 316-405s.
112
+
113
+ - docs: scope the "a load is not idempotent" claim in the README and in
114
+ `test_retry_policy` to the transport layer, which is where it is still true
115
+ and where those two were always talking about. Left unscoped they read as
116
+ repo-wide and contradict the call-layer retry above.
117
+
118
+ ### Added
119
+
120
+ - `HotdataError` carries `status_code`, `code` and `retry_after_seconds`. The
121
+ message is flattened and truncated for readability, so it could not serve as
122
+ a discriminator; these can.
123
+
11
124
  ## [0.12.1] - 2026-08-18
12
125
 
13
126
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.12.1
3
+ Version: 0.14.0
4
4
  Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -38,7 +38,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
38
38
 
39
39
  - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
40
40
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
41
- - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
41
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
42
42
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
43
43
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
44
44
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -10,7 +10,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
10
10
 
11
11
  - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
12
12
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
13
- - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
13
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
14
14
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
15
15
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
16
16
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -77,8 +77,17 @@ VectorMetric = Literal["l2", "cosine", "dot"]
77
77
  _INDEX_TYPES = frozenset(get_args(IndexType))
78
78
  _VECTOR_METRICS = frozenset(get_args(VectorMetric))
79
79
 
80
- _TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
81
- _RESULT_FAILURE = frozenset({"failed", "cancelled"})
80
+ # Query-run statuses that mean the run is over. `interrupted` belongs here --
81
+ # omitting it is what made an interrupted run wait out the full timeout -- and
82
+ # `cancelled`, listed here for a long time, is not a status this API sends.
83
+ #
84
+ # Enumerating the terminal side rather than the in-flight side is deliberate. An
85
+ # unrecognised status then keeps polling and costs one slow call, where treating
86
+ # it as terminal would fail every query the moment a status is added upstream.
87
+ # The timeout names the status it last saw, so a missing one is diagnosable
88
+ # without being dangerous.
89
+ _RUN_TERMINAL = frozenset({"succeeded", "failed", "interrupted"})
90
+ _RESULT_FAILURE = frozenset({"failed"})
82
91
  # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
83
92
  _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
84
93
 
@@ -918,7 +927,7 @@ class HotdataClient:
918
927
  last = None
919
928
  while time.monotonic() < deadline:
920
929
  last = runs.get_query_run(query_run_id)
921
- if last.status in _TERMINAL:
930
+ if last.status in _RUN_TERMINAL:
922
931
  return last
923
932
  time.sleep(interval_s)
924
933
  raise TimeoutError(
@@ -985,9 +994,11 @@ class HotdataClient:
985
994
  durable state rather than from a connection that has to stay alive. That
986
995
  also gives a caller a handle: the job id is returned on
987
996
  `LoadManagedTableResult`, so "did it land?" is answerable after a lost
988
- response. `append` stays non-retryable -- knowing the id makes the question
989
- answerable, it does not make a blind re-submission safe, and that call is
990
- the caller's to make.
997
+ response. That answer is a convenience rather than a precondition for
998
+ retrying: re-sending the same upload_id replays the server's receipt
999
+ instead of applying the load a second time, which is what makes a retry
1000
+ safe in every mode. It stops being safe for a caller that re-stages the
1001
+ upload, because a fresh upload id has no receipt to replay.
991
1002
 
992
1003
  `partially_succeeded` is terminal and carries a message, so it is raised
993
1004
  rather than returned -- a caller asked for a table's contents to be
@@ -0,0 +1,140 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections.abc import Mapping
5
+
6
+ from hotdata.rest import ApiException
7
+
8
+ # The API explains a 409 with a machine-readable code, and the two it sends
9
+ # mean opposite things to a retry policy. RESOURCE_LOCKED is a refusal taken
10
+ # before any work: the insert that would have created the unit of work lost a
11
+ # unique-constraint race, so nothing was claimed and nothing was written.
12
+ # CONFLICT is the opposite — the request cannot succeed as posted, so retrying
13
+ # spends the whole budget arriving at the same answer.
14
+ _TERMINAL_CONFLICT_CODE = "CONFLICT"
15
+
16
+
17
+ class HotdataError(RuntimeError):
18
+ """An API failure, carrying what a retry policy needs to decide.
19
+
20
+ The message cannot be the discriminator: it is flattened and truncated for
21
+ readability, so keying on it means substring-matching prose. ``status_code``
22
+ and ``code`` are the machine-readable form of the same answer, and
23
+ ``retry_after_seconds`` is the server's own estimate of how long the
24
+ condition it just refused will last.
25
+ """
26
+
27
+ def __init__(
28
+ self,
29
+ message: str,
30
+ *,
31
+ status_code: int | None = None,
32
+ code: str | None = None,
33
+ retry_after_seconds: float | None = None,
34
+ ) -> None:
35
+ super().__init__(message)
36
+ self.status_code = status_code
37
+ self.code = code
38
+ self.retry_after_seconds = retry_after_seconds
39
+
40
+
41
+ class HotdataTransientError(HotdataError):
42
+ pass
43
+
44
+
45
+ class HotdataTerminalError(HotdataError):
46
+ pass
47
+
48
+
49
+ def _error_code(body: object) -> str | None:
50
+ """The ``error.code`` an API error envelope carries, if this body is one.
51
+
52
+ Not every 409 comes from an endpoint that speaks the envelope — a failed
53
+ query result is reported as one and carries a result document instead — so
54
+ a missing code is ordinary, and callers fall back to the status.
55
+ """
56
+ if not isinstance(body, (str, bytes, bytearray)):
57
+ return None
58
+ try:
59
+ parsed: object = json.loads(body)
60
+ except ValueError:
61
+ return None
62
+ if not isinstance(parsed, Mapping):
63
+ return None
64
+ error: object = parsed.get("error")
65
+ if not isinstance(error, Mapping):
66
+ return None
67
+ code: object = error.get("code")
68
+ return code if isinstance(code, str) else None
69
+
70
+
71
+ def _retry_after_seconds(headers: object) -> float | None:
72
+ """``Retry-After`` as a number of seconds, when the response states one.
73
+
74
+ Only the delta-seconds form is read. That is what the API sends, and the
75
+ HTTP-date form would need a comparison against a server clock we do not
76
+ have to be worth anything.
77
+ """
78
+ if not isinstance(headers, Mapping):
79
+ return None
80
+ raw: object = headers.get("Retry-After")
81
+ if raw is None:
82
+ # The SDK hands us urllib3's case-insensitive mapping and the API sends
83
+ # the header lower-cased, so the direct hit is what normally answers.
84
+ # Fall back for any plain dict that reaches us instead — a missed
85
+ # header is silent, and silence here reads as "the server asked for
86
+ # nothing".
87
+ raw = next((v for k, v in headers.items() if str(k).lower() == "retry-after"), None)
88
+ if raw is None:
89
+ return None
90
+ try:
91
+ seconds = float(str(raw).strip())
92
+ except ValueError:
93
+ return None
94
+ return seconds if seconds >= 0 else None
95
+
96
+
97
+ def _error_class(status_code: int, code: str | None) -> type[HotdataError]:
98
+ if status_code == 409 and code == _TERMINAL_CONFLICT_CODE:
99
+ # The request cannot succeed as posted — an upload already consumed
100
+ # with nothing to replay, a receipt naming a different target, an
101
+ # incompatible column type. Every retry reaches the same 409.
102
+ return HotdataTerminalError
103
+ if status_code in (408, 409, 425, 429):
104
+ return HotdataTransientError
105
+ if status_code == 501:
106
+ # Not Implemented is a permanent capability gap (e.g. the storage
107
+ # backend cannot issue presigned URLs) — retrying cannot succeed.
108
+ return HotdataTerminalError
109
+ if 500 <= status_code <= 599:
110
+ return HotdataTransientError
111
+ return HotdataTerminalError
112
+
113
+
114
+ def classify_sdk_error(error: Exception) -> HotdataError:
115
+ if isinstance(error, HotdataError):
116
+ # Already classified. A caller that read transience off a typed status
117
+ # -- an interrupted query run, say -- knows more than this function can
118
+ # recover from the exception, and the fallback below would demote it to
119
+ # terminal and cost the retry.
120
+ return error
121
+ if isinstance(error, TimeoutError):
122
+ return HotdataTransientError(str(error))
123
+ if isinstance(error, ConnectionError):
124
+ return HotdataTransientError(str(error))
125
+ if isinstance(error, ApiException):
126
+ status_code = int(error.status or 0)
127
+ message = f"{status_code}: {error.reason or 'unknown error'}"
128
+ # The response body is where the API explains itself (e.g. which
129
+ # header is missing) — without it "400: Bad Request" is undebuggable.
130
+ body: object = getattr(error, "body", None)
131
+ if body:
132
+ message = f"{message} — {' '.join(str(body).split())[:500]}"
133
+ code = _error_code(body)
134
+ return _error_class(status_code, code)(
135
+ message,
136
+ status_code=status_code,
137
+ code=code,
138
+ retry_after_seconds=_retry_after_seconds(getattr(error, "headers", None)),
139
+ )
140
+ return HotdataTerminalError(str(error))
@@ -0,0 +1,313 @@
1
+ """Retry-wrapped managed-database client shared by Hotdata adapter packages.
2
+
3
+ Both hotdata-airflow and hotdata-dlt-destination import this module so that
4
+ the higher-level client logic (retries, Arrow queries, table management) lives
5
+ in one place rather than being duplicated per adapter.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import random
11
+ import time
12
+ from collections.abc import Callable
13
+ from typing import Any, TypeVar
14
+
15
+ import pyarrow as pa
16
+ from hotdata.api.query_api import QueryApi
17
+ from hotdata.api.query_runs_api import QueryRunsApi
18
+ from hotdata.arrow import ResultNotReadyError
19
+ from hotdata.arrow import ResultsApi as ArrowResultsApi
20
+ from hotdata.models.async_query_response import AsyncQueryResponse
21
+ from hotdata.models.query_request import QueryRequest
22
+ from hotdata.models.query_response import QueryResponse
23
+
24
+ from hotdata_framework.client import HotdataClient as RuntimeClient
25
+ from hotdata_framework.client import ManagedLoadMode
26
+ from hotdata_framework.databases import LoadManagedTableResult, ManagedDatabase
27
+ from hotdata_framework.errors import (
28
+ HotdataTransientError,
29
+ classify_sdk_error,
30
+ )
31
+
32
+ T = TypeVar("T")
33
+
34
+
35
+ class ManagedDatabaseClient:
36
+ """Managed-database client with bounded retries over hotdata-framework.
37
+
38
+ This is the shared client used by Hotdata adapter packages (Airflow,
39
+ dlt, etc.). It wraps the lower-level RuntimeClient with retry logic,
40
+ Arrow-based result fetching, and convenience helpers for the managed
41
+ database lifecycle.
42
+ """
43
+
44
+ _QUERY_TIMEOUT_SECONDS = 300.0
45
+ _POLL_INTERVAL_SECONDS = 0.4
46
+ _MAX_BACKOFF_SECONDS = 30.0
47
+ # Spread as a fraction of the wait, added on top of it. Half an interval is
48
+ # enough to decorrelate writers that started together without materially
49
+ # changing how long the budget lasts.
50
+ _RETRY_JITTER_FRACTION = 0.5
51
+
52
+ def __init__(
53
+ self,
54
+ *,
55
+ api_key: str,
56
+ workspace_id: str,
57
+ api_base_url: str,
58
+ max_retries: int,
59
+ retry_backoff_seconds: float,
60
+ request_timeout: float | tuple[float, float] | None = None,
61
+ ) -> None:
62
+ self._max_retries = max_retries
63
+ self._retry_backoff_seconds = retry_backoff_seconds
64
+ self._runtime = RuntimeClient(
65
+ api_key,
66
+ workspace_id,
67
+ host=api_base_url.rstrip("/"),
68
+ request_timeout=request_timeout,
69
+ )
70
+
71
+ def close(self) -> None:
72
+ self._runtime.close()
73
+
74
+ def ensure_managed_database(
75
+ self,
76
+ name: str,
77
+ *,
78
+ schema: str,
79
+ tables: list[str],
80
+ create_if_missing: bool,
81
+ ) -> ManagedDatabase:
82
+ def operation() -> ManagedDatabase:
83
+ try:
84
+ return self._runtime.resolve_managed_database(name)
85
+ except KeyError:
86
+ if not create_if_missing:
87
+ raise
88
+ return self._runtime.create_managed_database(
89
+ description=name,
90
+ schema=schema,
91
+ tables=sorted(set(tables)),
92
+ )
93
+
94
+ return self._request_with_retry(operation)
95
+
96
+ def table_is_synced(self, database: str, table: str, *, schema: str) -> bool:
97
+ for managed_table in self._runtime.list_managed_tables(database, schema=schema):
98
+ if managed_table.table == table:
99
+ return managed_table.synced
100
+ return False
101
+
102
+ def fetch_table(self, *, database: str, schema: str, table: str) -> pa.Table | None:
103
+ def operation() -> pa.Table | None:
104
+ if not self.table_is_synced(database, table, schema=schema):
105
+ return None
106
+ db = self._runtime.resolve_managed_database(database)
107
+ sql = f'SELECT * FROM "default"."{schema}"."{table}"'
108
+ result_id = self._query_database_scoped(sql, database_id=db.id)
109
+ if result_id is None:
110
+ return None
111
+ return self._fetch_result_arrow(result_id, database_id=db.id)
112
+
113
+ return self._request_with_retry(operation)
114
+
115
+ def _fetch_result_arrow(self, result_id: str, *, database_id: str) -> pa.Table:
116
+ """Fetch a ready result as Arrow, carrying the database scope.
117
+
118
+ Results of database-scoped queries are themselves database-scoped —
119
+ the results endpoints reject requests without the scope. The hotdata
120
+ 0.6.0 SDK exposes (and requires) ``x_database_id`` on the Arrow
121
+ helper directly.
122
+ """
123
+ arrow = ArrowResultsApi(self._runtime.api)
124
+ deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
125
+ while True:
126
+ try:
127
+ return arrow.get_result_arrow(result_id, x_database_id=database_id)
128
+ except ResultNotReadyError:
129
+ # Waiting on the run should already have made this unreachable:
130
+ # a run reports `succeeded` only once its result is saved and
131
+ # ready. Tolerating it anyway costs nothing and removes the need
132
+ # to take that ordering on trust. The Arrow endpoint answers a
133
+ # result that is not ready with a small refusal rather than with
134
+ # data, so waiting here is cheap in the way waiting on the JSON
135
+ # result body -- which is what this change removed -- is not.
136
+ if time.monotonic() >= deadline:
137
+ raise
138
+ time.sleep(self._POLL_INTERVAL_SECONDS)
139
+
140
+ def _query_database_scoped(self, sql: str, *, database_id: str) -> str | None:
141
+ raw = QueryApi(self._runtime.api).query(
142
+ # Asked asynchronously because this caller wants a result id, not
143
+ # rows. A synchronous submit always builds an inline preview of the
144
+ # result and sends it -- megabytes, on a path that then reads the
145
+ # whole result as Arrow anyway and never looks at the preview. The
146
+ # async reply carries a run id and nothing else, and there is no way
147
+ # to suppress the preview on a synchronous one.
148
+ #
149
+ # It also settles the types: the preview is JSON, which has no Arrow
150
+ # schema and renders non-finite floats as null, so it could not have
151
+ # substituted for the Arrow fetch even when it holds every row.
152
+ #
153
+ # `var_async` is the generated SDK's spelling of the wire field
154
+ # `async`, which is a Python keyword and so cannot be the attribute
155
+ # name.
156
+ QueryRequest(sql=sql, var_async=True),
157
+ x_database_id=database_id,
158
+ )
159
+ # Both reply shapes carry `query_run_id`, and the run is the readiness
160
+ # signal for either -- a synchronous reply (which `async_after_ms` can
161
+ # still produce) returns rows inline but goes on saving the full result
162
+ # in the background, so it is not the finish line either.
163
+ if isinstance(raw, (QueryResponse, AsyncQueryResponse)):
164
+ return self._await_query_run(raw.query_run_id, database_id=database_id)
165
+ # Returning nothing here would read as an empty table: `fetch_table`
166
+ # answers `None`, `fetch_table_rows` turns that into `[]`, and a
167
+ # read-modify-write load would write only its new batch over rows it
168
+ # believed were not there. A reply shape this client does not know is a
169
+ # reason to stop, not to report emptiness. `HotdataClient` raises on the
170
+ # same condition.
171
+ raise RuntimeError(f"Unexpected query response type: {type(raw)!r}")
172
+
173
+ def _await_query_run(self, query_run_id: str, *, database_id: str) -> str | None:
174
+ """Wait for a query run to finish; return the result id it produced.
175
+
176
+ The run is the whole wait. A run turns `succeeded` only after its result
177
+ has been saved and is `ready`, so `succeeded` needs no second check
178
+ against the result -- and asking the result endpoint instead would mean
179
+ downloading the entire result to read one field, which the server
180
+ refuses outright (413/429) once the result is large enough.
181
+
182
+ `result_id` comes off the run rather than off the query reply because a
183
+ `succeeded` run reports none when every row came back inline but the
184
+ result could not be saved for later retrieval.
185
+ """
186
+ runs = QueryRunsApi(self._runtime.api)
187
+ deadline = time.monotonic() + self._QUERY_TIMEOUT_SECONDS
188
+ last_status: str | None = None
189
+ while time.monotonic() < deadline:
190
+ # Runs (like results) of database-scoped queries are database-scoped.
191
+ run = runs.get_query_run(query_run_id, x_database_id=database_id)
192
+ last_status = run.status
193
+ if run.status == "succeeded":
194
+ if run.result_id is None:
195
+ # A run succeeds with no result id when its rows were
196
+ # returned inline but the result could not be saved for
197
+ # later retrieval. Returning nothing here would surface as
198
+ # an empty table -- `fetch_table` answers `None`, and
199
+ # `fetch_table_rows` turns that into `[]`, which is the same
200
+ # answer it gives for a table that does not exist. A
201
+ # read-modify-write load would then read no existing rows
202
+ # and write only its new batch, dropping what was there.
203
+ # Terminal rather than transient: re-running the query
204
+ # cannot save a result that was already discarded.
205
+ # `getattr` because this runs while building an error: if
206
+ # the field ever goes away, losing the explanation is a far
207
+ # better outcome than an AttributeError replacing the raise.
208
+ warning = getattr(run, "warning_message", None)
209
+ raise RuntimeError(
210
+ f"Query run {query_run_id} succeeded but its result was not "
211
+ f"saved, so the table cannot be read"
212
+ + (f": {warning}" if warning else "")
213
+ )
214
+ return run.result_id
215
+ if run.status == "interrupted":
216
+ # Terminal, but the server lost the run rather than rejecting
217
+ # the query, so it is the one failure here worth re-running.
218
+ # Raised pre-classified: `classify_sdk_error` cannot tell this
219
+ # apart from an ordinary RuntimeError and would call it terminal.
220
+ raise HotdataTransientError(
221
+ run.error_message or f"Query run {query_run_id} was interrupted"
222
+ )
223
+ if run.status == "failed":
224
+ raise RuntimeError(run.error_message or f"Query run {query_run_id} failed")
225
+ # Any other status keeps polling, including one this client has never
226
+ # seen. Treating an unrecognised status as terminal is the cheaper
227
+ # failure to diagnose and by far the more expensive one to suffer: a
228
+ # single status added upstream would then fail every query at once,
229
+ # where waiting costs one slow call. What made `interrupted`
230
+ # expensive was not the waiting, it was that the timeout never said
231
+ # which status it had waited on -- so the message now carries it.
232
+ time.sleep(self._POLL_INTERVAL_SECONDS)
233
+ raise TimeoutError(
234
+ f"Query run {query_run_id} did not finish within "
235
+ f"{self._QUERY_TIMEOUT_SECONDS}s (last status: {last_status})"
236
+ )
237
+
238
+ def fetch_table_rows(self, *, database: str, schema: str, table: str) -> list[dict[str, Any]]:
239
+ result = self.fetch_table(database=database, schema=schema, table=table)
240
+ return result.to_pylist() if result is not None else []
241
+
242
+ def upload_parquet(self, path: str) -> str:
243
+ return self._request_with_retry(lambda: self._runtime.upload_parquet(path))
244
+
245
+ def load_managed_table(
246
+ self,
247
+ database: str,
248
+ table: str,
249
+ *,
250
+ schema: str,
251
+ upload_id: str,
252
+ mode: ManagedLoadMode = "replace",
253
+ key: list[str] | None = None,
254
+ ) -> LoadManagedTableResult:
255
+ # Retryable in every mode, append included. A retry re-sends the SAME
256
+ # upload_id, and the server keys a receipt on it: a replay returns the
257
+ # committed result rather than applying the load a second time. So the
258
+ # invariant that makes this safe is the upload id, not the mode — a
259
+ # caller that re-stages the upload between attempts mints a new id,
260
+ # loses the receipt, and a retried append would then duplicate rows.
261
+ # This client stages once, in upload_parquet, outside the operation
262
+ # retried here. `HotdataClient.load_managed_table(file=...)` uploads
263
+ # inside the call and so does not hold the invariant; it is unwrapped,
264
+ # and retrying an append through it is the caller's to justify.
265
+ #
266
+ # `key` is the merge key for delete/update/upsert loads: when set it is
267
+ # matched per-load instead of a key declared at table creation. Omit it
268
+ # to use the table's declared key. Ignored for replace/append.
269
+ return self._request_with_retry(
270
+ lambda: self._runtime.load_managed_table(
271
+ database,
272
+ table,
273
+ schema=schema,
274
+ upload_id=upload_id,
275
+ mode=mode,
276
+ key=key,
277
+ )
278
+ )
279
+
280
+ def _request_with_retry(self, operation: Callable[[], T]) -> T:
281
+ max_attempts = self._max_retries
282
+ for attempt in range(1, max_attempts + 1):
283
+ try:
284
+ return operation()
285
+ except Exception as error:
286
+ mapped_error = classify_sdk_error(error.__cause__ or error)
287
+ if isinstance(mapped_error, HotdataTransientError) and attempt < max_attempts:
288
+ time.sleep(self._retry_delay(attempt, mapped_error.retry_after_seconds))
289
+ continue
290
+ raise mapped_error from error
291
+ raise RuntimeError("No retry attempts configured")
292
+
293
+ def _retry_delay(self, attempt: int, retry_after_seconds: float | None) -> float:
294
+ """A linear ramp, floored by the server's Retry-After and spread by jitter.
295
+
296
+ Retry-After is a floor rather than a replacement: it says how long the
297
+ condition just refused typically lasts, while the ramp is what gives up
298
+ eventually, and taking the larger of the two honours both. It is capped
299
+ like the ramp so a hostile or mistaken header cannot park an attempt for
300
+ an hour.
301
+
302
+ Jitter is added on top and never subtracted, so a stated Retry-After is
303
+ not undercut. It matters because the callers that collide are the ones
304
+ that started together: writers refused by one table's lock would retry
305
+ in lockstep on an identical ramp and re-collide every time.
306
+ _MAX_BACKOFF_SECONDS caps the ramp, deliberately not the jitter above
307
+ it — clamping the total would flatten every late attempt onto the same
308
+ value and re-correlate exactly the waits that most need spreading.
309
+ """
310
+ base = min(self._retry_backoff_seconds * attempt, self._MAX_BACKOFF_SECONDS)
311
+ if retry_after_seconds is not None:
312
+ base = max(base, min(retry_after_seconds, self._MAX_BACKOFF_SECONDS))
313
+ return base * (1.0 + random.random() * self._RETRY_JITTER_FRACTION)
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.12.1"
7
+ version = "0.14.0"
8
8
  description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"