hotdata-framework 0.12.0__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/publish.yml +15 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/release.yml +18 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/CHANGELOG.md +54 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/PKG-INFO +2 -2
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/README.md +1 -1
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/client.py +121 -1
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/databases.py +5 -0
- hotdata_framework-0.13.0/hotdata_framework/errors.py +134 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/managed_client.py +41 -9
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/pyproject.toml +1 -1
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_client.py +330 -1
- hotdata_framework-0.13.0/tests/test_errors.py +125 -0
- hotdata_framework-0.13.0/tests/test_managed_client.py +430 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_retry_policy.py +10 -4
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/uv.lock +1 -1
- hotdata_framework-0.12.0/hotdata_framework/errors.py +0 -40
- hotdata_framework-0.12.0/tests/test_errors.py +0 -48
- hotdata_framework-0.12.0/tests/test_managed_client.py +0 -248
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.gitignore +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/CONTRACT.md +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/RELEASING.md +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/__init__.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_contract.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_indexes.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_version.py +0 -0
|
@@ -31,6 +31,21 @@ jobs:
|
|
|
31
31
|
# The tag being released, whether it arrived by push or by dispatch.
|
|
32
32
|
TAG: ${{ inputs.tag || github.ref_name }}
|
|
33
33
|
steps:
|
|
34
|
+
# Before checkout, because checkout resolves the input as an arbitrary ref:
|
|
35
|
+
# a branch or SHA is fetched first and only rejected later by the version
|
|
36
|
+
# match below, which is also looser (`^v[0-9]`). Strict here so the dispatch
|
|
37
|
+
# contract matches release.yml — release.sh only ever produces X.Y.Z.
|
|
38
|
+
- name: Validate release tag format
|
|
39
|
+
if: github.event_name == 'workflow_dispatch'
|
|
40
|
+
env:
|
|
41
|
+
INPUT_TAG: ${{ inputs.tag }}
|
|
42
|
+
run: |
|
|
43
|
+
set -euo pipefail
|
|
44
|
+
if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
|
|
45
|
+
echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
|
|
46
|
+
exit 1
|
|
47
|
+
fi
|
|
48
|
+
|
|
34
49
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
35
50
|
with:
|
|
36
51
|
ref: ${{ inputs.tag || github.ref_name }}
|
|
@@ -25,6 +25,24 @@ jobs:
|
|
|
25
25
|
# The tag being released, whether it arrived by push or by dispatch.
|
|
26
26
|
TAG: ${{ inputs.tag || github.ref_name }}
|
|
27
27
|
steps:
|
|
28
|
+
# Before checkout, because checkout resolves the input as an arbitrary ref
|
|
29
|
+
# — and this workflow needs the guard more than publish.yml does. It holds
|
|
30
|
+
# `contents: write`, and action-gh-release CREATES a tag when tag_name does
|
|
31
|
+
# not resolve to one, so an unvalidated `tag: main` would check out cleanly
|
|
32
|
+
# and leave refs/tags/main plus a release named for it. The push path is
|
|
33
|
+
# constrained by the v[0-9]* filter; the dispatch path was not constrained
|
|
34
|
+
# at all.
|
|
35
|
+
- name: Validate release tag format
|
|
36
|
+
if: github.event_name == 'workflow_dispatch'
|
|
37
|
+
env:
|
|
38
|
+
INPUT_TAG: ${{ inputs.tag }}
|
|
39
|
+
run: |
|
|
40
|
+
set -euo pipefail
|
|
41
|
+
if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
|
|
42
|
+
echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
|
|
43
|
+
exit 1
|
|
44
|
+
fi
|
|
45
|
+
|
|
28
46
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
29
47
|
with:
|
|
30
48
|
ref: ${{ inputs.tag || github.ref_name }}
|
|
@@ -7,6 +7,60 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.13.0] - 2026-08-27
|
|
11
|
+
|
|
12
|
+
### Fixed
|
|
13
|
+
|
|
14
|
+
- fix(load): retry an `append` load instead of running it at most once.
|
|
15
|
+
|
|
16
|
+
`append` was excluded from retries on the grounds that it is not idempotent:
|
|
17
|
+
if the server commits but the response is lost, a retry would duplicate rows.
|
|
18
|
+
That is not how the server behaves. It keys a receipt on `upload_id`, and a
|
|
19
|
+
re-POST of the same id replays the committed result instead of applying the
|
|
20
|
+
load again — so what makes a retry safe is re-sending the same upload, not
|
|
21
|
+
the mode. This client stages once, in `upload_parquet`, outside the retried
|
|
22
|
+
operation, so the invariant holds for every mode.
|
|
23
|
+
|
|
24
|
+
The exclusion cost real availability. The destination serialises writes per
|
|
25
|
+
table and refuses rather than queues, so concurrent writers to one table get
|
|
26
|
+
`409 RESOURCE_LOCKED` — and an append had no budget to wait it out, whatever
|
|
27
|
+
`max_retries` the caller had configured.
|
|
28
|
+
|
|
29
|
+
`HotdataClient.load_managed_table(file=...)` uploads inside the call and so
|
|
30
|
+
does not hold the invariant. It is unwrapped and unaffected.
|
|
31
|
+
|
|
32
|
+
- fix(errors): classify a 409 by its `error.code` rather than by the status alone.
|
|
33
|
+
|
|
34
|
+
`CONFLICT` is now terminal: it means the request cannot succeed as posted, so
|
|
35
|
+
the previous behaviour spent the entire retry budget arriving at the same
|
|
36
|
+
answer. `RESOURCE_LOCKED` stays transient. A 409 with no error envelope — a
|
|
37
|
+
failed query result, say — is classified as before.
|
|
38
|
+
|
|
39
|
+
- fix(retry): honour `Retry-After`, and jitter the backoff.
|
|
40
|
+
|
|
41
|
+
`Retry-After` is taken as a floor on the ramp, capped like the ramp so a bad
|
|
42
|
+
header cannot park an attempt for an hour. Jitter of up to +50% is added on
|
|
43
|
+
top and never subtracted, so a stated `Retry-After` is not undercut. Without
|
|
44
|
+
it, writers that collided on one table retry in lockstep and collide again.
|
|
45
|
+
|
|
46
|
+
This lengthens a 20-attempt budget from 285s to roughly 316-405s.
|
|
47
|
+
|
|
48
|
+
- docs: scope the "a load is not idempotent" claim in the README and in
|
|
49
|
+
`test_retry_policy` to the transport layer, which is where it is still true
|
|
50
|
+
and where those two were always talking about. Left unscoped they read as
|
|
51
|
+
repo-wide and contradict the call-layer retry above.
|
|
52
|
+
|
|
53
|
+
### Added
|
|
54
|
+
|
|
55
|
+
- `HotdataError` carries `status_code`, `code` and `retry_after_seconds`. The
|
|
56
|
+
message is flattened and truncated for readability, so it could not serve as
|
|
57
|
+
a discriminator; these can.
|
|
58
|
+
|
|
59
|
+
## [0.12.1] - 2026-08-18
|
|
60
|
+
|
|
61
|
+
### Fixed
|
|
62
|
+
|
|
63
|
+
- fix(load): submit managed loads as a job and poll, instead of holding one request open
|
|
10
64
|
|
|
11
65
|
## [0.12.0] - 2026-08-11
|
|
12
66
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -38,7 +38,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
38
38
|
|
|
39
39
|
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
40
40
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
41
|
-
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a
|
|
41
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
|
|
42
42
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
43
43
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
44
44
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -10,7 +10,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
10
10
|
|
|
11
11
|
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
12
12
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
13
|
-
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a
|
|
13
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
|
|
14
14
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
15
15
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
16
16
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -25,6 +25,7 @@ from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
|
|
|
25
25
|
from hotdata.models.index_info_response import IndexInfoResponse
|
|
26
26
|
from hotdata.models.job_status_response import JobStatusResponse
|
|
27
27
|
from hotdata.models.load_managed_table_request import LoadManagedTableRequest
|
|
28
|
+
from hotdata.models.load_managed_table_response import LoadManagedTableResponse
|
|
28
29
|
from hotdata.models.query_request import QueryRequest
|
|
29
30
|
from hotdata.models.query_response import QueryResponse
|
|
30
31
|
from hotdata.models.submit_job_response import SubmitJobResponse
|
|
@@ -81,6 +82,37 @@ _RESULT_FAILURE = frozenset({"failed", "cancelled"})
|
|
|
81
82
|
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
82
83
|
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
83
84
|
|
|
85
|
+
# How long a load may finish INLINE before the server hands back a job instead.
|
|
86
|
+
# Small enough that a slow load stops holding a request open, large enough that
|
|
87
|
+
# the overwhelming majority never become jobs at all: dlt's bookkeeping tables
|
|
88
|
+
# (`_dlt_version`, `_dlt_loads`, `_dlt_pipeline_state`) settle in under a second,
|
|
89
|
+
# and paying a submit-then-poll round trip for those would be a regression.
|
|
90
|
+
_LOAD_INLINE_WAIT_MS = 10_000
|
|
91
|
+
|
|
92
|
+
# A load's own polling budget. Deliberately NOT the 300s used for queries and
|
|
93
|
+
# results: a load is the one operation here whose duration scales with the data,
|
|
94
|
+
# and reusing the query budget is what put a five-minute ceiling on it in the
|
|
95
|
+
# first place. Bounded rather than unbounded so a wedged job still surfaces.
|
|
96
|
+
_LOAD_JOB_TIMEOUT_S = 3600.0
|
|
97
|
+
|
|
98
|
+
# How long a poll tolerates CONTINUOUS status-check failure before giving up.
|
|
99
|
+
#
|
|
100
|
+
# Time, not a count: a count means whatever the caller's `interval_s` makes it,
|
|
101
|
+
# and callers differ -- five checks is eight seconds at the load interval and
|
|
102
|
+
# something else at the index one. The thing being survived is a gateway blip,
|
|
103
|
+
# and a rolling restart or a load balancer reconverging routinely serves 502s for
|
|
104
|
+
# longer than a few seconds. Too short and the poll aborts, the caller's retry
|
|
105
|
+
# re-submits the load, and the original job is still holding the table -- the
|
|
106
|
+
# door this tolerance exists to close.
|
|
107
|
+
#
|
|
108
|
+
# Still bounded by the poll's own deadline, so this only decides how a stretch of
|
|
109
|
+
# failures ends, never how long the wait can be.
|
|
110
|
+
_JOB_POLL_ERROR_GRACE_S = 120.0
|
|
111
|
+
|
|
112
|
+
# Failed checks back off rather than hammering at `interval_s`: whatever is
|
|
113
|
+
# serving 502s does not need the extra traffic.
|
|
114
|
+
_JOB_POLL_ERROR_MAX_BACKOFF_S = 15.0
|
|
115
|
+
|
|
84
116
|
|
|
85
117
|
@dataclass(frozen=True)
|
|
86
118
|
class ResultSummary:
|
|
@@ -418,10 +450,25 @@ class HotdataClient:
|
|
|
418
450
|
else:
|
|
419
451
|
assert file is not None
|
|
420
452
|
resolved_upload_id = self.upload_parquet(file)
|
|
453
|
+
# ASKED FOR AS A JOB, not as a held-open request. A load's duration scales
|
|
454
|
+
# with the data, and a single request that must survive minutes has to
|
|
455
|
+
# survive every layer between here and the engine -- CDN, gateway, socket
|
|
456
|
+
# read timeout -- any one of which ends it. When it ends, the server logs
|
|
457
|
+
# the load `abandoned` and DISCARDS work it had already done, while the
|
|
458
|
+
# table's write lock is still held against the retry that follows; the
|
|
459
|
+
# retry then collides with it (409 RESOURCE_LOCKED) and the pair can spin
|
|
460
|
+
# indefinitely without the load ever completing. Observed in production on
|
|
461
|
+
# a table whose load runs past five minutes.
|
|
462
|
+
#
|
|
463
|
+
# `async_after_ms` keeps the common case unchanged: the server answers 200
|
|
464
|
+
# with the result if it finishes inside the window, and only falls back to
|
|
465
|
+
# a job when it does not. So nothing pays for polling that did not need it.
|
|
421
466
|
request = LoadManagedTableRequest(
|
|
422
467
|
mode=mode,
|
|
423
468
|
upload_id=resolved_upload_id,
|
|
424
469
|
key=key,
|
|
470
|
+
var_async=True,
|
|
471
|
+
async_after_ms=_LOAD_INLINE_WAIT_MS,
|
|
425
472
|
)
|
|
426
473
|
try:
|
|
427
474
|
loaded = self.connections().load_managed_table(
|
|
@@ -432,12 +479,21 @@ class HotdataClient:
|
|
|
432
479
|
)
|
|
433
480
|
except ApiException as e:
|
|
434
481
|
raise RuntimeError(api_error_message(e)) from e
|
|
482
|
+
# Only the job branch is type-checked. The index path can also assert its
|
|
483
|
+
# inline type because it names it positively first; here the inline shape is
|
|
484
|
+
# read duck-typed, which callers and tests already rely on, so asserting it
|
|
485
|
+
# would narrow an interface this change has no business narrowing.
|
|
486
|
+
job_id: str | None = None
|
|
487
|
+
if isinstance(loaded, SubmitJobResponse):
|
|
488
|
+
job_id = loaded.id
|
|
489
|
+
loaded = self._load_response_from_job(job_id)
|
|
435
490
|
return LoadManagedTableResult(
|
|
436
491
|
connection_id=loaded.connection_id,
|
|
437
492
|
schema_name=loaded.schema_name,
|
|
438
493
|
table_name=loaded.table_name,
|
|
439
494
|
row_count=loaded.row_count,
|
|
440
495
|
full_name=f"{db.id}.{loaded.schema_name}.{loaded.table_name}",
|
|
496
|
+
job_id=job_id,
|
|
441
497
|
)
|
|
442
498
|
|
|
443
499
|
def add_managed_table(
|
|
@@ -880,11 +936,40 @@ class HotdataClient:
|
|
|
880
936
|
jobs = self._jobs_api()
|
|
881
937
|
deadline = time.monotonic() + timeout_s
|
|
882
938
|
last: JobStatusResponse | None = None
|
|
939
|
+
# When the current run of failures began, or None while checks succeed.
|
|
940
|
+
failing_since: float | None = None
|
|
941
|
+
error_backoff = interval_s
|
|
883
942
|
while time.monotonic() < deadline:
|
|
884
943
|
try:
|
|
885
944
|
last = jobs.get_job(job_id)
|
|
945
|
+
failing_since = None
|
|
946
|
+
error_backoff = interval_s
|
|
886
947
|
except ApiException as e:
|
|
887
|
-
|
|
948
|
+
# A failed STATUS CHECK is not a failed job. Aborting here throws
|
|
949
|
+
# away work that is still running, and the caller's retry then
|
|
950
|
+
# re-submits it while the original still holds its resources -- so
|
|
951
|
+
# one blip becomes a collision with the job it just abandoned. A
|
|
952
|
+
# load polls for up to `_LOAD_JOB_TIMEOUT_S`, so the longer the
|
|
953
|
+
# wait the more chances to hit it, which is exactly backwards.
|
|
954
|
+
#
|
|
955
|
+
# CONSECUTIVE failures are the signal: an isolated 502 is noise, a
|
|
956
|
+
# run of them means the API is gone and there is nothing to wait
|
|
957
|
+
# for. The poll's own deadline bounds the total wait regardless.
|
|
958
|
+
now = time.monotonic()
|
|
959
|
+
if failing_since is None:
|
|
960
|
+
failing_since = now
|
|
961
|
+
elif now - failing_since >= _JOB_POLL_ERROR_GRACE_S:
|
|
962
|
+
# Name the job. This is one of the two paths where the caller
|
|
963
|
+
# cannot tell whether the load landed, so the id is the only
|
|
964
|
+
# thing that makes the question answerable -- and it is exactly
|
|
965
|
+
# what a message like "502: Bad Gateway" leaves out.
|
|
966
|
+
raise RuntimeError(
|
|
967
|
+
f"Job {job_id} status checks failed for "
|
|
968
|
+
f"{_JOB_POLL_ERROR_GRACE_S:.0f}s: {api_error_message(e)}"
|
|
969
|
+
) from e
|
|
970
|
+
time.sleep(error_backoff)
|
|
971
|
+
error_backoff = min(error_backoff * 2, _JOB_POLL_ERROR_MAX_BACKOFF_S)
|
|
972
|
+
continue
|
|
888
973
|
if last.status in _JOB_TERMINAL:
|
|
889
974
|
return last
|
|
890
975
|
time.sleep(interval_s)
|
|
@@ -893,6 +978,41 @@ class HotdataClient:
|
|
|
893
978
|
f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
|
|
894
979
|
)
|
|
895
980
|
|
|
981
|
+
def _load_response_from_job(self, job_id: str) -> LoadManagedTableResponse:
|
|
982
|
+
"""The result of a load the server chose to run as a job.
|
|
983
|
+
|
|
984
|
+
Polling replaces waiting on the request, so the outcome is read from
|
|
985
|
+
durable state rather than from a connection that has to stay alive. That
|
|
986
|
+
also gives a caller a handle: the job id is returned on
|
|
987
|
+
`LoadManagedTableResult`, so "did it land?" is answerable after a lost
|
|
988
|
+
response. That answer is a convenience rather than a precondition for
|
|
989
|
+
retrying: re-sending the same upload_id replays the server's receipt
|
|
990
|
+
instead of applying the load a second time, which is what makes a retry
|
|
991
|
+
safe in every mode. It stops being safe for a caller that re-stages the
|
|
992
|
+
upload, because a fresh upload id has no receipt to replay.
|
|
993
|
+
|
|
994
|
+
`partially_succeeded` is terminal and carries a message, so it is raised
|
|
995
|
+
rather than returned -- a caller asked for a table's contents to be
|
|
996
|
+
replaced or appended to, and "some of it" is not an answer it can use.
|
|
997
|
+
"""
|
|
998
|
+
final = self._poll_job(job_id, timeout_s=_LOAD_JOB_TIMEOUT_S)
|
|
999
|
+
status = enum_value(final.status)
|
|
1000
|
+
if status != "succeeded":
|
|
1001
|
+
# The id goes in whether or not the server's message mentions it: the
|
|
1002
|
+
# caller is being told the load did not succeed, and "which load" is
|
|
1003
|
+
# the next thing it needs.
|
|
1004
|
+
detail = final.error_message or f"finished {status}"
|
|
1005
|
+
raise RuntimeError(f"load job {job_id} {status}: {detail}")
|
|
1006
|
+
# `result` is a oneOf wrapper today; tolerate the model arriving directly,
|
|
1007
|
+
# the same way the index path does rather than disagreeing with it.
|
|
1008
|
+
payload = getattr(final.result, "actual_instance", final.result)
|
|
1009
|
+
if not isinstance(payload, LoadManagedTableResponse):
|
|
1010
|
+
raise RuntimeError(
|
|
1011
|
+
f"load job {job_id} succeeded without a load result "
|
|
1012
|
+
f"(got {type(payload).__name__})"
|
|
1013
|
+
)
|
|
1014
|
+
return payload
|
|
1015
|
+
|
|
896
1016
|
def _wait_result_ready(
|
|
897
1017
|
self,
|
|
898
1018
|
result_id: str,
|
|
@@ -83,6 +83,11 @@ class LoadManagedTableResult:
|
|
|
83
83
|
table_name: str
|
|
84
84
|
row_count: int
|
|
85
85
|
full_name: str
|
|
86
|
+
# Set when the server ran the load as a background job. Carried for the same
|
|
87
|
+
# reason CreateIndexResult carries it: it is the only handle a caller has to
|
|
88
|
+
# ask "did that land?" after a lost response, and without it the question is
|
|
89
|
+
# unanswerable. `None` when the load finished inline.
|
|
90
|
+
job_id: str | None = None
|
|
86
91
|
|
|
87
92
|
def to_dict(self) -> dict[str, Any]:
|
|
88
93
|
return asdict(self)
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
|
|
6
|
+
from hotdata.rest import ApiException
|
|
7
|
+
|
|
8
|
+
# The API explains a 409 with a machine-readable code, and the two it sends
|
|
9
|
+
# mean opposite things to a retry policy. RESOURCE_LOCKED is a refusal taken
|
|
10
|
+
# before any work: the insert that would have created the unit of work lost a
|
|
11
|
+
# unique-constraint race, so nothing was claimed and nothing was written.
|
|
12
|
+
# CONFLICT is the opposite — the request cannot succeed as posted, so retrying
|
|
13
|
+
# spends the whole budget arriving at the same answer.
|
|
14
|
+
_TERMINAL_CONFLICT_CODE = "CONFLICT"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class HotdataError(RuntimeError):
|
|
18
|
+
"""An API failure, carrying what a retry policy needs to decide.
|
|
19
|
+
|
|
20
|
+
The message cannot be the discriminator: it is flattened and truncated for
|
|
21
|
+
readability, so keying on it means substring-matching prose. ``status_code``
|
|
22
|
+
and ``code`` are the machine-readable form of the same answer, and
|
|
23
|
+
``retry_after_seconds`` is the server's own estimate of how long the
|
|
24
|
+
condition it just refused will last.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
message: str,
|
|
30
|
+
*,
|
|
31
|
+
status_code: int | None = None,
|
|
32
|
+
code: str | None = None,
|
|
33
|
+
retry_after_seconds: float | None = None,
|
|
34
|
+
) -> None:
|
|
35
|
+
super().__init__(message)
|
|
36
|
+
self.status_code = status_code
|
|
37
|
+
self.code = code
|
|
38
|
+
self.retry_after_seconds = retry_after_seconds
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class HotdataTransientError(HotdataError):
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class HotdataTerminalError(HotdataError):
|
|
46
|
+
pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _error_code(body: object) -> str | None:
|
|
50
|
+
"""The ``error.code`` an API error envelope carries, if this body is one.
|
|
51
|
+
|
|
52
|
+
Not every 409 comes from an endpoint that speaks the envelope — a failed
|
|
53
|
+
query result is reported as one and carries a result document instead — so
|
|
54
|
+
a missing code is ordinary, and callers fall back to the status.
|
|
55
|
+
"""
|
|
56
|
+
if not isinstance(body, (str, bytes, bytearray)):
|
|
57
|
+
return None
|
|
58
|
+
try:
|
|
59
|
+
parsed: object = json.loads(body)
|
|
60
|
+
except ValueError:
|
|
61
|
+
return None
|
|
62
|
+
if not isinstance(parsed, Mapping):
|
|
63
|
+
return None
|
|
64
|
+
error: object = parsed.get("error")
|
|
65
|
+
if not isinstance(error, Mapping):
|
|
66
|
+
return None
|
|
67
|
+
code: object = error.get("code")
|
|
68
|
+
return code if isinstance(code, str) else None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _retry_after_seconds(headers: object) -> float | None:
|
|
72
|
+
"""``Retry-After`` as a number of seconds, when the response states one.
|
|
73
|
+
|
|
74
|
+
Only the delta-seconds form is read. That is what the API sends, and the
|
|
75
|
+
HTTP-date form would need a comparison against a server clock we do not
|
|
76
|
+
have to be worth anything.
|
|
77
|
+
"""
|
|
78
|
+
if not isinstance(headers, Mapping):
|
|
79
|
+
return None
|
|
80
|
+
raw: object = headers.get("Retry-After")
|
|
81
|
+
if raw is None:
|
|
82
|
+
# The SDK hands us urllib3's case-insensitive mapping and the API sends
|
|
83
|
+
# the header lower-cased, so the direct hit is what normally answers.
|
|
84
|
+
# Fall back for any plain dict that reaches us instead — a missed
|
|
85
|
+
# header is silent, and silence here reads as "the server asked for
|
|
86
|
+
# nothing".
|
|
87
|
+
raw = next((v for k, v in headers.items() if str(k).lower() == "retry-after"), None)
|
|
88
|
+
if raw is None:
|
|
89
|
+
return None
|
|
90
|
+
try:
|
|
91
|
+
seconds = float(str(raw).strip())
|
|
92
|
+
except ValueError:
|
|
93
|
+
return None
|
|
94
|
+
return seconds if seconds >= 0 else None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _error_class(status_code: int, code: str | None) -> type[HotdataError]:
|
|
98
|
+
if status_code == 409 and code == _TERMINAL_CONFLICT_CODE:
|
|
99
|
+
# The request cannot succeed as posted — an upload already consumed
|
|
100
|
+
# with nothing to replay, a receipt naming a different target, an
|
|
101
|
+
# incompatible column type. Every retry reaches the same 409.
|
|
102
|
+
return HotdataTerminalError
|
|
103
|
+
if status_code in (408, 409, 425, 429):
|
|
104
|
+
return HotdataTransientError
|
|
105
|
+
if status_code == 501:
|
|
106
|
+
# Not Implemented is a permanent capability gap (e.g. the storage
|
|
107
|
+
# backend cannot issue presigned URLs) — retrying cannot succeed.
|
|
108
|
+
return HotdataTerminalError
|
|
109
|
+
if 500 <= status_code <= 599:
|
|
110
|
+
return HotdataTransientError
|
|
111
|
+
return HotdataTerminalError
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def classify_sdk_error(error: Exception) -> HotdataError:
|
|
115
|
+
if isinstance(error, TimeoutError):
|
|
116
|
+
return HotdataTransientError(str(error))
|
|
117
|
+
if isinstance(error, ConnectionError):
|
|
118
|
+
return HotdataTransientError(str(error))
|
|
119
|
+
if isinstance(error, ApiException):
|
|
120
|
+
status_code = int(error.status or 0)
|
|
121
|
+
message = f"{status_code}: {error.reason or 'unknown error'}"
|
|
122
|
+
# The response body is where the API explains itself (e.g. which
|
|
123
|
+
# header is missing) — without it "400: Bad Request" is undebuggable.
|
|
124
|
+
body: object = getattr(error, "body", None)
|
|
125
|
+
if body:
|
|
126
|
+
message = f"{message} — {' '.join(str(body).split())[:500]}"
|
|
127
|
+
code = _error_code(body)
|
|
128
|
+
return _error_class(status_code, code)(
|
|
129
|
+
message,
|
|
130
|
+
status_code=status_code,
|
|
131
|
+
code=code,
|
|
132
|
+
retry_after_seconds=_retry_after_seconds(getattr(error, "headers", None)),
|
|
133
|
+
)
|
|
134
|
+
return HotdataTerminalError(str(error))
|
|
@@ -7,6 +7,7 @@ in one place rather than being duplicated per adapter.
|
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
9
9
|
|
|
10
|
+
import random
|
|
10
11
|
import time
|
|
11
12
|
from collections.abc import Callable
|
|
12
13
|
from typing import Any, Protocol, TypeVar
|
|
@@ -53,6 +54,10 @@ class ManagedDatabaseClient:
|
|
|
53
54
|
_QUERY_TIMEOUT_SECONDS = 300.0
|
|
54
55
|
_POLL_INTERVAL_SECONDS = 0.4
|
|
55
56
|
_MAX_BACKOFF_SECONDS = 30.0
|
|
57
|
+
# Spread as a fraction of the wait, added on top of it. Half an interval is
|
|
58
|
+
# enough to decorrelate writers that started together without materially
|
|
59
|
+
# changing how long the budget lasts.
|
|
60
|
+
_RETRY_JITTER_FRACTION = 0.5
|
|
56
61
|
|
|
57
62
|
def __init__(
|
|
58
63
|
self,
|
|
@@ -207,9 +212,16 @@ class ManagedDatabaseClient:
|
|
|
207
212
|
mode: ManagedLoadMode = "replace",
|
|
208
213
|
key: list[str] | None = None,
|
|
209
214
|
) -> LoadManagedTableResult:
|
|
210
|
-
#
|
|
211
|
-
#
|
|
212
|
-
#
|
|
215
|
+
# Retryable in every mode, append included. A retry re-sends the SAME
|
|
216
|
+
# upload_id, and the server keys a receipt on it: a replay returns the
|
|
217
|
+
# committed result rather than applying the load a second time. So the
|
|
218
|
+
# invariant that makes this safe is the upload id, not the mode — a
|
|
219
|
+
# caller that re-stages the upload between attempts mints a new id,
|
|
220
|
+
# loses the receipt, and a retried append would then duplicate rows.
|
|
221
|
+
# This client stages once, in upload_parquet, outside the operation
|
|
222
|
+
# retried here. `HotdataClient.load_managed_table(file=...)` uploads
|
|
223
|
+
# inside the call and so does not hold the invariant; it is unwrapped,
|
|
224
|
+
# and retrying an append through it is the caller's to justify.
|
|
213
225
|
#
|
|
214
226
|
# `key` is the merge key for delete/update/upsert loads: when set it is
|
|
215
227
|
# matched per-load instead of a key declared at table creation. Omit it
|
|
@@ -222,20 +234,40 @@ class ManagedDatabaseClient:
|
|
|
222
234
|
upload_id=upload_id,
|
|
223
235
|
mode=mode,
|
|
224
236
|
key=key,
|
|
225
|
-
)
|
|
226
|
-
retryable=(mode != "append"),
|
|
237
|
+
)
|
|
227
238
|
)
|
|
228
239
|
|
|
229
|
-
def _request_with_retry(self, operation: Callable[[], T]
|
|
230
|
-
max_attempts = self._max_retries
|
|
240
|
+
def _request_with_retry(self, operation: Callable[[], T]) -> T:
|
|
241
|
+
max_attempts = self._max_retries
|
|
231
242
|
for attempt in range(1, max_attempts + 1):
|
|
232
243
|
try:
|
|
233
244
|
return operation()
|
|
234
245
|
except Exception as error:
|
|
235
246
|
mapped_error = classify_sdk_error(error.__cause__ or error)
|
|
236
247
|
if isinstance(mapped_error, HotdataTransientError) and attempt < max_attempts:
|
|
237
|
-
|
|
238
|
-
time.sleep(backoff)
|
|
248
|
+
time.sleep(self._retry_delay(attempt, mapped_error.retry_after_seconds))
|
|
239
249
|
continue
|
|
240
250
|
raise mapped_error from error
|
|
241
251
|
raise RuntimeError("No retry attempts configured")
|
|
252
|
+
|
|
253
|
+
def _retry_delay(self, attempt: int, retry_after_seconds: float | None) -> float:
|
|
254
|
+
"""A linear ramp, floored by the server's Retry-After and spread by jitter.
|
|
255
|
+
|
|
256
|
+
Retry-After is a floor rather than a replacement: it says how long the
|
|
257
|
+
condition just refused typically lasts, while the ramp is what gives up
|
|
258
|
+
eventually, and taking the larger of the two honours both. It is capped
|
|
259
|
+
like the ramp so a hostile or mistaken header cannot park an attempt for
|
|
260
|
+
an hour.
|
|
261
|
+
|
|
262
|
+
Jitter is added on top and never subtracted, so a stated Retry-After is
|
|
263
|
+
not undercut. It matters because the callers that collide are the ones
|
|
264
|
+
that started together: writers refused by one table's lock would retry
|
|
265
|
+
in lockstep on an identical ramp and re-collide every time.
|
|
266
|
+
_MAX_BACKOFF_SECONDS caps the ramp, deliberately not the jitter above
|
|
267
|
+
it — clamping the total would flatten every late attempt onto the same
|
|
268
|
+
value and re-correlate exactly the waits that most need spreading.
|
|
269
|
+
"""
|
|
270
|
+
base = min(self._retry_backoff_seconds * attempt, self._MAX_BACKOFF_SECONDS)
|
|
271
|
+
if retry_after_seconds is not None:
|
|
272
|
+
base = max(base, min(retry_after_seconds, self._MAX_BACKOFF_SECONDS))
|
|
273
|
+
return base * (1.0 + random.random() * self._RETRY_JITTER_FRACTION)
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "hotdata-framework"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.13.0"
|
|
8
8
|
description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|