hotdata-framework 0.12.0__tar.gz → 0.13.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/publish.yml +15 -0
  2. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/release.yml +18 -0
  3. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/CHANGELOG.md +54 -0
  4. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/PKG-INFO +2 -2
  5. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/README.md +1 -1
  6. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/client.py +121 -1
  7. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/databases.py +5 -0
  8. hotdata_framework-0.13.0/hotdata_framework/errors.py +134 -0
  9. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/managed_client.py +41 -9
  10. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/pyproject.toml +1 -1
  11. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_client.py +330 -1
  12. hotdata_framework-0.13.0/tests/test_errors.py +125 -0
  13. hotdata_framework-0.13.0/tests/test_managed_client.py +430 -0
  14. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_retry_policy.py +10 -4
  15. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/uv.lock +1 -1
  16. hotdata_framework-0.12.0/hotdata_framework/errors.py +0 -40
  17. hotdata_framework-0.12.0/tests/test_errors.py +0 -48
  18. hotdata_framework-0.12.0/tests/test_managed_client.py +0 -248
  19. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/CODEOWNERS +0 -0
  20. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/dependabot.yml +0 -0
  21. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/check-release.yml +0 -0
  22. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/ci.yml +0 -0
  23. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.github/workflows/dependabot-automerge.yml +0 -0
  24. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/.gitignore +0 -0
  25. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/CONTRACT.md +0 -0
  26. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/RELEASING.md +0 -0
  27. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/examples/basic_usage.py +0 -0
  28. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/__init__.py +0 -0
  29. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/env.py +0 -0
  30. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/health.py +0 -0
  31. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/py.typed +0 -0
  32. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/hotdata_framework/result.py +0 -0
  33. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/check-release.py +0 -0
  34. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/extract-changelog.py +0 -0
  35. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/publish-workflow.sh +0 -0
  36. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/release.sh +0 -0
  37. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/scripts/update_changelog.py +0 -0
  38. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_contract.py +0 -0
  39. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_databases.py +0 -0
  40. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_health.py +0 -0
  41. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_indexes.py +0 -0
  42. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_request_timeout.py +0 -0
  43. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_result.py +0 -0
  44. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_update_changelog.py +0 -0
  45. {hotdata_framework-0.12.0 → hotdata_framework-0.13.0}/tests/test_version.py +0 -0
@@ -31,6 +31,21 @@ jobs:
31
31
  # The tag being released, whether it arrived by push or by dispatch.
32
32
  TAG: ${{ inputs.tag || github.ref_name }}
33
33
  steps:
34
+ # Before checkout, because checkout resolves the input as an arbitrary ref:
35
+ # a branch or SHA is fetched first and only rejected later by the version
36
+ # match below, which is also looser (`^v[0-9]`). Strict here so the dispatch
37
+ # contract matches release.yml — release.sh only ever produces X.Y.Z.
38
+ - name: Validate release tag format
39
+ if: github.event_name == 'workflow_dispatch'
40
+ env:
41
+ INPUT_TAG: ${{ inputs.tag }}
42
+ run: |
43
+ set -euo pipefail
44
+ if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
45
+ echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
46
+ exit 1
47
+ fi
48
+
34
49
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
35
50
  with:
36
51
  ref: ${{ inputs.tag || github.ref_name }}
@@ -25,6 +25,24 @@ jobs:
25
25
  # The tag being released, whether it arrived by push or by dispatch.
26
26
  TAG: ${{ inputs.tag || github.ref_name }}
27
27
  steps:
28
+ # Before checkout, because checkout resolves the input as an arbitrary ref
29
+ # — and this workflow needs the guard more than publish.yml does. It holds
30
+ # `contents: write`, and action-gh-release CREATES a tag when tag_name does
31
+ # not resolve to one, so an unvalidated `tag: main` would check out cleanly
32
+ # and leave refs/tags/main plus a release named for it. The push path is
33
+ # constrained by the v[0-9]* filter; the dispatch path was not constrained
34
+ # at all.
35
+ - name: Validate release tag format
36
+ if: github.event_name == 'workflow_dispatch'
37
+ env:
38
+ INPUT_TAG: ${{ inputs.tag }}
39
+ run: |
40
+ set -euo pipefail
41
+ if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
42
+ echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
43
+ exit 1
44
+ fi
45
+
28
46
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
29
47
  with:
30
48
  ref: ${{ inputs.tag || github.ref_name }}
@@ -7,6 +7,60 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.13.0] - 2026-08-27
11
+
12
+ ### Fixed
13
+
14
+ - fix(load): retry an `append` load instead of running it at most once.
15
+
16
+ `append` was excluded from retries on the grounds that it is not idempotent:
17
+ if the server commits but the response is lost, a retry would duplicate rows.
18
+ That is not how the server behaves. It keys a receipt on `upload_id`, and a
19
+ re-POST of the same id replays the committed result instead of applying the
20
+ load again — so what makes a retry safe is re-sending the same upload, not
21
+ the mode. This client stages once, in `upload_parquet`, outside the retried
22
+ operation, so the invariant holds for every mode.
23
+
24
+ The exclusion cost real availability. The destination serialises writes per
25
+ table and refuses rather than queues, so concurrent writers to one table get
26
+ `409 RESOURCE_LOCKED` — and an append had no budget to wait it out, whatever
27
+ `max_retries` the caller had configured.
28
+
29
+ `HotdataClient.load_managed_table(file=...)` uploads inside the call and so
30
+ does not hold the invariant. It is unwrapped and unaffected.
31
+
32
+ - fix(errors): classify a 409 by its `error.code` rather than by the status alone.
33
+
34
+ `CONFLICT` is now terminal: it means the request cannot succeed as posted, so
35
+ the previous behaviour spent the entire retry budget arriving at the same
36
+ answer. `RESOURCE_LOCKED` stays transient. A 409 with no error envelope — a
37
+ failed query result, say — is classified as before.
38
+
39
+ - fix(retry): honour `Retry-After`, and jitter the backoff.
40
+
41
+ `Retry-After` is taken as a floor on the ramp, capped like the ramp so a bad
42
+ header cannot park an attempt for an hour. Jitter of up to +50% is added on
43
+ top and never subtracted, so a stated `Retry-After` is not undercut. Without
44
+ it, writers that collided on one table retry in lockstep and collide again.
45
+
46
+ This lengthens a 20-attempt budget from 285s to roughly 316-405s.
47
+
48
+ - docs: scope the "a load is not idempotent" claim in the README and in
49
+ `test_retry_policy` to the transport layer, which is where it is still true
50
+ and where those two were always talking about. Left unscoped they read as
51
+ repo-wide and contradict the call-layer retry above.
52
+
53
+ ### Added
54
+
55
+ - `HotdataError` carries `status_code`, `code` and `retry_after_seconds`. The
56
+ message is flattened and truncated for readability, so it could not serve as
57
+ a discriminator; these can.
58
+
59
+ ## [0.12.1] - 2026-08-18
60
+
61
+ ### Fixed
62
+
63
+ - fix(load): submit managed loads as a job and poll, instead of holding one request open
10
64
 
11
65
  ## [0.12.0] - 2026-08-11
12
66
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.12.0
3
+ Version: 0.13.0
4
4
  Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -38,7 +38,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
38
38
 
39
39
  - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
40
40
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
41
- - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
41
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
42
42
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
43
43
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
44
44
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -10,7 +10,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
10
10
 
11
11
  - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
12
12
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
13
- - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
13
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a request is never blindly replayed on a response status. That is a claim about the transport, which cannot know what it would be replaying. `ManagedDatabaseClient` retries at the call layer, which can: a managed load is safe to re-send because it carries the same `upload_id` and the API replays its receipt for that id rather than applying the load twice.
14
14
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
15
15
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
16
16
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -25,6 +25,7 @@ from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
25
25
  from hotdata.models.index_info_response import IndexInfoResponse
26
26
  from hotdata.models.job_status_response import JobStatusResponse
27
27
  from hotdata.models.load_managed_table_request import LoadManagedTableRequest
28
+ from hotdata.models.load_managed_table_response import LoadManagedTableResponse
28
29
  from hotdata.models.query_request import QueryRequest
29
30
  from hotdata.models.query_response import QueryResponse
30
31
  from hotdata.models.submit_job_response import SubmitJobResponse
@@ -81,6 +82,37 @@ _RESULT_FAILURE = frozenset({"failed", "cancelled"})
81
82
  # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
82
83
  _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
83
84
 
85
+ # How long a load may finish INLINE before the server hands back a job instead.
86
+ # Small enough that a slow load stops holding a request open, large enough that
87
+ # the overwhelming majority never become jobs at all: dlt's bookkeeping tables
88
+ # (`_dlt_version`, `_dlt_loads`, `_dlt_pipeline_state`) settle in under a second,
89
+ # and paying a submit-then-poll round trip for those would be a regression.
90
+ _LOAD_INLINE_WAIT_MS = 10_000
91
+
92
+ # A load's own polling budget. Deliberately NOT the 300s used for queries and
93
+ # results: a load is the one operation here whose duration scales with the data,
94
+ # and reusing the query budget is what put a five-minute ceiling on it in the
95
+ # first place. Bounded rather than unbounded so a wedged job still surfaces.
96
+ _LOAD_JOB_TIMEOUT_S = 3600.0
97
+
98
+ # How long a poll tolerates CONTINUOUS status-check failure before giving up.
99
+ #
100
+ # Time, not a count: a count means whatever the caller's `interval_s` makes it,
101
+ # and callers differ -- five checks is eight seconds at the load interval and
102
+ # something else at the index one. The thing being survived is a gateway blip,
103
+ # and a rolling restart or a load balancer reconverging routinely serves 502s for
104
+ # longer than a few seconds. Too short and the poll aborts, the caller's retry
105
+ # re-submits the load, and the original job is still holding the table -- the
106
+ # door this tolerance exists to close.
107
+ #
108
+ # Still bounded by the poll's own deadline, so this only decides how a stretch of
109
+ # failures ends, never how long the wait can be.
110
+ _JOB_POLL_ERROR_GRACE_S = 120.0
111
+
112
+ # Failed checks back off rather than hammering at `interval_s`: whatever is
113
+ # serving 502s does not need the extra traffic.
114
+ _JOB_POLL_ERROR_MAX_BACKOFF_S = 15.0
115
+
84
116
 
85
117
  @dataclass(frozen=True)
86
118
  class ResultSummary:
@@ -418,10 +450,25 @@ class HotdataClient:
418
450
  else:
419
451
  assert file is not None
420
452
  resolved_upload_id = self.upload_parquet(file)
453
+ # ASKED FOR AS A JOB, not as a held-open request. A load's duration scales
454
+ # with the data, and a single request that must survive minutes has to
455
+ # survive every layer between here and the engine -- CDN, gateway, socket
456
+ # read timeout -- any one of which ends it. When it ends, the server logs
457
+ # the load `abandoned` and DISCARDS work it had already done, while the
458
+ # table's write lock is still held against the retry that follows; the
459
+ # retry then collides with it (409 RESOURCE_LOCKED) and the pair can spin
460
+ # indefinitely without the load ever completing. Observed in production on
461
+ # a table whose load runs past five minutes.
462
+ #
463
+ # `async_after_ms` keeps the common case unchanged: the server answers 200
464
+ # with the result if it finishes inside the window, and only falls back to
465
+ # a job when it does not. So nothing pays for polling that did not need it.
421
466
  request = LoadManagedTableRequest(
422
467
  mode=mode,
423
468
  upload_id=resolved_upload_id,
424
469
  key=key,
470
+ var_async=True,
471
+ async_after_ms=_LOAD_INLINE_WAIT_MS,
425
472
  )
426
473
  try:
427
474
  loaded = self.connections().load_managed_table(
@@ -432,12 +479,21 @@ class HotdataClient:
432
479
  )
433
480
  except ApiException as e:
434
481
  raise RuntimeError(api_error_message(e)) from e
482
+ # Only the job branch is type-checked. The index path can also assert its
483
+ # inline type because it names it positively first; here the inline shape is
484
+ # read duck-typed, which callers and tests already rely on, so asserting it
485
+ # would narrow an interface this change has no business narrowing.
486
+ job_id: str | None = None
487
+ if isinstance(loaded, SubmitJobResponse):
488
+ job_id = loaded.id
489
+ loaded = self._load_response_from_job(job_id)
435
490
  return LoadManagedTableResult(
436
491
  connection_id=loaded.connection_id,
437
492
  schema_name=loaded.schema_name,
438
493
  table_name=loaded.table_name,
439
494
  row_count=loaded.row_count,
440
495
  full_name=f"{db.id}.{loaded.schema_name}.{loaded.table_name}",
496
+ job_id=job_id,
441
497
  )
442
498
 
443
499
  def add_managed_table(
@@ -880,11 +936,40 @@ class HotdataClient:
880
936
  jobs = self._jobs_api()
881
937
  deadline = time.monotonic() + timeout_s
882
938
  last: JobStatusResponse | None = None
939
+ # When the current run of failures began, or None while checks succeed.
940
+ failing_since: float | None = None
941
+ error_backoff = interval_s
883
942
  while time.monotonic() < deadline:
884
943
  try:
885
944
  last = jobs.get_job(job_id)
945
+ failing_since = None
946
+ error_backoff = interval_s
886
947
  except ApiException as e:
887
- raise RuntimeError(api_error_message(e)) from e
948
+ # A failed STATUS CHECK is not a failed job. Aborting here throws
949
+ # away work that is still running, and the caller's retry then
950
+ # re-submits it while the original still holds its resources -- so
951
+ # one blip becomes a collision with the job it just abandoned. A
952
+ # load polls for up to `_LOAD_JOB_TIMEOUT_S`, so the longer the
953
+ # wait the more chances to hit it, which is exactly backwards.
954
+ #
955
+ # CONSECUTIVE failures are the signal: an isolated 502 is noise, a
956
+ # run of them means the API is gone and there is nothing to wait
957
+ # for. The poll's own deadline bounds the total wait regardless.
958
+ now = time.monotonic()
959
+ if failing_since is None:
960
+ failing_since = now
961
+ elif now - failing_since >= _JOB_POLL_ERROR_GRACE_S:
962
+ # Name the job. This is one of the two paths where the caller
963
+ # cannot tell whether the load landed, so the id is the only
964
+ # thing that makes the question answerable -- and it is exactly
965
+ # what a message like "502: Bad Gateway" leaves out.
966
+ raise RuntimeError(
967
+ f"Job {job_id} status checks failed for "
968
+ f"{_JOB_POLL_ERROR_GRACE_S:.0f}s: {api_error_message(e)}"
969
+ ) from e
970
+ time.sleep(error_backoff)
971
+ error_backoff = min(error_backoff * 2, _JOB_POLL_ERROR_MAX_BACKOFF_S)
972
+ continue
888
973
  if last.status in _JOB_TERMINAL:
889
974
  return last
890
975
  time.sleep(interval_s)
@@ -893,6 +978,41 @@ class HotdataClient:
893
978
  f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
894
979
  )
895
980
 
981
+ def _load_response_from_job(self, job_id: str) -> LoadManagedTableResponse:
982
+ """The result of a load the server chose to run as a job.
983
+
984
+ Polling replaces waiting on the request, so the outcome is read from
985
+ durable state rather than from a connection that has to stay alive. That
986
+ also gives a caller a handle: the job id is returned on
987
+ `LoadManagedTableResult`, so "did it land?" is answerable after a lost
988
+ response. That answer is a convenience rather than a precondition for
989
+ retrying: re-sending the same upload_id replays the server's receipt
990
+ instead of applying the load a second time, which is what makes a retry
991
+ safe in every mode. It stops being safe for a caller that re-stages the
992
+ upload, because a fresh upload id has no receipt to replay.
993
+
994
+ `partially_succeeded` is terminal and carries a message, so it is raised
995
+ rather than returned -- a caller asked for a table's contents to be
996
+ replaced or appended to, and "some of it" is not an answer it can use.
997
+ """
998
+ final = self._poll_job(job_id, timeout_s=_LOAD_JOB_TIMEOUT_S)
999
+ status = enum_value(final.status)
1000
+ if status != "succeeded":
1001
+ # The id goes in whether or not the server's message mentions it: the
1002
+ # caller is being told the load did not succeed, and "which load" is
1003
+ # the next thing it needs.
1004
+ detail = final.error_message or f"finished {status}"
1005
+ raise RuntimeError(f"load job {job_id} {status}: {detail}")
1006
+ # `result` is a oneOf wrapper today; tolerate the model arriving directly,
1007
+ # the same way the index path does rather than disagreeing with it.
1008
+ payload = getattr(final.result, "actual_instance", final.result)
1009
+ if not isinstance(payload, LoadManagedTableResponse):
1010
+ raise RuntimeError(
1011
+ f"load job {job_id} succeeded without a load result "
1012
+ f"(got {type(payload).__name__})"
1013
+ )
1014
+ return payload
1015
+
896
1016
  def _wait_result_ready(
897
1017
  self,
898
1018
  result_id: str,
@@ -83,6 +83,11 @@ class LoadManagedTableResult:
83
83
  table_name: str
84
84
  row_count: int
85
85
  full_name: str
86
+ # Set when the server ran the load as a background job. Carried for the same
87
+ # reason CreateIndexResult carries it: it is the only handle a caller has to
88
+ # ask "did that land?" after a lost response, and without it the question is
89
+ # unanswerable. `None` when the load finished inline.
90
+ job_id: str | None = None
86
91
 
87
92
  def to_dict(self) -> dict[str, Any]:
88
93
  return asdict(self)
@@ -0,0 +1,134 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections.abc import Mapping
5
+
6
+ from hotdata.rest import ApiException
7
+
8
+ # The API explains a 409 with a machine-readable code, and the two it sends
9
+ # mean opposite things to a retry policy. RESOURCE_LOCKED is a refusal taken
10
+ # before any work: the insert that would have created the unit of work lost a
11
+ # unique-constraint race, so nothing was claimed and nothing was written.
12
+ # CONFLICT is the opposite — the request cannot succeed as posted, so retrying
13
+ # spends the whole budget arriving at the same answer.
14
+ _TERMINAL_CONFLICT_CODE = "CONFLICT"
15
+
16
+
17
+ class HotdataError(RuntimeError):
18
+ """An API failure, carrying what a retry policy needs to decide.
19
+
20
+ The message cannot be the discriminator: it is flattened and truncated for
21
+ readability, so keying on it means substring-matching prose. ``status_code``
22
+ and ``code`` are the machine-readable form of the same answer, and
23
+ ``retry_after_seconds`` is the server's own estimate of how long the
24
+ condition it just refused will last.
25
+ """
26
+
27
+ def __init__(
28
+ self,
29
+ message: str,
30
+ *,
31
+ status_code: int | None = None,
32
+ code: str | None = None,
33
+ retry_after_seconds: float | None = None,
34
+ ) -> None:
35
+ super().__init__(message)
36
+ self.status_code = status_code
37
+ self.code = code
38
+ self.retry_after_seconds = retry_after_seconds
39
+
40
+
41
+ class HotdataTransientError(HotdataError):
42
+ pass
43
+
44
+
45
+ class HotdataTerminalError(HotdataError):
46
+ pass
47
+
48
+
49
+ def _error_code(body: object) -> str | None:
50
+ """The ``error.code`` an API error envelope carries, if this body is one.
51
+
52
+ Not every 409 comes from an endpoint that speaks the envelope — a failed
53
+ query result is reported as one and carries a result document instead — so
54
+ a missing code is ordinary, and callers fall back to the status.
55
+ """
56
+ if not isinstance(body, (str, bytes, bytearray)):
57
+ return None
58
+ try:
59
+ parsed: object = json.loads(body)
60
+ except ValueError:
61
+ return None
62
+ if not isinstance(parsed, Mapping):
63
+ return None
64
+ error: object = parsed.get("error")
65
+ if not isinstance(error, Mapping):
66
+ return None
67
+ code: object = error.get("code")
68
+ return code if isinstance(code, str) else None
69
+
70
+
71
+ def _retry_after_seconds(headers: object) -> float | None:
72
+ """``Retry-After`` as a number of seconds, when the response states one.
73
+
74
+ Only the delta-seconds form is read. That is what the API sends, and the
75
+ HTTP-date form would need a comparison against a server clock we do not
76
+ have to be worth anything.
77
+ """
78
+ if not isinstance(headers, Mapping):
79
+ return None
80
+ raw: object = headers.get("Retry-After")
81
+ if raw is None:
82
+ # The SDK hands us urllib3's case-insensitive mapping and the API sends
83
+ # the header lower-cased, so the direct hit is what normally answers.
84
+ # Fall back for any plain dict that reaches us instead — a missed
85
+ # header is silent, and silence here reads as "the server asked for
86
+ # nothing".
87
+ raw = next((v for k, v in headers.items() if str(k).lower() == "retry-after"), None)
88
+ if raw is None:
89
+ return None
90
+ try:
91
+ seconds = float(str(raw).strip())
92
+ except ValueError:
93
+ return None
94
+ return seconds if seconds >= 0 else None
95
+
96
+
97
+ def _error_class(status_code: int, code: str | None) -> type[HotdataError]:
98
+ if status_code == 409 and code == _TERMINAL_CONFLICT_CODE:
99
+ # The request cannot succeed as posted — an upload already consumed
100
+ # with nothing to replay, a receipt naming a different target, an
101
+ # incompatible column type. Every retry reaches the same 409.
102
+ return HotdataTerminalError
103
+ if status_code in (408, 409, 425, 429):
104
+ return HotdataTransientError
105
+ if status_code == 501:
106
+ # Not Implemented is a permanent capability gap (e.g. the storage
107
+ # backend cannot issue presigned URLs) — retrying cannot succeed.
108
+ return HotdataTerminalError
109
+ if 500 <= status_code <= 599:
110
+ return HotdataTransientError
111
+ return HotdataTerminalError
112
+
113
+
114
+ def classify_sdk_error(error: Exception) -> HotdataError:
115
+ if isinstance(error, TimeoutError):
116
+ return HotdataTransientError(str(error))
117
+ if isinstance(error, ConnectionError):
118
+ return HotdataTransientError(str(error))
119
+ if isinstance(error, ApiException):
120
+ status_code = int(error.status or 0)
121
+ message = f"{status_code}: {error.reason or 'unknown error'}"
122
+ # The response body is where the API explains itself (e.g. which
123
+ # header is missing) — without it "400: Bad Request" is undebuggable.
124
+ body: object = getattr(error, "body", None)
125
+ if body:
126
+ message = f"{message} — {' '.join(str(body).split())[:500]}"
127
+ code = _error_code(body)
128
+ return _error_class(status_code, code)(
129
+ message,
130
+ status_code=status_code,
131
+ code=code,
132
+ retry_after_seconds=_retry_after_seconds(getattr(error, "headers", None)),
133
+ )
134
+ return HotdataTerminalError(str(error))
@@ -7,6 +7,7 @@ in one place rather than being duplicated per adapter.
7
7
 
8
8
  from __future__ import annotations
9
9
 
10
+ import random
10
11
  import time
11
12
  from collections.abc import Callable
12
13
  from typing import Any, Protocol, TypeVar
@@ -53,6 +54,10 @@ class ManagedDatabaseClient:
53
54
  _QUERY_TIMEOUT_SECONDS = 300.0
54
55
  _POLL_INTERVAL_SECONDS = 0.4
55
56
  _MAX_BACKOFF_SECONDS = 30.0
57
+ # Spread as a fraction of the wait, added on top of it. Half an interval is
58
+ # enough to decorrelate writers that started together without materially
59
+ # changing how long the budget lasts.
60
+ _RETRY_JITTER_FRACTION = 0.5
56
61
 
57
62
  def __init__(
58
63
  self,
@@ -207,9 +212,16 @@ class ManagedDatabaseClient:
207
212
  mode: ManagedLoadMode = "replace",
208
213
  key: list[str] | None = None,
209
214
  ) -> LoadManagedTableResult:
210
- # append is the only non-idempotent mode: if the server commits the load
211
- # but the response is lost, a retry re-appends the same rows. Run it
212
- # at-most-once; every other mode is safe to retry.
215
+ # Retryable in every mode, append included. A retry re-sends the SAME
216
+ # upload_id, and the server keys a receipt on it: a replay returns the
217
+ # committed result rather than applying the load a second time. So the
218
+ # invariant that makes this safe is the upload id, not the mode — a
219
+ # caller that re-stages the upload between attempts mints a new id,
220
+ # loses the receipt, and a retried append would then duplicate rows.
221
+ # This client stages once, in upload_parquet, outside the operation
222
+ # retried here. `HotdataClient.load_managed_table(file=...)` uploads
223
+ # inside the call and so does not hold the invariant; it is unwrapped,
224
+ # and retrying an append through it is the caller's to justify.
213
225
  #
214
226
  # `key` is the merge key for delete/update/upsert loads: when set it is
215
227
  # matched per-load instead of a key declared at table creation. Omit it
@@ -222,20 +234,40 @@ class ManagedDatabaseClient:
222
234
  upload_id=upload_id,
223
235
  mode=mode,
224
236
  key=key,
225
- ),
226
- retryable=(mode != "append"),
237
+ )
227
238
  )
228
239
 
229
- def _request_with_retry(self, operation: Callable[[], T], *, retryable: bool = True) -> T:
230
- max_attempts = self._max_retries if retryable else 1
240
+ def _request_with_retry(self, operation: Callable[[], T]) -> T:
241
+ max_attempts = self._max_retries
231
242
  for attempt in range(1, max_attempts + 1):
232
243
  try:
233
244
  return operation()
234
245
  except Exception as error:
235
246
  mapped_error = classify_sdk_error(error.__cause__ or error)
236
247
  if isinstance(mapped_error, HotdataTransientError) and attempt < max_attempts:
237
- backoff = min(self._retry_backoff_seconds * attempt, self._MAX_BACKOFF_SECONDS)
238
- time.sleep(backoff)
248
+ time.sleep(self._retry_delay(attempt, mapped_error.retry_after_seconds))
239
249
  continue
240
250
  raise mapped_error from error
241
251
  raise RuntimeError("No retry attempts configured")
252
+
253
+ def _retry_delay(self, attempt: int, retry_after_seconds: float | None) -> float:
254
+ """A linear ramp, floored by the server's Retry-After and spread by jitter.
255
+
256
+ Retry-After is a floor rather than a replacement: it says how long the
257
+ condition just refused typically lasts, while the ramp is what gives up
258
+ eventually, and taking the larger of the two honours both. It is capped
259
+ like the ramp so a hostile or mistaken header cannot park an attempt for
260
+ an hour.
261
+
262
+ Jitter is added on top and never subtracted, so a stated Retry-After is
263
+ not undercut. It matters because the callers that collide are the ones
264
+ that started together: writers refused by one table's lock would retry
265
+ in lockstep on an identical ramp and re-collide every time.
266
+ _MAX_BACKOFF_SECONDS caps the ramp, deliberately not the jitter above
267
+ it — clamping the total would flatten every late attempt onto the same
268
+ value and re-correlate exactly the waits that most need spreading.
269
+ """
270
+ base = min(self._retry_backoff_seconds * attempt, self._MAX_BACKOFF_SECONDS)
271
+ if retry_after_seconds is not None:
272
+ base = max(base, min(retry_after_seconds, self._MAX_BACKOFF_SECONDS))
273
+ return base * (1.0 + random.random() * self._RETRY_JITTER_FRACTION)
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.12.0"
7
+ version = "0.13.0"
8
8
  description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"