hotdata-framework 0.12.0__tar.gz → 0.12.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/workflows/publish.yml +15 -0
  2. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/workflows/release.yml +18 -0
  3. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/CHANGELOG.md +6 -0
  4. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/PKG-INFO +1 -1
  5. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/client.py +119 -1
  6. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/databases.py +5 -0
  7. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/pyproject.toml +1 -1
  8. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_client.py +331 -1
  9. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/uv.lock +1 -1
  10. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/CODEOWNERS +0 -0
  11. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/dependabot.yml +0 -0
  12. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/workflows/check-release.yml +0 -0
  13. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/workflows/ci.yml +0 -0
  14. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.github/workflows/dependabot-automerge.yml +0 -0
  15. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/.gitignore +0 -0
  16. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/CONTRACT.md +0 -0
  17. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/README.md +0 -0
  18. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/RELEASING.md +0 -0
  19. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/examples/basic_usage.py +0 -0
  20. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/__init__.py +0 -0
  21. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/env.py +0 -0
  22. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/errors.py +0 -0
  23. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/health.py +0 -0
  24. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/managed_client.py +0 -0
  25. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/py.typed +0 -0
  26. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/hotdata_framework/result.py +0 -0
  27. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/scripts/check-release.py +0 -0
  28. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/scripts/extract-changelog.py +0 -0
  29. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/scripts/publish-workflow.sh +0 -0
  30. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/scripts/release.sh +0 -0
  31. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/scripts/update_changelog.py +0 -0
  32. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_contract.py +0 -0
  33. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_databases.py +0 -0
  34. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_errors.py +0 -0
  35. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_health.py +0 -0
  36. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_indexes.py +0 -0
  37. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_managed_client.py +0 -0
  38. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_request_timeout.py +0 -0
  39. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_result.py +0 -0
  40. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_retry_policy.py +0 -0
  41. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_update_changelog.py +0 -0
  42. {hotdata_framework-0.12.0 → hotdata_framework-0.12.1}/tests/test_version.py +0 -0
@@ -31,6 +31,21 @@ jobs:
31
31
  # The tag being released, whether it arrived by push or by dispatch.
32
32
  TAG: ${{ inputs.tag || github.ref_name }}
33
33
  steps:
34
+ # Before checkout, because checkout resolves the input as an arbitrary ref:
35
+ # a branch or SHA is fetched first and only rejected later by the version
36
+ # match below, which is also looser (`^v[0-9]`). Strict here so the dispatch
37
+ # contract matches release.yml — release.sh only ever produces X.Y.Z.
38
+ - name: Validate release tag format
39
+ if: github.event_name == 'workflow_dispatch'
40
+ env:
41
+ INPUT_TAG: ${{ inputs.tag }}
42
+ run: |
43
+ set -euo pipefail
44
+ if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
45
+ echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
46
+ exit 1
47
+ fi
48
+
34
49
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
35
50
  with:
36
51
  ref: ${{ inputs.tag || github.ref_name }}
@@ -25,6 +25,24 @@ jobs:
25
25
  # The tag being released, whether it arrived by push or by dispatch.
26
26
  TAG: ${{ inputs.tag || github.ref_name }}
27
27
  steps:
28
+ # Before checkout, because checkout resolves the input as an arbitrary ref
29
+ # — and this workflow needs the guard more than publish.yml does. It holds
30
+ # `contents: write`, and action-gh-release CREATES a tag when tag_name does
31
+ # not resolve to one, so an unvalidated `tag: main` would check out cleanly
32
+ # and leave refs/tags/main plus a release named for it. The push path is
33
+ # constrained by the v[0-9]* filter; the dispatch path was not constrained
34
+ # at all.
35
+ - name: Validate release tag format
36
+ if: github.event_name == 'workflow_dispatch'
37
+ env:
38
+ INPUT_TAG: ${{ inputs.tag }}
39
+ run: |
40
+ set -euo pipefail
41
+ if [[ ! "$INPUT_TAG" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
42
+ echo "tag must look like vX.Y.Z, got: $INPUT_TAG" >&2
43
+ exit 1
44
+ fi
45
+
28
46
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
29
47
  with:
30
48
  ref: ${{ inputs.tag || github.ref_name }}
@@ -8,6 +8,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
8
8
  ## [Unreleased]
9
9
 
10
10
 
11
+ ## [0.12.1] - 2026-08-18
12
+
13
+ ### Fixed
14
+
15
+ - fix(load): submit managed loads as a job and poll, instead of holding one request open
16
+
11
17
  ## [0.12.0] - 2026-08-11
12
18
 
13
19
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.12.0
3
+ Version: 0.12.1
4
4
  Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -25,6 +25,7 @@ from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
25
25
  from hotdata.models.index_info_response import IndexInfoResponse
26
26
  from hotdata.models.job_status_response import JobStatusResponse
27
27
  from hotdata.models.load_managed_table_request import LoadManagedTableRequest
28
+ from hotdata.models.load_managed_table_response import LoadManagedTableResponse
28
29
  from hotdata.models.query_request import QueryRequest
29
30
  from hotdata.models.query_response import QueryResponse
30
31
  from hotdata.models.submit_job_response import SubmitJobResponse
@@ -81,6 +82,37 @@ _RESULT_FAILURE = frozenset({"failed", "cancelled"})
81
82
  # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
82
83
  _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
83
84
 
85
+ # How long a load may finish INLINE before the server hands back a job instead.
86
+ # Small enough that a slow load stops holding a request open, large enough that
87
+ # the overwhelming majority never become jobs at all: dlt's bookkeeping tables
88
+ # (`_dlt_version`, `_dlt_loads`, `_dlt_pipeline_state`) settle in under a second,
89
+ # and paying a submit-then-poll round trip for those would be a regression.
90
+ _LOAD_INLINE_WAIT_MS = 10_000
91
+
92
+ # A load's own polling budget. Deliberately NOT the 300s used for queries and
93
+ # results: a load is the one operation here whose duration scales with the data,
94
+ # and reusing the query budget is what put a five-minute ceiling on it in the
95
+ # first place. Bounded rather than unbounded so a wedged job still surfaces.
96
+ _LOAD_JOB_TIMEOUT_S = 3600.0
97
+
98
+ # How long a poll tolerates CONTINUOUS status-check failure before giving up.
99
+ #
100
+ # Time, not a count: a count means whatever the caller's `interval_s` makes it,
101
+ # and callers differ -- five checks is eight seconds at the load interval and
102
+ # something else at the index one. The thing being survived is a gateway blip,
103
+ # and a rolling restart or a load balancer reconverging routinely serves 502s for
104
+ # longer than a few seconds. Too short and the poll aborts, the caller's retry
105
+ # re-submits the load, and the original job is still holding the table -- the
106
+ # door this tolerance exists to close.
107
+ #
108
+ # Still bounded by the poll's own deadline, so this only decides how a stretch of
109
+ # failures ends, never how long the wait can be.
110
+ _JOB_POLL_ERROR_GRACE_S = 120.0
111
+
112
+ # Failed checks back off rather than hammering at `interval_s`: whatever is
113
+ # serving 502s does not need the extra traffic.
114
+ _JOB_POLL_ERROR_MAX_BACKOFF_S = 15.0
115
+
84
116
 
85
117
  @dataclass(frozen=True)
86
118
  class ResultSummary:
@@ -418,10 +450,25 @@ class HotdataClient:
418
450
  else:
419
451
  assert file is not None
420
452
  resolved_upload_id = self.upload_parquet(file)
453
+ # ASKED FOR AS A JOB, not as a held-open request. A load's duration scales
454
+ # with the data, and a single request that must survive minutes has to
455
+ # survive every layer between here and the engine -- CDN, gateway, socket
456
+ # read timeout -- any one of which ends it. When it ends, the server logs
457
+ # the load `abandoned` and DISCARDS work it had already done, while the
458
+ # table's write lock is still held against the retry that follows; the
459
+ # retry then collides with it (409 RESOURCE_LOCKED) and the pair can spin
460
+ # indefinitely without the load ever completing. Observed in production on
461
+ # a table whose load runs past five minutes.
462
+ #
463
+ # `async_after_ms` keeps the common case unchanged: the server answers 200
464
+ # with the result if it finishes inside the window, and only falls back to
465
+ # a job when it does not. So nothing pays for polling that did not need it.
421
466
  request = LoadManagedTableRequest(
422
467
  mode=mode,
423
468
  upload_id=resolved_upload_id,
424
469
  key=key,
470
+ var_async=True,
471
+ async_after_ms=_LOAD_INLINE_WAIT_MS,
425
472
  )
426
473
  try:
427
474
  loaded = self.connections().load_managed_table(
@@ -432,12 +479,21 @@ class HotdataClient:
432
479
  )
433
480
  except ApiException as e:
434
481
  raise RuntimeError(api_error_message(e)) from e
482
+ # Only the job branch is type-checked. The index path can also assert its
483
+ # inline type because it names it positively first; here the inline shape is
484
+ # read duck-typed, which callers and tests already rely on, so asserting it
485
+ # would narrow an interface this change has no business narrowing.
486
+ job_id: str | None = None
487
+ if isinstance(loaded, SubmitJobResponse):
488
+ job_id = loaded.id
489
+ loaded = self._load_response_from_job(job_id)
435
490
  return LoadManagedTableResult(
436
491
  connection_id=loaded.connection_id,
437
492
  schema_name=loaded.schema_name,
438
493
  table_name=loaded.table_name,
439
494
  row_count=loaded.row_count,
440
495
  full_name=f"{db.id}.{loaded.schema_name}.{loaded.table_name}",
496
+ job_id=job_id,
441
497
  )
442
498
 
443
499
  def add_managed_table(
@@ -880,11 +936,40 @@ class HotdataClient:
880
936
  jobs = self._jobs_api()
881
937
  deadline = time.monotonic() + timeout_s
882
938
  last: JobStatusResponse | None = None
939
+ # When the current run of failures began, or None while checks succeed.
940
+ failing_since: float | None = None
941
+ error_backoff = interval_s
883
942
  while time.monotonic() < deadline:
884
943
  try:
885
944
  last = jobs.get_job(job_id)
945
+ failing_since = None
946
+ error_backoff = interval_s
886
947
  except ApiException as e:
887
- raise RuntimeError(api_error_message(e)) from e
948
+ # A failed STATUS CHECK is not a failed job. Aborting here throws
949
+ # away work that is still running, and the caller's retry then
950
+ # re-submits it while the original still holds its resources -- so
951
+ # one blip becomes a collision with the job it just abandoned. A
952
+ # load polls for up to `_LOAD_JOB_TIMEOUT_S`, so the longer the
953
+ # wait the more chances to hit it, which is exactly backwards.
954
+ #
955
+ # CONSECUTIVE failures are the signal: an isolated 502 is noise, a
956
+ # run of them means the API is gone and there is nothing to wait
957
+ # for. The poll's own deadline bounds the total wait regardless.
958
+ now = time.monotonic()
959
+ if failing_since is None:
960
+ failing_since = now
961
+ elif now - failing_since >= _JOB_POLL_ERROR_GRACE_S:
962
+ # Name the job. This is one of the two paths where the caller
963
+ # cannot tell whether the load landed, so the id is the only
964
+ # thing that makes the question answerable -- and it is exactly
965
+ # what a message like "502: Bad Gateway" leaves out.
966
+ raise RuntimeError(
967
+ f"Job {job_id} status checks failed for "
968
+ f"{_JOB_POLL_ERROR_GRACE_S:.0f}s: {api_error_message(e)}"
969
+ ) from e
970
+ time.sleep(error_backoff)
971
+ error_backoff = min(error_backoff * 2, _JOB_POLL_ERROR_MAX_BACKOFF_S)
972
+ continue
888
973
  if last.status in _JOB_TERMINAL:
889
974
  return last
890
975
  time.sleep(interval_s)
@@ -893,6 +978,39 @@ class HotdataClient:
893
978
  f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
894
979
  )
895
980
 
981
+ def _load_response_from_job(self, job_id: str) -> LoadManagedTableResponse:
982
+ """The result of a load the server chose to run as a job.
983
+
984
+ Polling replaces waiting on the request, so the outcome is read from
985
+ durable state rather than from a connection that has to stay alive. That
986
+ also gives a caller a handle: the job id is returned on
987
+ `LoadManagedTableResult`, so "did it land?" is answerable after a lost
988
+ response. `append` stays non-retryable -- knowing the id makes the question
989
+ answerable, it does not make a blind re-submission safe, and that call is
990
+ the caller's to make.
991
+
992
+ `partially_succeeded` is terminal and carries a message, so it is raised
993
+ rather than returned -- a caller asked for a table's contents to be
994
+ replaced or appended to, and "some of it" is not an answer it can use.
995
+ """
996
+ final = self._poll_job(job_id, timeout_s=_LOAD_JOB_TIMEOUT_S)
997
+ status = enum_value(final.status)
998
+ if status != "succeeded":
999
+ # The id goes in whether or not the server's message mentions it: the
1000
+ # caller is being told the load did not succeed, and "which load" is
1001
+ # the next thing it needs.
1002
+ detail = final.error_message or f"finished {status}"
1003
+ raise RuntimeError(f"load job {job_id} {status}: {detail}")
1004
+ # `result` is a oneOf wrapper today; tolerate the model arriving directly,
1005
+ # the same way the index path does rather than disagreeing with it.
1006
+ payload = getattr(final.result, "actual_instance", final.result)
1007
+ if not isinstance(payload, LoadManagedTableResponse):
1008
+ raise RuntimeError(
1009
+ f"load job {job_id} succeeded without a load result "
1010
+ f"(got {type(payload).__name__})"
1011
+ )
1012
+ return payload
1013
+
896
1014
  def _wait_result_ready(
897
1015
  self,
898
1016
  result_id: str,
@@ -83,6 +83,11 @@ class LoadManagedTableResult:
83
83
  table_name: str
84
84
  row_count: int
85
85
  full_name: str
86
+ # Set when the server ran the load as a background job. Carried for the same
87
+ # reason CreateIndexResult carries it: it is the only handle a caller has to
88
+ # ask "did that land?" after a lost response, and without it the question is
89
+ # unanswerable. `None` when the load finished inline.
90
+ job_id: str | None = None
86
91
 
87
92
  def to_dict(self) -> dict[str, Any]:
88
93
  return asdict(self)
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.12.0"
7
+ version = "0.12.1"
8
8
  description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -37,11 +37,16 @@ class _ForbiddenDatabasesApi:
37
37
 
38
38
 
39
39
  class _FakeConnectionsApi:
40
- def __init__(self) -> None:
40
+ def __init__(self, responses=None) -> None:
41
41
  self.load_calls: list[tuple[str, str, str]] = []
42
+ self.requests: list = []
43
+ self._responses = list(responses) if responses else None
42
44
 
43
45
  def load_managed_table(self, connection_id, schema, table, request):
44
46
  self.load_calls.append((connection_id, schema, table))
47
+ self.requests.append(request)
48
+ if self._responses:
49
+ return self._responses.pop(0)
45
50
  return SimpleNamespace(
46
51
  connection_id=connection_id,
47
52
  schema_name=schema,
@@ -544,3 +549,328 @@ def test_from_env_requires_an_api_key(monkeypatch: pytest.MonkeyPatch):
544
549
  monkeypatch.delenv("HOTDATA_API_KEY", raising=False)
545
550
  with pytest.raises(RuntimeError, match="HOTDATA_API_KEY"):
546
551
  HotdataClient.from_env()
552
+
553
+
554
+ # --------------------------------------------------------------------------
555
+ # A load is submitted as a job, not held open on one request
556
+ # --------------------------------------------------------------------------
557
+
558
+
559
+ def _load_response(rows=7):
560
+ from hotdata.models.load_managed_table_response import LoadManagedTableResponse
561
+
562
+ return LoadManagedTableResponse(
563
+ connection_id="conn_1", schema_name="public",
564
+ table_name="orders", row_count=rows,
565
+ arrow_schema_json="{}",
566
+ )
567
+
568
+
569
+ def test_a_load_asks_for_a_job_with_an_inline_window():
570
+ """The request itself is the fix. A load whose duration scales with the data
571
+ must not depend on one HTTP request surviving minutes through every layer
572
+ between here and the engine; when such a request dies the server discards the
573
+ work and leaves the table locked against the retry."""
574
+ from hotdata_framework.client import _LOAD_INLINE_WAIT_MS
575
+
576
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
577
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
578
+ connections = _FakeConnectionsApi()
579
+
580
+ with (
581
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
582
+ patch.object(client, "connections", return_value=connections),
583
+ ):
584
+ client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
585
+
586
+ req = connections.requests[0]
587
+ assert req.var_async is True, "load was not submitted as a job"
588
+ assert req.async_after_ms == _LOAD_INLINE_WAIT_MS
589
+
590
+
591
+ def test_a_load_that_finishes_inline_costs_no_polling():
592
+ """dlt's bookkeeping tables settle in under a second. Making those pay a
593
+ submit-then-poll round trip would be a regression, which is what
594
+ `async_after_ms` exists to prevent."""
595
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
596
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
597
+ connections = _FakeConnectionsApi()
598
+
599
+ def _forbidden_poll(*a, **k):
600
+ raise AssertionError("polled a load the server answered inline")
601
+
602
+ with (
603
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
604
+ patch.object(client, "connections", return_value=connections),
605
+ patch.object(client, "_poll_job", _forbidden_poll),
606
+ ):
607
+ result = client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
608
+
609
+ assert result.row_count == 3
610
+
611
+
612
+ def test_a_load_the_server_defers_is_polled_to_completion():
613
+ """The 202 path: the result comes from durable job state rather than from a
614
+ connection that had to stay alive to carry it."""
615
+ from hotdata.models.submit_job_response import SubmitJobResponse
616
+
617
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
618
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
619
+ connections = _FakeConnectionsApi(responses=[
620
+ SubmitJobResponse(id="jobs_1", status="running", status_url="/v1/jobs/jobs_1"),
621
+ ])
622
+ final = SimpleNamespace(
623
+ status="succeeded", error_message=None,
624
+ result=SimpleNamespace(actual_instance=_load_response(rows=91)),
625
+ )
626
+
627
+ with (
628
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
629
+ patch.object(client, "connections", return_value=connections),
630
+ patch.object(client, "_poll_job", return_value=final) as poll,
631
+ ):
632
+ result = client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
633
+
634
+ assert result.row_count == 91
635
+ assert poll.call_args.args[0] == "jobs_1"
636
+
637
+
638
+ def test_a_load_job_gets_its_own_budget_not_the_query_one():
639
+ """Reusing the 300s query budget is what put a five-minute ceiling on loads in
640
+ the first place; a load is the one operation whose duration scales with the
641
+ data."""
642
+ from hotdata.models.submit_job_response import SubmitJobResponse
643
+
644
+ from hotdata_framework.client import _LOAD_JOB_TIMEOUT_S
645
+
646
+ assert _LOAD_JOB_TIMEOUT_S > 300.0
647
+
648
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
649
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
650
+ connections = _FakeConnectionsApi(responses=[
651
+ SubmitJobResponse(id="jobs_1", status="running", status_url="/v1/jobs/jobs_1"),
652
+ ])
653
+ final = SimpleNamespace(
654
+ status="succeeded", error_message=None,
655
+ result=SimpleNamespace(actual_instance=_load_response()),
656
+ )
657
+
658
+ with (
659
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
660
+ patch.object(client, "connections", return_value=connections),
661
+ patch.object(client, "_poll_job", return_value=final) as poll,
662
+ ):
663
+ client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
664
+
665
+ assert poll.call_args.kwargs["timeout_s"] == _LOAD_JOB_TIMEOUT_S
666
+
667
+
668
+ def test_a_failed_load_job_raises_with_the_server_message():
669
+ from hotdata.models.submit_job_response import SubmitJobResponse
670
+
671
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
672
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
673
+ connections = _FakeConnectionsApi(responses=[
674
+ SubmitJobResponse(id="jobs_1", status="running", status_url="/v1/jobs/jobs_1"),
675
+ ])
676
+ final = SimpleNamespace(status="failed", error_message="disk full", result=None)
677
+
678
+ with (
679
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
680
+ patch.object(client, "connections", return_value=connections),
681
+ patch.object(client, "_poll_job", return_value=final),
682
+ ):
683
+ try:
684
+ client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
685
+ except RuntimeError as e:
686
+ assert "disk full" in str(e), e
687
+ else:
688
+ raise AssertionError("a failed load job did not raise")
689
+
690
+
691
+ def test_a_partially_succeeded_load_job_is_not_treated_as_success():
692
+ """A caller asked for a table's contents to be replaced or appended to;
693
+ "some of it" is not an answer it can use."""
694
+ from hotdata.models.submit_job_response import SubmitJobResponse
695
+
696
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
697
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
698
+ connections = _FakeConnectionsApi(responses=[
699
+ SubmitJobResponse(id="jobs_1", status="running", status_url="/v1/jobs/jobs_1"),
700
+ ])
701
+ final = SimpleNamespace(
702
+ status="partially_succeeded", error_message="3 rows rejected",
703
+ result=SimpleNamespace(actual_instance=_load_response()),
704
+ )
705
+
706
+ with (
707
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
708
+ patch.object(client, "connections", return_value=connections),
709
+ patch.object(client, "_poll_job", return_value=final),
710
+ ):
711
+ try:
712
+ client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
713
+ except RuntimeError as e:
714
+ assert "3 rows rejected" in str(e), e
715
+ else:
716
+ raise AssertionError("partially_succeeded was treated as success")
717
+
718
+
719
+ def test_a_transient_status_check_does_not_discard_a_running_job():
720
+ """The regression this guards. A load polls for up to an hour, so there are
721
+ hundreds of status checks; aborting on the first bad one throws away a job
722
+ that is still running, and the caller's retry then re-submits the load while
723
+ the original still holds the table -- the 409 spin, from a new direction."""
724
+ from hotdata.exceptions import ApiException
725
+
726
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
727
+ calls = {"n": 0}
728
+ final = SimpleNamespace(status="succeeded", error_message=None,
729
+ result=SimpleNamespace(actual_instance=_load_response(5)))
730
+
731
+ class _Flaky:
732
+ def get_job(self, job_id):
733
+ calls["n"] += 1
734
+ if calls["n"] < 3:
735
+ raise ApiException(status=502, reason="Bad Gateway")
736
+ return final
737
+
738
+ with (
739
+ patch.object(client, "_jobs_api", return_value=_Flaky()),
740
+ patch("time.sleep", lambda *_: None),
741
+ ):
742
+ got = client._poll_job("jobs_1", timeout_s=60.0, interval_s=0.01)
743
+
744
+ assert got is final, "a blipping status check aborted the poll"
745
+ assert calls["n"] == 3
746
+
747
+
748
+ def test_a_sustained_run_of_failed_status_checks_gives_up_naming_the_job():
749
+ """Tolerance is for blips, not for an API that has gone away -- and the message
750
+ has to name the job, because this is one of the two paths where the caller
751
+ cannot tell whether the load landed. `502: Bad Gateway` alone does not."""
752
+ from hotdata.exceptions import ApiException
753
+
754
+ from hotdata_framework.client import _JOB_POLL_ERROR_GRACE_S
755
+
756
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
757
+
758
+ class _Dead:
759
+ def get_job(self, job_id):
760
+ raise ApiException(status=502, reason="Bad Gateway")
761
+
762
+ # A FAKE CLOCK, advanced by the sleeps. The tolerance is measured in time, so
763
+ # a no-op `sleep` with a real clock would make this test wait out the whole
764
+ # grace period in wall time -- 120 seconds of a spinning loop to assert one
765
+ # message.
766
+ clock = {"t": 0.0}
767
+ with (
768
+ patch.object(client, "_jobs_api", return_value=_Dead()),
769
+ patch("time.monotonic", lambda: clock["t"]),
770
+ patch("time.sleep", lambda s: clock.__setitem__("t", clock["t"] + s)),
771
+ ):
772
+ try:
773
+ client._poll_job("jobs_1", timeout_s=6000.0, interval_s=1.0)
774
+ except RuntimeError as e:
775
+ assert "jobs_1" in str(e), f"gave up without naming the job: {e}"
776
+ else:
777
+ raise AssertionError("polled forever against a dead API")
778
+ assert clock["t"] >= _JOB_POLL_ERROR_GRACE_S, "gave up before the grace elapsed"
779
+ assert clock["t"] < 6000.0, "ran to the poll deadline instead of the grace"
780
+
781
+
782
+ def test_the_error_tolerance_is_measured_in_time_not_checks():
783
+ """A count means whatever the caller's `interval_s` makes it, and callers
784
+ differ. The thing being survived is a gateway blip, which lasts seconds to
785
+ minutes regardless of how often we happen to ask."""
786
+ from hotdata_framework.client import _JOB_POLL_ERROR_GRACE_S
787
+
788
+ # long enough to outlast a rolling restart / LB reconverge, not seconds
789
+ assert _JOB_POLL_ERROR_GRACE_S >= 60.0
790
+
791
+
792
+ def test_failed_status_checks_back_off_instead_of_hammering():
793
+ """Whatever is serving 502s does not need the extra traffic."""
794
+ from hotdata.exceptions import ApiException
795
+
796
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
797
+ sleeps: list[float] = []
798
+ calls = {"n": 0}
799
+ final = SimpleNamespace(status="succeeded", error_message=None,
800
+ result=SimpleNamespace(actual_instance=_load_response()))
801
+
802
+ class _Flaky:
803
+ def get_job(self, job_id):
804
+ calls["n"] += 1
805
+ if calls["n"] < 5:
806
+ raise ApiException(status=502, reason="Bad Gateway")
807
+ return final
808
+
809
+ with (
810
+ patch.object(client, "_jobs_api", return_value=_Flaky()),
811
+ patch("time.sleep", lambda s: sleeps.append(s)),
812
+ ):
813
+ client._poll_job("jobs_1", timeout_s=600.0, interval_s=1.0)
814
+
815
+ failed_waits = sleeps[:4]
816
+ assert failed_waits == sorted(failed_waits), f"did not back off: {failed_waits}"
817
+ assert failed_waits[-1] > failed_waits[0]
818
+
819
+
820
+ def test_a_failed_load_job_names_the_job_alongside_the_server_message():
821
+ from hotdata.models.submit_job_response import SubmitJobResponse
822
+
823
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
824
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
825
+ connections = _FakeConnectionsApi(responses=[
826
+ SubmitJobResponse(id="jobs_42", status="running", status_url="/v1/jobs/jobs_42"),
827
+ ])
828
+ final = SimpleNamespace(status="failed", error_message="disk full", result=None)
829
+
830
+ with (
831
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
832
+ patch.object(client, "connections", return_value=connections),
833
+ patch.object(client, "_poll_job", return_value=final),
834
+ ):
835
+ try:
836
+ client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
837
+ except RuntimeError as e:
838
+ assert "disk full" in str(e) and "jobs_42" in str(e), e
839
+ else:
840
+ raise AssertionError("a failed load job did not raise")
841
+
842
+
843
+ def test_a_deferred_load_returns_the_job_id_to_the_caller():
844
+ """`append` stays non-retryable, so the id is the only handle a caller has to
845
+ answer "did it land?" after a lost response -- the same reason
846
+ CreateIndexResult carries one."""
847
+ from hotdata.models.submit_job_response import SubmitJobResponse
848
+
849
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
850
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
851
+ connections = _FakeConnectionsApi(responses=[
852
+ SubmitJobResponse(id="jobs_77", status="running", status_url="/v1/jobs/jobs_77"),
853
+ ])
854
+ final = SimpleNamespace(status="succeeded", error_message=None,
855
+ result=SimpleNamespace(actual_instance=_load_response()))
856
+
857
+ with (
858
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
859
+ patch.object(client, "connections", return_value=connections),
860
+ patch.object(client, "_poll_job", return_value=final),
861
+ ):
862
+ result = client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
863
+
864
+ assert result.job_id == "jobs_77"
865
+
866
+
867
+ def test_an_inline_load_carries_no_job_id():
868
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
869
+ db = ManagedDatabase(id="db_1", description="mydb", default_connection_id="conn_1")
870
+ with (
871
+ patch.object(client, "_databases_api", return_value=_ForbiddenDatabasesApi()),
872
+ patch.object(client, "connections", return_value=_FakeConnectionsApi()),
873
+ ):
874
+ result = client.load_managed_table(db, "orders", schema="public", upload_id="up_1")
875
+ assert result.job_id is None
876
+
@@ -101,7 +101,7 @@ wheels = [
101
101
 
102
102
  [[package]]
103
103
  name = "hotdata-framework"
104
- version = "0.12.0"
104
+ version = "0.12.1"
105
105
  source = { editable = "." }
106
106
  dependencies = [
107
107
  { name = "hotdata" },