hotdata-framework 0.10.0__tar.gz → 0.11.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/CHANGELOG.md +61 -50
  2. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/CONTRACT.md +1 -3
  3. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/PKG-INFO +7 -8
  4. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/README.md +3 -4
  5. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/__init__.py +0 -2
  6. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/client.py +8 -13
  7. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/env.py +5 -12
  8. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/health.py +0 -2
  9. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/pyproject.toml +14 -4
  10. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/scripts/release.sh +41 -8
  11. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_client.py +103 -6
  12. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_contract.py +0 -1
  13. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_health.py +16 -3
  14. hotdata_framework-0.11.0/tests/test_retry_policy.py +63 -0
  15. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/uv.lock +2 -2
  16. hotdata_framework-0.10.0/hotdata_framework/http.py +0 -17
  17. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/CODEOWNERS +0 -0
  18. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/dependabot.yml +0 -0
  19. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/workflows/check-release.yml +0 -0
  20. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/workflows/ci.yml +0 -0
  21. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/workflows/dependabot-automerge.yml +0 -0
  22. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/workflows/publish.yml +0 -0
  23. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.github/workflows/release.yml +0 -0
  24. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/.gitignore +0 -0
  25. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/RELEASING.md +0 -0
  26. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/examples/basic_usage.py +0 -0
  27. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/databases.py +0 -0
  28. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/errors.py +0 -0
  29. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/managed_client.py +0 -0
  30. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/py.typed +0 -0
  31. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/hotdata_framework/result.py +0 -0
  32. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/scripts/check-release.py +0 -0
  33. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/scripts/extract-changelog.py +0 -0
  34. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/scripts/publish-workflow.sh +0 -0
  35. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/scripts/update_changelog.py +0 -0
  36. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_databases.py +0 -0
  37. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_errors.py +0 -0
  38. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_indexes.py +0 -0
  39. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_managed_client.py +0 -0
  40. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_request_timeout.py +0 -0
  41. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_result.py +0 -0
  42. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_update_changelog.py +0 -0
  43. {hotdata_framework-0.10.0 → hotdata_framework-0.11.0}/tests/test_version.py +0 -0
@@ -7,61 +7,72 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.11.0] - 2026-08-11
11
+
12
+ ### Changed
13
+
14
+ - Cap the `hotdata` dependency to the current minor (`>=0.8.0,<0.9`). This
15
+ package wraps a *generated* client, so an SDK minor can remove a model field
16
+ or a `Configuration` keyword this wrapper passes, and there is no regeneration
17
+ step here to surface it — an uncapped floor turns an SDK release into a break
18
+ in this package, in versions already published. Raise the cap deliberately
19
+ after running the suite against the new minor.
20
+
21
+ ### Removed
22
+
23
+ - **Breaking:** session/sandbox support is gone. `HotdataClient` no longer accepts
24
+ `session_id=`, `HotdataClient.session_id` is removed, `default_session_id()` and
25
+ the `HOTDATA_SANDBOX` read are gone, and `list_workspaces()`,
26
+ `resolve_workspace_selection()` and `pick_workspace()` lose their `session_id`
27
+ parameter — note that loss is **positional**, so a three-argument call raises an
28
+ arity `TypeError` rather than an unexpected-keyword one.
29
+ `workspace_health_lines()` no longer emits a `sandbox` line.
30
+
31
+ **Why now.** The server stopped enforcing session scoping some time ago, so the
32
+ value already reached nothing. What makes removal urgent rather than tidy is
33
+ that the SDK is dropping the `SessionId` security scheme: against that release
34
+ `Configuration(session_id=...)` raises `TypeError` instead of setting a header,
35
+ and this package passed it unconditionally — so every `HotdataClient(...)`
36
+ would fail at construction. This package still pins `hotdata<0.9`, so nothing
37
+ is broken today; the change is what lets the cap be raised later without a
38
+ second breaking release.
39
+
40
+ **Migrating.** Drop `session_id=` from `HotdataClient(...)`, stop reading
41
+ `client.session_id`, stop setting `HOTDATA_SANDBOX`, and pass two arguments to
42
+ the workspace helpers. Adapters that re-export session context in their own
43
+ signatures — a `session_id=` parameter, a `session_id` metadata key — need to
44
+ remove it from theirs too, which makes their own release breaking in turn.
45
+
46
+ - `hotdata_framework.http` and `default_http_retries()`. The module existed only
47
+ to build the `retries=` policy removed under Fixed below, and had no other
48
+ callers. It predates `hotdata._retry`, which supersedes it.
49
+
50
+ ### Fixed
51
+
52
+ - A `POST` is no longer replayed because of a response status. `HotdataClient`
53
+ passed its own `retries=` into `Configuration`, which replaced the generated
54
+ SDK's policy wholesale with one listing `POST` in `allowed_methods` alongside
55
+ a `(502, 503, 504)` forcelist — so an intermediary timing out a long request
56
+ produced a silent, identical re-`POST` while the server was still working on
57
+ the first one. For a load that is not idempotent: the duplicate collides with
58
+ the write lock the original holds and is refused.
59
+
60
+ The override is removed and the SDK's own default now applies. It is the
61
+ policy this wrapper was reaching for — `hotdata._retry` retries a
62
+ *pre-response* connection reset (the stale pooled socket case, where the
63
+ server did no work) on any method, while leaving read timeouts and status
64
+ retries idempotent-only.
10
65
 
11
66
  ## [0.10.0] - 2026-08-07
12
67
 
13
68
  ### Added
14
69
 
15
- - `create_index(database, table, columns=..., index_type=...)` builds an index on a
16
- managed table, bringing the framework client to parity with `hotdata indexes
17
- create` in the CLI. It covers all three index kinds the API accepts: `"bm25"` for
18
- full-text search, `"vector"` for nearest-neighbour search, and `"sorted"`.
19
- Previously the framework had no index API at all and callers had to drop to the raw
20
- `hotdata.IndexesApi`, which left a managed database's data loaded but not
21
- searchable: full-text queries error without an index, and vector queries run at
22
- full-scan speed. Like the other managed-table operations, `database` accepts a
23
- name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
24
- (non-managed) connection is not covered — the CLI's `--catalog` handles that.
25
-
26
- `index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
27
- derivation the CLI uses when `--name` is omitted, so both surfaces name the same
28
- index identically. `index_type` is required, unlike the API's `"sorted"` default,
29
- because the wrong kind only fails at query time.
30
-
31
- The server builds the index as a background job whose submit call reports success
32
- even when the build later fails, so `create_index` polls the job to a terminal
33
- state and raises `RuntimeError` carrying the job's `error_message`. Pass
34
- `wait=False` to return once the job is accepted (`status="pending"` plus a
35
- `job_id`) and own the outcome check yourself, as the CLI's `--async` does;
36
- `timeout_s` and `poll_interval_s` tune the wait.
37
-
38
- Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
39
- existing vector column, queried with a literal vector — and there `metric` must
40
- match the distance function the query uses (`cosine`→`cosine_distance`,
41
- `l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
42
- reverts to a full table scan rather than erroring. Setting
43
- `embedding_provider_id` indexes a *text* column instead: the provider embeds it
44
- into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
45
- and the server resolves the distance function itself.
46
-
47
- Argument combinations that the server would silently ignore raise `ValueError`
48
- before any request is sent: an unknown `index_type` or `metric`, a vector index
49
- with more than one column (the engine indexes only the first), and
50
- `metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
51
- non-vector index.
52
-
53
- Verified against `api.hotdata.dev` when this version was released: BM25 and vector
54
- indexes both build and report `ready`, and a BM25 index is used by full-text
55
- search. A *vector* index on a managed database was **not** picked up by the query
56
- planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
57
- planned as a full scan. That reproduces with an index created by `hotdata indexes
58
- create`, so it is an engine-side issue rather than a client one, but it means a
59
- vector index built through this method may not yet accelerate queries.
60
-
61
- - `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
62
- `hotdata_framework` and added to the public contract surface. Its `source_column`
63
- names the text column to query for a provider-backed vector index, and is `None`
64
- for BM25, sorted, and plain vector indexes.
70
+ - `create_index(database, table, columns=..., index_type=...)` builds a `bm25`,
71
+ `vector`, or `sorted` index on a managed table, matching `hotdata indexes create`
72
+ in the CLI. The build is a background job whose submit call reports success even
73
+ when the build later fails, so this polls the job and raises `RuntimeError` with
74
+ its error message; `wait=False` returns as soon as the job is accepted. Returns
75
+ `CreateIndexResult`, also exported.
65
76
 
66
77
  ## [0.9.0] - 2026-07-23
67
78
 
@@ -21,7 +21,6 @@ The supported import surface is:
21
21
  - `workspace_health_lines`
22
22
  - `default_api_key`
23
23
  - `default_host`
24
- - `default_session_id`
25
24
  - `explicit_workspace_id`
26
25
  - `list_workspaces`
27
26
  - `normalize_host`
@@ -43,7 +42,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
43
42
 
44
43
  ### `HotdataClient`
45
44
 
46
- - Represents runtime context: API key, host, workspace, optional session.
45
+ - Represents runtime context: API key, host, workspace.
47
46
  - `from_env()` resolves runtime context from env vars and selected workspace.
48
47
  - `execute_sql(sql)` returns `QueryResult` or raises `RuntimeError`/`TimeoutError`.
49
48
  - `get_result(result_id)` returns a ready `QueryResult` and waits for readiness when needed.
@@ -79,7 +78,6 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
79
78
 
80
79
  - `default_api_key()` reads `HOTDATA_API_KEY`.
81
80
  - `default_host()` reads `HOTDATA_API_URL` (default: `https://api.hotdata.dev`) and normalizes it.
82
- - `default_session_id()` reads `HOTDATA_SANDBOX`.
83
81
  - `explicit_workspace_id()` reads `HOTDATA_WORKSPACE` (workspace public id).
84
82
  - `pick_workspace()` prefers explicit env workspace, then active workspace, then first workspace.
85
83
  - `resolve_workspace_selection()` is the canonical workspace selection algorithm. It returns `WorkspaceSelection` with selected workspace id, selection source, and discovered workspaces when auto-selected.
@@ -1,7 +1,7 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.10.0
4
- Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
3
+ Version: 0.11.0
4
+ Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
7
7
  Project-URL: Repository, https://github.com/hotdata-dev/sdk-python-framework
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
21
21
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
22
  Classifier: Typing :: Typed
23
23
  Requires-Python: >=3.10
24
- Requires-Dist: hotdata>=0.8.0
24
+ Requires-Dist: hotdata<0.9,>=0.8.0
25
25
  Requires-Dist: pandas>=2.0
26
26
  Requires-Dist: pyarrow>=14.0
27
27
  Description-Content-Type: text/markdown
@@ -30,16 +30,15 @@ Description-Content-Type: text/markdown
30
30
 
31
31
  **A Python framework for building Hotdata integrations.**
32
32
 
33
- Shared runtime primitives for Hotdata integrations: workspace/session semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
33
+ Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
34
34
 
35
35
  Runtime boundary and guarantees are defined in `CONTRACT.md`.
36
36
 
37
37
  ## Features
38
38
 
39
- - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, `HOTDATA_WORKSPACE`, and `HOTDATA_SANDBOX`.
39
+ - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
40
40
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
41
- - **Sandbox/session propagation** — pass sandbox session context through the SDK via `X-Session-Id`.
42
- - **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
41
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
43
42
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
44
43
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
45
44
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -2,16 +2,15 @@
2
2
 
3
3
  **A Python framework for building Hotdata integrations.**
4
4
 
5
- Shared runtime primitives for Hotdata integrations: workspace/session semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
5
+ Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
6
6
 
7
7
  Runtime boundary and guarantees are defined in `CONTRACT.md`.
8
8
 
9
9
  ## Features
10
10
 
11
- - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, `HOTDATA_WORKSPACE`, and `HOTDATA_SANDBOX`.
11
+ - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
12
12
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
13
- - **Sandbox/session propagation** — pass sandbox session context through the SDK via `X-Session-Id`.
14
- - **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
13
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
15
14
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
16
15
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
17
16
  - **History helpers** — list recent results and query run history with normalized dataclasses.
@@ -20,7 +20,6 @@ from hotdata_framework.env import (
20
20
  WorkspaceSelection,
21
21
  default_api_key,
22
22
  default_host,
23
- default_session_id,
24
23
  explicit_workspace_id,
25
24
  list_workspaces,
26
25
  normalize_host,
@@ -61,7 +60,6 @@ __all__ = [
61
60
  "classify_sdk_error",
62
61
  "default_api_key",
63
62
  "default_host",
64
- "default_session_id",
65
63
  "explicit_workspace_id",
66
64
  "from_env",
67
65
  "is_parquet_path",
@@ -49,11 +49,9 @@ from hotdata_framework.databases import (
49
49
  from hotdata_framework.env import (
50
50
  default_api_key,
51
51
  default_host,
52
- default_session_id,
53
52
  normalize_host,
54
53
  pick_workspace,
55
54
  )
56
- from hotdata_framework.http import default_http_retries
57
55
  from hotdata_framework.result import QueryResult
58
56
 
59
57
  # Load modes the managed-table endpoint accepts: replace overwrites, append adds
@@ -150,19 +148,21 @@ class HotdataClient:
150
148
  workspace_id: str,
151
149
  *,
152
150
  host: str | None = None,
153
- session_id: str | None = None,
154
151
  request_timeout: float | tuple[float, float] | None = None,
155
152
  ) -> None:
156
153
  self._host = normalize_host(host) if host else default_host()
157
154
  self._api_key = api_key
158
155
  self._workspace_id = workspace_id
159
- self._session_id = session_id
156
+ # No `retries=`: the generated SDK's own default is the correct policy
157
+ # and passing one here replaces it wholesale. `hotdata._retry` retries a
158
+ # pre-response connection reset on any method — the stale pooled socket
159
+ # case this wrapper was reaching for — while leaving read timeouts and
160
+ # status retries idempotent-only, so a POST that may have reached the
161
+ # server is never replayed.
160
162
  self._config = Configuration(
161
163
  host=self._host,
162
164
  api_key=api_key,
163
165
  workspace_id=workspace_id,
164
- session_id=session_id,
165
- retries=default_http_retries(),
166
166
  )
167
167
  self._api = ApiClient(self._config)
168
168
  if request_timeout is not None:
@@ -174,9 +174,8 @@ class HotdataClient:
174
174
  if not api_key:
175
175
  raise RuntimeError("HOTDATA_API_KEY must be set.")
176
176
  host = default_host()
177
- session = default_session_id()
178
- workspace_id = pick_workspace(api_key, host, session)
179
- return cls(api_key, workspace_id, host=host, session_id=session)
177
+ workspace_id = pick_workspace(api_key, host)
178
+ return cls(api_key, workspace_id, host=host)
180
179
 
181
180
  @property
182
181
  def workspace_id(self) -> str:
@@ -186,10 +185,6 @@ class HotdataClient:
186
185
  def host(self) -> str:
187
186
  return self._host
188
187
 
189
- @property
190
- def session_id(self) -> str | None:
191
- return self._session_id
192
-
193
188
  @property
194
189
  def api(self) -> ApiClient:
195
190
  return self._api
@@ -31,16 +31,11 @@ def default_host() -> str:
31
31
  return normalize_host(raw)
32
32
 
33
33
 
34
- def default_session_id() -> str | None:
35
- return os.environ.get("HOTDATA_SANDBOX")
36
-
37
-
38
- def list_workspaces(api_key: str, host: str, session_id: str | None):
34
+ def list_workspaces(api_key: str, host: str):
39
35
  cfg = Configuration(
40
36
  host=host,
41
37
  api_key=api_key,
42
38
  workspace_id=None,
43
- session_id=session_id,
44
39
  )
45
40
  with ApiClient(cfg) as api:
46
41
  listing = WorkspacesApi(api).list_workspaces()
@@ -54,9 +49,7 @@ class WorkspaceSelection:
54
49
  workspaces: list
55
50
 
56
51
 
57
- def resolve_workspace_selection(
58
- api_key: str, host: str, session_id: str | None
59
- ) -> WorkspaceSelection:
52
+ def resolve_workspace_selection(api_key: str, host: str) -> WorkspaceSelection:
60
53
  explicit = explicit_workspace_id()
61
54
  if explicit:
62
55
  return WorkspaceSelection(
@@ -64,7 +57,7 @@ def resolve_workspace_selection(
64
57
  source="explicit_env",
65
58
  workspaces=[],
66
59
  )
67
- workspaces = list_workspaces(api_key, host, session_id)
60
+ workspaces = list_workspaces(api_key, host)
68
61
  if not workspaces:
69
62
  raise RuntimeError("No Hotdata workspaces found for this API key.")
70
63
  active = [w for w in workspaces if w.active]
@@ -76,6 +69,6 @@ def resolve_workspace_selection(
76
69
  )
77
70
 
78
71
 
79
- def pick_workspace(api_key: str, host: str, session_id: str | None) -> str:
80
- selection = resolve_workspace_selection(api_key, host, session_id)
72
+ def pick_workspace(api_key: str, host: str) -> str:
73
+ selection = resolve_workspace_selection(api_key, host)
81
74
  return selection.workspace_id
@@ -18,8 +18,6 @@ def workspace_health_lines(client: HotdataClient) -> tuple[bool, list[str]]:
18
18
  f"**workspace** `{client.workspace_id}`",
19
19
  f"**connections** {n}",
20
20
  ]
21
- if client.session_id:
22
- lines.append(f"**sandbox** `{client.session_id}`")
23
21
  return True, lines
24
22
  except ApiException as e:
25
23
  return False, [e.reason or str(e)]
@@ -4,8 +4,8 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.10.0"
8
- description = "Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases"
7
+ version = "0.11.0"
8
+ description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
11
  license = { text = "MIT" }
@@ -26,8 +26,18 @@ classifiers = [
26
26
  "Typing :: Typed",
27
27
  ]
28
28
  dependencies = [
29
- # 0.7.0 adds `key` to table decls (create_managed_database(keys=) / add_managed_table(key=))
30
- "hotdata>=0.8.0",
29
+ # Floor: 0.8.0 carries the per-load `key` on load_managed_table — the merge key
30
+ # for delete/update/upsert loads, matched per load rather than declared at table
31
+ # creation. 0.7.0 is what first put `key` on the table decls themselves
32
+ # (create_managed_database(keys=) / add_managed_table(key=)).
33
+ #
34
+ # Capped to the current minor. This package is a hand-written wrapper over a
35
+ # GENERATED client, so an SDK minor can remove a model field or a Configuration
36
+ # keyword we pass, and there is no regeneration step here to catch it — an
37
+ # uncapped floor turns someone else's release into a break in ours,
38
+ # with no commit of our own to point at. Raise the cap deliberately, after
39
+ # running the suite against the new minor.
40
+ "hotdata>=0.8.0,<0.9",
31
41
  "pandas>=2.0",
32
42
  "pyarrow>=14.0",
33
43
  ]
@@ -7,6 +7,39 @@ cd "$ROOT"
7
7
  die() { echo "error: $*" >&2; exit 1; }
8
8
  need() { command -v "$1" >/dev/null 2>&1 || die "$1 is required"; }
9
9
 
10
+ # An interpreter that can `import tomllib`, which is stdlib only from 3.11.
11
+ #
12
+ # `python3` carries no version guarantee, and this package's own requires-python
13
+ # is >=3.10 — so hardcoding it meant the release script could not run on the
14
+ # oldest Python the package claims to support. No workflow runs this script; only
15
+ # a human cutting a release does, which is why it went unnoticed for so long. The
16
+ # symptom was a bare ModuleNotFoundError, which reads as a broken checkout rather
17
+ # than a too-old interpreter.
18
+ #
19
+ # `uv` is the fallback because this repo already builds and tests through it, so
20
+ # a machine that can run the suite can run the release.
21
+ resolve_python() {
22
+ if command -v python3 >/dev/null 2>&1 && python3 -c "import tomllib" >/dev/null 2>&1; then
23
+ echo python3
24
+ elif command -v uv >/dev/null 2>&1; then
25
+ # `>=3.11` not `3.12`: an exact request downloads a managed interpreter when
26
+ # the machine's uv-visible one is 3.11 or 3.13, and fails outright under
27
+ # UV_PYTHON_DOWNLOADS=never. It reads like a redirect but is not one —
28
+ # $PY_BIN is expanded unquoted for word splitting, and bash does not rescan
29
+ # expansion results for redirection operators, so uv receives it literally.
30
+ echo "uv run --no-project --python >=3.11 python"
31
+ else
32
+ die "need python3 >= 3.11 (for tomllib) or uv; python3 is $(command -v python3 >/dev/null 2>&1 && python3 -V 2>&1 || echo absent)"
33
+ fi
34
+ }
35
+
36
+ # Deferred, not resolved at load: the commands that never touch Python must keep
37
+ # working without a usable interpreter — otherwise the message telling you which
38
+ # interpreter you need is itself gated on having it, and `--help` breaks on
39
+ # exactly the machine this resolution exists for. Empty default satisfies `set -u`
40
+ # if a helper is ever reached by another path.
41
+ PY_BIN=""
42
+
10
43
  usage() {
11
44
  cat <<'EOF'
12
45
  Usage:
@@ -24,7 +57,7 @@ EOF
24
57
  }
25
58
 
26
59
  get_version() {
27
- python3 - <<'PY'
60
+ $PY_BIN - <<'PY'
28
61
  import tomllib
29
62
  from pathlib import Path
30
63
  print(tomllib.loads(Path("pyproject.toml").read_text())["project"]["version"])
@@ -32,7 +65,7 @@ PY
32
65
  }
33
66
 
34
67
  get_pkg_name() {
35
- python3 - <<'PY'
68
+ $PY_BIN - <<'PY'
36
69
  import tomllib
37
70
  from pathlib import Path
38
71
  print(tomllib.loads(Path("pyproject.toml").read_text())["project"]["name"])
@@ -41,7 +74,7 @@ PY
41
74
 
42
75
  set_version() {
43
76
  local ver="$1"
44
- python3 - "$ver" <<'PY'
77
+ $PY_BIN - "$ver" <<'PY'
45
78
  import re, sys
46
79
  from pathlib import Path
47
80
  ver = sys.argv[1]
@@ -56,7 +89,7 @@ PY
56
89
 
57
90
  bump_version() {
58
91
  local kind="$1" current="$2"
59
- python3 - "$kind" "$current" <<'PY'
92
+ $PY_BIN - "$kind" "$current" <<'PY'
60
93
  import re, sys
61
94
  kind, current = sys.argv[1], sys.argv[2]
62
95
  match = re.match(r"^(\d+)\.(\d+)\.(\d+)(.*)$", current)
@@ -95,14 +128,14 @@ update_changelog() {
95
128
  local ver="$1"
96
129
  local date
97
130
  date="$(date +%Y-%m-%d)"
98
- python3 scripts/update_changelog.py "$ver" "$date"
131
+ $PY_BIN scripts/update_changelog.py "$ver" "$date"
99
132
  }
100
133
 
101
134
  cmd_prepare() {
102
135
  local bump="${1:-}"
103
136
  [[ -n "$bump" ]] || { usage; die "missing bump kind or explicit version"; }
104
137
  need gh
105
- need python3
138
+ PY_BIN="$(resolve_python)"
106
139
  ensure_clean
107
140
 
108
141
  local current new base branch pkg
@@ -151,7 +184,7 @@ After merge, run \`./scripts/release.sh publish\` from a clean \`${base}\` check
151
184
 
152
185
  cmd_publish() {
153
186
  need gh
154
- need python3
187
+ PY_BIN="$(resolve_python)"
155
188
  ensure_clean
156
189
 
157
190
  local base ver tag
@@ -166,7 +199,7 @@ cmd_publish() {
166
199
 
167
200
  git rev-parse "$tag" >/dev/null 2>&1 && die "tag $tag already exists"
168
201
  [[ -f CHANGELOG.md ]] || die "CHANGELOG.md is required"
169
- python3 - "$ver" <<'PY'
202
+ $PY_BIN - "$ver" <<'PY'
170
203
  import re, sys
171
204
  from pathlib import Path
172
205
  ver = sys.argv[1]
@@ -4,6 +4,7 @@ from types import SimpleNamespace
4
4
  from unittest.mock import patch
5
5
 
6
6
  import pytest
7
+ from hotdata import Configuration
7
8
  from hotdata.exceptions import ForbiddenException
8
9
 
9
10
  from hotdata_framework.client import HotdataClient
@@ -153,7 +154,7 @@ def test_normalize_host(raw: str, expected: str):
153
154
 
154
155
  def test_pick_workspace_prefers_env(monkeypatch: pytest.MonkeyPatch):
155
156
  monkeypatch.setenv("HOTDATA_WORKSPACE", "ws_explicit")
156
- assert pick_workspace("k", "https://api.hotdata.dev", None) == "ws_explicit"
157
+ assert pick_workspace("k", "https://api.hotdata.dev") == "ws_explicit"
157
158
 
158
159
 
159
160
  def test_resolve_workspace_selection_prefers_env_without_listing(
@@ -161,7 +162,7 @@ def test_resolve_workspace_selection_prefers_env_without_listing(
161
162
  ):
162
163
  monkeypatch.setenv("HOTDATA_WORKSPACE", "ws_explicit")
163
164
  with patch("hotdata_framework.env.list_workspaces") as listing:
164
- resolved = resolve_workspace_selection("k", "https://api.hotdata.dev", None)
165
+ resolved = resolve_workspace_selection("k", "https://api.hotdata.dev")
165
166
  listing.assert_not_called()
166
167
  assert resolved.workspace_id == "ws_explicit"
167
168
  assert resolved.source == "explicit_env"
@@ -180,7 +181,7 @@ def test_pick_workspace_chooses_first_active(monkeypatch: pytest.MonkeyPatch):
180
181
 
181
182
  with patch("hotdata_framework.env.WorkspacesApi") as Api:
182
183
  Api.return_value.list_workspaces.return_value = listing
183
- assert pick_workspace("k", "https://api.hotdata.dev", None) == "ws_2"
184
+ assert pick_workspace("k", "https://api.hotdata.dev") == "ws_2"
184
185
 
185
186
 
186
187
  def test_pick_workspace_falls_back_to_first(monkeypatch: pytest.MonkeyPatch):
@@ -194,7 +195,7 @@ def test_pick_workspace_falls_back_to_first(monkeypatch: pytest.MonkeyPatch):
194
195
 
195
196
  with patch("hotdata_framework.env.WorkspacesApi") as Api:
196
197
  Api.return_value.list_workspaces.return_value = listing
197
- assert pick_workspace("k", "https://api.hotdata.dev", None) == "ws_1"
198
+ assert pick_workspace("k", "https://api.hotdata.dev") == "ws_1"
198
199
 
199
200
 
200
201
  def test_resolve_workspace_selection_source_first(monkeypatch: pytest.MonkeyPatch):
@@ -206,7 +207,7 @@ def test_resolve_workspace_selection_source_first(monkeypatch: pytest.MonkeyPatc
206
207
  listing = SimpleNamespace(workspaces=items)
207
208
  with patch("hotdata_framework.env.WorkspacesApi") as Api:
208
209
  Api.return_value.list_workspaces.return_value = listing
209
- resolved = resolve_workspace_selection("k", "https://api.hotdata.dev", None)
210
+ resolved = resolve_workspace_selection("k", "https://api.hotdata.dev")
210
211
  assert resolved.workspace_id == "ws_1"
211
212
  assert resolved.source == "first"
212
213
  assert resolved.workspaces == items
@@ -225,7 +226,7 @@ def test_resolve_workspace_selection_returns_workspaces_and_source(
225
226
 
226
227
  with patch("hotdata_framework.env.WorkspacesApi") as Api:
227
228
  Api.return_value.list_workspaces.return_value = listing
228
- resolved = resolve_workspace_selection("k", "https://api.hotdata.dev", None)
229
+ resolved = resolve_workspace_selection("k", "https://api.hotdata.dev")
229
230
  assert resolved.workspace_id == "ws_2"
230
231
  assert resolved.source == "active"
231
232
  assert resolved.workspaces == items
@@ -447,3 +448,99 @@ def test_list_run_history_returns_normalized_items():
447
448
  assert out[0].execution_time_ms == 7
448
449
  assert out[0].to_dict()["result_id"] == "res_1"
449
450
  assert fake_runs.kwargs == {"limit": 5}
451
+
452
+
453
+ # ---------------------------------------------------------------------------
454
+ # Session removal — asserted at the wire, not at the signature
455
+ #
456
+ # A signature-shaped check ("does __init__ accept session_id?") is satisfied by
457
+ # any unrecognised keyword and says nothing about what goes on the request. The
458
+ # feature was proven revivable from the ENVIRONMENT with the whole suite green:
459
+ # read HOTDATA_SANDBOX, pass it to Configuration, and the X-Session-Id header is
460
+ # back on every call with no signature change to notice. So these assert on
461
+ # `Configuration.api_keys`, which is where the generated client actually decides
462
+ # to send the header, and they set HOTDATA_SANDBOX so an environment-sourced
463
+ # revival has something to find.
464
+
465
+
466
+ def _sandbox_set(monkeypatch: pytest.MonkeyPatch) -> str:
467
+ value = "sb_should_reach_nothing"
468
+ monkeypatch.setenv("HOTDATA_SANDBOX", value)
469
+ return value
470
+
471
+
472
+ def test_client_registers_no_session_header(monkeypatch: pytest.MonkeyPatch):
473
+ """The removal's observable contract: no X-Session-Id on anything this client
474
+ sends. `api_keys` is the SDK's own record of which security schemes it will
475
+ attach, so an empty SessionId slot is the header being absent."""
476
+ _sandbox_set(monkeypatch)
477
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
478
+ assert "SessionId" not in client.api.configuration.api_keys
479
+ assert client.api.configuration.api_keys == {"WorkspaceId": "ws"}
480
+
481
+
482
+ def test_client_rejects_session_id_by_name(monkeypatch: pytest.MonkeyPatch):
483
+ """`match=` matters: without it the assertion passes for ANY unknown keyword,
484
+ making it a typo detector rather than a statement about `session_id`."""
485
+ _sandbox_set(monkeypatch)
486
+ with pytest.raises(TypeError, match="session_id"):
487
+ HotdataClient("k", "ws", host="https://api.hotdata.dev", session_id="sb_x")
488
+ # The property is gone too — the CHANGELOG says so, and an adapter reading it
489
+ # should get AttributeError rather than a stale value.
490
+ assert not hasattr(HotdataClient("k", "ws", host="https://api.hotdata.dev"), "session_id")
491
+
492
+
493
+ def test_workspace_listing_registers_no_session_header(monkeypatch: pytest.MonkeyPatch):
494
+ """`list_workspaces` builds its OWN Configuration, so it is a second place the
495
+ header can come back — and it is the pre-auth call, made before any workspace
496
+ is known."""
497
+ _clear_workspace_env(monkeypatch)
498
+ _sandbox_set(monkeypatch)
499
+ seen: list[Configuration] = []
500
+
501
+ def spy(*args: object, **kwargs: object) -> Configuration:
502
+ # Assert on the CONFIG, not on the kwargs. `session_id=` is only one way
503
+ # back: `Configuration(api_keys={"SessionId": ...})` is a documented
504
+ # escape hatch that re-attaches the header with no such keyword, and a
505
+ # kwargs-only check waves it through.
506
+ cfg = Configuration(*args, **kwargs)
507
+ seen.append(cfg)
508
+ return cfg
509
+
510
+ listing = SimpleNamespace(workspaces=[SimpleNamespace(public_id="ws_1", active=True)])
511
+ with patch("hotdata_framework.env.Configuration", spy), patch(
512
+ "hotdata_framework.env.WorkspacesApi"
513
+ ) as Api:
514
+ Api.return_value.list_workspaces.return_value = listing
515
+ assert pick_workspace("k", "https://api.hotdata.dev") == "ws_1"
516
+ assert seen, "list_workspaces built no Configuration — the spy never fired"
517
+ for cfg in seen:
518
+ assert "SessionId" not in cfg.api_keys
519
+
520
+
521
+ def test_from_env_builds_a_client_without_a_session(monkeypatch: pytest.MonkeyPatch):
522
+ """`from_env` is the entry point every adapter uses, and its body had no
523
+ coverage at all: the module-level `from_env` test patches this classmethod
524
+ out. Leaving a stale 3-argument `pick_workspace(...)` call here broke it
525
+ unconditionally while the suite stayed green."""
526
+ _clear_workspace_env(monkeypatch)
527
+ _sandbox_set(monkeypatch)
528
+ monkeypatch.setenv("HOTDATA_API_KEY", "k_env")
529
+ monkeypatch.setenv("HOTDATA_API_URL", "https://api.hotdata.dev")
530
+
531
+ with patch("hotdata_framework.client.pick_workspace") as picked:
532
+ picked.return_value = "ws_from_env"
533
+ client = HotdataClient.from_env()
534
+
535
+ # Two arguments, positionally — the signature this change trimmed.
536
+ assert picked.call_args.args == ("k_env", "https://api.hotdata.dev")
537
+ assert picked.call_args.kwargs == {}
538
+ assert client.workspace_id == "ws_from_env"
539
+ assert client.host == "https://api.hotdata.dev"
540
+ assert "SessionId" not in client.api.configuration.api_keys
541
+
542
+
543
+ def test_from_env_requires_an_api_key(monkeypatch: pytest.MonkeyPatch):
544
+ monkeypatch.delenv("HOTDATA_API_KEY", raising=False)
545
+ with pytest.raises(RuntimeError, match="HOTDATA_API_KEY"):
546
+ HotdataClient.from_env()
@@ -28,7 +28,6 @@ def test_public_exports_contract():
28
28
  "classify_sdk_error",
29
29
  "default_api_key",
30
30
  "default_host",
31
- "default_session_id",
32
31
  "explicit_workspace_id",
33
32
  "from_env",
34
33
  "is_parquet_path",
@@ -2,6 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  from unittest.mock import patch
4
4
 
5
+ import pytest
5
6
  from hotdata.exceptions import ApiException
6
7
 
7
8
  from hotdata_framework.client import HotdataClient
@@ -22,8 +23,19 @@ def test_workspace_health_ok():
22
23
  assert any("reachable" in p for p in parts)
23
24
 
24
25
 
25
- def test_workspace_health_ok_includes_sandbox_when_session_set():
26
- client = HotdataClient("k", "ws", host="https://api.hotdata.dev", session_id="sb_test")
26
+ def test_workspace_health_no_longer_reports_a_session(monkeypatch: pytest.MonkeyPatch):
27
+ """Health output must not mention a session even when the environment offers
28
+ one.
29
+
30
+ HOTDATA_SANDBOX is set deliberately, and that is the whole point. An earlier
31
+ version of this test built a client with no session at all — so on a revert
32
+ that restored the property and the `if client.session_id:` line, the guard was
33
+ simply false, no `sandbox` line was emitted, and the assertion passed with the
34
+ regression fully present. Setting the variable is what makes the assertion
35
+ reachable, because a revival of this feature would read it from here.
36
+ """
37
+ monkeypatch.setenv("HOTDATA_SANDBOX", "sb_should_reach_nothing")
38
+ client = HotdataClient("k", "ws", host="https://api.hotdata.dev")
27
39
  listing = type("L", (), {"connections": [object()]})()
28
40
 
29
41
  class FakeConnectionsApi:
@@ -33,7 +45,8 @@ def test_workspace_health_ok_includes_sandbox_when_session_set():
33
45
  with patch.object(client, "connections", return_value=FakeConnectionsApi()):
34
46
  ok, parts = workspace_health_lines(client)
35
47
  assert ok is True
36
- assert any("sandbox" in p and "sb_test" in p for p in parts)
48
+ assert "sb_should_reach_nothing" not in " ".join(parts)
49
+ assert not any("sandbox" in p for p in parts)
37
50
 
38
51
 
39
52
  def test_workspace_health_api_error():
@@ -0,0 +1,63 @@
1
+ """A POST must never be replayed because of a response status.
2
+
3
+ A load is not idempotent: re-sending one that the server is still working on
4
+ collides with the write lock the first attempt holds, and the duplicate is
5
+ refused. The generated SDK already draws this line — ``hotdata._retry`` retries
6
+ a *pre-response* connection reset on any method (the stale pooled socket case,
7
+ where the server did no work) while leaving read timeouts and status retries
8
+ idempotent-only. This wrapper used to pass its own ``retries=`` into
9
+ ``Configuration``, which replaced that policy wholesale with one that listed
10
+ POST alongside a 502/503/504 forcelist — so a gateway timing out a long request
11
+ produced a silent duplicate of it.
12
+
13
+ These tests pin the resulting policy rather than the absence of an argument,
14
+ so re-introducing an override that is unsafe for POST fails here.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from hotdata_framework.client import HotdataClient
20
+
21
+ WS = "work_test0000000000000000000000"
22
+
23
+
24
+ def _retry_policy():
25
+ client = HotdataClient(api_key="hd_test", workspace_id=WS, host="https://example.invalid")
26
+ return client._config.retries
27
+
28
+
29
+ def test_no_status_forcelist() -> None:
30
+ """A status code means the request reached the server, so retrying one is a
31
+ decision about idempotency that the transport layer cannot make."""
32
+ retry = _retry_policy()
33
+ assert not retry.status_forcelist, retry.status_forcelist
34
+ assert retry.status == 0
35
+
36
+
37
+ def test_post_is_not_read_or_status_retried() -> None:
38
+ """``allowed_methods`` gates both read and status retries. POST outside it is
39
+ what stops a request the server may already be processing from being sent
40
+ twice."""
41
+ retry = _retry_policy()
42
+ assert "POST" not in retry.allowed_methods
43
+ assert "GET" in retry.allowed_methods
44
+
45
+
46
+ def test_pre_response_connection_reset_still_retries_any_method() -> None:
47
+ """The case the wrapper's own override was reaching for, kept: a reset before
48
+ any response means the request never landed, so a POST retry is safe."""
49
+ from urllib3.exceptions import ProtocolError
50
+
51
+ # `_is_connection_error` is urllib3-private, and urllib3 reaches this package
52
+ # only as an unpinned transitive dependency — so an upstream rename breaks
53
+ # this test for reasons unrelated to this repo. Accepted knowingly: it is the
54
+ # only hook that separates a reset from any other ProtocolError without
55
+ # standing up a live socket.
56
+ retry = _retry_policy()
57
+ cause = ConnectionResetError(54, "Connection reset by peer")
58
+ reset = ProtocolError("Connection aborted.", cause)
59
+ assert retry._is_connection_error(reset)
60
+ # A ProtocolError with no connection-level cause is NOT reclassified: only a
61
+ # reset proves the request never landed. Read timeouts never reach this branch
62
+ # at all — urllib3 raises those as ReadTimeoutError, and they stay method-gated.
63
+ assert not retry._is_connection_error(ProtocolError("Connection aborted.", None))
@@ -101,7 +101,7 @@ wheels = [
101
101
 
102
102
  [[package]]
103
103
  name = "hotdata-framework"
104
- version = "0.10.0"
104
+ version = "0.11.0"
105
105
  source = { editable = "." }
106
106
  dependencies = [
107
107
  { name = "hotdata" },
@@ -120,7 +120,7 @@ dev = [
120
120
 
121
121
  [package.metadata]
122
122
  requires-dist = [
123
- { name = "hotdata", specifier = ">=0.8.0" },
123
+ { name = "hotdata", specifier = ">=0.8.0,<0.9" },
124
124
  { name = "pandas", specifier = ">=2.0" },
125
125
  { name = "pyarrow", specifier = ">=14.0" },
126
126
  ]
@@ -1,17 +0,0 @@
1
- """HTTP client defaults for Hotdata SDK :class:`~hotdata.Configuration`."""
2
-
3
- from __future__ import annotations
4
-
5
- from urllib3.util.retry import Retry
6
-
7
-
8
- def default_http_retries() -> Retry:
9
- """Retry transient connection failures (e.g. stale pooled sockets)."""
10
- return Retry(
11
- total=3,
12
- connect=3,
13
- read=3,
14
- backoff_factor=0.2,
15
- status_forcelist=(502, 503, 504),
16
- allowed_methods=frozenset(["GET", "HEAD", "POST", "PUT", "DELETE", "OPTIONS", "PATCH"]),
17
- )