hotdata-framework 0.9.0__tar.gz → 0.11.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/CHANGELOG.md +66 -0
  2. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/CONTRACT.md +4 -4
  3. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/PKG-INFO +8 -8
  4. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/README.md +4 -4
  5. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/__init__.py +2 -2
  6. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/client.py +255 -14
  7. hotdata_framework-0.11.0/hotdata_framework/databases.py +114 -0
  8. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/env.py +5 -12
  9. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/health.py +0 -2
  10. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/pyproject.toml +14 -4
  11. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/release.sh +41 -8
  12. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_client.py +103 -6
  13. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_contract.py +1 -1
  14. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_health.py +16 -3
  15. hotdata_framework-0.11.0/tests/test_indexes.py +824 -0
  16. hotdata_framework-0.11.0/tests/test_retry_policy.py +63 -0
  17. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/uv.lock +2 -2
  18. hotdata_framework-0.9.0/hotdata_framework/databases.py +0 -66
  19. hotdata_framework-0.9.0/hotdata_framework/http.py +0 -17
  20. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/CODEOWNERS +0 -0
  21. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/dependabot.yml +0 -0
  22. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/check-release.yml +0 -0
  23. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/ci.yml +0 -0
  24. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/dependabot-automerge.yml +0 -0
  25. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/publish.yml +0 -0
  26. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/release.yml +0 -0
  27. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.gitignore +0 -0
  28. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/RELEASING.md +0 -0
  29. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/examples/basic_usage.py +0 -0
  30. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/errors.py +0 -0
  31. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/managed_client.py +0 -0
  32. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/py.typed +0 -0
  33. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/result.py +0 -0
  34. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/check-release.py +0 -0
  35. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/extract-changelog.py +0 -0
  36. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/publish-workflow.sh +0 -0
  37. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/update_changelog.py +0 -0
  38. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_databases.py +0 -0
  39. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_errors.py +0 -0
  40. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_managed_client.py +0 -0
  41. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_request_timeout.py +0 -0
  42. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_result.py +0 -0
  43. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_update_changelog.py +0 -0
  44. {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_version.py +0 -0
@@ -7,6 +7,72 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.11.0] - 2026-08-11
11
+
12
+ ### Changed
13
+
14
+ - Cap the `hotdata` dependency to the current minor (`>=0.8.0,<0.9`). This
15
+ package wraps a *generated* client, so an SDK minor can remove a model field
16
+ or a `Configuration` keyword this wrapper passes, and there is no regeneration
17
+ step here to surface it — an uncapped floor turns an SDK release into a break
18
+ in this package, in versions already published. Raise the cap deliberately
19
+ after running the suite against the new minor.
20
+
21
+ ### Removed
22
+
23
+ - **Breaking:** session/sandbox support is gone. `HotdataClient` no longer accepts
24
+ `session_id=`, `HotdataClient.session_id` is removed, `default_session_id()` and
25
+ the `HOTDATA_SANDBOX` read are gone, and `list_workspaces()`,
26
+ `resolve_workspace_selection()` and `pick_workspace()` lose their `session_id`
27
+ parameter — note that loss is **positional**, so a three-argument call raises an
28
+ arity `TypeError` rather than an unexpected-keyword one.
29
+ `workspace_health_lines()` no longer emits a `sandbox` line.
30
+
31
+ **Why now.** The server stopped enforcing session scoping some time ago, so the
32
+ value already reached nothing. What makes removal urgent rather than tidy is
33
+ that the SDK is dropping the `SessionId` security scheme: against that release
34
+ `Configuration(session_id=...)` raises `TypeError` instead of setting a header,
35
+ and this package passed it unconditionally — so every `HotdataClient(...)`
36
+ would fail at construction. This package still pins `hotdata<0.9`, so nothing
37
+ is broken today; the change is what lets the cap be raised later without a
38
+ second breaking release.
39
+
40
+ **Migrating.** Drop `session_id=` from `HotdataClient(...)`, stop reading
41
+ `client.session_id`, stop setting `HOTDATA_SANDBOX`, and pass two arguments to
42
+ the workspace helpers. Adapters that re-export session context in their own
43
+ signatures — a `session_id=` parameter, a `session_id` metadata key — need to
44
+ remove it from theirs too, which makes their own release breaking in turn.
45
+
46
+ - `hotdata_framework.http` and `default_http_retries()`. The module existed only
47
+ to build the `retries=` policy removed under Fixed below, and had no other
48
+ callers. It predates `hotdata._retry`, which supersedes it.
49
+
50
+ ### Fixed
51
+
52
+ - A `POST` is no longer replayed because of a response status. `HotdataClient`
53
+ passed its own `retries=` into `Configuration`, which replaced the generated
54
+ SDK's policy wholesale with one listing `POST` in `allowed_methods` alongside
55
+ a `(502, 503, 504)` forcelist — so an intermediary timing out a long request
56
+ produced a silent, identical re-`POST` while the server was still working on
57
+ the first one. For a load that is not idempotent: the duplicate collides with
58
+ the write lock the original holds and is refused.
59
+
60
+ The override is removed and the SDK's own default now applies. It is the
61
+ policy this wrapper was reaching for — `hotdata._retry` retries a
62
+ *pre-response* connection reset (the stale pooled socket case, where the
63
+ server did no work) on any method, while leaving read timeouts and status
64
+ retries idempotent-only.
65
+
66
+ ## [0.10.0] - 2026-08-07
67
+
68
+ ### Added
69
+
70
+ - `create_index(database, table, columns=..., index_type=...)` builds a `bm25`,
71
+ `vector`, or `sorted` index on a managed table, matching `hotdata indexes create`
72
+ in the CLI. The build is a background job whose submit call reports success even
73
+ when the build later fails, so this polls the job and raises `RuntimeError` with
74
+ its error message; `wait=False` returns as soon as the job is accepted. Returns
75
+ `CreateIndexResult`, also exported.
10
76
 
11
77
  ## [0.9.0] - 2026-07-23
12
78
 
@@ -21,7 +21,6 @@ The supported import surface is:
21
21
  - `workspace_health_lines`
22
22
  - `default_api_key`
23
23
  - `default_host`
24
- - `default_session_id`
25
24
  - `explicit_workspace_id`
26
25
  - `list_workspaces`
27
26
  - `normalize_host`
@@ -33,6 +32,7 @@ The supported import surface is:
33
32
  - `ManagedDatabase`
34
33
  - `ManagedTable`
35
34
  - `LoadManagedTableResult`
35
+ - `CreateIndexResult`
36
36
  - `DEFAULT_SCHEMA`
37
37
  - `is_parquet_path`
38
38
 
@@ -42,7 +42,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
42
42
 
43
43
  ### `HotdataClient`
44
44
 
45
- - Represents runtime context: API key, host, workspace, optional session.
45
+ - Represents runtime context: API key, host, workspace.
46
46
  - `from_env()` resolves runtime context from env vars and selected workspace.
47
47
  - `execute_sql(sql)` returns `QueryResult` or raises `RuntimeError`/`TimeoutError`.
48
48
  - `get_result(result_id)` returns a ready `QueryResult` and waits for readiness when needed.
@@ -63,7 +63,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
63
63
  - `upload_parquet(path)` uploads a local parquet file and returns an upload id.
64
64
  - `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
65
65
  - `delete_managed_table(database, table, schema=...)` deletes a managed table.
66
- - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
66
+ - `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
67
+ - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
67
68
 
68
69
  ### `QueryResult`
69
70
 
@@ -77,7 +78,6 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
77
78
 
78
79
  - `default_api_key()` reads `HOTDATA_API_KEY`.
79
80
  - `default_host()` reads `HOTDATA_API_URL` (default: `https://api.hotdata.dev`) and normalizes it.
80
- - `default_session_id()` reads `HOTDATA_SANDBOX`.
81
81
  - `explicit_workspace_id()` reads `HOTDATA_WORKSPACE` (workspace public id).
82
82
  - `pick_workspace()` prefers explicit env workspace, then active workspace, then first workspace.
83
83
  - `resolve_workspace_selection()` is the canonical workspace selection algorithm. It returns `WorkspaceSelection` with selected workspace id, selection source, and discovered workspaces when auto-selected.
@@ -1,7 +1,7 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.9.0
4
- Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
3
+ Version: 0.11.0
4
+ Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
7
7
  Project-URL: Repository, https://github.com/hotdata-dev/sdk-python-framework
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
21
21
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
22
  Classifier: Typing :: Typed
23
23
  Requires-Python: >=3.10
24
- Requires-Dist: hotdata>=0.8.0
24
+ Requires-Dist: hotdata<0.9,>=0.8.0
25
25
  Requires-Dist: pandas>=2.0
26
26
  Requires-Dist: pyarrow>=14.0
27
27
  Description-Content-Type: text/markdown
@@ -30,20 +30,20 @@ Description-Content-Type: text/markdown
30
30
 
31
31
  **A Python framework for building Hotdata integrations.**
32
32
 
33
- Shared runtime primitives for Hotdata integrations: workspace/session semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
33
+ Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
34
34
 
35
35
  Runtime boundary and guarantees are defined in `CONTRACT.md`.
36
36
 
37
37
  ## Features
38
38
 
39
- - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, `HOTDATA_WORKSPACE`, and `HOTDATA_SANDBOX`.
39
+ - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
40
40
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
41
- - **Sandbox/session propagation** — pass sandbox session context through the SDK via `X-Session-Id`.
42
- - **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
41
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
43
42
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
44
43
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
45
44
  - **History helpers** — list recent results and query run history with normalized dataclasses.
46
45
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
46
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
47
47
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
48
48
 
49
49
  Install:
@@ -2,20 +2,20 @@
2
2
 
3
3
  **A Python framework for building Hotdata integrations.**
4
4
 
5
- Shared runtime primitives for Hotdata integrations: workspace/session semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
5
+ Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
6
6
 
7
7
  Runtime boundary and guarantees are defined in `CONTRACT.md`.
8
8
 
9
9
  ## Features
10
10
 
11
- - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, `HOTDATA_WORKSPACE`, and `HOTDATA_SANDBOX`.
11
+ - **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
12
12
  - **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
13
- - **Sandbox/session propagation** — pass sandbox session context through the SDK via `X-Session-Id`.
14
- - **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
13
+ - **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
15
14
  - **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
16
15
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
17
16
  - **History helpers** — list recent results and query run history with normalized dataclasses.
18
17
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
18
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
19
19
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
20
20
 
21
21
  Install:
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
10
10
  )
11
11
  from hotdata_framework.databases import (
12
12
  DEFAULT_SCHEMA,
13
+ CreateIndexResult,
13
14
  LoadManagedTableResult,
14
15
  ManagedDatabase,
15
16
  ManagedTable,
@@ -19,7 +20,6 @@ from hotdata_framework.env import (
19
20
  WorkspaceSelection,
20
21
  default_api_key,
21
22
  default_host,
22
- default_session_id,
23
23
  explicit_workspace_id,
24
24
  list_workspaces,
25
25
  normalize_host,
@@ -43,6 +43,7 @@ except PackageNotFoundError:
43
43
 
44
44
  __all__ = [
45
45
  "DEFAULT_SCHEMA",
46
+ "CreateIndexResult",
46
47
  "HotdataClient",
47
48
  "HotdataError",
48
49
  "HotdataTerminalError",
@@ -59,7 +60,6 @@ __all__ = [
59
60
  "classify_sdk_error",
60
61
  "default_api_key",
61
62
  "default_host",
62
- "default_session_id",
63
63
  "explicit_workspace_id",
64
64
  "from_env",
65
65
  "is_parquet_path",
@@ -4,12 +4,14 @@ import functools
4
4
  import time
5
5
  from collections.abc import Iterator
6
6
  from dataclasses import asdict, dataclass
7
- from typing import Any, Literal
7
+ from typing import Any, Literal, get_args
8
8
 
9
9
  from hotdata import ApiClient, Configuration
10
10
  from hotdata.api.connections_api import ConnectionsApi
11
11
  from hotdata.api.databases_api import DatabasesApi
12
+ from hotdata.api.indexes_api import IndexesApi
12
13
  from hotdata.api.information_schema_api import InformationSchemaApi
14
+ from hotdata.api.jobs_api import JobsApi
13
15
  from hotdata.api.query_api import QueryApi
14
16
  from hotdata.api.query_runs_api import QueryRunsApi
15
17
  from hotdata.api.results_api import ResultsApi
@@ -20,40 +22,60 @@ from hotdata.exceptions import ApiException
20
22
  from hotdata.models.add_managed_table_request import AddManagedTableRequest
21
23
  from hotdata.models.async_query_response import AsyncQueryResponse
22
24
  from hotdata.models.create_database_request import CreateDatabaseRequest
25
+ from hotdata.models.create_index_request import CreateIndexRequest
23
26
  from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
24
27
  from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
28
+ from hotdata.models.index_info_response import IndexInfoResponse
29
+ from hotdata.models.job_status_response import JobStatusResponse
25
30
  from hotdata.models.load_managed_table_request import LoadManagedTableRequest
26
31
  from hotdata.models.query_request import QueryRequest
27
32
  from hotdata.models.query_response import QueryResponse
33
+ from hotdata.models.submit_job_response import SubmitJobResponse
28
34
  from hotdata.models.table_info import TableInfo
29
35
  from urllib3.exceptions import HTTPError as Urllib3HTTPError
30
36
  from urllib3.exceptions import ProtocolError
31
37
 
32
38
  from hotdata_framework.databases import (
33
39
  DEFAULT_SCHEMA,
40
+ CreateIndexResult,
34
41
  LoadManagedTableResult,
35
42
  ManagedDatabase,
36
43
  ManagedTable,
37
44
  api_error_message,
45
+ enum_value,
38
46
  is_parquet_path,
39
47
  managed_database_from_detail,
40
48
  )
41
49
  from hotdata_framework.env import (
42
50
  default_api_key,
43
51
  default_host,
44
- default_session_id,
45
52
  normalize_host,
46
53
  pick_workspace,
47
54
  )
48
- from hotdata_framework.http import default_http_retries
49
55
  from hotdata_framework.result import QueryResult
50
56
 
51
57
  # Load modes the managed-table endpoint accepts: replace overwrites, append adds
52
58
  # rows, delete/update/upsert match by the table's declared key.
53
59
  ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
54
60
 
61
+ # Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
62
+ # "bm25" backs full-text search and "vector" backs nearest-neighbour search.
63
+ IndexType = Literal["sorted", "bm25", "vector"]
64
+
65
+ # Distance metrics a vector index can be built with. Each one accelerates
66
+ # exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
67
+ # dot -> negative_dot_product. A hand-written query naming a different function
68
+ # falls back to a full scan; the provider-backed vector_distance path resolves
69
+ # the function from the index instead, so it cannot mismatch.
70
+ VectorMetric = Literal["l2", "cosine", "dot"]
71
+
72
+ _INDEX_TYPES = frozenset(get_args(IndexType))
73
+ _VECTOR_METRICS = frozenset(get_args(VectorMetric))
74
+
55
75
  _TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
56
76
  _RESULT_FAILURE = frozenset({"failed", "cancelled"})
77
+ # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
78
+ _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
57
79
 
58
80
 
59
81
  @dataclass(frozen=True)
@@ -126,19 +148,21 @@ class HotdataClient:
126
148
  workspace_id: str,
127
149
  *,
128
150
  host: str | None = None,
129
- session_id: str | None = None,
130
151
  request_timeout: float | tuple[float, float] | None = None,
131
152
  ) -> None:
132
153
  self._host = normalize_host(host) if host else default_host()
133
154
  self._api_key = api_key
134
155
  self._workspace_id = workspace_id
135
- self._session_id = session_id
156
+ # No `retries=`: the generated SDK's own default is the correct policy
157
+ # and passing one here replaces it wholesale. `hotdata._retry` retries a
158
+ # pre-response connection reset on any method — the stale pooled socket
159
+ # case this wrapper was reaching for — while leaving read timeouts and
160
+ # status retries idempotent-only, so a POST that may have reached the
161
+ # server is never replayed.
136
162
  self._config = Configuration(
137
163
  host=self._host,
138
164
  api_key=api_key,
139
165
  workspace_id=workspace_id,
140
- session_id=session_id,
141
- retries=default_http_retries(),
142
166
  )
143
167
  self._api = ApiClient(self._config)
144
168
  if request_timeout is not None:
@@ -150,9 +174,8 @@ class HotdataClient:
150
174
  if not api_key:
151
175
  raise RuntimeError("HOTDATA_API_KEY must be set.")
152
176
  host = default_host()
153
- session = default_session_id()
154
- workspace_id = pick_workspace(api_key, host, session)
155
- return cls(api_key, workspace_id, host=host, session_id=session)
177
+ workspace_id = pick_workspace(api_key, host)
178
+ return cls(api_key, workspace_id, host=host)
156
179
 
157
180
  @property
158
181
  def workspace_id(self) -> str:
@@ -162,10 +185,6 @@ class HotdataClient:
162
185
  def host(self) -> str:
163
186
  return self._host
164
187
 
165
- @property
166
- def session_id(self) -> str | None:
167
- return self._session_id
168
-
169
188
  @property
170
189
  def api(self) -> ApiClient:
171
190
  return self._api
@@ -188,6 +207,12 @@ class HotdataClient:
188
207
  def _information_schema(self) -> InformationSchemaApi:
189
208
  return InformationSchemaApi(self._api)
190
209
 
210
+ def _indexes_api(self) -> IndexesApi:
211
+ return IndexesApi(self._api)
212
+
213
+ def _jobs_api(self) -> JobsApi:
214
+ return JobsApi(self._api)
215
+
191
216
  def _query_api(self) -> QueryApi:
192
217
  return QueryApi(self._api)
193
218
 
@@ -427,6 +452,199 @@ class HotdataClient:
427
452
  except ApiException as e:
428
453
  raise RuntimeError(api_error_message(e)) from e
429
454
 
455
+ def create_index(
456
+ self,
457
+ database: str | ManagedDatabase,
458
+ table: str,
459
+ *,
460
+ schema: str = DEFAULT_SCHEMA,
461
+ index_name: str | None = None,
462
+ columns: list[str],
463
+ index_type: IndexType,
464
+ metric: VectorMetric | None = None,
465
+ dimensions: int | None = None,
466
+ embedding_provider_id: str | None = None,
467
+ output_column: str | None = None,
468
+ description: str | None = None,
469
+ wait: bool = True,
470
+ timeout_s: float = 300.0,
471
+ poll_interval_s: float = 2.0,
472
+ ) -> CreateIndexResult:
473
+ """Build an index on a managed table and wait for it to be ready.
474
+
475
+ The Python equivalent of ``hotdata indexes create``, scoped to managed
476
+ databases. Indexing a table on a plain (non-managed) connection is not
477
+ supported here; the CLI's ``--catalog`` flag covers that case.
478
+
479
+ ``index_type`` selects the index kind and is required: ``"bm25"`` for
480
+ full-text search (queries error outright without one), ``"vector"`` for
481
+ nearest-neighbour search (queries work without one, but only at
482
+ full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
483
+ to ``"sorted"``; this method makes the choice explicit instead, because
484
+ the wrong kind fails at query time rather than here.
485
+
486
+ ``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
487
+ derivation the CLI uses when ``--name`` is omitted, so both surfaces
488
+ name the same index identically.
489
+
490
+ There are two kinds of vector index, and they are queried differently:
491
+
492
+ * **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
493
+ vector column (a float list), and a query passes a literal vector:
494
+ ``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
495
+ distance function the caller writes — ``cosine`` serves
496
+ ``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
497
+ ``negative_dot_product``. A mismatch is not an error: the query
498
+ silently reverts to a full table scan. Omitting ``metric`` lets the
499
+ server choose (``l2`` for float-array columns), so pass it explicitly
500
+ whenever the query function is known.
501
+ * **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
502
+ provider ``sys_emb_openai``). ``columns`` is then the *source text*
503
+ column; the provider embeds it into ``output_column`` (default
504
+ ``{column}_embedding``) and the index is built over that. A query
505
+ passes text, not a vector — ``vector_distance(source_col, 'query')`` —
506
+ and the server resolves the matching distance function from the index
507
+ itself, so the metric-mismatch trap above does not apply. The returned
508
+ ``source_column`` names the column to query.
509
+
510
+ ``dimensions`` picks the output width for providers that support several;
511
+ it does not apply when indexing an existing vector column, whose width is
512
+ read from the data. ``description`` is a user-facing label for the
513
+ embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
514
+ index takes exactly one column, and every option in this paragraph — plus
515
+ ``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
516
+
517
+ The server builds the index as a background job. This method polls that
518
+ job to a terminal state and raises ``RuntimeError`` if it failed, because
519
+ the submit call itself reports success for builds that later fail. Pass
520
+ ``wait=False`` to return as soon as the job is accepted — the result then
521
+ carries ``status="pending"`` and a ``job_id``, and the caller owns
522
+ checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
523
+
524
+ Raises ``ValueError`` for an unusable argument combination,
525
+ ``RuntimeError`` if the API rejects the request or the build fails, and
526
+ ``TimeoutError`` if the build is still running after ``timeout_s``.
527
+ """
528
+ if not columns:
529
+ raise ValueError("create_index requires at least one column")
530
+ if index_type not in _INDEX_TYPES:
531
+ allowed = ", ".join(sorted(_INDEX_TYPES))
532
+ raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
533
+ if index_type != "vector":
534
+ vector_only = {
535
+ "metric": metric,
536
+ "dimensions": dimensions,
537
+ "embedding_provider_id": embedding_provider_id,
538
+ "output_column": output_column,
539
+ "description": description,
540
+ }
541
+ supplied = sorted(k for k, v in vector_only.items() if v is not None)
542
+ if supplied:
543
+ raise ValueError(
544
+ f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
545
+ f"vector indexes only (index_type={index_type!r})"
546
+ )
547
+ else:
548
+ if len(columns) != 1:
549
+ raise ValueError(
550
+ f"a vector index takes exactly one column (got {len(columns)}); "
551
+ "the engine indexes only the first"
552
+ )
553
+ if metric is not None and metric not in _VECTOR_METRICS:
554
+ allowed = ", ".join(sorted(_VECTOR_METRICS))
555
+ raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
556
+
557
+ # Matches the CLI's derivation so both surfaces name the same index
558
+ # identically: `hotdata indexes create` without --name.
559
+ resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
560
+
561
+ db = self._as_managed_database(database)
562
+ request = CreateIndexRequest(
563
+ index_name=resolved_name,
564
+ columns=list(columns),
565
+ index_type=index_type,
566
+ metric=metric,
567
+ dimensions=dimensions,
568
+ embedding_provider_id=embedding_provider_id,
569
+ output_column=output_column,
570
+ description=description,
571
+ var_async=True,
572
+ )
573
+ try:
574
+ submitted = self._indexes_api().create_index(
575
+ db.default_connection_id,
576
+ schema,
577
+ table,
578
+ request,
579
+ )
580
+ except ApiException as e:
581
+ raise RuntimeError(api_error_message(e)) from e
582
+
583
+ full_name = f"{db.id}.{schema}.{table}"
584
+
585
+ # A build the server finished inline answers 201 with the index itself;
586
+ # the async path answers 202 with a job to poll.
587
+ if isinstance(submitted, IndexInfoResponse):
588
+ return self._index_result(submitted, full_name, schema, table, job_id=None)
589
+
590
+ if not isinstance(submitted, SubmitJobResponse):
591
+ raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
592
+
593
+ job_id = submitted.id
594
+
595
+ def requested_result(status: str) -> CreateIndexResult:
596
+ """Echo the requested values, for the paths where the server hands
597
+ back a job rather than the built index."""
598
+ return CreateIndexResult(
599
+ full_name=full_name,
600
+ schema_name=schema,
601
+ table_name=table,
602
+ index_name=resolved_name,
603
+ index_type=index_type,
604
+ columns=list(columns),
605
+ metric=metric,
606
+ source_column=columns[0] if embedding_provider_id else None,
607
+ status=status,
608
+ job_id=job_id,
609
+ )
610
+
611
+ if not wait:
612
+ return requested_result(enum_value(submitted.status))
613
+
614
+ job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
615
+ status = enum_value(job.status)
616
+ if status != "succeeded":
617
+ detail = job.error_message or f"Index build {status}"
618
+ raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
619
+
620
+ # `result` is a oneOf wrapper today; tolerate the model arriving directly.
621
+ built = getattr(job.result, "actual_instance", job.result)
622
+ if isinstance(built, IndexInfoResponse):
623
+ return self._index_result(built, full_name, schema, table, job_id=job_id)
624
+ return requested_result("ready")
625
+
626
+ @staticmethod
627
+ def _index_result(
628
+ info: IndexInfoResponse,
629
+ full_name: str,
630
+ schema: str,
631
+ table: str,
632
+ *,
633
+ job_id: str | None,
634
+ ) -> CreateIndexResult:
635
+ return CreateIndexResult(
636
+ full_name=full_name,
637
+ schema_name=schema,
638
+ table_name=table,
639
+ index_name=info.index_name,
640
+ index_type=info.index_type,
641
+ columns=list(info.columns),
642
+ metric=info.metric,
643
+ source_column=info.source_column,
644
+ status=enum_value(info.status),
645
+ job_id=job_id,
646
+ )
647
+
430
648
  def list_recent_results(
431
649
  self,
432
650
  *,
@@ -560,6 +778,29 @@ class HotdataClient:
560
778
  f"(last status: {getattr(last, 'status', None)})"
561
779
  )
562
780
 
781
+ def _poll_job(
782
+ self,
783
+ job_id: str,
784
+ *,
785
+ timeout_s: float = 300.0,
786
+ interval_s: float = 2.0,
787
+ ) -> JobStatusResponse:
788
+ jobs = self._jobs_api()
789
+ deadline = time.monotonic() + timeout_s
790
+ last: JobStatusResponse | None = None
791
+ while time.monotonic() < deadline:
792
+ try:
793
+ last = jobs.get_job(job_id)
794
+ except ApiException as e:
795
+ raise RuntimeError(api_error_message(e)) from e
796
+ if last.status in _JOB_TERMINAL:
797
+ return last
798
+ time.sleep(interval_s)
799
+ last_status = enum_value(last.status) if last is not None else None
800
+ raise TimeoutError(
801
+ f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
802
+ )
803
+
563
804
  def _wait_result_ready(
564
805
  self,
565
806
  result_id: str,