hotdata-framework 0.9.0__tar.gz → 0.11.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/CHANGELOG.md +66 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/CONTRACT.md +4 -4
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/PKG-INFO +8 -8
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/README.md +4 -4
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/__init__.py +2 -2
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/client.py +255 -14
- hotdata_framework-0.11.0/hotdata_framework/databases.py +114 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/env.py +5 -12
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/health.py +0 -2
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/pyproject.toml +14 -4
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/release.sh +41 -8
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_client.py +103 -6
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_contract.py +1 -1
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_health.py +16 -3
- hotdata_framework-0.11.0/tests/test_indexes.py +824 -0
- hotdata_framework-0.11.0/tests/test_retry_policy.py +63 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/uv.lock +2 -2
- hotdata_framework-0.9.0/hotdata_framework/databases.py +0 -66
- hotdata_framework-0.9.0/hotdata_framework/http.py +0 -17
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/publish.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.github/workflows/release.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/.gitignore +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/RELEASING.md +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/errors.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/managed_client.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_managed_client.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.11.0}/tests/test_version.py +0 -0
|
@@ -7,6 +7,72 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.11.0] - 2026-08-11
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Cap the `hotdata` dependency to the current minor (`>=0.8.0,<0.9`). This
|
|
15
|
+
package wraps a *generated* client, so an SDK minor can remove a model field
|
|
16
|
+
or a `Configuration` keyword this wrapper passes, and there is no regeneration
|
|
17
|
+
step here to surface it — an uncapped floor turns an SDK release into a break
|
|
18
|
+
in this package, in versions already published. Raise the cap deliberately
|
|
19
|
+
after running the suite against the new minor.
|
|
20
|
+
|
|
21
|
+
### Removed
|
|
22
|
+
|
|
23
|
+
- **Breaking:** session/sandbox support is gone. `HotdataClient` no longer accepts
|
|
24
|
+
`session_id=`, `HotdataClient.session_id` is removed, `default_session_id()` and
|
|
25
|
+
the `HOTDATA_SANDBOX` read are gone, and `list_workspaces()`,
|
|
26
|
+
`resolve_workspace_selection()` and `pick_workspace()` lose their `session_id`
|
|
27
|
+
parameter — note that loss is **positional**, so a three-argument call raises an
|
|
28
|
+
arity `TypeError` rather than an unexpected-keyword one.
|
|
29
|
+
`workspace_health_lines()` no longer emits a `sandbox` line.
|
|
30
|
+
|
|
31
|
+
**Why now.** The server stopped enforcing session scoping some time ago, so the
|
|
32
|
+
value already reached nothing. What makes removal urgent rather than tidy is
|
|
33
|
+
that the SDK is dropping the `SessionId` security scheme: against that release
|
|
34
|
+
`Configuration(session_id=...)` raises `TypeError` instead of setting a header,
|
|
35
|
+
and this package passed it unconditionally — so every `HotdataClient(...)`
|
|
36
|
+
would fail at construction. This package still pins `hotdata<0.9`, so nothing
|
|
37
|
+
is broken today; the change is what lets the cap be raised later without a
|
|
38
|
+
second breaking release.
|
|
39
|
+
|
|
40
|
+
**Migrating.** Drop `session_id=` from `HotdataClient(...)`, stop reading
|
|
41
|
+
`client.session_id`, stop setting `HOTDATA_SANDBOX`, and pass two arguments to
|
|
42
|
+
the workspace helpers. Adapters that re-export session context in their own
|
|
43
|
+
signatures — a `session_id=` parameter, a `session_id` metadata key — need to
|
|
44
|
+
remove it from theirs too, which makes their own release breaking in turn.
|
|
45
|
+
|
|
46
|
+
- `hotdata_framework.http` and `default_http_retries()`. The module existed only
|
|
47
|
+
to build the `retries=` policy removed under Fixed below, and had no other
|
|
48
|
+
callers. It predates `hotdata._retry`, which supersedes it.
|
|
49
|
+
|
|
50
|
+
### Fixed
|
|
51
|
+
|
|
52
|
+
- A `POST` is no longer replayed because of a response status. `HotdataClient`
|
|
53
|
+
passed its own `retries=` into `Configuration`, which replaced the generated
|
|
54
|
+
SDK's policy wholesale with one listing `POST` in `allowed_methods` alongside
|
|
55
|
+
a `(502, 503, 504)` forcelist — so an intermediary timing out a long request
|
|
56
|
+
produced a silent, identical re-`POST` while the server was still working on
|
|
57
|
+
the first one. For a load that is not idempotent: the duplicate collides with
|
|
58
|
+
the write lock the original holds and is refused.
|
|
59
|
+
|
|
60
|
+
The override is removed and the SDK's own default now applies. It is the
|
|
61
|
+
policy this wrapper was reaching for — `hotdata._retry` retries a
|
|
62
|
+
*pre-response* connection reset (the stale pooled socket case, where the
|
|
63
|
+
server did no work) on any method, while leaving read timeouts and status
|
|
64
|
+
retries idempotent-only.
|
|
65
|
+
|
|
66
|
+
## [0.10.0] - 2026-08-07
|
|
67
|
+
|
|
68
|
+
### Added
|
|
69
|
+
|
|
70
|
+
- `create_index(database, table, columns=..., index_type=...)` builds a `bm25`,
|
|
71
|
+
`vector`, or `sorted` index on a managed table, matching `hotdata indexes create`
|
|
72
|
+
in the CLI. The build is a background job whose submit call reports success even
|
|
73
|
+
when the build later fails, so this polls the job and raises `RuntimeError` with
|
|
74
|
+
its error message; `wait=False` returns as soon as the job is accepted. Returns
|
|
75
|
+
`CreateIndexResult`, also exported.
|
|
10
76
|
|
|
11
77
|
## [0.9.0] - 2026-07-23
|
|
12
78
|
|
|
@@ -21,7 +21,6 @@ The supported import surface is:
|
|
|
21
21
|
- `workspace_health_lines`
|
|
22
22
|
- `default_api_key`
|
|
23
23
|
- `default_host`
|
|
24
|
-
- `default_session_id`
|
|
25
24
|
- `explicit_workspace_id`
|
|
26
25
|
- `list_workspaces`
|
|
27
26
|
- `normalize_host`
|
|
@@ -33,6 +32,7 @@ The supported import surface is:
|
|
|
33
32
|
- `ManagedDatabase`
|
|
34
33
|
- `ManagedTable`
|
|
35
34
|
- `LoadManagedTableResult`
|
|
35
|
+
- `CreateIndexResult`
|
|
36
36
|
- `DEFAULT_SCHEMA`
|
|
37
37
|
- `is_parquet_path`
|
|
38
38
|
|
|
@@ -42,7 +42,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
42
42
|
|
|
43
43
|
### `HotdataClient`
|
|
44
44
|
|
|
45
|
-
- Represents runtime context: API key, host, workspace
|
|
45
|
+
- Represents runtime context: API key, host, workspace.
|
|
46
46
|
- `from_env()` resolves runtime context from env vars and selected workspace.
|
|
47
47
|
- `execute_sql(sql)` returns `QueryResult` or raises `RuntimeError`/`TimeoutError`.
|
|
48
48
|
- `get_result(result_id)` returns a ready `QueryResult` and waits for readiness when needed.
|
|
@@ -63,7 +63,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
63
63
|
- `upload_parquet(path)` uploads a local parquet file and returns an upload id.
|
|
64
64
|
- `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
|
|
65
65
|
- `delete_managed_table(database, table, schema=...)` deletes a managed table.
|
|
66
|
-
-
|
|
66
|
+
- `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
|
|
67
|
+
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
67
68
|
|
|
68
69
|
### `QueryResult`
|
|
69
70
|
|
|
@@ -77,7 +78,6 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
77
78
|
|
|
78
79
|
- `default_api_key()` reads `HOTDATA_API_KEY`.
|
|
79
80
|
- `default_host()` reads `HOTDATA_API_URL` (default: `https://api.hotdata.dev`) and normalizes it.
|
|
80
|
-
- `default_session_id()` reads `HOTDATA_SANDBOX`.
|
|
81
81
|
- `explicit_workspace_id()` reads `HOTDATA_WORKSPACE` (workspace public id).
|
|
82
82
|
- `pick_workspace()` prefers explicit env workspace, then active workspace, then first workspace.
|
|
83
83
|
- `resolve_workspace_selection()` is the canonical workspace selection algorithm. It returns `WorkspaceSelection` with selected workspace id, selection source, and discovered workspaces when auto-selected.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Python framework for building Hotdata integrations: workspace
|
|
3
|
+
Version: 0.11.0
|
|
4
|
+
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
7
7
|
Project-URL: Repository, https://github.com/hotdata-dev/sdk-python-framework
|
|
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
|
21
21
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
22
|
Classifier: Typing :: Typed
|
|
23
23
|
Requires-Python: >=3.10
|
|
24
|
-
Requires-Dist: hotdata
|
|
24
|
+
Requires-Dist: hotdata<0.9,>=0.8.0
|
|
25
25
|
Requires-Dist: pandas>=2.0
|
|
26
26
|
Requires-Dist: pyarrow>=14.0
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
@@ -30,20 +30,20 @@ Description-Content-Type: text/markdown
|
|
|
30
30
|
|
|
31
31
|
**A Python framework for building Hotdata integrations.**
|
|
32
32
|
|
|
33
|
-
Shared runtime primitives for Hotdata integrations: workspace
|
|
33
|
+
Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
|
|
34
34
|
|
|
35
35
|
Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
36
36
|
|
|
37
37
|
## Features
|
|
38
38
|
|
|
39
|
-
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`,
|
|
39
|
+
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
40
40
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
41
|
-
- **
|
|
42
|
-
- **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
|
|
41
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
|
|
43
42
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
44
43
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
45
44
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
46
45
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
46
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
47
47
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
48
48
|
|
|
49
49
|
Install:
|
|
@@ -2,20 +2,20 @@
|
|
|
2
2
|
|
|
3
3
|
**A Python framework for building Hotdata integrations.**
|
|
4
4
|
|
|
5
|
-
Shared runtime primitives for Hotdata integrations: workspace
|
|
5
|
+
Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
|
|
6
6
|
|
|
7
7
|
Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
|
-
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`,
|
|
11
|
+
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
12
12
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
13
|
-
- **
|
|
14
|
-
- **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
|
|
13
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
|
|
15
14
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
16
15
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
17
16
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
18
17
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
18
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
19
19
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
20
20
|
|
|
21
21
|
Install:
|
|
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
|
|
|
10
10
|
)
|
|
11
11
|
from hotdata_framework.databases import (
|
|
12
12
|
DEFAULT_SCHEMA,
|
|
13
|
+
CreateIndexResult,
|
|
13
14
|
LoadManagedTableResult,
|
|
14
15
|
ManagedDatabase,
|
|
15
16
|
ManagedTable,
|
|
@@ -19,7 +20,6 @@ from hotdata_framework.env import (
|
|
|
19
20
|
WorkspaceSelection,
|
|
20
21
|
default_api_key,
|
|
21
22
|
default_host,
|
|
22
|
-
default_session_id,
|
|
23
23
|
explicit_workspace_id,
|
|
24
24
|
list_workspaces,
|
|
25
25
|
normalize_host,
|
|
@@ -43,6 +43,7 @@ except PackageNotFoundError:
|
|
|
43
43
|
|
|
44
44
|
__all__ = [
|
|
45
45
|
"DEFAULT_SCHEMA",
|
|
46
|
+
"CreateIndexResult",
|
|
46
47
|
"HotdataClient",
|
|
47
48
|
"HotdataError",
|
|
48
49
|
"HotdataTerminalError",
|
|
@@ -59,7 +60,6 @@ __all__ = [
|
|
|
59
60
|
"classify_sdk_error",
|
|
60
61
|
"default_api_key",
|
|
61
62
|
"default_host",
|
|
62
|
-
"default_session_id",
|
|
63
63
|
"explicit_workspace_id",
|
|
64
64
|
"from_env",
|
|
65
65
|
"is_parquet_path",
|
|
@@ -4,12 +4,14 @@ import functools
|
|
|
4
4
|
import time
|
|
5
5
|
from collections.abc import Iterator
|
|
6
6
|
from dataclasses import asdict, dataclass
|
|
7
|
-
from typing import Any, Literal
|
|
7
|
+
from typing import Any, Literal, get_args
|
|
8
8
|
|
|
9
9
|
from hotdata import ApiClient, Configuration
|
|
10
10
|
from hotdata.api.connections_api import ConnectionsApi
|
|
11
11
|
from hotdata.api.databases_api import DatabasesApi
|
|
12
|
+
from hotdata.api.indexes_api import IndexesApi
|
|
12
13
|
from hotdata.api.information_schema_api import InformationSchemaApi
|
|
14
|
+
from hotdata.api.jobs_api import JobsApi
|
|
13
15
|
from hotdata.api.query_api import QueryApi
|
|
14
16
|
from hotdata.api.query_runs_api import QueryRunsApi
|
|
15
17
|
from hotdata.api.results_api import ResultsApi
|
|
@@ -20,40 +22,60 @@ from hotdata.exceptions import ApiException
|
|
|
20
22
|
from hotdata.models.add_managed_table_request import AddManagedTableRequest
|
|
21
23
|
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
22
24
|
from hotdata.models.create_database_request import CreateDatabaseRequest
|
|
25
|
+
from hotdata.models.create_index_request import CreateIndexRequest
|
|
23
26
|
from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
|
|
24
27
|
from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
|
|
28
|
+
from hotdata.models.index_info_response import IndexInfoResponse
|
|
29
|
+
from hotdata.models.job_status_response import JobStatusResponse
|
|
25
30
|
from hotdata.models.load_managed_table_request import LoadManagedTableRequest
|
|
26
31
|
from hotdata.models.query_request import QueryRequest
|
|
27
32
|
from hotdata.models.query_response import QueryResponse
|
|
33
|
+
from hotdata.models.submit_job_response import SubmitJobResponse
|
|
28
34
|
from hotdata.models.table_info import TableInfo
|
|
29
35
|
from urllib3.exceptions import HTTPError as Urllib3HTTPError
|
|
30
36
|
from urllib3.exceptions import ProtocolError
|
|
31
37
|
|
|
32
38
|
from hotdata_framework.databases import (
|
|
33
39
|
DEFAULT_SCHEMA,
|
|
40
|
+
CreateIndexResult,
|
|
34
41
|
LoadManagedTableResult,
|
|
35
42
|
ManagedDatabase,
|
|
36
43
|
ManagedTable,
|
|
37
44
|
api_error_message,
|
|
45
|
+
enum_value,
|
|
38
46
|
is_parquet_path,
|
|
39
47
|
managed_database_from_detail,
|
|
40
48
|
)
|
|
41
49
|
from hotdata_framework.env import (
|
|
42
50
|
default_api_key,
|
|
43
51
|
default_host,
|
|
44
|
-
default_session_id,
|
|
45
52
|
normalize_host,
|
|
46
53
|
pick_workspace,
|
|
47
54
|
)
|
|
48
|
-
from hotdata_framework.http import default_http_retries
|
|
49
55
|
from hotdata_framework.result import QueryResult
|
|
50
56
|
|
|
51
57
|
# Load modes the managed-table endpoint accepts: replace overwrites, append adds
|
|
52
58
|
# rows, delete/update/upsert match by the table's declared key.
|
|
53
59
|
ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
|
|
54
60
|
|
|
61
|
+
# Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
|
|
62
|
+
# "bm25" backs full-text search and "vector" backs nearest-neighbour search.
|
|
63
|
+
IndexType = Literal["sorted", "bm25", "vector"]
|
|
64
|
+
|
|
65
|
+
# Distance metrics a vector index can be built with. Each one accelerates
|
|
66
|
+
# exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
|
|
67
|
+
# dot -> negative_dot_product. A hand-written query naming a different function
|
|
68
|
+
# falls back to a full scan; the provider-backed vector_distance path resolves
|
|
69
|
+
# the function from the index instead, so it cannot mismatch.
|
|
70
|
+
VectorMetric = Literal["l2", "cosine", "dot"]
|
|
71
|
+
|
|
72
|
+
_INDEX_TYPES = frozenset(get_args(IndexType))
|
|
73
|
+
_VECTOR_METRICS = frozenset(get_args(VectorMetric))
|
|
74
|
+
|
|
55
75
|
_TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
|
|
56
76
|
_RESULT_FAILURE = frozenset({"failed", "cancelled"})
|
|
77
|
+
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
78
|
+
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
57
79
|
|
|
58
80
|
|
|
59
81
|
@dataclass(frozen=True)
|
|
@@ -126,19 +148,21 @@ class HotdataClient:
|
|
|
126
148
|
workspace_id: str,
|
|
127
149
|
*,
|
|
128
150
|
host: str | None = None,
|
|
129
|
-
session_id: str | None = None,
|
|
130
151
|
request_timeout: float | tuple[float, float] | None = None,
|
|
131
152
|
) -> None:
|
|
132
153
|
self._host = normalize_host(host) if host else default_host()
|
|
133
154
|
self._api_key = api_key
|
|
134
155
|
self._workspace_id = workspace_id
|
|
135
|
-
|
|
156
|
+
# No `retries=`: the generated SDK's own default is the correct policy
|
|
157
|
+
# and passing one here replaces it wholesale. `hotdata._retry` retries a
|
|
158
|
+
# pre-response connection reset on any method — the stale pooled socket
|
|
159
|
+
# case this wrapper was reaching for — while leaving read timeouts and
|
|
160
|
+
# status retries idempotent-only, so a POST that may have reached the
|
|
161
|
+
# server is never replayed.
|
|
136
162
|
self._config = Configuration(
|
|
137
163
|
host=self._host,
|
|
138
164
|
api_key=api_key,
|
|
139
165
|
workspace_id=workspace_id,
|
|
140
|
-
session_id=session_id,
|
|
141
|
-
retries=default_http_retries(),
|
|
142
166
|
)
|
|
143
167
|
self._api = ApiClient(self._config)
|
|
144
168
|
if request_timeout is not None:
|
|
@@ -150,9 +174,8 @@ class HotdataClient:
|
|
|
150
174
|
if not api_key:
|
|
151
175
|
raise RuntimeError("HOTDATA_API_KEY must be set.")
|
|
152
176
|
host = default_host()
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
return cls(api_key, workspace_id, host=host, session_id=session)
|
|
177
|
+
workspace_id = pick_workspace(api_key, host)
|
|
178
|
+
return cls(api_key, workspace_id, host=host)
|
|
156
179
|
|
|
157
180
|
@property
|
|
158
181
|
def workspace_id(self) -> str:
|
|
@@ -162,10 +185,6 @@ class HotdataClient:
|
|
|
162
185
|
def host(self) -> str:
|
|
163
186
|
return self._host
|
|
164
187
|
|
|
165
|
-
@property
|
|
166
|
-
def session_id(self) -> str | None:
|
|
167
|
-
return self._session_id
|
|
168
|
-
|
|
169
188
|
@property
|
|
170
189
|
def api(self) -> ApiClient:
|
|
171
190
|
return self._api
|
|
@@ -188,6 +207,12 @@ class HotdataClient:
|
|
|
188
207
|
def _information_schema(self) -> InformationSchemaApi:
|
|
189
208
|
return InformationSchemaApi(self._api)
|
|
190
209
|
|
|
210
|
+
def _indexes_api(self) -> IndexesApi:
|
|
211
|
+
return IndexesApi(self._api)
|
|
212
|
+
|
|
213
|
+
def _jobs_api(self) -> JobsApi:
|
|
214
|
+
return JobsApi(self._api)
|
|
215
|
+
|
|
191
216
|
def _query_api(self) -> QueryApi:
|
|
192
217
|
return QueryApi(self._api)
|
|
193
218
|
|
|
@@ -427,6 +452,199 @@ class HotdataClient:
|
|
|
427
452
|
except ApiException as e:
|
|
428
453
|
raise RuntimeError(api_error_message(e)) from e
|
|
429
454
|
|
|
455
|
+
def create_index(
|
|
456
|
+
self,
|
|
457
|
+
database: str | ManagedDatabase,
|
|
458
|
+
table: str,
|
|
459
|
+
*,
|
|
460
|
+
schema: str = DEFAULT_SCHEMA,
|
|
461
|
+
index_name: str | None = None,
|
|
462
|
+
columns: list[str],
|
|
463
|
+
index_type: IndexType,
|
|
464
|
+
metric: VectorMetric | None = None,
|
|
465
|
+
dimensions: int | None = None,
|
|
466
|
+
embedding_provider_id: str | None = None,
|
|
467
|
+
output_column: str | None = None,
|
|
468
|
+
description: str | None = None,
|
|
469
|
+
wait: bool = True,
|
|
470
|
+
timeout_s: float = 300.0,
|
|
471
|
+
poll_interval_s: float = 2.0,
|
|
472
|
+
) -> CreateIndexResult:
|
|
473
|
+
"""Build an index on a managed table and wait for it to be ready.
|
|
474
|
+
|
|
475
|
+
The Python equivalent of ``hotdata indexes create``, scoped to managed
|
|
476
|
+
databases. Indexing a table on a plain (non-managed) connection is not
|
|
477
|
+
supported here; the CLI's ``--catalog`` flag covers that case.
|
|
478
|
+
|
|
479
|
+
``index_type`` selects the index kind and is required: ``"bm25"`` for
|
|
480
|
+
full-text search (queries error outright without one), ``"vector"`` for
|
|
481
|
+
nearest-neighbour search (queries work without one, but only at
|
|
482
|
+
full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
|
|
483
|
+
to ``"sorted"``; this method makes the choice explicit instead, because
|
|
484
|
+
the wrong kind fails at query time rather than here.
|
|
485
|
+
|
|
486
|
+
``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
|
|
487
|
+
derivation the CLI uses when ``--name`` is omitted, so both surfaces
|
|
488
|
+
name the same index identically.
|
|
489
|
+
|
|
490
|
+
There are two kinds of vector index, and they are queried differently:
|
|
491
|
+
|
|
492
|
+
* **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
|
|
493
|
+
vector column (a float list), and a query passes a literal vector:
|
|
494
|
+
``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
|
|
495
|
+
distance function the caller writes — ``cosine`` serves
|
|
496
|
+
``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
|
|
497
|
+
``negative_dot_product``. A mismatch is not an error: the query
|
|
498
|
+
silently reverts to a full table scan. Omitting ``metric`` lets the
|
|
499
|
+
server choose (``l2`` for float-array columns), so pass it explicitly
|
|
500
|
+
whenever the query function is known.
|
|
501
|
+
* **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
|
|
502
|
+
provider ``sys_emb_openai``). ``columns`` is then the *source text*
|
|
503
|
+
column; the provider embeds it into ``output_column`` (default
|
|
504
|
+
``{column}_embedding``) and the index is built over that. A query
|
|
505
|
+
passes text, not a vector — ``vector_distance(source_col, 'query')`` —
|
|
506
|
+
and the server resolves the matching distance function from the index
|
|
507
|
+
itself, so the metric-mismatch trap above does not apply. The returned
|
|
508
|
+
``source_column`` names the column to query.
|
|
509
|
+
|
|
510
|
+
``dimensions`` picks the output width for providers that support several;
|
|
511
|
+
it does not apply when indexing an existing vector column, whose width is
|
|
512
|
+
read from the data. ``description`` is a user-facing label for the
|
|
513
|
+
embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
|
|
514
|
+
index takes exactly one column, and every option in this paragraph — plus
|
|
515
|
+
``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
|
|
516
|
+
|
|
517
|
+
The server builds the index as a background job. This method polls that
|
|
518
|
+
job to a terminal state and raises ``RuntimeError`` if it failed, because
|
|
519
|
+
the submit call itself reports success for builds that later fail. Pass
|
|
520
|
+
``wait=False`` to return as soon as the job is accepted — the result then
|
|
521
|
+
carries ``status="pending"`` and a ``job_id``, and the caller owns
|
|
522
|
+
checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
|
|
523
|
+
|
|
524
|
+
Raises ``ValueError`` for an unusable argument combination,
|
|
525
|
+
``RuntimeError`` if the API rejects the request or the build fails, and
|
|
526
|
+
``TimeoutError`` if the build is still running after ``timeout_s``.
|
|
527
|
+
"""
|
|
528
|
+
if not columns:
|
|
529
|
+
raise ValueError("create_index requires at least one column")
|
|
530
|
+
if index_type not in _INDEX_TYPES:
|
|
531
|
+
allowed = ", ".join(sorted(_INDEX_TYPES))
|
|
532
|
+
raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
|
|
533
|
+
if index_type != "vector":
|
|
534
|
+
vector_only = {
|
|
535
|
+
"metric": metric,
|
|
536
|
+
"dimensions": dimensions,
|
|
537
|
+
"embedding_provider_id": embedding_provider_id,
|
|
538
|
+
"output_column": output_column,
|
|
539
|
+
"description": description,
|
|
540
|
+
}
|
|
541
|
+
supplied = sorted(k for k, v in vector_only.items() if v is not None)
|
|
542
|
+
if supplied:
|
|
543
|
+
raise ValueError(
|
|
544
|
+
f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
|
|
545
|
+
f"vector indexes only (index_type={index_type!r})"
|
|
546
|
+
)
|
|
547
|
+
else:
|
|
548
|
+
if len(columns) != 1:
|
|
549
|
+
raise ValueError(
|
|
550
|
+
f"a vector index takes exactly one column (got {len(columns)}); "
|
|
551
|
+
"the engine indexes only the first"
|
|
552
|
+
)
|
|
553
|
+
if metric is not None and metric not in _VECTOR_METRICS:
|
|
554
|
+
allowed = ", ".join(sorted(_VECTOR_METRICS))
|
|
555
|
+
raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
|
|
556
|
+
|
|
557
|
+
# Matches the CLI's derivation so both surfaces name the same index
|
|
558
|
+
# identically: `hotdata indexes create` without --name.
|
|
559
|
+
resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
|
|
560
|
+
|
|
561
|
+
db = self._as_managed_database(database)
|
|
562
|
+
request = CreateIndexRequest(
|
|
563
|
+
index_name=resolved_name,
|
|
564
|
+
columns=list(columns),
|
|
565
|
+
index_type=index_type,
|
|
566
|
+
metric=metric,
|
|
567
|
+
dimensions=dimensions,
|
|
568
|
+
embedding_provider_id=embedding_provider_id,
|
|
569
|
+
output_column=output_column,
|
|
570
|
+
description=description,
|
|
571
|
+
var_async=True,
|
|
572
|
+
)
|
|
573
|
+
try:
|
|
574
|
+
submitted = self._indexes_api().create_index(
|
|
575
|
+
db.default_connection_id,
|
|
576
|
+
schema,
|
|
577
|
+
table,
|
|
578
|
+
request,
|
|
579
|
+
)
|
|
580
|
+
except ApiException as e:
|
|
581
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
582
|
+
|
|
583
|
+
full_name = f"{db.id}.{schema}.{table}"
|
|
584
|
+
|
|
585
|
+
# A build the server finished inline answers 201 with the index itself;
|
|
586
|
+
# the async path answers 202 with a job to poll.
|
|
587
|
+
if isinstance(submitted, IndexInfoResponse):
|
|
588
|
+
return self._index_result(submitted, full_name, schema, table, job_id=None)
|
|
589
|
+
|
|
590
|
+
if not isinstance(submitted, SubmitJobResponse):
|
|
591
|
+
raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
|
|
592
|
+
|
|
593
|
+
job_id = submitted.id
|
|
594
|
+
|
|
595
|
+
def requested_result(status: str) -> CreateIndexResult:
|
|
596
|
+
"""Echo the requested values, for the paths where the server hands
|
|
597
|
+
back a job rather than the built index."""
|
|
598
|
+
return CreateIndexResult(
|
|
599
|
+
full_name=full_name,
|
|
600
|
+
schema_name=schema,
|
|
601
|
+
table_name=table,
|
|
602
|
+
index_name=resolved_name,
|
|
603
|
+
index_type=index_type,
|
|
604
|
+
columns=list(columns),
|
|
605
|
+
metric=metric,
|
|
606
|
+
source_column=columns[0] if embedding_provider_id else None,
|
|
607
|
+
status=status,
|
|
608
|
+
job_id=job_id,
|
|
609
|
+
)
|
|
610
|
+
|
|
611
|
+
if not wait:
|
|
612
|
+
return requested_result(enum_value(submitted.status))
|
|
613
|
+
|
|
614
|
+
job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
|
|
615
|
+
status = enum_value(job.status)
|
|
616
|
+
if status != "succeeded":
|
|
617
|
+
detail = job.error_message or f"Index build {status}"
|
|
618
|
+
raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
|
|
619
|
+
|
|
620
|
+
# `result` is a oneOf wrapper today; tolerate the model arriving directly.
|
|
621
|
+
built = getattr(job.result, "actual_instance", job.result)
|
|
622
|
+
if isinstance(built, IndexInfoResponse):
|
|
623
|
+
return self._index_result(built, full_name, schema, table, job_id=job_id)
|
|
624
|
+
return requested_result("ready")
|
|
625
|
+
|
|
626
|
+
@staticmethod
|
|
627
|
+
def _index_result(
|
|
628
|
+
info: IndexInfoResponse,
|
|
629
|
+
full_name: str,
|
|
630
|
+
schema: str,
|
|
631
|
+
table: str,
|
|
632
|
+
*,
|
|
633
|
+
job_id: str | None,
|
|
634
|
+
) -> CreateIndexResult:
|
|
635
|
+
return CreateIndexResult(
|
|
636
|
+
full_name=full_name,
|
|
637
|
+
schema_name=schema,
|
|
638
|
+
table_name=table,
|
|
639
|
+
index_name=info.index_name,
|
|
640
|
+
index_type=info.index_type,
|
|
641
|
+
columns=list(info.columns),
|
|
642
|
+
metric=info.metric,
|
|
643
|
+
source_column=info.source_column,
|
|
644
|
+
status=enum_value(info.status),
|
|
645
|
+
job_id=job_id,
|
|
646
|
+
)
|
|
647
|
+
|
|
430
648
|
def list_recent_results(
|
|
431
649
|
self,
|
|
432
650
|
*,
|
|
@@ -560,6 +778,29 @@ class HotdataClient:
|
|
|
560
778
|
f"(last status: {getattr(last, 'status', None)})"
|
|
561
779
|
)
|
|
562
780
|
|
|
781
|
+
def _poll_job(
|
|
782
|
+
self,
|
|
783
|
+
job_id: str,
|
|
784
|
+
*,
|
|
785
|
+
timeout_s: float = 300.0,
|
|
786
|
+
interval_s: float = 2.0,
|
|
787
|
+
) -> JobStatusResponse:
|
|
788
|
+
jobs = self._jobs_api()
|
|
789
|
+
deadline = time.monotonic() + timeout_s
|
|
790
|
+
last: JobStatusResponse | None = None
|
|
791
|
+
while time.monotonic() < deadline:
|
|
792
|
+
try:
|
|
793
|
+
last = jobs.get_job(job_id)
|
|
794
|
+
except ApiException as e:
|
|
795
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
796
|
+
if last.status in _JOB_TERMINAL:
|
|
797
|
+
return last
|
|
798
|
+
time.sleep(interval_s)
|
|
799
|
+
last_status = enum_value(last.status) if last is not None else None
|
|
800
|
+
raise TimeoutError(
|
|
801
|
+
f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
|
|
802
|
+
)
|
|
803
|
+
|
|
563
804
|
def _wait_result_ready(
|
|
564
805
|
self,
|
|
565
806
|
result_id: str,
|