hotdata-framework 0.8.0__tar.gz → 0.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hotdata_framework-0.10.0/.github/CODEOWNERS +1 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/CHANGELOG.md +68 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/CONTRACT.md +5 -2
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/PKG-INFO +2 -1
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/README.md +1 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/__init__.py +2 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/client.py +277 -15
- hotdata_framework-0.10.0/hotdata_framework/databases.py +114 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/pyproject.toml +1 -1
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_client.py +123 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_contract.py +1 -0
- hotdata_framework-0.10.0/tests/test_indexes.py +824 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/uv.lock +1 -1
- hotdata_framework-0.8.0/hotdata_framework/databases.py +0 -66
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/publish.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/release.yml +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.gitignore +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/RELEASING.md +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/errors.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/http.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/managed_client.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_managed_client.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_version.py +0 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
* @hotdata-dev/engineers
|
|
@@ -8,6 +8,74 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
## [0.10.0] - 2026-08-07
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- `create_index(database, table, columns=..., index_type=...)` builds an index on a
|
|
16
|
+
managed table, bringing the framework client to parity with `hotdata indexes
|
|
17
|
+
create` in the CLI. It covers all three index kinds the API accepts: `"bm25"` for
|
|
18
|
+
full-text search, `"vector"` for nearest-neighbour search, and `"sorted"`.
|
|
19
|
+
Previously the framework had no index API at all and callers had to drop to the raw
|
|
20
|
+
`hotdata.IndexesApi`, which left a managed database's data loaded but not
|
|
21
|
+
searchable: full-text queries error without an index, and vector queries run at
|
|
22
|
+
full-scan speed. Like the other managed-table operations, `database` accepts a
|
|
23
|
+
name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
|
|
24
|
+
(non-managed) connection is not covered — the CLI's `--catalog` handles that.
|
|
25
|
+
|
|
26
|
+
`index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
|
|
27
|
+
derivation the CLI uses when `--name` is omitted, so both surfaces name the same
|
|
28
|
+
index identically. `index_type` is required, unlike the API's `"sorted"` default,
|
|
29
|
+
because the wrong kind only fails at query time.
|
|
30
|
+
|
|
31
|
+
The server builds the index as a background job whose submit call reports success
|
|
32
|
+
even when the build later fails, so `create_index` polls the job to a terminal
|
|
33
|
+
state and raises `RuntimeError` carrying the job's `error_message`. Pass
|
|
34
|
+
`wait=False` to return once the job is accepted (`status="pending"` plus a
|
|
35
|
+
`job_id`) and own the outcome check yourself, as the CLI's `--async` does;
|
|
36
|
+
`timeout_s` and `poll_interval_s` tune the wait.
|
|
37
|
+
|
|
38
|
+
Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
|
|
39
|
+
existing vector column, queried with a literal vector — and there `metric` must
|
|
40
|
+
match the distance function the query uses (`cosine`→`cosine_distance`,
|
|
41
|
+
`l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
|
|
42
|
+
reverts to a full table scan rather than erroring. Setting
|
|
43
|
+
`embedding_provider_id` indexes a *text* column instead: the provider embeds it
|
|
44
|
+
into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
|
|
45
|
+
and the server resolves the distance function itself.
|
|
46
|
+
|
|
47
|
+
Argument combinations that the server would silently ignore raise `ValueError`
|
|
48
|
+
before any request is sent: an unknown `index_type` or `metric`, a vector index
|
|
49
|
+
with more than one column (the engine indexes only the first), and
|
|
50
|
+
`metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
|
|
51
|
+
non-vector index.
|
|
52
|
+
|
|
53
|
+
Verified against `api.hotdata.dev` when this version was released: BM25 and vector
|
|
54
|
+
indexes both build and report `ready`, and a BM25 index is used by full-text
|
|
55
|
+
search. A *vector* index on a managed database was **not** picked up by the query
|
|
56
|
+
planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
|
|
57
|
+
planned as a full scan. That reproduces with an index created by `hotdata indexes
|
|
58
|
+
create`, so it is an engine-side issue rather than a client one, but it means a
|
|
59
|
+
vector index built through this method may not yet accelerate queries.
|
|
60
|
+
|
|
61
|
+
- `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
|
|
62
|
+
`hotdata_framework` and added to the public contract surface. Its `source_column`
|
|
63
|
+
names the text column to query for a provider-backed vector index, and is `None`
|
|
64
|
+
for BM25, sorted, and plain vector indexes.
|
|
65
|
+
|
|
66
|
+
## [0.9.0] - 2026-07-23
|
|
67
|
+
|
|
68
|
+
### Added
|
|
69
|
+
|
|
70
|
+
- `list_managed_tables`, `load_managed_table`, `add_managed_table`,
|
|
71
|
+
`delete_managed_table`, `delete_managed_database`, and `execute_sql` accept an
|
|
72
|
+
already-resolved `ManagedDatabase` (as returned by `create_managed_database`)
|
|
73
|
+
in place of a name/id. When passed one, they skip the `get_database` /
|
|
74
|
+
`list_databases` read probe. This lets an API key scoped to create + load but
|
|
75
|
+
not read `/databases` bootstrap a managed database and load into it within a
|
|
76
|
+
single run: the caller holds the `ManagedDatabase` from `create` and drives
|
|
77
|
+
the load/add/query ops with zero reads. The name/id string path is unchanged.
|
|
78
|
+
|
|
11
79
|
## [0.8.0] - 2026-07-20
|
|
12
80
|
|
|
13
81
|
### Changed
|
|
@@ -33,6 +33,7 @@ The supported import surface is:
|
|
|
33
33
|
- `ManagedDatabase`
|
|
34
34
|
- `ManagedTable`
|
|
35
35
|
- `LoadManagedTableResult`
|
|
36
|
+
- `CreateIndexResult`
|
|
36
37
|
- `DEFAULT_SCHEMA`
|
|
37
38
|
- `is_parquet_path`
|
|
38
39
|
|
|
@@ -56,13 +57,15 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
56
57
|
adapters should pass `connection_id` when known.
|
|
57
58
|
- `uploads()` returns the uploads API wrapper for parquet staging.
|
|
58
59
|
- `list_managed_databases()` returns all databases via the `/databases` API.
|
|
59
|
-
- `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan).
|
|
60
|
-
- `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front.
|
|
60
|
+
- `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan). A `403` from `/databases` surfaces as `RuntimeError` (forbidden, not absent), preserving the underlying `ApiException` as `__cause__`.
|
|
61
|
+
- `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front. Returns a `ManagedDatabase` (id + `default_connection_id`) sufficient to load without a further read.
|
|
61
62
|
- `delete_managed_database(name_or_id)` deletes a database via the `/databases` API.
|
|
62
63
|
- `list_managed_tables(database, schema=...)` lists tables in a managed database.
|
|
63
64
|
- `upload_parquet(path)` uploads a local parquet file and returns an upload id.
|
|
64
65
|
- `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
|
|
65
66
|
- `delete_managed_table(database, table, schema=...)` deletes a managed table.
|
|
67
|
+
- `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
|
|
68
|
+
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
66
69
|
|
|
67
70
|
### `QueryResult`
|
|
68
71
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.10.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -44,6 +44,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
44
44
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
45
45
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
46
46
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
47
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
47
48
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
48
49
|
|
|
49
50
|
Install:
|
|
@@ -16,6 +16,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
16
16
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
17
17
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
18
18
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
19
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
19
20
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
20
21
|
|
|
21
22
|
Install:
|
|
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
|
|
|
10
10
|
)
|
|
11
11
|
from hotdata_framework.databases import (
|
|
12
12
|
DEFAULT_SCHEMA,
|
|
13
|
+
CreateIndexResult,
|
|
13
14
|
LoadManagedTableResult,
|
|
14
15
|
ManagedDatabase,
|
|
15
16
|
ManagedTable,
|
|
@@ -43,6 +44,7 @@ except PackageNotFoundError:
|
|
|
43
44
|
|
|
44
45
|
__all__ = [
|
|
45
46
|
"DEFAULT_SCHEMA",
|
|
47
|
+
"CreateIndexResult",
|
|
46
48
|
"HotdataClient",
|
|
47
49
|
"HotdataError",
|
|
48
50
|
"HotdataTerminalError",
|
|
@@ -4,12 +4,14 @@ import functools
|
|
|
4
4
|
import time
|
|
5
5
|
from collections.abc import Iterator
|
|
6
6
|
from dataclasses import asdict, dataclass
|
|
7
|
-
from typing import Any, Literal
|
|
7
|
+
from typing import Any, Literal, get_args
|
|
8
8
|
|
|
9
9
|
from hotdata import ApiClient, Configuration
|
|
10
10
|
from hotdata.api.connections_api import ConnectionsApi
|
|
11
11
|
from hotdata.api.databases_api import DatabasesApi
|
|
12
|
+
from hotdata.api.indexes_api import IndexesApi
|
|
12
13
|
from hotdata.api.information_schema_api import InformationSchemaApi
|
|
14
|
+
from hotdata.api.jobs_api import JobsApi
|
|
13
15
|
from hotdata.api.query_api import QueryApi
|
|
14
16
|
from hotdata.api.query_runs_api import QueryRunsApi
|
|
15
17
|
from hotdata.api.results_api import ResultsApi
|
|
@@ -20,21 +22,27 @@ from hotdata.exceptions import ApiException
|
|
|
20
22
|
from hotdata.models.add_managed_table_request import AddManagedTableRequest
|
|
21
23
|
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
22
24
|
from hotdata.models.create_database_request import CreateDatabaseRequest
|
|
25
|
+
from hotdata.models.create_index_request import CreateIndexRequest
|
|
23
26
|
from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
|
|
24
27
|
from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
|
|
28
|
+
from hotdata.models.index_info_response import IndexInfoResponse
|
|
29
|
+
from hotdata.models.job_status_response import JobStatusResponse
|
|
25
30
|
from hotdata.models.load_managed_table_request import LoadManagedTableRequest
|
|
26
31
|
from hotdata.models.query_request import QueryRequest
|
|
27
32
|
from hotdata.models.query_response import QueryResponse
|
|
33
|
+
from hotdata.models.submit_job_response import SubmitJobResponse
|
|
28
34
|
from hotdata.models.table_info import TableInfo
|
|
29
35
|
from urllib3.exceptions import HTTPError as Urllib3HTTPError
|
|
30
36
|
from urllib3.exceptions import ProtocolError
|
|
31
37
|
|
|
32
38
|
from hotdata_framework.databases import (
|
|
33
39
|
DEFAULT_SCHEMA,
|
|
40
|
+
CreateIndexResult,
|
|
34
41
|
LoadManagedTableResult,
|
|
35
42
|
ManagedDatabase,
|
|
36
43
|
ManagedTable,
|
|
37
44
|
api_error_message,
|
|
45
|
+
enum_value,
|
|
38
46
|
is_parquet_path,
|
|
39
47
|
managed_database_from_detail,
|
|
40
48
|
)
|
|
@@ -52,8 +60,24 @@ from hotdata_framework.result import QueryResult
|
|
|
52
60
|
# rows, delete/update/upsert match by the table's declared key.
|
|
53
61
|
ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
|
|
54
62
|
|
|
63
|
+
# Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
|
|
64
|
+
# "bm25" backs full-text search and "vector" backs nearest-neighbour search.
|
|
65
|
+
IndexType = Literal["sorted", "bm25", "vector"]
|
|
66
|
+
|
|
67
|
+
# Distance metrics a vector index can be built with. Each one accelerates
|
|
68
|
+
# exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
|
|
69
|
+
# dot -> negative_dot_product. A hand-written query naming a different function
|
|
70
|
+
# falls back to a full scan; the provider-backed vector_distance path resolves
|
|
71
|
+
# the function from the index instead, so it cannot mismatch.
|
|
72
|
+
VectorMetric = Literal["l2", "cosine", "dot"]
|
|
73
|
+
|
|
74
|
+
_INDEX_TYPES = frozenset(get_args(IndexType))
|
|
75
|
+
_VECTOR_METRICS = frozenset(get_args(VectorMetric))
|
|
76
|
+
|
|
55
77
|
_TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
|
|
56
78
|
_RESULT_FAILURE = frozenset({"failed", "cancelled"})
|
|
79
|
+
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
80
|
+
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
57
81
|
|
|
58
82
|
|
|
59
83
|
@dataclass(frozen=True)
|
|
@@ -188,6 +212,12 @@ class HotdataClient:
|
|
|
188
212
|
def _information_schema(self) -> InformationSchemaApi:
|
|
189
213
|
return InformationSchemaApi(self._api)
|
|
190
214
|
|
|
215
|
+
def _indexes_api(self) -> IndexesApi:
|
|
216
|
+
return IndexesApi(self._api)
|
|
217
|
+
|
|
218
|
+
def _jobs_api(self) -> JobsApi:
|
|
219
|
+
return JobsApi(self._api)
|
|
220
|
+
|
|
191
221
|
def _query_api(self) -> QueryApi:
|
|
192
222
|
return QueryApi(self._api)
|
|
193
223
|
|
|
@@ -241,6 +271,19 @@ class HotdataClient:
|
|
|
241
271
|
raise RuntimeError(api_error_message(e)) from e
|
|
242
272
|
return managed_database_from_detail(detail)
|
|
243
273
|
|
|
274
|
+
def _as_managed_database(self, database: str | ManagedDatabase) -> ManagedDatabase:
|
|
275
|
+
"""Return ``database`` as-is if it is already a resolved ``ManagedDatabase``,
|
|
276
|
+
otherwise resolve it by name or id.
|
|
277
|
+
|
|
278
|
+
Passing an already-resolved ``ManagedDatabase`` (e.g. the value returned by
|
|
279
|
+
:meth:`create_managed_database`) skips the id/name read probe, so callers
|
|
280
|
+
whose API key may create but not read ``/databases`` can drive loads without
|
|
281
|
+
a forbidden read.
|
|
282
|
+
"""
|
|
283
|
+
if isinstance(database, ManagedDatabase):
|
|
284
|
+
return database
|
|
285
|
+
return self.resolve_managed_database(database)
|
|
286
|
+
|
|
244
287
|
def create_managed_database(
|
|
245
288
|
self,
|
|
246
289
|
description: str | None = None,
|
|
@@ -275,8 +318,8 @@ class HotdataClient:
|
|
|
275
318
|
raise RuntimeError(api_error_message(e)) from e
|
|
276
319
|
return managed_database_from_detail(created)
|
|
277
320
|
|
|
278
|
-
def delete_managed_database(self, name_or_id: str) -> None:
|
|
279
|
-
db = self.
|
|
321
|
+
def delete_managed_database(self, name_or_id: str | ManagedDatabase) -> None:
|
|
322
|
+
db = self._as_managed_database(name_or_id)
|
|
280
323
|
try:
|
|
281
324
|
self._databases_api().delete_database(db.id)
|
|
282
325
|
except ApiException as e:
|
|
@@ -284,11 +327,11 @@ class HotdataClient:
|
|
|
284
327
|
|
|
285
328
|
def list_managed_tables(
|
|
286
329
|
self,
|
|
287
|
-
database: str,
|
|
330
|
+
database: str | ManagedDatabase,
|
|
288
331
|
*,
|
|
289
332
|
schema: str | None = None,
|
|
290
333
|
) -> list[ManagedTable]:
|
|
291
|
-
db = self.
|
|
334
|
+
db = self._as_managed_database(database)
|
|
292
335
|
rows: list[ManagedTable] = []
|
|
293
336
|
for t in self.iter_tables(connection_id=db.default_connection_id):
|
|
294
337
|
if schema is not None and t.var_schema != schema:
|
|
@@ -333,7 +376,7 @@ class HotdataClient:
|
|
|
333
376
|
|
|
334
377
|
def load_managed_table(
|
|
335
378
|
self,
|
|
336
|
-
database: str,
|
|
379
|
+
database: str | ManagedDatabase,
|
|
337
380
|
table: str,
|
|
338
381
|
*,
|
|
339
382
|
schema: str = DEFAULT_SCHEMA,
|
|
@@ -344,7 +387,7 @@ class HotdataClient:
|
|
|
344
387
|
) -> LoadManagedTableResult:
|
|
345
388
|
if (upload_id is None) == (file is None):
|
|
346
389
|
raise ValueError("Exactly one of upload_id or file is required")
|
|
347
|
-
db = self.
|
|
390
|
+
db = self._as_managed_database(database)
|
|
348
391
|
if upload_id is not None:
|
|
349
392
|
resolved_upload_id = upload_id
|
|
350
393
|
else:
|
|
@@ -374,7 +417,7 @@ class HotdataClient:
|
|
|
374
417
|
|
|
375
418
|
def add_managed_table(
|
|
376
419
|
self,
|
|
377
|
-
database: str,
|
|
420
|
+
database: str | ManagedDatabase,
|
|
378
421
|
table: str,
|
|
379
422
|
*,
|
|
380
423
|
schema: str = DEFAULT_SCHEMA,
|
|
@@ -387,7 +430,7 @@ class HotdataClient:
|
|
|
387
430
|
schema after creation without recreating it. ``key`` sets the
|
|
388
431
|
row-identity columns for delete/update/upsert; omit for keyless.
|
|
389
432
|
"""
|
|
390
|
-
db = self.
|
|
433
|
+
db = self._as_managed_database(database)
|
|
391
434
|
request = AddManagedTableRequest(name=table, key=list(key or []))
|
|
392
435
|
try:
|
|
393
436
|
self._databases_api().add_database_table(db.id, schema, request)
|
|
@@ -403,17 +446,210 @@ class HotdataClient:
|
|
|
403
446
|
|
|
404
447
|
def delete_managed_table(
|
|
405
448
|
self,
|
|
406
|
-
database: str,
|
|
449
|
+
database: str | ManagedDatabase,
|
|
407
450
|
table: str,
|
|
408
451
|
*,
|
|
409
452
|
schema: str = DEFAULT_SCHEMA,
|
|
410
453
|
) -> None:
|
|
411
|
-
db = self.
|
|
454
|
+
db = self._as_managed_database(database)
|
|
412
455
|
try:
|
|
413
456
|
self.connections().delete_managed_table(db.default_connection_id, schema, table)
|
|
414
457
|
except ApiException as e:
|
|
415
458
|
raise RuntimeError(api_error_message(e)) from e
|
|
416
459
|
|
|
460
|
+
def create_index(
|
|
461
|
+
self,
|
|
462
|
+
database: str | ManagedDatabase,
|
|
463
|
+
table: str,
|
|
464
|
+
*,
|
|
465
|
+
schema: str = DEFAULT_SCHEMA,
|
|
466
|
+
index_name: str | None = None,
|
|
467
|
+
columns: list[str],
|
|
468
|
+
index_type: IndexType,
|
|
469
|
+
metric: VectorMetric | None = None,
|
|
470
|
+
dimensions: int | None = None,
|
|
471
|
+
embedding_provider_id: str | None = None,
|
|
472
|
+
output_column: str | None = None,
|
|
473
|
+
description: str | None = None,
|
|
474
|
+
wait: bool = True,
|
|
475
|
+
timeout_s: float = 300.0,
|
|
476
|
+
poll_interval_s: float = 2.0,
|
|
477
|
+
) -> CreateIndexResult:
|
|
478
|
+
"""Build an index on a managed table and wait for it to be ready.
|
|
479
|
+
|
|
480
|
+
The Python equivalent of ``hotdata indexes create``, scoped to managed
|
|
481
|
+
databases. Indexing a table on a plain (non-managed) connection is not
|
|
482
|
+
supported here; the CLI's ``--catalog`` flag covers that case.
|
|
483
|
+
|
|
484
|
+
``index_type`` selects the index kind and is required: ``"bm25"`` for
|
|
485
|
+
full-text search (queries error outright without one), ``"vector"`` for
|
|
486
|
+
nearest-neighbour search (queries work without one, but only at
|
|
487
|
+
full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
|
|
488
|
+
to ``"sorted"``; this method makes the choice explicit instead, because
|
|
489
|
+
the wrong kind fails at query time rather than here.
|
|
490
|
+
|
|
491
|
+
``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
|
|
492
|
+
derivation the CLI uses when ``--name`` is omitted, so both surfaces
|
|
493
|
+
name the same index identically.
|
|
494
|
+
|
|
495
|
+
There are two kinds of vector index, and they are queried differently:
|
|
496
|
+
|
|
497
|
+
* **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
|
|
498
|
+
vector column (a float list), and a query passes a literal vector:
|
|
499
|
+
``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
|
|
500
|
+
distance function the caller writes — ``cosine`` serves
|
|
501
|
+
``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
|
|
502
|
+
``negative_dot_product``. A mismatch is not an error: the query
|
|
503
|
+
silently reverts to a full table scan. Omitting ``metric`` lets the
|
|
504
|
+
server choose (``l2`` for float-array columns), so pass it explicitly
|
|
505
|
+
whenever the query function is known.
|
|
506
|
+
* **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
|
|
507
|
+
provider ``sys_emb_openai``). ``columns`` is then the *source text*
|
|
508
|
+
column; the provider embeds it into ``output_column`` (default
|
|
509
|
+
``{column}_embedding``) and the index is built over that. A query
|
|
510
|
+
passes text, not a vector — ``vector_distance(source_col, 'query')`` —
|
|
511
|
+
and the server resolves the matching distance function from the index
|
|
512
|
+
itself, so the metric-mismatch trap above does not apply. The returned
|
|
513
|
+
``source_column`` names the column to query.
|
|
514
|
+
|
|
515
|
+
``dimensions`` picks the output width for providers that support several;
|
|
516
|
+
it does not apply when indexing an existing vector column, whose width is
|
|
517
|
+
read from the data. ``description`` is a user-facing label for the
|
|
518
|
+
embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
|
|
519
|
+
index takes exactly one column, and every option in this paragraph — plus
|
|
520
|
+
``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
|
|
521
|
+
|
|
522
|
+
The server builds the index as a background job. This method polls that
|
|
523
|
+
job to a terminal state and raises ``RuntimeError`` if it failed, because
|
|
524
|
+
the submit call itself reports success for builds that later fail. Pass
|
|
525
|
+
``wait=False`` to return as soon as the job is accepted — the result then
|
|
526
|
+
carries ``status="pending"`` and a ``job_id``, and the caller owns
|
|
527
|
+
checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
|
|
528
|
+
|
|
529
|
+
Raises ``ValueError`` for an unusable argument combination,
|
|
530
|
+
``RuntimeError`` if the API rejects the request or the build fails, and
|
|
531
|
+
``TimeoutError`` if the build is still running after ``timeout_s``.
|
|
532
|
+
"""
|
|
533
|
+
if not columns:
|
|
534
|
+
raise ValueError("create_index requires at least one column")
|
|
535
|
+
if index_type not in _INDEX_TYPES:
|
|
536
|
+
allowed = ", ".join(sorted(_INDEX_TYPES))
|
|
537
|
+
raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
|
|
538
|
+
if index_type != "vector":
|
|
539
|
+
vector_only = {
|
|
540
|
+
"metric": metric,
|
|
541
|
+
"dimensions": dimensions,
|
|
542
|
+
"embedding_provider_id": embedding_provider_id,
|
|
543
|
+
"output_column": output_column,
|
|
544
|
+
"description": description,
|
|
545
|
+
}
|
|
546
|
+
supplied = sorted(k for k, v in vector_only.items() if v is not None)
|
|
547
|
+
if supplied:
|
|
548
|
+
raise ValueError(
|
|
549
|
+
f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
|
|
550
|
+
f"vector indexes only (index_type={index_type!r})"
|
|
551
|
+
)
|
|
552
|
+
else:
|
|
553
|
+
if len(columns) != 1:
|
|
554
|
+
raise ValueError(
|
|
555
|
+
f"a vector index takes exactly one column (got {len(columns)}); "
|
|
556
|
+
"the engine indexes only the first"
|
|
557
|
+
)
|
|
558
|
+
if metric is not None and metric not in _VECTOR_METRICS:
|
|
559
|
+
allowed = ", ".join(sorted(_VECTOR_METRICS))
|
|
560
|
+
raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
|
|
561
|
+
|
|
562
|
+
# Matches the CLI's derivation so both surfaces name the same index
|
|
563
|
+
# identically: `hotdata indexes create` without --name.
|
|
564
|
+
resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
|
|
565
|
+
|
|
566
|
+
db = self._as_managed_database(database)
|
|
567
|
+
request = CreateIndexRequest(
|
|
568
|
+
index_name=resolved_name,
|
|
569
|
+
columns=list(columns),
|
|
570
|
+
index_type=index_type,
|
|
571
|
+
metric=metric,
|
|
572
|
+
dimensions=dimensions,
|
|
573
|
+
embedding_provider_id=embedding_provider_id,
|
|
574
|
+
output_column=output_column,
|
|
575
|
+
description=description,
|
|
576
|
+
var_async=True,
|
|
577
|
+
)
|
|
578
|
+
try:
|
|
579
|
+
submitted = self._indexes_api().create_index(
|
|
580
|
+
db.default_connection_id,
|
|
581
|
+
schema,
|
|
582
|
+
table,
|
|
583
|
+
request,
|
|
584
|
+
)
|
|
585
|
+
except ApiException as e:
|
|
586
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
587
|
+
|
|
588
|
+
full_name = f"{db.id}.{schema}.{table}"
|
|
589
|
+
|
|
590
|
+
# A build the server finished inline answers 201 with the index itself;
|
|
591
|
+
# the async path answers 202 with a job to poll.
|
|
592
|
+
if isinstance(submitted, IndexInfoResponse):
|
|
593
|
+
return self._index_result(submitted, full_name, schema, table, job_id=None)
|
|
594
|
+
|
|
595
|
+
if not isinstance(submitted, SubmitJobResponse):
|
|
596
|
+
raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
|
|
597
|
+
|
|
598
|
+
job_id = submitted.id
|
|
599
|
+
|
|
600
|
+
def requested_result(status: str) -> CreateIndexResult:
|
|
601
|
+
"""Echo the requested values, for the paths where the server hands
|
|
602
|
+
back a job rather than the built index."""
|
|
603
|
+
return CreateIndexResult(
|
|
604
|
+
full_name=full_name,
|
|
605
|
+
schema_name=schema,
|
|
606
|
+
table_name=table,
|
|
607
|
+
index_name=resolved_name,
|
|
608
|
+
index_type=index_type,
|
|
609
|
+
columns=list(columns),
|
|
610
|
+
metric=metric,
|
|
611
|
+
source_column=columns[0] if embedding_provider_id else None,
|
|
612
|
+
status=status,
|
|
613
|
+
job_id=job_id,
|
|
614
|
+
)
|
|
615
|
+
|
|
616
|
+
if not wait:
|
|
617
|
+
return requested_result(enum_value(submitted.status))
|
|
618
|
+
|
|
619
|
+
job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
|
|
620
|
+
status = enum_value(job.status)
|
|
621
|
+
if status != "succeeded":
|
|
622
|
+
detail = job.error_message or f"Index build {status}"
|
|
623
|
+
raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
|
|
624
|
+
|
|
625
|
+
# `result` is a oneOf wrapper today; tolerate the model arriving directly.
|
|
626
|
+
built = getattr(job.result, "actual_instance", job.result)
|
|
627
|
+
if isinstance(built, IndexInfoResponse):
|
|
628
|
+
return self._index_result(built, full_name, schema, table, job_id=job_id)
|
|
629
|
+
return requested_result("ready")
|
|
630
|
+
|
|
631
|
+
@staticmethod
|
|
632
|
+
def _index_result(
|
|
633
|
+
info: IndexInfoResponse,
|
|
634
|
+
full_name: str,
|
|
635
|
+
schema: str,
|
|
636
|
+
table: str,
|
|
637
|
+
*,
|
|
638
|
+
job_id: str | None,
|
|
639
|
+
) -> CreateIndexResult:
|
|
640
|
+
return CreateIndexResult(
|
|
641
|
+
full_name=full_name,
|
|
642
|
+
schema_name=schema,
|
|
643
|
+
table_name=table,
|
|
644
|
+
index_name=info.index_name,
|
|
645
|
+
index_type=info.index_type,
|
|
646
|
+
columns=list(info.columns),
|
|
647
|
+
metric=info.metric,
|
|
648
|
+
source_column=info.source_column,
|
|
649
|
+
status=enum_value(info.status),
|
|
650
|
+
job_id=job_id,
|
|
651
|
+
)
|
|
652
|
+
|
|
417
653
|
def list_recent_results(
|
|
418
654
|
self,
|
|
419
655
|
*,
|
|
@@ -547,6 +783,29 @@ class HotdataClient:
|
|
|
547
783
|
f"(last status: {getattr(last, 'status', None)})"
|
|
548
784
|
)
|
|
549
785
|
|
|
786
|
+
def _poll_job(
|
|
787
|
+
self,
|
|
788
|
+
job_id: str,
|
|
789
|
+
*,
|
|
790
|
+
timeout_s: float = 300.0,
|
|
791
|
+
interval_s: float = 2.0,
|
|
792
|
+
) -> JobStatusResponse:
|
|
793
|
+
jobs = self._jobs_api()
|
|
794
|
+
deadline = time.monotonic() + timeout_s
|
|
795
|
+
last: JobStatusResponse | None = None
|
|
796
|
+
while time.monotonic() < deadline:
|
|
797
|
+
try:
|
|
798
|
+
last = jobs.get_job(job_id)
|
|
799
|
+
except ApiException as e:
|
|
800
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
801
|
+
if last.status in _JOB_TERMINAL:
|
|
802
|
+
return last
|
|
803
|
+
time.sleep(interval_s)
|
|
804
|
+
last_status = enum_value(last.status) if last is not None else None
|
|
805
|
+
raise TimeoutError(
|
|
806
|
+
f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
|
|
807
|
+
)
|
|
808
|
+
|
|
550
809
|
def _wait_result_ready(
|
|
551
810
|
self,
|
|
552
811
|
result_id: str,
|
|
@@ -569,16 +828,19 @@ class HotdataClient:
|
|
|
569
828
|
f"(last status: {getattr(last, 'status', None)})"
|
|
570
829
|
)
|
|
571
830
|
|
|
572
|
-
def execute_sql(
|
|
831
|
+
def execute_sql(
|
|
832
|
+
self, sql: str, *, database: str | ManagedDatabase | None = None
|
|
833
|
+
) -> QueryResult:
|
|
573
834
|
"""Execute SQL and return a :class:`QueryResult`.
|
|
574
835
|
|
|
575
|
-
Pass ``database`` to scope the query to a managed database.
|
|
576
|
-
is resolved to a database ID once before the retry loop
|
|
836
|
+
Pass ``database`` to scope the query to a managed database. A name or
|
|
837
|
+
id is resolved to a database ID once before the retry loop; an
|
|
838
|
+
already-resolved ``ManagedDatabase`` is used as-is (no read probe). The
|
|
577
839
|
``X-Database-Id`` header is sent with every attempt. Inside a managed
|
|
578
840
|
database the built-in catalog is always ``"default"``, so table
|
|
579
841
|
references should use ``"default"."<schema>"."<table>"``.
|
|
580
842
|
"""
|
|
581
|
-
database_id = self.
|
|
843
|
+
database_id = self._as_managed_database(database).id if database else None
|
|
582
844
|
last_err: BaseException | None = None
|
|
583
845
|
for attempt in range(3):
|
|
584
846
|
try:
|