hotdata-framework 0.9.0__tar.gz → 0.10.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/CHANGELOG.md +55 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/CONTRACT.md +3 -1
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/PKG-INFO +2 -1
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/README.md +1 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/__init__.py +2 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/client.py +247 -1
- hotdata_framework-0.10.0/hotdata_framework/databases.py +114 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/pyproject.toml +1 -1
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_contract.py +1 -0
- hotdata_framework-0.10.0/tests/test_indexes.py +824 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/uv.lock +1 -1
- hotdata_framework-0.9.0/hotdata_framework/databases.py +0 -66
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/dependabot.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/publish.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/release.yml +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.gitignore +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/RELEASING.md +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/errors.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/http.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/managed_client.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_client.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_databases.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_managed_client.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_version.py +0 -0
|
@@ -8,6 +8,61 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
## [0.10.0] - 2026-08-07
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- `create_index(database, table, columns=..., index_type=...)` builds an index on a
|
|
16
|
+
managed table, bringing the framework client to parity with `hotdata indexes
|
|
17
|
+
create` in the CLI. It covers all three index kinds the API accepts: `"bm25"` for
|
|
18
|
+
full-text search, `"vector"` for nearest-neighbour search, and `"sorted"`.
|
|
19
|
+
Previously the framework had no index API at all and callers had to drop to the raw
|
|
20
|
+
`hotdata.IndexesApi`, which left a managed database's data loaded but not
|
|
21
|
+
searchable: full-text queries error without an index, and vector queries run at
|
|
22
|
+
full-scan speed. Like the other managed-table operations, `database` accepts a
|
|
23
|
+
name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
|
|
24
|
+
(non-managed) connection is not covered — the CLI's `--catalog` handles that.
|
|
25
|
+
|
|
26
|
+
`index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
|
|
27
|
+
derivation the CLI uses when `--name` is omitted, so both surfaces name the same
|
|
28
|
+
index identically. `index_type` is required, unlike the API's `"sorted"` default,
|
|
29
|
+
because the wrong kind only fails at query time.
|
|
30
|
+
|
|
31
|
+
The server builds the index as a background job whose submit call reports success
|
|
32
|
+
even when the build later fails, so `create_index` polls the job to a terminal
|
|
33
|
+
state and raises `RuntimeError` carrying the job's `error_message`. Pass
|
|
34
|
+
`wait=False` to return once the job is accepted (`status="pending"` plus a
|
|
35
|
+
`job_id`) and own the outcome check yourself, as the CLI's `--async` does;
|
|
36
|
+
`timeout_s` and `poll_interval_s` tune the wait.
|
|
37
|
+
|
|
38
|
+
Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
|
|
39
|
+
existing vector column, queried with a literal vector — and there `metric` must
|
|
40
|
+
match the distance function the query uses (`cosine`→`cosine_distance`,
|
|
41
|
+
`l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
|
|
42
|
+
reverts to a full table scan rather than erroring. Setting
|
|
43
|
+
`embedding_provider_id` indexes a *text* column instead: the provider embeds it
|
|
44
|
+
into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
|
|
45
|
+
and the server resolves the distance function itself.
|
|
46
|
+
|
|
47
|
+
Argument combinations that the server would silently ignore raise `ValueError`
|
|
48
|
+
before any request is sent: an unknown `index_type` or `metric`, a vector index
|
|
49
|
+
with more than one column (the engine indexes only the first), and
|
|
50
|
+
`metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
|
|
51
|
+
non-vector index.
|
|
52
|
+
|
|
53
|
+
Verified against `api.hotdata.dev` when this version was released: BM25 and vector
|
|
54
|
+
indexes both build and report `ready`, and a BM25 index is used by full-text
|
|
55
|
+
search. A *vector* index on a managed database was **not** picked up by the query
|
|
56
|
+
planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
|
|
57
|
+
planned as a full scan. That reproduces with an index created by `hotdata indexes
|
|
58
|
+
create`, so it is an engine-side issue rather than a client one, but it means a
|
|
59
|
+
vector index built through this method may not yet accelerate queries.
|
|
60
|
+
|
|
61
|
+
- `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
|
|
62
|
+
`hotdata_framework` and added to the public contract surface. Its `source_column`
|
|
63
|
+
names the text column to query for a provider-backed vector index, and is `None`
|
|
64
|
+
for BM25, sorted, and plain vector indexes.
|
|
65
|
+
|
|
11
66
|
## [0.9.0] - 2026-07-23
|
|
12
67
|
|
|
13
68
|
### Added
|
|
@@ -33,6 +33,7 @@ The supported import surface is:
|
|
|
33
33
|
- `ManagedDatabase`
|
|
34
34
|
- `ManagedTable`
|
|
35
35
|
- `LoadManagedTableResult`
|
|
36
|
+
- `CreateIndexResult`
|
|
36
37
|
- `DEFAULT_SCHEMA`
|
|
37
38
|
- `is_parquet_path`
|
|
38
39
|
|
|
@@ -63,7 +64,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
63
64
|
- `upload_parquet(path)` uploads a local parquet file and returns an upload id.
|
|
64
65
|
- `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
|
|
65
66
|
- `delete_managed_table(database, table, schema=...)` deletes a managed table.
|
|
66
|
-
-
|
|
67
|
+
- `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
|
|
68
|
+
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
67
69
|
|
|
68
70
|
### `QueryResult`
|
|
69
71
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.10.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -44,6 +44,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
44
44
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
45
45
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
46
46
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
47
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
47
48
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
48
49
|
|
|
49
50
|
Install:
|
|
@@ -16,6 +16,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
|
16
16
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
17
17
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
18
18
|
- **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
|
|
19
|
+
- **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
|
|
19
20
|
- **Health helpers** — build compact API/workspace health summaries for UI integrations.
|
|
20
21
|
|
|
21
22
|
Install:
|
|
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
|
|
|
10
10
|
)
|
|
11
11
|
from hotdata_framework.databases import (
|
|
12
12
|
DEFAULT_SCHEMA,
|
|
13
|
+
CreateIndexResult,
|
|
13
14
|
LoadManagedTableResult,
|
|
14
15
|
ManagedDatabase,
|
|
15
16
|
ManagedTable,
|
|
@@ -43,6 +44,7 @@ except PackageNotFoundError:
|
|
|
43
44
|
|
|
44
45
|
__all__ = [
|
|
45
46
|
"DEFAULT_SCHEMA",
|
|
47
|
+
"CreateIndexResult",
|
|
46
48
|
"HotdataClient",
|
|
47
49
|
"HotdataError",
|
|
48
50
|
"HotdataTerminalError",
|
|
@@ -4,12 +4,14 @@ import functools
|
|
|
4
4
|
import time
|
|
5
5
|
from collections.abc import Iterator
|
|
6
6
|
from dataclasses import asdict, dataclass
|
|
7
|
-
from typing import Any, Literal
|
|
7
|
+
from typing import Any, Literal, get_args
|
|
8
8
|
|
|
9
9
|
from hotdata import ApiClient, Configuration
|
|
10
10
|
from hotdata.api.connections_api import ConnectionsApi
|
|
11
11
|
from hotdata.api.databases_api import DatabasesApi
|
|
12
|
+
from hotdata.api.indexes_api import IndexesApi
|
|
12
13
|
from hotdata.api.information_schema_api import InformationSchemaApi
|
|
14
|
+
from hotdata.api.jobs_api import JobsApi
|
|
13
15
|
from hotdata.api.query_api import QueryApi
|
|
14
16
|
from hotdata.api.query_runs_api import QueryRunsApi
|
|
15
17
|
from hotdata.api.results_api import ResultsApi
|
|
@@ -20,21 +22,27 @@ from hotdata.exceptions import ApiException
|
|
|
20
22
|
from hotdata.models.add_managed_table_request import AddManagedTableRequest
|
|
21
23
|
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
22
24
|
from hotdata.models.create_database_request import CreateDatabaseRequest
|
|
25
|
+
from hotdata.models.create_index_request import CreateIndexRequest
|
|
23
26
|
from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
|
|
24
27
|
from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
|
|
28
|
+
from hotdata.models.index_info_response import IndexInfoResponse
|
|
29
|
+
from hotdata.models.job_status_response import JobStatusResponse
|
|
25
30
|
from hotdata.models.load_managed_table_request import LoadManagedTableRequest
|
|
26
31
|
from hotdata.models.query_request import QueryRequest
|
|
27
32
|
from hotdata.models.query_response import QueryResponse
|
|
33
|
+
from hotdata.models.submit_job_response import SubmitJobResponse
|
|
28
34
|
from hotdata.models.table_info import TableInfo
|
|
29
35
|
from urllib3.exceptions import HTTPError as Urllib3HTTPError
|
|
30
36
|
from urllib3.exceptions import ProtocolError
|
|
31
37
|
|
|
32
38
|
from hotdata_framework.databases import (
|
|
33
39
|
DEFAULT_SCHEMA,
|
|
40
|
+
CreateIndexResult,
|
|
34
41
|
LoadManagedTableResult,
|
|
35
42
|
ManagedDatabase,
|
|
36
43
|
ManagedTable,
|
|
37
44
|
api_error_message,
|
|
45
|
+
enum_value,
|
|
38
46
|
is_parquet_path,
|
|
39
47
|
managed_database_from_detail,
|
|
40
48
|
)
|
|
@@ -52,8 +60,24 @@ from hotdata_framework.result import QueryResult
|
|
|
52
60
|
# rows, delete/update/upsert match by the table's declared key.
|
|
53
61
|
ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
|
|
54
62
|
|
|
63
|
+
# Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
|
|
64
|
+
# "bm25" backs full-text search and "vector" backs nearest-neighbour search.
|
|
65
|
+
IndexType = Literal["sorted", "bm25", "vector"]
|
|
66
|
+
|
|
67
|
+
# Distance metrics a vector index can be built with. Each one accelerates
|
|
68
|
+
# exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
|
|
69
|
+
# dot -> negative_dot_product. A hand-written query naming a different function
|
|
70
|
+
# falls back to a full scan; the provider-backed vector_distance path resolves
|
|
71
|
+
# the function from the index instead, so it cannot mismatch.
|
|
72
|
+
VectorMetric = Literal["l2", "cosine", "dot"]
|
|
73
|
+
|
|
74
|
+
_INDEX_TYPES = frozenset(get_args(IndexType))
|
|
75
|
+
_VECTOR_METRICS = frozenset(get_args(VectorMetric))
|
|
76
|
+
|
|
55
77
|
_TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
|
|
56
78
|
_RESULT_FAILURE = frozenset({"failed", "cancelled"})
|
|
79
|
+
# Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
|
|
80
|
+
_JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
|
|
57
81
|
|
|
58
82
|
|
|
59
83
|
@dataclass(frozen=True)
|
|
@@ -188,6 +212,12 @@ class HotdataClient:
|
|
|
188
212
|
def _information_schema(self) -> InformationSchemaApi:
|
|
189
213
|
return InformationSchemaApi(self._api)
|
|
190
214
|
|
|
215
|
+
def _indexes_api(self) -> IndexesApi:
|
|
216
|
+
return IndexesApi(self._api)
|
|
217
|
+
|
|
218
|
+
def _jobs_api(self) -> JobsApi:
|
|
219
|
+
return JobsApi(self._api)
|
|
220
|
+
|
|
191
221
|
def _query_api(self) -> QueryApi:
|
|
192
222
|
return QueryApi(self._api)
|
|
193
223
|
|
|
@@ -427,6 +457,199 @@ class HotdataClient:
|
|
|
427
457
|
except ApiException as e:
|
|
428
458
|
raise RuntimeError(api_error_message(e)) from e
|
|
429
459
|
|
|
460
|
+
def create_index(
|
|
461
|
+
self,
|
|
462
|
+
database: str | ManagedDatabase,
|
|
463
|
+
table: str,
|
|
464
|
+
*,
|
|
465
|
+
schema: str = DEFAULT_SCHEMA,
|
|
466
|
+
index_name: str | None = None,
|
|
467
|
+
columns: list[str],
|
|
468
|
+
index_type: IndexType,
|
|
469
|
+
metric: VectorMetric | None = None,
|
|
470
|
+
dimensions: int | None = None,
|
|
471
|
+
embedding_provider_id: str | None = None,
|
|
472
|
+
output_column: str | None = None,
|
|
473
|
+
description: str | None = None,
|
|
474
|
+
wait: bool = True,
|
|
475
|
+
timeout_s: float = 300.0,
|
|
476
|
+
poll_interval_s: float = 2.0,
|
|
477
|
+
) -> CreateIndexResult:
|
|
478
|
+
"""Build an index on a managed table and wait for it to be ready.
|
|
479
|
+
|
|
480
|
+
The Python equivalent of ``hotdata indexes create``, scoped to managed
|
|
481
|
+
databases. Indexing a table on a plain (non-managed) connection is not
|
|
482
|
+
supported here; the CLI's ``--catalog`` flag covers that case.
|
|
483
|
+
|
|
484
|
+
``index_type`` selects the index kind and is required: ``"bm25"`` for
|
|
485
|
+
full-text search (queries error outright without one), ``"vector"`` for
|
|
486
|
+
nearest-neighbour search (queries work without one, but only at
|
|
487
|
+
full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
|
|
488
|
+
to ``"sorted"``; this method makes the choice explicit instead, because
|
|
489
|
+
the wrong kind fails at query time rather than here.
|
|
490
|
+
|
|
491
|
+
``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
|
|
492
|
+
derivation the CLI uses when ``--name`` is omitted, so both surfaces
|
|
493
|
+
name the same index identically.
|
|
494
|
+
|
|
495
|
+
There are two kinds of vector index, and they are queried differently:
|
|
496
|
+
|
|
497
|
+
* **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
|
|
498
|
+
vector column (a float list), and a query passes a literal vector:
|
|
499
|
+
``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
|
|
500
|
+
distance function the caller writes — ``cosine`` serves
|
|
501
|
+
``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
|
|
502
|
+
``negative_dot_product``. A mismatch is not an error: the query
|
|
503
|
+
silently reverts to a full table scan. Omitting ``metric`` lets the
|
|
504
|
+
server choose (``l2`` for float-array columns), so pass it explicitly
|
|
505
|
+
whenever the query function is known.
|
|
506
|
+
* **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
|
|
507
|
+
provider ``sys_emb_openai``). ``columns`` is then the *source text*
|
|
508
|
+
column; the provider embeds it into ``output_column`` (default
|
|
509
|
+
``{column}_embedding``) and the index is built over that. A query
|
|
510
|
+
passes text, not a vector — ``vector_distance(source_col, 'query')`` —
|
|
511
|
+
and the server resolves the matching distance function from the index
|
|
512
|
+
itself, so the metric-mismatch trap above does not apply. The returned
|
|
513
|
+
``source_column`` names the column to query.
|
|
514
|
+
|
|
515
|
+
``dimensions`` picks the output width for providers that support several;
|
|
516
|
+
it does not apply when indexing an existing vector column, whose width is
|
|
517
|
+
read from the data. ``description`` is a user-facing label for the
|
|
518
|
+
embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
|
|
519
|
+
index takes exactly one column, and every option in this paragraph — plus
|
|
520
|
+
``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
|
|
521
|
+
|
|
522
|
+
The server builds the index as a background job. This method polls that
|
|
523
|
+
job to a terminal state and raises ``RuntimeError`` if it failed, because
|
|
524
|
+
the submit call itself reports success for builds that later fail. Pass
|
|
525
|
+
``wait=False`` to return as soon as the job is accepted — the result then
|
|
526
|
+
carries ``status="pending"`` and a ``job_id``, and the caller owns
|
|
527
|
+
checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
|
|
528
|
+
|
|
529
|
+
Raises ``ValueError`` for an unusable argument combination,
|
|
530
|
+
``RuntimeError`` if the API rejects the request or the build fails, and
|
|
531
|
+
``TimeoutError`` if the build is still running after ``timeout_s``.
|
|
532
|
+
"""
|
|
533
|
+
if not columns:
|
|
534
|
+
raise ValueError("create_index requires at least one column")
|
|
535
|
+
if index_type not in _INDEX_TYPES:
|
|
536
|
+
allowed = ", ".join(sorted(_INDEX_TYPES))
|
|
537
|
+
raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
|
|
538
|
+
if index_type != "vector":
|
|
539
|
+
vector_only = {
|
|
540
|
+
"metric": metric,
|
|
541
|
+
"dimensions": dimensions,
|
|
542
|
+
"embedding_provider_id": embedding_provider_id,
|
|
543
|
+
"output_column": output_column,
|
|
544
|
+
"description": description,
|
|
545
|
+
}
|
|
546
|
+
supplied = sorted(k for k, v in vector_only.items() if v is not None)
|
|
547
|
+
if supplied:
|
|
548
|
+
raise ValueError(
|
|
549
|
+
f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
|
|
550
|
+
f"vector indexes only (index_type={index_type!r})"
|
|
551
|
+
)
|
|
552
|
+
else:
|
|
553
|
+
if len(columns) != 1:
|
|
554
|
+
raise ValueError(
|
|
555
|
+
f"a vector index takes exactly one column (got {len(columns)}); "
|
|
556
|
+
"the engine indexes only the first"
|
|
557
|
+
)
|
|
558
|
+
if metric is not None and metric not in _VECTOR_METRICS:
|
|
559
|
+
allowed = ", ".join(sorted(_VECTOR_METRICS))
|
|
560
|
+
raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
|
|
561
|
+
|
|
562
|
+
# Matches the CLI's derivation so both surfaces name the same index
|
|
563
|
+
# identically: `hotdata indexes create` without --name.
|
|
564
|
+
resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
|
|
565
|
+
|
|
566
|
+
db = self._as_managed_database(database)
|
|
567
|
+
request = CreateIndexRequest(
|
|
568
|
+
index_name=resolved_name,
|
|
569
|
+
columns=list(columns),
|
|
570
|
+
index_type=index_type,
|
|
571
|
+
metric=metric,
|
|
572
|
+
dimensions=dimensions,
|
|
573
|
+
embedding_provider_id=embedding_provider_id,
|
|
574
|
+
output_column=output_column,
|
|
575
|
+
description=description,
|
|
576
|
+
var_async=True,
|
|
577
|
+
)
|
|
578
|
+
try:
|
|
579
|
+
submitted = self._indexes_api().create_index(
|
|
580
|
+
db.default_connection_id,
|
|
581
|
+
schema,
|
|
582
|
+
table,
|
|
583
|
+
request,
|
|
584
|
+
)
|
|
585
|
+
except ApiException as e:
|
|
586
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
587
|
+
|
|
588
|
+
full_name = f"{db.id}.{schema}.{table}"
|
|
589
|
+
|
|
590
|
+
# A build the server finished inline answers 201 with the index itself;
|
|
591
|
+
# the async path answers 202 with a job to poll.
|
|
592
|
+
if isinstance(submitted, IndexInfoResponse):
|
|
593
|
+
return self._index_result(submitted, full_name, schema, table, job_id=None)
|
|
594
|
+
|
|
595
|
+
if not isinstance(submitted, SubmitJobResponse):
|
|
596
|
+
raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
|
|
597
|
+
|
|
598
|
+
job_id = submitted.id
|
|
599
|
+
|
|
600
|
+
def requested_result(status: str) -> CreateIndexResult:
|
|
601
|
+
"""Echo the requested values, for the paths where the server hands
|
|
602
|
+
back a job rather than the built index."""
|
|
603
|
+
return CreateIndexResult(
|
|
604
|
+
full_name=full_name,
|
|
605
|
+
schema_name=schema,
|
|
606
|
+
table_name=table,
|
|
607
|
+
index_name=resolved_name,
|
|
608
|
+
index_type=index_type,
|
|
609
|
+
columns=list(columns),
|
|
610
|
+
metric=metric,
|
|
611
|
+
source_column=columns[0] if embedding_provider_id else None,
|
|
612
|
+
status=status,
|
|
613
|
+
job_id=job_id,
|
|
614
|
+
)
|
|
615
|
+
|
|
616
|
+
if not wait:
|
|
617
|
+
return requested_result(enum_value(submitted.status))
|
|
618
|
+
|
|
619
|
+
job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
|
|
620
|
+
status = enum_value(job.status)
|
|
621
|
+
if status != "succeeded":
|
|
622
|
+
detail = job.error_message or f"Index build {status}"
|
|
623
|
+
raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
|
|
624
|
+
|
|
625
|
+
# `result` is a oneOf wrapper today; tolerate the model arriving directly.
|
|
626
|
+
built = getattr(job.result, "actual_instance", job.result)
|
|
627
|
+
if isinstance(built, IndexInfoResponse):
|
|
628
|
+
return self._index_result(built, full_name, schema, table, job_id=job_id)
|
|
629
|
+
return requested_result("ready")
|
|
630
|
+
|
|
631
|
+
@staticmethod
|
|
632
|
+
def _index_result(
|
|
633
|
+
info: IndexInfoResponse,
|
|
634
|
+
full_name: str,
|
|
635
|
+
schema: str,
|
|
636
|
+
table: str,
|
|
637
|
+
*,
|
|
638
|
+
job_id: str | None,
|
|
639
|
+
) -> CreateIndexResult:
|
|
640
|
+
return CreateIndexResult(
|
|
641
|
+
full_name=full_name,
|
|
642
|
+
schema_name=schema,
|
|
643
|
+
table_name=table,
|
|
644
|
+
index_name=info.index_name,
|
|
645
|
+
index_type=info.index_type,
|
|
646
|
+
columns=list(info.columns),
|
|
647
|
+
metric=info.metric,
|
|
648
|
+
source_column=info.source_column,
|
|
649
|
+
status=enum_value(info.status),
|
|
650
|
+
job_id=job_id,
|
|
651
|
+
)
|
|
652
|
+
|
|
430
653
|
def list_recent_results(
|
|
431
654
|
self,
|
|
432
655
|
*,
|
|
@@ -560,6 +783,29 @@ class HotdataClient:
|
|
|
560
783
|
f"(last status: {getattr(last, 'status', None)})"
|
|
561
784
|
)
|
|
562
785
|
|
|
786
|
+
def _poll_job(
|
|
787
|
+
self,
|
|
788
|
+
job_id: str,
|
|
789
|
+
*,
|
|
790
|
+
timeout_s: float = 300.0,
|
|
791
|
+
interval_s: float = 2.0,
|
|
792
|
+
) -> JobStatusResponse:
|
|
793
|
+
jobs = self._jobs_api()
|
|
794
|
+
deadline = time.monotonic() + timeout_s
|
|
795
|
+
last: JobStatusResponse | None = None
|
|
796
|
+
while time.monotonic() < deadline:
|
|
797
|
+
try:
|
|
798
|
+
last = jobs.get_job(job_id)
|
|
799
|
+
except ApiException as e:
|
|
800
|
+
raise RuntimeError(api_error_message(e)) from e
|
|
801
|
+
if last.status in _JOB_TERMINAL:
|
|
802
|
+
return last
|
|
803
|
+
time.sleep(interval_s)
|
|
804
|
+
last_status = enum_value(last.status) if last is not None else None
|
|
805
|
+
raise TimeoutError(
|
|
806
|
+
f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
|
|
807
|
+
)
|
|
808
|
+
|
|
563
809
|
def _wait_result_ready(
|
|
564
810
|
self,
|
|
565
811
|
result_id: str,
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Managed database helpers (Hotdata-owned catalogs with parquet table loads)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import asdict, dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from hotdata.exceptions import ApiException
|
|
10
|
+
|
|
11
|
+
DEFAULT_SCHEMA = "public"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ManagedDatabase:
|
|
16
|
+
id: str
|
|
17
|
+
description: str | None
|
|
18
|
+
default_connection_id: str
|
|
19
|
+
|
|
20
|
+
def to_dict(self) -> dict[str, Any]:
|
|
21
|
+
return asdict(self)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class ManagedTable:
|
|
26
|
+
full_name: str
|
|
27
|
+
schema: str
|
|
28
|
+
table: str
|
|
29
|
+
synced: bool
|
|
30
|
+
last_sync: str | None
|
|
31
|
+
|
|
32
|
+
def to_dict(self) -> dict[str, Any]:
|
|
33
|
+
return asdict(self)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class LoadManagedTableResult:
|
|
38
|
+
connection_id: str
|
|
39
|
+
schema_name: str
|
|
40
|
+
table_name: str
|
|
41
|
+
row_count: int
|
|
42
|
+
full_name: str
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> dict[str, Any]:
|
|
45
|
+
return asdict(self)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class CreateIndexResult:
|
|
50
|
+
"""An index created on a managed table.
|
|
51
|
+
|
|
52
|
+
``status`` is the server's own — ``"ready"`` once the index is built,
|
|
53
|
+
``"pending"`` while it is still building. ``job_id`` identifies the
|
|
54
|
+
background build job, and is ``None`` only when the server built the index
|
|
55
|
+
inline. A caller that passed ``wait=False`` always gets ``"pending"`` and
|
|
56
|
+
owns checking the job's outcome.
|
|
57
|
+
|
|
58
|
+
``source_column`` is set only for an embedding-backed vector index, where it
|
|
59
|
+
names the *text* column a query passes to ``vector_distance(col, 'text')``.
|
|
60
|
+
It is ``None`` for BM25, sorted, and plain (existing-vector-column) indexes.
|
|
61
|
+
|
|
62
|
+
``index_type``, ``columns``, and ``metric`` echo the requested values when the
|
|
63
|
+
server does not return the built index alongside the finished job — that is,
|
|
64
|
+
on the ``wait=False`` path and when a finished job carries no index payload.
|
|
65
|
+
Only when the server did return it does ``columns`` hold the *generated*
|
|
66
|
+
embedding column for an embedding-backed index; on the echoing paths
|
|
67
|
+
``columns[0]`` is the source text column, the same value as
|
|
68
|
+
``source_column``. Read ``status`` to tell the cases apart.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
full_name: str
|
|
72
|
+
schema_name: str
|
|
73
|
+
table_name: str
|
|
74
|
+
index_name: str
|
|
75
|
+
index_type: str
|
|
76
|
+
columns: list[str]
|
|
77
|
+
metric: str | None
|
|
78
|
+
source_column: str | None
|
|
79
|
+
status: str
|
|
80
|
+
job_id: str | None
|
|
81
|
+
|
|
82
|
+
def to_dict(self) -> dict[str, Any]:
|
|
83
|
+
return asdict(self)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def enum_value(value: Any) -> str:
|
|
87
|
+
"""Render an enum-or-str API field as its wire string.
|
|
88
|
+
|
|
89
|
+
The generated models type several status fields as ``str``-mixin enums,
|
|
90
|
+
whose ``str()`` is ``"JobStatus.FAILED"`` rather than ``"failed"``.
|
|
91
|
+
"""
|
|
92
|
+
inner = getattr(value, "value", value)
|
|
93
|
+
return str(inner)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def is_parquet_path(path: str) -> bool:
|
|
97
|
+
return Path(path).suffix.lower() == ".parquet"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def managed_database_from_detail(detail: Any) -> ManagedDatabase:
|
|
101
|
+
return ManagedDatabase(
|
|
102
|
+
id=str(detail.id),
|
|
103
|
+
description=detail.name,
|
|
104
|
+
default_connection_id=str(detail.default_connection_id),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def api_error_message(exc: ApiException) -> str:
|
|
109
|
+
reason = exc.reason or str(exc)
|
|
110
|
+
# Keep the response body: it carries the API's actual explanation.
|
|
111
|
+
body = getattr(exc, "body", None)
|
|
112
|
+
if body:
|
|
113
|
+
return f"{reason}: {' '.join(str(body).split())[:500]}"
|
|
114
|
+
return reason
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "hotdata-framework"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.10.0"
|
|
8
8
|
description = "Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|