hotdata-framework 0.8.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. hotdata_framework-0.10.0/.github/CODEOWNERS +1 -0
  2. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/CHANGELOG.md +68 -0
  3. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/CONTRACT.md +5 -2
  4. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/PKG-INFO +2 -1
  5. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/README.md +1 -0
  6. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/__init__.py +2 -0
  7. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/client.py +277 -15
  8. hotdata_framework-0.10.0/hotdata_framework/databases.py +114 -0
  9. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/pyproject.toml +1 -1
  10. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_client.py +123 -0
  11. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_contract.py +1 -0
  12. hotdata_framework-0.10.0/tests/test_indexes.py +824 -0
  13. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/uv.lock +1 -1
  14. hotdata_framework-0.8.0/hotdata_framework/databases.py +0 -66
  15. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/dependabot.yml +0 -0
  16. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/check-release.yml +0 -0
  17. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/ci.yml +0 -0
  18. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/dependabot-automerge.yml +0 -0
  19. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/publish.yml +0 -0
  20. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.github/workflows/release.yml +0 -0
  21. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/.gitignore +0 -0
  22. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/RELEASING.md +0 -0
  23. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/examples/basic_usage.py +0 -0
  24. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/env.py +0 -0
  25. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/errors.py +0 -0
  26. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/health.py +0 -0
  27. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/http.py +0 -0
  28. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/managed_client.py +0 -0
  29. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/py.typed +0 -0
  30. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/hotdata_framework/result.py +0 -0
  31. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/check-release.py +0 -0
  32. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/extract-changelog.py +0 -0
  33. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/publish-workflow.sh +0 -0
  34. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/release.sh +0 -0
  35. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/scripts/update_changelog.py +0 -0
  36. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_databases.py +0 -0
  37. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_errors.py +0 -0
  38. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_health.py +0 -0
  39. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_managed_client.py +0 -0
  40. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_request_timeout.py +0 -0
  41. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_result.py +0 -0
  42. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_update_changelog.py +0 -0
  43. {hotdata_framework-0.8.0 → hotdata_framework-0.10.0}/tests/test_version.py +0 -0
@@ -0,0 +1 @@
1
+ * @hotdata-dev/engineers
@@ -8,6 +8,74 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
8
8
  ## [Unreleased]
9
9
 
10
10
 
11
+ ## [0.10.0] - 2026-08-07
12
+
13
+ ### Added
14
+
15
+ - `create_index(database, table, columns=..., index_type=...)` builds an index on a
16
+ managed table, bringing the framework client to parity with `hotdata indexes
17
+ create` in the CLI. It covers all three index kinds the API accepts: `"bm25"` for
18
+ full-text search, `"vector"` for nearest-neighbour search, and `"sorted"`.
19
+ Previously the framework had no index API at all and callers had to drop to the raw
20
+ `hotdata.IndexesApi`, which left a managed database's data loaded but not
21
+ searchable: full-text queries error without an index, and vector queries run at
22
+ full-scan speed. Like the other managed-table operations, `database` accepts a
23
+ name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
24
+ (non-managed) connection is not covered — the CLI's `--catalog` handles that.
25
+
26
+ `index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
27
+ derivation the CLI uses when `--name` is omitted, so both surfaces name the same
28
+ index identically. `index_type` is required, unlike the API's `"sorted"` default,
29
+ because the wrong kind only fails at query time.
30
+
31
+ The server builds the index as a background job whose submit call reports success
32
+ even when the build later fails, so `create_index` polls the job to a terminal
33
+ state and raises `RuntimeError` carrying the job's `error_message`. Pass
34
+ `wait=False` to return once the job is accepted (`status="pending"` plus a
35
+ `job_id`) and own the outcome check yourself, as the CLI's `--async` does;
36
+ `timeout_s` and `poll_interval_s` tune the wait.
37
+
38
+ Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
39
+ existing vector column, queried with a literal vector — and there `metric` must
40
+ match the distance function the query uses (`cosine`→`cosine_distance`,
41
+ `l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
42
+ reverts to a full table scan rather than erroring. Setting
43
+ `embedding_provider_id` indexes a *text* column instead: the provider embeds it
44
+ into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
45
+ and the server resolves the distance function itself.
46
+
47
+ Argument combinations that the server would silently ignore raise `ValueError`
48
+ before any request is sent: an unknown `index_type` or `metric`, a vector index
49
+ with more than one column (the engine indexes only the first), and
50
+ `metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
51
+ non-vector index.
52
+
53
+ Verified against `api.hotdata.dev` when this version was released: BM25 and vector
54
+ indexes both build and report `ready`, and a BM25 index is used by full-text
55
+ search. A *vector* index on a managed database was **not** picked up by the query
56
+ planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
57
+ planned as a full scan. That reproduces with an index created by `hotdata indexes
58
+ create`, so it is an engine-side issue rather than a client one, but it means a
59
+ vector index built through this method may not yet accelerate queries.
60
+
61
+ - `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
62
+ `hotdata_framework` and added to the public contract surface. Its `source_column`
63
+ names the text column to query for a provider-backed vector index, and is `None`
64
+ for BM25, sorted, and plain vector indexes.
65
+
66
+ ## [0.9.0] - 2026-07-23
67
+
68
+ ### Added
69
+
70
+ - `list_managed_tables`, `load_managed_table`, `add_managed_table`,
71
+ `delete_managed_table`, `delete_managed_database`, and `execute_sql` accept an
72
+ already-resolved `ManagedDatabase` (as returned by `create_managed_database`)
73
+ in place of a name/id. When passed one, they skip the `get_database` /
74
+ `list_databases` read probe. This lets an API key scoped to create + load but
75
+ not read `/databases` bootstrap a managed database and load into it within a
76
+ single run: the caller holds the `ManagedDatabase` from `create` and drives
77
+ the load/add/query ops with zero reads. The name/id string path is unchanged.
78
+
11
79
  ## [0.8.0] - 2026-07-20
12
80
 
13
81
  ### Changed
@@ -33,6 +33,7 @@ The supported import surface is:
33
33
  - `ManagedDatabase`
34
34
  - `ManagedTable`
35
35
  - `LoadManagedTableResult`
36
+ - `CreateIndexResult`
36
37
  - `DEFAULT_SCHEMA`
37
38
  - `is_parquet_path`
38
39
 
@@ -56,13 +57,15 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
56
57
  adapters should pass `connection_id` when known.
57
58
  - `uploads()` returns the uploads API wrapper for parquet staging.
58
59
  - `list_managed_databases()` returns all databases via the `/databases` API.
59
- - `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan).
60
- - `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front.
60
+ - `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan). A `403` from `/databases` surfaces as `RuntimeError` (forbidden, not absent), preserving the underlying `ApiException` as `__cause__`.
61
+ - `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front. Returns a `ManagedDatabase` (id + `default_connection_id`) sufficient to load without a further read.
61
62
  - `delete_managed_database(name_or_id)` deletes a database via the `/databases` API.
62
63
  - `list_managed_tables(database, schema=...)` lists tables in a managed database.
63
64
  - `upload_parquet(path)` uploads a local parquet file and returns an upload id.
64
65
  - `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
65
66
  - `delete_managed_table(database, table, schema=...)` deletes a managed table.
67
+ - `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
68
+ - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
66
69
 
67
70
  ### `QueryResult`
68
71
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hotdata-framework
3
- Version: 0.8.0
3
+ Version: 0.10.0
4
4
  Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -44,6 +44,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
44
44
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
45
45
  - **History helpers** — list recent results and query run history with normalized dataclasses.
46
46
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
47
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
47
48
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
48
49
 
49
50
  Install:
@@ -16,6 +16,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
16
16
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
17
17
  - **History helpers** — list recent results and query run history with normalized dataclasses.
18
18
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
19
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
19
20
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
20
21
 
21
22
  Install:
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
10
10
  )
11
11
  from hotdata_framework.databases import (
12
12
  DEFAULT_SCHEMA,
13
+ CreateIndexResult,
13
14
  LoadManagedTableResult,
14
15
  ManagedDatabase,
15
16
  ManagedTable,
@@ -43,6 +44,7 @@ except PackageNotFoundError:
43
44
 
44
45
  __all__ = [
45
46
  "DEFAULT_SCHEMA",
47
+ "CreateIndexResult",
46
48
  "HotdataClient",
47
49
  "HotdataError",
48
50
  "HotdataTerminalError",
@@ -4,12 +4,14 @@ import functools
4
4
  import time
5
5
  from collections.abc import Iterator
6
6
  from dataclasses import asdict, dataclass
7
- from typing import Any, Literal
7
+ from typing import Any, Literal, get_args
8
8
 
9
9
  from hotdata import ApiClient, Configuration
10
10
  from hotdata.api.connections_api import ConnectionsApi
11
11
  from hotdata.api.databases_api import DatabasesApi
12
+ from hotdata.api.indexes_api import IndexesApi
12
13
  from hotdata.api.information_schema_api import InformationSchemaApi
14
+ from hotdata.api.jobs_api import JobsApi
13
15
  from hotdata.api.query_api import QueryApi
14
16
  from hotdata.api.query_runs_api import QueryRunsApi
15
17
  from hotdata.api.results_api import ResultsApi
@@ -20,21 +22,27 @@ from hotdata.exceptions import ApiException
20
22
  from hotdata.models.add_managed_table_request import AddManagedTableRequest
21
23
  from hotdata.models.async_query_response import AsyncQueryResponse
22
24
  from hotdata.models.create_database_request import CreateDatabaseRequest
25
+ from hotdata.models.create_index_request import CreateIndexRequest
23
26
  from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
24
27
  from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
28
+ from hotdata.models.index_info_response import IndexInfoResponse
29
+ from hotdata.models.job_status_response import JobStatusResponse
25
30
  from hotdata.models.load_managed_table_request import LoadManagedTableRequest
26
31
  from hotdata.models.query_request import QueryRequest
27
32
  from hotdata.models.query_response import QueryResponse
33
+ from hotdata.models.submit_job_response import SubmitJobResponse
28
34
  from hotdata.models.table_info import TableInfo
29
35
  from urllib3.exceptions import HTTPError as Urllib3HTTPError
30
36
  from urllib3.exceptions import ProtocolError
31
37
 
32
38
  from hotdata_framework.databases import (
33
39
  DEFAULT_SCHEMA,
40
+ CreateIndexResult,
34
41
  LoadManagedTableResult,
35
42
  ManagedDatabase,
36
43
  ManagedTable,
37
44
  api_error_message,
45
+ enum_value,
38
46
  is_parquet_path,
39
47
  managed_database_from_detail,
40
48
  )
@@ -52,8 +60,24 @@ from hotdata_framework.result import QueryResult
52
60
  # rows, delete/update/upsert match by the table's declared key.
53
61
  ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
54
62
 
63
+ # Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
64
+ # "bm25" backs full-text search and "vector" backs nearest-neighbour search.
65
+ IndexType = Literal["sorted", "bm25", "vector"]
66
+
67
+ # Distance metrics a vector index can be built with. Each one accelerates
68
+ # exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
69
+ # dot -> negative_dot_product. A hand-written query naming a different function
70
+ # falls back to a full scan; the provider-backed vector_distance path resolves
71
+ # the function from the index instead, so it cannot mismatch.
72
+ VectorMetric = Literal["l2", "cosine", "dot"]
73
+
74
+ _INDEX_TYPES = frozenset(get_args(IndexType))
75
+ _VECTOR_METRICS = frozenset(get_args(VectorMetric))
76
+
55
77
  _TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
56
78
  _RESULT_FAILURE = frozenset({"failed", "cancelled"})
79
+ # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
80
+ _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
57
81
 
58
82
 
59
83
  @dataclass(frozen=True)
@@ -188,6 +212,12 @@ class HotdataClient:
188
212
  def _information_schema(self) -> InformationSchemaApi:
189
213
  return InformationSchemaApi(self._api)
190
214
 
215
+ def _indexes_api(self) -> IndexesApi:
216
+ return IndexesApi(self._api)
217
+
218
+ def _jobs_api(self) -> JobsApi:
219
+ return JobsApi(self._api)
220
+
191
221
  def _query_api(self) -> QueryApi:
192
222
  return QueryApi(self._api)
193
223
 
@@ -241,6 +271,19 @@ class HotdataClient:
241
271
  raise RuntimeError(api_error_message(e)) from e
242
272
  return managed_database_from_detail(detail)
243
273
 
274
+ def _as_managed_database(self, database: str | ManagedDatabase) -> ManagedDatabase:
275
+ """Return ``database`` as-is if it is already a resolved ``ManagedDatabase``,
276
+ otherwise resolve it by name or id.
277
+
278
+ Passing an already-resolved ``ManagedDatabase`` (e.g. the value returned by
279
+ :meth:`create_managed_database`) skips the id/name read probe, so callers
280
+ whose API key may create but not read ``/databases`` can drive loads without
281
+ a forbidden read.
282
+ """
283
+ if isinstance(database, ManagedDatabase):
284
+ return database
285
+ return self.resolve_managed_database(database)
286
+
244
287
  def create_managed_database(
245
288
  self,
246
289
  description: str | None = None,
@@ -275,8 +318,8 @@ class HotdataClient:
275
318
  raise RuntimeError(api_error_message(e)) from e
276
319
  return managed_database_from_detail(created)
277
320
 
278
- def delete_managed_database(self, name_or_id: str) -> None:
279
- db = self.resolve_managed_database(name_or_id)
321
+ def delete_managed_database(self, name_or_id: str | ManagedDatabase) -> None:
322
+ db = self._as_managed_database(name_or_id)
280
323
  try:
281
324
  self._databases_api().delete_database(db.id)
282
325
  except ApiException as e:
@@ -284,11 +327,11 @@ class HotdataClient:
284
327
 
285
328
  def list_managed_tables(
286
329
  self,
287
- database: str,
330
+ database: str | ManagedDatabase,
288
331
  *,
289
332
  schema: str | None = None,
290
333
  ) -> list[ManagedTable]:
291
- db = self.resolve_managed_database(database)
334
+ db = self._as_managed_database(database)
292
335
  rows: list[ManagedTable] = []
293
336
  for t in self.iter_tables(connection_id=db.default_connection_id):
294
337
  if schema is not None and t.var_schema != schema:
@@ -333,7 +376,7 @@ class HotdataClient:
333
376
 
334
377
  def load_managed_table(
335
378
  self,
336
- database: str,
379
+ database: str | ManagedDatabase,
337
380
  table: str,
338
381
  *,
339
382
  schema: str = DEFAULT_SCHEMA,
@@ -344,7 +387,7 @@ class HotdataClient:
344
387
  ) -> LoadManagedTableResult:
345
388
  if (upload_id is None) == (file is None):
346
389
  raise ValueError("Exactly one of upload_id or file is required")
347
- db = self.resolve_managed_database(database)
390
+ db = self._as_managed_database(database)
348
391
  if upload_id is not None:
349
392
  resolved_upload_id = upload_id
350
393
  else:
@@ -374,7 +417,7 @@ class HotdataClient:
374
417
 
375
418
  def add_managed_table(
376
419
  self,
377
- database: str,
420
+ database: str | ManagedDatabase,
378
421
  table: str,
379
422
  *,
380
423
  schema: str = DEFAULT_SCHEMA,
@@ -387,7 +430,7 @@ class HotdataClient:
387
430
  schema after creation without recreating it. ``key`` sets the
388
431
  row-identity columns for delete/update/upsert; omit for keyless.
389
432
  """
390
- db = self.resolve_managed_database(database)
433
+ db = self._as_managed_database(database)
391
434
  request = AddManagedTableRequest(name=table, key=list(key or []))
392
435
  try:
393
436
  self._databases_api().add_database_table(db.id, schema, request)
@@ -403,17 +446,210 @@ class HotdataClient:
403
446
 
404
447
  def delete_managed_table(
405
448
  self,
406
- database: str,
449
+ database: str | ManagedDatabase,
407
450
  table: str,
408
451
  *,
409
452
  schema: str = DEFAULT_SCHEMA,
410
453
  ) -> None:
411
- db = self.resolve_managed_database(database)
454
+ db = self._as_managed_database(database)
412
455
  try:
413
456
  self.connections().delete_managed_table(db.default_connection_id, schema, table)
414
457
  except ApiException as e:
415
458
  raise RuntimeError(api_error_message(e)) from e
416
459
 
460
+ def create_index(
461
+ self,
462
+ database: str | ManagedDatabase,
463
+ table: str,
464
+ *,
465
+ schema: str = DEFAULT_SCHEMA,
466
+ index_name: str | None = None,
467
+ columns: list[str],
468
+ index_type: IndexType,
469
+ metric: VectorMetric | None = None,
470
+ dimensions: int | None = None,
471
+ embedding_provider_id: str | None = None,
472
+ output_column: str | None = None,
473
+ description: str | None = None,
474
+ wait: bool = True,
475
+ timeout_s: float = 300.0,
476
+ poll_interval_s: float = 2.0,
477
+ ) -> CreateIndexResult:
478
+ """Build an index on a managed table and wait for it to be ready.
479
+
480
+ The Python equivalent of ``hotdata indexes create``, scoped to managed
481
+ databases. Indexing a table on a plain (non-managed) connection is not
482
+ supported here; the CLI's ``--catalog`` flag covers that case.
483
+
484
+ ``index_type`` selects the index kind and is required: ``"bm25"`` for
485
+ full-text search (queries error outright without one), ``"vector"`` for
486
+ nearest-neighbour search (queries work without one, but only at
487
+ full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
488
+ to ``"sorted"``; this method makes the choice explicit instead, because
489
+ the wrong kind fails at query time rather than here.
490
+
491
+ ``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
492
+ derivation the CLI uses when ``--name`` is omitted, so both surfaces
493
+ name the same index identically.
494
+
495
+ There are two kinds of vector index, and they are queried differently:
496
+
497
+ * **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
498
+ vector column (a float list), and a query passes a literal vector:
499
+ ``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
500
+ distance function the caller writes — ``cosine`` serves
501
+ ``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
502
+ ``negative_dot_product``. A mismatch is not an error: the query
503
+ silently reverts to a full table scan. Omitting ``metric`` lets the
504
+ server choose (``l2`` for float-array columns), so pass it explicitly
505
+ whenever the query function is known.
506
+ * **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
507
+ provider ``sys_emb_openai``). ``columns`` is then the *source text*
508
+ column; the provider embeds it into ``output_column`` (default
509
+ ``{column}_embedding``) and the index is built over that. A query
510
+ passes text, not a vector — ``vector_distance(source_col, 'query')`` —
511
+ and the server resolves the matching distance function from the index
512
+ itself, so the metric-mismatch trap above does not apply. The returned
513
+ ``source_column`` names the column to query.
514
+
515
+ ``dimensions`` picks the output width for providers that support several;
516
+ it does not apply when indexing an existing vector column, whose width is
517
+ read from the data. ``description`` is a user-facing label for the
518
+ embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
519
+ index takes exactly one column, and every option in this paragraph — plus
520
+ ``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
521
+
522
+ The server builds the index as a background job. This method polls that
523
+ job to a terminal state and raises ``RuntimeError`` if it failed, because
524
+ the submit call itself reports success for builds that later fail. Pass
525
+ ``wait=False`` to return as soon as the job is accepted — the result then
526
+ carries ``status="pending"`` and a ``job_id``, and the caller owns
527
+ checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
528
+
529
+ Raises ``ValueError`` for an unusable argument combination,
530
+ ``RuntimeError`` if the API rejects the request or the build fails, and
531
+ ``TimeoutError`` if the build is still running after ``timeout_s``.
532
+ """
533
+ if not columns:
534
+ raise ValueError("create_index requires at least one column")
535
+ if index_type not in _INDEX_TYPES:
536
+ allowed = ", ".join(sorted(_INDEX_TYPES))
537
+ raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
538
+ if index_type != "vector":
539
+ vector_only = {
540
+ "metric": metric,
541
+ "dimensions": dimensions,
542
+ "embedding_provider_id": embedding_provider_id,
543
+ "output_column": output_column,
544
+ "description": description,
545
+ }
546
+ supplied = sorted(k for k, v in vector_only.items() if v is not None)
547
+ if supplied:
548
+ raise ValueError(
549
+ f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
550
+ f"vector indexes only (index_type={index_type!r})"
551
+ )
552
+ else:
553
+ if len(columns) != 1:
554
+ raise ValueError(
555
+ f"a vector index takes exactly one column (got {len(columns)}); "
556
+ "the engine indexes only the first"
557
+ )
558
+ if metric is not None and metric not in _VECTOR_METRICS:
559
+ allowed = ", ".join(sorted(_VECTOR_METRICS))
560
+ raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
561
+
562
+ # Matches the CLI's derivation so both surfaces name the same index
563
+ # identically: `hotdata indexes create` without --name.
564
+ resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
565
+
566
+ db = self._as_managed_database(database)
567
+ request = CreateIndexRequest(
568
+ index_name=resolved_name,
569
+ columns=list(columns),
570
+ index_type=index_type,
571
+ metric=metric,
572
+ dimensions=dimensions,
573
+ embedding_provider_id=embedding_provider_id,
574
+ output_column=output_column,
575
+ description=description,
576
+ var_async=True,
577
+ )
578
+ try:
579
+ submitted = self._indexes_api().create_index(
580
+ db.default_connection_id,
581
+ schema,
582
+ table,
583
+ request,
584
+ )
585
+ except ApiException as e:
586
+ raise RuntimeError(api_error_message(e)) from e
587
+
588
+ full_name = f"{db.id}.{schema}.{table}"
589
+
590
+ # A build the server finished inline answers 201 with the index itself;
591
+ # the async path answers 202 with a job to poll.
592
+ if isinstance(submitted, IndexInfoResponse):
593
+ return self._index_result(submitted, full_name, schema, table, job_id=None)
594
+
595
+ if not isinstance(submitted, SubmitJobResponse):
596
+ raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
597
+
598
+ job_id = submitted.id
599
+
600
+ def requested_result(status: str) -> CreateIndexResult:
601
+ """Echo the requested values, for the paths where the server hands
602
+ back a job rather than the built index."""
603
+ return CreateIndexResult(
604
+ full_name=full_name,
605
+ schema_name=schema,
606
+ table_name=table,
607
+ index_name=resolved_name,
608
+ index_type=index_type,
609
+ columns=list(columns),
610
+ metric=metric,
611
+ source_column=columns[0] if embedding_provider_id else None,
612
+ status=status,
613
+ job_id=job_id,
614
+ )
615
+
616
+ if not wait:
617
+ return requested_result(enum_value(submitted.status))
618
+
619
+ job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
620
+ status = enum_value(job.status)
621
+ if status != "succeeded":
622
+ detail = job.error_message or f"Index build {status}"
623
+ raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
624
+
625
+ # `result` is a oneOf wrapper today; tolerate the model arriving directly.
626
+ built = getattr(job.result, "actual_instance", job.result)
627
+ if isinstance(built, IndexInfoResponse):
628
+ return self._index_result(built, full_name, schema, table, job_id=job_id)
629
+ return requested_result("ready")
630
+
631
+ @staticmethod
632
+ def _index_result(
633
+ info: IndexInfoResponse,
634
+ full_name: str,
635
+ schema: str,
636
+ table: str,
637
+ *,
638
+ job_id: str | None,
639
+ ) -> CreateIndexResult:
640
+ return CreateIndexResult(
641
+ full_name=full_name,
642
+ schema_name=schema,
643
+ table_name=table,
644
+ index_name=info.index_name,
645
+ index_type=info.index_type,
646
+ columns=list(info.columns),
647
+ metric=info.metric,
648
+ source_column=info.source_column,
649
+ status=enum_value(info.status),
650
+ job_id=job_id,
651
+ )
652
+
417
653
  def list_recent_results(
418
654
  self,
419
655
  *,
@@ -547,6 +783,29 @@ class HotdataClient:
547
783
  f"(last status: {getattr(last, 'status', None)})"
548
784
  )
549
785
 
786
+ def _poll_job(
787
+ self,
788
+ job_id: str,
789
+ *,
790
+ timeout_s: float = 300.0,
791
+ interval_s: float = 2.0,
792
+ ) -> JobStatusResponse:
793
+ jobs = self._jobs_api()
794
+ deadline = time.monotonic() + timeout_s
795
+ last: JobStatusResponse | None = None
796
+ while time.monotonic() < deadline:
797
+ try:
798
+ last = jobs.get_job(job_id)
799
+ except ApiException as e:
800
+ raise RuntimeError(api_error_message(e)) from e
801
+ if last.status in _JOB_TERMINAL:
802
+ return last
803
+ time.sleep(interval_s)
804
+ last_status = enum_value(last.status) if last is not None else None
805
+ raise TimeoutError(
806
+ f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
807
+ )
808
+
550
809
  def _wait_result_ready(
551
810
  self,
552
811
  result_id: str,
@@ -569,16 +828,19 @@ class HotdataClient:
569
828
  f"(last status: {getattr(last, 'status', None)})"
570
829
  )
571
830
 
572
- def execute_sql(self, sql: str, *, database: str | None = None) -> QueryResult:
831
+ def execute_sql(
832
+ self, sql: str, *, database: str | ManagedDatabase | None = None
833
+ ) -> QueryResult:
573
834
  """Execute SQL and return a :class:`QueryResult`.
574
835
 
575
- Pass ``database`` to scope the query to a managed database. The name
576
- is resolved to a database ID once before the retry loop, and the
836
+ Pass ``database`` to scope the query to a managed database. A name or
837
+ id is resolved to a database ID once before the retry loop; an
838
+ already-resolved ``ManagedDatabase`` is used as-is (no read probe). The
577
839
  ``X-Database-Id`` header is sent with every attempt. Inside a managed
578
840
  database the built-in catalog is always ``"default"``, so table
579
841
  references should use ``"default"."<schema>"."<table>"``.
580
842
  """
581
- database_id = self.resolve_managed_database(database).id if database else None
843
+ database_id = self._as_managed_database(database).id if database else None
582
844
  last_err: BaseException | None = None
583
845
  for attempt in range(3):
584
846
  try: