hotdata-framework 0.9.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/CHANGELOG.md +55 -0
  2. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/CONTRACT.md +3 -1
  3. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/PKG-INFO +2 -1
  4. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/README.md +1 -0
  5. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/__init__.py +2 -0
  6. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/client.py +247 -1
  7. hotdata_framework-0.10.0/hotdata_framework/databases.py +114 -0
  8. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/pyproject.toml +1 -1
  9. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_contract.py +1 -0
  10. hotdata_framework-0.10.0/tests/test_indexes.py +824 -0
  11. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/uv.lock +1 -1
  12. hotdata_framework-0.9.0/hotdata_framework/databases.py +0 -66
  13. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/CODEOWNERS +0 -0
  14. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/dependabot.yml +0 -0
  15. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/check-release.yml +0 -0
  16. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/ci.yml +0 -0
  17. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/dependabot-automerge.yml +0 -0
  18. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/publish.yml +0 -0
  19. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.github/workflows/release.yml +0 -0
  20. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/.gitignore +0 -0
  21. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/RELEASING.md +0 -0
  22. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/examples/basic_usage.py +0 -0
  23. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/env.py +0 -0
  24. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/errors.py +0 -0
  25. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/health.py +0 -0
  26. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/http.py +0 -0
  27. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/managed_client.py +0 -0
  28. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/py.typed +0 -0
  29. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/hotdata_framework/result.py +0 -0
  30. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/check-release.py +0 -0
  31. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/extract-changelog.py +0 -0
  32. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/publish-workflow.sh +0 -0
  33. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/release.sh +0 -0
  34. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/scripts/update_changelog.py +0 -0
  35. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_client.py +0 -0
  36. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_databases.py +0 -0
  37. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_errors.py +0 -0
  38. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_health.py +0 -0
  39. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_managed_client.py +0 -0
  40. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_request_timeout.py +0 -0
  41. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_result.py +0 -0
  42. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_update_changelog.py +0 -0
  43. {hotdata_framework-0.9.0 → hotdata_framework-0.10.0}/tests/test_version.py +0 -0
@@ -8,6 +8,61 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
8
8
  ## [Unreleased]
9
9
 
10
10
 
11
+ ## [0.10.0] - 2026-08-07
12
+
13
+ ### Added
14
+
15
+ - `create_index(database, table, columns=..., index_type=...)` builds an index on a
16
+ managed table, bringing the framework client to parity with `hotdata indexes
17
+ create` in the CLI. It covers all three index kinds the API accepts: `"bm25"` for
18
+ full-text search, `"vector"` for nearest-neighbour search, and `"sorted"`.
19
+ Previously the framework had no index API at all and callers had to drop to the raw
20
+ `hotdata.IndexesApi`, which left a managed database's data loaded but not
21
+ searchable: full-text queries error without an index, and vector queries run at
22
+ full-scan speed. Like the other managed-table operations, `database` accepts a
23
+ name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
24
+ (non-managed) connection is not covered — the CLI's `--catalog` handles that.
25
+
26
+ `index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
27
+ derivation the CLI uses when `--name` is omitted, so both surfaces name the same
28
+ index identically. `index_type` is required, unlike the API's `"sorted"` default,
29
+ because the wrong kind only fails at query time.
30
+
31
+ The server builds the index as a background job whose submit call reports success
32
+ even when the build later fails, so `create_index` polls the job to a terminal
33
+ state and raises `RuntimeError` carrying the job's `error_message`. Pass
34
+ `wait=False` to return once the job is accepted (`status="pending"` plus a
35
+ `job_id`) and own the outcome check yourself, as the CLI's `--async` does;
36
+ `timeout_s` and `poll_interval_s` tune the wait.
37
+
38
+ Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
39
+ existing vector column, queried with a literal vector — and there `metric` must
40
+ match the distance function the query uses (`cosine`→`cosine_distance`,
41
+ `l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
42
+ reverts to a full table scan rather than erroring. Setting
43
+ `embedding_provider_id` indexes a *text* column instead: the provider embeds it
44
+ into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
45
+ and the server resolves the distance function itself.
46
+
47
+ Argument combinations that the server would silently ignore raise `ValueError`
48
+ before any request is sent: an unknown `index_type` or `metric`, a vector index
49
+ with more than one column (the engine indexes only the first), and
50
+ `metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
51
+ non-vector index.
52
+
53
+ Verified against `api.hotdata.dev` when this version was released: BM25 and vector
54
+ indexes both build and report `ready`, and a BM25 index is used by full-text
55
+ search. A *vector* index on a managed database was **not** picked up by the query
56
+ planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
57
+ planned as a full scan. That reproduces with an index created by `hotdata indexes
58
+ create`, so it is an engine-side issue rather than a client one, but it means a
59
+ vector index built through this method may not yet accelerate queries.
60
+
61
+ - `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
62
+ `hotdata_framework` and added to the public contract surface. Its `source_column`
63
+ names the text column to query for a provider-backed vector index, and is `None`
64
+ for BM25, sorted, and plain vector indexes.
65
+
11
66
  ## [0.9.0] - 2026-07-23
12
67
 
13
68
  ### Added
@@ -33,6 +33,7 @@ The supported import surface is:
33
33
  - `ManagedDatabase`
34
34
  - `ManagedTable`
35
35
  - `LoadManagedTableResult`
36
+ - `CreateIndexResult`
36
37
  - `DEFAULT_SCHEMA`
37
38
  - `is_parquet_path`
38
39
 
@@ -63,7 +64,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
63
64
  - `upload_parquet(path)` uploads a local parquet file and returns an upload id.
64
65
  - `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
65
66
  - `delete_managed_table(database, table, schema=...)` deletes a managed table.
66
- - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
67
+ - `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
68
+ - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
67
69
 
68
70
  ### `QueryResult`
69
71
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hotdata-framework
3
- Version: 0.9.0
3
+ Version: 0.10.0
4
4
  Summary: Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -44,6 +44,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
44
44
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
45
45
  - **History helpers** — list recent results and query run history with normalized dataclasses.
46
46
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
47
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
47
48
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
48
49
 
49
50
  Install:
@@ -16,6 +16,7 @@ Runtime boundary and guarantees are defined in `CONTRACT.md`.
16
16
  - **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
17
17
  - **History helpers** — list recent results and query run history with normalized dataclasses.
18
18
  - **Managed databases** — create Hotdata-owned catalogs, declare tables, upload parquet, and load managed tables (mirrors `hotdata databases` in the CLI).
19
+ - **Indexes** — build BM25, vector, or sorted indexes on managed tables, mirroring `hotdata indexes create` (managed databases only). Waits on the background build job and surfaces its failure, instead of reporting the phantom success the submit call returns.
19
20
  - **Health helpers** — build compact API/workspace health summaries for UI integrations.
20
21
 
21
22
  Install:
@@ -10,6 +10,7 @@ from hotdata_framework.client import (
10
10
  )
11
11
  from hotdata_framework.databases import (
12
12
  DEFAULT_SCHEMA,
13
+ CreateIndexResult,
13
14
  LoadManagedTableResult,
14
15
  ManagedDatabase,
15
16
  ManagedTable,
@@ -43,6 +44,7 @@ except PackageNotFoundError:
43
44
 
44
45
  __all__ = [
45
46
  "DEFAULT_SCHEMA",
47
+ "CreateIndexResult",
46
48
  "HotdataClient",
47
49
  "HotdataError",
48
50
  "HotdataTerminalError",
@@ -4,12 +4,14 @@ import functools
4
4
  import time
5
5
  from collections.abc import Iterator
6
6
  from dataclasses import asdict, dataclass
7
- from typing import Any, Literal
7
+ from typing import Any, Literal, get_args
8
8
 
9
9
  from hotdata import ApiClient, Configuration
10
10
  from hotdata.api.connections_api import ConnectionsApi
11
11
  from hotdata.api.databases_api import DatabasesApi
12
+ from hotdata.api.indexes_api import IndexesApi
12
13
  from hotdata.api.information_schema_api import InformationSchemaApi
14
+ from hotdata.api.jobs_api import JobsApi
13
15
  from hotdata.api.query_api import QueryApi
14
16
  from hotdata.api.query_runs_api import QueryRunsApi
15
17
  from hotdata.api.results_api import ResultsApi
@@ -20,21 +22,27 @@ from hotdata.exceptions import ApiException
20
22
  from hotdata.models.add_managed_table_request import AddManagedTableRequest
21
23
  from hotdata.models.async_query_response import AsyncQueryResponse
22
24
  from hotdata.models.create_database_request import CreateDatabaseRequest
25
+ from hotdata.models.create_index_request import CreateIndexRequest
23
26
  from hotdata.models.database_default_schema_decl import DatabaseDefaultSchemaDecl
24
27
  from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
28
+ from hotdata.models.index_info_response import IndexInfoResponse
29
+ from hotdata.models.job_status_response import JobStatusResponse
25
30
  from hotdata.models.load_managed_table_request import LoadManagedTableRequest
26
31
  from hotdata.models.query_request import QueryRequest
27
32
  from hotdata.models.query_response import QueryResponse
33
+ from hotdata.models.submit_job_response import SubmitJobResponse
28
34
  from hotdata.models.table_info import TableInfo
29
35
  from urllib3.exceptions import HTTPError as Urllib3HTTPError
30
36
  from urllib3.exceptions import ProtocolError
31
37
 
32
38
  from hotdata_framework.databases import (
33
39
  DEFAULT_SCHEMA,
40
+ CreateIndexResult,
34
41
  LoadManagedTableResult,
35
42
  ManagedDatabase,
36
43
  ManagedTable,
37
44
  api_error_message,
45
+ enum_value,
38
46
  is_parquet_path,
39
47
  managed_database_from_detail,
40
48
  )
@@ -52,8 +60,24 @@ from hotdata_framework.result import QueryResult
52
60
  # rows, delete/update/upsert match by the table's declared key.
53
61
  ManagedLoadMode = Literal["replace", "append", "delete", "update", "upsert"]
54
62
 
63
+ # Index kinds the indexes endpoint accepts. "sorted" is the server-side default;
64
+ # "bm25" backs full-text search and "vector" backs nearest-neighbour search.
65
+ IndexType = Literal["sorted", "bm25", "vector"]
66
+
67
+ # Distance metrics a vector index can be built with. Each one accelerates
68
+ # exactly one query function: cosine -> cosine_distance, l2 -> l2_distance,
69
+ # dot -> negative_dot_product. A hand-written query naming a different function
70
+ # falls back to a full scan; the provider-backed vector_distance path resolves
71
+ # the function from the index instead, so it cannot mismatch.
72
+ VectorMetric = Literal["l2", "cosine", "dot"]
73
+
74
+ _INDEX_TYPES = frozenset(get_args(IndexType))
75
+ _VECTOR_METRICS = frozenset(get_args(VectorMetric))
76
+
55
77
  _TERMINAL = frozenset({"succeeded", "failed", "cancelled"})
56
78
  _RESULT_FAILURE = frozenset({"failed", "cancelled"})
79
+ # Jobs have no "cancelled" state; "partially_succeeded" carries an error_message.
80
+ _JOB_TERMINAL = frozenset({"succeeded", "partially_succeeded", "failed"})
57
81
 
58
82
 
59
83
  @dataclass(frozen=True)
@@ -188,6 +212,12 @@ class HotdataClient:
188
212
  def _information_schema(self) -> InformationSchemaApi:
189
213
  return InformationSchemaApi(self._api)
190
214
 
215
+ def _indexes_api(self) -> IndexesApi:
216
+ return IndexesApi(self._api)
217
+
218
+ def _jobs_api(self) -> JobsApi:
219
+ return JobsApi(self._api)
220
+
191
221
  def _query_api(self) -> QueryApi:
192
222
  return QueryApi(self._api)
193
223
 
@@ -427,6 +457,199 @@ class HotdataClient:
427
457
  except ApiException as e:
428
458
  raise RuntimeError(api_error_message(e)) from e
429
459
 
460
+ def create_index(
461
+ self,
462
+ database: str | ManagedDatabase,
463
+ table: str,
464
+ *,
465
+ schema: str = DEFAULT_SCHEMA,
466
+ index_name: str | None = None,
467
+ columns: list[str],
468
+ index_type: IndexType,
469
+ metric: VectorMetric | None = None,
470
+ dimensions: int | None = None,
471
+ embedding_provider_id: str | None = None,
472
+ output_column: str | None = None,
473
+ description: str | None = None,
474
+ wait: bool = True,
475
+ timeout_s: float = 300.0,
476
+ poll_interval_s: float = 2.0,
477
+ ) -> CreateIndexResult:
478
+ """Build an index on a managed table and wait for it to be ready.
479
+
480
+ The Python equivalent of ``hotdata indexes create``, scoped to managed
481
+ databases. Indexing a table on a plain (non-managed) connection is not
482
+ supported here; the CLI's ``--catalog`` flag covers that case.
483
+
484
+ ``index_type`` selects the index kind and is required: ``"bm25"`` for
485
+ full-text search (queries error outright without one), ``"vector"`` for
486
+ nearest-neighbour search (queries work without one, but only at
487
+ full-scan speed), or ``"sorted"``. The API defaults an unspecified kind
488
+ to ``"sorted"``; this method makes the choice explicit instead, because
489
+ the wrong kind fails at query time rather than here.
490
+
491
+ ``index_name`` defaults to ``{table}_{columns}_{index_type}``, the same
492
+ derivation the CLI uses when ``--name`` is omitted, so both surfaces
493
+ name the same index identically.
494
+
495
+ There are two kinds of vector index, and they are queried differently:
496
+
497
+ * **Plain** — omit ``embedding_provider_id``. ``columns`` is the existing
498
+ vector column (a float list), and a query passes a literal vector:
499
+ ``cosine_distance(col, ARRAY[...])``. Here ``metric`` must match the
500
+ distance function the caller writes — ``cosine`` serves
501
+ ``cosine_distance``, ``l2`` serves ``l2_distance``, ``dot`` serves
502
+ ``negative_dot_product``. A mismatch is not an error: the query
503
+ silently reverts to a full table scan. Omitting ``metric`` lets the
504
+ server choose (``l2`` for float-array columns), so pass it explicitly
505
+ whenever the query function is known.
506
+ * **Provider-backed** — set ``embedding_provider_id`` (e.g. the system
507
+ provider ``sys_emb_openai``). ``columns`` is then the *source text*
508
+ column; the provider embeds it into ``output_column`` (default
509
+ ``{column}_embedding``) and the index is built over that. A query
510
+ passes text, not a vector — ``vector_distance(source_col, 'query')`` —
511
+ and the server resolves the matching distance function from the index
512
+ itself, so the metric-mismatch trap above does not apply. The returned
513
+ ``source_column`` names the column to query.
514
+
515
+ ``dimensions`` picks the output width for providers that support several;
516
+ it does not apply when indexing an existing vector column, whose width is
517
+ read from the data. ``description`` is a user-facing label for the
518
+ embedding (e.g. ``"product descriptions"``), stored alongside it. A vector
519
+ index takes exactly one column, and every option in this paragraph — plus
520
+ ``metric`` — is rejected for a non-vector ``index_type``, matching the CLI.
521
+
522
+ The server builds the index as a background job. This method polls that
523
+ job to a terminal state and raises ``RuntimeError`` if it failed, because
524
+ the submit call itself reports success for builds that later fail. Pass
525
+ ``wait=False`` to return as soon as the job is accepted — the result then
526
+ carries ``status="pending"`` and a ``job_id``, and the caller owns
527
+ checking the outcome (the CLI's ``--async`` plus ``hotdata jobs``).
528
+
529
+ Raises ``ValueError`` for an unusable argument combination,
530
+ ``RuntimeError`` if the API rejects the request or the build fails, and
531
+ ``TimeoutError`` if the build is still running after ``timeout_s``.
532
+ """
533
+ if not columns:
534
+ raise ValueError("create_index requires at least one column")
535
+ if index_type not in _INDEX_TYPES:
536
+ allowed = ", ".join(sorted(_INDEX_TYPES))
537
+ raise ValueError(f"index_type must be one of {allowed} (got {index_type!r})")
538
+ if index_type != "vector":
539
+ vector_only = {
540
+ "metric": metric,
541
+ "dimensions": dimensions,
542
+ "embedding_provider_id": embedding_provider_id,
543
+ "output_column": output_column,
544
+ "description": description,
545
+ }
546
+ supplied = sorted(k for k, v in vector_only.items() if v is not None)
547
+ if supplied:
548
+ raise ValueError(
549
+ f"{', '.join(supplied)} appl{'ies' if len(supplied) == 1 else 'y'} to "
550
+ f"vector indexes only (index_type={index_type!r})"
551
+ )
552
+ else:
553
+ if len(columns) != 1:
554
+ raise ValueError(
555
+ f"a vector index takes exactly one column (got {len(columns)}); "
556
+ "the engine indexes only the first"
557
+ )
558
+ if metric is not None and metric not in _VECTOR_METRICS:
559
+ allowed = ", ".join(sorted(_VECTOR_METRICS))
560
+ raise ValueError(f"metric must be one of {allowed} (got {metric!r})")
561
+
562
+ # Matches the CLI's derivation so both surfaces name the same index
563
+ # identically: `hotdata indexes create` without --name.
564
+ resolved_name = index_name or f"{table}_{'_'.join(columns)}_{index_type}"
565
+
566
+ db = self._as_managed_database(database)
567
+ request = CreateIndexRequest(
568
+ index_name=resolved_name,
569
+ columns=list(columns),
570
+ index_type=index_type,
571
+ metric=metric,
572
+ dimensions=dimensions,
573
+ embedding_provider_id=embedding_provider_id,
574
+ output_column=output_column,
575
+ description=description,
576
+ var_async=True,
577
+ )
578
+ try:
579
+ submitted = self._indexes_api().create_index(
580
+ db.default_connection_id,
581
+ schema,
582
+ table,
583
+ request,
584
+ )
585
+ except ApiException as e:
586
+ raise RuntimeError(api_error_message(e)) from e
587
+
588
+ full_name = f"{db.id}.{schema}.{table}"
589
+
590
+ # A build the server finished inline answers 201 with the index itself;
591
+ # the async path answers 202 with a job to poll.
592
+ if isinstance(submitted, IndexInfoResponse):
593
+ return self._index_result(submitted, full_name, schema, table, job_id=None)
594
+
595
+ if not isinstance(submitted, SubmitJobResponse):
596
+ raise RuntimeError(f"Unexpected create_index response type: {type(submitted)!r}")
597
+
598
+ job_id = submitted.id
599
+
600
+ def requested_result(status: str) -> CreateIndexResult:
601
+ """Echo the requested values, for the paths where the server hands
602
+ back a job rather than the built index."""
603
+ return CreateIndexResult(
604
+ full_name=full_name,
605
+ schema_name=schema,
606
+ table_name=table,
607
+ index_name=resolved_name,
608
+ index_type=index_type,
609
+ columns=list(columns),
610
+ metric=metric,
611
+ source_column=columns[0] if embedding_provider_id else None,
612
+ status=status,
613
+ job_id=job_id,
614
+ )
615
+
616
+ if not wait:
617
+ return requested_result(enum_value(submitted.status))
618
+
619
+ job = self._poll_job(job_id, timeout_s=timeout_s, interval_s=poll_interval_s)
620
+ status = enum_value(job.status)
621
+ if status != "succeeded":
622
+ detail = job.error_message or f"Index build {status}"
623
+ raise RuntimeError(f"Index {resolved_name!r} on {full_name}: {detail}")
624
+
625
+ # `result` is a oneOf wrapper today; tolerate the model arriving directly.
626
+ built = getattr(job.result, "actual_instance", job.result)
627
+ if isinstance(built, IndexInfoResponse):
628
+ return self._index_result(built, full_name, schema, table, job_id=job_id)
629
+ return requested_result("ready")
630
+
631
+ @staticmethod
632
+ def _index_result(
633
+ info: IndexInfoResponse,
634
+ full_name: str,
635
+ schema: str,
636
+ table: str,
637
+ *,
638
+ job_id: str | None,
639
+ ) -> CreateIndexResult:
640
+ return CreateIndexResult(
641
+ full_name=full_name,
642
+ schema_name=schema,
643
+ table_name=table,
644
+ index_name=info.index_name,
645
+ index_type=info.index_type,
646
+ columns=list(info.columns),
647
+ metric=info.metric,
648
+ source_column=info.source_column,
649
+ status=enum_value(info.status),
650
+ job_id=job_id,
651
+ )
652
+
430
653
  def list_recent_results(
431
654
  self,
432
655
  *,
@@ -560,6 +783,29 @@ class HotdataClient:
560
783
  f"(last status: {getattr(last, 'status', None)})"
561
784
  )
562
785
 
786
+ def _poll_job(
787
+ self,
788
+ job_id: str,
789
+ *,
790
+ timeout_s: float = 300.0,
791
+ interval_s: float = 2.0,
792
+ ) -> JobStatusResponse:
793
+ jobs = self._jobs_api()
794
+ deadline = time.monotonic() + timeout_s
795
+ last: JobStatusResponse | None = None
796
+ while time.monotonic() < deadline:
797
+ try:
798
+ last = jobs.get_job(job_id)
799
+ except ApiException as e:
800
+ raise RuntimeError(api_error_message(e)) from e
801
+ if last.status in _JOB_TERMINAL:
802
+ return last
803
+ time.sleep(interval_s)
804
+ last_status = enum_value(last.status) if last is not None else None
805
+ raise TimeoutError(
806
+ f"Job {job_id} did not finish within {timeout_s}s (last status: {last_status})"
807
+ )
808
+
563
809
  def _wait_result_ready(
564
810
  self,
565
811
  result_id: str,
@@ -0,0 +1,114 @@
1
+ """Managed database helpers (Hotdata-owned catalogs with parquet table loads)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import asdict, dataclass
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from hotdata.exceptions import ApiException
10
+
11
+ DEFAULT_SCHEMA = "public"
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class ManagedDatabase:
16
+ id: str
17
+ description: str | None
18
+ default_connection_id: str
19
+
20
+ def to_dict(self) -> dict[str, Any]:
21
+ return asdict(self)
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class ManagedTable:
26
+ full_name: str
27
+ schema: str
28
+ table: str
29
+ synced: bool
30
+ last_sync: str | None
31
+
32
+ def to_dict(self) -> dict[str, Any]:
33
+ return asdict(self)
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class LoadManagedTableResult:
38
+ connection_id: str
39
+ schema_name: str
40
+ table_name: str
41
+ row_count: int
42
+ full_name: str
43
+
44
+ def to_dict(self) -> dict[str, Any]:
45
+ return asdict(self)
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class CreateIndexResult:
50
+ """An index created on a managed table.
51
+
52
+ ``status`` is the server's own — ``"ready"`` once the index is built,
53
+ ``"pending"`` while it is still building. ``job_id`` identifies the
54
+ background build job, and is ``None`` only when the server built the index
55
+ inline. A caller that passed ``wait=False`` always gets ``"pending"`` and
56
+ owns checking the job's outcome.
57
+
58
+ ``source_column`` is set only for an embedding-backed vector index, where it
59
+ names the *text* column a query passes to ``vector_distance(col, 'text')``.
60
+ It is ``None`` for BM25, sorted, and plain (existing-vector-column) indexes.
61
+
62
+ ``index_type``, ``columns``, and ``metric`` echo the requested values when the
63
+ server does not return the built index alongside the finished job — that is,
64
+ on the ``wait=False`` path and when a finished job carries no index payload.
65
+ Only when the server did return it does ``columns`` hold the *generated*
66
+ embedding column for an embedding-backed index; on the echoing paths
67
+ ``columns[0]`` is the source text column, the same value as
68
+ ``source_column``. Read ``status`` to tell the cases apart.
69
+ """
70
+
71
+ full_name: str
72
+ schema_name: str
73
+ table_name: str
74
+ index_name: str
75
+ index_type: str
76
+ columns: list[str]
77
+ metric: str | None
78
+ source_column: str | None
79
+ status: str
80
+ job_id: str | None
81
+
82
+ def to_dict(self) -> dict[str, Any]:
83
+ return asdict(self)
84
+
85
+
86
+ def enum_value(value: Any) -> str:
87
+ """Render an enum-or-str API field as its wire string.
88
+
89
+ The generated models type several status fields as ``str``-mixin enums,
90
+ whose ``str()`` is ``"JobStatus.FAILED"`` rather than ``"failed"``.
91
+ """
92
+ inner = getattr(value, "value", value)
93
+ return str(inner)
94
+
95
+
96
+ def is_parquet_path(path: str) -> bool:
97
+ return Path(path).suffix.lower() == ".parquet"
98
+
99
+
100
+ def managed_database_from_detail(detail: Any) -> ManagedDatabase:
101
+ return ManagedDatabase(
102
+ id=str(detail.id),
103
+ description=detail.name,
104
+ default_connection_id=str(detail.default_connection_id),
105
+ )
106
+
107
+
108
+ def api_error_message(exc: ApiException) -> str:
109
+ reason = exc.reason or str(exc)
110
+ # Keep the response body: it carries the API's actual explanation.
111
+ body = getattr(exc, "body", None)
112
+ if body:
113
+ return f"{reason}: {' '.join(str(body).split())[:500]}"
114
+ return reason
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.9.0"
7
+ version = "0.10.0"
8
8
  description = "Python framework for building Hotdata integrations: workspace/session runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -11,6 +11,7 @@ from hotdata_framework.result import QueryResult
11
11
  def test_public_exports_contract():
12
12
  assert hr.__all__ == [
13
13
  "DEFAULT_SCHEMA",
14
+ "CreateIndexResult",
14
15
  "HotdataClient",
15
16
  "HotdataError",
16
17
  "HotdataTerminalError",