hotdata-framework 0.11.0__tar.gz → 0.12.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. hotdata_framework-0.12.0/.github/dependabot.yml +17 -0
  2. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/publish.yml +26 -5
  3. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/release.yml +24 -4
  4. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/CHANGELOG.md +33 -0
  5. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/CONTRACT.md +6 -1
  6. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/PKG-INFO +2 -2
  7. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/RELEASING.md +21 -0
  8. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/__init__.py +7 -0
  9. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/client.py +99 -7
  10. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/databases.py +43 -0
  11. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/pyproject.toml +2 -2
  12. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_contract.py +3 -0
  13. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_databases.py +173 -0
  14. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/uv.lock +5 -5
  15. hotdata_framework-0.11.0/.github/dependabot.yml +0 -8
  16. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/CODEOWNERS +0 -0
  17. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/check-release.yml +0 -0
  18. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/ci.yml +0 -0
  19. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/dependabot-automerge.yml +0 -0
  20. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.gitignore +0 -0
  21. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/README.md +0 -0
  22. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/examples/basic_usage.py +0 -0
  23. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/env.py +0 -0
  24. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/errors.py +0 -0
  25. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/health.py +0 -0
  26. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/managed_client.py +0 -0
  27. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/py.typed +0 -0
  28. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/result.py +0 -0
  29. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/check-release.py +0 -0
  30. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/extract-changelog.py +0 -0
  31. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/publish-workflow.sh +0 -0
  32. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/release.sh +0 -0
  33. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/update_changelog.py +0 -0
  34. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_client.py +0 -0
  35. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_errors.py +0 -0
  36. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_health.py +0 -0
  37. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_indexes.py +0 -0
  38. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_managed_client.py +0 -0
  39. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_request_timeout.py +0 -0
  40. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_result.py +0 -0
  41. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_retry_policy.py +0 -0
  42. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_update_changelog.py +0 -0
  43. {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_version.py +0 -0
@@ -0,0 +1,17 @@
1
+ version: 2
2
+ updates:
3
+ - package-ecosystem: uv
4
+ directory: "/"
5
+ schedule:
6
+ interval: daily
7
+ allow:
8
+ - dependency-name: hotdata
9
+
10
+ # Action pins had no watcher, which is how gh-action-pypi-publish sat at
11
+ # v1.13.0 until its bundled twine broke a release at upload time. Unlike the
12
+ # uv entry there is no `allow` filter: the point is to see every stale pin,
13
+ # not a chosen one.
14
+ - package-ecosystem: github-actions
15
+ directory: "/"
16
+ schedule:
17
+ interval: weekly
@@ -4,9 +4,20 @@ on:
4
4
  push:
5
5
  tags:
6
6
  - 'v[0-9]*'
7
+ # Retry an existing tag without moving it. A publish can fail for reasons that
8
+ # have nothing to do with the code — a stale action pin, a PyPI outage — and
9
+ # with only the tag trigger the choices were to delete and re-push the tag or
10
+ # to burn a version number on a CI fix. Neither is a good answer to
11
+ # "the upload failed, run it again".
12
+ workflow_dispatch:
13
+ inputs:
14
+ tag:
15
+ description: Existing tag to build and publish (e.g. v1.2.3)
16
+ required: true
17
+ type: string
7
18
 
8
19
  concurrency:
9
- group: pypi-publish-${{ github.ref_name }}
20
+ group: pypi-publish-${{ inputs.tag || github.ref_name }}
10
21
  cancel-in-progress: false
11
22
 
12
23
  permissions:
@@ -16,8 +27,13 @@ jobs:
16
27
  build:
17
28
  name: Build distribution
18
29
  runs-on: ubuntu-latest
30
+ env:
31
+ # The tag being released, whether it arrived by push or by dispatch.
32
+ TAG: ${{ inputs.tag || github.ref_name }}
19
33
  steps:
20
34
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
35
+ with:
36
+ ref: ${{ inputs.tag || github.ref_name }}
21
37
 
22
38
  - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
23
39
  with:
@@ -28,11 +44,11 @@ jobs:
28
44
 
29
45
  - name: Verify tag matches pyproject version
30
46
  run: |
31
- if [[ ! "$GITHUB_REF_NAME" =~ ^v[0-9] ]]; then
32
- echo "Release tag '$GITHUB_REF_NAME' must start with 'v' followed by a digit (e.g. v1.0.0)" >&2
47
+ if [[ ! "$TAG" =~ ^v[0-9] ]]; then
48
+ echo "Release tag '$TAG' must start with 'v' followed by a digit (e.g. v1.0.0)" >&2
33
49
  exit 1
34
50
  fi
35
- tag="${GITHUB_REF_NAME#v}"
51
+ tag="${TAG#v}"
36
52
  pkg_version=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['version'])")
37
53
  if [ "$tag" != "$pkg_version" ]; then
38
54
  echo "Release tag ($tag) does not match pyproject.toml version ($pkg_version)" >&2
@@ -65,5 +81,10 @@ jobs:
65
81
  name: dist
66
82
  path: dist/
67
83
 
84
+ # v1.13.0's bundled twine rejects `Metadata-Version: 2.5`, which current
85
+ # hatchling emits: `InvalidDistribution: '2.5' is not a valid metadata
86
+ # version`. The build job's own `twine check --strict` passes, because it
87
+ # pip-installs a current twine — so the failure appears only at upload,
88
+ # after the tag is already public.
68
89
  - name: Publish via Trusted Publishing
69
- uses: pypa/gh-action-pypi-publish@ed0c53931b1dc9bd32cbe73a98c7f6766f8a527e # v1.13.0
90
+ uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
@@ -4,6 +4,15 @@ on:
4
4
  push:
5
5
  tags:
6
6
  - 'v[0-9]*'
7
+ # Repair the Release for a tag that already exists, without moving it. See
8
+ # RELEASING.md, "If a release workflow fails" — added alongside this trigger,
9
+ # since a capability nobody can find is not much better than not having it.
10
+ workflow_dispatch:
11
+ inputs:
12
+ tag:
13
+ description: Existing tag to create or update a Release for (e.g. v1.2.3)
14
+ required: true
15
+ type: string
7
16
 
8
17
  permissions:
9
18
  contents: write
@@ -12,8 +21,13 @@ jobs:
12
21
  release:
13
22
  name: Create GitHub Release
14
23
  runs-on: ubuntu-latest
24
+ env:
25
+ # The tag being released, whether it arrived by push or by dispatch.
26
+ TAG: ${{ inputs.tag || github.ref_name }}
15
27
  steps:
16
28
  - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
29
+ with:
30
+ ref: ${{ inputs.tag || github.ref_name }}
17
31
 
18
32
  - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
19
33
  with:
@@ -23,7 +37,7 @@ jobs:
23
37
  id: meta
24
38
  run: |
25
39
  pkg_name=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['name'])")
26
- pkg_version="${GITHUB_REF_NAME#v}"
40
+ pkg_version="${TAG#v}"
27
41
  echo "name=${pkg_name}" >> "$GITHUB_OUTPUT"
28
42
  echo "version=${pkg_version}" >> "$GITHUB_OUTPUT"
29
43
 
@@ -31,7 +45,7 @@ jobs:
31
45
  id: notes
32
46
  run: |
33
47
  set -euo pipefail
34
- version="${GITHUB_REF_NAME#v}"
48
+ version="${TAG#v}"
35
49
  if [[ -f CHANGELOG.md ]]; then
36
50
  body="$(python scripts/extract-changelog.py "$version")"
37
51
  else
@@ -47,8 +61,14 @@ jobs:
47
61
  - name: Create GitHub Release
48
62
  uses: softprops/action-gh-release@da05d552573ad5aba039eaac05058a918a7bf631 # v2.2.2
49
63
  with:
50
- tag_name: ${{ github.ref_name }}
64
+ tag_name: ${{ inputs.tag || github.ref_name }}
51
65
  name: ${{ steps.meta.outputs.name }} ${{ steps.meta.outputs.version }}
52
66
  body: ${{ steps.notes.outputs.body }}
53
67
  generate_release_notes: false
54
- make_latest: true
68
+ # Let GitHub decide by tag date/semver rather than by run order. On a
69
+ # push the tag is the newest version and becomes latest; on a dispatch
70
+ # repairing an older tag, a newer release keeps the badge. Note `false`
71
+ # is not "leave alone" — it explicitly marks a release NOT latest, so
72
+ # gating on the event would demote the newest tag in the very case this
73
+ # trigger exists for: repairing its Release after a failed push run.
74
+ make_latest: legacy
@@ -7,6 +7,39 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+
11
+ ## [0.12.0] - 2026-08-11
12
+
13
+ ### Added
14
+
15
+ - Table storage layout, both directions. `add_managed_table()` and
16
+ `create_managed_database()` take `partition_by` / `sorted_by`, and
17
+ `managed_table_layout()` reads back what was actually declared as a
18
+ `TableLayout`. `TablePartitionKey` and `TableSortKey` are re-exported so
19
+ callers need one import.
20
+
21
+ Both halves matter because a layout is fixed when the table is created and
22
+ there is no alter path: a table declared without one keeps that shape until it
23
+ is recreated and its data rewritten. So declaring is not enough — a caller has
24
+ to be able to confirm it took, and to refuse to load when it cannot.
25
+
26
+ `managed_table_layout()` raises `KeyError` for a table that is not declared,
27
+ rather than returning an empty layout. "Not there" and "declared without a
28
+ layout" lead to opposite decisions for a caller.
29
+
30
+ Until now this package could not express a layout at all, which is why at least
31
+ one consumer hand-built the HTTP request instead. The generated key models are
32
+ passed through rather than wrapped, so the transform vocabulary stays exactly
33
+ the API's.
34
+
35
+ ### Changed
36
+
37
+ - Require `hotdata>=0.9.0,<0.10`. 0.9.0 is the first release whose models carry
38
+ `partition_by` / `sorted_by` on the add-table request, the create-database
39
+ table declarations, and the table-info response. On an older `hotdata` the
40
+ fields would be silently dropped by the model and the table declared without a
41
+ layout, returning success — which is the failure this feature exists to end.
42
+
10
43
  ## [0.11.0] - 2026-08-11
11
44
 
12
45
  ### Changed
@@ -31,6 +31,9 @@ The supported import surface is:
31
31
  - `WorkspaceSelection`
32
32
  - `ManagedDatabase`
33
33
  - `ManagedTable`
34
+ - `TableLayout`
35
+ - `TablePartitionKey`
36
+ - `TableSortKey`
34
37
  - `LoadManagedTableResult`
35
38
  - `CreateIndexResult`
36
39
  - `DEFAULT_SCHEMA`
@@ -56,6 +59,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
56
59
  adapters should pass `connection_id` when known.
57
60
  - `uploads()` returns the uploads API wrapper for parquet staging.
58
61
  - `list_managed_databases()` returns all databases via the `/databases` API.
62
+ - `add_managed_table(...)` and `create_managed_database(...)` accept `partition_by` / `sorted_by` to declare a table's storage layout. The layout is fixed when the table is created and cannot be altered afterwards, so omitting it is permanent for that table.
63
+ - `managed_table_layout(database, table, schema=...)` returns the declared layout as `TableLayout`. Empty lists mean no layout was declared — sound only because the table is resolved through a managed database. Raises `KeyError` when the table is not declared, keeping "absent" distinct from "declared without a layout".
59
64
  - `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan). A `403` from `/databases` surfaces as `RuntimeError` (forbidden, not absent), preserving the underlying `ApiException` as `__cause__`.
60
65
  - `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front. Returns a `ManagedDatabase` (id + `default_connection_id`) sufficient to load without a further read.
61
66
  - `delete_managed_database(name_or_id)` deletes a database via the `/databases` API.
@@ -64,7 +69,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
64
69
  - `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
65
70
  - `delete_managed_table(database, table, schema=...)` deletes a managed table.
66
71
  - `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
67
- - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
72
+ - The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, `managed_table_layout`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
68
73
 
69
74
  ### `QueryResult`
70
75
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hotdata-framework
3
- Version: 0.11.0
3
+ Version: 0.12.0
4
4
  Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
5
5
  Project-URL: Homepage, https://www.hotdata.dev
6
6
  Project-URL: Documentation, https://www.hotdata.dev/docs
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
21
21
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
22
  Classifier: Typing :: Typed
23
23
  Requires-Python: >=3.10
24
- Requires-Dist: hotdata<0.9,>=0.8.0
24
+ Requires-Dist: hotdata<0.10,>=0.9.0
25
25
  Requires-Dist: pandas>=2.0
26
26
  Requires-Dist: pyarrow>=14.0
27
27
  Description-Content-Type: text/markdown
@@ -34,6 +34,27 @@ Pushing a `vX.Y.Z` tag triggers two workflows:
34
34
  | `publish.yml` | Build wheel/sdist and publish to PyPI |
35
35
  | `release.yml` | Create the GitHub Release with notes from `CHANGELOG.md` |
36
36
 
37
+ ## If a release workflow fails
38
+
39
+ Both workflows also accept a manual re-run against an existing tag, so a failure
40
+ unrelated to the code — a stale action pin, a PyPI outage — does not require
41
+ deleting the tag or burning a version number:
42
+
43
+ ```bash
44
+ gh workflow run "Publish to PyPI" --ref main -f tag=vX.Y.Z
45
+ gh workflow run "GitHub Release" --ref main -f tag=vX.Y.Z
46
+ ```
47
+
48
+ `--ref main` selects the workflow *definition*, so a fix to the workflow file
49
+ itself is picked up; everything it runs — including `scripts/extract-changelog.py`
50
+ — still comes from the tag, since that is what is being built and released. The
51
+ two refs serve different purposes, which is why they can differ.
52
+
53
+ A version is only spent once PyPI has accepted an upload. If the publish failed
54
+ before that, the same version can still be published — check with
55
+ `curl -s -o /dev/null -w '%{http_code}' https://pypi.org/pypi/<pkg>/<version>/json`
56
+ returning 404.
57
+
37
58
  ## Enforcement
38
59
 
39
60
  - **PR check** (`check-release.yml`): if `pyproject.toml` version changes, `CHANGELOG.md` must contain a matching `## [X.Y.Z]` section.
@@ -2,6 +2,9 @@
2
2
 
3
3
  from importlib.metadata import PackageNotFoundError, version
4
4
 
5
+ from hotdata.models.table_partition_key import TablePartitionKey
6
+ from hotdata.models.table_sort_key import TableSortKey
7
+
5
8
  from hotdata_framework.client import (
6
9
  HotdataClient,
7
10
  ResultSummary,
@@ -14,6 +17,7 @@ from hotdata_framework.databases import (
14
17
  LoadManagedTableResult,
15
18
  ManagedDatabase,
16
19
  ManagedTable,
20
+ TableLayout,
17
21
  is_parquet_path,
18
22
  )
19
23
  from hotdata_framework.env import (
@@ -55,6 +59,9 @@ __all__ = [
55
59
  "QueryResult",
56
60
  "ResultSummary",
57
61
  "RunHistoryItem",
62
+ "TableLayout",
63
+ "TablePartitionKey",
64
+ "TableSortKey",
58
65
  "WorkspaceSelection",
59
66
  "__version__",
60
67
  "classify_sdk_error",
@@ -2,7 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  import functools
4
4
  import time
5
- from collections.abc import Iterator
5
+ from collections.abc import Iterator, Sequence
6
6
  from dataclasses import asdict, dataclass
7
7
  from typing import Any, Literal, get_args
8
8
 
@@ -15,9 +15,6 @@ from hotdata.api.jobs_api import JobsApi
15
15
  from hotdata.api.query_api import QueryApi
16
16
  from hotdata.api.query_runs_api import QueryRunsApi
17
17
  from hotdata.api.results_api import ResultsApi
18
- # The enriched wrapper (hotdata.uploads), NOT the generated hotdata.api class:
19
- # it adds the full upload_file orchestration used by upload_parquet.
20
- from hotdata.uploads import UploadError, UploadsApi
21
18
  from hotdata.exceptions import ApiException
22
19
  from hotdata.models.add_managed_table_request import AddManagedTableRequest
23
20
  from hotdata.models.async_query_response import AsyncQueryResponse
@@ -32,6 +29,12 @@ from hotdata.models.query_request import QueryRequest
32
29
  from hotdata.models.query_response import QueryResponse
33
30
  from hotdata.models.submit_job_response import SubmitJobResponse
34
31
  from hotdata.models.table_info import TableInfo
32
+ from hotdata.models.table_partition_key import TablePartitionKey
33
+ from hotdata.models.table_sort_key import TableSortKey
34
+
35
+ # The enriched wrapper (hotdata.uploads), NOT the generated hotdata.api class:
36
+ # it adds the full upload_file orchestration used by upload_parquet.
37
+ from hotdata.uploads import UploadError, UploadsApi
35
38
  from urllib3.exceptions import HTTPError as Urllib3HTTPError
36
39
  from urllib3.exceptions import ProtocolError
37
40
 
@@ -41,6 +44,7 @@ from hotdata_framework.databases import (
41
44
  LoadManagedTableResult,
42
45
  ManagedDatabase,
43
46
  ManagedTable,
47
+ TableLayout,
44
48
  api_error_message,
45
49
  enum_value,
46
50
  is_parquet_path,
@@ -286,18 +290,44 @@ class HotdataClient:
286
290
  schema: str = DEFAULT_SCHEMA,
287
291
  tables: list[str] | None = None,
288
292
  keys: dict[str, list[str]] | None = None,
293
+ partition_by: dict[str, Sequence[TablePartitionKey]] | None = None,
294
+ sorted_by: dict[str, Sequence[TableSortKey]] | None = None,
289
295
  expires_at: str | None = None,
290
296
  ) -> ManagedDatabase:
291
297
  """Create a managed database. ``keys`` maps a table to its key columns
292
- (enabling delete/update/upsert on it); omitted tables are keyless."""
298
+ (enabling delete/update/upsert on it); omitted tables are keyless.
299
+
300
+ ``partition_by`` and ``sorted_by`` are keyed the same way — table name to
301
+ that table's keys, in declaration order — so a database can be created
302
+ with its tables already laid out. Tables absent from the mapping get no
303
+ layout, and a layout cannot be added afterwards: it is fixed at table
304
+ creation, so a table created here without one stays that way.
305
+ """
293
306
  keys = keys or {}
307
+ partition_by = partition_by or {}
308
+ sorted_by = sorted_by or {}
309
+ # A layout aimed at a table that is not being created would otherwise be
310
+ # dropped in silence, and the table it was meant for created flat — which
311
+ # is permanent, since a layout is fixed at creation with no alter path. A
312
+ # typo'd `keys` entry costs nothing by comparison: load_managed_table
313
+ # takes `key=` per call, so it can be corrected later.
314
+ unknown = (set(partition_by) | set(sorted_by)) - set(tables or ())
315
+ if unknown:
316
+ raise ValueError(
317
+ f"layout given for tables not being created: {', '.join(sorted(unknown))}"
318
+ )
294
319
  schemas = None
295
320
  if tables:
296
321
  schemas = [
297
322
  DatabaseDefaultSchemaDecl(
298
323
  name=schema,
299
324
  tables=[
300
- DatabaseDefaultTableDecl(name=t, key=list(keys.get(t, [])))
325
+ DatabaseDefaultTableDecl(
326
+ name=t,
327
+ key=list(keys.get(t, [])),
328
+ partition_by=list(partition_by.get(t, ())) or None,
329
+ sorted_by=list(sorted_by.get(t, ())) or None,
330
+ )
301
331
  for t in tables
302
332
  ],
303
333
  )
@@ -417,6 +447,8 @@ class HotdataClient:
417
447
  *,
418
448
  schema: str = DEFAULT_SCHEMA,
419
449
  key: list[str] | None = None,
450
+ partition_by: Sequence[TablePartitionKey] | None = None,
451
+ sorted_by: Sequence[TableSortKey] | None = None,
420
452
  ) -> ManagedTable:
421
453
  """Declare a new table on an existing managed database.
422
454
 
@@ -424,9 +456,24 @@ class HotdataClient:
424
456
  :meth:`load_managed_table`. Use this to evolve a managed database's
425
457
  schema after creation without recreating it. ``key`` sets the
426
458
  row-identity columns for delete/update/upsert; omit for keyless.
459
+
460
+ ``partition_by`` and ``sorted_by`` declare the table's storage layout, in
461
+ the order given. THIS IS THE ONLY CHANCE TO SET IT: a layout is fixed
462
+ when the table is created and there is no alter path, so a table declared
463
+ without one keeps that shape until it is recreated and its data rewritten.
464
+ Confirm what was applied with :meth:`managed_table_layout`.
465
+
466
+ The generated key models are passed through rather than wrapped, so the
467
+ transform vocabulary and field names stay exactly the API's. Both are
468
+ re-exported from ``hotdata_framework`` so callers need one import.
427
469
  """
428
470
  db = self._as_managed_database(database)
429
- request = AddManagedTableRequest(name=table, key=list(key or []))
471
+ request = AddManagedTableRequest(
472
+ name=table,
473
+ key=list(key or []),
474
+ partition_by=list(partition_by) if partition_by else None,
475
+ sorted_by=list(sorted_by) if sorted_by else None,
476
+ )
430
477
  try:
431
478
  self._databases_api().add_database_table(db.id, schema, request)
432
479
  except ApiException as e:
@@ -439,6 +486,51 @@ class HotdataClient:
439
486
  last_sync=None,
440
487
  )
441
488
 
489
+ def managed_table_layout(
490
+ self,
491
+ database: str | ManagedDatabase,
492
+ table: str,
493
+ *,
494
+ schema: str = DEFAULT_SCHEMA,
495
+ ) -> TableLayout:
496
+ """Read back a managed table's declared storage layout.
497
+
498
+ The counterpart to the ``partition_by`` / ``sorted_by`` arguments on
499
+ :meth:`add_managed_table` and :meth:`create_managed_database`. Declaring a
500
+ layout is only half of it: it is fixed at table creation with no alter
501
+ path, so a caller that cares whether the layout took has to look, and a
502
+ caller that cannot confirm it should refuse to load rather than fill a
503
+ table it can never repair.
504
+
505
+ Empty lists here mean no layout was declared. That reading is sound
506
+ because the table is resolved through a managed database — the same fields
507
+ on a table discovered from an external connection are empty because its
508
+ layout belongs to the upstream system, which is not the same claim.
509
+
510
+ Raises KeyError when the table is not present on the database, so that
511
+ "no such table" is distinguishable from "declared without a layout"; the
512
+ two are very different for a caller deciding whether to load.
513
+ """
514
+ db = self._as_managed_database(database)
515
+ # Filtered server-side rather than paging iter_tables: this answers a
516
+ # single-table question, and a table sorting late in the listing would
517
+ # otherwise cost several round trips. include_columns is left off — the
518
+ # layout lives on the table row, not the columns.
519
+ resp = self._information_schema().information_schema(
520
+ connection_id=db.default_connection_id,
521
+ var_schema=schema,
522
+ table=table,
523
+ limit=1,
524
+ )
525
+ for info in resp.tables:
526
+ return TableLayout(
527
+ schema_name=schema,
528
+ table_name=table,
529
+ partition_by=list(info.partition_by or []),
530
+ sorted_by=list(info.sorted_by or []),
531
+ )
532
+ raise KeyError(f"{schema}.{table} is not declared on database {db.id}")
533
+
442
534
  def delete_managed_table(
443
535
  self,
444
536
  database: str | ManagedDatabase,
@@ -7,6 +7,8 @@ from pathlib import Path
7
7
  from typing import Any
8
8
 
9
9
  from hotdata.exceptions import ApiException
10
+ from hotdata.models.table_partition_key import TablePartitionKey
11
+ from hotdata.models.table_sort_key import TableSortKey
10
12
 
11
13
  DEFAULT_SCHEMA = "public"
12
14
 
@@ -33,6 +35,47 @@ class ManagedTable:
33
35
  return asdict(self)
34
36
 
35
37
 
38
+ @dataclass(frozen=True)
39
+ class TableLayout:
40
+ """A managed table's declared storage layout, as the server reports it.
41
+
42
+ Both lists carry the generated `TablePartitionKey` / `TableSortKey` models,
43
+ in the order they were declared. A layout is fixed when the table is created
44
+ and cannot be altered, so reading it back is the only way to confirm what was
45
+ actually applied — which is why this exists as a first-class return rather
46
+ than a field on `ManagedTable`, whose other fields describe sync state.
47
+
48
+ Empty lists mean no layout was declared. That reading is only safe because
49
+ this is resolved through a MANAGED database: the same fields on a table
50
+ discovered from an external connection are empty because its layout belongs
51
+ to the upstream system, which is "not known from here" rather than
52
+ "confirmed none".
53
+ """
54
+
55
+ schema_name: str
56
+ table_name: str
57
+ partition_by: list[TablePartitionKey]
58
+ sorted_by: list[TableSortKey]
59
+
60
+ # NO to_dict(), unlike every other dataclass here, and deliberately so.
61
+ # `asdict()` would copy the pydantic key models through untouched rather than
62
+ # flatten them, so it would not return a plain dict. Mapping each key through
63
+ # its own `to_dict()` does flatten, but returns `dict[str, Any]` and adds
64
+ # eight errors under this package's strict mypy settings; hand-building the
65
+ # dict from named fields avoids that but silently drops any field a later
66
+ # spec adds to the key models, which is the failure this whole feature exists
67
+ # to prevent. A caller wanting dicts can map `k.to_dict()` itself and own
68
+ # that choice.
69
+
70
+ @property
71
+ def is_partitioned(self) -> bool:
72
+ return bool(self.partition_by)
73
+
74
+ @property
75
+ def is_sorted(self) -> bool:
76
+ return bool(self.sorted_by)
77
+
78
+
36
79
  @dataclass(frozen=True)
37
80
  class LoadManagedTableResult:
38
81
  connection_id: str
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "hotdata-framework"
7
- version = "0.11.0"
7
+ version = "0.12.0"
8
8
  description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -37,7 +37,7 @@ dependencies = [
37
37
  # uncapped floor turns someone else's release into a break in ours,
38
38
  # with no commit of our own to point at. Raise the cap deliberately, after
39
39
  # running the suite against the new minor.
40
- "hotdata>=0.8.0,<0.9",
40
+ "hotdata>=0.9.0,<0.10",
41
41
  "pandas>=2.0",
42
42
  "pyarrow>=14.0",
43
43
  ]
@@ -23,6 +23,9 @@ def test_public_exports_contract():
23
23
  "QueryResult",
24
24
  "ResultSummary",
25
25
  "RunHistoryItem",
26
+ "TableLayout",
27
+ "TablePartitionKey",
28
+ "TableSortKey",
26
29
  "WorkspaceSelection",
27
30
  "__version__",
28
31
  "classify_sdk_error",
@@ -9,10 +9,13 @@ from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
9
9
 
10
10
  from hotdata_framework.client import HotdataClient
11
11
  from hotdata_framework.databases import (
12
+ ManagedDatabase,
12
13
  is_parquet_path,
13
14
  managed_database_from_detail,
14
15
  )
15
16
 
17
+ _MANAGED_DB = ManagedDatabase(id="db_1", description="d", default_connection_id="conn_1")
18
+
16
19
 
17
20
  def _decl_key_supported() -> bool:
18
21
  # `key` ships with the regenerated client; the key tests activate once it does.
@@ -383,3 +386,173 @@ class _Any:
383
386
 
384
387
  def __eq__(self, other: object) -> bool:
385
388
  return True
389
+
390
+
391
+ # ---------------------------------------------------------------------------
392
+ # Table layout: declaring it, and reading it back
393
+ #
394
+ # A layout is fixed when the table is created and there is no alter path, so a
395
+ # silently dropped argument produces a table that can never be repaired — only
396
+ # recreated with its data rewritten. That is why these assert on the REQUEST
397
+ # object reaching the API rather than on the call succeeding: the generated model
398
+ # ignores unknown fields, so passing `partition_by=` to a client whose model
399
+ # lacks it returns 201 and declares a layout-less table. That exact silent drop
400
+ # is why this package could not express a layout until now.
401
+
402
+
403
+ def _layout():
404
+ from hotdata_framework import TablePartitionKey, TableSortKey
405
+
406
+ return (
407
+ [TablePartitionKey(column="event_date", transform="identity")],
408
+ [TableSortKey(column="event_time", direction="desc", nulls="last")],
409
+ )
410
+
411
+
412
+ def test_add_managed_table_sends_the_layout():
413
+
414
+ parts, sorts = _layout()
415
+ client = _client()
416
+ captured = {}
417
+
418
+ class FakeDatabasesApi:
419
+ def add_database_table(self, db_id, schema, request):
420
+ captured["request"] = request
421
+ captured["schema"] = schema
422
+
423
+ with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()), patch.object(
424
+ client, "_as_managed_database", return_value=_MANAGED_DB
425
+ ):
426
+ client.add_managed_table("db_1", "files", key=["id"], partition_by=parts, sorted_by=sorts)
427
+
428
+ req = captured["request"]
429
+ # Serialised, because that is what actually goes on the wire — a field the
430
+ # model does not know about vanishes here rather than at the call site.
431
+ body = req.to_dict()
432
+ assert body["partition_by"] == [{"column": "event_date", "transform": "identity"}]
433
+ assert body["sorted_by"] == [{"column": "event_time", "direction": "desc", "nulls": "last"}]
434
+ assert body["key"] == ["id"]
435
+
436
+
437
+ def test_add_managed_table_omits_layout_when_not_asked():
438
+ """Omitted must stay omitted: sending empty arrays would declare "explicitly
439
+ no layout" on a server that distinguishes absent from empty."""
440
+
441
+ client = _client()
442
+ captured = {}
443
+
444
+ class FakeDatabasesApi:
445
+ def add_database_table(self, db_id, schema, request):
446
+ captured["request"] = request
447
+
448
+ with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()), patch.object(
449
+ client, "_as_managed_database", return_value=_MANAGED_DB
450
+ ):
451
+ client.add_managed_table("db_1", "files")
452
+
453
+ body = captured["request"].to_dict()
454
+ assert body.get("partition_by") is None
455
+ assert body.get("sorted_by") is None
456
+
457
+
458
+ def test_create_managed_database_sends_per_table_layout():
459
+
460
+ parts, sorts = _layout()
461
+ client = _client()
462
+ captured = {}
463
+
464
+ class FakeDatabasesApi:
465
+ def create_database(self, request):
466
+ captured["request"] = request
467
+ return _detail(id="db_new", description="demo")
468
+
469
+ with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()):
470
+ client.create_managed_database(
471
+ "demo",
472
+ tables=["files", "other"],
473
+ keys={"files": ["id"]},
474
+ partition_by={"files": parts},
475
+ sorted_by={"files": sorts},
476
+ )
477
+
478
+ decls = captured["request"].to_dict()["schemas"][0]["tables"]
479
+ by_name = {d["name"]: d for d in decls}
480
+ assert by_name["files"]["partition_by"] == [
481
+ {"column": "event_date", "transform": "identity"}
482
+ ]
483
+ assert by_name["files"]["sorted_by"] == [
484
+ {"column": "event_time", "direction": "desc", "nulls": "last"}
485
+ ]
486
+ # A table absent from the mapping gets no layout, not an empty one.
487
+ assert by_name["other"].get("partition_by") is None
488
+ assert by_name["other"].get("sorted_by") is None
489
+
490
+
491
+ def test_managed_table_layout_reads_it_back():
492
+ parts, sorts = _layout()
493
+ client = _client()
494
+ info = SimpleNamespace(
495
+ table="files", var_schema="public", partition_by=parts, sorted_by=sorts
496
+ )
497
+ captured = {}
498
+
499
+ class FakeInfoSchema:
500
+ def information_schema(self, **kwargs):
501
+ captured.update(kwargs)
502
+ return SimpleNamespace(tables=[info])
503
+
504
+ with patch.object(client, "_information_schema", return_value=FakeInfoSchema()), patch.object(
505
+ client, "_as_managed_database", return_value=_MANAGED_DB
506
+ ):
507
+ layout = client.managed_table_layout("db_1", "files")
508
+
509
+ assert layout.table_name == "files"
510
+ assert [p.column for p in layout.partition_by] == ["event_date"]
511
+ assert [s.column for s in layout.sorted_by] == ["event_time"]
512
+ assert layout.is_partitioned is True
513
+ assert layout.is_sorted is True
514
+ # Filtered server-side, not by paging the whole listing.
515
+ assert captured["var_schema"] == "public"
516
+ assert captured["table"] == "files"
517
+ assert captured["connection_id"] == _MANAGED_DB.default_connection_id
518
+
519
+
520
+ def test_managed_table_layout_distinguishes_absent_from_unpartitioned():
521
+ """KeyError for a table the server does not report, empty lists for one
522
+ declared without a layout. Collapsing the two would let a caller read "table
523
+ isn't there" as "confirmed no layout" and load into something unverified.
524
+
525
+ Each call gets a FRESH response via side_effect. Sharing one would leave the
526
+ second assertion inspecting a consumed result rather than the case named
527
+ here.
528
+ """
529
+ client = _client()
530
+ plain = SimpleNamespace(table="files", var_schema="public", partition_by=[], sorted_by=[])
531
+
532
+ def respond(**kwargs):
533
+ found = [plain] if kwargs.get("table") == "files" else []
534
+ return SimpleNamespace(tables=found)
535
+
536
+ api = SimpleNamespace(information_schema=respond)
537
+ with patch.object(client, "_information_schema", return_value=api), patch.object(
538
+ client, "_as_managed_database", return_value=_MANAGED_DB
539
+ ):
540
+ layout = client.managed_table_layout("db_1", "files")
541
+ assert layout.is_partitioned is False and layout.is_sorted is False
542
+
543
+ with pytest.raises(KeyError, match="missing"):
544
+ client.managed_table_layout("db_1", "missing")
545
+
546
+
547
+ def test_create_managed_database_refuses_layout_for_a_table_it_is_not_creating():
548
+ """A typo'd table name in the layout mapping would otherwise be dropped, and
549
+ the table it was meant for created flat — permanently, since a layout cannot
550
+ be added later."""
551
+ parts, _ = _layout()
552
+ client = _client()
553
+
554
+ with patch.object(client, "_databases_api") as dbs, pytest.raises(ValueError, match="fils"):
555
+ client.create_managed_database(
556
+ "demo", tables=["files"], partition_by={"fils": parts}
557
+ )
558
+ dbs.return_value.create_database.assert_not_called()
@@ -86,7 +86,7 @@ wheels = [
86
86
 
87
87
  [[package]]
88
88
  name = "hotdata"
89
- version = "0.8.0"
89
+ version = "0.9.0"
90
90
  source = { registry = "https://pypi.org/simple" }
91
91
  dependencies = [
92
92
  { name = "pydantic" },
@@ -94,14 +94,14 @@ dependencies = [
94
94
  { name = "typing-extensions" },
95
95
  { name = "urllib3" },
96
96
  ]
97
- sdist = { url = "https://files.pythonhosted.org/packages/8a/38/30ed3d1d99413e7684672fa424baef85a347db465b90c774443320cf1cea/hotdata-0.8.0.tar.gz", hash = "sha256:cdac515ffa193ed028491e4b7abcd1b6404d4e3f96986c9c0a2b2f407db42eb6", size = 216848, upload-time = "2026-07-20T06:30:37.659Z" }
97
+ sdist = { url = "https://files.pythonhosted.org/packages/6e/db/91d0e8f8a8bfc9a76b588c8dc2ecb054d3e70ebf2ee7a70cd6e4a99a7dec/hotdata-0.9.0.tar.gz", hash = "sha256:35e4a569b7223e025c26b0d3299c8b4cd493276f32826682de6df9e583cea4ef", size = 217368, upload-time = "2026-08-11T14:09:08.166Z" }
98
98
  wheels = [
99
- { url = "https://files.pythonhosted.org/packages/bc/0f/f2ccaeb0f910c3f8cd84a911d934704277f0d6e5a20f8b49cf0eedb2a8aa/hotdata-0.8.0-py3-none-any.whl", hash = "sha256:5d64bfc185a0e7e2bf34ebdcea50787d3418fb250dacc53e00dc74d64e99f753", size = 316923, upload-time = "2026-07-20T06:30:35.778Z" },
99
+ { url = "https://files.pythonhosted.org/packages/ec/48/9b2f5cbbcc92e6dc06a6980fcf4413b83484af302a78263381a22236c308/hotdata-0.9.0-py3-none-any.whl", hash = "sha256:9d942a0d78979552ad641419e5706c9703d5165db0cf3497ae49344cd032e635", size = 316070, upload-time = "2026-08-11T14:09:06.589Z" },
100
100
  ]
101
101
 
102
102
  [[package]]
103
103
  name = "hotdata-framework"
104
- version = "0.11.0"
104
+ version = "0.12.0"
105
105
  source = { editable = "." }
106
106
  dependencies = [
107
107
  { name = "hotdata" },
@@ -120,7 +120,7 @@ dev = [
120
120
 
121
121
  [package.metadata]
122
122
  requires-dist = [
123
- { name = "hotdata", specifier = ">=0.8.0,<0.9" },
123
+ { name = "hotdata", specifier = ">=0.9.0,<0.10" },
124
124
  { name = "pandas", specifier = ">=2.0" },
125
125
  { name = "pyarrow", specifier = ">=14.0" },
126
126
  ]
@@ -1,8 +0,0 @@
1
- version: 2
2
- updates:
3
- - package-ecosystem: uv
4
- directory: "/"
5
- schedule:
6
- interval: daily
7
- allow:
8
- - dependency-name: hotdata