hotdata-framework 0.11.0__tar.gz → 0.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hotdata_framework-0.12.0/.github/dependabot.yml +17 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/publish.yml +26 -5
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/release.yml +24 -4
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/CHANGELOG.md +33 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/CONTRACT.md +6 -1
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/PKG-INFO +2 -2
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/RELEASING.md +21 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/__init__.py +7 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/client.py +99 -7
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/databases.py +43 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/pyproject.toml +2 -2
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_contract.py +3 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_databases.py +173 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/uv.lock +5 -5
- hotdata_framework-0.11.0/.github/dependabot.yml +0 -8
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.gitignore +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/README.md +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/env.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/errors.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/health.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/managed_client.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/release.sh +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_client.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_health.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_indexes.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_managed_client.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_retry_policy.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/tests/test_version.py +0 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
version: 2
|
|
2
|
+
updates:
|
|
3
|
+
- package-ecosystem: uv
|
|
4
|
+
directory: "/"
|
|
5
|
+
schedule:
|
|
6
|
+
interval: daily
|
|
7
|
+
allow:
|
|
8
|
+
- dependency-name: hotdata
|
|
9
|
+
|
|
10
|
+
# Action pins had no watcher, which is how gh-action-pypi-publish sat at
|
|
11
|
+
# v1.13.0 until its bundled twine broke a release at upload time. Unlike the
|
|
12
|
+
# uv entry there is no `allow` filter: the point is to see every stale pin,
|
|
13
|
+
# not a chosen one.
|
|
14
|
+
- package-ecosystem: github-actions
|
|
15
|
+
directory: "/"
|
|
16
|
+
schedule:
|
|
17
|
+
interval: weekly
|
|
@@ -4,9 +4,20 @@ on:
|
|
|
4
4
|
push:
|
|
5
5
|
tags:
|
|
6
6
|
- 'v[0-9]*'
|
|
7
|
+
# Retry an existing tag without moving it. A publish can fail for reasons that
|
|
8
|
+
# have nothing to do with the code — a stale action pin, a PyPI outage — and
|
|
9
|
+
# with only the tag trigger the choices were to delete and re-push the tag or
|
|
10
|
+
# to burn a version number on a CI fix. Neither is a good answer to
|
|
11
|
+
# "the upload failed, run it again".
|
|
12
|
+
workflow_dispatch:
|
|
13
|
+
inputs:
|
|
14
|
+
tag:
|
|
15
|
+
description: Existing tag to build and publish (e.g. v1.2.3)
|
|
16
|
+
required: true
|
|
17
|
+
type: string
|
|
7
18
|
|
|
8
19
|
concurrency:
|
|
9
|
-
group: pypi-publish-${{ github.ref_name }}
|
|
20
|
+
group: pypi-publish-${{ inputs.tag || github.ref_name }}
|
|
10
21
|
cancel-in-progress: false
|
|
11
22
|
|
|
12
23
|
permissions:
|
|
@@ -16,8 +27,13 @@ jobs:
|
|
|
16
27
|
build:
|
|
17
28
|
name: Build distribution
|
|
18
29
|
runs-on: ubuntu-latest
|
|
30
|
+
env:
|
|
31
|
+
# The tag being released, whether it arrived by push or by dispatch.
|
|
32
|
+
TAG: ${{ inputs.tag || github.ref_name }}
|
|
19
33
|
steps:
|
|
20
34
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
35
|
+
with:
|
|
36
|
+
ref: ${{ inputs.tag || github.ref_name }}
|
|
21
37
|
|
|
22
38
|
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
|
23
39
|
with:
|
|
@@ -28,11 +44,11 @@ jobs:
|
|
|
28
44
|
|
|
29
45
|
- name: Verify tag matches pyproject version
|
|
30
46
|
run: |
|
|
31
|
-
if [[ ! "$
|
|
32
|
-
echo "Release tag '$
|
|
47
|
+
if [[ ! "$TAG" =~ ^v[0-9] ]]; then
|
|
48
|
+
echo "Release tag '$TAG' must start with 'v' followed by a digit (e.g. v1.0.0)" >&2
|
|
33
49
|
exit 1
|
|
34
50
|
fi
|
|
35
|
-
tag="${
|
|
51
|
+
tag="${TAG#v}"
|
|
36
52
|
pkg_version=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['version'])")
|
|
37
53
|
if [ "$tag" != "$pkg_version" ]; then
|
|
38
54
|
echo "Release tag ($tag) does not match pyproject.toml version ($pkg_version)" >&2
|
|
@@ -65,5 +81,10 @@ jobs:
|
|
|
65
81
|
name: dist
|
|
66
82
|
path: dist/
|
|
67
83
|
|
|
84
|
+
# v1.13.0's bundled twine rejects `Metadata-Version: 2.5`, which current
|
|
85
|
+
# hatchling emits: `InvalidDistribution: '2.5' is not a valid metadata
|
|
86
|
+
# version`. The build job's own `twine check --strict` passes, because it
|
|
87
|
+
# pip-installs a current twine — so the failure appears only at upload,
|
|
88
|
+
# after the tag is already public.
|
|
68
89
|
- name: Publish via Trusted Publishing
|
|
69
|
-
uses: pypa/gh-action-pypi-publish@
|
|
90
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -4,6 +4,15 @@ on:
|
|
|
4
4
|
push:
|
|
5
5
|
tags:
|
|
6
6
|
- 'v[0-9]*'
|
|
7
|
+
# Repair the Release for a tag that already exists, without moving it. See
|
|
8
|
+
# RELEASING.md, "If a release workflow fails" — added alongside this trigger,
|
|
9
|
+
# since a capability nobody can find is not much better than not having it.
|
|
10
|
+
workflow_dispatch:
|
|
11
|
+
inputs:
|
|
12
|
+
tag:
|
|
13
|
+
description: Existing tag to create or update a Release for (e.g. v1.2.3)
|
|
14
|
+
required: true
|
|
15
|
+
type: string
|
|
7
16
|
|
|
8
17
|
permissions:
|
|
9
18
|
contents: write
|
|
@@ -12,8 +21,13 @@ jobs:
|
|
|
12
21
|
release:
|
|
13
22
|
name: Create GitHub Release
|
|
14
23
|
runs-on: ubuntu-latest
|
|
24
|
+
env:
|
|
25
|
+
# The tag being released, whether it arrived by push or by dispatch.
|
|
26
|
+
TAG: ${{ inputs.tag || github.ref_name }}
|
|
15
27
|
steps:
|
|
16
28
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
29
|
+
with:
|
|
30
|
+
ref: ${{ inputs.tag || github.ref_name }}
|
|
17
31
|
|
|
18
32
|
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
|
19
33
|
with:
|
|
@@ -23,7 +37,7 @@ jobs:
|
|
|
23
37
|
id: meta
|
|
24
38
|
run: |
|
|
25
39
|
pkg_name=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['name'])")
|
|
26
|
-
pkg_version="${
|
|
40
|
+
pkg_version="${TAG#v}"
|
|
27
41
|
echo "name=${pkg_name}" >> "$GITHUB_OUTPUT"
|
|
28
42
|
echo "version=${pkg_version}" >> "$GITHUB_OUTPUT"
|
|
29
43
|
|
|
@@ -31,7 +45,7 @@ jobs:
|
|
|
31
45
|
id: notes
|
|
32
46
|
run: |
|
|
33
47
|
set -euo pipefail
|
|
34
|
-
version="${
|
|
48
|
+
version="${TAG#v}"
|
|
35
49
|
if [[ -f CHANGELOG.md ]]; then
|
|
36
50
|
body="$(python scripts/extract-changelog.py "$version")"
|
|
37
51
|
else
|
|
@@ -47,8 +61,14 @@ jobs:
|
|
|
47
61
|
- name: Create GitHub Release
|
|
48
62
|
uses: softprops/action-gh-release@da05d552573ad5aba039eaac05058a918a7bf631 # v2.2.2
|
|
49
63
|
with:
|
|
50
|
-
tag_name: ${{ github.ref_name }}
|
|
64
|
+
tag_name: ${{ inputs.tag || github.ref_name }}
|
|
51
65
|
name: ${{ steps.meta.outputs.name }} ${{ steps.meta.outputs.version }}
|
|
52
66
|
body: ${{ steps.notes.outputs.body }}
|
|
53
67
|
generate_release_notes: false
|
|
54
|
-
|
|
68
|
+
# Let GitHub decide by tag date/semver rather than by run order. On a
|
|
69
|
+
# push the tag is the newest version and becomes latest; on a dispatch
|
|
70
|
+
# repairing an older tag, a newer release keeps the badge. Note `false`
|
|
71
|
+
# is not "leave alone" — it explicitly marks a release NOT latest, so
|
|
72
|
+
# gating on the event would demote the newest tag in the very case this
|
|
73
|
+
# trigger exists for: repairing its Release after a failed push run.
|
|
74
|
+
make_latest: legacy
|
|
@@ -7,6 +7,39 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
|
|
11
|
+
## [0.12.0] - 2026-08-11
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Table storage layout, both directions. `add_managed_table()` and
|
|
16
|
+
`create_managed_database()` take `partition_by` / `sorted_by`, and
|
|
17
|
+
`managed_table_layout()` reads back what was actually declared as a
|
|
18
|
+
`TableLayout`. `TablePartitionKey` and `TableSortKey` are re-exported so
|
|
19
|
+
callers need one import.
|
|
20
|
+
|
|
21
|
+
Both halves matter because a layout is fixed when the table is created and
|
|
22
|
+
there is no alter path: a table declared without one keeps that shape until it
|
|
23
|
+
is recreated and its data rewritten. So declaring is not enough — a caller has
|
|
24
|
+
to be able to confirm it took, and to refuse to load when it cannot.
|
|
25
|
+
|
|
26
|
+
`managed_table_layout()` raises `KeyError` for a table that is not declared,
|
|
27
|
+
rather than returning an empty layout. "Not there" and "declared without a
|
|
28
|
+
layout" lead to opposite decisions for a caller.
|
|
29
|
+
|
|
30
|
+
Until now this package could not express a layout at all, which is why at least
|
|
31
|
+
one consumer hand-built the HTTP request instead. The generated key models are
|
|
32
|
+
passed through rather than wrapped, so the transform vocabulary stays exactly
|
|
33
|
+
the API's.
|
|
34
|
+
|
|
35
|
+
### Changed
|
|
36
|
+
|
|
37
|
+
- Require `hotdata>=0.9.0,<0.10`. 0.9.0 is the first release whose models carry
|
|
38
|
+
`partition_by` / `sorted_by` on the add-table request, the create-database
|
|
39
|
+
table declarations, and the table-info response. On an older `hotdata` the
|
|
40
|
+
fields would be silently dropped by the model and the table declared without a
|
|
41
|
+
layout, returning success — which is the failure this feature exists to end.
|
|
42
|
+
|
|
10
43
|
## [0.11.0] - 2026-08-11
|
|
11
44
|
|
|
12
45
|
### Changed
|
|
@@ -31,6 +31,9 @@ The supported import surface is:
|
|
|
31
31
|
- `WorkspaceSelection`
|
|
32
32
|
- `ManagedDatabase`
|
|
33
33
|
- `ManagedTable`
|
|
34
|
+
- `TableLayout`
|
|
35
|
+
- `TablePartitionKey`
|
|
36
|
+
- `TableSortKey`
|
|
34
37
|
- `LoadManagedTableResult`
|
|
35
38
|
- `CreateIndexResult`
|
|
36
39
|
- `DEFAULT_SCHEMA`
|
|
@@ -56,6 +59,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
56
59
|
adapters should pass `connection_id` when known.
|
|
57
60
|
- `uploads()` returns the uploads API wrapper for parquet staging.
|
|
58
61
|
- `list_managed_databases()` returns all databases via the `/databases` API.
|
|
62
|
+
- `add_managed_table(...)` and `create_managed_database(...)` accept `partition_by` / `sorted_by` to declare a table's storage layout. The layout is fixed when the table is created and cannot be altered afterwards, so omitting it is permanent for that table.
|
|
63
|
+
- `managed_table_layout(database, table, schema=...)` returns the declared layout as `TableLayout`. Empty lists mean no layout was declared — sound only because the table is resolved through a managed database. Raises `KeyError` when the table is not declared, keeping "absent" distinct from "declared without a layout".
|
|
59
64
|
- `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan). A `403` from `/databases` surfaces as `RuntimeError` (forbidden, not absent), preserving the underlying `ApiException` as `__cause__`.
|
|
60
65
|
- `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front. Returns a `ManagedDatabase` (id + `default_connection_id`) sufficient to load without a further read.
|
|
61
66
|
- `delete_managed_database(name_or_id)` deletes a database via the `/databases` API.
|
|
@@ -64,7 +69,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
64
69
|
- `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
|
|
65
70
|
- `delete_managed_table(database, table, schema=...)` deletes a managed table.
|
|
66
71
|
- `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
|
|
67
|
-
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
72
|
+
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, `managed_table_layout`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
68
73
|
|
|
69
74
|
### `QueryResult`
|
|
70
75
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.12.0
|
|
4
4
|
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
|
21
21
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
22
|
Classifier: Typing :: Typed
|
|
23
23
|
Requires-Python: >=3.10
|
|
24
|
-
Requires-Dist: hotdata<0.
|
|
24
|
+
Requires-Dist: hotdata<0.10,>=0.9.0
|
|
25
25
|
Requires-Dist: pandas>=2.0
|
|
26
26
|
Requires-Dist: pyarrow>=14.0
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
@@ -34,6 +34,27 @@ Pushing a `vX.Y.Z` tag triggers two workflows:
|
|
|
34
34
|
| `publish.yml` | Build wheel/sdist and publish to PyPI |
|
|
35
35
|
| `release.yml` | Create the GitHub Release with notes from `CHANGELOG.md` |
|
|
36
36
|
|
|
37
|
+
## If a release workflow fails
|
|
38
|
+
|
|
39
|
+
Both workflows also accept a manual re-run against an existing tag, so a failure
|
|
40
|
+
unrelated to the code — a stale action pin, a PyPI outage — does not require
|
|
41
|
+
deleting the tag or burning a version number:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
gh workflow run "Publish to PyPI" --ref main -f tag=vX.Y.Z
|
|
45
|
+
gh workflow run "GitHub Release" --ref main -f tag=vX.Y.Z
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`--ref main` selects the workflow *definition*, so a fix to the workflow file
|
|
49
|
+
itself is picked up; everything it runs — including `scripts/extract-changelog.py`
|
|
50
|
+
— still comes from the tag, since that is what is being built and released. The
|
|
51
|
+
two refs serve different purposes, which is why they can differ.
|
|
52
|
+
|
|
53
|
+
A version is only spent once PyPI has accepted an upload. If the publish failed
|
|
54
|
+
before that, the same version can still be published — check with
|
|
55
|
+
`curl -s -o /dev/null -w '%{http_code}' https://pypi.org/pypi/<pkg>/<version>/json`
|
|
56
|
+
returning 404.
|
|
57
|
+
|
|
37
58
|
## Enforcement
|
|
38
59
|
|
|
39
60
|
- **PR check** (`check-release.yml`): if `pyproject.toml` version changes, `CHANGELOG.md` must contain a matching `## [X.Y.Z]` section.
|
|
@@ -2,6 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
from importlib.metadata import PackageNotFoundError, version
|
|
4
4
|
|
|
5
|
+
from hotdata.models.table_partition_key import TablePartitionKey
|
|
6
|
+
from hotdata.models.table_sort_key import TableSortKey
|
|
7
|
+
|
|
5
8
|
from hotdata_framework.client import (
|
|
6
9
|
HotdataClient,
|
|
7
10
|
ResultSummary,
|
|
@@ -14,6 +17,7 @@ from hotdata_framework.databases import (
|
|
|
14
17
|
LoadManagedTableResult,
|
|
15
18
|
ManagedDatabase,
|
|
16
19
|
ManagedTable,
|
|
20
|
+
TableLayout,
|
|
17
21
|
is_parquet_path,
|
|
18
22
|
)
|
|
19
23
|
from hotdata_framework.env import (
|
|
@@ -55,6 +59,9 @@ __all__ = [
|
|
|
55
59
|
"QueryResult",
|
|
56
60
|
"ResultSummary",
|
|
57
61
|
"RunHistoryItem",
|
|
62
|
+
"TableLayout",
|
|
63
|
+
"TablePartitionKey",
|
|
64
|
+
"TableSortKey",
|
|
58
65
|
"WorkspaceSelection",
|
|
59
66
|
"__version__",
|
|
60
67
|
"classify_sdk_error",
|
|
@@ -2,7 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import functools
|
|
4
4
|
import time
|
|
5
|
-
from collections.abc import Iterator
|
|
5
|
+
from collections.abc import Iterator, Sequence
|
|
6
6
|
from dataclasses import asdict, dataclass
|
|
7
7
|
from typing import Any, Literal, get_args
|
|
8
8
|
|
|
@@ -15,9 +15,6 @@ from hotdata.api.jobs_api import JobsApi
|
|
|
15
15
|
from hotdata.api.query_api import QueryApi
|
|
16
16
|
from hotdata.api.query_runs_api import QueryRunsApi
|
|
17
17
|
from hotdata.api.results_api import ResultsApi
|
|
18
|
-
# The enriched wrapper (hotdata.uploads), NOT the generated hotdata.api class:
|
|
19
|
-
# it adds the full upload_file orchestration used by upload_parquet.
|
|
20
|
-
from hotdata.uploads import UploadError, UploadsApi
|
|
21
18
|
from hotdata.exceptions import ApiException
|
|
22
19
|
from hotdata.models.add_managed_table_request import AddManagedTableRequest
|
|
23
20
|
from hotdata.models.async_query_response import AsyncQueryResponse
|
|
@@ -32,6 +29,12 @@ from hotdata.models.query_request import QueryRequest
|
|
|
32
29
|
from hotdata.models.query_response import QueryResponse
|
|
33
30
|
from hotdata.models.submit_job_response import SubmitJobResponse
|
|
34
31
|
from hotdata.models.table_info import TableInfo
|
|
32
|
+
from hotdata.models.table_partition_key import TablePartitionKey
|
|
33
|
+
from hotdata.models.table_sort_key import TableSortKey
|
|
34
|
+
|
|
35
|
+
# The enriched wrapper (hotdata.uploads), NOT the generated hotdata.api class:
|
|
36
|
+
# it adds the full upload_file orchestration used by upload_parquet.
|
|
37
|
+
from hotdata.uploads import UploadError, UploadsApi
|
|
35
38
|
from urllib3.exceptions import HTTPError as Urllib3HTTPError
|
|
36
39
|
from urllib3.exceptions import ProtocolError
|
|
37
40
|
|
|
@@ -41,6 +44,7 @@ from hotdata_framework.databases import (
|
|
|
41
44
|
LoadManagedTableResult,
|
|
42
45
|
ManagedDatabase,
|
|
43
46
|
ManagedTable,
|
|
47
|
+
TableLayout,
|
|
44
48
|
api_error_message,
|
|
45
49
|
enum_value,
|
|
46
50
|
is_parquet_path,
|
|
@@ -286,18 +290,44 @@ class HotdataClient:
|
|
|
286
290
|
schema: str = DEFAULT_SCHEMA,
|
|
287
291
|
tables: list[str] | None = None,
|
|
288
292
|
keys: dict[str, list[str]] | None = None,
|
|
293
|
+
partition_by: dict[str, Sequence[TablePartitionKey]] | None = None,
|
|
294
|
+
sorted_by: dict[str, Sequence[TableSortKey]] | None = None,
|
|
289
295
|
expires_at: str | None = None,
|
|
290
296
|
) -> ManagedDatabase:
|
|
291
297
|
"""Create a managed database. ``keys`` maps a table to its key columns
|
|
292
|
-
(enabling delete/update/upsert on it); omitted tables are keyless.
|
|
298
|
+
(enabling delete/update/upsert on it); omitted tables are keyless.
|
|
299
|
+
|
|
300
|
+
``partition_by`` and ``sorted_by`` are keyed the same way — table name to
|
|
301
|
+
that table's keys, in declaration order — so a database can be created
|
|
302
|
+
with its tables already laid out. Tables absent from the mapping get no
|
|
303
|
+
layout, and a layout cannot be added afterwards: it is fixed at table
|
|
304
|
+
creation, so a table created here without one stays that way.
|
|
305
|
+
"""
|
|
293
306
|
keys = keys or {}
|
|
307
|
+
partition_by = partition_by or {}
|
|
308
|
+
sorted_by = sorted_by or {}
|
|
309
|
+
# A layout aimed at a table that is not being created would otherwise be
|
|
310
|
+
# dropped in silence, and the table it was meant for created flat — which
|
|
311
|
+
# is permanent, since a layout is fixed at creation with no alter path. A
|
|
312
|
+
# typo'd `keys` entry costs nothing by comparison: load_managed_table
|
|
313
|
+
# takes `key=` per call, so it can be corrected later.
|
|
314
|
+
unknown = (set(partition_by) | set(sorted_by)) - set(tables or ())
|
|
315
|
+
if unknown:
|
|
316
|
+
raise ValueError(
|
|
317
|
+
f"layout given for tables not being created: {', '.join(sorted(unknown))}"
|
|
318
|
+
)
|
|
294
319
|
schemas = None
|
|
295
320
|
if tables:
|
|
296
321
|
schemas = [
|
|
297
322
|
DatabaseDefaultSchemaDecl(
|
|
298
323
|
name=schema,
|
|
299
324
|
tables=[
|
|
300
|
-
DatabaseDefaultTableDecl(
|
|
325
|
+
DatabaseDefaultTableDecl(
|
|
326
|
+
name=t,
|
|
327
|
+
key=list(keys.get(t, [])),
|
|
328
|
+
partition_by=list(partition_by.get(t, ())) or None,
|
|
329
|
+
sorted_by=list(sorted_by.get(t, ())) or None,
|
|
330
|
+
)
|
|
301
331
|
for t in tables
|
|
302
332
|
],
|
|
303
333
|
)
|
|
@@ -417,6 +447,8 @@ class HotdataClient:
|
|
|
417
447
|
*,
|
|
418
448
|
schema: str = DEFAULT_SCHEMA,
|
|
419
449
|
key: list[str] | None = None,
|
|
450
|
+
partition_by: Sequence[TablePartitionKey] | None = None,
|
|
451
|
+
sorted_by: Sequence[TableSortKey] | None = None,
|
|
420
452
|
) -> ManagedTable:
|
|
421
453
|
"""Declare a new table on an existing managed database.
|
|
422
454
|
|
|
@@ -424,9 +456,24 @@ class HotdataClient:
|
|
|
424
456
|
:meth:`load_managed_table`. Use this to evolve a managed database's
|
|
425
457
|
schema after creation without recreating it. ``key`` sets the
|
|
426
458
|
row-identity columns for delete/update/upsert; omit for keyless.
|
|
459
|
+
|
|
460
|
+
``partition_by`` and ``sorted_by`` declare the table's storage layout, in
|
|
461
|
+
the order given. THIS IS THE ONLY CHANCE TO SET IT: a layout is fixed
|
|
462
|
+
when the table is created and there is no alter path, so a table declared
|
|
463
|
+
without one keeps that shape until it is recreated and its data rewritten.
|
|
464
|
+
Confirm what was applied with :meth:`managed_table_layout`.
|
|
465
|
+
|
|
466
|
+
The generated key models are passed through rather than wrapped, so the
|
|
467
|
+
transform vocabulary and field names stay exactly the API's. Both are
|
|
468
|
+
re-exported from ``hotdata_framework`` so callers need one import.
|
|
427
469
|
"""
|
|
428
470
|
db = self._as_managed_database(database)
|
|
429
|
-
request = AddManagedTableRequest(
|
|
471
|
+
request = AddManagedTableRequest(
|
|
472
|
+
name=table,
|
|
473
|
+
key=list(key or []),
|
|
474
|
+
partition_by=list(partition_by) if partition_by else None,
|
|
475
|
+
sorted_by=list(sorted_by) if sorted_by else None,
|
|
476
|
+
)
|
|
430
477
|
try:
|
|
431
478
|
self._databases_api().add_database_table(db.id, schema, request)
|
|
432
479
|
except ApiException as e:
|
|
@@ -439,6 +486,51 @@ class HotdataClient:
|
|
|
439
486
|
last_sync=None,
|
|
440
487
|
)
|
|
441
488
|
|
|
489
|
+
def managed_table_layout(
|
|
490
|
+
self,
|
|
491
|
+
database: str | ManagedDatabase,
|
|
492
|
+
table: str,
|
|
493
|
+
*,
|
|
494
|
+
schema: str = DEFAULT_SCHEMA,
|
|
495
|
+
) -> TableLayout:
|
|
496
|
+
"""Read back a managed table's declared storage layout.
|
|
497
|
+
|
|
498
|
+
The counterpart to the ``partition_by`` / ``sorted_by`` arguments on
|
|
499
|
+
:meth:`add_managed_table` and :meth:`create_managed_database`. Declaring a
|
|
500
|
+
layout is only half of it: it is fixed at table creation with no alter
|
|
501
|
+
path, so a caller that cares whether the layout took has to look, and a
|
|
502
|
+
caller that cannot confirm it should refuse to load rather than fill a
|
|
503
|
+
table it can never repair.
|
|
504
|
+
|
|
505
|
+
Empty lists here mean no layout was declared. That reading is sound
|
|
506
|
+
because the table is resolved through a managed database — the same fields
|
|
507
|
+
on a table discovered from an external connection are empty because its
|
|
508
|
+
layout belongs to the upstream system, which is not the same claim.
|
|
509
|
+
|
|
510
|
+
Raises KeyError when the table is not present on the database, so that
|
|
511
|
+
"no such table" is distinguishable from "declared without a layout"; the
|
|
512
|
+
two are very different for a caller deciding whether to load.
|
|
513
|
+
"""
|
|
514
|
+
db = self._as_managed_database(database)
|
|
515
|
+
# Filtered server-side rather than paging iter_tables: this answers a
|
|
516
|
+
# single-table question, and a table sorting late in the listing would
|
|
517
|
+
# otherwise cost several round trips. include_columns is left off — the
|
|
518
|
+
# layout lives on the table row, not the columns.
|
|
519
|
+
resp = self._information_schema().information_schema(
|
|
520
|
+
connection_id=db.default_connection_id,
|
|
521
|
+
var_schema=schema,
|
|
522
|
+
table=table,
|
|
523
|
+
limit=1,
|
|
524
|
+
)
|
|
525
|
+
for info in resp.tables:
|
|
526
|
+
return TableLayout(
|
|
527
|
+
schema_name=schema,
|
|
528
|
+
table_name=table,
|
|
529
|
+
partition_by=list(info.partition_by or []),
|
|
530
|
+
sorted_by=list(info.sorted_by or []),
|
|
531
|
+
)
|
|
532
|
+
raise KeyError(f"{schema}.{table} is not declared on database {db.id}")
|
|
533
|
+
|
|
442
534
|
def delete_managed_table(
|
|
443
535
|
self,
|
|
444
536
|
database: str | ManagedDatabase,
|
|
@@ -7,6 +7,8 @@ from pathlib import Path
|
|
|
7
7
|
from typing import Any
|
|
8
8
|
|
|
9
9
|
from hotdata.exceptions import ApiException
|
|
10
|
+
from hotdata.models.table_partition_key import TablePartitionKey
|
|
11
|
+
from hotdata.models.table_sort_key import TableSortKey
|
|
10
12
|
|
|
11
13
|
DEFAULT_SCHEMA = "public"
|
|
12
14
|
|
|
@@ -33,6 +35,47 @@ class ManagedTable:
|
|
|
33
35
|
return asdict(self)
|
|
34
36
|
|
|
35
37
|
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class TableLayout:
|
|
40
|
+
"""A managed table's declared storage layout, as the server reports it.
|
|
41
|
+
|
|
42
|
+
Both lists carry the generated `TablePartitionKey` / `TableSortKey` models,
|
|
43
|
+
in the order they were declared. A layout is fixed when the table is created
|
|
44
|
+
and cannot be altered, so reading it back is the only way to confirm what was
|
|
45
|
+
actually applied — which is why this exists as a first-class return rather
|
|
46
|
+
than a field on `ManagedTable`, whose other fields describe sync state.
|
|
47
|
+
|
|
48
|
+
Empty lists mean no layout was declared. That reading is only safe because
|
|
49
|
+
this is resolved through a MANAGED database: the same fields on a table
|
|
50
|
+
discovered from an external connection are empty because its layout belongs
|
|
51
|
+
to the upstream system, which is "not known from here" rather than
|
|
52
|
+
"confirmed none".
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
schema_name: str
|
|
56
|
+
table_name: str
|
|
57
|
+
partition_by: list[TablePartitionKey]
|
|
58
|
+
sorted_by: list[TableSortKey]
|
|
59
|
+
|
|
60
|
+
# NO to_dict(), unlike every other dataclass here, and deliberately so.
|
|
61
|
+
# `asdict()` would copy the pydantic key models through untouched rather than
|
|
62
|
+
# flatten them, so it would not return a plain dict. Mapping each key through
|
|
63
|
+
# its own `to_dict()` does flatten, but returns `dict[str, Any]` and adds
|
|
64
|
+
# eight errors under this package's strict mypy settings; hand-building the
|
|
65
|
+
# dict from named fields avoids that but silently drops any field a later
|
|
66
|
+
# spec adds to the key models, which is the failure this whole feature exists
|
|
67
|
+
# to prevent. A caller wanting dicts can map `k.to_dict()` itself and own
|
|
68
|
+
# that choice.
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def is_partitioned(self) -> bool:
|
|
72
|
+
return bool(self.partition_by)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def is_sorted(self) -> bool:
|
|
76
|
+
return bool(self.sorted_by)
|
|
77
|
+
|
|
78
|
+
|
|
36
79
|
@dataclass(frozen=True)
|
|
37
80
|
class LoadManagedTableResult:
|
|
38
81
|
connection_id: str
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "hotdata-framework"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.12.0"
|
|
8
8
|
description = "Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -37,7 +37,7 @@ dependencies = [
|
|
|
37
37
|
# uncapped floor turns someone else's release into a break in ours,
|
|
38
38
|
# with no commit of our own to point at. Raise the cap deliberately, after
|
|
39
39
|
# running the suite against the new minor.
|
|
40
|
-
"hotdata>=0.
|
|
40
|
+
"hotdata>=0.9.0,<0.10",
|
|
41
41
|
"pandas>=2.0",
|
|
42
42
|
"pyarrow>=14.0",
|
|
43
43
|
]
|
|
@@ -9,10 +9,13 @@ from hotdata.models.database_default_table_decl import DatabaseDefaultTableDecl
|
|
|
9
9
|
|
|
10
10
|
from hotdata_framework.client import HotdataClient
|
|
11
11
|
from hotdata_framework.databases import (
|
|
12
|
+
ManagedDatabase,
|
|
12
13
|
is_parquet_path,
|
|
13
14
|
managed_database_from_detail,
|
|
14
15
|
)
|
|
15
16
|
|
|
17
|
+
_MANAGED_DB = ManagedDatabase(id="db_1", description="d", default_connection_id="conn_1")
|
|
18
|
+
|
|
16
19
|
|
|
17
20
|
def _decl_key_supported() -> bool:
|
|
18
21
|
# `key` ships with the regenerated client; the key tests activate once it does.
|
|
@@ -383,3 +386,173 @@ class _Any:
|
|
|
383
386
|
|
|
384
387
|
def __eq__(self, other: object) -> bool:
|
|
385
388
|
return True
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
# ---------------------------------------------------------------------------
|
|
392
|
+
# Table layout: declaring it, and reading it back
|
|
393
|
+
#
|
|
394
|
+
# A layout is fixed when the table is created and there is no alter path, so a
|
|
395
|
+
# silently dropped argument produces a table that can never be repaired — only
|
|
396
|
+
# recreated with its data rewritten. That is why these assert on the REQUEST
|
|
397
|
+
# object reaching the API rather than on the call succeeding: the generated model
|
|
398
|
+
# ignores unknown fields, so passing `partition_by=` to a client whose model
|
|
399
|
+
# lacks it returns 201 and declares a layout-less table. That exact silent drop
|
|
400
|
+
# is why this package could not express a layout until now.
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _layout():
|
|
404
|
+
from hotdata_framework import TablePartitionKey, TableSortKey
|
|
405
|
+
|
|
406
|
+
return (
|
|
407
|
+
[TablePartitionKey(column="event_date", transform="identity")],
|
|
408
|
+
[TableSortKey(column="event_time", direction="desc", nulls="last")],
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def test_add_managed_table_sends_the_layout():
|
|
413
|
+
|
|
414
|
+
parts, sorts = _layout()
|
|
415
|
+
client = _client()
|
|
416
|
+
captured = {}
|
|
417
|
+
|
|
418
|
+
class FakeDatabasesApi:
|
|
419
|
+
def add_database_table(self, db_id, schema, request):
|
|
420
|
+
captured["request"] = request
|
|
421
|
+
captured["schema"] = schema
|
|
422
|
+
|
|
423
|
+
with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()), patch.object(
|
|
424
|
+
client, "_as_managed_database", return_value=_MANAGED_DB
|
|
425
|
+
):
|
|
426
|
+
client.add_managed_table("db_1", "files", key=["id"], partition_by=parts, sorted_by=sorts)
|
|
427
|
+
|
|
428
|
+
req = captured["request"]
|
|
429
|
+
# Serialised, because that is what actually goes on the wire — a field the
|
|
430
|
+
# model does not know about vanishes here rather than at the call site.
|
|
431
|
+
body = req.to_dict()
|
|
432
|
+
assert body["partition_by"] == [{"column": "event_date", "transform": "identity"}]
|
|
433
|
+
assert body["sorted_by"] == [{"column": "event_time", "direction": "desc", "nulls": "last"}]
|
|
434
|
+
assert body["key"] == ["id"]
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def test_add_managed_table_omits_layout_when_not_asked():
|
|
438
|
+
"""Omitted must stay omitted: sending empty arrays would declare "explicitly
|
|
439
|
+
no layout" on a server that distinguishes absent from empty."""
|
|
440
|
+
|
|
441
|
+
client = _client()
|
|
442
|
+
captured = {}
|
|
443
|
+
|
|
444
|
+
class FakeDatabasesApi:
|
|
445
|
+
def add_database_table(self, db_id, schema, request):
|
|
446
|
+
captured["request"] = request
|
|
447
|
+
|
|
448
|
+
with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()), patch.object(
|
|
449
|
+
client, "_as_managed_database", return_value=_MANAGED_DB
|
|
450
|
+
):
|
|
451
|
+
client.add_managed_table("db_1", "files")
|
|
452
|
+
|
|
453
|
+
body = captured["request"].to_dict()
|
|
454
|
+
assert body.get("partition_by") is None
|
|
455
|
+
assert body.get("sorted_by") is None
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def test_create_managed_database_sends_per_table_layout():
|
|
459
|
+
|
|
460
|
+
parts, sorts = _layout()
|
|
461
|
+
client = _client()
|
|
462
|
+
captured = {}
|
|
463
|
+
|
|
464
|
+
class FakeDatabasesApi:
|
|
465
|
+
def create_database(self, request):
|
|
466
|
+
captured["request"] = request
|
|
467
|
+
return _detail(id="db_new", description="demo")
|
|
468
|
+
|
|
469
|
+
with patch.object(client, "_databases_api", return_value=FakeDatabasesApi()):
|
|
470
|
+
client.create_managed_database(
|
|
471
|
+
"demo",
|
|
472
|
+
tables=["files", "other"],
|
|
473
|
+
keys={"files": ["id"]},
|
|
474
|
+
partition_by={"files": parts},
|
|
475
|
+
sorted_by={"files": sorts},
|
|
476
|
+
)
|
|
477
|
+
|
|
478
|
+
decls = captured["request"].to_dict()["schemas"][0]["tables"]
|
|
479
|
+
by_name = {d["name"]: d for d in decls}
|
|
480
|
+
assert by_name["files"]["partition_by"] == [
|
|
481
|
+
{"column": "event_date", "transform": "identity"}
|
|
482
|
+
]
|
|
483
|
+
assert by_name["files"]["sorted_by"] == [
|
|
484
|
+
{"column": "event_time", "direction": "desc", "nulls": "last"}
|
|
485
|
+
]
|
|
486
|
+
# A table absent from the mapping gets no layout, not an empty one.
|
|
487
|
+
assert by_name["other"].get("partition_by") is None
|
|
488
|
+
assert by_name["other"].get("sorted_by") is None
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def test_managed_table_layout_reads_it_back():
|
|
492
|
+
parts, sorts = _layout()
|
|
493
|
+
client = _client()
|
|
494
|
+
info = SimpleNamespace(
|
|
495
|
+
table="files", var_schema="public", partition_by=parts, sorted_by=sorts
|
|
496
|
+
)
|
|
497
|
+
captured = {}
|
|
498
|
+
|
|
499
|
+
class FakeInfoSchema:
|
|
500
|
+
def information_schema(self, **kwargs):
|
|
501
|
+
captured.update(kwargs)
|
|
502
|
+
return SimpleNamespace(tables=[info])
|
|
503
|
+
|
|
504
|
+
with patch.object(client, "_information_schema", return_value=FakeInfoSchema()), patch.object(
|
|
505
|
+
client, "_as_managed_database", return_value=_MANAGED_DB
|
|
506
|
+
):
|
|
507
|
+
layout = client.managed_table_layout("db_1", "files")
|
|
508
|
+
|
|
509
|
+
assert layout.table_name == "files"
|
|
510
|
+
assert [p.column for p in layout.partition_by] == ["event_date"]
|
|
511
|
+
assert [s.column for s in layout.sorted_by] == ["event_time"]
|
|
512
|
+
assert layout.is_partitioned is True
|
|
513
|
+
assert layout.is_sorted is True
|
|
514
|
+
# Filtered server-side, not by paging the whole listing.
|
|
515
|
+
assert captured["var_schema"] == "public"
|
|
516
|
+
assert captured["table"] == "files"
|
|
517
|
+
assert captured["connection_id"] == _MANAGED_DB.default_connection_id
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def test_managed_table_layout_distinguishes_absent_from_unpartitioned():
|
|
521
|
+
"""KeyError for a table the server does not report, empty lists for one
|
|
522
|
+
declared without a layout. Collapsing the two would let a caller read "table
|
|
523
|
+
isn't there" as "confirmed no layout" and load into something unverified.
|
|
524
|
+
|
|
525
|
+
Each call gets a FRESH response via side_effect. Sharing one would leave the
|
|
526
|
+
second assertion inspecting a consumed result rather than the case named
|
|
527
|
+
here.
|
|
528
|
+
"""
|
|
529
|
+
client = _client()
|
|
530
|
+
plain = SimpleNamespace(table="files", var_schema="public", partition_by=[], sorted_by=[])
|
|
531
|
+
|
|
532
|
+
def respond(**kwargs):
|
|
533
|
+
found = [plain] if kwargs.get("table") == "files" else []
|
|
534
|
+
return SimpleNamespace(tables=found)
|
|
535
|
+
|
|
536
|
+
api = SimpleNamespace(information_schema=respond)
|
|
537
|
+
with patch.object(client, "_information_schema", return_value=api), patch.object(
|
|
538
|
+
client, "_as_managed_database", return_value=_MANAGED_DB
|
|
539
|
+
):
|
|
540
|
+
layout = client.managed_table_layout("db_1", "files")
|
|
541
|
+
assert layout.is_partitioned is False and layout.is_sorted is False
|
|
542
|
+
|
|
543
|
+
with pytest.raises(KeyError, match="missing"):
|
|
544
|
+
client.managed_table_layout("db_1", "missing")
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def test_create_managed_database_refuses_layout_for_a_table_it_is_not_creating():
|
|
548
|
+
"""A typo'd table name in the layout mapping would otherwise be dropped, and
|
|
549
|
+
the table it was meant for created flat — permanently, since a layout cannot
|
|
550
|
+
be added later."""
|
|
551
|
+
parts, _ = _layout()
|
|
552
|
+
client = _client()
|
|
553
|
+
|
|
554
|
+
with patch.object(client, "_databases_api") as dbs, pytest.raises(ValueError, match="fils"):
|
|
555
|
+
client.create_managed_database(
|
|
556
|
+
"demo", tables=["files"], partition_by={"fils": parts}
|
|
557
|
+
)
|
|
558
|
+
dbs.return_value.create_database.assert_not_called()
|
|
@@ -86,7 +86,7 @@ wheels = [
|
|
|
86
86
|
|
|
87
87
|
[[package]]
|
|
88
88
|
name = "hotdata"
|
|
89
|
-
version = "0.
|
|
89
|
+
version = "0.9.0"
|
|
90
90
|
source = { registry = "https://pypi.org/simple" }
|
|
91
91
|
dependencies = [
|
|
92
92
|
{ name = "pydantic" },
|
|
@@ -94,14 +94,14 @@ dependencies = [
|
|
|
94
94
|
{ name = "typing-extensions" },
|
|
95
95
|
{ name = "urllib3" },
|
|
96
96
|
]
|
|
97
|
-
sdist = { url = "https://files.pythonhosted.org/packages/
|
|
97
|
+
sdist = { url = "https://files.pythonhosted.org/packages/6e/db/91d0e8f8a8bfc9a76b588c8dc2ecb054d3e70ebf2ee7a70cd6e4a99a7dec/hotdata-0.9.0.tar.gz", hash = "sha256:35e4a569b7223e025c26b0d3299c8b4cd493276f32826682de6df9e583cea4ef", size = 217368, upload-time = "2026-08-11T14:09:08.166Z" }
|
|
98
98
|
wheels = [
|
|
99
|
-
{ url = "https://files.pythonhosted.org/packages/
|
|
99
|
+
{ url = "https://files.pythonhosted.org/packages/ec/48/9b2f5cbbcc92e6dc06a6980fcf4413b83484af302a78263381a22236c308/hotdata-0.9.0-py3-none-any.whl", hash = "sha256:9d942a0d78979552ad641419e5706c9703d5165db0cf3497ae49344cd032e635", size = 316070, upload-time = "2026-08-11T14:09:06.589Z" },
|
|
100
100
|
]
|
|
101
101
|
|
|
102
102
|
[[package]]
|
|
103
103
|
name = "hotdata-framework"
|
|
104
|
-
version = "0.
|
|
104
|
+
version = "0.12.0"
|
|
105
105
|
source = { editable = "." }
|
|
106
106
|
dependencies = [
|
|
107
107
|
{ name = "hotdata" },
|
|
@@ -120,7 +120,7 @@ dev = [
|
|
|
120
120
|
|
|
121
121
|
[package.metadata]
|
|
122
122
|
requires-dist = [
|
|
123
|
-
{ name = "hotdata", specifier = ">=0.
|
|
123
|
+
{ name = "hotdata", specifier = ">=0.9.0,<0.10" },
|
|
124
124
|
{ name = "pandas", specifier = ">=2.0" },
|
|
125
125
|
{ name = "pyarrow", specifier = ">=14.0" },
|
|
126
126
|
]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{hotdata_framework-0.11.0 → hotdata_framework-0.12.0}/.github/workflows/dependabot-automerge.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|