hotdata-framework 0.10.0__tar.gz → 0.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hotdata_framework-0.12.0/.github/dependabot.yml +17 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/workflows/publish.yml +26 -5
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/workflows/release.yml +24 -4
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/CHANGELOG.md +94 -50
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/CONTRACT.md +7 -4
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/PKG-INFO +7 -8
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/README.md +3 -4
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/RELEASING.md +21 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/__init__.py +7 -2
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/client.py +107 -20
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/databases.py +43 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/env.py +5 -12
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/health.py +0 -2
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/pyproject.toml +14 -4
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/scripts/release.sh +41 -8
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_client.py +103 -6
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_contract.py +3 -1
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_databases.py +173 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_health.py +16 -3
- hotdata_framework-0.12.0/tests/test_retry_policy.py +63 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/uv.lock +5 -5
- hotdata_framework-0.10.0/.github/dependabot.yml +0 -8
- hotdata_framework-0.10.0/hotdata_framework/http.py +0 -17
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/CODEOWNERS +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/workflows/check-release.yml +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/workflows/ci.yml +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.github/workflows/dependabot-automerge.yml +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/.gitignore +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/examples/basic_usage.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/errors.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/managed_client.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/py.typed +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/hotdata_framework/result.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/scripts/check-release.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/scripts/extract-changelog.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/scripts/publish-workflow.sh +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/scripts/update_changelog.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_errors.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_indexes.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_managed_client.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_request_timeout.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_result.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_update_changelog.py +0 -0
- {hotdata_framework-0.10.0 → hotdata_framework-0.12.0}/tests/test_version.py +0 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
version: 2
|
|
2
|
+
updates:
|
|
3
|
+
- package-ecosystem: uv
|
|
4
|
+
directory: "/"
|
|
5
|
+
schedule:
|
|
6
|
+
interval: daily
|
|
7
|
+
allow:
|
|
8
|
+
- dependency-name: hotdata
|
|
9
|
+
|
|
10
|
+
# Action pins had no watcher, which is how gh-action-pypi-publish sat at
|
|
11
|
+
# v1.13.0 until its bundled twine broke a release at upload time. Unlike the
|
|
12
|
+
# uv entry there is no `allow` filter: the point is to see every stale pin,
|
|
13
|
+
# not a chosen one.
|
|
14
|
+
- package-ecosystem: github-actions
|
|
15
|
+
directory: "/"
|
|
16
|
+
schedule:
|
|
17
|
+
interval: weekly
|
|
@@ -4,9 +4,20 @@ on:
|
|
|
4
4
|
push:
|
|
5
5
|
tags:
|
|
6
6
|
- 'v[0-9]*'
|
|
7
|
+
# Retry an existing tag without moving it. A publish can fail for reasons that
|
|
8
|
+
# have nothing to do with the code — a stale action pin, a PyPI outage — and
|
|
9
|
+
# with only the tag trigger the choices were to delete and re-push the tag or
|
|
10
|
+
# to burn a version number on a CI fix. Neither is a good answer to
|
|
11
|
+
# "the upload failed, run it again".
|
|
12
|
+
workflow_dispatch:
|
|
13
|
+
inputs:
|
|
14
|
+
tag:
|
|
15
|
+
description: Existing tag to build and publish (e.g. v1.2.3)
|
|
16
|
+
required: true
|
|
17
|
+
type: string
|
|
7
18
|
|
|
8
19
|
concurrency:
|
|
9
|
-
group: pypi-publish-${{ github.ref_name }}
|
|
20
|
+
group: pypi-publish-${{ inputs.tag || github.ref_name }}
|
|
10
21
|
cancel-in-progress: false
|
|
11
22
|
|
|
12
23
|
permissions:
|
|
@@ -16,8 +27,13 @@ jobs:
|
|
|
16
27
|
build:
|
|
17
28
|
name: Build distribution
|
|
18
29
|
runs-on: ubuntu-latest
|
|
30
|
+
env:
|
|
31
|
+
# The tag being released, whether it arrived by push or by dispatch.
|
|
32
|
+
TAG: ${{ inputs.tag || github.ref_name }}
|
|
19
33
|
steps:
|
|
20
34
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
35
|
+
with:
|
|
36
|
+
ref: ${{ inputs.tag || github.ref_name }}
|
|
21
37
|
|
|
22
38
|
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
|
23
39
|
with:
|
|
@@ -28,11 +44,11 @@ jobs:
|
|
|
28
44
|
|
|
29
45
|
- name: Verify tag matches pyproject version
|
|
30
46
|
run: |
|
|
31
|
-
if [[ ! "$
|
|
32
|
-
echo "Release tag '$
|
|
47
|
+
if [[ ! "$TAG" =~ ^v[0-9] ]]; then
|
|
48
|
+
echo "Release tag '$TAG' must start with 'v' followed by a digit (e.g. v1.0.0)" >&2
|
|
33
49
|
exit 1
|
|
34
50
|
fi
|
|
35
|
-
tag="${
|
|
51
|
+
tag="${TAG#v}"
|
|
36
52
|
pkg_version=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['version'])")
|
|
37
53
|
if [ "$tag" != "$pkg_version" ]; then
|
|
38
54
|
echo "Release tag ($tag) does not match pyproject.toml version ($pkg_version)" >&2
|
|
@@ -65,5 +81,10 @@ jobs:
|
|
|
65
81
|
name: dist
|
|
66
82
|
path: dist/
|
|
67
83
|
|
|
84
|
+
# v1.13.0's bundled twine rejects `Metadata-Version: 2.5`, which current
|
|
85
|
+
# hatchling emits: `InvalidDistribution: '2.5' is not a valid metadata
|
|
86
|
+
# version`. The build job's own `twine check --strict` passes, because it
|
|
87
|
+
# pip-installs a current twine — so the failure appears only at upload,
|
|
88
|
+
# after the tag is already public.
|
|
68
89
|
- name: Publish via Trusted Publishing
|
|
69
|
-
uses: pypa/gh-action-pypi-publish@
|
|
90
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -4,6 +4,15 @@ on:
|
|
|
4
4
|
push:
|
|
5
5
|
tags:
|
|
6
6
|
- 'v[0-9]*'
|
|
7
|
+
# Repair the Release for a tag that already exists, without moving it. See
|
|
8
|
+
# RELEASING.md, "If a release workflow fails" — added alongside this trigger,
|
|
9
|
+
# since a capability nobody can find is not much better than not having it.
|
|
10
|
+
workflow_dispatch:
|
|
11
|
+
inputs:
|
|
12
|
+
tag:
|
|
13
|
+
description: Existing tag to create or update a Release for (e.g. v1.2.3)
|
|
14
|
+
required: true
|
|
15
|
+
type: string
|
|
7
16
|
|
|
8
17
|
permissions:
|
|
9
18
|
contents: write
|
|
@@ -12,8 +21,13 @@ jobs:
|
|
|
12
21
|
release:
|
|
13
22
|
name: Create GitHub Release
|
|
14
23
|
runs-on: ubuntu-latest
|
|
24
|
+
env:
|
|
25
|
+
# The tag being released, whether it arrived by push or by dispatch.
|
|
26
|
+
TAG: ${{ inputs.tag || github.ref_name }}
|
|
15
27
|
steps:
|
|
16
28
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
|
29
|
+
with:
|
|
30
|
+
ref: ${{ inputs.tag || github.ref_name }}
|
|
17
31
|
|
|
18
32
|
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
|
19
33
|
with:
|
|
@@ -23,7 +37,7 @@ jobs:
|
|
|
23
37
|
id: meta
|
|
24
38
|
run: |
|
|
25
39
|
pkg_name=$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text())['project']['name'])")
|
|
26
|
-
pkg_version="${
|
|
40
|
+
pkg_version="${TAG#v}"
|
|
27
41
|
echo "name=${pkg_name}" >> "$GITHUB_OUTPUT"
|
|
28
42
|
echo "version=${pkg_version}" >> "$GITHUB_OUTPUT"
|
|
29
43
|
|
|
@@ -31,7 +45,7 @@ jobs:
|
|
|
31
45
|
id: notes
|
|
32
46
|
run: |
|
|
33
47
|
set -euo pipefail
|
|
34
|
-
version="${
|
|
48
|
+
version="${TAG#v}"
|
|
35
49
|
if [[ -f CHANGELOG.md ]]; then
|
|
36
50
|
body="$(python scripts/extract-changelog.py "$version")"
|
|
37
51
|
else
|
|
@@ -47,8 +61,14 @@ jobs:
|
|
|
47
61
|
- name: Create GitHub Release
|
|
48
62
|
uses: softprops/action-gh-release@da05d552573ad5aba039eaac05058a918a7bf631 # v2.2.2
|
|
49
63
|
with:
|
|
50
|
-
tag_name: ${{ github.ref_name }}
|
|
64
|
+
tag_name: ${{ inputs.tag || github.ref_name }}
|
|
51
65
|
name: ${{ steps.meta.outputs.name }} ${{ steps.meta.outputs.version }}
|
|
52
66
|
body: ${{ steps.notes.outputs.body }}
|
|
53
67
|
generate_release_notes: false
|
|
54
|
-
|
|
68
|
+
# Let GitHub decide by tag date/semver rather than by run order. On a
|
|
69
|
+
# push the tag is the newest version and becomes latest; on a dispatch
|
|
70
|
+
# repairing an older tag, a newer release keeps the badge. Note `false`
|
|
71
|
+
# is not "leave alone" — it explicitly marks a release NOT latest, so
|
|
72
|
+
# gating on the event would demote the newest tag in the very case this
|
|
73
|
+
# trigger exists for: repairing its Release after a failed push run.
|
|
74
|
+
make_latest: legacy
|
|
@@ -8,60 +8,104 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
## [0.12.0] - 2026-08-11
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
|
|
15
|
+
- Table storage layout, both directions. `add_managed_table()` and
|
|
16
|
+
`create_managed_database()` take `partition_by` / `sorted_by`, and
|
|
17
|
+
`managed_table_layout()` reads back what was actually declared as a
|
|
18
|
+
`TableLayout`. `TablePartitionKey` and `TableSortKey` are re-exported so
|
|
19
|
+
callers need one import.
|
|
20
|
+
|
|
21
|
+
Both halves matter because a layout is fixed when the table is created and
|
|
22
|
+
there is no alter path: a table declared without one keeps that shape until it
|
|
23
|
+
is recreated and its data rewritten. So declaring is not enough — a caller has
|
|
24
|
+
to be able to confirm it took, and to refuse to load when it cannot.
|
|
25
|
+
|
|
26
|
+
`managed_table_layout()` raises `KeyError` for a table that is not declared,
|
|
27
|
+
rather than returning an empty layout. "Not there" and "declared without a
|
|
28
|
+
layout" lead to opposite decisions for a caller.
|
|
29
|
+
|
|
30
|
+
Until now this package could not express a layout at all, which is why at least
|
|
31
|
+
one consumer hand-built the HTTP request instead. The generated key models are
|
|
32
|
+
passed through rather than wrapped, so the transform vocabulary stays exactly
|
|
33
|
+
the API's.
|
|
34
|
+
|
|
35
|
+
### Changed
|
|
36
|
+
|
|
37
|
+
- Require `hotdata>=0.9.0,<0.10`. 0.9.0 is the first release whose models carry
|
|
38
|
+
`partition_by` / `sorted_by` on the add-table request, the create-database
|
|
39
|
+
table declarations, and the table-info response. On an older `hotdata` the
|
|
40
|
+
fields would be silently dropped by the model and the table declared without a
|
|
41
|
+
layout, returning success — which is the failure this feature exists to end.
|
|
42
|
+
|
|
43
|
+
## [0.11.0] - 2026-08-11
|
|
44
|
+
|
|
45
|
+
### Changed
|
|
46
|
+
|
|
47
|
+
- Cap the `hotdata` dependency to the current minor (`>=0.8.0,<0.9`). This
|
|
48
|
+
package wraps a *generated* client, so an SDK minor can remove a model field
|
|
49
|
+
or a `Configuration` keyword this wrapper passes, and there is no regeneration
|
|
50
|
+
step here to surface it — an uncapped floor turns an SDK release into a break
|
|
51
|
+
in this package, in versions already published. Raise the cap deliberately
|
|
52
|
+
after running the suite against the new minor.
|
|
53
|
+
|
|
54
|
+
### Removed
|
|
55
|
+
|
|
56
|
+
- **Breaking:** session/sandbox support is gone. `HotdataClient` no longer accepts
|
|
57
|
+
`session_id=`, `HotdataClient.session_id` is removed, `default_session_id()` and
|
|
58
|
+
the `HOTDATA_SANDBOX` read are gone, and `list_workspaces()`,
|
|
59
|
+
`resolve_workspace_selection()` and `pick_workspace()` lose their `session_id`
|
|
60
|
+
parameter — note that loss is **positional**, so a three-argument call raises an
|
|
61
|
+
arity `TypeError` rather than an unexpected-keyword one.
|
|
62
|
+
`workspace_health_lines()` no longer emits a `sandbox` line.
|
|
63
|
+
|
|
64
|
+
**Why now.** The server stopped enforcing session scoping some time ago, so the
|
|
65
|
+
value already reached nothing. What makes removal urgent rather than tidy is
|
|
66
|
+
that the SDK is dropping the `SessionId` security scheme: against that release
|
|
67
|
+
`Configuration(session_id=...)` raises `TypeError` instead of setting a header,
|
|
68
|
+
and this package passed it unconditionally — so every `HotdataClient(...)`
|
|
69
|
+
would fail at construction. This package still pins `hotdata<0.9`, so nothing
|
|
70
|
+
is broken today; the change is what lets the cap be raised later without a
|
|
71
|
+
second breaking release.
|
|
72
|
+
|
|
73
|
+
**Migrating.** Drop `session_id=` from `HotdataClient(...)`, stop reading
|
|
74
|
+
`client.session_id`, stop setting `HOTDATA_SANDBOX`, and pass two arguments to
|
|
75
|
+
the workspace helpers. Adapters that re-export session context in their own
|
|
76
|
+
signatures — a `session_id=` parameter, a `session_id` metadata key — need to
|
|
77
|
+
remove it from theirs too, which makes their own release breaking in turn.
|
|
78
|
+
|
|
79
|
+
- `hotdata_framework.http` and `default_http_retries()`. The module existed only
|
|
80
|
+
to build the `retries=` policy removed under Fixed below, and had no other
|
|
81
|
+
callers. It predates `hotdata._retry`, which supersedes it.
|
|
82
|
+
|
|
83
|
+
### Fixed
|
|
84
|
+
|
|
85
|
+
- A `POST` is no longer replayed because of a response status. `HotdataClient`
|
|
86
|
+
passed its own `retries=` into `Configuration`, which replaced the generated
|
|
87
|
+
SDK's policy wholesale with one listing `POST` in `allowed_methods` alongside
|
|
88
|
+
a `(502, 503, 504)` forcelist — so an intermediary timing out a long request
|
|
89
|
+
produced a silent, identical re-`POST` while the server was still working on
|
|
90
|
+
the first one. For a load that is not idempotent: the duplicate collides with
|
|
91
|
+
the write lock the original holds and is refused.
|
|
92
|
+
|
|
93
|
+
The override is removed and the SDK's own default now applies. It is the
|
|
94
|
+
policy this wrapper was reaching for — `hotdata._retry` retries a
|
|
95
|
+
*pre-response* connection reset (the stale pooled socket case, where the
|
|
96
|
+
server did no work) on any method, while leaving read timeouts and status
|
|
97
|
+
retries idempotent-only.
|
|
98
|
+
|
|
11
99
|
## [0.10.0] - 2026-08-07
|
|
12
100
|
|
|
13
101
|
### Added
|
|
14
102
|
|
|
15
|
-
- `create_index(database, table, columns=..., index_type=...)` builds
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
`
|
|
21
|
-
searchable: full-text queries error without an index, and vector queries run at
|
|
22
|
-
full-scan speed. Like the other managed-table operations, `database` accepts a
|
|
23
|
-
name/id or an already-resolved `ManagedDatabase`. Indexing a table on a plain
|
|
24
|
-
(non-managed) connection is not covered — the CLI's `--catalog` handles that.
|
|
25
|
-
|
|
26
|
-
`index_name` is optional and defaults to `{table}_{columns}_{index_type}`, the same
|
|
27
|
-
derivation the CLI uses when `--name` is omitted, so both surfaces name the same
|
|
28
|
-
index identically. `index_type` is required, unlike the API's `"sorted"` default,
|
|
29
|
-
because the wrong kind only fails at query time.
|
|
30
|
-
|
|
31
|
-
The server builds the index as a background job whose submit call reports success
|
|
32
|
-
even when the build later fails, so `create_index` polls the job to a terminal
|
|
33
|
-
state and raises `RuntimeError` carrying the job's `error_message`. Pass
|
|
34
|
-
`wait=False` to return once the job is accepted (`status="pending"` plus a
|
|
35
|
-
`job_id`) and own the outcome check yourself, as the CLI's `--async` does;
|
|
36
|
-
`timeout_s` and `poll_interval_s` tune the wait.
|
|
37
|
-
|
|
38
|
-
Both vector-index modes are supported. Omitting `embedding_provider_id` indexes an
|
|
39
|
-
existing vector column, queried with a literal vector — and there `metric` must
|
|
40
|
-
match the distance function the query uses (`cosine`→`cosine_distance`,
|
|
41
|
-
`l2`→`l2_distance`, `dot`→`negative_dot_product`), since a mismatch silently
|
|
42
|
-
reverts to a full table scan rather than erroring. Setting
|
|
43
|
-
`embedding_provider_id` indexes a *text* column instead: the provider embeds it
|
|
44
|
-
into `output_column`, queries pass text via `vector_distance(source_col, 'query')`,
|
|
45
|
-
and the server resolves the distance function itself.
|
|
46
|
-
|
|
47
|
-
Argument combinations that the server would silently ignore raise `ValueError`
|
|
48
|
-
before any request is sent: an unknown `index_type` or `metric`, a vector index
|
|
49
|
-
with more than one column (the engine indexes only the first), and
|
|
50
|
-
`metric`/`dimensions`/`embedding_provider_id`/`output_column`/`description` on a
|
|
51
|
-
non-vector index.
|
|
52
|
-
|
|
53
|
-
Verified against `api.hotdata.dev` when this version was released: BM25 and vector
|
|
54
|
-
indexes both build and report `ready`, and a BM25 index is used by full-text
|
|
55
|
-
search. A *vector* index on a managed database was **not** picked up by the query
|
|
56
|
-
planner at that time — a matching `cosine_distance(...) ORDER BY ... LIMIT k` still
|
|
57
|
-
planned as a full scan. That reproduces with an index created by `hotdata indexes
|
|
58
|
-
create`, so it is an engine-side issue rather than a client one, but it means a
|
|
59
|
-
vector index built through this method may not yet accelerate queries.
|
|
60
|
-
|
|
61
|
-
- `CreateIndexResult`, the frozen dataclass `create_index` returns, is exported from
|
|
62
|
-
`hotdata_framework` and added to the public contract surface. Its `source_column`
|
|
63
|
-
names the text column to query for a provider-backed vector index, and is `None`
|
|
64
|
-
for BM25, sorted, and plain vector indexes.
|
|
103
|
+
- `create_index(database, table, columns=..., index_type=...)` builds a `bm25`,
|
|
104
|
+
`vector`, or `sorted` index on a managed table, matching `hotdata indexes create`
|
|
105
|
+
in the CLI. The build is a background job whose submit call reports success even
|
|
106
|
+
when the build later fails, so this polls the job and raises `RuntimeError` with
|
|
107
|
+
its error message; `wait=False` returns as soon as the job is accepted. Returns
|
|
108
|
+
`CreateIndexResult`, also exported.
|
|
65
109
|
|
|
66
110
|
## [0.9.0] - 2026-07-23
|
|
67
111
|
|
|
@@ -21,7 +21,6 @@ The supported import surface is:
|
|
|
21
21
|
- `workspace_health_lines`
|
|
22
22
|
- `default_api_key`
|
|
23
23
|
- `default_host`
|
|
24
|
-
- `default_session_id`
|
|
25
24
|
- `explicit_workspace_id`
|
|
26
25
|
- `list_workspaces`
|
|
27
26
|
- `normalize_host`
|
|
@@ -32,6 +31,9 @@ The supported import surface is:
|
|
|
32
31
|
- `WorkspaceSelection`
|
|
33
32
|
- `ManagedDatabase`
|
|
34
33
|
- `ManagedTable`
|
|
34
|
+
- `TableLayout`
|
|
35
|
+
- `TablePartitionKey`
|
|
36
|
+
- `TableSortKey`
|
|
35
37
|
- `LoadManagedTableResult`
|
|
36
38
|
- `CreateIndexResult`
|
|
37
39
|
- `DEFAULT_SCHEMA`
|
|
@@ -43,7 +45,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
43
45
|
|
|
44
46
|
### `HotdataClient`
|
|
45
47
|
|
|
46
|
-
- Represents runtime context: API key, host, workspace
|
|
48
|
+
- Represents runtime context: API key, host, workspace.
|
|
47
49
|
- `from_env()` resolves runtime context from env vars and selected workspace.
|
|
48
50
|
- `execute_sql(sql)` returns `QueryResult` or raises `RuntimeError`/`TimeoutError`.
|
|
49
51
|
- `get_result(result_id)` returns a ready `QueryResult` and waits for readiness when needed.
|
|
@@ -57,6 +59,8 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
57
59
|
adapters should pass `connection_id` when known.
|
|
58
60
|
- `uploads()` returns the uploads API wrapper for parquet staging.
|
|
59
61
|
- `list_managed_databases()` returns all databases via the `/databases` API.
|
|
62
|
+
- `add_managed_table(...)` and `create_managed_database(...)` accept `partition_by` / `sorted_by` to declare a table's storage layout. The layout is fixed when the table is created and cannot be altered afterwards, so omitting it is permanent for that table.
|
|
63
|
+
- `managed_table_layout(database, table, schema=...)` returns the declared layout as `TableLayout`. Empty lists mean no layout was declared — sound only because the table is resolved through a managed database. Raises `KeyError` when the table is not declared, keeping "absent" distinct from "declared without a layout".
|
|
60
64
|
- `resolve_managed_database(name_or_id)` resolves a database by id (direct lookup) or description (list scan). A `403` from `/databases` surfaces as `RuntimeError` (forbidden, not absent), preserving the underlying `ApiException` as `__cause__`.
|
|
61
65
|
- `create_managed_database(description=..., schema=..., tables=..., expires_at=...)` creates a database via the `/databases` API and optionally declares tables up front. Returns a `ManagedDatabase` (id + `default_connection_id`) sufficient to load without a further read.
|
|
62
66
|
- `delete_managed_database(name_or_id)` deletes a database via the `/databases` API.
|
|
@@ -65,7 +69,7 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
65
69
|
- `load_managed_table(database, table, schema=..., upload_id=..., file=...)` publishes parquet data into a declared managed table.
|
|
66
70
|
- `delete_managed_table(database, table, schema=...)` deletes a managed table.
|
|
67
71
|
- `create_index(database, table, schema=..., columns=..., index_type=..., index_name=...)` builds a `"sorted"`, `"bm25"`, or `"vector"` index on a managed table and returns a `CreateIndexResult`. It is the framework-side equivalent of the CLI's `hotdata indexes create`; indexing a table on a plain (non-managed) connection is out of scope. `index_name` defaults to `{table}_{columns}_{index_type}`, matching the CLI's derivation when `--name` is omitted. `index_type` is required rather than defaulting to the API's `"sorted"`. The build runs as a background job; the call polls it to a terminal state and raises `RuntimeError` with the job's `error_message` when it fails, because the submit call reports success regardless. `wait=False` returns as soon as the job is accepted, with `status="pending"` and a `job_id` for the caller to poll. For `index_type="vector"`, omitting `embedding_provider_id` indexes an existing vector column and `metric` (`"l2"`, `"cosine"`, `"dot"`) selects the distance function the index accelerates — a query using a different function silently falls back to a full scan; setting `embedding_provider_id` indexes a source *text* column instead, and the returned `source_column` names the column to pass to `vector_distance`. Argument combinations the server would silently ignore raise `ValueError` before any request is sent.
|
|
68
|
-
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
72
|
+
- The `database` argument of `list_managed_tables`, `load_managed_table`, `add_managed_table`, `delete_managed_table`, `delete_managed_database`, `create_index`, `managed_table_layout`, and `execute_sql` accepts a name/id **or** an already-resolved `ManagedDatabase`. Passing a `ManagedDatabase` skips the name/id read probe, so a create-scoped key that cannot read `/databases` can load into a database it just created.
|
|
69
73
|
|
|
70
74
|
### `QueryResult`
|
|
71
75
|
|
|
@@ -79,7 +83,6 @@ Adapters should import from `hotdata_framework` and treat this surface as the st
|
|
|
79
83
|
|
|
80
84
|
- `default_api_key()` reads `HOTDATA_API_KEY`.
|
|
81
85
|
- `default_host()` reads `HOTDATA_API_URL` (default: `https://api.hotdata.dev`) and normalizes it.
|
|
82
|
-
- `default_session_id()` reads `HOTDATA_SANDBOX`.
|
|
83
86
|
- `explicit_workspace_id()` reads `HOTDATA_WORKSPACE` (workspace public id).
|
|
84
87
|
- `pick_workspace()` prefers explicit env workspace, then active workspace, then first workspace.
|
|
85
88
|
- `resolve_workspace_selection()` is the canonical workspace selection algorithm. It returns `WorkspaceSelection` with selected workspace id, selection source, and discovered workspaces when auto-selected.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: hotdata-framework
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Python framework for building Hotdata integrations: workspace
|
|
3
|
+
Version: 0.12.0
|
|
4
|
+
Summary: Python framework for building Hotdata integrations: workspace runtime, query execution, and managed databases
|
|
5
5
|
Project-URL: Homepage, https://www.hotdata.dev
|
|
6
6
|
Project-URL: Documentation, https://www.hotdata.dev/docs
|
|
7
7
|
Project-URL: Repository, https://github.com/hotdata-dev/sdk-python-framework
|
|
@@ -21,7 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
|
21
21
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
22
|
Classifier: Typing :: Typed
|
|
23
23
|
Requires-Python: >=3.10
|
|
24
|
-
Requires-Dist: hotdata
|
|
24
|
+
Requires-Dist: hotdata<0.10,>=0.9.0
|
|
25
25
|
Requires-Dist: pandas>=2.0
|
|
26
26
|
Requires-Dist: pyarrow>=14.0
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
@@ -30,16 +30,15 @@ Description-Content-Type: text/markdown
|
|
|
30
30
|
|
|
31
31
|
**A Python framework for building Hotdata integrations.**
|
|
32
32
|
|
|
33
|
-
Shared runtime primitives for Hotdata integrations: workspace
|
|
33
|
+
Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
|
|
34
34
|
|
|
35
35
|
Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
36
36
|
|
|
37
37
|
## Features
|
|
38
38
|
|
|
39
|
-
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`,
|
|
39
|
+
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
40
40
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
41
|
-
- **
|
|
42
|
-
- **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
|
|
41
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
|
|
43
42
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
44
43
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
45
44
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -2,16 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
**A Python framework for building Hotdata integrations.**
|
|
4
4
|
|
|
5
|
-
Shared runtime primitives for Hotdata integrations: workspace
|
|
5
|
+
Shared runtime primitives for Hotdata integrations: workspace semantics, execution context, query state, run history, and replayable result handles. Framework packages (Marimo, Jupyter, Streamlit, LangGraph) depend on this package.
|
|
6
6
|
|
|
7
7
|
Runtime boundary and guarantees are defined in `CONTRACT.md`.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
|
-
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`,
|
|
11
|
+
- **Environment-driven client setup** — create clients from `HOTDATA_API_KEY`, optional `HOTDATA_API_URL`, and `HOTDATA_WORKSPACE`.
|
|
12
12
|
- **Workspace resolution** — choose an explicit workspace from env, otherwise discover workspaces and select the active workspace or first available workspace.
|
|
13
|
-
- **
|
|
14
|
-
- **HTTP resilience** — configure SDK retries for transient connection failures and retry SQL execution on stale pooled sockets.
|
|
13
|
+
- **HTTP resilience** — retry SQL execution on stale pooled sockets. Transport-level retries are the SDK's own default, which this package leaves in place so a non-idempotent request is never replayed on a response status.
|
|
15
14
|
- **SQL execution helper** — run SQL through `POST /v1/query`, poll async query runs when needed, and return a `QueryResult`.
|
|
16
15
|
- **Result utilities** — convert query results to records, pandas DataFrames, or metadata dictionaries for adapter display layers.
|
|
17
16
|
- **History helpers** — list recent results and query run history with normalized dataclasses.
|
|
@@ -34,6 +34,27 @@ Pushing a `vX.Y.Z` tag triggers two workflows:
|
|
|
34
34
|
| `publish.yml` | Build wheel/sdist and publish to PyPI |
|
|
35
35
|
| `release.yml` | Create the GitHub Release with notes from `CHANGELOG.md` |
|
|
36
36
|
|
|
37
|
+
## If a release workflow fails
|
|
38
|
+
|
|
39
|
+
Both workflows also accept a manual re-run against an existing tag, so a failure
|
|
40
|
+
unrelated to the code — a stale action pin, a PyPI outage — does not require
|
|
41
|
+
deleting the tag or burning a version number:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
gh workflow run "Publish to PyPI" --ref main -f tag=vX.Y.Z
|
|
45
|
+
gh workflow run "GitHub Release" --ref main -f tag=vX.Y.Z
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`--ref main` selects the workflow *definition*, so a fix to the workflow file
|
|
49
|
+
itself is picked up; everything it runs — including `scripts/extract-changelog.py`
|
|
50
|
+
— still comes from the tag, since that is what is being built and released. The
|
|
51
|
+
two refs serve different purposes, which is why they can differ.
|
|
52
|
+
|
|
53
|
+
A version is only spent once PyPI has accepted an upload. If the publish failed
|
|
54
|
+
before that, the same version can still be published — check with
|
|
55
|
+
`curl -s -o /dev/null -w '%{http_code}' https://pypi.org/pypi/<pkg>/<version>/json`
|
|
56
|
+
returning 404.
|
|
57
|
+
|
|
37
58
|
## Enforcement
|
|
38
59
|
|
|
39
60
|
- **PR check** (`check-release.yml`): if `pyproject.toml` version changes, `CHANGELOG.md` must contain a matching `## [X.Y.Z]` section.
|
|
@@ -2,6 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
from importlib.metadata import PackageNotFoundError, version
|
|
4
4
|
|
|
5
|
+
from hotdata.models.table_partition_key import TablePartitionKey
|
|
6
|
+
from hotdata.models.table_sort_key import TableSortKey
|
|
7
|
+
|
|
5
8
|
from hotdata_framework.client import (
|
|
6
9
|
HotdataClient,
|
|
7
10
|
ResultSummary,
|
|
@@ -14,13 +17,13 @@ from hotdata_framework.databases import (
|
|
|
14
17
|
LoadManagedTableResult,
|
|
15
18
|
ManagedDatabase,
|
|
16
19
|
ManagedTable,
|
|
20
|
+
TableLayout,
|
|
17
21
|
is_parquet_path,
|
|
18
22
|
)
|
|
19
23
|
from hotdata_framework.env import (
|
|
20
24
|
WorkspaceSelection,
|
|
21
25
|
default_api_key,
|
|
22
26
|
default_host,
|
|
23
|
-
default_session_id,
|
|
24
27
|
explicit_workspace_id,
|
|
25
28
|
list_workspaces,
|
|
26
29
|
normalize_host,
|
|
@@ -56,12 +59,14 @@ __all__ = [
|
|
|
56
59
|
"QueryResult",
|
|
57
60
|
"ResultSummary",
|
|
58
61
|
"RunHistoryItem",
|
|
62
|
+
"TableLayout",
|
|
63
|
+
"TablePartitionKey",
|
|
64
|
+
"TableSortKey",
|
|
59
65
|
"WorkspaceSelection",
|
|
60
66
|
"__version__",
|
|
61
67
|
"classify_sdk_error",
|
|
62
68
|
"default_api_key",
|
|
63
69
|
"default_host",
|
|
64
|
-
"default_session_id",
|
|
65
70
|
"explicit_workspace_id",
|
|
66
71
|
"from_env",
|
|
67
72
|
"is_parquet_path",
|