api-dock 0.8.0__tar.gz → 0.8.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {api_dock-0.8.0 → api_dock-0.8.2}/PKG-INFO +106 -33
- {api_dock-0.8.0 → api_dock-0.8.2}/README.md +105 -32
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/database_config.py +130 -1
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/config.yaml +5 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/config.yaml +12 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/fast_api.py +32 -3
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/flask_api.py +25 -2
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/route_mapper.py +207 -32
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/sql_builder.py +251 -24
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/types.py +24 -1
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/PKG-INFO +106 -33
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/SOURCES.txt +2 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/pyproject.toml +1 -1
- api_dock-0.8.2/tests/test_runtime_settings.py +249 -0
- api_dock-0.8.2/tests/test_schema_unions.py +308 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/LICENSE.md +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/__init__.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/auth.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/cli.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/config.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/config_discovery.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/encryption.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/example_db.yaml +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/remotes/example_remote.yaml +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/listings.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/storage_auth.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/dependency_links.txt +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/entry_points.txt +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/requires.txt +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/top_level.txt +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/setup.cfg +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_inject_cookies.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_listings.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_proxy_pipeline.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_shared_database_config.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_sql_builder.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_sql_selector.py +0 -0
- {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: api_dock
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.2
|
|
4
4
|
Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
|
|
5
5
|
Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
|
|
6
6
|
License-Expression: BSD-3-Clause
|
|
@@ -219,19 +219,33 @@ remotes:
|
|
|
219
219
|
settings:
|
|
220
220
|
add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
|
|
221
221
|
follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
|
|
222
|
+
follow_redirects: true # Follow remote redirects (default: true)
|
|
222
223
|
timeout: 10 # Upstream request timeout in seconds (default: 10)
|
|
224
|
+
base_path: /dock # Also serve the API under this prefix (default: none)
|
|
225
|
+
duckdb: # Options for database queries (default: none)
|
|
226
|
+
memory_limit: 700MB
|
|
227
|
+
threads: 2
|
|
228
|
+
max_concurrent_queries: 2
|
|
223
229
|
```
|
|
224
230
|
|
|
225
|
-
###
|
|
231
|
+
### Settings
|
|
226
232
|
|
|
227
|
-
The optional `settings` section controls HTTP behavior:
|
|
233
|
+
The optional `settings` section controls HTTP and query behavior:
|
|
228
234
|
|
|
229
235
|
- **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
|
|
230
236
|
|
|
231
237
|
- **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
|
|
232
238
|
|
|
239
|
+
- **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
|
|
240
|
+
|
|
233
241
|
- **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
|
|
234
242
|
|
|
243
|
+
- **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
|
|
244
|
+
|
|
245
|
+
- **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
|
|
246
|
+
|
|
247
|
+
Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
|
|
248
|
+
|
|
235
249
|
### Catalog Endpoints (`expose`)
|
|
236
250
|
|
|
237
251
|
The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
|
|
@@ -563,6 +577,60 @@ Notes:
|
|
|
563
577
|
- Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
|
|
564
578
|
- Views are created only for the `[[schema.table]]` tables a query actually references.
|
|
565
579
|
|
|
580
|
+
#### Querying across schemas (`[[*.table]]`, schema groups)
|
|
581
|
+
|
|
582
|
+
Union references read the same table from several schemas at once:
|
|
583
|
+
|
|
584
|
+
| Reference | Reads |
|
|
585
|
+
|---|---|
|
|
586
|
+
| `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
|
|
587
|
+
| `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
|
|
588
|
+
| `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
|
|
589
|
+
| `[[group1!.detections]]` | that group, minus the current schema |
|
|
590
|
+
|
|
591
|
+
```yaml
|
|
592
|
+
# api_dock_config/databases/config.yaml
|
|
593
|
+
schema_groups: # named lists of shared schemas
|
|
594
|
+
birdnet_models:
|
|
595
|
+
- birdnet_2p4
|
|
596
|
+
- birdnet_bullfrog_2p4v0p5
|
|
597
|
+
```
|
|
598
|
+
|
|
599
|
+
- A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
|
|
600
|
+
- `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
|
|
601
|
+
- Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
|
|
602
|
+
|
|
603
|
+
**Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
|
|
604
|
+
|
|
605
|
+
```yaml
|
|
606
|
+
source_columns: [schema, name, version] # adds schema_name, name, version
|
|
607
|
+
source_columns: {schema: _schema, name: model} # pick a subset and rename
|
|
608
|
+
```
|
|
609
|
+
|
|
610
|
+
A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
|
|
611
|
+
|
|
612
|
+
**`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
|
|
613
|
+
|
|
614
|
+
Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
|
|
615
|
+
|
|
616
|
+
```yaml
|
|
617
|
+
routes:
|
|
618
|
+
- route: detections/{{id}}/overlaps
|
|
619
|
+
source_columns: [schema, name, version]
|
|
620
|
+
sql: |
|
|
621
|
+
WITH src AS (
|
|
622
|
+
SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
|
|
623
|
+
)
|
|
624
|
+
SELECT detections.*
|
|
625
|
+
FROM [[*.detections]] detections
|
|
626
|
+
JOIN src ON detections.recording_id = src.recording_id
|
|
627
|
+
AND detections.start_time < src.end_time
|
|
628
|
+
AND detections.end_time > src.start_time
|
|
629
|
+
WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
|
|
630
|
+
```
|
|
631
|
+
|
|
632
|
+
Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
|
|
633
|
+
|
|
566
634
|
#### Inline database configs (`slugs`)
|
|
567
635
|
|
|
568
636
|
Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
|
|
@@ -1543,47 +1611,52 @@ pixi run python scripts/hello_world.py
|
|
|
1543
1611
|
|
|
1544
1612
|
## Publishing a Release
|
|
1545
1613
|
|
|
1614
|
+
Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
|
|
1615
|
+
|
|
1546
1616
|
```bash
|
|
1547
|
-
# 0.
|
|
1617
|
+
# 0. Start from a clean, up-to-date main
|
|
1618
|
+
export VERSION=0.8.2 # the NEW version, no leading "v"
|
|
1619
|
+
git checkout main
|
|
1620
|
+
git pull origin main
|
|
1621
|
+
git status
|
|
1548
1622
|
|
|
1549
|
-
# 1.
|
|
1623
|
+
# 1. Set `version` in pyproject.toml to $VERSION
|
|
1550
1624
|
|
|
1551
|
-
# 2.
|
|
1625
|
+
# 2. Run the tests
|
|
1626
|
+
pixi run -e dev pytest -q
|
|
1627
|
+
|
|
1628
|
+
# 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
|
|
1629
|
+
export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
|
|
1552
1630
|
git add -A
|
|
1553
|
-
git commit -m "
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
|
|
1562
|
-
find . -name "*.pyc" -delete
|
|
1563
|
-
pixi run -e dev python -m build --wheel
|
|
1564
|
-
ls dist/*.whl
|
|
1565
|
-
|
|
1566
|
-
# 5. Create GitHub release with the wheel attached
|
|
1567
|
-
gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
|
|
1568
|
-
--title "v0.6.1" --notes "$(cat <<'EOF'
|
|
1631
|
+
git commit -m "v$VERSION: $COMMIT_MESSAGE"
|
|
1632
|
+
git tag "v$VERSION"
|
|
1633
|
+
git push origin main "v$VERSION"
|
|
1634
|
+
|
|
1635
|
+
# 4. Publish the GitHub Release; this triggers the PyPI upload.
|
|
1636
|
+
# Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
|
|
1637
|
+
gh release create "v$VERSION" \
|
|
1638
|
+
--title "v$VERSION" \
|
|
1639
|
+
--notes "$(cat <<'EOF'
|
|
1569
1640
|
* new features
|
|
1570
|
-
-
|
|
1571
|
-
-
|
|
1641
|
+
- `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
|
|
1642
|
+
- `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
|
|
1572
1643
|
* bug fixes
|
|
1573
|
-
-
|
|
1574
|
-
-
|
|
1575
|
-
- Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
|
|
1644
|
+
- Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
|
|
1645
|
+
- Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
|
|
1576
1646
|
* cleanup / other improvements
|
|
1577
|
-
-
|
|
1578
|
-
-
|
|
1579
|
-
- Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
|
|
1647
|
+
- README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
|
|
1648
|
+
- Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
|
|
1580
1649
|
EOF
|
|
1581
1650
|
)"
|
|
1582
1651
|
|
|
1583
|
-
#
|
|
1584
|
-
|
|
1585
|
-
|
|
1652
|
+
# 5. Watch the publish workflow, then confirm PyPI has the new version
|
|
1653
|
+
gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
|
|
1654
|
+
curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
|
|
1586
1655
|
|
|
1656
|
+
# 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
|
|
1657
|
+
# run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
|
|
1658
|
+
gh pr list --repo conda-forge/api_dock-feedstock --state open
|
|
1659
|
+
```
|
|
1587
1660
|
|
|
1588
1661
|
---
|
|
1589
1662
|
|
|
@@ -180,19 +180,33 @@ remotes:
|
|
|
180
180
|
settings:
|
|
181
181
|
add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
|
|
182
182
|
follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
|
|
183
|
+
follow_redirects: true # Follow remote redirects (default: true)
|
|
183
184
|
timeout: 10 # Upstream request timeout in seconds (default: 10)
|
|
185
|
+
base_path: /dock # Also serve the API under this prefix (default: none)
|
|
186
|
+
duckdb: # Options for database queries (default: none)
|
|
187
|
+
memory_limit: 700MB
|
|
188
|
+
threads: 2
|
|
189
|
+
max_concurrent_queries: 2
|
|
184
190
|
```
|
|
185
191
|
|
|
186
|
-
###
|
|
192
|
+
### Settings
|
|
187
193
|
|
|
188
|
-
The optional `settings` section controls HTTP behavior:
|
|
194
|
+
The optional `settings` section controls HTTP and query behavior:
|
|
189
195
|
|
|
190
196
|
- **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
|
|
191
197
|
|
|
192
198
|
- **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
|
|
193
199
|
|
|
200
|
+
- **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
|
|
201
|
+
|
|
194
202
|
- **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
|
|
195
203
|
|
|
204
|
+
- **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
|
|
205
|
+
|
|
206
|
+
- **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
|
|
207
|
+
|
|
208
|
+
Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
|
|
209
|
+
|
|
196
210
|
### Catalog Endpoints (`expose`)
|
|
197
211
|
|
|
198
212
|
The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
|
|
@@ -524,6 +538,60 @@ Notes:
|
|
|
524
538
|
- Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
|
|
525
539
|
- Views are created only for the `[[schema.table]]` tables a query actually references.
|
|
526
540
|
|
|
541
|
+
#### Querying across schemas (`[[*.table]]`, schema groups)
|
|
542
|
+
|
|
543
|
+
Union references read the same table from several schemas at once:
|
|
544
|
+
|
|
545
|
+
| Reference | Reads |
|
|
546
|
+
|---|---|
|
|
547
|
+
| `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
|
|
548
|
+
| `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
|
|
549
|
+
| `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
|
|
550
|
+
| `[[group1!.detections]]` | that group, minus the current schema |
|
|
551
|
+
|
|
552
|
+
```yaml
|
|
553
|
+
# api_dock_config/databases/config.yaml
|
|
554
|
+
schema_groups: # named lists of shared schemas
|
|
555
|
+
birdnet_models:
|
|
556
|
+
- birdnet_2p4
|
|
557
|
+
- birdnet_bullfrog_2p4v0p5
|
|
558
|
+
```
|
|
559
|
+
|
|
560
|
+
- A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
|
|
561
|
+
- `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
|
|
562
|
+
- Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
|
|
563
|
+
|
|
564
|
+
**Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
|
|
565
|
+
|
|
566
|
+
```yaml
|
|
567
|
+
source_columns: [schema, name, version] # adds schema_name, name, version
|
|
568
|
+
source_columns: {schema: _schema, name: model} # pick a subset and rename
|
|
569
|
+
```
|
|
570
|
+
|
|
571
|
+
A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
|
|
572
|
+
|
|
573
|
+
**`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
|
|
574
|
+
|
|
575
|
+
Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
|
|
576
|
+
|
|
577
|
+
```yaml
|
|
578
|
+
routes:
|
|
579
|
+
- route: detections/{{id}}/overlaps
|
|
580
|
+
source_columns: [schema, name, version]
|
|
581
|
+
sql: |
|
|
582
|
+
WITH src AS (
|
|
583
|
+
SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
|
|
584
|
+
)
|
|
585
|
+
SELECT detections.*
|
|
586
|
+
FROM [[*.detections]] detections
|
|
587
|
+
JOIN src ON detections.recording_id = src.recording_id
|
|
588
|
+
AND detections.start_time < src.end_time
|
|
589
|
+
AND detections.end_time > src.start_time
|
|
590
|
+
WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
|
|
591
|
+
```
|
|
592
|
+
|
|
593
|
+
Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
|
|
594
|
+
|
|
527
595
|
#### Inline database configs (`slugs`)
|
|
528
596
|
|
|
529
597
|
Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
|
|
@@ -1504,47 +1572,52 @@ pixi run python scripts/hello_world.py
|
|
|
1504
1572
|
|
|
1505
1573
|
## Publishing a Release
|
|
1506
1574
|
|
|
1575
|
+
Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
|
|
1576
|
+
|
|
1507
1577
|
```bash
|
|
1508
|
-
# 0.
|
|
1578
|
+
# 0. Start from a clean, up-to-date main
|
|
1579
|
+
export VERSION=0.8.2 # the NEW version, no leading "v"
|
|
1580
|
+
git checkout main
|
|
1581
|
+
git pull origin main
|
|
1582
|
+
git status
|
|
1509
1583
|
|
|
1510
|
-
# 1.
|
|
1584
|
+
# 1. Set `version` in pyproject.toml to $VERSION
|
|
1511
1585
|
|
|
1512
|
-
# 2.
|
|
1586
|
+
# 2. Run the tests
|
|
1587
|
+
pixi run -e dev pytest -q
|
|
1588
|
+
|
|
1589
|
+
# 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
|
|
1590
|
+
export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
|
|
1513
1591
|
git add -A
|
|
1514
|
-
git commit -m "
|
|
1515
|
-
|
|
1516
|
-
|
|
1517
|
-
|
|
1518
|
-
|
|
1519
|
-
|
|
1520
|
-
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
find . -name "*.pyc" -delete
|
|
1524
|
-
pixi run -e dev python -m build --wheel
|
|
1525
|
-
ls dist/*.whl
|
|
1526
|
-
|
|
1527
|
-
# 5. Create GitHub release with the wheel attached
|
|
1528
|
-
gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
|
|
1529
|
-
--title "v0.6.1" --notes "$(cat <<'EOF'
|
|
1592
|
+
git commit -m "v$VERSION: $COMMIT_MESSAGE"
|
|
1593
|
+
git tag "v$VERSION"
|
|
1594
|
+
git push origin main "v$VERSION"
|
|
1595
|
+
|
|
1596
|
+
# 4. Publish the GitHub Release; this triggers the PyPI upload.
|
|
1597
|
+
# Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
|
|
1598
|
+
gh release create "v$VERSION" \
|
|
1599
|
+
--title "v$VERSION" \
|
|
1600
|
+
--notes "$(cat <<'EOF'
|
|
1530
1601
|
* new features
|
|
1531
|
-
-
|
|
1532
|
-
-
|
|
1602
|
+
- `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
|
|
1603
|
+
- `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
|
|
1533
1604
|
* bug fixes
|
|
1534
|
-
-
|
|
1535
|
-
-
|
|
1536
|
-
- Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
|
|
1605
|
+
- Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
|
|
1606
|
+
- Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
|
|
1537
1607
|
* cleanup / other improvements
|
|
1538
|
-
-
|
|
1539
|
-
-
|
|
1540
|
-
- Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
|
|
1608
|
+
- README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
|
|
1609
|
+
- Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
|
|
1541
1610
|
EOF
|
|
1542
1611
|
)"
|
|
1543
1612
|
|
|
1544
|
-
#
|
|
1545
|
-
|
|
1546
|
-
|
|
1613
|
+
# 5. Watch the publish workflow, then confirm PyPI has the new version
|
|
1614
|
+
gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
|
|
1615
|
+
curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
|
|
1547
1616
|
|
|
1617
|
+
# 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
|
|
1618
|
+
# run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
|
|
1619
|
+
gh pr list --repo conda-forge/api_dock-feedstock --state open
|
|
1620
|
+
```
|
|
1548
1621
|
|
|
1549
1622
|
---
|
|
1550
1623
|
|
|
@@ -56,6 +56,15 @@ SLUG_NAME_KEY: str = "name"
|
|
|
56
56
|
SLUG_VERSION_KEY: str = "version"
|
|
57
57
|
SLUG_VERSIONS_KEY: str = "versions"
|
|
58
58
|
|
|
59
|
+
# Named groups of shared schemas: {group: [schema, ...]}, referenced in SQL as
|
|
60
|
+
# [[group.table]]. Group names may not collide with schema names.
|
|
61
|
+
SCHEMA_GROUPS_KEY: str = "schema_groups"
|
|
62
|
+
|
|
63
|
+
# Union selectors: [[*.table]] = every shared schema with that table; a trailing
|
|
64
|
+
# "!" ([[*!.table]], [[group!.table]]) drops the current version's schema.
|
|
65
|
+
ALL_SCHEMAS: str = "*"
|
|
66
|
+
EXCLUDE_SELF_SUFFIX: str = "!"
|
|
67
|
+
|
|
59
68
|
# Exclusion version wildcard: matches every version (and unversioned databases).
|
|
60
69
|
ALL_VERSIONS: str = "*"
|
|
61
70
|
|
|
@@ -212,7 +221,7 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
|
|
|
212
221
|
"""Load the whole shared database config file (``databases/config.yaml``).
|
|
213
222
|
|
|
214
223
|
Top-level keys: ``database`` (tables, ``meta``, ``schema``), ``slugs``
|
|
215
|
-
(inline database/version configs), ``routes`` and
|
|
224
|
+
(inline database/version configs), ``schema_groups``, ``routes`` and
|
|
216
225
|
``query_params`` (added to every database/version), and the
|
|
217
226
|
``route_inclusions`` / ``query_inclusions`` and ``route_exclusions`` /
|
|
218
227
|
``query_exclusions`` lists.
|
|
@@ -250,12 +259,15 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
|
|
|
250
259
|
ROUTE_EXCLUSIONS_KEY: list,
|
|
251
260
|
QUERY_EXCLUSIONS_KEY: list,
|
|
252
261
|
SLUGS_KEY: list,
|
|
262
|
+
SCHEMA_GROUPS_KEY: dict,
|
|
253
263
|
}
|
|
254
264
|
for key, expected in expected_types.items():
|
|
255
265
|
value = contents.get(key) or expected()
|
|
256
266
|
if not isinstance(value, expected):
|
|
257
267
|
raise ValueError(f"'{key}' in {shared_path} must be a {expected.__name__}")
|
|
258
268
|
normalized[key] = value
|
|
269
|
+
|
|
270
|
+
_validate_schema_groups(normalized, shared_path)
|
|
259
271
|
return normalized
|
|
260
272
|
|
|
261
273
|
|
|
@@ -418,6 +430,92 @@ def resolve_table_reference(
|
|
|
418
430
|
return None
|
|
419
431
|
|
|
420
432
|
|
|
433
|
+
def resolve_schema_union(
|
|
434
|
+
selector: str,
|
|
435
|
+
table_name: str,
|
|
436
|
+
shared_config: Optional[Dict[str, Any]] = None,
|
|
437
|
+
schema_groups: Optional[Dict[str, List[str]]] = None) -> Optional[List[TableReference]]:
|
|
438
|
+
"""Resolve a union selector (``*`` or a schema group) to its member tables.
|
|
439
|
+
|
|
440
|
+
Args:
|
|
441
|
+
selector: ``*`` for every shared schema, or a ``schema_groups`` name
|
|
442
|
+
(without any trailing ``!``).
|
|
443
|
+
table_name: Table to read from each member schema.
|
|
444
|
+
shared_config: The shared ``database`` mapping, or None.
|
|
445
|
+
schema_groups: The shared ``schema_groups`` mapping, or None.
|
|
446
|
+
|
|
447
|
+
Returns:
|
|
448
|
+
Qualified TableReferences in schema/group order, or None if the
|
|
449
|
+
selector is neither ``*`` nor a group (i.e. it names a single schema).
|
|
450
|
+
|
|
451
|
+
Raises:
|
|
452
|
+
ValueError: If no schema has the table (``*``), or a group member is
|
|
453
|
+
not a schema or lacks the table.
|
|
454
|
+
"""
|
|
455
|
+
schemas = (shared_config or {}).get(SHARED_SCHEMA_KEY) or {}
|
|
456
|
+
groups = schema_groups or {}
|
|
457
|
+
|
|
458
|
+
if selector == ALL_SCHEMAS:
|
|
459
|
+
members = [name for name, tables in schemas.items() if table_name in (tables or {})]
|
|
460
|
+
if not members:
|
|
461
|
+
raise ValueError(f"No shared schema has a table named '{table_name}'")
|
|
462
|
+
elif selector in groups:
|
|
463
|
+
members = list(groups[selector])
|
|
464
|
+
for schema_name in members:
|
|
465
|
+
if table_name not in (schemas.get(schema_name) or {}):
|
|
466
|
+
raise ValueError(
|
|
467
|
+
f"Schema '{schema_name}' in group '{selector}' has no table '{table_name}'"
|
|
468
|
+
)
|
|
469
|
+
else:
|
|
470
|
+
return None
|
|
471
|
+
|
|
472
|
+
references = []
|
|
473
|
+
for schema_name in members:
|
|
474
|
+
reference = resolve_table_reference(
|
|
475
|
+
f"{schema_name}{SCHEMA_SEPARATOR}{table_name}", {}, shared_config
|
|
476
|
+
)
|
|
477
|
+
if reference is None:
|
|
478
|
+
raise ValueError(f"Table '{schema_name}.{table_name}' not found")
|
|
479
|
+
references.append(reference)
|
|
480
|
+
return references
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def get_schema_sources(
|
|
484
|
+
database_names: List[str],
|
|
485
|
+
config_dir: Optional[str] = None) -> Dict[str, Tuple[str, Optional[str]]]:
|
|
486
|
+
"""Map each shared schema to the database/version that uses it.
|
|
487
|
+
|
|
488
|
+
Args:
|
|
489
|
+
database_names: Served database names (from the main config).
|
|
490
|
+
config_dir: Base config directory. If None, uses default.
|
|
491
|
+
|
|
492
|
+
Returns:
|
|
493
|
+
Schema name -> (database name, version or None). Schemas used by more
|
|
494
|
+
than one database/version are left out (their source is ambiguous).
|
|
495
|
+
"""
|
|
496
|
+
sources: Dict[str, Tuple[str, Optional[str]]] = {}
|
|
497
|
+
ambiguous = set()
|
|
498
|
+
|
|
499
|
+
for database_name in database_names:
|
|
500
|
+
if is_versioned_database(database_name, config_dir):
|
|
501
|
+
versions: List[Optional[str]] = list(get_database_versions(database_name, config_dir))
|
|
502
|
+
else:
|
|
503
|
+
versions = [None]
|
|
504
|
+
for version in versions:
|
|
505
|
+
try:
|
|
506
|
+
config = load_database_config(database_name, config_dir, version)
|
|
507
|
+
except FileNotFoundError:
|
|
508
|
+
continue
|
|
509
|
+
schema_name = config.get(DATABASE_SCHEMA_KEY) if isinstance(config, dict) else None
|
|
510
|
+
if not schema_name:
|
|
511
|
+
continue
|
|
512
|
+
if schema_name in sources:
|
|
513
|
+
ambiguous.add(schema_name)
|
|
514
|
+
sources[schema_name] = (database_name, version)
|
|
515
|
+
|
|
516
|
+
return {k: v for k, v in sources.items() if k not in ambiguous}
|
|
517
|
+
|
|
518
|
+
|
|
421
519
|
def get_local_table_references(
|
|
422
520
|
database_config: Dict[str, Any],
|
|
423
521
|
shared_config: Optional[Dict[str, Any]] = None) -> List[TableReference]:
|
|
@@ -789,6 +887,37 @@ def _load_yaml_file(file_path: str) -> Dict[str, Any]:
|
|
|
789
887
|
raise yaml.YAMLError(f"Invalid YAML in {file_path}: {e}")
|
|
790
888
|
|
|
791
889
|
|
|
890
|
+
def _validate_schema_groups(shared_file: Dict[str, Any], shared_path: str) -> None:
|
|
891
|
+
"""Validate the shared ``schema_groups`` mapping.
|
|
892
|
+
|
|
893
|
+
Args:
|
|
894
|
+
shared_file: The normalized shared config.
|
|
895
|
+
shared_path: Path of the shared config (for error messages).
|
|
896
|
+
|
|
897
|
+
Raises:
|
|
898
|
+
ValueError: If a group name isn't a plain identifier or collides with a
|
|
899
|
+
schema name, or a group isn't a non-empty list of known schemas.
|
|
900
|
+
"""
|
|
901
|
+
schemas = (shared_file.get(SHARED_CONFIG_KEY) or {}).get(SHARED_SCHEMA_KEY) or {}
|
|
902
|
+
for group, members in (shared_file.get(SCHEMA_GROUPS_KEY) or {}).items():
|
|
903
|
+
if not IDENTIFIER_PATTERN.match(str(group)):
|
|
904
|
+
raise ValueError(f"{SCHEMA_GROUPS_KEY}: invalid group name '{group}' in {shared_path}")
|
|
905
|
+
if group in schemas:
|
|
906
|
+
raise ValueError(
|
|
907
|
+
f"{SCHEMA_GROUPS_KEY}: '{group}' is also a schema name in {shared_path}"
|
|
908
|
+
)
|
|
909
|
+
if not isinstance(members, list) or not members:
|
|
910
|
+
raise ValueError(
|
|
911
|
+
f"{SCHEMA_GROUPS_KEY}: '{group}' must be a non-empty list in {shared_path}"
|
|
912
|
+
)
|
|
913
|
+
for member in members:
|
|
914
|
+
if member not in schemas:
|
|
915
|
+
raise ValueError(
|
|
916
|
+
f"{SCHEMA_GROUPS_KEY}: '{group}' names unknown schema '{member}' "
|
|
917
|
+
f"in {shared_path}"
|
|
918
|
+
)
|
|
919
|
+
|
|
920
|
+
|
|
792
921
|
def _is_selected(
|
|
793
922
|
inclusions: Any,
|
|
794
923
|
exclusions: Any,
|
|
@@ -15,6 +15,11 @@ settings:
|
|
|
15
15
|
add_trailing_slash: false # Set true to auto-append trailing slash to proxied paths
|
|
16
16
|
follow_redirects: true # Set false to pass 3xx redirects through to the client
|
|
17
17
|
timeout: 10 # Upstream request timeout (seconds); null/false to disable
|
|
18
|
+
# base_path: /dock # Also serve the API under this prefix (e.g. behind a CDN path)
|
|
19
|
+
# duckdb: # Database query options (DuckDB SET options + a concurrency cap)
|
|
20
|
+
# memory_limit: 700MB
|
|
21
|
+
# threads: 2
|
|
22
|
+
# max_concurrent_queries: 2
|
|
18
23
|
|
|
19
24
|
# Global route restrictions — applied to all remotes unless overridden per-remote.
|
|
20
25
|
# Uncomment to block DELETE on every remote:
|
|
@@ -36,6 +36,11 @@
|
|
|
36
36
|
# items:
|
|
37
37
|
# uri: s3://your-bucket/v2/items/**/*.parquet
|
|
38
38
|
#
|
|
39
|
+
# # named groups of schemas: [[all_items.items]] reads both; [[all_items!.items]]
|
|
40
|
+
# # skips the current version's schema; [[*.items]] reads every schema
|
|
41
|
+
# schema_groups:
|
|
42
|
+
# all_items: [items_v1, items_v2]
|
|
43
|
+
#
|
|
39
44
|
# # database/version configs defined inline instead of as files (list them under
|
|
40
45
|
# # `databases:` in the main config like any other database)
|
|
41
46
|
# slugs:
|
|
@@ -68,6 +73,13 @@
|
|
|
68
73
|
# - route: items/
|
|
69
74
|
# include: ['example_db'] # ONLY add this route to these slug/versions
|
|
70
75
|
# sql: SELECT [[items]].id FROM [[items]]
|
|
76
|
+
# # every items row in the same group as this one, from any schema, tagged with its source
|
|
77
|
+
# - route: items/{{id}}/related
|
|
78
|
+
# source_columns: [schema, name, version]
|
|
79
|
+
# sql: >
|
|
80
|
+
# SELECT items.* FROM [[*.items]] items
|
|
81
|
+
# WHERE items.group_id = (SELECT group_id FROM [[items]] WHERE id = {{id}})
|
|
82
|
+
# AND NOT (items.schema_name = {{self.schema}} AND items.id = {{id}})
|
|
71
83
|
#
|
|
72
84
|
# query_params:
|
|
73
85
|
# - limit:
|
|
@@ -16,9 +16,14 @@ import warnings
|
|
|
16
16
|
import httpx
|
|
17
17
|
from fastapi import FastAPI, Request
|
|
18
18
|
from fastapi.responses import JSONResponse, Response, StreamingResponse
|
|
19
|
-
from typing import Any, Dict, Optional
|
|
20
|
-
|
|
21
|
-
from api_dock.route_mapper import
|
|
19
|
+
from typing import Any, Callable, Dict, Optional
|
|
20
|
+
|
|
21
|
+
from api_dock.route_mapper import (
|
|
22
|
+
collect_multi_query_params,
|
|
23
|
+
HOP_BY_HOP_HEADERS,
|
|
24
|
+
RouteMapper,
|
|
25
|
+
strip_base_path,
|
|
26
|
+
)
|
|
22
27
|
from api_dock.types import PreparedRequest, ProxyResponse
|
|
23
28
|
|
|
24
29
|
|
|
@@ -66,12 +71,36 @@ def create_app(config_path: Optional[str] = None) -> FastAPI:
|
|
|
66
71
|
_add_remote_routes(app, route_mapper)
|
|
67
72
|
_add_error_handlers(app)
|
|
68
73
|
|
|
74
|
+
if route_mapper.base_path:
|
|
75
|
+
app.add_middleware(_StripBasePath, base_path=route_mapper.base_path)
|
|
76
|
+
|
|
69
77
|
return app
|
|
70
78
|
|
|
71
79
|
|
|
72
80
|
#
|
|
73
81
|
# INTERNAL
|
|
74
82
|
#
|
|
83
|
+
class _StripBasePath:
|
|
84
|
+
"""ASGI middleware serving the app under ``settings.base_path`` as well.
|
|
85
|
+
|
|
86
|
+
Requests whose path starts with the base path (e.g. ``/dock/birdnet/...``)
|
|
87
|
+
are routed as if it weren't there; other paths pass through unchanged.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
def __init__(self, app: Callable, base_path: str) -> None:
|
|
91
|
+
self.app = app
|
|
92
|
+
self.base_path = base_path
|
|
93
|
+
|
|
94
|
+
async def __call__(self, scope: Dict[str, Any], receive: Callable, send: Callable) -> None:
|
|
95
|
+
"""Rewrite the request path, then call the wrapped app."""
|
|
96
|
+
if scope.get("type") in ("http", "websocket"):
|
|
97
|
+
path = strip_base_path(scope.get("path", ""), self.base_path)
|
|
98
|
+
if path != scope.get("path"):
|
|
99
|
+
scope = {**scope, "path": path, "raw_path": path.encode()}
|
|
100
|
+
await self.app(scope, receive, send)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
75
104
|
def _add_main_routes(app: FastAPI, route_mapper: RouteMapper) -> None:
|
|
76
105
|
"""Add main API routes to the FastAPI app.
|
|
77
106
|
|