api-dock 0.8.0__tar.gz → 0.8.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {api_dock-0.8.0 → api_dock-0.8.2}/PKG-INFO +106 -33
  2. {api_dock-0.8.0 → api_dock-0.8.2}/README.md +105 -32
  3. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/database_config.py +130 -1
  4. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/config.yaml +5 -0
  5. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/config.yaml +12 -0
  6. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/fast_api.py +32 -3
  7. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/flask_api.py +25 -2
  8. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/route_mapper.py +207 -32
  9. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/sql_builder.py +251 -24
  10. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/types.py +24 -1
  11. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/PKG-INFO +106 -33
  12. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/SOURCES.txt +2 -0
  13. {api_dock-0.8.0 → api_dock-0.8.2}/pyproject.toml +1 -1
  14. api_dock-0.8.2/tests/test_runtime_settings.py +249 -0
  15. api_dock-0.8.2/tests/test_schema_unions.py +308 -0
  16. {api_dock-0.8.0 → api_dock-0.8.2}/LICENSE.md +0 -0
  17. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/__init__.py +0 -0
  18. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/auth.py +0 -0
  19. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/cli.py +0 -0
  20. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/config.py +0 -0
  21. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/config_discovery.py +0 -0
  22. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/encryption.py +0 -0
  23. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/example_db.yaml +0 -0
  24. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/example_api_dock_config/remotes/example_remote.yaml +0 -0
  25. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/listings.py +0 -0
  26. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock/storage_auth.py +0 -0
  27. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/dependency_links.txt +0 -0
  28. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/entry_points.txt +0 -0
  29. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/requires.txt +0 -0
  30. {api_dock-0.8.0 → api_dock-0.8.2}/api_dock.egg-info/top_level.txt +0 -0
  31. {api_dock-0.8.0 → api_dock-0.8.2}/setup.cfg +0 -0
  32. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_inject_cookies.py +0 -0
  33. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_listings.py +0 -0
  34. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_proxy_pipeline.py +0 -0
  35. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_shared_database_config.py +0 -0
  36. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_sql_builder.py +0 -0
  37. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_sql_selector.py +0 -0
  38. {api_dock-0.8.0 → api_dock-0.8.2}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: api_dock
3
- Version: 0.8.0
3
+ Version: 0.8.2
4
4
  Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
5
5
  Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
6
6
  License-Expression: BSD-3-Clause
@@ -219,19 +219,33 @@ remotes:
219
219
  settings:
220
220
  add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
221
221
  follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
222
+ follow_redirects: true # Follow remote redirects (default: true)
222
223
  timeout: 10 # Upstream request timeout in seconds (default: 10)
224
+ base_path: /dock # Also serve the API under this prefix (default: none)
225
+ duckdb: # Options for database queries (default: none)
226
+ memory_limit: 700MB
227
+ threads: 2
228
+ max_concurrent_queries: 2
223
229
  ```
224
230
 
225
- ### HTTP behavior Settings
231
+ ### Settings
226
232
 
227
- The optional `settings` section controls HTTP behavior:
233
+ The optional `settings` section controls HTTP and query behavior:
228
234
 
229
235
  - **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
230
236
 
231
237
  - **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
232
238
 
239
+ - **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
240
+
233
241
  - **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
234
242
 
243
+ - **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
244
+
245
+ - **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
246
+
247
+ Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
248
+
235
249
  ### Catalog Endpoints (`expose`)
236
250
 
237
251
  The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
@@ -563,6 +577,60 @@ Notes:
563
577
  - Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
564
578
  - Views are created only for the `[[schema.table]]` tables a query actually references.
565
579
 
580
+ #### Querying across schemas (`[[*.table]]`, schema groups)
581
+
582
+ Union references read the same table from several schemas at once:
583
+
584
+ | Reference | Reads |
585
+ |---|---|
586
+ | `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
587
+ | `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
588
+ | `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
589
+ | `[[group1!.detections]]` | that group, minus the current schema |
590
+
591
+ ```yaml
592
+ # api_dock_config/databases/config.yaml
593
+ schema_groups: # named lists of shared schemas
594
+ birdnet_models:
595
+ - birdnet_2p4
596
+ - birdnet_bullfrog_2p4v0p5
597
+ ```
598
+
599
+ - A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
600
+ - `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
601
+ - Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
602
+
603
+ **Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
604
+
605
+ ```yaml
606
+ source_columns: [schema, name, version] # adds schema_name, name, version
607
+ source_columns: {schema: _schema, name: model} # pick a subset and rename
608
+ ```
609
+
610
+ A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
611
+
612
+ **`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
613
+
614
+ Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
615
+
616
+ ```yaml
617
+ routes:
618
+ - route: detections/{{id}}/overlaps
619
+ source_columns: [schema, name, version]
620
+ sql: |
621
+ WITH src AS (
622
+ SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
623
+ )
624
+ SELECT detections.*
625
+ FROM [[*.detections]] detections
626
+ JOIN src ON detections.recording_id = src.recording_id
627
+ AND detections.start_time < src.end_time
628
+ AND detections.end_time > src.start_time
629
+ WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
630
+ ```
631
+
632
+ Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
633
+
566
634
  #### Inline database configs (`slugs`)
567
635
 
568
636
  Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
@@ -1543,47 +1611,52 @@ pixi run python scripts/hello_world.py
1543
1611
 
1544
1612
  ## Publishing a Release
1545
1613
 
1614
+ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
1615
+
1546
1616
  ```bash
1547
- # 0. Make sure you are on `main` and merged with any changes
1617
+ # 0. Start from a clean, up-to-date main
1618
+ export VERSION=0.8.2 # the NEW version, no leading "v"
1619
+ git checkout main
1620
+ git pull origin main
1621
+ git status
1548
1622
 
1549
- # 1. Bump version in pyproject.toml
1623
+ # 1. Set `version` in pyproject.toml to $VERSION
1550
1624
 
1551
- # 2. Commit everything
1625
+ # 2. Run the tests
1626
+ pixi run -e dev pytest -q
1627
+
1628
+ # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1629
+ export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
1552
1630
  git add -A
1553
- git commit -m "v0.6.1: stream proxy responses (fix large-response 502 + content-encoding)"
1554
-
1555
- # 3. Tag and push
1556
- git tag v0.6.1
1557
- git push origin main v0.6.1
1558
-
1559
- # 4. Build the wheel (requires the `dev` pixi environment)
1560
- rm -rf dist/
1561
- find . -name "__pycache__" -type d -exec rm -rf {} +
1562
- find . -name "*.pyc" -delete
1563
- pixi run -e dev python -m build --wheel
1564
- ls dist/*.whl
1565
-
1566
- # 5. Create GitHub release with the wheel attached
1567
- gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
1568
- --title "v0.6.1" --notes "$(cat <<'EOF'
1631
+ git commit -m "v$VERSION: $COMMIT_MESSAGE"
1632
+ git tag "v$VERSION"
1633
+ git push origin main "v$VERSION"
1634
+
1635
+ # 4. Publish the GitHub Release; this triggers the PyPI upload.
1636
+ # Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
1637
+ gh release create "v$VERSION" \
1638
+ --title "v$VERSION" \
1639
+ --notes "$(cat <<'EOF'
1569
1640
  * new features
1570
- - Remote proxy responses are now streamed (FastAPI) — upstream bytes are piped to the client as they arrive instead of being buffered fully in memory
1571
- - New `timeout` setting (default 10s) for the upstream request; set to `null`/`false` to disable
1641
+ - `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
1642
+ - `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
1572
1643
  * bug fixes
1573
- - Large upstream responses no longer return 502 — streamed via `StreamingResponse` instead of reading the whole body into memory
1574
- - `Content-Encoding` (gzip/br/deflate) is now preserved on compressed responses — raw bytes are streamed via `aiter_raw()` so the header stays valid and the client can decompress
1575
- - Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
1644
+ - Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
1645
+ - Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
1576
1646
  * cleanup / other improvements
1577
- - Added `PreparedRequest` dataclass and split route validation/resolution into `RouteMapper.prepare_remote_request()`; the FastAPI adapter issues the streaming HTTP call
1578
- - `map_route()` (buffered) retained for the Flask/sync path
1579
- - Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
1647
+ - README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
1648
+ - Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
1580
1649
  EOF
1581
1650
  )"
1582
1651
 
1583
- # 6. Publish to PyPI
1584
- pixi run -e dev python -m twine upload dist/*.whl
1585
- ```
1652
+ # 5. Watch the publish workflow, then confirm PyPI has the new version
1653
+ gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
1654
+ curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
1586
1655
 
1656
+ # 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
1657
+ # run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
1658
+ gh pr list --repo conda-forge/api_dock-feedstock --state open
1659
+ ```
1587
1660
 
1588
1661
  ---
1589
1662
 
@@ -180,19 +180,33 @@ remotes:
180
180
  settings:
181
181
  add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
182
182
  follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
183
+ follow_redirects: true # Follow remote redirects (default: true)
183
184
  timeout: 10 # Upstream request timeout in seconds (default: 10)
185
+ base_path: /dock # Also serve the API under this prefix (default: none)
186
+ duckdb: # Options for database queries (default: none)
187
+ memory_limit: 700MB
188
+ threads: 2
189
+ max_concurrent_queries: 2
184
190
  ```
185
191
 
186
- ### HTTP behavior Settings
192
+ ### Settings
187
193
 
188
- The optional `settings` section controls HTTP behavior:
194
+ The optional `settings` section controls HTTP and query behavior:
189
195
 
190
196
  - **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
191
197
 
192
198
  - **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
193
199
 
200
+ - **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
201
+
194
202
  - **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
195
203
 
204
+ - **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
205
+
206
+ - **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
207
+
208
+ Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
209
+
196
210
  ### Catalog Endpoints (`expose`)
197
211
 
198
212
  The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
@@ -524,6 +538,60 @@ Notes:
524
538
  - Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
525
539
  - Views are created only for the `[[schema.table]]` tables a query actually references.
526
540
 
541
+ #### Querying across schemas (`[[*.table]]`, schema groups)
542
+
543
+ Union references read the same table from several schemas at once:
544
+
545
+ | Reference | Reads |
546
+ |---|---|
547
+ | `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
548
+ | `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
549
+ | `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
550
+ | `[[group1!.detections]]` | that group, minus the current schema |
551
+
552
+ ```yaml
553
+ # api_dock_config/databases/config.yaml
554
+ schema_groups: # named lists of shared schemas
555
+ birdnet_models:
556
+ - birdnet_2p4
557
+ - birdnet_bullfrog_2p4v0p5
558
+ ```
559
+
560
+ - A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
561
+ - `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
562
+ - Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
563
+
564
+ **Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
565
+
566
+ ```yaml
567
+ source_columns: [schema, name, version] # adds schema_name, name, version
568
+ source_columns: {schema: _schema, name: model} # pick a subset and rename
569
+ ```
570
+
571
+ A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
572
+
573
+ **`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
574
+
575
+ Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
576
+
577
+ ```yaml
578
+ routes:
579
+ - route: detections/{{id}}/overlaps
580
+ source_columns: [schema, name, version]
581
+ sql: |
582
+ WITH src AS (
583
+ SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
584
+ )
585
+ SELECT detections.*
586
+ FROM [[*.detections]] detections
587
+ JOIN src ON detections.recording_id = src.recording_id
588
+ AND detections.start_time < src.end_time
589
+ AND detections.end_time > src.start_time
590
+ WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
591
+ ```
592
+
593
+ Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
594
+
527
595
  #### Inline database configs (`slugs`)
528
596
 
529
597
  Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
@@ -1504,47 +1572,52 @@ pixi run python scripts/hello_world.py
1504
1572
 
1505
1573
  ## Publishing a Release
1506
1574
 
1575
+ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
1576
+
1507
1577
  ```bash
1508
- # 0. Make sure you are on `main` and merged with any changes
1578
+ # 0. Start from a clean, up-to-date main
1579
+ export VERSION=0.8.2 # the NEW version, no leading "v"
1580
+ git checkout main
1581
+ git pull origin main
1582
+ git status
1509
1583
 
1510
- # 1. Bump version in pyproject.toml
1584
+ # 1. Set `version` in pyproject.toml to $VERSION
1511
1585
 
1512
- # 2. Commit everything
1586
+ # 2. Run the tests
1587
+ pixi run -e dev pytest -q
1588
+
1589
+ # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1590
+ export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
1513
1591
  git add -A
1514
- git commit -m "v0.6.1: stream proxy responses (fix large-response 502 + content-encoding)"
1515
-
1516
- # 3. Tag and push
1517
- git tag v0.6.1
1518
- git push origin main v0.6.1
1519
-
1520
- # 4. Build the wheel (requires the `dev` pixi environment)
1521
- rm -rf dist/
1522
- find . -name "__pycache__" -type d -exec rm -rf {} +
1523
- find . -name "*.pyc" -delete
1524
- pixi run -e dev python -m build --wheel
1525
- ls dist/*.whl
1526
-
1527
- # 5. Create GitHub release with the wheel attached
1528
- gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
1529
- --title "v0.6.1" --notes "$(cat <<'EOF'
1592
+ git commit -m "v$VERSION: $COMMIT_MESSAGE"
1593
+ git tag "v$VERSION"
1594
+ git push origin main "v$VERSION"
1595
+
1596
+ # 4. Publish the GitHub Release; this triggers the PyPI upload.
1597
+ # Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
1598
+ gh release create "v$VERSION" \
1599
+ --title "v$VERSION" \
1600
+ --notes "$(cat <<'EOF'
1530
1601
  * new features
1531
- - Remote proxy responses are now streamed (FastAPI) — upstream bytes are piped to the client as they arrive instead of being buffered fully in memory
1532
- - New `timeout` setting (default 10s) for the upstream request; set to `null`/`false` to disable
1602
+ - `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
1603
+ - `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
1533
1604
  * bug fixes
1534
- - Large upstream responses no longer return 502 — streamed via `StreamingResponse` instead of reading the whole body into memory
1535
- - `Content-Encoding` (gzip/br/deflate) is now preserved on compressed responses — raw bytes are streamed via `aiter_raw()` so the header stays valid and the client can decompress
1536
- - Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
1605
+ - Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
1606
+ - Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
1537
1607
  * cleanup / other improvements
1538
- - Added `PreparedRequest` dataclass and split route validation/resolution into `RouteMapper.prepare_remote_request()`; the FastAPI adapter issues the streaming HTTP call
1539
- - `map_route()` (buffered) retained for the Flask/sync path
1540
- - Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
1608
+ - README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
1609
+ - Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
1541
1610
  EOF
1542
1611
  )"
1543
1612
 
1544
- # 6. Publish to PyPI
1545
- pixi run -e dev python -m twine upload dist/*.whl
1546
- ```
1613
+ # 5. Watch the publish workflow, then confirm PyPI has the new version
1614
+ gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
1615
+ curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
1547
1616
 
1617
+ # 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
1618
+ # run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
1619
+ gh pr list --repo conda-forge/api_dock-feedstock --state open
1620
+ ```
1548
1621
 
1549
1622
  ---
1550
1623
 
@@ -56,6 +56,15 @@ SLUG_NAME_KEY: str = "name"
56
56
  SLUG_VERSION_KEY: str = "version"
57
57
  SLUG_VERSIONS_KEY: str = "versions"
58
58
 
59
+ # Named groups of shared schemas: {group: [schema, ...]}, referenced in SQL as
60
+ # [[group.table]]. Group names may not collide with schema names.
61
+ SCHEMA_GROUPS_KEY: str = "schema_groups"
62
+
63
+ # Union selectors: [[*.table]] = every shared schema with that table; a trailing
64
+ # "!" ([[*!.table]], [[group!.table]]) drops the current version's schema.
65
+ ALL_SCHEMAS: str = "*"
66
+ EXCLUDE_SELF_SUFFIX: str = "!"
67
+
59
68
  # Exclusion version wildcard: matches every version (and unversioned databases).
60
69
  ALL_VERSIONS: str = "*"
61
70
 
@@ -212,7 +221,7 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
212
221
  """Load the whole shared database config file (``databases/config.yaml``).
213
222
 
214
223
  Top-level keys: ``database`` (tables, ``meta``, ``schema``), ``slugs``
215
- (inline database/version configs), ``routes`` and
224
+ (inline database/version configs), ``schema_groups``, ``routes`` and
216
225
  ``query_params`` (added to every database/version), and the
217
226
  ``route_inclusions`` / ``query_inclusions`` and ``route_exclusions`` /
218
227
  ``query_exclusions`` lists.
@@ -250,12 +259,15 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
250
259
  ROUTE_EXCLUSIONS_KEY: list,
251
260
  QUERY_EXCLUSIONS_KEY: list,
252
261
  SLUGS_KEY: list,
262
+ SCHEMA_GROUPS_KEY: dict,
253
263
  }
254
264
  for key, expected in expected_types.items():
255
265
  value = contents.get(key) or expected()
256
266
  if not isinstance(value, expected):
257
267
  raise ValueError(f"'{key}' in {shared_path} must be a {expected.__name__}")
258
268
  normalized[key] = value
269
+
270
+ _validate_schema_groups(normalized, shared_path)
259
271
  return normalized
260
272
 
261
273
 
@@ -418,6 +430,92 @@ def resolve_table_reference(
418
430
  return None
419
431
 
420
432
 
433
+ def resolve_schema_union(
434
+ selector: str,
435
+ table_name: str,
436
+ shared_config: Optional[Dict[str, Any]] = None,
437
+ schema_groups: Optional[Dict[str, List[str]]] = None) -> Optional[List[TableReference]]:
438
+ """Resolve a union selector (``*`` or a schema group) to its member tables.
439
+
440
+ Args:
441
+ selector: ``*`` for every shared schema, or a ``schema_groups`` name
442
+ (without any trailing ``!``).
443
+ table_name: Table to read from each member schema.
444
+ shared_config: The shared ``database`` mapping, or None.
445
+ schema_groups: The shared ``schema_groups`` mapping, or None.
446
+
447
+ Returns:
448
+ Qualified TableReferences in schema/group order, or None if the
449
+ selector is neither ``*`` nor a group (i.e. it names a single schema).
450
+
451
+ Raises:
452
+ ValueError: If no schema has the table (``*``), or a group member is
453
+ not a schema or lacks the table.
454
+ """
455
+ schemas = (shared_config or {}).get(SHARED_SCHEMA_KEY) or {}
456
+ groups = schema_groups or {}
457
+
458
+ if selector == ALL_SCHEMAS:
459
+ members = [name for name, tables in schemas.items() if table_name in (tables or {})]
460
+ if not members:
461
+ raise ValueError(f"No shared schema has a table named '{table_name}'")
462
+ elif selector in groups:
463
+ members = list(groups[selector])
464
+ for schema_name in members:
465
+ if table_name not in (schemas.get(schema_name) or {}):
466
+ raise ValueError(
467
+ f"Schema '{schema_name}' in group '{selector}' has no table '{table_name}'"
468
+ )
469
+ else:
470
+ return None
471
+
472
+ references = []
473
+ for schema_name in members:
474
+ reference = resolve_table_reference(
475
+ f"{schema_name}{SCHEMA_SEPARATOR}{table_name}", {}, shared_config
476
+ )
477
+ if reference is None:
478
+ raise ValueError(f"Table '{schema_name}.{table_name}' not found")
479
+ references.append(reference)
480
+ return references
481
+
482
+
483
+ def get_schema_sources(
484
+ database_names: List[str],
485
+ config_dir: Optional[str] = None) -> Dict[str, Tuple[str, Optional[str]]]:
486
+ """Map each shared schema to the database/version that uses it.
487
+
488
+ Args:
489
+ database_names: Served database names (from the main config).
490
+ config_dir: Base config directory. If None, uses default.
491
+
492
+ Returns:
493
+ Schema name -> (database name, version or None). Schemas used by more
494
+ than one database/version are left out (their source is ambiguous).
495
+ """
496
+ sources: Dict[str, Tuple[str, Optional[str]]] = {}
497
+ ambiguous = set()
498
+
499
+ for database_name in database_names:
500
+ if is_versioned_database(database_name, config_dir):
501
+ versions: List[Optional[str]] = list(get_database_versions(database_name, config_dir))
502
+ else:
503
+ versions = [None]
504
+ for version in versions:
505
+ try:
506
+ config = load_database_config(database_name, config_dir, version)
507
+ except FileNotFoundError:
508
+ continue
509
+ schema_name = config.get(DATABASE_SCHEMA_KEY) if isinstance(config, dict) else None
510
+ if not schema_name:
511
+ continue
512
+ if schema_name in sources:
513
+ ambiguous.add(schema_name)
514
+ sources[schema_name] = (database_name, version)
515
+
516
+ return {k: v for k, v in sources.items() if k not in ambiguous}
517
+
518
+
421
519
  def get_local_table_references(
422
520
  database_config: Dict[str, Any],
423
521
  shared_config: Optional[Dict[str, Any]] = None) -> List[TableReference]:
@@ -789,6 +887,37 @@ def _load_yaml_file(file_path: str) -> Dict[str, Any]:
789
887
  raise yaml.YAMLError(f"Invalid YAML in {file_path}: {e}")
790
888
 
791
889
 
890
+ def _validate_schema_groups(shared_file: Dict[str, Any], shared_path: str) -> None:
891
+ """Validate the shared ``schema_groups`` mapping.
892
+
893
+ Args:
894
+ shared_file: The normalized shared config.
895
+ shared_path: Path of the shared config (for error messages).
896
+
897
+ Raises:
898
+ ValueError: If a group name isn't a plain identifier or collides with a
899
+ schema name, or a group isn't a non-empty list of known schemas.
900
+ """
901
+ schemas = (shared_file.get(SHARED_CONFIG_KEY) or {}).get(SHARED_SCHEMA_KEY) or {}
902
+ for group, members in (shared_file.get(SCHEMA_GROUPS_KEY) or {}).items():
903
+ if not IDENTIFIER_PATTERN.match(str(group)):
904
+ raise ValueError(f"{SCHEMA_GROUPS_KEY}: invalid group name '{group}' in {shared_path}")
905
+ if group in schemas:
906
+ raise ValueError(
907
+ f"{SCHEMA_GROUPS_KEY}: '{group}' is also a schema name in {shared_path}"
908
+ )
909
+ if not isinstance(members, list) or not members:
910
+ raise ValueError(
911
+ f"{SCHEMA_GROUPS_KEY}: '{group}' must be a non-empty list in {shared_path}"
912
+ )
913
+ for member in members:
914
+ if member not in schemas:
915
+ raise ValueError(
916
+ f"{SCHEMA_GROUPS_KEY}: '{group}' names unknown schema '{member}' "
917
+ f"in {shared_path}"
918
+ )
919
+
920
+
792
921
  def _is_selected(
793
922
  inclusions: Any,
794
923
  exclusions: Any,
@@ -15,6 +15,11 @@ settings:
15
15
  add_trailing_slash: false # Set true to auto-append trailing slash to proxied paths
16
16
  follow_redirects: true # Set false to pass 3xx redirects through to the client
17
17
  timeout: 10 # Upstream request timeout (seconds); null/false to disable
18
+ # base_path: /dock # Also serve the API under this prefix (e.g. behind a CDN path)
19
+ # duckdb: # Database query options (DuckDB SET options + a concurrency cap)
20
+ # memory_limit: 700MB
21
+ # threads: 2
22
+ # max_concurrent_queries: 2
18
23
 
19
24
  # Global route restrictions — applied to all remotes unless overridden per-remote.
20
25
  # Uncomment to block DELETE on every remote:
@@ -36,6 +36,11 @@
36
36
  # items:
37
37
  # uri: s3://your-bucket/v2/items/**/*.parquet
38
38
  #
39
+ # # named groups of schemas: [[all_items.items]] reads both; [[all_items!.items]]
40
+ # # skips the current version's schema; [[*.items]] reads every schema
41
+ # schema_groups:
42
+ # all_items: [items_v1, items_v2]
43
+ #
39
44
  # # database/version configs defined inline instead of as files (list them under
40
45
  # # `databases:` in the main config like any other database)
41
46
  # slugs:
@@ -68,6 +73,13 @@
68
73
  # - route: items/
69
74
  # include: ['example_db'] # ONLY add this route to these slug/versions
70
75
  # sql: SELECT [[items]].id FROM [[items]]
76
+ # # every items row in the same group as this one, from any schema, tagged with its source
77
+ # - route: items/{{id}}/related
78
+ # source_columns: [schema, name, version]
79
+ # sql: >
80
+ # SELECT items.* FROM [[*.items]] items
81
+ # WHERE items.group_id = (SELECT group_id FROM [[items]] WHERE id = {{id}})
82
+ # AND NOT (items.schema_name = {{self.schema}} AND items.id = {{id}})
71
83
  #
72
84
  # query_params:
73
85
  # - limit:
@@ -16,9 +16,14 @@ import warnings
16
16
  import httpx
17
17
  from fastapi import FastAPI, Request
18
18
  from fastapi.responses import JSONResponse, Response, StreamingResponse
19
- from typing import Any, Dict, Optional
20
-
21
- from api_dock.route_mapper import collect_multi_query_params, HOP_BY_HOP_HEADERS, RouteMapper
19
+ from typing import Any, Callable, Dict, Optional
20
+
21
+ from api_dock.route_mapper import (
22
+ collect_multi_query_params,
23
+ HOP_BY_HOP_HEADERS,
24
+ RouteMapper,
25
+ strip_base_path,
26
+ )
22
27
  from api_dock.types import PreparedRequest, ProxyResponse
23
28
 
24
29
 
@@ -66,12 +71,36 @@ def create_app(config_path: Optional[str] = None) -> FastAPI:
66
71
  _add_remote_routes(app, route_mapper)
67
72
  _add_error_handlers(app)
68
73
 
74
+ if route_mapper.base_path:
75
+ app.add_middleware(_StripBasePath, base_path=route_mapper.base_path)
76
+
69
77
  return app
70
78
 
71
79
 
72
80
  #
73
81
  # INTERNAL
74
82
  #
83
+ class _StripBasePath:
84
+ """ASGI middleware serving the app under ``settings.base_path`` as well.
85
+
86
+ Requests whose path starts with the base path (e.g. ``/dock/birdnet/...``)
87
+ are routed as if it weren't there; other paths pass through unchanged.
88
+ """
89
+
90
+ def __init__(self, app: Callable, base_path: str) -> None:
91
+ self.app = app
92
+ self.base_path = base_path
93
+
94
+ async def __call__(self, scope: Dict[str, Any], receive: Callable, send: Callable) -> None:
95
+ """Rewrite the request path, then call the wrapped app."""
96
+ if scope.get("type") in ("http", "websocket"):
97
+ path = strip_base_path(scope.get("path", ""), self.base_path)
98
+ if path != scope.get("path"):
99
+ scope = {**scope, "path": path, "raw_path": path.encode()}
100
+ await self.app(scope, receive, send)
101
+
102
+
103
+
75
104
  def _add_main_routes(app: FastAPI, route_mapper: RouteMapper) -> None:
76
105
  """Add main API routes to the FastAPI app.
77
106