api-dock 0.8.1__tar.gz → 0.8.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {api_dock-0.8.1 → api_dock-0.8.2}/PKG-INFO +25 -14
- {api_dock-0.8.1 → api_dock-0.8.2}/README.md +24 -13
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/config.yaml +5 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/fast_api.py +32 -3
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/flask_api.py +25 -2
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/route_mapper.py +191 -27
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/PKG-INFO +25 -14
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/SOURCES.txt +1 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/pyproject.toml +1 -1
- api_dock-0.8.2/tests/test_runtime_settings.py +249 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/LICENSE.md +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/__init__.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/auth.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/cli.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/config.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/config_discovery.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/database_config.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/encryption.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/config.yaml +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/example_db.yaml +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/remotes/example_remote.yaml +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/listings.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/sql_builder.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/storage_auth.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/types.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/dependency_links.txt +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/entry_points.txt +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/requires.txt +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/top_level.txt +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/setup.cfg +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_inject_cookies.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_listings.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_proxy_pipeline.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_schema_unions.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_shared_database_config.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_sql_builder.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_sql_selector.py +0 -0
- {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: api_dock
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.2
|
|
4
4
|
Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
|
|
5
5
|
Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
|
|
6
6
|
License-Expression: BSD-3-Clause
|
|
@@ -219,19 +219,33 @@ remotes:
|
|
|
219
219
|
settings:
|
|
220
220
|
add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
|
|
221
221
|
follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
|
|
222
|
+
follow_redirects: true # Follow remote redirects (default: true)
|
|
222
223
|
timeout: 10 # Upstream request timeout in seconds (default: 10)
|
|
224
|
+
base_path: /dock # Also serve the API under this prefix (default: none)
|
|
225
|
+
duckdb: # Options for database queries (default: none)
|
|
226
|
+
memory_limit: 700MB
|
|
227
|
+
threads: 2
|
|
228
|
+
max_concurrent_queries: 2
|
|
223
229
|
```
|
|
224
230
|
|
|
225
|
-
###
|
|
231
|
+
### Settings
|
|
226
232
|
|
|
227
|
-
The optional `settings` section controls HTTP behavior:
|
|
233
|
+
The optional `settings` section controls HTTP and query behavior:
|
|
228
234
|
|
|
229
235
|
- **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
|
|
230
236
|
|
|
231
237
|
- **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
|
|
232
238
|
|
|
239
|
+
- **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
|
|
240
|
+
|
|
233
241
|
- **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
|
|
234
242
|
|
|
243
|
+
- **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
|
|
244
|
+
|
|
245
|
+
- **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
|
|
246
|
+
|
|
247
|
+
Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
|
|
248
|
+
|
|
235
249
|
### Catalog Endpoints (`expose`)
|
|
236
250
|
|
|
237
251
|
The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
|
|
@@ -1601,7 +1615,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
|
|
|
1601
1615
|
|
|
1602
1616
|
```bash
|
|
1603
1617
|
# 0. Start from a clean, up-to-date main
|
|
1604
|
-
export VERSION=0.8.
|
|
1618
|
+
export VERSION=0.8.2 # the NEW version, no leading "v"
|
|
1605
1619
|
git checkout main
|
|
1606
1620
|
git pull origin main
|
|
1607
1621
|
git status
|
|
@@ -1612,7 +1626,7 @@ git status
|
|
|
1612
1626
|
pixi run -e dev pytest -q
|
|
1613
1627
|
|
|
1614
1628
|
# 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
|
|
1615
|
-
export COMMIT_MESSAGE='
|
|
1629
|
+
export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
|
|
1616
1630
|
git add -A
|
|
1617
1631
|
git commit -m "v$VERSION: $COMMIT_MESSAGE"
|
|
1618
1632
|
git tag "v$VERSION"
|
|
@@ -1624,17 +1638,14 @@ gh release create "v$VERSION" \
|
|
|
1624
1638
|
--title "v$VERSION" \
|
|
1625
1639
|
--notes "$(cat <<'EOF'
|
|
1626
1640
|
* new features
|
|
1627
|
-
-
|
|
1628
|
-
-
|
|
1629
|
-
- Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
|
|
1630
|
-
- `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
|
|
1631
|
-
- Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
|
|
1641
|
+
- `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
|
|
1642
|
+
- `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
|
|
1632
1643
|
* bug fixes
|
|
1633
|
-
-
|
|
1644
|
+
- Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
|
|
1645
|
+
- Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
|
|
1634
1646
|
* cleanup / other improvements
|
|
1635
|
-
-
|
|
1636
|
-
-
|
|
1637
|
-
- Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
|
|
1647
|
+
- README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
|
|
1648
|
+
- Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
|
|
1638
1649
|
EOF
|
|
1639
1650
|
)"
|
|
1640
1651
|
|
|
@@ -180,19 +180,33 @@ remotes:
|
|
|
180
180
|
settings:
|
|
181
181
|
add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
|
|
182
182
|
follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
|
|
183
|
+
follow_redirects: true # Follow remote redirects (default: true)
|
|
183
184
|
timeout: 10 # Upstream request timeout in seconds (default: 10)
|
|
185
|
+
base_path: /dock # Also serve the API under this prefix (default: none)
|
|
186
|
+
duckdb: # Options for database queries (default: none)
|
|
187
|
+
memory_limit: 700MB
|
|
188
|
+
threads: 2
|
|
189
|
+
max_concurrent_queries: 2
|
|
184
190
|
```
|
|
185
191
|
|
|
186
|
-
###
|
|
192
|
+
### Settings
|
|
187
193
|
|
|
188
|
-
The optional `settings` section controls HTTP behavior:
|
|
194
|
+
The optional `settings` section controls HTTP and query behavior:
|
|
189
195
|
|
|
190
196
|
- **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
|
|
191
197
|
|
|
192
198
|
- **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
|
|
193
199
|
|
|
200
|
+
- **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
|
|
201
|
+
|
|
194
202
|
- **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
|
|
195
203
|
|
|
204
|
+
- **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
|
|
205
|
+
|
|
206
|
+
- **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
|
|
207
|
+
|
|
208
|
+
Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
|
|
209
|
+
|
|
196
210
|
### Catalog Endpoints (`expose`)
|
|
197
211
|
|
|
198
212
|
The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
|
|
@@ -1562,7 +1576,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
|
|
|
1562
1576
|
|
|
1563
1577
|
```bash
|
|
1564
1578
|
# 0. Start from a clean, up-to-date main
|
|
1565
|
-
export VERSION=0.8.
|
|
1579
|
+
export VERSION=0.8.2 # the NEW version, no leading "v"
|
|
1566
1580
|
git checkout main
|
|
1567
1581
|
git pull origin main
|
|
1568
1582
|
git status
|
|
@@ -1573,7 +1587,7 @@ git status
|
|
|
1573
1587
|
pixi run -e dev pytest -q
|
|
1574
1588
|
|
|
1575
1589
|
# 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
|
|
1576
|
-
export COMMIT_MESSAGE='
|
|
1590
|
+
export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
|
|
1577
1591
|
git add -A
|
|
1578
1592
|
git commit -m "v$VERSION: $COMMIT_MESSAGE"
|
|
1579
1593
|
git tag "v$VERSION"
|
|
@@ -1585,17 +1599,14 @@ gh release create "v$VERSION" \
|
|
|
1585
1599
|
--title "v$VERSION" \
|
|
1586
1600
|
--notes "$(cat <<'EOF'
|
|
1587
1601
|
* new features
|
|
1588
|
-
-
|
|
1589
|
-
-
|
|
1590
|
-
- Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
|
|
1591
|
-
- `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
|
|
1592
|
-
- Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
|
|
1602
|
+
- `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
|
|
1603
|
+
- `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
|
|
1593
1604
|
* bug fixes
|
|
1594
|
-
-
|
|
1605
|
+
- Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
|
|
1606
|
+
- Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
|
|
1595
1607
|
* cleanup / other improvements
|
|
1596
|
-
-
|
|
1597
|
-
-
|
|
1598
|
-
- Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
|
|
1608
|
+
- README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
|
|
1609
|
+
- Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
|
|
1599
1610
|
EOF
|
|
1600
1611
|
)"
|
|
1601
1612
|
|
|
@@ -15,6 +15,11 @@ settings:
|
|
|
15
15
|
add_trailing_slash: false # Set true to auto-append trailing slash to proxied paths
|
|
16
16
|
follow_redirects: true # Set false to pass 3xx redirects through to the client
|
|
17
17
|
timeout: 10 # Upstream request timeout (seconds); null/false to disable
|
|
18
|
+
# base_path: /dock # Also serve the API under this prefix (e.g. behind a CDN path)
|
|
19
|
+
# duckdb: # Database query options (DuckDB SET options + a concurrency cap)
|
|
20
|
+
# memory_limit: 700MB
|
|
21
|
+
# threads: 2
|
|
22
|
+
# max_concurrent_queries: 2
|
|
18
23
|
|
|
19
24
|
# Global route restrictions — applied to all remotes unless overridden per-remote.
|
|
20
25
|
# Uncomment to block DELETE on every remote:
|
|
@@ -16,9 +16,14 @@ import warnings
|
|
|
16
16
|
import httpx
|
|
17
17
|
from fastapi import FastAPI, Request
|
|
18
18
|
from fastapi.responses import JSONResponse, Response, StreamingResponse
|
|
19
|
-
from typing import Any, Dict, Optional
|
|
20
|
-
|
|
21
|
-
from api_dock.route_mapper import
|
|
19
|
+
from typing import Any, Callable, Dict, Optional
|
|
20
|
+
|
|
21
|
+
from api_dock.route_mapper import (
|
|
22
|
+
collect_multi_query_params,
|
|
23
|
+
HOP_BY_HOP_HEADERS,
|
|
24
|
+
RouteMapper,
|
|
25
|
+
strip_base_path,
|
|
26
|
+
)
|
|
22
27
|
from api_dock.types import PreparedRequest, ProxyResponse
|
|
23
28
|
|
|
24
29
|
|
|
@@ -66,12 +71,36 @@ def create_app(config_path: Optional[str] = None) -> FastAPI:
|
|
|
66
71
|
_add_remote_routes(app, route_mapper)
|
|
67
72
|
_add_error_handlers(app)
|
|
68
73
|
|
|
74
|
+
if route_mapper.base_path:
|
|
75
|
+
app.add_middleware(_StripBasePath, base_path=route_mapper.base_path)
|
|
76
|
+
|
|
69
77
|
return app
|
|
70
78
|
|
|
71
79
|
|
|
72
80
|
#
|
|
73
81
|
# INTERNAL
|
|
74
82
|
#
|
|
83
|
+
class _StripBasePath:
|
|
84
|
+
"""ASGI middleware serving the app under ``settings.base_path`` as well.
|
|
85
|
+
|
|
86
|
+
Requests whose path starts with the base path (e.g. ``/dock/birdnet/...``)
|
|
87
|
+
are routed as if it weren't there; other paths pass through unchanged.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
def __init__(self, app: Callable, base_path: str) -> None:
|
|
91
|
+
self.app = app
|
|
92
|
+
self.base_path = base_path
|
|
93
|
+
|
|
94
|
+
async def __call__(self, scope: Dict[str, Any], receive: Callable, send: Callable) -> None:
|
|
95
|
+
"""Rewrite the request path, then call the wrapped app."""
|
|
96
|
+
if scope.get("type") in ("http", "websocket"):
|
|
97
|
+
path = strip_base_path(scope.get("path", ""), self.base_path)
|
|
98
|
+
if path != scope.get("path"):
|
|
99
|
+
scope = {**scope, "path": path, "raw_path": path.encode()}
|
|
100
|
+
await self.app(scope, receive, send)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
75
104
|
def _add_main_routes(app: FastAPI, route_mapper: RouteMapper) -> None:
|
|
76
105
|
"""Add main API routes to the FastAPI app.
|
|
77
106
|
|
|
@@ -14,9 +14,9 @@ License: BSD 3-Clause
|
|
|
14
14
|
import asyncio
|
|
15
15
|
import warnings
|
|
16
16
|
from flask import Flask, jsonify, request, Response as FlaskResponse
|
|
17
|
-
from typing import Any, Dict, Optional
|
|
17
|
+
from typing import Any, Callable, Dict, Optional
|
|
18
18
|
|
|
19
|
-
from api_dock.route_mapper import collect_multi_query_params, RouteMapper
|
|
19
|
+
from api_dock.route_mapper import collect_multi_query_params, RouteMapper, strip_base_path
|
|
20
20
|
|
|
21
21
|
|
|
22
22
|
#
|
|
@@ -52,12 +52,35 @@ def create_app(config_path: Optional[str] = None) -> Flask:
|
|
|
52
52
|
_add_main_routes(app, route_mapper)
|
|
53
53
|
_add_error_handlers(app)
|
|
54
54
|
|
|
55
|
+
if route_mapper.base_path:
|
|
56
|
+
app.wsgi_app = _strip_base_path(app.wsgi_app, route_mapper.base_path)
|
|
57
|
+
|
|
55
58
|
return app
|
|
56
59
|
|
|
57
60
|
|
|
58
61
|
#
|
|
59
62
|
# INTERNAL
|
|
60
63
|
#
|
|
64
|
+
def _strip_base_path(wsgi_app: Callable, base_path: str) -> Callable:
|
|
65
|
+
"""Wrap a WSGI app so it is also served under ``settings.base_path``.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
wsgi_app: The Flask WSGI app.
|
|
69
|
+
base_path: Normalized prefix, e.g. "/dock".
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
WSGI app that routes ``/dock/...`` as ``/...`` (other paths unchanged).
|
|
73
|
+
"""
|
|
74
|
+
def middleware(environ: Dict[str, Any], start_response: Callable) -> Any:
|
|
75
|
+
"""Strip the base path from PATH_INFO, then call the Flask app."""
|
|
76
|
+
path = environ.get("PATH_INFO", "")
|
|
77
|
+
stripped = strip_base_path(path, base_path)
|
|
78
|
+
if stripped != path:
|
|
79
|
+
environ["PATH_INFO"] = stripped
|
|
80
|
+
return wsgi_app(environ, start_response)
|
|
81
|
+
return middleware
|
|
82
|
+
|
|
83
|
+
|
|
61
84
|
def _add_main_routes(app: Flask, route_mapper: RouteMapper) -> None:
|
|
62
85
|
"""Add main API routes to the Flask app.
|
|
63
86
|
|
|
@@ -11,7 +11,10 @@ License: BSD 3-Clause
|
|
|
11
11
|
#
|
|
12
12
|
# IMPORTS
|
|
13
13
|
#
|
|
14
|
+
import asyncio
|
|
14
15
|
import json
|
|
16
|
+
import re
|
|
17
|
+
import threading
|
|
15
18
|
import httpx
|
|
16
19
|
import yaml
|
|
17
20
|
from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
|
|
@@ -54,9 +57,77 @@ HOP_BY_HOP_HEADERS: frozenset = frozenset({
|
|
|
54
57
|
})
|
|
55
58
|
|
|
56
59
|
|
|
60
|
+
# Request headers never forwarded to an upstream remote. Host must be the
|
|
61
|
+
# upstream's own (httpx sets it): forwarding the client's Host makes the upstream
|
|
62
|
+
# build redirects and absolute URLs that point back at the proxy. The rest are
|
|
63
|
+
# hop-by-hop headers, or (content-length) recomputed by httpx from the body.
|
|
64
|
+
EXCLUDED_REQUEST_HEADERS: frozenset = frozenset({
|
|
65
|
+
"connection",
|
|
66
|
+
"content-length",
|
|
67
|
+
"host",
|
|
68
|
+
"keep-alive",
|
|
69
|
+
"proxy-authorization",
|
|
70
|
+
"te",
|
|
71
|
+
"trailers",
|
|
72
|
+
"transfer-encoding",
|
|
73
|
+
"upgrade",
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
# `settings.duckdb` options. Every key is applied to each query's DuckDB
|
|
77
|
+
# connection as `SET <key> = <value>` (memory_limit, threads, temp_directory,
|
|
78
|
+
# ...), except max_concurrent_queries, which caps how many database queries run
|
|
79
|
+
# at once in this process (others wait their turn).
|
|
80
|
+
DUCKDB_SETTINGS_KEY: str = "duckdb"
|
|
81
|
+
MAX_CONCURRENT_QUERIES_KEY: str = "max_concurrent_queries"
|
|
82
|
+
|
|
83
|
+
# `settings.base_path`: an optional URL prefix (e.g. "/dock") the API is also
|
|
84
|
+
# served under, for when a proxy/CDN forwards a path prefix unchanged.
|
|
85
|
+
BASE_PATH_KEY: str = "base_path"
|
|
86
|
+
|
|
87
|
+
DUCKDB_OPTION_PATTERN: re.Pattern = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
88
|
+
|
|
89
|
+
|
|
57
90
|
#
|
|
58
91
|
# PUBLIC
|
|
59
92
|
#
|
|
93
|
+
def normalize_base_path(base_path: Any) -> Optional[str]:
|
|
94
|
+
"""Normalize a ``base_path`` setting to ``/prefix`` form.
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
base_path: Configured value (e.g. "dock", "/dock/"), or None/empty.
|
|
98
|
+
|
|
99
|
+
Returns:
|
|
100
|
+
"/prefix" without a trailing slash, or None if no prefix is set.
|
|
101
|
+
"""
|
|
102
|
+
if not base_path:
|
|
103
|
+
return None
|
|
104
|
+
stripped = str(base_path).strip().strip("/")
|
|
105
|
+
return f"/{stripped}" if stripped else None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def strip_base_path(path: str, base_path: Optional[str]) -> str:
|
|
109
|
+
"""Remove a base path prefix from a request path, if present.
|
|
110
|
+
|
|
111
|
+
Paths without the prefix are returned unchanged, so the API answers both
|
|
112
|
+
with and without it (e.g. direct calls and health checks still work).
|
|
113
|
+
|
|
114
|
+
Args:
|
|
115
|
+
path: Request path, e.g. "/dock/birdnet/latest/detections/".
|
|
116
|
+
base_path: Normalized prefix (see normalize_base_path), or None.
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
The path without the prefix ("/birdnet/latest/detections/"), "/" for
|
|
120
|
+
the prefix itself, or the original path.
|
|
121
|
+
"""
|
|
122
|
+
if not base_path:
|
|
123
|
+
return path
|
|
124
|
+
if path == base_path or path == f"{base_path}/":
|
|
125
|
+
return "/"
|
|
126
|
+
if path.startswith(f"{base_path}/"):
|
|
127
|
+
return path[len(base_path):]
|
|
128
|
+
return path
|
|
129
|
+
|
|
130
|
+
|
|
60
131
|
def collect_multi_query_params(items: Iterable[Tuple[str, str]]) -> Dict[str, List[str]]:
|
|
61
132
|
"""Group repeated query string pairs into a name-to-value-list mapping.
|
|
62
133
|
|
|
@@ -100,6 +171,11 @@ class RouteMapper:
|
|
|
100
171
|
self.database_names = get_database_names(self.config)
|
|
101
172
|
self.settings = get_settings(self.config)
|
|
102
173
|
self.listing_specs, self.listing_warnings = resolve_listing_specs(self.config)
|
|
174
|
+
self.base_path = normalize_base_path(self.settings.get(BASE_PATH_KEY))
|
|
175
|
+
self.duckdb_statements, max_queries = _duckdb_settings(
|
|
176
|
+
self.settings.get(DUCKDB_SETTINGS_KEY)
|
|
177
|
+
)
|
|
178
|
+
self._query_slots = threading.BoundedSemaphore(max_queries) if max_queries else None
|
|
103
179
|
|
|
104
180
|
def get_config_metadata(self) -> Dict[str, Any]:
|
|
105
181
|
"""Get API metadata from configuration.
|
|
@@ -251,7 +327,7 @@ class RouteMapper:
|
|
|
251
327
|
return PreparedRequest(
|
|
252
328
|
url=full_url,
|
|
253
329
|
method=method,
|
|
254
|
-
headers=headers or {},
|
|
330
|
+
headers=_filter_request_headers(headers or {}),
|
|
255
331
|
params=filtered_query_params,
|
|
256
332
|
cookies=filtered_cookies,
|
|
257
333
|
body=body,
|
|
@@ -492,35 +568,22 @@ class RouteMapper:
|
|
|
492
568
|
except (ValueError, yaml.YAMLError):
|
|
493
569
|
return _error_response(500, "SQL query error")
|
|
494
570
|
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
# Authenticate every local table (as before) plus any shared tables
|
|
501
|
-
# the query references, then expose [[schema.table]] refs as views.
|
|
502
|
-
auth_tables = get_local_table_references(database_config, shared_config)
|
|
503
|
-
local_names = {table.sql_name for table in auth_tables}
|
|
504
|
-
auth_tables += [ref for ref in table_refs if ref.sql_name not in local_names]
|
|
505
|
-
setup_table_storage_authentication(conn, auth_tables)
|
|
506
|
-
for statement in build_schema_view_statements(table_refs):
|
|
507
|
-
conn.execute(statement)
|
|
508
|
-
|
|
509
|
-
result = conn.execute(sql_query).fetchall()
|
|
510
|
-
columns = [desc[0] for desc in conn.description] if conn.description else []
|
|
511
|
-
conn.close()
|
|
512
|
-
|
|
513
|
-
response_data = []
|
|
514
|
-
for row in result:
|
|
515
|
-
row_dict = {}
|
|
516
|
-
for col, val in zip(columns, row):
|
|
517
|
-
row_dict[col] = _make_json_safe(val)
|
|
518
|
-
response_data.append(row_dict)
|
|
519
|
-
|
|
520
|
-
return _json_response(response_data)
|
|
571
|
+
# Authenticate every local table plus any shared tables the query
|
|
572
|
+
# references, then expose [[schema.table]] refs as views.
|
|
573
|
+
auth_tables = get_local_table_references(database_config, shared_config)
|
|
574
|
+
local_names = {table.sql_name for table in auth_tables}
|
|
575
|
+
auth_tables += [ref for ref in table_refs if ref.sql_name not in local_names]
|
|
521
576
|
|
|
577
|
+
try:
|
|
578
|
+
# DuckDB calls block; run them in a worker thread so one slow query
|
|
579
|
+
# doesn't stall every other request (and health checks) meanwhile.
|
|
580
|
+
response_data = await asyncio.to_thread(
|
|
581
|
+
self._run_query, sql_query, auth_tables,
|
|
582
|
+
build_schema_view_statements(table_refs),
|
|
583
|
+
)
|
|
522
584
|
except Exception:
|
|
523
585
|
return _error_response(500, "Database query error")
|
|
586
|
+
return _json_response(response_data)
|
|
524
587
|
|
|
525
588
|
def is_remote_name(self, name: str) -> bool:
|
|
526
589
|
"""Check if a given name is a configured remote name.
|
|
@@ -598,6 +661,51 @@ class RouteMapper:
|
|
|
598
661
|
except Exception as e:
|
|
599
662
|
return _error_response(500, f"Sync wrapper error: {str(e)}")
|
|
600
663
|
|
|
664
|
+
def _run_query(
|
|
665
|
+
self,
|
|
666
|
+
sql_query: str,
|
|
667
|
+
auth_tables: List[Any],
|
|
668
|
+
view_statements: List[str]) -> List[Dict[str, Any]]:
|
|
669
|
+
"""Execute a database query on a fresh DuckDB connection (blocking).
|
|
670
|
+
|
|
671
|
+
Applies the ``settings.duckdb`` options, storage authentication and
|
|
672
|
+
schema views, then runs the query. Honors max_concurrent_queries.
|
|
673
|
+
|
|
674
|
+
Args:
|
|
675
|
+
sql_query: The SQL to run.
|
|
676
|
+
auth_tables: TableReferences to set up storage authentication for.
|
|
677
|
+
view_statements: CREATE SCHEMA/VIEW statements to run first.
|
|
678
|
+
|
|
679
|
+
Returns:
|
|
680
|
+
Result rows as JSON-safe dicts.
|
|
681
|
+
"""
|
|
682
|
+
import duckdb
|
|
683
|
+
|
|
684
|
+
# getattr: RouteMappers built without __init__ (e.g. in tests) have neither.
|
|
685
|
+
slots = getattr(self, '_query_slots', None)
|
|
686
|
+
if slots is not None:
|
|
687
|
+
slots.acquire()
|
|
688
|
+
try:
|
|
689
|
+
conn = duckdb.connect(database=':memory:')
|
|
690
|
+
try:
|
|
691
|
+
for statement in getattr(self, 'duckdb_statements', []):
|
|
692
|
+
conn.execute(statement)
|
|
693
|
+
setup_table_storage_authentication(conn, auth_tables)
|
|
694
|
+
for statement in view_statements:
|
|
695
|
+
conn.execute(statement)
|
|
696
|
+
result = conn.execute(sql_query).fetchall()
|
|
697
|
+
columns = [desc[0] for desc in conn.description] if conn.description else []
|
|
698
|
+
finally:
|
|
699
|
+
conn.close()
|
|
700
|
+
finally:
|
|
701
|
+
if slots is not None:
|
|
702
|
+
slots.release()
|
|
703
|
+
|
|
704
|
+
return [
|
|
705
|
+
{column: _make_json_safe(value) for column, value in zip(columns, row)}
|
|
706
|
+
for row in result
|
|
707
|
+
]
|
|
708
|
+
|
|
601
709
|
def _is_remote_filename(self, filename: str) -> bool:
|
|
602
710
|
"""Check if a filename corresponds to a remote config file.
|
|
603
711
|
|
|
@@ -634,6 +742,62 @@ class RouteMapper:
|
|
|
634
742
|
#
|
|
635
743
|
# INTERNAL
|
|
636
744
|
#
|
|
745
|
+
def _duckdb_settings(options: Any) -> Tuple[List[str], Optional[int]]:
|
|
746
|
+
"""Turn ``settings.duckdb`` into SET statements and a concurrency cap.
|
|
747
|
+
|
|
748
|
+
Args:
|
|
749
|
+
options: Mapping of DuckDB option -> value, plus optional
|
|
750
|
+
max_concurrent_queries; or None.
|
|
751
|
+
|
|
752
|
+
Returns:
|
|
753
|
+
Tuple of (SET statements, max concurrent queries or None).
|
|
754
|
+
|
|
755
|
+
Raises:
|
|
756
|
+
ValueError: If options isn't a mapping, an option name isn't a plain
|
|
757
|
+
identifier, or max_concurrent_queries isn't a positive integer.
|
|
758
|
+
"""
|
|
759
|
+
if not options:
|
|
760
|
+
return ([], None)
|
|
761
|
+
if not isinstance(options, dict):
|
|
762
|
+
raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY} must be a mapping")
|
|
763
|
+
|
|
764
|
+
max_queries = options.get(MAX_CONCURRENT_QUERIES_KEY)
|
|
765
|
+
if max_queries is not None and (not isinstance(max_queries, int) or max_queries < 1):
|
|
766
|
+
raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY}.{MAX_CONCURRENT_QUERIES_KEY} "
|
|
767
|
+
"must be a positive integer")
|
|
768
|
+
|
|
769
|
+
statements = []
|
|
770
|
+
for name, value in options.items():
|
|
771
|
+
if name == MAX_CONCURRENT_QUERIES_KEY:
|
|
772
|
+
continue
|
|
773
|
+
if not DUCKDB_OPTION_PATTERN.match(str(name)):
|
|
774
|
+
raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY}: invalid option name '{name}'")
|
|
775
|
+
if isinstance(value, bool):
|
|
776
|
+
literal = "true" if value else "false"
|
|
777
|
+
elif isinstance(value, (int, float)):
|
|
778
|
+
literal = str(value)
|
|
779
|
+
else:
|
|
780
|
+
literal = "'" + str(value).replace("'", "''") + "'"
|
|
781
|
+
statements.append(f"SET {name} = {literal}")
|
|
782
|
+
return (statements, max_queries)
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def _filter_request_headers(headers: Dict[str, str]) -> Dict[str, str]:
|
|
786
|
+
"""Drop request headers that must not be forwarded upstream.
|
|
787
|
+
|
|
788
|
+
Args:
|
|
789
|
+
headers: Incoming client request headers.
|
|
790
|
+
|
|
791
|
+
Returns:
|
|
792
|
+
Headers safe to forward (see EXCLUDED_REQUEST_HEADERS).
|
|
793
|
+
"""
|
|
794
|
+
return {
|
|
795
|
+
key: value
|
|
796
|
+
for key, value in headers.items()
|
|
797
|
+
if key.lower() not in EXCLUDED_REQUEST_HEADERS
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
|
|
637
801
|
def _resolve_timeout(value: Any) -> Optional[float]:
|
|
638
802
|
"""Resolve the configured timeout to seconds, or None to disable it.
|
|
639
803
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: api_dock
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.2
|
|
4
4
|
Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
|
|
5
5
|
Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
|
|
6
6
|
License-Expression: BSD-3-Clause
|
|
@@ -219,19 +219,33 @@ remotes:
|
|
|
219
219
|
settings:
|
|
220
220
|
add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
|
|
221
221
|
follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
|
|
222
|
+
follow_redirects: true # Follow remote redirects (default: true)
|
|
222
223
|
timeout: 10 # Upstream request timeout in seconds (default: 10)
|
|
224
|
+
base_path: /dock # Also serve the API under this prefix (default: none)
|
|
225
|
+
duckdb: # Options for database queries (default: none)
|
|
226
|
+
memory_limit: 700MB
|
|
227
|
+
threads: 2
|
|
228
|
+
max_concurrent_queries: 2
|
|
223
229
|
```
|
|
224
230
|
|
|
225
|
-
###
|
|
231
|
+
### Settings
|
|
226
232
|
|
|
227
|
-
The optional `settings` section controls HTTP behavior:
|
|
233
|
+
The optional `settings` section controls HTTP and query behavior:
|
|
228
234
|
|
|
229
235
|
- **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
|
|
230
236
|
|
|
231
237
|
- **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
|
|
232
238
|
|
|
239
|
+
- **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
|
|
240
|
+
|
|
233
241
|
- **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
|
|
234
242
|
|
|
243
|
+
- **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
|
|
244
|
+
|
|
245
|
+
- **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
|
|
246
|
+
|
|
247
|
+
Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
|
|
248
|
+
|
|
235
249
|
### Catalog Endpoints (`expose`)
|
|
236
250
|
|
|
237
251
|
The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
|
|
@@ -1601,7 +1615,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
|
|
|
1601
1615
|
|
|
1602
1616
|
```bash
|
|
1603
1617
|
# 0. Start from a clean, up-to-date main
|
|
1604
|
-
export VERSION=0.8.
|
|
1618
|
+
export VERSION=0.8.2 # the NEW version, no leading "v"
|
|
1605
1619
|
git checkout main
|
|
1606
1620
|
git pull origin main
|
|
1607
1621
|
git status
|
|
@@ -1612,7 +1626,7 @@ git status
|
|
|
1612
1626
|
pixi run -e dev pytest -q
|
|
1613
1627
|
|
|
1614
1628
|
# 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
|
|
1615
|
-
export COMMIT_MESSAGE='
|
|
1629
|
+
export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
|
|
1616
1630
|
git add -A
|
|
1617
1631
|
git commit -m "v$VERSION: $COMMIT_MESSAGE"
|
|
1618
1632
|
git tag "v$VERSION"
|
|
@@ -1624,17 +1638,14 @@ gh release create "v$VERSION" \
|
|
|
1624
1638
|
--title "v$VERSION" \
|
|
1625
1639
|
--notes "$(cat <<'EOF'
|
|
1626
1640
|
* new features
|
|
1627
|
-
-
|
|
1628
|
-
-
|
|
1629
|
-
- Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
|
|
1630
|
-
- `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
|
|
1631
|
-
- Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
|
|
1641
|
+
- `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
|
|
1642
|
+
- `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
|
|
1632
1643
|
* bug fixes
|
|
1633
|
-
-
|
|
1644
|
+
- Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
|
|
1645
|
+
- Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
|
|
1634
1646
|
* cleanup / other improvements
|
|
1635
|
-
-
|
|
1636
|
-
-
|
|
1637
|
-
- Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
|
|
1647
|
+
- README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
|
|
1648
|
+
- Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
|
|
1638
1649
|
EOF
|
|
1639
1650
|
)"
|
|
1640
1651
|
|
|
@@ -29,6 +29,7 @@ api_dock/example_api_dock_config/remotes/example_remote.yaml
|
|
|
29
29
|
tests/test_inject_cookies.py
|
|
30
30
|
tests/test_listings.py
|
|
31
31
|
tests/test_proxy_pipeline.py
|
|
32
|
+
tests/test_runtime_settings.py
|
|
32
33
|
tests/test_schema_unions.py
|
|
33
34
|
tests/test_shared_database_config.py
|
|
34
35
|
tests/test_sql_builder.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "api_dock"
|
|
7
|
-
version = "0.8.
|
|
7
|
+
version = "0.8.2"
|
|
8
8
|
description = "A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "BSD-3-Clause"
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""
|
|
2
|
+
|
|
3
|
+
Tests for runtime settings and request handling.
|
|
4
|
+
|
|
5
|
+
Covers ``settings.duckdb`` (per-connection DuckDB options and
|
|
6
|
+
max_concurrent_queries), running database queries off the event loop,
|
|
7
|
+
``settings.base_path`` on the FastAPI and Flask apps, and the request headers
|
|
8
|
+
that are not forwarded to upstream remotes.
|
|
9
|
+
|
|
10
|
+
License: BSD 3-Clause
|
|
11
|
+
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
#
|
|
15
|
+
# IMPORTS
|
|
16
|
+
#
|
|
17
|
+
import asyncio
|
|
18
|
+
import json
|
|
19
|
+
import time
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Dict, List
|
|
22
|
+
from unittest.mock import MagicMock, patch
|
|
23
|
+
|
|
24
|
+
import pytest
|
|
25
|
+
import yaml
|
|
26
|
+
from fastapi.testclient import TestClient
|
|
27
|
+
|
|
28
|
+
from api_dock import fast_api, flask_api
|
|
29
|
+
from api_dock.route_mapper import (
|
|
30
|
+
_duckdb_settings,
|
|
31
|
+
_filter_request_headers,
|
|
32
|
+
normalize_base_path,
|
|
33
|
+
RouteMapper,
|
|
34
|
+
strip_base_path,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
#
|
|
39
|
+
# CONSTANTS
|
|
40
|
+
#
|
|
41
|
+
# A deliberately slow, dependency-free query (~0.5-2 s on one thread).
|
|
42
|
+
SLOW_SQL: str = "SELECT SUM(i * i) AS total FROM range(300000000) t(i)"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
#
|
|
46
|
+
# PUBLIC
|
|
47
|
+
#
|
|
48
|
+
class TestBasePathHelpers:
|
|
49
|
+
"""normalize_base_path / strip_base_path."""
|
|
50
|
+
|
|
51
|
+
@pytest.mark.parametrize("value,expected", [
|
|
52
|
+
(None, None), ("", None), ("/", None),
|
|
53
|
+
("dock", "/dock"), ("/dock/", "/dock"), ("/a/b", "/a/b"),
|
|
54
|
+
])
|
|
55
|
+
def test_normalize(self, value: Any, expected: Any) -> None:
|
|
56
|
+
assert normalize_base_path(value) == expected
|
|
57
|
+
|
|
58
|
+
@pytest.mark.parametrize("path,expected", [
|
|
59
|
+
("/dock/birdnet/latest/detections/", "/birdnet/latest/detections/"),
|
|
60
|
+
("/dock", "/"),
|
|
61
|
+
("/dock/", "/"),
|
|
62
|
+
("/birdnet/latest/", "/birdnet/latest/"),
|
|
63
|
+
("/docks/x", "/docks/x"),
|
|
64
|
+
])
|
|
65
|
+
def test_strip(self, path: str, expected: str) -> None:
|
|
66
|
+
assert strip_base_path(path, "/dock") == expected
|
|
67
|
+
|
|
68
|
+
def test_strip_without_base_path(self) -> None:
|
|
69
|
+
assert strip_base_path("/dock/x", None) == "/dock/x"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class TestDuckdbSettings:
|
|
73
|
+
"""``settings.duckdb`` parsing."""
|
|
74
|
+
|
|
75
|
+
def test_statements_and_cap(self) -> None:
|
|
76
|
+
statements, cap = _duckdb_settings({
|
|
77
|
+
"memory_limit": "700MB", "threads": 2, "preserve_insertion_order": False,
|
|
78
|
+
"temp_directory": "/tmp/it's", "max_concurrent_queries": 3,
|
|
79
|
+
})
|
|
80
|
+
assert statements == [
|
|
81
|
+
"SET memory_limit = '700MB'",
|
|
82
|
+
"SET threads = 2",
|
|
83
|
+
"SET preserve_insertion_order = false",
|
|
84
|
+
"SET temp_directory = '/tmp/it''s'",
|
|
85
|
+
]
|
|
86
|
+
assert cap == 3
|
|
87
|
+
|
|
88
|
+
def test_empty(self) -> None:
|
|
89
|
+
assert _duckdb_settings(None) == ([], None)
|
|
90
|
+
|
|
91
|
+
@pytest.mark.parametrize("options", [
|
|
92
|
+
{"bad name": 1}, {"max_concurrent_queries": 0}, {"max_concurrent_queries": "2"}, ["x"],
|
|
93
|
+
])
|
|
94
|
+
def test_invalid(self, options: Any) -> None:
|
|
95
|
+
with pytest.raises(ValueError):
|
|
96
|
+
_duckdb_settings(options)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class TestRequestHeaderFilter:
|
|
100
|
+
"""Headers not forwarded to remotes."""
|
|
101
|
+
|
|
102
|
+
def test_drops_host_and_hop_by_hop(self) -> None:
|
|
103
|
+
headers = {
|
|
104
|
+
"Host": "proxy.example.com", "Connection": "keep-alive", "Content-Length": "3",
|
|
105
|
+
"Accept": "application/json", "Content-Type": "application/json", "X-Custom": "1",
|
|
106
|
+
}
|
|
107
|
+
assert _filter_request_headers(headers) == {
|
|
108
|
+
"Accept": "application/json", "Content-Type": "application/json", "X-Custom": "1",
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
@pytest.mark.anyio
|
|
112
|
+
async def test_prepared_request_has_no_host(self) -> None:
|
|
113
|
+
mapper = RouteMapper.__new__(RouteMapper)
|
|
114
|
+
mapper.remote_names, mapper.database_names, mapper.settings = ["core"], [], {}
|
|
115
|
+
mapper.config = {"remotes": ["core"]}
|
|
116
|
+
with patch("api_dock.route_mapper.is_versioned_remote", return_value=False), \
|
|
117
|
+
patch("api_dock.route_mapper.is_route_allowed", return_value=True), \
|
|
118
|
+
patch("api_dock.route_mapper.find_remote_config",
|
|
119
|
+
return_value={"url": "https://api.example.com"}), \
|
|
120
|
+
patch("api_dock.route_mapper.filter_remote_query_params",
|
|
121
|
+
side_effect=lambda qp, *a, **kw: qp), \
|
|
122
|
+
patch("api_dock.route_mapper.find_route_mapping", return_value=None), \
|
|
123
|
+
patch("api_dock.route_mapper.filter_cookies_by_config", return_value={}):
|
|
124
|
+
prepared = await mapper.prepare_remote_request(
|
|
125
|
+
"core", "detections/", "GET", headers={"host": "proxy:8000", "accept": "*/*"},
|
|
126
|
+
)
|
|
127
|
+
assert prepared.headers == {"accept": "*/*"}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class TestQueryExecution:
|
|
131
|
+
"""Database queries: DuckDB options, worker threads, concurrency cap."""
|
|
132
|
+
|
|
133
|
+
@pytest.mark.anyio
|
|
134
|
+
async def test_duckdb_options_applied(
|
|
135
|
+
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
136
|
+
mapper = _make_mapper(tmp_path, monkeypatch, settings={
|
|
137
|
+
"duckdb": {"threads": 1, "memory_limit": "300MB"},
|
|
138
|
+
})
|
|
139
|
+
result = await mapper.map_database_route("db", "settings")
|
|
140
|
+
row = json.loads(result.content)[0]
|
|
141
|
+
assert row["threads"] == "1"
|
|
142
|
+
assert row["memory_limit"] != _default_duckdb_setting("memory_limit")
|
|
143
|
+
|
|
144
|
+
@pytest.mark.anyio
|
|
145
|
+
async def test_query_does_not_block_event_loop(
|
|
146
|
+
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
147
|
+
mapper = _make_mapper(tmp_path, monkeypatch, settings={"duckdb": {"threads": 1}})
|
|
148
|
+
gaps: List[float] = []
|
|
149
|
+
|
|
150
|
+
async def ticker() -> None:
|
|
151
|
+
"""Record the gap between event-loop wakeups (large = loop was blocked)."""
|
|
152
|
+
last = time.perf_counter()
|
|
153
|
+
while True:
|
|
154
|
+
await asyncio.sleep(0.01)
|
|
155
|
+
now = time.perf_counter()
|
|
156
|
+
gaps.append(now - last)
|
|
157
|
+
last = now
|
|
158
|
+
|
|
159
|
+
tick = asyncio.create_task(ticker())
|
|
160
|
+
start = time.perf_counter()
|
|
161
|
+
result = await mapper.map_database_route("db", "slow")
|
|
162
|
+
duration = time.perf_counter() - start
|
|
163
|
+
tick.cancel()
|
|
164
|
+
|
|
165
|
+
assert result.status_code == 200, result.content
|
|
166
|
+
assert duration > 0.2, 'query too fast to show blocking; raise SLOW_SQL range'
|
|
167
|
+
assert max(gaps) < duration / 2
|
|
168
|
+
|
|
169
|
+
@pytest.mark.anyio
|
|
170
|
+
async def test_concurrency_cap_acquires_and_releases(
|
|
171
|
+
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
172
|
+
mapper = _make_mapper(tmp_path, monkeypatch, settings={
|
|
173
|
+
"duckdb": {"max_concurrent_queries": 1},
|
|
174
|
+
})
|
|
175
|
+
slots = MagicMock()
|
|
176
|
+
mapper._query_slots = slots
|
|
177
|
+
await mapper.map_database_route("db", "settings")
|
|
178
|
+
slots.acquire.assert_called_once()
|
|
179
|
+
slots.release.assert_called_once()
|
|
180
|
+
|
|
181
|
+
@pytest.mark.anyio
|
|
182
|
+
async def test_failed_query_releases_slot(
|
|
183
|
+
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
184
|
+
mapper = _make_mapper(tmp_path, monkeypatch, settings={
|
|
185
|
+
"duckdb": {"max_concurrent_queries": 1},
|
|
186
|
+
})
|
|
187
|
+
result = await mapper.map_database_route("db", "broken")
|
|
188
|
+
assert result.status_code == 500
|
|
189
|
+
assert mapper._query_slots.acquire(blocking=False)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
class TestBasePathApps:
|
|
193
|
+
"""``settings.base_path`` on the FastAPI and Flask apps."""
|
|
194
|
+
|
|
195
|
+
def test_fastapi(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
196
|
+
config_path = _write_config(tmp_path, monkeypatch, settings={"base_path": "/dock"})
|
|
197
|
+
client = TestClient(fast_api.create_app(config_path))
|
|
198
|
+
for prefix in ("", "/dock"):
|
|
199
|
+
assert client.get(f"{prefix}/").json()["name"] == "t"
|
|
200
|
+
assert client.get(f"{prefix}/databases").json() == ["db"]
|
|
201
|
+
assert client.get(f"{prefix}/db/settings").status_code == 200
|
|
202
|
+
assert client.get("/docks/db/settings").status_code == 404
|
|
203
|
+
|
|
204
|
+
def test_flask(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
205
|
+
config_path = _write_config(tmp_path, monkeypatch, settings={"base_path": "dock"})
|
|
206
|
+
client = flask_api.create_app(config_path).test_client()
|
|
207
|
+
for prefix in ("", "/dock"):
|
|
208
|
+
assert client.get(f"{prefix}/").get_json()["name"] == "t"
|
|
209
|
+
assert client.get(f"{prefix}/databases").get_json() == ["db"]
|
|
210
|
+
assert client.get(f"{prefix}/db/settings").status_code == 200
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
#
|
|
214
|
+
# INTERNAL
|
|
215
|
+
#
|
|
216
|
+
def _write_config(tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
|
|
217
|
+
settings: Dict[str, Any]) -> str:
|
|
218
|
+
"""Write a one-database config under tmp_path, chdir there, return its path."""
|
|
219
|
+
config_dir = tmp_path / "api_dock_config"
|
|
220
|
+
(config_dir / "databases").mkdir(parents=True)
|
|
221
|
+
(config_dir / "config.yaml").write_text(yaml.safe_dump({
|
|
222
|
+
"name": "t", "databases": ["db"], "expose": {"databases": True, "dict": False},
|
|
223
|
+
"settings": settings,
|
|
224
|
+
}))
|
|
225
|
+
(config_dir / "databases" / "db.yaml").write_text(yaml.safe_dump({
|
|
226
|
+
"name": "db",
|
|
227
|
+
"routes": [
|
|
228
|
+
{"route": "settings", "sql": "SELECT current_setting('threads')::VARCHAR AS threads, "
|
|
229
|
+
"current_setting('memory_limit') AS memory_limit"},
|
|
230
|
+
{"route": "slow", "sql": SLOW_SQL},
|
|
231
|
+
{"route": "broken", "sql": "SELECT * FROM no_such_table"},
|
|
232
|
+
],
|
|
233
|
+
}))
|
|
234
|
+
monkeypatch.chdir(tmp_path)
|
|
235
|
+
return str(config_dir / "config.yaml")
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _default_duckdb_setting(name: str) -> str:
|
|
239
|
+
"""A DuckDB setting's value on a fresh connection with no options applied."""
|
|
240
|
+
import duckdb
|
|
241
|
+
|
|
242
|
+
with duckdb.connect() as conn:
|
|
243
|
+
return str(conn.execute(f"SELECT current_setting('{name}')").fetchone()[0])
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _make_mapper(tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
|
|
247
|
+
settings: Dict[str, Any]) -> RouteMapper:
|
|
248
|
+
"""Build a RouteMapper over the one-database test config."""
|
|
249
|
+
return RouteMapper(_write_config(tmp_path, monkeypatch, settings))
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/example_db.yaml
RENAMED
|
File without changes
|
{api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/remotes/example_remote.yaml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|