api-dock 0.8.1__tar.gz → 0.8.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {api_dock-0.8.1 → api_dock-0.8.2}/PKG-INFO +25 -14
  2. {api_dock-0.8.1 → api_dock-0.8.2}/README.md +24 -13
  3. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/config.yaml +5 -0
  4. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/fast_api.py +32 -3
  5. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/flask_api.py +25 -2
  6. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/route_mapper.py +191 -27
  7. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/PKG-INFO +25 -14
  8. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/SOURCES.txt +1 -0
  9. {api_dock-0.8.1 → api_dock-0.8.2}/pyproject.toml +1 -1
  10. api_dock-0.8.2/tests/test_runtime_settings.py +249 -0
  11. {api_dock-0.8.1 → api_dock-0.8.2}/LICENSE.md +0 -0
  12. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/__init__.py +0 -0
  13. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/auth.py +0 -0
  14. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/cli.py +0 -0
  15. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/config.py +0 -0
  16. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/config_discovery.py +0 -0
  17. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/database_config.py +0 -0
  18. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/encryption.py +0 -0
  19. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/config.yaml +0 -0
  20. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/databases/example_db.yaml +0 -0
  21. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/example_api_dock_config/remotes/example_remote.yaml +0 -0
  22. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/listings.py +0 -0
  23. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/sql_builder.py +0 -0
  24. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/storage_auth.py +0 -0
  25. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock/types.py +0 -0
  26. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/dependency_links.txt +0 -0
  27. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/entry_points.txt +0 -0
  28. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/requires.txt +0 -0
  29. {api_dock-0.8.1 → api_dock-0.8.2}/api_dock.egg-info/top_level.txt +0 -0
  30. {api_dock-0.8.1 → api_dock-0.8.2}/setup.cfg +0 -0
  31. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_inject_cookies.py +0 -0
  32. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_listings.py +0 -0
  33. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_proxy_pipeline.py +0 -0
  34. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_schema_unions.py +0 -0
  35. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_shared_database_config.py +0 -0
  36. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_sql_builder.py +0 -0
  37. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_sql_selector.py +0 -0
  38. {api_dock-0.8.1 → api_dock-0.8.2}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: api_dock
3
- Version: 0.8.1
3
+ Version: 0.8.2
4
4
  Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
5
5
  Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
6
6
  License-Expression: BSD-3-Clause
@@ -219,19 +219,33 @@ remotes:
219
219
  settings:
220
220
  add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
221
221
  follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
222
+ follow_redirects: true # Follow remote redirects (default: true)
222
223
  timeout: 10 # Upstream request timeout in seconds (default: 10)
224
+ base_path: /dock # Also serve the API under this prefix (default: none)
225
+ duckdb: # Options for database queries (default: none)
226
+ memory_limit: 700MB
227
+ threads: 2
228
+ max_concurrent_queries: 2
223
229
  ```
224
230
 
225
- ### HTTP behavior Settings
231
+ ### Settings
226
232
 
227
- The optional `settings` section controls HTTP behavior:
233
+ The optional `settings` section controls HTTP and query behavior:
228
234
 
229
235
  - **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
230
236
 
231
237
  - **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
232
238
 
239
+ - **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
240
+
233
241
  - **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
234
242
 
243
+ - **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
244
+
245
+ - **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
246
+
247
+ Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
248
+
235
249
  ### Catalog Endpoints (`expose`)
236
250
 
237
251
  The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
@@ -1601,7 +1615,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
1601
1615
 
1602
1616
  ```bash
1603
1617
  # 0. Start from a clean, up-to-date main
1604
- export VERSION=0.8.1 # the NEW version, no leading "v"
1618
+ export VERSION=0.8.2 # the NEW version, no leading "v"
1605
1619
  git checkout main
1606
1620
  git pull origin main
1607
1621
  git status
@@ -1612,7 +1626,7 @@ git status
1612
1626
  pixi run -e dev pytest -q
1613
1627
 
1614
1628
  # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1615
- export COMMIT_MESSAGE='cross-schema unions, schema groups, source columns'
1629
+ export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
1616
1630
  git add -A
1617
1631
  git commit -m "v$VERSION: $COMMIT_MESSAGE"
1618
1632
  git tag "v$VERSION"
@@ -1624,17 +1638,14 @@ gh release create "v$VERSION" \
1624
1638
  --title "v$VERSION" \
1625
1639
  --notes "$(cat <<'EOF'
1626
1640
  * new features
1627
- - Cross-schema unions: `[[*.table]]` reads a table from every shared schema that has it, and `[[*!.table]]` does the same minus the current database/version's schema
1628
- - Named `schema_groups` in `databases/config.yaml`, used as `[[group.table]]` / `[[group!.table]]` (validated: known schemas only, no group/schema name clashes)
1629
- - Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
1630
- - `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
1631
- - Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
1641
+ - `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
1642
+ - `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
1632
1643
  * bug fixes
1633
- - A `!` union that removes every member returns no rows (with the right columns) instead of failing, and only the schemas a union actually reads get views and storage credentials
1644
+ - Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
1645
+ - Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
1634
1646
  * cleanup / other improvements
1635
- - Added `SqlContext`; `build_sql_query()` / `build_sql_query_with_tables()` accept an optional `context`
1636
- - README: new "Querying across schemas" section with an overlaps example; example `databases/config.yaml` shows `schema_groups` and a union route
1637
- - Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
1647
+ - README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
1648
+ - Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
1638
1649
  EOF
1639
1650
  )"
1640
1651
 
@@ -180,19 +180,33 @@ remotes:
180
180
  settings:
181
181
  add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
182
182
  follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
183
+ follow_redirects: true # Follow remote redirects (default: true)
183
184
  timeout: 10 # Upstream request timeout in seconds (default: 10)
185
+ base_path: /dock # Also serve the API under this prefix (default: none)
186
+ duckdb: # Options for database queries (default: none)
187
+ memory_limit: 700MB
188
+ threads: 2
189
+ max_concurrent_queries: 2
184
190
  ```
185
191
 
186
- ### HTTP behavior Settings
192
+ ### Settings
187
193
 
188
- The optional `settings` section controls HTTP behavior:
194
+ The optional `settings` section controls HTTP and query behavior:
189
195
 
190
196
  - **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
191
197
 
192
198
  - **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
193
199
 
200
+ - **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
201
+
194
202
  - **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
195
203
 
204
+ - **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
205
+
206
+ - **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
207
+
208
+ Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
209
+
196
210
  ### Catalog Endpoints (`expose`)
197
211
 
198
212
  The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
@@ -1562,7 +1576,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
1562
1576
 
1563
1577
  ```bash
1564
1578
  # 0. Start from a clean, up-to-date main
1565
- export VERSION=0.8.1 # the NEW version, no leading "v"
1579
+ export VERSION=0.8.2 # the NEW version, no leading "v"
1566
1580
  git checkout main
1567
1581
  git pull origin main
1568
1582
  git status
@@ -1573,7 +1587,7 @@ git status
1573
1587
  pixi run -e dev pytest -q
1574
1588
 
1575
1589
  # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1576
- export COMMIT_MESSAGE='cross-schema unions, schema groups, source columns'
1590
+ export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
1577
1591
  git add -A
1578
1592
  git commit -m "v$VERSION: $COMMIT_MESSAGE"
1579
1593
  git tag "v$VERSION"
@@ -1585,17 +1599,14 @@ gh release create "v$VERSION" \
1585
1599
  --title "v$VERSION" \
1586
1600
  --notes "$(cat <<'EOF'
1587
1601
  * new features
1588
- - Cross-schema unions: `[[*.table]]` reads a table from every shared schema that has it, and `[[*!.table]]` does the same minus the current database/version's schema
1589
- - Named `schema_groups` in `databases/config.yaml`, used as `[[group.table]]` / `[[group!.table]]` (validated: known schemas only, no group/schema name clashes)
1590
- - Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
1591
- - `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
1592
- - Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
1602
+ - `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
1603
+ - `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
1593
1604
  * bug fixes
1594
- - A `!` union that removes every member returns no rows (with the right columns) instead of failing, and only the schemas a union actually reads get views and storage credentials
1605
+ - Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
1606
+ - Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
1595
1607
  * cleanup / other improvements
1596
- - Added `SqlContext`; `build_sql_query()` / `build_sql_query_with_tables()` accept an optional `context`
1597
- - README: new "Querying across schemas" section with an overlaps example; example `databases/config.yaml` shows `schema_groups` and a union route
1598
- - Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
1608
+ - README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
1609
+ - Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
1599
1610
  EOF
1600
1611
  )"
1601
1612
 
@@ -15,6 +15,11 @@ settings:
15
15
  add_trailing_slash: false # Set true to auto-append trailing slash to proxied paths
16
16
  follow_redirects: true # Set false to pass 3xx redirects through to the client
17
17
  timeout: 10 # Upstream request timeout (seconds); null/false to disable
18
+ # base_path: /dock # Also serve the API under this prefix (e.g. behind a CDN path)
19
+ # duckdb: # Database query options (DuckDB SET options + a concurrency cap)
20
+ # memory_limit: 700MB
21
+ # threads: 2
22
+ # max_concurrent_queries: 2
18
23
 
19
24
  # Global route restrictions — applied to all remotes unless overridden per-remote.
20
25
  # Uncomment to block DELETE on every remote:
@@ -16,9 +16,14 @@ import warnings
16
16
  import httpx
17
17
  from fastapi import FastAPI, Request
18
18
  from fastapi.responses import JSONResponse, Response, StreamingResponse
19
- from typing import Any, Dict, Optional
20
-
21
- from api_dock.route_mapper import collect_multi_query_params, HOP_BY_HOP_HEADERS, RouteMapper
19
+ from typing import Any, Callable, Dict, Optional
20
+
21
+ from api_dock.route_mapper import (
22
+ collect_multi_query_params,
23
+ HOP_BY_HOP_HEADERS,
24
+ RouteMapper,
25
+ strip_base_path,
26
+ )
22
27
  from api_dock.types import PreparedRequest, ProxyResponse
23
28
 
24
29
 
@@ -66,12 +71,36 @@ def create_app(config_path: Optional[str] = None) -> FastAPI:
66
71
  _add_remote_routes(app, route_mapper)
67
72
  _add_error_handlers(app)
68
73
 
74
+ if route_mapper.base_path:
75
+ app.add_middleware(_StripBasePath, base_path=route_mapper.base_path)
76
+
69
77
  return app
70
78
 
71
79
 
72
80
  #
73
81
  # INTERNAL
74
82
  #
83
+ class _StripBasePath:
84
+ """ASGI middleware serving the app under ``settings.base_path`` as well.
85
+
86
+ Requests whose path starts with the base path (e.g. ``/dock/birdnet/...``)
87
+ are routed as if it weren't there; other paths pass through unchanged.
88
+ """
89
+
90
+ def __init__(self, app: Callable, base_path: str) -> None:
91
+ self.app = app
92
+ self.base_path = base_path
93
+
94
+ async def __call__(self, scope: Dict[str, Any], receive: Callable, send: Callable) -> None:
95
+ """Rewrite the request path, then call the wrapped app."""
96
+ if scope.get("type") in ("http", "websocket"):
97
+ path = strip_base_path(scope.get("path", ""), self.base_path)
98
+ if path != scope.get("path"):
99
+ scope = {**scope, "path": path, "raw_path": path.encode()}
100
+ await self.app(scope, receive, send)
101
+
102
+
103
+
75
104
  def _add_main_routes(app: FastAPI, route_mapper: RouteMapper) -> None:
76
105
  """Add main API routes to the FastAPI app.
77
106
 
@@ -14,9 +14,9 @@ License: BSD 3-Clause
14
14
  import asyncio
15
15
  import warnings
16
16
  from flask import Flask, jsonify, request, Response as FlaskResponse
17
- from typing import Any, Dict, Optional
17
+ from typing import Any, Callable, Dict, Optional
18
18
 
19
- from api_dock.route_mapper import collect_multi_query_params, RouteMapper
19
+ from api_dock.route_mapper import collect_multi_query_params, RouteMapper, strip_base_path
20
20
 
21
21
 
22
22
  #
@@ -52,12 +52,35 @@ def create_app(config_path: Optional[str] = None) -> Flask:
52
52
  _add_main_routes(app, route_mapper)
53
53
  _add_error_handlers(app)
54
54
 
55
+ if route_mapper.base_path:
56
+ app.wsgi_app = _strip_base_path(app.wsgi_app, route_mapper.base_path)
57
+
55
58
  return app
56
59
 
57
60
 
58
61
  #
59
62
  # INTERNAL
60
63
  #
64
+ def _strip_base_path(wsgi_app: Callable, base_path: str) -> Callable:
65
+ """Wrap a WSGI app so it is also served under ``settings.base_path``.
66
+
67
+ Args:
68
+ wsgi_app: The Flask WSGI app.
69
+ base_path: Normalized prefix, e.g. "/dock".
70
+
71
+ Returns:
72
+ WSGI app that routes ``/dock/...`` as ``/...`` (other paths unchanged).
73
+ """
74
+ def middleware(environ: Dict[str, Any], start_response: Callable) -> Any:
75
+ """Strip the base path from PATH_INFO, then call the Flask app."""
76
+ path = environ.get("PATH_INFO", "")
77
+ stripped = strip_base_path(path, base_path)
78
+ if stripped != path:
79
+ environ["PATH_INFO"] = stripped
80
+ return wsgi_app(environ, start_response)
81
+ return middleware
82
+
83
+
61
84
  def _add_main_routes(app: Flask, route_mapper: RouteMapper) -> None:
62
85
  """Add main API routes to the Flask app.
63
86
 
@@ -11,7 +11,10 @@ License: BSD 3-Clause
11
11
  #
12
12
  # IMPORTS
13
13
  #
14
+ import asyncio
14
15
  import json
16
+ import re
17
+ import threading
15
18
  import httpx
16
19
  import yaml
17
20
  from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
@@ -54,9 +57,77 @@ HOP_BY_HOP_HEADERS: frozenset = frozenset({
54
57
  })
55
58
 
56
59
 
60
+ # Request headers never forwarded to an upstream remote. Host must be the
61
+ # upstream's own (httpx sets it): forwarding the client's Host makes the upstream
62
+ # build redirects and absolute URLs that point back at the proxy. The rest are
63
+ # hop-by-hop headers, or (content-length) recomputed by httpx from the body.
64
+ EXCLUDED_REQUEST_HEADERS: frozenset = frozenset({
65
+ "connection",
66
+ "content-length",
67
+ "host",
68
+ "keep-alive",
69
+ "proxy-authorization",
70
+ "te",
71
+ "trailers",
72
+ "transfer-encoding",
73
+ "upgrade",
74
+ })
75
+
76
+ # `settings.duckdb` options. Every key is applied to each query's DuckDB
77
+ # connection as `SET <key> = <value>` (memory_limit, threads, temp_directory,
78
+ # ...), except max_concurrent_queries, which caps how many database queries run
79
+ # at once in this process (others wait their turn).
80
+ DUCKDB_SETTINGS_KEY: str = "duckdb"
81
+ MAX_CONCURRENT_QUERIES_KEY: str = "max_concurrent_queries"
82
+
83
+ # `settings.base_path`: an optional URL prefix (e.g. "/dock") the API is also
84
+ # served under, for when a proxy/CDN forwards a path prefix unchanged.
85
+ BASE_PATH_KEY: str = "base_path"
86
+
87
+ DUCKDB_OPTION_PATTERN: re.Pattern = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
88
+
89
+
57
90
  #
58
91
  # PUBLIC
59
92
  #
93
+ def normalize_base_path(base_path: Any) -> Optional[str]:
94
+ """Normalize a ``base_path`` setting to ``/prefix`` form.
95
+
96
+ Args:
97
+ base_path: Configured value (e.g. "dock", "/dock/"), or None/empty.
98
+
99
+ Returns:
100
+ "/prefix" without a trailing slash, or None if no prefix is set.
101
+ """
102
+ if not base_path:
103
+ return None
104
+ stripped = str(base_path).strip().strip("/")
105
+ return f"/{stripped}" if stripped else None
106
+
107
+
108
+ def strip_base_path(path: str, base_path: Optional[str]) -> str:
109
+ """Remove a base path prefix from a request path, if present.
110
+
111
+ Paths without the prefix are returned unchanged, so the API answers both
112
+ with and without it (e.g. direct calls and health checks still work).
113
+
114
+ Args:
115
+ path: Request path, e.g. "/dock/birdnet/latest/detections/".
116
+ base_path: Normalized prefix (see normalize_base_path), or None.
117
+
118
+ Returns:
119
+ The path without the prefix ("/birdnet/latest/detections/"), "/" for
120
+ the prefix itself, or the original path.
121
+ """
122
+ if not base_path:
123
+ return path
124
+ if path == base_path or path == f"{base_path}/":
125
+ return "/"
126
+ if path.startswith(f"{base_path}/"):
127
+ return path[len(base_path):]
128
+ return path
129
+
130
+
60
131
  def collect_multi_query_params(items: Iterable[Tuple[str, str]]) -> Dict[str, List[str]]:
61
132
  """Group repeated query string pairs into a name-to-value-list mapping.
62
133
 
@@ -100,6 +171,11 @@ class RouteMapper:
100
171
  self.database_names = get_database_names(self.config)
101
172
  self.settings = get_settings(self.config)
102
173
  self.listing_specs, self.listing_warnings = resolve_listing_specs(self.config)
174
+ self.base_path = normalize_base_path(self.settings.get(BASE_PATH_KEY))
175
+ self.duckdb_statements, max_queries = _duckdb_settings(
176
+ self.settings.get(DUCKDB_SETTINGS_KEY)
177
+ )
178
+ self._query_slots = threading.BoundedSemaphore(max_queries) if max_queries else None
103
179
 
104
180
  def get_config_metadata(self) -> Dict[str, Any]:
105
181
  """Get API metadata from configuration.
@@ -251,7 +327,7 @@ class RouteMapper:
251
327
  return PreparedRequest(
252
328
  url=full_url,
253
329
  method=method,
254
- headers=headers or {},
330
+ headers=_filter_request_headers(headers or {}),
255
331
  params=filtered_query_params,
256
332
  cookies=filtered_cookies,
257
333
  body=body,
@@ -492,35 +568,22 @@ class RouteMapper:
492
568
  except (ValueError, yaml.YAMLError):
493
569
  return _error_response(500, "SQL query error")
494
570
 
495
- try:
496
- import duckdb
497
-
498
- conn = duckdb.connect(database=':memory:')
499
-
500
- # Authenticate every local table (as before) plus any shared tables
501
- # the query references, then expose [[schema.table]] refs as views.
502
- auth_tables = get_local_table_references(database_config, shared_config)
503
- local_names = {table.sql_name for table in auth_tables}
504
- auth_tables += [ref for ref in table_refs if ref.sql_name not in local_names]
505
- setup_table_storage_authentication(conn, auth_tables)
506
- for statement in build_schema_view_statements(table_refs):
507
- conn.execute(statement)
508
-
509
- result = conn.execute(sql_query).fetchall()
510
- columns = [desc[0] for desc in conn.description] if conn.description else []
511
- conn.close()
512
-
513
- response_data = []
514
- for row in result:
515
- row_dict = {}
516
- for col, val in zip(columns, row):
517
- row_dict[col] = _make_json_safe(val)
518
- response_data.append(row_dict)
519
-
520
- return _json_response(response_data)
571
+ # Authenticate every local table plus any shared tables the query
572
+ # references, then expose [[schema.table]] refs as views.
573
+ auth_tables = get_local_table_references(database_config, shared_config)
574
+ local_names = {table.sql_name for table in auth_tables}
575
+ auth_tables += [ref for ref in table_refs if ref.sql_name not in local_names]
521
576
 
577
+ try:
578
+ # DuckDB calls block; run them in a worker thread so one slow query
579
+ # doesn't stall every other request (and health checks) meanwhile.
580
+ response_data = await asyncio.to_thread(
581
+ self._run_query, sql_query, auth_tables,
582
+ build_schema_view_statements(table_refs),
583
+ )
522
584
  except Exception:
523
585
  return _error_response(500, "Database query error")
586
+ return _json_response(response_data)
524
587
 
525
588
  def is_remote_name(self, name: str) -> bool:
526
589
  """Check if a given name is a configured remote name.
@@ -598,6 +661,51 @@ class RouteMapper:
598
661
  except Exception as e:
599
662
  return _error_response(500, f"Sync wrapper error: {str(e)}")
600
663
 
664
+ def _run_query(
665
+ self,
666
+ sql_query: str,
667
+ auth_tables: List[Any],
668
+ view_statements: List[str]) -> List[Dict[str, Any]]:
669
+ """Execute a database query on a fresh DuckDB connection (blocking).
670
+
671
+ Applies the ``settings.duckdb`` options, storage authentication and
672
+ schema views, then runs the query. Honors max_concurrent_queries.
673
+
674
+ Args:
675
+ sql_query: The SQL to run.
676
+ auth_tables: TableReferences to set up storage authentication for.
677
+ view_statements: CREATE SCHEMA/VIEW statements to run first.
678
+
679
+ Returns:
680
+ Result rows as JSON-safe dicts.
681
+ """
682
+ import duckdb
683
+
684
+ # getattr: RouteMappers built without __init__ (e.g. in tests) have neither.
685
+ slots = getattr(self, '_query_slots', None)
686
+ if slots is not None:
687
+ slots.acquire()
688
+ try:
689
+ conn = duckdb.connect(database=':memory:')
690
+ try:
691
+ for statement in getattr(self, 'duckdb_statements', []):
692
+ conn.execute(statement)
693
+ setup_table_storage_authentication(conn, auth_tables)
694
+ for statement in view_statements:
695
+ conn.execute(statement)
696
+ result = conn.execute(sql_query).fetchall()
697
+ columns = [desc[0] for desc in conn.description] if conn.description else []
698
+ finally:
699
+ conn.close()
700
+ finally:
701
+ if slots is not None:
702
+ slots.release()
703
+
704
+ return [
705
+ {column: _make_json_safe(value) for column, value in zip(columns, row)}
706
+ for row in result
707
+ ]
708
+
601
709
  def _is_remote_filename(self, filename: str) -> bool:
602
710
  """Check if a filename corresponds to a remote config file.
603
711
 
@@ -634,6 +742,62 @@ class RouteMapper:
634
742
  #
635
743
  # INTERNAL
636
744
  #
745
+ def _duckdb_settings(options: Any) -> Tuple[List[str], Optional[int]]:
746
+ """Turn ``settings.duckdb`` into SET statements and a concurrency cap.
747
+
748
+ Args:
749
+ options: Mapping of DuckDB option -> value, plus optional
750
+ max_concurrent_queries; or None.
751
+
752
+ Returns:
753
+ Tuple of (SET statements, max concurrent queries or None).
754
+
755
+ Raises:
756
+ ValueError: If options isn't a mapping, an option name isn't a plain
757
+ identifier, or max_concurrent_queries isn't a positive integer.
758
+ """
759
+ if not options:
760
+ return ([], None)
761
+ if not isinstance(options, dict):
762
+ raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY} must be a mapping")
763
+
764
+ max_queries = options.get(MAX_CONCURRENT_QUERIES_KEY)
765
+ if max_queries is not None and (not isinstance(max_queries, int) or max_queries < 1):
766
+ raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY}.{MAX_CONCURRENT_QUERIES_KEY} "
767
+ "must be a positive integer")
768
+
769
+ statements = []
770
+ for name, value in options.items():
771
+ if name == MAX_CONCURRENT_QUERIES_KEY:
772
+ continue
773
+ if not DUCKDB_OPTION_PATTERN.match(str(name)):
774
+ raise ValueError(f"settings.{DUCKDB_SETTINGS_KEY}: invalid option name '{name}'")
775
+ if isinstance(value, bool):
776
+ literal = "true" if value else "false"
777
+ elif isinstance(value, (int, float)):
778
+ literal = str(value)
779
+ else:
780
+ literal = "'" + str(value).replace("'", "''") + "'"
781
+ statements.append(f"SET {name} = {literal}")
782
+ return (statements, max_queries)
783
+
784
+
785
+ def _filter_request_headers(headers: Dict[str, str]) -> Dict[str, str]:
786
+ """Drop request headers that must not be forwarded upstream.
787
+
788
+ Args:
789
+ headers: Incoming client request headers.
790
+
791
+ Returns:
792
+ Headers safe to forward (see EXCLUDED_REQUEST_HEADERS).
793
+ """
794
+ return {
795
+ key: value
796
+ for key, value in headers.items()
797
+ if key.lower() not in EXCLUDED_REQUEST_HEADERS
798
+ }
799
+
800
+
637
801
  def _resolve_timeout(value: Any) -> Optional[float]:
638
802
  """Resolve the configured timeout to seconds, or None to disable it.
639
803
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: api_dock
3
- Version: 0.8.1
3
+ Version: 0.8.2
4
4
  Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
5
5
  Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
6
6
  License-Expression: BSD-3-Clause
@@ -219,19 +219,33 @@ remotes:
219
219
  settings:
220
220
  add_trailing_slash: true # Auto-add trailing slash to paths (default: true)
221
221
  follow_protocol_downgrades: false # Allow HTTPS->HTTP redirects (default: false)
222
+ follow_redirects: true # Follow remote redirects (default: true)
222
223
  timeout: 10 # Upstream request timeout in seconds (default: 10)
224
+ base_path: /dock # Also serve the API under this prefix (default: none)
225
+ duckdb: # Options for database queries (default: none)
226
+ memory_limit: 700MB
227
+ threads: 2
228
+ max_concurrent_queries: 2
223
229
  ```
224
230
 
225
- ### HTTP behavior Settings
231
+ ### Settings
226
232
 
227
- The optional `settings` section controls HTTP behavior:
233
+ The optional `settings` section controls HTTP and query behavior:
228
234
 
229
235
  - **`add_trailing_slash`** (default: `true`): Automatically append a trailing slash to all proxied paths. This prevents 307/301 redirects from remote APIs that require trailing slashes (e.g., `/projects` → `/projects/`). Set to `false` to disable this behavior.
230
236
 
231
237
  - **`follow_protocol_downgrades`** (default: `false`): Control how HTTP redirects are handled. When `false` (recommended), HTTPS→HTTP redirects are blocked for security. When `true`, allows following redirects that downgrade from HTTPS to HTTP (not recommended for production).
232
238
 
239
+ - **`follow_redirects`** (default: `true`): Whether redirects from a remote are followed by API Dock (`true`) or passed through to the client with their `Location` header (`false`). Set it to `false` when a remote answers with redirects the client should follow itself, such as presigned S3 URLs for large files.
240
+
233
241
  - **`timeout`** (default: `10`): Upstream request timeout in seconds, applied to both the streaming and buffered proxy paths. Raise it for slow upstreams (e.g. large aggregation queries) that would otherwise return a 502 on timeout. Set to `null` or `false` to disable the timeout entirely (not recommended — a stalled upstream can hold the connection open indefinitely).
234
242
 
243
+ - **`base_path`** (default: none): An extra URL prefix the API is also served under, e.g. `/dock`. Use it when a proxy or CDN forwards a path on another domain without stripping it (say CloudFront routes `https://app.example.org/dock/*` to API Dock): `/dock/birdnet/latest/detections/` is then handled as `/birdnet/latest/detections/`. Paths without the prefix keep working, so direct calls and health checks are unaffected.
244
+
245
+ - **`duckdb`** (default: none): Options for the DuckDB connection each database query runs on. Every key except `max_concurrent_queries` is applied as `SET <key> = <value>`, so any [DuckDB setting](https://duckdb.org/docs/configuration/overview) works; the useful ones on small servers are `memory_limit` (DuckDB spills to disk or fails the query instead of exceeding it), `threads`, and `temp_directory`. `max_concurrent_queries` caps how many database queries run at once in the process; further queries wait their turn. Memory limits apply per query, so on a small instance set `memory_limit × max_concurrent_queries` below the instance's memory.
246
+
247
+ Database queries run in worker threads, so a slow query doesn't hold up other requests (including health checks on `/`).
248
+
235
249
  ### Catalog Endpoints (`expose`)
236
250
 
237
251
  The optional `expose` section adds read-only endpoints that list the models and versions of your configured databases, remotes, or both ("sources"). Listings are **opt-in** — with no `expose` key nothing is added.
@@ -1601,7 +1615,7 @@ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publis
1601
1615
 
1602
1616
  ```bash
1603
1617
  # 0. Start from a clean, up-to-date main
1604
- export VERSION=0.8.1 # the NEW version, no leading "v"
1618
+ export VERSION=0.8.2 # the NEW version, no leading "v"
1605
1619
  git checkout main
1606
1620
  git pull origin main
1607
1621
  git status
@@ -1612,7 +1626,7 @@ git status
1612
1626
  pixi run -e dev pytest -q
1613
1627
 
1614
1628
  # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1615
- export COMMIT_MESSAGE='cross-schema unions, schema groups, source columns'
1629
+ export COMMIT_MESSAGE='query worker threads, duckdb settings, base_path, proxy host fix'
1616
1630
  git add -A
1617
1631
  git commit -m "v$VERSION: $COMMIT_MESSAGE"
1618
1632
  git tag "v$VERSION"
@@ -1624,17 +1638,14 @@ gh release create "v$VERSION" \
1624
1638
  --title "v$VERSION" \
1625
1639
  --notes "$(cat <<'EOF'
1626
1640
  * new features
1627
- - Cross-schema unions: `[[*.table]]` reads a table from every shared schema that has it, and `[[*!.table]]` does the same minus the current database/version's schema
1628
- - Named `schema_groups` in `databases/config.yaml`, used as `[[group.table]]` / `[[group!.table]]` (validated: known schemas only, no group/schema name clashes)
1629
- - Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
1630
- - `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
1631
- - Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
1641
+ - `settings.duckdb`: DuckDB options applied to every database query (`memory_limit`, `threads`, `temp_directory`, or any other DuckDB setting), plus `max_concurrent_queries` to cap how many queries run at once
1642
+ - `settings.base_path`: also serve the API under a URL prefix (e.g. `/dock`), for a CDN/proxy path such as CloudFront routing `https://app.example.org/dock/*` to API Dock; unprefixed paths keep working
1632
1643
  * bug fixes
1633
- - A `!` union that removes every member returns no rows (with the right columns) instead of failing, and only the schemas a union actually reads get views and storage credentials
1644
+ - Database queries now run in worker threads, so one slow query no longer stalls every other request (including health checks on `/`)
1645
+ - Remote proxying no longer forwards the client's `Host` (or hop-by-hop) headers upstream; upstream redirects (e.g. trailing-slash 307s) no longer point back at the proxy with the wrong host and path
1634
1646
  * cleanup / other improvements
1635
- - Added `SqlContext`; `build_sql_query()` / `build_sql_query_with_tables()` accept an optional `context`
1636
- - README: new "Querying across schemas" section with an overlaps example; example `databases/config.yaml` shows `schema_groups` and a union route
1637
- - Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
1647
+ - README: document `base_path`, `duckdb`, and the previously undocumented `follow_redirects` setting
1648
+ - Test suite grew from 227 to 253 tests (`test_runtime_settings.py`)
1638
1649
  EOF
1639
1650
  )"
1640
1651
 
@@ -29,6 +29,7 @@ api_dock/example_api_dock_config/remotes/example_remote.yaml
29
29
  tests/test_inject_cookies.py
30
30
  tests/test_listings.py
31
31
  tests/test_proxy_pipeline.py
32
+ tests/test_runtime_settings.py
32
33
  tests/test_schema_unions.py
33
34
  tests/test_shared_database_config.py
34
35
  tests/test_sql_builder.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "api_dock"
7
- version = "0.8.1"
7
+ version = "0.8.2"
8
8
  description = "A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases"
9
9
  readme = "README.md"
10
10
  license = "BSD-3-Clause"
@@ -0,0 +1,249 @@
1
+ """
2
+
3
+ Tests for runtime settings and request handling.
4
+
5
+ Covers ``settings.duckdb`` (per-connection DuckDB options and
6
+ max_concurrent_queries), running database queries off the event loop,
7
+ ``settings.base_path`` on the FastAPI and Flask apps, and the request headers
8
+ that are not forwarded to upstream remotes.
9
+
10
+ License: BSD 3-Clause
11
+
12
+ """
13
+
14
+ #
15
+ # IMPORTS
16
+ #
17
+ import asyncio
18
+ import json
19
+ import time
20
+ from pathlib import Path
21
+ from typing import Any, Dict, List
22
+ from unittest.mock import MagicMock, patch
23
+
24
+ import pytest
25
+ import yaml
26
+ from fastapi.testclient import TestClient
27
+
28
+ from api_dock import fast_api, flask_api
29
+ from api_dock.route_mapper import (
30
+ _duckdb_settings,
31
+ _filter_request_headers,
32
+ normalize_base_path,
33
+ RouteMapper,
34
+ strip_base_path,
35
+ )
36
+
37
+
38
+ #
39
+ # CONSTANTS
40
+ #
41
+ # A deliberately slow, dependency-free query (~0.5-2 s on one thread).
42
+ SLOW_SQL: str = "SELECT SUM(i * i) AS total FROM range(300000000) t(i)"
43
+
44
+
45
+ #
46
+ # PUBLIC
47
+ #
48
+ class TestBasePathHelpers:
49
+ """normalize_base_path / strip_base_path."""
50
+
51
+ @pytest.mark.parametrize("value,expected", [
52
+ (None, None), ("", None), ("/", None),
53
+ ("dock", "/dock"), ("/dock/", "/dock"), ("/a/b", "/a/b"),
54
+ ])
55
+ def test_normalize(self, value: Any, expected: Any) -> None:
56
+ assert normalize_base_path(value) == expected
57
+
58
+ @pytest.mark.parametrize("path,expected", [
59
+ ("/dock/birdnet/latest/detections/", "/birdnet/latest/detections/"),
60
+ ("/dock", "/"),
61
+ ("/dock/", "/"),
62
+ ("/birdnet/latest/", "/birdnet/latest/"),
63
+ ("/docks/x", "/docks/x"),
64
+ ])
65
+ def test_strip(self, path: str, expected: str) -> None:
66
+ assert strip_base_path(path, "/dock") == expected
67
+
68
+ def test_strip_without_base_path(self) -> None:
69
+ assert strip_base_path("/dock/x", None) == "/dock/x"
70
+
71
+
72
+ class TestDuckdbSettings:
73
+ """``settings.duckdb`` parsing."""
74
+
75
+ def test_statements_and_cap(self) -> None:
76
+ statements, cap = _duckdb_settings({
77
+ "memory_limit": "700MB", "threads": 2, "preserve_insertion_order": False,
78
+ "temp_directory": "/tmp/it's", "max_concurrent_queries": 3,
79
+ })
80
+ assert statements == [
81
+ "SET memory_limit = '700MB'",
82
+ "SET threads = 2",
83
+ "SET preserve_insertion_order = false",
84
+ "SET temp_directory = '/tmp/it''s'",
85
+ ]
86
+ assert cap == 3
87
+
88
+ def test_empty(self) -> None:
89
+ assert _duckdb_settings(None) == ([], None)
90
+
91
+ @pytest.mark.parametrize("options", [
92
+ {"bad name": 1}, {"max_concurrent_queries": 0}, {"max_concurrent_queries": "2"}, ["x"],
93
+ ])
94
+ def test_invalid(self, options: Any) -> None:
95
+ with pytest.raises(ValueError):
96
+ _duckdb_settings(options)
97
+
98
+
99
+ class TestRequestHeaderFilter:
100
+ """Headers not forwarded to remotes."""
101
+
102
+ def test_drops_host_and_hop_by_hop(self) -> None:
103
+ headers = {
104
+ "Host": "proxy.example.com", "Connection": "keep-alive", "Content-Length": "3",
105
+ "Accept": "application/json", "Content-Type": "application/json", "X-Custom": "1",
106
+ }
107
+ assert _filter_request_headers(headers) == {
108
+ "Accept": "application/json", "Content-Type": "application/json", "X-Custom": "1",
109
+ }
110
+
111
+ @pytest.mark.anyio
112
+ async def test_prepared_request_has_no_host(self) -> None:
113
+ mapper = RouteMapper.__new__(RouteMapper)
114
+ mapper.remote_names, mapper.database_names, mapper.settings = ["core"], [], {}
115
+ mapper.config = {"remotes": ["core"]}
116
+ with patch("api_dock.route_mapper.is_versioned_remote", return_value=False), \
117
+ patch("api_dock.route_mapper.is_route_allowed", return_value=True), \
118
+ patch("api_dock.route_mapper.find_remote_config",
119
+ return_value={"url": "https://api.example.com"}), \
120
+ patch("api_dock.route_mapper.filter_remote_query_params",
121
+ side_effect=lambda qp, *a, **kw: qp), \
122
+ patch("api_dock.route_mapper.find_route_mapping", return_value=None), \
123
+ patch("api_dock.route_mapper.filter_cookies_by_config", return_value={}):
124
+ prepared = await mapper.prepare_remote_request(
125
+ "core", "detections/", "GET", headers={"host": "proxy:8000", "accept": "*/*"},
126
+ )
127
+ assert prepared.headers == {"accept": "*/*"}
128
+
129
+
130
+ class TestQueryExecution:
131
+ """Database queries: DuckDB options, worker threads, concurrency cap."""
132
+
133
+ @pytest.mark.anyio
134
+ async def test_duckdb_options_applied(
135
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
136
+ mapper = _make_mapper(tmp_path, monkeypatch, settings={
137
+ "duckdb": {"threads": 1, "memory_limit": "300MB"},
138
+ })
139
+ result = await mapper.map_database_route("db", "settings")
140
+ row = json.loads(result.content)[0]
141
+ assert row["threads"] == "1"
142
+ assert row["memory_limit"] != _default_duckdb_setting("memory_limit")
143
+
144
+ @pytest.mark.anyio
145
+ async def test_query_does_not_block_event_loop(
146
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
147
+ mapper = _make_mapper(tmp_path, monkeypatch, settings={"duckdb": {"threads": 1}})
148
+ gaps: List[float] = []
149
+
150
+ async def ticker() -> None:
151
+ """Record the gap between event-loop wakeups (large = loop was blocked)."""
152
+ last = time.perf_counter()
153
+ while True:
154
+ await asyncio.sleep(0.01)
155
+ now = time.perf_counter()
156
+ gaps.append(now - last)
157
+ last = now
158
+
159
+ tick = asyncio.create_task(ticker())
160
+ start = time.perf_counter()
161
+ result = await mapper.map_database_route("db", "slow")
162
+ duration = time.perf_counter() - start
163
+ tick.cancel()
164
+
165
+ assert result.status_code == 200, result.content
166
+ assert duration > 0.2, 'query too fast to show blocking; raise SLOW_SQL range'
167
+ assert max(gaps) < duration / 2
168
+
169
+ @pytest.mark.anyio
170
+ async def test_concurrency_cap_acquires_and_releases(
171
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
172
+ mapper = _make_mapper(tmp_path, monkeypatch, settings={
173
+ "duckdb": {"max_concurrent_queries": 1},
174
+ })
175
+ slots = MagicMock()
176
+ mapper._query_slots = slots
177
+ await mapper.map_database_route("db", "settings")
178
+ slots.acquire.assert_called_once()
179
+ slots.release.assert_called_once()
180
+
181
+ @pytest.mark.anyio
182
+ async def test_failed_query_releases_slot(
183
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
184
+ mapper = _make_mapper(tmp_path, monkeypatch, settings={
185
+ "duckdb": {"max_concurrent_queries": 1},
186
+ })
187
+ result = await mapper.map_database_route("db", "broken")
188
+ assert result.status_code == 500
189
+ assert mapper._query_slots.acquire(blocking=False)
190
+
191
+
192
+ class TestBasePathApps:
193
+ """``settings.base_path`` on the FastAPI and Flask apps."""
194
+
195
+ def test_fastapi(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
196
+ config_path = _write_config(tmp_path, monkeypatch, settings={"base_path": "/dock"})
197
+ client = TestClient(fast_api.create_app(config_path))
198
+ for prefix in ("", "/dock"):
199
+ assert client.get(f"{prefix}/").json()["name"] == "t"
200
+ assert client.get(f"{prefix}/databases").json() == ["db"]
201
+ assert client.get(f"{prefix}/db/settings").status_code == 200
202
+ assert client.get("/docks/db/settings").status_code == 404
203
+
204
+ def test_flask(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
205
+ config_path = _write_config(tmp_path, monkeypatch, settings={"base_path": "dock"})
206
+ client = flask_api.create_app(config_path).test_client()
207
+ for prefix in ("", "/dock"):
208
+ assert client.get(f"{prefix}/").get_json()["name"] == "t"
209
+ assert client.get(f"{prefix}/databases").get_json() == ["db"]
210
+ assert client.get(f"{prefix}/db/settings").status_code == 200
211
+
212
+
213
+ #
214
+ # INTERNAL
215
+ #
216
+ def _write_config(tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
217
+ settings: Dict[str, Any]) -> str:
218
+ """Write a one-database config under tmp_path, chdir there, return its path."""
219
+ config_dir = tmp_path / "api_dock_config"
220
+ (config_dir / "databases").mkdir(parents=True)
221
+ (config_dir / "config.yaml").write_text(yaml.safe_dump({
222
+ "name": "t", "databases": ["db"], "expose": {"databases": True, "dict": False},
223
+ "settings": settings,
224
+ }))
225
+ (config_dir / "databases" / "db.yaml").write_text(yaml.safe_dump({
226
+ "name": "db",
227
+ "routes": [
228
+ {"route": "settings", "sql": "SELECT current_setting('threads')::VARCHAR AS threads, "
229
+ "current_setting('memory_limit') AS memory_limit"},
230
+ {"route": "slow", "sql": SLOW_SQL},
231
+ {"route": "broken", "sql": "SELECT * FROM no_such_table"},
232
+ ],
233
+ }))
234
+ monkeypatch.chdir(tmp_path)
235
+ return str(config_dir / "config.yaml")
236
+
237
+
238
+ def _default_duckdb_setting(name: str) -> str:
239
+ """A DuckDB setting's value on a fresh connection with no options applied."""
240
+ import duckdb
241
+
242
+ with duckdb.connect() as conn:
243
+ return str(conn.execute(f"SELECT current_setting('{name}')").fetchone()[0])
244
+
245
+
246
+ def _make_mapper(tmp_path: Path, monkeypatch: pytest.MonkeyPatch,
247
+ settings: Dict[str, Any]) -> RouteMapper:
248
+ """Build a RouteMapper over the one-database test config."""
249
+ return RouteMapper(_write_config(tmp_path, monkeypatch, settings))
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes