api-dock 0.8.0__tar.gz → 0.8.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {api_dock-0.8.0 → api_dock-0.8.1}/PKG-INFO +93 -31
  2. {api_dock-0.8.0 → api_dock-0.8.1}/README.md +92 -30
  3. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/database_config.py +130 -1
  4. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/example_api_dock_config/databases/config.yaml +12 -0
  5. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/route_mapper.py +16 -5
  6. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/sql_builder.py +251 -24
  7. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/types.py +24 -1
  8. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/PKG-INFO +93 -31
  9. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/SOURCES.txt +1 -0
  10. {api_dock-0.8.0 → api_dock-0.8.1}/pyproject.toml +1 -1
  11. api_dock-0.8.1/tests/test_schema_unions.py +308 -0
  12. {api_dock-0.8.0 → api_dock-0.8.1}/LICENSE.md +0 -0
  13. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/__init__.py +0 -0
  14. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/auth.py +0 -0
  15. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/cli.py +0 -0
  16. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/config.py +0 -0
  17. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/config_discovery.py +0 -0
  18. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/encryption.py +0 -0
  19. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/example_api_dock_config/config.yaml +0 -0
  20. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/example_api_dock_config/databases/example_db.yaml +0 -0
  21. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/example_api_dock_config/remotes/example_remote.yaml +0 -0
  22. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/fast_api.py +0 -0
  23. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/flask_api.py +0 -0
  24. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/listings.py +0 -0
  25. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock/storage_auth.py +0 -0
  26. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/dependency_links.txt +0 -0
  27. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/entry_points.txt +0 -0
  28. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/requires.txt +0 -0
  29. {api_dock-0.8.0 → api_dock-0.8.1}/api_dock.egg-info/top_level.txt +0 -0
  30. {api_dock-0.8.0 → api_dock-0.8.1}/setup.cfg +0 -0
  31. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_inject_cookies.py +0 -0
  32. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_listings.py +0 -0
  33. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_proxy_pipeline.py +0 -0
  34. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_shared_database_config.py +0 -0
  35. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_sql_builder.py +0 -0
  36. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_sql_selector.py +0 -0
  37. {api_dock-0.8.0 → api_dock-0.8.1}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: api_dock
3
- Version: 0.8.0
3
+ Version: 0.8.1
4
4
  Summary: A flexible API gateway that allows you to proxy requests to multiple remote APIs and Databases
5
5
  Author-email: Brookie Guzder-Williams <bguzder-williams@berkeley.edu>
6
6
  License-Expression: BSD-3-Clause
@@ -563,6 +563,60 @@ Notes:
563
563
  - Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
564
564
  - Views are created only for the `[[schema.table]]` tables a query actually references.
565
565
 
566
+ #### Querying across schemas (`[[*.table]]`, schema groups)
567
+
568
+ Union references read the same table from several schemas at once:
569
+
570
+ | Reference | Reads |
571
+ |---|---|
572
+ | `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
573
+ | `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
574
+ | `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
575
+ | `[[group1!.detections]]` | that group, minus the current schema |
576
+
577
+ ```yaml
578
+ # api_dock_config/databases/config.yaml
579
+ schema_groups: # named lists of shared schemas
580
+ birdnet_models:
581
+ - birdnet_2p4
582
+ - birdnet_bullfrog_2p4v0p5
583
+ ```
584
+
585
+ - A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
586
+ - `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
587
+ - Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
588
+
589
+ **Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
590
+
591
+ ```yaml
592
+ source_columns: [schema, name, version] # adds schema_name, name, version
593
+ source_columns: {schema: _schema, name: model} # pick a subset and rename
594
+ ```
595
+
596
+ A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
597
+
598
+ **`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
599
+
600
+ Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
601
+
602
+ ```yaml
603
+ routes:
604
+ - route: detections/{{id}}/overlaps
605
+ source_columns: [schema, name, version]
606
+ sql: |
607
+ WITH src AS (
608
+ SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
609
+ )
610
+ SELECT detections.*
611
+ FROM [[*.detections]] detections
612
+ JOIN src ON detections.recording_id = src.recording_id
613
+ AND detections.start_time < src.end_time
614
+ AND detections.end_time > src.start_time
615
+ WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
616
+ ```
617
+
618
+ Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
619
+
566
620
  #### Inline database configs (`slugs`)
567
621
 
568
622
  Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
@@ -1543,47 +1597,55 @@ pixi run python scripts/hello_world.py
1543
1597
 
1544
1598
  ## Publishing a Release
1545
1599
 
1600
+ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
1601
+
1546
1602
  ```bash
1547
- # 0. Make sure you are on `main` and merged with any changes
1603
+ # 0. Start from a clean, up-to-date main
1604
+ export VERSION=0.8.1 # the NEW version, no leading "v"
1605
+ git checkout main
1606
+ git pull origin main
1607
+ git status
1548
1608
 
1549
- # 1. Bump version in pyproject.toml
1609
+ # 1. Set `version` in pyproject.toml to $VERSION
1550
1610
 
1551
- # 2. Commit everything
1611
+ # 2. Run the tests
1612
+ pixi run -e dev pytest -q
1613
+
1614
+ # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1615
+ export COMMIT_MESSAGE='cross-schema unions, schema groups, source columns'
1552
1616
  git add -A
1553
- git commit -m "v0.6.1: stream proxy responses (fix large-response 502 + content-encoding)"
1554
-
1555
- # 3. Tag and push
1556
- git tag v0.6.1
1557
- git push origin main v0.6.1
1558
-
1559
- # 4. Build the wheel (requires the `dev` pixi environment)
1560
- rm -rf dist/
1561
- find . -name "__pycache__" -type d -exec rm -rf {} +
1562
- find . -name "*.pyc" -delete
1563
- pixi run -e dev python -m build --wheel
1564
- ls dist/*.whl
1565
-
1566
- # 5. Create GitHub release with the wheel attached
1567
- gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
1568
- --title "v0.6.1" --notes "$(cat <<'EOF'
1617
+ git commit -m "v$VERSION: $COMMIT_MESSAGE"
1618
+ git tag "v$VERSION"
1619
+ git push origin main "v$VERSION"
1620
+
1621
+ # 4. Publish the GitHub Release; this triggers the PyPI upload.
1622
+ # Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
1623
+ gh release create "v$VERSION" \
1624
+ --title "v$VERSION" \
1625
+ --notes "$(cat <<'EOF'
1569
1626
  * new features
1570
- - Remote proxy responses are now streamed (FastAPI) — upstream bytes are piped to the client as they arrive instead of being buffered fully in memory
1571
- - New `timeout` setting (default 10s) for the upstream request; set to `null`/`false` to disable
1627
+ - Cross-schema unions: `[[*.table]]` reads a table from every shared schema that has it, and `[[*!.table]]` does the same minus the current database/version's schema
1628
+ - Named `schema_groups` in `databases/config.yaml`, used as `[[group.table]]` / `[[group!.table]]` (validated: known schemas only, no group/schema name clashes)
1629
+ - Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
1630
+ - `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
1631
+ - Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
1572
1632
  * bug fixes
1573
- - Large upstream responses no longer return 502 — streamed via `StreamingResponse` instead of reading the whole body into memory
1574
- - `Content-Encoding` (gzip/br/deflate) is now preserved on compressed responses — raw bytes are streamed via `aiter_raw()` so the header stays valid and the client can decompress
1575
- - Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
1633
+ - A `!` union that removes every member returns no rows (with the right columns) instead of failing, and only the schemas a union actually reads get views and storage credentials
1576
1634
  * cleanup / other improvements
1577
- - Added `PreparedRequest` dataclass and split route validation/resolution into `RouteMapper.prepare_remote_request()`; the FastAPI adapter issues the streaming HTTP call
1578
- - `map_route()` (buffered) retained for the Flask/sync path
1579
- - Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
1635
+ - Added `SqlContext`; `build_sql_query()` / `build_sql_query_with_tables()` accept an optional `context`
1636
+ - README: new "Querying across schemas" section with an overlaps example; example `databases/config.yaml` shows `schema_groups` and a union route
1637
+ - Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
1580
1638
  EOF
1581
1639
  )"
1582
1640
 
1583
- # 6. Publish to PyPI
1584
- pixi run -e dev python -m twine upload dist/*.whl
1585
- ```
1641
+ # 5. Watch the publish workflow, then confirm PyPI has the new version
1642
+ gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
1643
+ curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
1586
1644
 
1645
+ # 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
1646
+ # run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
1647
+ gh pr list --repo conda-forge/api_dock-feedstock --state open
1648
+ ```
1587
1649
 
1588
1650
  ---
1589
1651
 
@@ -524,6 +524,60 @@ Notes:
524
524
  - Storage credentials are set per table. Tables whose `region`/`public` differ from the rest get their own S3 secret scoped to their path, so one query can mix regions and public/private buckets.
525
525
  - Views are created only for the `[[schema.table]]` tables a query actually references.
526
526
 
527
+ #### Querying across schemas (`[[*.table]]`, schema groups)
528
+
529
+ Union references read the same table from several schemas at once:
530
+
531
+ | Reference | Reads |
532
+ |---|---|
533
+ | `[[*.detections]]` | every shared schema that has a `detections` table, including the current one |
534
+ | `[[*!.detections]]` | the same, minus the current database/version's `schema:` |
535
+ | `[[group1.detections]]` | the schemas listed in `schema_groups.group1` |
536
+ | `[[group1!.detections]]` | that group, minus the current schema |
537
+
538
+ ```yaml
539
+ # api_dock_config/databases/config.yaml
540
+ schema_groups: # named lists of shared schemas
541
+ birdnet_models:
542
+ - birdnet_2p4
543
+ - birdnet_bullfrog_2p4v0p5
544
+ ```
545
+
546
+ - A union expands, after `FROM`/`JOIN` only, to a parenthesized `UNION ALL BY NAME` over the member schemas, so give it an alias: `FROM [[*.detections]] detections`. Columns missing from some members come back as `NULL`.
547
+ - `*` skips schemas without the table. A group whose member lacks the table, a group naming an unknown schema, and a group sharing a name with a schema are all errors. `!` applies only to `*` and groups; with no `schema:`, it removes nothing.
548
+ - Every member gets its own S3 credentials (see above), so a union can mix regions and public/private buckets.
549
+
550
+ **Source columns.** Union rows carry only the tables' real columns unless the route asks for more with `source_columns`. The available facts are `schema` (the member schema), and `name` and `version` (the database/version whose `schema:` is that schema, or `NULL` if none or several use it):
551
+
552
+ ```yaml
553
+ source_columns: [schema, name, version] # adds schema_name, name, version
554
+ source_columns: {schema: _schema, name: model} # pick a subset and rename
555
+ ```
556
+
557
+ A source column that clashes with a real column raises an error. To use one for filtering without returning it, use DuckDB's `EXCLUDE`: `SELECT detections.* EXCLUDE (schema_name) ...`.
558
+
559
+ **`{{self.*}}` placeholders.** `{{self.schema}}`, `{{self.name}}` and `{{self.version}}` are the current database/version's schema, name and version, as SQL literals (`NULL` when unknown).
560
+
561
+ Together they make an "overlaps" route that every database/version can share. It returns every detection overlapping the given one, across all schemas, except that detection itself; other overlapping rows in the same schema are kept:
562
+
563
+ ```yaml
564
+ routes:
565
+ - route: detections/{{id}}/overlaps
566
+ source_columns: [schema, name, version]
567
+ sql: |
568
+ WITH src AS (
569
+ SELECT recording_id, start_time, end_time FROM [[detections]] WHERE id = {{id}}
570
+ )
571
+ SELECT detections.*
572
+ FROM [[*.detections]] detections
573
+ JOIN src ON detections.recording_id = src.recording_id
574
+ AND detections.start_time < src.end_time
575
+ AND detections.end_time > src.start_time
576
+ WHERE NOT (detections.schema_name = {{self.schema}} AND detections.id = {{id}})
577
+ ```
578
+
579
+ Aliasing the union as `detections` also lets shared filters such as `[[detections]].confidence >= {{confidence}}` apply to the overlapping rows.
580
+
527
581
  #### Inline database configs (`slugs`)
528
582
 
529
583
  Simple database/version configs (often just a description and a `schema`) can live in the shared file instead of in their own files. Config files keep working, and the two can be mixed, even for the same database:
@@ -1504,47 +1558,55 @@ pixi run python scripts/hello_world.py
1504
1558
 
1505
1559
  ## Publishing a Release
1506
1560
 
1561
+ Publishing a GitHub Release is what publishes to PyPI: `.github/workflows/publish_to_pypi.yml` runs on `release: published`, builds the sdist and wheel with `uv build`, and uploads them using PyPI trusted publishing (OIDC). There's no local build, no API token, and no `twine`. conda-forge follows automatically: its bot opens a version PR on [conda-forge/api_dock-feedstock](https://github.com/conda-forge/api_dock-feedstock), which a maintainer merges.
1562
+
1507
1563
  ```bash
1508
- # 0. Make sure you are on `main` and merged with any changes
1564
+ # 0. Start from a clean, up-to-date main
1565
+ export VERSION=0.8.1 # the NEW version, no leading "v"
1566
+ git checkout main
1567
+ git pull origin main
1568
+ git status
1509
1569
 
1510
- # 1. Bump version in pyproject.toml
1570
+ # 1. Set `version` in pyproject.toml to $VERSION
1511
1571
 
1512
- # 2. Commit everything
1572
+ # 2. Run the tests
1573
+ pixi run -e dev pytest -q
1574
+
1575
+ # 3. Commit, tag, push (the commit command adds the "v$VERSION: " prefix)
1576
+ export COMMIT_MESSAGE='cross-schema unions, schema groups, source columns'
1513
1577
  git add -A
1514
- git commit -m "v0.6.1: stream proxy responses (fix large-response 502 + content-encoding)"
1515
-
1516
- # 3. Tag and push
1517
- git tag v0.6.1
1518
- git push origin main v0.6.1
1519
-
1520
- # 4. Build the wheel (requires the `dev` pixi environment)
1521
- rm -rf dist/
1522
- find . -name "__pycache__" -type d -exec rm -rf {} +
1523
- find . -name "*.pyc" -delete
1524
- pixi run -e dev python -m build --wheel
1525
- ls dist/*.whl
1526
-
1527
- # 5. Create GitHub release with the wheel attached
1528
- gh release create v0.6.1 dist/api_dock-0.6.1-py3-none-any.whl \
1529
- --title "v0.6.1" --notes "$(cat <<'EOF'
1578
+ git commit -m "v$VERSION: $COMMIT_MESSAGE"
1579
+ git tag "v$VERSION"
1580
+ git push origin main "v$VERSION"
1581
+
1582
+ # 4. Publish the GitHub Release; this triggers the PyPI upload.
1583
+ # Don't use --draft (the workflow only runs on a published release); no wheel needs attaching.
1584
+ gh release create "v$VERSION" \
1585
+ --title "v$VERSION" \
1586
+ --notes "$(cat <<'EOF'
1530
1587
  * new features
1531
- - Remote proxy responses are now streamed (FastAPI) — upstream bytes are piped to the client as they arrive instead of being buffered fully in memory
1532
- - New `timeout` setting (default 10s) for the upstream request; set to `null`/`false` to disable
1588
+ - Cross-schema unions: `[[*.table]]` reads a table from every shared schema that has it, and `[[*!.table]]` does the same minus the current database/version's schema
1589
+ - Named `schema_groups` in `databases/config.yaml`, used as `[[group.table]]` / `[[group!.table]]` (validated: known schemas only, no group/schema name clashes)
1590
+ - Route `source_columns` adds where each union row came from (`schema`, `name`, `version`), with default names `schema_name`/`name`/`version` or your own; nothing is added by default
1591
+ - `{{self.schema}}`, `{{self.name}}`, and `{{self.version}}` placeholders for the database/version being queried
1592
+ - Together these support an "overlaps" route shared by every database/version (every detection overlapping a given one across all schemas, except that detection itself); shared query params such as `confidence`, `sort`, and `limit` apply to its rows
1533
1593
  * bug fixes
1534
- - Large upstream responses no longer return 502 — streamed via `StreamingResponse` instead of reading the whole body into memory
1535
- - `Content-Encoding` (gzip/br/deflate) is now preserved on compressed responses — raw bytes are streamed via `aiter_raw()` so the header stays valid and the client can decompress
1536
- - Slow upstreams (e.g. large aggregation queries) no longer 502 at httpx's hardcoded 5s default — the timeout is now configurable via the `timeout` setting
1594
+ - A `!` union that removes every member returns no rows (with the right columns) instead of failing, and only the schemas a union actually reads get views and storage credentials
1537
1595
  * cleanup / other improvements
1538
- - Added `PreparedRequest` dataclass and split route validation/resolution into `RouteMapper.prepare_remote_request()`; the FastAPI adapter issues the streaming HTTP call
1539
- - `map_route()` (buffered) retained for the Flask/sync path
1540
- - Added streaming test coverage (`TestStreamUpstream`, plus `prepare_remote_request` and streaming-header tests) — 53 tests total
1596
+ - Added `SqlContext`; `build_sql_query()` / `build_sql_query_with_tables()` accept an optional `context`
1597
+ - README: new "Querying across schemas" section with an overlaps example; example `databases/config.yaml` shows `schema_groups` and a union route
1598
+ - Test suite grew from 198 to 227 tests (`test_schema_unions.py`, including a real DuckDB end-to-end overlaps test)
1541
1599
  EOF
1542
1600
  )"
1543
1601
 
1544
- # 6. Publish to PyPI
1545
- pixi run -e dev python -m twine upload dist/*.whl
1546
- ```
1602
+ # 5. Watch the publish workflow, then confirm PyPI has the new version
1603
+ gh run watch "$(gh run list --workflow=publish_to_pypi.yml -L1 --json databaseId -q '.[0].databaseId')" --repo SchmidtDSE/api_dock
1604
+ curl -s https://pypi.org/pypi/api-dock/json | python3 -c "import sys,json; print('PyPI latest:', json.load(sys.stdin)['info']['version'])"
1547
1605
 
1606
+ # 6. conda-forge: once the bot opens the v$VERSION PR (usually within hours), check that the recipe's
1607
+ # run requirements match pyproject.toml dependencies (the bot only bumps version + sha256), then merge it
1608
+ gh pr list --repo conda-forge/api_dock-feedstock --state open
1609
+ ```
1548
1610
 
1549
1611
  ---
1550
1612
 
@@ -56,6 +56,15 @@ SLUG_NAME_KEY: str = "name"
56
56
  SLUG_VERSION_KEY: str = "version"
57
57
  SLUG_VERSIONS_KEY: str = "versions"
58
58
 
59
+ # Named groups of shared schemas: {group: [schema, ...]}, referenced in SQL as
60
+ # [[group.table]]. Group names may not collide with schema names.
61
+ SCHEMA_GROUPS_KEY: str = "schema_groups"
62
+
63
+ # Union selectors: [[*.table]] = every shared schema with that table; a trailing
64
+ # "!" ([[*!.table]], [[group!.table]]) drops the current version's schema.
65
+ ALL_SCHEMAS: str = "*"
66
+ EXCLUDE_SELF_SUFFIX: str = "!"
67
+
59
68
  # Exclusion version wildcard: matches every version (and unversioned databases).
60
69
  ALL_VERSIONS: str = "*"
61
70
 
@@ -212,7 +221,7 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
212
221
  """Load the whole shared database config file (``databases/config.yaml``).
213
222
 
214
223
  Top-level keys: ``database`` (tables, ``meta``, ``schema``), ``slugs``
215
- (inline database/version configs), ``routes`` and
224
+ (inline database/version configs), ``schema_groups``, ``routes`` and
216
225
  ``query_params`` (added to every database/version), and the
217
226
  ``route_inclusions`` / ``query_inclusions`` and ``route_exclusions`` /
218
227
  ``query_exclusions`` lists.
@@ -250,12 +259,15 @@ def load_shared_config(config_dir: Optional[str] = None) -> Dict[str, Any]:
250
259
  ROUTE_EXCLUSIONS_KEY: list,
251
260
  QUERY_EXCLUSIONS_KEY: list,
252
261
  SLUGS_KEY: list,
262
+ SCHEMA_GROUPS_KEY: dict,
253
263
  }
254
264
  for key, expected in expected_types.items():
255
265
  value = contents.get(key) or expected()
256
266
  if not isinstance(value, expected):
257
267
  raise ValueError(f"'{key}' in {shared_path} must be a {expected.__name__}")
258
268
  normalized[key] = value
269
+
270
+ _validate_schema_groups(normalized, shared_path)
259
271
  return normalized
260
272
 
261
273
 
@@ -418,6 +430,92 @@ def resolve_table_reference(
418
430
  return None
419
431
 
420
432
 
433
+ def resolve_schema_union(
434
+ selector: str,
435
+ table_name: str,
436
+ shared_config: Optional[Dict[str, Any]] = None,
437
+ schema_groups: Optional[Dict[str, List[str]]] = None) -> Optional[List[TableReference]]:
438
+ """Resolve a union selector (``*`` or a schema group) to its member tables.
439
+
440
+ Args:
441
+ selector: ``*`` for every shared schema, or a ``schema_groups`` name
442
+ (without any trailing ``!``).
443
+ table_name: Table to read from each member schema.
444
+ shared_config: The shared ``database`` mapping, or None.
445
+ schema_groups: The shared ``schema_groups`` mapping, or None.
446
+
447
+ Returns:
448
+ Qualified TableReferences in schema/group order, or None if the
449
+ selector is neither ``*`` nor a group (i.e. it names a single schema).
450
+
451
+ Raises:
452
+ ValueError: If no schema has the table (``*``), or a group member is
453
+ not a schema or lacks the table.
454
+ """
455
+ schemas = (shared_config or {}).get(SHARED_SCHEMA_KEY) or {}
456
+ groups = schema_groups or {}
457
+
458
+ if selector == ALL_SCHEMAS:
459
+ members = [name for name, tables in schemas.items() if table_name in (tables or {})]
460
+ if not members:
461
+ raise ValueError(f"No shared schema has a table named '{table_name}'")
462
+ elif selector in groups:
463
+ members = list(groups[selector])
464
+ for schema_name in members:
465
+ if table_name not in (schemas.get(schema_name) or {}):
466
+ raise ValueError(
467
+ f"Schema '{schema_name}' in group '{selector}' has no table '{table_name}'"
468
+ )
469
+ else:
470
+ return None
471
+
472
+ references = []
473
+ for schema_name in members:
474
+ reference = resolve_table_reference(
475
+ f"{schema_name}{SCHEMA_SEPARATOR}{table_name}", {}, shared_config
476
+ )
477
+ if reference is None:
478
+ raise ValueError(f"Table '{schema_name}.{table_name}' not found")
479
+ references.append(reference)
480
+ return references
481
+
482
+
483
+ def get_schema_sources(
484
+ database_names: List[str],
485
+ config_dir: Optional[str] = None) -> Dict[str, Tuple[str, Optional[str]]]:
486
+ """Map each shared schema to the database/version that uses it.
487
+
488
+ Args:
489
+ database_names: Served database names (from the main config).
490
+ config_dir: Base config directory. If None, uses default.
491
+
492
+ Returns:
493
+ Schema name -> (database name, version or None). Schemas used by more
494
+ than one database/version are left out (their source is ambiguous).
495
+ """
496
+ sources: Dict[str, Tuple[str, Optional[str]]] = {}
497
+ ambiguous = set()
498
+
499
+ for database_name in database_names:
500
+ if is_versioned_database(database_name, config_dir):
501
+ versions: List[Optional[str]] = list(get_database_versions(database_name, config_dir))
502
+ else:
503
+ versions = [None]
504
+ for version in versions:
505
+ try:
506
+ config = load_database_config(database_name, config_dir, version)
507
+ except FileNotFoundError:
508
+ continue
509
+ schema_name = config.get(DATABASE_SCHEMA_KEY) if isinstance(config, dict) else None
510
+ if not schema_name:
511
+ continue
512
+ if schema_name in sources:
513
+ ambiguous.add(schema_name)
514
+ sources[schema_name] = (database_name, version)
515
+
516
+ return {k: v for k, v in sources.items() if k not in ambiguous}
517
+
518
+
421
519
  def get_local_table_references(
422
520
  database_config: Dict[str, Any],
423
521
  shared_config: Optional[Dict[str, Any]] = None) -> List[TableReference]:
@@ -789,6 +887,37 @@ def _load_yaml_file(file_path: str) -> Dict[str, Any]:
789
887
  raise yaml.YAMLError(f"Invalid YAML in {file_path}: {e}")
790
888
 
791
889
 
890
+ def _validate_schema_groups(shared_file: Dict[str, Any], shared_path: str) -> None:
891
+ """Validate the shared ``schema_groups`` mapping.
892
+
893
+ Args:
894
+ shared_file: The normalized shared config.
895
+ shared_path: Path of the shared config (for error messages).
896
+
897
+ Raises:
898
+ ValueError: If a group name isn't a plain identifier or collides with a
899
+ schema name, or a group isn't a non-empty list of known schemas.
900
+ """
901
+ schemas = (shared_file.get(SHARED_CONFIG_KEY) or {}).get(SHARED_SCHEMA_KEY) or {}
902
+ for group, members in (shared_file.get(SCHEMA_GROUPS_KEY) or {}).items():
903
+ if not IDENTIFIER_PATTERN.match(str(group)):
904
+ raise ValueError(f"{SCHEMA_GROUPS_KEY}: invalid group name '{group}' in {shared_path}")
905
+ if group in schemas:
906
+ raise ValueError(
907
+ f"{SCHEMA_GROUPS_KEY}: '{group}' is also a schema name in {shared_path}"
908
+ )
909
+ if not isinstance(members, list) or not members:
910
+ raise ValueError(
911
+ f"{SCHEMA_GROUPS_KEY}: '{group}' must be a non-empty list in {shared_path}"
912
+ )
913
+ for member in members:
914
+ if member not in schemas:
915
+ raise ValueError(
916
+ f"{SCHEMA_GROUPS_KEY}: '{group}' names unknown schema '{member}' "
917
+ f"in {shared_path}"
918
+ )
919
+
920
+
792
921
  def _is_selected(
793
922
  inclusions: Any,
794
923
  exclusions: Any,
@@ -36,6 +36,11 @@
36
36
  # items:
37
37
  # uri: s3://your-bucket/v2/items/**/*.parquet
38
38
  #
39
+ # # named groups of schemas: [[all_items.items]] reads both; [[all_items!.items]]
40
+ # # skips the current version's schema; [[*.items]] reads every schema
41
+ # schema_groups:
42
+ # all_items: [items_v1, items_v2]
43
+ #
39
44
  # # database/version configs defined inline instead of as files (list them under
40
45
  # # `databases:` in the main config like any other database)
41
46
  # slugs:
@@ -68,6 +73,13 @@
68
73
  # - route: items/
69
74
  # include: ['example_db'] # ONLY add this route to these slug/versions
70
75
  # sql: SELECT [[items]].id FROM [[items]]
76
+ # # every items row in the same group as this one, from any schema, tagged with its source
77
+ # - route: items/{{id}}/related
78
+ # source_columns: [schema, name, version]
79
+ # sql: >
80
+ # SELECT items.* FROM [[*.items]] items
81
+ # WHERE items.group_id = (SELECT group_id FROM [[items]] WHERE id = {{id}})
82
+ # AND NOT (items.schema_name = {{self.schema}} AND items.id = {{id}})
71
83
  #
72
84
  # query_params:
73
85
  # - limit:
@@ -18,11 +18,11 @@ from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
18
18
 
19
19
  from api_dock.auth import validate_authentication
20
20
  from api_dock.config import filter_cookies_by_config, filter_remote_query_params, find_remote_config, find_route_mapping, get_authentication_config, get_database_names, get_remote_names, get_remote_versions, get_settings, is_route_allowed, is_versioned_remote, load_main_config, merge_inherited_config, resolve_latest_version
21
- from api_dock.database_config import apply_shared_definitions, find_database_route, get_database_versions, get_local_table_references, is_versioned_database, load_database_config, load_shared_config, merge_query_params, resolve_latest_database_version, SHARED_CONFIG_KEY
21
+ from api_dock.database_config import apply_shared_definitions, find_database_route, get_database_versions, get_local_table_references, get_schema_sources, is_versioned_database, load_database_config, load_shared_config, merge_query_params, resolve_latest_database_version, SCHEMA_GROUPS_KEY, SHARED_CONFIG_KEY
22
22
  from api_dock.listings import build_listing, resolve_listing_specs
23
- from api_dock.sql_builder import build_schema_view_statements, build_sql_query_with_tables, extract_path_parameters, process_query_parameters, SqlSelectionError
23
+ from api_dock.sql_builder import build_schema_view_statements, build_sql_query_with_tables, extract_path_parameters, process_query_parameters, SOURCE_COLUMNS_KEY, SqlSelectionError
24
24
  from api_dock.storage_auth import setup_table_storage_authentication
25
- from api_dock.types import PreparedRequest, ProxyResponse
25
+ from api_dock.types import PreparedRequest, ProxyResponse, SqlContext
26
26
 
27
27
 
28
28
  #
@@ -467,9 +467,20 @@ class RouteMapper:
467
467
  return _error_response(500, "Query parameter processing error")
468
468
 
469
469
  try:
470
+ # Schema -> name/version lookups load every database config, so only
471
+ # do them when the route asks for source columns.
472
+ context = SqlContext(
473
+ name=database_name,
474
+ version=version,
475
+ schema_groups=shared_file.get(SCHEMA_GROUPS_KEY) or {},
476
+ schema_sources=(
477
+ get_schema_sources(self.database_names)
478
+ if route_config.get(SOURCE_COLUMNS_KEY) else {}
479
+ ),
480
+ )
470
481
  sql_query, table_refs = build_sql_query_with_tables(
471
482
  route_config, database_config, path_params, query_params,
472
- filtered_cookies, multi_query_params, shared_config
483
+ filtered_cookies, multi_query_params, shared_config, context
473
484
  )
474
485
  except SqlSelectionError as e:
475
486
  return ProxyResponse(
@@ -478,7 +489,7 @@ class RouteMapper:
478
489
  content_type="application/json",
479
490
  error_message=str(e.response.get("error")) if e.response.get("error") else None,
480
491
  )
481
- except ValueError:
492
+ except (ValueError, yaml.YAMLError):
482
493
  return _error_response(500, "SQL query error")
483
494
 
484
495
  try: