pyoq-sql 1.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyoq/__init__.py +10 -0
- pyoq/__main__.py +5 -0
- pyoq/_native.pyi +5 -0
- pyoq/cli/__init__.py +5 -0
- pyoq/cli/commands.py +270 -0
- pyoq/cli/defaults.py +98 -0
- pyoq/cli/services.py +97 -0
- pyoq/config/__init__.py +31 -0
- pyoq/config/connection.py +161 -0
- pyoq/config/loader.py +289 -0
- pyoq/config/models.py +245 -0
- pyoq/config/values.py +142 -0
- pyoq/descriptors.py +165 -0
- pyoq/diagnostics/__init__.py +68 -0
- pyoq/diagnostics/budget.py +136 -0
- pyoq/diagnostics/events.py +137 -0
- pyoq/diagnostics/fingerprint.py +267 -0
- pyoq/diagnostics/instrumented.py +237 -0
- pyoq/diagnostics/metrics.py +61 -0
- pyoq/diagnostics/observation.py +227 -0
- pyoq/diagnostics/scoped.py +103 -0
- pyoq/django/__init__.py +15 -0
- pyoq/django/apps.py +17 -0
- pyoq/django/execution.py +317 -0
- pyoq/django/generation.py +59 -0
- pyoq/django/management/__init__.py +0 -0
- pyoq/django/management/commands/__init__.py +0 -0
- pyoq/django/management/commands/makemigrations.py +53 -0
- pyoq/django/management/commands/pyoq_codegen.py +75 -0
- pyoq/django/parameters.py +101 -0
- pyoq/django/schema.py +379 -0
- pyoq/django/settings.py +87 -0
- pyoq/django/timeouts.py +105 -0
- pyoq/dsl/__init__.py +64 -0
- pyoq/dsl/aio/__init__.py +31 -0
- pyoq/dsl/aio/context.py +295 -0
- pyoq/dsl/aio/queries.py +335 -0
- pyoq/dsl/aio/writes.py +368 -0
- pyoq/dsl/context.py +326 -0
- pyoq/dsl/entry.py +37 -0
- pyoq/dsl/labels.py +36 -0
- pyoq/dsl/queries.py +339 -0
- pyoq/dsl/result.py +164 -0
- pyoq/dsl/writes.py +360 -0
- pyoq/errors.py +317 -0
- pyoq/fastapi/__init__.py +32 -0
- pyoq/fastapi/dependencies.py +167 -0
- pyoq/fastapi/lifespan.py +119 -0
- pyoq/fetching/__init__.py +55 -0
- pyoq/fetching/collections.py +136 -0
- pyoq/fetching/execution.py +587 -0
- pyoq/fetching/joined.py +79 -0
- pyoq/fetching/nesting.py +183 -0
- pyoq/fetching/plans.py +541 -0
- pyoq/fetching/select_in.py +149 -0
- pyoq/fetching/tables.py +110 -0
- pyoq/generation/__init__.py +54 -0
- pyoq/generation/cleanup.py +44 -0
- pyoq/generation/contracts.py +248 -0
- pyoq/generation/drift.py +169 -0
- pyoq/generation/lock.py +33 -0
- pyoq/generation/manifest.py +114 -0
- pyoq/generation/model.py +1001 -0
- pyoq/generation/pipeline.py +119 -0
- pyoq/generation/rendering/__init__.py +5 -0
- pyoq/generation/rendering/domains.py +51 -0
- pyoq/generation/rendering/enums.py +29 -0
- pyoq/generation/rendering/exports.py +70 -0
- pyoq/generation/rendering/imports.py +63 -0
- pyoq/generation/rendering/package.py +56 -0
- pyoq/generation/rendering/relations.py +133 -0
- pyoq/generation/rendering/routines.py +396 -0
- pyoq/generation/rendering/rows.py +79 -0
- pyoq/generation/rendering/source.py +121 -0
- pyoq/generation/rendering/tables.py +300 -0
- pyoq/generation/rendering/writes.py +514 -0
- pyoq/generation/validation.py +27 -0
- pyoq/generation/writer.py +184 -0
- pyoq/hydration/__init__.py +24 -0
- pyoq/hydration/engine.py +155 -0
- pyoq/hydration/identity.py +194 -0
- pyoq/hydration/plan.py +116 -0
- pyoq/migrations/__init__.py +9 -0
- pyoq/migrations/alembic.py +106 -0
- pyoq/migrations/hooks.py +75 -0
- pyoq/naming.py +261 -0
- pyoq/policies/__init__.py +47 -0
- pyoq/policies/bypass.py +122 -0
- pyoq/policies/governed.py +430 -0
- pyoq/policies/model.py +242 -0
- pyoq/policies/rewriting.py +263 -0
- pyoq/py.typed +1 -0
- pyoq/query/__init__.py +312 -0
- pyoq/query/aggregates.py +172 -0
- pyoq/query/arrays.py +65 -0
- pyoq/query/binding.py +52 -0
- pyoq/query/capabilities.py +317 -0
- pyoq/query/casts.py +73 -0
- pyoq/query/choices.py +185 -0
- pyoq/query/decoding.py +360 -0
- pyoq/query/documents.py +56 -0
- pyoq/query/execution/__init__.py +63 -0
- pyoq/query/execution/aio/__init__.py +31 -0
- pyoq/query/execution/aio/operations.py +228 -0
- pyoq/query/execution/aio/pooling.py +233 -0
- pyoq/query/execution/aio/streaming.py +161 -0
- pyoq/query/execution/aio/transactions.py +105 -0
- pyoq/query/execution/batch.py +96 -0
- pyoq/query/execution/binding_style.py +30 -0
- pyoq/query/execution/compilation.py +48 -0
- pyoq/query/execution/context.py +61 -0
- pyoq/query/execution/control.py +50 -0
- pyoq/query/execution/operations.py +224 -0
- pyoq/query/execution/planning.py +107 -0
- pyoq/query/execution/pooling.py +279 -0
- pyoq/query/execution/results.py +36 -0
- pyoq/query/execution/streaming.py +178 -0
- pyoq/query/execution/transactions.py +95 -0
- pyoq/query/expressions.py +1200 -0
- pyoq/query/fields.py +60 -0
- pyoq/query/mysql/__init__.py +59 -0
- pyoq/query/mysql/aio/__init__.py +38 -0
- pyoq/query/mysql/aio/commands.py +389 -0
- pyoq/query/mysql/aio/driver.py +196 -0
- pyoq/query/mysql/aio/executor.py +123 -0
- pyoq/query/mysql/aio/factory.py +26 -0
- pyoq/query/mysql/aio/operations.py +38 -0
- pyoq/query/mysql/aio/pool.py +53 -0
- pyoq/query/mysql/aio/transactions.py +313 -0
- pyoq/query/mysql/commands.py +354 -0
- pyoq/query/mysql/compiler.py +134 -0
- pyoq/query/mysql/context.py +20 -0
- pyoq/query/mysql/executor.py +126 -0
- pyoq/query/mysql/expressions.py +244 -0
- pyoq/query/mysql/factory.py +46 -0
- pyoq/query/mysql/health.py +66 -0
- pyoq/query/mysql/identifiers.py +9 -0
- pyoq/query/mysql/model.py +79 -0
- pyoq/query/mysql/operations.py +43 -0
- pyoq/query/mysql/parameters.py +69 -0
- pyoq/query/mysql/planning.py +20 -0
- pyoq/query/mysql/pool.py +67 -0
- pyoq/query/mysql/transactions.py +331 -0
- pyoq/query/mysql/writes.py +73 -0
- pyoq/query/nodes.py +750 -0
- pyoq/query/postgres/__init__.py +48 -0
- pyoq/query/postgres/aio/__init__.py +25 -0
- pyoq/query/postgres/aio/bulk.py +56 -0
- pyoq/query/postgres/aio/commands.py +264 -0
- pyoq/query/postgres/aio/executor.py +152 -0
- pyoq/query/postgres/aio/factory.py +26 -0
- pyoq/query/postgres/aio/operations.py +26 -0
- pyoq/query/postgres/aio/pool.py +40 -0
- pyoq/query/postgres/aio/transactions.py +295 -0
- pyoq/query/postgres/bulk.py +62 -0
- pyoq/query/postgres/commands.py +238 -0
- pyoq/query/postgres/compiler.py +114 -0
- pyoq/query/postgres/context.py +20 -0
- pyoq/query/postgres/executor.py +147 -0
- pyoq/query/postgres/expressions.py +311 -0
- pyoq/query/postgres/factory.py +24 -0
- pyoq/query/postgres/health.py +24 -0
- pyoq/query/postgres/identifiers.py +9 -0
- pyoq/query/postgres/model.py +81 -0
- pyoq/query/postgres/operations.py +25 -0
- pyoq/query/postgres/parameters.py +71 -0
- pyoq/query/postgres/planning.py +20 -0
- pyoq/query/postgres/pool.py +52 -0
- pyoq/query/postgres/transactions.py +295 -0
- pyoq/query/postgres/writes.py +37 -0
- pyoq/query/projections.py +105 -0
- pyoq/query/raw.py +90 -0
- pyoq/query/recursion.py +265 -0
- pyoq/query/rendering/__init__.py +1 -0
- pyoq/query/rendering/expressions.py +913 -0
- pyoq/query/rendering/identifiers.py +40 -0
- pyoq/query/rendering/projections.py +63 -0
- pyoq/query/rendering/queries.py +334 -0
- pyoq/query/rendering/sources.py +66 -0
- pyoq/query/rendering/writes.py +176 -0
- pyoq/query/results.py +459 -0
- pyoq/query/routines.py +196 -0
- pyoq/query/rows.py +156 -0
- pyoq/query/select.py +793 -0
- pyoq/query/select_nodes.py +277 -0
- pyoq/query/sources.py +236 -0
- pyoq/query/sqlite/__init__.py +43 -0
- pyoq/query/sqlite/commands.py +201 -0
- pyoq/query/sqlite/compiler.py +139 -0
- pyoq/query/sqlite/context.py +20 -0
- pyoq/query/sqlite/executor.py +119 -0
- pyoq/query/sqlite/expressions.py +224 -0
- pyoq/query/sqlite/factory.py +32 -0
- pyoq/query/sqlite/health.py +28 -0
- pyoq/query/sqlite/identifiers.py +9 -0
- pyoq/query/sqlite/model.py +73 -0
- pyoq/query/sqlite/operations.py +36 -0
- pyoq/query/sqlite/parameters.py +50 -0
- pyoq/query/sqlite/planning.py +20 -0
- pyoq/query/sqlite/pool.py +50 -0
- pyoq/query/sqlite/streaming.py +13 -0
- pyoq/query/sqlite/transactions.py +274 -0
- pyoq/query/sqlite/writes.py +35 -0
- pyoq/query/statements.py +27 -0
- pyoq/query/values.py +23 -0
- pyoq/query/vendor.py +162 -0
- pyoq/query/windows.py +424 -0
- pyoq/query/write_nodes.py +174 -0
- pyoq/query/writes.py +628 -0
- pyoq/relations/__init__.py +66 -0
- pyoq/relations/batching.py +219 -0
- pyoq/relations/derivation.py +111 -0
- pyoq/relations/fetching.py +355 -0
- pyoq/relations/graph.py +245 -0
- pyoq/relations/loading.py +74 -0
- pyoq/relations/model.py +75 -0
- pyoq/relations/planning.py +206 -0
- pyoq/runtime/__init__.py +9 -0
- pyoq/runtime/kernels.py +25 -0
- pyoq/runtime/python.py +43 -0
- pyoq/runtime/selection.py +73 -0
- pyoq/sanic/__init__.py +32 -0
- pyoq/sanic/scope.py +197 -0
- pyoq/sanic/workers.py +129 -0
- pyoq/schema/__init__.py +108 -0
- pyoq/schema/codec.py +711 -0
- pyoq/schema/models.py +604 -0
- pyoq/schema/mysql/__init__.py +16 -0
- pyoq/schema/mysql/connection.py +73 -0
- pyoq/schema/mysql/dsn.py +72 -0
- pyoq/schema/mysql/records.py +354 -0
- pyoq/schema/mysql/reflection.py +309 -0
- pyoq/schema/mysql/source.py +30 -0
- pyoq/schema/mysql/sql.py +128 -0
- pyoq/schema/mysql/types.py +105 -0
- pyoq/schema/postgres/__init__.py +13 -0
- pyoq/schema/postgres/connection.py +63 -0
- pyoq/schema/postgres/records.py +384 -0
- pyoq/schema/postgres/reflection.py +466 -0
- pyoq/schema/postgres/source.py +30 -0
- pyoq/schema/postgres/sql.py +246 -0
- pyoq/schema/postgres/types.py +98 -0
- pyoq/schema/registry.py +45 -0
- pyoq/schema/source.py +15 -0
- pyoq/schema/sqlite/__init__.py +6 -0
- pyoq/schema/sqlite/connection.py +54 -0
- pyoq/schema/sqlite/records.py +167 -0
- pyoq/schema/sqlite/reflection.py +393 -0
- pyoq/schema/sqlite/source.py +30 -0
- pyoq/schema/sqlite/sql.py +254 -0
- pyoq/schema/sqlite/types.py +74 -0
- pyoq/serving/__init__.py +23 -0
- pyoq/serving/databases.py +107 -0
- pyoq/serving/opening.py +331 -0
- pyoq/snapshots/__init__.py +20 -0
- pyoq/snapshots/drift.py +312 -0
- pyoq/snapshots/files.py +96 -0
- pyoq/snapshots/routing.py +40 -0
- pyoq/snapshots/source.py +33 -0
- pyoq/tracing/__init__.py +5 -0
- pyoq/tracing/spans.py +89 -0
- pyoq/unset.py +14 -0
- pyoq_sql-1.0.2.dist-info/METADATA +3050 -0
- pyoq_sql-1.0.2.dist-info/RECORD +267 -0
- pyoq_sql-1.0.2.dist-info/WHEEL +4 -0
- pyoq_sql-1.0.2.dist-info/entry_points.txt +3 -0
- pyoq_sql-1.0.2.dist-info/licenses/LICENSE +373 -0
|
@@ -0,0 +1,3050 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pyoq-sql
|
|
3
|
+
Version: 1.0.2
|
|
4
|
+
Summary: A fully typed, database-first SQL toolkit for Python.
|
|
5
|
+
Project-URL: Homepage, https://teqpod.com/pyoq-sql/
|
|
6
|
+
Project-URL: Documentation, https://teqpod.com/pyoq-sql/overview/
|
|
7
|
+
Project-URL: Issues, https://github.com/Teqpod/pyoq-sql/issues
|
|
8
|
+
Project-URL: Security, https://github.com/Teqpod/pyoq-sql/security/advisories/new
|
|
9
|
+
License-Expression: MPL-2.0
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: database,query-builder,sql,typing
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Framework :: AsyncIO
|
|
14
|
+
Classifier: Framework :: Django
|
|
15
|
+
Classifier: Framework :: FastAPI
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
25
|
+
Classifier: Topic :: Database
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.11
|
|
28
|
+
Provides-Extra: django
|
|
29
|
+
Requires-Dist: django<7,>=4.2; extra == 'django'
|
|
30
|
+
Provides-Extra: fastapi
|
|
31
|
+
Requires-Dist: fastapi<1,>=0.115; extra == 'fastapi'
|
|
32
|
+
Provides-Extra: mysql
|
|
33
|
+
Requires-Dist: pymysql<2,>=1.1; extra == 'mysql'
|
|
34
|
+
Provides-Extra: mysql-async
|
|
35
|
+
Requires-Dist: asyncmy<1,>=0.2.14; extra == 'mysql-async'
|
|
36
|
+
Requires-Dist: pymysql<2,>=1.1; extra == 'mysql-async'
|
|
37
|
+
Provides-Extra: postgres
|
|
38
|
+
Requires-Dist: psycopg<4,>=3.2; extra == 'postgres'
|
|
39
|
+
Provides-Extra: sanic
|
|
40
|
+
Requires-Dist: sanic<26,>=24.6; extra == 'sanic'
|
|
41
|
+
Provides-Extra: tracing
|
|
42
|
+
Requires-Dist: opentelemetry-api<2,>=1.25; extra == 'tracing'
|
|
43
|
+
Description-Content-Type: text/markdown
|
|
44
|
+
|
|
45
|
+
# PyOQ
|
|
46
|
+
|
|
47
|
+
PyOQ is a fully typed, database-first SQL toolkit for Python. It is designed to
|
|
48
|
+
provide explicit query construction, generated schema types, predictable
|
|
49
|
+
execution, and efficient result hydration without hiding SQL behavior.
|
|
50
|
+
|
|
51
|
+
SQLite, PostgreSQL and MySQL are supported, synchronously and asynchronously.
|
|
52
|
+
Schema reflection generates row, key, insert, update, table, column, relation,
|
|
53
|
+
enum, domain and routine types that hold under strict type checking. Relations
|
|
54
|
+
are fetched by a plan that keeps the query count independent of the result,
|
|
55
|
+
and Django, FastAPI and Sanic each run on the same core rather than on a copy
|
|
56
|
+
of it.
|
|
57
|
+
|
|
58
|
+
Every public behaviour is covered by tests at complete statement and branch
|
|
59
|
+
coverage, run against real PostgreSQL and MySQL servers rather than recorded
|
|
60
|
+
stand-ins.
|
|
61
|
+
|
|
62
|
+
## Installation
|
|
63
|
+
|
|
64
|
+
```console
|
|
65
|
+
python -m pip install pyoq-sql
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The distribution name is `pyoq-sql`; the Python import package is `pyoq`.
|
|
69
|
+
|
|
70
|
+
Releases provide native platform wheels and a portable fallback wheel, so
|
|
71
|
+
installation does not require Rust. Building from the source archive does
|
|
72
|
+
require a supported Rust toolchain. When the portable wheel is used, select
|
|
73
|
+
its runtime explicitly:
|
|
74
|
+
|
|
75
|
+
```console
|
|
76
|
+
PYOQ_RUNTIME=python python -m your_application
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
The fallback preserves behavior and typing but is outside the native
|
|
80
|
+
performance contract. `PYOQ_RUNTIME=native` makes an unavailable native engine
|
|
81
|
+
a startup error. The default selects the native engine when present and emits
|
|
82
|
+
a runtime warning before using the fallback when it is absent.
|
|
83
|
+
|
|
84
|
+
## Package check
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
from pyoq import __version__
|
|
88
|
+
|
|
89
|
+
print(__version__)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Package architecture
|
|
93
|
+
|
|
94
|
+
Implementation code is grouped by capability. Each feature package owns its
|
|
95
|
+
public facade and private implementation details, while package-wide errors,
|
|
96
|
+
naming policy, typing metadata, and the native extension remain at the root.
|
|
97
|
+
|
|
98
|
+
```mermaid
|
|
99
|
+
flowchart TD
|
|
100
|
+
Package[pyoq] --> CLI[cli]
|
|
101
|
+
Package --> Config[config]
|
|
102
|
+
Package --> Generation[generation]
|
|
103
|
+
Package --> Query[query]
|
|
104
|
+
Package --> Runtime[runtime]
|
|
105
|
+
Package --> Schema[schema]
|
|
106
|
+
Package --> Serving[serving]
|
|
107
|
+
Package --> Errors[errors]
|
|
108
|
+
Package --> Naming[naming]
|
|
109
|
+
Runtime --> Native[private native extension]
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Applications should import feature APIs from their stable facades, such as
|
|
113
|
+
`pyoq.config`, `pyoq.schema`, `pyoq.generation`, and `pyoq.runtime`. Nested
|
|
114
|
+
modules separate component responsibilities and are not compatibility
|
|
115
|
+
surfaces.
|
|
116
|
+
|
|
117
|
+
## Configuration
|
|
118
|
+
|
|
119
|
+
Configuration is read from `pyproject.toml`. Generated code always lives in a
|
|
120
|
+
dedicated directory below the project root. Use a complete environment
|
|
121
|
+
reference when a data source name contains credentials.
|
|
122
|
+
|
|
123
|
+
```toml
|
|
124
|
+
[tool.pyoq]
|
|
125
|
+
codegen-directory = "generated/database"
|
|
126
|
+
codegen-package = "application.database"
|
|
127
|
+
selected-profile = "default"
|
|
128
|
+
|
|
129
|
+
[tool.pyoq.profiles.default]
|
|
130
|
+
dialect = "postgres"
|
|
131
|
+
host = "localhost"
|
|
132
|
+
port = 5432
|
|
133
|
+
database = "${POSTGRES_DATABASE}"
|
|
134
|
+
username = "${POSTGRES_USERNAME}"
|
|
135
|
+
password = "${POSTGRES_PASSWORD}"
|
|
136
|
+
minimum-pool-size = 1
|
|
137
|
+
maximum-pool-size = 8
|
|
138
|
+
pool-checkout-timeout = 3.0
|
|
139
|
+
|
|
140
|
+
[tool.pyoq.profiles.testing]
|
|
141
|
+
dialect = "sqlite"
|
|
142
|
+
database-path = "database/testing.sqlite"
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Open the selected profile through the fully typed lifecycle API:
|
|
146
|
+
|
|
147
|
+
```python
|
|
148
|
+
from pyoq import connect
|
|
149
|
+
from application.database.tables import ACCOUNT
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
with connect(profile="default") as database:
|
|
153
|
+
rows = database.select_from(ACCOUNT).limit(10).fetch_all()
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Omit `profile` to use `selected-profile`. The context creates one process-owned
|
|
157
|
+
pool from the profile limits and closes it on exit. A framework-free
|
|
158
|
+
application can instead pass a data source directly with
|
|
159
|
+
`connect(dsn, dialect=Dialect.MYSQL)`. An explicit `pool_policy` replaces the
|
|
160
|
+
complete profile pool policy for one call.
|
|
161
|
+
|
|
162
|
+
SQLite file paths use `database-path`. They are resolved relative to the project
|
|
163
|
+
root, must remain inside it, and are opened read-only during inspection. An
|
|
164
|
+
environment reference in the exact `${VARIABLE_NAME}` form may be used for any
|
|
165
|
+
structured server or pool value. Any other string is literal. Server profiles
|
|
166
|
+
may alternatively define one `dsn`, but cannot combine it with structured
|
|
167
|
+
connection fields.
|
|
168
|
+
|
|
169
|
+
`SecretValue` renders as `[REDACTED]`; access to its underlying value is always
|
|
170
|
+
explicit. An explicit `Configuration` object takes precedence over the file,
|
|
171
|
+
and the `selected_profile` argument can override profile selection. Unknown
|
|
172
|
+
settings, missing profiles, invalid dialects, unavailable environment values,
|
|
173
|
+
and paths outside the project root fail with exceptions from `pyoq.errors`.
|
|
174
|
+
|
|
175
|
+
## What each dialect can do
|
|
176
|
+
|
|
177
|
+
Every row here is one operation, not a family, because a family name hides
|
|
178
|
+
which parts of it exist. Every entry was taken from a running server rather
|
|
179
|
+
than from a specification.
|
|
180
|
+
|
|
181
|
+
**refused** means PyOQ raises before the statement is sent, naming what the
|
|
182
|
+
dialect has not got. It never emits SQL a database cannot run, and never
|
|
183
|
+
quietly does something narrower than what was asked. **written by PyOQ** means
|
|
184
|
+
the dialect has no such operation and PyOQ writes it out of what the dialect
|
|
185
|
+
does have, so one word means one thing on all three.
|
|
186
|
+
|
|
187
|
+
This table is generated from the capability matrix the source keeps. A
|
|
188
|
+
contract compiles every row against every dialect, checks that each answers
|
|
189
|
+
with the kind of value the row claims, and names the dialect-neutral contract
|
|
190
|
+
that runs it against a real server, so a claim here that is not true stops the
|
|
191
|
+
test suite.
|
|
192
|
+
|
|
193
|
+
<!-- capability matrix -->
|
|
194
|
+
|
|
195
|
+
| Operation | sqlite | postgres | mysql | reads back as | run by |
|
|
196
|
+
| --- | --- | --- | --- | --- | --- |
|
|
197
|
+
| window: row number | yes | yes | yes | numeric | `window_contract` |
|
|
198
|
+
| window: rank | yes | yes | yes | numeric | `window_contract` |
|
|
199
|
+
| window: dense rank | yes | yes | yes | numeric | `window_contract` |
|
|
200
|
+
| window: percent rank | yes | yes | yes | numeric | `window_contract` |
|
|
201
|
+
| window: cumulative distribution | yes | yes | yes | numeric | `window_contract` |
|
|
202
|
+
| window: ntile | yes | yes | yes | numeric | `window_contract` |
|
|
203
|
+
| window: lag | yes | yes | yes | numeric | `window_contract` |
|
|
204
|
+
| window: lead | yes | yes | yes | numeric | `window_contract` |
|
|
205
|
+
| window: first value | yes | yes | yes | numeric | `window_contract` |
|
|
206
|
+
| window: last value | yes | yes | yes | numeric | `window_contract` |
|
|
207
|
+
| window: nth value | yes | yes | yes | numeric | `window_contract` |
|
|
208
|
+
| window: an aggregate over a window | yes | yes | yes | numeric | `window_contract` |
|
|
209
|
+
| window: partition by | yes | yes | yes | numeric | `window_contract` |
|
|
210
|
+
| window: order by | yes | yes | yes | numeric | `window_contract` |
|
|
211
|
+
| window frame: rows | yes | yes | yes | numeric | `window_contract` |
|
|
212
|
+
| window frame: range | yes | yes | yes | numeric | `window_contract` |
|
|
213
|
+
| window frame: groups | yes | yes | refused | numeric | `window_contract` |
|
|
214
|
+
| window frame: exclude | yes | yes | refused | numeric | `window_contract` |
|
|
215
|
+
| window: declared by name | yes | yes | yes | numeric | `window_contract` |
|
|
216
|
+
| json: read a value | yes | yes | yes | json | `json_contract` |
|
|
217
|
+
| json: read text | yes | yes | yes | string | `json_contract` |
|
|
218
|
+
| json: how long an array is | yes | yes | yes | numeric | `json_contract` |
|
|
219
|
+
| json: whether a path is there | yes | yes | yes | boolean | `json_contract` |
|
|
220
|
+
| json: whether it holds a value | refused | yes | yes | boolean | `json_contract` |
|
|
221
|
+
| json: set | yes | yes | yes | json | `json_contract` |
|
|
222
|
+
| json: insert | yes | written by PyOQ | yes | json | `json_contract` |
|
|
223
|
+
| json: replace | yes | yes | yes | json | `json_contract` |
|
|
224
|
+
| json: merge all the way down | yes | refused | yes | json | `json_contract` |
|
|
225
|
+
| json: write over members, one level | refused | yes | refused | json | `json_contract` |
|
|
226
|
+
| json: remove | yes | yes | yes | json | `json_contract` |
|
|
227
|
+
| json: build an object | yes | yes | yes | json | `json_contract` |
|
|
228
|
+
| json: build an array | yes | yes | yes | json | `json_contract` |
|
|
229
|
+
| json: name the members | written by PyOQ | written by PyOQ | yes | json | `json_contract` |
|
|
230
|
+
| json: gather rows into an array | yes | yes | yes | json | `json_contract` |
|
|
231
|
+
| json: gather rows into an object | yes | yes | yes | json | `json_contract` |
|
|
232
|
+
| json: read an array as rows | yes | yes | yes | string | `json_contract` |
|
|
233
|
+
| array: holds a value | refused | yes | refused | boolean | `array_contract` |
|
|
234
|
+
| array: holds a value nowhere | refused | yes | refused | boolean | `array_contract` |
|
|
235
|
+
| array: holds every value | refused | yes | refused | boolean | `array_contract` |
|
|
236
|
+
| array: is among | refused | yes | refused | boolean | `array_contract` |
|
|
237
|
+
| array: shares a value | refused | yes | refused | boolean | `array_contract` |
|
|
238
|
+
| array: append | refused | yes | refused | other | `array_contract` |
|
|
239
|
+
| array: prepend | refused | yes | refused | other | `array_contract` |
|
|
240
|
+
| array: concatenate | refused | yes | refused | other | `array_contract` |
|
|
241
|
+
| array: without a value | refused | yes | refused | other | `array_contract` |
|
|
242
|
+
| array: replace a value | refused | yes | refused | other | `array_contract` |
|
|
243
|
+
| array: one element | refused | yes | refused | other | `array_contract` |
|
|
244
|
+
| array: how many elements | refused | yes | refused | numeric | `array_contract` |
|
|
245
|
+
| array: how many along a dimension | refused | yes | refused | numeric | `array_contract` |
|
|
246
|
+
| array: how many dimensions | refused | yes | refused | numeric | `array_contract` |
|
|
247
|
+
| array: built from values | refused | yes | refused | other | `array_contract` |
|
|
248
|
+
| lock: for update | refused | yes | yes | numeric | `locking_contract` |
|
|
249
|
+
| lock: for share | refused | yes | yes | numeric | `locking_contract` |
|
|
250
|
+
| lock: fail rather than wait | refused | yes | yes | numeric | `locking_contract` |
|
|
251
|
+
| lock: skip what is locked | refused | yes | yes | numeric | `locking_contract` |
|
|
252
|
+
| lock: named tables | refused | yes | yes | numeric | `locking_contract` |
|
|
253
|
+
| lock: named fields | refused | yes | yes | numeric | `locking_contract` |
|
|
254
|
+
| query: lateral source | refused | yes | yes | numeric | `lateral_contract` |
|
|
255
|
+
| query: recursive common table | yes | yes | yes | numeric | `recursion_contract` |
|
|
256
|
+
| query: row value | yes | yes | yes | boolean | `row_value_contract` |
|
|
257
|
+
| query: cast | yes | yes | yes | string | `cast_contract` |
|
|
258
|
+
| query: nulls first or last | yes | yes | written by PyOQ | numeric | `null_order_contract` |
|
|
259
|
+
| query: right join | refused | yes | yes | numeric | `joined_contract` |
|
|
260
|
+
| query: full join | refused | yes | refused | numeric | `joined_contract` |
|
|
261
|
+
| routine: run a stored procedure | refused | yes | yes | rows | `routine_contract` |
|
|
262
|
+
| routine: ask a function by name | yes | yes | yes | numeric | `routine_contract` |
|
|
263
|
+
| routine: more than one result set | refused | refused | refused | rows | `routine_contract` |
|
|
264
|
+
|
|
265
|
+
<!-- capability matrix -->
|
|
266
|
+
|
|
267
|
+
A JSON path is steps rather than text, because PostgreSQL reads another
|
|
268
|
+
dialect's written path as a member of that name and answers null without
|
|
269
|
+
complaining. PostgreSQL's own insert raises where a member already exists,
|
|
270
|
+
while the other two leave what is there, so PyOQ asks first on PostgreSQL.
|
|
271
|
+
|
|
272
|
+
Merging and concatenating are two operations, not one. `json_merge` is
|
|
273
|
+
merge-patch: a member given as null is removed, and an object inside an object
|
|
274
|
+
is merged rather than replaced. MySQL and SQLite do that themselves, and
|
|
275
|
+
PostgreSQL has no merge-patch at all. `json_concat` is one level deep: a member
|
|
276
|
+
is replaced whole and a null is stored as a null. PostgreSQL does that itself,
|
|
277
|
+
and the other two have nothing that means it. Each is refused where the
|
|
278
|
+
dialect does not have it, rather than quietly doing the other one.
|
|
279
|
+
|
|
280
|
+
Only PostgreSQL has an array type. SQLite will accept `TEXT[]` as a type name
|
|
281
|
+
with no array behind it, which is the kind of tolerance an emulation would be
|
|
282
|
+
built on, so the other two refuse instead.
|
|
283
|
+
|
|
284
|
+
SQLite locks a database rather than rows and keeps no stored procedures. A
|
|
285
|
+
timed lock wait is in no supported dialect: `FOR UPDATE WAIT` is a syntax error
|
|
286
|
+
on both PostgreSQL and MySQL. Null treatment on a window function is likewise
|
|
287
|
+
in none of them: `IGNORE NULLS` is refused by all three servers.
|
|
288
|
+
|
|
289
|
+
A procedure that answers with more than one result set is refused rather than
|
|
290
|
+
read down to its first, because a dropped result set is a silent one.
|
|
291
|
+
|
|
292
|
+
### Schema objects
|
|
293
|
+
|
|
294
|
+
| Object | SQLite | PostgreSQL | MySQL |
|
|
295
|
+
| --- | --- | --- | --- |
|
|
296
|
+
| tables, views, keys, relations, indexes, checks | yes | yes | yes |
|
|
297
|
+
| enums | none | reflected and generated | reflected per column |
|
|
298
|
+
| domains | none | reflected and generated | none |
|
|
299
|
+
| routines | none | reflected and generated | reflected with typed calls or typed rejection |
|
|
300
|
+
|
|
301
|
+
Only PostgreSQL keeps overloaded routines; MySQL refuses a second routine of
|
|
302
|
+
the same name, and SQLite keeps no routine catalog at all.
|
|
303
|
+
|
|
304
|
+
A procedure answers through the parameters it writes. A generated PostgreSQL
|
|
305
|
+
call asks the caller only for IN and INOUT values, supplies required OUT
|
|
306
|
+
placeholders internally, and reads the answer back with the catalog's types and
|
|
307
|
+
nullability. MySQL needs a session-variable adapter for OUT and INOUT values.
|
|
308
|
+
Those routines retain a generated, typed signature that raises
|
|
309
|
+
`UnsupportedQueryError` before any invalid SQL is sent.
|
|
310
|
+
|
|
311
|
+
A generated call names the schema the catalog said keeps the routine. Without
|
|
312
|
+
it a routine outside the search path resolves to nothing, and one whose name
|
|
313
|
+
another schema shares resolves to the wrong one.
|
|
314
|
+
|
|
315
|
+
## Supported dialects
|
|
316
|
+
|
|
317
|
+
Every configured dialect resolves to a schema source, so inspection and
|
|
318
|
+
generation work for all of them. SQLite is implemented end to end. PostgreSQL
|
|
319
|
+
has a compiler, type mapping, schema reflection, and both synchronous and
|
|
320
|
+
asynchronous drivers with pooling, transactions, savepoints, and streaming.
|
|
321
|
+
MySQL has all of that too, so all three dialects are now complete vertical
|
|
322
|
+
slices.
|
|
323
|
+
|
|
324
|
+
Database drivers are optional extras, so a SQLite installation stays dependency
|
|
325
|
+
free:
|
|
326
|
+
|
|
327
|
+
```console
|
|
328
|
+
python -m pip install "pyoq-sql[postgres]"
|
|
329
|
+
python -m pip install "pyoq-sql[mysql]"
|
|
330
|
+
python -m pip install "pyoq-sql[mysql-async]"
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
Without them, PostgreSQL and MySQL query construction and compilation still
|
|
334
|
+
work because they are pure Python. Only connecting requires a driver, and its
|
|
335
|
+
absence raises an explicit error naming the extra to install. PostgreSQL and
|
|
336
|
+
PostgreSQL and MySQL profiles supply their connection string through `dsn`, and
|
|
337
|
+
inspection opens the connection read-only.
|
|
338
|
+
|
|
339
|
+
A MySQL connection string names one database, which is reflected as the
|
|
340
|
+
snapshot's schema:
|
|
341
|
+
|
|
342
|
+
```console
|
|
343
|
+
mysql://user:password@host:3306/database
|
|
344
|
+
```
|
|
345
|
+
|
|
346
|
+
Query and fragment options are refused rather than silently ignored, so an
|
|
347
|
+
unsupported setting can never be mistaken for an applied one.
|
|
348
|
+
|
|
349
|
+
`schema_source_for()` resolves a dialect to its schema source, and
|
|
350
|
+
`supported_dialects()` reports every dialect that has one. Command services
|
|
351
|
+
route through the configured dialect at load time, so adding a dialect is a
|
|
352
|
+
registry entry rather than a change to the command layer.
|
|
353
|
+
|
|
354
|
+
The layers a dialect plugs into are deliberately neutral. `CompiledQuery`,
|
|
355
|
+
`StatementCompiler`, `SchemaSource`, `QueryOperations`, `BulkPlan`, and
|
|
356
|
+
`DatabaseCursor` carry no dialect assumptions, so a new database supplies a
|
|
357
|
+
compiler, a schema source, and a cursor-backed executor without reimplementing
|
|
358
|
+
result cardinality, bulk planning, or ordered multi-operation execution.
|
|
359
|
+
|
|
360
|
+
Execution is shared too. `ConnectionPool` owns the lease lifecycle, size and
|
|
361
|
+
timeout policy, idle reuse, invalidation, and close semantics; `RowStream` owns
|
|
362
|
+
bounded batch buffering, iteration ownership, and deterministic cleanup; and
|
|
363
|
+
`TransactionState` owns scope nesting, savepoint naming, owner-thread checks,
|
|
364
|
+
and stream interaction. A dialect supplies how it resets and validates a
|
|
365
|
+
connection, how it opens a cursor, and how it begins a transaction.
|
|
366
|
+
|
|
367
|
+
SQL rendering is shared as well. `ExpressionRenderer`, `QueryRenderer`, and
|
|
368
|
+
`WriteRenderer` own the standard SQL that every relational dialect writes the
|
|
369
|
+
same way: projections, sources, joins, common tables, ordering, set operations,
|
|
370
|
+
inserts, assignments, row scoping, conflict resolution, and returning. A dialect
|
|
371
|
+
declares only where it genuinely differs, such as how it spells distinctness,
|
|
372
|
+
how it extracts a date part, how it renders string predicates, which joins and
|
|
373
|
+
capabilities it allows, and how it paginates.
|
|
374
|
+
|
|
375
|
+
Parameter placeholders are part of that declaration. `ParameterStyle` covers
|
|
376
|
+
`?`, `$1`, and `%s`, and the compilation context escapes literal percent signs
|
|
377
|
+
in structural SQL when the driver style requires it.
|
|
378
|
+
|
|
379
|
+
`PostgresCompiler` shows the shape of a dialect built this way. It renders the
|
|
380
|
+
same query model as SQLite while spelling distinctness as `IS DISTINCT FROM`,
|
|
381
|
+
extracting date parts with `EXTRACT`, rendering string predicates with
|
|
382
|
+
`POSITION`, `LEFT`, and `RIGHT`, allowing RIGHT and FULL joins and a native
|
|
383
|
+
`NULLS` clause, emitting a bare `OFFSET` without a placeholder limit, and
|
|
384
|
+
budgeting at the 65535 parameter ceiling of the extended query protocol. It
|
|
385
|
+
rejects catalog-qualified identifiers and adding two temporal values, because
|
|
386
|
+
PostgreSQL has no operator for either.
|
|
387
|
+
|
|
388
|
+
MySQL has a compiler, type mapping, schema reflection, and synchronous and
|
|
389
|
+
asynchronous drivers. It shows how far a dialect can diverge while still using
|
|
390
|
+
the shared renderers: identifiers are quoted with backticks, string
|
|
391
|
+
concatenation renders as `CONCAT` because `||` means logical OR in MySQL,
|
|
392
|
+
distinctness uses the null-safe `<=>` operator, a bare `OFFSET` is given the
|
|
393
|
+
maximum row limit MySQL requires, explicit null placement is written by
|
|
394
|
+
ordering on whether a value is null because MySQL has no `NULLS` clause, and
|
|
395
|
+
upsert renders as a row alias with `ON DUPLICATE KEY UPDATE`. The row alias form requires
|
|
396
|
+
MySQL 8.0.19 or later; earlier servers and MariaDB use a different spelling
|
|
397
|
+
that PyOQ does not emit.
|
|
398
|
+
|
|
399
|
+
Where MySQL cannot express something, it says so instead of emulating it.
|
|
400
|
+
`RETURNING` is refused with a message pointing at a follow-up query. A conflict
|
|
401
|
+
target is refused because MySQL infers the key. A conflict condition is refused
|
|
402
|
+
because `ON DUPLICATE KEY UPDATE` has no `WHERE`. Do-nothing conflict
|
|
403
|
+
resolution is refused rather than emulated with `INSERT IGNORE`, which would
|
|
404
|
+
also swallow unrelated errors, or with a self-assignment, which would report a
|
|
405
|
+
different affected row count.
|
|
406
|
+
|
|
407
|
+
`postgres_type()` normalizes declared PostgreSQL type names into the shared
|
|
408
|
+
schema model, including serial types, spelled-out variants such as
|
|
409
|
+
`timestamp with time zone`, length and precision arguments, and array types
|
|
410
|
+
with their element type. A declaration it cannot parse stays opaque rather than
|
|
411
|
+
being guessed at.
|
|
412
|
+
|
|
413
|
+
MySQL executes through `MySQLExecutor` over a `MySQLPool`, with the same typed
|
|
414
|
+
operations SQLite and PostgreSQL use. Four dialect behaviors differ and are
|
|
415
|
+
documented rather than hidden.
|
|
416
|
+
|
|
417
|
+
An affected row count means matched rows. MySQL counts changed rows by default,
|
|
418
|
+
so an update writing a column's existing value would report nothing, and PyOQ
|
|
419
|
+
connects with the client flag that restores the meaning every other dialect
|
|
420
|
+
has.
|
|
421
|
+
|
|
422
|
+
An insert reports the first identifier it generated, not the last, because that
|
|
423
|
+
is what MySQL reports for a multi-row insert.
|
|
424
|
+
|
|
425
|
+
A statement timeout is enforced from the client. MySQL applies its own
|
|
426
|
+
`max_execution_time` only to read-only SELECT statements, so PyOQ sets that
|
|
427
|
+
and also runs a watchdog that asks the server to stop the statement once the
|
|
428
|
+
deadline passes. Cancellation works the same way, because MySQL has no
|
|
429
|
+
client-side cancel. Both open a short-lived second connection, and a statement
|
|
430
|
+
carrying neither a timeout nor a cancellation token starts no watchdog and pays
|
|
431
|
+
nothing.
|
|
432
|
+
|
|
433
|
+
A deadlock rolls the whole transaction back on the server while leaving the
|
|
434
|
+
connection healthy, and a later commit then succeeds having committed nothing.
|
|
435
|
+
PyOQ refuses to report that commit and raises instead, so work that was
|
|
436
|
+
discarded is never reported as saved. A failed statement that is not a deadlock
|
|
437
|
+
leaves a MySQL transaction usable, which is the opposite of PostgreSQL and
|
|
438
|
+
needs no savepoint to recover from.
|
|
439
|
+
|
|
440
|
+
Isolation is set before a transaction starts rather than as part of starting
|
|
441
|
+
it, and the default is MySQL's own repeatable read rather than a level PyOQ
|
|
442
|
+
imposes. Streaming reads through an unbuffered cursor, and a transaction allows
|
|
443
|
+
only one open stream at a time so that a second statement cannot silently
|
|
444
|
+
discard the rows the first has not read.
|
|
445
|
+
|
|
446
|
+
`MySQLAsyncExecutor` over a `MySQLAsyncPool` provides the same operations
|
|
447
|
+
without blocking, and behaves identically on every point above. Three things are
|
|
448
|
+
worth knowing about it.
|
|
449
|
+
|
|
450
|
+
It has its own extra, because its driver is a compiled one:
|
|
451
|
+
|
|
452
|
+
```console
|
|
453
|
+
python -m pip install "pyoq-sql[mysql-async]"
|
|
454
|
+
```
|
|
455
|
+
|
|
456
|
+
The synchronous extra stays pure Python. The asynchronous driver is compiled
|
|
457
|
+
because the read path is worth it: PyOQ adds only a few percent above its
|
|
458
|
+
driver on a large read, so the driver is the cost, and a pure-Python driver
|
|
459
|
+
measured roughly an order of magnitude slower end to end on the same query. A
|
|
460
|
+
throughput contract holds the read path to that standard so it cannot quietly
|
|
461
|
+
regress.
|
|
462
|
+
|
|
463
|
+
Its connections are encrypted. The synchronous driver negotiates TLS on its
|
|
464
|
+
own, and the asynchronous one is given the same settings so that choosing
|
|
465
|
+
`async` never quietly downgrades the transport. Encryption is what lets a
|
|
466
|
+
server's password check happen without a separate cryptography library, so a
|
|
467
|
+
server with TLS switched off needs `PyMySQL[rsa]` installed for either driver.
|
|
468
|
+
|
|
469
|
+
Its driver publishes only generated type stubs that leave most of its surface
|
|
470
|
+
unknown, and raises its own exception hierarchy rather than the synchronous
|
|
471
|
+
driver's. PyOQ therefore declares the shape it depends on and confines every
|
|
472
|
+
reference to the driver to one module. Failure classification is shared, since
|
|
473
|
+
every MySQL driver reports the same numeric codes while raising different
|
|
474
|
+
exception types.
|
|
475
|
+
|
|
476
|
+
`mysql_type()` normalizes declared MySQL type names, including width and
|
|
477
|
+
precision arguments, `unsigned`, `signed`, and `zerofill` modifiers, and the
|
|
478
|
+
inline `ENUM` and `SET` declarations that carry their values in the column type
|
|
479
|
+
rather than in a named type.
|
|
480
|
+
|
|
481
|
+
MySQL reflection reads the connected database through `information_schema`.
|
|
482
|
+
Auto-increment columns are reported as identity columns, stored and virtual
|
|
483
|
+
generated columns keep their generation expression, and column defaults are
|
|
484
|
+
normalized into valid SQL so that a literal default such as an empty string is
|
|
485
|
+
quoted while an expression default such as `json_object()` is not. Descending
|
|
486
|
+
and expression index terms, composite foreign keys with their update and delete
|
|
487
|
+
rules, and CHECK constraints are all carried across. The placeholder comment
|
|
488
|
+
MySQL stores on every view is discarded rather than reported as a comment.
|
|
489
|
+
|
|
490
|
+
PostgreSQL executes through `PostgresExecutor` over a `PostgresPool`, with the
|
|
491
|
+
same typed operations SQLite uses. Two dialect behaviors differ and are not
|
|
492
|
+
hidden. A statement timeout is enforced by the server through
|
|
493
|
+
`statement_timeout` rather than a client-side interrupt, and a cancellation
|
|
494
|
+
token is honored mid-statement by a watchdog that issues a server cancel.
|
|
495
|
+
Streaming uses a server-side cursor inside its own transaction, which is what
|
|
496
|
+
keeps memory bounded for large results.
|
|
497
|
+
|
|
498
|
+
PostgreSQL also executes asynchronously through `PostgresAsyncExecutor` over a
|
|
499
|
+
`PostgresAsyncPool`. The asynchronous surface mirrors the synchronous one
|
|
500
|
+
method for method, including pooling, transactions, savepoints, server-side
|
|
501
|
+
cursor streaming, server-enforced timeouts, and cancellation:
|
|
502
|
+
|
|
503
|
+
```python
|
|
504
|
+
async with PostgresAsyncPool(PostgresAsyncConnectionFactory(dsn)) as pool:
|
|
505
|
+
database = PostgresAsyncExecutor(pool)
|
|
506
|
+
async with database.transaction() as transaction:
|
|
507
|
+
await transaction.execute(statement)
|
|
508
|
+
async with database.stream(query) as rows:
|
|
509
|
+
async for row in rows:
|
|
510
|
+
...
|
|
511
|
+
```
|
|
512
|
+
|
|
513
|
+
Bulk writes use PostgreSQL pipeline mode on both paths. The planner still
|
|
514
|
+
splits a large insert into chunks that fit the parameter budget, but the chunks
|
|
515
|
+
are sent without waiting for each result, which removes a round trip per chunk.
|
|
516
|
+
This changes one behavior for the better and it is worth knowing: a pipelined
|
|
517
|
+
batch is atomic, so a failure discards the whole batch rather than leaving
|
|
518
|
+
earlier chunks applied. Sequential chunking keeps the earlier chunks, as it
|
|
519
|
+
always did.
|
|
520
|
+
|
|
521
|
+
Pipelining is governed by the `pipeline` compiler capability and by whether the
|
|
522
|
+
installed libpq supports it. When either says no, chunk execution falls back to
|
|
523
|
+
the sequential path with identical results.
|
|
524
|
+
|
|
525
|
+
Ownership is per task rather than per thread, so a stream or transaction used
|
|
526
|
+
from a different task fails explicitly instead of corrupting its state. Both
|
|
527
|
+
paths are held to the same dialect-neutral execution contract, so behavior does
|
|
528
|
+
not drift between them.
|
|
529
|
+
|
|
530
|
+
The most important difference is transactional: in PostgreSQL a failed
|
|
531
|
+
statement aborts the whole transaction, and every later statement in that
|
|
532
|
+
transaction fails until it ends. Recovery is a savepoint taken before the
|
|
533
|
+
statement that might fail:
|
|
534
|
+
|
|
535
|
+
```python
|
|
536
|
+
with database.transaction() as transaction:
|
|
537
|
+
with transaction.savepoint() as attempt:
|
|
538
|
+
attempt.execute(statement_that_may_conflict)
|
|
539
|
+
transaction.execute(next_statement)
|
|
540
|
+
```
|
|
541
|
+
|
|
542
|
+
PyOQ does not wrap every statement in an implicit savepoint to paper over this,
|
|
543
|
+
because that would add a hidden round trip to every write.
|
|
544
|
+
|
|
545
|
+
PostgreSQL reflection reads the system catalogs directly rather than
|
|
546
|
+
`information_schema`, so it recovers identity and generated columns, array
|
|
547
|
+
element types, per-column comments, foreign key referential actions, expression
|
|
548
|
+
and partial indexes with their sort direction, check constraints, materialized
|
|
549
|
+
views, and multiple schemas. Catalog rows are validated as they are read, so a
|
|
550
|
+
row of unexpected shape fails as a schema error instead of producing a silently
|
|
551
|
+
wrong model.
|
|
552
|
+
|
|
553
|
+
## Schema model
|
|
554
|
+
|
|
555
|
+
Database metadata is normalized into an immutable, dialect-neutral snapshot.
|
|
556
|
+
Source identifiers retain their exact case, spacing, and Unicode content. The
|
|
557
|
+
model does not apply Python naming rules or depend on database drivers and web
|
|
558
|
+
frameworks.
|
|
559
|
+
|
|
560
|
+
```mermaid
|
|
561
|
+
flowchart TD
|
|
562
|
+
Snapshot[SchemaSnapshot] --> Catalog[Catalog]
|
|
563
|
+
Catalog --> Schema[Schema]
|
|
564
|
+
Schema --> Table[Table]
|
|
565
|
+
Schema --> View[View]
|
|
566
|
+
Schema --> Enum[EnumType]
|
|
567
|
+
Table --> Column[Column]
|
|
568
|
+
Table --> Key[Key]
|
|
569
|
+
Table --> Relation[Relation]
|
|
570
|
+
Table --> Index[Index]
|
|
571
|
+
Table --> Check[CheckConstraint]
|
|
572
|
+
```
|
|
573
|
+
|
|
574
|
+
Construct snapshots directly when implementing metadata sources:
|
|
575
|
+
|
|
576
|
+
```python
|
|
577
|
+
from pyoq.config import DatabaseDialect
|
|
578
|
+
from pyoq.schema import (
|
|
579
|
+
Catalog,
|
|
580
|
+
Column,
|
|
581
|
+
Identifier,
|
|
582
|
+
Schema,
|
|
583
|
+
SchemaSnapshot,
|
|
584
|
+
SqlType,
|
|
585
|
+
Table,
|
|
586
|
+
TypeKind,
|
|
587
|
+
)
|
|
588
|
+
|
|
589
|
+
user_id = Column(
|
|
590
|
+
name=Identifier("id"),
|
|
591
|
+
data_type=SqlType(TypeKind.INTEGER, "INTEGER"),
|
|
592
|
+
nullable=False,
|
|
593
|
+
)
|
|
594
|
+
users = Table(name=Identifier("users"), columns=(user_id,))
|
|
595
|
+
snapshot = SchemaSnapshot(
|
|
596
|
+
dialect=DatabaseDialect.SQLITE,
|
|
597
|
+
catalogs=(Catalog(None, (Schema(None, tables=(users,)),)),),
|
|
598
|
+
)
|
|
599
|
+
```
|
|
600
|
+
|
|
601
|
+
`SchemaSnapshot.to_json()` produces compact, versioned, deterministic JSON with
|
|
602
|
+
Unicode preserved. `SchemaSnapshot.from_json()` performs strict field and type
|
|
603
|
+
validation before constructing model objects. The round trip preserves ordered
|
|
604
|
+
columns, nullability, raw defaults, identity and computed values, SQL type
|
|
605
|
+
details, qualified references, comments, indexes, checks, relations, enums,
|
|
606
|
+
views, and dialect capabilities.
|
|
607
|
+
|
|
608
|
+
Every model uses frozen slots. Duplicate identifiers, invalid type dimensions,
|
|
609
|
+
misaligned foreign-key columns, unknown local key or index columns, conflicting
|
|
610
|
+
generated-value metadata, and unsupported snapshot versions fail with typed
|
|
611
|
+
schema exceptions. Qualified relation targets may remain outside a snapshot so
|
|
612
|
+
metadata sources can represent intentionally partial introspection safely.
|
|
613
|
+
|
|
614
|
+
## Relations
|
|
615
|
+
|
|
616
|
+
A foreign key is one fact that can be read from either side, so PyOQ derives
|
|
617
|
+
both. `derive_relations()` turns a schema snapshot into typed relations that
|
|
618
|
+
carry direction, cardinality, and whether a to-one relation may resolve to
|
|
619
|
+
nothing:
|
|
620
|
+
|
|
621
|
+
```python
|
|
622
|
+
from pyoq.relations import derive_relations
|
|
623
|
+
|
|
624
|
+
for relation in derive_relations(snapshot):
|
|
625
|
+
print(relation.direction, relation.cardinality, relation.optional)
|
|
626
|
+
```
|
|
627
|
+
|
|
628
|
+
Cardinality and optionality are read from the schema rather than declared:
|
|
629
|
+
|
|
630
|
+
- A foreign key names at most one row of the table it points at, so following
|
|
631
|
+
it forward is always to-one. It is optional when any of its columns is
|
|
632
|
+
nullable, because a foreign key with a null part references nothing at all.
|
|
633
|
+
- Following it backwards is to-many, unless the referencing columns are
|
|
634
|
+
themselves unique, in which case it is a to-one that may be absent. A unique
|
|
635
|
+
key over part of the referencing columns is enough, since a subset that is
|
|
636
|
+
already unique makes the wider set unique too. This is what recognizes a
|
|
637
|
+
table whose primary key is also its foreign key as a one-to-one extension
|
|
638
|
+
rather than a collection.
|
|
639
|
+
- A to-many relation is never optional. The absence of children is an empty
|
|
640
|
+
collection, not a missing value.
|
|
641
|
+
|
|
642
|
+
A self-referencing foreign key produces both directions on the same table.
|
|
643
|
+
|
|
644
|
+
`build_relation_graph()` indexes those relations by the table they start from,
|
|
645
|
+
so a schema can be navigated in either direction:
|
|
646
|
+
|
|
647
|
+
```python
|
|
648
|
+
from pyoq.relations import build_relation_graph
|
|
649
|
+
|
|
650
|
+
graph = build_relation_graph(snapshot)
|
|
651
|
+
for relation in graph.to_many_from(graph.resolve(team_reference)):
|
|
652
|
+
print(relation.target.table.name.value)
|
|
653
|
+
```
|
|
654
|
+
|
|
655
|
+
Endpoints and lookups are resolved to the way the snapshot names a table, so a
|
|
656
|
+
foreign key written against a bare name reaches the qualified table it means,
|
|
657
|
+
and a caller need not know how the snapshot qualified anything. A bare name
|
|
658
|
+
that two schemas both carry is left alone, because it names neither.
|
|
659
|
+
|
|
660
|
+
A snapshot describes one database and a foreign key may point beyond it, so a
|
|
661
|
+
target the snapshot does not describe is reported through
|
|
662
|
+
`unresolved_targets` rather than refused. Such a table stays navigable, because
|
|
663
|
+
the table that declared the key still knows its side of it; only asking for the
|
|
664
|
+
missing table's definition fails, and it fails by name. A key naming a column
|
|
665
|
+
that a table it did resolve does not have is refused outright, because that is
|
|
666
|
+
corruption rather than partial coverage.
|
|
667
|
+
|
|
668
|
+
Generation emits both directions of every key as typed constants, each
|
|
669
|
+
carrying its cardinality and whether it may resolve to nothing:
|
|
670
|
+
|
|
671
|
+
```python
|
|
672
|
+
from generated.relations import (
|
|
673
|
+
EMPLOYEE_TEAM_TEAM_ID,
|
|
674
|
+
TEAM_EMPLOYEE_TEAM_ID_REVERSE,
|
|
675
|
+
)
|
|
676
|
+
|
|
677
|
+
EMPLOYEE_TEAM_TEAM_ID.to_one # True: an employee has one team
|
|
678
|
+
TEAM_EMPLOYEE_TEAM_ID_REVERSE.to_one # False: a team has many employees
|
|
679
|
+
```
|
|
680
|
+
|
|
681
|
+
A relation is named after the table across the key and the columns that carry
|
|
682
|
+
it, with the reverse direction saying so. Nothing in the name depends on where
|
|
683
|
+
a relation falls in a list, so adding a key to one table cannot rename the
|
|
684
|
+
relations already generated for another. A key that points at its own table
|
|
685
|
+
produces both directions on that table, which is why the direction is always
|
|
686
|
+
part of the name rather than only when it would otherwise collide. Where a
|
|
687
|
+
constraint has a name of its own, that name is used instead.
|
|
688
|
+
|
|
689
|
+
Two keys that no name can tell apart are reported as a naming collision rather
|
|
690
|
+
than resolved by guessing, and naming the constraint in the database is the fix.
|
|
691
|
+
|
|
692
|
+
### Fetch plans
|
|
693
|
+
|
|
694
|
+
A fetch plan says which relations a query brings back and how, and it is read
|
|
695
|
+
back before anything runs:
|
|
696
|
+
|
|
697
|
+
```python
|
|
698
|
+
from pyoq.relations import FetchPlan, FetchRequest, FetchStrategy
|
|
699
|
+
|
|
700
|
+
plan = FetchPlan(
|
|
701
|
+
team_reference,
|
|
702
|
+
(FetchRequest(employees, FetchStrategy.NESTED, (FetchRequest(badge),)),),
|
|
703
|
+
)
|
|
704
|
+
print("\n".join(plan.describe()))
|
|
705
|
+
```
|
|
706
|
+
|
|
707
|
+
```console
|
|
708
|
+
employee [] via nested
|
|
709
|
+
badge via auto
|
|
710
|
+
```
|
|
711
|
+
|
|
712
|
+
Four strategies are available. `joined` brings a relation back in the same
|
|
713
|
+
query, `nested` brings a collection back as one correlated result, `select_in`
|
|
714
|
+
issues one further query per level, and `auto` leaves the choice to be made
|
|
715
|
+
against the dialect's capabilities when the query is planned, so one plan can
|
|
716
|
+
take the best route each database offers.
|
|
717
|
+
|
|
718
|
+
`resolve_fetch_plan()` turns that intent into a decision against what a dialect
|
|
719
|
+
can actually do, and records why:
|
|
720
|
+
|
|
721
|
+
```console
|
|
722
|
+
posts ordered by id via select-in: this dialect cannot order inside an
|
|
723
|
+
aggregate, and this collection's order matters
|
|
724
|
+
users via joined: a to-one relation is one row of the join
|
|
725
|
+
```
|
|
726
|
+
|
|
727
|
+
A fetch request says what a collection is ordered by with `by_column()` and
|
|
728
|
+
`by_expression()`, each of which takes a direction:
|
|
729
|
+
|
|
730
|
+
```python
|
|
731
|
+
from pyoq.relations import by_column, by_expression
|
|
732
|
+
|
|
733
|
+
FetchRequest(
|
|
734
|
+
employees,
|
|
735
|
+
order_by=(by_expression("lower(name)"), by_column("id", descending=True)),
|
|
736
|
+
)
|
|
737
|
+
```
|
|
738
|
+
|
|
739
|
+
A column is checked against the related table, because the plan can see the
|
|
740
|
+
schema. An expression is named rather than carried, because a plan is built
|
|
741
|
+
below the layer that writes queries and does not reach up into one. The name
|
|
742
|
+
is what the plan reports when it explains itself, and naming it is enough to
|
|
743
|
+
plan with: whether a collection is ordered at all is what decides how it can
|
|
744
|
+
be fetched, not what it is ordered by. The expression itself is given to
|
|
745
|
+
whatever carries the plan out.
|
|
746
|
+
|
|
747
|
+
A to-one relation rides the join, because it is one row the join already
|
|
748
|
+
carries. A collection is nested where the dialect can aggregate one, and falls
|
|
749
|
+
back to select-in where it cannot, or where the collection's order matters and
|
|
750
|
+
the dialect cannot order inside an aggregate. That last case is measured rather
|
|
751
|
+
than assumed: SQLite from 3.44 and PostgreSQL both accept an order inside their
|
|
752
|
+
aggregate, while MySQL rejects it outright, and MySQL discards an order given in
|
|
753
|
+
a derived table instead of honouring it. An aggregate in the wrong order is
|
|
754
|
+
worse than a second query, so PyOQ takes the second query.
|
|
755
|
+
|
|
756
|
+
A plan that names a strategy the dialect cannot carry out is refused rather than
|
|
757
|
+
quietly changed, because a caller who asked for an ordered nested collection and
|
|
758
|
+
received an unordered one has no way to notice.
|
|
759
|
+
|
|
760
|
+
`validate_fetch_plan()` checks a plan against the relation graph once, before
|
|
761
|
+
any query runs, so a mistake is reported against the schema rather than part-way
|
|
762
|
+
through a fetch. A relation the table does not have is refused, a nested
|
|
763
|
+
relation is checked against its own table rather than the root, and fetching one
|
|
764
|
+
relation twice at the same level is refused. The same relation may appear at
|
|
765
|
+
different levels, because a key pointing back is a different fetch rather than a
|
|
766
|
+
repeat, and that is also why plan depth is bounded.
|
|
767
|
+
|
|
768
|
+
### Relation loading state
|
|
769
|
+
|
|
770
|
+
A relation is loaded, known to be absent, or was never fetched. Rows are
|
|
771
|
+
detached values, so reading one never reaches a database, which is what makes
|
|
772
|
+
the third state necessary: without it a relation nobody asked for would be
|
|
773
|
+
indistinguishable from one that resolved to nothing.
|
|
774
|
+
|
|
775
|
+
```python
|
|
776
|
+
team.employees.value # the fetched rows
|
|
777
|
+
person.manager.value # None, when the key resolved to nothing
|
|
778
|
+
person.badge.value # raises: this relation was not fetched
|
|
779
|
+
```
|
|
780
|
+
|
|
781
|
+
Reading a relation nobody fetched raises rather than answering, because any
|
|
782
|
+
answer would be a guess and no answer can be produced without I/O.
|
|
783
|
+
|
|
784
|
+
Run the model against a reflected database with:
|
|
785
|
+
|
|
786
|
+
```console
|
|
787
|
+
python examples/relation_model.py
|
|
788
|
+
```
|
|
789
|
+
|
|
790
|
+
## Row identity
|
|
791
|
+
|
|
792
|
+
A join repeats a parent row once per child, and an outer join invents a child
|
|
793
|
+
made entirely of nulls. Both are answered by asking what identifies a row, so
|
|
794
|
+
identity is decided before any row is built:
|
|
795
|
+
|
|
796
|
+
```python
|
|
797
|
+
from pyoq.hydration import IdentityMap, row_key_for
|
|
798
|
+
|
|
799
|
+
key = row_key_for(table, reference, projection.position_of)
|
|
800
|
+
identity = key.identify(result_row) # None when the result holds no row here
|
|
801
|
+
```
|
|
802
|
+
|
|
803
|
+
A primary key identifies a table. Failing that, a unique key no part of which
|
|
804
|
+
is nullable. A table offering neither is identified by everything projected
|
|
805
|
+
from it, which is the only sound answer left: rows it cannot tell apart are
|
|
806
|
+
rows nobody can tell apart. A null among the identifying values means the
|
|
807
|
+
result holds no row there rather than a row whose identity happens to be null,
|
|
808
|
+
which is how an absent child is recognized.
|
|
809
|
+
|
|
810
|
+
Identity is by value however a driver spells it, so buffers and arrays compare
|
|
811
|
+
by their contents. A value no dictionary can hold, or a key column the query
|
|
812
|
+
does not select, is reported against the table it belongs to rather than
|
|
813
|
+
surfacing as a bare type or lookup error from inside a fetch.
|
|
814
|
+
|
|
815
|
+
`IdentityMap` builds each distinct row once. It lives for one fetch and is
|
|
816
|
+
discarded with it, so a row it returns can never be a stale row from an earlier
|
|
817
|
+
fetch.
|
|
818
|
+
|
|
819
|
+
## Hydration
|
|
820
|
+
|
|
821
|
+
A join hands back one row per combination, so the same parent arrives once per
|
|
822
|
+
child and the same child can arrive under several parents. `hydrate()` reads a
|
|
823
|
+
result back into rows that carry what the plan asked for:
|
|
824
|
+
|
|
825
|
+
```python
|
|
826
|
+
from pyoq.hydration import hydrate
|
|
827
|
+
|
|
828
|
+
teams = hydrate(node, result_rows)
|
|
829
|
+
```
|
|
830
|
+
|
|
831
|
+
Rows come back in the order the result first mentioned them, a row the result
|
|
832
|
+
mentions twice is built once, and children keep the order they first appeared
|
|
833
|
+
in. A null among a child's identifying values means the result holds no child
|
|
834
|
+
there, so an outer join yields an empty collection or an absent value rather
|
|
835
|
+
than a row made of nulls.
|
|
836
|
+
|
|
837
|
+
The result is read twice: once to group it and once to construct it. A frozen
|
|
838
|
+
row cannot be handed its children after it exists, so its children have to be
|
|
839
|
+
known before it is built.
|
|
840
|
+
|
|
841
|
+
One table read at two places in a plan yields separate rows, because each
|
|
842
|
+
carries different children and one row cannot hold both. Within one place, a
|
|
843
|
+
child several parents share is one object.
|
|
844
|
+
|
|
845
|
+
A to-one relation that returns several distinct rows for one parent is
|
|
846
|
+
reported. The schema said those columns were unique and the result disagreed,
|
|
847
|
+
and discarding rows would hide the disagreement.
|
|
848
|
+
|
|
849
|
+
Throughput and memory are held to a contract:
|
|
850
|
+
|
|
851
|
+
```console
|
|
852
|
+
python -m benchmarks.hydration
|
|
853
|
+
```
|
|
854
|
+
|
|
855
|
+
## Query diagnostics
|
|
856
|
+
|
|
857
|
+
A query issued once per row of a previous result is invisible from inside the
|
|
858
|
+
loop that causes it. Recording what a scope executed makes it visible from
|
|
859
|
+
outside:
|
|
860
|
+
|
|
861
|
+
```python
|
|
862
|
+
from pyoq.diagnostics import QueryObserver
|
|
863
|
+
|
|
864
|
+
observer = QueryObserver()
|
|
865
|
+
...
|
|
866
|
+
for repeated in observer.repeated():
|
|
867
|
+
print(repeated.describe())
|
|
868
|
+
```
|
|
869
|
+
|
|
870
|
+
```console
|
|
871
|
+
4 executions of SELECT id FROM person WHERE team_id = ? from app/teams.py:31 in members
|
|
872
|
+
```
|
|
873
|
+
|
|
874
|
+
One query is recognized across the many times it is executed. Its bound values
|
|
875
|
+
are already placeholders, a list of any length collapses to one, a write chunked
|
|
876
|
+
to fit a parameter limit collapses however many chunks it took, and the three
|
|
877
|
+
placeholder styles reach the same shape, so one query compiled for three
|
|
878
|
+
dialects is one shape. Placeholders that are not a list stay apart, so a query
|
|
879
|
+
with two conditions never merges into one with a single condition.
|
|
880
|
+
|
|
881
|
+
A report carries the first place outside PyOQ that issued the query, which is
|
|
882
|
+
the caller's own code rather than the library's. An interpreter that offers no
|
|
883
|
+
frames reports no place rather than refusing to run.
|
|
884
|
+
|
|
885
|
+
Diagnostics are safe to log. A shape carries no bound value, so nothing a query
|
|
886
|
+
was asked about appears in a report, and that holds whether or not a parameter
|
|
887
|
+
was marked sensitive.
|
|
888
|
+
|
|
889
|
+
Diagnostics must not become the thing that exhausts a process, so the number of
|
|
890
|
+
distinct shapes and the number of places per shape are both capped. Executions
|
|
891
|
+
keep being counted after the caps are reached, and shapes seen beyond the cap
|
|
892
|
+
are counted through `unrecorded_shapes`.
|
|
893
|
+
|
|
894
|
+
An observer is safe to share. A pool hands connections to whichever thread asks,
|
|
895
|
+
so reading a report while several threads record must not fail.
|
|
896
|
+
|
|
897
|
+
### Query budgets
|
|
898
|
+
|
|
899
|
+
A repeated query that is only reported is a repeated query that still ships. A
|
|
900
|
+
budget turns the same observation into a refusal:
|
|
901
|
+
|
|
902
|
+
```python
|
|
903
|
+
from pyoq.diagnostics import QueryBudget, QueryScope
|
|
904
|
+
|
|
905
|
+
scope = QueryScope(QueryBudget(maximum_repeats=3))
|
|
906
|
+
```
|
|
907
|
+
|
|
908
|
+
```console
|
|
909
|
+
one statement ran 4 times in this scope, beyond the 3 it was allowed
|
|
910
|
+
from app/teams.py:31 in members: SELECT id FROM person WHERE team_id = ?
|
|
911
|
+
```
|
|
912
|
+
|
|
913
|
+
`maximum_repeats` is the one that catches an N+1 access, because the shape
|
|
914
|
+
executed once per row of an earlier result is the shape that repeats.
|
|
915
|
+
`maximum_queries` caps the scope as a whole. A refusal names the shape, the
|
|
916
|
+
count, and the place, and carries no bound value. The statement that caused it
|
|
917
|
+
is recorded before it is judged, so a report taken afterwards explains the
|
|
918
|
+
refusal.
|
|
919
|
+
|
|
920
|
+
Holding a budget costs almost nothing per statement, because only the shape
|
|
921
|
+
just recorded is judged rather than every shape a scope has seen.
|
|
922
|
+
|
|
923
|
+
### Select-in fetching
|
|
924
|
+
|
|
925
|
+
Select-in is the strategy that turns an N+1 access into a bounded number of
|
|
926
|
+
statements. `select_in_conditions()` produces one predicate per batch:
|
|
927
|
+
|
|
928
|
+
```python
|
|
929
|
+
from pyoq.fetching import select_in_conditions
|
|
930
|
+
|
|
931
|
+
for condition in select_in_conditions(
|
|
932
|
+
relation, columns, parent_keys, maximum_parameters=999
|
|
933
|
+
):
|
|
934
|
+
rows = database.many(select(...).from_(source).where(condition))
|
|
935
|
+
```
|
|
936
|
+
|
|
937
|
+
A key of one column becomes a membership test. A key of several becomes a
|
|
938
|
+
choice between equalities rather than a row value, because every dialect
|
|
939
|
+
renders the first alike and they do not all render the second alike.
|
|
940
|
+
|
|
941
|
+
The predicate is built from expression nodes rather than through the typed
|
|
942
|
+
query API. Keys come back from a database as values whose types the caller has
|
|
943
|
+
already checked, so routing them through a typed column would fight the column's
|
|
944
|
+
own type without protecting the query the caller wrote.
|
|
945
|
+
|
|
946
|
+
A contract proves the reduction on SQLite, PostgreSQL, and MySQL alike: the same
|
|
947
|
+
rows come back from far fewer statements than asking once per parent.
|
|
948
|
+
|
|
949
|
+
### Joined fetching
|
|
950
|
+
|
|
951
|
+
A to-one relation is one row the join already carries, so it costs no extra
|
|
952
|
+
statement. What it costs is knowing where each table's values land:
|
|
953
|
+
|
|
954
|
+
```python
|
|
955
|
+
from pyoq.fetching import join_condition, projection_layout
|
|
956
|
+
|
|
957
|
+
child_at, parent_at = projection_layout((2, 2))
|
|
958
|
+
query = (
|
|
959
|
+
select(*child_columns, *parent_columns)
|
|
960
|
+
.from_(table_source(children))
|
|
961
|
+
.left_join(table_source(parents))
|
|
962
|
+
.on(join_condition(child_key, parent_key))
|
|
963
|
+
)
|
|
964
|
+
```
|
|
965
|
+
|
|
966
|
+
Hydration reads a result by position, so the layout a projection is laid out
|
|
967
|
+
with and the layout it is read back by have to be the same one. Deriving both
|
|
968
|
+
from `projection_layout()` keeps them from drifting apart as a plan grows.
|
|
969
|
+
|
|
970
|
+
A left join leaves a to-one relation absent where nothing matched, which the
|
|
971
|
+
relation states already distinguish from a relation nobody fetched. This is
|
|
972
|
+
proved on SQLite, PostgreSQL, and MySQL alike: one query, hydrated back into
|
|
973
|
+
parents carrying their relation, with the unmatched half absent.
|
|
974
|
+
|
|
975
|
+
### Nested collections
|
|
976
|
+
|
|
977
|
+
A nested collection gathers a relation's rows into one value beside their
|
|
978
|
+
parent, so one statement returns the graph:
|
|
979
|
+
|
|
980
|
+
```console
|
|
981
|
+
SELECT "parent"."id",
|
|
982
|
+
(SELECT COALESCE(json_group_array(json_object('id', "child"."id")
|
|
983
|
+
ORDER BY "child"."id"), json_array())
|
|
984
|
+
FROM "child" WHERE ("child"."parent_id" = "parent"."id"))
|
|
985
|
+
FROM "parent"
|
|
986
|
+
```
|
|
987
|
+
|
|
988
|
+
Each dialect gathers rows with its own functions: `json_group_array` and
|
|
989
|
+
`json_object` on SQLite, `jsonb_agg` and `jsonb_build_object` on PostgreSQL,
|
|
990
|
+
`JSON_ARRAYAGG` and `JSON_OBJECT` on MySQL. An empty relation is coalesced to an
|
|
991
|
+
empty collection rather than left null, so a caller need not tell a relation
|
|
992
|
+
with no rows from a fetch that did not happen.
|
|
993
|
+
|
|
994
|
+
MySQL refuses an ordered collection rather than returning one in the wrong
|
|
995
|
+
order, and says to fetch it with select-in instead. Ordering inside a SQLite
|
|
996
|
+
aggregate needs SQLite 3.44 or later.
|
|
997
|
+
|
|
998
|
+
A collection reads one table, because a collection is the rows of one relation
|
|
999
|
+
and anything wider is a query rather than a relation.
|
|
1000
|
+
|
|
1001
|
+
`decode_collection()` reads the value back into rows. What a driver hands over
|
|
1002
|
+
differs: SQLite and MySQL return JSON text while PostgreSQL returns a list it
|
|
1003
|
+
has already decoded, and both arrive at the same rows. Values come back in the
|
|
1004
|
+
order their names were given, so a caller reads them like any other row.
|
|
1005
|
+
|
|
1006
|
+
Nothing is trusted to hold what it promised. A collection that is not a list of
|
|
1007
|
+
objects is refused, and a row missing a value it was asked for is reported by
|
|
1008
|
+
name and position. A value that is null is a value, because a column that held
|
|
1009
|
+
null is not a column that was missing. Values a query did not ask for are left
|
|
1010
|
+
alone, so a table gaining a column does not break a fetch.
|
|
1011
|
+
|
|
1012
|
+
`hydrate_collection()` turns that into a loaded relation. An empty collection is
|
|
1013
|
+
loaded and holds nothing, which is not the same as a relation nobody fetched.
|
|
1014
|
+
|
|
1015
|
+
### Relation batching
|
|
1016
|
+
|
|
1017
|
+
An N+1 access asks for one parent's children at a time. Gathering the keys first
|
|
1018
|
+
turns that into one question, and a dialect's parameter limit turns it into a
|
|
1019
|
+
bounded number of questions rather than one per parent:
|
|
1020
|
+
|
|
1021
|
+
```python
|
|
1022
|
+
from pyoq.relations import RelationBatchLoader
|
|
1023
|
+
|
|
1024
|
+
loader = RelationBatchLoader()
|
|
1025
|
+
for team in teams:
|
|
1026
|
+
loader.enqueue(members, team.key)
|
|
1027
|
+
batches = loader.batches(maximum_parameters=999)
|
|
1028
|
+
```
|
|
1029
|
+
|
|
1030
|
+
A key spanning several columns costs one bound value per column, so a limit
|
|
1031
|
+
counts values rather than keys, and a key wider than the limit is refused rather
|
|
1032
|
+
than split. Keys keep their order and never repeat. A key that is null reaches
|
|
1033
|
+
nothing, so it is counted through `skipped_keys` rather than asked about, which
|
|
1034
|
+
would cost a parameter to learn nothing.
|
|
1035
|
+
|
|
1036
|
+
A loader belongs to one scope, and each relation it holds is bounded on its own.
|
|
1037
|
+
|
|
1038
|
+
## One chain, from building to reading
|
|
1039
|
+
|
|
1040
|
+
A builder describes a statement and never reaches a database. That is what makes
|
|
1041
|
+
one safe to share, cache, and test, and it does not change. A context adds the
|
|
1042
|
+
other half: it holds the database to run against and hands back builders that
|
|
1043
|
+
carry it, so a chain can end in a fetch instead of being handed to an executor.
|
|
1044
|
+
|
|
1045
|
+
```python
|
|
1046
|
+
from pyoq.dsl import using
|
|
1047
|
+
|
|
1048
|
+
dsl = using(database)
|
|
1049
|
+
|
|
1050
|
+
dsl.select(TITLE_ID, TITLE_NAME, TITLE_PRICE) \
|
|
1051
|
+
.from_(table_source(TITLES)) \
|
|
1052
|
+
.where(TITLE_PRICE.gt(Decimal("10.00"))) \
|
|
1053
|
+
.order_by(TITLE_ID) \
|
|
1054
|
+
.fetch_all() \
|
|
1055
|
+
.to_list()
|
|
1056
|
+
```
|
|
1057
|
+
|
|
1058
|
+
A generated table names its own columns, so a query over all of them does not
|
|
1059
|
+
have to list them:
|
|
1060
|
+
|
|
1061
|
+
```python
|
|
1062
|
+
dsl.select_from(TITLES) \
|
|
1063
|
+
.where(TITLE_PRICE.gt(Decimal("10.00"))) \
|
|
1064
|
+
.order_by(TITLE_ID) \
|
|
1065
|
+
.fetch_all() \
|
|
1066
|
+
.into(TitleRow)
|
|
1067
|
+
```
|
|
1068
|
+
|
|
1069
|
+
The SQL is `SELECT *`. PyOQ still carries the column list, so the row keeps
|
|
1070
|
+
its names and each value is read back as the type the column holds, and that
|
|
1071
|
+
order is the order the generated row type takes its arguments in. A package
|
|
1072
|
+
that no longer matches its schema is refused rather than read by the wrong
|
|
1073
|
+
columns.
|
|
1074
|
+
|
|
1075
|
+
The shape a caller wants is asked for at the end rather than assembled around
|
|
1076
|
+
the call:
|
|
1077
|
+
|
|
1078
|
+
```python
|
|
1079
|
+
rows = dsl.select(...).from_(...).fetch_all()
|
|
1080
|
+
|
|
1081
|
+
rows.to_list() # [(1, "A Guide", Decimal("12.50")), ...]
|
|
1082
|
+
rows.into(Title) # [Title(1, "A Guide", Decimal("12.50")), ...]
|
|
1083
|
+
rows.map(lambda r: r[1])
|
|
1084
|
+
rows.to_dicts() # [{"id": 1, "name": "A Guide", ...}, ...]
|
|
1085
|
+
rows.to_dict() # one row keyed, or an error saying which way it failed
|
|
1086
|
+
rows.one() # exactly one row, or an error saying which way it failed
|
|
1087
|
+
rows.first() # the row at the front, or None
|
|
1088
|
+
```
|
|
1089
|
+
|
|
1090
|
+
A chain that ends in one row asks the same questions of it. The row is a
|
|
1091
|
+
tuple, so it still unpacks, indexes, and compares the way it always did:
|
|
1092
|
+
|
|
1093
|
+
```python
|
|
1094
|
+
row = dsl.select(TITLE_ID, TITLE_NAME).from_(...).fetch_one()
|
|
1095
|
+
|
|
1096
|
+
row.to_dict() # {"id": 1, "name": "A Guide"}
|
|
1097
|
+
row.to_list() # [1, "A Guide"]
|
|
1098
|
+
row.to_tuple() # (1, "A Guide"), typed as it was selected
|
|
1099
|
+
row.into(Title) # Title(1, "A Guide")
|
|
1100
|
+
identifier, name = row # still a tuple
|
|
1101
|
+
```
|
|
1102
|
+
|
|
1103
|
+
A row that may be missing is the one shape that had to be tested for before
|
|
1104
|
+
it could be built, so the chain takes the type instead:
|
|
1105
|
+
|
|
1106
|
+
```python
|
|
1107
|
+
dsl.select(...).where(...).fetch_optional_into(Title) # Title | None
|
|
1108
|
+
```
|
|
1109
|
+
|
|
1110
|
+
`into` spreads each row across a constructor in the order it was selected, so a
|
|
1111
|
+
dataclass or a named tuple needs nothing else. `to_dicts` keys by the name each
|
|
1112
|
+
column came back under, and refuses when a column has none rather than inventing
|
|
1113
|
+
one.
|
|
1114
|
+
|
|
1115
|
+
A query can end in a single row or a single value instead:
|
|
1116
|
+
|
|
1117
|
+
```python
|
|
1118
|
+
dsl.select(TITLE_NAME).from_(source).where(TITLE_ID.eq(1)).fetch_one()
|
|
1119
|
+
dsl.select(TITLE_NAME).from_(source).where(TITLE_ID.eq(1)).fetch_optional()
|
|
1120
|
+
dsl.select(TITLE_NAME).from_(source).where(TITLE_ID.eq(1)).fetch_value()
|
|
1121
|
+
```
|
|
1122
|
+
|
|
1123
|
+
`fetch_value` is only offered to a query that selected one column. That is a
|
|
1124
|
+
statement about the type rather than a check when it runs.
|
|
1125
|
+
|
|
1126
|
+
Writes chain the same way and end in the number of rows the database wrote:
|
|
1127
|
+
|
|
1128
|
+
```python
|
|
1129
|
+
dsl.insert_into(TITLES, TITLE_ID, TITLE_NAME).values(1, "A Guide").execute()
|
|
1130
|
+
dsl.update(TITLES).set(TITLE_PRICE, Decimal("20.00")).where(TITLE_ID.eq(1)).execute()
|
|
1131
|
+
dsl.delete_from(TITLES).where(TITLE_ID.eq(1)).execute()
|
|
1132
|
+
|
|
1133
|
+
dsl.insert_into(TITLES, TITLE_ID, TITLE_NAME) \
|
|
1134
|
+
.values(1, "A Guide") \
|
|
1135
|
+
.on_conflict_do_update(TITLE_ID) \
|
|
1136
|
+
.set(TITLE_NAME, "A Guide") \
|
|
1137
|
+
.returning(TITLE_ID, TITLE_NAME) \
|
|
1138
|
+
.fetch_all() \
|
|
1139
|
+
.into(Title)
|
|
1140
|
+
```
|
|
1141
|
+
|
|
1142
|
+
A set operation orders by a column it selected, because its operands may read
|
|
1143
|
+
different tables and a table-qualified name means nothing once they are
|
|
1144
|
+
combined. Naming anything else there is refused rather than sent to a server
|
|
1145
|
+
that would reject it.
|
|
1146
|
+
|
|
1147
|
+
Types are carried the whole way. `select(TITLE_ID, TITLE_NAME, TITLE_PRICE)`
|
|
1148
|
+
gives a chain over `tuple[int, str, Decimal]`, `fetch_all().to_list()` is a
|
|
1149
|
+
`list[tuple[int, str, Decimal]]`, and `fetch_one()` is the tuple itself.
|
|
1150
|
+
|
|
1151
|
+
Unwrapping a bound query with `query`, or a bound write with `statement`, gives
|
|
1152
|
+
back the ordinary builder, which cannot reach a database. Anything built that
|
|
1153
|
+
way still runs through `dsl.fetch(...)` or `dsl.execute(...)`.
|
|
1154
|
+
|
|
1155
|
+
### An asynchronous database
|
|
1156
|
+
|
|
1157
|
+
`using` answers the same question either way, so an asynchronous database gives
|
|
1158
|
+
an asynchronous context. Only the ending changes:
|
|
1159
|
+
|
|
1160
|
+
```python
|
|
1161
|
+
rows = await (
|
|
1162
|
+
using(database)
|
|
1163
|
+
.select(TITLE_ID, TITLE_NAME)
|
|
1164
|
+
.from_(table_source(TITLES))
|
|
1165
|
+
.order_by(TITLE_ID)
|
|
1166
|
+
.fetch_all()
|
|
1167
|
+
)
|
|
1168
|
+
rows.into(Title)
|
|
1169
|
+
|
|
1170
|
+
await using(database).insert_into(TITLES, TITLE_ID).values(1).execute()
|
|
1171
|
+
```
|
|
1172
|
+
|
|
1173
|
+
The chain is built the same way and describes the same statement. `fetch_all`,
|
|
1174
|
+
`fetch_one`, `fetch_optional`, `fetch_value`, and `execute` are awaited because
|
|
1175
|
+
the database is, and the types carry through exactly as they do synchronously.
|
|
1176
|
+
|
|
1177
|
+
## Django
|
|
1178
|
+
|
|
1179
|
+
Django 4.2 through 6.x, on any Python each of them supports.
|
|
1180
|
+
|
|
1181
|
+
[`docs/django.md`](docs/django.md) is a step by step guide from an empty project
|
|
1182
|
+
to typed reads, writes, transactions, policies, and events. What follows is the
|
|
1183
|
+
reference.
|
|
1184
|
+
|
|
1185
|
+
PyOQ runs on the connection Django already has open:
|
|
1186
|
+
|
|
1187
|
+
```python
|
|
1188
|
+
from pyoq.django import DjangoOperations
|
|
1189
|
+
|
|
1190
|
+
database = DjangoOperations() # or DjangoOperations("reporting")
|
|
1191
|
+
rows = database.many(select(...).from_(source))
|
|
1192
|
+
```
|
|
1193
|
+
|
|
1194
|
+
```console
|
|
1195
|
+
python -m pip install "pyoq-sql[django]"
|
|
1196
|
+
```
|
|
1197
|
+
|
|
1198
|
+
No pool is opened beside Django's. Two pools against one database is two views
|
|
1199
|
+
of what has been committed, and the point of running inside Django is that there
|
|
1200
|
+
is one. The connection is resolved for each statement rather than held, because
|
|
1201
|
+
Django gives each thread a different one and closes them between requests.
|
|
1202
|
+
|
|
1203
|
+
The dialect comes from the connection rather than from the caller, since it is
|
|
1204
|
+
the connection that is on the socket. Values are adapted the same way they are
|
|
1205
|
+
on PyOQ's own connections, because a backend does not care whose connection it
|
|
1206
|
+
is: a decimal or a date still has to reach it in the shape its driver accepts.
|
|
1207
|
+
A backend PyOQ has no dialect for is refused by name.
|
|
1208
|
+
|
|
1209
|
+
A timeout is applied where the backend has a setting for one, and the session is
|
|
1210
|
+
left exactly as it was found. Inside a transaction the setting is local, so the
|
|
1211
|
+
database reverts it when the transaction ends; outside one, the previous value is
|
|
1212
|
+
read first and put back after. A restore never replaces the failure that made it
|
|
1213
|
+
necessary, because a statement that failed can leave a connection unable to
|
|
1214
|
+
accept the next one. SQLite has no such setting, and MySQL's covers reads only,
|
|
1215
|
+
so a caller is given what the backend can honour rather than a promise it
|
|
1216
|
+
cannot keep. A caller's cancellation is honoured before and after a statement
|
|
1217
|
+
regardless.
|
|
1218
|
+
|
|
1219
|
+
Routing is a question about a model, because that is the only thing a Django
|
|
1220
|
+
router is given to decide on:
|
|
1221
|
+
|
|
1222
|
+
```python
|
|
1223
|
+
database = DjangoOperations.for_model(Report) # wherever this is routed
|
|
1224
|
+
database = DjangoOperations.for_model(Report, write=True)
|
|
1225
|
+
```
|
|
1226
|
+
|
|
1227
|
+
A caller without a model names the alias instead.
|
|
1228
|
+
|
|
1229
|
+
### Transactions
|
|
1230
|
+
|
|
1231
|
+
PyOQ takes part in the transaction Django opened and never opens one of its own.
|
|
1232
|
+
There is no commit or rollback on `DjangoOperations`, because the block that
|
|
1233
|
+
starts a transaction is the one that ends it:
|
|
1234
|
+
|
|
1235
|
+
```python
|
|
1236
|
+
with database.atomic():
|
|
1237
|
+
database.execute(insert_into(REPORTS, name).values("quarterly"))
|
|
1238
|
+
database.on_commit(lambda: notify("saved"))
|
|
1239
|
+
```
|
|
1240
|
+
|
|
1241
|
+
Ask the database for its own block rather than reaching for Django's. Django's
|
|
1242
|
+
`transaction.atomic()` covers the default alias unless told which to use, so a
|
|
1243
|
+
caller working against another database inside a bare block has no transaction
|
|
1244
|
+
there at all and nothing reports it. `database.atomic()` always means this
|
|
1245
|
+
database. The same holds for `database.on_commit()`, which waits for this
|
|
1246
|
+
database to commit.
|
|
1247
|
+
|
|
1248
|
+
Nesting a block is a savepoint, and `atomic(savepoint=False)`,
|
|
1249
|
+
`atomic(durable=True)`, and `on_commit(robust=True)` behave as Django defines
|
|
1250
|
+
them. `database.in_transaction` reports whether a transaction is open.
|
|
1251
|
+
|
|
1252
|
+
Test isolation needs nothing extra. A Django test case wraps its test in a block
|
|
1253
|
+
and rolls it back, and PyOQ writes through the connection that block belongs to.
|
|
1254
|
+
|
|
1255
|
+
An async view reaches a statement through Django's own bridging:
|
|
1256
|
+
|
|
1257
|
+
```python
|
|
1258
|
+
await sync_to_async(read_reports, thread_sensitive=True)(database)
|
|
1259
|
+
```
|
|
1260
|
+
|
|
1261
|
+
Django refuses a synchronous statement called directly from a coroutine, and
|
|
1262
|
+
PyOQ inherits that guard. Nothing here shares transaction state across the
|
|
1263
|
+
boundary, because a transaction belongs to the connection that a thread holds.
|
|
1264
|
+
|
|
1265
|
+
### Generating from migrations
|
|
1266
|
+
|
|
1267
|
+
Add PyOQ to the project and tell it where generated code belongs:
|
|
1268
|
+
|
|
1269
|
+
```python
|
|
1270
|
+
INSTALLED_APPS = ["pyoq.django", ...]
|
|
1271
|
+
|
|
1272
|
+
PYOQ = {
|
|
1273
|
+
"codegen_directory": "app/_generated",
|
|
1274
|
+
"codegen_package": "app._generated",
|
|
1275
|
+
}
|
|
1276
|
+
```
|
|
1277
|
+
|
|
1278
|
+
```bash
|
|
1279
|
+
python manage.py pyoq_codegen
|
|
1280
|
+
python manage.py pyoq_codegen catalog --database reporting
|
|
1281
|
+
python manage.py pyoq_codegen --dry-run
|
|
1282
|
+
python manage.py pyoq_codegen --check
|
|
1283
|
+
```
|
|
1284
|
+
|
|
1285
|
+
The schema comes from the migration files, not from a database. Migrations are
|
|
1286
|
+
what the schema is going to be, and a developer generating types has usually
|
|
1287
|
+
not applied them yet. Nothing connects: Django answers what column a field
|
|
1288
|
+
takes on a backend without consulting a server, so the alias selects the
|
|
1289
|
+
dialect rather than opening a socket.
|
|
1290
|
+
|
|
1291
|
+
What that yields is the schema the database will really have, which is not
|
|
1292
|
+
always the one the models appear to describe:
|
|
1293
|
+
|
|
1294
|
+
- A Django default is applied in Python, so the column has no default. PyOQ
|
|
1295
|
+
says so rather than inviting an insert the database would refuse. The model's
|
|
1296
|
+
default is kept as column metadata.
|
|
1297
|
+
- `on_delete` is carried out by Django, and the constraint it writes names no
|
|
1298
|
+
action at all. The relation records what the database will do, with the
|
|
1299
|
+
model's choice beside it.
|
|
1300
|
+
- A unique constraint carrying a condition is not a key. It is unique among the
|
|
1301
|
+
rows it matches and not among the others, and treating it as a key would
|
|
1302
|
+
describe a relation reaching many rows as reaching one.
|
|
1303
|
+
- A many-to-many field owns a table that migration state never lists, so PyOQ
|
|
1304
|
+
describes it from the field. A through model the project wrote is already a
|
|
1305
|
+
model and is left alone.
|
|
1306
|
+
|
|
1307
|
+
Generation itself is the same pipeline every other entry point uses, so
|
|
1308
|
+
staging, validation, the manifest, drift detection, and the atomic replacement
|
|
1309
|
+
behave identically here.
|
|
1310
|
+
|
|
1311
|
+
To generate after every migration, put PyOQ ahead of the app that supplies the
|
|
1312
|
+
original command and ask for it:
|
|
1313
|
+
|
|
1314
|
+
```python
|
|
1315
|
+
PYOQ = {..., "codegen_after_makemigrations": True}
|
|
1316
|
+
```
|
|
1317
|
+
|
|
1318
|
+
Django's own command runs first and unchanged. A run that writes no migration,
|
|
1319
|
+
such as `--dry-run` or `--check`, generates nothing: the files on disk still
|
|
1320
|
+
describe the schema before the change, and generating from them would look like
|
|
1321
|
+
it had worked.
|
|
1322
|
+
|
|
1323
|
+
### A project to read
|
|
1324
|
+
|
|
1325
|
+
[`examples/django_project`](examples/django_project) is a project laid out the
|
|
1326
|
+
way a real one is: settings, a URLconf, WSGI and ASGI entry points, the admin,
|
|
1327
|
+
two applications with models and migrations, a template, and views. It has two
|
|
1328
|
+
databases and a router, and its models are chosen to show what reaches the
|
|
1329
|
+
database and what stays in Python.
|
|
1330
|
+
|
|
1331
|
+
[`examples/django_integration.py`](examples/django_integration.py) applies its
|
|
1332
|
+
migrations, generates from them, writes through the ORM and PyOQ on the same
|
|
1333
|
+
connection, and then exercises the project through its own views, including an
|
|
1334
|
+
async one, along with transactions, savepoints, and commit hooks:
|
|
1335
|
+
|
|
1336
|
+
```bash
|
|
1337
|
+
python examples/django_integration.py
|
|
1338
|
+
```
|
|
1339
|
+
|
|
1340
|
+
## FastAPI
|
|
1341
|
+
|
|
1342
|
+
The pool belongs to the application, so the lifespan opens it once and gives it
|
|
1343
|
+
back once:
|
|
1344
|
+
|
|
1345
|
+
```python
|
|
1346
|
+
from pyoq.fastapi import SyncDatabase, database_lifespan
|
|
1347
|
+
|
|
1348
|
+
def open_notes() -> SyncDatabase:
|
|
1349
|
+
pool = SQLitePool(SQLiteConnectionFactory(path))
|
|
1350
|
+
return SyncDatabase(SQLiteExecutor(pool), pool.close)
|
|
1351
|
+
|
|
1352
|
+
app = FastAPI(lifespan=database_lifespan(synchronous={"default": open_notes}))
|
|
1353
|
+
```
|
|
1354
|
+
|
|
1355
|
+
A database that fails to open takes the application down, and the ones already
|
|
1356
|
+
open are given back first. Starting up halfway and serving requests against a
|
|
1357
|
+
partly opened application is worse than not starting.
|
|
1358
|
+
|
|
1359
|
+
Routes ask for what they need:
|
|
1360
|
+
|
|
1361
|
+
```python
|
|
1362
|
+
import pyoq.fastapi as pyoq_fastapi
|
|
1363
|
+
|
|
1364
|
+
Database = Annotated[QueryOperations, Depends(pyoq_fastapi.database())]
|
|
1365
|
+
Transaction = Annotated[QueryOperations, Depends(pyoq_fastapi.transaction())]
|
|
1366
|
+
|
|
1367
|
+
@app.get("/notes")
|
|
1368
|
+
def list_notes(database: Database) -> dict[str, object]:
|
|
1369
|
+
return {"notes": using(database).select(...).fetch_all().to_dicts()}
|
|
1370
|
+
|
|
1371
|
+
@app.post("/notes")
|
|
1372
|
+
def add_note(database: Transaction) -> dict[str, int]:
|
|
1373
|
+
return {"written": using(database).insert_into(...).values(...).execute()}
|
|
1374
|
+
```
|
|
1375
|
+
|
|
1376
|
+
A request that returns commits. A request that raises rolls back, and so does
|
|
1377
|
+
one the client gave up on: FastAPI tears the dependency down either way, and the
|
|
1378
|
+
transaction is told what tore it down. An `HTTPException` counts as raising,
|
|
1379
|
+
because a refused request is still a refused request.
|
|
1380
|
+
|
|
1381
|
+
Asking twice for the same dependency gives back the same object, so an override
|
|
1382
|
+
matches:
|
|
1383
|
+
|
|
1384
|
+
```python
|
|
1385
|
+
app.dependency_overrides[pyoq_fastapi.database()] = lambda: substitute
|
|
1386
|
+
```
|
|
1387
|
+
|
|
1388
|
+
### Which database a route may use
|
|
1389
|
+
|
|
1390
|
+
`database()` and `transaction()` are synchronous. `async_database()` and
|
|
1391
|
+
`async_transaction()` are their counterparts.
|
|
1392
|
+
|
|
1393
|
+
A synchronous transaction belongs to a synchronous route. FastAPI runs a
|
|
1394
|
+
synchronous dependency in a worker thread and an async route body on the event
|
|
1395
|
+
loop, and a transaction belongs to the thread that opened it. PyOQ refuses the
|
|
1396
|
+
mismatch rather than running a statement somewhere it does not belong, so an
|
|
1397
|
+
async route takes `async_transaction()`.
|
|
1398
|
+
|
|
1399
|
+
### Request budgets
|
|
1400
|
+
|
|
1401
|
+
A budget is per request, so one request cannot spend another's allowance:
|
|
1402
|
+
|
|
1403
|
+
```python
|
|
1404
|
+
Budgeted = Annotated[
|
|
1405
|
+
QueryOperations,
|
|
1406
|
+
Depends(pyoq_fastapi.scoped(QueryBudget(maximum_queries=20))),
|
|
1407
|
+
]
|
|
1408
|
+
```
|
|
1409
|
+
|
|
1410
|
+
Every read and write is recorded before the driver sees it, so the statement
|
|
1411
|
+
that goes beyond the allowance never reaches the database. `maximum_repeats` is
|
|
1412
|
+
the one that catches a query running once per row of an earlier result.
|
|
1413
|
+
|
|
1414
|
+
`ScopedOperations` and `AsyncScopedOperations` do the counting and are not
|
|
1415
|
+
specific to FastAPI. Either wraps any database:
|
|
1416
|
+
|
|
1417
|
+
```python
|
|
1418
|
+
from pyoq.diagnostics import QueryScope, ScopedOperations
|
|
1419
|
+
|
|
1420
|
+
scope = QueryScope(QueryBudget(maximum_queries=20))
|
|
1421
|
+
counted = ScopedOperations(database, scope)
|
|
1422
|
+
```
|
|
1423
|
+
|
|
1424
|
+
See [`examples/fastapi_application.py`](examples/fastapi_application.py) for a
|
|
1425
|
+
runnable application covering all of it.
|
|
1426
|
+
|
|
1427
|
+
## Sanic
|
|
1428
|
+
|
|
1429
|
+
Sanic starts its workers as separate processes. A connection made before they
|
|
1430
|
+
exist would be shared by all of them, and two processes talking over one socket
|
|
1431
|
+
corrupt each other's results. So every pool is opened from the listener that
|
|
1432
|
+
runs inside each worker:
|
|
1433
|
+
|
|
1434
|
+
```python
|
|
1435
|
+
from pyoq.sanic import AsyncDatabase, attach_databases
|
|
1436
|
+
|
|
1437
|
+
async def open_notes() -> AsyncDatabase:
|
|
1438
|
+
pool = await PostgresAsyncPool(PostgresAsyncConnectionFactory(dsn)).open()
|
|
1439
|
+
return AsyncDatabase(PostgresAsyncExecutor(pool), pool.close)
|
|
1440
|
+
|
|
1441
|
+
attach_databases(app, asynchronous={"default": open_notes})
|
|
1442
|
+
```
|
|
1443
|
+
|
|
1444
|
+
A database that fails to open takes the worker down, and the ones already open
|
|
1445
|
+
are given back first. Every one a worker opened is closed when it stops.
|
|
1446
|
+
|
|
1447
|
+
### One transaction per request
|
|
1448
|
+
|
|
1449
|
+
A transaction belongs to the block the handler enters:
|
|
1450
|
+
|
|
1451
|
+
```python
|
|
1452
|
+
import pyoq.sanic as pyoq_sanic
|
|
1453
|
+
|
|
1454
|
+
@app.post("/notes")
|
|
1455
|
+
async def add_note(request):
|
|
1456
|
+
async with pyoq_sanic.transaction(request) as database:
|
|
1457
|
+
...
|
|
1458
|
+
```
|
|
1459
|
+
|
|
1460
|
+
Leaving the block normally commits. Leaving it any other way rolls back: a
|
|
1461
|
+
handler that raised, and a request the server abandoned.
|
|
1462
|
+
|
|
1463
|
+
It is a block rather than a pair of middlewares for a reason. A server that is
|
|
1464
|
+
told the client has gone cancels the task it called the application on, and
|
|
1465
|
+
nothing after the handler runs, response middleware included. A transaction
|
|
1466
|
+
ended there would keep its connection for as long as the worker lives, inside an
|
|
1467
|
+
open transaction, holding whatever it locked. Python ends a block whatever
|
|
1468
|
+
happens to the task running it, so the connection goes back to the pool.
|
|
1469
|
+
|
|
1470
|
+
A commit that fails raises on the way out of the block, which Sanic answers the
|
|
1471
|
+
way it answers any other failure in a handler.
|
|
1472
|
+
|
|
1473
|
+
`synchronous_transaction` is the same for a database that is not awaited.
|
|
1474
|
+
|
|
1475
|
+
### Request budgets
|
|
1476
|
+
|
|
1477
|
+
```python
|
|
1478
|
+
pyoq_sanic.attach_request_budget(app, QueryBudget(maximum_queries=20))
|
|
1479
|
+
|
|
1480
|
+
@app.get("/notes")
|
|
1481
|
+
async def list_notes(request):
|
|
1482
|
+
database = pyoq_sanic.budgeted_of(request)
|
|
1483
|
+
```
|
|
1484
|
+
|
|
1485
|
+
A budget belongs to the request, so one request cannot spend another's.
|
|
1486
|
+
`attach_synchronous_request_budget` and `synchronous_budgeted_of` are the
|
|
1487
|
+
counterparts.
|
|
1488
|
+
|
|
1489
|
+
See [`examples/sanic_application.py`](examples/sanic_application.py) for a
|
|
1490
|
+
runnable application covering all of it.
|
|
1491
|
+
|
|
1492
|
+
## Policies
|
|
1493
|
+
|
|
1494
|
+
A rule written against SQL text can be walked around with an alias, a subquery,
|
|
1495
|
+
or a common table. A policy is written against the statement's own nodes, before
|
|
1496
|
+
any SQL exists, so every one of those routes carries it:
|
|
1497
|
+
|
|
1498
|
+
```python
|
|
1499
|
+
from pyoq.policies import SoftDelete, TenantScope, governed
|
|
1500
|
+
|
|
1501
|
+
scoped = governed(database, [TenantScope("tenant_id", 7), SoftDelete("deleted_at")])
|
|
1502
|
+
```
|
|
1503
|
+
|
|
1504
|
+
```sql
|
|
1505
|
+
SELECT COUNT(*) FROM "notes" AS "n"
|
|
1506
|
+
WHERE (("n"."tenant_id" = ?) AND ("n"."deleted_at" IS NULL))
|
|
1507
|
+
```
|
|
1508
|
+
|
|
1509
|
+
The condition follows the name the statement gave the table, joins are scoped
|
|
1510
|
+
beside it, a subquery and a common table are scoped inside themselves, and both
|
|
1511
|
+
halves of a set operation are scoped. A write carries the scope as a value, so a
|
|
1512
|
+
row cannot be created outside the scope that would then be unable to see it, and
|
|
1513
|
+
a caller that names the column itself has its value replaced rather than
|
|
1514
|
+
honoured. An update or a delete that said it meant every row now means every row
|
|
1515
|
+
the policy admits.
|
|
1516
|
+
|
|
1517
|
+
### What is available
|
|
1518
|
+
|
|
1519
|
+
- `TenantScope(column, value)` puts every read and every write inside one scope
|
|
1520
|
+
- `SoftDelete(column)` keeps a row marked deleted out of every read
|
|
1521
|
+
- `AllowedTables(names)` refuses a statement that touches anything else
|
|
1522
|
+
- `RowConstraint(table, build)` narrows one table by a condition of your own
|
|
1523
|
+
|
|
1524
|
+
### What cannot be governed
|
|
1525
|
+
|
|
1526
|
+
SQL that was compiled elsewhere has no statement left to read, so it is refused
|
|
1527
|
+
rather than quietly allowed. So is raw SQL written into an expression. A project
|
|
1528
|
+
that means to run them says so once:
|
|
1529
|
+
|
|
1530
|
+
```python
|
|
1531
|
+
governed(database, [...], raw_sql=True)
|
|
1532
|
+
```
|
|
1533
|
+
|
|
1534
|
+
### Transactions and streams
|
|
1535
|
+
|
|
1536
|
+
A governed database opens its own transactions and streams its own reads, both
|
|
1537
|
+
held to the same rules:
|
|
1538
|
+
|
|
1539
|
+
```python
|
|
1540
|
+
with scoped.transaction() as active:
|
|
1541
|
+
...
|
|
1542
|
+
with scoped.stream(query) as rows:
|
|
1543
|
+
...
|
|
1544
|
+
```
|
|
1545
|
+
|
|
1546
|
+
Without those a caller would reach past the policy to the database underneath to
|
|
1547
|
+
open one, and everything inside it would be ungoverned. A dialect's own
|
|
1548
|
+
streaming settings belong to the database underneath, which is where it was
|
|
1549
|
+
given them.
|
|
1550
|
+
|
|
1551
|
+
### Going around a policy
|
|
1552
|
+
|
|
1553
|
+
```python
|
|
1554
|
+
with administrative_bypass(scoped, "restoring a withdrawn note", audit) as free:
|
|
1555
|
+
...
|
|
1556
|
+
```
|
|
1557
|
+
|
|
1558
|
+
The ungoverned database is yielded rather than returned, so it belongs to the
|
|
1559
|
+
block and cannot be kept past it. A reason is required, and entering and leaving
|
|
1560
|
+
are both written to the audit.
|
|
1561
|
+
|
|
1562
|
+
See [`examples/policies.py`](examples/policies.py) for a runnable walkthrough.
|
|
1563
|
+
|
|
1564
|
+
## Reading a column as the type it was declared to hold
|
|
1565
|
+
|
|
1566
|
+
A driver answers with what its own protocol carries, which is not always what a
|
|
1567
|
+
column was declared to be. SQLite hands back a float for a decimal and a string
|
|
1568
|
+
for a date, and psycopg hands back a view of its buffer for bytes. A descriptor
|
|
1569
|
+
that says `Decimal` and yields a float is a promise the runtime does not keep,
|
|
1570
|
+
and money is the usual casualty.
|
|
1571
|
+
|
|
1572
|
+
A descriptor carries the type it was declared with, so a row is given back as
|
|
1573
|
+
the row was described:
|
|
1574
|
+
|
|
1575
|
+
```python
|
|
1576
|
+
PRICE = ColumnDescriptor[Decimal](..., value_type=Decimal)
|
|
1577
|
+
|
|
1578
|
+
row = using(database).select(PRICE).from_(source).fetch_one()
|
|
1579
|
+
# Decimal('12.50'), not 12.5
|
|
1580
|
+
```
|
|
1581
|
+
|
|
1582
|
+
Generated descriptors declare it for you. `Decimal`, `date`, `datetime`,
|
|
1583
|
+
`time`, `timedelta`, `UUID`, `bytes`, `bool`, `int`, `float`, `str`, and JSON
|
|
1584
|
+
are read from whatever a driver answered with. A generated enumeration reads
|
|
1585
|
+
back as a member of itself rather than as the string stored for it, and an
|
|
1586
|
+
array column reads back as the tuple its descriptor declares rather than as
|
|
1587
|
+
the list a driver hands over.
|
|
1588
|
+
|
|
1589
|
+
This happens on every path that answers with rows: `fetch_one`, `fetch_all`,
|
|
1590
|
+
scalars, bulk statements that return rows, streams, and the asynchronous
|
|
1591
|
+
counterpart of each. A driver decides what it carries and decoding decides
|
|
1592
|
+
what a caller is given, so two paths that decode differently would answer the
|
|
1593
|
+
same query with different types.
|
|
1594
|
+
|
|
1595
|
+
A column that declared nothing is handed back exactly as it came, so a caller
|
|
1596
|
+
that never said what it wanted pays nothing for the question.
|
|
1597
|
+
|
|
1598
|
+
A value that cannot be read as what it was declared to be raises, because
|
|
1599
|
+
handing back the wrong type silently is the defect this exists to stop. What a
|
|
1600
|
+
driver already lost is lost: PyOQ converts what it is given and cannot recover
|
|
1601
|
+
precision a database did not keep.
|
|
1602
|
+
|
|
1603
|
+
A nested collection arrives as JSON, which spells a date as a string and a
|
|
1604
|
+
decimal as a number, so it is told what its columns hold in the same way:
|
|
1605
|
+
|
|
1606
|
+
```python
|
|
1607
|
+
decode_collection(payload, ("id", "released"), (int, date))
|
|
1608
|
+
hydrate_collection(payload, ("id", "released"), construct, (int, date))
|
|
1609
|
+
```
|
|
1610
|
+
|
|
1611
|
+
## Observability
|
|
1612
|
+
|
|
1613
|
+
Instrumentation is a database that wraps another one, so a project that wants
|
|
1614
|
+
none holds the plain database and pays nothing at all:
|
|
1615
|
+
|
|
1616
|
+
```python
|
|
1617
|
+
from pyoq.diagnostics import CollectingSink, InstrumentedOperations
|
|
1618
|
+
|
|
1619
|
+
watched = InstrumentedOperations(database, sink)
|
|
1620
|
+
```
|
|
1621
|
+
|
|
1622
|
+
Every statement is reported before it reaches a driver and again when it is
|
|
1623
|
+
answered, at the one place they all pass. An event carries the shape of the
|
|
1624
|
+
statement, how many parameters it had, how long it took, and how many rows came
|
|
1625
|
+
back.
|
|
1626
|
+
|
|
1627
|
+
### What an event may carry
|
|
1628
|
+
|
|
1629
|
+
Nothing a query was asked about, unless a policy says so:
|
|
1630
|
+
|
|
1631
|
+
```python
|
|
1632
|
+
InstrumentedOperations(database, sink, EventPolicy(
|
|
1633
|
+
include_values=True, # never the ones marked sensitive
|
|
1634
|
+
include_failure_detail=True, # a driver names the offending value in it
|
|
1635
|
+
slow_after=0.5,
|
|
1636
|
+
))
|
|
1637
|
+
```
|
|
1638
|
+
|
|
1639
|
+
A failure is reported by the name of its type. PostgreSQL states the value that
|
|
1640
|
+
caused it inside the message it raises, so the message is withheld until a
|
|
1641
|
+
project decides otherwise. A value the compiler marked sensitive is left out
|
|
1642
|
+
even when a policy asks for values, because marking it was the decision that it
|
|
1643
|
+
must not be shown.
|
|
1644
|
+
|
|
1645
|
+
A shape has its literals taken out as well as its placeholders, so a statement
|
|
1646
|
+
written by hand can be logged as safely as one PyOQ built. The statement is read
|
|
1647
|
+
once, left to right, because what a character means depends on what it is
|
|
1648
|
+
inside: a quote inside a comment starts nothing, and two dashes inside a string
|
|
1649
|
+
are not a comment. Quoted strings, dollar quoted strings, and every kind of
|
|
1650
|
+
comment go, nested ones whole; quoted names stay, because a name is not a value.
|
|
1651
|
+
A number is taken out however it was written, and an opening that was never
|
|
1652
|
+
closed takes the rest of the statement with it, because nothing after it can be
|
|
1653
|
+
shown to be safe. What a shape carries is
|
|
1654
|
+
bounded, while its digest is taken from the whole statement, so two that differ
|
|
1655
|
+
only past the bound are still told apart.
|
|
1656
|
+
|
|
1657
|
+
### Tracing
|
|
1658
|
+
|
|
1659
|
+
```python
|
|
1660
|
+
from pyoq.tracing import TracingSink
|
|
1661
|
+
|
|
1662
|
+
watched = InstrumentedOperations(database, TracingSink())
|
|
1663
|
+
```
|
|
1664
|
+
|
|
1665
|
+
Each answered statement becomes a span carrying the same facts the event does.
|
|
1666
|
+
A span is opened and closed when the statement is answered, because that is the
|
|
1667
|
+
event that knows how long it took.
|
|
1668
|
+
|
|
1669
|
+
### Readings
|
|
1670
|
+
|
|
1671
|
+
```python
|
|
1672
|
+
from pyoq.diagnostics import metrics_of, shape_cache_metrics
|
|
1673
|
+
|
|
1674
|
+
metrics_of(pool) # open, idle, checked out, and how much is in use
|
|
1675
|
+
shape_cache_metrics() # how often a shape was already known
|
|
1676
|
+
```
|
|
1677
|
+
|
|
1678
|
+
Events say what happened; a reading says what is happening now. The shape cache
|
|
1679
|
+
is bounded, because the statements a process runs are not.
|
|
1680
|
+
|
|
1681
|
+
See [`examples/observability.py`](examples/observability.py) for a runnable
|
|
1682
|
+
walkthrough, and `benchmarks/observability.py` for the cost it is held to.
|
|
1683
|
+
|
|
1684
|
+
## Generated names
|
|
1685
|
+
|
|
1686
|
+
`NamingPolicy` converts source identifiers into valid Python names as one
|
|
1687
|
+
immutable batch. It supports snake case, Pascal case, and upper snake case.
|
|
1688
|
+
Normalization uses Unicode NFKC and Python's locale-independent case rules.
|
|
1689
|
+
Python keywords, soft keywords, and names reserved by a generated component
|
|
1690
|
+
receive trailing underscores until the name is available.
|
|
1691
|
+
|
|
1692
|
+
Each request has a stable key chosen by the generation planner. Scopes isolate
|
|
1693
|
+
namespaces such as table types, row fields, and enum members:
|
|
1694
|
+
|
|
1695
|
+
```python
|
|
1696
|
+
from pyoq.naming import (
|
|
1697
|
+
NameRequest,
|
|
1698
|
+
NamingPolicy,
|
|
1699
|
+
NamingScope,
|
|
1700
|
+
PythonNameStyle,
|
|
1701
|
+
)
|
|
1702
|
+
from pyoq.schema import Identifier
|
|
1703
|
+
|
|
1704
|
+
fields = NamingScope(
|
|
1705
|
+
name="user-fields",
|
|
1706
|
+
style=PythonNameStyle.SNAKE_CASE,
|
|
1707
|
+
requests=(
|
|
1708
|
+
NameRequest("first-name", Identifier("First Name")),
|
|
1709
|
+
NameRequest("class", Identifier("class")),
|
|
1710
|
+
),
|
|
1711
|
+
reserved_names=("builder", "build"),
|
|
1712
|
+
)
|
|
1713
|
+
names = NamingPolicy().resolve((fields,))
|
|
1714
|
+
|
|
1715
|
+
assert names.get("user-fields", "first-name") == "first_name"
|
|
1716
|
+
assert names.get("user-fields", "class") == "class_"
|
|
1717
|
+
```
|
|
1718
|
+
|
|
1719
|
+
Ordering requests or scopes differently does not change the result. Names that
|
|
1720
|
+
become equal after case conversion, Unicode normalization, punctuation removal,
|
|
1721
|
+
or reserved-name escaping are not renamed by sequence. Instead,
|
|
1722
|
+
`NamingCollisionError` reports every conflict in the batch before a renderer
|
|
1723
|
+
can receive a partial result. This keeps generated APIs stable when database
|
|
1724
|
+
reflection order changes.
|
|
1725
|
+
|
|
1726
|
+
## Schema snapshots and drift
|
|
1727
|
+
|
|
1728
|
+
A snapshot is the one canonical record of what a database holds. Written to a
|
|
1729
|
+
file it becomes reviewable in a diff and usable without a database, which is
|
|
1730
|
+
what lets generation and staleness checks run where no credentials exist.
|
|
1731
|
+
|
|
1732
|
+
```toml
|
|
1733
|
+
[tool.pyoq]
|
|
1734
|
+
schema-snapshot = "schema.json"
|
|
1735
|
+
```
|
|
1736
|
+
|
|
1737
|
+
With that set, generation reads the file instead of connecting. Nothing else
|
|
1738
|
+
changes: the same pipeline serves both and does not know which it got.
|
|
1739
|
+
|
|
1740
|
+
```bash
|
|
1741
|
+
pyoq snapshot # record what the database holds
|
|
1742
|
+
pyoq drift # compare the record against the database
|
|
1743
|
+
pyoq generate # generate from the record, with no connection
|
|
1744
|
+
```
|
|
1745
|
+
|
|
1746
|
+
Recording the same database twice writes the same bytes, so the file changes
|
|
1747
|
+
only when the schema does. It is written beside its destination and moved into
|
|
1748
|
+
place, so a reader never sees half a snapshot and a failed write leaves the
|
|
1749
|
+
previous record intact.
|
|
1750
|
+
|
|
1751
|
+
`pyoq drift` names what moved in terms of the schema rather than of the file,
|
|
1752
|
+
so the answer is the column that changed rather than the line that differs.
|
|
1753
|
+
|
|
1754
|
+
```text
|
|
1755
|
+
1 schema difference(s): added column warehouse.public.parts.weight
|
|
1756
|
+
```
|
|
1757
|
+
|
|
1758
|
+
Named things are matched by name and then by value, so a renamed table reads
|
|
1759
|
+
as one removed and one added rather than as a change to something that is no
|
|
1760
|
+
longer there.
|
|
1761
|
+
|
|
1762
|
+
### After a migration
|
|
1763
|
+
|
|
1764
|
+
A migration changes the schema, which makes the generated package stale and
|
|
1765
|
+
the recorded snapshot wrong. One call puts both back in step, in that order,
|
|
1766
|
+
because a package regenerated before the snapshot is written would be built
|
|
1767
|
+
from the schema that has just been replaced.
|
|
1768
|
+
|
|
1769
|
+
```python
|
|
1770
|
+
from pyoq.migrations import after_migration_at
|
|
1771
|
+
|
|
1772
|
+
after_migration_at("/path/to/project")
|
|
1773
|
+
```
|
|
1774
|
+
|
|
1775
|
+
Nothing there imports a migration tool, and no migration tool is a dependency
|
|
1776
|
+
of PyOQ.
|
|
1777
|
+
|
|
1778
|
+
It runs after the migration has been applied, which for Alembic means after
|
|
1779
|
+
`alembic upgrade`. A post-write hook is not that moment: Alembic runs those
|
|
1780
|
+
inside `alembic revision`, on the revision file it has just written, while
|
|
1781
|
+
the database is still whatever it was. Wiring PyOQ there would record the
|
|
1782
|
+
schema the migration is about to replace, so the installed command refuses a
|
|
1783
|
+
revision file and says where the step belongs.
|
|
1784
|
+
|
|
1785
|
+
Call it from `env.py`, once the migrations have run:
|
|
1786
|
+
|
|
1787
|
+
```python
|
|
1788
|
+
with connectable.connect() as connection:
|
|
1789
|
+
context.configure(connection=connection, target_metadata=target_metadata)
|
|
1790
|
+
with context.begin_transaction():
|
|
1791
|
+
context.run_migrations()
|
|
1792
|
+
after_migration_at(PROJECT_ROOT)
|
|
1793
|
+
```
|
|
1794
|
+
|
|
1795
|
+
Or run it as a step of its own:
|
|
1796
|
+
|
|
1797
|
+
```bash
|
|
1798
|
+
alembic upgrade head && pyoq-after-migration
|
|
1799
|
+
```
|
|
1800
|
+
|
|
1801
|
+
The command takes the project directory, or nothing and uses the working
|
|
1802
|
+
directory, walking upwards to the first directory holding a `pyproject.toml`.
|
|
1803
|
+
Django, a shell script, or a CI job call `after_migration_at` directly and get
|
|
1804
|
+
the same two steps.
|
|
1805
|
+
|
|
1806
|
+
## Generation pipeline
|
|
1807
|
+
|
|
1808
|
+
Generation is a staged application service assembled from small typed
|
|
1809
|
+
components. Schema loading, concern rendering, syntax validation, drift
|
|
1810
|
+
inspection, locking, cleanup, manifest storage, and package replacement have
|
|
1811
|
+
independent interfaces and one responsibility each.
|
|
1812
|
+
|
|
1813
|
+
```mermaid
|
|
1814
|
+
flowchart LR
|
|
1815
|
+
Source[Schema source] --> Snapshot[Immutable snapshot]
|
|
1816
|
+
Snapshot --> Planner[Generation planner]
|
|
1817
|
+
Planner --> Renderers[Concern renderers]
|
|
1818
|
+
Renderers --> Plan[Canonical plan]
|
|
1819
|
+
Plan --> Validator[Python syntax validator]
|
|
1820
|
+
Validator --> Lock[Project lock]
|
|
1821
|
+
Lock --> Drift[Drift checker]
|
|
1822
|
+
Drift --> DryRun[Dry run report]
|
|
1823
|
+
Drift --> Check[Drift check]
|
|
1824
|
+
Drift --> Writer[Atomic writer]
|
|
1825
|
+
Writer --> Stage[Staged package]
|
|
1826
|
+
Stage --> Manifest[Ownership manifest]
|
|
1827
|
+
Manifest --> Swap[Package replacement]
|
|
1828
|
+
```
|
|
1829
|
+
|
|
1830
|
+
Every rendered path is relative and portable. Plans sort paths before
|
|
1831
|
+
validation, normalize line endings, reject duplicate outputs from any concern,
|
|
1832
|
+
and hash UTF-8 bytes with SHA-256. Python and stub files are parsed before the
|
|
1833
|
+
project lock or generated directory can be changed.
|
|
1834
|
+
|
|
1835
|
+
The ownership manifest is compact, versioned, and deterministic. A generated
|
|
1836
|
+
file is replaceable or removable only when its path is recorded and its current
|
|
1837
|
+
checksum still matches the manifest. Existing paths without ownership records,
|
|
1838
|
+
modified generated files, symlinks, and unsafe parent paths fail closed.
|
|
1839
|
+
Unowned files elsewhere in the generated directory are copied forward without
|
|
1840
|
+
content changes. Cleanup considers only stale manifest entries.
|
|
1841
|
+
|
|
1842
|
+
Generation uses a fail-closed project lock at `.pyoq-generation-lock`. If a
|
|
1843
|
+
process is interrupted without releasing it, verify that no generation command
|
|
1844
|
+
is active before removing that directory.
|
|
1845
|
+
|
|
1846
|
+
Write mode copies the current package into a sibling staging directory, applies
|
|
1847
|
+
the validated plan there, and replaces the destination package only after the
|
|
1848
|
+
new manifest is complete. A failed replacement restores the previous package.
|
|
1849
|
+
Check mode fails on creates, updates, removals, ownership conflicts, or manifest
|
|
1850
|
+
drift. Dry-run mode returns the same categorized change counts without writing
|
|
1851
|
+
the generated package.
|
|
1852
|
+
|
|
1853
|
+
The complete in-memory example assembles every production component and runs
|
|
1854
|
+
dry-run, write, and check modes without a database or network connection:
|
|
1855
|
+
|
|
1856
|
+
```console
|
|
1857
|
+
python examples/generation_pipeline.py
|
|
1858
|
+
```
|
|
1859
|
+
|
|
1860
|
+
See [`examples/generation_pipeline.py`](examples/generation_pipeline.py) for
|
|
1861
|
+
the typed schema source and pipeline assembly. Database reflection is
|
|
1862
|
+
introduced in a later package phase. Until a built-in schema source is
|
|
1863
|
+
available, the default command-line services continue to report generation as
|
|
1864
|
+
unavailable.
|
|
1865
|
+
|
|
1866
|
+
## Generated database types
|
|
1867
|
+
|
|
1868
|
+
`GeneratedTypesRenderer` turns one immutable schema snapshot into a complete
|
|
1869
|
+
Python package. The renderer resolves all names and type mappings once, then
|
|
1870
|
+
passes that canonical model to focused file renderers.
|
|
1871
|
+
|
|
1872
|
+
```mermaid
|
|
1873
|
+
flowchart LR
|
|
1874
|
+
Snapshot[Schema snapshot] --> Model[Canonical generation model]
|
|
1875
|
+
Model --> Enums[enums.py]
|
|
1876
|
+
Model --> Rows[rows.py]
|
|
1877
|
+
Model --> Writes[writes.py]
|
|
1878
|
+
Model --> Tables[tables.py]
|
|
1879
|
+
Model --> Relations[relations.py]
|
|
1880
|
+
Model --> Facade[package facade]
|
|
1881
|
+
```
|
|
1882
|
+
|
|
1883
|
+
Add the renderer as one concern in the generation planner:
|
|
1884
|
+
|
|
1885
|
+
```python
|
|
1886
|
+
from pyoq.generation import GeneratedTypesRenderer, GenerationPlanner
|
|
1887
|
+
|
|
1888
|
+
planner = GenerationPlanner((GeneratedTypesRenderer(),))
|
|
1889
|
+
```
|
|
1890
|
+
|
|
1891
|
+
For a `user` table, generated names follow these roles:
|
|
1892
|
+
|
|
1893
|
+
| Generated name | Responsibility |
|
|
1894
|
+
|---|---|
|
|
1895
|
+
| `User` and `USER` | Typed table descriptor and its shared instance |
|
|
1896
|
+
| `UserRow` | Frozen result value containing every readable column |
|
|
1897
|
+
| `UserInsert` | Frozen insert value with required and omitted-field semantics |
|
|
1898
|
+
| `UserUpdate` | Frozen update value where every writable field may be omitted |
|
|
1899
|
+
| `UserInsertValues` | Required and optional dictionary shape for typed boundaries |
|
|
1900
|
+
| `UserUpdateValues` | Optional dictionary shape for update boundaries |
|
|
1901
|
+
| `UserPrimaryKey` | Frozen primary-key value |
|
|
1902
|
+
| `UserBuilder` | Immutable insert builder without `build()` until complete |
|
|
1903
|
+
| `UserUpdateBuilder` | Immutable update builder with concrete field setters |
|
|
1904
|
+
|
|
1905
|
+
Table and column descriptors support SQL-shaped discovery. Constants use the
|
|
1906
|
+
source database name while carrying its exact value as metadata:
|
|
1907
|
+
|
|
1908
|
+
```python
|
|
1909
|
+
from application.database import USER, User
|
|
1910
|
+
|
|
1911
|
+
identifier_column = USER.ID
|
|
1912
|
+
new_values = User.builder().tenant_id(7).name("Ada").build()
|
|
1913
|
+
```
|
|
1914
|
+
|
|
1915
|
+
Every writable column produces a concrete setter with its exact mapped Python
|
|
1916
|
+
type. Nullable setters accept `None`. Generated, identity, computed, and other
|
|
1917
|
+
read-only columns expose no insert or update setter. Calling a setter returns a
|
|
1918
|
+
new frozen builder and never performs database I/O.
|
|
1919
|
+
|
|
1920
|
+
Required fields use `Missing` and `Present` type states. Each required setter
|
|
1921
|
+
adds two overloads, so generated typing grows linearly with required-column
|
|
1922
|
+
count. The final required setter returns a distinct complete-builder subtype.
|
|
1923
|
+
Only that subtype defines `build()`. Mypy and Pyright therefore reject both an
|
|
1924
|
+
empty build and a partially complete build, while editors can omit `build()`
|
|
1925
|
+
from their completion lists. Runtime incomplete builder objects also lack that
|
|
1926
|
+
attribute.
|
|
1927
|
+
|
|
1928
|
+
`UserInsert` is a reusable value object. The column-oriented insert statement
|
|
1929
|
+
API accepts it through `values_many()` alongside positional multi-row values.
|
|
1930
|
+
Constructing it has no connection, transaction, or persistence side effect.
|
|
1931
|
+
|
|
1932
|
+
Each generated column descriptor records both its database name and the Python
|
|
1933
|
+
field name used by the generated row and write values. That pairing is what
|
|
1934
|
+
lets a write statement map a generated value onto the columns it selected
|
|
1935
|
+
without dynamic name guessing.
|
|
1936
|
+
|
|
1937
|
+
Primary and unique keys become frozen exact-type values. Relationships become
|
|
1938
|
+
typed descriptors containing aligned source and target columns, referential
|
|
1939
|
+
actions, and source names. Relationship descriptors describe metadata only;
|
|
1940
|
+
they do not trigger lazy loading or hidden queries.
|
|
1941
|
+
|
|
1942
|
+
SQL scalars map to narrow Python types, including `Decimal`, `UUID`, date and
|
|
1943
|
+
time values, immutable tuples for arrays, generated `StrEnum` classes, and a
|
|
1944
|
+
recursive `JsonValue` alias. Unknown source types map to `object`, never
|
|
1945
|
+
`Any`. Generated source is deterministic and passes Ruff formatting, Mypy
|
|
1946
|
+
strict mode, and Pyright strict mode without suppressions.
|
|
1947
|
+
|
|
1948
|
+
A domain is a named type with rules attached, and it maps to whatever it is
|
|
1949
|
+
written over: a domain over `text` generates `str`, and one over
|
|
1950
|
+
`numeric(12, 2)` generates `Decimal`. The schema keeps the name as well as the
|
|
1951
|
+
base, because a schema that forgot it would no longer describe the database it
|
|
1952
|
+
was read from. A domain that forbids null makes every column of it not
|
|
1953
|
+
nullable, whatever the column itself said, because the server refuses a null
|
|
1954
|
+
there either way. PostgreSQL has domains; SQLite and MySQL have none.
|
|
1955
|
+
|
|
1956
|
+
Run the in-memory rendering example without a database or network connection:
|
|
1957
|
+
|
|
1958
|
+
```console
|
|
1959
|
+
python examples/generated_types.py
|
|
1960
|
+
```
|
|
1961
|
+
|
|
1962
|
+
## Typed expressions
|
|
1963
|
+
|
|
1964
|
+
Fields, bound values, computed expressions, and conditions form an immutable
|
|
1965
|
+
typed expression tree. Named methods keep SQL semantics explicit and preserve
|
|
1966
|
+
the result type in Mypy and Pyright strict modes.
|
|
1967
|
+
|
|
1968
|
+
```python
|
|
1969
|
+
from decimal import Decimal
|
|
1970
|
+
|
|
1971
|
+
from pyoq.query import bind, field
|
|
1972
|
+
|
|
1973
|
+
USER_ID = field(int, "id", table_name="users")
|
|
1974
|
+
USER_NAME = field(str, "name", table_name="users")
|
|
1975
|
+
USER_BALANCE = field(Decimal, "balance", table_name="users")
|
|
1976
|
+
USER_TENANT = field(int, "tenant_id", table_name="users")
|
|
1977
|
+
|
|
1978
|
+
predicate = USER_ID.gt(0) & USER_NAME.starts_with("A")
|
|
1979
|
+
adjusted = USER_BALANCE.add(Decimal("5.00"))
|
|
1980
|
+
selected = USER_ID.in_(bind(1), bind(2))
|
|
1981
|
+
```
|
|
1982
|
+
|
|
1983
|
+
Generated column descriptors implement the same `Expression[T]` contract, so
|
|
1984
|
+
the generated database package is the primary field source. Comparisons,
|
|
1985
|
+
numeric arithmetic, string operations, temporal extraction and duration
|
|
1986
|
+
arithmetic, null predicates, ranges, membership, and boolean composition are
|
|
1987
|
+
available without converting descriptors or losing their value types.
|
|
1988
|
+
|
|
1989
|
+
```mermaid
|
|
1990
|
+
flowchart LR
|
|
1991
|
+
Field[Typed field] --> Node[Immutable expression node]
|
|
1992
|
+
Value[Python value] --> Bind[Bound value node]
|
|
1993
|
+
Bind --> Node
|
|
1994
|
+
Node --> Computed[Typed computed expression]
|
|
1995
|
+
Node --> Condition[Boolean condition]
|
|
1996
|
+
Raw[Explicit typed raw template] --> Node
|
|
1997
|
+
```
|
|
1998
|
+
|
|
1999
|
+
Python values supplied to expression methods always become bound-value nodes.
|
|
2000
|
+
They cannot become field names, operators, clauses, or raw SQL text. Null
|
|
2001
|
+
equality is normalized to `is_null()` or `is_not_null()` nodes. Explicit raw
|
|
2002
|
+
expressions use named placeholders that accept expression objects only:
|
|
2003
|
+
|
|
2004
|
+
```python
|
|
2005
|
+
from pyoq.query import raw
|
|
2006
|
+
|
|
2007
|
+
distance = raw(
|
|
2008
|
+
float,
|
|
2009
|
+
"distance({origin}, {target})",
|
|
2010
|
+
origin=USER_ID,
|
|
2011
|
+
target=bind(10),
|
|
2012
|
+
)
|
|
2013
|
+
```
|
|
2014
|
+
|
|
2015
|
+
Raw placeholders reject plain values, attribute access, indexing, conversion,
|
|
2016
|
+
and format specifications. Use `raw_condition()` when the template produces a
|
|
2017
|
+
boolean condition. SQL rendering and ordered parameter extraction are owned by
|
|
2018
|
+
the compiler introduced with statement construction.
|
|
2019
|
+
|
|
2020
|
+
Run the expression example with:
|
|
2021
|
+
|
|
2022
|
+
```console
|
|
2023
|
+
python examples/expressions.py
|
|
2024
|
+
```
|
|
2025
|
+
|
|
2026
|
+
### Choosing one value out of several
|
|
2027
|
+
|
|
2028
|
+
`case()` reads its branches in order and stops at the first that holds. The
|
|
2029
|
+
type of the whole expression is the type of its first result, so a later branch
|
|
2030
|
+
that disagrees is refused where it is written.
|
|
2031
|
+
|
|
2032
|
+
```python
|
|
2033
|
+
from pyoq.query import case, coalesce, greatest, least, nullif
|
|
2034
|
+
|
|
2035
|
+
tier = (
|
|
2036
|
+
case()
|
|
2037
|
+
.when(USER_BALANCE.gt(Decimal("500.00")), "premium")
|
|
2038
|
+
.when(USER_BALANCE.gt(Decimal("100.00")), "standard")
|
|
2039
|
+
.otherwise("basic")
|
|
2040
|
+
)
|
|
2041
|
+
```
|
|
2042
|
+
|
|
2043
|
+
`otherwise()` gives the value for rows no branch claimed, and the result is not
|
|
2044
|
+
nullable. `end()` closes the case without one, and the result is `T | None`,
|
|
2045
|
+
because SQL answers null for a row nothing matched:
|
|
2046
|
+
|
|
2047
|
+
```python
|
|
2048
|
+
flagged = case().when(USER_BALANCE.lt(Decimal("0.00")), "overdrawn").end()
|
|
2049
|
+
```
|
|
2050
|
+
|
|
2051
|
+
The other three answer the same question with fixed rules. `coalesce()` takes
|
|
2052
|
+
the first argument that is not null, and it drops the `None` from the type when
|
|
2053
|
+
the fallback cannot be null. `USER.NICKNAME` below is a generated descriptor for
|
|
2054
|
+
a nullable column, typed `Expression[str | None]`:
|
|
2055
|
+
|
|
2056
|
+
```python
|
|
2057
|
+
from generated.database.tables import USER
|
|
2058
|
+
|
|
2059
|
+
label = coalesce(USER.NICKNAME, "unknown")
|
|
2060
|
+
```
|
|
2061
|
+
|
|
2062
|
+
`label` is `ComputedExpression[str]`, so a row read through it needs no null
|
|
2063
|
+
check and no cast. `nullif()` answers null when its two arguments agree, `greatest()` takes the
|
|
2064
|
+
widest of its arguments and `least()` the narrowest:
|
|
2065
|
+
|
|
2066
|
+
```python
|
|
2067
|
+
blank_as_null = nullif(USER_NAME, "")
|
|
2068
|
+
floor_price = greatest(USER_BALANCE, Decimal("0.00"))
|
|
2069
|
+
capped = least(USER_BALANCE, Decimal("1000.00"))
|
|
2070
|
+
```
|
|
2071
|
+
|
|
2072
|
+
Every one of these is an expression like any other, so it can be selected,
|
|
2073
|
+
ordered by, grouped by, compared, or nested inside another. Values reach them
|
|
2074
|
+
as bound parameters, never as SQL text. SQLite has no `GREATEST` or `LEAST` and
|
|
2075
|
+
spells them `MAX` and `MIN` over several arguments, which the SQLite compiler
|
|
2076
|
+
emits without changing what the expression means.
|
|
2077
|
+
|
|
2078
|
+
### Asking for a value as another type
|
|
2079
|
+
|
|
2080
|
+
`cast()` gives the conversion to the database and declares what it answers
|
|
2081
|
+
with, so the value is read back as that type rather than as whatever the driver
|
|
2082
|
+
carried:
|
|
2083
|
+
|
|
2084
|
+
```python
|
|
2085
|
+
from pyoq.query import cast
|
|
2086
|
+
|
|
2087
|
+
as_number = cast(USER_NAME, Decimal)
|
|
2088
|
+
as_text = cast(USER_ID, str)
|
|
2089
|
+
```
|
|
2090
|
+
|
|
2091
|
+
`as_number` is `ComputedExpression[Decimal]`, and a row read through it holds a
|
|
2092
|
+
`Decimal` on every dialect. A cast is the only computed expression that names
|
|
2093
|
+
its own type, so it is the only one the row decoder can hold to a promise.
|
|
2094
|
+
|
|
2095
|
+
Each dialect names its own types, and the names were taken from running
|
|
2096
|
+
servers rather than from a specification:
|
|
2097
|
+
|
|
2098
|
+
| Target | SQLite | PostgreSQL | MySQL |
|
|
2099
|
+
| --- | --- | --- | --- |
|
|
2100
|
+
| `bool` | `INTEGER` | `BOOLEAN` | refused |
|
|
2101
|
+
| `int` | `INTEGER` | `INTEGER` | `SIGNED` |
|
|
2102
|
+
| `float` | `REAL` | `DOUBLE PRECISION` | `DOUBLE` |
|
|
2103
|
+
| `Decimal` | `NUMERIC` | `NUMERIC` | `DECIMAL(65, 30)` |
|
|
2104
|
+
| `str` | `TEXT` | `TEXT` | `CHAR` |
|
|
2105
|
+
| `bytes` | `BLOB` | `BYTEA` | `BINARY` |
|
|
2106
|
+
| `date` | refused | `DATE` | `DATE` |
|
|
2107
|
+
| `time` | refused | `TIME` | `TIME` |
|
|
2108
|
+
| `datetime` | refused | `TIMESTAMP` | `DATETIME` |
|
|
2109
|
+
| `timedelta` | refused | `INTERVAL` | refused |
|
|
2110
|
+
| `UUID` | refused | `UUID` | refused |
|
|
2111
|
+
| `dict`, `list` | refused | `JSONB` | `JSON` |
|
|
2112
|
+
|
|
2113
|
+
A dialect that has no such type refuses the cast. SQLite accepts
|
|
2114
|
+
`CAST(x AS DATE)` and answers with an integer, because an unfamiliar type name
|
|
2115
|
+
falls back to storage affinity there instead of being rejected, so PyOQ
|
|
2116
|
+
refuses rather than emitting a name that would return the wrong kind of value
|
|
2117
|
+
under a promise of the right one. MySQL is asked for its largest decimal
|
|
2118
|
+
precision, because a bare `DECIMAL` means `DECIMAL(10, 0)` and would answer
|
|
2119
|
+
`123` for `123.45`.
|
|
2120
|
+
|
|
2121
|
+
A target no database type answers to is refused where it is written, not when
|
|
2122
|
+
the query runs.
|
|
2123
|
+
|
|
2124
|
+
### Measuring a row against the rows around it
|
|
2125
|
+
|
|
2126
|
+
An aggregate collapses a group into one row. A window leaves the rows alone
|
|
2127
|
+
and gives each one an answer computed from its neighbours, so a rank, a
|
|
2128
|
+
running total, or the previous row's value can be selected beside the row
|
|
2129
|
+
itself.
|
|
2130
|
+
|
|
2131
|
+
The clauses read in the order SQL writes them: the function, then the window
|
|
2132
|
+
it looks through, then how that window is divided and ordered.
|
|
2133
|
+
|
|
2134
|
+
```python
|
|
2135
|
+
from pyoq.query import dense_rank, lag, ntile, rank, row_number, sum_
|
|
2136
|
+
|
|
2137
|
+
position = row_number().over().partition_by(USER_TENANT).order_by(
|
|
2138
|
+
USER_BALANCE.desc()
|
|
2139
|
+
)
|
|
2140
|
+
standing = rank().over().order_by(USER_BALANCE.desc())
|
|
2141
|
+
running = sum_(USER_BALANCE).over().partition_by(USER_TENANT).order_by(USER_ID)
|
|
2142
|
+
previous = lag(USER_BALANCE).over().order_by(USER_ID)
|
|
2143
|
+
```
|
|
2144
|
+
|
|
2145
|
+
`row_number()` counts from one and breaks ties arbitrarily. `rank()` gives
|
|
2146
|
+
tied rows the same position and leaves a gap after them; `dense_rank()` gives
|
|
2147
|
+
them the same position and leaves no gap. `ntile(n)` says which of `n` equal
|
|
2148
|
+
buckets a row falls in. `lag()` and `lead()` read the value that many rows
|
|
2149
|
+
back or ahead, and answer null past the edge of the window, so they are typed
|
|
2150
|
+
`T | None`.
|
|
2151
|
+
|
|
2152
|
+
Any aggregate can be taken over a window instead of over a group, through the
|
|
2153
|
+
same `over()`. `count().over()` counts every row in the window without naming
|
|
2154
|
+
a column.
|
|
2155
|
+
|
|
2156
|
+
A window with no `partition_by()` covers every row. A window orders rows the
|
|
2157
|
+
way a query does, using the same terms and the same rules, so asking for nulls
|
|
2158
|
+
last inside a window works on every dialect just as it does outside one.
|
|
2159
|
+
|
|
2160
|
+
Window functions require SQLite 3.25 or later. They are governed by the
|
|
2161
|
+
`window_functions` capability, so a build without them refuses the query
|
|
2162
|
+
rather than emitting SQL it cannot run.
|
|
2163
|
+
|
|
2164
|
+
### Comparing several values as one
|
|
2165
|
+
|
|
2166
|
+
A composite key is one key, and asking whether a row is among a set of them
|
|
2167
|
+
should read that way. `row()` puts columns side by side and compares them
|
|
2168
|
+
against tuples in one predicate, instead of an OR of ANDs a reader has to
|
|
2169
|
+
reassemble:
|
|
2170
|
+
|
|
2171
|
+
```python
|
|
2172
|
+
from pyoq.query import row
|
|
2173
|
+
|
|
2174
|
+
wanted = row(USER_TENANT, USER_NAME).in_((1, "Ada"), (2, "Grace"))
|
|
2175
|
+
exact = row(USER_TENANT, USER_NAME).eq((1, "Ada"))
|
|
2176
|
+
after = row(USER_TENANT, USER_ID).gt((1, 100))
|
|
2177
|
+
```
|
|
2178
|
+
|
|
2179
|
+
The tuples are checked against the columns by position, so a value of the
|
|
2180
|
+
wrong type, in the wrong order, or a tuple of the wrong width is refused where
|
|
2181
|
+
it is written rather than when the query runs. `row(USER_TENANT,
|
|
2182
|
+
USER_NAME).in_((1, 2))` does not type-check, because the second column is a
|
|
2183
|
+
string.
|
|
2184
|
+
|
|
2185
|
+
`in_()`, `not_in()`, `eq()`, `ne()`, `gt()`, `ge()`, `lt()`, and `le()` are
|
|
2186
|
+
available. The ordering comparisons compare left to right, the way SQL orders
|
|
2187
|
+
a row value, so `(tenant, id) > (1, 100)` means every row of a later tenant
|
|
2188
|
+
and the rows of tenant one after id 100. Matching an empty set of rows is the
|
|
2189
|
+
same nothing that an empty `in_()` already means.
|
|
2190
|
+
|
|
2191
|
+
A row value is two columns or more. SQL reads a single bracketed value as that
|
|
2192
|
+
value, so a row of one is refused. Every value travels bound, exactly as it
|
|
2193
|
+
does in any other predicate.
|
|
2194
|
+
|
|
2195
|
+
Membership in the rows of another query is a `semi_join()` rather than a row
|
|
2196
|
+
value, because that is where the join conditions and their typing already
|
|
2197
|
+
live.
|
|
2198
|
+
|
|
2199
|
+
### Holding the rows a query read
|
|
2200
|
+
|
|
2201
|
+
A row read inside a transaction can be changed by somebody else before the
|
|
2202
|
+
transaction acts on it. A lock holds it until the transaction ends, and the
|
|
2203
|
+
clauses read in the order SQL writes them:
|
|
2204
|
+
|
|
2205
|
+
```python
|
|
2206
|
+
job = (
|
|
2207
|
+
select(JOB_ID, JOB_PAYLOAD)
|
|
2208
|
+
.from_(JOBS)
|
|
2209
|
+
.where(JOB_STATE.eq("pending"))
|
|
2210
|
+
.limit(1)
|
|
2211
|
+
.for_update()
|
|
2212
|
+
.skip_locked()
|
|
2213
|
+
)
|
|
2214
|
+
```
|
|
2215
|
+
|
|
2216
|
+
That is a work queue: each worker takes a row nobody else holds, and passes
|
|
2217
|
+
over the ones already taken instead of waiting behind them.
|
|
2218
|
+
|
|
2219
|
+
`for_update()` holds the whole row. `for_share()` holds it against change
|
|
2220
|
+
while letting others read it. `for_no_key_update()` and `for_key_share()` are
|
|
2221
|
+
the weaker PostgreSQL locks that leave a key referenceable.
|
|
2222
|
+
|
|
2223
|
+
`nowait()` fails instead of waiting for a row somebody else holds, and
|
|
2224
|
+
`skip_locked()` passes over it. Waiting is what a lock does when told neither,
|
|
2225
|
+
so nothing is written for it. `of()` narrows the lock to some of what the
|
|
2226
|
+
query read, which is what keeps a join from holding rows it only looked at.
|
|
2227
|
+
|
|
2228
|
+
Each dialect refuses what it has not got, measured against running servers.
|
|
2229
|
+
SQLite locks the whole database rather than rows, so it refuses locking
|
|
2230
|
+
outright and points at a transaction. MySQL has `FOR UPDATE` and `FOR SHARE`
|
|
2231
|
+
with `NOWAIT`, `SKIP LOCKED`, and `OF`, but no weaker lock, so it refuses
|
|
2232
|
+
`for_no_key_update()` and `for_key_share()`. PostgreSQL has all four.
|
|
2233
|
+
|
|
2234
|
+
A lock belongs to the query that read the rows, so a set operation carries
|
|
2235
|
+
none.
|
|
2236
|
+
|
|
2237
|
+
### Reading inside a JSON value
|
|
2238
|
+
|
|
2239
|
+
A path is steps rather than text, because the three dialects do not agree on
|
|
2240
|
+
how a path is written, and PostgreSQL reads another's spelling as a member
|
|
2241
|
+
that is simply absent and answers null without complaining. A string step is a
|
|
2242
|
+
member and an integer step is an element:
|
|
2243
|
+
|
|
2244
|
+
```python
|
|
2245
|
+
country = EVENT_BODY.json_text("actor", "country")
|
|
2246
|
+
first_tag = EVENT_BODY.json_text("tags", 0)
|
|
2247
|
+
tag_count = EVENT_BODY.json_length("tags")
|
|
2248
|
+
```
|
|
2249
|
+
|
|
2250
|
+
`json_get()` answers with JSON and `json_text()` with what that JSON says, so
|
|
2251
|
+
`json_text()` is typed `str | None` and gives you text on every dialect even
|
|
2252
|
+
where the driver would have handed back the number it looked like.
|
|
2253
|
+
|
|
2254
|
+
`json_has()` asks whether anything is at a path at all, and `json_contains()`
|
|
2255
|
+
asks whether one JSON value holds another:
|
|
2256
|
+
|
|
2257
|
+
```python
|
|
2258
|
+
verified = EVENT_BODY.json_has("actor", "verified")
|
|
2259
|
+
from_london = EVENT_BODY.json_contains({"actor": {"city": "London"}})
|
|
2260
|
+
```
|
|
2261
|
+
|
|
2262
|
+
The path is a value, so it is bound like any other and never enters the SQL.
|
|
2263
|
+
Each dialect writes what it has:
|
|
2264
|
+
|
|
2265
|
+
| Asked | SQLite | PostgreSQL | MySQL |
|
|
2266
|
+
| --- | --- | --- | --- |
|
|
2267
|
+
| `json_get` | `-> '$.a.b'` | `#> '{a,b}'` | `JSON_EXTRACT` |
|
|
2268
|
+
| `json_text` | `JSON_EXTRACT` | `#>> '{a,b}'` | `JSON_UNQUOTE(JSON_EXTRACT(...))` |
|
|
2269
|
+
| `json_length` | `JSON_ARRAY_LENGTH` | `JSONB_ARRAY_LENGTH` | `JSON_LENGTH` |
|
|
2270
|
+
| `json_has` | `JSON_TYPE(...) IS NOT NULL` | `#> ... IS NOT NULL` | `JSON_CONTAINS_PATH` |
|
|
2271
|
+
| `json_contains` | refused | `@>` | `JSON_CONTAINS` |
|
|
2272
|
+
|
|
2273
|
+
SQLite has no containment operator, so it refuses rather than emulating one.
|
|
2274
|
+
Asking any of this of a column that is not JSON is refused where it is
|
|
2275
|
+
written.
|
|
2276
|
+
|
|
2277
|
+
### Asking about an array
|
|
2278
|
+
|
|
2279
|
+
An array column is generated as `tuple[T, ...]`, and these questions are typed
|
|
2280
|
+
by that element type, so a value of the wrong kind is refused where it is
|
|
2281
|
+
written:
|
|
2282
|
+
|
|
2283
|
+
```python
|
|
2284
|
+
tagged = POST_TAGS.has("python")
|
|
2285
|
+
untagged = POST_TAGS.lacks("draft")
|
|
2286
|
+
both = POST_TAGS.contains_all(("python", "sql"))
|
|
2287
|
+
any_of = POST_TAGS.overlaps(("python", "rust"))
|
|
2288
|
+
inside = POST_TAGS.contained_by(("python", "sql", "rust"))
|
|
2289
|
+
first = POST_TAGS.element(0)
|
|
2290
|
+
count = POST_TAGS.length()
|
|
2291
|
+
```
|
|
2292
|
+
|
|
2293
|
+
**Elements are counted from zero.** The column reads back as a tuple and a
|
|
2294
|
+
tuple counts from zero, so `POST_TAGS.element(0)` is the same element as
|
|
2295
|
+
`row.tags[0]`. SQL counts an array from one, and PyOQ writes that difference
|
|
2296
|
+
out rather than leaving a caller to remember it. Past the end there is no
|
|
2297
|
+
element, so the answer is null and the type is `T | None`.
|
|
2298
|
+
|
|
2299
|
+
`length()` is the same method that measures text, because how long a thing is
|
|
2300
|
+
is one question whichever kind of thing it is.
|
|
2301
|
+
|
|
2302
|
+
Only PostgreSQL has an array type. SQLite and MySQL refuse these and say to
|
|
2303
|
+
hold a collection as JSON or in a table of its own, rather than emulating an
|
|
2304
|
+
array they have not got.
|
|
2305
|
+
|
|
2306
|
+
### Calling a function this database has
|
|
2307
|
+
|
|
2308
|
+
Every database grows functions the others have not got, and a toolkit that
|
|
2309
|
+
offered only what all three share would be smaller than any of them. A vendor
|
|
2310
|
+
function is declared once with the type it answers with, and called like any
|
|
2311
|
+
other expression:
|
|
2312
|
+
|
|
2313
|
+
```python
|
|
2314
|
+
from pyoq.config import DatabaseDialect
|
|
2315
|
+
from pyoq.query import vendor_function, vendor_predicate
|
|
2316
|
+
|
|
2317
|
+
similarity = vendor_function(
|
|
2318
|
+
float, "similarity", dialects=(DatabaseDialect.POSTGRES,)
|
|
2319
|
+
)
|
|
2320
|
+
starts_with = vendor_predicate(
|
|
2321
|
+
"starts_with", dialects=(DatabaseDialect.POSTGRES,)
|
|
2322
|
+
)
|
|
2323
|
+
|
|
2324
|
+
close = similarity(USER_NAME, bind("Ada")).gt(0.3)
|
|
2325
|
+
prefixed = starts_with(USER_NAME, bind("Ad"))
|
|
2326
|
+
```
|
|
2327
|
+
|
|
2328
|
+
`dialects` says which databases the function exists on, and a compiler for any
|
|
2329
|
+
other refuses the query rather than sending SQL that cannot run. A declaration
|
|
2330
|
+
that names no dialect is written wherever it is asked for, because saying
|
|
2331
|
+
nothing means the caller did not say.
|
|
2332
|
+
|
|
2333
|
+
`vendor_function()` answers with the type given; `vendor_predicate()` answers
|
|
2334
|
+
with a `Condition`, so it composes with `and_()`, `or_()`, and `where()` like
|
|
2335
|
+
any other predicate.
|
|
2336
|
+
|
|
2337
|
+
Arguments are expressions and travel bound, exactly as they do everywhere
|
|
2338
|
+
else. The **name** is not a value, because no database takes a function name
|
|
2339
|
+
as a parameter, so it is written into the SQL and held to being a name: it
|
|
2340
|
+
must be an identifier, optionally qualified by the schema that holds it.
|
|
2341
|
+
Anything else is refused where it is declared.
|
|
2342
|
+
|
|
2343
|
+
Use `raw()` instead when what you need is not a call at all but a fragment of
|
|
2344
|
+
SQL with a shape of its own.
|
|
2345
|
+
|
|
2346
|
+
### Running a stored procedure
|
|
2347
|
+
|
|
2348
|
+
A procedure is invoked rather than selected from, so it is a statement rather
|
|
2349
|
+
than an expression:
|
|
2350
|
+
|
|
2351
|
+
```python
|
|
2352
|
+
from pyoq.query import call
|
|
2353
|
+
|
|
2354
|
+
database.execute(call("record_one", bind(7)))
|
|
2355
|
+
```
|
|
2356
|
+
|
|
2357
|
+
Arguments travel bound like any other value. The name is written into the SQL,
|
|
2358
|
+
because no database takes a routine name as a parameter, so it is held to
|
|
2359
|
+
being an identifier optionally qualified by the schema that holds it.
|
|
2360
|
+
|
|
2361
|
+
A procedure can answer with rows, and a schema does not describe what they
|
|
2362
|
+
hold, so nothing can be inferred. Naming the columns is what lets the rows be
|
|
2363
|
+
read back as the types they were said to be:
|
|
2364
|
+
|
|
2365
|
+
```python
|
|
2366
|
+
rows = database.many(call("read_one", bind(0)).returning(RECORDED_VALUE))
|
|
2367
|
+
```
|
|
2368
|
+
|
|
2369
|
+
PostgreSQL has no result set from a procedure and passes a value back through
|
|
2370
|
+
an `INOUT` parameter, which `CALL` answers with as one row. MySQL answers with
|
|
2371
|
+
whatever the procedure selected. SQLite keeps no stored procedures at all and
|
|
2372
|
+
refuses, because there is nothing for it to run and nothing to fall back on.
|
|
2373
|
+
|
|
2374
|
+
A stored **function** is a function, not a statement, so it is declared with
|
|
2375
|
+
`vendor_function()` and called wherever an expression goes.
|
|
2376
|
+
|
|
2377
|
+
## SELECT queries
|
|
2378
|
+
|
|
2379
|
+
Every join type is its own method and the clause qualifying it comes after,
|
|
2380
|
+
so a chain reads in the order SQL is written:
|
|
2381
|
+
|
|
2382
|
+
```python
|
|
2383
|
+
select(TITLE_NAME, PUBLISHER_NAME).from_(TITLES).left_join(PUBLISHERS).on(
|
|
2384
|
+
TITLE_PUBLISHER_ID.eq(PUBLISHER_ID)
|
|
2385
|
+
)
|
|
2386
|
+
```
|
|
2387
|
+
|
|
2388
|
+
| SQL | PyOQ |
|
|
2389
|
+
| --- | --- |
|
|
2390
|
+
| `INNER JOIN t ON c` | `.join(t).on(c)`, or `.inner_join(t).on(c)` |
|
|
2391
|
+
| `LEFT JOIN t ON c` | `.left_join(t).on(c)` |
|
|
2392
|
+
| `RIGHT JOIN t ON c` | `.right_join(t).on(c)` |
|
|
2393
|
+
| `FULL JOIN t ON c` | `.full_join(t).on(c)` |
|
|
2394
|
+
| `CROSS JOIN t` | `.cross_join(t)` |
|
|
2395
|
+
| `NATURAL JOIN t` | `.natural_join(t)` |
|
|
2396
|
+
| `NATURAL LEFT JOIN t` | `.natural_left_join(t)` |
|
|
2397
|
+
| `NATURAL RIGHT JOIN t` | `.natural_right_join(t)` |
|
|
2398
|
+
| `NATURAL FULL JOIN t` | `.natural_full_join(t)` |
|
|
2399
|
+
| `JOIN t USING (a, b)` | `.join(t).using("a", "b")` |
|
|
2400
|
+
| `WHERE EXISTS (...)` | `.semi_join(t).on(c)`, or `where(exists(q))` |
|
|
2401
|
+
| `WHERE NOT EXISTS (...)` | `.anti_join(t).on(c)`, or `where(not_exists(q))` |
|
|
2402
|
+
| `JOIN LATERAL (...)` | `.cross_join(q.as_lateral("name"))` |
|
|
2403
|
+
|
|
2404
|
+
A join is qualified once, by `ON`, `USING`, or `NATURAL`. `on_key()` takes a
|
|
2405
|
+
generated relationship descriptor and builds the `ON` from the foreign key the
|
|
2406
|
+
schema already declares, including one equality per column of a composite key.
|
|
2407
|
+
|
|
2408
|
+
A cross join and a natural join finish the chain on their own. Every other
|
|
2409
|
+
kind is not a query until it is qualified, which both type checkers enforce.
|
|
2410
|
+
|
|
2411
|
+
A semi join keeps rows that have a match without bringing the match back, and
|
|
2412
|
+
an anti join keeps rows that have none. No database writes either as a join,
|
|
2413
|
+
so neither does the SQL: they become `EXISTS` and `NOT EXISTS`, which every
|
|
2414
|
+
supported dialect understands. They neither collide with `where` nor depend
|
|
2415
|
+
on being written after it.
|
|
2416
|
+
|
|
2417
|
+
A lateral source may read the rows to its left, one row at a time. SQLite has
|
|
2418
|
+
none, so it refuses the query rather than evaluating the subquery once and
|
|
2419
|
+
quietly meaning something else.
|
|
2420
|
+
|
|
2421
|
+
A recursive table declares typed columns before either term is built. The
|
|
2422
|
+
column object is reused when reading from the recursive source, so autocomplete
|
|
2423
|
+
and static checking cannot replace its type with a caller assertion:
|
|
2424
|
+
|
|
2425
|
+
```python
|
|
2426
|
+
from pyoq.query import bind, column, recursive_table, select
|
|
2427
|
+
|
|
2428
|
+
number = column("n", int)
|
|
2429
|
+
walk = recursive_table("walk", number)
|
|
2430
|
+
walk_definition = walk.define(
|
|
2431
|
+
select(bind(1)),
|
|
2432
|
+
select(walk.field(number).add(1))
|
|
2433
|
+
.from_(walk)
|
|
2434
|
+
.where(walk.field(number).lt(5)),
|
|
2435
|
+
)
|
|
2436
|
+
numbers = select(walk.field(number)).with_(walk_definition).from_(walk_definition)
|
|
2437
|
+
```
|
|
2438
|
+
|
|
2439
|
+
A write returns every column its table declares with `returning_all()`, which
|
|
2440
|
+
reads the list generation put on the table.
|
|
2441
|
+
|
|
2442
|
+
`select()` retains the exact ordered projection tuple through eight fields.
|
|
2443
|
+
Every clause returns a new query and leaves its input reusable. Fields and
|
|
2444
|
+
generated column descriptors are the only normal structural inputs, while
|
|
2445
|
+
Python values remain bound expression values.
|
|
2446
|
+
|
|
2447
|
+
```python
|
|
2448
|
+
from pyoq.descriptors import TableDescriptor
|
|
2449
|
+
from pyoq.query import count, field, select
|
|
2450
|
+
|
|
2451
|
+
USERS = TableDescriptor[object, object, object]("users")
|
|
2452
|
+
USER_ID = field(int, "id", table_name="users")
|
|
2453
|
+
USER_NAME = field(str, "name", table_name="users")
|
|
2454
|
+
|
|
2455
|
+
active_users = (
|
|
2456
|
+
select(USER_ID.as_("user_id"), USER_NAME, count())
|
|
2457
|
+
.from_(USERS)
|
|
2458
|
+
.where(USER_ID.gt(0))
|
|
2459
|
+
.group_by(USER_ID, USER_NAME)
|
|
2460
|
+
.having(count().gt(0))
|
|
2461
|
+
.order_by(USER_NAME.asc())
|
|
2462
|
+
.limit(20)
|
|
2463
|
+
)
|
|
2464
|
+
```
|
|
2465
|
+
|
|
2466
|
+
A query source is a table descriptor instance, never a generated table class.
|
|
2467
|
+
The generated class is the typed column and builder namespace, and the
|
|
2468
|
+
generated constant beside it is the value that `from_()`, `join()`,
|
|
2469
|
+
`cross_join()`, and `table_source()` accept. Passing the class is rejected by
|
|
2470
|
+
both strict type checkers and, for untyped callers, by a `QueryValidationError`
|
|
2471
|
+
that states the instance requirement.
|
|
2472
|
+
|
|
2473
|
+
The immutable model covers projection aliases, distinct selection, table and
|
|
2474
|
+
aliased-table sources, inner and outer joins, cross joins, predicates,
|
|
2475
|
+
grouping, aggregate filters, ordering with explicit null placement,
|
|
2476
|
+
pagination, derived-table subqueries, common table expressions, and set
|
|
2477
|
+
operations. Aggregates preserve numeric and scalar result types while marking
|
|
2478
|
+
empty-set results nullable where required.
|
|
2479
|
+
|
|
2480
|
+
```mermaid
|
|
2481
|
+
flowchart LR
|
|
2482
|
+
Projection[Typed projections] --> Select[Immutable SELECT node]
|
|
2483
|
+
Source[Table or derived source] --> Select
|
|
2484
|
+
Join[Typed joins] --> Select
|
|
2485
|
+
Predicate[Boolean conditions] --> Select
|
|
2486
|
+
Select --> Derived[Subquery or common table]
|
|
2487
|
+
Select --> Set[Typed set operation]
|
|
2488
|
+
```
|
|
2489
|
+
|
|
2490
|
+
Set operations require the same projection type and validate projection counts
|
|
2491
|
+
again at runtime. Common table column lists must match the selected arity.
|
|
2492
|
+
Structural identifiers reject empty strings and null characters. The model
|
|
2493
|
+
does not render or execute SQL; compilation and execution own those separate
|
|
2494
|
+
responsibilities.
|
|
2495
|
+
|
|
2496
|
+
Run the SELECT construction example with:
|
|
2497
|
+
|
|
2498
|
+
```console
|
|
2499
|
+
python examples/select_queries.py
|
|
2500
|
+
```
|
|
2501
|
+
|
|
2502
|
+
## Typed INSERT statements
|
|
2503
|
+
|
|
2504
|
+
`insert_into(table, *columns)` mirrors SQL and types the values by the selected
|
|
2505
|
+
columns. Each `values()` call adds one row, and repeated calls compile into a
|
|
2506
|
+
single multi-row statement rather than separate statements. Statement
|
|
2507
|
+
construction is immutable and performs no I/O.
|
|
2508
|
+
|
|
2509
|
+
```python
|
|
2510
|
+
from pyoq.query import insert_into
|
|
2511
|
+
|
|
2512
|
+
created = (
|
|
2513
|
+
insert_into(USERS, USER_ID, USER_NAME).values(100, "Hermann").values(101, "Alfred")
|
|
2514
|
+
)
|
|
2515
|
+
```
|
|
2516
|
+
|
|
2517
|
+
The selected columns determine the exact positional types and arity that
|
|
2518
|
+
`values()` accepts, so a wrong order, type, or count is a static error in both
|
|
2519
|
+
strict type checkers. Values become bound parameters; typed expressions may be
|
|
2520
|
+
passed where a computed value is required.
|
|
2521
|
+
|
|
2522
|
+
Generated immutable insert values feed the same statement through
|
|
2523
|
+
`values_many()`:
|
|
2524
|
+
|
|
2525
|
+
```python
|
|
2526
|
+
statement = insert_into(USERS, USER_ID, USER_NAME).values_many(
|
|
2527
|
+
(User.builder().id(100).name("Hermann").build(),)
|
|
2528
|
+
)
|
|
2529
|
+
```
|
|
2530
|
+
|
|
2531
|
+
`values_many()` maps each generated value onto the selected columns by the
|
|
2532
|
+
field name recorded on the generated column descriptor. A value that leaves a
|
|
2533
|
+
selected column unset is rejected, because a selected column always requires a
|
|
2534
|
+
value. Handwritten `field()` columns carry no generated field name and are
|
|
2535
|
+
therefore positional only.
|
|
2536
|
+
|
|
2537
|
+
Generated columns cannot be written. Selecting one, or selecting a column the
|
|
2538
|
+
schema marks read-only, fails with a `QueryValidationError` before any SQL is
|
|
2539
|
+
produced.
|
|
2540
|
+
|
|
2541
|
+
`returning()` changes the terminal result type and produces a statement that
|
|
2542
|
+
result operations can read:
|
|
2543
|
+
|
|
2544
|
+
```python
|
|
2545
|
+
row = database.one(
|
|
2546
|
+
insert_into(USERS, USER_NAME).values("Hermann").returning(USER_ID, USER_NAME)
|
|
2547
|
+
)
|
|
2548
|
+
```
|
|
2549
|
+
|
|
2550
|
+
RETURNING requires SQLite 3.35 or later. It is governed by the `returning`
|
|
2551
|
+
compiler capability, and disabling that capability makes the clause fail closed
|
|
2552
|
+
with `UnsupportedQueryError`.
|
|
2553
|
+
|
|
2554
|
+
Statements without a returning clause execute through `execute()`, which
|
|
2555
|
+
reports the affected row count and the last inserted row identifier:
|
|
2556
|
+
|
|
2557
|
+
```python
|
|
2558
|
+
result = database.execute(insert_into(USERS, USER_ID, USER_NAME).values(100, "Hermann"))
|
|
2559
|
+
```
|
|
2560
|
+
|
|
2561
|
+
Run the typed INSERT example with:
|
|
2562
|
+
|
|
2563
|
+
```console
|
|
2564
|
+
python examples/typed_inserts.py
|
|
2565
|
+
```
|
|
2566
|
+
|
|
2567
|
+
## Typed UPDATE and DELETE
|
|
2568
|
+
|
|
2569
|
+
`update(table)` and `delete_from(table)` build the same kind of immutable
|
|
2570
|
+
statement. Assignments are typed by their column, and values become bound
|
|
2571
|
+
parameters.
|
|
2572
|
+
|
|
2573
|
+
```python
|
|
2574
|
+
from pyoq.query import delete_from, update
|
|
2575
|
+
|
|
2576
|
+
renamed = update(USERS).set(USER_NAME, "Ada Lovelace").where(USER_ID.eq(1))
|
|
2577
|
+
removed = delete_from(USERS).where(USER_ID.eq(1))
|
|
2578
|
+
```
|
|
2579
|
+
|
|
2580
|
+
A statement that would touch every row must say so. An UPDATE or DELETE with no
|
|
2581
|
+
WHERE condition fails with a `CompilationError` before any SQL reaches the
|
|
2582
|
+
database, so a forgotten condition cannot rewrite or empty a table:
|
|
2583
|
+
|
|
2584
|
+
```python
|
|
2585
|
+
update(USERS).set(USER_ACTIVE, False).all_rows()
|
|
2586
|
+
delete_from(USERS).all_rows()
|
|
2587
|
+
```
|
|
2588
|
+
|
|
2589
|
+
`all_rows()` and `where()` exclude each other. Defining both, or defining either
|
|
2590
|
+
twice, raises `QueryStateError`.
|
|
2591
|
+
|
|
2592
|
+
Generated immutable update values assign only the fields they actually set:
|
|
2593
|
+
|
|
2594
|
+
```python
|
|
2595
|
+
statement = update(USERS).set_values(
|
|
2596
|
+
User.update_builder().name("Ada").build(),
|
|
2597
|
+
USER_NAME,
|
|
2598
|
+
USER_ACTIVE,
|
|
2599
|
+
)
|
|
2600
|
+
```
|
|
2601
|
+
|
|
2602
|
+
Unset fields are skipped, which is what makes an update value a partial update.
|
|
2603
|
+
A value that leaves every named column unset is rejected, because that statement
|
|
2604
|
+
would assign nothing. As with inserts, generated columns and read-only columns
|
|
2605
|
+
cannot be assigned.
|
|
2606
|
+
|
|
2607
|
+
`returning()` is available on every write form and produces the same typed
|
|
2608
|
+
terminal statement:
|
|
2609
|
+
|
|
2610
|
+
```python
|
|
2611
|
+
row = database.one(
|
|
2612
|
+
delete_from(USERS).where(USER_ID.eq(1)).returning(USER_ID, USER_NAME)
|
|
2613
|
+
)
|
|
2614
|
+
```
|
|
2615
|
+
|
|
2616
|
+
Run the typed UPDATE and DELETE example with:
|
|
2617
|
+
|
|
2618
|
+
```console
|
|
2619
|
+
python examples/typed_updates.py
|
|
2620
|
+
```
|
|
2621
|
+
|
|
2622
|
+
## Conflict resolution
|
|
2623
|
+
|
|
2624
|
+
An INSERT can resolve a uniqueness conflict instead of failing. The conflict
|
|
2625
|
+
target names the columns whose constraint is being resolved:
|
|
2626
|
+
|
|
2627
|
+
```python
|
|
2628
|
+
from pyoq.query import excluded, insert_into
|
|
2629
|
+
|
|
2630
|
+
ignored = (
|
|
2631
|
+
insert_into(USERS, USER_ID, USER_NAME)
|
|
2632
|
+
.values(1, "Ada")
|
|
2633
|
+
.on_conflict_do_nothing(USER_ID)
|
|
2634
|
+
)
|
|
2635
|
+
|
|
2636
|
+
merged = (
|
|
2637
|
+
insert_into(USERS, USER_ID, USER_NAME)
|
|
2638
|
+
.values(1, "Ada Lovelace")
|
|
2639
|
+
.on_conflict_do_update(USER_ID)
|
|
2640
|
+
.set(USER_NAME, excluded(USER_NAME))
|
|
2641
|
+
)
|
|
2642
|
+
```
|
|
2643
|
+
|
|
2644
|
+
`excluded()` refers to the row the statement tried to insert, keeping the value
|
|
2645
|
+
type of the column it names. It is how a conflicting update reads the new value
|
|
2646
|
+
rather than the stored one.
|
|
2647
|
+
|
|
2648
|
+
A conflict target must be one of the columns the statement inserts, so a typo or
|
|
2649
|
+
a column from another table fails with a `QueryValidationError` rather than
|
|
2650
|
+
producing a statement that never matches. `on_conflict_do_nothing()` may omit the
|
|
2651
|
+
target, which resolves a conflict on any constraint. Resolution that updates
|
|
2652
|
+
always requires a target, because SQLite cannot infer one.
|
|
2653
|
+
|
|
2654
|
+
A conflicting update can also filter which rows it touches:
|
|
2655
|
+
|
|
2656
|
+
```python
|
|
2657
|
+
merged = (
|
|
2658
|
+
insert_into(USERS, USER_ID, USER_NAME)
|
|
2659
|
+
.values(1, "Ada Lovelace")
|
|
2660
|
+
.on_conflict_do_update(USER_ID)
|
|
2661
|
+
.set(USER_NAME, excluded(USER_NAME))
|
|
2662
|
+
.where(USER_NAME.ne("locked"))
|
|
2663
|
+
)
|
|
2664
|
+
```
|
|
2665
|
+
|
|
2666
|
+
Conflict resolution can be defined once per statement, and an update resolution
|
|
2667
|
+
that assigns nothing fails at compilation. The `upsert` compiler capability
|
|
2668
|
+
disables the whole feature and makes it fail closed with
|
|
2669
|
+
`UnsupportedQueryError`.
|
|
2670
|
+
|
|
2671
|
+
Run the conflict resolution example with:
|
|
2672
|
+
|
|
2673
|
+
```console
|
|
2674
|
+
python examples/typed_upserts.py
|
|
2675
|
+
```
|
|
2676
|
+
|
|
2677
|
+
## Bulk and multi-operation writes
|
|
2678
|
+
|
|
2679
|
+
SQLite binds a bounded number of parameters per statement, so a large multi-row
|
|
2680
|
+
insert cannot be one statement. The bulk planner splits it into the fewest
|
|
2681
|
+
statements that each stay inside the budget, and it does so without executing
|
|
2682
|
+
anything:
|
|
2683
|
+
|
|
2684
|
+
```python
|
|
2685
|
+
plan = database.plan_bulk(statement)
|
|
2686
|
+
|
|
2687
|
+
plan.statement_count
|
|
2688
|
+
plan.input_rows
|
|
2689
|
+
plan.row_counts
|
|
2690
|
+
```
|
|
2691
|
+
|
|
2692
|
+
Chunks are contiguous and cover every input row exactly once, so each planned
|
|
2693
|
+
statement maps back to a known range of input rows through `start` and
|
|
2694
|
+
`length`. That is the correlation the plan guarantees. SQLite does not define
|
|
2695
|
+
the row order a `RETURNING` clause produces, so returned rows correlate to a
|
|
2696
|
+
chunk rather than to an individual input row.
|
|
2697
|
+
|
|
2698
|
+
Planning accounts for the parameters a statement spends outside its rows.
|
|
2699
|
+
Conflict assignments, conflict conditions, and returning projections reserve
|
|
2700
|
+
their share of the budget before rows are packed, and rows whose values are not
|
|
2701
|
+
plain bound parameters are measured individually rather than assumed.
|
|
2702
|
+
|
|
2703
|
+
```python
|
|
2704
|
+
result = database.execute_bulk(statement)
|
|
2705
|
+
|
|
2706
|
+
result.rows_affected
|
|
2707
|
+
result.statement_count
|
|
2708
|
+
result.row_counts
|
|
2709
|
+
```
|
|
2710
|
+
|
|
2711
|
+
A bulk write is only atomic inside a transaction. Executed directly, each
|
|
2712
|
+
planned statement commits on its own, so a failure in a later chunk leaves the
|
|
2713
|
+
earlier chunks applied. Wrap the call in a transaction when the whole batch must
|
|
2714
|
+
succeed or fail together:
|
|
2715
|
+
|
|
2716
|
+
```python
|
|
2717
|
+
with database.transaction() as transaction:
|
|
2718
|
+
transaction.execute_bulk(statement)
|
|
2719
|
+
```
|
|
2720
|
+
|
|
2721
|
+
`many_bulk()` runs the same plan and returns the returned rows from every chunk.
|
|
2722
|
+
|
|
2723
|
+
Ordered multi-operation execution runs a sequence of write statements in order
|
|
2724
|
+
under an explicit budget, so an unbounded batch cannot be submitted by accident:
|
|
2725
|
+
|
|
2726
|
+
```python
|
|
2727
|
+
from pyoq.query.execution import OperationBudget
|
|
2728
|
+
|
|
2729
|
+
results = database.execute_all(
|
|
2730
|
+
(first_statement, second_statement),
|
|
2731
|
+
budget=OperationBudget(maximum_operations=8),
|
|
2732
|
+
)
|
|
2733
|
+
```
|
|
2734
|
+
|
|
2735
|
+
An empty sequence executes nothing and returns an empty result tuple. Execution
|
|
2736
|
+
stops at the first failure, and the same atomicity rule applies: use a
|
|
2737
|
+
transaction when the earlier operations must not survive a later failure.
|
|
2738
|
+
|
|
2739
|
+
Run the bulk write example with:
|
|
2740
|
+
|
|
2741
|
+
```console
|
|
2742
|
+
python examples/bulk_writes.py
|
|
2743
|
+
```
|
|
2744
|
+
|
|
2745
|
+
## SQLite compilation
|
|
2746
|
+
|
|
2747
|
+
`SQLiteCompiler` renders a query into immutable SQL text and a separate ordered
|
|
2748
|
+
parameter tuple. Python values never enter the SQL string. Pagination values
|
|
2749
|
+
use the same parameter path, and sensitive bound values are recorded by their
|
|
2750
|
+
zero-based parameter positions.
|
|
2751
|
+
|
|
2752
|
+
```python
|
|
2753
|
+
from pyoq.query import bind, select
|
|
2754
|
+
from pyoq.query.sqlite import SQLiteCompiler
|
|
2755
|
+
|
|
2756
|
+
compiled = SQLiteCompiler().compile(
|
|
2757
|
+
select(USER_ID, USER_NAME)
|
|
2758
|
+
.from_(USERS)
|
|
2759
|
+
.where(USER_ID.gt(bind(100, sensitive=True)))
|
|
2760
|
+
.limit(20)
|
|
2761
|
+
)
|
|
2762
|
+
|
|
2763
|
+
assert compiled.parameters == (100, 20)
|
|
2764
|
+
assert compiled.sensitive_parameter_indexes == frozenset({0})
|
|
2765
|
+
```
|
|
2766
|
+
|
|
2767
|
+
Identifiers are quoted component by component. Operator expressions are
|
|
2768
|
+
parenthesized deterministically, null equality is rendered with `IS NULL`, and
|
|
2769
|
+
empty membership predicates become constant boolean expressions. Derived
|
|
2770
|
+
tables, common tables, joins, aggregates, and compound queries preserve the
|
|
2771
|
+
textual order of their parameters.
|
|
2772
|
+
|
|
2773
|
+
```mermaid
|
|
2774
|
+
flowchart LR
|
|
2775
|
+
Query[Immutable query tree] --> Capability[SQLite capability checks]
|
|
2776
|
+
Capability --> Compiler[SQLite compiler]
|
|
2777
|
+
Compiler --> SQL[Quoted SQL structure]
|
|
2778
|
+
Compiler --> Parameters[Ordered parameter tuple]
|
|
2779
|
+
Compiler --> Sensitive[Sensitive parameter positions]
|
|
2780
|
+
```
|
|
2781
|
+
|
|
2782
|
+
Compiler capabilities are explicit and immutable. RIGHT JOIN and FULL JOIN are
|
|
2783
|
+
disabled by default because support depends on the SQLite runtime. Recursive
|
|
2784
|
+
common tables, explicit null ordering, raw expressions, write returning,
|
|
2785
|
+
conflict resolution, and the maximum parameter budget can also be restricted. A query requiring a
|
|
2786
|
+
disabled or unrepresentable feature fails with an exception from `pyoq.errors`.
|
|
2787
|
+
|
|
2788
|
+
The compiler also renders write statements. An INSERT compiles into one
|
|
2789
|
+
statement whose rows share the same quoted column list, an UPDATE renders its
|
|
2790
|
+
assignments in the order they were added, and every value in either form is a
|
|
2791
|
+
bound parameter. An UPDATE or DELETE that has neither a WHERE condition nor an
|
|
2792
|
+
explicit full-table opt-in fails here rather than reaching the database.
|
|
2793
|
+
|
|
2794
|
+
Temporal arithmetic is not translated into SQLite numeric operators. Use an
|
|
2795
|
+
explicit typed raw expression when the intended SQLite date function is known.
|
|
2796
|
+
Raw expression text cannot introduce bind markers, statement separators, or
|
|
2797
|
+
SQL comment syntax. Catalog-qualified identifiers also fail because SQLite has
|
|
2798
|
+
no matching catalog level.
|
|
2799
|
+
|
|
2800
|
+
Run the compiler example with:
|
|
2801
|
+
|
|
2802
|
+
```console
|
|
2803
|
+
python examples/sqlite_compilation.py
|
|
2804
|
+
```
|
|
2805
|
+
|
|
2806
|
+
## Synchronous SQLite execution
|
|
2807
|
+
|
|
2808
|
+
`SQLitePool` provides bounded, exclusive connection leases. One checked-out
|
|
2809
|
+
connection belongs to one caller until the lease exits. Application exceptions
|
|
2810
|
+
return healthy connections, while failures that indicate a broken connection
|
|
2811
|
+
discard only that connection. Checkout waits are bounded by policy.
|
|
2812
|
+
|
|
2813
|
+
```python
|
|
2814
|
+
from pyoq.query import select
|
|
2815
|
+
from pyoq.query.sqlite import (
|
|
2816
|
+
SQLiteConnectionFactory,
|
|
2817
|
+
SQLiteExecutor,
|
|
2818
|
+
SQLitePool,
|
|
2819
|
+
SQLitePoolPolicy,
|
|
2820
|
+
)
|
|
2821
|
+
|
|
2822
|
+
pool = SQLitePool(
|
|
2823
|
+
SQLiteConnectionFactory("application.sqlite"),
|
|
2824
|
+
SQLitePoolPolicy(maximum_size=10, checkout_timeout=5.0),
|
|
2825
|
+
)
|
|
2826
|
+
db = SQLiteExecutor(pool)
|
|
2827
|
+
|
|
2828
|
+
rows: list[tuple[int, str]] = db.many(
|
|
2829
|
+
select(USER_ID, USER_NAME).from_(USERS).order_by(USER_ID)
|
|
2830
|
+
)
|
|
2831
|
+
selected: tuple[int, str] = db.one(
|
|
2832
|
+
select(USER_ID, USER_NAME).from_(USERS).where(USER_ID.eq(1))
|
|
2833
|
+
)
|
|
2834
|
+
name: str = db.scalar(select(USER_NAME).from_(USERS).where(USER_ID.eq(1)))
|
|
2835
|
+
|
|
2836
|
+
pool.close()
|
|
2837
|
+
```
|
|
2838
|
+
|
|
2839
|
+
`one` requires exactly one row, `one_or_none` accepts zero or one, `many`
|
|
2840
|
+
returns all rows, and `scalar` requires a single projected value. Their return
|
|
2841
|
+
types follow the query projection. `execute` accepts an immutable compiled
|
|
2842
|
+
statement and returns affected-row metadata. Insert identifiers are populated
|
|
2843
|
+
only for statements carrying explicit insert classification, which prevents
|
|
2844
|
+
stale driver metadata from escaping.
|
|
2845
|
+
|
|
2846
|
+
SQLite-compatible scalar parameters are adapted before a connection is
|
|
2847
|
+
checked out. Text, numbers, bytes, temporal values, decimal values, UUIDs,
|
|
2848
|
+
enums, memory views, and JSON containers have deterministic representations.
|
|
2849
|
+
Unsupported values fail with `ParameterBindingError` before pool capacity is
|
|
2850
|
+
used.
|
|
2851
|
+
|
|
2852
|
+
```mermaid
|
|
2853
|
+
flowchart LR
|
|
2854
|
+
Query[Typed query] --> Compiler[SQLite compiler]
|
|
2855
|
+
Compiler --> Adapter[Parameter adapter]
|
|
2856
|
+
Adapter --> Pool[Exclusive pool lease]
|
|
2857
|
+
Pool --> Driver[SQLite driver]
|
|
2858
|
+
Driver --> Cardinality[Typed result policy]
|
|
2859
|
+
Cardinality --> Result[Row, scalar, or execute result]
|
|
2860
|
+
```
|
|
2861
|
+
|
|
2862
|
+
Run the execution example with:
|
|
2863
|
+
|
|
2864
|
+
```console
|
|
2865
|
+
python examples/sqlite_execution.py
|
|
2866
|
+
```
|
|
2867
|
+
|
|
2868
|
+
## Transactions and bounded streaming
|
|
2869
|
+
|
|
2870
|
+
`SQLiteTransaction` owns one pool lease for its complete context. Normal exit
|
|
2871
|
+
commits, while every exception derived from `BaseException` rolls back. A
|
|
2872
|
+
transaction can only be used by its owner thread, and only its innermost active
|
|
2873
|
+
scope may issue work. Nested scopes use generated savepoint identifiers.
|
|
2874
|
+
|
|
2875
|
+
```python
|
|
2876
|
+
from pyoq.query.sqlite import TransactionMode
|
|
2877
|
+
|
|
2878
|
+
with db.transaction(TransactionMode.IMMEDIATE) as transaction:
|
|
2879
|
+
transaction.execute(FIRST_INSERT)
|
|
2880
|
+
|
|
2881
|
+
try:
|
|
2882
|
+
with transaction.savepoint() as nested:
|
|
2883
|
+
nested.execute(OPTIONAL_INSERT)
|
|
2884
|
+
raise ValueError("discard optional work")
|
|
2885
|
+
except ValueError:
|
|
2886
|
+
pass
|
|
2887
|
+
|
|
2888
|
+
transaction.execute(FINAL_INSERT)
|
|
2889
|
+
```
|
|
2890
|
+
|
|
2891
|
+
Statement failures that leave the SQLite connection healthy can be handled
|
|
2892
|
+
inside the transaction. Connection failures make the complete transaction
|
|
2893
|
+
unusable. A commit or savepoint failure discards the connection because its
|
|
2894
|
+
outcome cannot be assumed safely.
|
|
2895
|
+
|
|
2896
|
+
`stream` returns a typed iterator that fetches at most the configured batch
|
|
2897
|
+
size. It checks out no connection until it is entered or iterated, returns the
|
|
2898
|
+
lease automatically on exhaustion, and closes on iteration failure. Use its
|
|
2899
|
+
context manager whenever iteration may stop early.
|
|
2900
|
+
|
|
2901
|
+
```python
|
|
2902
|
+
from pyoq.query.execution import CancellationToken, ExecutionControl
|
|
2903
|
+
from pyoq.query.sqlite import SQLiteStreamPolicy
|
|
2904
|
+
|
|
2905
|
+
token = CancellationToken()
|
|
2906
|
+
policy = SQLiteStreamPolicy(
|
|
2907
|
+
batch_size=128,
|
|
2908
|
+
control=ExecutionControl(
|
|
2909
|
+
timeout=2.5,
|
|
2910
|
+
cancellation_token=token,
|
|
2911
|
+
progress_steps=500,
|
|
2912
|
+
),
|
|
2913
|
+
)
|
|
2914
|
+
|
|
2915
|
+
with db.stream(
|
|
2916
|
+
select(USER_ID, USER_NAME).from_(USERS).order_by(USER_ID),
|
|
2917
|
+
policy=policy,
|
|
2918
|
+
) as rows:
|
|
2919
|
+
for user_id, user_name in rows:
|
|
2920
|
+
consume(user_id, user_name)
|
|
2921
|
+
```
|
|
2922
|
+
|
|
2923
|
+
Timeout and cancellation checks run before execution, through SQLite progress
|
|
2924
|
+
callbacks, and between streamed rows. `QueryTimeoutError` and
|
|
2925
|
+
`QueryCancelledError` are distinct typed failures. Cancellation is cooperative:
|
|
2926
|
+
another thread may call `token.cancel()`, but it must not use the owned
|
|
2927
|
+
connection or consume the row stream.
|
|
2928
|
+
|
|
2929
|
+
```mermaid
|
|
2930
|
+
flowchart TD
|
|
2931
|
+
Executor[SQLiteExecutor] --> Lease[Exclusive pool lease]
|
|
2932
|
+
Lease --> Transaction[Owned transaction]
|
|
2933
|
+
Transaction --> Savepoint[Nested savepoint]
|
|
2934
|
+
Transaction --> Stream[Bounded row stream]
|
|
2935
|
+
Stream --> Batch[At most batch_size buffered rows]
|
|
2936
|
+
Control[Timeout or cancellation] --> Progress[SQLite progress callback]
|
|
2937
|
+
Progress --> Cleanup[Cursor, handler, transaction, and lease cleanup]
|
|
2938
|
+
Batch --> Cleanup
|
|
2939
|
+
Savepoint --> Cleanup
|
|
2940
|
+
```
|
|
2941
|
+
|
|
2942
|
+
Run the transaction and streaming example with:
|
|
2943
|
+
|
|
2944
|
+
```console
|
|
2945
|
+
python examples/sqlite_transactions.py
|
|
2946
|
+
```
|
|
2947
|
+
|
|
2948
|
+
## Command line
|
|
2949
|
+
|
|
2950
|
+
The package installs the `pyoq` command and also supports `python -m pyoq`.
|
|
2951
|
+
|
|
2952
|
+
```console
|
|
2953
|
+
pyoq --version
|
|
2954
|
+
pyoq generate --project-root .
|
|
2955
|
+
pyoq generate --project-root . --check
|
|
2956
|
+
pyoq generate --project-root . --dry-run
|
|
2957
|
+
pyoq inspect --project-root .
|
|
2958
|
+
```
|
|
2959
|
+
|
|
2960
|
+
Both database commands accept `--config` for a configuration file below the
|
|
2961
|
+
project root and `--profile` for explicit profile selection. `--check` and
|
|
2962
|
+
`--dry-run` are mutually exclusive.
|
|
2963
|
+
|
|
2964
|
+
Command exit codes are stable:
|
|
2965
|
+
|
|
2966
|
+
- `0`: success
|
|
2967
|
+
- `1`: command service failure
|
|
2968
|
+
- `2`: invalid command usage
|
|
2969
|
+
- `3`: configuration failure
|
|
2970
|
+
- `4`: requested operation is unavailable
|
|
2971
|
+
|
|
2972
|
+
SQLite generation and inspection are available through the default command
|
|
2973
|
+
services. Inspection reads tables, views, columns, defaults, generated values,
|
|
2974
|
+
primary and unique keys, indexes, CHECK constraints, and foreign-key
|
|
2975
|
+
relationships. Metadata is normalized before rendering, and the connection is
|
|
2976
|
+
closed before generated files are checked or written.
|
|
2977
|
+
|
|
2978
|
+
Generation writes six deterministic, fully typed modules and an ownership
|
|
2979
|
+
manifest into `codegen-directory`. Repeating a write against an unchanged
|
|
2980
|
+
database leaves the generated directory untouched. `--check` fails on drift,
|
|
2981
|
+
while `--dry-run` reports planned creation, update, removal, and conflict counts
|
|
2982
|
+
without changing files. Generation runs against SQLite, PostgreSQL, and
|
|
2983
|
+
MySQL.
|
|
2984
|
+
|
|
2985
|
+
Run the complete SQLite reflection and generation example with:
|
|
2986
|
+
|
|
2987
|
+
```console
|
|
2988
|
+
python examples/sqlite_generation.py
|
|
2989
|
+
```
|
|
2990
|
+
|
|
2991
|
+
## Performance contract
|
|
2992
|
+
|
|
2993
|
+
Run the native and portable kernel comparison from an editable native build:
|
|
2994
|
+
|
|
2995
|
+
```console
|
|
2996
|
+
python -m benchmarks.performance_contract
|
|
2997
|
+
```
|
|
2998
|
+
|
|
2999
|
+
The command checks relative throughput, traced peak memory, retained
|
|
3000
|
+
allocations, and native artifact size. It exits unsuccessfully when a measured
|
|
3001
|
+
kernel exceeds its calibrated budget.
|
|
3002
|
+
|
|
3003
|
+
## Distribution checks
|
|
3004
|
+
|
|
3005
|
+
Build the native wheel with the release backend:
|
|
3006
|
+
|
|
3007
|
+
```console
|
|
3008
|
+
maturin build --release --locked --out wheelhouse
|
|
3009
|
+
```
|
|
3010
|
+
|
|
3011
|
+
Build the universal portable wheel without compiling the native extension:
|
|
3012
|
+
|
|
3013
|
+
```console
|
|
3014
|
+
hatchling build -t wheel -d wheelhouse
|
|
3015
|
+
```
|
|
3016
|
+
|
|
3017
|
+
When both artifact kinds are present, validate their tags, native contents,
|
|
3018
|
+
license, type information, and shared package metadata:
|
|
3019
|
+
|
|
3020
|
+
```console
|
|
3021
|
+
python -m scripts.verify_wheels wheelhouse
|
|
3022
|
+
```
|
|
3023
|
+
|
|
3024
|
+
Continuous integration builds native wheels for glibc and musl Linux on x86-64
|
|
3025
|
+
and ARM64, macOS on Apple Silicon and x86-64, and Windows on x86-64 and ARM64.
|
|
3026
|
+
The stable native ABI is installed and imported on every supported Python
|
|
3027
|
+
version. The universal wheel is also installed through an incompatible-platform
|
|
3028
|
+
resolver test to prove that it does not require Rust.
|
|
3029
|
+
|
|
3030
|
+
## Design goals
|
|
3031
|
+
|
|
3032
|
+
- Fully typed public APIs compatible with mypy and Pyright strict modes.
|
|
3033
|
+
- Database-first schema generation with deterministic output.
|
|
3034
|
+
- Explicit, immutable SQL query models and bound parameters.
|
|
3035
|
+
- PostgreSQL, MySQL, and SQLite support.
|
|
3036
|
+
- Django, FastAPI, and Sanic integrations.
|
|
3037
|
+
- Connection pooling, transaction safety, and observable query execution.
|
|
3038
|
+
- Fetch planning that prevents hidden I/O and N+1 query behavior.
|
|
3039
|
+
- Native hot paths with measured throughput, allocation, and peak-memory gates.
|
|
3040
|
+
|
|
3041
|
+
## Supported Python
|
|
3042
|
+
|
|
3043
|
+
PyOQ requires Python 3.11 or newer.
|
|
3044
|
+
|
|
3045
|
+
## License
|
|
3046
|
+
|
|
3047
|
+
PyOQ is licensed under the Mozilla Public License 2.0. Applications may use it
|
|
3048
|
+
commercially, including as part of larger proprietary products. Changes to
|
|
3049
|
+
covered PyOQ source files remain subject to the MPL terms. See `LICENSE` for the
|
|
3050
|
+
complete terms.
|