ddxdb 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ddxdb-0.1.0/Cargo.toml +58 -0
- ddxdb-0.1.0/PKG-INFO +185 -0
- ddxdb-0.1.0/README.md +153 -0
- ddxdb-0.1.0/crates/ddx-core/CHANGELOG.md +189 -0
- ddxdb-0.1.0/crates/ddx-core/Cargo.toml +34 -0
- ddxdb-0.1.0/crates/ddx-core/README.md +64 -0
- ddxdb-0.1.0/crates/ddx-core/src/colref.rs +300 -0
- ddxdb-0.1.0/crates/ddx-core/src/constructors.rs +566 -0
- ddxdb-0.1.0/crates/ddx-core/src/ddx.rs +181 -0
- ddxdb-0.1.0/crates/ddx-core/src/engine.rs +535 -0
- ddxdb-0.1.0/crates/ddx-core/src/error.rs +65 -0
- ddxdb-0.1.0/crates/ddx-core/src/lib.rs +80 -0
- ddxdb-0.1.0/crates/ddx-core/src/rewrite.rs +804 -0
- ddxdb-0.1.0/crates/ddx-core/src/test_utils.rs +984 -0
- ddxdb-0.1.0/crates/ddx-core/tests/rewrite.rs +395 -0
- ddxdb-0.1.0/crates/ddx-core/tests/roundtrip.rs +138 -0
- ddxdb-0.1.0/crates/ddx-core/tests/rules.rs +307 -0
- ddxdb-0.1.0/crates/ddx-core/tests/simulation.rs +970 -0
- ddxdb-0.1.0/ddxdb/__init__.py +93 -0
- ddxdb-0.1.0/ddxdb/datafusion.py +61 -0
- ddxdb-0.1.0/pyproject.toml +50 -0
- ddxdb-0.1.0/python/ddxdb/Cargo.lock +278 -0
- ddxdb-0.1.0/python/ddxdb/Cargo.toml +39 -0
- ddxdb-0.1.0/python/ddxdb/README.md +153 -0
- ddxdb-0.1.0/python/ddxdb/src/lib.rs +240 -0
- ddxdb-0.1.0/python/ddxdb/tests/test_ddxdb.py +317 -0
ddxdb-0.1.0/Cargo.toml
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# ddx — portable autograd for composable databases.
|
|
2
|
+
#
|
|
3
|
+
# One engine-neutral differentiation core (`ddx-core`, v1 / M0) plus thin
|
|
4
|
+
# per-engine adapters. See docs/design.md §6 for the layout and the
|
|
5
|
+
# dependency policy: `ddx-core` depends on `sqlparser` only; the heavy
|
|
6
|
+
# per-engine dependencies (datafusion, duckdb) are quarantined in the
|
|
7
|
+
# adapter crates.
|
|
8
|
+
[workspace]
|
|
9
|
+
resolver = "2"
|
|
10
|
+
members = [
|
|
11
|
+
"crates/ddx-core",
|
|
12
|
+
"crates/ddx-ad",
|
|
13
|
+
"crates/ddx-datafusion",
|
|
14
|
+
]
|
|
15
|
+
|
|
16
|
+
[workspace.package]
|
|
17
|
+
version = "0.0.0"
|
|
18
|
+
edition = "2021"
|
|
19
|
+
license = "Apache-2.0"
|
|
20
|
+
repository = "https://github.com/xqlsystems/ddx"
|
|
21
|
+
authors = ["Alexander Merose <al@merose.com>"]
|
|
22
|
+
# MSRV. Enforced by the CI `msrv` matrix entry. The floor is set by a
|
|
23
|
+
# transitive build-dependency (ar_archive_writer, which uses edition 2024);
|
|
24
|
+
# ddx-core's own code needs far less, but the resolved tree does not build below
|
|
25
|
+
# this. Bump only with a matching CI change.
|
|
26
|
+
rust-version = "1.88"
|
|
27
|
+
|
|
28
|
+
[workspace.dependencies]
|
|
29
|
+
# The single load-bearing dependency of the core. Pinned exactly (see the
|
|
30
|
+
# `sqlparser` version policy in docs/design.md §6): a bump is a breaking
|
|
31
|
+
# release of ddx-core. 0.62 is the version every M0 spike was verified
|
|
32
|
+
# against (spikes/sqlparser-spike, decision-log G1/G3).
|
|
33
|
+
sqlparser = { version = "=0.62.0", features = ["visitor"] }
|
|
34
|
+
|
|
35
|
+
# The DataFusion adapter's engine dependency (M2), quarantined to
|
|
36
|
+
# `ddx-datafusion` — neither core crate may depend on it.
|
|
37
|
+
#
|
|
38
|
+
# The version is load-bearing for Path B, not incidental. The bridge unparses a
|
|
39
|
+
# bound DataFusion `Expr` into a `sqlparser::ast::Expr` and hands it to
|
|
40
|
+
# ddx-core, so the two crates MUST resolve the *identical* `sqlparser` — a
|
|
41
|
+
# mismatch makes them two unrelated Rust types and the bridge stops compiling
|
|
42
|
+
# datafusion 54.x requires `sqlparser ^0.62.0`,
|
|
43
|
+
# which unifies with the exact `=0.62.0` pin above; 53.x wanted ^0.61 and would
|
|
44
|
+
# NOT. `crates/ddx-datafusion/tests/sqlparser_pin.rs` asserts the resolved tree
|
|
45
|
+
# still has exactly one `sqlparser`, so a future bump fails loudly at the pin
|
|
46
|
+
# instead of confusingly at the bridge.
|
|
47
|
+
datafusion = { version = "54" }
|
|
48
|
+
|
|
49
|
+
# Internal crates, referenced by path within the workspace.
|
|
50
|
+
#
|
|
51
|
+
# `ddx-core` carries a `version` as well as a `path`: cargo uses the path for
|
|
52
|
+
# workspace builds and the version for what it writes into a published
|
|
53
|
+
# dependent's manifest, and it refuses to publish a crate whose path dependency
|
|
54
|
+
# has no version. The floor is 0.2.1 rather than 0.2 because `grad(abs(u), x)`
|
|
55
|
+
# returned a zero gradient for a NULL row in 0.2.0, and an adapter that resolved
|
|
56
|
+
# to it would hand users that bug.
|
|
57
|
+
ddx-core = { path = "crates/ddx-core", version = "0.2.1" }
|
|
58
|
+
ddx-ad = { path = "crates/ddx-ad" }
|
ddxdb-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ddxdb
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Classifier: Development Status :: 3 - Alpha
|
|
5
|
+
Classifier: Intended Audience :: Science/Research
|
|
6
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
12
|
+
Classifier: Programming Language :: Rust
|
|
13
|
+
Classifier: Topic :: Database
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
15
|
+
Requires-Dist: datafusion>=45 ; extra == 'datafusion'
|
|
16
|
+
Requires-Dist: duckdb>=1.0 ; extra == 'duckdb'
|
|
17
|
+
Requires-Dist: pytest>=7 ; extra == 'test'
|
|
18
|
+
Requires-Dist: datafusion>=45 ; extra == 'test'
|
|
19
|
+
Requires-Dist: duckdb>=1.0 ; extra == 'test'
|
|
20
|
+
Provides-Extra: datafusion
|
|
21
|
+
Provides-Extra: duckdb
|
|
22
|
+
Provides-Extra: test
|
|
23
|
+
Summary: SQL-portable autograd: write calculus in SQL, get derivatives back as columns.
|
|
24
|
+
Keywords: sql,autograd,derivatives,datafusion,duckdb
|
|
25
|
+
Author-email: Alexander Merose <al@merose.com>
|
|
26
|
+
License: Apache-2.0
|
|
27
|
+
Requires-Python: >=3.10
|
|
28
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
29
|
+
Project-URL: Homepage, https://github.com/xqlsystems/ddx
|
|
30
|
+
Project-URL: Repository, https://github.com/xqlsystems/ddx
|
|
31
|
+
|
|
32
|
+
# ddxdb
|
|
33
|
+
|
|
34
|
+
Write calculus directly in SQL and let the database evaluate the derivative, row
|
|
35
|
+
by row, alongside everything else:
|
|
36
|
+
|
|
37
|
+
```sql
|
|
38
|
+
SELECT i, grad(x * y, x) AS dfdx, grad(x * y, y) AS dfdy FROM g
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
`grad` and `jvp` are **markers**, not row functions. They are rewritten away into
|
|
42
|
+
ordinary derivative SQL *before* the engine sees them, so what runs is a plain
|
|
43
|
+
expression — the relational equivalent of `jax.vmap(jax.grad(f))`, with the rows
|
|
44
|
+
as the batch dimension.
|
|
45
|
+
|
|
46
|
+
This is the Python distribution of [`ddx`](https://github.com/xqlsystems/ddx), a
|
|
47
|
+
thin wrapper over the `ddx-core` engine.
|
|
48
|
+
|
|
49
|
+
## Install
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install ddxdb # everything below except Context
|
|
53
|
+
pip install "ddxdb[datafusion]" # + the DataFusion Context
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## `rewrite_sql` is the whole library
|
|
57
|
+
|
|
58
|
+
Text in, text out — so it works with **any** engine that accepts SQL. Pass the
|
|
59
|
+
result wherever you would have passed the original:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
import ddxdb
|
|
63
|
+
|
|
64
|
+
ddxdb.rewrite_sql("SELECT grad(sin(x), x) AS d FROM t")
|
|
65
|
+
# 'SELECT (cos(x)) AS d FROM t'
|
|
66
|
+
|
|
67
|
+
con.sql(ddxdb.rewrite_sql(q, "duckdb")) # DuckDB
|
|
68
|
+
session.sql(ddxdb.rewrite_sql(q, "spark")) # Spark
|
|
69
|
+
ctx.sql(ddxdb.rewrite_sql(q)) # DataFusion
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Accepted dialects: `generic`, `datafusion`, `postgres`, `ansi`, `snowflake`,
|
|
73
|
+
`oracle`, `duckdb`, `mysql`, `sqlite`, `bigquery`, `redshift`, `hive`, `spark`,
|
|
74
|
+
`databricks`, `mssql`, `teradata`, `clickhouse`.
|
|
75
|
+
|
|
76
|
+
Pick the one that matches the engine you will run on, not just the one that
|
|
77
|
+
parses your SQL. The dialect also decides which column an identifier *names*,
|
|
78
|
+
and engines disagree three ways:
|
|
79
|
+
|
|
80
|
+
| | unquoted `X` means | so `"X"` is |
|
|
81
|
+
|---|---|---|
|
|
82
|
+
| Postgres, DataFusion, generic, ansi | `"x"` | a different column |
|
|
83
|
+
| Snowflake, Oracle | `"X"` | the same column |
|
|
84
|
+
| DuckDB, Spark, MySQL, SQLite, BigQuery, Redshift, Hive, Databricks, SQL Server, Teradata | any casing | the same column |
|
|
85
|
+
| ClickHouse | `X` exactly | the same column, and `"x"` is not |
|
|
86
|
+
|
|
87
|
+
Getting this wrong does not raise. `grad("X" * "X", X)` is `2X` on Snowflake and
|
|
88
|
+
`0` on Postgres — both correct, for different engines — so ddx keeps a table
|
|
89
|
+
rather than a default, and refuses a dialect whose rule it has not established.
|
|
90
|
+
|
|
91
|
+
Because the rewrite happens in *your* process, on *your* connection, it sees
|
|
92
|
+
your temp tables, session settings, and open transaction. DuckDB's in-database
|
|
93
|
+
`ddx('<sql>')` table function cannot: it executes on a separate inner
|
|
94
|
+
connection.
|
|
95
|
+
|
|
96
|
+
## `Context`, for DataFusion
|
|
97
|
+
|
|
98
|
+
A real `SessionContext` subclass whose `.sql()` rewrites first — every inherited
|
|
99
|
+
method, property and constructor argument works unchanged:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
ctx = ddxdb.Context()
|
|
103
|
+
ctx.sql("SELECT grad(x * x, x) AS d FROM t").collect() # → 2x
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
It lives in `ddxdb.datafusion` (a subclass needs its base class at import time,
|
|
107
|
+
so it cannot sit beside `rewrite_sql` without dragging DataFusion in) and is
|
|
108
|
+
re-exported as `ddxdb.Context`, imported on first use. `import ddxdb` still needs
|
|
109
|
+
no engine.
|
|
110
|
+
|
|
111
|
+
There is sugar for DataFusion and not for other engines because DataFusion is
|
|
112
|
+
ddx's integration target. Everything else uses the one-liner above, which is why
|
|
113
|
+
there are no per-engine helpers here to drift out of date.
|
|
114
|
+
|
|
115
|
+
## What you can write
|
|
116
|
+
|
|
117
|
+
`+ - * /`; the chain rule for the trig / inverse-trig / exp / log / hyperbolic
|
|
118
|
+
set plus `abs`; `power` with a constant base or exponent. Higher order falls out
|
|
119
|
+
of nesting — `grad(grad(f, x), x)` just works. Differentiating through an
|
|
120
|
+
aggregate is linearity, so the marker goes *inside* it, which is what makes a
|
|
121
|
+
gradient-descent step expressible in SQL:
|
|
122
|
+
|
|
123
|
+
```sql
|
|
124
|
+
SELECT theta - 0.01 * AVG(grad(loss, theta)) FROM batch
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
A marker rewrites in place, so it is legal anywhere a scalar expression is —
|
|
128
|
+
including inside a recursive CTE, which is how a whole training loop fits in one
|
|
129
|
+
query.
|
|
130
|
+
|
|
131
|
+
## One other function
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
ddxdb.differentiate_sql("x * y", "x") # 'y' — the derivative as text
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
The escape hatch, for assembling SQL where a marker cannot reach — inside a
|
|
138
|
+
recursive term you are building programmatically, or a query some other tool
|
|
139
|
+
emits. Everything else should use `rewrite_sql`.
|
|
140
|
+
|
|
141
|
+
## Errors are typed
|
|
142
|
+
|
|
143
|
+
An unsupported construct is always an error, never a silently wrong number —
|
|
144
|
+
this is a numerical-correctness library, and a plausible-looking wrong
|
|
145
|
+
derivative is the worst thing it could produce. The kind of failure is a class,
|
|
146
|
+
so you can catch the one you can act on:
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
try:
|
|
150
|
+
ddxdb.rewrite_sql(query)
|
|
151
|
+
except ddxdb.UnsupportedExpression:
|
|
152
|
+
... # no rule for something in there — fall back
|
|
153
|
+
except ddxdb.AmbiguousColumn:
|
|
154
|
+
... # the query needs a qualifier — a fix the caller makes
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
All of them derive from `ddxdb.DdxError`. The full set is
|
|
158
|
+
`UnsupportedExpression`, `InvalidMarker`, `AmbiguousColumn`,
|
|
159
|
+
`ProjectionBoundary` and `SqlParseError`.
|
|
160
|
+
|
|
161
|
+
## One thing to know
|
|
162
|
+
|
|
163
|
+
**`grad` does not see through a CTE or a view.** Differentiation stops at column
|
|
164
|
+
references, so a column computed upstream is a constant to it:
|
|
165
|
+
|
|
166
|
+
```sql
|
|
167
|
+
WITH v AS (SELECT x, sin(x) AS s FROM t)
|
|
168
|
+
SELECT grad(s * x, x) FROM v -- ds/dx is treated as 0
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
That is defensible relational semantics and a real trap, so ddx refuses the
|
|
172
|
+
worst case rather than quietly dropping the term: referencing a computed CTE
|
|
173
|
+
alias as a non-`wrt` term raises `ProjectionBoundary` and tells you to
|
|
174
|
+
differentiate inside the CTE instead. Differentiating *with respect to* such an
|
|
175
|
+
alias is fine — every occurrence is then the differentiation leaf, and
|
|
176
|
+
`grad(s * s, s)` is exactly `2s`.
|
|
177
|
+
|
|
178
|
+
## Development
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
pip install maturin pytest
|
|
182
|
+
maturin develop --uv
|
|
183
|
+
python -m pytest tests/
|
|
184
|
+
```
|
|
185
|
+
|
ddxdb-0.1.0/README.md
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# ddxdb
|
|
2
|
+
|
|
3
|
+
Write calculus directly in SQL and let the database evaluate the derivative, row
|
|
4
|
+
by row, alongside everything else:
|
|
5
|
+
|
|
6
|
+
```sql
|
|
7
|
+
SELECT i, grad(x * y, x) AS dfdx, grad(x * y, y) AS dfdy FROM g
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
`grad` and `jvp` are **markers**, not row functions. They are rewritten away into
|
|
11
|
+
ordinary derivative SQL *before* the engine sees them, so what runs is a plain
|
|
12
|
+
expression — the relational equivalent of `jax.vmap(jax.grad(f))`, with the rows
|
|
13
|
+
as the batch dimension.
|
|
14
|
+
|
|
15
|
+
This is the Python distribution of [`ddx`](https://github.com/xqlsystems/ddx), a
|
|
16
|
+
thin wrapper over the `ddx-core` engine.
|
|
17
|
+
|
|
18
|
+
## Install
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
pip install ddxdb # everything below except Context
|
|
22
|
+
pip install "ddxdb[datafusion]" # + the DataFusion Context
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## `rewrite_sql` is the whole library
|
|
26
|
+
|
|
27
|
+
Text in, text out — so it works with **any** engine that accepts SQL. Pass the
|
|
28
|
+
result wherever you would have passed the original:
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
import ddxdb
|
|
32
|
+
|
|
33
|
+
ddxdb.rewrite_sql("SELECT grad(sin(x), x) AS d FROM t")
|
|
34
|
+
# 'SELECT (cos(x)) AS d FROM t'
|
|
35
|
+
|
|
36
|
+
con.sql(ddxdb.rewrite_sql(q, "duckdb")) # DuckDB
|
|
37
|
+
session.sql(ddxdb.rewrite_sql(q, "spark")) # Spark
|
|
38
|
+
ctx.sql(ddxdb.rewrite_sql(q)) # DataFusion
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Accepted dialects: `generic`, `datafusion`, `postgres`, `ansi`, `snowflake`,
|
|
42
|
+
`oracle`, `duckdb`, `mysql`, `sqlite`, `bigquery`, `redshift`, `hive`, `spark`,
|
|
43
|
+
`databricks`, `mssql`, `teradata`, `clickhouse`.
|
|
44
|
+
|
|
45
|
+
Pick the one that matches the engine you will run on, not just the one that
|
|
46
|
+
parses your SQL. The dialect also decides which column an identifier *names*,
|
|
47
|
+
and engines disagree three ways:
|
|
48
|
+
|
|
49
|
+
| | unquoted `X` means | so `"X"` is |
|
|
50
|
+
|---|---|---|
|
|
51
|
+
| Postgres, DataFusion, generic, ansi | `"x"` | a different column |
|
|
52
|
+
| Snowflake, Oracle | `"X"` | the same column |
|
|
53
|
+
| DuckDB, Spark, MySQL, SQLite, BigQuery, Redshift, Hive, Databricks, SQL Server, Teradata | any casing | the same column |
|
|
54
|
+
| ClickHouse | `X` exactly | the same column, and `"x"` is not |
|
|
55
|
+
|
|
56
|
+
Getting this wrong does not raise. `grad("X" * "X", X)` is `2X` on Snowflake and
|
|
57
|
+
`0` on Postgres — both correct, for different engines — so ddx keeps a table
|
|
58
|
+
rather than a default, and refuses a dialect whose rule it has not established.
|
|
59
|
+
|
|
60
|
+
Because the rewrite happens in *your* process, on *your* connection, it sees
|
|
61
|
+
your temp tables, session settings, and open transaction. DuckDB's in-database
|
|
62
|
+
`ddx('<sql>')` table function cannot: it executes on a separate inner
|
|
63
|
+
connection.
|
|
64
|
+
|
|
65
|
+
## `Context`, for DataFusion
|
|
66
|
+
|
|
67
|
+
A real `SessionContext` subclass whose `.sql()` rewrites first — every inherited
|
|
68
|
+
method, property and constructor argument works unchanged:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
ctx = ddxdb.Context()
|
|
72
|
+
ctx.sql("SELECT grad(x * x, x) AS d FROM t").collect() # → 2x
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
It lives in `ddxdb.datafusion` (a subclass needs its base class at import time,
|
|
76
|
+
so it cannot sit beside `rewrite_sql` without dragging DataFusion in) and is
|
|
77
|
+
re-exported as `ddxdb.Context`, imported on first use. `import ddxdb` still needs
|
|
78
|
+
no engine.
|
|
79
|
+
|
|
80
|
+
There is sugar for DataFusion and not for other engines because DataFusion is
|
|
81
|
+
ddx's integration target. Everything else uses the one-liner above, which is why
|
|
82
|
+
there are no per-engine helpers here to drift out of date.
|
|
83
|
+
|
|
84
|
+
## What you can write
|
|
85
|
+
|
|
86
|
+
`+ - * /`; the chain rule for the trig / inverse-trig / exp / log / hyperbolic
|
|
87
|
+
set plus `abs`; `power` with a constant base or exponent. Higher order falls out
|
|
88
|
+
of nesting — `grad(grad(f, x), x)` just works. Differentiating through an
|
|
89
|
+
aggregate is linearity, so the marker goes *inside* it, which is what makes a
|
|
90
|
+
gradient-descent step expressible in SQL:
|
|
91
|
+
|
|
92
|
+
```sql
|
|
93
|
+
SELECT theta - 0.01 * AVG(grad(loss, theta)) FROM batch
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
A marker rewrites in place, so it is legal anywhere a scalar expression is —
|
|
97
|
+
including inside a recursive CTE, which is how a whole training loop fits in one
|
|
98
|
+
query.
|
|
99
|
+
|
|
100
|
+
## One other function
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
ddxdb.differentiate_sql("x * y", "x") # 'y' — the derivative as text
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The escape hatch, for assembling SQL where a marker cannot reach — inside a
|
|
107
|
+
recursive term you are building programmatically, or a query some other tool
|
|
108
|
+
emits. Everything else should use `rewrite_sql`.
|
|
109
|
+
|
|
110
|
+
## Errors are typed
|
|
111
|
+
|
|
112
|
+
An unsupported construct is always an error, never a silently wrong number —
|
|
113
|
+
this is a numerical-correctness library, and a plausible-looking wrong
|
|
114
|
+
derivative is the worst thing it could produce. The kind of failure is a class,
|
|
115
|
+
so you can catch the one you can act on:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
try:
|
|
119
|
+
ddxdb.rewrite_sql(query)
|
|
120
|
+
except ddxdb.UnsupportedExpression:
|
|
121
|
+
... # no rule for something in there — fall back
|
|
122
|
+
except ddxdb.AmbiguousColumn:
|
|
123
|
+
... # the query needs a qualifier — a fix the caller makes
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
All of them derive from `ddxdb.DdxError`. The full set is
|
|
127
|
+
`UnsupportedExpression`, `InvalidMarker`, `AmbiguousColumn`,
|
|
128
|
+
`ProjectionBoundary` and `SqlParseError`.
|
|
129
|
+
|
|
130
|
+
## One thing to know
|
|
131
|
+
|
|
132
|
+
**`grad` does not see through a CTE or a view.** Differentiation stops at column
|
|
133
|
+
references, so a column computed upstream is a constant to it:
|
|
134
|
+
|
|
135
|
+
```sql
|
|
136
|
+
WITH v AS (SELECT x, sin(x) AS s FROM t)
|
|
137
|
+
SELECT grad(s * x, x) FROM v -- ds/dx is treated as 0
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
That is defensible relational semantics and a real trap, so ddx refuses the
|
|
141
|
+
worst case rather than quietly dropping the term: referencing a computed CTE
|
|
142
|
+
alias as a non-`wrt` term raises `ProjectionBoundary` and tells you to
|
|
143
|
+
differentiate inside the CTE instead. Differentiating *with respect to* such an
|
|
144
|
+
alias is fine — every occurrence is then the differentiation leaf, and
|
|
145
|
+
`grad(s * s, s)` is exactly `2s`.
|
|
146
|
+
|
|
147
|
+
## Development
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
pip install maturin pytest
|
|
151
|
+
maturin develop --uv
|
|
152
|
+
python -m pytest tests/
|
|
153
|
+
```
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to `ddx-core` are documented here.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
Entries below the first release are maintained automatically by
|
|
8
|
+
[release-plz](https://release-plz.dev/).
|
|
9
|
+
|
|
10
|
+
## [Unreleased]
|
|
11
|
+
|
|
12
|
+
## [0.2.1](https://github.com/xqlsystems/ddx/compare/ddx-core-v0.2.0...ddx-core-v0.2.1) - 2026-08-09
|
|
13
|
+
|
|
14
|
+
### Fixed
|
|
15
|
+
|
|
16
|
+
- **`grad(abs(u), x)` reported a zero gradient for a row whose value was NULL.**
|
|
17
|
+
The derivative of `abs` is emitted as a `CASE` on the sign of `u`, and SQL
|
|
18
|
+
comparisons are three-valued: against NULL they are NULL, not false. A row with
|
|
19
|
+
no value therefore answered *none* of the branches and fell through to `ELSE`,
|
|
20
|
+
which returned `0`:
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
x = [1.0, NULL, 3.0] grad(abs(x), x) -> [1.0, 0.0, 1.0]
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
**Who is affected.** Anyone differentiating an expression containing `abs` over
|
|
27
|
+
data that has gaps — on any engine; this was not engine-specific. The failure
|
|
28
|
+
is silent and the value is plausible, which is what makes it worth upgrading
|
|
29
|
+
for: a zero gradient is indistinguishable from a parameter that genuinely does
|
|
30
|
+
not move, so in a training loop a hole in the batch reads as a converged
|
|
31
|
+
weight. Expressions without `abs` were never affected, because arithmetic
|
|
32
|
+
propagates NULL on its own.
|
|
33
|
+
|
|
34
|
+
**What changed in the output.** The emitted `CASE` now states the kink as its
|
|
35
|
+
own condition and reserves `ELSE` for "no comparison answered":
|
|
36
|
+
|
|
37
|
+
```sql
|
|
38
|
+
-- before
|
|
39
|
+
CASE WHEN u > 0 THEN 1.0 WHEN u < 0 THEN -1.0 ELSE 0.0 END
|
|
40
|
+
-- after
|
|
41
|
+
CASE WHEN u > 0 THEN 1.0 WHEN u < 0 THEN -1.0 WHEN u = 0 THEN 0.0
|
|
42
|
+
ELSE CAST(NULL AS DOUBLE) END
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
The pinned convention `abs'(0) = 0` is unchanged, and every non-NULL input
|
|
46
|
+
gives the same answer as before ([#60](https://github.com/xqlsystems/ddx/pull/60)).
|
|
47
|
+
|
|
48
|
+
- **The same `CASE` came back as `DECIMAL` on DuckDB.** Its branches were bare
|
|
49
|
+
literals, which DuckDB types `DECIMAL(2,1)`, so `grad(abs(x), x)` was the one
|
|
50
|
+
derivative arriving in a different type — with different arithmetic under it —
|
|
51
|
+
from everything else ddx emits. The typed `ELSE` above fixes the whole
|
|
52
|
+
expression at `DOUBLE` ([#60](https://github.com/xqlsystems/ddx/pull/60)).
|
|
53
|
+
|
|
54
|
+
### Added
|
|
55
|
+
|
|
56
|
+
- `Ddx::unary_rule_names()` and `RuleRegistry::unary_names()` — the unary
|
|
57
|
+
function names the engine can differentiate, read from the rule registry and
|
|
58
|
+
including any added with `register`. Useful for deciding whether to hand ddx an
|
|
59
|
+
expression at all; note that a name being present does not by itself make an
|
|
60
|
+
*expression* differentiable, since the surrounding constructs matter too, so
|
|
61
|
+
catching the typed error remains the general answer. **Requires 0.2.1**
|
|
62
|
+
([#60](https://github.com/xqlsystems/ddx/pull/60)).
|
|
63
|
+
|
|
64
|
+
## [0.2.0](https://github.com/xqlsystems/ddx/compare/ddx-core-v0.1.3...ddx-core-v0.2.0) - 2026-08-09
|
|
65
|
+
|
|
66
|
+
### Added
|
|
67
|
+
|
|
68
|
+
- **`IdentCasing` gained two policies, for engines the existing two describe
|
|
69
|
+
incorrectly.** `IdentCasing` says how a dialect folds identifiers before ddx
|
|
70
|
+
compares them, and it offered only "fold unquoted to lowercase, quoted keeps
|
|
71
|
+
case" (Postgres, DataFusion) and "fold everything" (DuckDB). Two families of
|
|
72
|
+
engine fit neither:
|
|
73
|
+
|
|
74
|
+
- `IdentCasing::FoldUnquotedUpper` — unquoted identifiers fold to *uppercase*,
|
|
75
|
+
quoted keep case (Snowflake, Oracle). The same shape as `FoldUnquoted` with
|
|
76
|
+
the opposite target, which is not a cosmetic difference: bare `X` names the
|
|
77
|
+
column `"X"` here and `"x"` there.
|
|
78
|
+
- `IdentCasing::FoldNone` — nothing folds; `x` and `X` are different columns
|
|
79
|
+
(ClickHouse).
|
|
80
|
+
|
|
81
|
+
**Who is affected.** Anyone differentiating SQL aimed at Snowflake, Oracle or
|
|
82
|
+
ClickHouse while passing `FoldUnquoted`, which was the closest available
|
|
83
|
+
choice. The consequence was silent: the `wrt` column stopped matching its own
|
|
84
|
+
occurrences and the derivative came back `0`, or — worse, on the upper-folding
|
|
85
|
+
engines — matched the *other* column and returned a confident wrong nonzero
|
|
86
|
+
value. Nothing changes for existing `FoldUnquoted` or `FoldAll` callers; the
|
|
87
|
+
keys those two produce are unchanged ([#58](https://github.com/xqlsystems/ddx/pull/58)).
|
|
88
|
+
|
|
89
|
+
### Changed
|
|
90
|
+
|
|
91
|
+
- **BREAKING: `IdentCasing` is now `#[non_exhaustive]`.** Code outside this crate
|
|
92
|
+
that matches on it exhaustively must add a wildcard arm:
|
|
93
|
+
|
|
94
|
+
```rust
|
|
95
|
+
match casing {
|
|
96
|
+
IdentCasing::FoldUnquoted => ...,
|
|
97
|
+
IdentCasing::FoldAll => ...,
|
|
98
|
+
_ => ..., // now required
|
|
99
|
+
}
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Constructing and comparing variants is unaffected. More engines than these
|
|
103
|
+
four families exist, and each one found is another variant — marking the enum
|
|
104
|
+
non-exhaustive now makes those additive rather than a breaking release per
|
|
105
|
+
engine ([#58](https://github.com/xqlsystems/ddx/pull/58)).
|
|
106
|
+
|
|
107
|
+
- **BREAKING: `IdentCasing::FoldAll`'s discriminant moved from 1 to 2**, because
|
|
108
|
+
`FoldUnquotedUpper` was inserted before it to keep the variants grouped by
|
|
109
|
+
family. This affects only code casting the enum to an integer (`as isize`);
|
|
110
|
+
the enum has no `#[repr]`, so those values were never guaranteed
|
|
111
|
+
([#58](https://github.com/xqlsystems/ddx/pull/58)).
|
|
112
|
+
|
|
113
|
+
## [0.1.3](https://github.com/xqlsystems/ddx/compare/ddx-core-v0.1.2...ddx-core-v0.1.3) - 2026-08-08
|
|
114
|
+
|
|
115
|
+
### Other
|
|
116
|
+
|
|
117
|
+
- *(ddx-core)* skip points where a divisor has cancelled to rounding noise ([#56](https://github.com/xqlsystems/ddx/pull/56))
|
|
118
|
+
|
|
119
|
+
## [0.1.2](https://github.com/xqlsystems/ddx/compare/ddx-core-v0.1.1...ddx-core-v0.1.2) - 2026-08-08
|
|
120
|
+
|
|
121
|
+
### Fixed
|
|
122
|
+
|
|
123
|
+
- **Rendering a derivative could re-associate it, producing a wrong number in
|
|
124
|
+
valid SQL.** A derivative whose right operand bound as tightly as its parent
|
|
125
|
+
lost its parentheses: `a * (b / c)` was written as `a * b / c`, which reads
|
|
126
|
+
back as `(a * b) / c`. The two agree in exact arithmetic and diverge without
|
|
127
|
+
bound as `c` approaches zero, so a result could be off by any amount — the
|
|
128
|
+
continuous fuzz caught one wrong by twenty-four orders of magnitude
|
|
129
|
+
([#53](https://github.com/xqlsystems/ddx/pull/53)).
|
|
130
|
+
|
|
131
|
+
**Who is affected.** Anyone who takes the *text* of a derivative and reparses
|
|
132
|
+
it — `differentiate_sql`, or `rewrite_sql` output fed to an engine, which is
|
|
133
|
+
the normal path. Derivatives consumed as an `Expr` in memory were never
|
|
134
|
+
affected, because nothing reparsed them. The trigger needs a quotient inside a
|
|
135
|
+
product or a nested quotient, which the quotient and chain rules build
|
|
136
|
+
routinely, so upgrading is worthwhile even if you have not seen a bad value:
|
|
137
|
+
the error is silent and only large where a denominator is near zero.
|
|
138
|
+
|
|
139
|
+
Emitted SQL now carries a few more parentheses on same-precedence chains.
|
|
140
|
+
Nothing else about the derivatives changed.
|
|
141
|
+
|
|
142
|
+
## [0.1.1](https://github.com/xqlsystems/ddx/compare/ddx-core-v0.1.0...ddx-core-v0.1.1) - 2026-08-08
|
|
143
|
+
|
|
144
|
+
### Added
|
|
145
|
+
|
|
146
|
+
- `power` with a constant base or exponent now accepts one written as a *cast*
|
|
147
|
+
literal — `power(x, CAST(3 AS DOUBLE))` differentiates where it previously
|
|
148
|
+
returned `NotImplemented`. A cast to a numeric type is recognised as the
|
|
149
|
+
constant it wraps, which matters because query engines inject these: type
|
|
150
|
+
coercion rewrites `power(x, 3)` over a `DOUBLE` column into
|
|
151
|
+
`power(CAST(x AS DOUBLE), CAST(3 AS DOUBLE))` before ddx ever sees it. Casts
|
|
152
|
+
to non-numeric types are still not constants (`CAST(1 AS VARCHAR)` is the
|
|
153
|
+
string `'1'`).
|
|
154
|
+
- `ddx_core::test_utils`, behind the off-by-default `test-utils` feature: the
|
|
155
|
+
expression generator, reference interpreter, numeric conditioning gates and
|
|
156
|
+
failure reporter that `ddx-core`'s own property suite runs on. Exposed so that
|
|
157
|
+
crates building on `ddx-core` can fuzz against the *same* generator rather
|
|
158
|
+
than inventing their own. Test support, not API — **semver-exempt**, and
|
|
159
|
+
compiled only when you turn the feature on, so a default build of `ddx-core`
|
|
160
|
+
is byte-for-byte unaffected by it.
|
|
161
|
+
|
|
162
|
+
### Notes
|
|
163
|
+
|
|
164
|
+
- The engine itself is unchanged: no differentiation rule was added, removed or
|
|
165
|
+
altered, and every derivative `ddx-core` emitted at 0.1.0 it still emits.
|
|
166
|
+
- This release accompanies the first real `ddx-datafusion` adapter
|
|
167
|
+
([#49](https://github.com/xqlsystems/ddx/pull/49)), which is not yet published.
|
|
168
|
+
|
|
169
|
+
## [0.1.0] - 2026-07-26
|
|
170
|
+
|
|
171
|
+
### Added
|
|
172
|
+
|
|
173
|
+
- Initial release: the v1 scalar differentiation engine (design.md Milestone 0).
|
|
174
|
+
- `Ddx` — the engine object: `rewrite_sql` (the whole `grad`/`jvp` marker path,
|
|
175
|
+
byte-identical outside the marker), `explain` (preview a rewrite without
|
|
176
|
+
running it), `differentiate` / `jvp` / `differentiate_sql` (the lower-level
|
|
177
|
+
"calculus compiler" surface), and `register` for user-defined rules.
|
|
178
|
+
- Per-dialect identifier folding (`Ddx::for_datafusion()` / `for_duckdb()`) and
|
|
179
|
+
an extensible, name-keyed rule registry.
|
|
180
|
+
- Differentiation surface: `+ - * /`; the unary chain rule for the trig /
|
|
181
|
+
inverse-trig / exp / log / hyperbolic set plus `abs`; `power` with a constant
|
|
182
|
+
base or exponent; higher-order via nesting; through-aggregate via linearity.
|
|
183
|
+
Unsupported constructs are typed errors with actionable guidance — never a
|
|
184
|
+
silently-wrong number.
|
|
185
|
+
- Depends on `sqlparser` only (pinned `=0.62.0`, re-exported as
|
|
186
|
+
`ddx_core::sqlparser`).
|
|
187
|
+
|
|
188
|
+
[Unreleased]: https://github.com/xqlsystems/ddx/compare/ddx-core-v0.1.0...HEAD
|
|
189
|
+
[0.1.0]: https://github.com/xqlsystems/ddx/releases/tag/ddx-core-v0.1.0
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
[package]
|
|
2
|
+
name = "ddx-core"
|
|
3
|
+
description = "Engine-neutral symbolic differentiation of SQL scalar expressions: `grad` & `jvp`."
|
|
4
|
+
# Versioned independently of the workspace (the scaffold crates stay at the
|
|
5
|
+
# workspace 0.0.0). This is the first publishable crate; release-plz manages
|
|
6
|
+
# bumps from here (SemVer). See CONTRIBUTING.md → Releases.
|
|
7
|
+
version = "0.2.1"
|
|
8
|
+
edition.workspace = true
|
|
9
|
+
license.workspace = true
|
|
10
|
+
repository.workspace = true
|
|
11
|
+
authors.workspace = true
|
|
12
|
+
rust-version.workspace = true
|
|
13
|
+
readme = "README.md"
|
|
14
|
+
keywords = ["sql", "autograd", "differentiation", "datafusion", "duckdb"]
|
|
15
|
+
categories = ["mathematics", "database"]
|
|
16
|
+
|
|
17
|
+
[dependencies]
|
|
18
|
+
# ddx-core re-exports `sqlparser` (see lib.rs) so downstream adapters can't
|
|
19
|
+
# accidentally link a mismatched version. This is the *only* non-trivial
|
|
20
|
+
# dependency of the core, by design.
|
|
21
|
+
sqlparser = { workspace = true }
|
|
22
|
+
|
|
23
|
+
[features]
|
|
24
|
+
# The shared simulation harness (`ddx_core::test_utils`): generator, reference
|
|
25
|
+
# interpreter, conditioning gates, failure reporter. Off by default so it never
|
|
26
|
+
# ships in a normal build; enabled by every crate that fuzzes, so they all
|
|
27
|
+
# generate the *same* expressions. Semver-exempt — it is test support, not API.
|
|
28
|
+
test-utils = []
|
|
29
|
+
|
|
30
|
+
[dev-dependencies]
|
|
31
|
+
# ddx-core's own integration tests need the harness too, and an integration test
|
|
32
|
+
# does not inherit its crate's features. A self dev-dependency is the standard
|
|
33
|
+
# way to say "build me again, with this feature, for my own tests".
|
|
34
|
+
ddx-core = { path = ".", features = ["test-utils"] }
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# ddx-core
|
|
2
|
+
|
|
3
|
+
Engine-neutral symbolic differentiation of SQL scalar expressions — the v1
|
|
4
|
+
core of [`ddx`](https://github.com/xqlsystems/ddx), "autograd for composable
|
|
5
|
+
databases." Write calculus directly in SQL and let the engine evaluate the
|
|
6
|
+
derivative per row (the relational equivalent of `jax.vmap(jax.grad(f))`):
|
|
7
|
+
|
|
8
|
+
```sql
|
|
9
|
+
SELECT i, grad(x * y, x) AS dfdx, grad(x * y, y) AS dfdy FROM g
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
`grad`/`jvp` are **markers**, not row functions: they carry a differentiation
|
|
13
|
+
request through parsing and are always rewritten away *before* execution.
|
|
14
|
+
|
|
15
|
+
```rust
|
|
16
|
+
use ddx_core::Ddx;
|
|
17
|
+
use ddx_core::sqlparser::dialect::GenericDialect;
|
|
18
|
+
|
|
19
|
+
let ddx = Ddx::new();
|
|
20
|
+
let out = ddx.rewrite_sql("SELECT grad(sin(x), x) AS d FROM t", &GenericDialect {})?;
|
|
21
|
+
assert_eq!(out, "SELECT (cos(x)) AS d FROM t");
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
The engine differentiates [`sqlparser::ast::Expr`](https://docs.rs/sqlparser)
|
|
25
|
+
directly — the AST *is* the IR, there is no bespoke representation. The primary
|
|
26
|
+
dependency is `sqlparser`, **re-exported** as `ddx_core::sqlparser`
|
|
27
|
+
so downstream adapters cannot link a mismatched version.
|
|
28
|
+
|
|
29
|
+
## What it supports
|
|
30
|
+
|
|
31
|
+
`+ - * /`; the unary chain rule for the trig / inverse-trig / exp / log /
|
|
32
|
+
hyperbolic set plus `abs`; `power` with a constant base or exponent;
|
|
33
|
+
higher-order via nesting; through-aggregate via linearity
|
|
34
|
+
(`AVG(grad(loss, theta))`). Custom unary rules are registrable
|
|
35
|
+
(`ddx.register("myfn", rule)` — the rule supplies `f'(u)`, the engine applies
|
|
36
|
+
the chain rule). Anything else is a typed `DiffError`, never a silently-wrong
|
|
37
|
+
number.
|
|
38
|
+
|
|
39
|
+
Scalar `vjp` is deliberately **not** here: the name is reserved for the
|
|
40
|
+
query-level reverse-mode operation in `ddx-ad` (design.md §3.6, §4).
|
|
41
|
+
|
|
42
|
+
## Correctness properties worth knowing
|
|
43
|
+
|
|
44
|
+
- **Identifier folding is per-dialect** (`Ddx::for_datafusion()` vs
|
|
45
|
+
`Ddx::for_duckdb()`): unquoted identifiers always fold case; DuckDB folds
|
|
46
|
+
quoted ones too. `grad(Temp*Temp, temp)` matches.
|
|
47
|
+
- **Ambiguity is a hard error, not a guess**: a `wrt` that can't be pinned
|
|
48
|
+
syntactically (`grad(a.x*b.x, x)`) errors rather than differentiating the
|
|
49
|
+
wrong column.
|
|
50
|
+
- **`div` forces floating-point division** (`CAST(<numerator> AS DOUBLE)`), so
|
|
51
|
+
integer columns don't silently truncate the derivative.
|
|
52
|
+
- **0/1-folding follows JAX's `Zero`-tangent convention** and differs from
|
|
53
|
+
unfolded SQL only on NULL-bearing rows (documented, tested).
|
|
54
|
+
|
|
55
|
+
See [`../../docs/design.md`](../../docs/design.md) §3 for the full rationale and
|
|
56
|
+
the decision log (`F#`/`G#`) behind each of these.
|
|
57
|
+
|
|
58
|
+
## Status
|
|
59
|
+
|
|
60
|
+
M0 (the scalar core) — implemented. The per-engine adapters (`ddx-datafusion`,
|
|
61
|
+
`ddx-duckdb`), the Python wheel (`ddxdb`), and the query-level v2 engine
|
|
62
|
+
(`ddx-ad`) are later milestones; see the design doc's §8.
|
|
63
|
+
|
|
64
|
+
Licensed under Apache-2.0.
|