interlaced 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interlaced-1.0.1/LICENSE +21 -0
- interlaced-1.0.1/MANIFEST.in +2 -0
- interlaced-1.0.1/PKG-INFO +246 -0
- interlaced-1.0.1/README.md +198 -0
- interlaced-1.0.1/pyproject.toml +155 -0
- interlaced-1.0.1/setup.cfg +4 -0
- interlaced-1.0.1/src/interlace/__init__.py +26 -0
- interlaced-1.0.1/src/interlace/checks/__init__.py +10 -0
- interlaced-1.0.1/src/interlace/checks/builtin.py +152 -0
- interlaced-1.0.1/src/interlace/checks/runner.py +103 -0
- interlaced-1.0.1/src/interlace/checks/spec.py +113 -0
- interlaced-1.0.1/src/interlace/cli/__init__.py +7 -0
- interlaced-1.0.1/src/interlace/cli/main.py +1337 -0
- interlaced-1.0.1/src/interlace/config/__init__.py +7 -0
- interlaced-1.0.1/src/interlace/config/config.py +216 -0
- interlaced-1.0.1/src/interlace/contracts.py +35 -0
- interlaced-1.0.1/src/interlace/dsl/__init__.py +16 -0
- interlaced-1.0.1/src/interlace/dsl/decorators.py +226 -0
- interlaced-1.0.1/src/interlace/dsl/discovery.py +75 -0
- interlaced-1.0.1/src/interlace/dsl/sql_config.py +47 -0
- interlaced-1.0.1/src/interlace/engines/__init__.py +16 -0
- interlaced-1.0.1/src/interlace/engines/base.py +86 -0
- interlaced-1.0.1/src/interlace/engines/duckdb.py +306 -0
- interlaced-1.0.1/src/interlace/engines/postgres.py +181 -0
- interlaced-1.0.1/src/interlace/engines/quack.py +113 -0
- interlaced-1.0.1/src/interlace/engines/registry.py +125 -0
- interlaced-1.0.1/src/interlace/exceptions.py +66 -0
- interlaced-1.0.1/src/interlace/exports.py +162 -0
- interlaced-1.0.1/src/interlace/graph/__init__.py +17 -0
- interlaced-1.0.1/src/interlace/graph/column_lineage.py +69 -0
- interlaced-1.0.1/src/interlace/graph/dag.py +81 -0
- interlaced-1.0.1/src/interlace/graph/project.py +252 -0
- interlaced-1.0.1/src/interlace/graph/selectors.py +87 -0
- interlaced-1.0.1/src/interlace/ir/__init__.py +24 -0
- interlaced-1.0.1/src/interlace/ir/canonicalize.py +71 -0
- interlaced-1.0.1/src/interlace/ir/fingerprint.py +55 -0
- interlaced-1.0.1/src/interlace/ir/relation.py +71 -0
- interlaced-1.0.1/src/interlace/ir/schema.py +23 -0
- interlaced-1.0.1/src/interlace/plan/__init__.py +23 -0
- interlaced-1.0.1/src/interlace/plan/apply.py +625 -0
- interlaced-1.0.1/src/interlace/plan/differ.py +385 -0
- interlaced-1.0.1/src/interlace/plan/plan.py +192 -0
- interlaced-1.0.1/src/interlace/plan/resolve.py +74 -0
- interlaced-1.0.1/src/interlace/plan/run.py +113 -0
- interlaced-1.0.1/src/interlace/project.py +278 -0
- interlaced-1.0.1/src/interlace/py.typed +0 -0
- interlaced-1.0.1/src/interlace/runtime/__init__.py +6 -0
- interlaced-1.0.1/src/interlace/runtime/handles.py +48 -0
- interlaced-1.0.1/src/interlace/runtime/python_model.py +175 -0
- interlaced-1.0.1/src/interlace/scaffold.py +83 -0
- interlaced-1.0.1/src/interlace/scheduler/__init__.py +17 -0
- interlaced-1.0.1/src/interlace/scheduler/engine.py +66 -0
- interlaced-1.0.1/src/interlace/scheduler/triggers.py +84 -0
- interlaced-1.0.1/src/interlace/scheduler/worker.py +207 -0
- interlaced-1.0.1/src/interlace/service/__init__.py +7 -0
- interlaced-1.0.1/src/interlace/service/app.py +1508 -0
- interlaced-1.0.1/src/interlace/service/auth.py +42 -0
- interlaced-1.0.1/src/interlace/service/ui/app.css +356 -0
- interlaced-1.0.1/src/interlace/service/ui/favicon.svg +1 -0
- interlaced-1.0.1/src/interlace/service/ui/index.html +65 -0
- interlaced-1.0.1/src/interlace/service/ui/js/api.js +113 -0
- interlaced-1.0.1/src/interlace/service/ui/js/app.js +274 -0
- interlaced-1.0.1/src/interlace/service/ui/js/dag.js +433 -0
- interlaced-1.0.1/src/interlace/service/ui/js/ui.js +209 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/checks.js +0 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/environments.js +231 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/lineage.js +118 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/models.js +264 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/overview.js +93 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/plan.js +172 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/query.js +242 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/runs.js +264 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/streams.js +198 -0
- interlaced-1.0.1/src/interlace/service/ui/js/views/system.js +294 -0
- interlaced-1.0.1/src/interlace/state/__init__.py +18 -0
- interlaced-1.0.1/src/interlace/state/interval.py +148 -0
- interlaced-1.0.1/src/interlace/state/janitor.py +216 -0
- interlaced-1.0.1/src/interlace/state/snapshot.py +47 -0
- interlaced-1.0.1/src/interlace/state/store.py +1061 -0
- interlaced-1.0.1/src/interlace/strategies/__init__.py +64 -0
- interlaced-1.0.1/src/interlace/strategies/base.py +71 -0
- interlaced-1.0.1/src/interlace/strategies/full.py +38 -0
- interlaced-1.0.1/src/interlace/strategies/full_merge.py +98 -0
- interlaced-1.0.1/src/interlace/strategies/incremental_by_time.py +65 -0
- interlaced-1.0.1/src/interlace/strategies/merge_by_key.py +69 -0
- interlaced-1.0.1/src/interlace/strategies/scd_type_2.py +122 -0
- interlaced-1.0.1/src/interlace/strategies/view.py +33 -0
- interlaced-1.0.1/src/interlace/streaming/__init__.py +18 -0
- interlaced-1.0.1/src/interlace/streaming/log.py +323 -0
- interlaced-1.0.1/src/interlace/streaming/materializer.py +196 -0
- interlaced-1.0.1/src/interlace/streaming/schema.py +229 -0
- interlaced-1.0.1/src/interlaced.egg-info/PKG-INFO +246 -0
- interlaced-1.0.1/src/interlaced.egg-info/SOURCES.txt +95 -0
- interlaced-1.0.1/src/interlaced.egg-info/dependency_links.txt +1 -0
- interlaced-1.0.1/src/interlaced.egg-info/entry_points.txt +2 -0
- interlaced-1.0.1/src/interlaced.egg-info/requires.txt +38 -0
- interlaced-1.0.1/src/interlaced.egg-info/top_level.txt +1 -0
interlaced-1.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025-2026 Interlace Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: interlaced
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion
|
|
5
|
+
Author-email: Mark <mark@interlace.sh>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: data,pipeline,orchestration,transformation,etl,streaming,dbt,sqlmesh,duckdb
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Topic :: Software Development :: Build Tools
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Programming Language :: SQL
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Requires-Dist: sqlglot<31.0,>=25.0
|
|
17
|
+
Requires-Dist: duckdb>=1.5.3
|
|
18
|
+
Requires-Dist: pyarrow>=17.0
|
|
19
|
+
Requires-Dist: pydantic<3.0,>=2.5
|
|
20
|
+
Requires-Dist: typer<1.0,>=0.12
|
|
21
|
+
Requires-Dist: rich<15.0,>=13.0
|
|
22
|
+
Requires-Dist: cronsim<3.0,>=2.5
|
|
23
|
+
Requires-Dist: tenacity<10.0,>=8.2
|
|
24
|
+
Requires-Dist: pyyaml<7.0,>=6.0
|
|
25
|
+
Provides-Extra: service
|
|
26
|
+
Requires-Dist: litestar<3.0,>=2.12; extra == "service"
|
|
27
|
+
Requires-Dist: uvicorn<1.0,>=0.30; extra == "service"
|
|
28
|
+
Requires-Dist: msgspec<1.0,>=0.18; extra == "service"
|
|
29
|
+
Provides-Extra: adbc
|
|
30
|
+
Requires-Dist: adbc-driver-manager<2.0,>=1.2; extra == "adbc"
|
|
31
|
+
Requires-Dist: adbc-driver-postgresql<2.0,>=1.2; extra == "adbc"
|
|
32
|
+
Provides-Extra: postgres
|
|
33
|
+
Requires-Dist: psycopg[binary]<4.0,>=3.1; extra == "postgres"
|
|
34
|
+
Provides-Extra: polars
|
|
35
|
+
Requires-Dist: polars<2.0,>=1.0; extra == "polars"
|
|
36
|
+
Provides-Extra: pandas
|
|
37
|
+
Requires-Dist: pandas<4.0,>=2.0; extra == "pandas"
|
|
38
|
+
Provides-Extra: all
|
|
39
|
+
Requires-Dist: interlaced[adbc,polars,postgres,service]; extra == "all"
|
|
40
|
+
Provides-Extra: dev
|
|
41
|
+
Requires-Dist: pytest<10.0,>=8.0; extra == "dev"
|
|
42
|
+
Requires-Dist: httpx<1.0,>=0.27; extra == "dev"
|
|
43
|
+
Requires-Dist: pytest-asyncio<2.0,>=1.0; extra == "dev"
|
|
44
|
+
Requires-Dist: ruff<1.0,>=0.6; extra == "dev"
|
|
45
|
+
Requires-Dist: black<27.0,>=24.0; extra == "dev"
|
|
46
|
+
Requires-Dist: mypy<2.0,>=1.11; extra == "dev"
|
|
47
|
+
Dynamic: license-file
|
|
48
|
+
|
|
49
|
+
# interlace
|
|
50
|
+
|
|
51
|
+
**Python/SQL-first data platform: transformation, orchestration, and durable streaming — one process.**
|
|
52
|
+
|
|
53
|
+
interlace is an independent, MIT-licensed alternative to dbt/SQLMesh that also replaces the
|
|
54
|
+
orchestrator (no Airflow) and the ingestion layer (Cloudflare-Pipelines-style durable streams).
|
|
55
|
+
Models are `.sql` files or Python functions; state is versioned snapshots with virtual
|
|
56
|
+
environments and a terraform-style plan/apply; everything runs in a single daemon on
|
|
57
|
+
DuckDB + DuckLake by default.
|
|
58
|
+
|
|
59
|
+
> **Status: 1.0.** Requires Python 3.12+.
|
|
60
|
+
> The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
uv pip install "interlaced[service]" # extras: service, adbc, postgres, polars, pandas, all
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Sixty seconds
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
interlace init my-project && cd my-project
|
|
70
|
+
interlace plan # terraform-style preview: added / breaking / non-breaking / reuse
|
|
71
|
+
interlace apply # build changed models, run checks, promote the environment
|
|
72
|
+
interlace serve # the daemon: web UI (/ui) + HTTP API + scheduler + streams, one process
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Every model builds into a fingerprinted physical table (`interlace__main.orders__a1b2c3`);
|
|
76
|
+
environments are views over those tables, so promotion and rollback are atomic view swaps and a
|
|
77
|
+
dev environment reuses prod's tables for free. **Production is the unprefixed namespace** —
|
|
78
|
+
consumers query `main.orders`; sandboxes are prefixed (`dev__main.orders`). Commands default to
|
|
79
|
+
prod; pass `--env dev` while developing.
|
|
80
|
+
|
|
81
|
+
## Models
|
|
82
|
+
|
|
83
|
+
**SQL** — a file per model; upstreams referenced by model name, dependencies inferred by parsing
|
|
84
|
+
(sqlglot), config in a leading comment block:
|
|
85
|
+
|
|
86
|
+
```sql
|
|
87
|
+
/* interlace:
|
|
88
|
+
strategy: scd_type_2
|
|
89
|
+
key: customer_id
|
|
90
|
+
schedule: {cron: "0 * * * *"}
|
|
91
|
+
checks:
|
|
92
|
+
- not_null: customer_id
|
|
93
|
+
- unique: customer_id
|
|
94
|
+
*/
|
|
95
|
+
SELECT customer_id, name, tier FROM raw_customers
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
**Python** — functions whose parameters name their upstreams; data crosses as Arrow
|
|
99
|
+
(never pandas), streamed with bounded memory:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
from interlace import model
|
|
103
|
+
|
|
104
|
+
@model(strategy="merge_by_key", key="order_id", cursor="updated_at")
|
|
105
|
+
def orders(cursor, this):
|
|
106
|
+
"""Incremental API extract: `cursor` is max(updated_at) already in the
|
|
107
|
+
warehouse (None on first run); `this` is the previous materialisation."""
|
|
108
|
+
rows = fetch_orders(since=cursor) # your code
|
|
109
|
+
return pyarrow.Table.from_pylist(rows) # or RecordBatchReader / generator of batches
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
**Strategies:** `full`, `view`, `ephemeral` (CTE-inlined), `merge_by_key` (upsert),
|
|
113
|
+
`full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
|
|
114
|
+
(windowed, interval-ledger backfill/catchup), `scd_type_2` (history with validity windows).
|
|
115
|
+
|
|
116
|
+
## Plan / apply
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
$ interlace plan
|
|
120
|
+
Model Change Category Build
|
|
121
|
+
orders modified non_breaking rebuild
|
|
122
|
+
order_stats modified non_breaking reuse <- output provably identical: not rebuilt
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
- Changes classify **breaking / non-breaking / forward-only**; a plan with breaking changes
|
|
126
|
+
refuses to apply without `--force`. Downstream models whose output is provably identical
|
|
127
|
+
(column-pruned impact analysis) **reuse their existing tables** instead of rebuilding — an
|
|
128
|
+
improvement over model-granular invalidation.
|
|
129
|
+
- `apply --forward-only` lets history-keeping models (scd2/merge/incremental) survive a
|
|
130
|
+
definition change: the existing table is copied to the new version, the new logic applies to
|
|
131
|
+
the copy going forward, and checks gate before views move.
|
|
132
|
+
- **Checks gate promotion**: 10 built-in types (not_null, unique, accepted_values, row_count,
|
|
133
|
+
freshness, expression, relationships, pattern, range, sql) plus `@check` Python functions —
|
|
134
|
+
an error-severity failure blocks before the environment view moves. `interlace checks run`
|
|
135
|
+
re-runs them ad hoc against any environment's promoted tables.
|
|
136
|
+
- `interlace gc` removes snapshots no environment references (reference-aware: tables shared
|
|
137
|
+
through reuse survive).
|
|
138
|
+
|
|
139
|
+
## Streaming
|
|
140
|
+
|
|
141
|
+
Declare a stream; POST to it; rows are durable (SQLite WAL log) before the 200, deduplicated by
|
|
142
|
+
idempotency key, and materialized exactly-once into `streams.<name>` — a micro-batch flusher
|
|
143
|
+
commits the data and the watermark in one warehouse transaction, and SQL models just read the
|
|
144
|
+
table. A flush triggers the models that consume the stream.
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
from interlace import stream
|
|
148
|
+
|
|
149
|
+
@stream("orders", schema={"order_id": "string", "total": "double"},
|
|
150
|
+
idempotency_key="order_id", retention="7d", on_schema_drift="evolve")
|
|
151
|
+
def orders(event): ...
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
curl -X POST localhost:8000/streams/orders -d '{"order_id": "o1", "total": 49.5}'
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Schema drift is yours to choose: `reject` (400), `evolve` (new columns appear), or
|
|
159
|
+
`quarantine` (bad events divert to `<stream>__quarantine`). When the warehouse falls behind,
|
|
160
|
+
publishes get **429 backpressure** instead of unbounded backlog.
|
|
161
|
+
|
|
162
|
+
## Reverse ETL
|
|
163
|
+
|
|
164
|
+
Attach external databases and deliver model results into them — the live table is never
|
|
165
|
+
dropped, keyed modes reuse the same merge strategies:
|
|
166
|
+
|
|
167
|
+
```yaml
|
|
168
|
+
# interlace.yaml
|
|
169
|
+
attach:
|
|
170
|
+
crm: "postgres:host=... dbname=crm"
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
```sql
|
|
174
|
+
/* interlace: {export: {to: table, target: crm.public.accounts, mode: merge_by_key, key: id}} */
|
|
175
|
+
SELECT id, tier, lifetime_value FROM account_summary
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
File exports (`to: parquet|csv|json`) work the same way. Sinks are **environment-gated**: by
|
|
179
|
+
default the side effect fires only from prod — a dev apply never writes to a live external
|
|
180
|
+
table (opt in with `environments: [dev, prod]`).
|
|
181
|
+
|
|
182
|
+
## Multi-engine
|
|
183
|
+
|
|
184
|
+
Models run on **named engines**: DuckDB/DuckLake by default, Postgres natively over ADBC
|
|
185
|
+
(`pip install 'interlaced[adbc]'`), with per-model pinning:
|
|
186
|
+
|
|
187
|
+
```yaml
|
|
188
|
+
engines:
|
|
189
|
+
pg: {type: postgres, database: "${PG_DSN}"}
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
```sql
|
|
193
|
+
/* interlace: {engine: pg, strategy: merge_by_key, key: id} */
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Strategies execute *inside* the pinned engine (no DuckDB middleman); cross-engine dependencies
|
|
197
|
+
appear as explicit **transfer** lines in the plan and move as Arrow (or a federated `ATTACH`
|
|
198
|
+
fast lane when possible). Contract: `docs/architecture/MULTI_ENGINE.md`.
|
|
199
|
+
|
|
200
|
+
## The daemon
|
|
201
|
+
|
|
202
|
+
`interlace serve` runs everything in one process:
|
|
203
|
+
|
|
204
|
+
- the **web UI** at `/ui` (in-package, zero build step) — ten views: overview, lineage canvas
|
|
205
|
+
with column-level tracing, models, plan/apply with SQL diffs, live runs, query console,
|
|
206
|
+
streams, checks, environments, and system — live over SSE;
|
|
207
|
+
- the **HTTP API** (Litestar + msgspec, OpenAPI at `/schema/scalar`) with the same surface as
|
|
208
|
+
the CLI: plan/apply, runs, checks, streams, engines, schedules, lineage, query, gc;
|
|
209
|
+
- the **scheduler**: cron/interval triggers enqueue onto a **durable run queue** (leases,
|
|
210
|
+
retries, cooperative cancellation — `interlace cancel <id>` or `POST /runs/{id}/cancel`);
|
|
211
|
+
- **stream ingestion** and retention.
|
|
212
|
+
|
|
213
|
+
Scoped API keys (`interlace apikey create ci --scope read`) lock it down; a durable event log
|
|
214
|
+
backs `GET /events/stream` (SSE with `Last-Event-ID` replay).
|
|
215
|
+
|
|
216
|
+
Add `--quack quack:localhost:4213` to serve the warehouse itself over DuckDB's quack protocol —
|
|
217
|
+
other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by setting
|
|
218
|
+
`database: quack:localhost:4213`.
|
|
219
|
+
|
|
220
|
+
## Architecture in five lines
|
|
221
|
+
|
|
222
|
+
- The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
|
|
223
|
+
are AST builders and dialect appears only at `transpile()`.
|
|
224
|
+
- Storage defaults to **DuckLake** (Parquet + SQL catalog) opened as DuckDB's primary database.
|
|
225
|
+
- Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
|
|
226
|
+
scale-out swap.
|
|
227
|
+
- Streams live in their own durable log; the materializer commits data + watermark in one
|
|
228
|
+
warehouse transaction — exactly-once without distributed coordination.
|
|
229
|
+
- No Jinja, no pandas in core, no external orchestrator.
|
|
230
|
+
|
|
231
|
+
The full design rationale lives in `docs/architecture/v2-design.md`.
|
|
232
|
+
|
|
233
|
+
## Development
|
|
234
|
+
|
|
235
|
+
Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
|
|
236
|
+
[moon](https://moonrepo.dev/moon), `uv` owns the virtualenv:
|
|
237
|
+
|
|
238
|
+
```bash
|
|
239
|
+
proto install
|
|
240
|
+
moon run interlace:sync # install deps
|
|
241
|
+
moon run interlace:test # 350+ tests
|
|
242
|
+
moon run interlace:check # black + ruff (CI equivalent)
|
|
243
|
+
moon run interlace:typecheck # mypy
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
MIT licensed.
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
# interlace
|
|
2
|
+
|
|
3
|
+
**Python/SQL-first data platform: transformation, orchestration, and durable streaming — one process.**
|
|
4
|
+
|
|
5
|
+
interlace is an independent, MIT-licensed alternative to dbt/SQLMesh that also replaces the
|
|
6
|
+
orchestrator (no Airflow) and the ingestion layer (Cloudflare-Pipelines-style durable streams).
|
|
7
|
+
Models are `.sql` files or Python functions; state is versioned snapshots with virtual
|
|
8
|
+
environments and a terraform-style plan/apply; everything runs in a single daemon on
|
|
9
|
+
DuckDB + DuckLake by default.
|
|
10
|
+
|
|
11
|
+
> **Status: 1.0.** Requires Python 3.12+.
|
|
12
|
+
> The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
uv pip install "interlaced[service]" # extras: service, adbc, postgres, polars, pandas, all
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
## Sixty seconds
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
interlace init my-project && cd my-project
|
|
22
|
+
interlace plan # terraform-style preview: added / breaking / non-breaking / reuse
|
|
23
|
+
interlace apply # build changed models, run checks, promote the environment
|
|
24
|
+
interlace serve # the daemon: web UI (/ui) + HTTP API + scheduler + streams, one process
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Every model builds into a fingerprinted physical table (`interlace__main.orders__a1b2c3`);
|
|
28
|
+
environments are views over those tables, so promotion and rollback are atomic view swaps and a
|
|
29
|
+
dev environment reuses prod's tables for free. **Production is the unprefixed namespace** —
|
|
30
|
+
consumers query `main.orders`; sandboxes are prefixed (`dev__main.orders`). Commands default to
|
|
31
|
+
prod; pass `--env dev` while developing.
|
|
32
|
+
|
|
33
|
+
## Models
|
|
34
|
+
|
|
35
|
+
**SQL** — a file per model; upstreams referenced by model name, dependencies inferred by parsing
|
|
36
|
+
(sqlglot), config in a leading comment block:
|
|
37
|
+
|
|
38
|
+
```sql
|
|
39
|
+
/* interlace:
|
|
40
|
+
strategy: scd_type_2
|
|
41
|
+
key: customer_id
|
|
42
|
+
schedule: {cron: "0 * * * *"}
|
|
43
|
+
checks:
|
|
44
|
+
- not_null: customer_id
|
|
45
|
+
- unique: customer_id
|
|
46
|
+
*/
|
|
47
|
+
SELECT customer_id, name, tier FROM raw_customers
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
**Python** — functions whose parameters name their upstreams; data crosses as Arrow
|
|
51
|
+
(never pandas), streamed with bounded memory:
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from interlace import model
|
|
55
|
+
|
|
56
|
+
@model(strategy="merge_by_key", key="order_id", cursor="updated_at")
|
|
57
|
+
def orders(cursor, this):
|
|
58
|
+
"""Incremental API extract: `cursor` is max(updated_at) already in the
|
|
59
|
+
warehouse (None on first run); `this` is the previous materialisation."""
|
|
60
|
+
rows = fetch_orders(since=cursor) # your code
|
|
61
|
+
return pyarrow.Table.from_pylist(rows) # or RecordBatchReader / generator of batches
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
**Strategies:** `full`, `view`, `ephemeral` (CTE-inlined), `merge_by_key` (upsert),
|
|
65
|
+
`full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
|
|
66
|
+
(windowed, interval-ledger backfill/catchup), `scd_type_2` (history with validity windows).
|
|
67
|
+
|
|
68
|
+
## Plan / apply
|
|
69
|
+
|
|
70
|
+
```
|
|
71
|
+
$ interlace plan
|
|
72
|
+
Model Change Category Build
|
|
73
|
+
orders modified non_breaking rebuild
|
|
74
|
+
order_stats modified non_breaking reuse <- output provably identical: not rebuilt
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
- Changes classify **breaking / non-breaking / forward-only**; a plan with breaking changes
|
|
78
|
+
refuses to apply without `--force`. Downstream models whose output is provably identical
|
|
79
|
+
(column-pruned impact analysis) **reuse their existing tables** instead of rebuilding — an
|
|
80
|
+
improvement over model-granular invalidation.
|
|
81
|
+
- `apply --forward-only` lets history-keeping models (scd2/merge/incremental) survive a
|
|
82
|
+
definition change: the existing table is copied to the new version, the new logic applies to
|
|
83
|
+
the copy going forward, and checks gate before views move.
|
|
84
|
+
- **Checks gate promotion**: 10 built-in types (not_null, unique, accepted_values, row_count,
|
|
85
|
+
freshness, expression, relationships, pattern, range, sql) plus `@check` Python functions —
|
|
86
|
+
an error-severity failure blocks before the environment view moves. `interlace checks run`
|
|
87
|
+
re-runs them ad hoc against any environment's promoted tables.
|
|
88
|
+
- `interlace gc` removes snapshots no environment references (reference-aware: tables shared
|
|
89
|
+
through reuse survive).
|
|
90
|
+
|
|
91
|
+
## Streaming
|
|
92
|
+
|
|
93
|
+
Declare a stream; POST to it; rows are durable (SQLite WAL log) before the 200, deduplicated by
|
|
94
|
+
idempotency key, and materialized exactly-once into `streams.<name>` — a micro-batch flusher
|
|
95
|
+
commits the data and the watermark in one warehouse transaction, and SQL models just read the
|
|
96
|
+
table. A flush triggers the models that consume the stream.
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from interlace import stream
|
|
100
|
+
|
|
101
|
+
@stream("orders", schema={"order_id": "string", "total": "double"},
|
|
102
|
+
idempotency_key="order_id", retention="7d", on_schema_drift="evolve")
|
|
103
|
+
def orders(event): ...
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
curl -X POST localhost:8000/streams/orders -d '{"order_id": "o1", "total": 49.5}'
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Schema drift is yours to choose: `reject` (400), `evolve` (new columns appear), or
|
|
111
|
+
`quarantine` (bad events divert to `<stream>__quarantine`). When the warehouse falls behind,
|
|
112
|
+
publishes get **429 backpressure** instead of unbounded backlog.
|
|
113
|
+
|
|
114
|
+
## Reverse ETL
|
|
115
|
+
|
|
116
|
+
Attach external databases and deliver model results into them — the live table is never
|
|
117
|
+
dropped, keyed modes reuse the same merge strategies:
|
|
118
|
+
|
|
119
|
+
```yaml
|
|
120
|
+
# interlace.yaml
|
|
121
|
+
attach:
|
|
122
|
+
crm: "postgres:host=... dbname=crm"
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
```sql
|
|
126
|
+
/* interlace: {export: {to: table, target: crm.public.accounts, mode: merge_by_key, key: id}} */
|
|
127
|
+
SELECT id, tier, lifetime_value FROM account_summary
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
File exports (`to: parquet|csv|json`) work the same way. Sinks are **environment-gated**: by
|
|
131
|
+
default the side effect fires only from prod — a dev apply never writes to a live external
|
|
132
|
+
table (opt in with `environments: [dev, prod]`).
|
|
133
|
+
|
|
134
|
+
## Multi-engine
|
|
135
|
+
|
|
136
|
+
Models run on **named engines**: DuckDB/DuckLake by default, Postgres natively over ADBC
|
|
137
|
+
(`pip install 'interlaced[adbc]'`), with per-model pinning:
|
|
138
|
+
|
|
139
|
+
```yaml
|
|
140
|
+
engines:
|
|
141
|
+
pg: {type: postgres, database: "${PG_DSN}"}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
```sql
|
|
145
|
+
/* interlace: {engine: pg, strategy: merge_by_key, key: id} */
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Strategies execute *inside* the pinned engine (no DuckDB middleman); cross-engine dependencies
|
|
149
|
+
appear as explicit **transfer** lines in the plan and move as Arrow (or a federated `ATTACH`
|
|
150
|
+
fast lane when possible). Contract: `docs/architecture/MULTI_ENGINE.md`.
|
|
151
|
+
|
|
152
|
+
## The daemon
|
|
153
|
+
|
|
154
|
+
`interlace serve` runs everything in one process:
|
|
155
|
+
|
|
156
|
+
- the **web UI** at `/ui` (in-package, zero build step) — ten views: overview, lineage canvas
|
|
157
|
+
with column-level tracing, models, plan/apply with SQL diffs, live runs, query console,
|
|
158
|
+
streams, checks, environments, and system — live over SSE;
|
|
159
|
+
- the **HTTP API** (Litestar + msgspec, OpenAPI at `/schema/scalar`) with the same surface as
|
|
160
|
+
the CLI: plan/apply, runs, checks, streams, engines, schedules, lineage, query, gc;
|
|
161
|
+
- the **scheduler**: cron/interval triggers enqueue onto a **durable run queue** (leases,
|
|
162
|
+
retries, cooperative cancellation — `interlace cancel <id>` or `POST /runs/{id}/cancel`);
|
|
163
|
+
- **stream ingestion** and retention.
|
|
164
|
+
|
|
165
|
+
Scoped API keys (`interlace apikey create ci --scope read`) lock it down; a durable event log
|
|
166
|
+
backs `GET /events/stream` (SSE with `Last-Event-ID` replay).
|
|
167
|
+
|
|
168
|
+
Add `--quack quack:localhost:4213` to serve the warehouse itself over DuckDB's quack protocol —
|
|
169
|
+
other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by setting
|
|
170
|
+
`database: quack:localhost:4213`.
|
|
171
|
+
|
|
172
|
+
## Architecture in five lines
|
|
173
|
+
|
|
174
|
+
- The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
|
|
175
|
+
are AST builders and dialect appears only at `transpile()`.
|
|
176
|
+
- Storage defaults to **DuckLake** (Parquet + SQL catalog) opened as DuckDB's primary database.
|
|
177
|
+
- Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
|
|
178
|
+
scale-out swap.
|
|
179
|
+
- Streams live in their own durable log; the materializer commits data + watermark in one
|
|
180
|
+
warehouse transaction — exactly-once without distributed coordination.
|
|
181
|
+
- No Jinja, no pandas in core, no external orchestrator.
|
|
182
|
+
|
|
183
|
+
The full design rationale lives in `docs/architecture/v2-design.md`.
|
|
184
|
+
|
|
185
|
+
## Development
|
|
186
|
+
|
|
187
|
+
Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
|
|
188
|
+
[moon](https://moonrepo.dev/moon), `uv` owns the virtualenv:
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
proto install
|
|
192
|
+
moon run interlace:sync # install deps
|
|
193
|
+
moon run interlace:test # 350+ tests
|
|
194
|
+
moon run interlace:check # black + ruff (CI equivalent)
|
|
195
|
+
moon run interlace:typecheck # mypy
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
MIT licensed.
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "interlaced"
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{name = "Mark", email = "mark@interlace.sh"}
|
|
14
|
+
]
|
|
15
|
+
keywords = ["data", "pipeline", "orchestration", "transformation", "etl", "streaming", "dbt", "sqlmesh", "duckdb"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Topic :: Software Development :: Build Tools",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: SQL",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
# Core: the IR + engine + state + plan spine (Phase 1). Everything is Arrow- and
|
|
25
|
+
# sqlglot-native. See docs/architecture/v2-design.md.
|
|
26
|
+
dependencies = [
|
|
27
|
+
"sqlglot>=25.0,<31.0", # canonical IR, transpilation, semantic diff, column lineage
|
|
28
|
+
"duckdb>=1.5.3", # default engine + federation hub; DuckLake storage + quack serving
|
|
29
|
+
"pyarrow>=17.0", # the wire format: RecordBatchReader everywhere
|
|
30
|
+
"pydantic>=2.5,<3.0", # config + manifest validation (cold paths only)
|
|
31
|
+
"typer>=0.12,<1.0", # CLI
|
|
32
|
+
"rich>=13.0,<15.0", # display, strictly an event subscriber
|
|
33
|
+
"cronsim>=2.5,<3.0", # cron parsing for the trigger engine
|
|
34
|
+
"tenacity>=8.2,<10.0", # retries (tasks, DuckLake commit conflicts, transfers)
|
|
35
|
+
"pyyaml>=6.0,<7.0", # project config (config + env overlays)
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.scripts]
|
|
39
|
+
interlace = "interlace.cli.main:main"
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
# Service + orchestration daemon (Phase 2).
|
|
43
|
+
service = [
|
|
44
|
+
"litestar>=2.12,<3.0",
|
|
45
|
+
"uvicorn>=0.30,<1.0",
|
|
46
|
+
"msgspec>=0.18,<1.0", # the wire types (litestar serialises msgspec structs natively)
|
|
47
|
+
]
|
|
48
|
+
# Remote engines via Arrow-native transfer (Phase 4).
|
|
49
|
+
adbc = [
|
|
50
|
+
"adbc-driver-manager>=1.2,<2.0",
|
|
51
|
+
"adbc-driver-postgresql>=1.2,<2.0",
|
|
52
|
+
]
|
|
53
|
+
postgres = ["psycopg[binary]>=3.1,<4.0"]
|
|
54
|
+
polars = ["polars>=1.0,<2.0"]
|
|
55
|
+
pandas = ["pandas>=2.0,<4.0"]
|
|
56
|
+
all = ["interlaced[service,adbc,postgres,polars]"]
|
|
57
|
+
dev = [
|
|
58
|
+
"pytest>=8.0,<10.0",
|
|
59
|
+
"httpx>=0.27,<1.0", # litestar's TestClient transport
|
|
60
|
+
"pytest-asyncio>=1.0,<2.0",
|
|
61
|
+
"ruff>=0.6,<1.0",
|
|
62
|
+
"black>=24.0,<27.0",
|
|
63
|
+
"mypy>=1.11,<2.0",
|
|
64
|
+
]
|
|
65
|
+
|
|
66
|
+
[tool.setuptools.packages.find]
|
|
67
|
+
where = ["src"]
|
|
68
|
+
|
|
69
|
+
[tool.setuptools.package-data]
|
|
70
|
+
interlace = ["py.typed", "service/ui/*", "service/ui/**/*"]
|
|
71
|
+
|
|
72
|
+
[tool.black]
|
|
73
|
+
line-length = 120
|
|
74
|
+
target-version = ['py312']
|
|
75
|
+
|
|
76
|
+
[tool.ruff]
|
|
77
|
+
line-length = 120
|
|
78
|
+
target-version = "py312"
|
|
79
|
+
|
|
80
|
+
[tool.ruff.lint]
|
|
81
|
+
select = [
|
|
82
|
+
"E", # pycodestyle errors
|
|
83
|
+
"W", # pycodestyle warnings
|
|
84
|
+
"F", # pyflakes
|
|
85
|
+
"I", # isort
|
|
86
|
+
"B", # flake8-bugbear
|
|
87
|
+
"C4", # flake8-comprehensions
|
|
88
|
+
"UP", # pyupgrade
|
|
89
|
+
]
|
|
90
|
+
ignore = [
|
|
91
|
+
"E501", # line too long (handled by formatter)
|
|
92
|
+
"B008", # function calls in argument defaults
|
|
93
|
+
"C901", # too complex
|
|
94
|
+
]
|
|
95
|
+
|
|
96
|
+
[tool.ruff.lint.per-file-ignores]
|
|
97
|
+
"__init__.py" = ["F401"] # Allow unused imports in __init__.py
|
|
98
|
+
|
|
99
|
+
[tool.mypy]
|
|
100
|
+
python_version = "3.12"
|
|
101
|
+
warn_return_any = true
|
|
102
|
+
warn_unused_configs = true
|
|
103
|
+
disallow_untyped_defs = true
|
|
104
|
+
disallow_incomplete_defs = true
|
|
105
|
+
check_untyped_defs = true
|
|
106
|
+
no_implicit_optional = true
|
|
107
|
+
warn_redundant_casts = true
|
|
108
|
+
warn_unused_ignores = true
|
|
109
|
+
|
|
110
|
+
[[tool.mypy.overrides]]
|
|
111
|
+
module = [
|
|
112
|
+
"sqlglot.*",
|
|
113
|
+
"pyarrow.*",
|
|
114
|
+
"yaml",
|
|
115
|
+
]
|
|
116
|
+
ignore_missing_imports = true
|
|
117
|
+
|
|
118
|
+
[tool.coverage.run]
|
|
119
|
+
source = ["src/interlace"]
|
|
120
|
+
omit = [
|
|
121
|
+
"*/tests/*",
|
|
122
|
+
"*/test_*.py",
|
|
123
|
+
"*/__pycache__/*",
|
|
124
|
+
"*/__init__.py",
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
[tool.coverage.report]
|
|
128
|
+
exclude_lines = [
|
|
129
|
+
"pragma: no cover",
|
|
130
|
+
"def __repr__",
|
|
131
|
+
"raise AssertionError",
|
|
132
|
+
"raise NotImplementedError",
|
|
133
|
+
"if __name__ == .__main__.:",
|
|
134
|
+
"if TYPE_CHECKING:",
|
|
135
|
+
"@abstractmethod",
|
|
136
|
+
]
|
|
137
|
+
|
|
138
|
+
[tool.pytest.ini_options]
|
|
139
|
+
testpaths = ["tests"]
|
|
140
|
+
python_files = ["test_*.py", "*_test.py"]
|
|
141
|
+
python_classes = ["Test*"]
|
|
142
|
+
python_functions = ["test_*"]
|
|
143
|
+
addopts = [
|
|
144
|
+
"-v",
|
|
145
|
+
"--strict-markers",
|
|
146
|
+
"--tb=short",
|
|
147
|
+
]
|
|
148
|
+
markers = [
|
|
149
|
+
"unit: Unit tests (fast, isolated)",
|
|
150
|
+
"integration: Integration tests (slower, require database)",
|
|
151
|
+
"slow: Slow tests (may take a long time)",
|
|
152
|
+
"requires_db: Needs a reachable external database (e.g. Postgres)",
|
|
153
|
+
]
|
|
154
|
+
asyncio_mode = "auto"
|
|
155
|
+
asyncio_default_fixture_loop_scope = "function"
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Interlace v2 — Python/SQL-first data platform.
|
|
2
|
+
|
|
3
|
+
Transformation (sqlmesh-grade snapshots, virtual environments, plan/apply),
|
|
4
|
+
built-in orchestration (durable work queue + unified triggers), and durable
|
|
5
|
+
streaming ingestion — in one process. See docs/architecture/v2-design.md.
|
|
6
|
+
|
|
7
|
+
This package is under active greenfield construction; the public surface is the
|
|
8
|
+
``@model`` / ``@stream`` / ``@check`` decorators plus the core IR types.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from interlace.dsl.decorators import check, model, stream
|
|
14
|
+
from interlace.ir.relation import EngineRef, SqlRelation, TableRef
|
|
15
|
+
|
|
16
|
+
__version__ = "1.0.0"
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"EngineRef",
|
|
20
|
+
"SqlRelation",
|
|
21
|
+
"TableRef",
|
|
22
|
+
"__version__",
|
|
23
|
+
"check",
|
|
24
|
+
"model",
|
|
25
|
+
"stream",
|
|
26
|
+
]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Data-quality checks. Results gate promotion: an ``error``-severity failure
|
|
2
|
+
aborts the apply before the environment is promoted.
|
|
3
|
+
|
|
4
|
+
Import :mod:`interlace.checks.runner` for execution — kept out of this package
|
|
5
|
+
init so declaring checks (spec) never drags in the runtime.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from interlace.checks.spec import CheckSpec, parse_checks
|
|
9
|
+
|
|
10
|
+
__all__ = ["CheckSpec", "parse_checks"]
|