interlaced 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. interlaced-1.0.1/LICENSE +21 -0
  2. interlaced-1.0.1/MANIFEST.in +2 -0
  3. interlaced-1.0.1/PKG-INFO +246 -0
  4. interlaced-1.0.1/README.md +198 -0
  5. interlaced-1.0.1/pyproject.toml +155 -0
  6. interlaced-1.0.1/setup.cfg +4 -0
  7. interlaced-1.0.1/src/interlace/__init__.py +26 -0
  8. interlaced-1.0.1/src/interlace/checks/__init__.py +10 -0
  9. interlaced-1.0.1/src/interlace/checks/builtin.py +152 -0
  10. interlaced-1.0.1/src/interlace/checks/runner.py +103 -0
  11. interlaced-1.0.1/src/interlace/checks/spec.py +113 -0
  12. interlaced-1.0.1/src/interlace/cli/__init__.py +7 -0
  13. interlaced-1.0.1/src/interlace/cli/main.py +1337 -0
  14. interlaced-1.0.1/src/interlace/config/__init__.py +7 -0
  15. interlaced-1.0.1/src/interlace/config/config.py +216 -0
  16. interlaced-1.0.1/src/interlace/contracts.py +35 -0
  17. interlaced-1.0.1/src/interlace/dsl/__init__.py +16 -0
  18. interlaced-1.0.1/src/interlace/dsl/decorators.py +226 -0
  19. interlaced-1.0.1/src/interlace/dsl/discovery.py +75 -0
  20. interlaced-1.0.1/src/interlace/dsl/sql_config.py +47 -0
  21. interlaced-1.0.1/src/interlace/engines/__init__.py +16 -0
  22. interlaced-1.0.1/src/interlace/engines/base.py +86 -0
  23. interlaced-1.0.1/src/interlace/engines/duckdb.py +306 -0
  24. interlaced-1.0.1/src/interlace/engines/postgres.py +181 -0
  25. interlaced-1.0.1/src/interlace/engines/quack.py +113 -0
  26. interlaced-1.0.1/src/interlace/engines/registry.py +125 -0
  27. interlaced-1.0.1/src/interlace/exceptions.py +66 -0
  28. interlaced-1.0.1/src/interlace/exports.py +162 -0
  29. interlaced-1.0.1/src/interlace/graph/__init__.py +17 -0
  30. interlaced-1.0.1/src/interlace/graph/column_lineage.py +69 -0
  31. interlaced-1.0.1/src/interlace/graph/dag.py +81 -0
  32. interlaced-1.0.1/src/interlace/graph/project.py +252 -0
  33. interlaced-1.0.1/src/interlace/graph/selectors.py +87 -0
  34. interlaced-1.0.1/src/interlace/ir/__init__.py +24 -0
  35. interlaced-1.0.1/src/interlace/ir/canonicalize.py +71 -0
  36. interlaced-1.0.1/src/interlace/ir/fingerprint.py +55 -0
  37. interlaced-1.0.1/src/interlace/ir/relation.py +71 -0
  38. interlaced-1.0.1/src/interlace/ir/schema.py +23 -0
  39. interlaced-1.0.1/src/interlace/plan/__init__.py +23 -0
  40. interlaced-1.0.1/src/interlace/plan/apply.py +625 -0
  41. interlaced-1.0.1/src/interlace/plan/differ.py +385 -0
  42. interlaced-1.0.1/src/interlace/plan/plan.py +192 -0
  43. interlaced-1.0.1/src/interlace/plan/resolve.py +74 -0
  44. interlaced-1.0.1/src/interlace/plan/run.py +113 -0
  45. interlaced-1.0.1/src/interlace/project.py +278 -0
  46. interlaced-1.0.1/src/interlace/py.typed +0 -0
  47. interlaced-1.0.1/src/interlace/runtime/__init__.py +6 -0
  48. interlaced-1.0.1/src/interlace/runtime/handles.py +48 -0
  49. interlaced-1.0.1/src/interlace/runtime/python_model.py +175 -0
  50. interlaced-1.0.1/src/interlace/scaffold.py +83 -0
  51. interlaced-1.0.1/src/interlace/scheduler/__init__.py +17 -0
  52. interlaced-1.0.1/src/interlace/scheduler/engine.py +66 -0
  53. interlaced-1.0.1/src/interlace/scheduler/triggers.py +84 -0
  54. interlaced-1.0.1/src/interlace/scheduler/worker.py +207 -0
  55. interlaced-1.0.1/src/interlace/service/__init__.py +7 -0
  56. interlaced-1.0.1/src/interlace/service/app.py +1508 -0
  57. interlaced-1.0.1/src/interlace/service/auth.py +42 -0
  58. interlaced-1.0.1/src/interlace/service/ui/app.css +356 -0
  59. interlaced-1.0.1/src/interlace/service/ui/favicon.svg +1 -0
  60. interlaced-1.0.1/src/interlace/service/ui/index.html +65 -0
  61. interlaced-1.0.1/src/interlace/service/ui/js/api.js +113 -0
  62. interlaced-1.0.1/src/interlace/service/ui/js/app.js +274 -0
  63. interlaced-1.0.1/src/interlace/service/ui/js/dag.js +433 -0
  64. interlaced-1.0.1/src/interlace/service/ui/js/ui.js +209 -0
  65. interlaced-1.0.1/src/interlace/service/ui/js/views/checks.js +0 -0
  66. interlaced-1.0.1/src/interlace/service/ui/js/views/environments.js +231 -0
  67. interlaced-1.0.1/src/interlace/service/ui/js/views/lineage.js +118 -0
  68. interlaced-1.0.1/src/interlace/service/ui/js/views/models.js +264 -0
  69. interlaced-1.0.1/src/interlace/service/ui/js/views/overview.js +93 -0
  70. interlaced-1.0.1/src/interlace/service/ui/js/views/plan.js +172 -0
  71. interlaced-1.0.1/src/interlace/service/ui/js/views/query.js +242 -0
  72. interlaced-1.0.1/src/interlace/service/ui/js/views/runs.js +264 -0
  73. interlaced-1.0.1/src/interlace/service/ui/js/views/streams.js +198 -0
  74. interlaced-1.0.1/src/interlace/service/ui/js/views/system.js +294 -0
  75. interlaced-1.0.1/src/interlace/state/__init__.py +18 -0
  76. interlaced-1.0.1/src/interlace/state/interval.py +148 -0
  77. interlaced-1.0.1/src/interlace/state/janitor.py +216 -0
  78. interlaced-1.0.1/src/interlace/state/snapshot.py +47 -0
  79. interlaced-1.0.1/src/interlace/state/store.py +1061 -0
  80. interlaced-1.0.1/src/interlace/strategies/__init__.py +64 -0
  81. interlaced-1.0.1/src/interlace/strategies/base.py +71 -0
  82. interlaced-1.0.1/src/interlace/strategies/full.py +38 -0
  83. interlaced-1.0.1/src/interlace/strategies/full_merge.py +98 -0
  84. interlaced-1.0.1/src/interlace/strategies/incremental_by_time.py +65 -0
  85. interlaced-1.0.1/src/interlace/strategies/merge_by_key.py +69 -0
  86. interlaced-1.0.1/src/interlace/strategies/scd_type_2.py +122 -0
  87. interlaced-1.0.1/src/interlace/strategies/view.py +33 -0
  88. interlaced-1.0.1/src/interlace/streaming/__init__.py +18 -0
  89. interlaced-1.0.1/src/interlace/streaming/log.py +323 -0
  90. interlaced-1.0.1/src/interlace/streaming/materializer.py +196 -0
  91. interlaced-1.0.1/src/interlace/streaming/schema.py +229 -0
  92. interlaced-1.0.1/src/interlaced.egg-info/PKG-INFO +246 -0
  93. interlaced-1.0.1/src/interlaced.egg-info/SOURCES.txt +95 -0
  94. interlaced-1.0.1/src/interlaced.egg-info/dependency_links.txt +1 -0
  95. interlaced-1.0.1/src/interlaced.egg-info/entry_points.txt +2 -0
  96. interlaced-1.0.1/src/interlaced.egg-info/requires.txt +38 -0
  97. interlaced-1.0.1/src/interlaced.egg-info/top_level.txt +1 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025-2026 Interlace Contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,2 @@
1
+ prune tests
2
+ prune tasks
@@ -0,0 +1,246 @@
1
+ Metadata-Version: 2.4
2
+ Name: interlaced
3
+ Version: 1.0.1
4
+ Summary: Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion
5
+ Author-email: Mark <mark@interlace.sh>
6
+ License-Expression: MIT
7
+ Keywords: data,pipeline,orchestration,transformation,etl,streaming,dbt,sqlmesh,duckdb
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Topic :: Software Development :: Build Tools
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: SQL
13
+ Requires-Python: >=3.12
14
+ Description-Content-Type: text/markdown
15
+ License-File: LICENSE
16
+ Requires-Dist: sqlglot<31.0,>=25.0
17
+ Requires-Dist: duckdb>=1.5.3
18
+ Requires-Dist: pyarrow>=17.0
19
+ Requires-Dist: pydantic<3.0,>=2.5
20
+ Requires-Dist: typer<1.0,>=0.12
21
+ Requires-Dist: rich<15.0,>=13.0
22
+ Requires-Dist: cronsim<3.0,>=2.5
23
+ Requires-Dist: tenacity<10.0,>=8.2
24
+ Requires-Dist: pyyaml<7.0,>=6.0
25
+ Provides-Extra: service
26
+ Requires-Dist: litestar<3.0,>=2.12; extra == "service"
27
+ Requires-Dist: uvicorn<1.0,>=0.30; extra == "service"
28
+ Requires-Dist: msgspec<1.0,>=0.18; extra == "service"
29
+ Provides-Extra: adbc
30
+ Requires-Dist: adbc-driver-manager<2.0,>=1.2; extra == "adbc"
31
+ Requires-Dist: adbc-driver-postgresql<2.0,>=1.2; extra == "adbc"
32
+ Provides-Extra: postgres
33
+ Requires-Dist: psycopg[binary]<4.0,>=3.1; extra == "postgres"
34
+ Provides-Extra: polars
35
+ Requires-Dist: polars<2.0,>=1.0; extra == "polars"
36
+ Provides-Extra: pandas
37
+ Requires-Dist: pandas<4.0,>=2.0; extra == "pandas"
38
+ Provides-Extra: all
39
+ Requires-Dist: interlaced[adbc,polars,postgres,service]; extra == "all"
40
+ Provides-Extra: dev
41
+ Requires-Dist: pytest<10.0,>=8.0; extra == "dev"
42
+ Requires-Dist: httpx<1.0,>=0.27; extra == "dev"
43
+ Requires-Dist: pytest-asyncio<2.0,>=1.0; extra == "dev"
44
+ Requires-Dist: ruff<1.0,>=0.6; extra == "dev"
45
+ Requires-Dist: black<27.0,>=24.0; extra == "dev"
46
+ Requires-Dist: mypy<2.0,>=1.11; extra == "dev"
47
+ Dynamic: license-file
48
+
49
+ # interlace
50
+
51
+ **Python/SQL-first data platform: transformation, orchestration, and durable streaming — one process.**
52
+
53
+ interlace is an independent, MIT-licensed alternative to dbt/SQLMesh that also replaces the
54
+ orchestrator (no Airflow) and the ingestion layer (Cloudflare-Pipelines-style durable streams).
55
+ Models are `.sql` files or Python functions; state is versioned snapshots with virtual
56
+ environments and a terraform-style plan/apply; everything runs in a single daemon on
57
+ DuckDB + DuckLake by default.
58
+
59
+ > **Status: 1.0.** Requires Python 3.12+.
60
+ > The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
61
+
62
+ ```bash
63
+ uv pip install "interlaced[service]" # extras: service, adbc, postgres, polars, pandas, all
64
+ ```
65
+
66
+ ## Sixty seconds
67
+
68
+ ```bash
69
+ interlace init my-project && cd my-project
70
+ interlace plan # terraform-style preview: added / breaking / non-breaking / reuse
71
+ interlace apply # build changed models, run checks, promote the environment
72
+ interlace serve # the daemon: web UI (/ui) + HTTP API + scheduler + streams, one process
73
+ ```
74
+
75
+ Every model builds into a fingerprinted physical table (`interlace__main.orders__a1b2c3`);
76
+ environments are views over those tables, so promotion and rollback are atomic view swaps and a
77
+ dev environment reuses prod's tables for free. **Production is the unprefixed namespace** —
78
+ consumers query `main.orders`; sandboxes are prefixed (`dev__main.orders`). Commands default to
79
+ prod; pass `--env dev` while developing.
80
+
81
+ ## Models
82
+
83
+ **SQL** — a file per model; upstreams referenced by model name, dependencies inferred by parsing
84
+ (sqlglot), config in a leading comment block:
85
+
86
+ ```sql
87
+ /* interlace:
88
+ strategy: scd_type_2
89
+ key: customer_id
90
+ schedule: {cron: "0 * * * *"}
91
+ checks:
92
+ - not_null: customer_id
93
+ - unique: customer_id
94
+ */
95
+ SELECT customer_id, name, tier FROM raw_customers
96
+ ```
97
+
98
+ **Python** — functions whose parameters name their upstreams; data crosses as Arrow
99
+ (never pandas), streamed with bounded memory:
100
+
101
+ ```python
102
+ from interlace import model
103
+
104
+ @model(strategy="merge_by_key", key="order_id", cursor="updated_at")
105
+ def orders(cursor, this):
106
+ """Incremental API extract: `cursor` is max(updated_at) already in the
107
+ warehouse (None on first run); `this` is the previous materialisation."""
108
+ rows = fetch_orders(since=cursor) # your code
109
+ return pyarrow.Table.from_pylist(rows) # or RecordBatchReader / generator of batches
110
+ ```
111
+
112
+ **Strategies:** `full`, `view`, `ephemeral` (CTE-inlined), `merge_by_key` (upsert),
113
+ `full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
114
+ (windowed, interval-ledger backfill/catchup), `scd_type_2` (history with validity windows).
115
+
116
+ ## Plan / apply
117
+
118
+ ```
119
+ $ interlace plan
120
+ Model Change Category Build
121
+ orders modified non_breaking rebuild
122
+ order_stats modified non_breaking reuse <- output provably identical: not rebuilt
123
+ ```
124
+
125
+ - Changes classify **breaking / non-breaking / forward-only**; a plan with breaking changes
126
+ refuses to apply without `--force`. Downstream models whose output is provably identical
127
+ (column-pruned impact analysis) **reuse their existing tables** instead of rebuilding — an
128
+ improvement over model-granular invalidation.
129
+ - `apply --forward-only` lets history-keeping models (scd2/merge/incremental) survive a
130
+ definition change: the existing table is copied to the new version, the new logic applies to
131
+ the copy going forward, and checks gate before views move.
132
+ - **Checks gate promotion**: 10 built-in types (not_null, unique, accepted_values, row_count,
133
+ freshness, expression, relationships, pattern, range, sql) plus `@check` Python functions —
134
+ an error-severity failure blocks before the environment view moves. `interlace checks run`
135
+ re-runs them ad hoc against any environment's promoted tables.
136
+ - `interlace gc` removes snapshots no environment references (reference-aware: tables shared
137
+ through reuse survive).
138
+
139
+ ## Streaming
140
+
141
+ Declare a stream; POST to it; rows are durable (SQLite WAL log) before the 200, deduplicated by
142
+ idempotency key, and materialized exactly-once into `streams.<name>` — a micro-batch flusher
143
+ commits the data and the watermark in one warehouse transaction, and SQL models just read the
144
+ table. A flush triggers the models that consume the stream.
145
+
146
+ ```python
147
+ from interlace import stream
148
+
149
+ @stream("orders", schema={"order_id": "string", "total": "double"},
150
+ idempotency_key="order_id", retention="7d", on_schema_drift="evolve")
151
+ def orders(event): ...
152
+ ```
153
+
154
+ ```bash
155
+ curl -X POST localhost:8000/streams/orders -d '{"order_id": "o1", "total": 49.5}'
156
+ ```
157
+
158
+ Schema drift is yours to choose: `reject` (400), `evolve` (new columns appear), or
159
+ `quarantine` (bad events divert to `<stream>__quarantine`). When the warehouse falls behind,
160
+ publishes get **429 backpressure** instead of unbounded backlog.
161
+
162
+ ## Reverse ETL
163
+
164
+ Attach external databases and deliver model results into them — the live table is never
165
+ dropped, keyed modes reuse the same merge strategies:
166
+
167
+ ```yaml
168
+ # interlace.yaml
169
+ attach:
170
+ crm: "postgres:host=... dbname=crm"
171
+ ```
172
+
173
+ ```sql
174
+ /* interlace: {export: {to: table, target: crm.public.accounts, mode: merge_by_key, key: id}} */
175
+ SELECT id, tier, lifetime_value FROM account_summary
176
+ ```
177
+
178
+ File exports (`to: parquet|csv|json`) work the same way. Sinks are **environment-gated**: by
179
+ default the side effect fires only from prod — a dev apply never writes to a live external
180
+ table (opt in with `environments: [dev, prod]`).
181
+
182
+ ## Multi-engine
183
+
184
+ Models run on **named engines**: DuckDB/DuckLake by default, Postgres natively over ADBC
185
+ (`pip install 'interlaced[adbc]'`), with per-model pinning:
186
+
187
+ ```yaml
188
+ engines:
189
+ pg: {type: postgres, database: "${PG_DSN}"}
190
+ ```
191
+
192
+ ```sql
193
+ /* interlace: {engine: pg, strategy: merge_by_key, key: id} */
194
+ ```
195
+
196
+ Strategies execute *inside* the pinned engine (no DuckDB middleman); cross-engine dependencies
197
+ appear as explicit **transfer** lines in the plan and move as Arrow (or a federated `ATTACH`
198
+ fast lane when possible). Contract: `docs/architecture/MULTI_ENGINE.md`.
199
+
200
+ ## The daemon
201
+
202
+ `interlace serve` runs everything in one process:
203
+
204
+ - the **web UI** at `/ui` (in-package, zero build step) — ten views: overview, lineage canvas
205
+ with column-level tracing, models, plan/apply with SQL diffs, live runs, query console,
206
+ streams, checks, environments, and system — live over SSE;
207
+ - the **HTTP API** (Litestar + msgspec, OpenAPI at `/schema/scalar`) with the same surface as
208
+ the CLI: plan/apply, runs, checks, streams, engines, schedules, lineage, query, gc;
209
+ - the **scheduler**: cron/interval triggers enqueue onto a **durable run queue** (leases,
210
+ retries, cooperative cancellation — `interlace cancel <id>` or `POST /runs/{id}/cancel`);
211
+ - **stream ingestion** and retention.
212
+
213
+ Scoped API keys (`interlace apikey create ci --scope read`) lock it down; a durable event log
214
+ backs `GET /events/stream` (SSE with `Last-Event-ID` replay).
215
+
216
+ Add `--quack quack:localhost:4213` to serve the warehouse itself over DuckDB's quack protocol —
217
+ other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by setting
218
+ `database: quack:localhost:4213`.
219
+
220
+ ## Architecture in five lines
221
+
222
+ - The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
223
+ are AST builders and dialect appears only at `transpile()`.
224
+ - Storage defaults to **DuckLake** (Parquet + SQL catalog) opened as DuckDB's primary database.
225
+ - Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
226
+ scale-out swap.
227
+ - Streams live in their own durable log; the materializer commits data + watermark in one
228
+ warehouse transaction — exactly-once without distributed coordination.
229
+ - No Jinja, no pandas in core, no external orchestrator.
230
+
231
+ The full design rationale lives in `docs/architecture/v2-design.md`.
232
+
233
+ ## Development
234
+
235
+ Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
236
+ [moon](https://moonrepo.dev/moon), `uv` owns the virtualenv:
237
+
238
+ ```bash
239
+ proto install
240
+ moon run interlace:sync # install deps
241
+ moon run interlace:test # 350+ tests
242
+ moon run interlace:check # black + ruff (CI equivalent)
243
+ moon run interlace:typecheck # mypy
244
+ ```
245
+
246
+ MIT licensed.
@@ -0,0 +1,198 @@
1
+ # interlace
2
+
3
+ **Python/SQL-first data platform: transformation, orchestration, and durable streaming — one process.**
4
+
5
+ interlace is an independent, MIT-licensed alternative to dbt/SQLMesh that also replaces the
6
+ orchestrator (no Airflow) and the ingestion layer (Cloudflare-Pipelines-style durable streams).
7
+ Models are `.sql` files or Python functions; state is versioned snapshots with virtual
8
+ environments and a terraform-style plan/apply; everything runs in a single daemon on
9
+ DuckDB + DuckLake by default.
10
+
11
+ > **Status: 1.0.** Requires Python 3.12+.
12
+ > The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
13
+
14
+ ```bash
15
+ uv pip install "interlaced[service]" # extras: service, adbc, postgres, polars, pandas, all
16
+ ```
17
+
18
+ ## Sixty seconds
19
+
20
+ ```bash
21
+ interlace init my-project && cd my-project
22
+ interlace plan # terraform-style preview: added / breaking / non-breaking / reuse
23
+ interlace apply # build changed models, run checks, promote the environment
24
+ interlace serve # the daemon: web UI (/ui) + HTTP API + scheduler + streams, one process
25
+ ```
26
+
27
+ Every model builds into a fingerprinted physical table (`interlace__main.orders__a1b2c3`);
28
+ environments are views over those tables, so promotion and rollback are atomic view swaps and a
29
+ dev environment reuses prod's tables for free. **Production is the unprefixed namespace** —
30
+ consumers query `main.orders`; sandboxes are prefixed (`dev__main.orders`). Commands default to
31
+ prod; pass `--env dev` while developing.
32
+
33
+ ## Models
34
+
35
+ **SQL** — a file per model; upstreams referenced by model name, dependencies inferred by parsing
36
+ (sqlglot), config in a leading comment block:
37
+
38
+ ```sql
39
+ /* interlace:
40
+ strategy: scd_type_2
41
+ key: customer_id
42
+ schedule: {cron: "0 * * * *"}
43
+ checks:
44
+ - not_null: customer_id
45
+ - unique: customer_id
46
+ */
47
+ SELECT customer_id, name, tier FROM raw_customers
48
+ ```
49
+
50
+ **Python** — functions whose parameters name their upstreams; data crosses as Arrow
51
+ (never pandas), streamed with bounded memory:
52
+
53
+ ```python
54
+ from interlace import model
55
+
56
+ @model(strategy="merge_by_key", key="order_id", cursor="updated_at")
57
+ def orders(cursor, this):
58
+ """Incremental API extract: `cursor` is max(updated_at) already in the
59
+ warehouse (None on first run); `this` is the previous materialisation."""
60
+ rows = fetch_orders(since=cursor) # your code
61
+ return pyarrow.Table.from_pylist(rows) # or RecordBatchReader / generator of batches
62
+ ```
63
+
64
+ **Strategies:** `full`, `view`, `ephemeral` (CTE-inlined), `merge_by_key` (upsert),
65
+ `full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
66
+ (windowed, interval-ledger backfill/catchup), `scd_type_2` (history with validity windows).
67
+
68
+ ## Plan / apply
69
+
70
+ ```
71
+ $ interlace plan
72
+ Model Change Category Build
73
+ orders modified non_breaking rebuild
74
+ order_stats modified non_breaking reuse <- output provably identical: not rebuilt
75
+ ```
76
+
77
+ - Changes classify **breaking / non-breaking / forward-only**; a plan with breaking changes
78
+ refuses to apply without `--force`. Downstream models whose output is provably identical
79
+ (column-pruned impact analysis) **reuse their existing tables** instead of rebuilding — an
80
+ improvement over model-granular invalidation.
81
+ - `apply --forward-only` lets history-keeping models (scd2/merge/incremental) survive a
82
+ definition change: the existing table is copied to the new version, the new logic applies to
83
+ the copy going forward, and checks gate before views move.
84
+ - **Checks gate promotion**: 10 built-in types (not_null, unique, accepted_values, row_count,
85
+ freshness, expression, relationships, pattern, range, sql) plus `@check` Python functions —
86
+ an error-severity failure blocks before the environment view moves. `interlace checks run`
87
+ re-runs them ad hoc against any environment's promoted tables.
88
+ - `interlace gc` removes snapshots no environment references (reference-aware: tables shared
89
+ through reuse survive).
90
+
91
+ ## Streaming
92
+
93
+ Declare a stream; POST to it; rows are durable (SQLite WAL log) before the 200, deduplicated by
94
+ idempotency key, and materialized exactly-once into `streams.<name>` — a micro-batch flusher
95
+ commits the data and the watermark in one warehouse transaction, and SQL models just read the
96
+ table. A flush triggers the models that consume the stream.
97
+
98
+ ```python
99
+ from interlace import stream
100
+
101
+ @stream("orders", schema={"order_id": "string", "total": "double"},
102
+ idempotency_key="order_id", retention="7d", on_schema_drift="evolve")
103
+ def orders(event): ...
104
+ ```
105
+
106
+ ```bash
107
+ curl -X POST localhost:8000/streams/orders -d '{"order_id": "o1", "total": 49.5}'
108
+ ```
109
+
110
+ Schema drift is yours to choose: `reject` (400), `evolve` (new columns appear), or
111
+ `quarantine` (bad events divert to `<stream>__quarantine`). When the warehouse falls behind,
112
+ publishes get **429 backpressure** instead of unbounded backlog.
113
+
114
+ ## Reverse ETL
115
+
116
+ Attach external databases and deliver model results into them — the live table is never
117
+ dropped, keyed modes reuse the same merge strategies:
118
+
119
+ ```yaml
120
+ # interlace.yaml
121
+ attach:
122
+ crm: "postgres:host=... dbname=crm"
123
+ ```
124
+
125
+ ```sql
126
+ /* interlace: {export: {to: table, target: crm.public.accounts, mode: merge_by_key, key: id}} */
127
+ SELECT id, tier, lifetime_value FROM account_summary
128
+ ```
129
+
130
+ File exports (`to: parquet|csv|json`) work the same way. Sinks are **environment-gated**: by
131
+ default the side effect fires only from prod — a dev apply never writes to a live external
132
+ table (opt in with `environments: [dev, prod]`).
133
+
134
+ ## Multi-engine
135
+
136
+ Models run on **named engines**: DuckDB/DuckLake by default, Postgres natively over ADBC
137
+ (`pip install 'interlaced[adbc]'`), with per-model pinning:
138
+
139
+ ```yaml
140
+ engines:
141
+ pg: {type: postgres, database: "${PG_DSN}"}
142
+ ```
143
+
144
+ ```sql
145
+ /* interlace: {engine: pg, strategy: merge_by_key, key: id} */
146
+ ```
147
+
148
+ Strategies execute *inside* the pinned engine (no DuckDB middleman); cross-engine dependencies
149
+ appear as explicit **transfer** lines in the plan and move as Arrow (or a federated `ATTACH`
150
+ fast lane when possible). Contract: `docs/architecture/MULTI_ENGINE.md`.
151
+
152
+ ## The daemon
153
+
154
+ `interlace serve` runs everything in one process:
155
+
156
+ - the **web UI** at `/ui` (in-package, zero build step) — ten views: overview, lineage canvas
157
+ with column-level tracing, models, plan/apply with SQL diffs, live runs, query console,
158
+ streams, checks, environments, and system — live over SSE;
159
+ - the **HTTP API** (Litestar + msgspec, OpenAPI at `/schema/scalar`) with the same surface as
160
+ the CLI: plan/apply, runs, checks, streams, engines, schedules, lineage, query, gc;
161
+ - the **scheduler**: cron/interval triggers enqueue onto a **durable run queue** (leases,
162
+ retries, cooperative cancellation — `interlace cancel <id>` or `POST /runs/{id}/cancel`);
163
+ - **stream ingestion** and retention.
164
+
165
+ Scoped API keys (`interlace apikey create ci --scope read`) lock it down; a durable event log
166
+ backs `GET /events/stream` (SSE with `Last-Event-ID` replay).
167
+
168
+ Add `--quack quack:localhost:4213` to serve the warehouse itself over DuckDB's quack protocol —
169
+ other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by setting
170
+ `database: quack:localhost:4213`.
171
+
172
+ ## Architecture in five lines
173
+
174
+ - The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
175
+ are AST builders and dialect appears only at `transpile()`.
176
+ - Storage defaults to **DuckLake** (Parquet + SQL catalog) opened as DuckDB's primary database.
177
+ - Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
178
+ scale-out swap.
179
+ - Streams live in their own durable log; the materializer commits data + watermark in one
180
+ warehouse transaction — exactly-once without distributed coordination.
181
+ - No Jinja, no pandas in core, no external orchestrator.
182
+
183
+ The full design rationale lives in `docs/architecture/v2-design.md`.
184
+
185
+ ## Development
186
+
187
+ Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
188
+ [moon](https://moonrepo.dev/moon), `uv` owns the virtualenv:
189
+
190
+ ```bash
191
+ proto install
192
+ moon run interlace:sync # install deps
193
+ moon run interlace:test # 350+ tests
194
+ moon run interlace:check # black + ruff (CI equivalent)
195
+ moon run interlace:typecheck # mypy
196
+ ```
197
+
198
+ MIT licensed.
@@ -0,0 +1,155 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "interlaced"
7
+ version = "1.0.1"
8
+ description = "Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion"
9
+ readme = "README.md"
10
+ requires-python = ">=3.12"
11
+ license = "MIT"
12
+ authors = [
13
+ {name = "Mark", email = "mark@interlace.sh"}
14
+ ]
15
+ keywords = ["data", "pipeline", "orchestration", "transformation", "etl", "streaming", "dbt", "sqlmesh", "duckdb"]
16
+ classifiers = [
17
+ "Development Status :: 4 - Beta",
18
+ "Intended Audience :: Developers",
19
+ "Topic :: Software Development :: Build Tools",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: SQL",
22
+ ]
23
+
24
+ # Core: the IR + engine + state + plan spine (Phase 1). Everything is Arrow- and
25
+ # sqlglot-native. See docs/architecture/v2-design.md.
26
+ dependencies = [
27
+ "sqlglot>=25.0,<31.0", # canonical IR, transpilation, semantic diff, column lineage
28
+ "duckdb>=1.5.3", # default engine + federation hub; DuckLake storage + quack serving
29
+ "pyarrow>=17.0", # the wire format: RecordBatchReader everywhere
30
+ "pydantic>=2.5,<3.0", # config + manifest validation (cold paths only)
31
+ "typer>=0.12,<1.0", # CLI
32
+ "rich>=13.0,<15.0", # display, strictly an event subscriber
33
+ "cronsim>=2.5,<3.0", # cron parsing for the trigger engine
34
+ "tenacity>=8.2,<10.0", # retries (tasks, DuckLake commit conflicts, transfers)
35
+ "pyyaml>=6.0,<7.0", # project config (config + env overlays)
36
+ ]
37
+
38
+ [project.scripts]
39
+ interlace = "interlace.cli.main:main"
40
+
41
+ [project.optional-dependencies]
42
+ # Service + orchestration daemon (Phase 2).
43
+ service = [
44
+ "litestar>=2.12,<3.0",
45
+ "uvicorn>=0.30,<1.0",
46
+ "msgspec>=0.18,<1.0", # the wire types (litestar serialises msgspec structs natively)
47
+ ]
48
+ # Remote engines via Arrow-native transfer (Phase 4).
49
+ adbc = [
50
+ "adbc-driver-manager>=1.2,<2.0",
51
+ "adbc-driver-postgresql>=1.2,<2.0",
52
+ ]
53
+ postgres = ["psycopg[binary]>=3.1,<4.0"]
54
+ polars = ["polars>=1.0,<2.0"]
55
+ pandas = ["pandas>=2.0,<4.0"]
56
+ all = ["interlaced[service,adbc,postgres,polars]"]
57
+ dev = [
58
+ "pytest>=8.0,<10.0",
59
+ "httpx>=0.27,<1.0", # litestar's TestClient transport
60
+ "pytest-asyncio>=1.0,<2.0",
61
+ "ruff>=0.6,<1.0",
62
+ "black>=24.0,<27.0",
63
+ "mypy>=1.11,<2.0",
64
+ ]
65
+
66
+ [tool.setuptools.packages.find]
67
+ where = ["src"]
68
+
69
+ [tool.setuptools.package-data]
70
+ interlace = ["py.typed", "service/ui/*", "service/ui/**/*"]
71
+
72
+ [tool.black]
73
+ line-length = 120
74
+ target-version = ['py312']
75
+
76
+ [tool.ruff]
77
+ line-length = 120
78
+ target-version = "py312"
79
+
80
+ [tool.ruff.lint]
81
+ select = [
82
+ "E", # pycodestyle errors
83
+ "W", # pycodestyle warnings
84
+ "F", # pyflakes
85
+ "I", # isort
86
+ "B", # flake8-bugbear
87
+ "C4", # flake8-comprehensions
88
+ "UP", # pyupgrade
89
+ ]
90
+ ignore = [
91
+ "E501", # line too long (handled by formatter)
92
+ "B008", # function calls in argument defaults
93
+ "C901", # too complex
94
+ ]
95
+
96
+ [tool.ruff.lint.per-file-ignores]
97
+ "__init__.py" = ["F401"] # Allow unused imports in __init__.py
98
+
99
+ [tool.mypy]
100
+ python_version = "3.12"
101
+ warn_return_any = true
102
+ warn_unused_configs = true
103
+ disallow_untyped_defs = true
104
+ disallow_incomplete_defs = true
105
+ check_untyped_defs = true
106
+ no_implicit_optional = true
107
+ warn_redundant_casts = true
108
+ warn_unused_ignores = true
109
+
110
+ [[tool.mypy.overrides]]
111
+ module = [
112
+ "sqlglot.*",
113
+ "pyarrow.*",
114
+ "yaml",
115
+ ]
116
+ ignore_missing_imports = true
117
+
118
+ [tool.coverage.run]
119
+ source = ["src/interlace"]
120
+ omit = [
121
+ "*/tests/*",
122
+ "*/test_*.py",
123
+ "*/__pycache__/*",
124
+ "*/__init__.py",
125
+ ]
126
+
127
+ [tool.coverage.report]
128
+ exclude_lines = [
129
+ "pragma: no cover",
130
+ "def __repr__",
131
+ "raise AssertionError",
132
+ "raise NotImplementedError",
133
+ "if __name__ == .__main__.:",
134
+ "if TYPE_CHECKING:",
135
+ "@abstractmethod",
136
+ ]
137
+
138
+ [tool.pytest.ini_options]
139
+ testpaths = ["tests"]
140
+ python_files = ["test_*.py", "*_test.py"]
141
+ python_classes = ["Test*"]
142
+ python_functions = ["test_*"]
143
+ addopts = [
144
+ "-v",
145
+ "--strict-markers",
146
+ "--tb=short",
147
+ ]
148
+ markers = [
149
+ "unit: Unit tests (fast, isolated)",
150
+ "integration: Integration tests (slower, require database)",
151
+ "slow: Slow tests (may take a long time)",
152
+ "requires_db: Needs a reachable external database (e.g. Postgres)",
153
+ ]
154
+ asyncio_mode = "auto"
155
+ asyncio_default_fixture_loop_scope = "function"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,26 @@
1
+ """Interlace v2 — Python/SQL-first data platform.
2
+
3
+ Transformation (sqlmesh-grade snapshots, virtual environments, plan/apply),
4
+ built-in orchestration (durable work queue + unified triggers), and durable
5
+ streaming ingestion — in one process. See docs/architecture/v2-design.md.
6
+
7
+ This package is under active greenfield construction; the public surface is the
8
+ ``@model`` / ``@stream`` / ``@check`` decorators plus the core IR types.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from interlace.dsl.decorators import check, model, stream
14
+ from interlace.ir.relation import EngineRef, SqlRelation, TableRef
15
+
16
+ __version__ = "1.0.0"
17
+
18
+ __all__ = [
19
+ "EngineRef",
20
+ "SqlRelation",
21
+ "TableRef",
22
+ "__version__",
23
+ "check",
24
+ "model",
25
+ "stream",
26
+ ]
@@ -0,0 +1,10 @@
1
+ """Data-quality checks. Results gate promotion: an ``error``-severity failure
2
+ aborts the apply before the environment is promoted.
3
+
4
+ Import :mod:`interlace.checks.runner` for execution — kept out of this package
5
+ init so declaring checks (spec) never drags in the runtime.
6
+ """
7
+
8
+ from interlace.checks.spec import CheckSpec, parse_checks
9
+
10
+ __all__ = ["CheckSpec", "parse_checks"]