interlaced 2.3.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interlaced-2.3.0/src/interlaced.egg-info → interlaced-2.4.0}/PKG-INFO +47 -11
- {interlaced-2.3.0 → interlaced-2.4.0}/README.md +43 -9
- {interlaced-2.3.0 → interlaced-2.4.0}/pyproject.toml +6 -2
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/config/config.py +8 -7
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/dsl/decorators.py +12 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/duckdb.py +54 -3
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/apply.py +117 -24
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/project.py +6 -1
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/scaffold.py +19 -2
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/append.py +12 -4
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/base.py +13 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/hash_merge.py +40 -18
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/incremental.py +6 -0
- interlaced-2.4.0/src/interlace/strategies/merge.py +188 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/scd.py +9 -1
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/interlace.yaml +2 -1
- interlaced-2.4.0/src/interlace/templates/quickstart/interlace.yaml +6 -0
- {interlaced-2.3.0 → interlaced-2.4.0/src/interlaced.egg-info}/PKG-INFO +47 -11
- interlaced-2.3.0/src/interlace/strategies/merge.py +0 -115
- interlaced-2.3.0/src/interlace/templates/quickstart/interlace.yaml +0 -6
- {interlaced-2.3.0 → interlaced-2.4.0}/LICENSE +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/MANIFEST.in +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/setup.cfg +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/checks/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/checks/builtin.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/checks/runner.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/checks/spec.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/cli/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/cli/main.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/config/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/contracts.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/dsl/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/dsl/discovery.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/dsl/sql_config.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/adbc.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/base.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/bigquery.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/postgres.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/quack.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/redshift.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/registry.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/snowflake.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/engines/spark.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/exceptions.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/graph/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/graph/column_lineage.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/graph/dag.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/graph/project.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/graph/selectors.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/ir/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/ir/canonicalize.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/ir/fingerprint.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/ir/relation.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/differ.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/plan.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/resolve.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/plan/run.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/py.typed +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/query.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/runtime/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/runtime/handles.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/runtime/python_model.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/scheduler/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/scheduler/engine.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/scheduler/triggers.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/scheduler/worker.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/app.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/auth.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/app.css +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/favicon.svg +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/index.html +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/api.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/app.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/dag.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/timeline.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/ui.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/checks.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/environments.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/lineage.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/models.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/overview.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/plan.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/query.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/runs.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/streams.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/service/ui/js/views/system.js +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/sinks.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/sources/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/sources/auth.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/sources/rest.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/state/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/state/interval.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/state/janitor.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/state/snapshot.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/state/store.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/full_merge.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/replace.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/replace_in_place.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/strategies/view.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/streaming/__init__.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/streaming/log.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/streaming/materializer.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/streaming/schema.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/README.md +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/generate.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/models/events.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/models/events_by_minute.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/models/events_by_type.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/models/top_users.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/models/user_spend.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/events/template.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/github/README.md +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/github/interlace.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/github/models/github_issues.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/github/models/issues_by_state.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/github/template.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/README.md +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/docker-compose.yml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/init/seed.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/interlace.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/models/orders.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/models/orders_by_status.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/postgres/template.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/quickstart/README.md +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/quickstart/models/enriched_events.py +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/quickstart/models/event_summary.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/quickstart/models/raw_events.sql +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlace/templates/quickstart/template.yaml +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlaced.egg-info/SOURCES.txt +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlaced.egg-info/dependency_links.txt +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlaced.egg-info/entry_points.txt +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlaced.egg-info/requires.txt +0 -0
- {interlaced-2.3.0 → interlaced-2.4.0}/src/interlaced.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: interlaced
|
|
3
|
-
Version: 2.
|
|
4
|
-
Summary: Python
|
|
3
|
+
Version: 2.4.0
|
|
4
|
+
Summary: Python and SQL models in one DAG. Transformation, orchestration and durable streaming in one process, on DuckDB and Postgres.
|
|
5
5
|
Author-email: Mark <mark@interlace.sh>
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://interlace.sh
|
|
@@ -13,8 +13,10 @@ Keywords: data,pipeline,orchestration,transformation,etl,streaming,dbt,sqlmesh,d
|
|
|
13
13
|
Classifier: Development Status :: 5 - Production/Stable
|
|
14
14
|
Classifier: Intended Audience :: Developers
|
|
15
15
|
Classifier: Topic :: Software Development :: Build Tools
|
|
16
|
+
Classifier: Topic :: Database
|
|
16
17
|
Classifier: Programming Language :: Python :: 3.12
|
|
17
18
|
Classifier: Programming Language :: SQL
|
|
19
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
18
20
|
Requires-Python: >=3.12
|
|
19
21
|
Description-Content-Type: text/markdown
|
|
20
22
|
License-File: LICENSE
|
|
@@ -64,15 +66,41 @@ Dynamic: license-file
|
|
|
64
66
|
|
|
65
67
|
# interlace
|
|
66
68
|
|
|
67
|
-
**Python
|
|
69
|
+
**Python and SQL models are the same kind of node in one DAG.**
|
|
68
70
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
71
|
+
A `.py` model sits mid-graph with SQL either side, in both directions, with no bridge and no
|
|
72
|
+
separate runtime — running in-process on DuckDB and Postgres, not only on a cloud warehouse.
|
|
73
|
+
The Python model stays a plain function: call it in a test with no warehouse and no session.
|
|
74
|
+
|
|
75
|
+
How that compares with dbt and SQLMesh, including where they are ahead, is on
|
|
76
|
+
[interlace.sh/why](https://interlace.sh/why).
|
|
77
|
+
|
|
78
|
+
```sql
|
|
79
|
+
-- models/raw_events.sql SQL
|
|
80
|
+
SELECT event_id, user_id, kind, amount, country, ts FROM read_parquet('events/*.parquet')
|
|
81
|
+
```
|
|
82
|
+
```python
|
|
83
|
+
# models/enriched_events.py Python, mid-DAG
|
|
84
|
+
@model() # the parameter name IS the dependency — no depends_on
|
|
85
|
+
def enriched_events(raw_events):
|
|
86
|
+
for batch in raw_events.reader(): # Arrow in, Arrow out, bounded memory
|
|
87
|
+
yield add_revenue(batch)
|
|
88
|
+
```
|
|
89
|
+
```sql
|
|
90
|
+
-- models/event_summary.sql SQL again, straight over the Python
|
|
91
|
+
SELECT country, count(*) FILTER (WHERE is_conversion) AS conversions
|
|
92
|
+
FROM enriched_events GROUP BY country
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
`interlace init` scaffolds exactly this shape, runnable, with no external source.
|
|
96
|
+
|
|
97
|
+
That is the wedge. The rest is the reveal: interlace is an independent, MIT-licensed alternative
|
|
98
|
+
to dbt/SQLMesh that also replaces the orchestrator (no Airflow) and the ingestion layer
|
|
99
|
+
(Cloudflare-Pipelines-style durable streams). State is versioned snapshots with virtual
|
|
72
100
|
environments and a terraform-style plan/apply; everything runs in a single daemon on
|
|
73
|
-
DuckDB
|
|
101
|
+
DuckDB by default (DuckLake one config line away).
|
|
74
102
|
|
|
75
|
-
> **
|
|
103
|
+
> **2.x — see [releases](https://github.com/interlace-sh/interlace/releases).** Requires Python 3.12+.
|
|
76
104
|
> The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
|
|
77
105
|
|
|
78
106
|
```bash
|
|
@@ -80,6 +108,13 @@ pip install 'interlaced[service]' # the CLI + daemon; core CLI only: pip insta
|
|
|
80
108
|
# more extras: [adbc] postgres/redshift · [spark] · [polars] · [all]
|
|
81
109
|
```
|
|
82
110
|
|
|
111
|
+
> **Platforms.** Developed on Linux; CI runs Linux only. Nothing in the codebase is
|
|
112
|
+
> platform-specific — no `fork`, no signal handling, no POSIX-only calls, no shelling out — and
|
|
113
|
+
> every dependency ships macOS and Windows wheels, so both are expected to work. But
|
|
114
|
+
> **neither is tested**, so treat them as unverified rather than supported. If you run interlace
|
|
115
|
+
> on macOS or Windows, please open an issue either way; that is the fastest route to changing
|
|
116
|
+
> this paragraph.
|
|
117
|
+
|
|
83
118
|
## Sixty seconds
|
|
84
119
|
|
|
85
120
|
```bash
|
|
@@ -210,7 +245,7 @@ prod, so a dev apply never writes to a live external table (opt in with
|
|
|
210
245
|
|
|
211
246
|
## Multi-engine
|
|
212
247
|
|
|
213
|
-
Models run on **named engines**: DuckDB
|
|
248
|
+
Models run on **named engines**: DuckDB by default (DuckLake opt-in), Postgres natively over ADBC
|
|
214
249
|
(`pip install 'interlaced[adbc]'`), Spark (beta, `[spark]` extra), plus alpha adapters for
|
|
215
250
|
MotherDuck, Redshift, Snowflake and BigQuery (wired and dialect-correct, not yet run against a
|
|
216
251
|
live account), with per-model pinning:
|
|
@@ -252,7 +287,8 @@ other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by
|
|
|
252
287
|
|
|
253
288
|
- The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
|
|
254
289
|
are AST builders and dialect appears only at `transpile()`.
|
|
255
|
-
- Storage defaults to **DuckLake** (Parquet + SQL catalog
|
|
290
|
+
- Storage defaults to a plain **DuckDB** file; **DuckLake** (Parquet + SQL catalog, and
|
|
291
|
+
concurrent writers) is one `database: ducklake:…` line away.
|
|
256
292
|
- Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
|
|
257
293
|
scale-out swap.
|
|
258
294
|
- Streams live in their own durable log; the materializer commits data + watermark in one
|
|
@@ -269,7 +305,7 @@ Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
|
|
|
269
305
|
```bash
|
|
270
306
|
proto install
|
|
271
307
|
moon run interlace:sync # install deps
|
|
272
|
-
moon run interlace:test #
|
|
308
|
+
moon run interlace:test # 500+ tests
|
|
273
309
|
moon run interlace:check # black + ruff (CI equivalent)
|
|
274
310
|
moon run interlace:typecheck # mypy
|
|
275
311
|
```
|
|
@@ -1,14 +1,40 @@
|
|
|
1
1
|
# interlace
|
|
2
2
|
|
|
3
|
-
**Python
|
|
3
|
+
**Python and SQL models are the same kind of node in one DAG.**
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
5
|
+
A `.py` model sits mid-graph with SQL either side, in both directions, with no bridge and no
|
|
6
|
+
separate runtime — running in-process on DuckDB and Postgres, not only on a cloud warehouse.
|
|
7
|
+
The Python model stays a plain function: call it in a test with no warehouse and no session.
|
|
8
|
+
|
|
9
|
+
How that compares with dbt and SQLMesh, including where they are ahead, is on
|
|
10
|
+
[interlace.sh/why](https://interlace.sh/why).
|
|
11
|
+
|
|
12
|
+
```sql
|
|
13
|
+
-- models/raw_events.sql SQL
|
|
14
|
+
SELECT event_id, user_id, kind, amount, country, ts FROM read_parquet('events/*.parquet')
|
|
15
|
+
```
|
|
16
|
+
```python
|
|
17
|
+
# models/enriched_events.py Python, mid-DAG
|
|
18
|
+
@model() # the parameter name IS the dependency — no depends_on
|
|
19
|
+
def enriched_events(raw_events):
|
|
20
|
+
for batch in raw_events.reader(): # Arrow in, Arrow out, bounded memory
|
|
21
|
+
yield add_revenue(batch)
|
|
22
|
+
```
|
|
23
|
+
```sql
|
|
24
|
+
-- models/event_summary.sql SQL again, straight over the Python
|
|
25
|
+
SELECT country, count(*) FILTER (WHERE is_conversion) AS conversions
|
|
26
|
+
FROM enriched_events GROUP BY country
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
`interlace init` scaffolds exactly this shape, runnable, with no external source.
|
|
30
|
+
|
|
31
|
+
That is the wedge. The rest is the reveal: interlace is an independent, MIT-licensed alternative
|
|
32
|
+
to dbt/SQLMesh that also replaces the orchestrator (no Airflow) and the ingestion layer
|
|
33
|
+
(Cloudflare-Pipelines-style durable streams). State is versioned snapshots with virtual
|
|
8
34
|
environments and a terraform-style plan/apply; everything runs in a single daemon on
|
|
9
|
-
DuckDB
|
|
35
|
+
DuckDB by default (DuckLake one config line away).
|
|
10
36
|
|
|
11
|
-
> **
|
|
37
|
+
> **2.x — see [releases](https://github.com/interlace-sh/interlace/releases).** Requires Python 3.12+.
|
|
12
38
|
> The package is published to PyPI as **`interlaced`**; the import name and CLI are `interlace`.
|
|
13
39
|
|
|
14
40
|
```bash
|
|
@@ -16,6 +42,13 @@ pip install 'interlaced[service]' # the CLI + daemon; core CLI only: pip insta
|
|
|
16
42
|
# more extras: [adbc] postgres/redshift · [spark] · [polars] · [all]
|
|
17
43
|
```
|
|
18
44
|
|
|
45
|
+
> **Platforms.** Developed on Linux; CI runs Linux only. Nothing in the codebase is
|
|
46
|
+
> platform-specific — no `fork`, no signal handling, no POSIX-only calls, no shelling out — and
|
|
47
|
+
> every dependency ships macOS and Windows wheels, so both are expected to work. But
|
|
48
|
+
> **neither is tested**, so treat them as unverified rather than supported. If you run interlace
|
|
49
|
+
> on macOS or Windows, please open an issue either way; that is the fastest route to changing
|
|
50
|
+
> this paragraph.
|
|
51
|
+
|
|
19
52
|
## Sixty seconds
|
|
20
53
|
|
|
21
54
|
```bash
|
|
@@ -146,7 +179,7 @@ prod, so a dev apply never writes to a live external table (opt in with
|
|
|
146
179
|
|
|
147
180
|
## Multi-engine
|
|
148
181
|
|
|
149
|
-
Models run on **named engines**: DuckDB
|
|
182
|
+
Models run on **named engines**: DuckDB by default (DuckLake opt-in), Postgres natively over ADBC
|
|
150
183
|
(`pip install 'interlaced[adbc]'`), Spark (beta, `[spark]` extra), plus alpha adapters for
|
|
151
184
|
MotherDuck, Redshift, Snowflake and BigQuery (wired and dialect-correct, not yet run against a
|
|
152
185
|
live account), with per-model pinning:
|
|
@@ -188,7 +221,8 @@ other processes (CLI runs, ad-hoc DuckDB clients) then share it concurrently by
|
|
|
188
221
|
|
|
189
222
|
- The IR is a **sqlglot AST**; the wire format is an **Arrow RecordBatchReader**; strategies
|
|
190
223
|
are AST builders and dialect appears only at `transpile()`.
|
|
191
|
-
- Storage defaults to **DuckLake** (Parquet + SQL catalog
|
|
224
|
+
- Storage defaults to a plain **DuckDB** file; **DuckLake** (Parquet + SQL catalog, and
|
|
225
|
+
concurrent writers) is one `database: ducklake:…` line away.
|
|
192
226
|
- Control plane (snapshots, intervals, queue, events, keys) is **SQLite WAL**; Postgres is the
|
|
193
227
|
scale-out swap.
|
|
194
228
|
- Streams live in their own durable log; the materializer commits data + watermark in one
|
|
@@ -205,7 +239,7 @@ Toolchain is pinned with [proto](https://moonrepo.dev/proto), tasks run via
|
|
|
205
239
|
```bash
|
|
206
240
|
proto install
|
|
207
241
|
moon run interlace:sync # install deps
|
|
208
|
-
moon run interlace:test #
|
|
242
|
+
moon run interlace:test # 500+ tests
|
|
209
243
|
moon run interlace:check # black + ruff (CI equivalent)
|
|
210
244
|
moon run interlace:typecheck # mypy
|
|
211
245
|
```
|
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "interlaced"
|
|
7
|
-
version = "2.
|
|
8
|
-
description = "Python
|
|
7
|
+
version = "2.4.0"
|
|
8
|
+
description = "Python and SQL models in one DAG. Transformation, orchestration and durable streaming in one process, on DuckDB and Postgres."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.12"
|
|
11
11
|
license = "MIT"
|
|
@@ -17,8 +17,12 @@ classifiers = [
|
|
|
17
17
|
"Development Status :: 5 - Production/Stable",
|
|
18
18
|
"Intended Audience :: Developers",
|
|
19
19
|
"Topic :: Software Development :: Build Tools",
|
|
20
|
+
"Topic :: Database",
|
|
20
21
|
"Programming Language :: Python :: 3.12",
|
|
21
22
|
"Programming Language :: SQL",
|
|
23
|
+
# Linux is what CI runs and what this is developed on. Nothing in the codebase
|
|
24
|
+
# is platform-specific, but macOS and Windows are untested — see the README.
|
|
25
|
+
"Operating System :: POSIX :: Linux",
|
|
22
26
|
]
|
|
23
27
|
|
|
24
28
|
# Core: the IR + engine + state + plan spine (Phase 1). Everything is Arrow- and
|
|
@@ -84,7 +84,7 @@ class EngineConfig(BaseModel):
|
|
|
84
84
|
declaring them fails at open until an adapter ships.
|
|
85
85
|
"""
|
|
86
86
|
|
|
87
|
-
type: str = "
|
|
87
|
+
type: str = "duckdb"
|
|
88
88
|
# Path / URI for DuckDB-family engines. Also accepted on the project top level
|
|
89
89
|
# as ``database:`` (synthesised into the ``default`` engine).
|
|
90
90
|
database: str | None = None
|
|
@@ -117,13 +117,14 @@ class ProjectConfig(BaseModel):
|
|
|
117
117
|
default_engine: str = "default"
|
|
118
118
|
engines: dict[str, EngineConfig] = Field(default_factory=dict)
|
|
119
119
|
state_path: str = ".interlace/state.db" # SQLite control-plane database
|
|
120
|
-
# The warehouse. Default is
|
|
121
|
-
# Also accepted: a DuckLake catalog
|
|
122
|
-
#
|
|
123
|
-
#
|
|
124
|
-
#
|
|
120
|
+
# The warehouse. Default is a plain DuckDB file — simplest, single-process.
|
|
121
|
+
# Also accepted: a DuckLake catalog (``ducklake:.interlace/warehouse.ducklake``, or
|
|
122
|
+
# hosted in a SQL DB: ``ducklake:postgres:dbname=... host=...`` — pair with
|
|
123
|
+
# data_path/metadata_schema) which serialises catalog writes so `interlace serve`
|
|
124
|
+
# and a separate CLI can share the warehouse concurrently; ":memory:"; or
|
|
125
|
+
# "quack:<host>:<port>" to connect to a warehouse served by `interlace serve --quack`.
|
|
125
126
|
# When ``engines.default`` is not set, these top-level fields synthesise it.
|
|
126
|
-
database: str = "
|
|
127
|
+
database: str = ".interlace/warehouse.duckdb"
|
|
127
128
|
# The warehouse catalog's ATTACH alias (defaults to ``name``). Set it when a
|
|
128
129
|
# schema inside the warehouse shares the project name — see EngineConfig.alias.
|
|
129
130
|
alias: str | None = None
|
|
@@ -114,6 +114,18 @@ class ModelDef:
|
|
|
114
114
|
schedule: dict[str, str] | None = None # {"cron": "0 * * * *"} or {"every": "5m"} for `interlace serve`
|
|
115
115
|
checks: tuple[CheckSpec, ...] = () # data-quality checks; error severity gates promotion
|
|
116
116
|
|
|
117
|
+
def __post_init__(self) -> None:
|
|
118
|
+
# `@model(checks=…)` normalises through parse_checks, but a ModelDef built
|
|
119
|
+
# directly — the dynamic-model path, which is what generated models and dbt
|
|
120
|
+
# migrations use — stored the dicts raw and only failed at compile time with
|
|
121
|
+
# `AttributeError: 'dict' object has no attribute 'type'`. Normalise here too,
|
|
122
|
+
# so one spelling works on both surfaces and a bad check fails at declaration.
|
|
123
|
+
# Passed through as-is, not as list(...): parse_checks already handles a bare
|
|
124
|
+
# CheckSpec and reports a non-list clearly, both of which list() would mangle
|
|
125
|
+
# (TypeError on a CheckSpec; a dict silently degraded to its keys). Always run,
|
|
126
|
+
# so `checks=[]` normalises to the declared tuple rather than staying a list.
|
|
127
|
+
self.checks = parse_checks(self.checks, self.name)
|
|
128
|
+
|
|
117
129
|
@property
|
|
118
130
|
def is_terminal(self) -> bool:
|
|
119
131
|
"""A terminal model delivers into an external destination (table/file):
|
|
@@ -28,6 +28,7 @@ from __future__ import annotations
|
|
|
28
28
|
|
|
29
29
|
import asyncio
|
|
30
30
|
import contextlib
|
|
31
|
+
import re
|
|
31
32
|
import threading
|
|
32
33
|
from collections.abc import Iterator, Sequence
|
|
33
34
|
from uuid import uuid4
|
|
@@ -38,6 +39,7 @@ import tenacity
|
|
|
38
39
|
from sqlglot import exp
|
|
39
40
|
|
|
40
41
|
from interlace.engines.base import EngineAdapter, EngineCaps, LoadMode
|
|
42
|
+
from interlace.exceptions import ConfigurationError
|
|
41
43
|
from interlace.ir.relation import TableRef
|
|
42
44
|
|
|
43
45
|
_DUCKDB_CAPS = EngineCaps(
|
|
@@ -58,6 +60,39 @@ _commit_retry = tenacity.retry(
|
|
|
58
60
|
)
|
|
59
61
|
|
|
60
62
|
|
|
63
|
+
@contextlib.contextmanager
|
|
64
|
+
def _clean_lock_error(database: str) -> Iterator[None]:
|
|
65
|
+
"""Translate DuckDB's file-lock conflict into one actionable line.
|
|
66
|
+
|
|
67
|
+
A DuckDB/DuckLake database is held by a single process. The common way to hit
|
|
68
|
+
that is running a CLI command (`interlace query`, `plan`, `apply`) while
|
|
69
|
+
`interlace serve` is up — an obvious thing to do, since serve is the daemon
|
|
70
|
+
and query is the console's CLI counterpart. Raw, that surfaces as a dozen
|
|
71
|
+
frames ending in `duckdb.IOException`, which names neither the cause nor the
|
|
72
|
+
fix. The fix is `--quack`, and it is already documented; the error just never
|
|
73
|
+
said so.
|
|
74
|
+
"""
|
|
75
|
+
try:
|
|
76
|
+
yield
|
|
77
|
+
except duckdb.IOException as exc:
|
|
78
|
+
message = str(exc)
|
|
79
|
+
# "Conflicting lock is held in <exe> (PID n)" is the Linux rendering — DuckDB
|
|
80
|
+
# names the holder from /proc/locks. Elsewhere the message is the bare "Could
|
|
81
|
+
# not set lock on file", and a same-process conflict says "already held", so
|
|
82
|
+
# match all three or macOS/Windows keep the raw traceback.
|
|
83
|
+
if not any(marker in message for marker in ("Conflicting lock", "Could not set lock", "already held")):
|
|
84
|
+
raise
|
|
85
|
+
holder = re.search(r"\(PID (\d+)\)", message)
|
|
86
|
+
held_by = f" (PID {holder.group(1)})" if holder else ""
|
|
87
|
+
raise ConfigurationError(
|
|
88
|
+
f"the warehouse {database!r} is already open in another process{held_by}. "
|
|
89
|
+
"DuckDB allows one process at a time — stop `interlace serve`, or serve the "
|
|
90
|
+
"warehouse over the quack protocol (`interlace serve --quack quack:localhost:4213`) "
|
|
91
|
+
"and point this process at `database: quack:localhost:4213` to share it.",
|
|
92
|
+
details={"database": database},
|
|
93
|
+
) from None
|
|
94
|
+
|
|
95
|
+
|
|
61
96
|
def _affected(cur: duckdb.DuckDBPyConnection) -> int:
|
|
62
97
|
"""DML/CTAS/COPY return their affected-row count as a one-cell result; DDL returns
|
|
63
98
|
nothing. Never raises — row stats are best-effort decoration, not correctness."""
|
|
@@ -108,7 +143,9 @@ class DuckDBAdapter(EngineAdapter):
|
|
|
108
143
|
|
|
109
144
|
@classmethod
|
|
110
145
|
def connect(cls, path: str) -> DuckDBAdapter:
|
|
111
|
-
|
|
146
|
+
with _clean_lock_error(path):
|
|
147
|
+
conn = duckdb.connect(path)
|
|
148
|
+
return cls(conn, serialise_writes=path.startswith("ducklake:"))
|
|
112
149
|
|
|
113
150
|
@classmethod
|
|
114
151
|
def connect_ducklake(
|
|
@@ -139,7 +176,8 @@ class DuckDBAdapter(EngineAdapter):
|
|
|
139
176
|
options_sql = f" ({', '.join(options)})" if options else ""
|
|
140
177
|
escaped = catalog.replace("'", "''")
|
|
141
178
|
alias_sql = exp.to_identifier(alias).sql("duckdb")
|
|
142
|
-
|
|
179
|
+
with _clean_lock_error(catalog):
|
|
180
|
+
conn.execute(f"ATTACH IF NOT EXISTS '{escaped}' AS {alias_sql}{options_sql}")
|
|
143
181
|
conn.execute(f"USE {alias_sql}")
|
|
144
182
|
# LOAD, secrets, and ATTACH are all instance-wide — they carry into every
|
|
145
183
|
# cursor and must run ONCE (re-running CREATE OR REPLACE SECRET per cursor
|
|
@@ -162,10 +200,23 @@ class DuckDBAdapter(EngineAdapter):
|
|
|
162
200
|
with contextlib.suppress(Exception):
|
|
163
201
|
self._conn.interrupt()
|
|
164
202
|
|
|
203
|
+
def search_files_from(self, directory: str) -> None:
|
|
204
|
+
"""Resolve relative read paths (``read_csv_auto('seeds/x.csv')``) against
|
|
205
|
+
``directory`` — the project root — as well as the process CWD.
|
|
206
|
+
|
|
207
|
+
Additive: a CWD-relative path still resolves, so this only ever widens what a
|
|
208
|
+
model can find. GLOBAL scope because a plain ``SET`` is session-scoped and would
|
|
209
|
+
not reach the per-task cursors that actually run the queries. Reads only —
|
|
210
|
+
``COPY`` targets stay CWD-relative, which is why exports resolve their own paths
|
|
211
|
+
against the root (``plan.apply._resolve_export_path``)."""
|
|
212
|
+
escaped = directory.replace("'", "''")
|
|
213
|
+
self._conn.execute(f"SET GLOBAL file_search_path='{escaped}'")
|
|
214
|
+
|
|
165
215
|
def attach(self, alias: str, uri: str) -> None:
|
|
166
216
|
"""ATTACH another database (duckdb/sqlite/postgres/... URI) under ``alias``."""
|
|
167
217
|
escaped = uri.replace("'", "''")
|
|
168
|
-
|
|
218
|
+
with _clean_lock_error(uri): # attaching a held duckdb/ducklake file conflicts just like opening one
|
|
219
|
+
self._conn.execute(f"ATTACH IF NOT EXISTS '{escaped}' AS {exp.to_identifier(alias).sql('duckdb')}")
|
|
169
220
|
self._attached.append(alias)
|
|
170
221
|
if uri.startswith("ducklake:"): # writes may now reach a DuckLake catalog (e.g. table sinks)
|
|
171
222
|
if isinstance(self._write_lock, contextlib.nullcontext):
|
|
@@ -14,6 +14,7 @@ from __future__ import annotations
|
|
|
14
14
|
import asyncio
|
|
15
15
|
import contextlib
|
|
16
16
|
import logging
|
|
17
|
+
import re
|
|
17
18
|
import time
|
|
18
19
|
from collections.abc import Callable, Mapping
|
|
19
20
|
from dataclasses import dataclass, field, replace
|
|
@@ -107,9 +108,9 @@ async def _merge_python_output(
|
|
|
107
108
|
pre_statements: list[exp.Expression] = []
|
|
108
109
|
columns: list[str] | None = None
|
|
109
110
|
if exists:
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
)
|
|
111
|
+
alignment = await _align_stage_to_target(engine, stage, target, strategy)
|
|
112
|
+
pre_statements = alignment.pre_statements
|
|
113
|
+
source, columns = alignment.for_strategy(strategy)
|
|
113
114
|
elif isinstance(strategy, HashMerge):
|
|
114
115
|
# hash_merge builds its _hash from the column list; on the first build the target
|
|
115
116
|
# doesn't exist yet (so no align pass ran), so take the columns from the staged output
|
|
@@ -127,21 +128,54 @@ async def _merge_python_output(
|
|
|
127
128
|
return written, interval
|
|
128
129
|
|
|
129
130
|
|
|
131
|
+
@dataclass(frozen=True)
|
|
132
|
+
class Alignment:
|
|
133
|
+
"""A staged source fitted to an existing target, in both widths.
|
|
134
|
+
|
|
135
|
+
``source``/``columns`` cover the target's full column set — what a strategy that
|
|
136
|
+
binds positionally or compares whole rows (``INSERT ... SELECT *``, ``EXCEPT``)
|
|
137
|
+
needs, with columns the model doesn't produce NULL-filled in.
|
|
138
|
+
``produced_source``/``produced`` cover only the model's own columns; a strategy
|
|
139
|
+
that names its columns takes these instead, so the rest of the target is left
|
|
140
|
+
alone on an update and takes its DEFAULTs on an insert. ``unproduced`` names the
|
|
141
|
+
difference — target columns this model has no value for."""
|
|
142
|
+
|
|
143
|
+
pre_statements: list[exp.Expression]
|
|
144
|
+
source: exp.Query
|
|
145
|
+
columns: list[str]
|
|
146
|
+
produced_source: exp.Query
|
|
147
|
+
produced: list[str]
|
|
148
|
+
unproduced: list[str]
|
|
149
|
+
|
|
150
|
+
def for_strategy(self, strategy: Strategy) -> tuple[exp.Query, list[str]]:
|
|
151
|
+
"""The (source, column list) pair this strategy should be planned against."""
|
|
152
|
+
if strategy.writes_named_columns:
|
|
153
|
+
return self.produced_source, self.produced
|
|
154
|
+
return self.source, self.columns
|
|
155
|
+
|
|
156
|
+
|
|
130
157
|
async def _align_stage_to_target(
|
|
131
|
-
engine: EngineAdapter, stage: TableRef, target: TableRef,
|
|
132
|
-
) ->
|
|
158
|
+
engine: EngineAdapter, stage: TableRef, target: TableRef, strategy: Strategy
|
|
159
|
+
) -> Alignment:
|
|
133
160
|
"""Align a staged source to an EXISTING target: additive ALTERs for new columns,
|
|
134
|
-
widening type promotions in place, and
|
|
135
|
-
target's final column set (NULL-fill vanished columns, cast type drift).
|
|
136
|
-
(pre-statements, aligned source select, target column order).
|
|
161
|
+
widening type promotions in place, and projections over the stage fitted to the
|
|
162
|
+
target's final column set (NULL-fill vanished columns, cast type drift).
|
|
137
163
|
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
164
|
+
The strategy's managed bookkeeping columns (e.g. scd's validity pair, hash_merge's
|
|
165
|
+
``_hash``) live on the target but never in the model's output, so they are held out
|
|
166
|
+
of both projections — the strategy owns them, and must find them already there."""
|
|
141
167
|
target_columns = await engine.describe(target)
|
|
168
|
+
_require_managed_columns(strategy, target, target_columns)
|
|
169
|
+
exclude = strategy.managed_columns
|
|
142
170
|
for column in exclude:
|
|
143
171
|
target_columns.pop(column, None)
|
|
144
172
|
stage_columns = await engine.describe(stage)
|
|
173
|
+
if clash := [c for c in stage_columns if c in exclude]:
|
|
174
|
+
raise PlanError(
|
|
175
|
+
f"model output column {clash[0]!r} collides with a column the "
|
|
176
|
+
f"{_strategy_name(strategy)} strategy manages — rename it in the model",
|
|
177
|
+
details={"target": target.to_expr().sql(), "columns": clash},
|
|
178
|
+
)
|
|
145
179
|
target_expr = target.to_expr()
|
|
146
180
|
pre_statements: list[exp.Expression] = []
|
|
147
181
|
for column, dtype in stage_columns.items():
|
|
@@ -170,16 +204,60 @@ async def _align_stage_to_target(
|
|
|
170
204
|
# Any remaining type mismatch (e.g. a numeric field arriving as VARCHAR) is cast
|
|
171
205
|
# to the target's type — deterministic, and loudly fails the run on values that
|
|
172
206
|
# genuinely don't convert rather than silently corrupting the column.
|
|
173
|
-
projection = []
|
|
207
|
+
projection: list[exp.Expression] = []
|
|
208
|
+
produced_projection: list[exp.Expression] = []
|
|
209
|
+
produced: list[str] = []
|
|
210
|
+
unproduced: list[str] = []
|
|
174
211
|
for column, dtype in target_columns.items():
|
|
175
212
|
if column not in stage_columns:
|
|
176
213
|
projection.append(exp.alias_(exp.Cast(this=exp.Null(), to=exp.DataType.build(dtype)), column))
|
|
177
|
-
|
|
178
|
-
|
|
214
|
+
unproduced.append(column)
|
|
215
|
+
continue
|
|
216
|
+
if stage_columns[column] != dtype:
|
|
217
|
+
fitted: exp.Expression = exp.alias_(exp.Cast(this=exp.column(column), to=exp.DataType.build(dtype)), column)
|
|
179
218
|
else:
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
219
|
+
fitted = exp.column(column)
|
|
220
|
+
projection.append(fitted)
|
|
221
|
+
produced_projection.append(fitted.copy())
|
|
222
|
+
produced.append(column)
|
|
223
|
+
return Alignment(
|
|
224
|
+
pre_statements=pre_statements,
|
|
225
|
+
source=exp.select(*projection).from_(stage.to_expr()),
|
|
226
|
+
columns=list(target_columns),
|
|
227
|
+
produced_source=exp.select(*produced_projection).from_(stage.to_expr()),
|
|
228
|
+
produced=produced,
|
|
229
|
+
unproduced=unproduced,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _strategy_name(strategy: Strategy) -> str:
|
|
234
|
+
"""The strategy's config keyword (``HashMerge`` -> ``hash_merge``), for messages."""
|
|
235
|
+
return re.sub(r"(?<!^)(?=[A-Z])", "_", type(strategy).__name__).lower()
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _require_managed_columns(strategy: Strategy, target: TableRef, target_columns: Mapping[str, str]) -> None:
|
|
239
|
+
"""A strategy that keeps bookkeeping columns can only take over a table that already
|
|
240
|
+
carries them. Missing ones would otherwise surface as an engine binder error deep in
|
|
241
|
+
the strategy's UPDATE — most likely an externally-owned table that interlace did not
|
|
242
|
+
create, or a strategy swapped under an existing delivery target.
|
|
243
|
+
|
|
244
|
+
Nothing to say when the target describes empty: it isn't really there (a snapshot row
|
|
245
|
+
can outlive its table), so the caller is not taking over anything. Matching is
|
|
246
|
+
case-insensitive — an engine that folds unquoted identifiers (Snowflake, BigQuery)
|
|
247
|
+
stores ``_hash`` as ``_HASH`` and would otherwise trip this on every run."""
|
|
248
|
+
if not target_columns:
|
|
249
|
+
return
|
|
250
|
+
present = {c.casefold() for c in target_columns}
|
|
251
|
+
missing = [c for c in strategy.managed_columns if c.casefold() not in present]
|
|
252
|
+
if missing:
|
|
253
|
+
name = _strategy_name(strategy)
|
|
254
|
+
qualified = target.to_expr().sql()
|
|
255
|
+
raise PlanError(
|
|
256
|
+
f"{name} needs its bookkeeping column(s) {', '.join(missing)} on {qualified}, which already "
|
|
257
|
+
f"exists without them — interlace did not create this table. Use strategy: merge, or add "
|
|
258
|
+
f"the column(s) to it.",
|
|
259
|
+
details={"target": qualified, "missing": missing, "strategy": name},
|
|
260
|
+
)
|
|
183
261
|
|
|
184
262
|
|
|
185
263
|
async def _deliver_table(
|
|
@@ -194,10 +272,11 @@ async def _deliver_table(
|
|
|
194
272
|
|
|
195
273
|
The external target is never dropped (grants and readers survive). When it already
|
|
196
274
|
exists the source is staged in the warehouse and aligned to the target (additive
|
|
197
|
-
ALTERs, widening,
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
275
|
+
ALTERs, widening, casts) so a model that grows or reorders columns evolves the
|
|
276
|
+
destination instead of breaking it or positionally corrupting it. A strategy that
|
|
277
|
+
names its columns is then planned against the model's own columns alone, leaving the
|
|
278
|
+
rest of the target untouched (see :class:`Alignment`); a whole-row one gets the source
|
|
279
|
+
widened to the target with NULLs, and binds positionally in the target's order.
|
|
201
280
|
|
|
202
281
|
Two cases skip staging and run the strategy directly against the target: the first
|
|
203
282
|
delivery (the ensure-create matches the source), and any windowed ``incremental``
|
|
@@ -213,9 +292,23 @@ async def _deliver_table(
|
|
|
213
292
|
stage = TableRef(schema=XFER_SCHEMA, name=f"{model.name}__sink_stage")
|
|
214
293
|
await engine.create_schema(stage.schema)
|
|
215
294
|
await engine.execute(exp.Create(this=stage.to_expr(), kind="TABLE", replace=True, expression=resolved.copy()))
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
)
|
|
295
|
+
alignment = await _align_stage_to_target(engine, stage, target, strategy)
|
|
296
|
+
pre_statements = alignment.pre_statements
|
|
297
|
+
aligned, columns = alignment.for_strategy(strategy)
|
|
298
|
+
if alignment.unproduced and not strategy.writes_named_columns:
|
|
299
|
+
# A whole-row strategy rewrites rows entire, so any column the model doesn't
|
|
300
|
+
# produce is reset on every run — and the deleting ones (replace, full_merge)
|
|
301
|
+
# drop rows it doesn't supply at all. Fine when interlace owns the table (this
|
|
302
|
+
# is just a widened source); destructive when it shares it with another writer.
|
|
303
|
+
logger.warning(
|
|
304
|
+
"%s: strategy %s writes whole rows into %s — it resets columns this model does not "
|
|
305
|
+
"produce (%s) on every delivery, and (replace, full_merge) deletes rows it does not "
|
|
306
|
+
"supply. Use merge, hash_merge or append if another writer owns part of this table.",
|
|
307
|
+
model.name,
|
|
308
|
+
model.strategy,
|
|
309
|
+
target.to_expr().sql(),
|
|
310
|
+
", ".join(alignment.unproduced),
|
|
311
|
+
)
|
|
219
312
|
statements = strategy.plan_statements(SqlRelation(ast=aligned), target, engine.caps, interval, columns)
|
|
220
313
|
# One transaction may write only ONE attached database: the delivery batch writes the
|
|
221
314
|
# external target; the stage lives in the warehouse and is dropped separately (a
|
|
@@ -182,7 +182,7 @@ class Project:
|
|
|
182
182
|
if not database.startswith("md:"):
|
|
183
183
|
database = f"md:{database}"
|
|
184
184
|
return DuckDBAdapter.connect(database)
|
|
185
|
-
database = cfg.database or "
|
|
185
|
+
database = cfg.database or ".interlace/warehouse.duckdb"
|
|
186
186
|
if cfg.type == "quack" or database.startswith("quack:"):
|
|
187
187
|
from interlace.engines.quack import QuackAdapter
|
|
188
188
|
|
|
@@ -209,6 +209,11 @@ class Project:
|
|
|
209
209
|
database = str(path)
|
|
210
210
|
engine = DuckDBAdapter.connect(database)
|
|
211
211
|
|
|
212
|
+
# A model's relative read path (`read_csv_auto('seeds/x.csv')`) is documented to
|
|
213
|
+
# resolve against the project root. DuckDB resolves against the process CWD, which
|
|
214
|
+
# is only the same thing when you happen to run from the root — not under
|
|
215
|
+
# `--path`, `interlace serve`, or the scheduler.
|
|
216
|
+
engine.search_files_from(str(self.root))
|
|
212
217
|
for alias, uri in cfg.attach.items():
|
|
213
218
|
target = uri
|
|
214
219
|
if uri.startswith(("postgres:", "postgres://", "postgresql://")):
|
|
@@ -72,10 +72,27 @@ def scaffold_project(root: Path, name: str | None = None, template: str = DEFAUL
|
|
|
72
72
|
|
|
73
73
|
written: list[Path] = []
|
|
74
74
|
for item in sorted(source.rglob("*")):
|
|
75
|
-
if item.name == _META_FILE or not item.is_file():
|
|
75
|
+
if item.name == _META_FILE or not item.is_file() or _is_install_artefact(item):
|
|
76
76
|
continue
|
|
77
77
|
target = root / item.relative_to(source)
|
|
78
78
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
79
|
-
|
|
79
|
+
try:
|
|
80
|
+
text = item.read_text(encoding="utf-8")
|
|
81
|
+
except UnicodeDecodeError:
|
|
82
|
+
# A template may legitimately carry a binary fixture (parquet, sqlite).
|
|
83
|
+
target.write_bytes(item.read_bytes())
|
|
84
|
+
else:
|
|
85
|
+
target.write_text(text.replace(_NAME_TOKEN, project_name), encoding="utf-8")
|
|
80
86
|
written.append(target)
|
|
81
87
|
return written
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _is_install_artefact(item: Path) -> bool:
|
|
91
|
+
"""Templates are real projects, so they contain ``.py`` model files — and pip
|
|
92
|
+
byte-compiles every ``.py`` in the wheel at install time. That leaves
|
|
93
|
+
``__pycache__/*.pyc`` sitting inside the *installed* template tree, which is an
|
|
94
|
+
artefact of the install and never part of the template. Copying one into a new
|
|
95
|
+
project is wrong, and reading it as text raised ``UnicodeDecodeError`` on the
|
|
96
|
+
first command a pip user ever ran. (uv does not byte-compile by default, which
|
|
97
|
+
is why this only reproduced via pip.)"""
|
|
98
|
+
return "__pycache__" in item.parts or item.suffix in {".pyc", ".pyo"}
|