interlaced 2.1.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interlaced-2.1.0/src/interlaced.egg-info → interlaced-2.3.0}/PKG-INFO +7 -6
- {interlaced-2.1.0 → interlaced-2.3.0}/README.md +3 -2
- {interlaced-2.1.0 → interlaced-2.3.0}/pyproject.toml +4 -4
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/cli/main.py +5 -4
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/dsl/decorators.py +4 -4
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/dsl/discovery.py +1 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/spark.py +1 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/graph/project.py +2 -2
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/ir/canonicalize.py +4 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/apply.py +60 -11
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/differ.py +19 -4
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/plan.py +21 -4
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/run.py +1 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/app.py +132 -22
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/app.css +39 -0
- interlaced-2.3.0/src/interlace/service/ui/favicon.svg +28 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/index.html +3 -3
- interlaced-2.3.0/src/interlace/service/ui/js/timeline.js +157 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/ui.js +4 -3
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/overview.js +34 -15
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/query.js +8 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/runs.js +116 -83
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/state/janitor.py +17 -3
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/state/store.py +53 -13
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/__init__.py +16 -4
- interlaced-2.3.0/src/interlace/strategies/hash_merge.py +119 -0
- interlaced-2.3.0/src/interlace/strategies/incremental.py +89 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/template.yaml +2 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/github/template.yaml +2 -1
- interlaced-2.3.0/src/interlace/templates/postgres/template.yaml +4 -0
- {interlaced-2.1.0 → interlaced-2.3.0/src/interlaced.egg-info}/PKG-INFO +7 -6
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlaced.egg-info/SOURCES.txt +3 -1
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlaced.egg-info/requires.txt +3 -3
- interlaced-2.1.0/src/interlace/service/ui/favicon.svg +0 -1
- interlaced-2.1.0/src/interlace/strategies/incremental_by_time.py +0 -64
- interlaced-2.1.0/src/interlace/templates/postgres/template.yaml +0 -3
- {interlaced-2.1.0 → interlaced-2.3.0}/LICENSE +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/MANIFEST.in +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/setup.cfg +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/checks/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/checks/builtin.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/checks/runner.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/checks/spec.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/cli/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/config/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/config/config.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/contracts.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/dsl/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/dsl/sql_config.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/adbc.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/base.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/bigquery.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/duckdb.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/postgres.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/quack.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/redshift.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/registry.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/engines/snowflake.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/exceptions.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/graph/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/graph/column_lineage.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/graph/dag.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/graph/selectors.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/ir/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/ir/fingerprint.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/ir/relation.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/plan/resolve.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/project.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/py.typed +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/query.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/runtime/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/runtime/handles.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/runtime/python_model.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/scaffold.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/scheduler/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/scheduler/engine.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/scheduler/triggers.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/scheduler/worker.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/auth.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/api.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/app.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/dag.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/checks.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/environments.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/lineage.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/models.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/plan.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/streams.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/system.js +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/sinks.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/sources/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/sources/auth.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/sources/rest.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/state/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/state/interval.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/state/snapshot.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/append.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/base.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/full_merge.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/merge.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/replace.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/replace_in_place.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/scd.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/strategies/view.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/streaming/__init__.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/streaming/log.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/streaming/materializer.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/streaming/schema.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/README.md +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/generate.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/interlace.yaml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events_by_minute.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events_by_type.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/models/top_users.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/events/models/user_spend.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/github/README.md +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/github/interlace.yaml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/github/models/github_issues.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/github/models/issues_by_state.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/README.md +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/docker-compose.yml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/init/seed.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/interlace.yaml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/models/orders.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/postgres/models/orders_by_status.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/README.md +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/interlace.yaml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/enriched_events.py +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/event_summary.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/raw_events.sql +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/template.yaml +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlaced.egg-info/dependency_links.txt +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlaced.egg-info/entry_points.txt +0 -0
- {interlaced-2.1.0 → interlaced-2.3.0}/src/interlaced.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: interlaced
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion
|
|
5
5
|
Author-email: Mark <mark@interlace.sh>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -18,12 +18,12 @@ Classifier: Programming Language :: SQL
|
|
|
18
18
|
Requires-Python: >=3.12
|
|
19
19
|
Description-Content-Type: text/markdown
|
|
20
20
|
License-File: LICENSE
|
|
21
|
-
Requires-Dist: sqlglot<
|
|
21
|
+
Requires-Dist: sqlglot<30.0,>=25.0
|
|
22
22
|
Requires-Dist: duckdb>=1.5.3
|
|
23
23
|
Requires-Dist: pyarrow>=17.0
|
|
24
24
|
Requires-Dist: pydantic<3.0,>=2.5
|
|
25
25
|
Requires-Dist: typer<1.0,>=0.12
|
|
26
|
-
Requires-Dist: rich<
|
|
26
|
+
Requires-Dist: rich<16.0,>=13.0
|
|
27
27
|
Requires-Dist: cronsim<3.0,>=2.5
|
|
28
28
|
Requires-Dist: tenacity<10.0,>=8.2
|
|
29
29
|
Requires-Dist: pyyaml<7.0,>=6.0
|
|
@@ -59,7 +59,7 @@ Requires-Dist: httpx<1.0,>=0.27; extra == "dev"
|
|
|
59
59
|
Requires-Dist: pytest-asyncio<2.0,>=1.0; extra == "dev"
|
|
60
60
|
Requires-Dist: ruff<1.0,>=0.6; extra == "dev"
|
|
61
61
|
Requires-Dist: black<27.0,>=24.0; extra == "dev"
|
|
62
|
-
Requires-Dist: mypy<
|
|
62
|
+
Requires-Dist: mypy<3.0,>=1.11; extra == "dev"
|
|
63
63
|
Dynamic: license-file
|
|
64
64
|
|
|
65
65
|
# interlace
|
|
@@ -127,8 +127,9 @@ def orders(cursor, this):
|
|
|
127
127
|
```
|
|
128
128
|
|
|
129
129
|
**Strategies:** `replace`, `view`, `ephemeral` (CTE-inlined), `merge` (upsert),
|
|
130
|
-
`full_merge` (full-state source applied as a minimal diff), `
|
|
131
|
-
|
|
130
|
+
`full_merge` (full-state source applied as a minimal diff), `incremental` (one time window
|
|
131
|
+
at a time — rewrites the window, or upserts within it if you give it a `key`), `scd`
|
|
132
|
+
(history with validity windows).
|
|
132
133
|
|
|
133
134
|
## Plan / apply
|
|
134
135
|
|
|
@@ -63,8 +63,9 @@ def orders(cursor, this):
|
|
|
63
63
|
```
|
|
64
64
|
|
|
65
65
|
**Strategies:** `replace`, `view`, `ephemeral` (CTE-inlined), `merge` (upsert),
|
|
66
|
-
`full_merge` (full-state source applied as a minimal diff), `
|
|
67
|
-
|
|
66
|
+
`full_merge` (full-state source applied as a minimal diff), `incremental` (one time window
|
|
67
|
+
at a time — rewrites the window, or upserts within it if you give it a `key`), `scd`
|
|
68
|
+
(history with validity windows).
|
|
68
69
|
|
|
69
70
|
## Plan / apply
|
|
70
71
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "interlaced"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.3.0"
|
|
8
8
|
description = "Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.12"
|
|
@@ -24,12 +24,12 @@ classifiers = [
|
|
|
24
24
|
# Core: the IR + engine + state + plan spine (Phase 1). Everything is Arrow- and
|
|
25
25
|
# sqlglot-native. See docs/architecture/architecture.md.
|
|
26
26
|
dependencies = [
|
|
27
|
-
"sqlglot>=25.0,<
|
|
27
|
+
"sqlglot>=25.0,<30.0", # canonical IR, transpilation, semantic diff, column lineage
|
|
28
28
|
"duckdb>=1.5.3", # default engine + federation hub; DuckLake storage + quack serving
|
|
29
29
|
"pyarrow>=17.0", # the wire format: RecordBatchReader everywhere
|
|
30
30
|
"pydantic>=2.5,<3.0", # config + manifest validation (cold paths only)
|
|
31
31
|
"typer>=0.12,<1.0", # CLI
|
|
32
|
-
"rich>=13.0,<
|
|
32
|
+
"rich>=13.0,<16.0", # display, strictly an event subscriber
|
|
33
33
|
"cronsim>=2.5,<3.0", # cron parsing for the trigger engine
|
|
34
34
|
"tenacity>=8.2,<10.0", # retries (tasks, DuckLake commit conflicts, transfers)
|
|
35
35
|
"pyyaml>=6.0,<7.0", # project config (config + env overlays)
|
|
@@ -78,7 +78,7 @@ dev = [
|
|
|
78
78
|
"pytest-asyncio>=1.0,<2.0",
|
|
79
79
|
"ruff>=0.6,<1.0",
|
|
80
80
|
"black>=24.0,<27.0",
|
|
81
|
-
"mypy>=1.11,<
|
|
81
|
+
"mypy>=1.11,<3.0",
|
|
82
82
|
]
|
|
83
83
|
|
|
84
84
|
[tool.setuptools.packages.find]
|
|
@@ -107,7 +107,7 @@ _END = typer.Option("", "--end", help="Window end (ISO), for incremental models.
|
|
|
107
107
|
_FORWARD_ONLY = typer.Option(
|
|
108
108
|
False,
|
|
109
109
|
"--forward-only",
|
|
110
|
-
help="Modified history-keeping models (merge/full_merge/scd/
|
|
110
|
+
help="Modified history-keeping models (merge/full_merge/scd/incremental) carry their history "
|
|
111
111
|
"forward: it is copied to the new version, the new logic applies to the copy, and checks gate "
|
|
112
112
|
"before views move. Requires a shape-compatible change.",
|
|
113
113
|
)
|
|
@@ -205,7 +205,7 @@ async def _render_empty_incrementals(result: ApplyResult, compiled: CompiledProj
|
|
|
205
205
|
|
|
206
206
|
for name in result.built:
|
|
207
207
|
model = compiled.models[name]
|
|
208
|
-
if model.strategy != "
|
|
208
|
+
if model.strategy != "incremental" or model.is_terminal:
|
|
209
209
|
continue
|
|
210
210
|
counts = result.rows.get(name)
|
|
211
211
|
if counts is not None and (counts.inserted or counts.updated):
|
|
@@ -252,7 +252,8 @@ def init(
|
|
|
252
252
|
table.add_column("Description", style="dim")
|
|
253
253
|
table.add_column("Needs", style="dim")
|
|
254
254
|
for info in list_templates():
|
|
255
|
-
|
|
255
|
+
# escape: a description may contain [sources]-style brackets Rich would eat as markup
|
|
256
|
+
table.add_row(info.name, escape(info.description), ", ".join(info.requires_env) or "—")
|
|
256
257
|
console.print(table)
|
|
257
258
|
return
|
|
258
259
|
try:
|
|
@@ -403,7 +404,7 @@ def run(
|
|
|
403
404
|
) -> None:
|
|
404
405
|
"""Force-build models and promote, ignoring change detection.
|
|
405
406
|
|
|
406
|
-
For
|
|
407
|
+
For incremental models, --start/--end set the catchup window
|
|
407
408
|
(default: the latest grain interval).
|
|
408
409
|
"""
|
|
409
410
|
asyncio.run(_execute(environment, path, select, start, end, restate=False, parallelism=parallelism))
|
|
@@ -22,7 +22,7 @@ ModelFn = Callable[..., Any]
|
|
|
22
22
|
# (a fingerprinted snapshot read through an environment view); `table`/`file` are
|
|
23
23
|
# terminal deliveries into a destination interlace does not own.
|
|
24
24
|
_MATERIALISATIONS = frozenset({"virtual", "view", "ephemeral", "table", "file"})
|
|
25
|
-
_KEYED_STRATEGIES = frozenset({"merge", "full_merge", "scd"})
|
|
25
|
+
_KEYED_STRATEGIES = frozenset({"merge", "full_merge", "hash_merge", "scd"})
|
|
26
26
|
_DRIFT_MODES = frozenset({"evolve", "reject", "quarantine"})
|
|
27
27
|
|
|
28
28
|
|
|
@@ -94,9 +94,9 @@ class ModelDef:
|
|
|
94
94
|
dialect: str | None = None
|
|
95
95
|
engine: str | None = None # named engine from config (None → project default_engine)
|
|
96
96
|
depends_on: tuple[str, ...] = ()
|
|
97
|
-
interval: str | None = None # grain for
|
|
98
|
-
time_column: str | None = None # partition column for
|
|
99
|
-
# First-build window for
|
|
97
|
+
interval: str | None = None # grain for incremental (e.g. "1d")
|
|
98
|
+
time_column: str | None = None # partition column for incremental
|
|
99
|
+
# First-build window for incremental: "auto" derives [min, max] of the
|
|
100
100
|
# time column from the source at apply time and fills it as ONE interval;
|
|
101
101
|
# "none" keeps only the latest grain window; an ISO date pins the start.
|
|
102
102
|
backfill: str = "auto"
|
|
@@ -64,7 +64,7 @@ def _sql_model(default_name: str, sql: str, config: dict[str, Any], default_dial
|
|
|
64
64
|
depends_on=_as_tuple(config.get("depends_on") or ()),
|
|
65
65
|
interval=config.get("interval"),
|
|
66
66
|
time_column=config.get("time_column"),
|
|
67
|
-
backfill=config.get("backfill", "auto"), # first-build window for
|
|
67
|
+
backfill=config.get("backfill", "auto"), # first-build window for incremental
|
|
68
68
|
tags=_as_tuple(config.get("tags") or ()),
|
|
69
69
|
owner=config.get("owner"),
|
|
70
70
|
description=config.get("description"),
|
|
@@ -7,7 +7,7 @@ come back as Arrow via ``DataFrame.toArrow()``, and Arrow loads go in through
|
|
|
7
7
|
(``local[*]``, for tests) or a remote one (Spark Connect / a shared session).
|
|
8
8
|
|
|
9
9
|
**Strategy support.** ``replace``, ``append`` and ``view`` run on any Spark
|
|
10
|
-
catalog. ``merge`` (native ``MERGE``) and ``
|
|
10
|
+
catalog. ``merge`` (native ``MERGE``) and ``incremental`` (windowed
|
|
11
11
|
``DELETE`` by literal predicate + ``INSERT``) need a catalog with row-level
|
|
12
12
|
mutations — Delta Lake or Iceberg — configured on the session you hand the
|
|
13
13
|
adapter (the tests use a Delta-backed local session). ``scd`` and ``full_merge``
|
|
@@ -44,9 +44,9 @@ class CompiledModel:
|
|
|
44
44
|
materialise: str
|
|
45
45
|
strategy: str
|
|
46
46
|
key: tuple[str, ...] # business key for keyed strategies (merge)
|
|
47
|
-
time_column: str | None # partition column for
|
|
47
|
+
time_column: str | None # partition column for incremental
|
|
48
48
|
cursor: str | None # column whose max is injected into a Python model's `cursor` param
|
|
49
|
-
interval: str | None # grain for
|
|
49
|
+
interval: str | None # grain for incremental (e.g. "1d")
|
|
50
50
|
tags: tuple[str, ...] # for tag: selection
|
|
51
51
|
schedule: dict[str, str] | None # cron/interval schedule for the trigger engine
|
|
52
52
|
columns: dict[str, str | None] | None # output contract validated at apply time
|
|
@@ -8,6 +8,8 @@ pruning and the ``impact`` command).
|
|
|
8
8
|
|
|
9
9
|
from __future__ import annotations
|
|
10
10
|
|
|
11
|
+
from typing import cast
|
|
12
|
+
|
|
11
13
|
import sqlglot
|
|
12
14
|
from sqlglot import exp
|
|
13
15
|
|
|
@@ -86,4 +88,5 @@ def resolve_references(ast: exp.Expression, mapping: dict[str, TableRef]) -> exp
|
|
|
86
88
|
node.set("catalog", exp.to_identifier(target.catalog) if target.catalog else None)
|
|
87
89
|
return node
|
|
88
90
|
|
|
89
|
-
|
|
91
|
+
# sqlglot 29 loosened transform()'s return annotation; it is an Expression.
|
|
92
|
+
return cast("exp.Expression", ast.transform(rewrite))
|
|
@@ -36,8 +36,9 @@ from interlace.runtime.python_model import build_python_model, run_python_model
|
|
|
36
36
|
from interlace.sinks import file_statements, target_ref
|
|
37
37
|
from interlace.state.interval import Interval
|
|
38
38
|
from interlace.state.store import StateStore
|
|
39
|
-
from interlace.strategies import Strategy, resolve_strategy
|
|
39
|
+
from interlace.strategies import Incremental, Strategy, resolve_strategy
|
|
40
40
|
from interlace.strategies.base import RowCounts
|
|
41
|
+
from interlace.strategies.hash_merge import HashMerge
|
|
41
42
|
|
|
42
43
|
|
|
43
44
|
@dataclass
|
|
@@ -86,7 +87,9 @@ async def _merge_python_output(
|
|
|
86
87
|
reader: pa.RecordBatchReader,
|
|
87
88
|
*,
|
|
88
89
|
exists: bool,
|
|
89
|
-
|
|
90
|
+
interval: Interval | None = None,
|
|
91
|
+
bootstrap: bool = False,
|
|
92
|
+
) -> tuple[RowCounts, Interval | None]:
|
|
90
93
|
"""Stage a Python model's Arrow output and apply its keyed strategy in SQL.
|
|
91
94
|
|
|
92
95
|
The output lands in a stage table (CREATE OR REPLACE, so a crashed run's
|
|
@@ -107,12 +110,21 @@ async def _merge_python_output(
|
|
|
107
110
|
pre_statements, source, columns = await _align_stage_to_target(
|
|
108
111
|
engine, stage, target, exclude=strategy.managed_columns
|
|
109
112
|
)
|
|
113
|
+
elif isinstance(strategy, HashMerge):
|
|
114
|
+
# hash_merge builds its _hash from the column list; on the first build the target
|
|
115
|
+
# doesn't exist yet (so no align pass ran), so take the columns from the staged output
|
|
116
|
+
columns = [c for c in await engine.describe(stage) if c not in strategy.managed_columns]
|
|
110
117
|
|
|
111
118
|
relation = SqlRelation(ast=source)
|
|
112
|
-
|
|
119
|
+
if isinstance(strategy, Incremental) and bootstrap:
|
|
120
|
+
# First build of a keyed incremental Python model: the range comes from the
|
|
121
|
+
# staged output, since there is no query to probe the way a SQL model has.
|
|
122
|
+
interval = await _bootstrap_window(model, exp.select("*").from_(stage_table.copy()), engine)
|
|
123
|
+
statements = strategy.plan_statements(relation, target, engine.caps, interval, columns)
|
|
113
124
|
drop_stage = exp.Drop(this=stage_table.copy(), kind="TABLE", exists=True)
|
|
114
125
|
counts = await engine.execute_all([*pre_statements, *statements, drop_stage])
|
|
115
|
-
|
|
126
|
+
written = strategy.row_counts(counts[len(pre_statements) : len(pre_statements) + len(statements)])
|
|
127
|
+
return written, interval
|
|
116
128
|
|
|
117
129
|
|
|
118
130
|
async def _align_stage_to_target(
|
|
@@ -178,7 +190,7 @@ async def _deliver_table(
|
|
|
178
190
|
interval: Interval | None,
|
|
179
191
|
) -> RowCounts:
|
|
180
192
|
"""Deliver ``resolved`` into an external table (``materialise: table``) via
|
|
181
|
-
``strategy`` (replace / append / merge / full_merge /
|
|
193
|
+
``strategy`` (replace / append / merge / full_merge / incremental).
|
|
182
194
|
|
|
183
195
|
The external target is never dropped (grants and readers survive). When it already
|
|
184
196
|
exists the source is staged in the warehouse and aligned to the target (additive
|
|
@@ -188,7 +200,7 @@ async def _deliver_table(
|
|
|
188
200
|
order exactly.
|
|
189
201
|
|
|
190
202
|
Two cases skip staging and run the strategy directly against the target: the first
|
|
191
|
-
delivery (the ensure-create matches the source), and any windowed ``
|
|
203
|
+
delivery (the ensure-create matches the source), and any windowed ``incremental``
|
|
192
204
|
delivery (``interval`` set). An incremental window is grain-scoped and stays
|
|
193
205
|
schema-stable within a fingerprint, so staging the *whole* source once per window
|
|
194
206
|
would make a wide backfill/restate O(windows × source) — the pathological case."""
|
|
@@ -409,15 +421,38 @@ async def _run_backfill(
|
|
|
409
421
|
target_engine = registry.require(model.engine, model=model.name)
|
|
410
422
|
resolution = await _stage_cross_engine_inputs(model, compiled, registry, physical, staged, stage_lock, result)
|
|
411
423
|
|
|
424
|
+
if task.reuse_existing and await target_engine.table_exists(snapshot.physical_table):
|
|
425
|
+
# fingerprint already materialised (e.g. by another environment): the content-
|
|
426
|
+
# addressed table exists, so skip the (re)build compute — record the snapshot for
|
|
427
|
+
# this env's promotion, gate on checks against the existing table, and let the
|
|
428
|
+
# caller swap this environment's view onto the shared table. The table_exists guard
|
|
429
|
+
# is what makes the differ's optimistic reuse safe: a snapshot row can outlive its
|
|
430
|
+
# table (an in-memory warehouse across processes, a gc'd table) — then we fall
|
|
431
|
+
# through and build for real.
|
|
432
|
+
await state.add_snapshot(snapshot) # idempotent — the row may already exist
|
|
433
|
+
if snapshot.name not in result.built and snapshot.name not in result.reused:
|
|
434
|
+
result.reused.append(snapshot.name)
|
|
435
|
+
await _gate_checks(model, compiled, target_engine, state, plan.environment, result, resolution)
|
|
436
|
+
result.timings[snapshot.name] = result.timings.get(snapshot.name, 0.0) + (time.perf_counter() - task_started)
|
|
437
|
+
return
|
|
438
|
+
|
|
412
439
|
if model.ast is None: # Python model: run the function, load Arrow into the snapshot table
|
|
413
440
|
if model.materialise != "virtual":
|
|
414
441
|
raise PlanError(
|
|
415
442
|
f"Python model {snapshot.name!r} must materialise as virtual; table/file (write a SQL model "
|
|
416
443
|
f"over its output), view and ephemeral are not supported for Python models"
|
|
417
444
|
)
|
|
418
|
-
if model.strategy == "
|
|
445
|
+
if model.strategy == "incremental" and not model.key:
|
|
446
|
+
# Keyed is supported: the window bounds which staged rows are upserted.
|
|
447
|
+
# Unkeyed is not, and deliberately. For a SQL model the window predicate
|
|
448
|
+
# is pushed into the query so the engine only computes the window; a
|
|
449
|
+
# Python function has already computed everything by the time we could
|
|
450
|
+
# filter it, so an unkeyed windowed rewrite would look incremental while
|
|
451
|
+
# doing the full work every run. Bound the fetch with cursor= instead.
|
|
419
452
|
raise PlanError(
|
|
420
|
-
f"Python model {snapshot.name!r} cannot use
|
|
453
|
+
f"Python model {snapshot.name!r} cannot use incremental without a key: the function "
|
|
454
|
+
f"runs in full before the window can be applied, so the window would not save any work. "
|
|
455
|
+
f"Add key= to upsert the window's rows, or use cursor= with merge to bound the fetch"
|
|
421
456
|
)
|
|
422
457
|
recorded_self = await state.get_snapshot(snapshot.name, snapshot.fingerprint)
|
|
423
458
|
previous = recorded_self.physical_table if recorded_self is not None else None
|
|
@@ -432,10 +467,21 @@ async def _run_backfill(
|
|
|
432
467
|
result.record_rows(snapshot.name, RowCounts(inserted=loaded))
|
|
433
468
|
else: # keyed strategy: stage the Arrow output, then merge it in SQL
|
|
434
469
|
reader = await run_python_model(model, compiled, target_engine, resolution, previous)
|
|
435
|
-
merged = await _merge_python_output(
|
|
436
|
-
model,
|
|
470
|
+
merged, filled_window = await _merge_python_output(
|
|
471
|
+
model,
|
|
472
|
+
target_engine,
|
|
473
|
+
snapshot.physical_table,
|
|
474
|
+
reader,
|
|
475
|
+
exists=previous is not None,
|
|
476
|
+
interval=task.interval,
|
|
477
|
+
bootstrap=task.bootstrap,
|
|
437
478
|
)
|
|
438
479
|
result.record_rows(snapshot.name, merged)
|
|
480
|
+
if filled_window is not None: # incremental: accumulate the window in the ledger
|
|
481
|
+
filled = await state.get_intervals(snapshot.name, snapshot.fingerprint)
|
|
482
|
+
for carried in snapshot.intervals:
|
|
483
|
+
filled = filled.add(carried)
|
|
484
|
+
snapshot = replace(snapshot, intervals=filled.add(filled_window))
|
|
439
485
|
if model.columns:
|
|
440
486
|
validate_contract(model.name, await target_engine.describe(snapshot.physical_table), model.columns)
|
|
441
487
|
await state.add_snapshot(snapshot)
|
|
@@ -669,7 +715,10 @@ async def apply(
|
|
|
669
715
|
|
|
670
716
|
mapping = {name: compiled.models[name].fingerprint for name in plan.promote}
|
|
671
717
|
await state.promote(plan.environment, mapping)
|
|
672
|
-
|
|
718
|
+
# ephemeral models are tracked in the mapping (so re-plans stay clean) but are inlined
|
|
719
|
+
# into consumers — they have no promotable table/view, so the user-facing count omits
|
|
720
|
+
# them, keeping "promoted N" consistent with the N build rows shown
|
|
721
|
+
result.promoted = sum(1 for name in mapping if compiled.models[name].materialise != "ephemeral")
|
|
673
722
|
|
|
674
723
|
# deleted models: drop their env view and demote them, or the view serves the
|
|
675
724
|
# last snapshot forever and pins it against gc
|
|
@@ -243,7 +243,7 @@ def _schedule_reuse(plan: Plan, model: CompiledModel, previous: Snapshot, enviro
|
|
|
243
243
|
)
|
|
244
244
|
|
|
245
245
|
|
|
246
|
-
_HISTORY_STRATEGIES = frozenset({"merge", "full_merge", "scd", "
|
|
246
|
+
_HISTORY_STRATEGIES = frozenset({"merge", "full_merge", "hash_merge", "scd", "incremental"})
|
|
247
247
|
"""Strategies whose targets accumulate state a rebuild would destroy."""
|
|
248
248
|
|
|
249
249
|
|
|
@@ -283,7 +283,7 @@ async def diff(
|
|
|
283
283
|
classification still runs over the whole graph so downstream categories are correct.
|
|
284
284
|
|
|
285
285
|
``forward_only``: modified models whose strategy accumulates history
|
|
286
|
-
(merge / full_merge / scd /
|
|
286
|
+
(merge / full_merge / scd / incremental) inherit their
|
|
287
287
|
previous physical table and interval ledger instead of starting fresh — the
|
|
288
288
|
new logic applies going forward, history survives. Requires the new query to
|
|
289
289
|
stay shape-compatible with the existing table.
|
|
@@ -303,6 +303,13 @@ async def diff(
|
|
|
303
303
|
for name, fingerprint in current.items()
|
|
304
304
|
if name in compiled.models and fingerprint != compiled.models[name].fingerprint
|
|
305
305
|
)
|
|
306
|
+
# Fingerprints already materialised (by any prior apply — most usefully another
|
|
307
|
+
# environment): building these again would recompute an identical, content-addressed
|
|
308
|
+
# table, so schedule a reuse (record + view-swap) instead of a rebuild.
|
|
309
|
+
already_built = await state.get_snapshots((name, compiled.models[name].fingerprint) for name in selected)
|
|
310
|
+
|
|
311
|
+
def is_materialised(model: CompiledModel) -> bool:
|
|
312
|
+
return (model.name, model.fingerprint) in already_built
|
|
306
313
|
|
|
307
314
|
for model in compiled.ordered(): # topo order: upstream impact known before downstream
|
|
308
315
|
previous_fingerprint = current.get(model.name)
|
|
@@ -311,7 +318,13 @@ async def diff(
|
|
|
311
318
|
impact[model.name] = "semantic"
|
|
312
319
|
if model.name in selected:
|
|
313
320
|
plan.changes.append(ModelChange(model.name, ChangeType.ADDED, None, None, model.fingerprint))
|
|
314
|
-
schedule_build(
|
|
321
|
+
schedule_build(
|
|
322
|
+
plan,
|
|
323
|
+
model,
|
|
324
|
+
snapshot_of(model, ChangeCategory.BREAKING),
|
|
325
|
+
environment,
|
|
326
|
+
reuse_existing=is_materialised(model),
|
|
327
|
+
)
|
|
315
328
|
continue
|
|
316
329
|
|
|
317
330
|
if previous_fingerprint == model.fingerprint:
|
|
@@ -373,7 +386,9 @@ async def diff(
|
|
|
373
386
|
)
|
|
374
387
|
schedule_build(plan, model, snapshot, environment, seed_from=previous.physical_table) # type: ignore[union-attr]
|
|
375
388
|
elif rebuild:
|
|
376
|
-
schedule_build(
|
|
389
|
+
schedule_build(
|
|
390
|
+
plan, model, snapshot_of(model, category), environment, reuse_existing=is_materialised(model)
|
|
391
|
+
)
|
|
377
392
|
else:
|
|
378
393
|
_schedule_reuse(plan, model, previous, environment) # type: ignore[arg-type] # previous is not None here
|
|
379
394
|
|
|
@@ -55,6 +55,10 @@ class BackfillTask:
|
|
|
55
55
|
# the strategy runs — history moves to the new fingerprint, the old table stays
|
|
56
56
|
# as the rollback until gc.
|
|
57
57
|
seed_from: TableRef | None = None
|
|
58
|
+
# This exact fingerprint is already materialised (a prior apply, often in another
|
|
59
|
+
# environment): the physical table exists and its content is fingerprint-pinned, so
|
|
60
|
+
# apply skips the build compute and only gates checks + swaps this env's view.
|
|
61
|
+
reuse_existing: bool = False
|
|
58
62
|
|
|
59
63
|
|
|
60
64
|
@dataclass(frozen=True)
|
|
@@ -121,16 +125,29 @@ def env_view(environment: str, model_name: str) -> TableRef:
|
|
|
121
125
|
|
|
122
126
|
|
|
123
127
|
def schedule_build(
|
|
124
|
-
plan: Plan,
|
|
128
|
+
plan: Plan,
|
|
129
|
+
model: CompiledModel,
|
|
130
|
+
snapshot: Snapshot,
|
|
131
|
+
environment: str,
|
|
132
|
+
*,
|
|
133
|
+
seed_from: TableRef | None = None,
|
|
134
|
+
reuse_existing: bool = False,
|
|
125
135
|
) -> None:
|
|
126
136
|
"""Add the right tasks for a model: ephemeral builds nothing; a terminal
|
|
127
137
|
table/file builds (delivers) but gets no environment view; a virtual/view model
|
|
128
138
|
builds and is repointed by an environment view.
|
|
129
139
|
|
|
130
|
-
An
|
|
140
|
+
An incremental model (virtual, or a terminal ``table``) cannot build
|
|
131
141
|
without a window, so an apply fills the latest grain interval — the same default
|
|
132
142
|
as ``interlace run`` — leaving history to ``run --start/--end``.
|
|
143
|
+
|
|
144
|
+
``reuse_existing`` (the fingerprint is already materialised) skips the compute for a
|
|
145
|
+
plain virtual/view build — never for a terminal delivery, a forward-only seed, or an
|
|
146
|
+
incremental window, which must always run.
|
|
133
147
|
"""
|
|
148
|
+
# only a plain (non-seeded, non-windowed) virtual/view build can skip its compute:
|
|
149
|
+
# a terminal always delivers, a seed must copy history, an interval must fill
|
|
150
|
+
reuse = reuse_existing and seed_from is None and model.materialise in ("virtual", "view")
|
|
134
151
|
if model.materialise == "ephemeral": # inlined into consumers, never built
|
|
135
152
|
return
|
|
136
153
|
wants_view = model.materialise in ("virtual", "view") # terminal table/file has no env view
|
|
@@ -143,7 +160,7 @@ def schedule_build(
|
|
|
143
160
|
ViewSwap(env_view(environment, model.name), snapshot.physical_table, engine=model.engine)
|
|
144
161
|
)
|
|
145
162
|
|
|
146
|
-
if model.strategy == "
|
|
163
|
+
if model.strategy == "incremental": # virtual or terminal table: windowed delete+insert
|
|
147
164
|
from datetime import datetime
|
|
148
165
|
|
|
149
166
|
from interlace.state.interval import latest_complete_window, parse_grain
|
|
@@ -164,7 +181,7 @@ def schedule_build(
|
|
|
164
181
|
plan.backfills.append(BackfillTask(snapshot=snapshot, interval=window, seed_from=seed_from))
|
|
165
182
|
add_view()
|
|
166
183
|
return
|
|
167
|
-
plan.backfills.append(BackfillTask(snapshot=snapshot, seed_from=seed_from))
|
|
184
|
+
plan.backfills.append(BackfillTask(snapshot=snapshot, seed_from=seed_from, reuse_existing=reuse))
|
|
168
185
|
add_view()
|
|
169
186
|
|
|
170
187
|
|
|
@@ -70,7 +70,7 @@ async def run_plan(
|
|
|
70
70
|
|
|
71
71
|
# incremental into the interlace-owned virtual plane, or into a terminal
|
|
72
72
|
# `table` (windowed delete+insert against the external target)
|
|
73
|
-
is_incremental = model.strategy == "
|
|
73
|
+
is_incremental = model.strategy == "incremental" and model.materialise != "ephemeral"
|
|
74
74
|
wants_view = model.materialise in ("virtual", "view") # terminal table has no env view
|
|
75
75
|
if is_incremental:
|
|
76
76
|
grain = parse_grain(model.interval or "1d")
|