interlaced 2.2.0__tar.gz → 2.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. {interlaced-2.2.0/src/interlaced.egg-info → interlaced-2.3.0}/PKG-INFO +7 -6
  2. {interlaced-2.2.0 → interlaced-2.3.0}/README.md +3 -2
  3. {interlaced-2.2.0 → interlaced-2.3.0}/pyproject.toml +4 -4
  4. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/cli/main.py +3 -3
  5. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/dsl/decorators.py +3 -3
  6. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/dsl/discovery.py +1 -1
  7. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/spark.py +1 -1
  8. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/graph/project.py +2 -2
  9. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/ir/canonicalize.py +4 -1
  10. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/apply.py +36 -10
  11. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/differ.py +2 -2
  12. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/plan.py +2 -2
  13. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/run.py +1 -1
  14. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/app.py +8 -4
  15. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/__init__.py +10 -4
  16. interlaced-2.3.0/src/interlace/strategies/incremental.py +89 -0
  17. {interlaced-2.2.0 → interlaced-2.3.0/src/interlaced.egg-info}/PKG-INFO +7 -6
  18. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlaced.egg-info/SOURCES.txt +1 -1
  19. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlaced.egg-info/requires.txt +3 -3
  20. interlaced-2.2.0/src/interlace/strategies/incremental_by_time.py +0 -64
  21. {interlaced-2.2.0 → interlaced-2.3.0}/LICENSE +0 -0
  22. {interlaced-2.2.0 → interlaced-2.3.0}/MANIFEST.in +0 -0
  23. {interlaced-2.2.0 → interlaced-2.3.0}/setup.cfg +0 -0
  24. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/__init__.py +0 -0
  25. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/checks/__init__.py +0 -0
  26. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/checks/builtin.py +0 -0
  27. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/checks/runner.py +0 -0
  28. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/checks/spec.py +0 -0
  29. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/cli/__init__.py +0 -0
  30. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/config/__init__.py +0 -0
  31. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/config/config.py +0 -0
  32. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/contracts.py +0 -0
  33. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/dsl/__init__.py +0 -0
  34. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/dsl/sql_config.py +0 -0
  35. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/__init__.py +0 -0
  36. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/adbc.py +0 -0
  37. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/base.py +0 -0
  38. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/bigquery.py +0 -0
  39. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/duckdb.py +0 -0
  40. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/postgres.py +0 -0
  41. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/quack.py +0 -0
  42. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/redshift.py +0 -0
  43. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/registry.py +0 -0
  44. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/engines/snowflake.py +0 -0
  45. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/exceptions.py +0 -0
  46. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/graph/__init__.py +0 -0
  47. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/graph/column_lineage.py +0 -0
  48. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/graph/dag.py +0 -0
  49. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/graph/selectors.py +0 -0
  50. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/ir/__init__.py +0 -0
  51. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/ir/fingerprint.py +0 -0
  52. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/ir/relation.py +0 -0
  53. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/__init__.py +0 -0
  54. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/plan/resolve.py +0 -0
  55. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/project.py +0 -0
  56. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/py.typed +0 -0
  57. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/query.py +0 -0
  58. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/runtime/__init__.py +0 -0
  59. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/runtime/handles.py +0 -0
  60. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/runtime/python_model.py +0 -0
  61. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/scaffold.py +0 -0
  62. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/scheduler/__init__.py +0 -0
  63. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/scheduler/engine.py +0 -0
  64. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/scheduler/triggers.py +0 -0
  65. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/scheduler/worker.py +0 -0
  66. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/__init__.py +0 -0
  67. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/auth.py +0 -0
  68. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/app.css +0 -0
  69. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/favicon.svg +0 -0
  70. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/index.html +0 -0
  71. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/api.js +0 -0
  72. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/app.js +0 -0
  73. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/dag.js +0 -0
  74. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/timeline.js +0 -0
  75. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/ui.js +0 -0
  76. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/checks.js +0 -0
  77. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/environments.js +0 -0
  78. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/lineage.js +0 -0
  79. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/models.js +0 -0
  80. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/overview.js +0 -0
  81. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/plan.js +0 -0
  82. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/query.js +0 -0
  83. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/runs.js +0 -0
  84. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/streams.js +0 -0
  85. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/service/ui/js/views/system.js +0 -0
  86. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/sinks.py +0 -0
  87. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/sources/__init__.py +0 -0
  88. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/sources/auth.py +0 -0
  89. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/sources/rest.py +0 -0
  90. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/state/__init__.py +0 -0
  91. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/state/interval.py +0 -0
  92. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/state/janitor.py +0 -0
  93. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/state/snapshot.py +0 -0
  94. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/state/store.py +0 -0
  95. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/append.py +0 -0
  96. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/base.py +0 -0
  97. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/full_merge.py +0 -0
  98. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/hash_merge.py +0 -0
  99. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/merge.py +0 -0
  100. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/replace.py +0 -0
  101. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/replace_in_place.py +0 -0
  102. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/scd.py +0 -0
  103. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/strategies/view.py +0 -0
  104. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/streaming/__init__.py +0 -0
  105. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/streaming/log.py +0 -0
  106. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/streaming/materializer.py +0 -0
  107. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/streaming/schema.py +0 -0
  108. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/README.md +0 -0
  109. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/generate.py +0 -0
  110. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/interlace.yaml +0 -0
  111. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events.py +0 -0
  112. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events_by_minute.sql +0 -0
  113. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/models/events_by_type.sql +0 -0
  114. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/models/top_users.sql +0 -0
  115. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/models/user_spend.sql +0 -0
  116. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/events/template.yaml +0 -0
  117. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/github/README.md +0 -0
  118. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/github/interlace.yaml +0 -0
  119. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/github/models/github_issues.py +0 -0
  120. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/github/models/issues_by_state.sql +0 -0
  121. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/github/template.yaml +0 -0
  122. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/README.md +0 -0
  123. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/docker-compose.yml +0 -0
  124. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/init/seed.sql +0 -0
  125. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/interlace.yaml +0 -0
  126. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/models/orders.py +0 -0
  127. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/models/orders_by_status.sql +0 -0
  128. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/postgres/template.yaml +0 -0
  129. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/README.md +0 -0
  130. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/interlace.yaml +0 -0
  131. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/enriched_events.py +0 -0
  132. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/event_summary.sql +0 -0
  133. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/models/raw_events.sql +0 -0
  134. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlace/templates/quickstart/template.yaml +0 -0
  135. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlaced.egg-info/dependency_links.txt +0 -0
  136. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlaced.egg-info/entry_points.txt +0 -0
  137. {interlaced-2.2.0 → interlaced-2.3.0}/src/interlaced.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: interlaced
3
- Version: 2.2.0
3
+ Version: 2.3.0
4
4
  Summary: Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion
5
5
  Author-email: Mark <mark@interlace.sh>
6
6
  License-Expression: MIT
@@ -18,12 +18,12 @@ Classifier: Programming Language :: SQL
18
18
  Requires-Python: >=3.12
19
19
  Description-Content-Type: text/markdown
20
20
  License-File: LICENSE
21
- Requires-Dist: sqlglot<29.0,>=25.0
21
+ Requires-Dist: sqlglot<30.0,>=25.0
22
22
  Requires-Dist: duckdb>=1.5.3
23
23
  Requires-Dist: pyarrow>=17.0
24
24
  Requires-Dist: pydantic<3.0,>=2.5
25
25
  Requires-Dist: typer<1.0,>=0.12
26
- Requires-Dist: rich<15.0,>=13.0
26
+ Requires-Dist: rich<16.0,>=13.0
27
27
  Requires-Dist: cronsim<3.0,>=2.5
28
28
  Requires-Dist: tenacity<10.0,>=8.2
29
29
  Requires-Dist: pyyaml<7.0,>=6.0
@@ -59,7 +59,7 @@ Requires-Dist: httpx<1.0,>=0.27; extra == "dev"
59
59
  Requires-Dist: pytest-asyncio<2.0,>=1.0; extra == "dev"
60
60
  Requires-Dist: ruff<1.0,>=0.6; extra == "dev"
61
61
  Requires-Dist: black<27.0,>=24.0; extra == "dev"
62
- Requires-Dist: mypy<2.0,>=1.11; extra == "dev"
62
+ Requires-Dist: mypy<3.0,>=1.11; extra == "dev"
63
63
  Dynamic: license-file
64
64
 
65
65
  # interlace
@@ -127,8 +127,9 @@ def orders(cursor, this):
127
127
  ```
128
128
 
129
129
  **Strategies:** `replace`, `view`, `ephemeral` (CTE-inlined), `merge` (upsert),
130
- `full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
131
- (windowed, interval-ledger backfill/catchup), `scd` (history with validity windows).
130
+ `full_merge` (full-state source applied as a minimal diff), `incremental` (one time window
131
+ at a time — rewrites the window, or upserts within it if you give it a `key`), `scd`
132
+ (history with validity windows).
132
133
 
133
134
  ## Plan / apply
134
135
 
@@ -63,8 +63,9 @@ def orders(cursor, this):
63
63
  ```
64
64
 
65
65
  **Strategies:** `replace`, `view`, `ephemeral` (CTE-inlined), `merge` (upsert),
66
- `full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
67
- (windowed, interval-ledger backfill/catchup), `scd` (history with validity windows).
66
+ `full_merge` (full-state source applied as a minimal diff), `incremental` (one time window
67
+ at a time — rewrites the window, or upserts within it if you give it a `key`), `scd`
68
+ (history with validity windows).
68
69
 
69
70
  ## Plan / apply
70
71
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "interlaced"
7
- version = "2.2.0"
7
+ version = "2.3.0"
8
8
  description = "Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.12"
@@ -24,12 +24,12 @@ classifiers = [
24
24
  # Core: the IR + engine + state + plan spine (Phase 1). Everything is Arrow- and
25
25
  # sqlglot-native. See docs/architecture/architecture.md.
26
26
  dependencies = [
27
- "sqlglot>=25.0,<29.0", # canonical IR, transpilation, semantic diff, column lineage
27
+ "sqlglot>=25.0,<30.0", # canonical IR, transpilation, semantic diff, column lineage
28
28
  "duckdb>=1.5.3", # default engine + federation hub; DuckLake storage + quack serving
29
29
  "pyarrow>=17.0", # the wire format: RecordBatchReader everywhere
30
30
  "pydantic>=2.5,<3.0", # config + manifest validation (cold paths only)
31
31
  "typer>=0.12,<1.0", # CLI
32
- "rich>=13.0,<15.0", # display, strictly an event subscriber
32
+ "rich>=13.0,<16.0", # display, strictly an event subscriber
33
33
  "cronsim>=2.5,<3.0", # cron parsing for the trigger engine
34
34
  "tenacity>=8.2,<10.0", # retries (tasks, DuckLake commit conflicts, transfers)
35
35
  "pyyaml>=6.0,<7.0", # project config (config + env overlays)
@@ -78,7 +78,7 @@ dev = [
78
78
  "pytest-asyncio>=1.0,<2.0",
79
79
  "ruff>=0.6,<1.0",
80
80
  "black>=24.0,<27.0",
81
- "mypy>=1.11,<2.0",
81
+ "mypy>=1.11,<3.0",
82
82
  ]
83
83
 
84
84
  [tool.setuptools.packages.find]
@@ -107,7 +107,7 @@ _END = typer.Option("", "--end", help="Window end (ISO), for incremental models.
107
107
  _FORWARD_ONLY = typer.Option(
108
108
  False,
109
109
  "--forward-only",
110
- help="Modified history-keeping models (merge/full_merge/scd/incremental_by_time) carry their history "
110
+ help="Modified history-keeping models (merge/full_merge/scd/incremental) carry their history "
111
111
  "forward: it is copied to the new version, the new logic applies to the copy, and checks gate "
112
112
  "before views move. Requires a shape-compatible change.",
113
113
  )
@@ -205,7 +205,7 @@ async def _render_empty_incrementals(result: ApplyResult, compiled: CompiledProj
205
205
 
206
206
  for name in result.built:
207
207
  model = compiled.models[name]
208
- if model.strategy != "incremental_by_time" or model.is_terminal:
208
+ if model.strategy != "incremental" or model.is_terminal:
209
209
  continue
210
210
  counts = result.rows.get(name)
211
211
  if counts is not None and (counts.inserted or counts.updated):
@@ -404,7 +404,7 @@ def run(
404
404
  ) -> None:
405
405
  """Force-build models and promote, ignoring change detection.
406
406
 
407
- For incremental_by_time models, --start/--end set the catchup window
407
+ For incremental models, --start/--end set the catchup window
408
408
  (default: the latest grain interval).
409
409
  """
410
410
  asyncio.run(_execute(environment, path, select, start, end, restate=False, parallelism=parallelism))
@@ -94,9 +94,9 @@ class ModelDef:
94
94
  dialect: str | None = None
95
95
  engine: str | None = None # named engine from config (None → project default_engine)
96
96
  depends_on: tuple[str, ...] = ()
97
- interval: str | None = None # grain for incremental_by_time (e.g. "1d")
98
- time_column: str | None = None # partition column for incremental_by_time
99
- # First-build window for incremental_by_time: "auto" derives [min, max] of the
97
+ interval: str | None = None # grain for incremental (e.g. "1d")
98
+ time_column: str | None = None # partition column for incremental
99
+ # First-build window for incremental: "auto" derives [min, max] of the
100
100
  # time column from the source at apply time and fills it as ONE interval;
101
101
  # "none" keeps only the latest grain window; an ISO date pins the start.
102
102
  backfill: str = "auto"
@@ -64,7 +64,7 @@ def _sql_model(default_name: str, sql: str, config: dict[str, Any], default_dial
64
64
  depends_on=_as_tuple(config.get("depends_on") or ()),
65
65
  interval=config.get("interval"),
66
66
  time_column=config.get("time_column"),
67
- backfill=config.get("backfill", "auto"), # first-build window for incremental_by_time
67
+ backfill=config.get("backfill", "auto"), # first-build window for incremental
68
68
  tags=_as_tuple(config.get("tags") or ()),
69
69
  owner=config.get("owner"),
70
70
  description=config.get("description"),
@@ -7,7 +7,7 @@ come back as Arrow via ``DataFrame.toArrow()``, and Arrow loads go in through
7
7
  (``local[*]``, for tests) or a remote one (Spark Connect / a shared session).
8
8
 
9
9
  **Strategy support.** ``replace``, ``append`` and ``view`` run on any Spark
10
- catalog. ``merge`` (native ``MERGE``) and ``incremental_by_time`` (windowed
10
+ catalog. ``merge`` (native ``MERGE``) and ``incremental`` (windowed
11
11
  ``DELETE`` by literal predicate + ``INSERT``) need a catalog with row-level
12
12
  mutations — Delta Lake or Iceberg — configured on the session you hand the
13
13
  adapter (the tests use a Delta-backed local session). ``scd`` and ``full_merge``
@@ -44,9 +44,9 @@ class CompiledModel:
44
44
  materialise: str
45
45
  strategy: str
46
46
  key: tuple[str, ...] # business key for keyed strategies (merge)
47
- time_column: str | None # partition column for incremental_by_time
47
+ time_column: str | None # partition column for incremental
48
48
  cursor: str | None # column whose max is injected into a Python model's `cursor` param
49
- interval: str | None # grain for incremental_by_time (e.g. "1d")
49
+ interval: str | None # grain for incremental (e.g. "1d")
50
50
  tags: tuple[str, ...] # for tag: selection
51
51
  schedule: dict[str, str] | None # cron/interval schedule for the trigger engine
52
52
  columns: dict[str, str | None] | None # output contract validated at apply time
@@ -8,6 +8,8 @@ pruning and the ``impact`` command).
8
8
 
9
9
  from __future__ import annotations
10
10
 
11
+ from typing import cast
12
+
11
13
  import sqlglot
12
14
  from sqlglot import exp
13
15
 
@@ -86,4 +88,5 @@ def resolve_references(ast: exp.Expression, mapping: dict[str, TableRef]) -> exp
86
88
  node.set("catalog", exp.to_identifier(target.catalog) if target.catalog else None)
87
89
  return node
88
90
 
89
- return ast.transform(rewrite)
91
+ # sqlglot 29 loosened transform()'s return annotation; it is an Expression.
92
+ return cast("exp.Expression", ast.transform(rewrite))
@@ -36,7 +36,7 @@ from interlace.runtime.python_model import build_python_model, run_python_model
36
36
  from interlace.sinks import file_statements, target_ref
37
37
  from interlace.state.interval import Interval
38
38
  from interlace.state.store import StateStore
39
- from interlace.strategies import Strategy, resolve_strategy
39
+ from interlace.strategies import Incremental, Strategy, resolve_strategy
40
40
  from interlace.strategies.base import RowCounts
41
41
  from interlace.strategies.hash_merge import HashMerge
42
42
 
@@ -87,7 +87,9 @@ async def _merge_python_output(
87
87
  reader: pa.RecordBatchReader,
88
88
  *,
89
89
  exists: bool,
90
- ) -> RowCounts:
90
+ interval: Interval | None = None,
91
+ bootstrap: bool = False,
92
+ ) -> tuple[RowCounts, Interval | None]:
91
93
  """Stage a Python model's Arrow output and apply its keyed strategy in SQL.
92
94
 
93
95
  The output lands in a stage table (CREATE OR REPLACE, so a crashed run's
@@ -114,10 +116,15 @@ async def _merge_python_output(
114
116
  columns = [c for c in await engine.describe(stage) if c not in strategy.managed_columns]
115
117
 
116
118
  relation = SqlRelation(ast=source)
117
- statements = strategy.plan_statements(relation, target, engine.caps, None, columns)
119
+ if isinstance(strategy, Incremental) and bootstrap:
120
+ # First build of a keyed incremental Python model: the range comes from the
121
+ # staged output, since there is no query to probe the way a SQL model has.
122
+ interval = await _bootstrap_window(model, exp.select("*").from_(stage_table.copy()), engine)
123
+ statements = strategy.plan_statements(relation, target, engine.caps, interval, columns)
118
124
  drop_stage = exp.Drop(this=stage_table.copy(), kind="TABLE", exists=True)
119
125
  counts = await engine.execute_all([*pre_statements, *statements, drop_stage])
120
- return strategy.row_counts(counts[len(pre_statements) : len(pre_statements) + len(statements)])
126
+ written = strategy.row_counts(counts[len(pre_statements) : len(pre_statements) + len(statements)])
127
+ return written, interval
121
128
 
122
129
 
123
130
  async def _align_stage_to_target(
@@ -183,7 +190,7 @@ async def _deliver_table(
183
190
  interval: Interval | None,
184
191
  ) -> RowCounts:
185
192
  """Deliver ``resolved`` into an external table (``materialise: table``) via
186
- ``strategy`` (replace / append / merge / full_merge / incremental_by_time).
193
+ ``strategy`` (replace / append / merge / full_merge / incremental).
187
194
 
188
195
  The external target is never dropped (grants and readers survive). When it already
189
196
  exists the source is staged in the warehouse and aligned to the target (additive
@@ -193,7 +200,7 @@ async def _deliver_table(
193
200
  order exactly.
194
201
 
195
202
  Two cases skip staging and run the strategy directly against the target: the first
196
- delivery (the ensure-create matches the source), and any windowed ``incremental_by_time``
203
+ delivery (the ensure-create matches the source), and any windowed ``incremental``
197
204
  delivery (``interval`` set). An incremental window is grain-scoped and stays
198
205
  schema-stable within a fingerprint, so staging the *whole* source once per window
199
206
  would make a wide backfill/restate O(windows × source) — the pathological case."""
@@ -435,9 +442,17 @@ async def _run_backfill(
435
442
  f"Python model {snapshot.name!r} must materialise as virtual; table/file (write a SQL model "
436
443
  f"over its output), view and ephemeral are not supported for Python models"
437
444
  )
438
- if model.strategy == "incremental_by_time":
445
+ if model.strategy == "incremental" and not model.key:
446
+ # Keyed is supported: the window bounds which staged rows are upserted.
447
+ # Unkeyed is not, and deliberately. For a SQL model the window predicate
448
+ # is pushed into the query so the engine only computes the window; a
449
+ # Python function has already computed everything by the time we could
450
+ # filter it, so an unkeyed windowed rewrite would look incremental while
451
+ # doing the full work every run. Bound the fetch with cursor= instead.
439
452
  raise PlanError(
440
- f"Python model {snapshot.name!r} cannot use incremental_by_time; " f"use cursor= with merge instead"
453
+ f"Python model {snapshot.name!r} cannot use incremental without a key: the function "
454
+ f"runs in full before the window can be applied, so the window would not save any work. "
455
+ f"Add key= to upsert the window's rows, or use cursor= with merge to bound the fetch"
441
456
  )
442
457
  recorded_self = await state.get_snapshot(snapshot.name, snapshot.fingerprint)
443
458
  previous = recorded_self.physical_table if recorded_self is not None else None
@@ -452,10 +467,21 @@ async def _run_backfill(
452
467
  result.record_rows(snapshot.name, RowCounts(inserted=loaded))
453
468
  else: # keyed strategy: stage the Arrow output, then merge it in SQL
454
469
  reader = await run_python_model(model, compiled, target_engine, resolution, previous)
455
- merged = await _merge_python_output(
456
- model, target_engine, snapshot.physical_table, reader, exists=previous is not None
470
+ merged, filled_window = await _merge_python_output(
471
+ model,
472
+ target_engine,
473
+ snapshot.physical_table,
474
+ reader,
475
+ exists=previous is not None,
476
+ interval=task.interval,
477
+ bootstrap=task.bootstrap,
457
478
  )
458
479
  result.record_rows(snapshot.name, merged)
480
+ if filled_window is not None: # incremental: accumulate the window in the ledger
481
+ filled = await state.get_intervals(snapshot.name, snapshot.fingerprint)
482
+ for carried in snapshot.intervals:
483
+ filled = filled.add(carried)
484
+ snapshot = replace(snapshot, intervals=filled.add(filled_window))
459
485
  if model.columns:
460
486
  validate_contract(model.name, await target_engine.describe(snapshot.physical_table), model.columns)
461
487
  await state.add_snapshot(snapshot)
@@ -243,7 +243,7 @@ def _schedule_reuse(plan: Plan, model: CompiledModel, previous: Snapshot, enviro
243
243
  )
244
244
 
245
245
 
246
- _HISTORY_STRATEGIES = frozenset({"merge", "full_merge", "hash_merge", "scd", "incremental_by_time"})
246
+ _HISTORY_STRATEGIES = frozenset({"merge", "full_merge", "hash_merge", "scd", "incremental"})
247
247
  """Strategies whose targets accumulate state a rebuild would destroy."""
248
248
 
249
249
 
@@ -283,7 +283,7 @@ async def diff(
283
283
  classification still runs over the whole graph so downstream categories are correct.
284
284
 
285
285
  ``forward_only``: modified models whose strategy accumulates history
286
- (merge / full_merge / scd / incremental_by_time) inherit their
286
+ (merge / full_merge / scd / incremental) inherit their
287
287
  previous physical table and interval ledger instead of starting fresh — the
288
288
  new logic applies going forward, history survives. Requires the new query to
289
289
  stay shape-compatible with the existing table.
@@ -137,7 +137,7 @@ def schedule_build(
137
137
  table/file builds (delivers) but gets no environment view; a virtual/view model
138
138
  builds and is repointed by an environment view.
139
139
 
140
- An incremental_by_time model (virtual, or a terminal ``table``) cannot build
140
+ An incremental model (virtual, or a terminal ``table``) cannot build
141
141
  without a window, so an apply fills the latest grain interval — the same default
142
142
  as ``interlace run`` — leaving history to ``run --start/--end``.
143
143
 
@@ -160,7 +160,7 @@ def schedule_build(
160
160
  ViewSwap(env_view(environment, model.name), snapshot.physical_table, engine=model.engine)
161
161
  )
162
162
 
163
- if model.strategy == "incremental_by_time": # virtual or terminal table: windowed delete+insert
163
+ if model.strategy == "incremental": # virtual or terminal table: windowed delete+insert
164
164
  from datetime import datetime
165
165
 
166
166
  from interlace.state.interval import latest_complete_window, parse_grain
@@ -70,7 +70,7 @@ async def run_plan(
70
70
 
71
71
  # incremental into the interlace-owned virtual plane, or into a terminal
72
72
  # `table` (windowed delete+insert against the external target)
73
- is_incremental = model.strategy == "incremental_by_time" and model.materialise != "ephemeral"
73
+ is_incremental = model.strategy == "incremental" and model.materialise != "ephemeral"
74
74
  wants_view = model.materialise in ("virtual", "view") # terminal table has no env view
75
75
  if is_incremental:
76
76
  grain = parse_grain(model.interval or "1d")
@@ -1618,10 +1618,14 @@ def create_app(
1618
1618
 
1619
1619
  async def send_wrapper(message: Message) -> None:
1620
1620
  if message["type"] == "http.response.start":
1621
- headers = message.setdefault("headers", [])
1622
- headers.append((b"content-security-policy", ui_csp.encode()))
1623
- headers.append((b"x-content-type-options", b"nosniff"))
1624
- headers.append((b"x-frame-options", b"DENY"))
1621
+ # ASGI types `headers` as an Iterable, not a list, so copy into one
1622
+ # rather than appending to whatever the server happened to pass.
1623
+ message["headers"] = [
1624
+ *message.get("headers", []),
1625
+ (b"content-security-policy", ui_csp.encode()),
1626
+ (b"x-content-type-options", b"nosniff"),
1627
+ (b"x-frame-options", b"DENY"),
1628
+ ]
1625
1629
  await send(message)
1626
1630
 
1627
1631
  await app(scope, receive, send_wrapper)
@@ -9,7 +9,7 @@ from interlace.strategies.append import Append
9
9
  from interlace.strategies.base import Strategy, table_expr
10
10
  from interlace.strategies.full_merge import FullMerge
11
11
  from interlace.strategies.hash_merge import HashMerge
12
- from interlace.strategies.incremental_by_time import IncrementalByTime
12
+ from interlace.strategies.incremental import Incremental
13
13
  from interlace.strategies.merge import Merge
14
14
  from interlace.strategies.replace import Replace
15
15
  from interlace.strategies.replace_in_place import ReplaceInPlace
@@ -20,7 +20,7 @@ __all__ = [
20
20
  "Append",
21
21
  "FullMerge",
22
22
  "HashMerge",
23
- "IncrementalByTime",
23
+ "Incremental",
24
24
  "Merge",
25
25
  "Replace",
26
26
  "ReplaceInPlace",
@@ -72,9 +72,15 @@ def resolve_strategy(
72
72
  raise PlanError("hash_merge requires a key", details={"materialise": materialise})
73
73
  return HashMerge(tuple(key))
74
74
  if strategy == "incremental_by_time":
75
+ raise PlanError(
76
+ "strategy: incremental_by_time was renamed to incremental — the behaviour is unchanged, "
77
+ "and `key:` now additionally makes it upsert within the window instead of rewriting it",
78
+ details={"materialise": materialise, "strategy": strategy},
79
+ )
80
+ if strategy == "incremental":
75
81
  if not time_column:
76
- raise PlanError("incremental_by_time requires a time_column", details={"materialise": materialise})
77
- return IncrementalByTime(time_column)
82
+ raise PlanError("incremental requires a time_column", details={"materialise": materialise})
83
+ return Incremental(time_column, tuple(key))
78
84
  if strategy == "scd":
79
85
  if not key:
80
86
  raise PlanError("scd requires a key", details={"materialise": materialise})
@@ -0,0 +1,89 @@
1
+ """Incremental strategy — one time window at a time.
2
+
3
+ Both modes read only the rows whose ``time_column`` falls in the window
4
+ ``[start, end)`` the scheduler/planner supplies. They differ in what the window
5
+ means for the target:
6
+
7
+ - **without ``key`` (the default)** — the window is *authoritative*. Delete
8
+ everything already in ``[start, end)``, then insert the window's rows. A row
9
+ that vanished from the source disappears from the target. Delete-then-reinsert
10
+ is what makes reprocessing a window idempotent, and therefore what makes
11
+ backfill and ``restate`` safe.
12
+
13
+ - **with ``key``** — the window only *bounds what is read*. Rows are upserted by
14
+ key, so a target row inside the window that the source no longer produces is
15
+ left alone. This is the mode for late-arriving corrections to rows you have
16
+ already published, where the window is a cheap way to avoid rescanning history
17
+ rather than a statement about what the period should contain.
18
+
19
+ The grain (``interval`` config) lives with the planner, not here.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from collections.abc import Sequence
25
+ from typing import cast
26
+
27
+ from sqlglot import exp
28
+
29
+ from interlace.engines.base import EngineCaps
30
+ from interlace.exceptions import PlanError
31
+ from interlace.ir.relation import SqlRelation, TableRef
32
+ from interlace.state.interval import Interval
33
+ from interlace.strategies.base import RowCounts, Strategy, _at, table_expr
34
+ from interlace.strategies.merge import Merge
35
+
36
+
37
+ class Incremental(Strategy):
38
+ """``DELETE`` the window + ``INSERT`` it, or — with a ``key`` — upsert within it."""
39
+
40
+ def __init__(self, time_column: str, key: tuple[str, ...] = ()) -> None:
41
+ if not time_column:
42
+ raise PlanError("incremental requires a time_column")
43
+ self.time_column = time_column
44
+ self.key = key
45
+ # Keyed mode is a merge whose source happens to be window-filtered, so the
46
+ # upsert itself (native MERGE, or the portable DELETE+INSERT fallback) is
47
+ # the one in Merge rather than a second copy of it here.
48
+ self._merge = Merge(key) if key else None
49
+
50
+ def plan_statements(
51
+ self,
52
+ relation: SqlRelation,
53
+ target: TableRef,
54
+ caps: EngineCaps,
55
+ interval: Interval | None = None,
56
+ columns: Sequence[str] | None = None,
57
+ ) -> list[exp.Expression]:
58
+ if interval is None:
59
+ raise PlanError("incremental requires an interval to process")
60
+ query = relation.ast
61
+ table = table_expr(target)
62
+
63
+ def derived() -> exp.Subquery:
64
+ return cast("exp.Query", query.copy()).subquery("_s")
65
+
66
+ def window() -> exp.Expression:
67
+ column = exp.column(self.time_column)
68
+ return exp.And(
69
+ this=exp.GTE(this=column.copy(), expression=exp.Literal.string(interval.start.isoformat())),
70
+ expression=exp.LT(this=column.copy(), expression=exp.Literal.string(interval.end.isoformat())),
71
+ )
72
+
73
+ if self._merge is not None:
74
+ # The window bounds the source; the key decides what is written.
75
+ windowed = exp.select("*").from_(derived()).where(window())
76
+ return self._merge.plan_statements(SqlRelation(ast=windowed), target, caps, interval, columns)
77
+
78
+ ensure = exp.Create(
79
+ this=table.copy(), kind="TABLE", exists=True, expression=exp.select("*").from_(derived()).limit(0)
80
+ )
81
+ delete = exp.Delete(this=table.copy(), where=exp.Where(this=window()))
82
+ insert = exp.Insert(this=table.copy(), expression=exp.select("*").from_(derived()).where(window()))
83
+ return [ensure, delete, insert]
84
+
85
+ def row_counts(self, counts: Sequence[int]) -> RowCounts:
86
+ if self._merge is not None:
87
+ return self._merge.row_counts(counts)
88
+ # [ensure, delete window, insert window]: catchup deletes 0; restate rewrites
89
+ return RowCounts(inserted=_at(counts, 2), deleted=_at(counts, 1))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: interlaced
3
- Version: 2.2.0
3
+ Version: 2.3.0
4
4
  Summary: Python/SQL-first data platform: transformation, built-in orchestration, and durable streaming ingestion
5
5
  Author-email: Mark <mark@interlace.sh>
6
6
  License-Expression: MIT
@@ -18,12 +18,12 @@ Classifier: Programming Language :: SQL
18
18
  Requires-Python: >=3.12
19
19
  Description-Content-Type: text/markdown
20
20
  License-File: LICENSE
21
- Requires-Dist: sqlglot<29.0,>=25.0
21
+ Requires-Dist: sqlglot<30.0,>=25.0
22
22
  Requires-Dist: duckdb>=1.5.3
23
23
  Requires-Dist: pyarrow>=17.0
24
24
  Requires-Dist: pydantic<3.0,>=2.5
25
25
  Requires-Dist: typer<1.0,>=0.12
26
- Requires-Dist: rich<15.0,>=13.0
26
+ Requires-Dist: rich<16.0,>=13.0
27
27
  Requires-Dist: cronsim<3.0,>=2.5
28
28
  Requires-Dist: tenacity<10.0,>=8.2
29
29
  Requires-Dist: pyyaml<7.0,>=6.0
@@ -59,7 +59,7 @@ Requires-Dist: httpx<1.0,>=0.27; extra == "dev"
59
59
  Requires-Dist: pytest-asyncio<2.0,>=1.0; extra == "dev"
60
60
  Requires-Dist: ruff<1.0,>=0.6; extra == "dev"
61
61
  Requires-Dist: black<27.0,>=24.0; extra == "dev"
62
- Requires-Dist: mypy<2.0,>=1.11; extra == "dev"
62
+ Requires-Dist: mypy<3.0,>=1.11; extra == "dev"
63
63
  Dynamic: license-file
64
64
 
65
65
  # interlace
@@ -127,8 +127,9 @@ def orders(cursor, this):
127
127
  ```
128
128
 
129
129
  **Strategies:** `replace`, `view`, `ephemeral` (CTE-inlined), `merge` (upsert),
130
- `full_merge` (full-state source applied as a minimal diff), `incremental_by_time`
131
- (windowed, interval-ledger backfill/catchup), `scd` (history with validity windows).
130
+ `full_merge` (full-state source applied as a minimal diff), `incremental` (one time window
131
+ at a time — rewrites the window, or upserts within it if you give it a `key`), `scd`
132
+ (history with validity windows).
132
133
 
133
134
  ## Plan / apply
134
135
 
@@ -89,7 +89,7 @@ src/interlace/strategies/append.py
89
89
  src/interlace/strategies/base.py
90
90
  src/interlace/strategies/full_merge.py
91
91
  src/interlace/strategies/hash_merge.py
92
- src/interlace/strategies/incremental_by_time.py
92
+ src/interlace/strategies/incremental.py
93
93
  src/interlace/strategies/merge.py
94
94
  src/interlace/strategies/replace.py
95
95
  src/interlace/strategies/replace_in_place.py
@@ -1,9 +1,9 @@
1
- sqlglot<29.0,>=25.0
1
+ sqlglot<30.0,>=25.0
2
2
  duckdb>=1.5.3
3
3
  pyarrow>=17.0
4
4
  pydantic<3.0,>=2.5
5
5
  typer<1.0,>=0.12
6
- rich<15.0,>=13.0
6
+ rich<16.0,>=13.0
7
7
  cronsim<3.0,>=2.5
8
8
  tenacity<10.0,>=8.2
9
9
  pyyaml<7.0,>=6.0
@@ -29,7 +29,7 @@ httpx<1.0,>=0.27
29
29
  pytest-asyncio<2.0,>=1.0
30
30
  ruff<1.0,>=0.6
31
31
  black<27.0,>=24.0
32
- mypy<2.0,>=1.11
32
+ mypy<3.0,>=1.11
33
33
 
34
34
  [pandas]
35
35
  pandas<4.0,>=2.0
@@ -1,64 +0,0 @@
1
- """Incremental-by-time strategy.
2
-
3
- Processes one time window ``[start, end)`` at a time: ensure the target exists,
4
- delete that window, then insert the query filtered to it. Re-processing a window
5
- is idempotent (delete + reinsert), which is what makes backfill and catchup safe.
6
- The concrete interval comes from the scheduler/planner; the grain (``interval``
7
- config) lives there, not here.
8
- """
9
-
10
- from __future__ import annotations
11
-
12
- from collections.abc import Sequence
13
- from typing import cast
14
-
15
- from sqlglot import exp
16
-
17
- from interlace.engines.base import EngineCaps
18
- from interlace.exceptions import PlanError
19
- from interlace.ir.relation import SqlRelation, TableRef
20
- from interlace.state.interval import Interval
21
- from interlace.strategies.base import RowCounts, Strategy, _at, table_expr
22
-
23
-
24
- class IncrementalByTime(Strategy):
25
- """``CREATE IF NOT EXISTS`` + ``DELETE`` the window + ``INSERT`` the window's rows."""
26
-
27
- def __init__(self, time_column: str) -> None:
28
- if not time_column:
29
- raise PlanError("incremental_by_time requires a time_column")
30
- self.time_column = time_column
31
-
32
- def plan_statements(
33
- self,
34
- relation: SqlRelation,
35
- target: TableRef,
36
- caps: EngineCaps,
37
- interval: Interval | None = None,
38
- columns: Sequence[str] | None = None,
39
- ) -> list[exp.Expression]:
40
- if interval is None:
41
- raise PlanError("incremental_by_time requires an interval to process")
42
- query = relation.ast
43
- table = table_expr(target)
44
-
45
- def derived() -> exp.Subquery:
46
- return cast("exp.Query", query.copy()).subquery("_s")
47
-
48
- def window() -> exp.Expression:
49
- column = exp.column(self.time_column)
50
- return exp.And(
51
- this=exp.GTE(this=column.copy(), expression=exp.Literal.string(interval.start.isoformat())),
52
- expression=exp.LT(this=column.copy(), expression=exp.Literal.string(interval.end.isoformat())),
53
- )
54
-
55
- ensure = exp.Create(
56
- this=table.copy(), kind="TABLE", exists=True, expression=exp.select("*").from_(derived()).limit(0)
57
- )
58
- delete = exp.Delete(this=table.copy(), where=exp.Where(this=window()))
59
- insert = exp.Insert(this=table.copy(), expression=exp.select("*").from_(derived()).where(window()))
60
- return [ensure, delete, insert]
61
-
62
- def row_counts(self, counts: Sequence[int]) -> RowCounts:
63
- # [ensure, delete window, insert window]: catchup deletes 0; restate rewrites
64
- return RowCounts(inserted=_at(counts, 2), deleted=_at(counts, 1))
File without changes
File without changes
File without changes