duckpipe 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- duckpipe/__init__.py +41 -0
- duckpipe/cli.py +393 -0
- duckpipe/dag.py +164 -0
- duckpipe/fingerprint.py +68 -0
- duckpipe/remote.py +199 -0
- duckpipe/scheduler.py +502 -0
- duckpipe/state.py +540 -0
- duckpipe/task.py +127 -0
- duckpipe-0.1.0.dist-info/METADATA +320 -0
- duckpipe-0.1.0.dist-info/RECORD +13 -0
- duckpipe-0.1.0.dist-info/WHEEL +4 -0
- duckpipe-0.1.0.dist-info/entry_points.txt +3 -0
- duckpipe-0.1.0.dist-info/licenses/LICENSE +21 -0
duckpipe/__init__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""DuckPipe: a serverless-first, DuckDB-native pipeline orchestrator.
|
|
2
|
+
|
|
3
|
+
No scheduler daemon, no central metadata database required, no broker
|
|
4
|
+
-- a run is a process that starts, does work, records what it did to a
|
|
5
|
+
``.duckdb`` file, and exits::
|
|
6
|
+
|
|
7
|
+
from duckpipe import task, run
|
|
8
|
+
|
|
9
|
+
@task
|
|
10
|
+
def extract():
|
|
11
|
+
return duckdb.sql("select * from read_parquet('data.parquet')")
|
|
12
|
+
|
|
13
|
+
@task(cache=True)
|
|
14
|
+
def transform(rel=extract): # `rel=extract` infers the dependency
|
|
15
|
+
return rel.filter("amount > 0")
|
|
16
|
+
|
|
17
|
+
if __name__ == "__main__":
|
|
18
|
+
run(__file__)
|
|
19
|
+
|
|
20
|
+
Then from a shell: ``duckpipe run pipeline.py``.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from duckpipe.dag import DAG, CycleError, build_dag
|
|
26
|
+
from duckpipe.remote import StateLockedError
|
|
27
|
+
from duckpipe.scheduler import RunSummary, run
|
|
28
|
+
from duckpipe.task import Task, task
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"task",
|
|
32
|
+
"run",
|
|
33
|
+
"build_dag",
|
|
34
|
+
"Task",
|
|
35
|
+
"DAG",
|
|
36
|
+
"CycleError",
|
|
37
|
+
"RunSummary",
|
|
38
|
+
"StateLockedError",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
__version__ = "0.1.0"
|
duckpipe/cli.py
ADDED
|
@@ -0,0 +1,393 @@
|
|
|
1
|
+
"""``duckpipe`` command-line interface.
|
|
2
|
+
|
|
3
|
+
Every subcommand is a thin wrapper over the same public functions
|
|
4
|
+
(``duckpipe.run``, ``duckpipe.build_dag``) any other trigger -- cron, CI,
|
|
5
|
+
a Lambda handler -- would call directly (DESIGN.md sec 5, sec 9).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import functools
|
|
11
|
+
import re
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Annotated
|
|
15
|
+
|
|
16
|
+
import typer
|
|
17
|
+
from rich.console import Console
|
|
18
|
+
from rich.table import Table
|
|
19
|
+
|
|
20
|
+
from duckpipe.dag import CycleError, DuplicateTaskNameError, build_dag
|
|
21
|
+
from duckpipe.fingerprint import resolve_fingerprints
|
|
22
|
+
from duckpipe.remote import StateLockedError
|
|
23
|
+
from duckpipe.scheduler import (
|
|
24
|
+
UpstreamNotCachedError,
|
|
25
|
+
UpstreamNotReadyError,
|
|
26
|
+
_default_db_path,
|
|
27
|
+
would_skip,
|
|
28
|
+
)
|
|
29
|
+
from duckpipe.scheduler import run as run_pipeline
|
|
30
|
+
from duckpipe.state import is_ducklake
|
|
31
|
+
|
|
32
|
+
app = typer.Typer(
|
|
33
|
+
name="duckpipe",
|
|
34
|
+
help="A serverless-first, DuckDB-native pipeline orchestrator.",
|
|
35
|
+
no_args_is_help=True,
|
|
36
|
+
pretty_exceptions_enable=False,
|
|
37
|
+
)
|
|
38
|
+
console = Console()
|
|
39
|
+
|
|
40
|
+
_STATUS_STYLE = {"success": "green", "skipped": "yellow", "failed": "red"}
|
|
41
|
+
|
|
42
|
+
# A malformed pipeline (a cycle, a duplicate task name, a plain typo that
|
|
43
|
+
# blows up at import time), a locked state_uri, or a scoped (--only) run
|
|
44
|
+
# dispatched out of order is a user-actionable situation, not a DuckPipe
|
|
45
|
+
# bug -- it should read as one short line, not a wall of framework frames.
|
|
46
|
+
_USER_FACING_ERRORS = (
|
|
47
|
+
CycleError,
|
|
48
|
+
DuplicateTaskNameError,
|
|
49
|
+
ImportError,
|
|
50
|
+
SyntaxError,
|
|
51
|
+
StateLockedError,
|
|
52
|
+
UpstreamNotReadyError,
|
|
53
|
+
UpstreamNotCachedError,
|
|
54
|
+
ValueError,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
_DUCKLAKE_FILE_PREFIXES = ("ducklake:sqlite:", "ducklake:duckdb:")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _state_probably_exists(db_path: str | Path) -> bool:
|
|
61
|
+
"""A cheap existence check that never opens (and so never creates) the
|
|
62
|
+
store, for commands like `show`/`stats` that shouldn't conjure a
|
|
63
|
+
fresh, empty state file just by asking about one that isn't there yet.
|
|
64
|
+
"""
|
|
65
|
+
if is_ducklake(db_path):
|
|
66
|
+
for prefix in _DUCKLAKE_FILE_PREFIXES:
|
|
67
|
+
if db_path.startswith(prefix):
|
|
68
|
+
return Path(db_path[len(prefix) :]).exists()
|
|
69
|
+
return True # a live catalog (e.g. Postgres) -- can't check cheaply
|
|
70
|
+
return Path(db_path).exists()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
_MERMAID_STATUS_CLASS = {
|
|
74
|
+
"success": "success",
|
|
75
|
+
"skipped": "skipped",
|
|
76
|
+
"failed": "failed",
|
|
77
|
+
"upstream_failed": "failed",
|
|
78
|
+
}
|
|
79
|
+
_MERMAID_CLASS_DEFS = {
|
|
80
|
+
"success": "classDef success fill:#d4f7dc,stroke:#2f9e44,color:#1a1a1a",
|
|
81
|
+
"skipped": "classDef skipped fill:#fff3cd,stroke:#d9a400,color:#1a1a1a",
|
|
82
|
+
"failed": "classDef failed fill:#f8d7da,stroke:#d64545,color:#1a1a1a",
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _mermaid_node_id(name: str) -> str:
|
|
87
|
+
# Task names are usually valid identifiers already (a function's own
|
|
88
|
+
# __name__, or an explicit name=...), but a Mermaid node id has its
|
|
89
|
+
# own, stricter rules -- sanitize rather than assume, so an unusual
|
|
90
|
+
# name= never produces a broken diagram.
|
|
91
|
+
return "t_" + re.sub(r"[^A-Za-z0-9_]", "_", name)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _mermaid_diagram(order: list, last_status: dict[str, tuple[str, str]]) -> str:
|
|
95
|
+
"""A `flowchart TD` a task's real dependency edges, one node per task
|
|
96
|
+
(labeled with its real name, not the sanitized id), colored by last
|
|
97
|
+
recorded status when state exists -- pastes straight into any
|
|
98
|
+
Markdown that renders Mermaid (GitHub, GitLab, many others)."""
|
|
99
|
+
lines = ["flowchart TD"]
|
|
100
|
+
for t in order:
|
|
101
|
+
label = t.name.replace('"', """)
|
|
102
|
+
lines.append(f' {_mermaid_node_id(t.name)}["{label}"]')
|
|
103
|
+
for t in order:
|
|
104
|
+
for up in t.upstream_tasks():
|
|
105
|
+
lines.append(f" {_mermaid_node_id(up.name)} --> {_mermaid_node_id(t.name)}")
|
|
106
|
+
|
|
107
|
+
used_classes: list[str] = []
|
|
108
|
+
for t in order:
|
|
109
|
+
status = last_status.get(t.name, (None, None))[0]
|
|
110
|
+
cls = _MERMAID_STATUS_CLASS.get(status)
|
|
111
|
+
if cls:
|
|
112
|
+
lines.append(f" class {_mermaid_node_id(t.name)} {cls}")
|
|
113
|
+
if cls not in used_classes:
|
|
114
|
+
used_classes.append(cls)
|
|
115
|
+
for cls in used_classes:
|
|
116
|
+
lines.append(f" {_MERMAID_CLASS_DEFS[cls]}")
|
|
117
|
+
return "\n".join(lines)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _friendly_errors[F: Callable[..., object]](command: F) -> F:
|
|
121
|
+
@functools.wraps(command)
|
|
122
|
+
def wrapper(*args: object, **kwargs: object) -> object:
|
|
123
|
+
try:
|
|
124
|
+
return command(*args, **kwargs)
|
|
125
|
+
except typer.Exit:
|
|
126
|
+
raise
|
|
127
|
+
except _USER_FACING_ERRORS as exc:
|
|
128
|
+
console.print(f"[red]error:[/red] {exc}")
|
|
129
|
+
raise typer.Exit(code=1) from None
|
|
130
|
+
except Exception as exc: # noqa: BLE001 - last-resort CLI-friendly fallback
|
|
131
|
+
console.print(f"[red]error:[/red] {type(exc).__name__}: {exc}")
|
|
132
|
+
raise typer.Exit(code=1) from None
|
|
133
|
+
|
|
134
|
+
return wrapper
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@app.command()
|
|
138
|
+
@_friendly_errors
|
|
139
|
+
def run(
|
|
140
|
+
pipeline: Annotated[Path, typer.Argument(exists=True, help="Path to a pipeline .py module")],
|
|
141
|
+
db: Annotated[
|
|
142
|
+
str | None,
|
|
143
|
+
typer.Option(
|
|
144
|
+
"--db",
|
|
145
|
+
help="State file path, or a ducklake:... catalog "
|
|
146
|
+
"(default: duckpipe.db next to the pipeline)",
|
|
147
|
+
),
|
|
148
|
+
] = None,
|
|
149
|
+
state_uri: Annotated[
|
|
150
|
+
str | None,
|
|
151
|
+
typer.Option("--state-uri", help="fsspec URI to sync state to/from (S3/GCS/Azure/local)"),
|
|
152
|
+
] = None,
|
|
153
|
+
force: Annotated[
|
|
154
|
+
bool, typer.Option("--force", help="Ignore fingerprints; re-run every task")
|
|
155
|
+
] = False,
|
|
156
|
+
max_workers: Annotated[
|
|
157
|
+
int | None, typer.Option("--max-workers", help="Cap concurrent tasks")
|
|
158
|
+
] = None,
|
|
159
|
+
no_lock: Annotated[
|
|
160
|
+
bool,
|
|
161
|
+
typer.Option(
|
|
162
|
+
"--no-lock",
|
|
163
|
+
help="With --state-uri, skip the advisory lock and allow racing (not recommended)",
|
|
164
|
+
),
|
|
165
|
+
] = False,
|
|
166
|
+
only: Annotated[
|
|
167
|
+
str | None,
|
|
168
|
+
typer.Option("--only", help="Run just this one task -- for distributed dispatch"),
|
|
169
|
+
] = None,
|
|
170
|
+
run_id: Annotated[
|
|
171
|
+
str | None,
|
|
172
|
+
typer.Option("--run-id", help="Shared run id so --only calls for one run group together"),
|
|
173
|
+
] = None,
|
|
174
|
+
data_path: Annotated[
|
|
175
|
+
str | None,
|
|
176
|
+
typer.Option(
|
|
177
|
+
"--data-path", help="DuckLake DATA_PATH (only with --db ducklake:...; usually auto)"
|
|
178
|
+
),
|
|
179
|
+
] = None,
|
|
180
|
+
) -> None:
|
|
181
|
+
"""Run a pipeline module end-to-end, or (with --only) exactly one task."""
|
|
182
|
+
summary = run_pipeline(
|
|
183
|
+
pipeline,
|
|
184
|
+
db_path=db,
|
|
185
|
+
state_uri=state_uri,
|
|
186
|
+
force=force,
|
|
187
|
+
max_workers=max_workers,
|
|
188
|
+
lock=not no_lock,
|
|
189
|
+
only=only,
|
|
190
|
+
run_id=run_id,
|
|
191
|
+
data_path=data_path,
|
|
192
|
+
)
|
|
193
|
+
table = Table(title=f"run {summary.run_id} ({summary.db_path})")
|
|
194
|
+
table.add_column("task")
|
|
195
|
+
table.add_column("status")
|
|
196
|
+
table.add_column("detail")
|
|
197
|
+
for name, status in summary.statuses.items():
|
|
198
|
+
style = _STATUS_STYLE.get(status, "")
|
|
199
|
+
rendered = f"[{style}]{status}[/{style}]" if style else status
|
|
200
|
+
table.add_row(name, rendered, summary.errors.get(name, ""))
|
|
201
|
+
console.print(table)
|
|
202
|
+
if not summary.success:
|
|
203
|
+
raise typer.Exit(code=1)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@app.command()
|
|
207
|
+
@_friendly_errors
|
|
208
|
+
def show(
|
|
209
|
+
pipeline: Annotated[Path, typer.Argument(exists=True, help="Path to a pipeline .py module")],
|
|
210
|
+
db: Annotated[
|
|
211
|
+
str | None, typer.Option("--db", help="State file to read last-run status from")
|
|
212
|
+
] = None,
|
|
213
|
+
as_json: Annotated[
|
|
214
|
+
bool,
|
|
215
|
+
typer.Option(
|
|
216
|
+
"--json", help="Machine-readable output -- topological order + next-run preview"
|
|
217
|
+
),
|
|
218
|
+
] = False,
|
|
219
|
+
mermaid: Annotated[
|
|
220
|
+
bool,
|
|
221
|
+
typer.Option(
|
|
222
|
+
"--mermaid", help="Print a Mermaid flowchart of the DAG, colored by last status"
|
|
223
|
+
),
|
|
224
|
+
] = False,
|
|
225
|
+
) -> None:
|
|
226
|
+
"""Print the resolved DAG, each task's last-run status, and what would
|
|
227
|
+
happen if you ran it again right now -- a dry-run preview of the same
|
|
228
|
+
fingerprint check `run` itself uses, so you can see what's stale
|
|
229
|
+
before spending the time to re-run it. `--json` is the discovery
|
|
230
|
+
primitive a coordinator dispatching `--only <task>` calls needs
|
|
231
|
+
(DESIGN.md sec 8): topological order plus which tasks would skip.
|
|
232
|
+
`--mermaid` prints a flowchart instead -- paste it into any Markdown
|
|
233
|
+
that renders Mermaid."""
|
|
234
|
+
if as_json and mermaid:
|
|
235
|
+
raise ValueError("--json and --mermaid are two different output modes -- pick one")
|
|
236
|
+
dag = build_dag(pipeline)
|
|
237
|
+
order = dag.topological_order()
|
|
238
|
+
fingerprints = resolve_fingerprints(order)
|
|
239
|
+
db_path: str | Path = db or _default_db_path(pipeline)
|
|
240
|
+
|
|
241
|
+
last_status: dict[str, tuple[str, str]] = {}
|
|
242
|
+
next_run: dict[str, str] = {}
|
|
243
|
+
if _state_probably_exists(db_path):
|
|
244
|
+
from duckpipe.state import StateStore
|
|
245
|
+
|
|
246
|
+
# read_only: `show` never needs to write, and must never block on
|
|
247
|
+
# (or be blocked by) a pipeline that's still running.
|
|
248
|
+
with StateStore(db_path, read_only=True) as store:
|
|
249
|
+
for t in order:
|
|
250
|
+
row = store.last_status(t.name)
|
|
251
|
+
if row:
|
|
252
|
+
last_status[t.name] = (row[0], str(row[1]))
|
|
253
|
+
if would_skip(store, t, fingerprints[t.name]):
|
|
254
|
+
next_run[t.name] = "skip (unchanged)"
|
|
255
|
+
elif not t.cache:
|
|
256
|
+
next_run[t.name] = "run (not cached)"
|
|
257
|
+
else:
|
|
258
|
+
next_run[t.name] = "run (changed)"
|
|
259
|
+
else:
|
|
260
|
+
next_run = {t.name: "run (no prior state)" for t in order}
|
|
261
|
+
|
|
262
|
+
if mermaid:
|
|
263
|
+
print(_mermaid_diagram(order, last_status))
|
|
264
|
+
return
|
|
265
|
+
|
|
266
|
+
if as_json:
|
|
267
|
+
import json
|
|
268
|
+
|
|
269
|
+
print(
|
|
270
|
+
json.dumps(
|
|
271
|
+
[
|
|
272
|
+
{
|
|
273
|
+
"task": t.name,
|
|
274
|
+
"depends_on": sorted(u.name for u in t.upstream_tasks()),
|
|
275
|
+
"next_run": next_run[t.name],
|
|
276
|
+
}
|
|
277
|
+
for t in order
|
|
278
|
+
]
|
|
279
|
+
)
|
|
280
|
+
)
|
|
281
|
+
return
|
|
282
|
+
|
|
283
|
+
table = Table(title=f"DAG: {pipeline}")
|
|
284
|
+
table.add_column("task")
|
|
285
|
+
table.add_column("depends on")
|
|
286
|
+
table.add_column("last status")
|
|
287
|
+
table.add_column("last run")
|
|
288
|
+
table.add_column("next run")
|
|
289
|
+
for t in order:
|
|
290
|
+
deps = ", ".join(sorted(u.name for u in t.upstream_tasks())) or "-"
|
|
291
|
+
status, ts = last_status.get(t.name, ("-", "-"))
|
|
292
|
+
style = _STATUS_STYLE.get(status, "")
|
|
293
|
+
rendered = f"[{style}]{status}[/{style}]" if style else status
|
|
294
|
+
next_style = "yellow" if next_run[t.name].startswith("skip") else ""
|
|
295
|
+
next_rendered = (
|
|
296
|
+
f"[{next_style}]{next_run[t.name]}[/{next_style}]" if next_style else next_run[t.name]
|
|
297
|
+
)
|
|
298
|
+
table.add_row(t.name, deps, rendered, ts, next_rendered)
|
|
299
|
+
console.print(table)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
@app.command()
|
|
303
|
+
@_friendly_errors
|
|
304
|
+
def compact(
|
|
305
|
+
state_uri: Annotated[str, typer.Argument(help="fsspec URI of the state file to compact")],
|
|
306
|
+
no_lock: Annotated[
|
|
307
|
+
bool, typer.Option("--no-lock", help="Skip the advisory lock (not recommended)")
|
|
308
|
+
] = False,
|
|
309
|
+
) -> None:
|
|
310
|
+
"""Fold pending per-task deltas from `--only` runs into the canonical
|
|
311
|
+
state file, and clean them up. Nothing needs this for correctness --
|
|
312
|
+
every invocation already absorbs pending deltas itself -- but a purely
|
|
313
|
+
distributed workflow (many `--only` workers, no whole-run ever) never
|
|
314
|
+
otherwise re-uploads the canonical file, so its `.pending/` directory
|
|
315
|
+
only grows. Run this periodically (e.g. from cron) if that's your
|
|
316
|
+
workflow."""
|
|
317
|
+
from duckpipe.remote import compact as compact_state
|
|
318
|
+
|
|
319
|
+
compact_state(state_uri, lock=not no_lock)
|
|
320
|
+
console.print(f"[green]compacted[/green] {state_uri}")
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
@app.command()
|
|
324
|
+
@_friendly_errors
|
|
325
|
+
def stats(
|
|
326
|
+
db: Annotated[str, typer.Argument(help="Path to a state file, or a ducklake:... catalog")],
|
|
327
|
+
limit: Annotated[int, typer.Option(help="Number of recent runs to show")] = 20,
|
|
328
|
+
snapshots: Annotated[
|
|
329
|
+
bool,
|
|
330
|
+
typer.Option(
|
|
331
|
+
"--snapshots", help="Show DuckLake snapshot history (time travel) instead of stats"
|
|
332
|
+
),
|
|
333
|
+
] = False,
|
|
334
|
+
) -> None:
|
|
335
|
+
"""Show recent pipeline runs and per-task duration stats from the state
|
|
336
|
+
file. Against a DuckLake-backed store (``--db ducklake:...``), pass
|
|
337
|
+
``--snapshots`` to see every commit instead -- the time-travel-over-
|
|
338
|
+
run-history payoff of that backend (DESIGN.md sec 8, Phase 3b)."""
|
|
339
|
+
from duckpipe.state import StateStore
|
|
340
|
+
|
|
341
|
+
# read_only: never blocks on, or is blocked by, a pipeline still
|
|
342
|
+
# running against the same state file.
|
|
343
|
+
with StateStore(db, read_only=True) as store:
|
|
344
|
+
if snapshots:
|
|
345
|
+
_print_snapshots(store)
|
|
346
|
+
return
|
|
347
|
+
|
|
348
|
+
runs = store.con.execute(
|
|
349
|
+
"SELECT run_id, module_path, started_at, ended_at, status "
|
|
350
|
+
"FROM pipeline_runs ORDER BY started_at DESC LIMIT ?",
|
|
351
|
+
[limit],
|
|
352
|
+
).fetchall()
|
|
353
|
+
runs_table = Table(title="recent pipeline runs")
|
|
354
|
+
for col in ("run_id", "module", "started_at", "ended_at", "status"):
|
|
355
|
+
runs_table.add_column(col)
|
|
356
|
+
for row in runs:
|
|
357
|
+
runs_table.add_row(*(str(v) for v in row))
|
|
358
|
+
console.print(runs_table)
|
|
359
|
+
|
|
360
|
+
task_stats = store.con.execute(
|
|
361
|
+
"SELECT task_name, runs, avg_duration_ms, max_duration_ms, skipped_count, failed_count "
|
|
362
|
+
"FROM v_task_stats ORDER BY avg_duration_ms DESC"
|
|
363
|
+
).fetchall()
|
|
364
|
+
stats_table = Table(title="per-task stats (from v_task_stats)")
|
|
365
|
+
for col in ("task", "runs", "avg_ms", "max_ms", "skipped", "failed"):
|
|
366
|
+
stats_table.add_column(col)
|
|
367
|
+
for row in task_stats:
|
|
368
|
+
stats_table.add_row(*(f"{v:.1f}" if isinstance(v, float) else str(v) for v in row))
|
|
369
|
+
console.print(stats_table)
|
|
370
|
+
|
|
371
|
+
if store.is_ducklake:
|
|
372
|
+
console.print("[dim]DuckLake-backed -- see full history with --snapshots[/dim]")
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _print_snapshots(store) -> None:
|
|
376
|
+
columns = ("snapshot_id", "snapshot_time", "author", "commit_message")
|
|
377
|
+
table = Table(title="snapshot history (time travel)")
|
|
378
|
+
for col in columns:
|
|
379
|
+
table.add_column(col)
|
|
380
|
+
for snap in store.snapshots():
|
|
381
|
+
table.add_row(*(str(snap[col]) if snap[col] is not None else "-" for col in columns))
|
|
382
|
+
console.print(table)
|
|
383
|
+
console.print(
|
|
384
|
+
"[dim]query any table AT (VERSION => snapshot_id) to see state as of that point[/dim]"
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def main() -> None:
|
|
389
|
+
app()
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
if __name__ == "__main__":
|
|
393
|
+
main()
|
duckpipe/dag.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Build a DAG from a pipeline module.
|
|
2
|
+
|
|
3
|
+
A pipeline is just a Python module (DESIGN.md sec 5); DuckPipe discovers
|
|
4
|
+
its tasks by importing it once and scanning the module namespace for
|
|
5
|
+
``Task`` instances -- module-level attributes directly, and one level into
|
|
6
|
+
list/tuple/set/dict values, which is where the "plain Python loop
|
|
7
|
+
generating uniquely identified task instances" fan-out pattern (sec 4)
|
|
8
|
+
naturally puts its tasks.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import importlib.util
|
|
14
|
+
import sys
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from types import ModuleType
|
|
18
|
+
|
|
19
|
+
from duckpipe.task import Task
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class CycleError(RuntimeError):
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class DuplicateTaskNameError(RuntimeError):
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class DAG:
|
|
32
|
+
module: ModuleType
|
|
33
|
+
tasks: dict[str, Task] = field(default_factory=dict)
|
|
34
|
+
|
|
35
|
+
def topological_order(self) -> list[Task]:
|
|
36
|
+
indegree = {name: 0 for name in self.tasks}
|
|
37
|
+
children: dict[str, list[str]] = {name: [] for name in self.tasks}
|
|
38
|
+
for t in self.tasks.values():
|
|
39
|
+
for up in t.upstream_tasks():
|
|
40
|
+
indegree[t.name] += 1
|
|
41
|
+
children[up.name].append(t.name)
|
|
42
|
+
|
|
43
|
+
ready = sorted(name for name, deg in indegree.items() if deg == 0)
|
|
44
|
+
order: list[Task] = []
|
|
45
|
+
while ready:
|
|
46
|
+
name = ready.pop(0)
|
|
47
|
+
order.append(self.tasks[name])
|
|
48
|
+
newly_ready = []
|
|
49
|
+
for child in children[name]:
|
|
50
|
+
indegree[child] -= 1
|
|
51
|
+
if indegree[child] == 0:
|
|
52
|
+
newly_ready.append(child)
|
|
53
|
+
ready = sorted(ready + newly_ready)
|
|
54
|
+
|
|
55
|
+
if len(order) != len(self.tasks):
|
|
56
|
+
remaining = sorted(set(self.tasks) - {t.name for t in order})
|
|
57
|
+
raise CycleError(f"cycle detected among tasks: {remaining}")
|
|
58
|
+
return order
|
|
59
|
+
|
|
60
|
+
def roots(self) -> list[Task]:
|
|
61
|
+
return [t for t in self.tasks.values() if not t.upstream_tasks()]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _discover_tasks(module: ModuleType) -> dict[str, Task]:
|
|
65
|
+
found: dict[str, Task] = {}
|
|
66
|
+
|
|
67
|
+
def add(t: Task) -> None:
|
|
68
|
+
if t.name in found and found[t.name] is not t:
|
|
69
|
+
raise DuplicateTaskNameError(
|
|
70
|
+
f"two distinct tasks are both named {t.name!r}; "
|
|
71
|
+
"give one an explicit name=... to disambiguate"
|
|
72
|
+
)
|
|
73
|
+
found[t.name] = t
|
|
74
|
+
|
|
75
|
+
for value in vars(module).values():
|
|
76
|
+
if isinstance(value, Task):
|
|
77
|
+
add(value)
|
|
78
|
+
elif isinstance(value, list | tuple | set):
|
|
79
|
+
for item in value:
|
|
80
|
+
if isinstance(item, Task):
|
|
81
|
+
add(item)
|
|
82
|
+
elif isinstance(value, dict):
|
|
83
|
+
for item in value.values():
|
|
84
|
+
if isinstance(item, Task):
|
|
85
|
+
add(item)
|
|
86
|
+
|
|
87
|
+
return found
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _package_context(path: Path) -> tuple[Path, str] | None:
|
|
91
|
+
"""If ``path`` sits inside a real Python package (an ``__init__.py``
|
|
92
|
+
beside it, and so on up the tree), return ``(root_parent, dotted_name)``
|
|
93
|
+
so it can be imported through the normal import system -- the only way
|
|
94
|
+
a relative import between sibling task files (``from .extract import
|
|
95
|
+
extract``) resolves correctly. Returns ``None`` for a standalone
|
|
96
|
+
script (no ``__init__.py`` beside it) -- the common case, and the one
|
|
97
|
+
``load_module`` leaves entirely unchanged.
|
|
98
|
+
"""
|
|
99
|
+
if not (path.parent / "__init__.py").exists():
|
|
100
|
+
return None
|
|
101
|
+
parts = [] if path.name == "__init__.py" else [path.stem]
|
|
102
|
+
current = path.parent
|
|
103
|
+
while (current / "__init__.py").exists():
|
|
104
|
+
parts.append(current.name)
|
|
105
|
+
current = current.parent
|
|
106
|
+
parts.reverse()
|
|
107
|
+
return current, ".".join(parts)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def load_module(path: str | Path) -> ModuleType:
|
|
111
|
+
"""Import a pipeline file. A plain standalone script (no adjacent
|
|
112
|
+
``__init__.py``) always loads fresh, under a unique synthetic name --
|
|
113
|
+
calling this twice on the same edited-in-place file never serves a
|
|
114
|
+
stale cached version. A file that's a real package member instead
|
|
115
|
+
goes through ``importlib.import_module`` so relative imports between
|
|
116
|
+
sibling task-definition files resolve exactly like any other Python
|
|
117
|
+
package -- ordinary ``sys.modules`` caching then applies too, the
|
|
118
|
+
same as importing that package any other way. Splitting a pipeline's
|
|
119
|
+
tasks across multiple files needs no DuckPipe-specific mechanism
|
|
120
|
+
either way (DESIGN.md tenet #3): it's just Python composition, with
|
|
121
|
+
one entrypoint module DuckPipe is pointed at.
|
|
122
|
+
"""
|
|
123
|
+
path = Path(path).resolve()
|
|
124
|
+
package = _package_context(path)
|
|
125
|
+
if package is not None:
|
|
126
|
+
root_parent, dotted_name = package
|
|
127
|
+
root_parent_str = str(root_parent)
|
|
128
|
+
if root_parent_str not in sys.path:
|
|
129
|
+
sys.path.insert(0, root_parent_str)
|
|
130
|
+
return importlib.import_module(dotted_name)
|
|
131
|
+
|
|
132
|
+
module_name = f"duckpipe_pipeline_{path.stem}_{abs(hash(str(path)))}"
|
|
133
|
+
spec = importlib.util.spec_from_file_location(module_name, path)
|
|
134
|
+
if spec is None or spec.loader is None:
|
|
135
|
+
raise ImportError(f"could not load pipeline module from {path}")
|
|
136
|
+
module = importlib.util.module_from_spec(spec)
|
|
137
|
+
sys.modules[module_name] = module
|
|
138
|
+
spec.loader.exec_module(module)
|
|
139
|
+
return module
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def build_dag(source: str | Path | ModuleType) -> DAG:
|
|
143
|
+
"""Import (if needed) a pipeline and resolve its task graph.
|
|
144
|
+
|
|
145
|
+
Raises ``CycleError`` immediately if the graph is malformed, so callers
|
|
146
|
+
never have to discover that mid-run.
|
|
147
|
+
"""
|
|
148
|
+
module = source if isinstance(source, ModuleType) else load_module(source)
|
|
149
|
+
tasks = _discover_tasks(module)
|
|
150
|
+
|
|
151
|
+
# A task may reference an upstream Task that isn't itself bound to a
|
|
152
|
+
# module-level name (e.g. constructed inline) -- pull those in too so
|
|
153
|
+
# topological_order() sees the full graph.
|
|
154
|
+
frontier = list(tasks.values())
|
|
155
|
+
while frontier:
|
|
156
|
+
t = frontier.pop()
|
|
157
|
+
for up in t.upstream_tasks():
|
|
158
|
+
if up.name not in tasks:
|
|
159
|
+
tasks[up.name] = up
|
|
160
|
+
frontier.append(up)
|
|
161
|
+
|
|
162
|
+
dag = DAG(module=module, tasks=tasks)
|
|
163
|
+
dag.topological_order() # raises CycleError early if malformed
|
|
164
|
+
return dag
|
duckpipe/fingerprint.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Content fingerprinting for tasks (DESIGN.md tenet #6, open question #2).
|
|
2
|
+
|
|
3
|
+
A task's fingerprint hashes together:
|
|
4
|
+
|
|
5
|
+
- its own function source code
|
|
6
|
+
- its declarative config (retries, cache, cache_backend, depends_on names)
|
|
7
|
+
- the resolved fingerprints of its upstream tasks
|
|
8
|
+
- any values in ``extra_fingerprint`` the user opted to include
|
|
9
|
+
|
|
10
|
+
It deliberately never hashes a task's *return value* -- fingerprinting is
|
|
11
|
+
data-blind, per tenet #2. One accepted consequence (open question #2,
|
|
12
|
+
resolved for v1): an upstream *external* state change -- a source file
|
|
13
|
+
that changed on disk without any task code changing -- is invisible to
|
|
14
|
+
it. ``extra_fingerprint`` is the documented, opt-in escape hatch: pass
|
|
15
|
+
anything hashable (a file's mtime, a config dict, an API version string)
|
|
16
|
+
that should also invalidate the cache when it changes.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import hashlib
|
|
22
|
+
import inspect
|
|
23
|
+
from collections.abc import Callable
|
|
24
|
+
from typing import TYPE_CHECKING, Any
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from duckpipe.task import Task
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def fingerprint_source(func: Callable[..., Any]) -> str:
|
|
31
|
+
try:
|
|
32
|
+
source = inspect.getsource(func)
|
|
33
|
+
except (OSError, TypeError):
|
|
34
|
+
# Dynamically generated function with no retrievable source (e.g.
|
|
35
|
+
# built via exec/eval) -- fall back to bytecode identity.
|
|
36
|
+
source = repr(func.__code__.co_code)
|
|
37
|
+
return hashlib.sha256(source.encode("utf-8")).hexdigest()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _combine(*parts: str) -> str:
|
|
41
|
+
h = hashlib.sha256()
|
|
42
|
+
for p in parts:
|
|
43
|
+
h.update(p.encode("utf-8"))
|
|
44
|
+
h.update(b"\0")
|
|
45
|
+
return h.hexdigest()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def fingerprint_task(task: Task, upstream_fingerprints: dict[str, str]) -> str:
|
|
49
|
+
config_repr = repr(
|
|
50
|
+
(
|
|
51
|
+
task.retries,
|
|
52
|
+
task.cache,
|
|
53
|
+
task.cache_backend,
|
|
54
|
+
sorted(t.name for t in task.depends_on),
|
|
55
|
+
[repr(v) for v in task.extra_fingerprint],
|
|
56
|
+
)
|
|
57
|
+
)
|
|
58
|
+
upstream_repr = [f"{name}={fp}" for name, fp in sorted(upstream_fingerprints.items())]
|
|
59
|
+
return _combine(task.source_fingerprint, config_repr, *upstream_repr)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def resolve_fingerprints(order: list[Task]) -> dict[str, str]:
|
|
63
|
+
"""Compute every task's fingerprint in topological order."""
|
|
64
|
+
resolved: dict[str, str] = {}
|
|
65
|
+
for t in order:
|
|
66
|
+
upstream_fps = {up.name: resolved[up.name] for up in t.upstream_tasks()}
|
|
67
|
+
resolved[t.name] = fingerprint_task(t, upstream_fps)
|
|
68
|
+
return resolved
|