duckpipe 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
duckpipe/__init__.py ADDED
@@ -0,0 +1,41 @@
1
+ """DuckPipe: a serverless-first, DuckDB-native pipeline orchestrator.
2
+
3
+ No scheduler daemon, no central metadata database required, no broker
4
+ -- a run is a process that starts, does work, records what it did to a
5
+ ``.duckdb`` file, and exits::
6
+
7
+ from duckpipe import task, run
8
+
9
+ @task
10
+ def extract():
11
+ return duckdb.sql("select * from read_parquet('data.parquet')")
12
+
13
+ @task(cache=True)
14
+ def transform(rel=extract): # `rel=extract` infers the dependency
15
+ return rel.filter("amount > 0")
16
+
17
+ if __name__ == "__main__":
18
+ run(__file__)
19
+
20
+ Then from a shell: ``duckpipe run pipeline.py``.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from duckpipe.dag import DAG, CycleError, build_dag
26
+ from duckpipe.remote import StateLockedError
27
+ from duckpipe.scheduler import RunSummary, run
28
+ from duckpipe.task import Task, task
29
+
30
+ __all__ = [
31
+ "task",
32
+ "run",
33
+ "build_dag",
34
+ "Task",
35
+ "DAG",
36
+ "CycleError",
37
+ "RunSummary",
38
+ "StateLockedError",
39
+ ]
40
+
41
+ __version__ = "0.1.0"
duckpipe/cli.py ADDED
@@ -0,0 +1,393 @@
1
+ """``duckpipe`` command-line interface.
2
+
3
+ Every subcommand is a thin wrapper over the same public functions
4
+ (``duckpipe.run``, ``duckpipe.build_dag``) any other trigger -- cron, CI,
5
+ a Lambda handler -- would call directly (DESIGN.md sec 5, sec 9).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import functools
11
+ import re
12
+ from collections.abc import Callable
13
+ from pathlib import Path
14
+ from typing import Annotated
15
+
16
+ import typer
17
+ from rich.console import Console
18
+ from rich.table import Table
19
+
20
+ from duckpipe.dag import CycleError, DuplicateTaskNameError, build_dag
21
+ from duckpipe.fingerprint import resolve_fingerprints
22
+ from duckpipe.remote import StateLockedError
23
+ from duckpipe.scheduler import (
24
+ UpstreamNotCachedError,
25
+ UpstreamNotReadyError,
26
+ _default_db_path,
27
+ would_skip,
28
+ )
29
+ from duckpipe.scheduler import run as run_pipeline
30
+ from duckpipe.state import is_ducklake
31
+
32
+ app = typer.Typer(
33
+ name="duckpipe",
34
+ help="A serverless-first, DuckDB-native pipeline orchestrator.",
35
+ no_args_is_help=True,
36
+ pretty_exceptions_enable=False,
37
+ )
38
+ console = Console()
39
+
40
+ _STATUS_STYLE = {"success": "green", "skipped": "yellow", "failed": "red"}
41
+
42
+ # A malformed pipeline (a cycle, a duplicate task name, a plain typo that
43
+ # blows up at import time), a locked state_uri, or a scoped (--only) run
44
+ # dispatched out of order is a user-actionable situation, not a DuckPipe
45
+ # bug -- it should read as one short line, not a wall of framework frames.
46
+ _USER_FACING_ERRORS = (
47
+ CycleError,
48
+ DuplicateTaskNameError,
49
+ ImportError,
50
+ SyntaxError,
51
+ StateLockedError,
52
+ UpstreamNotReadyError,
53
+ UpstreamNotCachedError,
54
+ ValueError,
55
+ )
56
+
57
+ _DUCKLAKE_FILE_PREFIXES = ("ducklake:sqlite:", "ducklake:duckdb:")
58
+
59
+
60
+ def _state_probably_exists(db_path: str | Path) -> bool:
61
+ """A cheap existence check that never opens (and so never creates) the
62
+ store, for commands like `show`/`stats` that shouldn't conjure a
63
+ fresh, empty state file just by asking about one that isn't there yet.
64
+ """
65
+ if is_ducklake(db_path):
66
+ for prefix in _DUCKLAKE_FILE_PREFIXES:
67
+ if db_path.startswith(prefix):
68
+ return Path(db_path[len(prefix) :]).exists()
69
+ return True # a live catalog (e.g. Postgres) -- can't check cheaply
70
+ return Path(db_path).exists()
71
+
72
+
73
+ _MERMAID_STATUS_CLASS = {
74
+ "success": "success",
75
+ "skipped": "skipped",
76
+ "failed": "failed",
77
+ "upstream_failed": "failed",
78
+ }
79
+ _MERMAID_CLASS_DEFS = {
80
+ "success": "classDef success fill:#d4f7dc,stroke:#2f9e44,color:#1a1a1a",
81
+ "skipped": "classDef skipped fill:#fff3cd,stroke:#d9a400,color:#1a1a1a",
82
+ "failed": "classDef failed fill:#f8d7da,stroke:#d64545,color:#1a1a1a",
83
+ }
84
+
85
+
86
+ def _mermaid_node_id(name: str) -> str:
87
+ # Task names are usually valid identifiers already (a function's own
88
+ # __name__, or an explicit name=...), but a Mermaid node id has its
89
+ # own, stricter rules -- sanitize rather than assume, so an unusual
90
+ # name= never produces a broken diagram.
91
+ return "t_" + re.sub(r"[^A-Za-z0-9_]", "_", name)
92
+
93
+
94
+ def _mermaid_diagram(order: list, last_status: dict[str, tuple[str, str]]) -> str:
95
+ """A `flowchart TD` a task's real dependency edges, one node per task
96
+ (labeled with its real name, not the sanitized id), colored by last
97
+ recorded status when state exists -- pastes straight into any
98
+ Markdown that renders Mermaid (GitHub, GitLab, many others)."""
99
+ lines = ["flowchart TD"]
100
+ for t in order:
101
+ label = t.name.replace('"', """)
102
+ lines.append(f' {_mermaid_node_id(t.name)}["{label}"]')
103
+ for t in order:
104
+ for up in t.upstream_tasks():
105
+ lines.append(f" {_mermaid_node_id(up.name)} --> {_mermaid_node_id(t.name)}")
106
+
107
+ used_classes: list[str] = []
108
+ for t in order:
109
+ status = last_status.get(t.name, (None, None))[0]
110
+ cls = _MERMAID_STATUS_CLASS.get(status)
111
+ if cls:
112
+ lines.append(f" class {_mermaid_node_id(t.name)} {cls}")
113
+ if cls not in used_classes:
114
+ used_classes.append(cls)
115
+ for cls in used_classes:
116
+ lines.append(f" {_MERMAID_CLASS_DEFS[cls]}")
117
+ return "\n".join(lines)
118
+
119
+
120
+ def _friendly_errors[F: Callable[..., object]](command: F) -> F:
121
+ @functools.wraps(command)
122
+ def wrapper(*args: object, **kwargs: object) -> object:
123
+ try:
124
+ return command(*args, **kwargs)
125
+ except typer.Exit:
126
+ raise
127
+ except _USER_FACING_ERRORS as exc:
128
+ console.print(f"[red]error:[/red] {exc}")
129
+ raise typer.Exit(code=1) from None
130
+ except Exception as exc: # noqa: BLE001 - last-resort CLI-friendly fallback
131
+ console.print(f"[red]error:[/red] {type(exc).__name__}: {exc}")
132
+ raise typer.Exit(code=1) from None
133
+
134
+ return wrapper
135
+
136
+
137
+ @app.command()
138
+ @_friendly_errors
139
+ def run(
140
+ pipeline: Annotated[Path, typer.Argument(exists=True, help="Path to a pipeline .py module")],
141
+ db: Annotated[
142
+ str | None,
143
+ typer.Option(
144
+ "--db",
145
+ help="State file path, or a ducklake:... catalog "
146
+ "(default: duckpipe.db next to the pipeline)",
147
+ ),
148
+ ] = None,
149
+ state_uri: Annotated[
150
+ str | None,
151
+ typer.Option("--state-uri", help="fsspec URI to sync state to/from (S3/GCS/Azure/local)"),
152
+ ] = None,
153
+ force: Annotated[
154
+ bool, typer.Option("--force", help="Ignore fingerprints; re-run every task")
155
+ ] = False,
156
+ max_workers: Annotated[
157
+ int | None, typer.Option("--max-workers", help="Cap concurrent tasks")
158
+ ] = None,
159
+ no_lock: Annotated[
160
+ bool,
161
+ typer.Option(
162
+ "--no-lock",
163
+ help="With --state-uri, skip the advisory lock and allow racing (not recommended)",
164
+ ),
165
+ ] = False,
166
+ only: Annotated[
167
+ str | None,
168
+ typer.Option("--only", help="Run just this one task -- for distributed dispatch"),
169
+ ] = None,
170
+ run_id: Annotated[
171
+ str | None,
172
+ typer.Option("--run-id", help="Shared run id so --only calls for one run group together"),
173
+ ] = None,
174
+ data_path: Annotated[
175
+ str | None,
176
+ typer.Option(
177
+ "--data-path", help="DuckLake DATA_PATH (only with --db ducklake:...; usually auto)"
178
+ ),
179
+ ] = None,
180
+ ) -> None:
181
+ """Run a pipeline module end-to-end, or (with --only) exactly one task."""
182
+ summary = run_pipeline(
183
+ pipeline,
184
+ db_path=db,
185
+ state_uri=state_uri,
186
+ force=force,
187
+ max_workers=max_workers,
188
+ lock=not no_lock,
189
+ only=only,
190
+ run_id=run_id,
191
+ data_path=data_path,
192
+ )
193
+ table = Table(title=f"run {summary.run_id} ({summary.db_path})")
194
+ table.add_column("task")
195
+ table.add_column("status")
196
+ table.add_column("detail")
197
+ for name, status in summary.statuses.items():
198
+ style = _STATUS_STYLE.get(status, "")
199
+ rendered = f"[{style}]{status}[/{style}]" if style else status
200
+ table.add_row(name, rendered, summary.errors.get(name, ""))
201
+ console.print(table)
202
+ if not summary.success:
203
+ raise typer.Exit(code=1)
204
+
205
+
206
+ @app.command()
207
+ @_friendly_errors
208
+ def show(
209
+ pipeline: Annotated[Path, typer.Argument(exists=True, help="Path to a pipeline .py module")],
210
+ db: Annotated[
211
+ str | None, typer.Option("--db", help="State file to read last-run status from")
212
+ ] = None,
213
+ as_json: Annotated[
214
+ bool,
215
+ typer.Option(
216
+ "--json", help="Machine-readable output -- topological order + next-run preview"
217
+ ),
218
+ ] = False,
219
+ mermaid: Annotated[
220
+ bool,
221
+ typer.Option(
222
+ "--mermaid", help="Print a Mermaid flowchart of the DAG, colored by last status"
223
+ ),
224
+ ] = False,
225
+ ) -> None:
226
+ """Print the resolved DAG, each task's last-run status, and what would
227
+ happen if you ran it again right now -- a dry-run preview of the same
228
+ fingerprint check `run` itself uses, so you can see what's stale
229
+ before spending the time to re-run it. `--json` is the discovery
230
+ primitive a coordinator dispatching `--only <task>` calls needs
231
+ (DESIGN.md sec 8): topological order plus which tasks would skip.
232
+ `--mermaid` prints a flowchart instead -- paste it into any Markdown
233
+ that renders Mermaid."""
234
+ if as_json and mermaid:
235
+ raise ValueError("--json and --mermaid are two different output modes -- pick one")
236
+ dag = build_dag(pipeline)
237
+ order = dag.topological_order()
238
+ fingerprints = resolve_fingerprints(order)
239
+ db_path: str | Path = db or _default_db_path(pipeline)
240
+
241
+ last_status: dict[str, tuple[str, str]] = {}
242
+ next_run: dict[str, str] = {}
243
+ if _state_probably_exists(db_path):
244
+ from duckpipe.state import StateStore
245
+
246
+ # read_only: `show` never needs to write, and must never block on
247
+ # (or be blocked by) a pipeline that's still running.
248
+ with StateStore(db_path, read_only=True) as store:
249
+ for t in order:
250
+ row = store.last_status(t.name)
251
+ if row:
252
+ last_status[t.name] = (row[0], str(row[1]))
253
+ if would_skip(store, t, fingerprints[t.name]):
254
+ next_run[t.name] = "skip (unchanged)"
255
+ elif not t.cache:
256
+ next_run[t.name] = "run (not cached)"
257
+ else:
258
+ next_run[t.name] = "run (changed)"
259
+ else:
260
+ next_run = {t.name: "run (no prior state)" for t in order}
261
+
262
+ if mermaid:
263
+ print(_mermaid_diagram(order, last_status))
264
+ return
265
+
266
+ if as_json:
267
+ import json
268
+
269
+ print(
270
+ json.dumps(
271
+ [
272
+ {
273
+ "task": t.name,
274
+ "depends_on": sorted(u.name for u in t.upstream_tasks()),
275
+ "next_run": next_run[t.name],
276
+ }
277
+ for t in order
278
+ ]
279
+ )
280
+ )
281
+ return
282
+
283
+ table = Table(title=f"DAG: {pipeline}")
284
+ table.add_column("task")
285
+ table.add_column("depends on")
286
+ table.add_column("last status")
287
+ table.add_column("last run")
288
+ table.add_column("next run")
289
+ for t in order:
290
+ deps = ", ".join(sorted(u.name for u in t.upstream_tasks())) or "-"
291
+ status, ts = last_status.get(t.name, ("-", "-"))
292
+ style = _STATUS_STYLE.get(status, "")
293
+ rendered = f"[{style}]{status}[/{style}]" if style else status
294
+ next_style = "yellow" if next_run[t.name].startswith("skip") else ""
295
+ next_rendered = (
296
+ f"[{next_style}]{next_run[t.name]}[/{next_style}]" if next_style else next_run[t.name]
297
+ )
298
+ table.add_row(t.name, deps, rendered, ts, next_rendered)
299
+ console.print(table)
300
+
301
+
302
+ @app.command()
303
+ @_friendly_errors
304
+ def compact(
305
+ state_uri: Annotated[str, typer.Argument(help="fsspec URI of the state file to compact")],
306
+ no_lock: Annotated[
307
+ bool, typer.Option("--no-lock", help="Skip the advisory lock (not recommended)")
308
+ ] = False,
309
+ ) -> None:
310
+ """Fold pending per-task deltas from `--only` runs into the canonical
311
+ state file, and clean them up. Nothing needs this for correctness --
312
+ every invocation already absorbs pending deltas itself -- but a purely
313
+ distributed workflow (many `--only` workers, no whole-run ever) never
314
+ otherwise re-uploads the canonical file, so its `.pending/` directory
315
+ only grows. Run this periodically (e.g. from cron) if that's your
316
+ workflow."""
317
+ from duckpipe.remote import compact as compact_state
318
+
319
+ compact_state(state_uri, lock=not no_lock)
320
+ console.print(f"[green]compacted[/green] {state_uri}")
321
+
322
+
323
+ @app.command()
324
+ @_friendly_errors
325
+ def stats(
326
+ db: Annotated[str, typer.Argument(help="Path to a state file, or a ducklake:... catalog")],
327
+ limit: Annotated[int, typer.Option(help="Number of recent runs to show")] = 20,
328
+ snapshots: Annotated[
329
+ bool,
330
+ typer.Option(
331
+ "--snapshots", help="Show DuckLake snapshot history (time travel) instead of stats"
332
+ ),
333
+ ] = False,
334
+ ) -> None:
335
+ """Show recent pipeline runs and per-task duration stats from the state
336
+ file. Against a DuckLake-backed store (``--db ducklake:...``), pass
337
+ ``--snapshots`` to see every commit instead -- the time-travel-over-
338
+ run-history payoff of that backend (DESIGN.md sec 8, Phase 3b)."""
339
+ from duckpipe.state import StateStore
340
+
341
+ # read_only: never blocks on, or is blocked by, a pipeline still
342
+ # running against the same state file.
343
+ with StateStore(db, read_only=True) as store:
344
+ if snapshots:
345
+ _print_snapshots(store)
346
+ return
347
+
348
+ runs = store.con.execute(
349
+ "SELECT run_id, module_path, started_at, ended_at, status "
350
+ "FROM pipeline_runs ORDER BY started_at DESC LIMIT ?",
351
+ [limit],
352
+ ).fetchall()
353
+ runs_table = Table(title="recent pipeline runs")
354
+ for col in ("run_id", "module", "started_at", "ended_at", "status"):
355
+ runs_table.add_column(col)
356
+ for row in runs:
357
+ runs_table.add_row(*(str(v) for v in row))
358
+ console.print(runs_table)
359
+
360
+ task_stats = store.con.execute(
361
+ "SELECT task_name, runs, avg_duration_ms, max_duration_ms, skipped_count, failed_count "
362
+ "FROM v_task_stats ORDER BY avg_duration_ms DESC"
363
+ ).fetchall()
364
+ stats_table = Table(title="per-task stats (from v_task_stats)")
365
+ for col in ("task", "runs", "avg_ms", "max_ms", "skipped", "failed"):
366
+ stats_table.add_column(col)
367
+ for row in task_stats:
368
+ stats_table.add_row(*(f"{v:.1f}" if isinstance(v, float) else str(v) for v in row))
369
+ console.print(stats_table)
370
+
371
+ if store.is_ducklake:
372
+ console.print("[dim]DuckLake-backed -- see full history with --snapshots[/dim]")
373
+
374
+
375
+ def _print_snapshots(store) -> None:
376
+ columns = ("snapshot_id", "snapshot_time", "author", "commit_message")
377
+ table = Table(title="snapshot history (time travel)")
378
+ for col in columns:
379
+ table.add_column(col)
380
+ for snap in store.snapshots():
381
+ table.add_row(*(str(snap[col]) if snap[col] is not None else "-" for col in columns))
382
+ console.print(table)
383
+ console.print(
384
+ "[dim]query any table AT (VERSION => snapshot_id) to see state as of that point[/dim]"
385
+ )
386
+
387
+
388
+ def main() -> None:
389
+ app()
390
+
391
+
392
+ if __name__ == "__main__":
393
+ main()
duckpipe/dag.py ADDED
@@ -0,0 +1,164 @@
1
+ """Build a DAG from a pipeline module.
2
+
3
+ A pipeline is just a Python module (DESIGN.md sec 5); DuckPipe discovers
4
+ its tasks by importing it once and scanning the module namespace for
5
+ ``Task`` instances -- module-level attributes directly, and one level into
6
+ list/tuple/set/dict values, which is where the "plain Python loop
7
+ generating uniquely identified task instances" fan-out pattern (sec 4)
8
+ naturally puts its tasks.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import importlib.util
14
+ import sys
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from types import ModuleType
18
+
19
+ from duckpipe.task import Task
20
+
21
+
22
+ class CycleError(RuntimeError):
23
+ pass
24
+
25
+
26
+ class DuplicateTaskNameError(RuntimeError):
27
+ pass
28
+
29
+
30
+ @dataclass
31
+ class DAG:
32
+ module: ModuleType
33
+ tasks: dict[str, Task] = field(default_factory=dict)
34
+
35
+ def topological_order(self) -> list[Task]:
36
+ indegree = {name: 0 for name in self.tasks}
37
+ children: dict[str, list[str]] = {name: [] for name in self.tasks}
38
+ for t in self.tasks.values():
39
+ for up in t.upstream_tasks():
40
+ indegree[t.name] += 1
41
+ children[up.name].append(t.name)
42
+
43
+ ready = sorted(name for name, deg in indegree.items() if deg == 0)
44
+ order: list[Task] = []
45
+ while ready:
46
+ name = ready.pop(0)
47
+ order.append(self.tasks[name])
48
+ newly_ready = []
49
+ for child in children[name]:
50
+ indegree[child] -= 1
51
+ if indegree[child] == 0:
52
+ newly_ready.append(child)
53
+ ready = sorted(ready + newly_ready)
54
+
55
+ if len(order) != len(self.tasks):
56
+ remaining = sorted(set(self.tasks) - {t.name for t in order})
57
+ raise CycleError(f"cycle detected among tasks: {remaining}")
58
+ return order
59
+
60
+ def roots(self) -> list[Task]:
61
+ return [t for t in self.tasks.values() if not t.upstream_tasks()]
62
+
63
+
64
+ def _discover_tasks(module: ModuleType) -> dict[str, Task]:
65
+ found: dict[str, Task] = {}
66
+
67
+ def add(t: Task) -> None:
68
+ if t.name in found and found[t.name] is not t:
69
+ raise DuplicateTaskNameError(
70
+ f"two distinct tasks are both named {t.name!r}; "
71
+ "give one an explicit name=... to disambiguate"
72
+ )
73
+ found[t.name] = t
74
+
75
+ for value in vars(module).values():
76
+ if isinstance(value, Task):
77
+ add(value)
78
+ elif isinstance(value, list | tuple | set):
79
+ for item in value:
80
+ if isinstance(item, Task):
81
+ add(item)
82
+ elif isinstance(value, dict):
83
+ for item in value.values():
84
+ if isinstance(item, Task):
85
+ add(item)
86
+
87
+ return found
88
+
89
+
90
+ def _package_context(path: Path) -> tuple[Path, str] | None:
91
+ """If ``path`` sits inside a real Python package (an ``__init__.py``
92
+ beside it, and so on up the tree), return ``(root_parent, dotted_name)``
93
+ so it can be imported through the normal import system -- the only way
94
+ a relative import between sibling task files (``from .extract import
95
+ extract``) resolves correctly. Returns ``None`` for a standalone
96
+ script (no ``__init__.py`` beside it) -- the common case, and the one
97
+ ``load_module`` leaves entirely unchanged.
98
+ """
99
+ if not (path.parent / "__init__.py").exists():
100
+ return None
101
+ parts = [] if path.name == "__init__.py" else [path.stem]
102
+ current = path.parent
103
+ while (current / "__init__.py").exists():
104
+ parts.append(current.name)
105
+ current = current.parent
106
+ parts.reverse()
107
+ return current, ".".join(parts)
108
+
109
+
110
+ def load_module(path: str | Path) -> ModuleType:
111
+ """Import a pipeline file. A plain standalone script (no adjacent
112
+ ``__init__.py``) always loads fresh, under a unique synthetic name --
113
+ calling this twice on the same edited-in-place file never serves a
114
+ stale cached version. A file that's a real package member instead
115
+ goes through ``importlib.import_module`` so relative imports between
116
+ sibling task-definition files resolve exactly like any other Python
117
+ package -- ordinary ``sys.modules`` caching then applies too, the
118
+ same as importing that package any other way. Splitting a pipeline's
119
+ tasks across multiple files needs no DuckPipe-specific mechanism
120
+ either way (DESIGN.md tenet #3): it's just Python composition, with
121
+ one entrypoint module DuckPipe is pointed at.
122
+ """
123
+ path = Path(path).resolve()
124
+ package = _package_context(path)
125
+ if package is not None:
126
+ root_parent, dotted_name = package
127
+ root_parent_str = str(root_parent)
128
+ if root_parent_str not in sys.path:
129
+ sys.path.insert(0, root_parent_str)
130
+ return importlib.import_module(dotted_name)
131
+
132
+ module_name = f"duckpipe_pipeline_{path.stem}_{abs(hash(str(path)))}"
133
+ spec = importlib.util.spec_from_file_location(module_name, path)
134
+ if spec is None or spec.loader is None:
135
+ raise ImportError(f"could not load pipeline module from {path}")
136
+ module = importlib.util.module_from_spec(spec)
137
+ sys.modules[module_name] = module
138
+ spec.loader.exec_module(module)
139
+ return module
140
+
141
+
142
+ def build_dag(source: str | Path | ModuleType) -> DAG:
143
+ """Import (if needed) a pipeline and resolve its task graph.
144
+
145
+ Raises ``CycleError`` immediately if the graph is malformed, so callers
146
+ never have to discover that mid-run.
147
+ """
148
+ module = source if isinstance(source, ModuleType) else load_module(source)
149
+ tasks = _discover_tasks(module)
150
+
151
+ # A task may reference an upstream Task that isn't itself bound to a
152
+ # module-level name (e.g. constructed inline) -- pull those in too so
153
+ # topological_order() sees the full graph.
154
+ frontier = list(tasks.values())
155
+ while frontier:
156
+ t = frontier.pop()
157
+ for up in t.upstream_tasks():
158
+ if up.name not in tasks:
159
+ tasks[up.name] = up
160
+ frontier.append(up)
161
+
162
+ dag = DAG(module=module, tasks=tasks)
163
+ dag.topological_order() # raises CycleError early if malformed
164
+ return dag
@@ -0,0 +1,68 @@
1
+ """Content fingerprinting for tasks (DESIGN.md tenet #6, open question #2).
2
+
3
+ A task's fingerprint hashes together:
4
+
5
+ - its own function source code
6
+ - its declarative config (retries, cache, cache_backend, depends_on names)
7
+ - the resolved fingerprints of its upstream tasks
8
+ - any values in ``extra_fingerprint`` the user opted to include
9
+
10
+ It deliberately never hashes a task's *return value* -- fingerprinting is
11
+ data-blind, per tenet #2. One accepted consequence (open question #2,
12
+ resolved for v1): an upstream *external* state change -- a source file
13
+ that changed on disk without any task code changing -- is invisible to
14
+ it. ``extra_fingerprint`` is the documented, opt-in escape hatch: pass
15
+ anything hashable (a file's mtime, a config dict, an API version string)
16
+ that should also invalidate the cache when it changes.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import hashlib
22
+ import inspect
23
+ from collections.abc import Callable
24
+ from typing import TYPE_CHECKING, Any
25
+
26
+ if TYPE_CHECKING:
27
+ from duckpipe.task import Task
28
+
29
+
30
+ def fingerprint_source(func: Callable[..., Any]) -> str:
31
+ try:
32
+ source = inspect.getsource(func)
33
+ except (OSError, TypeError):
34
+ # Dynamically generated function with no retrievable source (e.g.
35
+ # built via exec/eval) -- fall back to bytecode identity.
36
+ source = repr(func.__code__.co_code)
37
+ return hashlib.sha256(source.encode("utf-8")).hexdigest()
38
+
39
+
40
+ def _combine(*parts: str) -> str:
41
+ h = hashlib.sha256()
42
+ for p in parts:
43
+ h.update(p.encode("utf-8"))
44
+ h.update(b"\0")
45
+ return h.hexdigest()
46
+
47
+
48
+ def fingerprint_task(task: Task, upstream_fingerprints: dict[str, str]) -> str:
49
+ config_repr = repr(
50
+ (
51
+ task.retries,
52
+ task.cache,
53
+ task.cache_backend,
54
+ sorted(t.name for t in task.depends_on),
55
+ [repr(v) for v in task.extra_fingerprint],
56
+ )
57
+ )
58
+ upstream_repr = [f"{name}={fp}" for name, fp in sorted(upstream_fingerprints.items())]
59
+ return _combine(task.source_fingerprint, config_repr, *upstream_repr)
60
+
61
+
62
+ def resolve_fingerprints(order: list[Task]) -> dict[str, str]:
63
+ """Compute every task's fingerprint in topological order."""
64
+ resolved: dict[str, str] = {}
65
+ for t in order:
66
+ upstream_fps = {up.name: resolved[up.name] for up in t.upstream_tasks()}
67
+ resolved[t.name] = fingerprint_task(t, upstream_fps)
68
+ return resolved