vibedata-dlt-core 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ dist/
5
+ build/
6
+ .venv/
7
+ .env
8
+ .pytest_cache/
@@ -0,0 +1,25 @@
1
+ Metadata-Version: 2.4
2
+ Name: vibedata-dlt-core
3
+ Version: 0.1.0
4
+ Summary: VibeData dlt runtime — cross-platform shared core (audit + pipeline context). Depended on by every platform's runtime; never installed to run a pipeline on its own.
5
+ Author-email: Vibedata <eng@acceleratedata.ai>
6
+ License: MIT
7
+ Requires-Python: >=3.11
8
+ Requires-Dist: dlt>=1.0.0
9
+ Provides-Extra: dev
10
+ Requires-Dist: pytest-mock>=3.12; extra == 'dev'
11
+ Requires-Dist: pytest>=8.0; extra == 'dev'
12
+ Description-Content-Type: text/markdown
13
+
14
+ # vibedata-dlt-core
15
+
16
+ Cross-platform shared core for the VibeData dlt runtime. Provides the
17
+ platform-agnostic building blocks every execution unit depends on:
18
+
19
+ - **audit** — extract dlt trace metrics into the `audit_runs` / `audit_table_loads` delta tables.
20
+ - **pipeline_context** — resolve the run's pipeline/connection context from the project's dlt config.
21
+
22
+ Import namespace: `vibedata.dlt.core`. This is a dependency package — it is not
23
+ installed to run a pipeline on its own. Platform cores
24
+ (`vibedata-dlt-fabric-core`, `vibedata-dlt-duckdb-core`) and the execution-unit
25
+ distributions build on it.
@@ -0,0 +1,12 @@
1
+ # vibedata-dlt-core
2
+
3
+ Cross-platform shared core for the VibeData dlt runtime. Provides the
4
+ platform-agnostic building blocks every execution unit depends on:
5
+
6
+ - **audit** — extract dlt trace metrics into the `audit_runs` / `audit_table_loads` delta tables.
7
+ - **pipeline_context** — resolve the run's pipeline/connection context from the project's dlt config.
8
+
9
+ Import namespace: `vibedata.dlt.core`. This is a dependency package — it is not
10
+ installed to run a pipeline on its own. Platform cores
11
+ (`vibedata-dlt-fabric-core`, `vibedata-dlt-duckdb-core`) and the execution-unit
12
+ distributions build on it.
@@ -0,0 +1,29 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "vibedata-dlt-core"
7
+ version = "0.1.0"
8
+ description = "VibeData dlt runtime — cross-platform shared core (audit + pipeline context). Depended on by every platform's runtime; never installed to run a pipeline on its own."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Vibedata", email = "eng@acceleratedata.ai" }]
13
+
14
+ dependencies = [
15
+ "dlt>=1.0.0",
16
+ ]
17
+
18
+ [project.optional-dependencies]
19
+ dev = ["pytest>=8.0", "pytest-mock>=3.12"]
20
+
21
+ # PEP 420 namespace package: `vibedata` and `vibedata.dlt` carry no __init__.py
22
+ # so other distributions (vibedata-dlt-fabric-core, vibedata-dlt-duckdb-core, …)
23
+ # can contribute their own subpackages to the same namespace.
24
+ [tool.hatch.build.targets.wheel]
25
+ packages = ["src/vibedata"]
26
+
27
+ [tool.pytest.ini_options]
28
+ testpaths = ["tests"]
29
+ pythonpath = ["src"]
@@ -0,0 +1,25 @@
1
+ """VibeData dlt runtime — cross-platform shared core.
2
+
3
+ Platform-agnostic building blocks shared by every execution unit:
4
+
5
+ - ``audit``: extract dlt trace metrics into the ``audit_runs`` /
6
+ ``audit_table_loads`` delta tables.
7
+ - ``pipeline_context``: resolve the run's pipeline/connection context from
8
+ the project's dlt config.
9
+
10
+ Platform packages (``vibedata.dlt.fabric_core``, ``vibedata.dlt.duckdb_core``)
11
+ and the execution-unit facades (``vibedata.dlt.fabric``, ``vibedata.dlt.duckdb``)
12
+ build on these. This package is a dependency, not a runnable distribution.
13
+ """
14
+
15
+ from .audit import build_audit_data, persist_audit
16
+ from .pipeline_context import PipelineContext, load_pipeline_context
17
+ from .tuning import register_pipeline_tuning
18
+
19
+ __all__ = [
20
+ "build_audit_data",
21
+ "persist_audit",
22
+ "PipelineContext",
23
+ "load_pipeline_context",
24
+ "register_pipeline_tuning",
25
+ ]
@@ -0,0 +1,231 @@
1
+ """
2
+ Audit — Pipeline run tracking via delta tables.
3
+
4
+ Extracts runtime metrics from dlt's in-memory trace object and persists
5
+ them into two append-only delta tables:
6
+
7
+ - audit_runs: one row per pipeline run (timing, status, row counts)
8
+ - audit_table_loads: one row per table per run (per-table metrics)
9
+
10
+ Both tables share `load_id` as a foreign key.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import logging
16
+ import uuid
17
+ from typing import Any, Optional
18
+
19
+ logger = logging.getLogger(__name__)
20
+
21
+
22
+ def build_audit_data(
23
+ trace: Any,
24
+ error_message: Optional[str] = None,
25
+ ) -> tuple[dict[str, Any], list[dict[str, Any]]]:
26
+ """Extract audit_runs row and audit_table_loads rows from a pipeline trace.
27
+
28
+ Handles partial traces gracefully — when pipeline.run() throws during
29
+ extract/normalize/load, the trace step's step_info is the Exception
30
+ object (no .metrics attribute). Those steps are skipped safely.
31
+
32
+ Args:
33
+ trace: The captured pipeline trace (pipeline.last_trace).
34
+ error_message: Optional error message from a caught exception.
35
+
36
+ Returns:
37
+ (audit_run dict, list of audit_table_loads dicts)
38
+ """
39
+ extract_metrics: dict[str, dict] = {}
40
+ normalize_metrics: dict[str, dict] = {}
41
+ load_job_metrics: dict[str, dict] = {}
42
+ step_timing: dict[str, dict] = {}
43
+ load_id = None
44
+ pipeline_name = None
45
+ destination_name = None
46
+ dataset_name = None
47
+ first_run = None
48
+
49
+ for step in trace.steps:
50
+ si = step.step_info
51
+ step_name = step.step
52
+
53
+ if step_name == "run":
54
+ continue
55
+
56
+ step_timing[step_name] = {
57
+ "started_at": step.started_at.isoformat() if step.started_at else None,
58
+ "finished_at": step.finished_at.isoformat() if step.finished_at else None,
59
+ }
60
+
61
+ if not hasattr(si, "metrics"):
62
+ continue
63
+
64
+ for lid, metric_list in si.metrics.items():
65
+ if load_id is None:
66
+ load_id = lid
67
+
68
+ for m in metric_list:
69
+ if step_name == "extract" and "table_metrics" in m:
70
+ for tname, tm in m["table_metrics"].items():
71
+ if tname.startswith("_dlt_"):
72
+ continue
73
+ extract_metrics[tname] = {
74
+ "items_count": tm.items_count,
75
+ "file_size": tm.file_size,
76
+ }
77
+
78
+ elif step_name == "normalize" and "table_metrics" in m:
79
+ for tname, tm in m["table_metrics"].items():
80
+ if tname.startswith("_dlt_"):
81
+ continue
82
+ normalize_metrics[tname] = {
83
+ "items_count": tm.items_count,
84
+ "file_size": tm.file_size,
85
+ }
86
+
87
+ elif step_name == "load" and "job_metrics" in m:
88
+ for jid, jm in m["job_metrics"].items():
89
+ tname = jm.table_name
90
+ if tname.startswith("_dlt_"):
91
+ continue
92
+ if tname not in load_job_metrics:
93
+ load_job_metrics[tname] = {
94
+ "state": jm.state,
95
+ "elapsed": 0.0,
96
+ "failed_message": None,
97
+ }
98
+ load_job_metrics[tname]["elapsed"] += (
99
+ (jm.finished_at - jm.started_at).total_seconds()
100
+ if jm.started_at and jm.finished_at
101
+ else 0.0
102
+ )
103
+ if jm.state != "completed":
104
+ load_job_metrics[tname]["state"] = jm.state
105
+ if hasattr(jm, "failed_message") and jm.failed_message:
106
+ load_job_metrics[tname]["failed_message"] = str(
107
+ jm.failed_message
108
+ )
109
+
110
+ if step_name == "extract" and pipeline_name is None:
111
+ try:
112
+ pipeline_name = si.pipeline.pipeline_name
113
+ first_run = si.first_run
114
+ except AttributeError:
115
+ pass
116
+ if step_name == "load":
117
+ try:
118
+ destination_name = si.destination_name
119
+ dataset_name = si.dataset_name
120
+ except AttributeError:
121
+ pass
122
+
123
+ if load_id is None:
124
+ load_id = f"failed_{uuid.uuid4().hex[:16]}"
125
+
126
+ failed_tables = [
127
+ t for t, m in load_job_metrics.items() if m["state"] != "completed"
128
+ ]
129
+ if error_message:
130
+ status = "failed"
131
+ else:
132
+ status = "failed" if failed_tables else "completed"
133
+
134
+ table_error_messages = [
135
+ load_job_metrics[t]["failed_message"]
136
+ for t in failed_tables
137
+ if load_job_metrics[t]["failed_message"]
138
+ ]
139
+ all_error_parts = []
140
+ if error_message:
141
+ all_error_parts.append(error_message)
142
+ all_error_parts.extend(table_error_messages)
143
+
144
+ total_rows = sum(m["items_count"] for m in extract_metrics.values())
145
+
146
+ audit_run = {
147
+ "load_id": load_id,
148
+ "pipeline_name": pipeline_name,
149
+ "destination_name": destination_name,
150
+ "dataset_name": dataset_name,
151
+ "first_run": first_run,
152
+ "status": status,
153
+ "started_at": trace.started_at.isoformat() if trace.started_at else None,
154
+ "finished_at": trace.finished_at.isoformat() if trace.finished_at else None,
155
+ "extract_started_at": step_timing.get("extract", {}).get("started_at"),
156
+ "extract_finished_at": step_timing.get("extract", {}).get("finished_at"),
157
+ "normalize_started_at": step_timing.get("normalize", {}).get("started_at"),
158
+ "normalize_finished_at": step_timing.get("normalize", {}).get("finished_at"),
159
+ "load_started_at": step_timing.get("load", {}).get("started_at"),
160
+ "load_finished_at": step_timing.get("load", {}).get("finished_at"),
161
+ "total_rows_extracted": total_rows,
162
+ "error_message": "; ".join(all_error_parts) if all_error_parts else None,
163
+ }
164
+
165
+ all_tables = set(extract_metrics) | set(normalize_metrics) | set(load_job_metrics)
166
+
167
+ audit_table_loads = []
168
+ for tname in sorted(all_tables):
169
+ row = {
170
+ "load_id": load_id,
171
+ "table_name": tname,
172
+ "rows_extracted": extract_metrics.get(tname, {}).get("items_count"),
173
+ "rows_normalized": normalize_metrics.get(tname, {}).get("items_count"),
174
+ "file_size": normalize_metrics.get(tname, {}).get("file_size"),
175
+ "load_state": load_job_metrics.get(tname, {}).get("state"),
176
+ "load_elapsed": load_job_metrics.get(tname, {}).get("elapsed"),
177
+ "failed_message": load_job_metrics.get(tname, {}).get("failed_message"),
178
+ }
179
+ audit_table_loads.append(row)
180
+
181
+ return audit_run, audit_table_loads
182
+
183
+
184
+ def persist_audit(
185
+ pipeline: Any,
186
+ trace: Any,
187
+ table_format: Optional[str] = "delta",
188
+ error_message: Optional[str] = None,
189
+ ) -> dict[str, Any]:
190
+ """Build audit data from trace and persist to audit tables.
191
+
192
+ Args:
193
+ pipeline: The dlt pipeline instance.
194
+ trace: The captured pipeline trace (pipeline.last_trace).
195
+ table_format: Table format (default "delta").
196
+ error_message: Optional error message from a caught exception.
197
+
198
+ Returns:
199
+ Dict with audit summary (load_id, status, tables_tracked).
200
+ """
201
+ audit_run, audit_table_loads = build_audit_data(trace, error_message=error_message)
202
+
203
+ run_kwargs: dict[str, Any] = {
204
+ "table_name": "audit_runs",
205
+ "write_disposition": "append",
206
+ }
207
+ if table_format:
208
+ run_kwargs["table_format"] = table_format
209
+ pipeline.run([audit_run], **run_kwargs)
210
+
211
+ if audit_table_loads:
212
+ load_kwargs: dict[str, Any] = {
213
+ "table_name": "audit_table_loads",
214
+ "write_disposition": "append",
215
+ }
216
+ if table_format:
217
+ load_kwargs["table_format"] = table_format
218
+ pipeline.run(audit_table_loads, **load_kwargs)
219
+
220
+ logger.info(
221
+ "Audit persisted: load_id=%s status=%s tables=%d",
222
+ audit_run["load_id"],
223
+ audit_run["status"],
224
+ len(audit_table_loads),
225
+ )
226
+
227
+ return {
228
+ "load_id": audit_run["load_id"],
229
+ "status": audit_run["status"],
230
+ "tables_tracked": len(audit_table_loads),
231
+ }
@@ -0,0 +1,67 @@
1
+ """
2
+ Pipeline-folder context — the run contract for the two-registry layout:
3
+
4
+ ingestion/
5
+ ├── connections/<connection_name>/.dlt/config.toml # dlt project dir at run time
6
+ └── pipelines/<pipeline_name>/
7
+ ├── pipeline.py # pure logic; runs with cwd = this folder
8
+ └── config.toml # [pipeline].connection + optional dlt tuning tables
9
+
10
+ `load_pipeline_context()` reads ./config.toml, resolves the referenced
11
+ connection folder, and returns the tuning tables (everything except the
12
+ SDK-owned [pipeline] table, which must never resolve as dlt config).
13
+ Returns None when there is no pipeline config — callers fall back to the
14
+ legacy single-project behavior.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import os
20
+ import tomllib
21
+ from dataclasses import dataclass
22
+ from typing import Any
23
+
24
+ PIPELINE_CONFIG_FILE = 'config.toml'
25
+
26
+
27
+ @dataclass
28
+ class PipelineContext:
29
+ connection_name: str
30
+ """Absolute path to ingestion/connections/<name>/ — the DLT_PROJECT_DIR."""
31
+ connection_dir: str
32
+ """dlt tuning tables from the pipeline config ([pipeline] stripped)."""
33
+ tuning: dict[str, Any]
34
+
35
+
36
+ def load_pipeline_context(cwd: str = '.') -> PipelineContext | None:
37
+ config_path = os.path.join(cwd, PIPELINE_CONFIG_FILE)
38
+ try:
39
+ with open(config_path, 'rb') as f:
40
+ cfg = tomllib.load(f)
41
+ except FileNotFoundError:
42
+ return None
43
+
44
+ pipeline_table = cfg.get('pipeline')
45
+ if not isinstance(pipeline_table, dict):
46
+ return None
47
+ connection_name = pipeline_table.get('connection')
48
+ if not isinstance(connection_name, str) or not connection_name.strip():
49
+ return None
50
+ connection_name = connection_name.strip()
51
+
52
+ # Fail fast on a dangling reference — referential integrity invariant.
53
+ connection_dir = os.path.abspath(
54
+ os.path.join(cwd, '..', '..', 'connections', connection_name)
55
+ )
56
+ if not os.path.isdir(connection_dir):
57
+ raise FileNotFoundError(
58
+ f"Pipeline references connection '{connection_name}' but "
59
+ f'{connection_dir} does not exist.'
60
+ )
61
+
62
+ tuning = {key: value for key, value in cfg.items() if key != 'pipeline'}
63
+ return PipelineContext(
64
+ connection_name=connection_name,
65
+ connection_dir=connection_dir,
66
+ tuning=tuning,
67
+ )
@@ -0,0 +1,24 @@
1
+ """Cross-platform pipeline-tuning provider.
2
+
3
+ The pipeline's ``config.toml`` tuning tables (everything except the SDK-owned
4
+ ``[pipeline]`` table) register as the LOWEST-precedence dlt config provider, so
5
+ ``env > connection toml > pipeline toml``. Identical for every platform, so it
6
+ lives in core rather than each platform's setup flow.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+ from typing import Any
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ def register_pipeline_tuning(tuning: dict[str, Any]) -> None:
18
+ """Register the pipeline config.toml tuning tables as a dlt config provider."""
19
+ import dlt
20
+ from dlt.common.configuration.providers import CustomLoaderDocProvider
21
+
22
+ provider = CustomLoaderDocProvider("pipeline_config", lambda: tuning, supports_secrets=False)
23
+ dlt.config.register_provider(provider)
24
+ logger.info("Registered pipeline_config provider (%d tables)", len(tuning))
@@ -0,0 +1,97 @@
1
+ """Tests for audit module."""
2
+
3
+ from unittest.mock import MagicMock
4
+ from datetime import datetime, timezone
5
+
6
+ from vibedata.dlt.core.audit import build_audit_data
7
+
8
+
9
+ def _make_mock_trace(*, has_error=False):
10
+ """Create a minimal mock dlt trace for testing."""
11
+ trace = MagicMock()
12
+ trace.started_at = datetime(2024, 1, 1, 12, 0, 0, tzinfo=timezone.utc)
13
+ trace.finished_at = datetime(2024, 1, 1, 12, 5, 0, tzinfo=timezone.utc)
14
+
15
+ # Extract step
16
+ extract_step = MagicMock()
17
+ extract_step.step = "extract"
18
+ extract_step.started_at = trace.started_at
19
+ extract_step.finished_at = datetime(2024, 1, 1, 12, 1, 0, tzinfo=timezone.utc)
20
+
21
+ extract_info = MagicMock()
22
+ extract_info.pipeline.pipeline_name = "test_pipeline"
23
+ extract_info.first_run = True
24
+
25
+ table_metric = MagicMock()
26
+ table_metric.items_count = 100
27
+ table_metric.file_size = 5000
28
+
29
+ extract_info.metrics = {
30
+ "load_1": [{"table_metrics": {"contacts": table_metric}}]
31
+ }
32
+ extract_step.step_info = extract_info
33
+
34
+ # Load step
35
+ load_step = MagicMock()
36
+ load_step.step = "load"
37
+ load_step.started_at = datetime(2024, 1, 1, 12, 3, 0, tzinfo=timezone.utc)
38
+ load_step.finished_at = datetime(2024, 1, 1, 12, 5, 0, tzinfo=timezone.utc)
39
+
40
+ load_info = MagicMock()
41
+ load_info.destination_name = "filesystem"
42
+ load_info.dataset_name = "salesforce"
43
+
44
+ job_metric = MagicMock()
45
+ job_metric.table_name = "contacts"
46
+ job_metric.state = "failed" if has_error else "completed"
47
+ job_metric.started_at = load_step.started_at
48
+ job_metric.finished_at = load_step.finished_at
49
+ job_metric.failed_message = "Load error" if has_error else None
50
+
51
+ load_info.metrics = {
52
+ "load_1": [{"job_metrics": {"job_1": job_metric}}]
53
+ }
54
+ load_step.step_info = load_info
55
+
56
+ trace.steps = [extract_step, load_step]
57
+ return trace
58
+
59
+
60
+ class TestBuildAuditData:
61
+ def test_successful_run(self):
62
+ trace = _make_mock_trace()
63
+ audit_run, table_loads = build_audit_data(trace)
64
+
65
+ assert audit_run["status"] == "completed"
66
+ assert audit_run["pipeline_name"] == "test_pipeline"
67
+ assert audit_run["total_rows_extracted"] == 100
68
+ assert audit_run["error_message"] is None
69
+ assert len(table_loads) == 1
70
+ assert table_loads[0]["table_name"] == "contacts"
71
+ assert table_loads[0]["rows_extracted"] == 100
72
+
73
+ def test_failed_run(self):
74
+ trace = _make_mock_trace(has_error=True)
75
+ audit_run, table_loads = build_audit_data(trace)
76
+
77
+ assert audit_run["status"] == "failed"
78
+ assert "Load error" in audit_run["error_message"]
79
+
80
+ def test_forced_error_message(self):
81
+ trace = _make_mock_trace()
82
+ audit_run, _ = build_audit_data(trace, error_message="Pipeline crashed")
83
+
84
+ assert audit_run["status"] == "failed"
85
+ assert "Pipeline crashed" in audit_run["error_message"]
86
+
87
+ def test_empty_trace(self):
88
+ trace = MagicMock()
89
+ trace.started_at = None
90
+ trace.finished_at = None
91
+ trace.steps = []
92
+
93
+ audit_run, table_loads = build_audit_data(trace)
94
+
95
+ assert audit_run["load_id"].startswith("failed_")
96
+ assert audit_run["status"] == "completed"
97
+ assert table_loads == []