vibedata-dlt-core 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vibedata_dlt_core-0.1.0/.gitignore +8 -0
- vibedata_dlt_core-0.1.0/PKG-INFO +25 -0
- vibedata_dlt_core-0.1.0/README.md +12 -0
- vibedata_dlt_core-0.1.0/pyproject.toml +29 -0
- vibedata_dlt_core-0.1.0/src/vibedata/dlt/core/__init__.py +25 -0
- vibedata_dlt_core-0.1.0/src/vibedata/dlt/core/audit.py +231 -0
- vibedata_dlt_core-0.1.0/src/vibedata/dlt/core/pipeline_context.py +67 -0
- vibedata_dlt_core-0.1.0/src/vibedata/dlt/core/tuning.py +24 -0
- vibedata_dlt_core-0.1.0/tests/test_audit.py +97 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: vibedata-dlt-core
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: VibeData dlt runtime — cross-platform shared core (audit + pipeline context). Depended on by every platform's runtime; never installed to run a pipeline on its own.
|
|
5
|
+
Author-email: Vibedata <eng@acceleratedata.ai>
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Requires-Dist: dlt>=1.0.0
|
|
9
|
+
Provides-Extra: dev
|
|
10
|
+
Requires-Dist: pytest-mock>=3.12; extra == 'dev'
|
|
11
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
|
|
14
|
+
# vibedata-dlt-core
|
|
15
|
+
|
|
16
|
+
Cross-platform shared core for the VibeData dlt runtime. Provides the
|
|
17
|
+
platform-agnostic building blocks every execution unit depends on:
|
|
18
|
+
|
|
19
|
+
- **audit** — extract dlt trace metrics into the `audit_runs` / `audit_table_loads` delta tables.
|
|
20
|
+
- **pipeline_context** — resolve the run's pipeline/connection context from the project's dlt config.
|
|
21
|
+
|
|
22
|
+
Import namespace: `vibedata.dlt.core`. This is a dependency package — it is not
|
|
23
|
+
installed to run a pipeline on its own. Platform cores
|
|
24
|
+
(`vibedata-dlt-fabric-core`, `vibedata-dlt-duckdb-core`) and the execution-unit
|
|
25
|
+
distributions build on it.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# vibedata-dlt-core
|
|
2
|
+
|
|
3
|
+
Cross-platform shared core for the VibeData dlt runtime. Provides the
|
|
4
|
+
platform-agnostic building blocks every execution unit depends on:
|
|
5
|
+
|
|
6
|
+
- **audit** — extract dlt trace metrics into the `audit_runs` / `audit_table_loads` delta tables.
|
|
7
|
+
- **pipeline_context** — resolve the run's pipeline/connection context from the project's dlt config.
|
|
8
|
+
|
|
9
|
+
Import namespace: `vibedata.dlt.core`. This is a dependency package — it is not
|
|
10
|
+
installed to run a pipeline on its own. Platform cores
|
|
11
|
+
(`vibedata-dlt-fabric-core`, `vibedata-dlt-duckdb-core`) and the execution-unit
|
|
12
|
+
distributions build on it.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "vibedata-dlt-core"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "VibeData dlt runtime — cross-platform shared core (audit + pipeline context). Depended on by every platform's runtime; never installed to run a pipeline on its own."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Vibedata", email = "eng@acceleratedata.ai" }]
|
|
13
|
+
|
|
14
|
+
dependencies = [
|
|
15
|
+
"dlt>=1.0.0",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
dev = ["pytest>=8.0", "pytest-mock>=3.12"]
|
|
20
|
+
|
|
21
|
+
# PEP 420 namespace package: `vibedata` and `vibedata.dlt` carry no __init__.py
|
|
22
|
+
# so other distributions (vibedata-dlt-fabric-core, vibedata-dlt-duckdb-core, …)
|
|
23
|
+
# can contribute their own subpackages to the same namespace.
|
|
24
|
+
[tool.hatch.build.targets.wheel]
|
|
25
|
+
packages = ["src/vibedata"]
|
|
26
|
+
|
|
27
|
+
[tool.pytest.ini_options]
|
|
28
|
+
testpaths = ["tests"]
|
|
29
|
+
pythonpath = ["src"]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""VibeData dlt runtime — cross-platform shared core.
|
|
2
|
+
|
|
3
|
+
Platform-agnostic building blocks shared by every execution unit:
|
|
4
|
+
|
|
5
|
+
- ``audit``: extract dlt trace metrics into the ``audit_runs`` /
|
|
6
|
+
``audit_table_loads`` delta tables.
|
|
7
|
+
- ``pipeline_context``: resolve the run's pipeline/connection context from
|
|
8
|
+
the project's dlt config.
|
|
9
|
+
|
|
10
|
+
Platform packages (``vibedata.dlt.fabric_core``, ``vibedata.dlt.duckdb_core``)
|
|
11
|
+
and the execution-unit facades (``vibedata.dlt.fabric``, ``vibedata.dlt.duckdb``)
|
|
12
|
+
build on these. This package is a dependency, not a runnable distribution.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from .audit import build_audit_data, persist_audit
|
|
16
|
+
from .pipeline_context import PipelineContext, load_pipeline_context
|
|
17
|
+
from .tuning import register_pipeline_tuning
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"build_audit_data",
|
|
21
|
+
"persist_audit",
|
|
22
|
+
"PipelineContext",
|
|
23
|
+
"load_pipeline_context",
|
|
24
|
+
"register_pipeline_tuning",
|
|
25
|
+
]
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Audit — Pipeline run tracking via delta tables.
|
|
3
|
+
|
|
4
|
+
Extracts runtime metrics from dlt's in-memory trace object and persists
|
|
5
|
+
them into two append-only delta tables:
|
|
6
|
+
|
|
7
|
+
- audit_runs: one row per pipeline run (timing, status, row counts)
|
|
8
|
+
- audit_table_loads: one row per table per run (per-table metrics)
|
|
9
|
+
|
|
10
|
+
Both tables share `load_id` as a foreign key.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
import uuid
|
|
17
|
+
from typing import Any, Optional
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def build_audit_data(
|
|
23
|
+
trace: Any,
|
|
24
|
+
error_message: Optional[str] = None,
|
|
25
|
+
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
|
26
|
+
"""Extract audit_runs row and audit_table_loads rows from a pipeline trace.
|
|
27
|
+
|
|
28
|
+
Handles partial traces gracefully — when pipeline.run() throws during
|
|
29
|
+
extract/normalize/load, the trace step's step_info is the Exception
|
|
30
|
+
object (no .metrics attribute). Those steps are skipped safely.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
trace: The captured pipeline trace (pipeline.last_trace).
|
|
34
|
+
error_message: Optional error message from a caught exception.
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
(audit_run dict, list of audit_table_loads dicts)
|
|
38
|
+
"""
|
|
39
|
+
extract_metrics: dict[str, dict] = {}
|
|
40
|
+
normalize_metrics: dict[str, dict] = {}
|
|
41
|
+
load_job_metrics: dict[str, dict] = {}
|
|
42
|
+
step_timing: dict[str, dict] = {}
|
|
43
|
+
load_id = None
|
|
44
|
+
pipeline_name = None
|
|
45
|
+
destination_name = None
|
|
46
|
+
dataset_name = None
|
|
47
|
+
first_run = None
|
|
48
|
+
|
|
49
|
+
for step in trace.steps:
|
|
50
|
+
si = step.step_info
|
|
51
|
+
step_name = step.step
|
|
52
|
+
|
|
53
|
+
if step_name == "run":
|
|
54
|
+
continue
|
|
55
|
+
|
|
56
|
+
step_timing[step_name] = {
|
|
57
|
+
"started_at": step.started_at.isoformat() if step.started_at else None,
|
|
58
|
+
"finished_at": step.finished_at.isoformat() if step.finished_at else None,
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
if not hasattr(si, "metrics"):
|
|
62
|
+
continue
|
|
63
|
+
|
|
64
|
+
for lid, metric_list in si.metrics.items():
|
|
65
|
+
if load_id is None:
|
|
66
|
+
load_id = lid
|
|
67
|
+
|
|
68
|
+
for m in metric_list:
|
|
69
|
+
if step_name == "extract" and "table_metrics" in m:
|
|
70
|
+
for tname, tm in m["table_metrics"].items():
|
|
71
|
+
if tname.startswith("_dlt_"):
|
|
72
|
+
continue
|
|
73
|
+
extract_metrics[tname] = {
|
|
74
|
+
"items_count": tm.items_count,
|
|
75
|
+
"file_size": tm.file_size,
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
elif step_name == "normalize" and "table_metrics" in m:
|
|
79
|
+
for tname, tm in m["table_metrics"].items():
|
|
80
|
+
if tname.startswith("_dlt_"):
|
|
81
|
+
continue
|
|
82
|
+
normalize_metrics[tname] = {
|
|
83
|
+
"items_count": tm.items_count,
|
|
84
|
+
"file_size": tm.file_size,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
elif step_name == "load" and "job_metrics" in m:
|
|
88
|
+
for jid, jm in m["job_metrics"].items():
|
|
89
|
+
tname = jm.table_name
|
|
90
|
+
if tname.startswith("_dlt_"):
|
|
91
|
+
continue
|
|
92
|
+
if tname not in load_job_metrics:
|
|
93
|
+
load_job_metrics[tname] = {
|
|
94
|
+
"state": jm.state,
|
|
95
|
+
"elapsed": 0.0,
|
|
96
|
+
"failed_message": None,
|
|
97
|
+
}
|
|
98
|
+
load_job_metrics[tname]["elapsed"] += (
|
|
99
|
+
(jm.finished_at - jm.started_at).total_seconds()
|
|
100
|
+
if jm.started_at and jm.finished_at
|
|
101
|
+
else 0.0
|
|
102
|
+
)
|
|
103
|
+
if jm.state != "completed":
|
|
104
|
+
load_job_metrics[tname]["state"] = jm.state
|
|
105
|
+
if hasattr(jm, "failed_message") and jm.failed_message:
|
|
106
|
+
load_job_metrics[tname]["failed_message"] = str(
|
|
107
|
+
jm.failed_message
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
if step_name == "extract" and pipeline_name is None:
|
|
111
|
+
try:
|
|
112
|
+
pipeline_name = si.pipeline.pipeline_name
|
|
113
|
+
first_run = si.first_run
|
|
114
|
+
except AttributeError:
|
|
115
|
+
pass
|
|
116
|
+
if step_name == "load":
|
|
117
|
+
try:
|
|
118
|
+
destination_name = si.destination_name
|
|
119
|
+
dataset_name = si.dataset_name
|
|
120
|
+
except AttributeError:
|
|
121
|
+
pass
|
|
122
|
+
|
|
123
|
+
if load_id is None:
|
|
124
|
+
load_id = f"failed_{uuid.uuid4().hex[:16]}"
|
|
125
|
+
|
|
126
|
+
failed_tables = [
|
|
127
|
+
t for t, m in load_job_metrics.items() if m["state"] != "completed"
|
|
128
|
+
]
|
|
129
|
+
if error_message:
|
|
130
|
+
status = "failed"
|
|
131
|
+
else:
|
|
132
|
+
status = "failed" if failed_tables else "completed"
|
|
133
|
+
|
|
134
|
+
table_error_messages = [
|
|
135
|
+
load_job_metrics[t]["failed_message"]
|
|
136
|
+
for t in failed_tables
|
|
137
|
+
if load_job_metrics[t]["failed_message"]
|
|
138
|
+
]
|
|
139
|
+
all_error_parts = []
|
|
140
|
+
if error_message:
|
|
141
|
+
all_error_parts.append(error_message)
|
|
142
|
+
all_error_parts.extend(table_error_messages)
|
|
143
|
+
|
|
144
|
+
total_rows = sum(m["items_count"] for m in extract_metrics.values())
|
|
145
|
+
|
|
146
|
+
audit_run = {
|
|
147
|
+
"load_id": load_id,
|
|
148
|
+
"pipeline_name": pipeline_name,
|
|
149
|
+
"destination_name": destination_name,
|
|
150
|
+
"dataset_name": dataset_name,
|
|
151
|
+
"first_run": first_run,
|
|
152
|
+
"status": status,
|
|
153
|
+
"started_at": trace.started_at.isoformat() if trace.started_at else None,
|
|
154
|
+
"finished_at": trace.finished_at.isoformat() if trace.finished_at else None,
|
|
155
|
+
"extract_started_at": step_timing.get("extract", {}).get("started_at"),
|
|
156
|
+
"extract_finished_at": step_timing.get("extract", {}).get("finished_at"),
|
|
157
|
+
"normalize_started_at": step_timing.get("normalize", {}).get("started_at"),
|
|
158
|
+
"normalize_finished_at": step_timing.get("normalize", {}).get("finished_at"),
|
|
159
|
+
"load_started_at": step_timing.get("load", {}).get("started_at"),
|
|
160
|
+
"load_finished_at": step_timing.get("load", {}).get("finished_at"),
|
|
161
|
+
"total_rows_extracted": total_rows,
|
|
162
|
+
"error_message": "; ".join(all_error_parts) if all_error_parts else None,
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
all_tables = set(extract_metrics) | set(normalize_metrics) | set(load_job_metrics)
|
|
166
|
+
|
|
167
|
+
audit_table_loads = []
|
|
168
|
+
for tname in sorted(all_tables):
|
|
169
|
+
row = {
|
|
170
|
+
"load_id": load_id,
|
|
171
|
+
"table_name": tname,
|
|
172
|
+
"rows_extracted": extract_metrics.get(tname, {}).get("items_count"),
|
|
173
|
+
"rows_normalized": normalize_metrics.get(tname, {}).get("items_count"),
|
|
174
|
+
"file_size": normalize_metrics.get(tname, {}).get("file_size"),
|
|
175
|
+
"load_state": load_job_metrics.get(tname, {}).get("state"),
|
|
176
|
+
"load_elapsed": load_job_metrics.get(tname, {}).get("elapsed"),
|
|
177
|
+
"failed_message": load_job_metrics.get(tname, {}).get("failed_message"),
|
|
178
|
+
}
|
|
179
|
+
audit_table_loads.append(row)
|
|
180
|
+
|
|
181
|
+
return audit_run, audit_table_loads
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def persist_audit(
|
|
185
|
+
pipeline: Any,
|
|
186
|
+
trace: Any,
|
|
187
|
+
table_format: Optional[str] = "delta",
|
|
188
|
+
error_message: Optional[str] = None,
|
|
189
|
+
) -> dict[str, Any]:
|
|
190
|
+
"""Build audit data from trace and persist to audit tables.
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
pipeline: The dlt pipeline instance.
|
|
194
|
+
trace: The captured pipeline trace (pipeline.last_trace).
|
|
195
|
+
table_format: Table format (default "delta").
|
|
196
|
+
error_message: Optional error message from a caught exception.
|
|
197
|
+
|
|
198
|
+
Returns:
|
|
199
|
+
Dict with audit summary (load_id, status, tables_tracked).
|
|
200
|
+
"""
|
|
201
|
+
audit_run, audit_table_loads = build_audit_data(trace, error_message=error_message)
|
|
202
|
+
|
|
203
|
+
run_kwargs: dict[str, Any] = {
|
|
204
|
+
"table_name": "audit_runs",
|
|
205
|
+
"write_disposition": "append",
|
|
206
|
+
}
|
|
207
|
+
if table_format:
|
|
208
|
+
run_kwargs["table_format"] = table_format
|
|
209
|
+
pipeline.run([audit_run], **run_kwargs)
|
|
210
|
+
|
|
211
|
+
if audit_table_loads:
|
|
212
|
+
load_kwargs: dict[str, Any] = {
|
|
213
|
+
"table_name": "audit_table_loads",
|
|
214
|
+
"write_disposition": "append",
|
|
215
|
+
}
|
|
216
|
+
if table_format:
|
|
217
|
+
load_kwargs["table_format"] = table_format
|
|
218
|
+
pipeline.run(audit_table_loads, **load_kwargs)
|
|
219
|
+
|
|
220
|
+
logger.info(
|
|
221
|
+
"Audit persisted: load_id=%s status=%s tables=%d",
|
|
222
|
+
audit_run["load_id"],
|
|
223
|
+
audit_run["status"],
|
|
224
|
+
len(audit_table_loads),
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
return {
|
|
228
|
+
"load_id": audit_run["load_id"],
|
|
229
|
+
"status": audit_run["status"],
|
|
230
|
+
"tables_tracked": len(audit_table_loads),
|
|
231
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Pipeline-folder context — the run contract for the two-registry layout:
|
|
3
|
+
|
|
4
|
+
ingestion/
|
|
5
|
+
├── connections/<connection_name>/.dlt/config.toml # dlt project dir at run time
|
|
6
|
+
└── pipelines/<pipeline_name>/
|
|
7
|
+
├── pipeline.py # pure logic; runs with cwd = this folder
|
|
8
|
+
└── config.toml # [pipeline].connection + optional dlt tuning tables
|
|
9
|
+
|
|
10
|
+
`load_pipeline_context()` reads ./config.toml, resolves the referenced
|
|
11
|
+
connection folder, and returns the tuning tables (everything except the
|
|
12
|
+
SDK-owned [pipeline] table, which must never resolve as dlt config).
|
|
13
|
+
Returns None when there is no pipeline config — callers fall back to the
|
|
14
|
+
legacy single-project behavior.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import os
|
|
20
|
+
import tomllib
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
PIPELINE_CONFIG_FILE = 'config.toml'
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class PipelineContext:
|
|
29
|
+
connection_name: str
|
|
30
|
+
"""Absolute path to ingestion/connections/<name>/ — the DLT_PROJECT_DIR."""
|
|
31
|
+
connection_dir: str
|
|
32
|
+
"""dlt tuning tables from the pipeline config ([pipeline] stripped)."""
|
|
33
|
+
tuning: dict[str, Any]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load_pipeline_context(cwd: str = '.') -> PipelineContext | None:
|
|
37
|
+
config_path = os.path.join(cwd, PIPELINE_CONFIG_FILE)
|
|
38
|
+
try:
|
|
39
|
+
with open(config_path, 'rb') as f:
|
|
40
|
+
cfg = tomllib.load(f)
|
|
41
|
+
except FileNotFoundError:
|
|
42
|
+
return None
|
|
43
|
+
|
|
44
|
+
pipeline_table = cfg.get('pipeline')
|
|
45
|
+
if not isinstance(pipeline_table, dict):
|
|
46
|
+
return None
|
|
47
|
+
connection_name = pipeline_table.get('connection')
|
|
48
|
+
if not isinstance(connection_name, str) or not connection_name.strip():
|
|
49
|
+
return None
|
|
50
|
+
connection_name = connection_name.strip()
|
|
51
|
+
|
|
52
|
+
# Fail fast on a dangling reference — referential integrity invariant.
|
|
53
|
+
connection_dir = os.path.abspath(
|
|
54
|
+
os.path.join(cwd, '..', '..', 'connections', connection_name)
|
|
55
|
+
)
|
|
56
|
+
if not os.path.isdir(connection_dir):
|
|
57
|
+
raise FileNotFoundError(
|
|
58
|
+
f"Pipeline references connection '{connection_name}' but "
|
|
59
|
+
f'{connection_dir} does not exist.'
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
tuning = {key: value for key, value in cfg.items() if key != 'pipeline'}
|
|
63
|
+
return PipelineContext(
|
|
64
|
+
connection_name=connection_name,
|
|
65
|
+
connection_dir=connection_dir,
|
|
66
|
+
tuning=tuning,
|
|
67
|
+
)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Cross-platform pipeline-tuning provider.
|
|
2
|
+
|
|
3
|
+
The pipeline's ``config.toml`` tuning tables (everything except the SDK-owned
|
|
4
|
+
``[pipeline]`` table) register as the LOWEST-precedence dlt config provider, so
|
|
5
|
+
``env > connection toml > pipeline toml``. Identical for every platform, so it
|
|
6
|
+
lives in core rather than each platform's setup flow.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def register_pipeline_tuning(tuning: dict[str, Any]) -> None:
|
|
18
|
+
"""Register the pipeline config.toml tuning tables as a dlt config provider."""
|
|
19
|
+
import dlt
|
|
20
|
+
from dlt.common.configuration.providers import CustomLoaderDocProvider
|
|
21
|
+
|
|
22
|
+
provider = CustomLoaderDocProvider("pipeline_config", lambda: tuning, supports_secrets=False)
|
|
23
|
+
dlt.config.register_provider(provider)
|
|
24
|
+
logger.info("Registered pipeline_config provider (%d tables)", len(tuning))
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Tests for audit module."""
|
|
2
|
+
|
|
3
|
+
from unittest.mock import MagicMock
|
|
4
|
+
from datetime import datetime, timezone
|
|
5
|
+
|
|
6
|
+
from vibedata.dlt.core.audit import build_audit_data
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _make_mock_trace(*, has_error=False):
|
|
10
|
+
"""Create a minimal mock dlt trace for testing."""
|
|
11
|
+
trace = MagicMock()
|
|
12
|
+
trace.started_at = datetime(2024, 1, 1, 12, 0, 0, tzinfo=timezone.utc)
|
|
13
|
+
trace.finished_at = datetime(2024, 1, 1, 12, 5, 0, tzinfo=timezone.utc)
|
|
14
|
+
|
|
15
|
+
# Extract step
|
|
16
|
+
extract_step = MagicMock()
|
|
17
|
+
extract_step.step = "extract"
|
|
18
|
+
extract_step.started_at = trace.started_at
|
|
19
|
+
extract_step.finished_at = datetime(2024, 1, 1, 12, 1, 0, tzinfo=timezone.utc)
|
|
20
|
+
|
|
21
|
+
extract_info = MagicMock()
|
|
22
|
+
extract_info.pipeline.pipeline_name = "test_pipeline"
|
|
23
|
+
extract_info.first_run = True
|
|
24
|
+
|
|
25
|
+
table_metric = MagicMock()
|
|
26
|
+
table_metric.items_count = 100
|
|
27
|
+
table_metric.file_size = 5000
|
|
28
|
+
|
|
29
|
+
extract_info.metrics = {
|
|
30
|
+
"load_1": [{"table_metrics": {"contacts": table_metric}}]
|
|
31
|
+
}
|
|
32
|
+
extract_step.step_info = extract_info
|
|
33
|
+
|
|
34
|
+
# Load step
|
|
35
|
+
load_step = MagicMock()
|
|
36
|
+
load_step.step = "load"
|
|
37
|
+
load_step.started_at = datetime(2024, 1, 1, 12, 3, 0, tzinfo=timezone.utc)
|
|
38
|
+
load_step.finished_at = datetime(2024, 1, 1, 12, 5, 0, tzinfo=timezone.utc)
|
|
39
|
+
|
|
40
|
+
load_info = MagicMock()
|
|
41
|
+
load_info.destination_name = "filesystem"
|
|
42
|
+
load_info.dataset_name = "salesforce"
|
|
43
|
+
|
|
44
|
+
job_metric = MagicMock()
|
|
45
|
+
job_metric.table_name = "contacts"
|
|
46
|
+
job_metric.state = "failed" if has_error else "completed"
|
|
47
|
+
job_metric.started_at = load_step.started_at
|
|
48
|
+
job_metric.finished_at = load_step.finished_at
|
|
49
|
+
job_metric.failed_message = "Load error" if has_error else None
|
|
50
|
+
|
|
51
|
+
load_info.metrics = {
|
|
52
|
+
"load_1": [{"job_metrics": {"job_1": job_metric}}]
|
|
53
|
+
}
|
|
54
|
+
load_step.step_info = load_info
|
|
55
|
+
|
|
56
|
+
trace.steps = [extract_step, load_step]
|
|
57
|
+
return trace
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class TestBuildAuditData:
|
|
61
|
+
def test_successful_run(self):
|
|
62
|
+
trace = _make_mock_trace()
|
|
63
|
+
audit_run, table_loads = build_audit_data(trace)
|
|
64
|
+
|
|
65
|
+
assert audit_run["status"] == "completed"
|
|
66
|
+
assert audit_run["pipeline_name"] == "test_pipeline"
|
|
67
|
+
assert audit_run["total_rows_extracted"] == 100
|
|
68
|
+
assert audit_run["error_message"] is None
|
|
69
|
+
assert len(table_loads) == 1
|
|
70
|
+
assert table_loads[0]["table_name"] == "contacts"
|
|
71
|
+
assert table_loads[0]["rows_extracted"] == 100
|
|
72
|
+
|
|
73
|
+
def test_failed_run(self):
|
|
74
|
+
trace = _make_mock_trace(has_error=True)
|
|
75
|
+
audit_run, table_loads = build_audit_data(trace)
|
|
76
|
+
|
|
77
|
+
assert audit_run["status"] == "failed"
|
|
78
|
+
assert "Load error" in audit_run["error_message"]
|
|
79
|
+
|
|
80
|
+
def test_forced_error_message(self):
|
|
81
|
+
trace = _make_mock_trace()
|
|
82
|
+
audit_run, _ = build_audit_data(trace, error_message="Pipeline crashed")
|
|
83
|
+
|
|
84
|
+
assert audit_run["status"] == "failed"
|
|
85
|
+
assert "Pipeline crashed" in audit_run["error_message"]
|
|
86
|
+
|
|
87
|
+
def test_empty_trace(self):
|
|
88
|
+
trace = MagicMock()
|
|
89
|
+
trace.started_at = None
|
|
90
|
+
trace.finished_at = None
|
|
91
|
+
trace.steps = []
|
|
92
|
+
|
|
93
|
+
audit_run, table_loads = build_audit_data(trace)
|
|
94
|
+
|
|
95
|
+
assert audit_run["load_id"].startswith("failed_")
|
|
96
|
+
assert audit_run["status"] == "completed"
|
|
97
|
+
assert table_loads == []
|