data-joinery 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_joinery/.agents/skills/data-joinery/SKILL.md +165 -0
- data_joinery/__init__.py +34 -0
- data_joinery/backends/__init__.py +17 -0
- data_joinery/backends/base.py +144 -0
- data_joinery/backends/polars.py +380 -0
- data_joinery/backends/spark.py +592 -0
- data_joinery/cli.py +19 -0
- data_joinery/contract.py +148 -0
- data_joinery/dbt.py +19 -0
- data_joinery/dependencies.py +49 -0
- data_joinery/model_schema.py +84 -0
- data_joinery/pipeline.py +391 -0
- data_joinery/schema_types.py +62 -0
- data_joinery/schemas.py +110 -0
- data_joinery/transform.py +194 -0
- data_joinery/type_inspection.py +123 -0
- data_joinery/utils.py +64 -0
- data_joinery/visualisation.py +12 -0
- data_joinery-0.1.0.dist-info/METADATA +15 -0
- data_joinery-0.1.0.dist-info/RECORD +22 -0
- data_joinery-0.1.0.dist-info/WHEEL +4 -0
- data_joinery-0.1.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: data-joinery
|
|
3
|
+
description: Data Joinery best practices and conventions. Use when working with Data Joinery. Keeps Data Joinery code clean and up to date with the latest features and patterns, updated with new versions. Write new code or refactor and update old code.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Data Joinery
|
|
7
|
+
|
|
8
|
+
Data Joinery is a schema-first framework for constructing testable PySpark transformations and validated pipeline DAGs. Use this skill when adding or changing schemas, decorated transformations, or `Pipeline` definitions.
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from data_joinery import (
|
|
13
|
+
Context,
|
|
14
|
+
Pipeline,
|
|
15
|
+
PipelineContext,
|
|
16
|
+
Project,
|
|
17
|
+
ProjectCast,
|
|
18
|
+
ProjectTopLevel,
|
|
19
|
+
Strict,
|
|
20
|
+
StrictNull,
|
|
21
|
+
transform,
|
|
22
|
+
)
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Define Schemas First
|
|
26
|
+
|
|
27
|
+
Define a Python dataclass or Pydantic model for every meaningful DataFrame boundary. Each model field represents a Spark column. Dataclasses are the usual choice; Pydantic is useful when its validation rules also help construct valid test fixtures, but Data Joinery only enforces the resulting Spark data types.
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from dataclasses import dataclass
|
|
31
|
+
from datetime import date
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class Customer:
|
|
36
|
+
snapshot_date: date
|
|
37
|
+
customer_id: str
|
|
38
|
+
name: str
|
|
39
|
+
is_active: bool
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
- Use nested dataclasses and typed lists for nested Spark structs and arrays.
|
|
43
|
+
- Python types map to Spark types by default. Use `typing.Annotated` on a schema field only when an explicit Spark type is required, such as `DecimalType`.
|
|
44
|
+
- Fields are nullable by default. Do not treat non-nullable schema metadata as a data-quality constraint.
|
|
45
|
+
- Keep schemas focused on the DataFrame contract for a boundary, rather than reusing a broad source-table model throughout a pipeline.
|
|
46
|
+
- Use `Schema(Model).create_dataframe(spark, instances)` to create typed DataFrame fixtures in tests.
|
|
47
|
+
|
|
48
|
+
## Build Transformations
|
|
49
|
+
|
|
50
|
+
Decorate each transformation with `@transform`. Annotate every DataFrame input and output as `Annotated[DataFrame, Mode(Schema)]`; this is both documentation and a runtime contract. Transformations may also return non-DataFrame values: annotate those with their ordinary Python type, which the pipeline checks for type compatibility without schema coercion.
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from typing import Annotated
|
|
54
|
+
|
|
55
|
+
from pyspark.sql import DataFrame
|
|
56
|
+
from data_joinery import Project, Strict, transform
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@transform
|
|
60
|
+
def filter_active_customers(
|
|
61
|
+
customers: Annotated[DataFrame, Project(Customer)],
|
|
62
|
+
) -> Annotated[DataFrame, Strict(Customer)]:
|
|
63
|
+
return customers.filter(customers.is_active)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Choose the coercion mode deliberately:
|
|
67
|
+
|
|
68
|
+
- `Strict(Model)`: requires exactly the model's fields and types; ignores nullability. Prefer this at stable internal boundaries and for transformation outputs.
|
|
69
|
+
- `StrictNull(Model)`: as strict, including nullability. Use only when nullability metadata is itself part of the contract.
|
|
70
|
+
- `Project(Model)`: recursively selects the model's fields, removes extras, and does not cast or invent missing fields. Prefer this for inputs from wide or nested sources.
|
|
71
|
+
- `ProjectTopLevel(Model)`: projects only top-level fields and requires nested struct fields to match.
|
|
72
|
+
- `ProjectCast(Model)`: recursively projects and uses Spark casts. Reserve it for external or loosely typed inputs where conversion is intentional; casts can still fail when evaluated.
|
|
73
|
+
|
|
74
|
+
The usual robust pattern is a permissive input and a strict output. Keep transformations small and DataFrame-focused; Data Joinery validates the contracts before Spark's lazy computation, unless the function itself performs an action such as `collect()`.
|
|
75
|
+
|
|
76
|
+
Source transforms may accept `SparkSession` and return a contracted DataFrame. Sink transforms may accept contracted DataFrames and return `None`.
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from pyspark.sql import DataFrame, SparkSession
|
|
80
|
+
from typing import Annotated
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@transform
|
|
84
|
+
def read_customers(
|
|
85
|
+
spark: SparkSession,
|
|
86
|
+
) -> Annotated[DataFrame, Project(Customer)]:
|
|
87
|
+
return spark.read.parquet("/data/customers")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@transform
|
|
91
|
+
def write_customers(
|
|
92
|
+
customers: Annotated[DataFrame, Strict(Customer)],
|
|
93
|
+
) -> None:
|
|
94
|
+
customers.write.mode("overwrite").parquet("/data/active-customers")
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Outside a pipeline, decorated transformations remain callable directly and can be used with `DataFrame.transform`.
|
|
98
|
+
|
|
99
|
+
## Wire A Pipeline
|
|
100
|
+
|
|
101
|
+
Create a `Pipeline`, add each decorated transform as a `Step`, then connect the steps. A `Step` is an occurrence of a transform, so the same transform may be added more than once with distinct names.
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
pipeline = Pipeline()
|
|
105
|
+
read_step = pipeline.add_step(read_customers)
|
|
106
|
+
filter_step = pipeline.add_step(filter_active_customers)
|
|
107
|
+
write_step = pipeline.add_step(write_customers)
|
|
108
|
+
|
|
109
|
+
pipeline.connect(read_step, filter_step)
|
|
110
|
+
pipeline.connect(filter_step, write_step)
|
|
111
|
+
|
|
112
|
+
pipeline.run(spark=spark)
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
- Pipeline steps support `SparkSession`, contract-annotated `DataFrame` parameters, typed non-DataFrame parameters, and `Context()`-annotated parameters. Do not add ordinary configuration parameters.
|
|
116
|
+
- A source step has no DataFrame inputs and must declare `SparkSession`.
|
|
117
|
+
- `connect()` validates the upstream output against a compatible downstream input immediately and rejects cycles.
|
|
118
|
+
- When an upstream output could satisfy more than one input parameter, disambiguate with `pipeline.connect(upstream, downstream, param="parameter_name")`.
|
|
119
|
+
- Use `connect_many([...], downstream)` for a transform with multiple independently produced DataFrame inputs.
|
|
120
|
+
- Give repeated transforms explicit, unique names with `add_step(transform, name="...")`.
|
|
121
|
+
- `run()` returns a dictionary of step outputs keyed by step name; sink steps returning `None` are omitted.
|
|
122
|
+
|
|
123
|
+
## Inject Context Deliberately
|
|
124
|
+
|
|
125
|
+
Use `Context()` for run-specific configuration or dependencies that should be supplied by the pipeline runner rather than carried as DataFrame data: paths, a run date, feature configuration, a client wrapper, or credentials wrappers. Do not use it for values that should be DataFrame columns or for ordinary transform calls outside a pipeline.
|
|
126
|
+
|
|
127
|
+
Define a dedicated type for each context value. Context resolution is by type and cannot disambiguate two values of the same type; never inject bare `str`, `int`, or `date` values.
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
from dataclasses import dataclass
|
|
131
|
+
from pathlib import Path
|
|
132
|
+
from typing import Annotated
|
|
133
|
+
|
|
134
|
+
from pyspark.sql import DataFrame, SparkSession
|
|
135
|
+
from data_joinery import Context, PipelineContext, Project, transform
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@dataclass
|
|
139
|
+
class CustomerPaths:
|
|
140
|
+
input_path: Path
|
|
141
|
+
output_path: Path
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@transform
|
|
145
|
+
def read_customers(
|
|
146
|
+
spark: SparkSession,
|
|
147
|
+
paths: Annotated[CustomerPaths, Context()],
|
|
148
|
+
) -> Annotated[DataFrame, Project(Customer)]:
|
|
149
|
+
return spark.read.parquet(str(paths.input_path))
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
context = PipelineContext(values=[CustomerPaths(Path("/in"), Path("/out"))])
|
|
153
|
+
pipeline.run(spark=spark, context=context)
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
For a semantic value based on a built-in, create a distinct subclass, such as `class RunDate(date): pass`, and annotate the parameter with that type. The pipeline raises an execution error when a required context value is absent.
|
|
157
|
+
|
|
158
|
+
## Checklist
|
|
159
|
+
|
|
160
|
+
1. Define or refine the schemas at each DataFrame boundary.
|
|
161
|
+
2. Add `@transform` and explicit input/output contracts before writing transformation logic.
|
|
162
|
+
3. Select coercion modes based on ownership and data quality at the boundary.
|
|
163
|
+
4. Unit-test transforms with typed fixtures created from schema models.
|
|
164
|
+
5. Add transforms to a `Pipeline`, connect the DAG, and run it with a `SparkSession`.
|
|
165
|
+
6. Introduce typed `Context()` only for dependencies or configuration supplied at pipeline execution time.
|
data_joinery/__init__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
from .cli import main
|
|
2
|
+
from .contract import (
|
|
3
|
+
Project,
|
|
4
|
+
ProjectCast,
|
|
5
|
+
ProjectTopLevel,
|
|
6
|
+
Strict,
|
|
7
|
+
)
|
|
8
|
+
from .dbt import Dbt
|
|
9
|
+
from .dependencies import Context, SparkContext
|
|
10
|
+
from .pipeline import (
|
|
11
|
+
Pipeline,
|
|
12
|
+
PipelineExecutionError,
|
|
13
|
+
PipelineOverrideError,
|
|
14
|
+
Step,
|
|
15
|
+
)
|
|
16
|
+
from .schemas import Schema
|
|
17
|
+
from .transform import transform
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Context",
|
|
21
|
+
"Dbt",
|
|
22
|
+
"Pipeline",
|
|
23
|
+
"PipelineExecutionError",
|
|
24
|
+
"PipelineOverrideError",
|
|
25
|
+
"Project",
|
|
26
|
+
"ProjectCast",
|
|
27
|
+
"ProjectTopLevel",
|
|
28
|
+
"Schema",
|
|
29
|
+
"SparkContext",
|
|
30
|
+
"Step",
|
|
31
|
+
"Strict",
|
|
32
|
+
"main",
|
|
33
|
+
"transform",
|
|
34
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from .base import (
|
|
2
|
+
DataFrameBackend,
|
|
3
|
+
backend_for_frame_type,
|
|
4
|
+
backend_for_schema_type,
|
|
5
|
+
backend_for_value,
|
|
6
|
+
get_backend,
|
|
7
|
+
register_backend,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"DataFrameBackend",
|
|
12
|
+
"backend_for_frame_type",
|
|
13
|
+
"backend_for_schema_type",
|
|
14
|
+
"backend_for_value",
|
|
15
|
+
"get_backend",
|
|
16
|
+
"register_backend",
|
|
17
|
+
]
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
from collections.abc import Sequence
|
|
2
|
+
from importlib import import_module
|
|
3
|
+
from importlib.util import find_spec
|
|
4
|
+
from typing import Any, Protocol
|
|
5
|
+
|
|
6
|
+
from ..model_schema import ModelSchema
|
|
7
|
+
from ..schema_types import CoercionMode
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class DataFrameBackend(Protocol):
|
|
11
|
+
name: str
|
|
12
|
+
|
|
13
|
+
@property
|
|
14
|
+
def dataframe_types(self) -> tuple[type, ...]: ...
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def schema_types(self) -> tuple[type, ...]: ...
|
|
18
|
+
|
|
19
|
+
def compile_schema(self, schema: ModelSchema[Any]) -> object: ...
|
|
20
|
+
|
|
21
|
+
def coerce_dataframe(
|
|
22
|
+
self,
|
|
23
|
+
dataframe: object,
|
|
24
|
+
schema: ModelSchema[Any],
|
|
25
|
+
mode: CoercionMode,
|
|
26
|
+
) -> object: ...
|
|
27
|
+
|
|
28
|
+
def create_dataframe(
|
|
29
|
+
self,
|
|
30
|
+
rows: Sequence[object],
|
|
31
|
+
schema: ModelSchema[Any],
|
|
32
|
+
**kwargs: object,
|
|
33
|
+
) -> object: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
_BACKENDS: dict[str, DataFrameBackend] = {}
|
|
37
|
+
_BUILTINS_LOADED = False
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def register_backend(backend: DataFrameBackend) -> None:
|
|
41
|
+
existing = _BACKENDS.get(backend.name)
|
|
42
|
+
if existing is backend:
|
|
43
|
+
return
|
|
44
|
+
if existing is not None:
|
|
45
|
+
raise ValueError(f"Backend '{backend.name}' is already registered")
|
|
46
|
+
|
|
47
|
+
for registered in _BACKENDS.values():
|
|
48
|
+
_raise_for_type_overlap(
|
|
49
|
+
backend,
|
|
50
|
+
registered,
|
|
51
|
+
attribute="dataframe_types",
|
|
52
|
+
kind="dataframe",
|
|
53
|
+
)
|
|
54
|
+
_raise_for_type_overlap(
|
|
55
|
+
backend,
|
|
56
|
+
registered,
|
|
57
|
+
attribute="schema_types",
|
|
58
|
+
kind="schema",
|
|
59
|
+
)
|
|
60
|
+
_BACKENDS[backend.name] = backend
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _raise_for_type_overlap(
|
|
64
|
+
backend: DataFrameBackend,
|
|
65
|
+
registered: DataFrameBackend,
|
|
66
|
+
*,
|
|
67
|
+
attribute: str,
|
|
68
|
+
kind: str,
|
|
69
|
+
) -> None:
|
|
70
|
+
candidate_types = getattr(backend, attribute)
|
|
71
|
+
registered_types = getattr(registered, attribute)
|
|
72
|
+
for candidate in candidate_types:
|
|
73
|
+
for registered_type in registered_types:
|
|
74
|
+
if issubclass(candidate, registered_type) or issubclass(
|
|
75
|
+
registered_type, candidate
|
|
76
|
+
):
|
|
77
|
+
raise TypeError(
|
|
78
|
+
f"Backend '{backend.name}' has ambiguous {kind} type "
|
|
79
|
+
f"{candidate!r} with backend '{registered.name}'"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _load_builtin_backends() -> None:
|
|
84
|
+
global _BUILTINS_LOADED
|
|
85
|
+
if _BUILTINS_LOADED:
|
|
86
|
+
return
|
|
87
|
+
_BUILTINS_LOADED = True
|
|
88
|
+
import_module("data_joinery.backends.spark")
|
|
89
|
+
if find_spec("polars") is not None:
|
|
90
|
+
import_module("data_joinery.backends.polars")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def get_backend(name: str) -> DataFrameBackend:
|
|
94
|
+
if name not in _BACKENDS:
|
|
95
|
+
_load_builtin_backends()
|
|
96
|
+
try:
|
|
97
|
+
return _BACKENDS[name]
|
|
98
|
+
except KeyError as error:
|
|
99
|
+
raise LookupError(
|
|
100
|
+
f"No dataframe backend named '{name}' is registered"
|
|
101
|
+
) from error
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def backend_for_frame_type(frame_type: type) -> DataFrameBackend | None:
|
|
105
|
+
if not isinstance(frame_type, type):
|
|
106
|
+
return None
|
|
107
|
+
backend = _find_backend_for_type(frame_type)
|
|
108
|
+
if backend is not None:
|
|
109
|
+
return backend
|
|
110
|
+
_load_builtin_backends()
|
|
111
|
+
return _find_backend_for_type(frame_type)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def backend_for_schema_type(schema_type: type) -> DataFrameBackend | None:
|
|
115
|
+
if not isinstance(schema_type, type):
|
|
116
|
+
return None
|
|
117
|
+
backend = _find_backend_for_type(schema_type, attribute="schema_types")
|
|
118
|
+
if backend is not None:
|
|
119
|
+
return backend
|
|
120
|
+
_load_builtin_backends()
|
|
121
|
+
return _find_backend_for_type(schema_type, attribute="schema_types")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _find_backend_for_type(
|
|
125
|
+
frame_type: type,
|
|
126
|
+
*,
|
|
127
|
+
attribute: str = "dataframe_types",
|
|
128
|
+
) -> DataFrameBackend | None:
|
|
129
|
+
matches = [
|
|
130
|
+
backend
|
|
131
|
+
for backend in _BACKENDS.values()
|
|
132
|
+
if any(
|
|
133
|
+
issubclass(frame_type, candidate)
|
|
134
|
+
for candidate in getattr(backend, attribute)
|
|
135
|
+
)
|
|
136
|
+
]
|
|
137
|
+
if len(matches) > 1:
|
|
138
|
+
names = ", ".join(sorted(backend.name for backend in matches))
|
|
139
|
+
raise LookupError(f"Multiple dataframe backends match {frame_type!r}: {names}")
|
|
140
|
+
return matches[0] if matches else None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def backend_for_value(value: object) -> DataFrameBackend | None:
|
|
144
|
+
return backend_for_frame_type(type(value))
|