data-joinery 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ Metadata-Version: 2.3
2
+ Name: data-joinery
3
+ Version: 0.1.0
4
+ Summary: Schema safety for DataFrame transformations
5
+ Author: Connor Charles
6
+ Author-email: Connor Charles <ccharles.gb@gmail.com>
7
+ Requires-Dist: pyspark>=4.2.0
8
+ Requires-Dist: rustworkx>=0.18.1
9
+ Requires-Dist: polars>=1.0.0 ; extra == 'polars'
10
+ Requires-Dist: pydantic>=2.0.0 ; extra == 'pydantic'
11
+ Requires-Dist: matplotlib>=3.10.8 ; extra == 'vis'
12
+ Requires-Python: >=3.12
13
+ Provides-Extra: polars
14
+ Provides-Extra: pydantic
15
+ Provides-Extra: vis
@@ -0,0 +1,68 @@
1
+ [project]
2
+ name = "data-joinery"
3
+ version = "0.1.0"
4
+ description = "Schema safety for DataFrame transformations"
5
+ requires-python = ">=3.12"
6
+ dependencies = [
7
+ "pyspark>=4.2.0",
8
+ "rustworkx>=0.18.1",
9
+ ]
10
+
11
+ [[project.authors]]
12
+ name = "Connor Charles"
13
+ email = "ccharles.gb@gmail.com"
14
+
15
+ [project.optional-dependencies]
16
+ polars = ["polars>=1.0.0"]
17
+ pydantic = ["pydantic>=2.0.0"]
18
+ vis = ["matplotlib>=3.10.8"]
19
+
20
+ [project.scripts]
21
+ data-joinery = "data_joinery:main"
22
+
23
+ [tool.pyrefly]
24
+ project-includes = [
25
+ "**/*.py*",
26
+ "**/*.ipynb",
27
+ ]
28
+ search-path = [
29
+ "src",
30
+ "docs_src",
31
+ ]
32
+
33
+ [tool.pytest-layout-enforcer]
34
+ enabled = true
35
+ feature-pattern = "[a-z][a-z0-9_]*"
36
+ feature-separator = "__"
37
+ require-tests = false
38
+ include-init = false
39
+ exclude-sources = []
40
+ exclude-tests = []
41
+
42
+ [[tool.pytest-layout-enforcer.layouts]]
43
+ source = "src/data_joinery"
44
+ tests = "tests"
45
+
46
+ [tool.ruff.format]
47
+ exclude = ["*.md"]
48
+
49
+ [build-system]
50
+ requires = ["uv_build>=0.9.27,<0.10.0"]
51
+ build-backend = "uv_build"
52
+
53
+ [dependency-groups]
54
+ dev = [
55
+ "deptry>=0.25.1",
56
+ "ipykernel>=7.3.0",
57
+ "mkdocstrings-python>=2.0.8",
58
+ "pandas>=2.0.0,<3.0.0",
59
+ "polars>=1.0.0",
60
+ "pyarrow>=25.0.1",
61
+ "pydantic>=2.0.0",
62
+ "pyrefly>=1.2.0",
63
+ "pytest>=9.1.0",
64
+ "pytest-layout-enforcer>=0.1.0",
65
+ "ruff>=0.15.17",
66
+ "scikit-learn>=1.9.1",
67
+ "zensical>=0.0.45",
68
+ ]
@@ -0,0 +1,74 @@
1
+ [project]
2
+ name = "data-joinery"
3
+ version = "0.1.0"
4
+ description = "Schema safety for DataFrame transformations"
5
+ authors = [
6
+ { name = "Connor Charles", email = "ccharles.gb@gmail.com" }
7
+ ]
8
+ requires-python = ">=3.12"
9
+ dependencies = [
10
+ "pyspark>=4.2.0",
11
+ "rustworkx>=0.18.1",
12
+ ]
13
+
14
+ [tool.pyrefly]
15
+ project-includes = [
16
+ "**/*.py*",
17
+ "**/*.ipynb",
18
+ ]
19
+ search-path = [
20
+ "src",
21
+ "docs_src",
22
+ ]
23
+
24
+ [project.optional-dependencies]
25
+ polars = [
26
+ "polars>=1.0.0",
27
+ ]
28
+ pydantic = [
29
+ "pydantic>=2.0.0",
30
+ ]
31
+ vis = [
32
+ "matplotlib>=3.10.8",
33
+ ]
34
+
35
+ [project.scripts]
36
+ data-joinery = "data_joinery:main"
37
+
38
+ [build-system]
39
+ requires = ["uv_build>=0.9.27,<0.10.0"]
40
+ build-backend = "uv_build"
41
+
42
+ [dependency-groups]
43
+ dev = [
44
+ "deptry>=0.25.1",
45
+ "ipykernel>=7.3.0",
46
+ "mkdocstrings-python>=2.0.8",
47
+ "pandas>=2.0.0,<3.0.0",
48
+ "polars>=1.0.0",
49
+ "pyarrow>=25.0.1",
50
+ "pydantic>=2.0.0",
51
+ "pyrefly>=1.2.0",
52
+ "pytest>=9.1.0",
53
+ "pytest-layout-enforcer>=0.1.0",
54
+ "ruff>=0.15.17",
55
+ "scikit-learn>=1.9.1",
56
+ "zensical>=0.0.45",
57
+ ]
58
+
59
+ [tool.pytest-layout-enforcer]
60
+ enabled = true
61
+ feature-pattern = "[a-z][a-z0-9_]*"
62
+ feature-separator = "__"
63
+ require-tests = false
64
+ include-init = false
65
+
66
+ exclude-sources = []
67
+ exclude-tests = []
68
+
69
+ [[tool.pytest-layout-enforcer.layouts]]
70
+ source = "src/data_joinery"
71
+ tests = "tests"
72
+
73
+ [tool.ruff.format]
74
+ exclude = ["*.md"]
@@ -0,0 +1,165 @@
1
+ ---
2
+ name: data-joinery
3
+ description: Data Joinery best practices and conventions. Use when working with Data Joinery. Keeps Data Joinery code clean and up to date with the latest features and patterns, updated with new versions. Write new code or refactor and update old code.
4
+ ---
5
+
6
+ # Data Joinery
7
+
8
+ Data Joinery is a schema-first framework for constructing testable PySpark transformations and validated pipeline DAGs. Use this skill when adding or changing schemas, decorated transformations, or `Pipeline` definitions.
9
+
10
+
11
+ ```python
12
+ from data_joinery import (
13
+ Context,
14
+ Pipeline,
15
+ PipelineContext,
16
+ Project,
17
+ ProjectCast,
18
+ ProjectTopLevel,
19
+ Strict,
20
+ StrictNull,
21
+ transform,
22
+ )
23
+ ```
24
+
25
+ ## Define Schemas First
26
+
27
+ Define a Python dataclass or Pydantic model for every meaningful DataFrame boundary. Each model field represents a Spark column. Dataclasses are the usual choice; Pydantic is useful when its validation rules also help construct valid test fixtures, but Data Joinery only enforces the resulting Spark data types.
28
+
29
+ ```python
30
+ from dataclasses import dataclass
31
+ from datetime import date
32
+
33
+
34
+ @dataclass
35
+ class Customer:
36
+ snapshot_date: date
37
+ customer_id: str
38
+ name: str
39
+ is_active: bool
40
+ ```
41
+
42
+ - Use nested dataclasses and typed lists for nested Spark structs and arrays.
43
+ - Python types map to Spark types by default. Use `typing.Annotated` on a schema field only when an explicit Spark type is required, such as `DecimalType`.
44
+ - Fields are nullable by default. Do not treat non-nullable schema metadata as a data-quality constraint.
45
+ - Keep schemas focused on the DataFrame contract for a boundary, rather than reusing a broad source-table model throughout a pipeline.
46
+ - Use `Schema(Model).create_dataframe(spark, instances)` to create typed DataFrame fixtures in tests.
47
+
48
+ ## Build Transformations
49
+
50
+ Decorate each transformation with `@transform`. Annotate every DataFrame input and output as `Annotated[DataFrame, Mode(Schema)]`; this is both documentation and a runtime contract. Transformations may also return non-DataFrame values: annotate those with their ordinary Python type, which the pipeline checks for type compatibility without schema coercion.
51
+
52
+ ```python
53
+ from typing import Annotated
54
+
55
+ from pyspark.sql import DataFrame
56
+ from data_joinery import Project, Strict, transform
57
+
58
+
59
+ @transform
60
+ def filter_active_customers(
61
+ customers: Annotated[DataFrame, Project(Customer)],
62
+ ) -> Annotated[DataFrame, Strict(Customer)]:
63
+ return customers.filter(customers.is_active)
64
+ ```
65
+
66
+ Choose the coercion mode deliberately:
67
+
68
+ - `Strict(Model)`: requires exactly the model's fields and types; ignores nullability. Prefer this at stable internal boundaries and for transformation outputs.
69
+ - `StrictNull(Model)`: as strict, including nullability. Use only when nullability metadata is itself part of the contract.
70
+ - `Project(Model)`: recursively selects the model's fields, removes extras, and does not cast or invent missing fields. Prefer this for inputs from wide or nested sources.
71
+ - `ProjectTopLevel(Model)`: projects only top-level fields and requires nested struct fields to match.
72
+ - `ProjectCast(Model)`: recursively projects and uses Spark casts. Reserve it for external or loosely typed inputs where conversion is intentional; casts can still fail when evaluated.
73
+
74
+ The usual robust pattern is a permissive input and a strict output. Keep transformations small and DataFrame-focused; Data Joinery validates the contracts before Spark's lazy computation, unless the function itself performs an action such as `collect()`.
75
+
76
+ Source transforms may accept `SparkSession` and return a contracted DataFrame. Sink transforms may accept contracted DataFrames and return `None`.
77
+
78
+ ```python
79
+ from pyspark.sql import DataFrame, SparkSession
80
+ from typing import Annotated
81
+
82
+
83
+ @transform
84
+ def read_customers(
85
+ spark: SparkSession,
86
+ ) -> Annotated[DataFrame, Project(Customer)]:
87
+ return spark.read.parquet("/data/customers")
88
+
89
+
90
+ @transform
91
+ def write_customers(
92
+ customers: Annotated[DataFrame, Strict(Customer)],
93
+ ) -> None:
94
+ customers.write.mode("overwrite").parquet("/data/active-customers")
95
+ ```
96
+
97
+ Outside a pipeline, decorated transformations remain callable directly and can be used with `DataFrame.transform`.
98
+
99
+ ## Wire A Pipeline
100
+
101
+ Create a `Pipeline`, add each decorated transform as a `Step`, then connect the steps. A `Step` is an occurrence of a transform, so the same transform may be added more than once with distinct names.
102
+
103
+ ```python
104
+ pipeline = Pipeline()
105
+ read_step = pipeline.add_step(read_customers)
106
+ filter_step = pipeline.add_step(filter_active_customers)
107
+ write_step = pipeline.add_step(write_customers)
108
+
109
+ pipeline.connect(read_step, filter_step)
110
+ pipeline.connect(filter_step, write_step)
111
+
112
+ pipeline.run(spark=spark)
113
+ ```
114
+
115
+ - Pipeline steps support `SparkSession`, contract-annotated `DataFrame` parameters, typed non-DataFrame parameters, and `Context()`-annotated parameters. Do not add ordinary configuration parameters.
116
+ - A source step has no DataFrame inputs and must declare `SparkSession`.
117
+ - `connect()` validates the upstream output against a compatible downstream input immediately and rejects cycles.
118
+ - When an upstream output could satisfy more than one input parameter, disambiguate with `pipeline.connect(upstream, downstream, param="parameter_name")`.
119
+ - Use `connect_many([...], downstream)` for a transform with multiple independently produced DataFrame inputs.
120
+ - Give repeated transforms explicit, unique names with `add_step(transform, name="...")`.
121
+ - `run()` returns a dictionary of step outputs keyed by step name; sink steps returning `None` are omitted.
122
+
123
+ ## Inject Context Deliberately
124
+
125
+ Use `Context()` for run-specific configuration or dependencies that should be supplied by the pipeline runner rather than carried as DataFrame data: paths, a run date, feature configuration, a client wrapper, or credentials wrappers. Do not use it for values that should be DataFrame columns or for ordinary transform calls outside a pipeline.
126
+
127
+ Define a dedicated type for each context value. Context resolution is by type and cannot disambiguate two values of the same type; never inject bare `str`, `int`, or `date` values.
128
+
129
+ ```python
130
+ from dataclasses import dataclass
131
+ from pathlib import Path
132
+ from typing import Annotated
133
+
134
+ from pyspark.sql import DataFrame, SparkSession
135
+ from data_joinery import Context, PipelineContext, Project, transform
136
+
137
+
138
+ @dataclass
139
+ class CustomerPaths:
140
+ input_path: Path
141
+ output_path: Path
142
+
143
+
144
+ @transform
145
+ def read_customers(
146
+ spark: SparkSession,
147
+ paths: Annotated[CustomerPaths, Context()],
148
+ ) -> Annotated[DataFrame, Project(Customer)]:
149
+ return spark.read.parquet(str(paths.input_path))
150
+
151
+
152
+ context = PipelineContext(values=[CustomerPaths(Path("/in"), Path("/out"))])
153
+ pipeline.run(spark=spark, context=context)
154
+ ```
155
+
156
+ For a semantic value based on a built-in, create a distinct subclass, such as `class RunDate(date): pass`, and annotate the parameter with that type. The pipeline raises an execution error when a required context value is absent.
157
+
158
+ ## Checklist
159
+
160
+ 1. Define or refine the schemas at each DataFrame boundary.
161
+ 2. Add `@transform` and explicit input/output contracts before writing transformation logic.
162
+ 3. Select coercion modes based on ownership and data quality at the boundary.
163
+ 4. Unit-test transforms with typed fixtures created from schema models.
164
+ 5. Add transforms to a `Pipeline`, connect the DAG, and run it with a `SparkSession`.
165
+ 6. Introduce typed `Context()` only for dependencies or configuration supplied at pipeline execution time.
@@ -0,0 +1,34 @@
1
+ from .cli import main
2
+ from .contract import (
3
+ Project,
4
+ ProjectCast,
5
+ ProjectTopLevel,
6
+ Strict,
7
+ )
8
+ from .dbt import Dbt
9
+ from .dependencies import Context, SparkContext
10
+ from .pipeline import (
11
+ Pipeline,
12
+ PipelineExecutionError,
13
+ PipelineOverrideError,
14
+ Step,
15
+ )
16
+ from .schemas import Schema
17
+ from .transform import transform
18
+
19
+ __all__ = [
20
+ "Context",
21
+ "Dbt",
22
+ "Pipeline",
23
+ "PipelineExecutionError",
24
+ "PipelineOverrideError",
25
+ "Project",
26
+ "ProjectCast",
27
+ "ProjectTopLevel",
28
+ "Schema",
29
+ "SparkContext",
30
+ "Step",
31
+ "Strict",
32
+ "main",
33
+ "transform",
34
+ ]
@@ -0,0 +1,17 @@
1
+ from .base import (
2
+ DataFrameBackend,
3
+ backend_for_frame_type,
4
+ backend_for_schema_type,
5
+ backend_for_value,
6
+ get_backend,
7
+ register_backend,
8
+ )
9
+
10
+ __all__ = [
11
+ "DataFrameBackend",
12
+ "backend_for_frame_type",
13
+ "backend_for_schema_type",
14
+ "backend_for_value",
15
+ "get_backend",
16
+ "register_backend",
17
+ ]
@@ -0,0 +1,144 @@
1
+ from collections.abc import Sequence
2
+ from importlib import import_module
3
+ from importlib.util import find_spec
4
+ from typing import Any, Protocol
5
+
6
+ from ..model_schema import ModelSchema
7
+ from ..schema_types import CoercionMode
8
+
9
+
10
+ class DataFrameBackend(Protocol):
11
+ name: str
12
+
13
+ @property
14
+ def dataframe_types(self) -> tuple[type, ...]: ...
15
+
16
+ @property
17
+ def schema_types(self) -> tuple[type, ...]: ...
18
+
19
+ def compile_schema(self, schema: ModelSchema[Any]) -> object: ...
20
+
21
+ def coerce_dataframe(
22
+ self,
23
+ dataframe: object,
24
+ schema: ModelSchema[Any],
25
+ mode: CoercionMode,
26
+ ) -> object: ...
27
+
28
+ def create_dataframe(
29
+ self,
30
+ rows: Sequence[object],
31
+ schema: ModelSchema[Any],
32
+ **kwargs: object,
33
+ ) -> object: ...
34
+
35
+
36
+ _BACKENDS: dict[str, DataFrameBackend] = {}
37
+ _BUILTINS_LOADED = False
38
+
39
+
40
+ def register_backend(backend: DataFrameBackend) -> None:
41
+ existing = _BACKENDS.get(backend.name)
42
+ if existing is backend:
43
+ return
44
+ if existing is not None:
45
+ raise ValueError(f"Backend '{backend.name}' is already registered")
46
+
47
+ for registered in _BACKENDS.values():
48
+ _raise_for_type_overlap(
49
+ backend,
50
+ registered,
51
+ attribute="dataframe_types",
52
+ kind="dataframe",
53
+ )
54
+ _raise_for_type_overlap(
55
+ backend,
56
+ registered,
57
+ attribute="schema_types",
58
+ kind="schema",
59
+ )
60
+ _BACKENDS[backend.name] = backend
61
+
62
+
63
+ def _raise_for_type_overlap(
64
+ backend: DataFrameBackend,
65
+ registered: DataFrameBackend,
66
+ *,
67
+ attribute: str,
68
+ kind: str,
69
+ ) -> None:
70
+ candidate_types = getattr(backend, attribute)
71
+ registered_types = getattr(registered, attribute)
72
+ for candidate in candidate_types:
73
+ for registered_type in registered_types:
74
+ if issubclass(candidate, registered_type) or issubclass(
75
+ registered_type, candidate
76
+ ):
77
+ raise TypeError(
78
+ f"Backend '{backend.name}' has ambiguous {kind} type "
79
+ f"{candidate!r} with backend '{registered.name}'"
80
+ )
81
+
82
+
83
+ def _load_builtin_backends() -> None:
84
+ global _BUILTINS_LOADED
85
+ if _BUILTINS_LOADED:
86
+ return
87
+ _BUILTINS_LOADED = True
88
+ import_module("data_joinery.backends.spark")
89
+ if find_spec("polars") is not None:
90
+ import_module("data_joinery.backends.polars")
91
+
92
+
93
+ def get_backend(name: str) -> DataFrameBackend:
94
+ if name not in _BACKENDS:
95
+ _load_builtin_backends()
96
+ try:
97
+ return _BACKENDS[name]
98
+ except KeyError as error:
99
+ raise LookupError(
100
+ f"No dataframe backend named '{name}' is registered"
101
+ ) from error
102
+
103
+
104
+ def backend_for_frame_type(frame_type: type) -> DataFrameBackend | None:
105
+ if not isinstance(frame_type, type):
106
+ return None
107
+ backend = _find_backend_for_type(frame_type)
108
+ if backend is not None:
109
+ return backend
110
+ _load_builtin_backends()
111
+ return _find_backend_for_type(frame_type)
112
+
113
+
114
+ def backend_for_schema_type(schema_type: type) -> DataFrameBackend | None:
115
+ if not isinstance(schema_type, type):
116
+ return None
117
+ backend = _find_backend_for_type(schema_type, attribute="schema_types")
118
+ if backend is not None:
119
+ return backend
120
+ _load_builtin_backends()
121
+ return _find_backend_for_type(schema_type, attribute="schema_types")
122
+
123
+
124
+ def _find_backend_for_type(
125
+ frame_type: type,
126
+ *,
127
+ attribute: str = "dataframe_types",
128
+ ) -> DataFrameBackend | None:
129
+ matches = [
130
+ backend
131
+ for backend in _BACKENDS.values()
132
+ if any(
133
+ issubclass(frame_type, candidate)
134
+ for candidate in getattr(backend, attribute)
135
+ )
136
+ ]
137
+ if len(matches) > 1:
138
+ names = ", ".join(sorted(backend.name for backend in matches))
139
+ raise LookupError(f"Multiple dataframe backends match {frame_type!r}: {names}")
140
+ return matches[0] if matches else None
141
+
142
+
143
+ def backend_for_value(value: object) -> DataFrameBackend | None:
144
+ return backend_for_frame_type(type(value))