overture-schema-pyspark 0.1.1.dev0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. overture_schema_pyspark-0.1.1.dev0/PKG-INFO +12 -0
  2. overture_schema_pyspark-0.1.1.dev0/pyproject.toml +28 -0
  3. overture_schema_pyspark-0.1.1.dev0/pyproject.toml.orig +30 -0
  4. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/__init__.py +52 -0
  5. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/_pyspark_version.py +62 -0
  6. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/_registry.py +104 -0
  7. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/check.py +59 -0
  8. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/cli.py +309 -0
  9. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/__init__.py +1 -0
  10. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/_schema_structs.py +22 -0
  11. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/column_patterns.py +169 -0
  12. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/constraint_expressions.py +589 -0
  13. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/addresses/address.py +494 -0
  14. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/bathymetry.py +383 -0
  15. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/infrastructure.py +617 -0
  16. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land.py +597 -0
  17. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land_cover.py +383 -0
  18. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land_use.py +617 -0
  19. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/water.py +578 -0
  20. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/buildings/building.py +678 -0
  21. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/buildings/building_part.py +688 -0
  22. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division.py +1109 -0
  23. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division_area.py +740 -0
  24. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division_boundary.py +618 -0
  25. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/places/place.py +1100 -0
  26. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/transportation/connector.py +333 -0
  27. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/transportation/segment.py +2852 -0
  28. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/py.typed +0 -0
  29. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/schema_check.py +124 -0
  30. overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/validate.py +425 -0
@@ -0,0 +1,12 @@
1
+ Metadata-Version: 2.4
2
+ Name: overture-schema-pyspark
3
+ Version: 0.1.1.dev0
4
+ Summary: PySpark validation expressions for Overture Maps data
5
+ License-Expression: MIT
6
+ Requires-Dist: click>=8.0
7
+ Requires-Dist: importlib-resources>=6.2 ; python_full_version < '3.13'
8
+ Requires-Dist: overture-schema-system>=0.1.1
9
+ Requires-Dist: packaging>=22
10
+ Requires-Dist: pyspark>=3.4 ; extra == 'spark'
11
+ Requires-Python: >=3.10
12
+ Provides-Extra: spark
@@ -0,0 +1,28 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.32,<0.13"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ dependencies = [
7
+ "click>=8.0",
8
+ "importlib-resources>=6.2; python_version < '3.13'",
9
+ "overture-schema-system>=0.1.1",
10
+ "packaging>=22",
11
+ ]
12
+ description = "PySpark validation expressions for Overture Maps data"
13
+ license = "MIT"
14
+ name = "overture-schema-pyspark"
15
+ requires-python = ">=3.10"
16
+ version = "0.1.1.dev0"
17
+
18
+ [project.optional-dependencies]
19
+ spark = ["pyspark>=3.4"]
20
+
21
+ [project.scripts]
22
+ overture-validate = "overture.schema.pyspark.cli:validate_cli"
23
+
24
+ [tool.uv.build-backend]
25
+ module-name = "overture.schema.pyspark"
26
+
27
+ [tool.uv.sources.overture-schema-system]
28
+ workspace = true
@@ -0,0 +1,30 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.32,<0.13"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ dependencies = [
7
+ "click>=8.0",
8
+ # 6.2 is the first release whose `files()` resolves a namespace portion
9
+ # inside a zip; the stdlib gained the same fix in 3.13.
10
+ "importlib-resources>=6.2; python_version < '3.13'",
11
+ "overture-schema-system>=0.1.1",
12
+ "packaging>=22",
13
+ ]
14
+ description = "PySpark validation expressions for Overture Maps data"
15
+ license = "MIT"
16
+ name = "overture-schema-pyspark"
17
+ requires-python = ">=3.10"
18
+ version = "0.1.1.dev0"
19
+
20
+ [project.optional-dependencies]
21
+ spark = ["pyspark>=3.4"]
22
+
23
+ [project.scripts]
24
+ overture-validate = "overture.schema.pyspark.cli:validate_cli"
25
+
26
+ [tool.uv.build-backend]
27
+ module-name = "overture.schema.pyspark"
28
+
29
+ [tool.uv.sources]
30
+ overture-schema-system = { workspace = true }
@@ -0,0 +1,52 @@
1
+ """PySpark validation expressions for Overture Maps data."""
2
+
3
+ from ._pyspark_version import pyspark_version_problem
4
+
5
+ # pyspark is an optional extra (the `spark` extra): a bare install lets this
6
+ # package's metadata and console script resolve, but every module below needs
7
+ # pyspark itself to do anything. Probe for it here -- this module runs on any
8
+ # import of any submodule -- so a bare install gets an actionable message. A
9
+ # pyspark that is present but broken still raises its own error, because the
10
+ # imports below run for real.
11
+ try:
12
+ import pyspark
13
+ except ModuleNotFoundError as exc:
14
+ if exc.name != "pyspark":
15
+ raise
16
+ raise ModuleNotFoundError(
17
+ "overture-schema-pyspark requires PySpark, which isn't installed. "
18
+ "Install it with `pip install overture-schema-pyspark[spark]`, or run "
19
+ "in an environment that already provides PySpark (e.g. a Spark cluster)."
20
+ ) from exc
21
+
22
+ # Installing without the extra leaves no resolver to enforce the version floor
23
+ # declared alongside it, so enforce it here, against the PySpark that actually
24
+ # turned up. The floor is read back out of this package's own metadata.
25
+ if _problem := pyspark_version_problem(getattr(pyspark, "__version__", None)):
26
+ raise ImportError(_problem)
27
+
28
+ from .check import Check, CheckShape
29
+ from .schema_check import SchemaMismatch, compare_schemas
30
+ from .validate import (
31
+ ValidationResult,
32
+ evaluate_checks,
33
+ explain_errors,
34
+ filter_errors,
35
+ model_keys,
36
+ model_names,
37
+ validate_model,
38
+ )
39
+
40
+ __all__ = [
41
+ "Check",
42
+ "CheckShape",
43
+ "SchemaMismatch",
44
+ "ValidationResult",
45
+ "compare_schemas",
46
+ "evaluate_checks",
47
+ "explain_errors",
48
+ "model_keys",
49
+ "model_names",
50
+ "filter_errors",
51
+ "validate_model",
52
+ ]
@@ -0,0 +1,62 @@
1
+ """The PySpark version floor this package declares, checked at import time."""
2
+
3
+ import importlib.metadata
4
+
5
+ from packaging.requirements import Requirement
6
+ from packaging.specifiers import SpecifierSet
7
+ from packaging.utils import canonicalize_name
8
+
9
+ # The distribution this module ships in. A name that stopped resolving would
10
+ # disable the check below in silence -- no floor found reads exactly like a
11
+ # floor satisfied -- so a test pins that this one finds real metadata.
12
+ _DISTRIBUTION = "overture-schema-pyspark"
13
+
14
+
15
+ def declared_pyspark_specifier() -> SpecifierSet | None:
16
+ """Return the PySpark version range this package declares, if readable.
17
+
18
+ The range lives in pyproject.toml (`pyspark>=N` on the `spark` extra) and
19
+ reaches the installed distribution as a `Requires-Dist` entry. Reading it
20
+ back from there rather than restating it here means one declaration, so a
21
+ check against it cannot drift from what the package actually requires.
22
+
23
+ Returns None when the metadata isn't installed -- a source tree run
24
+ without an install -- because there is then no declaration to enforce.
25
+ """
26
+ try:
27
+ declared = importlib.metadata.requires(_DISTRIBUTION) or ()
28
+ except importlib.metadata.PackageNotFoundError:
29
+ return None
30
+ for raw in declared:
31
+ requirement = Requirement(raw)
32
+ if canonicalize_name(requirement.name) == "pyspark":
33
+ return requirement.specifier
34
+ return None
35
+
36
+
37
+ def pyspark_version_problem(version: str | None) -> str | None:
38
+ """Describe how `version` falls outside the declared range, or None.
39
+
40
+ Installing without the `spark` extra is the case the extra exists for --
41
+ a runtime that provides its own PySpark -- and it is also the case no
42
+ resolver sees, so nothing enforces the range at install time. This is
43
+ where it gets enforced instead.
44
+
45
+ `version` is `pyspark.__version__` rather than the version recorded in
46
+ PySpark's own metadata, because a PySpark supplied by a Spark
47
+ distribution is on `sys.path` without a `dist-info` directory to read.
48
+ A PySpark that reports no version passes: unjudgeable is not out of range.
49
+ """
50
+ specifier = declared_pyspark_specifier()
51
+ if version is None or specifier is None:
52
+ return None
53
+ # Prereleases count: Spark ships release candidates and dev builds, and a
54
+ # `>=` floor is a statement about the release they belong to.
55
+ if specifier.contains(version, prereleases=True):
56
+ return None
57
+ return (
58
+ f"overture-schema-pyspark requires PySpark {specifier}, but PySpark "
59
+ f"{version} is installed. Upgrade the PySpark in this environment, or "
60
+ f"install this package with its extra (`pip install "
61
+ f"overture-schema-pyspark[spark]`) to let the resolver choose one."
62
+ )
@@ -0,0 +1,104 @@
1
+ """Runtime registry of feature validations.
2
+
3
+ Built at import time by walking the generated `expressions.generated`
4
+ namespace and collecting every module that exposes the
5
+ codegen-emitted `ENTRY_POINT` and `MODEL_VALIDATION` constants.
6
+
7
+ The generated tree is the runtime source of truth: the registry
8
+ contains exactly what was generated, regardless of which theme
9
+ packages are installed alongside the pyspark package. A missing
10
+ `expressions/generated/` subtree simply yields an empty registry --
11
+ the package still imports cleanly.
12
+
13
+ The tree is read through `importlib.resources`, which resolves a
14
+ namespace portion whether it is a directory on disk or a member of an
15
+ archive. Any wheel left unextracted on `sys.path` is zipimported and
16
+ `pathlib` cannot traverse into it -- Spark ships wheels that way with
17
+ `--py-files`, as does AWS Glue with `--extra-py-files`.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import importlib
23
+ import logging
24
+ import sys
25
+ from collections.abc import Iterator
26
+
27
+ if sys.version_info >= (3, 13):
28
+ from importlib.resources import files
29
+ from importlib.resources.abc import Traversable
30
+ else:
31
+ # `importlib.resources.files` raises `NotADirectoryError` for a namespace
32
+ # package with any non-directory portion through Python 3.12; the backport
33
+ # carries the 3.13 fix. Runtimes known to sit below it, where this is always
34
+ # the branch taken: AWS Glue 4.0 (Python 3.10) and Glue 5.0 (3.11). Other
35
+ # runtimes that ship wheels unextracted belong on that list -- if you hit
36
+ # this somewhere else, please add what you find.
37
+ from importlib_resources import files
38
+ from importlib_resources.abc import Traversable
39
+
40
+ from .check import ModelValidation
41
+
42
+ logger = logging.getLogger(__name__)
43
+
44
+ _GENERATED_ROOT = "overture.schema.pyspark.expressions.generated"
45
+
46
+
47
+ def _iter_generated_module_names(root: str = _GENERATED_ROOT) -> list[str]:
48
+ """Return the dotted names of every generated module under `root`.
49
+
50
+ The generated tree is PEP 420 (no `__init__.py`), so its subdirectories
51
+ are namespace packages, which `pkgutil.walk_packages` skips. It is walked
52
+ as resources instead: every `.py` below `root`, keyed to a dotted name.
53
+ `files` multiplexes every portion of the namespace, so a tree assembled
54
+ from more than one distribution is walked whole.
55
+ """
56
+
57
+ def walk(node: Traversable, prefix: tuple[str, ...]) -> Iterator[str]:
58
+ for child in node.iterdir():
59
+ if child.is_dir():
60
+ yield from walk(child, (*prefix, child.name))
61
+ elif child.name.endswith(".py") and child.name != "__init__.py":
62
+ yield ".".join([root, *prefix, child.name[: -len(".py")]])
63
+
64
+ try:
65
+ anchor = files(root)
66
+ except ModuleNotFoundError:
67
+ return []
68
+ return sorted(walk(anchor, ()))
69
+
70
+
71
+ def _walk() -> tuple[dict[str, ModelValidation], dict[str, dict[str, str]]]:
72
+ """Walk the generated tree and collect registry + partition map.
73
+
74
+ Returns a `(registry, partition_map)` pair:
75
+
76
+ * `registry` keys every feature by its `ENTRY_POINT` value.
77
+ * `partition_map` keys partitioned features by entry-point, mapping
78
+ to a Hive partition dict (e.g. `{"theme": "places", "type":
79
+ "place"}`) for path construction. Features with no `PARTITIONS`
80
+ data (empty dict) are omitted; the codegen only sets `PARTITIONS`
81
+ when the data lake organizes the feature by Hive partitions.
82
+ `type` is appended here from the module file name so consumers
83
+ get a complete partition path without the codegen having to
84
+ duplicate the type value.
85
+ """
86
+ registry: dict[str, ModelValidation] = {}
87
+ partition_map: dict[str, dict[str, str]] = {}
88
+
89
+ for name in _iter_generated_module_names():
90
+ module = importlib.import_module(name)
91
+ entry_point = getattr(module, "ENTRY_POINT", None)
92
+ validation = getattr(module, "MODEL_VALIDATION", None)
93
+ if entry_point is None or validation is None:
94
+ continue
95
+ registry[entry_point] = validation
96
+ partitions = getattr(module, "PARTITIONS", None) or {}
97
+ if partitions:
98
+ feature_type = name.rsplit(".", 1)[-1]
99
+ partition_map[entry_point] = {**partitions, "type": feature_type}
100
+
101
+ return registry, partition_map
102
+
103
+
104
+ REGISTRY, PARTITION_MAP = _walk()
@@ -0,0 +1,59 @@
1
+ """Check dataclass — interface between expression builders and composition."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from dataclasses import dataclass
7
+ from enum import Enum
8
+
9
+ from overture.schema.system.geometric import GeometryType
10
+ from pyspark.sql import Column
11
+ from pyspark.sql.types import StructType
12
+
13
+
14
+ class CheckShape(Enum):
15
+ """How the composition layer handles a check expression."""
16
+
17
+ SCALAR = "scalar" # expression returns nullable string
18
+ ARRAY = "array" # expression returns array<string>
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class Check:
23
+ """One validation check.
24
+
25
+ `field` identifies what the check is about (for error column naming
26
+ and report grouping), not how to access the data. The expression in
27
+ `expr` already encodes the access pattern.
28
+
29
+ `expr` and `read_columns` are two views of one computation, and each is
30
+ a "column" in a different sense. `read_columns` are real columns of the
31
+ underlying schema model -- the top-level columns the check must read to
32
+ evaluate. There is always at least one; a model-level constraint that
33
+ spans fields names several, plus any discriminator a variant gate reads.
34
+ `expr` is a *virtual column*: it is not a column of the schema model but
35
+ one synthesized by the generated validation machinery to hold the
36
+ composed expression the Spark engine evaluates. The two travel together
37
+ because the builder knows the read-set as it composes `expr`; recording
38
+ it is surer than recovering it from the finished `Column`.
39
+
40
+ `validate_model` drops a check when any column in `read_columns` is
41
+ skipped or structurally absent, so an unresolvable `F.col()` never
42
+ reaches Spark; it also treats these as the columns a check can be
43
+ suppressed by name.
44
+ """
45
+
46
+ field: str
47
+ name: str
48
+ expr: Column
49
+ shape: CheckShape
50
+ read_columns: frozenset[str]
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class ModelValidation:
55
+ """Pairs an expected schema with check builders for a feature type."""
56
+
57
+ schema: StructType
58
+ checks: Callable[[], list[Check]]
59
+ geometry_types: tuple[GeometryType, ...] = ()
@@ -0,0 +1,309 @@
1
+ """CLI entry point for validation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+ from collections.abc import Collection, Mapping
7
+ from dataclasses import dataclass
8
+
9
+ import click
10
+
11
+ from overture.schema.system.discovery import resolve_entry_point_key
12
+ from overture.schema.system.geometric import GeometryType
13
+ from pyspark.errors import AnalysisException
14
+ from pyspark.sql import DataFrame, SparkSession
15
+
16
+ from ._registry import PARTITION_MAP, REGISTRY
17
+ from .validate import (
18
+ explain_errors,
19
+ model_names,
20
+ validate_model,
21
+ )
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class ReadSpec:
26
+ """Parquet read plan.
27
+
28
+ `data_path` selects the files to read; `base_path`, when set, tells
29
+ Spark where to start discovering Hive partition columns.
30
+ """
31
+
32
+ data_path: str
33
+ base_path: str | None = None
34
+
35
+
36
+ def absent_column(exc: AnalysisException, columns: Collection[str]) -> str | None:
37
+ """Return the top-level column named by an unresolved-column error, if absent.
38
+
39
+ Returns the column name only when `exc` is an `UNRESOLVED_COLUMN` error
40
+ whose target is genuinely missing from `columns` -- the case a re-run with
41
+ `--skip-columns` resolves. Every other `AnalysisException` (a struct field
42
+ accessed on a scalar, a type mismatch, an unresolved column that is in fact
43
+ present) returns None, marking it a generator or expression bug to surface
44
+ rather than steer toward `--skip-columns`.
45
+
46
+ Parameters
47
+ ----------
48
+ exc
49
+ The exception raised while Spark planned the check expressions.
50
+ columns
51
+ The data's top-level column names (`df.columns`).
52
+ """
53
+ # getCondition() is pyspark 4.0+; the 3.4 floor exposes the same value as
54
+ # getErrorClass() (renamed, then deprecated, in 4.0).
55
+ get_condition = getattr(exc, "getCondition", None)
56
+ condition = get_condition() if get_condition is not None else exc.getErrorClass()
57
+ if condition is None or not condition.startswith("UNRESOLVED_COLUMN"):
58
+ return None
59
+ object_name = (exc.getMessageParameters() or {}).get("objectName")
60
+ if not object_name:
61
+ return None
62
+ # objectName is backtick-quoted, e.g. `phantom` or `bbox`.`xmin`; the
63
+ # top-level segment is the column df.columns would carry.
64
+ top_level = object_name.split(".", 1)[0].strip("`")
65
+ return top_level if top_level not in columns else None
66
+
67
+
68
+ def resolve_read(path: str, partitions: Mapping[str, str] | None) -> ReadSpec:
69
+ """Determine read strategy from path structure.
70
+
71
+ The partition map is an ordered Hive hierarchy
72
+ (`{"theme": "buildings", "type": "building"}`). A path supplies a
73
+ prefix of it; the leaves below the deepest level already present are
74
+ appended so the read always lands on a single feature type. Cases:
75
+
76
+ 1. **Individual file** (`*.parquet`) or no partitions -- read
77
+ directly; data already contains the partition columns inline.
78
+ 2. **Release root** (no partition directories) -- append the full
79
+ partition path and set `basePath` to the original path.
80
+ 3. **Partial partition path** (`theme=X/`) -- append the missing
81
+ leaves (`type=Y`) so a single feature's checks aren't run against
82
+ every type sharing the theme directory.
83
+ 4. **Leaf partition path** (`theme=X/type=Y/`) -- nothing to append;
84
+ read it directly with `basePath` derived.
85
+ """
86
+ stripped = path.rstrip("/")
87
+
88
+ # Individual file or no partition mapping — data has partition columns inline
89
+ if stripped.endswith(".parquet") or not partitions:
90
+ return ReadSpec(data_path=path)
91
+
92
+ keys = list(partitions)
93
+ # Partition levels already present in the path, in hierarchy order.
94
+ present = [i for i, key in enumerate(keys) if f"/{key}=" in stripped]
95
+ depth = present[-1] + 1 if present else 0 # count of levels already filled
96
+ leaves = "/".join(f"{key}={partitions[key]}" for key in keys[depth:])
97
+
98
+ if not present:
99
+ # Release root — append the full partition path; it is the base.
100
+ return ReadSpec(data_path=f"{stripped}/{leaves}", base_path=stripped)
101
+
102
+ # Path already contains partition directories: the base is the release
103
+ # root (before the first one); append any leaves below the deepest
104
+ # present level (none for a leaf path, which then reads as-is).
105
+ base_idx = stripped.find(f"/{keys[present[0]]}=")
106
+ data_path = f"{stripped}/{leaves}" if leaves else path
107
+ return ReadSpec(data_path=data_path, base_path=stripped[:base_idx])
108
+
109
+
110
+ def read_feature(spark: SparkSession, spec: ReadSpec) -> DataFrame:
111
+ """Read a DataFrame according to a ReadSpec."""
112
+ reader = spark.read
113
+ if spec.base_path:
114
+ reader = reader.option("basePath", spec.base_path)
115
+ return reader.parquet(spec.data_path)
116
+
117
+
118
+ _S3A_DEFAULTS: dict[str, str] = {
119
+ "spark.jars.packages": "org.apache.hadoop:hadoop-aws:3.4.1",
120
+ "spark.hadoop.fs.s3a.impl": "org.apache.hadoop.fs.s3a.S3AFileSystem",
121
+ "spark.hadoop.fs.s3a.aws.credentials.provider": (
122
+ "org.apache.hadoop.fs.s3a.AnonymousAWSCredentialsProvider"
123
+ ),
124
+ }
125
+
126
+ _LARGE_GEOMETRY_TYPES = frozenset(
127
+ {
128
+ GeometryType.LINE_STRING,
129
+ GeometryType.MULTI_LINE_STRING,
130
+ GeometryType.POLYGON,
131
+ GeometryType.MULTI_POLYGON,
132
+ GeometryType.GEOMETRY_COLLECTION,
133
+ }
134
+ )
135
+
136
+
137
+ def _may_have_large_geometry(feature_key: str) -> bool:
138
+ """Whether a registered feature's geometries may be large.
139
+
140
+ Returns True when the registered geometry types include
141
+ (multi)linestrings, (multi)polygons, or geometry collections,
142
+ or when geometry types are unspecified (safe default).
143
+ """
144
+ validation = REGISTRY[feature_key]
145
+ if not validation.geometry_types:
146
+ return True
147
+ return bool(set(validation.geometry_types) & _LARGE_GEOMETRY_TYPES)
148
+
149
+
150
+ def _spark_config(path: str, conf: tuple[str, ...], feature_key: str) -> dict[str, str]:
151
+ """Build Spark config dict with safe defaults.
152
+
153
+ Disables the vectorized Parquet reader for features with large
154
+ geometries (polygons, linestrings) to avoid OOM on WKB binary
155
+ columns. Adds S3A credentials for `s3a://` paths. User-supplied
156
+ `--conf` values override any defaults.
157
+ """
158
+ config: dict[str, str] = {}
159
+ if _may_have_large_geometry(feature_key):
160
+ config["spark.sql.parquet.enableVectorizedReader"] = "false"
161
+ if path.startswith("s3a://"):
162
+ config.update(_S3A_DEFAULTS)
163
+ for pair in conf:
164
+ key, _, value = pair.partition("=")
165
+ config[key] = value
166
+ return config
167
+
168
+
169
+ @click.command("overture-validate")
170
+ @click.argument("feature_type")
171
+ @click.argument("path")
172
+ @click.option("-o", "--output", default=None, help="Output path for validated Parquet.")
173
+ @click.option(
174
+ "--head",
175
+ "head_n",
176
+ default=20,
177
+ type=int,
178
+ show_default=True,
179
+ help="Error rows to display.",
180
+ )
181
+ @click.option("--conf", multiple=True, help="Spark config key=value pairs.")
182
+ @click.option(
183
+ "--count-only",
184
+ is_flag=True,
185
+ default=False,
186
+ help="Report error count only; skip explain/unpivot.",
187
+ )
188
+ @click.option(
189
+ "--skip-schema-check",
190
+ is_flag=True,
191
+ default=False,
192
+ help="Warn on schema mismatches instead of aborting.",
193
+ )
194
+ @click.option(
195
+ "--skip-columns",
196
+ multiple=True,
197
+ help="Columns declared absent from data; skips their checks.",
198
+ )
199
+ @click.option(
200
+ "--ignore-extra-columns",
201
+ multiple=True,
202
+ help="Extra data columns to ignore in schema comparison.",
203
+ )
204
+ @click.option(
205
+ "--suppress",
206
+ "suppress_specs",
207
+ multiple=True,
208
+ help="Suppress checks: FIELD (all checks) or FIELD:CHECK (specific).",
209
+ )
210
+ def validate_cli(
211
+ feature_type: str,
212
+ path: str,
213
+ output: str | None,
214
+ head_n: int,
215
+ conf: tuple[str, ...],
216
+ count_only: bool,
217
+ skip_schema_check: bool,
218
+ skip_columns: tuple[str, ...],
219
+ ignore_extra_columns: tuple[str, ...],
220
+ suppress_specs: tuple[str, ...],
221
+ ) -> None:
222
+ """Validate Overture data at PATH and write annotated Parquet."""
223
+ try:
224
+ resolved = resolve_entry_point_key(feature_type, REGISTRY)
225
+ except ValueError:
226
+ click.echo(
227
+ f"Unknown type '{feature_type}'. Known: {', '.join(model_names())}",
228
+ err=True,
229
+ )
230
+ sys.exit(1)
231
+
232
+ builder = SparkSession.builder
233
+ for key, value in _spark_config(path, conf, resolved).items():
234
+ builder = builder.config(key, value)
235
+ spark = builder.getOrCreate()
236
+ spark.sparkContext.setLogLevel("ERROR")
237
+
238
+ spec = resolve_read(path, PARTITION_MAP.get(resolved))
239
+ df = read_feature(spark, spec)
240
+
241
+ suppress: list[str | tuple[str, str]] = []
242
+ for s in suppress_specs:
243
+ if ":" in s:
244
+ field, name = s.split(":", 1)
245
+ suppress.append((field, name))
246
+ else:
247
+ suppress.append(s)
248
+
249
+ try:
250
+ result = validate_model(
251
+ df,
252
+ resolved,
253
+ skip_columns=skip_columns,
254
+ ignore_extra_columns=ignore_extra_columns,
255
+ suppress=suppress,
256
+ )
257
+ except ValueError as e:
258
+ click.echo(str(e), err=True)
259
+ sys.exit(1)
260
+ except AnalysisException as e:
261
+ # Backstop, narrowed to the one cause `--skip-columns` can address: a
262
+ # check that names a column missing from the data. validate_model
263
+ # already drops checks for skipped and schema-absent columns, so this
264
+ # fires only on a column outside the expected schema -- offer the
265
+ # operator the skip lever and name the column. Every other
266
+ # AnalysisException (a type mismatch, a struct field read off a scalar)
267
+ # is a generator bug `--skip-columns` cannot fix; let it propagate as a
268
+ # traceback rather than mask it behind the skip hint.
269
+ column = absent_column(e, df.columns)
270
+ if column is None:
271
+ raise
272
+ click.echo(
273
+ f"A check references column '{column}', absent from the data at {path}.",
274
+ err=True,
275
+ )
276
+ click.echo(
277
+ f"Re-run with `--skip-columns {column}` to skip its checks, "
278
+ "or `--skip-schema-check`.",
279
+ err=True,
280
+ )
281
+ sys.exit(1)
282
+
283
+ if result.schema_mismatches:
284
+ click.echo(f"Schema mismatches for {resolved}:", err=True)
285
+ for m in result.schema_mismatches:
286
+ click.echo(f" {m.path}: expected {m.expected}, got {m.actual}", err=True)
287
+ if result.absent_columns:
288
+ flags = " ".join(f"--skip-columns {c}" for c in result.absent_columns)
289
+ click.echo(
290
+ f" Re-run with `{flags}` to skip missing columns.",
291
+ err=True,
292
+ )
293
+ if not skip_schema_check:
294
+ sys.exit(1)
295
+
296
+ total_rows, error_count = result.row_counts()
297
+ click.echo(f"{error_count} / {total_rows} rows with errors", err=True)
298
+
299
+ if error_count > 0:
300
+ if not count_only:
301
+ explained = explain_errors(result.evaluated, result.checks).drop("geometry")
302
+ if output and head_n > 0:
303
+ explained = explained.cache()
304
+ if output:
305
+ explained.write.mode("overwrite").parquet(output)
306
+ click.echo(f"Written to {output}", err=True)
307
+ if head_n > 0:
308
+ explained.show(head_n, truncate=False)
309
+ sys.exit(1)
@@ -0,0 +1 @@
1
+ """Expression builders and reusable PySpark column patterns."""
@@ -0,0 +1,22 @@
1
+ """Hand-written Spark StructType fragments for types the codegen can't generate.
2
+
3
+ The codegen builds feature schemas by walking Pydantic `BaseModel`
4
+ subclasses. `BBox` is a plain class, not a `BaseModel`, so extraction
5
+ can't reach it -- `BBOX_STRUCT` is hand-written here to fill the gap.
6
+ Every other nested type is a `BaseModel` and gets generated directly
7
+ into each feature module, which is why this file holds only the one
8
+ struct.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from pyspark.sql.types import DoubleType, StructField, StructType
14
+
15
+ BBOX_STRUCT = StructType(
16
+ [
17
+ StructField("xmin", DoubleType(), True),
18
+ StructField("xmax", DoubleType(), True),
19
+ StructField("ymin", DoubleType(), True),
20
+ StructField("ymax", DoubleType(), True),
21
+ ]
22
+ )