overture-schema-pyspark 0.1.1.dev0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- overture_schema_pyspark-0.1.1.dev0/PKG-INFO +12 -0
- overture_schema_pyspark-0.1.1.dev0/pyproject.toml +28 -0
- overture_schema_pyspark-0.1.1.dev0/pyproject.toml.orig +30 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/__init__.py +52 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/_pyspark_version.py +62 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/_registry.py +104 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/check.py +59 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/cli.py +309 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/__init__.py +1 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/_schema_structs.py +22 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/column_patterns.py +169 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/constraint_expressions.py +589 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/addresses/address.py +494 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/bathymetry.py +383 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/infrastructure.py +617 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land.py +597 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land_cover.py +383 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/land_use.py +617 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/base/water.py +578 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/buildings/building.py +678 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/buildings/building_part.py +688 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division.py +1109 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division_area.py +740 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/divisions/division_boundary.py +618 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/places/place.py +1100 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/transportation/connector.py +333 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/expressions/generated/overture/schema/transportation/segment.py +2852 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/py.typed +0 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/schema_check.py +124 -0
- overture_schema_pyspark-0.1.1.dev0/src/overture/schema/pyspark/validate.py +425 -0
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: overture-schema-pyspark
|
|
3
|
+
Version: 0.1.1.dev0
|
|
4
|
+
Summary: PySpark validation expressions for Overture Maps data
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Requires-Dist: click>=8.0
|
|
7
|
+
Requires-Dist: importlib-resources>=6.2 ; python_full_version < '3.13'
|
|
8
|
+
Requires-Dist: overture-schema-system>=0.1.1
|
|
9
|
+
Requires-Dist: packaging>=22
|
|
10
|
+
Requires-Dist: pyspark>=3.4 ; extra == 'spark'
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Provides-Extra: spark
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.32,<0.13"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
dependencies = [
|
|
7
|
+
"click>=8.0",
|
|
8
|
+
"importlib-resources>=6.2; python_version < '3.13'",
|
|
9
|
+
"overture-schema-system>=0.1.1",
|
|
10
|
+
"packaging>=22",
|
|
11
|
+
]
|
|
12
|
+
description = "PySpark validation expressions for Overture Maps data"
|
|
13
|
+
license = "MIT"
|
|
14
|
+
name = "overture-schema-pyspark"
|
|
15
|
+
requires-python = ">=3.10"
|
|
16
|
+
version = "0.1.1.dev0"
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
spark = ["pyspark>=3.4"]
|
|
20
|
+
|
|
21
|
+
[project.scripts]
|
|
22
|
+
overture-validate = "overture.schema.pyspark.cli:validate_cli"
|
|
23
|
+
|
|
24
|
+
[tool.uv.build-backend]
|
|
25
|
+
module-name = "overture.schema.pyspark"
|
|
26
|
+
|
|
27
|
+
[tool.uv.sources.overture-schema-system]
|
|
28
|
+
workspace = true
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.32,<0.13"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
dependencies = [
|
|
7
|
+
"click>=8.0",
|
|
8
|
+
# 6.2 is the first release whose `files()` resolves a namespace portion
|
|
9
|
+
# inside a zip; the stdlib gained the same fix in 3.13.
|
|
10
|
+
"importlib-resources>=6.2; python_version < '3.13'",
|
|
11
|
+
"overture-schema-system>=0.1.1",
|
|
12
|
+
"packaging>=22",
|
|
13
|
+
]
|
|
14
|
+
description = "PySpark validation expressions for Overture Maps data"
|
|
15
|
+
license = "MIT"
|
|
16
|
+
name = "overture-schema-pyspark"
|
|
17
|
+
requires-python = ">=3.10"
|
|
18
|
+
version = "0.1.1.dev0"
|
|
19
|
+
|
|
20
|
+
[project.optional-dependencies]
|
|
21
|
+
spark = ["pyspark>=3.4"]
|
|
22
|
+
|
|
23
|
+
[project.scripts]
|
|
24
|
+
overture-validate = "overture.schema.pyspark.cli:validate_cli"
|
|
25
|
+
|
|
26
|
+
[tool.uv.build-backend]
|
|
27
|
+
module-name = "overture.schema.pyspark"
|
|
28
|
+
|
|
29
|
+
[tool.uv.sources]
|
|
30
|
+
overture-schema-system = { workspace = true }
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""PySpark validation expressions for Overture Maps data."""
|
|
2
|
+
|
|
3
|
+
from ._pyspark_version import pyspark_version_problem
|
|
4
|
+
|
|
5
|
+
# pyspark is an optional extra (the `spark` extra): a bare install lets this
|
|
6
|
+
# package's metadata and console script resolve, but every module below needs
|
|
7
|
+
# pyspark itself to do anything. Probe for it here -- this module runs on any
|
|
8
|
+
# import of any submodule -- so a bare install gets an actionable message. A
|
|
9
|
+
# pyspark that is present but broken still raises its own error, because the
|
|
10
|
+
# imports below run for real.
|
|
11
|
+
try:
|
|
12
|
+
import pyspark
|
|
13
|
+
except ModuleNotFoundError as exc:
|
|
14
|
+
if exc.name != "pyspark":
|
|
15
|
+
raise
|
|
16
|
+
raise ModuleNotFoundError(
|
|
17
|
+
"overture-schema-pyspark requires PySpark, which isn't installed. "
|
|
18
|
+
"Install it with `pip install overture-schema-pyspark[spark]`, or run "
|
|
19
|
+
"in an environment that already provides PySpark (e.g. a Spark cluster)."
|
|
20
|
+
) from exc
|
|
21
|
+
|
|
22
|
+
# Installing without the extra leaves no resolver to enforce the version floor
|
|
23
|
+
# declared alongside it, so enforce it here, against the PySpark that actually
|
|
24
|
+
# turned up. The floor is read back out of this package's own metadata.
|
|
25
|
+
if _problem := pyspark_version_problem(getattr(pyspark, "__version__", None)):
|
|
26
|
+
raise ImportError(_problem)
|
|
27
|
+
|
|
28
|
+
from .check import Check, CheckShape
|
|
29
|
+
from .schema_check import SchemaMismatch, compare_schemas
|
|
30
|
+
from .validate import (
|
|
31
|
+
ValidationResult,
|
|
32
|
+
evaluate_checks,
|
|
33
|
+
explain_errors,
|
|
34
|
+
filter_errors,
|
|
35
|
+
model_keys,
|
|
36
|
+
model_names,
|
|
37
|
+
validate_model,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"Check",
|
|
42
|
+
"CheckShape",
|
|
43
|
+
"SchemaMismatch",
|
|
44
|
+
"ValidationResult",
|
|
45
|
+
"compare_schemas",
|
|
46
|
+
"evaluate_checks",
|
|
47
|
+
"explain_errors",
|
|
48
|
+
"model_keys",
|
|
49
|
+
"model_names",
|
|
50
|
+
"filter_errors",
|
|
51
|
+
"validate_model",
|
|
52
|
+
]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""The PySpark version floor this package declares, checked at import time."""
|
|
2
|
+
|
|
3
|
+
import importlib.metadata
|
|
4
|
+
|
|
5
|
+
from packaging.requirements import Requirement
|
|
6
|
+
from packaging.specifiers import SpecifierSet
|
|
7
|
+
from packaging.utils import canonicalize_name
|
|
8
|
+
|
|
9
|
+
# The distribution this module ships in. A name that stopped resolving would
|
|
10
|
+
# disable the check below in silence -- no floor found reads exactly like a
|
|
11
|
+
# floor satisfied -- so a test pins that this one finds real metadata.
|
|
12
|
+
_DISTRIBUTION = "overture-schema-pyspark"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def declared_pyspark_specifier() -> SpecifierSet | None:
|
|
16
|
+
"""Return the PySpark version range this package declares, if readable.
|
|
17
|
+
|
|
18
|
+
The range lives in pyproject.toml (`pyspark>=N` on the `spark` extra) and
|
|
19
|
+
reaches the installed distribution as a `Requires-Dist` entry. Reading it
|
|
20
|
+
back from there rather than restating it here means one declaration, so a
|
|
21
|
+
check against it cannot drift from what the package actually requires.
|
|
22
|
+
|
|
23
|
+
Returns None when the metadata isn't installed -- a source tree run
|
|
24
|
+
without an install -- because there is then no declaration to enforce.
|
|
25
|
+
"""
|
|
26
|
+
try:
|
|
27
|
+
declared = importlib.metadata.requires(_DISTRIBUTION) or ()
|
|
28
|
+
except importlib.metadata.PackageNotFoundError:
|
|
29
|
+
return None
|
|
30
|
+
for raw in declared:
|
|
31
|
+
requirement = Requirement(raw)
|
|
32
|
+
if canonicalize_name(requirement.name) == "pyspark":
|
|
33
|
+
return requirement.specifier
|
|
34
|
+
return None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def pyspark_version_problem(version: str | None) -> str | None:
|
|
38
|
+
"""Describe how `version` falls outside the declared range, or None.
|
|
39
|
+
|
|
40
|
+
Installing without the `spark` extra is the case the extra exists for --
|
|
41
|
+
a runtime that provides its own PySpark -- and it is also the case no
|
|
42
|
+
resolver sees, so nothing enforces the range at install time. This is
|
|
43
|
+
where it gets enforced instead.
|
|
44
|
+
|
|
45
|
+
`version` is `pyspark.__version__` rather than the version recorded in
|
|
46
|
+
PySpark's own metadata, because a PySpark supplied by a Spark
|
|
47
|
+
distribution is on `sys.path` without a `dist-info` directory to read.
|
|
48
|
+
A PySpark that reports no version passes: unjudgeable is not out of range.
|
|
49
|
+
"""
|
|
50
|
+
specifier = declared_pyspark_specifier()
|
|
51
|
+
if version is None or specifier is None:
|
|
52
|
+
return None
|
|
53
|
+
# Prereleases count: Spark ships release candidates and dev builds, and a
|
|
54
|
+
# `>=` floor is a statement about the release they belong to.
|
|
55
|
+
if specifier.contains(version, prereleases=True):
|
|
56
|
+
return None
|
|
57
|
+
return (
|
|
58
|
+
f"overture-schema-pyspark requires PySpark {specifier}, but PySpark "
|
|
59
|
+
f"{version} is installed. Upgrade the PySpark in this environment, or "
|
|
60
|
+
f"install this package with its extra (`pip install "
|
|
61
|
+
f"overture-schema-pyspark[spark]`) to let the resolver choose one."
|
|
62
|
+
)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Runtime registry of feature validations.
|
|
2
|
+
|
|
3
|
+
Built at import time by walking the generated `expressions.generated`
|
|
4
|
+
namespace and collecting every module that exposes the
|
|
5
|
+
codegen-emitted `ENTRY_POINT` and `MODEL_VALIDATION` constants.
|
|
6
|
+
|
|
7
|
+
The generated tree is the runtime source of truth: the registry
|
|
8
|
+
contains exactly what was generated, regardless of which theme
|
|
9
|
+
packages are installed alongside the pyspark package. A missing
|
|
10
|
+
`expressions/generated/` subtree simply yields an empty registry --
|
|
11
|
+
the package still imports cleanly.
|
|
12
|
+
|
|
13
|
+
The tree is read through `importlib.resources`, which resolves a
|
|
14
|
+
namespace portion whether it is a directory on disk or a member of an
|
|
15
|
+
archive. Any wheel left unextracted on `sys.path` is zipimported and
|
|
16
|
+
`pathlib` cannot traverse into it -- Spark ships wheels that way with
|
|
17
|
+
`--py-files`, as does AWS Glue with `--extra-py-files`.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import importlib
|
|
23
|
+
import logging
|
|
24
|
+
import sys
|
|
25
|
+
from collections.abc import Iterator
|
|
26
|
+
|
|
27
|
+
if sys.version_info >= (3, 13):
|
|
28
|
+
from importlib.resources import files
|
|
29
|
+
from importlib.resources.abc import Traversable
|
|
30
|
+
else:
|
|
31
|
+
# `importlib.resources.files` raises `NotADirectoryError` for a namespace
|
|
32
|
+
# package with any non-directory portion through Python 3.12; the backport
|
|
33
|
+
# carries the 3.13 fix. Runtimes known to sit below it, where this is always
|
|
34
|
+
# the branch taken: AWS Glue 4.0 (Python 3.10) and Glue 5.0 (3.11). Other
|
|
35
|
+
# runtimes that ship wheels unextracted belong on that list -- if you hit
|
|
36
|
+
# this somewhere else, please add what you find.
|
|
37
|
+
from importlib_resources import files
|
|
38
|
+
from importlib_resources.abc import Traversable
|
|
39
|
+
|
|
40
|
+
from .check import ModelValidation
|
|
41
|
+
|
|
42
|
+
logger = logging.getLogger(__name__)
|
|
43
|
+
|
|
44
|
+
_GENERATED_ROOT = "overture.schema.pyspark.expressions.generated"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _iter_generated_module_names(root: str = _GENERATED_ROOT) -> list[str]:
|
|
48
|
+
"""Return the dotted names of every generated module under `root`.
|
|
49
|
+
|
|
50
|
+
The generated tree is PEP 420 (no `__init__.py`), so its subdirectories
|
|
51
|
+
are namespace packages, which `pkgutil.walk_packages` skips. It is walked
|
|
52
|
+
as resources instead: every `.py` below `root`, keyed to a dotted name.
|
|
53
|
+
`files` multiplexes every portion of the namespace, so a tree assembled
|
|
54
|
+
from more than one distribution is walked whole.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def walk(node: Traversable, prefix: tuple[str, ...]) -> Iterator[str]:
|
|
58
|
+
for child in node.iterdir():
|
|
59
|
+
if child.is_dir():
|
|
60
|
+
yield from walk(child, (*prefix, child.name))
|
|
61
|
+
elif child.name.endswith(".py") and child.name != "__init__.py":
|
|
62
|
+
yield ".".join([root, *prefix, child.name[: -len(".py")]])
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
anchor = files(root)
|
|
66
|
+
except ModuleNotFoundError:
|
|
67
|
+
return []
|
|
68
|
+
return sorted(walk(anchor, ()))
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _walk() -> tuple[dict[str, ModelValidation], dict[str, dict[str, str]]]:
|
|
72
|
+
"""Walk the generated tree and collect registry + partition map.
|
|
73
|
+
|
|
74
|
+
Returns a `(registry, partition_map)` pair:
|
|
75
|
+
|
|
76
|
+
* `registry` keys every feature by its `ENTRY_POINT` value.
|
|
77
|
+
* `partition_map` keys partitioned features by entry-point, mapping
|
|
78
|
+
to a Hive partition dict (e.g. `{"theme": "places", "type":
|
|
79
|
+
"place"}`) for path construction. Features with no `PARTITIONS`
|
|
80
|
+
data (empty dict) are omitted; the codegen only sets `PARTITIONS`
|
|
81
|
+
when the data lake organizes the feature by Hive partitions.
|
|
82
|
+
`type` is appended here from the module file name so consumers
|
|
83
|
+
get a complete partition path without the codegen having to
|
|
84
|
+
duplicate the type value.
|
|
85
|
+
"""
|
|
86
|
+
registry: dict[str, ModelValidation] = {}
|
|
87
|
+
partition_map: dict[str, dict[str, str]] = {}
|
|
88
|
+
|
|
89
|
+
for name in _iter_generated_module_names():
|
|
90
|
+
module = importlib.import_module(name)
|
|
91
|
+
entry_point = getattr(module, "ENTRY_POINT", None)
|
|
92
|
+
validation = getattr(module, "MODEL_VALIDATION", None)
|
|
93
|
+
if entry_point is None or validation is None:
|
|
94
|
+
continue
|
|
95
|
+
registry[entry_point] = validation
|
|
96
|
+
partitions = getattr(module, "PARTITIONS", None) or {}
|
|
97
|
+
if partitions:
|
|
98
|
+
feature_type = name.rsplit(".", 1)[-1]
|
|
99
|
+
partition_map[entry_point] = {**partitions, "type": feature_type}
|
|
100
|
+
|
|
101
|
+
return registry, partition_map
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
REGISTRY, PARTITION_MAP = _walk()
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Check dataclass — interface between expression builders and composition."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from enum import Enum
|
|
8
|
+
|
|
9
|
+
from overture.schema.system.geometric import GeometryType
|
|
10
|
+
from pyspark.sql import Column
|
|
11
|
+
from pyspark.sql.types import StructType
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class CheckShape(Enum):
|
|
15
|
+
"""How the composition layer handles a check expression."""
|
|
16
|
+
|
|
17
|
+
SCALAR = "scalar" # expression returns nullable string
|
|
18
|
+
ARRAY = "array" # expression returns array<string>
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class Check:
|
|
23
|
+
"""One validation check.
|
|
24
|
+
|
|
25
|
+
`field` identifies what the check is about (for error column naming
|
|
26
|
+
and report grouping), not how to access the data. The expression in
|
|
27
|
+
`expr` already encodes the access pattern.
|
|
28
|
+
|
|
29
|
+
`expr` and `read_columns` are two views of one computation, and each is
|
|
30
|
+
a "column" in a different sense. `read_columns` are real columns of the
|
|
31
|
+
underlying schema model -- the top-level columns the check must read to
|
|
32
|
+
evaluate. There is always at least one; a model-level constraint that
|
|
33
|
+
spans fields names several, plus any discriminator a variant gate reads.
|
|
34
|
+
`expr` is a *virtual column*: it is not a column of the schema model but
|
|
35
|
+
one synthesized by the generated validation machinery to hold the
|
|
36
|
+
composed expression the Spark engine evaluates. The two travel together
|
|
37
|
+
because the builder knows the read-set as it composes `expr`; recording
|
|
38
|
+
it is surer than recovering it from the finished `Column`.
|
|
39
|
+
|
|
40
|
+
`validate_model` drops a check when any column in `read_columns` is
|
|
41
|
+
skipped or structurally absent, so an unresolvable `F.col()` never
|
|
42
|
+
reaches Spark; it also treats these as the columns a check can be
|
|
43
|
+
suppressed by name.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
field: str
|
|
47
|
+
name: str
|
|
48
|
+
expr: Column
|
|
49
|
+
shape: CheckShape
|
|
50
|
+
read_columns: frozenset[str]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class ModelValidation:
|
|
55
|
+
"""Pairs an expected schema with check builders for a feature type."""
|
|
56
|
+
|
|
57
|
+
schema: StructType
|
|
58
|
+
checks: Callable[[], list[Check]]
|
|
59
|
+
geometry_types: tuple[GeometryType, ...] = ()
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
"""CLI entry point for validation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from collections.abc import Collection, Mapping
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
import click
|
|
10
|
+
|
|
11
|
+
from overture.schema.system.discovery import resolve_entry_point_key
|
|
12
|
+
from overture.schema.system.geometric import GeometryType
|
|
13
|
+
from pyspark.errors import AnalysisException
|
|
14
|
+
from pyspark.sql import DataFrame, SparkSession
|
|
15
|
+
|
|
16
|
+
from ._registry import PARTITION_MAP, REGISTRY
|
|
17
|
+
from .validate import (
|
|
18
|
+
explain_errors,
|
|
19
|
+
model_names,
|
|
20
|
+
validate_model,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class ReadSpec:
|
|
26
|
+
"""Parquet read plan.
|
|
27
|
+
|
|
28
|
+
`data_path` selects the files to read; `base_path`, when set, tells
|
|
29
|
+
Spark where to start discovering Hive partition columns.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
data_path: str
|
|
33
|
+
base_path: str | None = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def absent_column(exc: AnalysisException, columns: Collection[str]) -> str | None:
|
|
37
|
+
"""Return the top-level column named by an unresolved-column error, if absent.
|
|
38
|
+
|
|
39
|
+
Returns the column name only when `exc` is an `UNRESOLVED_COLUMN` error
|
|
40
|
+
whose target is genuinely missing from `columns` -- the case a re-run with
|
|
41
|
+
`--skip-columns` resolves. Every other `AnalysisException` (a struct field
|
|
42
|
+
accessed on a scalar, a type mismatch, an unresolved column that is in fact
|
|
43
|
+
present) returns None, marking it a generator or expression bug to surface
|
|
44
|
+
rather than steer toward `--skip-columns`.
|
|
45
|
+
|
|
46
|
+
Parameters
|
|
47
|
+
----------
|
|
48
|
+
exc
|
|
49
|
+
The exception raised while Spark planned the check expressions.
|
|
50
|
+
columns
|
|
51
|
+
The data's top-level column names (`df.columns`).
|
|
52
|
+
"""
|
|
53
|
+
# getCondition() is pyspark 4.0+; the 3.4 floor exposes the same value as
|
|
54
|
+
# getErrorClass() (renamed, then deprecated, in 4.0).
|
|
55
|
+
get_condition = getattr(exc, "getCondition", None)
|
|
56
|
+
condition = get_condition() if get_condition is not None else exc.getErrorClass()
|
|
57
|
+
if condition is None or not condition.startswith("UNRESOLVED_COLUMN"):
|
|
58
|
+
return None
|
|
59
|
+
object_name = (exc.getMessageParameters() or {}).get("objectName")
|
|
60
|
+
if not object_name:
|
|
61
|
+
return None
|
|
62
|
+
# objectName is backtick-quoted, e.g. `phantom` or `bbox`.`xmin`; the
|
|
63
|
+
# top-level segment is the column df.columns would carry.
|
|
64
|
+
top_level = object_name.split(".", 1)[0].strip("`")
|
|
65
|
+
return top_level if top_level not in columns else None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def resolve_read(path: str, partitions: Mapping[str, str] | None) -> ReadSpec:
|
|
69
|
+
"""Determine read strategy from path structure.
|
|
70
|
+
|
|
71
|
+
The partition map is an ordered Hive hierarchy
|
|
72
|
+
(`{"theme": "buildings", "type": "building"}`). A path supplies a
|
|
73
|
+
prefix of it; the leaves below the deepest level already present are
|
|
74
|
+
appended so the read always lands on a single feature type. Cases:
|
|
75
|
+
|
|
76
|
+
1. **Individual file** (`*.parquet`) or no partitions -- read
|
|
77
|
+
directly; data already contains the partition columns inline.
|
|
78
|
+
2. **Release root** (no partition directories) -- append the full
|
|
79
|
+
partition path and set `basePath` to the original path.
|
|
80
|
+
3. **Partial partition path** (`theme=X/`) -- append the missing
|
|
81
|
+
leaves (`type=Y`) so a single feature's checks aren't run against
|
|
82
|
+
every type sharing the theme directory.
|
|
83
|
+
4. **Leaf partition path** (`theme=X/type=Y/`) -- nothing to append;
|
|
84
|
+
read it directly with `basePath` derived.
|
|
85
|
+
"""
|
|
86
|
+
stripped = path.rstrip("/")
|
|
87
|
+
|
|
88
|
+
# Individual file or no partition mapping — data has partition columns inline
|
|
89
|
+
if stripped.endswith(".parquet") or not partitions:
|
|
90
|
+
return ReadSpec(data_path=path)
|
|
91
|
+
|
|
92
|
+
keys = list(partitions)
|
|
93
|
+
# Partition levels already present in the path, in hierarchy order.
|
|
94
|
+
present = [i for i, key in enumerate(keys) if f"/{key}=" in stripped]
|
|
95
|
+
depth = present[-1] + 1 if present else 0 # count of levels already filled
|
|
96
|
+
leaves = "/".join(f"{key}={partitions[key]}" for key in keys[depth:])
|
|
97
|
+
|
|
98
|
+
if not present:
|
|
99
|
+
# Release root — append the full partition path; it is the base.
|
|
100
|
+
return ReadSpec(data_path=f"{stripped}/{leaves}", base_path=stripped)
|
|
101
|
+
|
|
102
|
+
# Path already contains partition directories: the base is the release
|
|
103
|
+
# root (before the first one); append any leaves below the deepest
|
|
104
|
+
# present level (none for a leaf path, which then reads as-is).
|
|
105
|
+
base_idx = stripped.find(f"/{keys[present[0]]}=")
|
|
106
|
+
data_path = f"{stripped}/{leaves}" if leaves else path
|
|
107
|
+
return ReadSpec(data_path=data_path, base_path=stripped[:base_idx])
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def read_feature(spark: SparkSession, spec: ReadSpec) -> DataFrame:
|
|
111
|
+
"""Read a DataFrame according to a ReadSpec."""
|
|
112
|
+
reader = spark.read
|
|
113
|
+
if spec.base_path:
|
|
114
|
+
reader = reader.option("basePath", spec.base_path)
|
|
115
|
+
return reader.parquet(spec.data_path)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
_S3A_DEFAULTS: dict[str, str] = {
|
|
119
|
+
"spark.jars.packages": "org.apache.hadoop:hadoop-aws:3.4.1",
|
|
120
|
+
"spark.hadoop.fs.s3a.impl": "org.apache.hadoop.fs.s3a.S3AFileSystem",
|
|
121
|
+
"spark.hadoop.fs.s3a.aws.credentials.provider": (
|
|
122
|
+
"org.apache.hadoop.fs.s3a.AnonymousAWSCredentialsProvider"
|
|
123
|
+
),
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
_LARGE_GEOMETRY_TYPES = frozenset(
|
|
127
|
+
{
|
|
128
|
+
GeometryType.LINE_STRING,
|
|
129
|
+
GeometryType.MULTI_LINE_STRING,
|
|
130
|
+
GeometryType.POLYGON,
|
|
131
|
+
GeometryType.MULTI_POLYGON,
|
|
132
|
+
GeometryType.GEOMETRY_COLLECTION,
|
|
133
|
+
}
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _may_have_large_geometry(feature_key: str) -> bool:
|
|
138
|
+
"""Whether a registered feature's geometries may be large.
|
|
139
|
+
|
|
140
|
+
Returns True when the registered geometry types include
|
|
141
|
+
(multi)linestrings, (multi)polygons, or geometry collections,
|
|
142
|
+
or when geometry types are unspecified (safe default).
|
|
143
|
+
"""
|
|
144
|
+
validation = REGISTRY[feature_key]
|
|
145
|
+
if not validation.geometry_types:
|
|
146
|
+
return True
|
|
147
|
+
return bool(set(validation.geometry_types) & _LARGE_GEOMETRY_TYPES)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _spark_config(path: str, conf: tuple[str, ...], feature_key: str) -> dict[str, str]:
|
|
151
|
+
"""Build Spark config dict with safe defaults.
|
|
152
|
+
|
|
153
|
+
Disables the vectorized Parquet reader for features with large
|
|
154
|
+
geometries (polygons, linestrings) to avoid OOM on WKB binary
|
|
155
|
+
columns. Adds S3A credentials for `s3a://` paths. User-supplied
|
|
156
|
+
`--conf` values override any defaults.
|
|
157
|
+
"""
|
|
158
|
+
config: dict[str, str] = {}
|
|
159
|
+
if _may_have_large_geometry(feature_key):
|
|
160
|
+
config["spark.sql.parquet.enableVectorizedReader"] = "false"
|
|
161
|
+
if path.startswith("s3a://"):
|
|
162
|
+
config.update(_S3A_DEFAULTS)
|
|
163
|
+
for pair in conf:
|
|
164
|
+
key, _, value = pair.partition("=")
|
|
165
|
+
config[key] = value
|
|
166
|
+
return config
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@click.command("overture-validate")
|
|
170
|
+
@click.argument("feature_type")
|
|
171
|
+
@click.argument("path")
|
|
172
|
+
@click.option("-o", "--output", default=None, help="Output path for validated Parquet.")
|
|
173
|
+
@click.option(
|
|
174
|
+
"--head",
|
|
175
|
+
"head_n",
|
|
176
|
+
default=20,
|
|
177
|
+
type=int,
|
|
178
|
+
show_default=True,
|
|
179
|
+
help="Error rows to display.",
|
|
180
|
+
)
|
|
181
|
+
@click.option("--conf", multiple=True, help="Spark config key=value pairs.")
|
|
182
|
+
@click.option(
|
|
183
|
+
"--count-only",
|
|
184
|
+
is_flag=True,
|
|
185
|
+
default=False,
|
|
186
|
+
help="Report error count only; skip explain/unpivot.",
|
|
187
|
+
)
|
|
188
|
+
@click.option(
|
|
189
|
+
"--skip-schema-check",
|
|
190
|
+
is_flag=True,
|
|
191
|
+
default=False,
|
|
192
|
+
help="Warn on schema mismatches instead of aborting.",
|
|
193
|
+
)
|
|
194
|
+
@click.option(
|
|
195
|
+
"--skip-columns",
|
|
196
|
+
multiple=True,
|
|
197
|
+
help="Columns declared absent from data; skips their checks.",
|
|
198
|
+
)
|
|
199
|
+
@click.option(
|
|
200
|
+
"--ignore-extra-columns",
|
|
201
|
+
multiple=True,
|
|
202
|
+
help="Extra data columns to ignore in schema comparison.",
|
|
203
|
+
)
|
|
204
|
+
@click.option(
|
|
205
|
+
"--suppress",
|
|
206
|
+
"suppress_specs",
|
|
207
|
+
multiple=True,
|
|
208
|
+
help="Suppress checks: FIELD (all checks) or FIELD:CHECK (specific).",
|
|
209
|
+
)
|
|
210
|
+
def validate_cli(
|
|
211
|
+
feature_type: str,
|
|
212
|
+
path: str,
|
|
213
|
+
output: str | None,
|
|
214
|
+
head_n: int,
|
|
215
|
+
conf: tuple[str, ...],
|
|
216
|
+
count_only: bool,
|
|
217
|
+
skip_schema_check: bool,
|
|
218
|
+
skip_columns: tuple[str, ...],
|
|
219
|
+
ignore_extra_columns: tuple[str, ...],
|
|
220
|
+
suppress_specs: tuple[str, ...],
|
|
221
|
+
) -> None:
|
|
222
|
+
"""Validate Overture data at PATH and write annotated Parquet."""
|
|
223
|
+
try:
|
|
224
|
+
resolved = resolve_entry_point_key(feature_type, REGISTRY)
|
|
225
|
+
except ValueError:
|
|
226
|
+
click.echo(
|
|
227
|
+
f"Unknown type '{feature_type}'. Known: {', '.join(model_names())}",
|
|
228
|
+
err=True,
|
|
229
|
+
)
|
|
230
|
+
sys.exit(1)
|
|
231
|
+
|
|
232
|
+
builder = SparkSession.builder
|
|
233
|
+
for key, value in _spark_config(path, conf, resolved).items():
|
|
234
|
+
builder = builder.config(key, value)
|
|
235
|
+
spark = builder.getOrCreate()
|
|
236
|
+
spark.sparkContext.setLogLevel("ERROR")
|
|
237
|
+
|
|
238
|
+
spec = resolve_read(path, PARTITION_MAP.get(resolved))
|
|
239
|
+
df = read_feature(spark, spec)
|
|
240
|
+
|
|
241
|
+
suppress: list[str | tuple[str, str]] = []
|
|
242
|
+
for s in suppress_specs:
|
|
243
|
+
if ":" in s:
|
|
244
|
+
field, name = s.split(":", 1)
|
|
245
|
+
suppress.append((field, name))
|
|
246
|
+
else:
|
|
247
|
+
suppress.append(s)
|
|
248
|
+
|
|
249
|
+
try:
|
|
250
|
+
result = validate_model(
|
|
251
|
+
df,
|
|
252
|
+
resolved,
|
|
253
|
+
skip_columns=skip_columns,
|
|
254
|
+
ignore_extra_columns=ignore_extra_columns,
|
|
255
|
+
suppress=suppress,
|
|
256
|
+
)
|
|
257
|
+
except ValueError as e:
|
|
258
|
+
click.echo(str(e), err=True)
|
|
259
|
+
sys.exit(1)
|
|
260
|
+
except AnalysisException as e:
|
|
261
|
+
# Backstop, narrowed to the one cause `--skip-columns` can address: a
|
|
262
|
+
# check that names a column missing from the data. validate_model
|
|
263
|
+
# already drops checks for skipped and schema-absent columns, so this
|
|
264
|
+
# fires only on a column outside the expected schema -- offer the
|
|
265
|
+
# operator the skip lever and name the column. Every other
|
|
266
|
+
# AnalysisException (a type mismatch, a struct field read off a scalar)
|
|
267
|
+
# is a generator bug `--skip-columns` cannot fix; let it propagate as a
|
|
268
|
+
# traceback rather than mask it behind the skip hint.
|
|
269
|
+
column = absent_column(e, df.columns)
|
|
270
|
+
if column is None:
|
|
271
|
+
raise
|
|
272
|
+
click.echo(
|
|
273
|
+
f"A check references column '{column}', absent from the data at {path}.",
|
|
274
|
+
err=True,
|
|
275
|
+
)
|
|
276
|
+
click.echo(
|
|
277
|
+
f"Re-run with `--skip-columns {column}` to skip its checks, "
|
|
278
|
+
"or `--skip-schema-check`.",
|
|
279
|
+
err=True,
|
|
280
|
+
)
|
|
281
|
+
sys.exit(1)
|
|
282
|
+
|
|
283
|
+
if result.schema_mismatches:
|
|
284
|
+
click.echo(f"Schema mismatches for {resolved}:", err=True)
|
|
285
|
+
for m in result.schema_mismatches:
|
|
286
|
+
click.echo(f" {m.path}: expected {m.expected}, got {m.actual}", err=True)
|
|
287
|
+
if result.absent_columns:
|
|
288
|
+
flags = " ".join(f"--skip-columns {c}" for c in result.absent_columns)
|
|
289
|
+
click.echo(
|
|
290
|
+
f" Re-run with `{flags}` to skip missing columns.",
|
|
291
|
+
err=True,
|
|
292
|
+
)
|
|
293
|
+
if not skip_schema_check:
|
|
294
|
+
sys.exit(1)
|
|
295
|
+
|
|
296
|
+
total_rows, error_count = result.row_counts()
|
|
297
|
+
click.echo(f"{error_count} / {total_rows} rows with errors", err=True)
|
|
298
|
+
|
|
299
|
+
if error_count > 0:
|
|
300
|
+
if not count_only:
|
|
301
|
+
explained = explain_errors(result.evaluated, result.checks).drop("geometry")
|
|
302
|
+
if output and head_n > 0:
|
|
303
|
+
explained = explained.cache()
|
|
304
|
+
if output:
|
|
305
|
+
explained.write.mode("overwrite").parquet(output)
|
|
306
|
+
click.echo(f"Written to {output}", err=True)
|
|
307
|
+
if head_n > 0:
|
|
308
|
+
explained.show(head_n, truncate=False)
|
|
309
|
+
sys.exit(1)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Expression builders and reusable PySpark column patterns."""
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Hand-written Spark StructType fragments for types the codegen can't generate.
|
|
2
|
+
|
|
3
|
+
The codegen builds feature schemas by walking Pydantic `BaseModel`
|
|
4
|
+
subclasses. `BBox` is a plain class, not a `BaseModel`, so extraction
|
|
5
|
+
can't reach it -- `BBOX_STRUCT` is hand-written here to fill the gap.
|
|
6
|
+
Every other nested type is a `BaseModel` and gets generated directly
|
|
7
|
+
into each feature module, which is why this file holds only the one
|
|
8
|
+
struct.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from pyspark.sql.types import DoubleType, StructField, StructType
|
|
14
|
+
|
|
15
|
+
BBOX_STRUCT = StructType(
|
|
16
|
+
[
|
|
17
|
+
StructField("xmin", DoubleType(), True),
|
|
18
|
+
StructField("xmax", DoubleType(), True),
|
|
19
|
+
StructField("ymin", DoubleType(), True),
|
|
20
|
+
StructField("ymax", DoubleType(), True),
|
|
21
|
+
]
|
|
22
|
+
)
|