overture-spark 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- overture_spark-0.1.0/.python-version +1 -0
- overture_spark-0.1.0/LICENSE +21 -0
- overture_spark-0.1.0/PKG-INFO +74 -0
- overture_spark-0.1.0/README.md +47 -0
- overture_spark-0.1.0/overture_spark/__init__.py +300 -0
- overture_spark-0.1.0/overture_spark/job.py +277 -0
- overture_spark-0.1.0/overture_spark/test_area.py +161 -0
- overture_spark-0.1.0/overture_spark.egg-info/PKG-INFO +74 -0
- overture_spark-0.1.0/overture_spark.egg-info/SOURCES.txt +19 -0
- overture_spark-0.1.0/overture_spark.egg-info/dependency_links.txt +1 -0
- overture_spark-0.1.0/overture_spark.egg-info/requires.txt +9 -0
- overture_spark-0.1.0/overture_spark.egg-info/scm_file_list.json +16 -0
- overture_spark-0.1.0/overture_spark.egg-info/scm_version.json +8 -0
- overture_spark-0.1.0/overture_spark.egg-info/top_level.txt +1 -0
- overture_spark-0.1.0/pyproject.toml +94 -0
- overture_spark-0.1.0/setup.cfg +4 -0
- overture_spark-0.1.0/tests/__init__.py +0 -0
- overture_spark-0.1.0/tests/test_job.py +289 -0
- overture_spark-0.1.0/tests/test_module.py +452 -0
- overture_spark-0.1.0/tests/test_test_area.py +245 -0
- overture_spark-0.1.0/uv.lock +2097 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.11
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Overture Maps
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: overture-spark
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Portable, framework-agnostic Spark job runtime helpers.
|
|
5
|
+
Author: Overture Maps Foundation
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/OvertureMaps/overture-core
|
|
8
|
+
Project-URL: Source, https://github.com/OvertureMaps/overture-core/tree/main/packages/overture_spark
|
|
9
|
+
Project-URL: Issues, https://github.com/OvertureMaps/overture-core/issues
|
|
10
|
+
Keywords: overture,overturemaps,spark
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Provides-Extra: sql-spark
|
|
20
|
+
Requires-Dist: pyspark==3.5.0; extra == "sql-spark"
|
|
21
|
+
Requires-Dist: apache-sedona==1.7.0; extra == "sql-spark"
|
|
22
|
+
Requires-Dist: moto[server]>=5.0.25; extra == "sql-spark"
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: pytest>=9.1.1; extra == "dev"
|
|
25
|
+
Requires-Dist: pytest-cov>=7.0.0; extra == "dev"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# overture-spark
|
|
29
|
+
|
|
30
|
+
[](https://pypi.org/project/overture-spark/)
|
|
31
|
+
[](https://pypi.org/project/overture-spark/)
|
|
32
|
+
|
|
33
|
+
Portable, framework-agnostic runtime helpers for Overture's Spark jobs. Moved from `tf-data-platform`'s `overture_spark` module (see [ops-team#535](https://github.com/OvertureMaps/ops-team/issues/535), [ops-team#536](https://github.com/OvertureMaps/ops-team/issues/536)); other `overture_spark` modules follow in later PRs (see [ops-team#532](https://github.com/OvertureMaps/ops-team/issues/532)).
|
|
34
|
+
|
|
35
|
+
It contains:
|
|
36
|
+
- `SparkSedonaJob` (in `job.py`) — cluster-side base class for jobs: platform detection, parameter parsing, logging, and the Sedona `SparkSession` lifecycle. Subclass it and implement `execute_job()`.
|
|
37
|
+
- `test_area` — helpers for a job's optional `test_area` param (a bbox or WKT polygon that scopes a dev/test run to a small region instead of the full planet).
|
|
38
|
+
|
|
39
|
+
## Writing a job
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from overture_spark.job import SparkSedonaJob
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class CollectionJob(SparkSedonaJob):
|
|
46
|
+
def execute_job(self) -> None:
|
|
47
|
+
path = self.get_param("input_path")
|
|
48
|
+
self.log(f"Processing {path}")
|
|
49
|
+
df = self.apply_test_area_filter(self.spark.read.parquet(path))
|
|
50
|
+
# ... your logic
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
`getSparkSedonaSession` (in `overture_spark/__init__.py`) is the one call in this package that actually needs a real Spark/Sedona session — it, and everything downstream of it, imports `pyspark`/`apache-sedona` lazily and only there. Platform detection, version/JAR helpers, and `SparkSedonaJob`'s parameter/logging logic are plain Python and importable without either installed.
|
|
54
|
+
|
|
55
|
+
## Testing locally
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
cd packages/overture_spark
|
|
59
|
+
uv sync --extra dev
|
|
60
|
+
uv run pytest -v -m "not spark"
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Tests marked `@pytest.mark.spark` need a real Spark/Sedona session (a JVM, plus Sedona's native JAR resolution), so they're excluded from the routine `pytest` run above and from `overture-spark[dev]`. Install the `sql-spark` extra to run them:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
uv sync --extra dev --extra sql-spark
|
|
67
|
+
uv run pytest -v --cov --cov-fail-under=95
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
CI mirrors this split: the routine `Test Python` workflow runs `pytest -m "not spark"` for this package (no coverage gate, since the Spark-only code paths aren't exercised), and the `SQL Engines` workflow's `spark (overture_spark)` job installs the `sql-spark` extra and Java, then runs the full suite with the 95% coverage floor.
|
|
71
|
+
|
|
72
|
+
## Publishing
|
|
73
|
+
|
|
74
|
+
See [`PACKAGE_VERSIONING.md`](../PACKAGE_VERSIONING.md) for how a version bump here turns into a PyPI release.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# overture-spark
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/overture-spark/)
|
|
4
|
+
[](https://pypi.org/project/overture-spark/)
|
|
5
|
+
|
|
6
|
+
Portable, framework-agnostic runtime helpers for Overture's Spark jobs. Moved from `tf-data-platform`'s `overture_spark` module (see [ops-team#535](https://github.com/OvertureMaps/ops-team/issues/535), [ops-team#536](https://github.com/OvertureMaps/ops-team/issues/536)); other `overture_spark` modules follow in later PRs (see [ops-team#532](https://github.com/OvertureMaps/ops-team/issues/532)).
|
|
7
|
+
|
|
8
|
+
It contains:
|
|
9
|
+
- `SparkSedonaJob` (in `job.py`) — cluster-side base class for jobs: platform detection, parameter parsing, logging, and the Sedona `SparkSession` lifecycle. Subclass it and implement `execute_job()`.
|
|
10
|
+
- `test_area` — helpers for a job's optional `test_area` param (a bbox or WKT polygon that scopes a dev/test run to a small region instead of the full planet).
|
|
11
|
+
|
|
12
|
+
## Writing a job
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
from overture_spark.job import SparkSedonaJob
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class CollectionJob(SparkSedonaJob):
|
|
19
|
+
def execute_job(self) -> None:
|
|
20
|
+
path = self.get_param("input_path")
|
|
21
|
+
self.log(f"Processing {path}")
|
|
22
|
+
df = self.apply_test_area_filter(self.spark.read.parquet(path))
|
|
23
|
+
# ... your logic
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
`getSparkSedonaSession` (in `overture_spark/__init__.py`) is the one call in this package that actually needs a real Spark/Sedona session — it, and everything downstream of it, imports `pyspark`/`apache-sedona` lazily and only there. Platform detection, version/JAR helpers, and `SparkSedonaJob`'s parameter/logging logic are plain Python and importable without either installed.
|
|
27
|
+
|
|
28
|
+
## Testing locally
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
cd packages/overture_spark
|
|
32
|
+
uv sync --extra dev
|
|
33
|
+
uv run pytest -v -m "not spark"
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Tests marked `@pytest.mark.spark` need a real Spark/Sedona session (a JVM, plus Sedona's native JAR resolution), so they're excluded from the routine `pytest` run above and from `overture-spark[dev]`. Install the `sql-spark` extra to run them:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
uv sync --extra dev --extra sql-spark
|
|
40
|
+
uv run pytest -v --cov --cov-fail-under=95
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
CI mirrors this split: the routine `Test Python` workflow runs `pytest -m "not spark"` for this package (no coverage gate, since the Spark-only code paths aren't exercised), and the `SQL Engines` workflow's `spark (overture_spark)` job installs the `sql-spark` extra and Java, then runs the full suite with the 95% coverage floor.
|
|
44
|
+
|
|
45
|
+
## Publishing
|
|
46
|
+
|
|
47
|
+
See [`PACKAGE_VERSIONING.md`](../PACKAGE_VERSIONING.md) for how a version bump here turns into a PyPI release.
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
"""Cluster-side Spark/Sedona runtime for Overture jobs.
|
|
2
|
+
|
|
3
|
+
Installed and imported **on** the Spark cluster (Glue / Databricks / Wherobots /
|
|
4
|
+
local) and runs **inside** the executor. This is deliberately separate from the
|
|
5
|
+
Airflow orchestration layer:
|
|
6
|
+
|
|
7
|
+
- Orchestration (launch, cluster sizing, asset staging) lives in the external,
|
|
8
|
+
Overture-agnostic ``overture-airflow-provider``.
|
|
9
|
+
- This package is the cluster-execution-time side: it builds the configured
|
|
10
|
+
Sedona ``SparkSession`` (``getSparkSedonaSession``), resolves Sedona/GeoTools
|
|
11
|
+
JAR coordinates (``SparkSedona``), and detects the platform
|
|
12
|
+
(``SparkPlatform.autodetect``). ``SparkSedonaJob`` (in ``job.py``) is the base
|
|
13
|
+
class user jobs subclass; the provider's runners import those job classes and
|
|
14
|
+
call ``run()``.
|
|
15
|
+
|
|
16
|
+
Rule of thumb: launches a job via Airflow/AWS APIs -> provider/factory; runs inside
|
|
17
|
+
the Spark job -> here.
|
|
18
|
+
|
|
19
|
+
``pyspark``/``apache-sedona`` are only imported lazily, inside
|
|
20
|
+
``getSparkSedonaSession()``, the one function that actually builds a Spark
|
|
21
|
+
session. Everything else in this module (platform detection, version/JAR
|
|
22
|
+
helpers) is plain Python, so it — and the tests that only exercise it — never
|
|
23
|
+
pays the JVM cost of installing/starting Spark. Callers that do need a real
|
|
24
|
+
session install the ``sql-spark`` extra.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import os
|
|
28
|
+
import sys
|
|
29
|
+
from enum import IntEnum, auto
|
|
30
|
+
from importlib.util import find_spec
|
|
31
|
+
from typing import TYPE_CHECKING, Dict, List
|
|
32
|
+
|
|
33
|
+
if TYPE_CHECKING:
|
|
34
|
+
from pyspark.sql import SparkSession
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class PackageNotFoundError(Exception):
|
|
38
|
+
def __init__(self, package_name):
|
|
39
|
+
super().__init__(
|
|
40
|
+
f"Package '{package_name}' is not installed or could not be found."
|
|
41
|
+
)
|
|
42
|
+
self.package_name = package_name
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def get_package_version(package_name):
|
|
46
|
+
from importlib.metadata import PackageNotFoundError as _DistNotFound
|
|
47
|
+
from importlib.metadata import version
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
return version(package_name)
|
|
51
|
+
except _DistNotFound:
|
|
52
|
+
raise PackageNotFoundError(package_name)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class SparkPlatform(IntEnum):
|
|
56
|
+
GLUE = auto()
|
|
57
|
+
DATABRICKS = auto()
|
|
58
|
+
WHEROBOTS = auto()
|
|
59
|
+
LOCAL = auto()
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def autodetect(cls):
|
|
63
|
+
if cls.isRunningInGlue():
|
|
64
|
+
return SparkPlatform.GLUE
|
|
65
|
+
elif cls.isRunningInDatabricks():
|
|
66
|
+
return SparkPlatform.DATABRICKS
|
|
67
|
+
elif cls.isRunningInWherobots():
|
|
68
|
+
return SparkPlatform.WHEROBOTS
|
|
69
|
+
else:
|
|
70
|
+
return SparkPlatform.LOCAL
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def from_str(cls, name):
|
|
74
|
+
# Normalize the input name to uppercase
|
|
75
|
+
normalized_name = name.upper()
|
|
76
|
+
if normalized_name in cls.__members__:
|
|
77
|
+
return cls[normalized_name]
|
|
78
|
+
raise ValueError(f"{name} is not a valid platform")
|
|
79
|
+
|
|
80
|
+
@classmethod
|
|
81
|
+
def isRunningInDatabricks(cls):
|
|
82
|
+
if "DATABRICKS_RUNTIME_VERSION" not in os.environ:
|
|
83
|
+
return False
|
|
84
|
+
return find_spec("pyspark.dbutils") is not None
|
|
85
|
+
|
|
86
|
+
@classmethod
|
|
87
|
+
def isRunningInDatabricksNotebook(cls):
|
|
88
|
+
# dbutils is injected into the *notebook's* execution namespace
|
|
89
|
+
# (__main__ when run through Databricks' REPL), not into whichever
|
|
90
|
+
# module happens to define this classmethod. A bare `dbutils` name
|
|
91
|
+
# lookup here only ever resolves against overture_spark's own
|
|
92
|
+
# globals, so it always raises NameError and always returns False,
|
|
93
|
+
# even inside a real Databricks notebook.
|
|
94
|
+
main = sys.modules.get("__main__")
|
|
95
|
+
return main is not None and hasattr(main, "dbutils")
|
|
96
|
+
|
|
97
|
+
@classmethod
|
|
98
|
+
def isRunningInGlue(cls):
|
|
99
|
+
try:
|
|
100
|
+
return find_spec("awsglue.utils") is not None
|
|
101
|
+
except ModuleNotFoundError:
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
@classmethod
|
|
105
|
+
def isRunningInWherobots(cls):
|
|
106
|
+
marker_env_vars = (
|
|
107
|
+
"WHEROBOTS_RUNTIME_VERSION",
|
|
108
|
+
"WHEROBOTS_RUNTIME",
|
|
109
|
+
"WHEROBOTS_ENV",
|
|
110
|
+
"WHEROBOTS_CLUSTER_ID",
|
|
111
|
+
"WHEROBOTS_JOB_ID",
|
|
112
|
+
)
|
|
113
|
+
if any(os.getenv(var) for var in marker_env_vars):
|
|
114
|
+
return True
|
|
115
|
+
|
|
116
|
+
marker_paths = (
|
|
117
|
+
"/home/wherobots/run-scripts/run_submit.py",
|
|
118
|
+
"/opt/conda/envs/wherobots",
|
|
119
|
+
)
|
|
120
|
+
if any(os.path.exists(path) for path in marker_paths):
|
|
121
|
+
return True
|
|
122
|
+
|
|
123
|
+
if "wherobots" in (sys.prefix or "").lower():
|
|
124
|
+
return True
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
return find_spec("wherobots.db") is not None
|
|
128
|
+
except ModuleNotFoundError:
|
|
129
|
+
return False
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class SparkSedona:
|
|
133
|
+
@classmethod
|
|
134
|
+
def getPySparkVersion(cls) -> str:
|
|
135
|
+
return get_package_version("pyspark")
|
|
136
|
+
|
|
137
|
+
@classmethod
|
|
138
|
+
def getSedonaVersion(cls) -> str:
|
|
139
|
+
return get_package_version("apache-sedona")
|
|
140
|
+
|
|
141
|
+
@classmethod
|
|
142
|
+
def getSparkVersionForSedona(
|
|
143
|
+
cls, py_spark_version: str, sedona_version: str
|
|
144
|
+
) -> str:
|
|
145
|
+
# https://sedona.apache.org/latest-snapshot/setup/install-python/#prepare-sedona-spark-jar
|
|
146
|
+
spark_v = py_spark_version or cls.getPySparkVersion()
|
|
147
|
+
sparkMajorVersion, sparkMinorVersion = spark_v.split(".")[:2]
|
|
148
|
+
sedona_v = sedona_version or cls.getSedonaVersion()
|
|
149
|
+
sedonaMajorVersion, sedonaMinorVersion = sedona_v.split(".")[:2]
|
|
150
|
+
if sparkMajorVersion != "3":
|
|
151
|
+
raise RuntimeError("I'm only supporting spark 3")
|
|
152
|
+
if (
|
|
153
|
+
int(sparkMinorVersion) <= 3
|
|
154
|
+
and int(sedonaMajorVersion) == 1
|
|
155
|
+
and int(sedonaMinorVersion) <= 6
|
|
156
|
+
):
|
|
157
|
+
sparkMajorMinorVersion = f"{sparkMajorVersion}.0"
|
|
158
|
+
else:
|
|
159
|
+
sparkMajorMinorVersion = f"{sparkMajorVersion}.{sparkMinorVersion}"
|
|
160
|
+
return sparkMajorMinorVersion
|
|
161
|
+
|
|
162
|
+
@classmethod
|
|
163
|
+
def getGeotoolsWrapperVersion(cls, sedona_version: str) -> str:
|
|
164
|
+
# see https://repo1.maven.org/maven2/org/datasyslab/geotools-wrapper/
|
|
165
|
+
geotoolsVersionMap = {
|
|
166
|
+
"1.5.3": "28.2",
|
|
167
|
+
"1.6.1": "28.2",
|
|
168
|
+
"1.7.0": "28.5",
|
|
169
|
+
"1.7.1": "28.5",
|
|
170
|
+
"1.7.2": "28.5",
|
|
171
|
+
"1.8.0": "33.1",
|
|
172
|
+
"1.8.1": "33.1",
|
|
173
|
+
}
|
|
174
|
+
return geotoolsVersionMap[sedona_version]
|
|
175
|
+
|
|
176
|
+
@classmethod
|
|
177
|
+
def getSedonaJarPackages(
|
|
178
|
+
cls,
|
|
179
|
+
spark_platform: SparkPlatform,
|
|
180
|
+
scala_version: str,
|
|
181
|
+
py_spark_version: str,
|
|
182
|
+
sedona_version: str,
|
|
183
|
+
) -> List[str]:
|
|
184
|
+
if spark_platform == SparkPlatform.DATABRICKS:
|
|
185
|
+
# on databricks, sedona packages need to be configured in cluster setup
|
|
186
|
+
return []
|
|
187
|
+
|
|
188
|
+
suffix = "-shaded" if spark_platform != SparkPlatform.LOCAL else ""
|
|
189
|
+
packages = [
|
|
190
|
+
f"org.apache.sedona:sedona-spark{suffix}-{cls.getSparkVersionForSedona(py_spark_version=py_spark_version, sedona_version=sedona_version)}_{scala_version}:{sedona_version}",
|
|
191
|
+
f"org.datasyslab:geotools-wrapper:{sedona_version}-{cls.getGeotoolsWrapperVersion(sedona_version=sedona_version)}",
|
|
192
|
+
]
|
|
193
|
+
return packages
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def getSparkSedonaSession(
|
|
197
|
+
spark_platform: SparkPlatform = None,
|
|
198
|
+
app_name: str = "SparkSedonaApp",
|
|
199
|
+
extra_spark_conf: Dict[str, str] = None,
|
|
200
|
+
extra_packages: List[str] = None,
|
|
201
|
+
) -> "SparkSession":
|
|
202
|
+
# Lazy: this is the one call site that actually needs pyspark/sedona
|
|
203
|
+
# installed and starts a JVM, see the module docstring.
|
|
204
|
+
from pyspark.sql import SparkSession
|
|
205
|
+
from sedona.spark import KryoSerializer, SedonaContext, SedonaKryoRegistrator
|
|
206
|
+
|
|
207
|
+
spark_platform = spark_platform or SparkPlatform.autodetect()
|
|
208
|
+
extra_spark_conf = extra_spark_conf or {}
|
|
209
|
+
extra_packages = extra_packages or []
|
|
210
|
+
if (
|
|
211
|
+
spark_platform == SparkPlatform.DATABRICKS
|
|
212
|
+
and SparkPlatform.isRunningInDatabricksNotebook()
|
|
213
|
+
):
|
|
214
|
+
global spark
|
|
215
|
+
if len(extra_spark_conf) > 0:
|
|
216
|
+
raise RuntimeError(
|
|
217
|
+
"You need to set extra spark configuration when configuring the databricks cluster"
|
|
218
|
+
)
|
|
219
|
+
else:
|
|
220
|
+
# init a new spark session
|
|
221
|
+
builder = SparkSession.builder.appName(app_name)
|
|
222
|
+
if spark_platform == SparkPlatform.LOCAL:
|
|
223
|
+
hadoopAwsHelper = HadoopAwsInstaller()
|
|
224
|
+
jars = (
|
|
225
|
+
SparkSedona.getSedonaJarPackages(
|
|
226
|
+
spark_platform=spark_platform,
|
|
227
|
+
py_spark_version=SparkSedona.getPySparkVersion(),
|
|
228
|
+
sedona_version=SparkSedona.getSedonaVersion(),
|
|
229
|
+
scala_version="2.12",
|
|
230
|
+
)
|
|
231
|
+
+ hadoopAwsHelper.install()
|
|
232
|
+
+ extra_packages
|
|
233
|
+
)
|
|
234
|
+
(
|
|
235
|
+
builder.master("local[*]") # * = use all available cores
|
|
236
|
+
.config("spark.serializer", KryoSerializer.getName)
|
|
237
|
+
.config("spark.kryo.registrator", SedonaKryoRegistrator.getName)
|
|
238
|
+
.config(
|
|
239
|
+
"spark.jars.repositories",
|
|
240
|
+
"https://artifacts.unidata.ucar.edu/repository/unidata-all",
|
|
241
|
+
)
|
|
242
|
+
.config("spark.jars.packages", ",".join(jars))
|
|
243
|
+
.config(
|
|
244
|
+
"spark.hadoop.fs.s3.impl", "org.apache.hadoop.fs.s3a.S3AFileSystem"
|
|
245
|
+
)
|
|
246
|
+
)
|
|
247
|
+
else: # pragma: no cover - non-local platforms need a real cluster to exercise
|
|
248
|
+
# sedona jars for other (non-local) spark platforms are configured in SparkSedonaOperator*
|
|
249
|
+
pass
|
|
250
|
+
builder.config("spark.sql.parquet.outputTimestampType", "TIMESTAMP_MICROS")
|
|
251
|
+
for k, v in extra_spark_conf.items():
|
|
252
|
+
builder.config(k, v)
|
|
253
|
+
spark = builder.getOrCreate()
|
|
254
|
+
|
|
255
|
+
spark = SedonaContext.create(spark)
|
|
256
|
+
is_glue = spark_platform == SparkPlatform.GLUE
|
|
257
|
+
if is_glue: # pragma: no cover - needs a real Glue job runtime
|
|
258
|
+
from awsglue.context import GlueContext
|
|
259
|
+
|
|
260
|
+
spark = GlueContext(spark).spark_session
|
|
261
|
+
return spark
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def isLocalSpark(spark):
|
|
265
|
+
return spark.conf.get("spark.master").startswith("local")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
class HadoopAwsInstaller:
|
|
269
|
+
def __init__(self):
|
|
270
|
+
# Resolve pyspark's own install location via its module spec instead
|
|
271
|
+
# of site.getsitepackages(), which only looks at the interpreter's
|
|
272
|
+
# site-packages directories and returns nothing outside a venv (e.g.
|
|
273
|
+
# a system-Python install, or pyspark installed via --user). That
|
|
274
|
+
# left _getInstalledVersion() unable to find the bundled jars, and
|
|
275
|
+
# install() silently built an invalid "hadoop-aws:None" coordinate.
|
|
276
|
+
pyspark_spec = find_spec("pyspark")
|
|
277
|
+
if pyspark_spec and pyspark_spec.submodule_search_locations:
|
|
278
|
+
self.jarPath = os.path.join(
|
|
279
|
+
pyspark_spec.submodule_search_locations[0], "jars"
|
|
280
|
+
)
|
|
281
|
+
else:
|
|
282
|
+
self.jarPath = None
|
|
283
|
+
|
|
284
|
+
def install(self):
|
|
285
|
+
hadoopVersion = self._getInstalledVersion()
|
|
286
|
+
if not hadoopVersion:
|
|
287
|
+
raise RuntimeError(
|
|
288
|
+
"Could not determine the installed Hadoop version from "
|
|
289
|
+
"pyspark's bundled jars; is pyspark installed?"
|
|
290
|
+
)
|
|
291
|
+
return [f"org.apache.hadoop:hadoop-aws:{hadoopVersion}"]
|
|
292
|
+
|
|
293
|
+
def _getInstalledVersion(self):
|
|
294
|
+
if not self.jarPath:
|
|
295
|
+
return None
|
|
296
|
+
for root, dirs, files in os.walk(self.jarPath):
|
|
297
|
+
for file in files:
|
|
298
|
+
if file.startswith("hadoop-client-api-"):
|
|
299
|
+
return os.path.splitext(file)[0].replace("hadoop-client-api-", "")
|
|
300
|
+
return None
|