polymo 0.7.0__tar.gz → 0.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- polymo-0.7.2/PKG-INFO +154 -0
- polymo-0.7.2/README.md +105 -0
- {polymo-0.7.0 → polymo-0.7.2}/pyproject.toml +10 -2
- polymo-0.7.2/src/polymo/builder/static/favicon.png +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/main.js +2 -2
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/rest_client.py +121 -19
- polymo-0.7.0/PKG-INFO +0 -170
- polymo-0.7.0/README.md +0 -127
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/__init__.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/__init__.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/app.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/countries.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/githubrepos.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/jsonplaceholder.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/jsonplaceholder_endpoints.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/multi_endpoints.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/examples/weather.yml +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/favicon.ico +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/index.html +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/logo.png +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/logo192.png +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/main.css +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/static/mockServiceWorker.js +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/builder/templates/index.html +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/cli.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/config.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/datasource.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/py.typed +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/scripts/__init__.py +0 -0
- {polymo-0.7.0 → polymo-0.7.2}/src/polymo/scripts/smoke.py +0 -0
polymo-0.7.2/PKG-INFO
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: polymo
|
|
3
|
+
Version: 0.7.2
|
|
4
|
+
Summary: Declarative REST API ingestion for PySpark
|
|
5
|
+
Keywords: spark,pyspark,rest,api,ingestion,data-engineering,etl,http
|
|
6
|
+
Author: Daniel Tom
|
|
7
|
+
Author-email: Daniel Tom <d.e.tom89@gmail.com>
|
|
8
|
+
License: BSD-3
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Intended Audience :: Information Technology
|
|
12
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
13
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
14
|
+
Classifier: Topic :: Database
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Operating System :: OS Independent
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Dist: fsspec>=2025.9.0
|
|
24
|
+
Requires-Dist: httpx>=0.26
|
|
25
|
+
Requires-Dist: jinja2>=3.1.6
|
|
26
|
+
Requires-Dist: pyyaml>=6.0.1
|
|
27
|
+
Requires-Dist: pyspark>=4 ; extra == 'benchmark'
|
|
28
|
+
Requires-Dist: pyarrow>=13 ; extra == 'benchmark'
|
|
29
|
+
Requires-Dist: httpx>=0.26 ; extra == 'benchmark'
|
|
30
|
+
Requires-Dist: notebook>=7.4.7 ; extra == 'benchmark'
|
|
31
|
+
Requires-Dist: fastapi>=0.110 ; extra == 'benchmark'
|
|
32
|
+
Requires-Dist: uvicorn>=0.24 ; extra == 'benchmark'
|
|
33
|
+
Requires-Dist: fastapi>=0.110 ; extra == 'builder'
|
|
34
|
+
Requires-Dist: uvicorn>=0.24 ; extra == 'builder'
|
|
35
|
+
Requires-Dist: jinja2>=3.1.6 ; extra == 'builder'
|
|
36
|
+
Requires-Dist: pyspark>=4 ; extra == 'builder'
|
|
37
|
+
Requires-Dist: pyarrow>=13 ; extra == 'builder'
|
|
38
|
+
Requires-Dist: pyspark>=4 ; extra == 'smoke'
|
|
39
|
+
Requires-Dist: pyarrow>=13 ; extra == 'smoke'
|
|
40
|
+
Requires-Python: >=3.10
|
|
41
|
+
Project-URL: Documentation, https://dan1elt0m.github.io/polymo/
|
|
42
|
+
Project-URL: Homepage, https://github.com/dan1elt0m/polymo
|
|
43
|
+
Project-URL: Issues, https://github.com/dan1elt0m/polymo/issues
|
|
44
|
+
Project-URL: Repository, https://github.com/dan1elt0m/polymo.git
|
|
45
|
+
Provides-Extra: benchmark
|
|
46
|
+
Provides-Extra: builder
|
|
47
|
+
Provides-Extra: smoke
|
|
48
|
+
Description-Content-Type: text/markdown
|
|
49
|
+
|
|
50
|
+
<p align="center">
|
|
51
|
+
<img src="builder-ui/public/logo.png" alt="Polymo" width="220">
|
|
52
|
+
</p>
|
|
53
|
+
|
|
54
|
+
# Welcome to Polymo
|
|
55
|
+
|
|
56
|
+
Polymo makes it super easy to ingest APIs with Pyspark. It's like slicing cake.
|
|
57
|
+
|
|
58
|
+
My vision is that API ingestion doesn't need heavy, third party tools or hard to maintain custom code.
|
|
59
|
+
The heck, you don't even need Pyspark skills.
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
<!-- Centered clickable screenshot -->
|
|
63
|
+
<p align="center">
|
|
64
|
+
<a href="docs/ui.png">
|
|
65
|
+
<img src="docs/ui.png" alt="Polymo Builder UI - connector preview screen" width="860">
|
|
66
|
+
</a>
|
|
67
|
+
</p>
|
|
68
|
+
|
|
69
|
+
## How does it work?
|
|
70
|
+
|
|
71
|
+
Define a config file manually or use the recommended, lightweight builder UI.
|
|
72
|
+
Once you are happy with your config, all you need to do is register the Polymo reader and tell Spark where to find the config:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from pyspark.sql import SparkSession
|
|
76
|
+
from polymo import ApiReader
|
|
77
|
+
|
|
78
|
+
spark = SparkSession.builder.getOrCreate()
|
|
79
|
+
spark.dataSource.register(ApiReader)
|
|
80
|
+
|
|
81
|
+
df = (
|
|
82
|
+
spark.read.format("polymo")
|
|
83
|
+
.option("config_path", "./config.yml") # YAML you saved from the Builder
|
|
84
|
+
.option("token", "YOUR_TOKEN") # Only if the API needs one
|
|
85
|
+
.load()
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
df.show()
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Structured Streaming works out of the box aswell:
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
stream_df = (
|
|
95
|
+
spark.readStream.format("polymo")
|
|
96
|
+
.option("config_path", "./config.yml")
|
|
97
|
+
.option("stream_batch_size", 100)
|
|
98
|
+
.option("stream_progress_path", "/tmp/polymo-progress.json")
|
|
99
|
+
.load()
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
query = stream_df.writeStream.format("memory").outputMode("append").queryName("polymo")
|
|
103
|
+
query.show()
|
|
104
|
+
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Does it perform? Polymo can read in batches (pages in parallel) and therefore is much faster than row based solutions like UDFs.
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
It's still early days, but Polymo already supports a lot of features!
|
|
111
|
+
|
|
112
|
+
- Various Authentication options
|
|
113
|
+
- Many Pagination patterns, plus automatic partition-aware reading when totals are exposed.
|
|
114
|
+
- Several partitioning stategies for parallel Spark reads.
|
|
115
|
+
- Incremental sync support with cursor parameters, JSON state files on local or remote storage, optional memory caching, and overrideable state keys.
|
|
116
|
+
- Schema controls that auto-infer types or accept Spark SQL schemas, along with record selectors, filtering expressions, and schema-based casting for nested responses.
|
|
117
|
+
- Structured Streaming compatibility with `spark.readStream`, tunable batch sizing, durable progress tracking, and a streaming smoke test mode.
|
|
118
|
+
- Error handling through configurable retry counts, status code lists, timeout handling, and exponential backoff settings.
|
|
119
|
+
- Jinja templating of query parameters gives you a ton of flexibility
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
## How to start?
|
|
123
|
+
Locally you probably want to install polymo with the UI:
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
pip install "polymo[builder]"
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
This comes with UI deps such as pyspark
|
|
130
|
+
|
|
131
|
+
Running Polymo on an existing cluster in for instance databricks doesnt require these deps.
|
|
132
|
+
In that case, just install the bare minimum depa with
|
|
133
|
+
```bash
|
|
134
|
+
pip install polymo
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Launch the builder UI
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
polymo builder
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
#### (Optional) Run the Builder in Docker
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
docker compose up --build builder
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
- The service listens on port `8000`; open <http://localhost:8000> once Uvicorn reports it is running.
|
|
150
|
+
|
|
151
|
+
## Where to Next
|
|
152
|
+
Read the docs [here](https://dan1elt0m.github.io/polymo/)
|
|
153
|
+
|
|
154
|
+
Contributions and early feedback welcome!
|
polymo-0.7.2/README.md
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="builder-ui/public/logo.png" alt="Polymo" width="220">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
# Welcome to Polymo
|
|
6
|
+
|
|
7
|
+
Polymo makes it super easy to ingest APIs with Pyspark. It's like slicing cake.
|
|
8
|
+
|
|
9
|
+
My vision is that API ingestion doesn't need heavy, third party tools or hard to maintain custom code.
|
|
10
|
+
The heck, you don't even need Pyspark skills.
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
<!-- Centered clickable screenshot -->
|
|
14
|
+
<p align="center">
|
|
15
|
+
<a href="docs/ui.png">
|
|
16
|
+
<img src="docs/ui.png" alt="Polymo Builder UI - connector preview screen" width="860">
|
|
17
|
+
</a>
|
|
18
|
+
</p>
|
|
19
|
+
|
|
20
|
+
## How does it work?
|
|
21
|
+
|
|
22
|
+
Define a config file manually or use the recommended, lightweight builder UI.
|
|
23
|
+
Once you are happy with your config, all you need to do is register the Polymo reader and tell Spark where to find the config:
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
from pyspark.sql import SparkSession
|
|
27
|
+
from polymo import ApiReader
|
|
28
|
+
|
|
29
|
+
spark = SparkSession.builder.getOrCreate()
|
|
30
|
+
spark.dataSource.register(ApiReader)
|
|
31
|
+
|
|
32
|
+
df = (
|
|
33
|
+
spark.read.format("polymo")
|
|
34
|
+
.option("config_path", "./config.yml") # YAML you saved from the Builder
|
|
35
|
+
.option("token", "YOUR_TOKEN") # Only if the API needs one
|
|
36
|
+
.load()
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
df.show()
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Structured Streaming works out of the box aswell:
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
stream_df = (
|
|
46
|
+
spark.readStream.format("polymo")
|
|
47
|
+
.option("config_path", "./config.yml")
|
|
48
|
+
.option("stream_batch_size", 100)
|
|
49
|
+
.option("stream_progress_path", "/tmp/polymo-progress.json")
|
|
50
|
+
.load()
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
query = stream_df.writeStream.format("memory").outputMode("append").queryName("polymo")
|
|
54
|
+
query.show()
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Does it perform? Polymo can read in batches (pages in parallel) and therefore is much faster than row based solutions like UDFs.
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
It's still early days, but Polymo already supports a lot of features!
|
|
62
|
+
|
|
63
|
+
- Various Authentication options
|
|
64
|
+
- Many Pagination patterns, plus automatic partition-aware reading when totals are exposed.
|
|
65
|
+
- Several partitioning stategies for parallel Spark reads.
|
|
66
|
+
- Incremental sync support with cursor parameters, JSON state files on local or remote storage, optional memory caching, and overrideable state keys.
|
|
67
|
+
- Schema controls that auto-infer types or accept Spark SQL schemas, along with record selectors, filtering expressions, and schema-based casting for nested responses.
|
|
68
|
+
- Structured Streaming compatibility with `spark.readStream`, tunable batch sizing, durable progress tracking, and a streaming smoke test mode.
|
|
69
|
+
- Error handling through configurable retry counts, status code lists, timeout handling, and exponential backoff settings.
|
|
70
|
+
- Jinja templating of query parameters gives you a ton of flexibility
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
## How to start?
|
|
74
|
+
Locally you probably want to install polymo with the UI:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
pip install "polymo[builder]"
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
This comes with UI deps such as pyspark
|
|
81
|
+
|
|
82
|
+
Running Polymo on an existing cluster in for instance databricks doesnt require these deps.
|
|
83
|
+
In that case, just install the bare minimum depa with
|
|
84
|
+
```bash
|
|
85
|
+
pip install polymo
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Launch the builder UI
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
polymo builder
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
#### (Optional) Run the Builder in Docker
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
docker compose up --build builder
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
- The service listens on port `8000`; open <http://localhost:8000> once Uvicorn reports it is running.
|
|
101
|
+
|
|
102
|
+
## Where to Next
|
|
103
|
+
Read the docs [here](https://dan1elt0m.github.io/polymo/)
|
|
104
|
+
|
|
105
|
+
Contributions and early feedback welcome!
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "polymo"
|
|
3
|
-
version = "0.7.
|
|
3
|
+
version = "0.7.2"
|
|
4
4
|
description = "Declarative REST API ingestion for PySpark"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
authors = [
|
|
@@ -29,7 +29,6 @@ dependencies = [
|
|
|
29
29
|
"fsspec>=2025.9.0",
|
|
30
30
|
"httpx>=0.26",
|
|
31
31
|
"jinja2>=3.1.6",
|
|
32
|
-
"notebook>=7.4.7",
|
|
33
32
|
"pyyaml>=6.0.1",
|
|
34
33
|
]
|
|
35
34
|
|
|
@@ -41,6 +40,15 @@ builder = [
|
|
|
41
40
|
"pyspark>=4",
|
|
42
41
|
"pyarrow>=13",
|
|
43
42
|
]
|
|
43
|
+
benchmark = [
|
|
44
|
+
"pyspark>=4",
|
|
45
|
+
"pyarrow>=13",
|
|
46
|
+
"httpx>=0.26",
|
|
47
|
+
"notebook>=7.4.7",
|
|
48
|
+
"fastapi>=0.110",
|
|
49
|
+
"uvicorn>=0.24"
|
|
50
|
+
]
|
|
51
|
+
|
|
44
52
|
smoke = [
|
|
45
53
|
"pyspark>=4",
|
|
46
54
|
"pyarrow>=13",
|
|
Binary file
|