geodeploy 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geodeploy/__init__.py +52 -0
- geodeploy/__main__.py +13 -0
- geodeploy/admin.py +191 -0
- geodeploy/catalog.py +96 -0
- geodeploy/cli/__init__.py +7 -0
- geodeploy/cli/commands/__init__.py +2 -0
- geodeploy/cli/commands/_common.py +233 -0
- geodeploy/cli/commands/admin.py +317 -0
- geodeploy/cli/commands/auth.py +280 -0
- geodeploy/cli/commands/browse.py +224 -0
- geodeploy/cli/commands/catalog.py +107 -0
- geodeploy/cli/commands/imports.py +125 -0
- geodeploy/cli/commands/jobs.py +38 -0
- geodeploy/cli/commands/layers.py +536 -0
- geodeploy/cli/commands/portals.py +515 -0
- geodeploy/cli/commands/sources.py +97 -0
- geodeploy/cli/commands/upload.py +154 -0
- geodeploy/cli/main.py +263 -0
- geodeploy/cli/output.py +320 -0
- geodeploy/client.py +327 -0
- geodeploy/config.py +438 -0
- geodeploy/errors.py +93 -0
- geodeploy/imports.py +65 -0
- geodeploy/jobs.py +73 -0
- geodeploy/layers.py +433 -0
- geodeploy/portals.py +280 -0
- geodeploy/py.typed +0 -0
- geodeploy/sources.py +70 -0
- geodeploy/styles.py +467 -0
- geodeploy/transport.py +355 -0
- geodeploy/uploads.py +474 -0
- geodeploy-1.3.0.dist-info/METADATA +102 -0
- geodeploy-1.3.0.dist-info/RECORD +38 -0
- geodeploy-1.3.0.dist-info/WHEEL +5 -0
- geodeploy-1.3.0.dist-info/entry_points.txt +2 -0
- geodeploy-1.3.0.dist-info/licenses/LICENSE +202 -0
- geodeploy-1.3.0.dist-info/licenses/NOTICE +11 -0
- geodeploy-1.3.0.dist-info/top_level.txt +1 -0
geodeploy/uploads.py
ADDED
|
@@ -0,0 +1,474 @@
|
|
|
1
|
+
"""Uploads — one entry point, five routes, chosen for you.
|
|
2
|
+
|
|
3
|
+
GeoDeploy has five ways in, and picking the wrong one fails in ways that are hard to read:
|
|
4
|
+
|
|
5
|
+
=========================== ==========================================================
|
|
6
|
+
route when
|
|
7
|
+
=========================== ==========================================================
|
|
8
|
+
``vector-api`` .zip/.geojson/.json/.gpkg under the direct-upload threshold
|
|
9
|
+
``csv-api`` a small .csv, with X/Y or WKT columns → PostGIS
|
|
10
|
+
``large-vector`` any of the above at or over the threshold → GeoParquet
|
|
11
|
+
``geoparquet`` .parquet/.geoparquet, at any size
|
|
12
|
+
``raster-api`` / ``raster-large`` .tif/.tiff, small / large
|
|
13
|
+
=========================== ==========================================================
|
|
14
|
+
|
|
15
|
+
**The threshold is 48 MB, not the API's 2 GB.** A file POSTed through the API is buffered by
|
|
16
|
+
whatever proxy sits in front of the instance, and Cloudflare's free tier cuts a request body at
|
|
17
|
+
100 MB — which surfaced as a bare "Network error" at 1 % with *nothing in the API log*, because the
|
|
18
|
+
request never arrived. Above 48 MB everything goes direct-to-storage in 48 MB presigned parts, so
|
|
19
|
+
no single request approaches an edge limit whatever the total size. `ui/src/composables/useUpload.js`
|
|
20
|
+
makes the same decision with the same numbers; the two must stay in step.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import csv as _csv
|
|
25
|
+
import os
|
|
26
|
+
import threading
|
|
27
|
+
from typing import Any, Callable, Dict, List, Optional, Sequence
|
|
28
|
+
|
|
29
|
+
from .errors import ValidationError
|
|
30
|
+
from .transport import MultipartBody, ProgressReader, Request
|
|
31
|
+
|
|
32
|
+
#: At or above this, bypass the API and upload direct to object storage. See the module docstring.
|
|
33
|
+
LARGE_UPLOAD_THRESHOLD = 48 * 1024 * 1024
|
|
34
|
+
|
|
35
|
+
#: Above this a single presigned PUT is replaced by chunked parts. Must stay <= the server's
|
|
36
|
+
#: PART_SIZE (48 MB), which is itself sized to clear a CDN's ~100 MB request-body cap.
|
|
37
|
+
CHUNK_THRESHOLD = 48 * 1024 * 1024
|
|
38
|
+
|
|
39
|
+
#: Parts uploaded at once. Four saturates a normal uplink without making progress unreadable.
|
|
40
|
+
PART_CONCURRENCY = 4
|
|
41
|
+
|
|
42
|
+
#: Retries for ONE part. Parts are the only safely retryable piece of a big upload: each is an
|
|
43
|
+
#: idempotent PUT to its own presigned URL, so a reset at 90 % costs one part, not the whole file.
|
|
44
|
+
PART_RETRIES = 3
|
|
45
|
+
|
|
46
|
+
VECTOR_API_EXTENSIONS = frozenset((".zip", ".geojson", ".json", ".gpkg"))
|
|
47
|
+
GEOPARQUET_EXTENSIONS = frozenset((".parquet", ".geoparquet"))
|
|
48
|
+
LARGE_VECTOR_EXTENSIONS = frozenset(VECTOR_API_EXTENSIONS | {".csv"})
|
|
49
|
+
RASTER_EXTENSIONS = frozenset((".tif", ".tiff"))
|
|
50
|
+
|
|
51
|
+
#: Column names that are almost always coordinates. Used ONLY to offer a default for a CSV that
|
|
52
|
+
#: was given no geometry options — the choice is always reported, never silent.
|
|
53
|
+
X_NAMES = ("longitude", "lon", "lng", "long", "x", "easting", "east", "xcoord", "x_coord")
|
|
54
|
+
Y_NAMES = ("latitude", "lat", "y", "northing", "north", "ycoord", "y_coord")
|
|
55
|
+
WKT_NAMES = ("wkt", "geom", "geometry", "the_geom", "wkt_geom", "geometry_wkt")
|
|
56
|
+
|
|
57
|
+
DELIMITERS = {",": "comma", ";": "semicolon", "\t": "tab", "|": "pipe", " ": "space"}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class UploadPlan(object):
|
|
61
|
+
"""What will happen to one file, decided before a byte moves.
|
|
62
|
+
|
|
63
|
+
Separated from the doing so the CLI can show `--dry-run` and a plugin can explain the route to
|
|
64
|
+
a user before committing to a multi-gigabyte transfer.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
__slots__ = ("path", "route", "layer_type", "name", "size", "csv_opts", "chunked", "reason")
|
|
68
|
+
|
|
69
|
+
def __init__(self, path: str, route: str, layer_type: str, name: str, size: int,
|
|
70
|
+
csv_opts: Optional[Dict[str, Any]] = None, chunked: bool = False,
|
|
71
|
+
reason: str = ""):
|
|
72
|
+
self.path = path
|
|
73
|
+
self.route = route
|
|
74
|
+
self.layer_type = layer_type
|
|
75
|
+
self.name = name
|
|
76
|
+
self.size = size
|
|
77
|
+
self.csv_opts = csv_opts
|
|
78
|
+
self.chunked = chunked
|
|
79
|
+
self.reason = reason
|
|
80
|
+
|
|
81
|
+
def as_dict(self) -> Dict[str, Any]:
|
|
82
|
+
return {"path": self.path, "route": self.route, "layer_type": self.layer_type,
|
|
83
|
+
"name": self.name, "size": self.size, "chunked": self.chunked,
|
|
84
|
+
"csv_opts": self.csv_opts, "reason": self.reason}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class UploadResult(object):
|
|
88
|
+
"""The outcome for one file: which layer it became, and the job that is (or was) building it."""
|
|
89
|
+
|
|
90
|
+
__slots__ = ("plan", "job", "layer_id", "job_id", "final")
|
|
91
|
+
|
|
92
|
+
def __init__(self, plan: UploadPlan, job: Dict[str, Any]):
|
|
93
|
+
self.plan = plan
|
|
94
|
+
self.job = job or {}
|
|
95
|
+
self.layer_id = self.job.get("layer_id")
|
|
96
|
+
self.job_id = self.job.get("id")
|
|
97
|
+
#: The last status seen when `wait=True`; None when the caller did not wait.
|
|
98
|
+
self.final = None # type: Optional[Dict[str, Any]]
|
|
99
|
+
|
|
100
|
+
def as_dict(self) -> Dict[str, Any]:
|
|
101
|
+
out = {"file": self.plan.path, "name": self.plan.name, "route": self.plan.route,
|
|
102
|
+
"layer_type": self.plan.layer_type, "layer_id": self.layer_id,
|
|
103
|
+
"job_id": self.job_id, "size": self.plan.size}
|
|
104
|
+
if self.final is not None:
|
|
105
|
+
out["status"] = self.final.get("status")
|
|
106
|
+
out["error_message"] = self.final.get("error_message")
|
|
107
|
+
return out
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def detect_layer_type(path: str) -> str:
|
|
111
|
+
return "raster" if os.path.splitext(path)[1].lower() in RASTER_EXTENSIONS else "vector"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def sniff_csv(path: str, sample_bytes: int = 64 * 1024) -> Dict[str, Any]:
|
|
115
|
+
"""Header + delimiter of a CSV, and a guess at its geometry columns.
|
|
116
|
+
|
|
117
|
+
A guess, offered — never applied silently. The CLI prints "using x=lon, y=lat" so a file whose
|
|
118
|
+
`x`/`y` are something else entirely (a grid reference, a pixel index) is caught by the person
|
|
119
|
+
who knows, not discovered later as a layer sitting off the coast of Ghana.
|
|
120
|
+
"""
|
|
121
|
+
with open(path, "r", encoding="utf-8-sig", errors="replace", newline="") as fh:
|
|
122
|
+
sample = fh.read(sample_bytes)
|
|
123
|
+
delimiter = ","
|
|
124
|
+
try:
|
|
125
|
+
delimiter = _csv.Sniffer().sniff(sample, delimiters=",;\t|").delimiter
|
|
126
|
+
except _csv.Error:
|
|
127
|
+
pass
|
|
128
|
+
header = [] # type: List[str]
|
|
129
|
+
for row in _csv.reader(sample.splitlines()[:1], delimiter=delimiter):
|
|
130
|
+
header = [c.strip() for c in row]
|
|
131
|
+
break
|
|
132
|
+
lowered = {c.lower(): c for c in header}
|
|
133
|
+
guess = {"x_column": None, "y_column": None, "wkt_column": None}
|
|
134
|
+
for name in WKT_NAMES:
|
|
135
|
+
if name in lowered:
|
|
136
|
+
guess["wkt_column"] = lowered[name]
|
|
137
|
+
break
|
|
138
|
+
if not guess["wkt_column"]:
|
|
139
|
+
for name in X_NAMES:
|
|
140
|
+
if name in lowered:
|
|
141
|
+
guess["x_column"] = lowered[name]
|
|
142
|
+
break
|
|
143
|
+
for name in Y_NAMES:
|
|
144
|
+
if name in lowered:
|
|
145
|
+
guess["y_column"] = lowered[name]
|
|
146
|
+
break
|
|
147
|
+
if not (guess["x_column"] and guess["y_column"]):
|
|
148
|
+
guess["x_column"] = guess["y_column"] = None
|
|
149
|
+
return {"header": header, "delimiter": DELIMITERS.get(delimiter, "comma"), "guess": guess}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class Uploads(object):
|
|
153
|
+
"""Upload files and register them as layers."""
|
|
154
|
+
|
|
155
|
+
def __init__(self, client: Any):
|
|
156
|
+
self._c = client
|
|
157
|
+
|
|
158
|
+
# ── planning ────────────────────────────────────────────────────────────────────────────────
|
|
159
|
+
|
|
160
|
+
def plan(self, path: str, layer_type: Optional[str] = None, name: Optional[str] = None,
|
|
161
|
+
x_column: Optional[str] = None, y_column: Optional[str] = None,
|
|
162
|
+
wkt_column: Optional[str] = None, srid: int = 4326,
|
|
163
|
+
delimiter: Optional[str] = None, guess_csv: bool = True) -> UploadPlan:
|
|
164
|
+
"""Decide the route for one file, or raise with a message that says what to do instead."""
|
|
165
|
+
if not os.path.isfile(path):
|
|
166
|
+
raise ValidationError(400, "No such file: {0}".format(path))
|
|
167
|
+
size = os.path.getsize(path)
|
|
168
|
+
if size == 0:
|
|
169
|
+
raise ValidationError(400, "{0} is empty.".format(path))
|
|
170
|
+
ext = os.path.splitext(path)[1].lower()
|
|
171
|
+
kind = layer_type or detect_layer_type(path)
|
|
172
|
+
base = name or os.path.splitext(os.path.basename(path))[0]
|
|
173
|
+
|
|
174
|
+
if kind == "raster":
|
|
175
|
+
if ext not in RASTER_EXTENSIONS:
|
|
176
|
+
raise ValidationError(400, "{0} is not a GeoTIFF (.tif/.tiff).".format(path))
|
|
177
|
+
if size >= LARGE_UPLOAD_THRESHOLD:
|
|
178
|
+
return UploadPlan(path, "raster-large", "raster", base, size, chunked=True,
|
|
179
|
+
reason="over the {0} MB direct-upload threshold"
|
|
180
|
+
.format(LARGE_UPLOAD_THRESHOLD // (1024 * 1024)))
|
|
181
|
+
return UploadPlan(path, "raster-api", "raster", base, size)
|
|
182
|
+
|
|
183
|
+
if ext in GEOPARQUET_EXTENSIONS:
|
|
184
|
+
return UploadPlan(path, "geoparquet", "vector", base, size,
|
|
185
|
+
chunked=size > CHUNK_THRESHOLD,
|
|
186
|
+
reason="GeoParquet is always uploaded direct to storage")
|
|
187
|
+
|
|
188
|
+
if ext not in LARGE_VECTOR_EXTENSIONS:
|
|
189
|
+
raise ValidationError(
|
|
190
|
+
400, "Unsupported file type {0}. Vector: {1}, {2}; raster: {3}.".format(
|
|
191
|
+
ext or "(none)", ", ".join(sorted(LARGE_VECTOR_EXTENSIONS)),
|
|
192
|
+
", ".join(sorted(GEOPARQUET_EXTENSIONS)), ", ".join(sorted(RASTER_EXTENSIONS))))
|
|
193
|
+
|
|
194
|
+
csv_opts = None
|
|
195
|
+
if ext == ".csv":
|
|
196
|
+
csv_opts = self._csv_options(path, x_column, y_column, wkt_column, srid, delimiter,
|
|
197
|
+
guess_csv)
|
|
198
|
+
|
|
199
|
+
if size >= LARGE_UPLOAD_THRESHOLD:
|
|
200
|
+
return UploadPlan(path, "large-vector", "vector", base, size, csv_opts,
|
|
201
|
+
chunked=size > CHUNK_THRESHOLD,
|
|
202
|
+
reason="over the {0} MB direct-upload threshold — converts to "
|
|
203
|
+
"GeoParquet in the background"
|
|
204
|
+
.format(LARGE_UPLOAD_THRESHOLD // (1024 * 1024)))
|
|
205
|
+
if ext == ".csv":
|
|
206
|
+
return UploadPlan(path, "csv-api", "vector", base, size, csv_opts)
|
|
207
|
+
return UploadPlan(path, "vector-api", "vector", base, size)
|
|
208
|
+
|
|
209
|
+
def _csv_options(self, path: str, x_column, y_column, wkt_column, srid, delimiter,
|
|
210
|
+
guess_csv: bool) -> Dict[str, Any]:
|
|
211
|
+
opts = {"x_column": x_column, "y_column": y_column, "wkt_column": wkt_column,
|
|
212
|
+
"srid": int(srid or 4326), "delimiter": delimiter or "comma", "guessed": False}
|
|
213
|
+
if wkt_column or (x_column and y_column):
|
|
214
|
+
return opts
|
|
215
|
+
if not guess_csv:
|
|
216
|
+
raise ValidationError(400, "A CSV needs geometry columns: --x and --y, or --wkt.")
|
|
217
|
+
sniffed = sniff_csv(path)
|
|
218
|
+
guess = sniffed["guess"]
|
|
219
|
+
if not (guess["wkt_column"] or (guess["x_column"] and guess["y_column"])):
|
|
220
|
+
raise ValidationError(
|
|
221
|
+
400, "Could not find geometry columns in {0}. Pass --x/--y or --wkt. "
|
|
222
|
+
"Columns: {1}".format(os.path.basename(path),
|
|
223
|
+
", ".join(sniffed["header"][:20]) or "(none read)"))
|
|
224
|
+
opts.update(guess)
|
|
225
|
+
opts["guessed"] = True
|
|
226
|
+
if not delimiter:
|
|
227
|
+
opts["delimiter"] = sniffed["delimiter"]
|
|
228
|
+
return opts
|
|
229
|
+
|
|
230
|
+
# ── uploading ───────────────────────────────────────────────────────────────────────────────
|
|
231
|
+
|
|
232
|
+
def upload(self, path: str, layer_type: Optional[str] = None, name: Optional[str] = None,
|
|
233
|
+
wait: bool = False, on_progress: Optional[Callable[[int, int], None]] = None,
|
|
234
|
+
on_job: Optional[Callable[[Dict[str, Any]], None]] = None,
|
|
235
|
+
cancel: Optional[Callable[[], bool]] = None,
|
|
236
|
+
plan: Optional[UploadPlan] = None, **csv_kw: Any) -> UploadResult:
|
|
237
|
+
"""Upload one file and register it. Returns as soon as the ingest job is QUEUED.
|
|
238
|
+
|
|
239
|
+
`wait=True` blocks until the job finishes and raises `jobs.JobFailed` if it did not — which
|
|
240
|
+
is what a script wants, since a queued job says nothing about whether the data was readable.
|
|
241
|
+
"""
|
|
242
|
+
p = plan or self.plan(path, layer_type, name, **csv_kw)
|
|
243
|
+
route = p.route
|
|
244
|
+
if route == "vector-api":
|
|
245
|
+
job = self._post_file("/data/vector/upload", p, on_progress, cancel)
|
|
246
|
+
elif route == "raster-api":
|
|
247
|
+
job = self._post_file("/data/raster/upload", p, on_progress, cancel)
|
|
248
|
+
elif route == "csv-api":
|
|
249
|
+
job = self._post_file("/data/vector/upload-csv", p, on_progress, cancel,
|
|
250
|
+
fields=_csv_fields(p))
|
|
251
|
+
elif route == "geoparquet":
|
|
252
|
+
key = self._to_storage(p, "geoparquet", on_progress, cancel)
|
|
253
|
+
job = self._c.post("/data/vector/geoparquet/complete",
|
|
254
|
+
{"s3_key": key, "name": p.name, "file_size": p.size})
|
|
255
|
+
elif route == "large-vector":
|
|
256
|
+
key = self._to_storage(p, "large", on_progress, cancel)
|
|
257
|
+
body = {"s3_key": key, "name": p.name, "file_size": p.size}
|
|
258
|
+
if p.csv_opts:
|
|
259
|
+
body.update(_csv_fields(p))
|
|
260
|
+
job = self._c.post("/data/vector/large/complete", body)
|
|
261
|
+
elif route == "raster-large":
|
|
262
|
+
key = self._to_storage(p, "raster", on_progress, cancel)
|
|
263
|
+
job = self._c.post("/data/raster/large/complete",
|
|
264
|
+
{"s3_key": key, "name": p.name, "file_size": p.size})
|
|
265
|
+
else: # pragma: no cover - plan() cannot produce anything else
|
|
266
|
+
raise ValidationError(400, "Unknown upload route {0!r}.".format(route))
|
|
267
|
+
|
|
268
|
+
result = UploadResult(p, job)
|
|
269
|
+
if wait and result.job_id:
|
|
270
|
+
result.final = self._c.jobs.wait(result.job_id, p.layer_type,
|
|
271
|
+
on_progress=on_job)
|
|
272
|
+
return result
|
|
273
|
+
|
|
274
|
+
def upload_many(self, paths: Sequence[str], layer_type: Optional[str] = None,
|
|
275
|
+
wait: bool = False, concurrency: int = 1,
|
|
276
|
+
on_file: Optional[Callable[[UploadPlan], None]] = None,
|
|
277
|
+
on_progress: Optional[Callable[[str, int, int], None]] = None,
|
|
278
|
+
on_job: Optional[Callable[[str, Dict[str, Any]], None]] = None,
|
|
279
|
+
on_error: Optional[Callable[[str, Exception], None]] = None,
|
|
280
|
+
stop_on_error: bool = False, **kw: Any) -> List[Any]:
|
|
281
|
+
"""Upload several files. Every file is PLANNED first, so a bad argument fails before the
|
|
282
|
+
first byte rather than after the first three uploads.
|
|
283
|
+
|
|
284
|
+
Files run sequentially by default: each large file already uses four parallel part uploads,
|
|
285
|
+
and stacking file-level concurrency on top mostly redistributes the same bandwidth while
|
|
286
|
+
making progress output unreadable. Raise `concurrency` for many small files.
|
|
287
|
+
"""
|
|
288
|
+
plans = [self.plan(p, layer_type, **kw) for p in paths]
|
|
289
|
+
results = [] # type: List[Any]
|
|
290
|
+
|
|
291
|
+
def run(p: UploadPlan):
|
|
292
|
+
if on_file:
|
|
293
|
+
on_file(p)
|
|
294
|
+
return self.upload(
|
|
295
|
+
p.path, plan=p, wait=wait,
|
|
296
|
+
on_progress=(lambda done, total: on_progress(p.path, done, total)) if on_progress else None,
|
|
297
|
+
on_job=(lambda st: on_job(p.path, st)) if on_job else None)
|
|
298
|
+
|
|
299
|
+
if concurrency > 1 and len(plans) > 1:
|
|
300
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
301
|
+
with ThreadPoolExecutor(max_workers=min(concurrency, len(plans))) as pool:
|
|
302
|
+
futures = [(p, pool.submit(run, p)) for p in plans]
|
|
303
|
+
for p, future in futures:
|
|
304
|
+
try:
|
|
305
|
+
results.append(future.result())
|
|
306
|
+
except Exception as exc: # noqa: BLE001 - reported per file, not swallowed
|
|
307
|
+
if on_error:
|
|
308
|
+
on_error(p.path, exc)
|
|
309
|
+
results.append(exc)
|
|
310
|
+
return results
|
|
311
|
+
|
|
312
|
+
for p in plans:
|
|
313
|
+
try:
|
|
314
|
+
results.append(run(p))
|
|
315
|
+
except Exception as exc: # noqa: BLE001
|
|
316
|
+
if on_error:
|
|
317
|
+
on_error(p.path, exc)
|
|
318
|
+
results.append(exc)
|
|
319
|
+
if stop_on_error:
|
|
320
|
+
break
|
|
321
|
+
return results
|
|
322
|
+
|
|
323
|
+
# ── the two transports ──────────────────────────────────────────────────────────────────────
|
|
324
|
+
|
|
325
|
+
def _post_file(self, path: str, plan: UploadPlan,
|
|
326
|
+
on_progress: Optional[Callable[[int, int], None]],
|
|
327
|
+
cancel: Optional[Callable[[], bool]] = None,
|
|
328
|
+
fields: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
329
|
+
"""multipart/form-data through the API — streamed from disk, never buffered."""
|
|
330
|
+
body = MultipartBody(fields=fields, file_path=plan.path,
|
|
331
|
+
filename=os.path.basename(plan.path),
|
|
332
|
+
on_progress=on_progress, cancel=cancel)
|
|
333
|
+
return self._c.request("POST", path, body=body, content_type=body.content_type,
|
|
334
|
+
timeout=self._c.upload_timeout)
|
|
335
|
+
|
|
336
|
+
def _to_storage(self, plan: UploadPlan, kind: str,
|
|
337
|
+
on_progress: Optional[Callable[[int, int], None]],
|
|
338
|
+
cancel: Optional[Callable[[], bool]] = None) -> str:
|
|
339
|
+
"""Direct-to-storage upload; returns the object key to register. Chunked when big."""
|
|
340
|
+
if plan.chunked:
|
|
341
|
+
return self._chunked(plan, kind, on_progress, cancel)
|
|
342
|
+
endpoint = ("/data/vector/geoparquet/presign" if kind == "geoparquet"
|
|
343
|
+
else "/data/vector/large/presign")
|
|
344
|
+
pre = self._c.post(endpoint, {"filename": os.path.basename(plan.path),
|
|
345
|
+
"name": plan.name, "file_size": plan.size})
|
|
346
|
+
with open(plan.path, "rb") as fh:
|
|
347
|
+
reader = ProgressReader(fh, plan.size, on_progress, cancel)
|
|
348
|
+
response = self._c.send_absolute("PUT", pre["upload_url"], reader,
|
|
349
|
+
{"Content-Type": "application/octet-stream"})
|
|
350
|
+
if response.status >= 400:
|
|
351
|
+
from .errors import from_status
|
|
352
|
+
raise from_status(response.status,
|
|
353
|
+
"Storage rejected the upload: {0}".format(response.text[:300]),
|
|
354
|
+
response.url)
|
|
355
|
+
return pre["s3_key"]
|
|
356
|
+
|
|
357
|
+
def _chunked(self, plan: UploadPlan, kind: str,
|
|
358
|
+
on_progress: Optional[Callable[[int, int], None]],
|
|
359
|
+
cancel: Optional[Callable[[], bool]] = None) -> str:
|
|
360
|
+
"""initiate → PUT every part (in parallel, with retries) → assemble.
|
|
361
|
+
|
|
362
|
+
A failure aborts the multipart upload so the staged parts are not left paying for storage
|
|
363
|
+
on someone's S3 bill.
|
|
364
|
+
"""
|
|
365
|
+
base = "/data/raster" if kind == "raster" else "/data/vector"
|
|
366
|
+
init = self._c.post(base + "/upload/multipart/initiate",
|
|
367
|
+
{"filename": os.path.basename(plan.path), "file_size": plan.size,
|
|
368
|
+
"kind": kind})
|
|
369
|
+
key, upload_id = init["s3_key"], init["upload_id"]
|
|
370
|
+
part_size, parts = init["part_size"], init["parts"]
|
|
371
|
+
|
|
372
|
+
sent = [0] * len(parts)
|
|
373
|
+
lock = threading.Lock()
|
|
374
|
+
results = [None] * len(parts) # type: List[Optional[Dict[str, Any]]]
|
|
375
|
+
|
|
376
|
+
def report():
|
|
377
|
+
if on_progress:
|
|
378
|
+
on_progress(min(sum(sent), plan.size), plan.size)
|
|
379
|
+
|
|
380
|
+
def put(index: int) -> None:
|
|
381
|
+
part = parts[index]
|
|
382
|
+
number = int(part["part_number"])
|
|
383
|
+
offset = (number - 1) * part_size
|
|
384
|
+
length = min(part_size, plan.size - offset)
|
|
385
|
+
last_error = None
|
|
386
|
+
for attempt in range(PART_RETRIES):
|
|
387
|
+
if cancel is not None and cancel():
|
|
388
|
+
raise _Cancelled()
|
|
389
|
+
try:
|
|
390
|
+
with open(plan.path, "rb") as fh:
|
|
391
|
+
fh.seek(offset)
|
|
392
|
+
chunk = _Slice(fh, length)
|
|
393
|
+
|
|
394
|
+
def progress(done, _total, i=index):
|
|
395
|
+
with lock:
|
|
396
|
+
sent[i] = done
|
|
397
|
+
report()
|
|
398
|
+
|
|
399
|
+
reader = ProgressReader(chunk, length, progress, cancel)
|
|
400
|
+
response = self._c.send_absolute("PUT", part["url"], reader)
|
|
401
|
+
if response.status < 400:
|
|
402
|
+
etag = (response.headers.get("etag") or "").strip('"')
|
|
403
|
+
if not etag:
|
|
404
|
+
raise ValidationError(
|
|
405
|
+
502, "Storage accepted part {0} but returned no ETag, so the "
|
|
406
|
+
"upload cannot be assembled.".format(number))
|
|
407
|
+
results[index] = {"part_number": number, "etag": etag}
|
|
408
|
+
with lock:
|
|
409
|
+
sent[index] = length
|
|
410
|
+
report()
|
|
411
|
+
return
|
|
412
|
+
last_error = "HTTP {0}: {1}".format(response.status, response.text[:200])
|
|
413
|
+
except _Cancelled:
|
|
414
|
+
raise
|
|
415
|
+
except Exception as exc: # noqa: BLE001 - retried below
|
|
416
|
+
last_error = str(exc)
|
|
417
|
+
with lock:
|
|
418
|
+
sent[index] = 0
|
|
419
|
+
report()
|
|
420
|
+
raise ValidationError(502, "Part {0} of {1} failed after {2} attempts: {3}".format(
|
|
421
|
+
number, len(parts), PART_RETRIES, last_error))
|
|
422
|
+
|
|
423
|
+
try:
|
|
424
|
+
if len(parts) > 1 and PART_CONCURRENCY > 1:
|
|
425
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
426
|
+
with ThreadPoolExecutor(max_workers=min(PART_CONCURRENCY, len(parts))) as pool:
|
|
427
|
+
for _ in pool.map(put, range(len(parts))):
|
|
428
|
+
pass
|
|
429
|
+
else:
|
|
430
|
+
for index in range(len(parts)):
|
|
431
|
+
put(index)
|
|
432
|
+
self._c.post(base + "/upload/multipart/complete",
|
|
433
|
+
{"s3_key": key, "upload_id": upload_id,
|
|
434
|
+
"parts": [r for r in results if r]})
|
|
435
|
+
except BaseException:
|
|
436
|
+
try:
|
|
437
|
+
self._c.post(base + "/upload/multipart/abort",
|
|
438
|
+
{"s3_key": key, "upload_id": upload_id})
|
|
439
|
+
except Exception: # noqa: BLE001 - the original failure is what matters
|
|
440
|
+
pass
|
|
441
|
+
raise
|
|
442
|
+
return key
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
class _Cancelled(Exception):
|
|
446
|
+
"""Internal: a cancel callback said stop. Turned into a TransportError by the caller."""
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
class _Slice(object):
|
|
450
|
+
"""A read-only window onto an open file — one multipart part, without copying it."""
|
|
451
|
+
|
|
452
|
+
def __init__(self, fh, length: int):
|
|
453
|
+
self._fh = fh
|
|
454
|
+
self._left = length
|
|
455
|
+
|
|
456
|
+
def read(self, size: int = -1) -> bytes:
|
|
457
|
+
if self._left <= 0:
|
|
458
|
+
return b""
|
|
459
|
+
want = self._left if size is None or size < 0 else min(size, self._left)
|
|
460
|
+
chunk = self._fh.read(want)
|
|
461
|
+
self._left -= len(chunk)
|
|
462
|
+
return chunk
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def _csv_fields(plan: UploadPlan) -> Dict[str, Any]:
|
|
466
|
+
"""The CSV geometry options, in the shape both CSV routes expect (form fields / JSON body)."""
|
|
467
|
+
opts = dict(plan.csv_opts or {})
|
|
468
|
+
opts.pop("guessed", None)
|
|
469
|
+
fields = {"name": plan.name, "srid": opts.get("srid", 4326),
|
|
470
|
+
"delimiter": opts.get("delimiter", "comma")}
|
|
471
|
+
for key in ("x_column", "y_column", "wkt_column"):
|
|
472
|
+
if opts.get(key):
|
|
473
|
+
fields[key] = opts[key]
|
|
474
|
+
return fields
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: geodeploy
|
|
3
|
+
Version: 1.3.0
|
|
4
|
+
Summary: Command-line client and Python API for GeoDeploy — upload spatial data, build and publish portals.
|
|
5
|
+
Author: Koffi Dodji Noumonvi
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/bravemaster3/GeoDeploy
|
|
8
|
+
Project-URL: Documentation, https://docs-geodeploy.kndev.org/cli/
|
|
9
|
+
Project-URL: Repository, https://github.com/bravemaster3/GeoDeploy
|
|
10
|
+
Project-URL: Issues, https://github.com/bravemaster3/GeoDeploy/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/bravemaster3/GeoDeploy/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: gis,geospatial,postgis,geoparquet,cog,maplibre,qgis,stac,ogc-api-features,geodeploy
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Intended Audience :: System Administrators
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: GIS
|
|
25
|
+
Classifier: Topic :: Utilities
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.9
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
License-File: LICENSE
|
|
30
|
+
License-File: NOTICE
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# geodeploy
|
|
36
|
+
|
|
37
|
+
**Command-line client and Python API for [GeoDeploy](https://github.com/bravemaster3/GeoDeploy)** —
|
|
38
|
+
the self-hosted spatial data platform and geoportal builder.
|
|
39
|
+
|
|
40
|
+
Upload data, style it, build a portal and publish it, without opening a browser:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install geodeploy
|
|
44
|
+
|
|
45
|
+
geodeploy login https://geodeploy.example.org --token gdp_…
|
|
46
|
+
geodeploy upload roads.gpkg sites.csv dem.tif --wait
|
|
47
|
+
geodeploy portals create "Field sites 2026" --publish
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
**No dependencies. Python 3.9+.** Every request goes through the standard library, so installing
|
|
51
|
+
this pulls in nothing else — and a QGIS plugin can vendor the same client without asking anyone to
|
|
52
|
+
pip-install into QGIS.
|
|
53
|
+
|
|
54
|
+
If your shell cannot find `geodeploy` after installing, pip's scripts directory is not on your
|
|
55
|
+
`PATH` — `python -m geodeploy` works regardless, and `pipx install geodeploy` or a virtual
|
|
56
|
+
environment avoids it.
|
|
57
|
+
|
|
58
|
+
## What it covers
|
|
59
|
+
|
|
60
|
+
- **Uploads** — shapefile, GeoPackage, GeoJSON, CSV (X/Y or WKT), GeoParquet, GeoTIFF. Many files
|
|
61
|
+
in one command; the route is chosen per file, and anything over 48 MB goes direct to object
|
|
62
|
+
storage in parallel presigned parts, so a multi-gigabyte upload survives a proxy that caps
|
|
63
|
+
request bodies.
|
|
64
|
+
- **Layers** — list, inspect, rename, share (STAC / OGC API - Features opt-in), share links per
|
|
65
|
+
tool, download, delete, restart a stalled ingest.
|
|
66
|
+
- **Symbology** — colour, opacity, outlines, dashes, markers, raster colormaps and stretches, plus
|
|
67
|
+
data-driven styling: `--color-field pop --classify jenks --classes 6 --ramp magma`, proportional
|
|
68
|
+
size, and attribute-driven 3D. Classification is computed by the instance, with the same code the
|
|
69
|
+
portal editor uses, so the CLI and the editor can never disagree about a class.
|
|
70
|
+
- **Portals** — create (web map, story map or catalog), arrange and style layers, folders, About
|
|
71
|
+
page, assets, access tiers, publish, and a whole-configuration JSON round trip for version
|
|
72
|
+
control.
|
|
73
|
+
- **Everything else** — external WMS/XYZ/WFS services, registering data already in PostGIS or the
|
|
74
|
+
bucket, the public STAC and OGC API - Features catalog, ingest jobs, users, and instance
|
|
75
|
+
administration (health, services, updates, backups, activity log).
|
|
76
|
+
|
|
77
|
+
Every command takes `--json`, where stdout is exactly one JSON document, and exit codes distinguish
|
|
78
|
+
authentication (3) from network (4) from server (5) so a scheduled job can alert on the right thing.
|
|
79
|
+
|
|
80
|
+
## As a library
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from geodeploy import Client
|
|
84
|
+
|
|
85
|
+
gd = Client("https://geodeploy.example.org", token="gdp_…")
|
|
86
|
+
result = gd.uploads.upload("roads.gpkg", wait=True)
|
|
87
|
+
portal = gd.portals.create("Roads")
|
|
88
|
+
gd.portals.add_layer(portal["id"], result.layer_id, "vector", {"color": "#e11d48"})
|
|
89
|
+
gd.portals.publish(portal["id"])
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Typed exceptions (`AuthError`, `PermissionError_`, `NotFoundError`, `ValidationError`,
|
|
93
|
+
`ServerError`, `TransportError`, `JobFailed`), progress callbacks and cancellation on uploads, and a
|
|
94
|
+
swappable transport so a desktop application can route requests through its own network stack.
|
|
95
|
+
|
|
96
|
+
## Documentation
|
|
97
|
+
|
|
98
|
+
Full guide: **<https://docs-geodeploy.kndev.org/cli/>**
|
|
99
|
+
|
|
100
|
+
Source and issues: **<https://github.com/bravemaster3/GeoDeploy>**
|
|
101
|
+
|
|
102
|
+
Licensed under the Apache License 2.0.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
geodeploy/__init__.py,sha256=_SvWSaGYI_HlTfk7EzbFVVWwkDhhxCIxrE7Jh7jc3r4,1533
|
|
2
|
+
geodeploy/__main__.py,sha256=rvVpk_W4tJQrI1InsnKmuJAtFCgvfemzjVixJ1qN8Ck,346
|
|
3
|
+
geodeploy/admin.py,sha256=u7F9M2qBIPwx5tbXsbc2YXsQInZYKxfAle-gfCUhTWc,9818
|
|
4
|
+
geodeploy/catalog.py,sha256=TwkAT1NLGrYrN-siMkI8FpULz4QxoCwbYB0gxDnPpts,5336
|
|
5
|
+
geodeploy/client.py,sha256=quxJxFXwUWIH3UsAVxZATyKtbjLItLex_63BhEV2doc,15990
|
|
6
|
+
geodeploy/config.py,sha256=dgMxs0hX8sLQhJCbl42wL6N6LiY-E2yeu67aIT7mhQc,18901
|
|
7
|
+
geodeploy/errors.py,sha256=bJTxAD8a3ldjGCQjg_uSkfLuiM47fg4woFp4NbU_4jc,3611
|
|
8
|
+
geodeploy/imports.py,sha256=JEoHprUmqQu4Id9ll2Ko3lvIEjRTps9zh-YrPQj6LRs,3733
|
|
9
|
+
geodeploy/jobs.py,sha256=VywegdPGLCQR0vhXTI4VESsRTZFT6r_X1b6S3NjTGyE,3169
|
|
10
|
+
geodeploy/layers.py,sha256=-pSkzg5jKdUXgm_ExFBNLswAYBVh8Vyj92t60y1qmq8,22264
|
|
11
|
+
geodeploy/portals.py,sha256=QhVO0tdsyisj4Dlmw2TZvrrUljz4zFmgUK3Rgjv0_1E,15303
|
|
12
|
+
geodeploy/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
+
geodeploy/sources.py,sha256=rEkyRb7-6-qfBmuoUVKrUJMORTkzBSF56r4xxMtZvCc,3459
|
|
14
|
+
geodeploy/styles.py,sha256=XzM3o9vcjLWIBlWvUv6vb7aqAgRRM4qjq5dlvmty7bs,20763
|
|
15
|
+
geodeploy/transport.py,sha256=E1rLQ4ILt9Z4P9bcRJaKWdy-iFcwNaSeBLWEb6e2pLw,15461
|
|
16
|
+
geodeploy/uploads.py,sha256=92sH3VqV4cdFzUD65vekR6Kk0Rlyw75tmNN7f9NEPCg,23669
|
|
17
|
+
geodeploy/cli/__init__.py,sha256=X4w5xNAFqRxJ5ahCVv6Hx0ZECYC1k7xbwRnvGINP_BM,349
|
|
18
|
+
geodeploy/cli/main.py,sha256=F_wpSreSv5TfaRqygio2rKRIxHf9Urjhqs_ioAFh1xk,12718
|
|
19
|
+
geodeploy/cli/output.py,sha256=cHsBSj4ViKvN5HB5mZ1SVjBbLs_cIylrs5eWhgCSexE,13767
|
|
20
|
+
geodeploy/cli/commands/__init__.py,sha256=Sx5RGUwS0fxFEq1AN6XgUPmsRBpzmfu2Wju77qtc9ug,108
|
|
21
|
+
geodeploy/cli/commands/_common.py,sha256=5SMtig-9cQp1XSKu7OJ9r_N2yLrXtgZX_GuHU1aDWjc,12302
|
|
22
|
+
geodeploy/cli/commands/admin.py,sha256=LyHE-fLp5cj_i4gs7Te6sTZJswVdUJiMMVu-ibd0fRs,15483
|
|
23
|
+
geodeploy/cli/commands/auth.py,sha256=EhDemns2-s0XEPjTvR4aC2ie0T3X5VXdEXbaO0g2AO4,13692
|
|
24
|
+
geodeploy/cli/commands/browse.py,sha256=HXa-S3u8XxWSMMgyajPE7WqwaN3vmcBSEWS6KWyiqcE,9865
|
|
25
|
+
geodeploy/cli/commands/catalog.py,sha256=q47Ofyepl38lc0uW9Rv9FnEI7RqwZd-5EPcsHUik9Xs,4528
|
|
26
|
+
geodeploy/cli/commands/imports.py,sha256=3jbObyu8vDNsVO51xcjlaK1Q57m0ZrnDI36hkju0CCM,5920
|
|
27
|
+
geodeploy/cli/commands/jobs.py,sha256=TBIKr5qJMnm4DFlA7dWu5woyYvXPEEv2_MNuKAI4U8E,1728
|
|
28
|
+
geodeploy/cli/commands/layers.py,sha256=ofSV_J7R3JkuD_963IY4UgMg1Hcv9Xoqhm7HDV73dIg,25033
|
|
29
|
+
geodeploy/cli/commands/portals.py,sha256=NFd7j0_JhlsHwFOscpxXal3gHLB-OYMSUyJeR3g-FrY,23527
|
|
30
|
+
geodeploy/cli/commands/sources.py,sha256=o__PCtODJ0KXUQCf1xQdE7SME2TgyK7P6Chn5xzF_PQ,4065
|
|
31
|
+
geodeploy/cli/commands/upload.py,sha256=tj9g_Z5Rr65UG8E-VJWrFB-pjv67kGkY-Ul-7faFQAs,7827
|
|
32
|
+
geodeploy-1.3.0.dist-info/licenses/LICENSE,sha256=t4XqfCgddCh3KUtaOfF8RdWGRI50I96VmKdpW4fpmms,11351
|
|
33
|
+
geodeploy-1.3.0.dist-info/licenses/NOTICE,sha256=Po22nWikm2_1s2z1pKDxV9ykhHeAgtNcdAeeOTFMQPY,464
|
|
34
|
+
geodeploy-1.3.0.dist-info/METADATA,sha256=bhr8xwGsZXPdvcE0UkvR6ZQpLARs-7z7QimbJlgZyNs,4729
|
|
35
|
+
geodeploy-1.3.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
36
|
+
geodeploy-1.3.0.dist-info/entry_points.txt,sha256=_z6fwX2EZR2TPahf4z56UYYshQjqqgOTsWncrm_4Kvw,54
|
|
37
|
+
geodeploy-1.3.0.dist-info/top_level.txt,sha256=-PsREGzr8T0P_FpPUHd7WpbPDP-tKGKXa8peAcLQF5c,10
|
|
38
|
+
geodeploy-1.3.0.dist-info/RECORD,,
|