geodeploy 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
geodeploy/transport.py ADDED
@@ -0,0 +1,355 @@
1
+ """HTTP transport — stdlib only, and swappable.
2
+
3
+ Two things this module exists for:
4
+
5
+ 1. **No dependencies.** `urllib.request` sends every request this client makes, including a 10 GB
6
+ presigned PUT, because a file object passed as the body is streamed by `http.client` in 8 KB
7
+ blocks rather than read into memory.
8
+
9
+ 2. **A seam for a host that has its own network stack.** A QGIS plugin should go through
10
+ `QgsNetworkAccessManager` so it inherits the user's proxy, their CA bundle and their
11
+ authentication configuration — none of which urllib knows about. Anything with a
12
+ `send(Request) -> Response` method can be passed as `Client(transport=…)`, so that plugin
13
+ supplies ~30 lines and reuses every other line in this package.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import gzip
18
+ import io
19
+ import json
20
+ import os
21
+ import socket
22
+ import ssl
23
+ import time
24
+ import urllib.error
25
+ import urllib.request
26
+ from typing import Any, Callable, Dict, Iterable, Optional, Union
27
+
28
+ from .errors import TransportError
29
+
30
+ #: Bytes read per chunk when streaming a file body. Matches http.client's own block size, so a
31
+ #: progress callback fires at the same rate the socket is actually fed.
32
+ BLOCK = 64 * 1024
33
+
34
+
35
+ class Request:
36
+ """One HTTP request, fully resolved: absolute URL, final headers, body ready to send."""
37
+
38
+ __slots__ = ("method", "url", "headers", "body", "timeout")
39
+
40
+ def __init__(self, method: str, url: str, headers: Optional[Dict[str, str]] = None,
41
+ body: Union[bytes, io.IOBase, None] = None, timeout: Optional[float] = None):
42
+ self.method = method.upper()
43
+ self.url = url
44
+ self.headers = dict(headers or {})
45
+ #: bytes, or a file-like object with `read(n)` (streamed — never loaded whole).
46
+ self.body = body
47
+ self.timeout = timeout
48
+
49
+ def __repr__(self) -> str: # pragma: no cover - debugging aid
50
+ return "<Request {0} {1}>".format(self.method, self.url)
51
+
52
+
53
+ class Response:
54
+ """One HTTP answer. `content` is always bytes; decoding is the caller's business."""
55
+
56
+ __slots__ = ("status", "headers", "content", "url")
57
+
58
+ def __init__(self, status: int, headers: Dict[str, str], content: bytes, url: str = ""):
59
+ self.status = status
60
+ #: Lower-cased keys — HTTP header names are case-insensitive and callers should not have to
61
+ #: remember whether this instance's nginx wrote `ETag` or `etag`.
62
+ self.headers = {str(k).lower(): v for k, v in (headers or {}).items()}
63
+ self.content = content
64
+ self.url = url
65
+
66
+ @property
67
+ def text(self) -> str:
68
+ return self.content.decode("utf-8", "replace")
69
+
70
+ def json(self) -> Any:
71
+ if not self.content:
72
+ return None
73
+ return json.loads(self.text)
74
+
75
+ @property
76
+ def ok(self) -> bool:
77
+ return 200 <= self.status < 400
78
+
79
+ def __repr__(self) -> str: # pragma: no cover - debugging aid
80
+ return "<Response {0} {1}>".format(self.status, self.url)
81
+
82
+
83
+ class ProgressReader(io.RawIOBase):
84
+ """A file wrapper that reports how much of it has been sent.
85
+
86
+ Wrapping the READER rather than counting before the request is what makes the number honest:
87
+ it advances as the socket drains, so a stalled upload stops moving instead of showing 100 %
88
+ while the connection hangs.
89
+ """
90
+
91
+ def __init__(self, fh, total: int, on_progress: Optional[Callable[[int, int], None]] = None,
92
+ cancel: Optional[Callable[[], bool]] = None):
93
+ self._fh = fh
94
+ self._total = total
95
+ self._sent = 0
96
+ self._on_progress = on_progress
97
+ self._cancel = cancel
98
+
99
+ def read(self, size: int = -1) -> bytes: # noqa: D102 - file protocol
100
+ if self._cancel is not None and self._cancel():
101
+ # http.client turns this into a failed request, which is what a cancel should look
102
+ # like: the upload stops mid-flight and the caller gets an exception, not a silent
103
+ # truncated object on the far end.
104
+ raise TransportError("Upload cancelled.")
105
+ chunk = self._fh.read(BLOCK if size is None or size < 0 else size)
106
+ if chunk:
107
+ self._sent += len(chunk)
108
+ if self._on_progress:
109
+ self._on_progress(self._sent, self._total)
110
+ return chunk
111
+
112
+ def readable(self) -> bool:
113
+ return True
114
+
115
+ def __len__(self) -> int:
116
+ return self._total
117
+
118
+
119
+ class MultipartBody(io.RawIOBase):
120
+ """`multipart/form-data` built as a STREAM, so a 2 GB upload never enters memory.
121
+
122
+ `urllib` will happily take a bytes body, and every "upload a file" example does exactly that —
123
+ which is fine until the file is a GeoPackage. This reads the parts in order (preamble, file
124
+ from disk, epilogue) and reports a real `Content-Length`, which the API needs since it does not
125
+ accept chunked transfer.
126
+ """
127
+
128
+ def __init__(self, fields: Optional[Dict[str, Any]] = None,
129
+ file_field: str = "file", file_path: Optional[str] = None,
130
+ filename: Optional[str] = None, content_type: str = "application/octet-stream",
131
+ boundary: Optional[str] = None,
132
+ on_progress: Optional[Callable[[int, int], None]] = None,
133
+ cancel: Optional[Callable[[], bool]] = None):
134
+ self.boundary = boundary or ("gd" + os.urandom(16).hex())
135
+ self._path = file_path
136
+ self._fh = None
137
+ self._on_progress = on_progress
138
+ self._cancel = cancel
139
+ self._sent = 0
140
+
141
+ pre = io.BytesIO()
142
+ for key, value in (fields or {}).items():
143
+ if value is None:
144
+ continue
145
+ pre.write(self._dash())
146
+ pre.write('Content-Disposition: form-data; name="{0}"\r\n\r\n'.format(key).encode())
147
+ pre.write(str(value).encode("utf-8"))
148
+ pre.write(b"\r\n")
149
+ if file_path is not None:
150
+ pre.write(self._dash())
151
+ pre.write(
152
+ 'Content-Disposition: form-data; name="{0}"; filename="{1}"\r\n'
153
+ .format(file_field, filename or os.path.basename(file_path)).encode("utf-8"))
154
+ pre.write("Content-Type: {0}\r\n\r\n".format(content_type).encode())
155
+ self._pre = pre.getvalue()
156
+ self._post = b"\r\n" + self._dash(final=True) if file_path is not None else self._dash(final=True)
157
+ self._file_size = os.path.getsize(file_path) if file_path is not None else 0
158
+ self._stage = 0 # 0 preamble, 1 file, 2 epilogue, 3 done
159
+ self._offset = 0
160
+
161
+ def _dash(self, final: bool = False) -> bytes:
162
+ return "--{0}{1}\r\n".format(self.boundary, "--" if final else "").encode()
163
+
164
+ @property
165
+ def content_type(self) -> str:
166
+ return "multipart/form-data; boundary={0}".format(self.boundary)
167
+
168
+ def __len__(self) -> int:
169
+ return len(self._pre) + self._file_size + len(self._post)
170
+
171
+ def read(self, size: int = -1) -> bytes: # noqa: D102 - file protocol
172
+ if self._cancel is not None and self._cancel():
173
+ raise TransportError("Upload cancelled.")
174
+ want = BLOCK if size is None or size < 0 else size
175
+ if self._stage == 0:
176
+ chunk = self._pre[self._offset:self._offset + want]
177
+ self._offset += len(chunk)
178
+ if self._offset >= len(self._pre):
179
+ self._stage, self._offset = (1 if self._path else 2), 0
180
+ return self._advance(chunk)
181
+ if self._stage == 1:
182
+ if self._fh is None:
183
+ self._fh = open(self._path, "rb")
184
+ chunk = self._fh.read(want)
185
+ if not chunk:
186
+ self._fh.close()
187
+ self._fh = None
188
+ self._stage, self._offset = 2, 0
189
+ return self.read(size)
190
+ return self._advance(chunk)
191
+ if self._stage == 2:
192
+ chunk = self._post[self._offset:self._offset + want]
193
+ self._offset += len(chunk)
194
+ if self._offset >= len(self._post):
195
+ self._stage = 3
196
+ return self._advance(chunk)
197
+ return b""
198
+
199
+ def _advance(self, chunk: bytes) -> bytes:
200
+ self._sent += len(chunk)
201
+ if self._on_progress and chunk:
202
+ self._on_progress(min(self._sent, len(self)), len(self))
203
+ return chunk
204
+
205
+ def readable(self) -> bool:
206
+ return True
207
+
208
+ def close(self) -> None: # pragma: no cover - cleanup path
209
+ if self._fh is not None:
210
+ try:
211
+ self._fh.close()
212
+ finally:
213
+ self._fh = None
214
+ super().close()
215
+
216
+
217
+ class _MethodRequest(urllib.request.Request):
218
+ """urllib picks GET/POST from whether there is a body; PUT and DELETE need saying out loud."""
219
+
220
+ def __init__(self, *args, **kwargs):
221
+ self._method = kwargs.pop("method_", "GET")
222
+ urllib.request.Request.__init__(self, *args, **kwargs)
223
+
224
+ def get_method(self) -> str:
225
+ return self._method
226
+
227
+
228
+ class UrllibTransport:
229
+ """The default transport: `urllib.request`, no dependencies.
230
+
231
+ Retries are deliberately narrow — connection-level failures and 502/503/504, which mean the
232
+ request did not run or the gateway gave up. A 4xx is never retried (it will fail identically),
233
+ and neither is a non-idempotent request that actually reached the app.
234
+ """
235
+
236
+ #: Statuses worth a second attempt: an nginx/gateway answer, not the application's.
237
+ RETRY_STATUS = (502, 503, 504)
238
+
239
+ def __init__(self, verify_tls: bool = True, retries: int = 2, backoff: float = 0.75,
240
+ ca_bundle: Optional[str] = None):
241
+ self.retries = max(0, int(retries))
242
+ self.backoff = backoff
243
+ if verify_tls:
244
+ self._ctx = ssl.create_default_context(cafile=ca_bundle) if ca_bundle else None
245
+ else:
246
+ # For a self-signed instance on a lab network. The CLI only reaches this via an
247
+ # explicit --insecure, and it says so on every run.
248
+ ctx = ssl.create_default_context()
249
+ ctx.check_hostname = False
250
+ ctx.verify_mode = ssl.CERT_NONE
251
+ self._ctx = ctx
252
+ # No cookie jar and no redirect-following for anything but GET: a 307 on a streamed upload
253
+ # cannot be replayed (the file object is already partly consumed), so it must surface.
254
+ self._opener = urllib.request.build_opener(urllib.request.HTTPSHandler(context=self._ctx)
255
+ if self._ctx else urllib.request.HTTPSHandler())
256
+
257
+ def send(self, request: Request) -> Response:
258
+ streamed = not isinstance(request.body, (bytes, bytearray, type(None)))
259
+ attempts = 1 if streamed else self.retries + 1
260
+ last_exc = None # type: Optional[Exception]
261
+ for attempt in range(attempts):
262
+ try:
263
+ return self._send_once(request)
264
+ except TransportError as exc:
265
+ last_exc = exc
266
+ if attempt == attempts - 1:
267
+ raise
268
+ time.sleep(self.backoff * (2 ** attempt))
269
+ except _RetryStatus as exc:
270
+ last_exc = exc
271
+ if attempt == attempts - 1:
272
+ return exc.response
273
+ time.sleep(self.backoff * (2 ** attempt))
274
+ raise last_exc if last_exc else TransportError("Request failed") # pragma: no cover
275
+
276
+ def _send_once(self, request: Request) -> Response:
277
+ headers = dict(request.headers)
278
+ body = request.body
279
+ if body is not None and not isinstance(body, (bytes, bytearray)):
280
+ # Streamed body: urllib will not guess a length for a file object, and without
281
+ # Content-Length http.client falls back to chunked encoding, which S3 presigned PUTs
282
+ # reject outright (the signature covers the exact length).
283
+ length = len(body) if hasattr(body, "__len__") else None
284
+ if length is not None:
285
+ headers.setdefault("Content-Length", str(length))
286
+ req = _MethodRequest(request.url, data=body, headers=headers, method_=request.method)
287
+ try:
288
+ with self._opener.open(req, timeout=request.timeout) as resp:
289
+ content = resp.read()
290
+ if (resp.headers.get("Content-Encoding") or "").lower() == "gzip":
291
+ content = gzip.decompress(content)
292
+ out = Response(resp.status, dict(resp.headers), content, request.url)
293
+ except urllib.error.HTTPError as exc:
294
+ content = exc.read() or b""
295
+ if (exc.headers.get("Content-Encoding") or "").lower() == "gzip":
296
+ try:
297
+ content = gzip.decompress(content)
298
+ except OSError: # pragma: no cover - a lying header
299
+ pass
300
+ out = Response(exc.code, dict(exc.headers or {}), content, request.url)
301
+ if exc.code in self.RETRY_STATUS:
302
+ raise _RetryStatus(out)
303
+ return out
304
+ except urllib.error.URLError as exc:
305
+ raise TransportError(_reason(exc.reason, request.url)) from exc
306
+ except (socket.timeout, TimeoutError) as exc:
307
+ raise TransportError("Timed out talking to {0}".format(request.url)) from exc
308
+ except ConnectionError as exc: # reset mid-body, broken pipe on a big upload
309
+ raise TransportError("Connection lost talking to {0}: {1}".format(request.url, exc)) from exc
310
+ return out
311
+
312
+ def stream(self, request: Request, sink, chunk: int = BLOCK) -> Response:
313
+ """GET straight to a writable file object — for downloads that must not enter memory."""
314
+ req = _MethodRequest(request.url, data=None, headers=dict(request.headers),
315
+ method_=request.method)
316
+ try:
317
+ with self._opener.open(req, timeout=request.timeout) as resp:
318
+ total = int(resp.headers.get("Content-Length") or 0)
319
+ read = 0
320
+ while True:
321
+ block = resp.read(chunk)
322
+ if not block:
323
+ break
324
+ sink.write(block)
325
+ read += len(block)
326
+ return Response(resp.status, dict(resp.headers), b"", request.url)
327
+ except urllib.error.HTTPError as exc:
328
+ return Response(exc.code, dict(exc.headers or {}), exc.read() or b"", request.url)
329
+ except urllib.error.URLError as exc:
330
+ raise TransportError(_reason(exc.reason, request.url)) from exc
331
+
332
+
333
+ class _RetryStatus(Exception):
334
+ """Internal: a gateway status that `send` may retry, carrying the response if it will not."""
335
+
336
+ def __init__(self, response: Response):
337
+ self.response = response
338
+ Exception.__init__(self, "HTTP {0}".format(response.status))
339
+
340
+
341
+ def _reason(reason: Any, url: str) -> str:
342
+ """Turn urllib's reason object into something that names the instance that failed.
343
+
344
+ "[Errno 11001] getaddrinfo failed" tells a user nothing; which URL could not be reached tells
345
+ them they typed the host wrong, which is the actual cause most of the time.
346
+ """
347
+ if isinstance(reason, ssl.SSLError):
348
+ return ("TLS error talking to {0}: {1}. If this is a self-signed instance, pass --insecure."
349
+ .format(url, reason))
350
+ return "Could not reach {0}: {1}".format(url, reason)
351
+
352
+
353
+ def iter_chunks(paths: Iterable[str]) -> Iterable[bytes]: # pragma: no cover - reserved
354
+ """Placeholder kept out of the public API on purpose; multipart uploads read per part."""
355
+ raise NotImplementedError