localbucket 1.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: localbucket
|
|
3
|
+
Version: 1.2.1
|
|
4
|
+
Summary: Minimal S3-compatible server that stores buckets as plain folders on disk, for local development and testing.
|
|
5
|
+
Project-URL: Homepage, https://github.com/selcuk/localbucket
|
|
6
|
+
Project-URL: Issues, https://github.com/selcuk/localbucket/issues
|
|
7
|
+
Author: Selcuk Ayguney
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: aws,boto3,django,emulator,mock,s3,testing
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: HTTP Servers
|
|
18
|
+
Classifier: Topic :: Software Development :: Testing
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# LocalBucket
|
|
23
|
+
|
|
24
|
+
LocalBucket is a minimal S3-compatible HTTP server that stores buckets as
|
|
25
|
+
folders on a local directory. It is a single Python script for Python 3.10
|
|
26
|
+
and newer, with no dependencies beyond the standard library, meant for
|
|
27
|
+
local development and testing against S3 clients such as boto3 and the AWS
|
|
28
|
+
CLI.
|
|
29
|
+
|
|
30
|
+
Each bucket is a folder under the data directory and each object is a
|
|
31
|
+
regular file, so you can inspect and change the stored data with ordinary
|
|
32
|
+
file tools.
|
|
33
|
+
|
|
34
|
+
## Installation
|
|
35
|
+
|
|
36
|
+
```sh
|
|
37
|
+
uv tool install localbucket # or: pipx install localbucket
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
You can also run it without installing, with `uvx localbucket ./data`, or
|
|
41
|
+
copy `localbucket.py` anywhere and run it with `python3 localbucket.py`.
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
```sh
|
|
46
|
+
localbucket ./data
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
This serves `./data` on `http://127.0.0.1:9000`. Use `--host` and `--port`
|
|
50
|
+
to change the address.
|
|
51
|
+
|
|
52
|
+
Point a client at the server with path-style addressing. Credentials are not
|
|
53
|
+
checked, so any access key and secret will do:
|
|
54
|
+
|
|
55
|
+
```sh
|
|
56
|
+
aws --endpoint-url http://127.0.0.1:9000 s3 mb s3://mybucket
|
|
57
|
+
aws --endpoint-url http://127.0.0.1:9000 s3 cp file.txt s3://mybucket/
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
import boto3
|
|
62
|
+
from botocore.config import Config
|
|
63
|
+
|
|
64
|
+
s3 = boto3.client(
|
|
65
|
+
"s3",
|
|
66
|
+
endpoint_url="http://127.0.0.1:9000",
|
|
67
|
+
aws_access_key_id="test",
|
|
68
|
+
aws_secret_access_key="test",
|
|
69
|
+
region_name="us-east-1",
|
|
70
|
+
config=Config(s3={"addressing_style": "path"}),
|
|
71
|
+
)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Supported operations
|
|
75
|
+
|
|
76
|
+
- Buckets: ListBuckets, CreateBucket, HeadBucket, and DeleteBucket (empty
|
|
77
|
+
buckets only)
|
|
78
|
+
- Objects: ListObjects, ListObjectsV2, PutObject (including
|
|
79
|
+
`If-None-Match: *`), GetObject and HeadObject (including `Range` and
|
|
80
|
+
`response-*` header overrides such as `response-content-disposition`),
|
|
81
|
+
CopyObject, DeleteObject
|
|
82
|
+
- Multipart uploads: CreateMultipartUpload, UploadPart, UploadPartCopy,
|
|
83
|
+
CompleteMultipartUpload, AbortMultipartUpload
|
|
84
|
+
- GetObjectTagging, which always returns an empty tag set because tags are
|
|
85
|
+
not stored
|
|
86
|
+
|
|
87
|
+
Requests with any other query parameter, such as ACL or tagging changes,
|
|
88
|
+
return `501 NotImplemented` and do not change anything. Virtual-hosted-style
|
|
89
|
+
URLs, authentication and versioning are not supported. Object metadata is
|
|
90
|
+
not stored either: `Content-Type` is guessed from the key's file extension.
|
|
91
|
+
|
|
92
|
+
Unfinished multipart uploads are kept in `DATA_ROOT/.localbucket-uploads`.
|
|
93
|
+
|
|
94
|
+
CORS is open: requests from any origin are allowed, including `OPTIONS`
|
|
95
|
+
preflights, so browser code on your development site can call LocalBucket
|
|
96
|
+
directly. Any web page you open can also read from it while it runs, so
|
|
97
|
+
keep it on `127.0.0.1` and don't store anything sensitive in it.
|
|
98
|
+
|
|
99
|
+
## Testing
|
|
100
|
+
|
|
101
|
+
The test suite starts LocalBucket on a temporary directory and checks its
|
|
102
|
+
behaviour with boto3. It is a [uv](https://docs.astral.sh/uv/) script that
|
|
103
|
+
declares its own dependencies:
|
|
104
|
+
|
|
105
|
+
```sh
|
|
106
|
+
uv run test_localbucket.py
|
|
107
|
+
```
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
localbucket.py,sha256=ZHvaI12CKY3iNrdlj_W115BZWUzWt6h7tG0cjRYYIcc,47107
|
|
2
|
+
localbucket-1.2.1.dist-info/METADATA,sha256=5-UQJ2FVuoyqY-i5bLmebSHtmd5epj560646L58gxyw,3716
|
|
3
|
+
localbucket-1.2.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
4
|
+
localbucket-1.2.1.dist-info/entry_points.txt,sha256=94lO3GSUIS8-mbMq4Qo2B75lqhxNvrB1JD5NGmycxPk,49
|
|
5
|
+
localbucket-1.2.1.dist-info/licenses/LICENSE,sha256=bXAkS--71fEU6n463k4j8cCf4xunLiI0qoYzftanVRU,1071
|
|
6
|
+
localbucket-1.2.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Selcuk Ayguney
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
localbucket.py
ADDED
|
@@ -0,0 +1,1202 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""A minimal HTTP server that imitates part of the AWS S3 API on a local directory."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import base64
|
|
8
|
+
import hashlib
|
|
9
|
+
import io
|
|
10
|
+
import json
|
|
11
|
+
import logging
|
|
12
|
+
import mimetypes
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import shutil
|
|
16
|
+
import sys
|
|
17
|
+
import tempfile
|
|
18
|
+
import threading
|
|
19
|
+
import time
|
|
20
|
+
import uuid
|
|
21
|
+
import xml.etree.ElementTree as ET
|
|
22
|
+
from collections.abc import Set as AbstractSet
|
|
23
|
+
from datetime import datetime, timezone
|
|
24
|
+
from email.utils import formatdate, parsedate_to_datetime
|
|
25
|
+
from http import HTTPStatus
|
|
26
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from urllib.parse import parse_qs, quote, unquote, urlsplit
|
|
29
|
+
from xml.sax.saxutils import escape
|
|
30
|
+
|
|
31
|
+
__version__ = "1.2.1"
|
|
32
|
+
BUCKET_NAME_RE = re.compile(r"^[a-z0-9][a-z0-9.-]{1,61}[a-z0-9]$")
|
|
33
|
+
S3_NAMESPACE = "http://s3.amazonaws.com/doc/2006-03-01/"
|
|
34
|
+
TEMP_PREFIX = ".localbucket-"
|
|
35
|
+
MAX_KEYS_LIMIT = 1000
|
|
36
|
+
UPLOADS_DIR = ".localbucket-uploads"
|
|
37
|
+
UPLOAD_ID_RE = re.compile(r"^[0-9a-f]{32}$")
|
|
38
|
+
UPLOAD_META = "upload.json"
|
|
39
|
+
MIN_PART_SIZE = 5 * 1024 * 1024
|
|
40
|
+
MAX_PART_NUMBER = 10000
|
|
41
|
+
BLOCK_SIZE = 1024 * 1024
|
|
42
|
+
RANGE_RE = re.compile(r"^bytes=(\d*)-(\d*)$")
|
|
43
|
+
ALWAYS_ALLOWED_QUERY = {
|
|
44
|
+
"x-id",
|
|
45
|
+
"X-Amz-Algorithm",
|
|
46
|
+
"X-Amz-Credential",
|
|
47
|
+
"X-Amz-Date",
|
|
48
|
+
"X-Amz-Expires",
|
|
49
|
+
"X-Amz-SignedHeaders",
|
|
50
|
+
"X-Amz-Signature",
|
|
51
|
+
"X-Amz-Security-Token",
|
|
52
|
+
"AWSAccessKeyId",
|
|
53
|
+
"Signature",
|
|
54
|
+
"Expires",
|
|
55
|
+
}
|
|
56
|
+
LIST_OBJECTS_QUERY = {"prefix", "delimiter", "marker", "max-keys", "encoding-type"}
|
|
57
|
+
LIST_OBJECTS_V2_QUERY = {
|
|
58
|
+
"list-type",
|
|
59
|
+
"prefix",
|
|
60
|
+
"delimiter",
|
|
61
|
+
"max-keys",
|
|
62
|
+
"start-after",
|
|
63
|
+
"continuation-token",
|
|
64
|
+
"encoding-type",
|
|
65
|
+
}
|
|
66
|
+
logger = logging.getLogger("localbucket")
|
|
67
|
+
RESPONSE_HEADER_OVERRIDES = {
|
|
68
|
+
"response-cache-control": "Cache-Control",
|
|
69
|
+
"response-content-disposition": "Content-Disposition",
|
|
70
|
+
"response-content-encoding": "Content-Encoding",
|
|
71
|
+
"response-content-language": "Content-Language",
|
|
72
|
+
"response-content-type": "Content-Type",
|
|
73
|
+
"response-expires": "Expires",
|
|
74
|
+
}
|
|
75
|
+
CORS_EXPOSE_HEADERS = (
|
|
76
|
+
"Accept-Ranges, Content-Disposition, Content-Encoding, Content-Range, ETag, "
|
|
77
|
+
"x-amz-bucket-region, x-amz-request-id"
|
|
78
|
+
)
|
|
79
|
+
BANNER = r"""
|
|
80
|
+
__ ______ __ __
|
|
81
|
+
/ / ____ _________ _/ / __ )__ _______/ /_____ / /_
|
|
82
|
+
/ / / __ \/ ___/ __ `/ / __ / / / / ___/ //_/ _ \/ __/
|
|
83
|
+
/ /___/ /_/ / /__/ /_/ / / /_/ / /_/ / /__/ ,< / __/ /_
|
|
84
|
+
/_____/\____/\___/\__,_/_/_____/\__,_/\___/_/|_|\___/\__/
|
|
85
|
+
"""
|
|
86
|
+
HELP_EPILOG = """\
|
|
87
|
+
supported operations (path-style URLs only, credentials are not checked):
|
|
88
|
+
ListBuckets, CreateBucket, DeleteBucket (empty buckets only), HeadBucket,
|
|
89
|
+
ListObjects, ListObjectsV2, PutObject (with If-None-Match: *), CopyObject,
|
|
90
|
+
GetObject and HeadObject (with Range and response-* header overrides),
|
|
91
|
+
DeleteObject, CreateMultipartUpload, UploadPart, UploadPartCopy,
|
|
92
|
+
CompleteMultipartUpload, AbortMultipartUpload,
|
|
93
|
+
GetObjectTagging (always an empty tag set, because tags are not stored)
|
|
94
|
+
CORS requests from any origin are allowed, including OPTIONS preflights.
|
|
95
|
+
Unfinished multipart uploads are kept in DATA_ROOT/.localbucket-uploads.
|
|
96
|
+
|
|
97
|
+
example:
|
|
98
|
+
localbucket ./data
|
|
99
|
+
aws --endpoint-url http://127.0.0.1:9000 s3 cp file.txt s3://mybucket/
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class S3Error(Exception):
|
|
104
|
+
"""An S3-style error that is returned to the client as an XML document."""
|
|
105
|
+
|
|
106
|
+
def __init__(self, status: HTTPStatus, code: str, message: str):
|
|
107
|
+
super().__init__(message)
|
|
108
|
+
self.status = status
|
|
109
|
+
self.code = code
|
|
110
|
+
self.message = message
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def read_chunked(stream) -> bytes:
|
|
114
|
+
"""Decode an HTTP chunked or aws-chunked body, dropping signatures and trailers."""
|
|
115
|
+
body = bytearray()
|
|
116
|
+
while True:
|
|
117
|
+
line = stream.readline()
|
|
118
|
+
if not line:
|
|
119
|
+
raise S3Error(
|
|
120
|
+
HTTPStatus.BAD_REQUEST,
|
|
121
|
+
"IncompleteBody",
|
|
122
|
+
"Chunked body ended unexpectedly.",
|
|
123
|
+
)
|
|
124
|
+
size_text = line.split(b";", 1)[0].strip()
|
|
125
|
+
try:
|
|
126
|
+
size = int(size_text, 16)
|
|
127
|
+
except ValueError:
|
|
128
|
+
raise S3Error(
|
|
129
|
+
HTTPStatus.BAD_REQUEST, "InvalidRequest", "Invalid chunk size."
|
|
130
|
+
) from None
|
|
131
|
+
if size == 0:
|
|
132
|
+
while stream.readline().strip():
|
|
133
|
+
pass
|
|
134
|
+
return bytes(body)
|
|
135
|
+
chunk = stream.read(size)
|
|
136
|
+
if len(chunk) != size:
|
|
137
|
+
raise S3Error(
|
|
138
|
+
HTTPStatus.BAD_REQUEST,
|
|
139
|
+
"IncompleteBody",
|
|
140
|
+
"Chunk is shorter than declared.",
|
|
141
|
+
)
|
|
142
|
+
body += chunk
|
|
143
|
+
stream.readline()
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def iso_timestamp(epoch: float) -> str:
|
|
147
|
+
"""Format a Unix timestamp the way S3 formats dates in XML responses."""
|
|
148
|
+
return datetime.fromtimestamp(epoch, timezone.utc).strftime(
|
|
149
|
+
"%Y-%m-%dT%H:%M:%S.000Z"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
_etag_cache: dict[str, tuple[tuple[int, int, int, int], str]] = {}
|
|
154
|
+
_etag_lock = threading.Lock()
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def stat_signature(path: Path) -> tuple[int, int, int, int]:
|
|
158
|
+
"""Return a file's identity, size and mtime, which change when it is rewritten."""
|
|
159
|
+
stat = path.stat()
|
|
160
|
+
return stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def object_etag(path: Path) -> str:
|
|
164
|
+
"""Return a file's MD5 ETag, reusing the cached value while it is unchanged."""
|
|
165
|
+
signature = stat_signature(path)
|
|
166
|
+
with _etag_lock:
|
|
167
|
+
cached = _etag_cache.get(str(path))
|
|
168
|
+
if cached and cached[0] == signature:
|
|
169
|
+
return cached[1]
|
|
170
|
+
digest = hashlib.md5()
|
|
171
|
+
with path.open("rb") as file:
|
|
172
|
+
while block := file.read(1024 * 1024):
|
|
173
|
+
digest.update(block)
|
|
174
|
+
etag = digest.hexdigest()
|
|
175
|
+
with _etag_lock:
|
|
176
|
+
_etag_cache[str(path)] = (signature, etag)
|
|
177
|
+
return etag
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def remember_etag(path: Path, etag: str):
|
|
181
|
+
"""Cache the ETag of a file that has just been written."""
|
|
182
|
+
signature = stat_signature(path)
|
|
183
|
+
with _etag_lock:
|
|
184
|
+
_etag_cache[str(path)] = (signature, etag)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def etag_matches(header: str, etag: str) -> bool:
|
|
188
|
+
"""Return whether an If-Match style header lists the given ETag or the wildcard."""
|
|
189
|
+
candidates = {
|
|
190
|
+
value.strip().removeprefix("W/").strip('"') for value in header.split(",")
|
|
191
|
+
}
|
|
192
|
+
return "*" in candidates or etag in candidates
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def http_date(header: str | None) -> int | None:
|
|
196
|
+
"""Parse an HTTP date header to a Unix timestamp, or None if missing or invalid."""
|
|
197
|
+
if not header:
|
|
198
|
+
return None
|
|
199
|
+
try:
|
|
200
|
+
return int(parsedate_to_datetime(header).timestamp())
|
|
201
|
+
except (TypeError, ValueError, IndexError):
|
|
202
|
+
return None
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def write_bytes(file, data: bytes) -> str:
|
|
206
|
+
"""Write data to a file and return its hex MD5 digest."""
|
|
207
|
+
file.write(data)
|
|
208
|
+
return hashlib.md5(data).hexdigest()
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def copy_ranges(destination, ranges: list[tuple[Path, int, int]]) -> str:
|
|
212
|
+
"""Write (path, start, length) file ranges to destination; return their hex MD5."""
|
|
213
|
+
digest = hashlib.md5()
|
|
214
|
+
for source, start, length in ranges:
|
|
215
|
+
with source.open("rb") as file:
|
|
216
|
+
file.seek(start)
|
|
217
|
+
remaining = length
|
|
218
|
+
while remaining > 0:
|
|
219
|
+
block = file.read(min(remaining, BLOCK_SIZE))
|
|
220
|
+
if not block:
|
|
221
|
+
raise S3Error(
|
|
222
|
+
HTTPStatus.INTERNAL_SERVER_ERROR,
|
|
223
|
+
"InternalError",
|
|
224
|
+
"A source file changed while it was being copied.",
|
|
225
|
+
)
|
|
226
|
+
destination.write(block)
|
|
227
|
+
digest.update(block)
|
|
228
|
+
remaining -= len(block)
|
|
229
|
+
return digest.hexdigest()
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def parse_range(header: str, size: int) -> tuple[int, int] | None:
|
|
233
|
+
"""Return the Range header's inclusive byte range, or None for the whole object."""
|
|
234
|
+
match = RANGE_RE.match(header.strip())
|
|
235
|
+
if not match or match.groups() == ("", ""):
|
|
236
|
+
return None
|
|
237
|
+
first, last = match.groups()
|
|
238
|
+
if first:
|
|
239
|
+
start = int(first)
|
|
240
|
+
if last and int(last) < start:
|
|
241
|
+
return None
|
|
242
|
+
end = min(int(last), size - 1) if last else size - 1
|
|
243
|
+
else:
|
|
244
|
+
length = int(last)
|
|
245
|
+
if length == 0:
|
|
246
|
+
raise invalid_range_error()
|
|
247
|
+
start, end = max(size - length, 0), size - 1
|
|
248
|
+
if start >= size:
|
|
249
|
+
raise invalid_range_error()
|
|
250
|
+
return start, end
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def invalid_range_error() -> S3Error:
|
|
254
|
+
"""Build the error for a Range header that starts past the end of the object."""
|
|
255
|
+
return S3Error(
|
|
256
|
+
HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE,
|
|
257
|
+
"InvalidRange",
|
|
258
|
+
"The requested range is not satisfiable",
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def sub_element(parent: ET.Element, tag: str, text: str) -> ET.Element:
|
|
263
|
+
"""Add a child element containing the given text."""
|
|
264
|
+
child = ET.SubElement(parent, tag)
|
|
265
|
+
child.text = text
|
|
266
|
+
return child
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def bucket_dirs(data_root: Path) -> list[Path]:
|
|
270
|
+
"""Return the bucket directories under the data root, sorted by name."""
|
|
271
|
+
return sorted(
|
|
272
|
+
(
|
|
273
|
+
entry
|
|
274
|
+
for entry in data_root.iterdir()
|
|
275
|
+
if entry.is_dir() and BUCKET_NAME_RE.match(entry.name)
|
|
276
|
+
),
|
|
277
|
+
key=lambda p: p.name,
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def log_buckets(data_root: Path):
|
|
282
|
+
"""Log the buckets under the data root as a comma-separated list."""
|
|
283
|
+
buckets = [entry.name for entry in bucket_dirs(data_root)]
|
|
284
|
+
logger.info("Buckets: %s", ", ".join(buckets) or "none")
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
class S3Handler(BaseHTTPRequestHandler):
|
|
288
|
+
"""Handles path-style S3 requests for buckets and objects."""
|
|
289
|
+
|
|
290
|
+
server_version = "LocalBucket"
|
|
291
|
+
protocol_version = "HTTP/1.1"
|
|
292
|
+
data_root: Path
|
|
293
|
+
query: dict[str, str]
|
|
294
|
+
|
|
295
|
+
def log_message(self, format: str, *args):
|
|
296
|
+
"""Send a request log line to the localbucket logger."""
|
|
297
|
+
message = format % args
|
|
298
|
+
control_chars = getattr(self, "_control_char_table", None)
|
|
299
|
+
if control_chars:
|
|
300
|
+
message = message.translate(control_chars)
|
|
301
|
+
logger.info("%s %s", self.address_string(), message)
|
|
302
|
+
|
|
303
|
+
def do_PUT(self):
|
|
304
|
+
"""Create a bucket or write an object."""
|
|
305
|
+
self._dispatch(self._put)
|
|
306
|
+
|
|
307
|
+
def do_GET(self):
|
|
308
|
+
"""List buckets, list a bucket's objects or read an object."""
|
|
309
|
+
self._dispatch(self._get_request)
|
|
310
|
+
|
|
311
|
+
def do_HEAD(self):
|
|
312
|
+
"""Check a bucket exists or return an object's metadata without its body."""
|
|
313
|
+
self._dispatch(self._head_request)
|
|
314
|
+
|
|
315
|
+
def do_POST(self):
|
|
316
|
+
"""Start or complete a multipart upload."""
|
|
317
|
+
self._dispatch(self._post_request)
|
|
318
|
+
|
|
319
|
+
def do_DELETE(self):
|
|
320
|
+
"""Delete an object or abort a multipart upload."""
|
|
321
|
+
self._dispatch(self._delete_request)
|
|
322
|
+
|
|
323
|
+
def do_OPTIONS(self):
|
|
324
|
+
"""Answer a CORS preflight request, allowing any origin, method and header."""
|
|
325
|
+
headers = {
|
|
326
|
+
"Access-Control-Allow-Methods": "GET, HEAD, PUT, POST, DELETE",
|
|
327
|
+
"Access-Control-Max-Age": "3000",
|
|
328
|
+
}
|
|
329
|
+
requested = self.headers.get("Access-Control-Request-Headers")
|
|
330
|
+
if requested:
|
|
331
|
+
headers["Access-Control-Allow-Headers"] = requested
|
|
332
|
+
self._send(HTTPStatus.OK, headers=headers)
|
|
333
|
+
|
|
334
|
+
def _dispatch(self, action):
|
|
335
|
+
"""Parse bucket, key and query from the URL, run it and send errors as XML."""
|
|
336
|
+
try:
|
|
337
|
+
url = urlsplit(self.path)
|
|
338
|
+
path = unquote(url.path)
|
|
339
|
+
bucket, _, key = path.lstrip("/").partition("/")
|
|
340
|
+
parsed = parse_qs(url.query, keep_blank_values=True)
|
|
341
|
+
self.query = {name: values[0] for name, values in parsed.items()}
|
|
342
|
+
action(bucket, key)
|
|
343
|
+
except S3Error as error:
|
|
344
|
+
self._send_error(error)
|
|
345
|
+
|
|
346
|
+
def _get_request(self, bucket: str, key: str):
|
|
347
|
+
"""Route a GET request to ListBuckets, ListObjects(V2) or GetObject."""
|
|
348
|
+
if not bucket:
|
|
349
|
+
self._check_query("ListBuckets")
|
|
350
|
+
self._list_buckets()
|
|
351
|
+
elif not key:
|
|
352
|
+
if self.query.get("list-type") == "2":
|
|
353
|
+
self._check_query("ListObjectsV2", LIST_OBJECTS_V2_QUERY)
|
|
354
|
+
self._list_objects_v2(bucket)
|
|
355
|
+
else:
|
|
356
|
+
self._check_query("ListObjects", LIST_OBJECTS_QUERY)
|
|
357
|
+
self._list_objects(bucket)
|
|
358
|
+
elif "tagging" in self.query:
|
|
359
|
+
self._check_query("GetObjectTagging", {"tagging"})
|
|
360
|
+
if not self._object_path(bucket, key).is_file():
|
|
361
|
+
raise S3Error(
|
|
362
|
+
HTTPStatus.NOT_FOUND,
|
|
363
|
+
"NoSuchKey",
|
|
364
|
+
"The specified key does not exist.",
|
|
365
|
+
)
|
|
366
|
+
root = ET.Element("Tagging", xmlns=S3_NAMESPACE)
|
|
367
|
+
ET.SubElement(root, "TagSet")
|
|
368
|
+
self._send_xml(root)
|
|
369
|
+
else:
|
|
370
|
+
self._check_query("GetObject", RESPONSE_HEADER_OVERRIDES.keys())
|
|
371
|
+
self._get(bucket, key, include_body=True)
|
|
372
|
+
|
|
373
|
+
def _head_request(self, bucket: str, key: str):
|
|
374
|
+
"""Route a HEAD request to HeadBucket or HeadObject."""
|
|
375
|
+
if not bucket:
|
|
376
|
+
raise self._not_implemented_error()
|
|
377
|
+
if not key:
|
|
378
|
+
self._check_query("HeadBucket")
|
|
379
|
+
self._bucket_dir(bucket)
|
|
380
|
+
self._send(HTTPStatus.OK, headers={"x-amz-bucket-region": "us-east-1"})
|
|
381
|
+
else:
|
|
382
|
+
self._check_query("HeadObject", RESPONSE_HEADER_OVERRIDES.keys())
|
|
383
|
+
self._get(bucket, key, include_body=False)
|
|
384
|
+
|
|
385
|
+
def _delete_request(self, bucket: str, key: str):
|
|
386
|
+
"""Route DELETE to DeleteBucket, DeleteObject or AbortMultipartUpload."""
|
|
387
|
+
if not bucket:
|
|
388
|
+
raise self._not_implemented_error()
|
|
389
|
+
if not key:
|
|
390
|
+
self._check_query("DeleteBucket")
|
|
391
|
+
self._delete_bucket(bucket)
|
|
392
|
+
elif "uploadId" in self.query:
|
|
393
|
+
self._check_query("AbortMultipartUpload", {"uploadId"})
|
|
394
|
+
shutil.rmtree(self._upload_dir(bucket, key), ignore_errors=True)
|
|
395
|
+
self._send(HTTPStatus.NO_CONTENT)
|
|
396
|
+
else:
|
|
397
|
+
self._check_query("DeleteObject")
|
|
398
|
+
self._delete_object(bucket, key)
|
|
399
|
+
|
|
400
|
+
def _post_request(self, bucket: str, key: str):
|
|
401
|
+
"""Route a POST request to CreateMultipartUpload or CompleteMultipartUpload."""
|
|
402
|
+
if not bucket or not key:
|
|
403
|
+
raise self._not_implemented_error()
|
|
404
|
+
if "uploads" in self.query:
|
|
405
|
+
self._check_query("CreateMultipartUpload", {"uploads"})
|
|
406
|
+
self._read_body()
|
|
407
|
+
self._create_multipart_upload(bucket, key)
|
|
408
|
+
elif "uploadId" in self.query:
|
|
409
|
+
self._check_query("CompleteMultipartUpload", {"uploadId"})
|
|
410
|
+
self._complete_multipart_upload(bucket, key, self._read_body())
|
|
411
|
+
else:
|
|
412
|
+
raise self._not_implemented_error()
|
|
413
|
+
|
|
414
|
+
def _check_query(self, operation: str, allowed: AbstractSet[str] = frozenset()):
|
|
415
|
+
"""Raise NotImplemented for query parameters the operation does not support."""
|
|
416
|
+
unsupported = sorted(set(self.query) - ALWAYS_ALLOWED_QUERY - allowed)
|
|
417
|
+
if unsupported:
|
|
418
|
+
self.log_message(
|
|
419
|
+
"rejected %s with unsupported query parameters: %s",
|
|
420
|
+
operation,
|
|
421
|
+
", ".join(unsupported),
|
|
422
|
+
)
|
|
423
|
+
raise self._not_implemented_error()
|
|
424
|
+
|
|
425
|
+
def _list_buckets(self):
|
|
426
|
+
"""Send the list of all buckets under the data root."""
|
|
427
|
+
root = ET.Element("ListAllMyBucketsResult", xmlns=S3_NAMESPACE)
|
|
428
|
+
owner = ET.SubElement(root, "Owner")
|
|
429
|
+
sub_element(owner, "ID", "localbucket")
|
|
430
|
+
sub_element(owner, "DisplayName", "LocalBucket")
|
|
431
|
+
buckets = ET.SubElement(root, "Buckets")
|
|
432
|
+
for entry in bucket_dirs(self.data_root):
|
|
433
|
+
stat = entry.stat()
|
|
434
|
+
created = getattr(stat, "st_birthtime", stat.st_ctime)
|
|
435
|
+
bucket = ET.SubElement(buckets, "Bucket")
|
|
436
|
+
sub_element(bucket, "Name", entry.name)
|
|
437
|
+
sub_element(bucket, "CreationDate", iso_timestamp(created))
|
|
438
|
+
self._send_xml(root)
|
|
439
|
+
|
|
440
|
+
def _list_objects(self, bucket: str):
|
|
441
|
+
"""Send one page of objects, with prefix, delimiter, marker and paging."""
|
|
442
|
+
bucket_dir = self._bucket_dir(bucket)
|
|
443
|
+
prefix = self.query.get("prefix", "")
|
|
444
|
+
delimiter = self.query.get("delimiter", "")
|
|
445
|
+
marker = self.query.get("marker", "")
|
|
446
|
+
max_keys = self._max_keys()
|
|
447
|
+
marker_is_prefix = bool(delimiter) and marker.endswith(delimiter)
|
|
448
|
+
contents, common_prefixes, truncated, last_entry = self._list_entries(
|
|
449
|
+
bucket_dir, prefix, delimiter, marker, marker_is_prefix, max_keys
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
encode = self._list_encoder()
|
|
453
|
+
root = ET.Element("ListBucketResult", xmlns=S3_NAMESPACE)
|
|
454
|
+
sub_element(root, "Name", bucket)
|
|
455
|
+
sub_element(root, "Prefix", encode(prefix))
|
|
456
|
+
sub_element(root, "Marker", encode(marker))
|
|
457
|
+
if delimiter:
|
|
458
|
+
sub_element(root, "Delimiter", encode(delimiter))
|
|
459
|
+
sub_element(root, "MaxKeys", str(max_keys))
|
|
460
|
+
sub_element(root, "IsTruncated", "true" if truncated else "false")
|
|
461
|
+
if truncated:
|
|
462
|
+
sub_element(root, "NextMarker", encode(last_entry[2:]))
|
|
463
|
+
if self.query.get("encoding-type") == "url":
|
|
464
|
+
sub_element(root, "EncodingType", "url")
|
|
465
|
+
self._add_list_entries(root, contents, common_prefixes, encode)
|
|
466
|
+
self._send_xml(root)
|
|
467
|
+
|
|
468
|
+
def _list_objects_v2(self, bucket: str):
|
|
469
|
+
"""Send one page of objects, with prefix, delimiter, start-after and paging."""
|
|
470
|
+
bucket_dir = self._bucket_dir(bucket)
|
|
471
|
+
prefix = self.query.get("prefix", "")
|
|
472
|
+
delimiter = self.query.get("delimiter", "")
|
|
473
|
+
start_after = self.query.get("start-after", "")
|
|
474
|
+
token = self.query.get("continuation-token")
|
|
475
|
+
max_keys = self._max_keys()
|
|
476
|
+
|
|
477
|
+
marker, marker_is_prefix = start_after, False
|
|
478
|
+
if token is not None:
|
|
479
|
+
try:
|
|
480
|
+
decoded = base64.urlsafe_b64decode(token.encode()).decode()
|
|
481
|
+
except ValueError:
|
|
482
|
+
decoded = ""
|
|
483
|
+
if decoded[:2] not in ("K:", "P:"):
|
|
484
|
+
raise S3Error(
|
|
485
|
+
HTTPStatus.BAD_REQUEST,
|
|
486
|
+
"InvalidArgument",
|
|
487
|
+
"The continuation token provided is incorrect.",
|
|
488
|
+
)
|
|
489
|
+
marker, marker_is_prefix = decoded[2:], decoded.startswith("P:")
|
|
490
|
+
contents, common_prefixes, truncated, last_entry = self._list_entries(
|
|
491
|
+
bucket_dir, prefix, delimiter, marker, marker_is_prefix, max_keys
|
|
492
|
+
)
|
|
493
|
+
|
|
494
|
+
encode = self._list_encoder()
|
|
495
|
+
root = ET.Element("ListBucketResult", xmlns=S3_NAMESPACE)
|
|
496
|
+
sub_element(root, "Name", bucket)
|
|
497
|
+
sub_element(root, "Prefix", encode(prefix))
|
|
498
|
+
if delimiter:
|
|
499
|
+
sub_element(root, "Delimiter", encode(delimiter))
|
|
500
|
+
sub_element(root, "MaxKeys", str(max_keys))
|
|
501
|
+
sub_element(root, "KeyCount", str(len(contents) + len(common_prefixes)))
|
|
502
|
+
sub_element(root, "IsTruncated", "true" if truncated else "false")
|
|
503
|
+
if token is not None:
|
|
504
|
+
sub_element(root, "ContinuationToken", token)
|
|
505
|
+
if truncated:
|
|
506
|
+
next_token = base64.urlsafe_b64encode(last_entry.encode()).decode()
|
|
507
|
+
sub_element(root, "NextContinuationToken", next_token)
|
|
508
|
+
if start_after:
|
|
509
|
+
sub_element(root, "StartAfter", encode(start_after))
|
|
510
|
+
if self.query.get("encoding-type") == "url":
|
|
511
|
+
sub_element(root, "EncodingType", "url")
|
|
512
|
+
self._add_list_entries(root, contents, common_prefixes, encode)
|
|
513
|
+
self._send_xml(root)
|
|
514
|
+
|
|
515
|
+
def _max_keys(self) -> int:
|
|
516
|
+
"""Return the max-keys query parameter, capped at the S3 limit of 1000."""
|
|
517
|
+
try:
|
|
518
|
+
max_keys = min(
|
|
519
|
+
int(self.query.get("max-keys", MAX_KEYS_LIMIT)), MAX_KEYS_LIMIT
|
|
520
|
+
)
|
|
521
|
+
except ValueError:
|
|
522
|
+
max_keys = -1
|
|
523
|
+
if max_keys < 0:
|
|
524
|
+
raise S3Error(
|
|
525
|
+
HTTPStatus.BAD_REQUEST,
|
|
526
|
+
"InvalidArgument",
|
|
527
|
+
"Provided max-keys not an integer or within integer range.",
|
|
528
|
+
)
|
|
529
|
+
return max_keys
|
|
530
|
+
|
|
531
|
+
def _list_encoder(self):
|
|
532
|
+
"""Return a function that URL-encodes listed names if encoding-type=url."""
|
|
533
|
+
if self.query.get("encoding-type") == "url":
|
|
534
|
+
return lambda text: quote(text, safe="/")
|
|
535
|
+
return lambda text: text
|
|
536
|
+
|
|
537
|
+
def _list_entries(
|
|
538
|
+
self,
|
|
539
|
+
bucket_dir: Path,
|
|
540
|
+
prefix: str,
|
|
541
|
+
delimiter: str,
|
|
542
|
+
marker: str,
|
|
543
|
+
marker_is_prefix: bool,
|
|
544
|
+
max_keys: int,
|
|
545
|
+
) -> tuple[list[tuple[str, Path]], list[str], bool, str]:
|
|
546
|
+
"""Return one page of keys and common prefixes that come after the marker."""
|
|
547
|
+
contents: list[tuple[str, Path]] = []
|
|
548
|
+
common_prefixes: list[str] = []
|
|
549
|
+
last_entry = ""
|
|
550
|
+
truncated = False
|
|
551
|
+
keys = []
|
|
552
|
+
for directory, _, files in os.walk(bucket_dir):
|
|
553
|
+
for name in files:
|
|
554
|
+
if name.startswith(TEMP_PREFIX):
|
|
555
|
+
continue
|
|
556
|
+
path = Path(directory, name)
|
|
557
|
+
keys.append((path.relative_to(bucket_dir).as_posix(), path))
|
|
558
|
+
keys.sort(key=lambda item: item[0].encode())
|
|
559
|
+
for key, path in keys:
|
|
560
|
+
if not key.startswith(prefix):
|
|
561
|
+
continue
|
|
562
|
+
if marker and (
|
|
563
|
+
key.encode() <= marker.encode()
|
|
564
|
+
or (marker_is_prefix and key.startswith(marker))
|
|
565
|
+
):
|
|
566
|
+
continue
|
|
567
|
+
common = ""
|
|
568
|
+
if delimiter:
|
|
569
|
+
index = key.find(delimiter, len(prefix))
|
|
570
|
+
if index >= 0:
|
|
571
|
+
common = key[: index + len(delimiter)]
|
|
572
|
+
if common and common_prefixes and common_prefixes[-1] == common:
|
|
573
|
+
continue
|
|
574
|
+
if len(contents) + len(common_prefixes) >= max_keys:
|
|
575
|
+
truncated = max_keys > 0
|
|
576
|
+
break
|
|
577
|
+
if common:
|
|
578
|
+
common_prefixes.append(common)
|
|
579
|
+
last_entry = "P:" + common
|
|
580
|
+
else:
|
|
581
|
+
contents.append((key, path))
|
|
582
|
+
last_entry = "K:" + key
|
|
583
|
+
return contents, common_prefixes, truncated, last_entry
|
|
584
|
+
|
|
585
|
+
def _add_list_entries(
|
|
586
|
+
self,
|
|
587
|
+
root: ET.Element,
|
|
588
|
+
contents: list[tuple[str, Path]],
|
|
589
|
+
common_prefixes: list[str],
|
|
590
|
+
encode,
|
|
591
|
+
):
|
|
592
|
+
"""Add Contents and CommonPrefixes elements to a listing result."""
|
|
593
|
+
for key, path in contents:
|
|
594
|
+
item = ET.SubElement(root, "Contents")
|
|
595
|
+
stat = path.stat()
|
|
596
|
+
sub_element(item, "Key", encode(key))
|
|
597
|
+
sub_element(item, "LastModified", iso_timestamp(stat.st_mtime))
|
|
598
|
+
sub_element(item, "ETag", f'"{object_etag(path)}"')
|
|
599
|
+
sub_element(item, "Size", str(stat.st_size))
|
|
600
|
+
sub_element(item, "StorageClass", "STANDARD")
|
|
601
|
+
for common in common_prefixes:
|
|
602
|
+
item = ET.SubElement(root, "CommonPrefixes")
|
|
603
|
+
sub_element(item, "Prefix", encode(common))
|
|
604
|
+
|
|
605
|
+
def _delete_object(self, bucket: str, key: str):
|
|
606
|
+
"""Delete an object if it exists and remove any folders left empty."""
|
|
607
|
+
bucket_dir = self._bucket_dir(bucket).resolve()
|
|
608
|
+
target = self._object_path(bucket, key)
|
|
609
|
+
if target.is_file():
|
|
610
|
+
target.unlink(missing_ok=True)
|
|
611
|
+
parent = target.parent
|
|
612
|
+
while parent != bucket_dir:
|
|
613
|
+
try:
|
|
614
|
+
parent.rmdir()
|
|
615
|
+
except OSError:
|
|
616
|
+
break
|
|
617
|
+
parent = parent.parent
|
|
618
|
+
self._send(HTTPStatus.NO_CONTENT)
|
|
619
|
+
|
|
620
|
+
def _delete_bucket(self, bucket: str):
|
|
621
|
+
"""Delete an empty bucket, or raise BucketNotEmpty if it holds any files."""
|
|
622
|
+
bucket_dir = self._bucket_dir(bucket)
|
|
623
|
+
not_empty = S3Error(
|
|
624
|
+
HTTPStatus.CONFLICT,
|
|
625
|
+
"BucketNotEmpty",
|
|
626
|
+
"The bucket you tried to delete is not empty",
|
|
627
|
+
)
|
|
628
|
+
directories = []
|
|
629
|
+
for directory, _, files in os.walk(bucket_dir, topdown=False):
|
|
630
|
+
if any(not name.startswith(TEMP_PREFIX) for name in files):
|
|
631
|
+
raise not_empty
|
|
632
|
+
directories.append(directory)
|
|
633
|
+
for directory in directories:
|
|
634
|
+
try:
|
|
635
|
+
os.rmdir(directory)
|
|
636
|
+
except OSError:
|
|
637
|
+
raise not_empty from None
|
|
638
|
+
self._delete_bucket_uploads(bucket)
|
|
639
|
+
self._send(HTTPStatus.NO_CONTENT)
|
|
640
|
+
log_buckets(self.data_root)
|
|
641
|
+
|
|
642
|
+
def _delete_bucket_uploads(self, bucket: str):
|
|
643
|
+
"""Remove the unfinished multipart uploads that belong to a bucket."""
|
|
644
|
+
uploads_root = self.data_root / UPLOADS_DIR
|
|
645
|
+
if not uploads_root.is_dir():
|
|
646
|
+
return
|
|
647
|
+
for upload_dir in uploads_root.iterdir():
|
|
648
|
+
try:
|
|
649
|
+
meta = json.loads((upload_dir / UPLOAD_META).read_text())
|
|
650
|
+
except (OSError, ValueError):
|
|
651
|
+
continue
|
|
652
|
+
if isinstance(meta, dict) and meta.get("bucket") == bucket:
|
|
653
|
+
shutil.rmtree(upload_dir, ignore_errors=True)
|
|
654
|
+
|
|
655
|
+
def _put(self, bucket: str, key: str):
|
|
656
|
+
"""Route PUT to CreateBucket, PutObject, CopyObject or UploadPart(Copy)."""
|
|
657
|
+
if not bucket:
|
|
658
|
+
raise self._not_implemented_error()
|
|
659
|
+
copy_source = "x-amz-copy-source" in self.headers
|
|
660
|
+
if key and ("uploadId" in self.query or "partNumber" in self.query):
|
|
661
|
+
self._check_query(
|
|
662
|
+
"UploadPartCopy" if copy_source else "UploadPart",
|
|
663
|
+
{"uploadId", "partNumber"},
|
|
664
|
+
)
|
|
665
|
+
body = self._read_body()
|
|
666
|
+
if copy_source:
|
|
667
|
+
self._upload_part_copy(bucket, key)
|
|
668
|
+
else:
|
|
669
|
+
etag = self._store_part(
|
|
670
|
+
bucket, key, lambda file: write_bytes(file, body)
|
|
671
|
+
)
|
|
672
|
+
self._send(HTTPStatus.OK, headers={"ETag": f'"{etag}"'})
|
|
673
|
+
return
|
|
674
|
+
self._check_query(
|
|
675
|
+
("CopyObject" if copy_source else "PutObject") if key else "CreateBucket"
|
|
676
|
+
)
|
|
677
|
+
body = self._read_body()
|
|
678
|
+
if not key:
|
|
679
|
+
self._create_bucket(bucket)
|
|
680
|
+
elif copy_source:
|
|
681
|
+
self._copy_object(bucket, key)
|
|
682
|
+
else:
|
|
683
|
+
etag = self._write_object(bucket, key, lambda file: write_bytes(file, body))
|
|
684
|
+
self._send(HTTPStatus.OK, headers={"ETag": f'"{etag}"'})
|
|
685
|
+
|
|
686
|
+
def _create_bucket(self, bucket: str):
|
|
687
|
+
"""Create the bucket directory under the data root."""
|
|
688
|
+
if not BUCKET_NAME_RE.match(bucket) or ".." in bucket:
|
|
689
|
+
raise S3Error(
|
|
690
|
+
HTTPStatus.BAD_REQUEST,
|
|
691
|
+
"InvalidBucketName",
|
|
692
|
+
"The specified bucket is not valid.",
|
|
693
|
+
)
|
|
694
|
+
bucket_dir = self.data_root / bucket
|
|
695
|
+
try:
|
|
696
|
+
bucket_dir.mkdir()
|
|
697
|
+
except FileExistsError:
|
|
698
|
+
raise S3Error(
|
|
699
|
+
HTTPStatus.CONFLICT,
|
|
700
|
+
"BucketAlreadyOwnedByYou",
|
|
701
|
+
"Your previous request to create the named bucket succeeded and you "
|
|
702
|
+
"already own it.",
|
|
703
|
+
) from None
|
|
704
|
+
self._send(HTTPStatus.OK, headers={"Location": f"/{bucket}"})
|
|
705
|
+
log_buckets(self.data_root)
|
|
706
|
+
|
|
707
|
+
def _copy_object(self, bucket: str, key: str):
|
|
708
|
+
"""Copy the object named in the x-amz-copy-source header to this key."""
|
|
709
|
+
source = self._copy_source()
|
|
710
|
+
target = self._object_path(bucket, key)
|
|
711
|
+
directive = self.headers.get("x-amz-metadata-directive", "COPY").strip().upper()
|
|
712
|
+
if source == target and directive != "REPLACE":
|
|
713
|
+
raise S3Error(
|
|
714
|
+
HTTPStatus.BAD_REQUEST,
|
|
715
|
+
"InvalidRequest",
|
|
716
|
+
"This copy request is illegal because it is trying to copy an object "
|
|
717
|
+
"to itself without changing the object's metadata, storage class, "
|
|
718
|
+
"website redirect location or encryption attributes.",
|
|
719
|
+
)
|
|
720
|
+
ranges = [(source, 0, source.stat().st_size)]
|
|
721
|
+
etag = self._write_object(bucket, key, lambda file: copy_ranges(file, ranges))
|
|
722
|
+
root = ET.Element("CopyObjectResult", xmlns=S3_NAMESPACE)
|
|
723
|
+
sub_element(root, "LastModified", iso_timestamp(target.stat().st_mtime))
|
|
724
|
+
sub_element(root, "ETag", f'"{etag}"')
|
|
725
|
+
self._send_xml(root)
|
|
726
|
+
|
|
727
|
+
def _copy_source(self) -> Path:
|
|
728
|
+
"""Return the file named by x-amz-copy-source, rejecting unsupported options."""
|
|
729
|
+
source, _, source_query = self.headers["x-amz-copy-source"].partition("?")
|
|
730
|
+
if source_query:
|
|
731
|
+
raise self._not_implemented_error()
|
|
732
|
+
bucket, _, key = unquote(source).lstrip("/").partition("/")
|
|
733
|
+
if not bucket or not key:
|
|
734
|
+
raise S3Error(
|
|
735
|
+
HTTPStatus.BAD_REQUEST,
|
|
736
|
+
"InvalidArgument",
|
|
737
|
+
"Copy Source must mention the source bucket and key: "
|
|
738
|
+
"sourcebucket/sourcekey",
|
|
739
|
+
)
|
|
740
|
+
path = self._object_path(bucket, key)
|
|
741
|
+
if not path.is_file():
|
|
742
|
+
raise S3Error(
|
|
743
|
+
HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
|
|
744
|
+
)
|
|
745
|
+
self._check_copy_conditions(path)
|
|
746
|
+
return path
|
|
747
|
+
|
|
748
|
+
def _check_copy_conditions(self, source: Path):
|
|
749
|
+
"""Raise PreconditionFailed unless the x-amz-copy-source-if-* headers hold."""
|
|
750
|
+
if_match = self.headers.get("x-amz-copy-source-if-match")
|
|
751
|
+
if_none_match = self.headers.get("x-amz-copy-source-if-none-match")
|
|
752
|
+
modified_since = http_date(
|
|
753
|
+
self.headers.get("x-amz-copy-source-if-modified-since")
|
|
754
|
+
)
|
|
755
|
+
unmodified_since = http_date(
|
|
756
|
+
self.headers.get("x-amz-copy-source-if-unmodified-since")
|
|
757
|
+
)
|
|
758
|
+
etag = object_etag(source)
|
|
759
|
+
modified = int(source.stat().st_mtime)
|
|
760
|
+
if if_match is not None:
|
|
761
|
+
if not etag_matches(if_match, etag):
|
|
762
|
+
raise self._precondition_failed_error()
|
|
763
|
+
elif unmodified_since is not None and modified > unmodified_since:
|
|
764
|
+
raise self._precondition_failed_error()
|
|
765
|
+
if if_none_match is not None:
|
|
766
|
+
if etag_matches(if_none_match, etag):
|
|
767
|
+
raise self._precondition_failed_error()
|
|
768
|
+
elif modified_since is not None and modified <= modified_since:
|
|
769
|
+
raise self._precondition_failed_error()
|
|
770
|
+
|
|
771
|
+
def _create_multipart_upload(self, bucket: str, key: str):
|
|
772
|
+
"""Start a multipart upload and send its upload ID."""
|
|
773
|
+
self._object_path(bucket, key)
|
|
774
|
+
upload_id = uuid.uuid4().hex
|
|
775
|
+
upload_dir = self.data_root / UPLOADS_DIR / upload_id
|
|
776
|
+
upload_dir.mkdir(parents=True)
|
|
777
|
+
(upload_dir / UPLOAD_META).write_text(
|
|
778
|
+
json.dumps({"bucket": bucket, "key": key})
|
|
779
|
+
)
|
|
780
|
+
root = ET.Element("InitiateMultipartUploadResult", xmlns=S3_NAMESPACE)
|
|
781
|
+
sub_element(root, "Bucket", bucket)
|
|
782
|
+
sub_element(root, "Key", key)
|
|
783
|
+
sub_element(root, "UploadId", upload_id)
|
|
784
|
+
self._send_xml(root)
|
|
785
|
+
|
|
786
|
+
def _upload_part_copy(self, bucket: str, key: str):
|
|
787
|
+
"""Store all or part of an existing object as one part of a multipart upload."""
|
|
788
|
+
source = self._copy_source()
|
|
789
|
+
size = source.stat().st_size
|
|
790
|
+
header = self.headers.get("x-amz-copy-source-range")
|
|
791
|
+
if header is None:
|
|
792
|
+
start, length = 0, size
|
|
793
|
+
else:
|
|
794
|
+
match = RANGE_RE.match(header.strip())
|
|
795
|
+
if (
|
|
796
|
+
not match
|
|
797
|
+
or not all(match.groups())
|
|
798
|
+
or int(match[1]) > int(match[2])
|
|
799
|
+
or int(match[2]) >= size
|
|
800
|
+
):
|
|
801
|
+
raise S3Error(
|
|
802
|
+
HTTPStatus.BAD_REQUEST,
|
|
803
|
+
"InvalidArgument",
|
|
804
|
+
f"Range specified is not valid for source object of size: {size}",
|
|
805
|
+
)
|
|
806
|
+
start, length = int(match[1]), int(match[2]) - int(match[1]) + 1
|
|
807
|
+
etag = self._store_part(
|
|
808
|
+
bucket, key, lambda file: copy_ranges(file, [(source, start, length)])
|
|
809
|
+
)
|
|
810
|
+
root = ET.Element("CopyPartResult", xmlns=S3_NAMESPACE)
|
|
811
|
+
sub_element(root, "LastModified", iso_timestamp(time.time()))
|
|
812
|
+
sub_element(root, "ETag", f'"{etag}"')
|
|
813
|
+
self._send_xml(root)
|
|
814
|
+
|
|
815
|
+
def _complete_multipart_upload(self, bucket: str, key: str, body: bytes):
|
|
816
|
+
"""Join the listed parts into the final object and delete the upload."""
|
|
817
|
+
upload_dir = self._upload_dir(bucket, key)
|
|
818
|
+
try:
|
|
819
|
+
parts = ET.fromstring(body).findall("{*}Part")
|
|
820
|
+
except ET.ParseError:
|
|
821
|
+
parts = []
|
|
822
|
+
if not parts:
|
|
823
|
+
raise self._malformed_xml_error()
|
|
824
|
+
ranges = []
|
|
825
|
+
previous = 0
|
|
826
|
+
for part in parts:
|
|
827
|
+
try:
|
|
828
|
+
number = int(part.findtext("{*}PartNumber") or "")
|
|
829
|
+
except ValueError:
|
|
830
|
+
raise self._malformed_xml_error() from None
|
|
831
|
+
if number <= previous:
|
|
832
|
+
raise S3Error(
|
|
833
|
+
HTTPStatus.BAD_REQUEST,
|
|
834
|
+
"InvalidPartOrder",
|
|
835
|
+
"The list of parts was not in ascending order. The parts list "
|
|
836
|
+
"must be specified in order by part number.",
|
|
837
|
+
)
|
|
838
|
+
previous = number
|
|
839
|
+
path = upload_dir / f"{number:05d}"
|
|
840
|
+
etag = (part.findtext("{*}ETag") or "").strip().strip('"')
|
|
841
|
+
if not path.is_file() or etag != object_etag(path):
|
|
842
|
+
raise S3Error(
|
|
843
|
+
HTTPStatus.BAD_REQUEST,
|
|
844
|
+
"InvalidPart",
|
|
845
|
+
"One or more of the specified parts could not be found. The part "
|
|
846
|
+
"may not have been uploaded, or the specified entity tag may not "
|
|
847
|
+
"match the part's entity tag.",
|
|
848
|
+
)
|
|
849
|
+
ranges.append((path, 0, path.stat().st_size))
|
|
850
|
+
if any(length < MIN_PART_SIZE for _, _, length in ranges[:-1]):
|
|
851
|
+
raise S3Error(
|
|
852
|
+
HTTPStatus.BAD_REQUEST,
|
|
853
|
+
"EntityTooSmall",
|
|
854
|
+
"Your proposed upload is smaller than the minimum allowed object size.",
|
|
855
|
+
)
|
|
856
|
+
etag = self._write_object(bucket, key, lambda file: copy_ranges(file, ranges))
|
|
857
|
+
shutil.rmtree(upload_dir, ignore_errors=True)
|
|
858
|
+
root = ET.Element("CompleteMultipartUploadResult", xmlns=S3_NAMESPACE)
|
|
859
|
+
location = f"http://{self.headers.get('Host', '')}/{quote(bucket)}/{quote(key)}"
|
|
860
|
+
sub_element(root, "Location", location)
|
|
861
|
+
sub_element(root, "Bucket", bucket)
|
|
862
|
+
sub_element(root, "Key", key)
|
|
863
|
+
sub_element(root, "ETag", f'"{etag}"')
|
|
864
|
+
self._send_xml(root)
|
|
865
|
+
|
|
866
|
+
def _upload_dir(self, bucket: str, key: str) -> Path:
|
|
867
|
+
"""Return the uploadId's folder, checking the upload belongs to this key."""
|
|
868
|
+
self._bucket_dir(bucket)
|
|
869
|
+
upload_id = self.query.get("uploadId", "")
|
|
870
|
+
upload_dir = self.data_root / UPLOADS_DIR / upload_id
|
|
871
|
+
meta = None
|
|
872
|
+
if UPLOAD_ID_RE.match(upload_id):
|
|
873
|
+
try:
|
|
874
|
+
meta = json.loads((upload_dir / UPLOAD_META).read_text())
|
|
875
|
+
except (OSError, ValueError):
|
|
876
|
+
meta = None
|
|
877
|
+
if (
|
|
878
|
+
not isinstance(meta, dict)
|
|
879
|
+
or meta.get("bucket") != bucket
|
|
880
|
+
or meta.get("key") != key
|
|
881
|
+
):
|
|
882
|
+
raise self._no_such_upload_error()
|
|
883
|
+
return upload_dir
|
|
884
|
+
|
|
885
|
+
def _store_part(self, bucket: str, key: str, write) -> str:
|
|
886
|
+
"""Write one multipart upload part via a temporary file and return its ETag."""
|
|
887
|
+
upload_dir = self._upload_dir(bucket, key)
|
|
888
|
+
try:
|
|
889
|
+
number = int(self.query.get("partNumber", ""))
|
|
890
|
+
except ValueError:
|
|
891
|
+
number = 0
|
|
892
|
+
if not 1 <= number <= MAX_PART_NUMBER:
|
|
893
|
+
raise S3Error(
|
|
894
|
+
HTTPStatus.BAD_REQUEST,
|
|
895
|
+
"InvalidArgument",
|
|
896
|
+
f"Part number must be an integer between 1 and {MAX_PART_NUMBER}, "
|
|
897
|
+
"inclusive",
|
|
898
|
+
)
|
|
899
|
+
part_path = upload_dir / f"{number:05d}"
|
|
900
|
+
try:
|
|
901
|
+
fd, tmp_name = tempfile.mkstemp(dir=upload_dir, prefix=TEMP_PREFIX)
|
|
902
|
+
except FileNotFoundError:
|
|
903
|
+
raise self._no_such_upload_error() from None
|
|
904
|
+
try:
|
|
905
|
+
with os.fdopen(fd, "wb") as tmp:
|
|
906
|
+
etag = write(tmp)
|
|
907
|
+
try:
|
|
908
|
+
os.replace(tmp_name, part_path)
|
|
909
|
+
except FileNotFoundError:
|
|
910
|
+
raise self._no_such_upload_error() from None
|
|
911
|
+
except BaseException:
|
|
912
|
+
Path(tmp_name).unlink(missing_ok=True)
|
|
913
|
+
raise
|
|
914
|
+
remember_etag(part_path, etag)
|
|
915
|
+
return etag
|
|
916
|
+
|
|
917
|
+
def _write_object(self, bucket: str, key: str, write) -> str:
|
|
918
|
+
"""Write an object honouring If-None-Match and return its ETag."""
|
|
919
|
+
target = self._object_path(bucket, key)
|
|
920
|
+
if_none_match = self.headers.get("If-None-Match", "").strip() == "*"
|
|
921
|
+
if if_none_match and target.exists():
|
|
922
|
+
raise self._precondition_failed_error()
|
|
923
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
924
|
+
fd, tmp_name = tempfile.mkstemp(dir=target.parent, prefix=TEMP_PREFIX)
|
|
925
|
+
try:
|
|
926
|
+
with os.fdopen(fd, "wb") as tmp:
|
|
927
|
+
etag = write(tmp)
|
|
928
|
+
if if_none_match:
|
|
929
|
+
try:
|
|
930
|
+
os.link(tmp_name, target)
|
|
931
|
+
except FileExistsError:
|
|
932
|
+
raise self._precondition_failed_error() from None
|
|
933
|
+
os.unlink(tmp_name)
|
|
934
|
+
else:
|
|
935
|
+
os.replace(tmp_name, target)
|
|
936
|
+
except BaseException:
|
|
937
|
+
Path(tmp_name).unlink(missing_ok=True)
|
|
938
|
+
raise
|
|
939
|
+
remember_etag(target, etag)
|
|
940
|
+
return etag
|
|
941
|
+
|
|
942
|
+
def _get(self, bucket: str, key: str, include_body: bool):
|
|
943
|
+
"""Send the object's content and metadata, or just the requested byte range."""
|
|
944
|
+
if not bucket or not key:
|
|
945
|
+
raise self._not_implemented_error()
|
|
946
|
+
overrides = self._response_header_overrides()
|
|
947
|
+
target = self._object_path(bucket, key)
|
|
948
|
+
if not target.is_file():
|
|
949
|
+
raise S3Error(
|
|
950
|
+
HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
|
|
951
|
+
)
|
|
952
|
+
try:
|
|
953
|
+
file = target.open("rb")
|
|
954
|
+
except FileNotFoundError:
|
|
955
|
+
raise S3Error(
|
|
956
|
+
HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
|
|
957
|
+
) from None
|
|
958
|
+
with file:
|
|
959
|
+
stat = os.fstat(file.fileno())
|
|
960
|
+
content_type = (
|
|
961
|
+
mimetypes.guess_type(target.name)[0] or "application/octet-stream"
|
|
962
|
+
)
|
|
963
|
+
headers = {
|
|
964
|
+
"Content-Type": content_type,
|
|
965
|
+
"ETag": f'"{object_etag(target)}"',
|
|
966
|
+
"Last-Modified": formatdate(stat.st_mtime, usegmt=True),
|
|
967
|
+
"Accept-Ranges": "bytes",
|
|
968
|
+
**overrides,
|
|
969
|
+
}
|
|
970
|
+
status, start, length = HTTPStatus.OK, 0, stat.st_size
|
|
971
|
+
byte_range = parse_range(self.headers.get("Range", ""), stat.st_size)
|
|
972
|
+
if byte_range:
|
|
973
|
+
start, end = byte_range
|
|
974
|
+
status, length = HTTPStatus.PARTIAL_CONTENT, end - start + 1
|
|
975
|
+
headers["Content-Range"] = f"bytes {start}-{end}/{stat.st_size}"
|
|
976
|
+
self._send_file(status, file, start, length, headers, include_body)
|
|
977
|
+
|
|
978
|
+
def _response_header_overrides(self) -> dict[str, str]:
|
|
979
|
+
"""Return the headers that response-* query parameters ask to override."""
|
|
980
|
+
overrides = {}
|
|
981
|
+
for name, header in RESPONSE_HEADER_OVERRIDES.items():
|
|
982
|
+
value = self.query.get(name)
|
|
983
|
+
if value is None:
|
|
984
|
+
continue
|
|
985
|
+
if any(ord(char) < 32 or ord(char) == 127 for char in value):
|
|
986
|
+
raise S3Error(
|
|
987
|
+
HTTPStatus.BAD_REQUEST,
|
|
988
|
+
"InvalidArgument",
|
|
989
|
+
f"Invalid value for {name}.",
|
|
990
|
+
)
|
|
991
|
+
overrides[header] = value
|
|
992
|
+
return overrides
|
|
993
|
+
|
|
994
|
+
def _not_implemented_error(self) -> S3Error:
|
|
995
|
+
"""Build the error returned for unsupported operations."""
|
|
996
|
+
return S3Error(
|
|
997
|
+
HTTPStatus.NOT_IMPLEMENTED,
|
|
998
|
+
"NotImplemented",
|
|
999
|
+
"A header or operation you provided implies functionality that is not "
|
|
1000
|
+
"implemented.",
|
|
1001
|
+
)
|
|
1002
|
+
|
|
1003
|
+
def _precondition_failed_error(self) -> S3Error:
|
|
1004
|
+
"""Build the error for a conditional write that finds an existing object."""
|
|
1005
|
+
return S3Error(
|
|
1006
|
+
HTTPStatus.PRECONDITION_FAILED,
|
|
1007
|
+
"PreconditionFailed",
|
|
1008
|
+
"At least one of the pre-conditions you specified did not hold",
|
|
1009
|
+
)
|
|
1010
|
+
|
|
1011
|
+
def _malformed_xml_error(self) -> S3Error:
|
|
1012
|
+
"""Build the error for a request body that is not the expected XML."""
|
|
1013
|
+
return S3Error(
|
|
1014
|
+
HTTPStatus.BAD_REQUEST,
|
|
1015
|
+
"MalformedXML",
|
|
1016
|
+
"The XML you provided was not well-formed or did not validate against "
|
|
1017
|
+
"our published schema",
|
|
1018
|
+
)
|
|
1019
|
+
|
|
1020
|
+
def _no_such_upload_error(self) -> S3Error:
|
|
1021
|
+
"""Build the error for an upload ID with no active upload for this key."""
|
|
1022
|
+
return S3Error(
|
|
1023
|
+
HTTPStatus.NOT_FOUND,
|
|
1024
|
+
"NoSuchUpload",
|
|
1025
|
+
"The specified upload does not exist. The upload ID may be invalid, or "
|
|
1026
|
+
"the upload may have been aborted or completed.",
|
|
1027
|
+
)
|
|
1028
|
+
|
|
1029
|
+
def _bucket_dir(self, bucket: str) -> Path:
|
|
1030
|
+
"""Return a bucket's directory, raising NoSuchBucket if it does not exist."""
|
|
1031
|
+
bucket_dir = self.data_root / bucket
|
|
1032
|
+
if not BUCKET_NAME_RE.match(bucket) or not bucket_dir.is_dir():
|
|
1033
|
+
raise S3Error(
|
|
1034
|
+
HTTPStatus.NOT_FOUND,
|
|
1035
|
+
"NoSuchBucket",
|
|
1036
|
+
"The specified bucket does not exist.",
|
|
1037
|
+
)
|
|
1038
|
+
return bucket_dir
|
|
1039
|
+
|
|
1040
|
+
def _object_path(self, bucket: str, key: str) -> Path:
|
|
1041
|
+
"""Resolve a key's path, checking the bucket exists and the key stays inside."""
|
|
1042
|
+
bucket_dir = self._bucket_dir(bucket)
|
|
1043
|
+
parts = key.split("/")
|
|
1044
|
+
if key.endswith("/") or any(part in ("", ".", "..") for part in parts):
|
|
1045
|
+
raise S3Error(
|
|
1046
|
+
HTTPStatus.BAD_REQUEST,
|
|
1047
|
+
"InvalidArgument",
|
|
1048
|
+
"This server does not support this key.",
|
|
1049
|
+
)
|
|
1050
|
+
target = bucket_dir.joinpath(*parts).resolve()
|
|
1051
|
+
if not target.is_relative_to(bucket_dir.resolve()):
|
|
1052
|
+
raise S3Error(
|
|
1053
|
+
HTTPStatus.BAD_REQUEST,
|
|
1054
|
+
"InvalidArgument",
|
|
1055
|
+
"This server does not support this key.",
|
|
1056
|
+
)
|
|
1057
|
+
return target
|
|
1058
|
+
|
|
1059
|
+
def _read_body(self) -> bytes:
|
|
1060
|
+
"""Read the request body, decoding HTTP chunked and aws-chunked encodings."""
|
|
1061
|
+
if "chunked" in self.headers.get("Transfer-Encoding", "").lower():
|
|
1062
|
+
body = read_chunked(self.rfile)
|
|
1063
|
+
else:
|
|
1064
|
+
length = int(self.headers.get("Content-Length") or 0)
|
|
1065
|
+
body = self.rfile.read(length)
|
|
1066
|
+
content_sha = self.headers.get("x-amz-content-sha256", "")
|
|
1067
|
+
content_encoding = self.headers.get("Content-Encoding", "")
|
|
1068
|
+
if content_sha.startswith("STREAMING-") or "aws-chunked" in content_encoding:
|
|
1069
|
+
body = read_chunked(io.BytesIO(body))
|
|
1070
|
+
return body
|
|
1071
|
+
|
|
1072
|
+
def _send_headers(self, status: HTTPStatus, length: int, headers=None):
|
|
1073
|
+
"""Send the status line and headers, including the standard S3 ones."""
|
|
1074
|
+
self.send_response(status)
|
|
1075
|
+
self.send_header("x-amz-request-id", uuid.uuid4().hex.upper()[:16])
|
|
1076
|
+
self.send_header("Content-Length", str(length))
|
|
1077
|
+
origin = self.headers.get("Origin")
|
|
1078
|
+
if origin:
|
|
1079
|
+
self.send_header("Access-Control-Allow-Origin", origin)
|
|
1080
|
+
self.send_header("Access-Control-Expose-Headers", CORS_EXPOSE_HEADERS)
|
|
1081
|
+
self.send_header("Vary", "Origin")
|
|
1082
|
+
for name, value in (headers or {}).items():
|
|
1083
|
+
self.send_header(name, value)
|
|
1084
|
+
self.end_headers()
|
|
1085
|
+
|
|
1086
|
+
def _send(
|
|
1087
|
+
self,
|
|
1088
|
+
status: HTTPStatus,
|
|
1089
|
+
body: bytes = b"",
|
|
1090
|
+
headers=None,
|
|
1091
|
+
include_body: bool = True,
|
|
1092
|
+
):
|
|
1093
|
+
"""Send a response with the standard S3 headers."""
|
|
1094
|
+
self._send_headers(status, len(body), headers)
|
|
1095
|
+
if include_body and body:
|
|
1096
|
+
self.wfile.write(body)
|
|
1097
|
+
|
|
1098
|
+
def _send_file(
|
|
1099
|
+
self,
|
|
1100
|
+
status: HTTPStatus,
|
|
1101
|
+
file,
|
|
1102
|
+
start: int,
|
|
1103
|
+
length: int,
|
|
1104
|
+
headers,
|
|
1105
|
+
include_body: bool,
|
|
1106
|
+
):
|
|
1107
|
+
"""Send a response whose body is a byte range of an open file."""
|
|
1108
|
+
self._send_headers(status, length, headers)
|
|
1109
|
+
if not include_body:
|
|
1110
|
+
return
|
|
1111
|
+
file.seek(start)
|
|
1112
|
+
remaining = length
|
|
1113
|
+
while remaining > 0:
|
|
1114
|
+
block = file.read(min(remaining, BLOCK_SIZE))
|
|
1115
|
+
if not block:
|
|
1116
|
+
self.close_connection = True
|
|
1117
|
+
return
|
|
1118
|
+
self.wfile.write(block)
|
|
1119
|
+
remaining -= len(block)
|
|
1120
|
+
|
|
1121
|
+
def _send_xml(self, root: ET.Element):
|
|
1122
|
+
"""Send a successful XML response."""
|
|
1123
|
+
body = ET.tostring(root, encoding="UTF-8", xml_declaration=True)
|
|
1124
|
+
self._send(HTTPStatus.OK, body, {"Content-Type": "application/xml"})
|
|
1125
|
+
|
|
1126
|
+
def _send_error(self, error: S3Error):
|
|
1127
|
+
"""Send an S3-style XML error document."""
|
|
1128
|
+
xml = (
|
|
1129
|
+
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
1130
|
+
f"<Error><Code>{error.code}</Code>"
|
|
1131
|
+
f"<Message>{escape(error.message)}</Message>"
|
|
1132
|
+
f"<Resource>{escape(urlsplit(self.path).path)}</Resource></Error>"
|
|
1133
|
+
).encode()
|
|
1134
|
+
self.close_connection = True
|
|
1135
|
+
self._send(
|
|
1136
|
+
error.status,
|
|
1137
|
+
xml,
|
|
1138
|
+
{"Content-Type": "application/xml", "Connection": "close"},
|
|
1139
|
+
include_body=self.command != "HEAD",
|
|
1140
|
+
)
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
def main():
|
|
1144
|
+
"""Parse the command line and run the server."""
|
|
1145
|
+
parser = argparse.ArgumentParser(
|
|
1146
|
+
prog="localbucket",
|
|
1147
|
+
description=(
|
|
1148
|
+
f"LocalBucket {__version__}: a minimal S3-compatible HTTP server that "
|
|
1149
|
+
f"stores buckets as folders in DATA_ROOT."
|
|
1150
|
+
),
|
|
1151
|
+
epilog=HELP_EPILOG,
|
|
1152
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
1153
|
+
)
|
|
1154
|
+
parser.add_argument(
|
|
1155
|
+
"data_root",
|
|
1156
|
+
metavar="DATA_ROOT",
|
|
1157
|
+
type=Path,
|
|
1158
|
+
help="directory that holds the buckets",
|
|
1159
|
+
)
|
|
1160
|
+
parser.add_argument(
|
|
1161
|
+
"--host", default="127.0.0.1", help="address to listen on (default: 127.0.0.1)"
|
|
1162
|
+
)
|
|
1163
|
+
parser.add_argument(
|
|
1164
|
+
"--port", type=int, default=9000, help="port to listen on (default: 9000)"
|
|
1165
|
+
)
|
|
1166
|
+
parser.add_argument(
|
|
1167
|
+
"--version", action="version", version=f"LocalBucket {__version__}"
|
|
1168
|
+
)
|
|
1169
|
+
if len(sys.argv) == 1:
|
|
1170
|
+
parser.print_help()
|
|
1171
|
+
sys.exit(2)
|
|
1172
|
+
args = parser.parse_args()
|
|
1173
|
+
logging.basicConfig(
|
|
1174
|
+
level=logging.INFO,
|
|
1175
|
+
format="%(asctime)s [%(levelname)s]: %(message)s",
|
|
1176
|
+
datefmt="%Y-%m-%d %H:%M:%S",
|
|
1177
|
+
)
|
|
1178
|
+
|
|
1179
|
+
data_root = args.data_root.resolve()
|
|
1180
|
+
data_root.mkdir(parents=True, exist_ok=True)
|
|
1181
|
+
S3Handler.data_root = data_root
|
|
1182
|
+
|
|
1183
|
+
server = ThreadingHTTPServer((args.host, args.port), S3Handler)
|
|
1184
|
+
print(BANNER, file=sys.stderr)
|
|
1185
|
+
logger.info(
|
|
1186
|
+
"LocalBucket %s serving %s on http://%s:%s",
|
|
1187
|
+
__version__,
|
|
1188
|
+
data_root,
|
|
1189
|
+
args.host,
|
|
1190
|
+
args.port,
|
|
1191
|
+
)
|
|
1192
|
+
log_buckets(data_root)
|
|
1193
|
+
try:
|
|
1194
|
+
server.serve_forever()
|
|
1195
|
+
except KeyboardInterrupt:
|
|
1196
|
+
pass
|
|
1197
|
+
finally:
|
|
1198
|
+
server.server_close()
|
|
1199
|
+
|
|
1200
|
+
|
|
1201
|
+
if __name__ == "__main__":
|
|
1202
|
+
main()
|