localbucket 1.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,107 @@
1
+ Metadata-Version: 2.5
2
+ Name: localbucket
3
+ Version: 1.2.1
4
+ Summary: Minimal S3-compatible server that stores buckets as plain folders on disk, for local development and testing.
5
+ Project-URL: Homepage, https://github.com/selcuk/localbucket
6
+ Project-URL: Issues, https://github.com/selcuk/localbucket/issues
7
+ Author: Selcuk Ayguney
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: aws,boto3,django,emulator,mock,s3,testing
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Internet :: WWW/HTTP :: HTTP Servers
18
+ Classifier: Topic :: Software Development :: Testing
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+
22
+ # LocalBucket
23
+
24
+ LocalBucket is a minimal S3-compatible HTTP server that stores buckets as
25
+ folders on a local directory. It is a single Python script for Python 3.10
26
+ and newer, with no dependencies beyond the standard library, meant for
27
+ local development and testing against S3 clients such as boto3 and the AWS
28
+ CLI.
29
+
30
+ Each bucket is a folder under the data directory and each object is a
31
+ regular file, so you can inspect and change the stored data with ordinary
32
+ file tools.
33
+
34
+ ## Installation
35
+
36
+ ```sh
37
+ uv tool install localbucket # or: pipx install localbucket
38
+ ```
39
+
40
+ You can also run it without installing, with `uvx localbucket ./data`, or
41
+ copy `localbucket.py` anywhere and run it with `python3 localbucket.py`.
42
+
43
+ ## Usage
44
+
45
+ ```sh
46
+ localbucket ./data
47
+ ```
48
+
49
+ This serves `./data` on `http://127.0.0.1:9000`. Use `--host` and `--port`
50
+ to change the address.
51
+
52
+ Point a client at the server with path-style addressing. Credentials are not
53
+ checked, so any access key and secret will do:
54
+
55
+ ```sh
56
+ aws --endpoint-url http://127.0.0.1:9000 s3 mb s3://mybucket
57
+ aws --endpoint-url http://127.0.0.1:9000 s3 cp file.txt s3://mybucket/
58
+ ```
59
+
60
+ ```python
61
+ import boto3
62
+ from botocore.config import Config
63
+
64
+ s3 = boto3.client(
65
+ "s3",
66
+ endpoint_url="http://127.0.0.1:9000",
67
+ aws_access_key_id="test",
68
+ aws_secret_access_key="test",
69
+ region_name="us-east-1",
70
+ config=Config(s3={"addressing_style": "path"}),
71
+ )
72
+ ```
73
+
74
+ ## Supported operations
75
+
76
+ - Buckets: ListBuckets, CreateBucket, HeadBucket, and DeleteBucket (empty
77
+ buckets only)
78
+ - Objects: ListObjects, ListObjectsV2, PutObject (including
79
+ `If-None-Match: *`), GetObject and HeadObject (including `Range` and
80
+ `response-*` header overrides such as `response-content-disposition`),
81
+ CopyObject, DeleteObject
82
+ - Multipart uploads: CreateMultipartUpload, UploadPart, UploadPartCopy,
83
+ CompleteMultipartUpload, AbortMultipartUpload
84
+ - GetObjectTagging, which always returns an empty tag set because tags are
85
+ not stored
86
+
87
+ Requests with any other query parameter, such as ACL or tagging changes,
88
+ return `501 NotImplemented` and do not change anything. Virtual-hosted-style
89
+ URLs, authentication and versioning are not supported. Object metadata is
90
+ not stored either: `Content-Type` is guessed from the key's file extension.
91
+
92
+ Unfinished multipart uploads are kept in `DATA_ROOT/.localbucket-uploads`.
93
+
94
+ CORS is open: requests from any origin are allowed, including `OPTIONS`
95
+ preflights, so browser code on your development site can call LocalBucket
96
+ directly. Any web page you open can also read from it while it runs, so
97
+ keep it on `127.0.0.1` and don't store anything sensitive in it.
98
+
99
+ ## Testing
100
+
101
+ The test suite starts LocalBucket on a temporary directory and checks its
102
+ behaviour with boto3. It is a [uv](https://docs.astral.sh/uv/) script that
103
+ declares its own dependencies:
104
+
105
+ ```sh
106
+ uv run test_localbucket.py
107
+ ```
@@ -0,0 +1,6 @@
1
+ localbucket.py,sha256=ZHvaI12CKY3iNrdlj_W115BZWUzWt6h7tG0cjRYYIcc,47107
2
+ localbucket-1.2.1.dist-info/METADATA,sha256=5-UQJ2FVuoyqY-i5bLmebSHtmd5epj560646L58gxyw,3716
3
+ localbucket-1.2.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
4
+ localbucket-1.2.1.dist-info/entry_points.txt,sha256=94lO3GSUIS8-mbMq4Qo2B75lqhxNvrB1JD5NGmycxPk,49
5
+ localbucket-1.2.1.dist-info/licenses/LICENSE,sha256=bXAkS--71fEU6n463k4j8cCf4xunLiI0qoYzftanVRU,1071
6
+ localbucket-1.2.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ localbucket = localbucket:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Selcuk Ayguney
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
localbucket.py ADDED
@@ -0,0 +1,1202 @@
1
+ #!/usr/bin/env python3
2
+ """A minimal HTTP server that imitates part of the AWS S3 API on a local directory."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import base64
8
+ import hashlib
9
+ import io
10
+ import json
11
+ import logging
12
+ import mimetypes
13
+ import os
14
+ import re
15
+ import shutil
16
+ import sys
17
+ import tempfile
18
+ import threading
19
+ import time
20
+ import uuid
21
+ import xml.etree.ElementTree as ET
22
+ from collections.abc import Set as AbstractSet
23
+ from datetime import datetime, timezone
24
+ from email.utils import formatdate, parsedate_to_datetime
25
+ from http import HTTPStatus
26
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
27
+ from pathlib import Path
28
+ from urllib.parse import parse_qs, quote, unquote, urlsplit
29
+ from xml.sax.saxutils import escape
30
+
31
+ __version__ = "1.2.1"
32
+ BUCKET_NAME_RE = re.compile(r"^[a-z0-9][a-z0-9.-]{1,61}[a-z0-9]$")
33
+ S3_NAMESPACE = "http://s3.amazonaws.com/doc/2006-03-01/"
34
+ TEMP_PREFIX = ".localbucket-"
35
+ MAX_KEYS_LIMIT = 1000
36
+ UPLOADS_DIR = ".localbucket-uploads"
37
+ UPLOAD_ID_RE = re.compile(r"^[0-9a-f]{32}$")
38
+ UPLOAD_META = "upload.json"
39
+ MIN_PART_SIZE = 5 * 1024 * 1024
40
+ MAX_PART_NUMBER = 10000
41
+ BLOCK_SIZE = 1024 * 1024
42
+ RANGE_RE = re.compile(r"^bytes=(\d*)-(\d*)$")
43
+ ALWAYS_ALLOWED_QUERY = {
44
+ "x-id",
45
+ "X-Amz-Algorithm",
46
+ "X-Amz-Credential",
47
+ "X-Amz-Date",
48
+ "X-Amz-Expires",
49
+ "X-Amz-SignedHeaders",
50
+ "X-Amz-Signature",
51
+ "X-Amz-Security-Token",
52
+ "AWSAccessKeyId",
53
+ "Signature",
54
+ "Expires",
55
+ }
56
+ LIST_OBJECTS_QUERY = {"prefix", "delimiter", "marker", "max-keys", "encoding-type"}
57
+ LIST_OBJECTS_V2_QUERY = {
58
+ "list-type",
59
+ "prefix",
60
+ "delimiter",
61
+ "max-keys",
62
+ "start-after",
63
+ "continuation-token",
64
+ "encoding-type",
65
+ }
66
+ logger = logging.getLogger("localbucket")
67
+ RESPONSE_HEADER_OVERRIDES = {
68
+ "response-cache-control": "Cache-Control",
69
+ "response-content-disposition": "Content-Disposition",
70
+ "response-content-encoding": "Content-Encoding",
71
+ "response-content-language": "Content-Language",
72
+ "response-content-type": "Content-Type",
73
+ "response-expires": "Expires",
74
+ }
75
+ CORS_EXPOSE_HEADERS = (
76
+ "Accept-Ranges, Content-Disposition, Content-Encoding, Content-Range, ETag, "
77
+ "x-amz-bucket-region, x-amz-request-id"
78
+ )
79
+ BANNER = r"""
80
+ __ ______ __ __
81
+ / / ____ _________ _/ / __ )__ _______/ /_____ / /_
82
+ / / / __ \/ ___/ __ `/ / __ / / / / ___/ //_/ _ \/ __/
83
+ / /___/ /_/ / /__/ /_/ / / /_/ / /_/ / /__/ ,< / __/ /_
84
+ /_____/\____/\___/\__,_/_/_____/\__,_/\___/_/|_|\___/\__/
85
+ """
86
+ HELP_EPILOG = """\
87
+ supported operations (path-style URLs only, credentials are not checked):
88
+ ListBuckets, CreateBucket, DeleteBucket (empty buckets only), HeadBucket,
89
+ ListObjects, ListObjectsV2, PutObject (with If-None-Match: *), CopyObject,
90
+ GetObject and HeadObject (with Range and response-* header overrides),
91
+ DeleteObject, CreateMultipartUpload, UploadPart, UploadPartCopy,
92
+ CompleteMultipartUpload, AbortMultipartUpload,
93
+ GetObjectTagging (always an empty tag set, because tags are not stored)
94
+ CORS requests from any origin are allowed, including OPTIONS preflights.
95
+ Unfinished multipart uploads are kept in DATA_ROOT/.localbucket-uploads.
96
+
97
+ example:
98
+ localbucket ./data
99
+ aws --endpoint-url http://127.0.0.1:9000 s3 cp file.txt s3://mybucket/
100
+ """
101
+
102
+
103
+ class S3Error(Exception):
104
+ """An S3-style error that is returned to the client as an XML document."""
105
+
106
+ def __init__(self, status: HTTPStatus, code: str, message: str):
107
+ super().__init__(message)
108
+ self.status = status
109
+ self.code = code
110
+ self.message = message
111
+
112
+
113
+ def read_chunked(stream) -> bytes:
114
+ """Decode an HTTP chunked or aws-chunked body, dropping signatures and trailers."""
115
+ body = bytearray()
116
+ while True:
117
+ line = stream.readline()
118
+ if not line:
119
+ raise S3Error(
120
+ HTTPStatus.BAD_REQUEST,
121
+ "IncompleteBody",
122
+ "Chunked body ended unexpectedly.",
123
+ )
124
+ size_text = line.split(b";", 1)[0].strip()
125
+ try:
126
+ size = int(size_text, 16)
127
+ except ValueError:
128
+ raise S3Error(
129
+ HTTPStatus.BAD_REQUEST, "InvalidRequest", "Invalid chunk size."
130
+ ) from None
131
+ if size == 0:
132
+ while stream.readline().strip():
133
+ pass
134
+ return bytes(body)
135
+ chunk = stream.read(size)
136
+ if len(chunk) != size:
137
+ raise S3Error(
138
+ HTTPStatus.BAD_REQUEST,
139
+ "IncompleteBody",
140
+ "Chunk is shorter than declared.",
141
+ )
142
+ body += chunk
143
+ stream.readline()
144
+
145
+
146
+ def iso_timestamp(epoch: float) -> str:
147
+ """Format a Unix timestamp the way S3 formats dates in XML responses."""
148
+ return datetime.fromtimestamp(epoch, timezone.utc).strftime(
149
+ "%Y-%m-%dT%H:%M:%S.000Z"
150
+ )
151
+
152
+
153
+ _etag_cache: dict[str, tuple[tuple[int, int, int, int], str]] = {}
154
+ _etag_lock = threading.Lock()
155
+
156
+
157
+ def stat_signature(path: Path) -> tuple[int, int, int, int]:
158
+ """Return a file's identity, size and mtime, which change when it is rewritten."""
159
+ stat = path.stat()
160
+ return stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns
161
+
162
+
163
+ def object_etag(path: Path) -> str:
164
+ """Return a file's MD5 ETag, reusing the cached value while it is unchanged."""
165
+ signature = stat_signature(path)
166
+ with _etag_lock:
167
+ cached = _etag_cache.get(str(path))
168
+ if cached and cached[0] == signature:
169
+ return cached[1]
170
+ digest = hashlib.md5()
171
+ with path.open("rb") as file:
172
+ while block := file.read(1024 * 1024):
173
+ digest.update(block)
174
+ etag = digest.hexdigest()
175
+ with _etag_lock:
176
+ _etag_cache[str(path)] = (signature, etag)
177
+ return etag
178
+
179
+
180
+ def remember_etag(path: Path, etag: str):
181
+ """Cache the ETag of a file that has just been written."""
182
+ signature = stat_signature(path)
183
+ with _etag_lock:
184
+ _etag_cache[str(path)] = (signature, etag)
185
+
186
+
187
+ def etag_matches(header: str, etag: str) -> bool:
188
+ """Return whether an If-Match style header lists the given ETag or the wildcard."""
189
+ candidates = {
190
+ value.strip().removeprefix("W/").strip('"') for value in header.split(",")
191
+ }
192
+ return "*" in candidates or etag in candidates
193
+
194
+
195
+ def http_date(header: str | None) -> int | None:
196
+ """Parse an HTTP date header to a Unix timestamp, or None if missing or invalid."""
197
+ if not header:
198
+ return None
199
+ try:
200
+ return int(parsedate_to_datetime(header).timestamp())
201
+ except (TypeError, ValueError, IndexError):
202
+ return None
203
+
204
+
205
+ def write_bytes(file, data: bytes) -> str:
206
+ """Write data to a file and return its hex MD5 digest."""
207
+ file.write(data)
208
+ return hashlib.md5(data).hexdigest()
209
+
210
+
211
+ def copy_ranges(destination, ranges: list[tuple[Path, int, int]]) -> str:
212
+ """Write (path, start, length) file ranges to destination; return their hex MD5."""
213
+ digest = hashlib.md5()
214
+ for source, start, length in ranges:
215
+ with source.open("rb") as file:
216
+ file.seek(start)
217
+ remaining = length
218
+ while remaining > 0:
219
+ block = file.read(min(remaining, BLOCK_SIZE))
220
+ if not block:
221
+ raise S3Error(
222
+ HTTPStatus.INTERNAL_SERVER_ERROR,
223
+ "InternalError",
224
+ "A source file changed while it was being copied.",
225
+ )
226
+ destination.write(block)
227
+ digest.update(block)
228
+ remaining -= len(block)
229
+ return digest.hexdigest()
230
+
231
+
232
+ def parse_range(header: str, size: int) -> tuple[int, int] | None:
233
+ """Return the Range header's inclusive byte range, or None for the whole object."""
234
+ match = RANGE_RE.match(header.strip())
235
+ if not match or match.groups() == ("", ""):
236
+ return None
237
+ first, last = match.groups()
238
+ if first:
239
+ start = int(first)
240
+ if last and int(last) < start:
241
+ return None
242
+ end = min(int(last), size - 1) if last else size - 1
243
+ else:
244
+ length = int(last)
245
+ if length == 0:
246
+ raise invalid_range_error()
247
+ start, end = max(size - length, 0), size - 1
248
+ if start >= size:
249
+ raise invalid_range_error()
250
+ return start, end
251
+
252
+
253
+ def invalid_range_error() -> S3Error:
254
+ """Build the error for a Range header that starts past the end of the object."""
255
+ return S3Error(
256
+ HTTPStatus.REQUESTED_RANGE_NOT_SATISFIABLE,
257
+ "InvalidRange",
258
+ "The requested range is not satisfiable",
259
+ )
260
+
261
+
262
+ def sub_element(parent: ET.Element, tag: str, text: str) -> ET.Element:
263
+ """Add a child element containing the given text."""
264
+ child = ET.SubElement(parent, tag)
265
+ child.text = text
266
+ return child
267
+
268
+
269
+ def bucket_dirs(data_root: Path) -> list[Path]:
270
+ """Return the bucket directories under the data root, sorted by name."""
271
+ return sorted(
272
+ (
273
+ entry
274
+ for entry in data_root.iterdir()
275
+ if entry.is_dir() and BUCKET_NAME_RE.match(entry.name)
276
+ ),
277
+ key=lambda p: p.name,
278
+ )
279
+
280
+
281
+ def log_buckets(data_root: Path):
282
+ """Log the buckets under the data root as a comma-separated list."""
283
+ buckets = [entry.name for entry in bucket_dirs(data_root)]
284
+ logger.info("Buckets: %s", ", ".join(buckets) or "none")
285
+
286
+
287
+ class S3Handler(BaseHTTPRequestHandler):
288
+ """Handles path-style S3 requests for buckets and objects."""
289
+
290
+ server_version = "LocalBucket"
291
+ protocol_version = "HTTP/1.1"
292
+ data_root: Path
293
+ query: dict[str, str]
294
+
295
+ def log_message(self, format: str, *args):
296
+ """Send a request log line to the localbucket logger."""
297
+ message = format % args
298
+ control_chars = getattr(self, "_control_char_table", None)
299
+ if control_chars:
300
+ message = message.translate(control_chars)
301
+ logger.info("%s %s", self.address_string(), message)
302
+
303
+ def do_PUT(self):
304
+ """Create a bucket or write an object."""
305
+ self._dispatch(self._put)
306
+
307
+ def do_GET(self):
308
+ """List buckets, list a bucket's objects or read an object."""
309
+ self._dispatch(self._get_request)
310
+
311
+ def do_HEAD(self):
312
+ """Check a bucket exists or return an object's metadata without its body."""
313
+ self._dispatch(self._head_request)
314
+
315
+ def do_POST(self):
316
+ """Start or complete a multipart upload."""
317
+ self._dispatch(self._post_request)
318
+
319
+ def do_DELETE(self):
320
+ """Delete an object or abort a multipart upload."""
321
+ self._dispatch(self._delete_request)
322
+
323
+ def do_OPTIONS(self):
324
+ """Answer a CORS preflight request, allowing any origin, method and header."""
325
+ headers = {
326
+ "Access-Control-Allow-Methods": "GET, HEAD, PUT, POST, DELETE",
327
+ "Access-Control-Max-Age": "3000",
328
+ }
329
+ requested = self.headers.get("Access-Control-Request-Headers")
330
+ if requested:
331
+ headers["Access-Control-Allow-Headers"] = requested
332
+ self._send(HTTPStatus.OK, headers=headers)
333
+
334
+ def _dispatch(self, action):
335
+ """Parse bucket, key and query from the URL, run it and send errors as XML."""
336
+ try:
337
+ url = urlsplit(self.path)
338
+ path = unquote(url.path)
339
+ bucket, _, key = path.lstrip("/").partition("/")
340
+ parsed = parse_qs(url.query, keep_blank_values=True)
341
+ self.query = {name: values[0] for name, values in parsed.items()}
342
+ action(bucket, key)
343
+ except S3Error as error:
344
+ self._send_error(error)
345
+
346
+ def _get_request(self, bucket: str, key: str):
347
+ """Route a GET request to ListBuckets, ListObjects(V2) or GetObject."""
348
+ if not bucket:
349
+ self._check_query("ListBuckets")
350
+ self._list_buckets()
351
+ elif not key:
352
+ if self.query.get("list-type") == "2":
353
+ self._check_query("ListObjectsV2", LIST_OBJECTS_V2_QUERY)
354
+ self._list_objects_v2(bucket)
355
+ else:
356
+ self._check_query("ListObjects", LIST_OBJECTS_QUERY)
357
+ self._list_objects(bucket)
358
+ elif "tagging" in self.query:
359
+ self._check_query("GetObjectTagging", {"tagging"})
360
+ if not self._object_path(bucket, key).is_file():
361
+ raise S3Error(
362
+ HTTPStatus.NOT_FOUND,
363
+ "NoSuchKey",
364
+ "The specified key does not exist.",
365
+ )
366
+ root = ET.Element("Tagging", xmlns=S3_NAMESPACE)
367
+ ET.SubElement(root, "TagSet")
368
+ self._send_xml(root)
369
+ else:
370
+ self._check_query("GetObject", RESPONSE_HEADER_OVERRIDES.keys())
371
+ self._get(bucket, key, include_body=True)
372
+
373
+ def _head_request(self, bucket: str, key: str):
374
+ """Route a HEAD request to HeadBucket or HeadObject."""
375
+ if not bucket:
376
+ raise self._not_implemented_error()
377
+ if not key:
378
+ self._check_query("HeadBucket")
379
+ self._bucket_dir(bucket)
380
+ self._send(HTTPStatus.OK, headers={"x-amz-bucket-region": "us-east-1"})
381
+ else:
382
+ self._check_query("HeadObject", RESPONSE_HEADER_OVERRIDES.keys())
383
+ self._get(bucket, key, include_body=False)
384
+
385
+ def _delete_request(self, bucket: str, key: str):
386
+ """Route DELETE to DeleteBucket, DeleteObject or AbortMultipartUpload."""
387
+ if not bucket:
388
+ raise self._not_implemented_error()
389
+ if not key:
390
+ self._check_query("DeleteBucket")
391
+ self._delete_bucket(bucket)
392
+ elif "uploadId" in self.query:
393
+ self._check_query("AbortMultipartUpload", {"uploadId"})
394
+ shutil.rmtree(self._upload_dir(bucket, key), ignore_errors=True)
395
+ self._send(HTTPStatus.NO_CONTENT)
396
+ else:
397
+ self._check_query("DeleteObject")
398
+ self._delete_object(bucket, key)
399
+
400
+ def _post_request(self, bucket: str, key: str):
401
+ """Route a POST request to CreateMultipartUpload or CompleteMultipartUpload."""
402
+ if not bucket or not key:
403
+ raise self._not_implemented_error()
404
+ if "uploads" in self.query:
405
+ self._check_query("CreateMultipartUpload", {"uploads"})
406
+ self._read_body()
407
+ self._create_multipart_upload(bucket, key)
408
+ elif "uploadId" in self.query:
409
+ self._check_query("CompleteMultipartUpload", {"uploadId"})
410
+ self._complete_multipart_upload(bucket, key, self._read_body())
411
+ else:
412
+ raise self._not_implemented_error()
413
+
414
+ def _check_query(self, operation: str, allowed: AbstractSet[str] = frozenset()):
415
+ """Raise NotImplemented for query parameters the operation does not support."""
416
+ unsupported = sorted(set(self.query) - ALWAYS_ALLOWED_QUERY - allowed)
417
+ if unsupported:
418
+ self.log_message(
419
+ "rejected %s with unsupported query parameters: %s",
420
+ operation,
421
+ ", ".join(unsupported),
422
+ )
423
+ raise self._not_implemented_error()
424
+
425
+ def _list_buckets(self):
426
+ """Send the list of all buckets under the data root."""
427
+ root = ET.Element("ListAllMyBucketsResult", xmlns=S3_NAMESPACE)
428
+ owner = ET.SubElement(root, "Owner")
429
+ sub_element(owner, "ID", "localbucket")
430
+ sub_element(owner, "DisplayName", "LocalBucket")
431
+ buckets = ET.SubElement(root, "Buckets")
432
+ for entry in bucket_dirs(self.data_root):
433
+ stat = entry.stat()
434
+ created = getattr(stat, "st_birthtime", stat.st_ctime)
435
+ bucket = ET.SubElement(buckets, "Bucket")
436
+ sub_element(bucket, "Name", entry.name)
437
+ sub_element(bucket, "CreationDate", iso_timestamp(created))
438
+ self._send_xml(root)
439
+
440
+ def _list_objects(self, bucket: str):
441
+ """Send one page of objects, with prefix, delimiter, marker and paging."""
442
+ bucket_dir = self._bucket_dir(bucket)
443
+ prefix = self.query.get("prefix", "")
444
+ delimiter = self.query.get("delimiter", "")
445
+ marker = self.query.get("marker", "")
446
+ max_keys = self._max_keys()
447
+ marker_is_prefix = bool(delimiter) and marker.endswith(delimiter)
448
+ contents, common_prefixes, truncated, last_entry = self._list_entries(
449
+ bucket_dir, prefix, delimiter, marker, marker_is_prefix, max_keys
450
+ )
451
+
452
+ encode = self._list_encoder()
453
+ root = ET.Element("ListBucketResult", xmlns=S3_NAMESPACE)
454
+ sub_element(root, "Name", bucket)
455
+ sub_element(root, "Prefix", encode(prefix))
456
+ sub_element(root, "Marker", encode(marker))
457
+ if delimiter:
458
+ sub_element(root, "Delimiter", encode(delimiter))
459
+ sub_element(root, "MaxKeys", str(max_keys))
460
+ sub_element(root, "IsTruncated", "true" if truncated else "false")
461
+ if truncated:
462
+ sub_element(root, "NextMarker", encode(last_entry[2:]))
463
+ if self.query.get("encoding-type") == "url":
464
+ sub_element(root, "EncodingType", "url")
465
+ self._add_list_entries(root, contents, common_prefixes, encode)
466
+ self._send_xml(root)
467
+
468
+ def _list_objects_v2(self, bucket: str):
469
+ """Send one page of objects, with prefix, delimiter, start-after and paging."""
470
+ bucket_dir = self._bucket_dir(bucket)
471
+ prefix = self.query.get("prefix", "")
472
+ delimiter = self.query.get("delimiter", "")
473
+ start_after = self.query.get("start-after", "")
474
+ token = self.query.get("continuation-token")
475
+ max_keys = self._max_keys()
476
+
477
+ marker, marker_is_prefix = start_after, False
478
+ if token is not None:
479
+ try:
480
+ decoded = base64.urlsafe_b64decode(token.encode()).decode()
481
+ except ValueError:
482
+ decoded = ""
483
+ if decoded[:2] not in ("K:", "P:"):
484
+ raise S3Error(
485
+ HTTPStatus.BAD_REQUEST,
486
+ "InvalidArgument",
487
+ "The continuation token provided is incorrect.",
488
+ )
489
+ marker, marker_is_prefix = decoded[2:], decoded.startswith("P:")
490
+ contents, common_prefixes, truncated, last_entry = self._list_entries(
491
+ bucket_dir, prefix, delimiter, marker, marker_is_prefix, max_keys
492
+ )
493
+
494
+ encode = self._list_encoder()
495
+ root = ET.Element("ListBucketResult", xmlns=S3_NAMESPACE)
496
+ sub_element(root, "Name", bucket)
497
+ sub_element(root, "Prefix", encode(prefix))
498
+ if delimiter:
499
+ sub_element(root, "Delimiter", encode(delimiter))
500
+ sub_element(root, "MaxKeys", str(max_keys))
501
+ sub_element(root, "KeyCount", str(len(contents) + len(common_prefixes)))
502
+ sub_element(root, "IsTruncated", "true" if truncated else "false")
503
+ if token is not None:
504
+ sub_element(root, "ContinuationToken", token)
505
+ if truncated:
506
+ next_token = base64.urlsafe_b64encode(last_entry.encode()).decode()
507
+ sub_element(root, "NextContinuationToken", next_token)
508
+ if start_after:
509
+ sub_element(root, "StartAfter", encode(start_after))
510
+ if self.query.get("encoding-type") == "url":
511
+ sub_element(root, "EncodingType", "url")
512
+ self._add_list_entries(root, contents, common_prefixes, encode)
513
+ self._send_xml(root)
514
+
515
+ def _max_keys(self) -> int:
516
+ """Return the max-keys query parameter, capped at the S3 limit of 1000."""
517
+ try:
518
+ max_keys = min(
519
+ int(self.query.get("max-keys", MAX_KEYS_LIMIT)), MAX_KEYS_LIMIT
520
+ )
521
+ except ValueError:
522
+ max_keys = -1
523
+ if max_keys < 0:
524
+ raise S3Error(
525
+ HTTPStatus.BAD_REQUEST,
526
+ "InvalidArgument",
527
+ "Provided max-keys not an integer or within integer range.",
528
+ )
529
+ return max_keys
530
+
531
+ def _list_encoder(self):
532
+ """Return a function that URL-encodes listed names if encoding-type=url."""
533
+ if self.query.get("encoding-type") == "url":
534
+ return lambda text: quote(text, safe="/")
535
+ return lambda text: text
536
+
537
+ def _list_entries(
538
+ self,
539
+ bucket_dir: Path,
540
+ prefix: str,
541
+ delimiter: str,
542
+ marker: str,
543
+ marker_is_prefix: bool,
544
+ max_keys: int,
545
+ ) -> tuple[list[tuple[str, Path]], list[str], bool, str]:
546
+ """Return one page of keys and common prefixes that come after the marker."""
547
+ contents: list[tuple[str, Path]] = []
548
+ common_prefixes: list[str] = []
549
+ last_entry = ""
550
+ truncated = False
551
+ keys = []
552
+ for directory, _, files in os.walk(bucket_dir):
553
+ for name in files:
554
+ if name.startswith(TEMP_PREFIX):
555
+ continue
556
+ path = Path(directory, name)
557
+ keys.append((path.relative_to(bucket_dir).as_posix(), path))
558
+ keys.sort(key=lambda item: item[0].encode())
559
+ for key, path in keys:
560
+ if not key.startswith(prefix):
561
+ continue
562
+ if marker and (
563
+ key.encode() <= marker.encode()
564
+ or (marker_is_prefix and key.startswith(marker))
565
+ ):
566
+ continue
567
+ common = ""
568
+ if delimiter:
569
+ index = key.find(delimiter, len(prefix))
570
+ if index >= 0:
571
+ common = key[: index + len(delimiter)]
572
+ if common and common_prefixes and common_prefixes[-1] == common:
573
+ continue
574
+ if len(contents) + len(common_prefixes) >= max_keys:
575
+ truncated = max_keys > 0
576
+ break
577
+ if common:
578
+ common_prefixes.append(common)
579
+ last_entry = "P:" + common
580
+ else:
581
+ contents.append((key, path))
582
+ last_entry = "K:" + key
583
+ return contents, common_prefixes, truncated, last_entry
584
+
585
+ def _add_list_entries(
586
+ self,
587
+ root: ET.Element,
588
+ contents: list[tuple[str, Path]],
589
+ common_prefixes: list[str],
590
+ encode,
591
+ ):
592
+ """Add Contents and CommonPrefixes elements to a listing result."""
593
+ for key, path in contents:
594
+ item = ET.SubElement(root, "Contents")
595
+ stat = path.stat()
596
+ sub_element(item, "Key", encode(key))
597
+ sub_element(item, "LastModified", iso_timestamp(stat.st_mtime))
598
+ sub_element(item, "ETag", f'"{object_etag(path)}"')
599
+ sub_element(item, "Size", str(stat.st_size))
600
+ sub_element(item, "StorageClass", "STANDARD")
601
+ for common in common_prefixes:
602
+ item = ET.SubElement(root, "CommonPrefixes")
603
+ sub_element(item, "Prefix", encode(common))
604
+
605
+ def _delete_object(self, bucket: str, key: str):
606
+ """Delete an object if it exists and remove any folders left empty."""
607
+ bucket_dir = self._bucket_dir(bucket).resolve()
608
+ target = self._object_path(bucket, key)
609
+ if target.is_file():
610
+ target.unlink(missing_ok=True)
611
+ parent = target.parent
612
+ while parent != bucket_dir:
613
+ try:
614
+ parent.rmdir()
615
+ except OSError:
616
+ break
617
+ parent = parent.parent
618
+ self._send(HTTPStatus.NO_CONTENT)
619
+
620
+ def _delete_bucket(self, bucket: str):
621
+ """Delete an empty bucket, or raise BucketNotEmpty if it holds any files."""
622
+ bucket_dir = self._bucket_dir(bucket)
623
+ not_empty = S3Error(
624
+ HTTPStatus.CONFLICT,
625
+ "BucketNotEmpty",
626
+ "The bucket you tried to delete is not empty",
627
+ )
628
+ directories = []
629
+ for directory, _, files in os.walk(bucket_dir, topdown=False):
630
+ if any(not name.startswith(TEMP_PREFIX) for name in files):
631
+ raise not_empty
632
+ directories.append(directory)
633
+ for directory in directories:
634
+ try:
635
+ os.rmdir(directory)
636
+ except OSError:
637
+ raise not_empty from None
638
+ self._delete_bucket_uploads(bucket)
639
+ self._send(HTTPStatus.NO_CONTENT)
640
+ log_buckets(self.data_root)
641
+
642
+ def _delete_bucket_uploads(self, bucket: str):
643
+ """Remove the unfinished multipart uploads that belong to a bucket."""
644
+ uploads_root = self.data_root / UPLOADS_DIR
645
+ if not uploads_root.is_dir():
646
+ return
647
+ for upload_dir in uploads_root.iterdir():
648
+ try:
649
+ meta = json.loads((upload_dir / UPLOAD_META).read_text())
650
+ except (OSError, ValueError):
651
+ continue
652
+ if isinstance(meta, dict) and meta.get("bucket") == bucket:
653
+ shutil.rmtree(upload_dir, ignore_errors=True)
654
+
655
+ def _put(self, bucket: str, key: str):
656
+ """Route PUT to CreateBucket, PutObject, CopyObject or UploadPart(Copy)."""
657
+ if not bucket:
658
+ raise self._not_implemented_error()
659
+ copy_source = "x-amz-copy-source" in self.headers
660
+ if key and ("uploadId" in self.query or "partNumber" in self.query):
661
+ self._check_query(
662
+ "UploadPartCopy" if copy_source else "UploadPart",
663
+ {"uploadId", "partNumber"},
664
+ )
665
+ body = self._read_body()
666
+ if copy_source:
667
+ self._upload_part_copy(bucket, key)
668
+ else:
669
+ etag = self._store_part(
670
+ bucket, key, lambda file: write_bytes(file, body)
671
+ )
672
+ self._send(HTTPStatus.OK, headers={"ETag": f'"{etag}"'})
673
+ return
674
+ self._check_query(
675
+ ("CopyObject" if copy_source else "PutObject") if key else "CreateBucket"
676
+ )
677
+ body = self._read_body()
678
+ if not key:
679
+ self._create_bucket(bucket)
680
+ elif copy_source:
681
+ self._copy_object(bucket, key)
682
+ else:
683
+ etag = self._write_object(bucket, key, lambda file: write_bytes(file, body))
684
+ self._send(HTTPStatus.OK, headers={"ETag": f'"{etag}"'})
685
+
686
+ def _create_bucket(self, bucket: str):
687
+ """Create the bucket directory under the data root."""
688
+ if not BUCKET_NAME_RE.match(bucket) or ".." in bucket:
689
+ raise S3Error(
690
+ HTTPStatus.BAD_REQUEST,
691
+ "InvalidBucketName",
692
+ "The specified bucket is not valid.",
693
+ )
694
+ bucket_dir = self.data_root / bucket
695
+ try:
696
+ bucket_dir.mkdir()
697
+ except FileExistsError:
698
+ raise S3Error(
699
+ HTTPStatus.CONFLICT,
700
+ "BucketAlreadyOwnedByYou",
701
+ "Your previous request to create the named bucket succeeded and you "
702
+ "already own it.",
703
+ ) from None
704
+ self._send(HTTPStatus.OK, headers={"Location": f"/{bucket}"})
705
+ log_buckets(self.data_root)
706
+
707
+ def _copy_object(self, bucket: str, key: str):
708
+ """Copy the object named in the x-amz-copy-source header to this key."""
709
+ source = self._copy_source()
710
+ target = self._object_path(bucket, key)
711
+ directive = self.headers.get("x-amz-metadata-directive", "COPY").strip().upper()
712
+ if source == target and directive != "REPLACE":
713
+ raise S3Error(
714
+ HTTPStatus.BAD_REQUEST,
715
+ "InvalidRequest",
716
+ "This copy request is illegal because it is trying to copy an object "
717
+ "to itself without changing the object's metadata, storage class, "
718
+ "website redirect location or encryption attributes.",
719
+ )
720
+ ranges = [(source, 0, source.stat().st_size)]
721
+ etag = self._write_object(bucket, key, lambda file: copy_ranges(file, ranges))
722
+ root = ET.Element("CopyObjectResult", xmlns=S3_NAMESPACE)
723
+ sub_element(root, "LastModified", iso_timestamp(target.stat().st_mtime))
724
+ sub_element(root, "ETag", f'"{etag}"')
725
+ self._send_xml(root)
726
+
727
+ def _copy_source(self) -> Path:
728
+ """Return the file named by x-amz-copy-source, rejecting unsupported options."""
729
+ source, _, source_query = self.headers["x-amz-copy-source"].partition("?")
730
+ if source_query:
731
+ raise self._not_implemented_error()
732
+ bucket, _, key = unquote(source).lstrip("/").partition("/")
733
+ if not bucket or not key:
734
+ raise S3Error(
735
+ HTTPStatus.BAD_REQUEST,
736
+ "InvalidArgument",
737
+ "Copy Source must mention the source bucket and key: "
738
+ "sourcebucket/sourcekey",
739
+ )
740
+ path = self._object_path(bucket, key)
741
+ if not path.is_file():
742
+ raise S3Error(
743
+ HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
744
+ )
745
+ self._check_copy_conditions(path)
746
+ return path
747
+
748
+ def _check_copy_conditions(self, source: Path):
749
+ """Raise PreconditionFailed unless the x-amz-copy-source-if-* headers hold."""
750
+ if_match = self.headers.get("x-amz-copy-source-if-match")
751
+ if_none_match = self.headers.get("x-amz-copy-source-if-none-match")
752
+ modified_since = http_date(
753
+ self.headers.get("x-amz-copy-source-if-modified-since")
754
+ )
755
+ unmodified_since = http_date(
756
+ self.headers.get("x-amz-copy-source-if-unmodified-since")
757
+ )
758
+ etag = object_etag(source)
759
+ modified = int(source.stat().st_mtime)
760
+ if if_match is not None:
761
+ if not etag_matches(if_match, etag):
762
+ raise self._precondition_failed_error()
763
+ elif unmodified_since is not None and modified > unmodified_since:
764
+ raise self._precondition_failed_error()
765
+ if if_none_match is not None:
766
+ if etag_matches(if_none_match, etag):
767
+ raise self._precondition_failed_error()
768
+ elif modified_since is not None and modified <= modified_since:
769
+ raise self._precondition_failed_error()
770
+
771
+ def _create_multipart_upload(self, bucket: str, key: str):
772
+ """Start a multipart upload and send its upload ID."""
773
+ self._object_path(bucket, key)
774
+ upload_id = uuid.uuid4().hex
775
+ upload_dir = self.data_root / UPLOADS_DIR / upload_id
776
+ upload_dir.mkdir(parents=True)
777
+ (upload_dir / UPLOAD_META).write_text(
778
+ json.dumps({"bucket": bucket, "key": key})
779
+ )
780
+ root = ET.Element("InitiateMultipartUploadResult", xmlns=S3_NAMESPACE)
781
+ sub_element(root, "Bucket", bucket)
782
+ sub_element(root, "Key", key)
783
+ sub_element(root, "UploadId", upload_id)
784
+ self._send_xml(root)
785
+
786
+ def _upload_part_copy(self, bucket: str, key: str):
787
+ """Store all or part of an existing object as one part of a multipart upload."""
788
+ source = self._copy_source()
789
+ size = source.stat().st_size
790
+ header = self.headers.get("x-amz-copy-source-range")
791
+ if header is None:
792
+ start, length = 0, size
793
+ else:
794
+ match = RANGE_RE.match(header.strip())
795
+ if (
796
+ not match
797
+ or not all(match.groups())
798
+ or int(match[1]) > int(match[2])
799
+ or int(match[2]) >= size
800
+ ):
801
+ raise S3Error(
802
+ HTTPStatus.BAD_REQUEST,
803
+ "InvalidArgument",
804
+ f"Range specified is not valid for source object of size: {size}",
805
+ )
806
+ start, length = int(match[1]), int(match[2]) - int(match[1]) + 1
807
+ etag = self._store_part(
808
+ bucket, key, lambda file: copy_ranges(file, [(source, start, length)])
809
+ )
810
+ root = ET.Element("CopyPartResult", xmlns=S3_NAMESPACE)
811
+ sub_element(root, "LastModified", iso_timestamp(time.time()))
812
+ sub_element(root, "ETag", f'"{etag}"')
813
+ self._send_xml(root)
814
+
815
+ def _complete_multipart_upload(self, bucket: str, key: str, body: bytes):
816
+ """Join the listed parts into the final object and delete the upload."""
817
+ upload_dir = self._upload_dir(bucket, key)
818
+ try:
819
+ parts = ET.fromstring(body).findall("{*}Part")
820
+ except ET.ParseError:
821
+ parts = []
822
+ if not parts:
823
+ raise self._malformed_xml_error()
824
+ ranges = []
825
+ previous = 0
826
+ for part in parts:
827
+ try:
828
+ number = int(part.findtext("{*}PartNumber") or "")
829
+ except ValueError:
830
+ raise self._malformed_xml_error() from None
831
+ if number <= previous:
832
+ raise S3Error(
833
+ HTTPStatus.BAD_REQUEST,
834
+ "InvalidPartOrder",
835
+ "The list of parts was not in ascending order. The parts list "
836
+ "must be specified in order by part number.",
837
+ )
838
+ previous = number
839
+ path = upload_dir / f"{number:05d}"
840
+ etag = (part.findtext("{*}ETag") or "").strip().strip('"')
841
+ if not path.is_file() or etag != object_etag(path):
842
+ raise S3Error(
843
+ HTTPStatus.BAD_REQUEST,
844
+ "InvalidPart",
845
+ "One or more of the specified parts could not be found. The part "
846
+ "may not have been uploaded, or the specified entity tag may not "
847
+ "match the part's entity tag.",
848
+ )
849
+ ranges.append((path, 0, path.stat().st_size))
850
+ if any(length < MIN_PART_SIZE for _, _, length in ranges[:-1]):
851
+ raise S3Error(
852
+ HTTPStatus.BAD_REQUEST,
853
+ "EntityTooSmall",
854
+ "Your proposed upload is smaller than the minimum allowed object size.",
855
+ )
856
+ etag = self._write_object(bucket, key, lambda file: copy_ranges(file, ranges))
857
+ shutil.rmtree(upload_dir, ignore_errors=True)
858
+ root = ET.Element("CompleteMultipartUploadResult", xmlns=S3_NAMESPACE)
859
+ location = f"http://{self.headers.get('Host', '')}/{quote(bucket)}/{quote(key)}"
860
+ sub_element(root, "Location", location)
861
+ sub_element(root, "Bucket", bucket)
862
+ sub_element(root, "Key", key)
863
+ sub_element(root, "ETag", f'"{etag}"')
864
+ self._send_xml(root)
865
+
866
+ def _upload_dir(self, bucket: str, key: str) -> Path:
867
+ """Return the uploadId's folder, checking the upload belongs to this key."""
868
+ self._bucket_dir(bucket)
869
+ upload_id = self.query.get("uploadId", "")
870
+ upload_dir = self.data_root / UPLOADS_DIR / upload_id
871
+ meta = None
872
+ if UPLOAD_ID_RE.match(upload_id):
873
+ try:
874
+ meta = json.loads((upload_dir / UPLOAD_META).read_text())
875
+ except (OSError, ValueError):
876
+ meta = None
877
+ if (
878
+ not isinstance(meta, dict)
879
+ or meta.get("bucket") != bucket
880
+ or meta.get("key") != key
881
+ ):
882
+ raise self._no_such_upload_error()
883
+ return upload_dir
884
+
885
+ def _store_part(self, bucket: str, key: str, write) -> str:
886
+ """Write one multipart upload part via a temporary file and return its ETag."""
887
+ upload_dir = self._upload_dir(bucket, key)
888
+ try:
889
+ number = int(self.query.get("partNumber", ""))
890
+ except ValueError:
891
+ number = 0
892
+ if not 1 <= number <= MAX_PART_NUMBER:
893
+ raise S3Error(
894
+ HTTPStatus.BAD_REQUEST,
895
+ "InvalidArgument",
896
+ f"Part number must be an integer between 1 and {MAX_PART_NUMBER}, "
897
+ "inclusive",
898
+ )
899
+ part_path = upload_dir / f"{number:05d}"
900
+ try:
901
+ fd, tmp_name = tempfile.mkstemp(dir=upload_dir, prefix=TEMP_PREFIX)
902
+ except FileNotFoundError:
903
+ raise self._no_such_upload_error() from None
904
+ try:
905
+ with os.fdopen(fd, "wb") as tmp:
906
+ etag = write(tmp)
907
+ try:
908
+ os.replace(tmp_name, part_path)
909
+ except FileNotFoundError:
910
+ raise self._no_such_upload_error() from None
911
+ except BaseException:
912
+ Path(tmp_name).unlink(missing_ok=True)
913
+ raise
914
+ remember_etag(part_path, etag)
915
+ return etag
916
+
917
+ def _write_object(self, bucket: str, key: str, write) -> str:
918
+ """Write an object honouring If-None-Match and return its ETag."""
919
+ target = self._object_path(bucket, key)
920
+ if_none_match = self.headers.get("If-None-Match", "").strip() == "*"
921
+ if if_none_match and target.exists():
922
+ raise self._precondition_failed_error()
923
+ target.parent.mkdir(parents=True, exist_ok=True)
924
+ fd, tmp_name = tempfile.mkstemp(dir=target.parent, prefix=TEMP_PREFIX)
925
+ try:
926
+ with os.fdopen(fd, "wb") as tmp:
927
+ etag = write(tmp)
928
+ if if_none_match:
929
+ try:
930
+ os.link(tmp_name, target)
931
+ except FileExistsError:
932
+ raise self._precondition_failed_error() from None
933
+ os.unlink(tmp_name)
934
+ else:
935
+ os.replace(tmp_name, target)
936
+ except BaseException:
937
+ Path(tmp_name).unlink(missing_ok=True)
938
+ raise
939
+ remember_etag(target, etag)
940
+ return etag
941
+
942
+ def _get(self, bucket: str, key: str, include_body: bool):
943
+ """Send the object's content and metadata, or just the requested byte range."""
944
+ if not bucket or not key:
945
+ raise self._not_implemented_error()
946
+ overrides = self._response_header_overrides()
947
+ target = self._object_path(bucket, key)
948
+ if not target.is_file():
949
+ raise S3Error(
950
+ HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
951
+ )
952
+ try:
953
+ file = target.open("rb")
954
+ except FileNotFoundError:
955
+ raise S3Error(
956
+ HTTPStatus.NOT_FOUND, "NoSuchKey", "The specified key does not exist."
957
+ ) from None
958
+ with file:
959
+ stat = os.fstat(file.fileno())
960
+ content_type = (
961
+ mimetypes.guess_type(target.name)[0] or "application/octet-stream"
962
+ )
963
+ headers = {
964
+ "Content-Type": content_type,
965
+ "ETag": f'"{object_etag(target)}"',
966
+ "Last-Modified": formatdate(stat.st_mtime, usegmt=True),
967
+ "Accept-Ranges": "bytes",
968
+ **overrides,
969
+ }
970
+ status, start, length = HTTPStatus.OK, 0, stat.st_size
971
+ byte_range = parse_range(self.headers.get("Range", ""), stat.st_size)
972
+ if byte_range:
973
+ start, end = byte_range
974
+ status, length = HTTPStatus.PARTIAL_CONTENT, end - start + 1
975
+ headers["Content-Range"] = f"bytes {start}-{end}/{stat.st_size}"
976
+ self._send_file(status, file, start, length, headers, include_body)
977
+
978
+ def _response_header_overrides(self) -> dict[str, str]:
979
+ """Return the headers that response-* query parameters ask to override."""
980
+ overrides = {}
981
+ for name, header in RESPONSE_HEADER_OVERRIDES.items():
982
+ value = self.query.get(name)
983
+ if value is None:
984
+ continue
985
+ if any(ord(char) < 32 or ord(char) == 127 for char in value):
986
+ raise S3Error(
987
+ HTTPStatus.BAD_REQUEST,
988
+ "InvalidArgument",
989
+ f"Invalid value for {name}.",
990
+ )
991
+ overrides[header] = value
992
+ return overrides
993
+
994
+ def _not_implemented_error(self) -> S3Error:
995
+ """Build the error returned for unsupported operations."""
996
+ return S3Error(
997
+ HTTPStatus.NOT_IMPLEMENTED,
998
+ "NotImplemented",
999
+ "A header or operation you provided implies functionality that is not "
1000
+ "implemented.",
1001
+ )
1002
+
1003
+ def _precondition_failed_error(self) -> S3Error:
1004
+ """Build the error for a conditional write that finds an existing object."""
1005
+ return S3Error(
1006
+ HTTPStatus.PRECONDITION_FAILED,
1007
+ "PreconditionFailed",
1008
+ "At least one of the pre-conditions you specified did not hold",
1009
+ )
1010
+
1011
+ def _malformed_xml_error(self) -> S3Error:
1012
+ """Build the error for a request body that is not the expected XML."""
1013
+ return S3Error(
1014
+ HTTPStatus.BAD_REQUEST,
1015
+ "MalformedXML",
1016
+ "The XML you provided was not well-formed or did not validate against "
1017
+ "our published schema",
1018
+ )
1019
+
1020
+ def _no_such_upload_error(self) -> S3Error:
1021
+ """Build the error for an upload ID with no active upload for this key."""
1022
+ return S3Error(
1023
+ HTTPStatus.NOT_FOUND,
1024
+ "NoSuchUpload",
1025
+ "The specified upload does not exist. The upload ID may be invalid, or "
1026
+ "the upload may have been aborted or completed.",
1027
+ )
1028
+
1029
+ def _bucket_dir(self, bucket: str) -> Path:
1030
+ """Return a bucket's directory, raising NoSuchBucket if it does not exist."""
1031
+ bucket_dir = self.data_root / bucket
1032
+ if not BUCKET_NAME_RE.match(bucket) or not bucket_dir.is_dir():
1033
+ raise S3Error(
1034
+ HTTPStatus.NOT_FOUND,
1035
+ "NoSuchBucket",
1036
+ "The specified bucket does not exist.",
1037
+ )
1038
+ return bucket_dir
1039
+
1040
+ def _object_path(self, bucket: str, key: str) -> Path:
1041
+ """Resolve a key's path, checking the bucket exists and the key stays inside."""
1042
+ bucket_dir = self._bucket_dir(bucket)
1043
+ parts = key.split("/")
1044
+ if key.endswith("/") or any(part in ("", ".", "..") for part in parts):
1045
+ raise S3Error(
1046
+ HTTPStatus.BAD_REQUEST,
1047
+ "InvalidArgument",
1048
+ "This server does not support this key.",
1049
+ )
1050
+ target = bucket_dir.joinpath(*parts).resolve()
1051
+ if not target.is_relative_to(bucket_dir.resolve()):
1052
+ raise S3Error(
1053
+ HTTPStatus.BAD_REQUEST,
1054
+ "InvalidArgument",
1055
+ "This server does not support this key.",
1056
+ )
1057
+ return target
1058
+
1059
+ def _read_body(self) -> bytes:
1060
+ """Read the request body, decoding HTTP chunked and aws-chunked encodings."""
1061
+ if "chunked" in self.headers.get("Transfer-Encoding", "").lower():
1062
+ body = read_chunked(self.rfile)
1063
+ else:
1064
+ length = int(self.headers.get("Content-Length") or 0)
1065
+ body = self.rfile.read(length)
1066
+ content_sha = self.headers.get("x-amz-content-sha256", "")
1067
+ content_encoding = self.headers.get("Content-Encoding", "")
1068
+ if content_sha.startswith("STREAMING-") or "aws-chunked" in content_encoding:
1069
+ body = read_chunked(io.BytesIO(body))
1070
+ return body
1071
+
1072
+ def _send_headers(self, status: HTTPStatus, length: int, headers=None):
1073
+ """Send the status line and headers, including the standard S3 ones."""
1074
+ self.send_response(status)
1075
+ self.send_header("x-amz-request-id", uuid.uuid4().hex.upper()[:16])
1076
+ self.send_header("Content-Length", str(length))
1077
+ origin = self.headers.get("Origin")
1078
+ if origin:
1079
+ self.send_header("Access-Control-Allow-Origin", origin)
1080
+ self.send_header("Access-Control-Expose-Headers", CORS_EXPOSE_HEADERS)
1081
+ self.send_header("Vary", "Origin")
1082
+ for name, value in (headers or {}).items():
1083
+ self.send_header(name, value)
1084
+ self.end_headers()
1085
+
1086
+ def _send(
1087
+ self,
1088
+ status: HTTPStatus,
1089
+ body: bytes = b"",
1090
+ headers=None,
1091
+ include_body: bool = True,
1092
+ ):
1093
+ """Send a response with the standard S3 headers."""
1094
+ self._send_headers(status, len(body), headers)
1095
+ if include_body and body:
1096
+ self.wfile.write(body)
1097
+
1098
+ def _send_file(
1099
+ self,
1100
+ status: HTTPStatus,
1101
+ file,
1102
+ start: int,
1103
+ length: int,
1104
+ headers,
1105
+ include_body: bool,
1106
+ ):
1107
+ """Send a response whose body is a byte range of an open file."""
1108
+ self._send_headers(status, length, headers)
1109
+ if not include_body:
1110
+ return
1111
+ file.seek(start)
1112
+ remaining = length
1113
+ while remaining > 0:
1114
+ block = file.read(min(remaining, BLOCK_SIZE))
1115
+ if not block:
1116
+ self.close_connection = True
1117
+ return
1118
+ self.wfile.write(block)
1119
+ remaining -= len(block)
1120
+
1121
+ def _send_xml(self, root: ET.Element):
1122
+ """Send a successful XML response."""
1123
+ body = ET.tostring(root, encoding="UTF-8", xml_declaration=True)
1124
+ self._send(HTTPStatus.OK, body, {"Content-Type": "application/xml"})
1125
+
1126
+ def _send_error(self, error: S3Error):
1127
+ """Send an S3-style XML error document."""
1128
+ xml = (
1129
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
1130
+ f"<Error><Code>{error.code}</Code>"
1131
+ f"<Message>{escape(error.message)}</Message>"
1132
+ f"<Resource>{escape(urlsplit(self.path).path)}</Resource></Error>"
1133
+ ).encode()
1134
+ self.close_connection = True
1135
+ self._send(
1136
+ error.status,
1137
+ xml,
1138
+ {"Content-Type": "application/xml", "Connection": "close"},
1139
+ include_body=self.command != "HEAD",
1140
+ )
1141
+
1142
+
1143
+ def main():
1144
+ """Parse the command line and run the server."""
1145
+ parser = argparse.ArgumentParser(
1146
+ prog="localbucket",
1147
+ description=(
1148
+ f"LocalBucket {__version__}: a minimal S3-compatible HTTP server that "
1149
+ f"stores buckets as folders in DATA_ROOT."
1150
+ ),
1151
+ epilog=HELP_EPILOG,
1152
+ formatter_class=argparse.RawDescriptionHelpFormatter,
1153
+ )
1154
+ parser.add_argument(
1155
+ "data_root",
1156
+ metavar="DATA_ROOT",
1157
+ type=Path,
1158
+ help="directory that holds the buckets",
1159
+ )
1160
+ parser.add_argument(
1161
+ "--host", default="127.0.0.1", help="address to listen on (default: 127.0.0.1)"
1162
+ )
1163
+ parser.add_argument(
1164
+ "--port", type=int, default=9000, help="port to listen on (default: 9000)"
1165
+ )
1166
+ parser.add_argument(
1167
+ "--version", action="version", version=f"LocalBucket {__version__}"
1168
+ )
1169
+ if len(sys.argv) == 1:
1170
+ parser.print_help()
1171
+ sys.exit(2)
1172
+ args = parser.parse_args()
1173
+ logging.basicConfig(
1174
+ level=logging.INFO,
1175
+ format="%(asctime)s [%(levelname)s]: %(message)s",
1176
+ datefmt="%Y-%m-%d %H:%M:%S",
1177
+ )
1178
+
1179
+ data_root = args.data_root.resolve()
1180
+ data_root.mkdir(parents=True, exist_ok=True)
1181
+ S3Handler.data_root = data_root
1182
+
1183
+ server = ThreadingHTTPServer((args.host, args.port), S3Handler)
1184
+ print(BANNER, file=sys.stderr)
1185
+ logger.info(
1186
+ "LocalBucket %s serving %s on http://%s:%s",
1187
+ __version__,
1188
+ data_root,
1189
+ args.host,
1190
+ args.port,
1191
+ )
1192
+ log_buckets(data_root)
1193
+ try:
1194
+ server.serve_forever()
1195
+ except KeyboardInterrupt:
1196
+ pass
1197
+ finally:
1198
+ server.server_close()
1199
+
1200
+
1201
+ if __name__ == "__main__":
1202
+ main()