maxminddb 3.2.0__cp313-cp313-android_24_arm64_v8a.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- maxminddb/__init__.py +91 -0
- maxminddb/const.py +47 -0
- maxminddb/decoder.py +356 -0
- maxminddb/errors.py +5 -0
- maxminddb/extension.cpython-313-aarch64-linux-android.so +0 -0
- maxminddb/extension.pyi +117 -0
- maxminddb/file.py +74 -0
- maxminddb/py.typed +0 -0
- maxminddb/reader.py +383 -0
- maxminddb/types.py +15 -0
- maxminddb-3.2.0.dist-info/METADATA +154 -0
- maxminddb-3.2.0.dist-info/RECORD +15 -0
- maxminddb-3.2.0.dist-info/WHEEL +5 -0
- maxminddb-3.2.0.dist-info/licenses/LICENSE +202 -0
- maxminddb-3.2.0.dist-info/top_level.txt +1 -0
maxminddb/__init__.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Module for reading MaxMind DB files."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from importlib.metadata import version
|
|
6
|
+
from typing import IO, TYPE_CHECKING, AnyStr, cast
|
|
7
|
+
|
|
8
|
+
from .const import (
|
|
9
|
+
MODE_AUTO,
|
|
10
|
+
MODE_FD,
|
|
11
|
+
MODE_FILE,
|
|
12
|
+
MODE_MEMORY,
|
|
13
|
+
MODE_MMAP,
|
|
14
|
+
MODE_MMAP_EXT,
|
|
15
|
+
)
|
|
16
|
+
from .decoder import InvalidDatabaseError
|
|
17
|
+
from .reader import Reader
|
|
18
|
+
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
import os
|
|
21
|
+
|
|
22
|
+
try:
|
|
23
|
+
from . import extension as _extension
|
|
24
|
+
except ImportError:
|
|
25
|
+
_extension = None # type: ignore[assignment]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"MODE_AUTO",
|
|
30
|
+
"MODE_FD",
|
|
31
|
+
"MODE_FILE",
|
|
32
|
+
"MODE_MEMORY",
|
|
33
|
+
"MODE_MMAP",
|
|
34
|
+
"MODE_MMAP_EXT",
|
|
35
|
+
"InvalidDatabaseError",
|
|
36
|
+
"Reader",
|
|
37
|
+
"open_database",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def open_database(
|
|
42
|
+
database: AnyStr | int | os.PathLike | IO,
|
|
43
|
+
mode: int = MODE_AUTO,
|
|
44
|
+
) -> Reader:
|
|
45
|
+
"""Open a MaxMind DB database.
|
|
46
|
+
|
|
47
|
+
Arguments:
|
|
48
|
+
database: A path to a valid MaxMind DB file such as a GeoIP database
|
|
49
|
+
file, or a file descriptor in the case of MODE_FD.
|
|
50
|
+
mode: mode to open the database with. Valid mode are:
|
|
51
|
+
* MODE_MMAP_EXT - use the C extension with memory map.
|
|
52
|
+
* MODE_MMAP - read from memory map. Pure Python.
|
|
53
|
+
* MODE_FILE - read database as standard file. Pure Python.
|
|
54
|
+
* MODE_MEMORY - load database into memory. Pure Python.
|
|
55
|
+
* MODE_FD - the param passed via database is a file descriptor, not
|
|
56
|
+
a path. This mode implies MODE_MEMORY.
|
|
57
|
+
* MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that
|
|
58
|
+
order. Default mode.
|
|
59
|
+
|
|
60
|
+
"""
|
|
61
|
+
if mode not in (
|
|
62
|
+
MODE_AUTO,
|
|
63
|
+
MODE_FD,
|
|
64
|
+
MODE_FILE,
|
|
65
|
+
MODE_MEMORY,
|
|
66
|
+
MODE_MMAP,
|
|
67
|
+
MODE_MMAP_EXT,
|
|
68
|
+
):
|
|
69
|
+
msg = f"Unsupported open mode: {mode}"
|
|
70
|
+
raise ValueError(msg)
|
|
71
|
+
|
|
72
|
+
has_extension = _extension and hasattr(_extension, "Reader")
|
|
73
|
+
use_extension = has_extension if mode == MODE_AUTO else mode == MODE_MMAP_EXT
|
|
74
|
+
|
|
75
|
+
if not use_extension:
|
|
76
|
+
return Reader(database, mode)
|
|
77
|
+
|
|
78
|
+
if not has_extension:
|
|
79
|
+
msg = "MODE_MMAP_EXT requires the maxminddb.extension module to be available"
|
|
80
|
+
raise ValueError(
|
|
81
|
+
msg,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
# The C type exposes the same API as the Python Reader, so for type
|
|
85
|
+
# checking purposes, pretend it is one. (Ideally this would be a subclass
|
|
86
|
+
# of, or share a common parent class with, the Python Reader
|
|
87
|
+
# implementation.)
|
|
88
|
+
return cast("Reader", _extension.Reader(database, mode))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
__version__ = version("maxminddb")
|
maxminddb/const.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Constants used in the API."""
|
|
2
|
+
|
|
3
|
+
from enum import IntEnum
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Mode(IntEnum):
|
|
7
|
+
"""Database open modes.
|
|
8
|
+
|
|
9
|
+
These modes control how the MaxMind DB file is opened and read.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
AUTO = 0
|
|
13
|
+
"""Try MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Default mode."""
|
|
14
|
+
|
|
15
|
+
MMAP_EXT = 1
|
|
16
|
+
"""Use the C extension with memory map."""
|
|
17
|
+
|
|
18
|
+
MMAP = 2
|
|
19
|
+
"""Read from memory map. Pure Python."""
|
|
20
|
+
|
|
21
|
+
FILE = 4
|
|
22
|
+
"""Read database as standard file. Pure Python."""
|
|
23
|
+
|
|
24
|
+
MEMORY = 8
|
|
25
|
+
"""Load database into memory. Pure Python."""
|
|
26
|
+
|
|
27
|
+
FD = 16
|
|
28
|
+
"""Database is a file descriptor, not a path. This mode implies MODE_MEMORY."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# Backward compatibility: export both enum members and old-style constants
|
|
32
|
+
MODE_AUTO = Mode.AUTO
|
|
33
|
+
MODE_MMAP_EXT = Mode.MMAP_EXT
|
|
34
|
+
MODE_MMAP = Mode.MMAP
|
|
35
|
+
MODE_FILE = Mode.FILE
|
|
36
|
+
MODE_MEMORY = Mode.MEMORY
|
|
37
|
+
MODE_FD = Mode.FD
|
|
38
|
+
|
|
39
|
+
__all__ = [
|
|
40
|
+
"MODE_AUTO",
|
|
41
|
+
"MODE_FD",
|
|
42
|
+
"MODE_FILE",
|
|
43
|
+
"MODE_MEMORY",
|
|
44
|
+
"MODE_MMAP",
|
|
45
|
+
"MODE_MMAP_EXT",
|
|
46
|
+
"Mode",
|
|
47
|
+
]
|
maxminddb/decoder.py
ADDED
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
"""Decoder for the MaxMind DB data section."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import struct
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import TYPE_CHECKING
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
import mmap
|
|
11
|
+
except ImportError:
|
|
12
|
+
mmap = None # type: ignore[assignment]
|
|
13
|
+
|
|
14
|
+
from maxminddb.errors import InvalidDatabaseError
|
|
15
|
+
|
|
16
|
+
if TYPE_CHECKING:
|
|
17
|
+
from maxminddb.file import FileBuffer
|
|
18
|
+
from maxminddb.types import Record
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# Per-lookup value limit recommended by the MaxMind DB specification. It stops
|
|
22
|
+
# pointer fan-out, where nested containers share targets that would otherwise
|
|
23
|
+
# cost 2**depth decode operations. The root costs one value. Arrays charge each
|
|
24
|
+
# element, maps charge each key and value, and pointers cost no extra value.
|
|
25
|
+
# Real records decode a few hundred values, leaving a wide margin.
|
|
26
|
+
# An explicit depth limit catches container cycles and overly nested data.
|
|
27
|
+
# Python's recursion limit may fire first, which decode converts to the same
|
|
28
|
+
# error. The explicit limit also applies when callers raise Python's limit.
|
|
29
|
+
_MAX_VALUES = 1 << 16
|
|
30
|
+
_MAX_DEPTH = 512
|
|
31
|
+
# Per-lookup limit on the total string and bytes payload materialized, matching
|
|
32
|
+
# libmaxminddb and the Go reader. It stops a payload amplification, where many
|
|
33
|
+
# pointers to one large value would otherwise materialize N * size bytes from a
|
|
34
|
+
# small file. Each string or bytes value is charged its length wherever it is
|
|
35
|
+
# decoded, so re-decoding a shared target through another pointer recharges.
|
|
36
|
+
_MAX_PAYLOAD_BYTES = 1 << 21
|
|
37
|
+
# The widest fixed-width integer the format defines is the 16-byte uint128; a
|
|
38
|
+
# declared size past that is malformed and could copy attacker-controlled bytes.
|
|
39
|
+
_MAX_UINT_BYTES = 16
|
|
40
|
+
_MAX_INT32_BYTES = 4
|
|
41
|
+
# Added to a pointer value, by pointer size. A 4-byte pointer adds nothing.
|
|
42
|
+
_POINTER_VALUE_OFFSETS = (0, 0, 2048, 526336)
|
|
43
|
+
_TOO_MANY_VALUES = (
|
|
44
|
+
"The MaxMind DB file's data section exceeds the maximum number of values"
|
|
45
|
+
)
|
|
46
|
+
_TOO_DEEP = "The MaxMind DB file's data section exceeds the maximum depth"
|
|
47
|
+
_TOO_LARGE = "The MaxMind DB file's data section exceeds the maximum payload size"
|
|
48
|
+
_BAD_DATA = (
|
|
49
|
+
"The MaxMind DB file's data section contains bad data "
|
|
50
|
+
"(unknown data type or corrupt data)"
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class _DecodeBudget:
|
|
56
|
+
"""Shared counters for one record or metadata decode."""
|
|
57
|
+
|
|
58
|
+
values_left: int
|
|
59
|
+
depth: int
|
|
60
|
+
payload_left: int
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class Decoder:
|
|
64
|
+
"""Decoder for the data section of the MaxMind DB."""
|
|
65
|
+
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
database_buffer: FileBuffer | mmap.mmap | bytes,
|
|
69
|
+
pointer_base: int = 0,
|
|
70
|
+
pointer_test: bool = False, # noqa: FBT001, FBT002
|
|
71
|
+
) -> None:
|
|
72
|
+
"""Create a Decoder for a MaxMind DB.
|
|
73
|
+
|
|
74
|
+
Arguments:
|
|
75
|
+
database_buffer: an mmap'd MaxMind DB file.
|
|
76
|
+
pointer_base: the base number to use when decoding a pointer
|
|
77
|
+
pointer_test: used for internal unit testing of pointer code
|
|
78
|
+
|
|
79
|
+
"""
|
|
80
|
+
self._pointer_test = pointer_test
|
|
81
|
+
self._buffer = database_buffer
|
|
82
|
+
self._pointer_base = pointer_base
|
|
83
|
+
|
|
84
|
+
def _decode_array(
|
|
85
|
+
self,
|
|
86
|
+
size: int,
|
|
87
|
+
offset: int,
|
|
88
|
+
budget: _DecodeBudget,
|
|
89
|
+
) -> tuple[list[Record], int]:
|
|
90
|
+
remaining = budget.values_left - size
|
|
91
|
+
if remaining < 0:
|
|
92
|
+
raise InvalidDatabaseError(_TOO_MANY_VALUES)
|
|
93
|
+
budget.values_left = remaining
|
|
94
|
+
depth = budget.depth + 1
|
|
95
|
+
if depth > _MAX_DEPTH:
|
|
96
|
+
raise InvalidDatabaseError(_TOO_DEEP)
|
|
97
|
+
budget.depth = depth
|
|
98
|
+
array = []
|
|
99
|
+
decode = self._decode
|
|
100
|
+
for _ in range(size):
|
|
101
|
+
(value, offset) = decode(offset, budget, False) # noqa: FBT003
|
|
102
|
+
array.append(value)
|
|
103
|
+
budget.depth -= 1
|
|
104
|
+
return array, offset
|
|
105
|
+
|
|
106
|
+
def _decode_boolean(
|
|
107
|
+
self,
|
|
108
|
+
size: int,
|
|
109
|
+
offset: int,
|
|
110
|
+
_budget: _DecodeBudget,
|
|
111
|
+
) -> tuple[bool, int]:
|
|
112
|
+
return size != 0, offset
|
|
113
|
+
|
|
114
|
+
def _decode_bytes(
|
|
115
|
+
self,
|
|
116
|
+
size: int,
|
|
117
|
+
offset: int,
|
|
118
|
+
budget: _DecodeBudget,
|
|
119
|
+
) -> tuple[bytes, int]:
|
|
120
|
+
# Charge the payload before copying so a crafted size cannot force a
|
|
121
|
+
# large allocation, and so pointers reusing one target recharge.
|
|
122
|
+
remaining = budget.payload_left - size
|
|
123
|
+
if remaining < 0:
|
|
124
|
+
raise InvalidDatabaseError(_TOO_LARGE)
|
|
125
|
+
budget.payload_left = remaining
|
|
126
|
+
new_offset = offset + size
|
|
127
|
+
return self._buffer[offset:new_offset], new_offset
|
|
128
|
+
|
|
129
|
+
def _decode_double(
|
|
130
|
+
self,
|
|
131
|
+
size: int,
|
|
132
|
+
offset: int,
|
|
133
|
+
_budget: _DecodeBudget,
|
|
134
|
+
) -> tuple[float, int]:
|
|
135
|
+
self._verify_size(size, 8)
|
|
136
|
+
new_offset = offset + size
|
|
137
|
+
packed_bytes = self._buffer[offset:new_offset]
|
|
138
|
+
(value,) = struct.unpack(b"!d", packed_bytes)
|
|
139
|
+
return value, new_offset
|
|
140
|
+
|
|
141
|
+
def _decode_float(
|
|
142
|
+
self,
|
|
143
|
+
size: int,
|
|
144
|
+
offset: int,
|
|
145
|
+
_budget: _DecodeBudget,
|
|
146
|
+
) -> tuple[float, int]:
|
|
147
|
+
self._verify_size(size, 4)
|
|
148
|
+
new_offset = offset + size
|
|
149
|
+
packed_bytes = self._buffer[offset:new_offset]
|
|
150
|
+
(value,) = struct.unpack(b"!f", packed_bytes)
|
|
151
|
+
return value, new_offset
|
|
152
|
+
|
|
153
|
+
def _decode_int32(
|
|
154
|
+
self,
|
|
155
|
+
size: int,
|
|
156
|
+
offset: int,
|
|
157
|
+
_budget: _DecodeBudget,
|
|
158
|
+
) -> tuple[int, int]:
|
|
159
|
+
if size > _MAX_INT32_BYTES:
|
|
160
|
+
raise InvalidDatabaseError(_BAD_DATA)
|
|
161
|
+
if size == 0:
|
|
162
|
+
return 0, offset
|
|
163
|
+
new_offset = offset + size
|
|
164
|
+
packed_bytes = self._buffer[offset:new_offset]
|
|
165
|
+
|
|
166
|
+
if size != 4:
|
|
167
|
+
packed_bytes = packed_bytes.rjust(4, b"\x00")
|
|
168
|
+
(value,) = struct.unpack(b"!i", packed_bytes)
|
|
169
|
+
return value, new_offset
|
|
170
|
+
|
|
171
|
+
def _decode_map(
|
|
172
|
+
self,
|
|
173
|
+
size: int,
|
|
174
|
+
offset: int,
|
|
175
|
+
budget: _DecodeBudget,
|
|
176
|
+
) -> tuple[dict[str, Record], int]:
|
|
177
|
+
# A map entry decodes a key and a value, so it costs two values.
|
|
178
|
+
remaining = budget.values_left - size * 2
|
|
179
|
+
if remaining < 0:
|
|
180
|
+
raise InvalidDatabaseError(_TOO_MANY_VALUES)
|
|
181
|
+
budget.values_left = remaining
|
|
182
|
+
depth = budget.depth + 1
|
|
183
|
+
if depth > _MAX_DEPTH:
|
|
184
|
+
raise InvalidDatabaseError(_TOO_DEEP)
|
|
185
|
+
budget.depth = depth
|
|
186
|
+
container: dict[str, Record] = {}
|
|
187
|
+
decode = self._decode
|
|
188
|
+
for _ in range(size):
|
|
189
|
+
(key, offset) = decode(offset, budget, False) # noqa: FBT003
|
|
190
|
+
(value, offset) = decode(offset, budget, False) # noqa: FBT003
|
|
191
|
+
container[key] = value # type: ignore[index]
|
|
192
|
+
budget.depth -= 1
|
|
193
|
+
return container, offset
|
|
194
|
+
|
|
195
|
+
def _decode_pointer(
|
|
196
|
+
self,
|
|
197
|
+
size: int,
|
|
198
|
+
offset: int,
|
|
199
|
+
budget: _DecodeBudget,
|
|
200
|
+
) -> tuple[Record, int]:
|
|
201
|
+
pointer_size = (size >> 3) + 1
|
|
202
|
+
new_offset = offset + pointer_size
|
|
203
|
+
pointer_bytes = self._buffer[offset:new_offset]
|
|
204
|
+
if len(pointer_bytes) != pointer_size:
|
|
205
|
+
raise InvalidDatabaseError(_BAD_DATA)
|
|
206
|
+
pointer = int.from_bytes(pointer_bytes, "big")
|
|
207
|
+
if pointer_size < 4:
|
|
208
|
+
# The low three bits of the ctrl byte are the high bits of the
|
|
209
|
+
# pointer, and sizes 2 and 3 add a fixed offset.
|
|
210
|
+
pointer |= (size & 0x7) << (pointer_size << 3)
|
|
211
|
+
pointer += _POINTER_VALUE_OFFSETS[pointer_size]
|
|
212
|
+
pointer += self._pointer_base
|
|
213
|
+
|
|
214
|
+
if self._pointer_test:
|
|
215
|
+
return pointer, new_offset
|
|
216
|
+
|
|
217
|
+
# The value at the pointer's position was charged by its containing
|
|
218
|
+
# array or map, so the target costs nothing more. Only the depth changes.
|
|
219
|
+
depth = budget.depth + 1
|
|
220
|
+
if depth > _MAX_DEPTH:
|
|
221
|
+
raise InvalidDatabaseError(_TOO_DEEP)
|
|
222
|
+
budget.depth = depth
|
|
223
|
+
(value, _) = self._decode(pointer, budget, True) # noqa: FBT003
|
|
224
|
+
budget.depth -= 1
|
|
225
|
+
return value, new_offset
|
|
226
|
+
|
|
227
|
+
def _decode_uint(
|
|
228
|
+
self,
|
|
229
|
+
size: int,
|
|
230
|
+
offset: int,
|
|
231
|
+
_budget: _DecodeBudget,
|
|
232
|
+
) -> tuple[int, int]:
|
|
233
|
+
# Reject a declared size past the widest defined unsigned integer before
|
|
234
|
+
# copying, so a crafted size cannot force a large allocation.
|
|
235
|
+
if size > _MAX_UINT_BYTES:
|
|
236
|
+
raise InvalidDatabaseError(_BAD_DATA)
|
|
237
|
+
new_offset = offset + size
|
|
238
|
+
uint_bytes = self._buffer[offset:new_offset]
|
|
239
|
+
return int.from_bytes(uint_bytes, "big"), new_offset
|
|
240
|
+
|
|
241
|
+
def decode(self, offset: int) -> tuple[Record, int]:
|
|
242
|
+
"""Decode a section of the data section starting at offset.
|
|
243
|
+
|
|
244
|
+
Arguments:
|
|
245
|
+
offset: the location of the data structure to decode
|
|
246
|
+
|
|
247
|
+
"""
|
|
248
|
+
# Each call gets its own budget, shared by recursive calls. Charge the
|
|
249
|
+
# root here.
|
|
250
|
+
try:
|
|
251
|
+
return self._decode(
|
|
252
|
+
offset,
|
|
253
|
+
_DecodeBudget(_MAX_VALUES - 1, 0, _MAX_PAYLOAD_BYTES),
|
|
254
|
+
False, # noqa: FBT003
|
|
255
|
+
)
|
|
256
|
+
except RecursionError as ex:
|
|
257
|
+
raise InvalidDatabaseError(_TOO_DEEP) from ex
|
|
258
|
+
except (IndexError, struct.error) as ex:
|
|
259
|
+
# Convert failed buffer indexing and fixed-width unpacking.
|
|
260
|
+
raise InvalidDatabaseError(_BAD_DATA) from ex
|
|
261
|
+
|
|
262
|
+
# Keep type dispatch inline to avoid another call for every decoded value.
|
|
263
|
+
# The positional booleans are intentional: keywords and omitted defaults
|
|
264
|
+
# prevented CPython from using its fastest call path in our benchmarks.
|
|
265
|
+
# pointer_target rejects pointers to other pointers.
|
|
266
|
+
def _decode( # noqa: C901, PLR0911, PLR0912
|
|
267
|
+
self,
|
|
268
|
+
offset: int,
|
|
269
|
+
budget: _DecodeBudget,
|
|
270
|
+
pointer_target: bool, # noqa: FBT001
|
|
271
|
+
) -> tuple[Record, int]:
|
|
272
|
+
new_offset = offset + 1
|
|
273
|
+
ctrl_byte = self._buffer[offset]
|
|
274
|
+
type_num = ctrl_byte >> 5
|
|
275
|
+
# Extended type
|
|
276
|
+
if not type_num:
|
|
277
|
+
(type_num, new_offset) = self._read_extended(new_offset)
|
|
278
|
+
|
|
279
|
+
size = ctrl_byte & 0x1F
|
|
280
|
+
# Sizes under 29 are stored in the ctrl byte, and a pointer's size bits
|
|
281
|
+
# are not a size. Skip the call for that common case.
|
|
282
|
+
if size >= 29 and type_num != 1:
|
|
283
|
+
(size, new_offset) = self._size_from_ctrl_byte(size, new_offset)
|
|
284
|
+
# Put common types first to reduce comparisons during real lookups.
|
|
285
|
+
match type_num:
|
|
286
|
+
case 2:
|
|
287
|
+
# Strings are most of the values in a real database. Decode them
|
|
288
|
+
# here to save a method call.
|
|
289
|
+
# Charge the payload before copying so a crafted size cannot force
|
|
290
|
+
# a large allocation, and so pointers reusing one target recharge.
|
|
291
|
+
remaining = budget.payload_left - size
|
|
292
|
+
if remaining < 0:
|
|
293
|
+
raise InvalidDatabaseError(_TOO_LARGE)
|
|
294
|
+
budget.payload_left = remaining
|
|
295
|
+
end = new_offset + size
|
|
296
|
+
return self._buffer[new_offset:end].decode("utf-8"), end
|
|
297
|
+
case 1:
|
|
298
|
+
if pointer_target:
|
|
299
|
+
raise InvalidDatabaseError(_BAD_DATA)
|
|
300
|
+
return self._decode_pointer(size, new_offset, budget)
|
|
301
|
+
case 7:
|
|
302
|
+
return self._decode_map(size, new_offset, budget)
|
|
303
|
+
case 6 | 5 | 9 | 10: # uint32, uint16, uint64, uint128
|
|
304
|
+
return self._decode_uint(size, new_offset, budget)
|
|
305
|
+
case 11:
|
|
306
|
+
return self._decode_array(size, new_offset, budget)
|
|
307
|
+
case 3:
|
|
308
|
+
return self._decode_double(size, new_offset, budget)
|
|
309
|
+
case 4:
|
|
310
|
+
return self._decode_bytes(size, new_offset, budget)
|
|
311
|
+
case 8:
|
|
312
|
+
return self._decode_int32(size, new_offset, budget)
|
|
313
|
+
case 14:
|
|
314
|
+
return self._decode_boolean(size, new_offset, budget)
|
|
315
|
+
case 15:
|
|
316
|
+
return self._decode_float(size, new_offset, budget)
|
|
317
|
+
case _:
|
|
318
|
+
msg = f"Unexpected type number ({type_num}) encountered"
|
|
319
|
+
raise InvalidDatabaseError(msg)
|
|
320
|
+
|
|
321
|
+
def _read_extended(self, offset: int) -> tuple[int, int]:
|
|
322
|
+
next_byte = self._buffer[offset]
|
|
323
|
+
type_num = next_byte + 7
|
|
324
|
+
if type_num < 7:
|
|
325
|
+
msg = (
|
|
326
|
+
"Something went horribly wrong in the decoder. An "
|
|
327
|
+
f"extended type resolved to a type number < 8 ({type_num})"
|
|
328
|
+
)
|
|
329
|
+
raise InvalidDatabaseError(
|
|
330
|
+
msg,
|
|
331
|
+
)
|
|
332
|
+
return type_num, offset + 1
|
|
333
|
+
|
|
334
|
+
@staticmethod
|
|
335
|
+
def _verify_size(expected: int, actual: int) -> None:
|
|
336
|
+
if expected != actual:
|
|
337
|
+
raise InvalidDatabaseError(_BAD_DATA)
|
|
338
|
+
|
|
339
|
+
def _size_from_ctrl_byte(self, size: int, offset: int) -> tuple[int, int]:
|
|
340
|
+
# Called only for size codes 29 to 31, which are followed by size bytes.
|
|
341
|
+
if size == 29:
|
|
342
|
+
size = 29 + self._buffer[offset]
|
|
343
|
+
return size, offset + 1
|
|
344
|
+
|
|
345
|
+
# Using unpack rather than int_from_bytes as it is faster
|
|
346
|
+
# here and below.
|
|
347
|
+
if size == 30:
|
|
348
|
+
new_offset = offset + 2
|
|
349
|
+
size_bytes = self._buffer[offset:new_offset]
|
|
350
|
+
size = 285 + struct.unpack(b"!H", size_bytes)[0]
|
|
351
|
+
return size, new_offset
|
|
352
|
+
|
|
353
|
+
new_offset = offset + 3
|
|
354
|
+
size_bytes = self._buffer[offset:new_offset]
|
|
355
|
+
size = struct.unpack(b"!I", b"\x00" + size_bytes)[0] + 65821
|
|
356
|
+
return size, new_offset
|
maxminddb/errors.py
ADDED
|
Binary file
|
maxminddb/extension.pyi
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""C extension database reader and related classes."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network
|
|
5
|
+
from os import PathLike
|
|
6
|
+
from typing import IO, Any, AnyStr
|
|
7
|
+
|
|
8
|
+
from typing_extensions import Self
|
|
9
|
+
|
|
10
|
+
from maxminddb.types import Record
|
|
11
|
+
|
|
12
|
+
class Reader:
|
|
13
|
+
"""A C extension implementation of a reader for the MaxMind DB format.
|
|
14
|
+
|
|
15
|
+
IP addresses can be looked up using the ``get`` method.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
closed: bool = ...
|
|
19
|
+
|
|
20
|
+
def __init__(
|
|
21
|
+
self,
|
|
22
|
+
database: AnyStr | int | PathLike | IO,
|
|
23
|
+
mode: int = ...,
|
|
24
|
+
) -> None:
|
|
25
|
+
"""Reader for the MaxMind DB file format.
|
|
26
|
+
|
|
27
|
+
Arguments:
|
|
28
|
+
database: A path to a valid MaxMind DB file such as a GeoIP database
|
|
29
|
+
file, or a file descriptor in the case of MODE_FD.
|
|
30
|
+
mode: mode to open the database with. The only supported modes are
|
|
31
|
+
MODE_AUTO and MODE_MMAP_EXT.
|
|
32
|
+
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def close(self) -> None:
|
|
36
|
+
"""Close the MaxMind DB file and returns the resources to the system."""
|
|
37
|
+
|
|
38
|
+
def get(self, ip_address: str | IPv6Address | IPv4Address) -> Record | None:
|
|
39
|
+
"""Return the record for the ip_address in the MaxMind DB.
|
|
40
|
+
|
|
41
|
+
Arguments:
|
|
42
|
+
ip_address: an IP address in the standard string notation
|
|
43
|
+
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def get_with_prefix_len(
|
|
47
|
+
self,
|
|
48
|
+
ip_address: str | IPv6Address | IPv4Address,
|
|
49
|
+
) -> tuple[Record | None, int]:
|
|
50
|
+
"""Return a tuple with the record and the associated prefix length.
|
|
51
|
+
|
|
52
|
+
Arguments:
|
|
53
|
+
ip_address: an IP address in the standard string notation
|
|
54
|
+
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def metadata(self) -> Metadata:
|
|
58
|
+
"""Return the metadata associated with the MaxMind DB file."""
|
|
59
|
+
|
|
60
|
+
def __iter__(self) -> Iterator[tuple[IPv4Network | IPv6Network, Record]]: ...
|
|
61
|
+
def __enter__(self) -> Self: ...
|
|
62
|
+
def __exit__(self, *args) -> None: ... # noqa: ANN002
|
|
63
|
+
|
|
64
|
+
class Metadata:
|
|
65
|
+
"""Metadata for the MaxMind DB reader."""
|
|
66
|
+
|
|
67
|
+
binary_format_major_version: int
|
|
68
|
+
"""
|
|
69
|
+
The major version number of the binary format used when creating the
|
|
70
|
+
database.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
binary_format_minor_version: int
|
|
74
|
+
"""
|
|
75
|
+
The minor version number of the binary format used when creating the
|
|
76
|
+
database.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
build_epoch: int
|
|
80
|
+
"""
|
|
81
|
+
The Unix epoch for the build time of the database.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
database_type: str
|
|
85
|
+
"""
|
|
86
|
+
A string identifying the database type, e.g., "GeoIP2-City".
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
description: dict[str, str]
|
|
90
|
+
"""
|
|
91
|
+
A map from locales to text descriptions of the database.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
ip_version: int
|
|
95
|
+
"""
|
|
96
|
+
The IP version of the data in a database. A value of "4" means the
|
|
97
|
+
database only supports IPv4. A database with a value of "6" may support
|
|
98
|
+
both IPv4 and IPv6 lookups.
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
languages: list[str]
|
|
102
|
+
"""
|
|
103
|
+
A list of locale codes supported by the database.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
node_count: int
|
|
107
|
+
"""
|
|
108
|
+
The number of nodes in the database.
|
|
109
|
+
"""
|
|
110
|
+
|
|
111
|
+
record_size: int
|
|
112
|
+
"""
|
|
113
|
+
The bit size of a record in the search tree.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
def __init__(self, **kwargs: Any) -> None: # noqa: ANN401
|
|
117
|
+
"""Create new Metadata object. kwargs are key/value pairs from spec."""
|
maxminddb/file.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""For internal use only. It provides a slice-like file reader."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from typing import overload
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
from multiprocessing import Lock
|
|
10
|
+
except ImportError:
|
|
11
|
+
from threading import Lock # type: ignore[assignment]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class FileBuffer:
|
|
15
|
+
"""A slice-able file reader."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, database: str) -> None:
|
|
18
|
+
"""Create FileBuffer."""
|
|
19
|
+
self._handle = open(database, "rb") # noqa: SIM115
|
|
20
|
+
self._size = os.fstat(self._handle.fileno()).st_size
|
|
21
|
+
if not hasattr(os, "pread"):
|
|
22
|
+
self._lock = Lock()
|
|
23
|
+
|
|
24
|
+
@overload
|
|
25
|
+
def __getitem__(self, index: int) -> int: ...
|
|
26
|
+
|
|
27
|
+
@overload
|
|
28
|
+
def __getitem__(self, index: slice) -> bytes: ...
|
|
29
|
+
|
|
30
|
+
def __getitem__(self, index: slice | int) -> bytes | int:
|
|
31
|
+
"""Get item by index."""
|
|
32
|
+
if isinstance(index, slice):
|
|
33
|
+
return self._read(index.stop - index.start, index.start)
|
|
34
|
+
if isinstance(index, int):
|
|
35
|
+
return self._read(1, index)[0]
|
|
36
|
+
msg = "Invalid argument type."
|
|
37
|
+
raise TypeError(msg)
|
|
38
|
+
|
|
39
|
+
def rfind(self, needle: bytes, start: int) -> int:
|
|
40
|
+
"""Reverse find needle from start."""
|
|
41
|
+
pos = self._read(self._size - start - 1, start).rfind(needle)
|
|
42
|
+
if pos == -1:
|
|
43
|
+
return pos
|
|
44
|
+
return start + pos
|
|
45
|
+
|
|
46
|
+
def size(self) -> int:
|
|
47
|
+
"""Size of file."""
|
|
48
|
+
return self._size
|
|
49
|
+
|
|
50
|
+
def close(self) -> None:
|
|
51
|
+
"""Close file."""
|
|
52
|
+
self._handle.close()
|
|
53
|
+
|
|
54
|
+
if hasattr(os, "pread"): # type: ignore[attr-defined]
|
|
55
|
+
|
|
56
|
+
def _read(self, buffersize: int, offset: int) -> bytes:
|
|
57
|
+
"""Read that uses pread."""
|
|
58
|
+
return os.pread(self._handle.fileno(), buffersize, offset) # type: ignore[attr-defined]
|
|
59
|
+
|
|
60
|
+
else:
|
|
61
|
+
|
|
62
|
+
def _read(self, buffersize: int, offset: int) -> bytes:
|
|
63
|
+
"""Read with a lock.
|
|
64
|
+
|
|
65
|
+
This lock is necessary as after a fork, the different processes
|
|
66
|
+
will share the same file table entry, even if we dup the fd, and
|
|
67
|
+
as such the same offsets. There does not appear to be a way to
|
|
68
|
+
duplicate the file table entry and we cannot re-open based on the
|
|
69
|
+
original path as that file may have replaced with another or
|
|
70
|
+
unlinked.
|
|
71
|
+
"""
|
|
72
|
+
with self._lock:
|
|
73
|
+
self._handle.seek(offset)
|
|
74
|
+
return self._handle.read(buffersize)
|
maxminddb/py.typed
ADDED
|
File without changes
|