fpr-ff1 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fpr_ff1/__init__.py +23 -0
- fpr_ff1/_exceptions.py +34 -0
- fpr_ff1/_ff1.py +549 -0
- fpr_ff1/py.typed +0 -0
- fpr_ff1-0.1.0.dist-info/METADATA +348 -0
- fpr_ff1-0.1.0.dist-info/RECORD +8 -0
- fpr_ff1-0.1.0.dist-info/WHEEL +4 -0
- fpr_ff1-0.1.0.dist-info/licenses/LICENSE +21 -0
fpr_ff1/__init__.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""FF1 format-preserving encryption (NIST SP 800-38G)."""
|
|
2
|
+
|
|
3
|
+
from fpr_ff1._exceptions import (
|
|
4
|
+
AlphabetError,
|
|
5
|
+
FF1Error,
|
|
6
|
+
KeyLengthError,
|
|
7
|
+
LengthError,
|
|
8
|
+
RadixError,
|
|
9
|
+
TweakLengthError,
|
|
10
|
+
ValueRangeError,
|
|
11
|
+
)
|
|
12
|
+
from fpr_ff1._ff1 import FF1
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"FF1",
|
|
16
|
+
"AlphabetError",
|
|
17
|
+
"FF1Error",
|
|
18
|
+
"KeyLengthError",
|
|
19
|
+
"LengthError",
|
|
20
|
+
"RadixError",
|
|
21
|
+
"TweakLengthError",
|
|
22
|
+
"ValueRangeError",
|
|
23
|
+
]
|
fpr_ff1/_exceptions.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Typed exceptions for the FF1 implementation."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class FF1Error(Exception):
|
|
5
|
+
"""Base for all FF1 errors."""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class KeyLengthError(FF1Error):
|
|
9
|
+
"""Key length is not 16, 24, or 32 bytes."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class RadixError(FF1Error):
|
|
13
|
+
"""Radix is outside ``2 <= radix < 2**16``."""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class LengthError(FF1Error):
|
|
17
|
+
"""Input length is outside the valid domain for the radix."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class ValueRangeError(FF1Error):
|
|
21
|
+
"""Input data is outside the domain the instance accepts.
|
|
22
|
+
|
|
23
|
+
Raised for a numeral outside ``[0, radix)``, and for a character that does
|
|
24
|
+
not appear in the configured alphabet. Both are faults in the *data* being
|
|
25
|
+
encrypted; malformed *configuration* raises :class:`AlphabetError` instead.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TweakLengthError(FF1Error):
|
|
30
|
+
"""Tweak length is outside the configured bounds."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class AlphabetError(FF1Error):
|
|
34
|
+
"""Alphabet length or uniqueness does not match the radix."""
|
fpr_ff1/_ff1.py
ADDED
|
@@ -0,0 +1,549 @@
|
|
|
1
|
+
"""FF1 implementation following NIST SP 800-38G Algorithm 7."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import operator
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from typing import ClassVar, NamedTuple, SupportsIndex, cast
|
|
8
|
+
|
|
9
|
+
from cryptography.hazmat.primitives.ciphers import (
|
|
10
|
+
Cipher,
|
|
11
|
+
CipherContext,
|
|
12
|
+
algorithms,
|
|
13
|
+
modes,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
from fpr_ff1._exceptions import (
|
|
17
|
+
AlphabetError,
|
|
18
|
+
FF1Error,
|
|
19
|
+
KeyLengthError,
|
|
20
|
+
LengthError,
|
|
21
|
+
RadixError,
|
|
22
|
+
TweakLengthError,
|
|
23
|
+
ValueRangeError,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
#: One round's intermediate values, as published in the NIST sample document.
|
|
27
|
+
#: Test-only; see :meth:`FF1._encrypt_traced`.
|
|
28
|
+
type TraceRecord = dict[str, object]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _require_int(value: object, name: str, error: type[FF1Error]) -> int:
|
|
32
|
+
"""Return ``value`` as an ``int``, rejecting anything not losslessly integral.
|
|
33
|
+
|
|
34
|
+
Uses ``operator.index()``, Python's own "this is an integer" protocol:
|
|
35
|
+
``float`` and ``Decimal`` deliberately do not implement it, while ``int``,
|
|
36
|
+
``IntEnum`` and NumPy integers do. Comparison alone is not enough of a
|
|
37
|
+
gate -- ``1.0 < 10`` is ``True``, so a float numeral would otherwise pass
|
|
38
|
+
validation and fail much later with an ``AttributeError`` from outside the
|
|
39
|
+
:class:`FF1Error` hierarchy.
|
|
40
|
+
|
|
41
|
+
``bool`` is rejected explicitly even though it implements ``__index__``:
|
|
42
|
+
silently encrypting ``True`` as ``1`` is the coercion the contract forbids,
|
|
43
|
+
and a sequence of booleans reaching this point is a caller mistake.
|
|
44
|
+
|
|
45
|
+
Converting rather than merely checking also normalises NumPy integers to
|
|
46
|
+
Python ``int``, which matters: fixed-width integers would overflow silently
|
|
47
|
+
in the big-integer arithmetic downstream.
|
|
48
|
+
"""
|
|
49
|
+
if isinstance(value, bool):
|
|
50
|
+
raise error(f"{name} must be an integer, not bool")
|
|
51
|
+
try:
|
|
52
|
+
# The cast asserts only that __index__ *might* exist; TypeError below
|
|
53
|
+
# is the actual gate.
|
|
54
|
+
return operator.index(cast("SupportsIndex", value))
|
|
55
|
+
except TypeError:
|
|
56
|
+
raise error(f"{name} must be an integer, got {type(value).__name__}") from None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _validate_tweak_bounds(
|
|
60
|
+
min_tweak_len: int | None, max_tweak_len: int | None
|
|
61
|
+
) -> tuple[int | None, int | None]:
|
|
62
|
+
"""Type- and sanity-check the configured tweak length bounds.
|
|
63
|
+
|
|
64
|
+
These are *configuration* faults, distinct from a tweak that merely
|
|
65
|
+
violates an otherwise valid bound, and they are caught at construction so
|
|
66
|
+
an unusable instance can never be built.
|
|
67
|
+
|
|
68
|
+
Raises:
|
|
69
|
+
TweakLengthError: if a bound is not an integer, is negative, or if the
|
|
70
|
+
two bounds are mutually unsatisfiable.
|
|
71
|
+
"""
|
|
72
|
+
low = (
|
|
73
|
+
None
|
|
74
|
+
if min_tweak_len is None
|
|
75
|
+
else _require_int(min_tweak_len, "min_tweak_len", TweakLengthError)
|
|
76
|
+
)
|
|
77
|
+
high = (
|
|
78
|
+
None
|
|
79
|
+
if max_tweak_len is None
|
|
80
|
+
else _require_int(max_tweak_len, "max_tweak_len", TweakLengthError)
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
for name, bound in (("min_tweak_len", low), ("max_tweak_len", high)):
|
|
84
|
+
if bound is not None and bound < 0:
|
|
85
|
+
# A negative bound is inert rather than harmful, but it silently
|
|
86
|
+
# means "no constraint" -- which is not what the caller asked for.
|
|
87
|
+
raise TweakLengthError(f"{name} must be non-negative, got {bound}")
|
|
88
|
+
|
|
89
|
+
if low is not None and high is not None and low > high:
|
|
90
|
+
raise TweakLengthError(
|
|
91
|
+
f"min_tweak_len {low} exceeds max_tweak_len {high}; "
|
|
92
|
+
"no tweak length could satisfy both bounds"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
return low, high
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _require_bytes(value: object, name: str, error: type[FF1Error]) -> bytes:
|
|
99
|
+
"""Return ``value`` as ``bytes``, rejecting non-bytes-like input."""
|
|
100
|
+
if not isinstance(value, bytes | bytearray | memoryview):
|
|
101
|
+
raise error(f"{name} must be bytes-like, got {type(value).__name__}")
|
|
102
|
+
# Normalise to immutable bytes: a caller holding the bytearray must not be
|
|
103
|
+
# able to mutate a tweak or key after construction.
|
|
104
|
+
return bytes(cast("bytes | bytearray | memoryview[int]", value))
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class _Aes(NamedTuple):
|
|
108
|
+
"""AES objects shared for the lifetime of one :class:`FF1` instance.
|
|
109
|
+
|
|
110
|
+
``algorithm`` and ``cbc_zero_iv`` are immutable configuration, reused to
|
|
111
|
+
avoid rebuilding them on every PRF call. ``ecb_encryptor`` is a live
|
|
112
|
+
context, but ECB carries no chaining state so repeated ``update()`` calls
|
|
113
|
+
are safe. A CBC *encryptor* is never held here: it would carry chaining
|
|
114
|
+
state between PRF invocations.
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
algorithm: algorithms.AES
|
|
118
|
+
cbc_zero_iv: modes.CBC
|
|
119
|
+
ecb_encryptor: CipherContext
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class FF1:
|
|
123
|
+
"""FF1 format-preserving encryption primitive and string wrapper."""
|
|
124
|
+
|
|
125
|
+
# SP 800-38G requires "minlen <= n <= maxlen < 2**32", so the largest
|
|
126
|
+
# admissible length is 2**32 - 1, not 2**32. Failing closed on the
|
|
127
|
+
# boundary matches the project's stance elsewhere; the excluded value is
|
|
128
|
+
# unconstructable in practice (a 2**32-element list needs tens of GB).
|
|
129
|
+
_MAX_LEN: ClassVar[int] = 2**32 - 1
|
|
130
|
+
_RADIX_MIN: ClassVar[int] = 2
|
|
131
|
+
_RADIX_MAX_EXCLUSIVE: ClassVar[int] = 2**16
|
|
132
|
+
|
|
133
|
+
def __init__(
|
|
134
|
+
self,
|
|
135
|
+
key: bytes,
|
|
136
|
+
radix: int,
|
|
137
|
+
*,
|
|
138
|
+
alphabet: str | None = None,
|
|
139
|
+
tweak: bytes = b"",
|
|
140
|
+
min_tweak_len: int | None = None,
|
|
141
|
+
max_tweak_len: int | None = None,
|
|
142
|
+
) -> None:
|
|
143
|
+
"""Create an FF1 instance.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
key: AES key; must be 16, 24, or 32 bytes.
|
|
147
|
+
radix: Numeral base, ``2 <= radix < 2**16``.
|
|
148
|
+
alphabet: Optional string of exactly ``radix`` unique characters.
|
|
149
|
+
Required for the string interface.
|
|
150
|
+
tweak: Default tweak used when not provided per call.
|
|
151
|
+
min_tweak_len: Optional inclusive lower bound on tweak length.
|
|
152
|
+
max_tweak_len: Optional inclusive upper bound on tweak length.
|
|
153
|
+
|
|
154
|
+
Raises:
|
|
155
|
+
KeyLengthError: if the key is not bytes-like or has an invalid length.
|
|
156
|
+
RadixError: if the radix is not an integer or is out of range.
|
|
157
|
+
TweakLengthError: if the tweak is not bytes-like or out of bounds.
|
|
158
|
+
AlphabetError: if the alphabet is not a string or is malformed.
|
|
159
|
+
"""
|
|
160
|
+
# Types are checked before values throughout: a wrong type is the more
|
|
161
|
+
# fundamental fault, and reporting a range error for a float would be
|
|
162
|
+
# actively misleading.
|
|
163
|
+
key = _require_bytes(key, "key", KeyLengthError)
|
|
164
|
+
if len(key) not in {16, 24, 32}:
|
|
165
|
+
raise KeyLengthError(f"key must be 16, 24, or 32 bytes, got {len(key)}")
|
|
166
|
+
|
|
167
|
+
radix = _require_int(radix, "radix", RadixError)
|
|
168
|
+
if radix < self._RADIX_MIN or radix >= self._RADIX_MAX_EXCLUSIVE:
|
|
169
|
+
raise RadixError(
|
|
170
|
+
f"radix must satisfy {self._RADIX_MIN} <= radix < {self._RADIX_MAX_EXCLUSIVE}, "
|
|
171
|
+
f"got {radix!r}"
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
self._key = key
|
|
175
|
+
self._radix = radix
|
|
176
|
+
|
|
177
|
+
# Bounds are validated before the default tweak is checked against
|
|
178
|
+
# them, so an unsatisfiable configuration is reported as such rather
|
|
179
|
+
# than as whichever bound the default tweak happened to violate first.
|
|
180
|
+
self._min_tweak_len, self._max_tweak_len = _validate_tweak_bounds(
|
|
181
|
+
min_tweak_len, max_tweak_len
|
|
182
|
+
)
|
|
183
|
+
tweak = _require_bytes(tweak, "tweak", TweakLengthError)
|
|
184
|
+
self._validate_tweak(tweak)
|
|
185
|
+
self._default_tweak = tweak
|
|
186
|
+
|
|
187
|
+
self._alphabet: str | None = None
|
|
188
|
+
self._char_to_index: dict[str, int] | None = None
|
|
189
|
+
self._index_to_char: list[str] | None = None
|
|
190
|
+
if alphabet is not None:
|
|
191
|
+
# Checked at runtime despite the annotation: type hints are not
|
|
192
|
+
# enforced, and a list alphabet silently worked before this guard.
|
|
193
|
+
if not isinstance(alphabet, str): # pyright: ignore[reportUnnecessaryIsInstance]
|
|
194
|
+
raise AlphabetError(f"alphabet must be a str, got {type(alphabet).__name__}")
|
|
195
|
+
if len(alphabet) != radix:
|
|
196
|
+
raise AlphabetError(f"alphabet length {len(alphabet)} does not match radix {radix}")
|
|
197
|
+
# Uniqueness is by Unicode code point. Two visually identical but
|
|
198
|
+
# differently-normalised symbols (e.g. precomposed vs decomposed
|
|
199
|
+
# accents) are distinct here; normalisation is the caller's job.
|
|
200
|
+
if len(set(alphabet)) != len(alphabet):
|
|
201
|
+
raise AlphabetError("alphabet contains duplicate characters")
|
|
202
|
+
self._alphabet = alphabet
|
|
203
|
+
self._char_to_index = {ch: i for i, ch in enumerate(alphabet)}
|
|
204
|
+
self._index_to_char = list(alphabet)
|
|
205
|
+
|
|
206
|
+
# Every legal radix admits a feasible length: min_length peaks at 20
|
|
207
|
+
# (radix 2), far below _MAX_LEN, so no infeasibility check is needed.
|
|
208
|
+
self._min_length = _min_length(radix)
|
|
209
|
+
|
|
210
|
+
# Cipher objects reused across calls. Do not call finalize() on the
|
|
211
|
+
# ECB encryptor; it must stay alive for the instance's lifetime.
|
|
212
|
+
algorithm = algorithms.AES(key)
|
|
213
|
+
self._aes = _Aes(
|
|
214
|
+
algorithm=algorithm,
|
|
215
|
+
cbc_zero_iv=modes.CBC(b"\x00" * 16),
|
|
216
|
+
ecb_encryptor=Cipher(algorithm, modes.ECB()).encryptor(),
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
@property
|
|
220
|
+
def min_length(self) -> int:
|
|
221
|
+
"""Minimum permitted input length for this radix."""
|
|
222
|
+
return self._min_length
|
|
223
|
+
|
|
224
|
+
@property
|
|
225
|
+
def max_length(self) -> int:
|
|
226
|
+
"""Maximum permitted input length for this radix.
|
|
227
|
+
|
|
228
|
+
``2**32 - 1``: SP 800-38G specifies ``maxlen < 2**32``.
|
|
229
|
+
"""
|
|
230
|
+
return self._MAX_LEN
|
|
231
|
+
|
|
232
|
+
def _validate_tweak(self, tweak: bytes) -> None:
|
|
233
|
+
if self._min_tweak_len is not None and len(tweak) < self._min_tweak_len:
|
|
234
|
+
raise TweakLengthError(f"tweak length {len(tweak)} below minimum {self._min_tweak_len}")
|
|
235
|
+
if self._max_tweak_len is not None and len(tweak) > self._max_tweak_len:
|
|
236
|
+
raise TweakLengthError(f"tweak length {len(tweak)} above maximum {self._max_tweak_len}")
|
|
237
|
+
|
|
238
|
+
def _validate_length(self, n: int, inout: str) -> None:
|
|
239
|
+
if n < self._min_length:
|
|
240
|
+
raise LengthError(
|
|
241
|
+
f"{inout} length {n} below minimum {self._min_length} for radix {self._radix}"
|
|
242
|
+
)
|
|
243
|
+
if n > self._MAX_LEN:
|
|
244
|
+
raise LengthError(f"{inout} length {n} above maximum {self._MAX_LEN}")
|
|
245
|
+
|
|
246
|
+
def _coerce_numerals(self, x: Sequence[int], inout: str) -> list[int]:
|
|
247
|
+
"""Type-check, normalise and range-check numerals in a single pass.
|
|
248
|
+
|
|
249
|
+
Returns true Python ``int`` values, so fixed-width integers from other
|
|
250
|
+
numeric libraries cannot reach the big-integer arithmetic downstream
|
|
251
|
+
and overflow silently.
|
|
252
|
+
"""
|
|
253
|
+
radix = self._radix
|
|
254
|
+
numerals: list[int] = []
|
|
255
|
+
for idx, value in enumerate(x):
|
|
256
|
+
numeral = _require_int(value, f"{inout}[{idx}]", ValueRangeError)
|
|
257
|
+
if numeral < 0 or numeral >= radix:
|
|
258
|
+
raise ValueRangeError(f"{inout}[{idx}]={numeral!r} out of range for radix {radix}")
|
|
259
|
+
numerals.append(numeral)
|
|
260
|
+
return numerals
|
|
261
|
+
|
|
262
|
+
def _prepare(
|
|
263
|
+
self, x: Sequence[int], tweak: bytes | None, inout: str
|
|
264
|
+
) -> tuple[list[int], bytes]:
|
|
265
|
+
"""Validate and normalise the inputs shared by encrypt and decrypt."""
|
|
266
|
+
t = (
|
|
267
|
+
self._default_tweak
|
|
268
|
+
if tweak is None
|
|
269
|
+
else _require_bytes(tweak, "tweak", TweakLengthError)
|
|
270
|
+
)
|
|
271
|
+
# Check the length before materialising the sequence: an over-long
|
|
272
|
+
# input must be rejected without first allocating a copy of it.
|
|
273
|
+
if not hasattr(x, "__len__"):
|
|
274
|
+
raise TypeError(
|
|
275
|
+
f"{inout} must be a Sequence[int] with a known length, got "
|
|
276
|
+
f"{type(x).__name__}; wrap it with list(...) if it is an iterator"
|
|
277
|
+
)
|
|
278
|
+
self._validate_length(len(x), inout)
|
|
279
|
+
numerals = self._coerce_numerals(x, inout)
|
|
280
|
+
self._validate_tweak(t)
|
|
281
|
+
return numerals, t
|
|
282
|
+
|
|
283
|
+
def _alphabet_maps(self, numeral_method: str) -> tuple[dict[str, int], list[str]]:
|
|
284
|
+
"""Return the alphabet lookup tables, or raise if none was configured."""
|
|
285
|
+
if self._char_to_index is None or self._index_to_char is None:
|
|
286
|
+
raise FF1Error(
|
|
287
|
+
f"alphabet required for string interface; use {numeral_method} "
|
|
288
|
+
"for the numeral interface"
|
|
289
|
+
)
|
|
290
|
+
return self._char_to_index, self._index_to_char
|
|
291
|
+
|
|
292
|
+
def _decode_str(self, s: str, char_to_index: dict[str, int]) -> list[int]:
|
|
293
|
+
"""Map characters to numerals, rejecting anything outside the alphabet."""
|
|
294
|
+
# Checked at runtime despite the annotation; see _require_int.
|
|
295
|
+
if not isinstance(s, str): # pyright: ignore[reportUnnecessaryIsInstance]
|
|
296
|
+
raise ValueRangeError(f"input must be a str, got {type(s).__name__}")
|
|
297
|
+
numerals: list[int] = []
|
|
298
|
+
for idx, ch in enumerate(s):
|
|
299
|
+
value = char_to_index.get(ch)
|
|
300
|
+
if value is None:
|
|
301
|
+
raise ValueRangeError(f"character {ch!r} at index {idx} is not in the alphabet")
|
|
302
|
+
numerals.append(value)
|
|
303
|
+
return numerals
|
|
304
|
+
|
|
305
|
+
def encrypt_numerals(self, x: Sequence[int], tweak: bytes | None = None) -> list[int]:
|
|
306
|
+
"""Encrypt a sequence of numerals.
|
|
307
|
+
|
|
308
|
+
Args:
|
|
309
|
+
x: List of integers in ``[0, radix)``.
|
|
310
|
+
tweak: Tweak bytes; defaults to the instance tweak.
|
|
311
|
+
|
|
312
|
+
Returns:
|
|
313
|
+
Encrypted numeral sequence of the same length.
|
|
314
|
+
|
|
315
|
+
Raises:
|
|
316
|
+
LengthError: if the input length is outside the valid domain.
|
|
317
|
+
ValueRangeError: if any numeral is outside ``[0, radix)``.
|
|
318
|
+
TweakLengthError: if the tweak is out of bounds.
|
|
319
|
+
"""
|
|
320
|
+
numerals, t = self._prepare(x, tweak, "plaintext")
|
|
321
|
+
return _ff1(self._aes, self._radix, numerals, t, encrypt=True)
|
|
322
|
+
|
|
323
|
+
def decrypt_numerals(self, x: Sequence[int], tweak: bytes | None = None) -> list[int]:
|
|
324
|
+
"""Decrypt a sequence of numerals.
|
|
325
|
+
|
|
326
|
+
Args:
|
|
327
|
+
x: List of integers in ``[0, radix)``.
|
|
328
|
+
tweak: Tweak bytes; defaults to the instance tweak.
|
|
329
|
+
|
|
330
|
+
Returns:
|
|
331
|
+
Decrypted numeral sequence of the same length.
|
|
332
|
+
|
|
333
|
+
Raises:
|
|
334
|
+
LengthError: if the input length is outside the valid domain.
|
|
335
|
+
ValueRangeError: if any numeral is outside ``[0, radix)``.
|
|
336
|
+
TweakLengthError: if the tweak is out of bounds.
|
|
337
|
+
"""
|
|
338
|
+
numerals, t = self._prepare(x, tweak, "ciphertext")
|
|
339
|
+
return _ff1(self._aes, self._radix, numerals, t, encrypt=False)
|
|
340
|
+
|
|
341
|
+
def _encrypt_traced(
|
|
342
|
+
self, x: Sequence[int], tweak: bytes | None = None
|
|
343
|
+
) -> tuple[list[int], list[TraceRecord]]:
|
|
344
|
+
"""Encrypt, also returning the per-round intermediates.
|
|
345
|
+
|
|
346
|
+
Test-only conformance hook, deliberately kept off the public methods:
|
|
347
|
+
the NIST sample document publishes ``P``, ``Q``, ``R``, ``S``, ``y``,
|
|
348
|
+
``m``, ``c`` and ``C`` for every round, and two compensating bugs can
|
|
349
|
+
agree on the final output while disagreeing here.
|
|
350
|
+
|
|
351
|
+
Not exported from the package and not part of the supported API.
|
|
352
|
+
"""
|
|
353
|
+
numerals, t = self._prepare(x, tweak, "plaintext")
|
|
354
|
+
trace: list[TraceRecord] = []
|
|
355
|
+
return _ff1(self._aes, self._radix, numerals, t, encrypt=True, _trace=trace), trace
|
|
356
|
+
|
|
357
|
+
def encrypt(self, s: str, tweak: bytes | None = None) -> str:
|
|
358
|
+
"""Encrypt a string using the configured alphabet.
|
|
359
|
+
|
|
360
|
+
Raises:
|
|
361
|
+
FF1Error: if no alphabet was configured at construction.
|
|
362
|
+
ValueRangeError: if a character is absent from the alphabet.
|
|
363
|
+
"""
|
|
364
|
+
char_to_index, index_to_char = self._alphabet_maps("encrypt_numerals")
|
|
365
|
+
numerals = self._decode_str(s, char_to_index)
|
|
366
|
+
encrypted = self.encrypt_numerals(numerals, tweak)
|
|
367
|
+
return "".join(index_to_char[i] for i in encrypted)
|
|
368
|
+
|
|
369
|
+
def decrypt(self, s: str, tweak: bytes | None = None) -> str:
|
|
370
|
+
"""Decrypt a string using the configured alphabet.
|
|
371
|
+
|
|
372
|
+
Raises:
|
|
373
|
+
FF1Error: if no alphabet was configured at construction.
|
|
374
|
+
ValueRangeError: if a character is absent from the alphabet.
|
|
375
|
+
"""
|
|
376
|
+
char_to_index, index_to_char = self._alphabet_maps("decrypt_numerals")
|
|
377
|
+
numerals = self._decode_str(s, char_to_index)
|
|
378
|
+
decrypted = self.decrypt_numerals(numerals, tweak)
|
|
379
|
+
return "".join(index_to_char[i] for i in decrypted)
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
_MIN_DOMAIN = 1_000_000
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _min_length(radix: int) -> int:
|
|
386
|
+
"""Return the smallest n with radix**n >= 1_000_000."""
|
|
387
|
+
n = 1
|
|
388
|
+
value = radix
|
|
389
|
+
while value < _MIN_DOMAIN:
|
|
390
|
+
value *= radix
|
|
391
|
+
n += 1
|
|
392
|
+
return n
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _num_radix(radix: int, numerals: Sequence[int]) -> int:
|
|
396
|
+
"""Decode a sequence of numerals as a big-endian base-radix integer."""
|
|
397
|
+
value = 0
|
|
398
|
+
for x in numerals:
|
|
399
|
+
value = value * radix + x
|
|
400
|
+
return value
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _str_radix(value: int, radix: int, length: int) -> list[int]:
|
|
404
|
+
"""Encode a non-negative integer as ``length`` big-endian base-radix numerals."""
|
|
405
|
+
out = [0] * length
|
|
406
|
+
for i in range(length - 1, -1, -1):
|
|
407
|
+
out[i] = value % radix
|
|
408
|
+
value //= radix
|
|
409
|
+
return out
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _prf(aes: _Aes, data: bytes) -> bytes:
|
|
413
|
+
"""SP 800-38G Algorithm 6 (PRF): CBC-MAC with a zero IV.
|
|
414
|
+
|
|
415
|
+
Invoked from Algorithm 7 step 6.ii as ``PRF(P || Q)``.
|
|
416
|
+
"""
|
|
417
|
+
# data is already 16-byte aligned by callers. A fresh encryptor per call
|
|
418
|
+
# is required: CBC chaining state must never persist between PRF calls.
|
|
419
|
+
encryptor = Cipher(aes.algorithm, aes.cbc_zero_iv).encryptor()
|
|
420
|
+
result = encryptor.update(data) + encryptor.finalize()
|
|
421
|
+
return result[-16:]
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _ff1(
|
|
425
|
+
aes: _Aes,
|
|
426
|
+
radix: int,
|
|
427
|
+
x: list[int],
|
|
428
|
+
tweak: bytes,
|
|
429
|
+
*,
|
|
430
|
+
encrypt: bool,
|
|
431
|
+
_trace: list[TraceRecord] | None = None,
|
|
432
|
+
) -> list[int]:
|
|
433
|
+
"""SP 800-38G Algorithm 7 core."""
|
|
434
|
+
n = len(x)
|
|
435
|
+
|
|
436
|
+
# Step 1: u = floor(n/2), v = n - u
|
|
437
|
+
u = n // 2
|
|
438
|
+
v = n - u
|
|
439
|
+
|
|
440
|
+
# Step 2: A = X[1..u], B = X[u+1..n]
|
|
441
|
+
a = x[:u]
|
|
442
|
+
b_side = x[u:]
|
|
443
|
+
|
|
444
|
+
# Step 3: b = ceil(ceil(v * log2(radix)) / 8) -- derived from v, not u.
|
|
445
|
+
# The bit length uses exact integer arithmetic; never math.log2.
|
|
446
|
+
b = ((radix**v - 1).bit_length() + 7) // 8
|
|
447
|
+
|
|
448
|
+
# Step 4: d = 4 * ceil(b/4) + 4
|
|
449
|
+
d = 4 * ((b + 3) // 4) + 4
|
|
450
|
+
|
|
451
|
+
t = len(tweak)
|
|
452
|
+
pad = (-(t + b + 1)) % 16
|
|
453
|
+
|
|
454
|
+
# Step 5: P is loop-invariant, so it is built once here rather than
|
|
455
|
+
# rebuilt on each of the ten rounds.
|
|
456
|
+
p_block = (
|
|
457
|
+
bytes([1, 2, 1])
|
|
458
|
+
+ _encode_uint(radix, 3)
|
|
459
|
+
+ bytes([10, u % 256])
|
|
460
|
+
+ _encode_uint(n, 4)
|
|
461
|
+
+ _encode_uint(t, 4)
|
|
462
|
+
)
|
|
463
|
+
|
|
464
|
+
rounds = range(10) if encrypt else range(9, -1, -1)
|
|
465
|
+
|
|
466
|
+
for i in rounds:
|
|
467
|
+
# Step 6.i: Q = T || [0]^pad || [i]^1 || [NUM_radix(B)]^b
|
|
468
|
+
# (decrypt builds Q from A instead of B)
|
|
469
|
+
if encrypt:
|
|
470
|
+
q_block = (
|
|
471
|
+
tweak + bytes([0]) * pad + bytes([i]) + _encode_uint(_num_radix(radix, b_side), b)
|
|
472
|
+
)
|
|
473
|
+
else:
|
|
474
|
+
q_block = tweak + bytes([0]) * pad + bytes([i]) + _encode_uint(_num_radix(radix, a), b)
|
|
475
|
+
|
|
476
|
+
# Step 6.ii: R = PRF(P || Q)
|
|
477
|
+
r_block = _prf(aes, p_block + q_block)
|
|
478
|
+
|
|
479
|
+
# Step 6.iii: S is the first d bytes of
|
|
480
|
+
# R || CIPH_K(R XOR [1]^16) || CIPH_K(R XOR [2]^16) || ...
|
|
481
|
+
# Each expansion block is a SINGLE forward-cipher block over R XOR the
|
|
482
|
+
# 16-byte encoding of j. It is not a PRF, and j is not concatenated
|
|
483
|
+
# onto R -- both mistakes produce a non-16-byte-aligned input and are
|
|
484
|
+
# invisible to the NIST samples, none of which reach d > 16.
|
|
485
|
+
s_block = r_block
|
|
486
|
+
j = 1
|
|
487
|
+
while len(s_block) < d:
|
|
488
|
+
xored = bytes(p ^ q for p, q in zip(r_block, _encode_uint(j, 16), strict=True))
|
|
489
|
+
s_block += aes.ecb_encryptor.update(xored)
|
|
490
|
+
j += 1
|
|
491
|
+
# Truncate to d BYTES, not d bits.
|
|
492
|
+
s_block = s_block[:d]
|
|
493
|
+
|
|
494
|
+
# Step 6.iv: y = NUM(S)
|
|
495
|
+
y = int.from_bytes(s_block, byteorder="big")
|
|
496
|
+
|
|
497
|
+
# Step 6.v: parity rule is identical for encrypt and decrypt
|
|
498
|
+
m = u if i % 2 == 0 else v
|
|
499
|
+
|
|
500
|
+
# Step 6.vi: c = (NUM_radix(A) + y) mod radix**m (decrypt subtracts
|
|
501
|
+
# y from NUM_radix(B) instead)
|
|
502
|
+
if encrypt:
|
|
503
|
+
c = (_num_radix(radix, a) + y) % (radix**m)
|
|
504
|
+
else:
|
|
505
|
+
c = (_num_radix(radix, b_side) - y) % (radix**m)
|
|
506
|
+
|
|
507
|
+
# Step 6.vii: C = STR^m_radix(c)
|
|
508
|
+
c_block = _str_radix(c, radix, m)
|
|
509
|
+
|
|
510
|
+
if _trace is not None:
|
|
511
|
+
_trace.append(
|
|
512
|
+
{
|
|
513
|
+
"i": i,
|
|
514
|
+
"u": u,
|
|
515
|
+
"v": v,
|
|
516
|
+
"b": b,
|
|
517
|
+
"d": d,
|
|
518
|
+
"P": list(p_block),
|
|
519
|
+
"Q": list(q_block),
|
|
520
|
+
"R": list(r_block),
|
|
521
|
+
"S": list(s_block),
|
|
522
|
+
"y": y,
|
|
523
|
+
"m": m,
|
|
524
|
+
"c": c,
|
|
525
|
+
"C": list(c_block),
|
|
526
|
+
"A_before": list(a),
|
|
527
|
+
"B_before": list(b_side),
|
|
528
|
+
}
|
|
529
|
+
)
|
|
530
|
+
|
|
531
|
+
# Steps 6.viii and 6.ix: A = B, B = C (decrypt assigns B = A, A = C)
|
|
532
|
+
if encrypt:
|
|
533
|
+
a = b_side
|
|
534
|
+
b_side = c_block
|
|
535
|
+
else:
|
|
536
|
+
b_side = a
|
|
537
|
+
a = c_block
|
|
538
|
+
|
|
539
|
+
# Step 7: return A || B
|
|
540
|
+
return a + b_side
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _encode_uint(value: int, length: int) -> bytes:
|
|
544
|
+
"""Encode a non-negative integer as a big-endian ``length``-byte string.
|
|
545
|
+
|
|
546
|
+
Raises ``OverflowError`` if the value does not fit, which is deliberate:
|
|
547
|
+
silently truncating would corrupt Q and produce non-conformant ciphertext.
|
|
548
|
+
"""
|
|
549
|
+
return value.to_bytes(length, byteorder="big")
|
fpr_ff1/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: fpr-ff1
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Format-preserving encryption: NIST SP 800-38G FF1 for Python.
|
|
5
|
+
Project-URL: Homepage, https://github.com/joelee/fpr-ff1
|
|
6
|
+
Project-URL: Repository, https://github.com/joelee/fpr-ff1
|
|
7
|
+
Project-URL: Issues, https://github.com/joelee/fpr-ff1/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/joelee/fpr-ff1/blob/main/CHANGELOG.md
|
|
9
|
+
Project-URL: Security, https://github.com/joelee/fpr-ff1/blob/main/SECURITY.md
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: cryptography,ff1,format-preserving-encryption,fpe,nist,sp800-38g
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
21
|
+
Classifier: Topic :: Security :: Cryptography
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: <3.15,>=3.12
|
|
24
|
+
Requires-Dist: cryptography>=44.0.0
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# fpr-ff1
|
|
28
|
+
|
|
29
|
+
A small, correct Python implementation of **FF1**, the format-preserving encryption mode from NIST SP 800-38G.
|
|
30
|
+
|
|
31
|
+
This package is intentionally just the algorithm: no accounts, no network, no key management, and no FF3/FF3-1 modes.
|
|
32
|
+
|
|
33
|
+
## Install
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install fpr-ff1
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Quick start
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from fpr_ff1 import FF1
|
|
43
|
+
|
|
44
|
+
ff1 = FF1(
|
|
45
|
+
key=b"\x00" * 16,
|
|
46
|
+
radix=10,
|
|
47
|
+
alphabet="0123456789",
|
|
48
|
+
tweak=b"",
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
encrypted = ff1.encrypt("123456")
|
|
52
|
+
decrypted = ff1.decrypt(encrypted)
|
|
53
|
+
assert decrypted == "123456"
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Features
|
|
57
|
+
|
|
58
|
+
- Pure Python with a single runtime dependency: `cryptography`.
|
|
59
|
+
- Conformance-tested against the NIST SP 800-38G sample vectors.
|
|
60
|
+
- No floating-point arithmetic in the FF1 core.
|
|
61
|
+
- Tightened domain limits from the SP 800-38G Rev. 1 second public draft:
|
|
62
|
+
- radix range `2 <= radix < 2**16`
|
|
63
|
+
- minimum domain `radix ** minlen >= 1_000_000`
|
|
64
|
+
- maximum length `2 ** 32 - 1` (SP 800-38G specifies `maxlen < 2 ** 32`)
|
|
65
|
+
- AES keys of 128, 192, or 256 bits only
|
|
66
|
+
- Strongly typed public API with typed exceptions rooted at `FF1Error`.
|
|
67
|
+
|
|
68
|
+
## Why you can trust this implementation
|
|
69
|
+
|
|
70
|
+
Format-preserving encryption is unusually easy to get *almost* right. A subtly wrong FF1 still
|
|
71
|
+
round-trips perfectly — `decrypt(encrypt(x)) == x` — while producing ciphertext no conformant
|
|
72
|
+
implementation can read. By the time anyone notices, the data is written. So conformance here is
|
|
73
|
+
not a checkbox; it is the entire product, and it is evidenced rather than asserted.
|
|
74
|
+
|
|
75
|
+
**708 tests. 100% line and branch coverage, enforced — the build fails below it.**
|
|
76
|
+
|
|
77
|
+
### Conformance is proven at the round level, not just the output level
|
|
78
|
+
|
|
79
|
+
All nine published NIST sample vectors pass in both directions. That alone is a weak statement:
|
|
80
|
+
nine input/output pairs can be satisfied by two bugs that cancel out.
|
|
81
|
+
|
|
82
|
+
So this package also asserts the **per-round intermediate values** the NIST sample document
|
|
83
|
+
publishes — `P`, `Q`, `R`, `S`, `y`, `m`, `c` and `C`, plus the derived `u`, `v`, `b` and `d` — for
|
|
84
|
+
**every round of every sample**, 90 rounds in total. Compensating bugs survive an output test. They
|
|
85
|
+
do not survive this one.
|
|
86
|
+
|
|
87
|
+
The vectors are transcribed from the NIST document and stored as data files. They are never
|
|
88
|
+
regenerated from this implementation, which would make them a record of whatever the code does
|
|
89
|
+
rather than of what the standard requires.
|
|
90
|
+
|
|
91
|
+
### Radices without published vectors are proven against an independent implementation
|
|
92
|
+
|
|
93
|
+
NIST publishes vectors for radix 10 and 36 only. Every other radix has none, so agreement with an
|
|
94
|
+
independent implementation is the only correctness evidence available — expected values authored
|
|
95
|
+
from this code would test nothing and lock in any bug permanently.
|
|
96
|
+
|
|
97
|
+
`fpr-ff1` is therefore differential-tested against `ubiq_security_fpe` across radices **2, 10, 16,
|
|
98
|
+
32, 36, 62, 256 and 65535**, including every length where the algorithm's internal block structure
|
|
99
|
+
changes. The oracle is itself validated against all nine NIST vectors before a single comparison is
|
|
100
|
+
trusted.
|
|
101
|
+
|
|
102
|
+
### Bijectivity is verified exhaustively, not sampled
|
|
103
|
+
|
|
104
|
+
For two domains small enough to enumerate completely — radix 2 at length 20 (1,048,576 values) and
|
|
105
|
+
radix 10 at length 6 (1,000,000 values) — every point is encrypted and the image checked to be the
|
|
106
|
+
full domain, with no gaps and no collisions. That is the strongest correctness statement available
|
|
107
|
+
for a permutation, and it is run in CI rather than kept as a manual check.
|
|
108
|
+
|
|
109
|
+
### The known failure modes are tested for by name
|
|
110
|
+
|
|
111
|
+
Published FF1 bugs cluster in a few places. Each has a dedicated test:
|
|
112
|
+
|
|
113
|
+
| Known failure mode | How it is prevented |
|
|
114
|
+
|---|---|
|
|
115
|
+
| Floating-point `ceil(v · log₂(radix))` — the Bouncy Castle bug class | Exact integer arithmetic; an AST scan fails the build if `math.log`, `ceil`, `/` or any float literal appears in the core |
|
|
116
|
+
| `b` derived from `u` instead of `v` | Asserted against the traced value, not a round-trip — which passes either way |
|
|
117
|
+
| Wrong `S` expansion when `d > 16` | Differential cases at each block-count transition; no NIST sample reaches this branch |
|
|
118
|
+
| Mirrored parity rule in decrypt | Encrypt and decrypt share one code path for it |
|
|
119
|
+
| Silent coercion of bad input | Every rejection raises a typed exception; a 50-case sweep asserts nothing escapes as a bare `AttributeError` or `KeyError` |
|
|
120
|
+
|
|
121
|
+
### Migration is safe by construction
|
|
122
|
+
|
|
123
|
+
Output is byte-identical to `ubiq_security_fpe`, verified in **both** directions — old ciphertext
|
|
124
|
+
decrypts with this library, and new ciphertext decrypts with the old one. Existing encrypted data
|
|
125
|
+
stays readable, and a rollback strands nothing. See [Migrating](#migrating-from-ubiq_security_fpe).
|
|
126
|
+
|
|
127
|
+
### What this is *not*
|
|
128
|
+
|
|
129
|
+
Passing the published sample vectors is **conformance evidence, not FIPS validation**. This package
|
|
130
|
+
is not FIPS 140 validated and makes no such claim. It also does not attempt key zeroization, and
|
|
131
|
+
offers no constant-time guarantee — see [`SECURITY.md`](SECURITY.md) for the full statement of
|
|
132
|
+
limitations.
|
|
133
|
+
|
|
134
|
+
## Why FF1 only — and why FF3 is excluded
|
|
135
|
+
|
|
136
|
+
SP 800-38G originally specified two modes, FF1 and FF3. FF3 was revised to FF3-1 after an attack
|
|
137
|
+
on the original construction, but Beyne subsequently demonstrated a weakness in the **tweak
|
|
138
|
+
schedule** that affects FF3 and FF3-1 alike — the repair did not address the underlying problem.
|
|
139
|
+
|
|
140
|
+
The **February 2025 second public draft of SP 800-38G Rev. 1 removes FF3 entirely**, leaving FF1
|
|
141
|
+
as the only approved format-preserving mode.
|
|
142
|
+
|
|
143
|
+
`fpr-ff1` will therefore never implement FF3 or FF3-1. This is a deliberate feature, not an
|
|
144
|
+
omission: there is no configuration flag, no opt-in, and no plan to add one. If you need FF3 you
|
|
145
|
+
need a different library, and you should first satisfy yourself that you actually need a mode
|
|
146
|
+
NIST has withdrawn.
|
|
147
|
+
|
|
148
|
+
## Domain limits are stricter than the 2016 text
|
|
149
|
+
|
|
150
|
+
This package implements SP 800-38G (2016, updated 2019) as the normative algorithm, but enforces
|
|
151
|
+
the **tightened constraints from the Rev. 1 second public draft**:
|
|
152
|
+
|
|
153
|
+
| Constraint | This package | SP 800-38G (2016) |
|
|
154
|
+
|---|---|---|
|
|
155
|
+
| Minimum domain | `radix ** minlen >= 1_000_000` | `radix ** minlen >= 100` |
|
|
156
|
+
| Maximum length | `2 ** 32 - 1` | `2 ** 32 - 1` |
|
|
157
|
+
| Key sizes | 128, 192, 256 bits | same |
|
|
158
|
+
| Radix | `2 <= radix < 2 ** 16` | same |
|
|
159
|
+
| Rounds | exactly 10 | same |
|
|
160
|
+
|
|
161
|
+
The minimum-domain rule is the one that will bite. A domain of only 100 values is trivially
|
|
162
|
+
enumerable, so this package **fails closed** and rejects it. Concretely, `min_length` is 6 for
|
|
163
|
+
radix 10 and 4 for radix 36 — inputs shorter than that raise `LengthError`, even though some
|
|
164
|
+
older libraries (including `ubiq_security_fpe`) accept them.
|
|
165
|
+
|
|
166
|
+
Rev. 1 is still a draft. If it is finalised with different limits, that will be a breaking change
|
|
167
|
+
and a major version.
|
|
168
|
+
|
|
169
|
+
## Scope
|
|
170
|
+
|
|
171
|
+
- **In scope:** FF1 encryption and decryption, numeral and string interfaces, alphabet handling, parameter validation, tests, documentation.
|
|
172
|
+
- **Out of scope, permanently:** FF3/FF3-1, identifier generation, persistence, checksums, key generation/storage/derivation, application-specific defaults or alphabets.
|
|
173
|
+
|
|
174
|
+
## Supported Python versions
|
|
175
|
+
|
|
176
|
+
**3.12, 3.13 and 3.14.** The upper bound in `requires-python` is deliberate: it matches the
|
|
177
|
+
versions actually exercised in CI on Linux, macOS and Windows. It is raised as part of a release
|
|
178
|
+
once a newer Python is in the matrix and green, rather than being left open and assumed to work.
|
|
179
|
+
|
|
180
|
+
## Roadmap
|
|
181
|
+
|
|
182
|
+
| Version | Focus |
|
|
183
|
+
|---|---|
|
|
184
|
+
| **1.0** | **Pure Python.** Conformance, a stable API, and a single runtime dependency (`cryptography`). No compiled extension, no optional backends — one code path, and it is the one the vectors test. |
|
|
185
|
+
| **2.0** | **Optional accelerated backend.** An opt-in faster path for high-throughput callers, with the pure-Python implementation retained as the reference and the default. |
|
|
186
|
+
|
|
187
|
+
The 2.0 backend is explicitly *not* a 1.0 concern. An accelerated path is only worth having once
|
|
188
|
+
the reference implementation is settled and there is a conformance suite strong enough to prove
|
|
189
|
+
the two agree bit for bit — which is the point of the differential and interoperability tests.
|
|
190
|
+
|
|
191
|
+
Nothing in the roadmap changes the scope boundary above. FF3 and FF3-1 remain permanently out of
|
|
192
|
+
scope, and no release will add key management.
|
|
193
|
+
|
|
194
|
+
## API
|
|
195
|
+
|
|
196
|
+
### `FF1(key, radix, *, alphabet=None, tweak=b"", min_tweak_len=None, max_tweak_len=None)`
|
|
197
|
+
|
|
198
|
+
| Parameter | Description |
|
|
199
|
+
|---|---|
|
|
200
|
+
| `key` | 16, 24, or 32 bytes. |
|
|
201
|
+
| `radix` | Integer base of the numeral system. |
|
|
202
|
+
| `alphabet` | Optional string of exactly `radix` unique characters; enables `encrypt`/`decrypt`. |
|
|
203
|
+
| `tweak` | Default tweak used when not supplied per call. |
|
|
204
|
+
| `min_tweak_len` / `max_tweak_len` | Optional per-instance tweak length bounds. |
|
|
205
|
+
|
|
206
|
+
### Numeral interface
|
|
207
|
+
|
|
208
|
+
The primitive interface works on integers in `[0, radix)`.
|
|
209
|
+
|
|
210
|
+
```python
|
|
211
|
+
ciphertext = ff1.encrypt_numerals([1, 2, 3, 4, 5, 6])
|
|
212
|
+
plaintext = ff1.decrypt_numerals(ciphertext)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
**Accepted numeral types.** Anything losslessly integral — `int`, `IntEnum`, and integers from
|
|
216
|
+
other numeric libraries such as NumPy, which are normalised to Python `int` so fixed-width values
|
|
217
|
+
cannot overflow in the internal big-integer arithmetic.
|
|
218
|
+
|
|
219
|
+
`float`, `Decimal`, `Fraction` and `str` are **rejected** with `ValueRangeError`, even when they
|
|
220
|
+
compare equal to a valid numeral: `1.0 < 10` is `True`, so comparison alone is not a type check.
|
|
221
|
+
|
|
222
|
+
`bool` is **rejected deliberately**. `True` would otherwise encrypt silently as `1`, and a list of
|
|
223
|
+
booleans arriving here is a caller mistake, not an intent to encrypt ones and zeros.
|
|
224
|
+
|
|
225
|
+
The input must be a `Sequence` — something with a known length. A generator raises `TypeError`
|
|
226
|
+
(not `FF1Error`), because that is misuse of the API rather than bad data; wrap it in `list(...)`.
|
|
227
|
+
|
|
228
|
+
### String interface
|
|
229
|
+
|
|
230
|
+
When `alphabet` is provided, the string interface maps characters to numerals and back.
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
ff1.encrypt("123456")
|
|
234
|
+
ff1.decrypt("654321")
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
**Alphabet uniqueness is by Unicode code point.** FF1 operates on code points, so normalisation is
|
|
238
|
+
the caller's responsibility. Precomposed `é` (U+00E9) and decomposed `é` (U+0065 U+0301) are
|
|
239
|
+
visually identical but count as two distinct symbols, and an alphabet containing both is accepted.
|
|
240
|
+
If your alphabet comes from user input or an external source, normalise it first:
|
|
241
|
+
|
|
242
|
+
```python
|
|
243
|
+
import unicodedata
|
|
244
|
+
alphabet = unicodedata.normalize("NFC", alphabet)
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
### Exceptions
|
|
248
|
+
|
|
249
|
+
Every rejection raises a typed exception derived from `FF1Error`. Nothing is silently truncated,
|
|
250
|
+
padded, coerced or clamped.
|
|
251
|
+
|
|
252
|
+
| Exception | Raised when |
|
|
253
|
+
|---|---|
|
|
254
|
+
| `KeyLengthError` | key is not 16, 24 or 32 bytes |
|
|
255
|
+
| `RadixError` | radix outside `2 <= radix < 2**16` |
|
|
256
|
+
| `LengthError` | input length outside `[min_length, max_length]` |
|
|
257
|
+
| `ValueRangeError` | a numeral outside `[0, radix)`, or a character absent from the alphabet |
|
|
258
|
+
| `TweakLengthError` | tweak outside the configured bounds |
|
|
259
|
+
| `AlphabetError` | alphabet length mismatched to radix, or containing duplicates |
|
|
260
|
+
|
|
261
|
+
`AlphabetError` signals malformed *configuration* (caught at construction); `ValueRangeError`
|
|
262
|
+
signals malformed *data* (caught per call). They are deliberately distinct so callers can handle
|
|
263
|
+
a programming error differently from a bad input record.
|
|
264
|
+
|
|
265
|
+
## Migrating from `ubiq_security_fpe`
|
|
266
|
+
|
|
267
|
+
`fpr-ff1` exists to replace `ubiq_security_fpe`, which was deprecated in favour of a SaaS client
|
|
268
|
+
and is no longer maintained. **The two produce identical ciphertext for identical inputs**, so
|
|
269
|
+
existing encrypted data stays readable — no re-encryption, no migration window, no rollback risk.
|
|
270
|
+
|
|
271
|
+
That claim is enforced by `tests/test_interoperability.py`, which checks both directions (old
|
|
272
|
+
ciphertext decrypts with the new library and vice versa) across all three key sizes, tweaked and
|
|
273
|
+
untweaked. Migration safety is treated as a correctness obligation, not a promise.
|
|
274
|
+
|
|
275
|
+
**No compatibility shim ships, deliberately.** A `Context(...)` / `.Encrypt()` drop-in would mean
|
|
276
|
+
maintaining a permanent second API, in a naming style this project does not use, mirroring a
|
|
277
|
+
library that is itself deprecated. The migration below is three mechanical edits per call site,
|
|
278
|
+
and the part that would actually be hard — identical ciphertext — is already done.
|
|
279
|
+
|
|
280
|
+
### API mapping
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
# before
|
|
284
|
+
from ubiq_security_fpe import ff1
|
|
285
|
+
ctx = ff1.Context(key, tweak, twk_min_len, twk_max_len, radix, alphabet)
|
|
286
|
+
ciphertext = ctx.Encrypt(plaintext, None)
|
|
287
|
+
plaintext = ctx.Decrypt(ciphertext, None)
|
|
288
|
+
|
|
289
|
+
# after
|
|
290
|
+
from fpr_ff1 import FF1
|
|
291
|
+
ctx = FF1(key, radix, alphabet=alphabet, tweak=tweak,
|
|
292
|
+
min_tweak_len=twk_min_len, max_tweak_len=twk_max_len)
|
|
293
|
+
ciphertext = ctx.encrypt(plaintext)
|
|
294
|
+
plaintext = ctx.decrypt(ciphertext)
|
|
295
|
+
```
|
|
296
|
+
|
|
297
|
+
| `ubiq_security_fpe` | `fpr-ff1` |
|
|
298
|
+
|---|---|
|
|
299
|
+
| `ff1.Context(key, twk, twk_min_len, twk_max_len, radix, alpha)` | `FF1(key, radix, alphabet=..., tweak=..., min_tweak_len=..., max_tweak_len=...)` |
|
|
300
|
+
| `ctx.Encrypt(pt, twk)` | `ctx.encrypt(pt, twk)` |
|
|
301
|
+
| `ctx.Decrypt(ct, twk)` | `ctx.decrypt(ct, twk)` |
|
|
302
|
+
| — | `ctx.encrypt_numerals(...)` / `ctx.decrypt_numerals(...)` (no alphabet needed) |
|
|
303
|
+
| `RuntimeError` for every rejection | typed exceptions under `FF1Error` |
|
|
304
|
+
|
|
305
|
+
### Behaviour changes to check before you switch
|
|
306
|
+
|
|
307
|
+
1. **Shorter inputs are rejected.** `fpr-ff1` enforces `radix ** minlen >= 1_000_000`; the legacy
|
|
308
|
+
library used the same rule, but if you relied on any library using the 2016 `>= 100` bound,
|
|
309
|
+
inputs below `ctx.min_length` now raise `LengthError`. Check `ctx.min_length` for your radix.
|
|
310
|
+
2. **Errors are typed.** Rejections raise `KeyLengthError`, `RadixError`, `LengthError`,
|
|
311
|
+
`ValueRangeError`, `TweakLengthError` or `AlphabetError` — all subclasses of `FF1Error` —
|
|
312
|
+
rather than bare `RuntimeError`. Catch `FF1Error` if you want the old catch-all behaviour.
|
|
313
|
+
3. **No `M2Crypto` dependency.** `fpr-ff1` depends only on `cryptography`.
|
|
314
|
+
4. **Alphabet is validated at construction.** A wrong-length alphabet or one with duplicate
|
|
315
|
+
characters raises `AlphabetError` immediately rather than misbehaving later.
|
|
316
|
+
|
|
317
|
+
## FIPS disclaimer
|
|
318
|
+
|
|
319
|
+
Passing the published NIST sample vectors is evidence of conformance. It is **not** FIPS validation. This package makes no claims of FIPS 140 conformance.
|
|
320
|
+
|
|
321
|
+
## Key material
|
|
322
|
+
|
|
323
|
+
`fpr-ff1` does not attempt to zeroize key material. Python `bytes` are immutable and the interpreter may copy them during garbage collection.
|
|
324
|
+
|
|
325
|
+
## Development
|
|
326
|
+
|
|
327
|
+
Requires Python 3.12, `uv`, and `just`.
|
|
328
|
+
|
|
329
|
+
```bash
|
|
330
|
+
just setup # create venv and install deps
|
|
331
|
+
just quality # format check, lint, typecheck, tests
|
|
332
|
+
just build # quality gate + uv build
|
|
333
|
+
just secrets # gitleaks scan (must be installed locally)
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
## Documentation
|
|
337
|
+
|
|
338
|
+
- `docs/architecture.md` — design and module overview
|
|
339
|
+
- `docs/developer-guide.md` — setup, commands, testing, and CI
|
|
340
|
+
- `docs/directory-structure.md` — repository layout
|
|
341
|
+
- `docs/configuration.md` — `FF1` constructor parameters and runtime constraints
|
|
342
|
+
- `docs/backlog.md` — active and completed work
|
|
343
|
+
- `CHANGELOG.md` — release history, including behaviour changes that affect accepted inputs
|
|
344
|
+
- `SECURITY.md` — disclosure process and known limitations
|
|
345
|
+
|
|
346
|
+
## License
|
|
347
|
+
|
|
348
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
fpr_ff1/__init__.py,sha256=kbnpTIr-F_D9Cp2_5QVcIL9GxT2twJ_HyckTbnA8T40,422
|
|
2
|
+
fpr_ff1/_exceptions.py,sha256=kCUG4KYiwsjiGLkespCJsPSnXvD1_pRp7Ke7VxfQSLk,913
|
|
3
|
+
fpr_ff1/_ff1.py,sha256=DfYAciuBIkIzO0SklMWxM2pgFgCS670eMvggwiwdaVg,21545
|
|
4
|
+
fpr_ff1/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
fpr_ff1-0.1.0.dist-info/METADATA,sha256=ECH4p073Xk3KoJTT98iynfEjyy1NrceX_OWvfZYzGDg,16289
|
|
6
|
+
fpr_ff1-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
7
|
+
fpr_ff1-0.1.0.dist-info/licenses/LICENSE,sha256=FGQ9O0FIhqnsoDFIFBbwJXU_T9LEJnVeU6KRLn0Vs-I,1065
|
|
8
|
+
fpr_ff1-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Joel Lee
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|