fpr-ff1 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
fpr_ff1/__init__.py ADDED
@@ -0,0 +1,23 @@
1
+ """FF1 format-preserving encryption (NIST SP 800-38G)."""
2
+
3
+ from fpr_ff1._exceptions import (
4
+ AlphabetError,
5
+ FF1Error,
6
+ KeyLengthError,
7
+ LengthError,
8
+ RadixError,
9
+ TweakLengthError,
10
+ ValueRangeError,
11
+ )
12
+ from fpr_ff1._ff1 import FF1
13
+
14
+ __all__ = [
15
+ "FF1",
16
+ "AlphabetError",
17
+ "FF1Error",
18
+ "KeyLengthError",
19
+ "LengthError",
20
+ "RadixError",
21
+ "TweakLengthError",
22
+ "ValueRangeError",
23
+ ]
fpr_ff1/_exceptions.py ADDED
@@ -0,0 +1,34 @@
1
+ """Typed exceptions for the FF1 implementation."""
2
+
3
+
4
+ class FF1Error(Exception):
5
+ """Base for all FF1 errors."""
6
+
7
+
8
+ class KeyLengthError(FF1Error):
9
+ """Key length is not 16, 24, or 32 bytes."""
10
+
11
+
12
+ class RadixError(FF1Error):
13
+ """Radix is outside ``2 <= radix < 2**16``."""
14
+
15
+
16
+ class LengthError(FF1Error):
17
+ """Input length is outside the valid domain for the radix."""
18
+
19
+
20
+ class ValueRangeError(FF1Error):
21
+ """Input data is outside the domain the instance accepts.
22
+
23
+ Raised for a numeral outside ``[0, radix)``, and for a character that does
24
+ not appear in the configured alphabet. Both are faults in the *data* being
25
+ encrypted; malformed *configuration* raises :class:`AlphabetError` instead.
26
+ """
27
+
28
+
29
+ class TweakLengthError(FF1Error):
30
+ """Tweak length is outside the configured bounds."""
31
+
32
+
33
+ class AlphabetError(FF1Error):
34
+ """Alphabet length or uniqueness does not match the radix."""
fpr_ff1/_ff1.py ADDED
@@ -0,0 +1,549 @@
1
+ """FF1 implementation following NIST SP 800-38G Algorithm 7."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import operator
6
+ from collections.abc import Sequence
7
+ from typing import ClassVar, NamedTuple, SupportsIndex, cast
8
+
9
+ from cryptography.hazmat.primitives.ciphers import (
10
+ Cipher,
11
+ CipherContext,
12
+ algorithms,
13
+ modes,
14
+ )
15
+
16
+ from fpr_ff1._exceptions import (
17
+ AlphabetError,
18
+ FF1Error,
19
+ KeyLengthError,
20
+ LengthError,
21
+ RadixError,
22
+ TweakLengthError,
23
+ ValueRangeError,
24
+ )
25
+
26
+ #: One round's intermediate values, as published in the NIST sample document.
27
+ #: Test-only; see :meth:`FF1._encrypt_traced`.
28
+ type TraceRecord = dict[str, object]
29
+
30
+
31
+ def _require_int(value: object, name: str, error: type[FF1Error]) -> int:
32
+ """Return ``value`` as an ``int``, rejecting anything not losslessly integral.
33
+
34
+ Uses ``operator.index()``, Python's own "this is an integer" protocol:
35
+ ``float`` and ``Decimal`` deliberately do not implement it, while ``int``,
36
+ ``IntEnum`` and NumPy integers do. Comparison alone is not enough of a
37
+ gate -- ``1.0 < 10`` is ``True``, so a float numeral would otherwise pass
38
+ validation and fail much later with an ``AttributeError`` from outside the
39
+ :class:`FF1Error` hierarchy.
40
+
41
+ ``bool`` is rejected explicitly even though it implements ``__index__``:
42
+ silently encrypting ``True`` as ``1`` is the coercion the contract forbids,
43
+ and a sequence of booleans reaching this point is a caller mistake.
44
+
45
+ Converting rather than merely checking also normalises NumPy integers to
46
+ Python ``int``, which matters: fixed-width integers would overflow silently
47
+ in the big-integer arithmetic downstream.
48
+ """
49
+ if isinstance(value, bool):
50
+ raise error(f"{name} must be an integer, not bool")
51
+ try:
52
+ # The cast asserts only that __index__ *might* exist; TypeError below
53
+ # is the actual gate.
54
+ return operator.index(cast("SupportsIndex", value))
55
+ except TypeError:
56
+ raise error(f"{name} must be an integer, got {type(value).__name__}") from None
57
+
58
+
59
+ def _validate_tweak_bounds(
60
+ min_tweak_len: int | None, max_tweak_len: int | None
61
+ ) -> tuple[int | None, int | None]:
62
+ """Type- and sanity-check the configured tweak length bounds.
63
+
64
+ These are *configuration* faults, distinct from a tweak that merely
65
+ violates an otherwise valid bound, and they are caught at construction so
66
+ an unusable instance can never be built.
67
+
68
+ Raises:
69
+ TweakLengthError: if a bound is not an integer, is negative, or if the
70
+ two bounds are mutually unsatisfiable.
71
+ """
72
+ low = (
73
+ None
74
+ if min_tweak_len is None
75
+ else _require_int(min_tweak_len, "min_tweak_len", TweakLengthError)
76
+ )
77
+ high = (
78
+ None
79
+ if max_tweak_len is None
80
+ else _require_int(max_tweak_len, "max_tweak_len", TweakLengthError)
81
+ )
82
+
83
+ for name, bound in (("min_tweak_len", low), ("max_tweak_len", high)):
84
+ if bound is not None and bound < 0:
85
+ # A negative bound is inert rather than harmful, but it silently
86
+ # means "no constraint" -- which is not what the caller asked for.
87
+ raise TweakLengthError(f"{name} must be non-negative, got {bound}")
88
+
89
+ if low is not None and high is not None and low > high:
90
+ raise TweakLengthError(
91
+ f"min_tweak_len {low} exceeds max_tweak_len {high}; "
92
+ "no tweak length could satisfy both bounds"
93
+ )
94
+
95
+ return low, high
96
+
97
+
98
+ def _require_bytes(value: object, name: str, error: type[FF1Error]) -> bytes:
99
+ """Return ``value`` as ``bytes``, rejecting non-bytes-like input."""
100
+ if not isinstance(value, bytes | bytearray | memoryview):
101
+ raise error(f"{name} must be bytes-like, got {type(value).__name__}")
102
+ # Normalise to immutable bytes: a caller holding the bytearray must not be
103
+ # able to mutate a tweak or key after construction.
104
+ return bytes(cast("bytes | bytearray | memoryview[int]", value))
105
+
106
+
107
+ class _Aes(NamedTuple):
108
+ """AES objects shared for the lifetime of one :class:`FF1` instance.
109
+
110
+ ``algorithm`` and ``cbc_zero_iv`` are immutable configuration, reused to
111
+ avoid rebuilding them on every PRF call. ``ecb_encryptor`` is a live
112
+ context, but ECB carries no chaining state so repeated ``update()`` calls
113
+ are safe. A CBC *encryptor* is never held here: it would carry chaining
114
+ state between PRF invocations.
115
+ """
116
+
117
+ algorithm: algorithms.AES
118
+ cbc_zero_iv: modes.CBC
119
+ ecb_encryptor: CipherContext
120
+
121
+
122
+ class FF1:
123
+ """FF1 format-preserving encryption primitive and string wrapper."""
124
+
125
+ # SP 800-38G requires "minlen <= n <= maxlen < 2**32", so the largest
126
+ # admissible length is 2**32 - 1, not 2**32. Failing closed on the
127
+ # boundary matches the project's stance elsewhere; the excluded value is
128
+ # unconstructable in practice (a 2**32-element list needs tens of GB).
129
+ _MAX_LEN: ClassVar[int] = 2**32 - 1
130
+ _RADIX_MIN: ClassVar[int] = 2
131
+ _RADIX_MAX_EXCLUSIVE: ClassVar[int] = 2**16
132
+
133
+ def __init__(
134
+ self,
135
+ key: bytes,
136
+ radix: int,
137
+ *,
138
+ alphabet: str | None = None,
139
+ tweak: bytes = b"",
140
+ min_tweak_len: int | None = None,
141
+ max_tweak_len: int | None = None,
142
+ ) -> None:
143
+ """Create an FF1 instance.
144
+
145
+ Args:
146
+ key: AES key; must be 16, 24, or 32 bytes.
147
+ radix: Numeral base, ``2 <= radix < 2**16``.
148
+ alphabet: Optional string of exactly ``radix`` unique characters.
149
+ Required for the string interface.
150
+ tweak: Default tweak used when not provided per call.
151
+ min_tweak_len: Optional inclusive lower bound on tweak length.
152
+ max_tweak_len: Optional inclusive upper bound on tweak length.
153
+
154
+ Raises:
155
+ KeyLengthError: if the key is not bytes-like or has an invalid length.
156
+ RadixError: if the radix is not an integer or is out of range.
157
+ TweakLengthError: if the tweak is not bytes-like or out of bounds.
158
+ AlphabetError: if the alphabet is not a string or is malformed.
159
+ """
160
+ # Types are checked before values throughout: a wrong type is the more
161
+ # fundamental fault, and reporting a range error for a float would be
162
+ # actively misleading.
163
+ key = _require_bytes(key, "key", KeyLengthError)
164
+ if len(key) not in {16, 24, 32}:
165
+ raise KeyLengthError(f"key must be 16, 24, or 32 bytes, got {len(key)}")
166
+
167
+ radix = _require_int(radix, "radix", RadixError)
168
+ if radix < self._RADIX_MIN or radix >= self._RADIX_MAX_EXCLUSIVE:
169
+ raise RadixError(
170
+ f"radix must satisfy {self._RADIX_MIN} <= radix < {self._RADIX_MAX_EXCLUSIVE}, "
171
+ f"got {radix!r}"
172
+ )
173
+
174
+ self._key = key
175
+ self._radix = radix
176
+
177
+ # Bounds are validated before the default tweak is checked against
178
+ # them, so an unsatisfiable configuration is reported as such rather
179
+ # than as whichever bound the default tweak happened to violate first.
180
+ self._min_tweak_len, self._max_tweak_len = _validate_tweak_bounds(
181
+ min_tweak_len, max_tweak_len
182
+ )
183
+ tweak = _require_bytes(tweak, "tweak", TweakLengthError)
184
+ self._validate_tweak(tweak)
185
+ self._default_tweak = tweak
186
+
187
+ self._alphabet: str | None = None
188
+ self._char_to_index: dict[str, int] | None = None
189
+ self._index_to_char: list[str] | None = None
190
+ if alphabet is not None:
191
+ # Checked at runtime despite the annotation: type hints are not
192
+ # enforced, and a list alphabet silently worked before this guard.
193
+ if not isinstance(alphabet, str): # pyright: ignore[reportUnnecessaryIsInstance]
194
+ raise AlphabetError(f"alphabet must be a str, got {type(alphabet).__name__}")
195
+ if len(alphabet) != radix:
196
+ raise AlphabetError(f"alphabet length {len(alphabet)} does not match radix {radix}")
197
+ # Uniqueness is by Unicode code point. Two visually identical but
198
+ # differently-normalised symbols (e.g. precomposed vs decomposed
199
+ # accents) are distinct here; normalisation is the caller's job.
200
+ if len(set(alphabet)) != len(alphabet):
201
+ raise AlphabetError("alphabet contains duplicate characters")
202
+ self._alphabet = alphabet
203
+ self._char_to_index = {ch: i for i, ch in enumerate(alphabet)}
204
+ self._index_to_char = list(alphabet)
205
+
206
+ # Every legal radix admits a feasible length: min_length peaks at 20
207
+ # (radix 2), far below _MAX_LEN, so no infeasibility check is needed.
208
+ self._min_length = _min_length(radix)
209
+
210
+ # Cipher objects reused across calls. Do not call finalize() on the
211
+ # ECB encryptor; it must stay alive for the instance's lifetime.
212
+ algorithm = algorithms.AES(key)
213
+ self._aes = _Aes(
214
+ algorithm=algorithm,
215
+ cbc_zero_iv=modes.CBC(b"\x00" * 16),
216
+ ecb_encryptor=Cipher(algorithm, modes.ECB()).encryptor(),
217
+ )
218
+
219
+ @property
220
+ def min_length(self) -> int:
221
+ """Minimum permitted input length for this radix."""
222
+ return self._min_length
223
+
224
+ @property
225
+ def max_length(self) -> int:
226
+ """Maximum permitted input length for this radix.
227
+
228
+ ``2**32 - 1``: SP 800-38G specifies ``maxlen < 2**32``.
229
+ """
230
+ return self._MAX_LEN
231
+
232
+ def _validate_tweak(self, tweak: bytes) -> None:
233
+ if self._min_tweak_len is not None and len(tweak) < self._min_tweak_len:
234
+ raise TweakLengthError(f"tweak length {len(tweak)} below minimum {self._min_tweak_len}")
235
+ if self._max_tweak_len is not None and len(tweak) > self._max_tweak_len:
236
+ raise TweakLengthError(f"tweak length {len(tweak)} above maximum {self._max_tweak_len}")
237
+
238
+ def _validate_length(self, n: int, inout: str) -> None:
239
+ if n < self._min_length:
240
+ raise LengthError(
241
+ f"{inout} length {n} below minimum {self._min_length} for radix {self._radix}"
242
+ )
243
+ if n > self._MAX_LEN:
244
+ raise LengthError(f"{inout} length {n} above maximum {self._MAX_LEN}")
245
+
246
+ def _coerce_numerals(self, x: Sequence[int], inout: str) -> list[int]:
247
+ """Type-check, normalise and range-check numerals in a single pass.
248
+
249
+ Returns true Python ``int`` values, so fixed-width integers from other
250
+ numeric libraries cannot reach the big-integer arithmetic downstream
251
+ and overflow silently.
252
+ """
253
+ radix = self._radix
254
+ numerals: list[int] = []
255
+ for idx, value in enumerate(x):
256
+ numeral = _require_int(value, f"{inout}[{idx}]", ValueRangeError)
257
+ if numeral < 0 or numeral >= radix:
258
+ raise ValueRangeError(f"{inout}[{idx}]={numeral!r} out of range for radix {radix}")
259
+ numerals.append(numeral)
260
+ return numerals
261
+
262
+ def _prepare(
263
+ self, x: Sequence[int], tweak: bytes | None, inout: str
264
+ ) -> tuple[list[int], bytes]:
265
+ """Validate and normalise the inputs shared by encrypt and decrypt."""
266
+ t = (
267
+ self._default_tweak
268
+ if tweak is None
269
+ else _require_bytes(tweak, "tweak", TweakLengthError)
270
+ )
271
+ # Check the length before materialising the sequence: an over-long
272
+ # input must be rejected without first allocating a copy of it.
273
+ if not hasattr(x, "__len__"):
274
+ raise TypeError(
275
+ f"{inout} must be a Sequence[int] with a known length, got "
276
+ f"{type(x).__name__}; wrap it with list(...) if it is an iterator"
277
+ )
278
+ self._validate_length(len(x), inout)
279
+ numerals = self._coerce_numerals(x, inout)
280
+ self._validate_tweak(t)
281
+ return numerals, t
282
+
283
+ def _alphabet_maps(self, numeral_method: str) -> tuple[dict[str, int], list[str]]:
284
+ """Return the alphabet lookup tables, or raise if none was configured."""
285
+ if self._char_to_index is None or self._index_to_char is None:
286
+ raise FF1Error(
287
+ f"alphabet required for string interface; use {numeral_method} "
288
+ "for the numeral interface"
289
+ )
290
+ return self._char_to_index, self._index_to_char
291
+
292
+ def _decode_str(self, s: str, char_to_index: dict[str, int]) -> list[int]:
293
+ """Map characters to numerals, rejecting anything outside the alphabet."""
294
+ # Checked at runtime despite the annotation; see _require_int.
295
+ if not isinstance(s, str): # pyright: ignore[reportUnnecessaryIsInstance]
296
+ raise ValueRangeError(f"input must be a str, got {type(s).__name__}")
297
+ numerals: list[int] = []
298
+ for idx, ch in enumerate(s):
299
+ value = char_to_index.get(ch)
300
+ if value is None:
301
+ raise ValueRangeError(f"character {ch!r} at index {idx} is not in the alphabet")
302
+ numerals.append(value)
303
+ return numerals
304
+
305
+ def encrypt_numerals(self, x: Sequence[int], tweak: bytes | None = None) -> list[int]:
306
+ """Encrypt a sequence of numerals.
307
+
308
+ Args:
309
+ x: List of integers in ``[0, radix)``.
310
+ tweak: Tweak bytes; defaults to the instance tweak.
311
+
312
+ Returns:
313
+ Encrypted numeral sequence of the same length.
314
+
315
+ Raises:
316
+ LengthError: if the input length is outside the valid domain.
317
+ ValueRangeError: if any numeral is outside ``[0, radix)``.
318
+ TweakLengthError: if the tweak is out of bounds.
319
+ """
320
+ numerals, t = self._prepare(x, tweak, "plaintext")
321
+ return _ff1(self._aes, self._radix, numerals, t, encrypt=True)
322
+
323
+ def decrypt_numerals(self, x: Sequence[int], tweak: bytes | None = None) -> list[int]:
324
+ """Decrypt a sequence of numerals.
325
+
326
+ Args:
327
+ x: List of integers in ``[0, radix)``.
328
+ tweak: Tweak bytes; defaults to the instance tweak.
329
+
330
+ Returns:
331
+ Decrypted numeral sequence of the same length.
332
+
333
+ Raises:
334
+ LengthError: if the input length is outside the valid domain.
335
+ ValueRangeError: if any numeral is outside ``[0, radix)``.
336
+ TweakLengthError: if the tweak is out of bounds.
337
+ """
338
+ numerals, t = self._prepare(x, tweak, "ciphertext")
339
+ return _ff1(self._aes, self._radix, numerals, t, encrypt=False)
340
+
341
+ def _encrypt_traced(
342
+ self, x: Sequence[int], tweak: bytes | None = None
343
+ ) -> tuple[list[int], list[TraceRecord]]:
344
+ """Encrypt, also returning the per-round intermediates.
345
+
346
+ Test-only conformance hook, deliberately kept off the public methods:
347
+ the NIST sample document publishes ``P``, ``Q``, ``R``, ``S``, ``y``,
348
+ ``m``, ``c`` and ``C`` for every round, and two compensating bugs can
349
+ agree on the final output while disagreeing here.
350
+
351
+ Not exported from the package and not part of the supported API.
352
+ """
353
+ numerals, t = self._prepare(x, tweak, "plaintext")
354
+ trace: list[TraceRecord] = []
355
+ return _ff1(self._aes, self._radix, numerals, t, encrypt=True, _trace=trace), trace
356
+
357
+ def encrypt(self, s: str, tweak: bytes | None = None) -> str:
358
+ """Encrypt a string using the configured alphabet.
359
+
360
+ Raises:
361
+ FF1Error: if no alphabet was configured at construction.
362
+ ValueRangeError: if a character is absent from the alphabet.
363
+ """
364
+ char_to_index, index_to_char = self._alphabet_maps("encrypt_numerals")
365
+ numerals = self._decode_str(s, char_to_index)
366
+ encrypted = self.encrypt_numerals(numerals, tweak)
367
+ return "".join(index_to_char[i] for i in encrypted)
368
+
369
+ def decrypt(self, s: str, tweak: bytes | None = None) -> str:
370
+ """Decrypt a string using the configured alphabet.
371
+
372
+ Raises:
373
+ FF1Error: if no alphabet was configured at construction.
374
+ ValueRangeError: if a character is absent from the alphabet.
375
+ """
376
+ char_to_index, index_to_char = self._alphabet_maps("decrypt_numerals")
377
+ numerals = self._decode_str(s, char_to_index)
378
+ decrypted = self.decrypt_numerals(numerals, tweak)
379
+ return "".join(index_to_char[i] for i in decrypted)
380
+
381
+
382
+ _MIN_DOMAIN = 1_000_000
383
+
384
+
385
+ def _min_length(radix: int) -> int:
386
+ """Return the smallest n with radix**n >= 1_000_000."""
387
+ n = 1
388
+ value = radix
389
+ while value < _MIN_DOMAIN:
390
+ value *= radix
391
+ n += 1
392
+ return n
393
+
394
+
395
+ def _num_radix(radix: int, numerals: Sequence[int]) -> int:
396
+ """Decode a sequence of numerals as a big-endian base-radix integer."""
397
+ value = 0
398
+ for x in numerals:
399
+ value = value * radix + x
400
+ return value
401
+
402
+
403
+ def _str_radix(value: int, radix: int, length: int) -> list[int]:
404
+ """Encode a non-negative integer as ``length`` big-endian base-radix numerals."""
405
+ out = [0] * length
406
+ for i in range(length - 1, -1, -1):
407
+ out[i] = value % radix
408
+ value //= radix
409
+ return out
410
+
411
+
412
+ def _prf(aes: _Aes, data: bytes) -> bytes:
413
+ """SP 800-38G Algorithm 6 (PRF): CBC-MAC with a zero IV.
414
+
415
+ Invoked from Algorithm 7 step 6.ii as ``PRF(P || Q)``.
416
+ """
417
+ # data is already 16-byte aligned by callers. A fresh encryptor per call
418
+ # is required: CBC chaining state must never persist between PRF calls.
419
+ encryptor = Cipher(aes.algorithm, aes.cbc_zero_iv).encryptor()
420
+ result = encryptor.update(data) + encryptor.finalize()
421
+ return result[-16:]
422
+
423
+
424
+ def _ff1(
425
+ aes: _Aes,
426
+ radix: int,
427
+ x: list[int],
428
+ tweak: bytes,
429
+ *,
430
+ encrypt: bool,
431
+ _trace: list[TraceRecord] | None = None,
432
+ ) -> list[int]:
433
+ """SP 800-38G Algorithm 7 core."""
434
+ n = len(x)
435
+
436
+ # Step 1: u = floor(n/2), v = n - u
437
+ u = n // 2
438
+ v = n - u
439
+
440
+ # Step 2: A = X[1..u], B = X[u+1..n]
441
+ a = x[:u]
442
+ b_side = x[u:]
443
+
444
+ # Step 3: b = ceil(ceil(v * log2(radix)) / 8) -- derived from v, not u.
445
+ # The bit length uses exact integer arithmetic; never math.log2.
446
+ b = ((radix**v - 1).bit_length() + 7) // 8
447
+
448
+ # Step 4: d = 4 * ceil(b/4) + 4
449
+ d = 4 * ((b + 3) // 4) + 4
450
+
451
+ t = len(tweak)
452
+ pad = (-(t + b + 1)) % 16
453
+
454
+ # Step 5: P is loop-invariant, so it is built once here rather than
455
+ # rebuilt on each of the ten rounds.
456
+ p_block = (
457
+ bytes([1, 2, 1])
458
+ + _encode_uint(radix, 3)
459
+ + bytes([10, u % 256])
460
+ + _encode_uint(n, 4)
461
+ + _encode_uint(t, 4)
462
+ )
463
+
464
+ rounds = range(10) if encrypt else range(9, -1, -1)
465
+
466
+ for i in rounds:
467
+ # Step 6.i: Q = T || [0]^pad || [i]^1 || [NUM_radix(B)]^b
468
+ # (decrypt builds Q from A instead of B)
469
+ if encrypt:
470
+ q_block = (
471
+ tweak + bytes([0]) * pad + bytes([i]) + _encode_uint(_num_radix(radix, b_side), b)
472
+ )
473
+ else:
474
+ q_block = tweak + bytes([0]) * pad + bytes([i]) + _encode_uint(_num_radix(radix, a), b)
475
+
476
+ # Step 6.ii: R = PRF(P || Q)
477
+ r_block = _prf(aes, p_block + q_block)
478
+
479
+ # Step 6.iii: S is the first d bytes of
480
+ # R || CIPH_K(R XOR [1]^16) || CIPH_K(R XOR [2]^16) || ...
481
+ # Each expansion block is a SINGLE forward-cipher block over R XOR the
482
+ # 16-byte encoding of j. It is not a PRF, and j is not concatenated
483
+ # onto R -- both mistakes produce a non-16-byte-aligned input and are
484
+ # invisible to the NIST samples, none of which reach d > 16.
485
+ s_block = r_block
486
+ j = 1
487
+ while len(s_block) < d:
488
+ xored = bytes(p ^ q for p, q in zip(r_block, _encode_uint(j, 16), strict=True))
489
+ s_block += aes.ecb_encryptor.update(xored)
490
+ j += 1
491
+ # Truncate to d BYTES, not d bits.
492
+ s_block = s_block[:d]
493
+
494
+ # Step 6.iv: y = NUM(S)
495
+ y = int.from_bytes(s_block, byteorder="big")
496
+
497
+ # Step 6.v: parity rule is identical for encrypt and decrypt
498
+ m = u if i % 2 == 0 else v
499
+
500
+ # Step 6.vi: c = (NUM_radix(A) + y) mod radix**m (decrypt subtracts
501
+ # y from NUM_radix(B) instead)
502
+ if encrypt:
503
+ c = (_num_radix(radix, a) + y) % (radix**m)
504
+ else:
505
+ c = (_num_radix(radix, b_side) - y) % (radix**m)
506
+
507
+ # Step 6.vii: C = STR^m_radix(c)
508
+ c_block = _str_radix(c, radix, m)
509
+
510
+ if _trace is not None:
511
+ _trace.append(
512
+ {
513
+ "i": i,
514
+ "u": u,
515
+ "v": v,
516
+ "b": b,
517
+ "d": d,
518
+ "P": list(p_block),
519
+ "Q": list(q_block),
520
+ "R": list(r_block),
521
+ "S": list(s_block),
522
+ "y": y,
523
+ "m": m,
524
+ "c": c,
525
+ "C": list(c_block),
526
+ "A_before": list(a),
527
+ "B_before": list(b_side),
528
+ }
529
+ )
530
+
531
+ # Steps 6.viii and 6.ix: A = B, B = C (decrypt assigns B = A, A = C)
532
+ if encrypt:
533
+ a = b_side
534
+ b_side = c_block
535
+ else:
536
+ b_side = a
537
+ a = c_block
538
+
539
+ # Step 7: return A || B
540
+ return a + b_side
541
+
542
+
543
+ def _encode_uint(value: int, length: int) -> bytes:
544
+ """Encode a non-negative integer as a big-endian ``length``-byte string.
545
+
546
+ Raises ``OverflowError`` if the value does not fit, which is deliberate:
547
+ silently truncating would corrupt Q and produce non-conformant ciphertext.
548
+ """
549
+ return value.to_bytes(length, byteorder="big")
fpr_ff1/py.typed ADDED
File without changes
@@ -0,0 +1,348 @@
1
+ Metadata-Version: 2.5
2
+ Name: fpr-ff1
3
+ Version: 0.1.0
4
+ Summary: Format-preserving encryption: NIST SP 800-38G FF1 for Python.
5
+ Project-URL: Homepage, https://github.com/joelee/fpr-ff1
6
+ Project-URL: Repository, https://github.com/joelee/fpr-ff1
7
+ Project-URL: Issues, https://github.com/joelee/fpr-ff1/issues
8
+ Project-URL: Changelog, https://github.com/joelee/fpr-ff1/blob/main/CHANGELOG.md
9
+ Project-URL: Security, https://github.com/joelee/fpr-ff1/blob/main/SECURITY.md
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: cryptography,ff1,format-preserving-encryption,fpe,nist,sp800-38g
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Programming Language :: Python :: 3.14
20
+ Classifier: Programming Language :: Python :: Implementation :: CPython
21
+ Classifier: Topic :: Security :: Cryptography
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: <3.15,>=3.12
24
+ Requires-Dist: cryptography>=44.0.0
25
+ Description-Content-Type: text/markdown
26
+
27
+ # fpr-ff1
28
+
29
+ A small, correct Python implementation of **FF1**, the format-preserving encryption mode from NIST SP 800-38G.
30
+
31
+ This package is intentionally just the algorithm: no accounts, no network, no key management, and no FF3/FF3-1 modes.
32
+
33
+ ## Install
34
+
35
+ ```bash
36
+ pip install fpr-ff1
37
+ ```
38
+
39
+ ## Quick start
40
+
41
+ ```python
42
+ from fpr_ff1 import FF1
43
+
44
+ ff1 = FF1(
45
+ key=b"\x00" * 16,
46
+ radix=10,
47
+ alphabet="0123456789",
48
+ tweak=b"",
49
+ )
50
+
51
+ encrypted = ff1.encrypt("123456")
52
+ decrypted = ff1.decrypt(encrypted)
53
+ assert decrypted == "123456"
54
+ ```
55
+
56
+ ## Features
57
+
58
+ - Pure Python with a single runtime dependency: `cryptography`.
59
+ - Conformance-tested against the NIST SP 800-38G sample vectors.
60
+ - No floating-point arithmetic in the FF1 core.
61
+ - Tightened domain limits from the SP 800-38G Rev. 1 second public draft:
62
+ - radix range `2 <= radix < 2**16`
63
+ - minimum domain `radix ** minlen >= 1_000_000`
64
+ - maximum length `2 ** 32 - 1` (SP 800-38G specifies `maxlen < 2 ** 32`)
65
+ - AES keys of 128, 192, or 256 bits only
66
+ - Strongly typed public API with typed exceptions rooted at `FF1Error`.
67
+
68
+ ## Why you can trust this implementation
69
+
70
+ Format-preserving encryption is unusually easy to get *almost* right. A subtly wrong FF1 still
71
+ round-trips perfectly — `decrypt(encrypt(x)) == x` — while producing ciphertext no conformant
72
+ implementation can read. By the time anyone notices, the data is written. So conformance here is
73
+ not a checkbox; it is the entire product, and it is evidenced rather than asserted.
74
+
75
+ **708 tests. 100% line and branch coverage, enforced — the build fails below it.**
76
+
77
+ ### Conformance is proven at the round level, not just the output level
78
+
79
+ All nine published NIST sample vectors pass in both directions. That alone is a weak statement:
80
+ nine input/output pairs can be satisfied by two bugs that cancel out.
81
+
82
+ So this package also asserts the **per-round intermediate values** the NIST sample document
83
+ publishes — `P`, `Q`, `R`, `S`, `y`, `m`, `c` and `C`, plus the derived `u`, `v`, `b` and `d` — for
84
+ **every round of every sample**, 90 rounds in total. Compensating bugs survive an output test. They
85
+ do not survive this one.
86
+
87
+ The vectors are transcribed from the NIST document and stored as data files. They are never
88
+ regenerated from this implementation, which would make them a record of whatever the code does
89
+ rather than of what the standard requires.
90
+
91
+ ### Radices without published vectors are proven against an independent implementation
92
+
93
+ NIST publishes vectors for radix 10 and 36 only. Every other radix has none, so agreement with an
94
+ independent implementation is the only correctness evidence available — expected values authored
95
+ from this code would test nothing and lock in any bug permanently.
96
+
97
+ `fpr-ff1` is therefore differential-tested against `ubiq_security_fpe` across radices **2, 10, 16,
98
+ 32, 36, 62, 256 and 65535**, including every length where the algorithm's internal block structure
99
+ changes. The oracle is itself validated against all nine NIST vectors before a single comparison is
100
+ trusted.
101
+
102
+ ### Bijectivity is verified exhaustively, not sampled
103
+
104
+ For two domains small enough to enumerate completely — radix 2 at length 20 (1,048,576 values) and
105
+ radix 10 at length 6 (1,000,000 values) — every point is encrypted and the image checked to be the
106
+ full domain, with no gaps and no collisions. That is the strongest correctness statement available
107
+ for a permutation, and it is run in CI rather than kept as a manual check.
108
+
109
+ ### The known failure modes are tested for by name
110
+
111
+ Published FF1 bugs cluster in a few places. Each has a dedicated test:
112
+
113
+ | Known failure mode | How it is prevented |
114
+ |---|---|
115
+ | Floating-point `ceil(v · log₂(radix))` — the Bouncy Castle bug class | Exact integer arithmetic; an AST scan fails the build if `math.log`, `ceil`, `/` or any float literal appears in the core |
116
+ | `b` derived from `u` instead of `v` | Asserted against the traced value, not a round-trip — which passes either way |
117
+ | Wrong `S` expansion when `d > 16` | Differential cases at each block-count transition; no NIST sample reaches this branch |
118
+ | Mirrored parity rule in decrypt | Encrypt and decrypt share one code path for it |
119
+ | Silent coercion of bad input | Every rejection raises a typed exception; a 50-case sweep asserts nothing escapes as a bare `AttributeError` or `KeyError` |
120
+
121
+ ### Migration is safe by construction
122
+
123
+ Output is byte-identical to `ubiq_security_fpe`, verified in **both** directions — old ciphertext
124
+ decrypts with this library, and new ciphertext decrypts with the old one. Existing encrypted data
125
+ stays readable, and a rollback strands nothing. See [Migrating](#migrating-from-ubiq_security_fpe).
126
+
127
+ ### What this is *not*
128
+
129
+ Passing the published sample vectors is **conformance evidence, not FIPS validation**. This package
130
+ is not FIPS 140 validated and makes no such claim. It also does not attempt key zeroization, and
131
+ offers no constant-time guarantee — see [`SECURITY.md`](SECURITY.md) for the full statement of
132
+ limitations.
133
+
134
+ ## Why FF1 only — and why FF3 is excluded
135
+
136
+ SP 800-38G originally specified two modes, FF1 and FF3. FF3 was revised to FF3-1 after an attack
137
+ on the original construction, but Beyne subsequently demonstrated a weakness in the **tweak
138
+ schedule** that affects FF3 and FF3-1 alike — the repair did not address the underlying problem.
139
+
140
+ The **February 2025 second public draft of SP 800-38G Rev. 1 removes FF3 entirely**, leaving FF1
141
+ as the only approved format-preserving mode.
142
+
143
+ `fpr-ff1` will therefore never implement FF3 or FF3-1. This is a deliberate feature, not an
144
+ omission: there is no configuration flag, no opt-in, and no plan to add one. If you need FF3 you
145
+ need a different library, and you should first satisfy yourself that you actually need a mode
146
+ NIST has withdrawn.
147
+
148
+ ## Domain limits are stricter than the 2016 text
149
+
150
+ This package implements SP 800-38G (2016, updated 2019) as the normative algorithm, but enforces
151
+ the **tightened constraints from the Rev. 1 second public draft**:
152
+
153
+ | Constraint | This package | SP 800-38G (2016) |
154
+ |---|---|---|
155
+ | Minimum domain | `radix ** minlen >= 1_000_000` | `radix ** minlen >= 100` |
156
+ | Maximum length | `2 ** 32 - 1` | `2 ** 32 - 1` |
157
+ | Key sizes | 128, 192, 256 bits | same |
158
+ | Radix | `2 <= radix < 2 ** 16` | same |
159
+ | Rounds | exactly 10 | same |
160
+
161
+ The minimum-domain rule is the one that will bite. A domain of only 100 values is trivially
162
+ enumerable, so this package **fails closed** and rejects it. Concretely, `min_length` is 6 for
163
+ radix 10 and 4 for radix 36 — inputs shorter than that raise `LengthError`, even though some
164
+ older libraries (including `ubiq_security_fpe`) accept them.
165
+
166
+ Rev. 1 is still a draft. If it is finalised with different limits, that will be a breaking change
167
+ and a major version.
168
+
169
+ ## Scope
170
+
171
+ - **In scope:** FF1 encryption and decryption, numeral and string interfaces, alphabet handling, parameter validation, tests, documentation.
172
+ - **Out of scope, permanently:** FF3/FF3-1, identifier generation, persistence, checksums, key generation/storage/derivation, application-specific defaults or alphabets.
173
+
174
+ ## Supported Python versions
175
+
176
+ **3.12, 3.13 and 3.14.** The upper bound in `requires-python` is deliberate: it matches the
177
+ versions actually exercised in CI on Linux, macOS and Windows. It is raised as part of a release
178
+ once a newer Python is in the matrix and green, rather than being left open and assumed to work.
179
+
180
+ ## Roadmap
181
+
182
+ | Version | Focus |
183
+ |---|---|
184
+ | **1.0** | **Pure Python.** Conformance, a stable API, and a single runtime dependency (`cryptography`). No compiled extension, no optional backends — one code path, and it is the one the vectors test. |
185
+ | **2.0** | **Optional accelerated backend.** An opt-in faster path for high-throughput callers, with the pure-Python implementation retained as the reference and the default. |
186
+
187
+ The 2.0 backend is explicitly *not* a 1.0 concern. An accelerated path is only worth having once
188
+ the reference implementation is settled and there is a conformance suite strong enough to prove
189
+ the two agree bit for bit — which is the point of the differential and interoperability tests.
190
+
191
+ Nothing in the roadmap changes the scope boundary above. FF3 and FF3-1 remain permanently out of
192
+ scope, and no release will add key management.
193
+
194
+ ## API
195
+
196
+ ### `FF1(key, radix, *, alphabet=None, tweak=b"", min_tweak_len=None, max_tweak_len=None)`
197
+
198
+ | Parameter | Description |
199
+ |---|---|
200
+ | `key` | 16, 24, or 32 bytes. |
201
+ | `radix` | Integer base of the numeral system. |
202
+ | `alphabet` | Optional string of exactly `radix` unique characters; enables `encrypt`/`decrypt`. |
203
+ | `tweak` | Default tweak used when not supplied per call. |
204
+ | `min_tweak_len` / `max_tweak_len` | Optional per-instance tweak length bounds. |
205
+
206
+ ### Numeral interface
207
+
208
+ The primitive interface works on integers in `[0, radix)`.
209
+
210
+ ```python
211
+ ciphertext = ff1.encrypt_numerals([1, 2, 3, 4, 5, 6])
212
+ plaintext = ff1.decrypt_numerals(ciphertext)
213
+ ```
214
+
215
+ **Accepted numeral types.** Anything losslessly integral — `int`, `IntEnum`, and integers from
216
+ other numeric libraries such as NumPy, which are normalised to Python `int` so fixed-width values
217
+ cannot overflow in the internal big-integer arithmetic.
218
+
219
+ `float`, `Decimal`, `Fraction` and `str` are **rejected** with `ValueRangeError`, even when they
220
+ compare equal to a valid numeral: `1.0 < 10` is `True`, so comparison alone is not a type check.
221
+
222
+ `bool` is **rejected deliberately**. `True` would otherwise encrypt silently as `1`, and a list of
223
+ booleans arriving here is a caller mistake, not an intent to encrypt ones and zeros.
224
+
225
+ The input must be a `Sequence` — something with a known length. A generator raises `TypeError`
226
+ (not `FF1Error`), because that is misuse of the API rather than bad data; wrap it in `list(...)`.
227
+
228
+ ### String interface
229
+
230
+ When `alphabet` is provided, the string interface maps characters to numerals and back.
231
+
232
+ ```python
233
+ ff1.encrypt("123456")
234
+ ff1.decrypt("654321")
235
+ ```
236
+
237
+ **Alphabet uniqueness is by Unicode code point.** FF1 operates on code points, so normalisation is
238
+ the caller's responsibility. Precomposed `é` (U+00E9) and decomposed `é` (U+0065 U+0301) are
239
+ visually identical but count as two distinct symbols, and an alphabet containing both is accepted.
240
+ If your alphabet comes from user input or an external source, normalise it first:
241
+
242
+ ```python
243
+ import unicodedata
244
+ alphabet = unicodedata.normalize("NFC", alphabet)
245
+ ```
246
+
247
+ ### Exceptions
248
+
249
+ Every rejection raises a typed exception derived from `FF1Error`. Nothing is silently truncated,
250
+ padded, coerced or clamped.
251
+
252
+ | Exception | Raised when |
253
+ |---|---|
254
+ | `KeyLengthError` | key is not 16, 24 or 32 bytes |
255
+ | `RadixError` | radix outside `2 <= radix < 2**16` |
256
+ | `LengthError` | input length outside `[min_length, max_length]` |
257
+ | `ValueRangeError` | a numeral outside `[0, radix)`, or a character absent from the alphabet |
258
+ | `TweakLengthError` | tweak outside the configured bounds |
259
+ | `AlphabetError` | alphabet length mismatched to radix, or containing duplicates |
260
+
261
+ `AlphabetError` signals malformed *configuration* (caught at construction); `ValueRangeError`
262
+ signals malformed *data* (caught per call). They are deliberately distinct so callers can handle
263
+ a programming error differently from a bad input record.
264
+
265
+ ## Migrating from `ubiq_security_fpe`
266
+
267
+ `fpr-ff1` exists to replace `ubiq_security_fpe`, which was deprecated in favour of a SaaS client
268
+ and is no longer maintained. **The two produce identical ciphertext for identical inputs**, so
269
+ existing encrypted data stays readable — no re-encryption, no migration window, no rollback risk.
270
+
271
+ That claim is enforced by `tests/test_interoperability.py`, which checks both directions (old
272
+ ciphertext decrypts with the new library and vice versa) across all three key sizes, tweaked and
273
+ untweaked. Migration safety is treated as a correctness obligation, not a promise.
274
+
275
+ **No compatibility shim ships, deliberately.** A `Context(...)` / `.Encrypt()` drop-in would mean
276
+ maintaining a permanent second API, in a naming style this project does not use, mirroring a
277
+ library that is itself deprecated. The migration below is three mechanical edits per call site,
278
+ and the part that would actually be hard — identical ciphertext — is already done.
279
+
280
+ ### API mapping
281
+
282
+ ```python
283
+ # before
284
+ from ubiq_security_fpe import ff1
285
+ ctx = ff1.Context(key, tweak, twk_min_len, twk_max_len, radix, alphabet)
286
+ ciphertext = ctx.Encrypt(plaintext, None)
287
+ plaintext = ctx.Decrypt(ciphertext, None)
288
+
289
+ # after
290
+ from fpr_ff1 import FF1
291
+ ctx = FF1(key, radix, alphabet=alphabet, tweak=tweak,
292
+ min_tweak_len=twk_min_len, max_tweak_len=twk_max_len)
293
+ ciphertext = ctx.encrypt(plaintext)
294
+ plaintext = ctx.decrypt(ciphertext)
295
+ ```
296
+
297
+ | `ubiq_security_fpe` | `fpr-ff1` |
298
+ |---|---|
299
+ | `ff1.Context(key, twk, twk_min_len, twk_max_len, radix, alpha)` | `FF1(key, radix, alphabet=..., tweak=..., min_tweak_len=..., max_tweak_len=...)` |
300
+ | `ctx.Encrypt(pt, twk)` | `ctx.encrypt(pt, twk)` |
301
+ | `ctx.Decrypt(ct, twk)` | `ctx.decrypt(ct, twk)` |
302
+ | — | `ctx.encrypt_numerals(...)` / `ctx.decrypt_numerals(...)` (no alphabet needed) |
303
+ | `RuntimeError` for every rejection | typed exceptions under `FF1Error` |
304
+
305
+ ### Behaviour changes to check before you switch
306
+
307
+ 1. **Shorter inputs are rejected.** `fpr-ff1` enforces `radix ** minlen >= 1_000_000`; the legacy
308
+ library used the same rule, but if you relied on any library using the 2016 `>= 100` bound,
309
+ inputs below `ctx.min_length` now raise `LengthError`. Check `ctx.min_length` for your radix.
310
+ 2. **Errors are typed.** Rejections raise `KeyLengthError`, `RadixError`, `LengthError`,
311
+ `ValueRangeError`, `TweakLengthError` or `AlphabetError` — all subclasses of `FF1Error` —
312
+ rather than bare `RuntimeError`. Catch `FF1Error` if you want the old catch-all behaviour.
313
+ 3. **No `M2Crypto` dependency.** `fpr-ff1` depends only on `cryptography`.
314
+ 4. **Alphabet is validated at construction.** A wrong-length alphabet or one with duplicate
315
+ characters raises `AlphabetError` immediately rather than misbehaving later.
316
+
317
+ ## FIPS disclaimer
318
+
319
+ Passing the published NIST sample vectors is evidence of conformance. It is **not** FIPS validation. This package makes no claims of FIPS 140 conformance.
320
+
321
+ ## Key material
322
+
323
+ `fpr-ff1` does not attempt to zeroize key material. Python `bytes` are immutable and the interpreter may copy them during garbage collection.
324
+
325
+ ## Development
326
+
327
+ Requires Python 3.12, `uv`, and `just`.
328
+
329
+ ```bash
330
+ just setup # create venv and install deps
331
+ just quality # format check, lint, typecheck, tests
332
+ just build # quality gate + uv build
333
+ just secrets # gitleaks scan (must be installed locally)
334
+ ```
335
+
336
+ ## Documentation
337
+
338
+ - `docs/architecture.md` — design and module overview
339
+ - `docs/developer-guide.md` — setup, commands, testing, and CI
340
+ - `docs/directory-structure.md` — repository layout
341
+ - `docs/configuration.md` — `FF1` constructor parameters and runtime constraints
342
+ - `docs/backlog.md` — active and completed work
343
+ - `CHANGELOG.md` — release history, including behaviour changes that affect accepted inputs
344
+ - `SECURITY.md` — disclosure process and known limitations
345
+
346
+ ## License
347
+
348
+ MIT
@@ -0,0 +1,8 @@
1
+ fpr_ff1/__init__.py,sha256=kbnpTIr-F_D9Cp2_5QVcIL9GxT2twJ_HyckTbnA8T40,422
2
+ fpr_ff1/_exceptions.py,sha256=kCUG4KYiwsjiGLkespCJsPSnXvD1_pRp7Ke7VxfQSLk,913
3
+ fpr_ff1/_ff1.py,sha256=DfYAciuBIkIzO0SklMWxM2pgFgCS670eMvggwiwdaVg,21545
4
+ fpr_ff1/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ fpr_ff1-0.1.0.dist-info/METADATA,sha256=ECH4p073Xk3KoJTT98iynfEjyy1NrceX_OWvfZYzGDg,16289
6
+ fpr_ff1-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
7
+ fpr_ff1-0.1.0.dist-info/licenses/LICENSE,sha256=FGQ9O0FIhqnsoDFIFBbwJXU_T9LEJnVeU6KRLn0Vs-I,1065
8
+ fpr_ff1-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Joel Lee
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.