baseh 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- baseh/__init__.py +62 -0
- baseh/basen.py +36 -0
- baseh/blocklist.py +41 -0
- baseh/checksum.py +36 -0
- baseh/codec.py +242 -0
- baseh/errors.py +36 -0
- baseh/feistel.py +119 -0
- baseh/profile.py +205 -0
- baseh/profiles.py +146 -0
- baseh/zero.py +52 -0
- baseh-1.1.0.dist-info/METADATA +125 -0
- baseh-1.1.0.dist-info/RECORD +14 -0
- baseh-1.1.0.dist-info/WHEEL +5 -0
- baseh-1.1.0.dist-info/top_level.txt +1 -0
baseh/__init__.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""baseH codec, Python implementation.
|
|
2
|
+
|
|
3
|
+
Public API mirrors spec section 12: the Baseh codec class, BasehError with a
|
|
4
|
+
stable .code attribute and the frozen profile tier helpers (baseh_medium_v1
|
|
5
|
+
is the default tier; the _p variants enable feistel-v1 permutation).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .blocklist import DEFAULT_BLOCKLIST
|
|
9
|
+
from .codec import CONFUSION_MAPS, Baseh, DecodeResult, generate_candidates
|
|
10
|
+
from .errors import (
|
|
11
|
+
AMBIGUOUS_INPUT,
|
|
12
|
+
BLOCKED_CODE,
|
|
13
|
+
INVALID_CHARACTER,
|
|
14
|
+
INVALID_CHECKSUM,
|
|
15
|
+
INVALID_LENGTH,
|
|
16
|
+
INVALID_PROFILE,
|
|
17
|
+
OUT_OF_RANGE,
|
|
18
|
+
PERMUTATION_FAILURE,
|
|
19
|
+
TOO_MANY_CANDIDATES,
|
|
20
|
+
BasehError,
|
|
21
|
+
)
|
|
22
|
+
from .profiles import (
|
|
23
|
+
baseh_heavy_p_v1,
|
|
24
|
+
baseh_heavy_v1,
|
|
25
|
+
baseh_light_p_v1,
|
|
26
|
+
baseh_light_v1,
|
|
27
|
+
baseh_medium_p_v1,
|
|
28
|
+
baseh_medium_v1,
|
|
29
|
+
baseh_minimum_p_v1,
|
|
30
|
+
baseh_minimum_v1,
|
|
31
|
+
)
|
|
32
|
+
from .zero import from_code, to_code
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"Baseh",
|
|
36
|
+
"BasehError",
|
|
37
|
+
"DecodeResult",
|
|
38
|
+
"CONFUSION_MAPS",
|
|
39
|
+
"DEFAULT_BLOCKLIST",
|
|
40
|
+
"generate_candidates",
|
|
41
|
+
"baseh_minimum_v1",
|
|
42
|
+
"baseh_minimum_p_v1",
|
|
43
|
+
"baseh_light_v1",
|
|
44
|
+
"baseh_light_p_v1",
|
|
45
|
+
"baseh_medium_v1",
|
|
46
|
+
"baseh_medium_p_v1",
|
|
47
|
+
"baseh_heavy_v1",
|
|
48
|
+
"baseh_heavy_p_v1",
|
|
49
|
+
"INVALID_PROFILE",
|
|
50
|
+
"OUT_OF_RANGE",
|
|
51
|
+
"PERMUTATION_FAILURE",
|
|
52
|
+
"INVALID_LENGTH",
|
|
53
|
+
"INVALID_CHARACTER",
|
|
54
|
+
"INVALID_CHECKSUM",
|
|
55
|
+
"AMBIGUOUS_INPUT",
|
|
56
|
+
"TOO_MANY_CANDIDATES",
|
|
57
|
+
"BLOCKED_CODE",
|
|
58
|
+
"to_code",
|
|
59
|
+
"from_code",
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
__version__ = "1.0.0"
|
baseh/basen.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Fixed-length base-N encode and decode, spec sections 5.1 through 5.3."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .errors import INVALID_CHARACTER, OUT_OF_RANGE, BasehError
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def alphabet_index(alphabet: str) -> dict:
|
|
9
|
+
return {ch: i for i, ch in enumerate(alphabet)}
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def encode_base_n(value: int, alphabet: str, length: int) -> str:
|
|
13
|
+
"""Fixed-length base-N encode, most significant digit first."""
|
|
14
|
+
base = len(alphabet)
|
|
15
|
+
capacity = base ** length
|
|
16
|
+
if value < 0 or value >= capacity:
|
|
17
|
+
raise BasehError(OUT_OF_RANGE, "value is outside the fixed-length capacity")
|
|
18
|
+
out = [""] * length
|
|
19
|
+
v = value
|
|
20
|
+
for pos in range(length - 1, -1, -1):
|
|
21
|
+
out[pos] = alphabet[v % base]
|
|
22
|
+
v //= base
|
|
23
|
+
return "".join(out)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def decode_base_n(text: str, alphabet: str, index: dict | None = None) -> int:
|
|
27
|
+
base = len(alphabet)
|
|
28
|
+
if index is None:
|
|
29
|
+
index = alphabet_index(alphabet)
|
|
30
|
+
value = 0
|
|
31
|
+
for ch in text:
|
|
32
|
+
digit = index.get(ch)
|
|
33
|
+
if digit is None:
|
|
34
|
+
raise BasehError(INVALID_CHARACTER, f"Symbol {ch!r} is not in the alphabet")
|
|
35
|
+
value = value * base + digit
|
|
36
|
+
return value
|
baseh/blocklist.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Profanity safety primitives, spec section 18."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from .errors import INVALID_PROFILE, BasehError
|
|
8
|
+
|
|
9
|
+
# Spec 18.2 default list. Deliberately small; applications extend it.
|
|
10
|
+
DEFAULT_BLOCKLIST = (
|
|
11
|
+
"CRAP", "TWAT", "SHAG", "DAMN", "FCK", "FUC",
|
|
12
|
+
"SHT", "CNT", "TWT", "DCK", "AZZ", "BCH",
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_WORD = re.compile(r"^[A-Za-z]{2,32}$")
|
|
16
|
+
_VOWELS = frozenset("AEIOU")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def effective_blocklist(profanity: dict) -> list:
|
|
20
|
+
"""Spec 18.2: replacement semantics, then augmentation, uppercased and
|
|
21
|
+
deduplicated. Raises BasehError INVALID_PROFILE on a malformed entry."""
|
|
22
|
+
base = list(profanity["words"]) if "words" in profanity else list(DEFAULT_BLOCKLIST)
|
|
23
|
+
words = base + list(profanity.get("extraWords") or [])
|
|
24
|
+
out: list = []
|
|
25
|
+
for word in words:
|
|
26
|
+
if not isinstance(word, str) or not _WORD.match(word):
|
|
27
|
+
raise BasehError(
|
|
28
|
+
INVALID_PROFILE,
|
|
29
|
+
"Invalid baseH profile: blocklist entries must be 2 through 32 ASCII letters",
|
|
30
|
+
False,
|
|
31
|
+
)
|
|
32
|
+
upper = word.upper()
|
|
33
|
+
if upper not in out:
|
|
34
|
+
out.append(upper)
|
|
35
|
+
return out
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def strip_vowels(alphabet_norm: str) -> str:
|
|
39
|
+
"""Spec 18.1: vowels removed for no-vowels mode, applied after case
|
|
40
|
+
normalization."""
|
|
41
|
+
return "".join(ch for ch in alphabet_norm if ch not in _VOWELS)
|
baseh/checksum.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Version 1 rolling polynomial checksum, spec section 6.2."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .basen import alphabet_index, encode_base_n
|
|
6
|
+
from .errors import INVALID_CHARACTER, BasehError
|
|
7
|
+
from .profile import PreparedProfile
|
|
8
|
+
|
|
9
|
+
_INITIAL_STATE = 17
|
|
10
|
+
_MULTIPLIER = 37
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def checksum_value(profile: PreparedProfile, body: str, body_index: dict) -> int:
|
|
14
|
+
"""Return the checksum value in [0, modulus)."""
|
|
15
|
+
modulus = profile.checksum_modulus
|
|
16
|
+
state = _INITIAL_STATE
|
|
17
|
+
for byte in profile.profile_id.encode("ascii"):
|
|
18
|
+
state = (state * _MULTIPLIER + byte + 1) % modulus
|
|
19
|
+
state = (state * _MULTIPLIER) % modulus
|
|
20
|
+
for pos, ch in enumerate(body):
|
|
21
|
+
value = body_index.get(ch)
|
|
22
|
+
if value is None:
|
|
23
|
+
raise BasehError(
|
|
24
|
+
INVALID_CHARACTER, f"Symbol {ch!r} is not in the body alphabet"
|
|
25
|
+
)
|
|
26
|
+
state = (state * _MULTIPLIER + value + pos + 1) % modulus
|
|
27
|
+
return state
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def calculate_checksum(profile: PreparedProfile, body: str) -> str:
|
|
31
|
+
"""Compute the expected checksum string for a normalized body."""
|
|
32
|
+
if profile.checksum_length == 0:
|
|
33
|
+
return ""
|
|
34
|
+
index = alphabet_index(profile.body_alphabet_norm)
|
|
35
|
+
value = checksum_value(profile, body, index)
|
|
36
|
+
return encode_base_n(value, profile.checksum_alphabet_norm, profile.checksum_length)
|
baseh/codec.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Full encode and decode flows, spec sections 8, 9, 10, 11 and 12."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from .basen import alphabet_index, decode_base_n, encode_base_n
|
|
8
|
+
from .checksum import calculate_checksum
|
|
9
|
+
from .errors import (
|
|
10
|
+
AMBIGUOUS_INPUT,
|
|
11
|
+
BLOCKED_CODE,
|
|
12
|
+
INVALID_CHARACTER,
|
|
13
|
+
INVALID_CHECKSUM,
|
|
14
|
+
INVALID_LENGTH,
|
|
15
|
+
OUT_OF_RANGE,
|
|
16
|
+
TOO_MANY_CANDIDATES,
|
|
17
|
+
BasehError,
|
|
18
|
+
)
|
|
19
|
+
from .feistel import FeistelKey, inverse_permute, permute
|
|
20
|
+
from .profile import PreparedProfile, prepare_profile
|
|
21
|
+
|
|
22
|
+
# Built-in spoken-confusion candidate maps, spec 3.3. Body symbols only.
|
|
23
|
+
CONFUSION_MAPS = {
|
|
24
|
+
"light": {"B": ["D"], "D": ["B"], "P": ["T"], "T": ["P"]},
|
|
25
|
+
"medium": {
|
|
26
|
+
"B": ["D"], "D": ["B"], "P": ["T"], "T": ["P"],
|
|
27
|
+
"M": ["N"], "N": ["M"], "V": ["W"], "W": ["V"],
|
|
28
|
+
},
|
|
29
|
+
"heavy": {
|
|
30
|
+
"B": ["D"], "D": ["B"], "P": ["T"], "T": ["P"],
|
|
31
|
+
"M": ["N"], "N": ["M"], "V": ["W"], "W": ["V"],
|
|
32
|
+
"F": ["S"], "S": ["F"], "C": ["G"], "G": ["C"],
|
|
33
|
+
},
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
_ASCII_WS = "\t\n\v\f\r "
|
|
37
|
+
_MAX_CANDIDATES = 64
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class DecodeResult:
|
|
42
|
+
id: int
|
|
43
|
+
canonical_code: str
|
|
44
|
+
corrected: bool
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def normalize(input: str, profile: PreparedProfile, accept_spaces: bool = False) -> str:
|
|
48
|
+
"""Spec 3.1 normalization, steps 1-7. Returns the raw unformatted string."""
|
|
49
|
+
if not isinstance(input, str):
|
|
50
|
+
raise BasehError(INVALID_CHARACTER, "input must be a string")
|
|
51
|
+
s = input.strip(_ASCII_WS)
|
|
52
|
+
if profile.separator:
|
|
53
|
+
s = s.replace(profile.separator, "")
|
|
54
|
+
if accept_spaces:
|
|
55
|
+
s = s.replace(" ", "")
|
|
56
|
+
if not profile.case_sensitive:
|
|
57
|
+
s = s.upper()
|
|
58
|
+
if profile.aliases_norm:
|
|
59
|
+
s = "".join(profile.aliases_norm.get(ch, ch) for ch in s)
|
|
60
|
+
allowed = set(profile.body_alphabet_norm) | set(profile.checksum_alphabet_norm)
|
|
61
|
+
for ch in s:
|
|
62
|
+
if ch not in allowed:
|
|
63
|
+
raise BasehError(INVALID_CHARACTER, f"Symbol {ch!r} is not accepted")
|
|
64
|
+
expected = profile.body_length + profile.checksum_length
|
|
65
|
+
# Spec 3.4: a code that lost leading zero body symbols is re-padded with
|
|
66
|
+
# the body zero symbol. The checksum symbols always remain, so the split
|
|
67
|
+
# point is unambiguous. A fully stripped no-checksum code would be empty
|
|
68
|
+
# and stays a length error.
|
|
69
|
+
if len(s) < expected and len(s) >= max(profile.checksum_length, 1):
|
|
70
|
+
zero = profile.body_alphabet_norm[0]
|
|
71
|
+
s = zero * (expected - len(s)) + s
|
|
72
|
+
if len(s) != expected:
|
|
73
|
+
raise BasehError(INVALID_LENGTH, f"Expected {expected} symbols, got {len(s)}")
|
|
74
|
+
return s
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def format_raw(raw: str, profile: PreparedProfile) -> str:
|
|
78
|
+
if not profile.separator:
|
|
79
|
+
return raw
|
|
80
|
+
parts = []
|
|
81
|
+
offset = 0
|
|
82
|
+
for size in profile.grouping:
|
|
83
|
+
parts.append(raw[offset : offset + size])
|
|
84
|
+
offset += size
|
|
85
|
+
return profile.separator.join(parts)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def generate_candidates(body: str, confusion_map: dict, max_edits: int = 1) -> list:
|
|
89
|
+
"""Spec 10. Substitution-only candidate generation, capped and deduplicated."""
|
|
90
|
+
if max_edits == 0:
|
|
91
|
+
return []
|
|
92
|
+
results: set = set()
|
|
93
|
+
for pos, source in enumerate(body):
|
|
94
|
+
for replacement in confusion_map.get(source, ()):
|
|
95
|
+
candidate = body[:pos] + replacement + body[pos + 1 :]
|
|
96
|
+
results.add(candidate)
|
|
97
|
+
if len(results) > _MAX_CANDIDATES:
|
|
98
|
+
raise BasehError(
|
|
99
|
+
TOO_MANY_CANDIDATES,
|
|
100
|
+
"Candidate generation exceeded 64 entries",
|
|
101
|
+
False,
|
|
102
|
+
)
|
|
103
|
+
return list(results)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class Baseh:
|
|
107
|
+
"""Codec bound to one validated profile. The profile is validated once at
|
|
108
|
+
construction per spec 2.2, never per encode or decode."""
|
|
109
|
+
|
|
110
|
+
def __init__(self, profile: dict) -> None:
|
|
111
|
+
if isinstance(profile, PreparedProfile):
|
|
112
|
+
self._profile = profile
|
|
113
|
+
else:
|
|
114
|
+
self._profile = prepare_profile(profile)
|
|
115
|
+
self._body_index = alphabet_index(self._profile.body_alphabet_norm)
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
def profile(self) -> PreparedProfile:
|
|
119
|
+
return self._profile
|
|
120
|
+
|
|
121
|
+
def capacity(self) -> int:
|
|
122
|
+
return self._profile.capacity
|
|
123
|
+
|
|
124
|
+
def _feistel_key(self) -> FeistelKey:
|
|
125
|
+
perm = self._profile.permutation
|
|
126
|
+
return FeistelKey(
|
|
127
|
+
profile_id=self._profile.profile_id,
|
|
128
|
+
key_bytes=perm.key_bytes,
|
|
129
|
+
rounds=perm.rounds,
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
def encode(self, id: int) -> str:
|
|
133
|
+
"""Spec 8, including the section 18.2 blocked-substring scan."""
|
|
134
|
+
if isinstance(id, bool) or not isinstance(id, int):
|
|
135
|
+
raise BasehError(OUT_OF_RANGE, "id must be an integer")
|
|
136
|
+
value = id
|
|
137
|
+
if value < 0 or value >= self._profile.capacity:
|
|
138
|
+
raise BasehError(OUT_OF_RANGE, f"ID {value} is outside the profile capacity")
|
|
139
|
+
if self._profile.permutation.enabled:
|
|
140
|
+
value = permute(value, self._profile.capacity, self._feistel_key())
|
|
141
|
+
body = encode_base_n(
|
|
142
|
+
value, self._profile.body_alphabet_norm, self._profile.body_length
|
|
143
|
+
)
|
|
144
|
+
checksum = calculate_checksum(self._profile, body)
|
|
145
|
+
raw = body + checksum
|
|
146
|
+
# Spec 18.2: case-insensitive substring scan over the raw code.
|
|
147
|
+
if self._profile.blocklist:
|
|
148
|
+
upper = raw.upper()
|
|
149
|
+
if any(word in upper for word in self._profile.blocklist):
|
|
150
|
+
raise BasehError(
|
|
151
|
+
BLOCKED_CODE,
|
|
152
|
+
"The generated reference contains a blocked substring",
|
|
153
|
+
False,
|
|
154
|
+
)
|
|
155
|
+
return format_raw(raw, self._profile)
|
|
156
|
+
|
|
157
|
+
def decode(
|
|
158
|
+
self,
|
|
159
|
+
input: str,
|
|
160
|
+
*,
|
|
161
|
+
accept_spaces: bool = False,
|
|
162
|
+
try_correction: bool = False,
|
|
163
|
+
confusion_profile: str = "none",
|
|
164
|
+
max_corrections: int = 1,
|
|
165
|
+
) -> DecodeResult:
|
|
166
|
+
"""Spec 9."""
|
|
167
|
+
raw = normalize(input, self._profile, accept_spaces)
|
|
168
|
+
body = raw[: self._profile.body_length]
|
|
169
|
+
supplied_checksum = raw[self._profile.body_length :]
|
|
170
|
+
|
|
171
|
+
# Spec 3.1 validates union membership before the split. There is no
|
|
172
|
+
# per-region membership check: a checksum-region symbol outside the
|
|
173
|
+
# checksum alphabet fails as INVALID_CHECKSUM and a body symbol
|
|
174
|
+
# outside the body alphabet fails in the checksum or base-N work as
|
|
175
|
+
# INVALID_CHARACTER.
|
|
176
|
+
|
|
177
|
+
if calculate_checksum(self._profile, body) != supplied_checksum:
|
|
178
|
+
if not try_correction or max_corrections == 0:
|
|
179
|
+
raise BasehError(
|
|
180
|
+
INVALID_CHECKSUM, "The reference code did not pass validation"
|
|
181
|
+
)
|
|
182
|
+
if confusion_profile == "none":
|
|
183
|
+
raw_map: dict = {}
|
|
184
|
+
elif confusion_profile in CONFUSION_MAPS:
|
|
185
|
+
raw_map = CONFUSION_MAPS[confusion_profile]
|
|
186
|
+
else:
|
|
187
|
+
raise ValueError(
|
|
188
|
+
f"unknown confusion profile: {confusion_profile!r}"
|
|
189
|
+
)
|
|
190
|
+
# Spec 10: replacements that are not body alphabet symbols are
|
|
191
|
+
# dropped before candidate generation. A suggested symbol the
|
|
192
|
+
# alphabet cannot contain (say a spoken drop on a stripped-alphabet
|
|
193
|
+
# profile) could never validate; generating it anyway would raise
|
|
194
|
+
# INVALID_CHARACTER from the checksum step instead of reporting an
|
|
195
|
+
# honest INVALID_CHECKSUM.
|
|
196
|
+
body_set = set(self._profile.body_alphabet_norm)
|
|
197
|
+
confusion_map = {}
|
|
198
|
+
for source, replacements in raw_map.items():
|
|
199
|
+
kept = [r for r in replacements if r in body_set]
|
|
200
|
+
if kept:
|
|
201
|
+
confusion_map[source] = kept
|
|
202
|
+
valid: set = set()
|
|
203
|
+
for candidate in generate_candidates(body, confusion_map, max_corrections):
|
|
204
|
+
if calculate_checksum(self._profile, candidate) == supplied_checksum:
|
|
205
|
+
valid.add(candidate)
|
|
206
|
+
if not valid:
|
|
207
|
+
raise BasehError(
|
|
208
|
+
INVALID_CHECKSUM, "The reference code did not pass validation"
|
|
209
|
+
)
|
|
210
|
+
if len(valid) > 1:
|
|
211
|
+
raise BasehError(
|
|
212
|
+
AMBIGUOUS_INPUT,
|
|
213
|
+
"The reference code matches more than one record",
|
|
214
|
+
False,
|
|
215
|
+
)
|
|
216
|
+
body = next(iter(valid))
|
|
217
|
+
|
|
218
|
+
value = decode_base_n(
|
|
219
|
+
body, self._profile.body_alphabet_norm, self._body_index
|
|
220
|
+
)
|
|
221
|
+
if self._profile.permutation.enabled:
|
|
222
|
+
value = inverse_permute(
|
|
223
|
+
value, self._profile.capacity, self._feistel_key()
|
|
224
|
+
)
|
|
225
|
+
canonical_code = self.encode(value)
|
|
226
|
+
if self._profile.separator:
|
|
227
|
+
canonical_raw = canonical_code.replace(self._profile.separator, "")
|
|
228
|
+
else:
|
|
229
|
+
canonical_raw = canonical_code
|
|
230
|
+
return DecodeResult(
|
|
231
|
+
id=value,
|
|
232
|
+
canonical_code=canonical_code,
|
|
233
|
+
corrected=(raw != canonical_raw),
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
def validate(self, input: str, **options) -> dict:
|
|
237
|
+
"""Spec 12.4. Never raises on user input."""
|
|
238
|
+
try:
|
|
239
|
+
result = self.decode(input, **options)
|
|
240
|
+
return {"valid": True, "canonical_code": result.canonical_code}
|
|
241
|
+
except BasehError as err:
|
|
242
|
+
return {"valid": False, "reason": err.code}
|
baseh/errors.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Error type and codes defined by the baseH codec specification."""
|
|
2
|
+
|
|
3
|
+
INVALID_PROFILE = "INVALID_PROFILE"
|
|
4
|
+
OUT_OF_RANGE = "OUT_OF_RANGE"
|
|
5
|
+
PERMUTATION_FAILURE = "PERMUTATION_FAILURE"
|
|
6
|
+
INVALID_LENGTH = "INVALID_LENGTH"
|
|
7
|
+
INVALID_CHARACTER = "INVALID_CHARACTER"
|
|
8
|
+
INVALID_CHECKSUM = "INVALID_CHECKSUM"
|
|
9
|
+
AMBIGUOUS_INPUT = "AMBIGUOUS_INPUT"
|
|
10
|
+
TOO_MANY_CANDIDATES = "TOO_MANY_CANDIDATES"
|
|
11
|
+
BLOCKED_CODE = "BLOCKED_CODE"
|
|
12
|
+
|
|
13
|
+
ERROR_CODES = frozenset(
|
|
14
|
+
{
|
|
15
|
+
INVALID_PROFILE,
|
|
16
|
+
OUT_OF_RANGE,
|
|
17
|
+
PERMUTATION_FAILURE,
|
|
18
|
+
INVALID_LENGTH,
|
|
19
|
+
INVALID_CHARACTER,
|
|
20
|
+
INVALID_CHECKSUM,
|
|
21
|
+
AMBIGUOUS_INPUT,
|
|
22
|
+
TOO_MANY_CANDIDATES,
|
|
23
|
+
BLOCKED_CODE,
|
|
24
|
+
}
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class BasehError(Exception):
|
|
29
|
+
"""Codec error carrying a stable, machine-readable code."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, code: str, message: str, safe_for_customer: bool = True) -> None:
|
|
32
|
+
super().__init__(message)
|
|
33
|
+
if code not in ERROR_CODES:
|
|
34
|
+
raise ValueError(f"unknown baseH error code: {code!r}")
|
|
35
|
+
self.code = code
|
|
36
|
+
self.safe_for_customer = safe_for_customer
|
baseh/feistel.py
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Feistel-v1 reversible permutation, spec section 7.3.
|
|
2
|
+
|
|
3
|
+
Message bytes, half widths, low-N-bits truncation and cycle walking are
|
|
4
|
+
implemented exactly as written in section 7.3. The vectors in
|
|
5
|
+
vectors/feistel-vectors.json are the ground truth.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import hmac
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
from .errors import PERMUTATION_FAILURE, BasehError
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class FeistelKey:
|
|
19
|
+
profile_id: str
|
|
20
|
+
key_bytes: bytes
|
|
21
|
+
rounds: int
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_TAG = b"BASEH-FEISTEL-V1"
|
|
25
|
+
_MAX_WALKS = 1000
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _bit_length(capacity: int) -> int:
|
|
29
|
+
return (capacity - 1).bit_length()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _low_bits(digest: bytes, n: int) -> int:
|
|
33
|
+
"""Interpret the first ceil(n/8) digest bytes as big-endian, mask to n bits."""
|
|
34
|
+
byte_count = (n + 7) // 8
|
|
35
|
+
value = int.from_bytes(digest[:byte_count], "big")
|
|
36
|
+
return value & ((1 << n) - 1)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _round_message(profile_id: str, round_number: int, right: int, wr: int) -> bytes:
|
|
40
|
+
pid_bytes = profile_id.encode("ascii")
|
|
41
|
+
right_bytes = right.to_bytes((wr + 7) // 8, "big")
|
|
42
|
+
return (
|
|
43
|
+
_TAG
|
|
44
|
+
+ b"\x00"
|
|
45
|
+
+ pid_bytes
|
|
46
|
+
+ b"\x00"
|
|
47
|
+
+ bytes([round_number])
|
|
48
|
+
+ right_bytes
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _run_rounds(left: int, right: int, key, w0: int, w1: int) -> tuple:
|
|
53
|
+
for i in range(key.rounds):
|
|
54
|
+
even = i % 2 == 0
|
|
55
|
+
wr = w1 if even else w0
|
|
56
|
+
wl = w0 if even else w1
|
|
57
|
+
digest = hmac.new(
|
|
58
|
+
key.key_bytes, _round_message(key.profile_id, i, right, wr), hashlib.sha256
|
|
59
|
+
).digest()
|
|
60
|
+
f = _low_bits(digest, wl)
|
|
61
|
+
left, right = right, left ^ f
|
|
62
|
+
return left, right
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _run_inverse(left: int, right: int, key, w0: int, w1: int) -> tuple:
|
|
66
|
+
for i in range(key.rounds - 1, -1, -1):
|
|
67
|
+
even = i % 2 == 0
|
|
68
|
+
wr = w1 if even else w0
|
|
69
|
+
wl = w0 if even else w1
|
|
70
|
+
digest = hmac.new(
|
|
71
|
+
key.key_bytes, _round_message(key.profile_id, i, left, wr), hashlib.sha256
|
|
72
|
+
).digest()
|
|
73
|
+
f = _low_bits(digest, wl)
|
|
74
|
+
left, right = right ^ f, left
|
|
75
|
+
return left, right
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _split(value: int, w1: int) -> tuple:
|
|
79
|
+
return value >> w1, value & ((1 << w1) - 1)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _combine(left: int, right: int, w1: int) -> int:
|
|
83
|
+
return (left << w1) | right
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def permute(value: int, capacity: int, key) -> int:
|
|
87
|
+
"""Forward permutation with cycle walking, spec 7.3 steps 1-7."""
|
|
88
|
+
bits = _bit_length(capacity)
|
|
89
|
+
w1 = bits // 2
|
|
90
|
+
w0 = bits - w1
|
|
91
|
+
v = value
|
|
92
|
+
for _ in range(_MAX_WALKS):
|
|
93
|
+
left, right = _split(v, w1)
|
|
94
|
+
left, right = _run_rounds(left, right, key, w0, w1)
|
|
95
|
+
out = _combine(left, right, w1)
|
|
96
|
+
if out < capacity:
|
|
97
|
+
return out
|
|
98
|
+
v = out
|
|
99
|
+
raise BasehError(
|
|
100
|
+
PERMUTATION_FAILURE, "Feistel cycle walking exceeded 1000 iterations", False
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def inverse_permute(value: int, capacity: int, key) -> int:
|
|
105
|
+
"""Inverse permutation with identical cycle walking, spec 7.3."""
|
|
106
|
+
bits = _bit_length(capacity)
|
|
107
|
+
w1 = bits // 2
|
|
108
|
+
w0 = bits - w1
|
|
109
|
+
v = value
|
|
110
|
+
for _ in range(_MAX_WALKS):
|
|
111
|
+
left, right = _split(v, w1)
|
|
112
|
+
left, right = _run_inverse(left, right, key, w0, w1)
|
|
113
|
+
out = _combine(left, right, w1)
|
|
114
|
+
if out < capacity:
|
|
115
|
+
return out
|
|
116
|
+
v = out
|
|
117
|
+
raise BasehError(
|
|
118
|
+
PERMUTATION_FAILURE, "Feistel cycle walking exceeded 1000 iterations", False
|
|
119
|
+
)
|
baseh/profile.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Profile validation per spec section 2.2 and derived precomputed values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from .blocklist import effective_blocklist, strip_vowels
|
|
8
|
+
from .errors import INVALID_PROFILE, BasehError
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _fail(reason: str) -> None:
|
|
12
|
+
raise BasehError(INVALID_PROFILE, f"Invalid baseH profile: {reason}", False)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _is_ascii_char(ch: str) -> bool:
|
|
16
|
+
return len(ch) == 1 and 0x20 <= ord(ch) <= 0x7E
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _is_ascii(text: str) -> bool:
|
|
20
|
+
return all(0x20 <= ord(ch) <= 0x7E for ch in text)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _is_int(value) -> bool:
|
|
24
|
+
return isinstance(value, int) and not isinstance(value, bool)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class PreparedPermutation:
|
|
29
|
+
enabled: bool
|
|
30
|
+
key_id: str = ""
|
|
31
|
+
key_bytes: bytes = b""
|
|
32
|
+
rounds: int = 8
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class PreparedProfile:
|
|
37
|
+
"""Validated profile with derived case-normalized values."""
|
|
38
|
+
|
|
39
|
+
profile_id: str
|
|
40
|
+
body_alphabet_norm: str
|
|
41
|
+
checksum_alphabet_norm: str
|
|
42
|
+
body_length: int
|
|
43
|
+
checksum_length: int
|
|
44
|
+
case_sensitive: bool
|
|
45
|
+
separator: str
|
|
46
|
+
grouping: tuple
|
|
47
|
+
aliases_norm: dict
|
|
48
|
+
permutation: PreparedPermutation
|
|
49
|
+
checksum_modulus: int
|
|
50
|
+
capacity: int
|
|
51
|
+
blocklist: tuple
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _norm(case_sensitive: bool, ch: str) -> str:
|
|
55
|
+
return ch if case_sensitive else ch.upper()
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def prepare_profile(profile) -> PreparedProfile:
|
|
59
|
+
"""Validate a profile dict per spec 2.2. Raises BasehError INVALID_PROFILE."""
|
|
60
|
+
if not isinstance(profile, dict):
|
|
61
|
+
_fail("profile is required")
|
|
62
|
+
|
|
63
|
+
profile_id = profile.get("profileId")
|
|
64
|
+
if not isinstance(profile_id, str) or len(profile_id) == 0:
|
|
65
|
+
_fail("profileId must be non-empty")
|
|
66
|
+
if not _is_ascii(profile_id):
|
|
67
|
+
_fail("profileId must be ASCII")
|
|
68
|
+
|
|
69
|
+
case_sensitive = profile.get("caseSensitive") is True
|
|
70
|
+
|
|
71
|
+
body_alphabet = profile.get("bodyAlphabet")
|
|
72
|
+
if not isinstance(body_alphabet, str) or len(body_alphabet) < 2:
|
|
73
|
+
_fail("bodyAlphabet needs at least two symbols")
|
|
74
|
+
for ch in body_alphabet:
|
|
75
|
+
if not _is_ascii_char(ch):
|
|
76
|
+
_fail(f"body alphabet symbol is not single ASCII: {ch!r}")
|
|
77
|
+
body_norm = "".join(_norm(case_sensitive, ch) for ch in body_alphabet)
|
|
78
|
+
if len(set(body_norm)) != len(body_norm):
|
|
79
|
+
_fail("body alphabet symbols must be unique after case normalization")
|
|
80
|
+
|
|
81
|
+
body_length = profile.get("bodyLength")
|
|
82
|
+
if not _is_int(body_length) or body_length < 1 or body_length > 32:
|
|
83
|
+
_fail("bodyLength must be an integer from 1 through 32")
|
|
84
|
+
|
|
85
|
+
checksum_length = profile.get("checksumLength")
|
|
86
|
+
if not _is_int(checksum_length) or checksum_length < 0 or checksum_length > 8:
|
|
87
|
+
_fail("checksumLength must be an integer from 0 through 8")
|
|
88
|
+
|
|
89
|
+
checksum_alphabet = profile.get("checksumAlphabet") or ""
|
|
90
|
+
if not isinstance(checksum_alphabet, str):
|
|
91
|
+
_fail("checksumAlphabet must be a string")
|
|
92
|
+
if checksum_length > 0:
|
|
93
|
+
if len(checksum_alphabet) < 2:
|
|
94
|
+
_fail("checksumAlphabet needs at least two symbols when checksumLength is positive")
|
|
95
|
+
for ch in checksum_alphabet:
|
|
96
|
+
if not _is_ascii_char(ch):
|
|
97
|
+
_fail(f"checksum alphabet symbol is not single ASCII: {ch!r}")
|
|
98
|
+
checksum_norm = "".join(_norm(case_sensitive, ch) for ch in checksum_alphabet)
|
|
99
|
+
if len(set(checksum_norm)) != len(checksum_norm):
|
|
100
|
+
_fail("checksum alphabet symbols must be unique after case normalization")
|
|
101
|
+
|
|
102
|
+
# Spec 18. no-vowels strips vowels before every downstream rule; blocklist
|
|
103
|
+
# only arms the encode-time scan.
|
|
104
|
+
profanity = profile.get("profanity") or {"mode": "none"}
|
|
105
|
+
if not isinstance(profanity, dict) or profanity.get("mode") not in (
|
|
106
|
+
"none",
|
|
107
|
+
"no-vowels",
|
|
108
|
+
"blocklist",
|
|
109
|
+
):
|
|
110
|
+
_fail("profanity mode must be none, no-vowels or blocklist")
|
|
111
|
+
if profanity["mode"] == "no-vowels":
|
|
112
|
+
body_norm = strip_vowels(body_norm)
|
|
113
|
+
checksum_norm = strip_vowels(checksum_norm)
|
|
114
|
+
if len(body_norm) < 2:
|
|
115
|
+
_fail("no-vowels mode leaves the body alphabet with fewer than two symbols")
|
|
116
|
+
if checksum_length > 0 and len(checksum_norm) < 2:
|
|
117
|
+
_fail("no-vowels mode leaves the checksum alphabet with fewer than two symbols")
|
|
118
|
+
blocklist = (
|
|
119
|
+
effective_blocklist(profanity) if profanity["mode"] == "blocklist" else []
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
separator = profile.get("separator") or ""
|
|
123
|
+
if not isinstance(separator, str):
|
|
124
|
+
_fail("separator must be a string")
|
|
125
|
+
for ch in separator:
|
|
126
|
+
if ch in body_norm or ch in checksum_norm:
|
|
127
|
+
_fail("separator must not occur in either alphabet")
|
|
128
|
+
|
|
129
|
+
aliases = profile.get("aliases") or {}
|
|
130
|
+
if not isinstance(aliases, dict):
|
|
131
|
+
_fail("aliases must be a mapping")
|
|
132
|
+
aliases_norm: dict = {}
|
|
133
|
+
canonical_set = set(body_norm) | set(checksum_norm)
|
|
134
|
+
for src, tgt in aliases.items():
|
|
135
|
+
if not isinstance(src, str) or not _is_ascii_char(src):
|
|
136
|
+
_fail(f"alias source is not single ASCII: {src!r}")
|
|
137
|
+
if not isinstance(tgt, str) or not _is_ascii_char(tgt):
|
|
138
|
+
_fail(f"alias target is not single ASCII: {tgt!r}")
|
|
139
|
+
s_norm = _norm(case_sensitive, src)
|
|
140
|
+
t_norm = _norm(case_sensitive, tgt)
|
|
141
|
+
if s_norm in canonical_set:
|
|
142
|
+
_fail(f"alias source {src!r} is already a canonical symbol")
|
|
143
|
+
if t_norm not in canonical_set:
|
|
144
|
+
_fail(f"alias target {tgt!r} is not a canonical symbol")
|
|
145
|
+
if s_norm in aliases_norm:
|
|
146
|
+
_fail(f"duplicate alias source {s_norm!r} after case normalization")
|
|
147
|
+
if any(_norm(case_sensitive, key) == t_norm for key in aliases):
|
|
148
|
+
_fail(f"alias chain forbidden: target {t_norm} is also an alias source")
|
|
149
|
+
aliases_norm[s_norm] = t_norm
|
|
150
|
+
|
|
151
|
+
grouping = profile.get("grouping")
|
|
152
|
+
if not isinstance(grouping, (list, tuple)):
|
|
153
|
+
_fail("grouping must be empty when separator is empty")
|
|
154
|
+
if separator == "":
|
|
155
|
+
if len(grouping) != 0:
|
|
156
|
+
_fail("grouping must be empty when separator is empty")
|
|
157
|
+
else:
|
|
158
|
+
group_sum = 0
|
|
159
|
+
for g in grouping:
|
|
160
|
+
if not _is_int(g) or g < 1:
|
|
161
|
+
_fail("group sizes must sum to bodyLength + checksumLength")
|
|
162
|
+
group_sum += g
|
|
163
|
+
if group_sum != body_length + checksum_length:
|
|
164
|
+
_fail("group sizes must sum to bodyLength + checksumLength")
|
|
165
|
+
|
|
166
|
+
permutation = profile.get("permutation") or {"enabled": False}
|
|
167
|
+
if not isinstance(permutation, dict):
|
|
168
|
+
_fail("permutation must be a mapping")
|
|
169
|
+
if permutation.get("enabled"):
|
|
170
|
+
if permutation.get("algorithm") != "feistel-v1":
|
|
171
|
+
_fail("unknown permutation algorithm")
|
|
172
|
+
key_id = permutation.get("keyId")
|
|
173
|
+
if not isinstance(key_id, str) or len(key_id) == 0:
|
|
174
|
+
_fail("permutation requires a keyId")
|
|
175
|
+
key_bytes = permutation.get("keyBytes")
|
|
176
|
+
if not isinstance(key_bytes, (bytes, bytearray)) or len(key_bytes) == 0:
|
|
177
|
+
_fail("permutation requires key material")
|
|
178
|
+
rounds = permutation.get("rounds")
|
|
179
|
+
if not _is_int(rounds) or rounds < 4 or rounds > 16 or rounds % 2 != 0:
|
|
180
|
+
_fail("Feistel rounds must be an even integer from 4 through 16")
|
|
181
|
+
perm = PreparedPermutation(
|
|
182
|
+
enabled=True,
|
|
183
|
+
key_id=key_id,
|
|
184
|
+
key_bytes=bytes(key_bytes),
|
|
185
|
+
rounds=rounds,
|
|
186
|
+
)
|
|
187
|
+
else:
|
|
188
|
+
perm = PreparedPermutation(enabled=False)
|
|
189
|
+
|
|
190
|
+
modulus_base = len(checksum_norm) if checksum_norm else 1
|
|
191
|
+
return PreparedProfile(
|
|
192
|
+
profile_id=profile_id,
|
|
193
|
+
body_alphabet_norm=body_norm,
|
|
194
|
+
checksum_alphabet_norm=checksum_norm,
|
|
195
|
+
body_length=body_length,
|
|
196
|
+
checksum_length=checksum_length,
|
|
197
|
+
case_sensitive=case_sensitive,
|
|
198
|
+
separator=separator,
|
|
199
|
+
grouping=tuple(grouping),
|
|
200
|
+
aliases_norm=aliases_norm,
|
|
201
|
+
permutation=perm,
|
|
202
|
+
checksum_modulus=modulus_base ** checksum_length,
|
|
203
|
+
capacity=len(body_norm) ** body_length,
|
|
204
|
+
blocklist=tuple(blocklist),
|
|
205
|
+
)
|
baseh/profiles.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Frozen profile tiers, spec section 17.
|
|
2
|
+
|
|
3
|
+
Four tiers built from the full alphanumeric set with cumulative visual and
|
|
4
|
+
spoken strips:
|
|
5
|
+
|
|
6
|
+
Minimum 36 symbols, no checksum 2,176,782,336 ids
|
|
7
|
+
Light 31 symbols, 1 checksum 887,503,681 ids
|
|
8
|
+
Medium 28 symbols, 1 checksum 481,890,304 ids (default)
|
|
9
|
+
Heavy 26 symbols, 1 checksum 308,915,776 ids
|
|
10
|
+
|
|
11
|
+
All four are 6 body symbols, case-insensitive, run the default profanity
|
|
12
|
+
blocklist and keep the typed O/I/L aliases where possible. Minimum also
|
|
13
|
+
uses a hyphen delimiter; the rest have none. The _p variants are identical
|
|
14
|
+
but enable feistel-v1 permutation and require caller-supplied key material.
|
|
15
|
+
|
|
16
|
+
Each helper returns a freshly-built mutable profile dict on every call, so
|
|
17
|
+
callers can load a default and modify it.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
_OIL_ALIASES = {"O": "0", "I": "1", "L": "1"}
|
|
23
|
+
|
|
24
|
+
_TIERS = {
|
|
25
|
+
"minimum": {
|
|
26
|
+
"profileId": "baseh-minimum",
|
|
27
|
+
"bodyAlphabet": "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ",
|
|
28
|
+
"checksumAlphabet": "",
|
|
29
|
+
"checksumLength": 0,
|
|
30
|
+
"separator": "-",
|
|
31
|
+
"grouping": [3, 3],
|
|
32
|
+
"aliases": {},
|
|
33
|
+
},
|
|
34
|
+
"light": {
|
|
35
|
+
"profileId": "baseh-light",
|
|
36
|
+
"bodyAlphabet": "0123456789ABCEFGHJKMNPQRSUVWXYZ",
|
|
37
|
+
"checksumAlphabet": "234679ACEFGHJKMNPQRUVWXY",
|
|
38
|
+
"checksumLength": 1,
|
|
39
|
+
"separator": "",
|
|
40
|
+
"grouping": [],
|
|
41
|
+
"aliases": {**_OIL_ALIASES, "D": "B", "T": "P"},
|
|
42
|
+
},
|
|
43
|
+
"medium": {
|
|
44
|
+
"profileId": "baseh-medium",
|
|
45
|
+
"bodyAlphabet": "0123456789ACDEFGHJKMPQRUVXYZ",
|
|
46
|
+
"checksumAlphabet": "234679ACDEFGHJKMPQRUVXY",
|
|
47
|
+
"checksumLength": 1,
|
|
48
|
+
"separator": "",
|
|
49
|
+
"grouping": [],
|
|
50
|
+
# B and S are dropped for looking like 8 and 5; since they can never
|
|
51
|
+
# be issued, a typed B is always an 8 and a typed S always a 5.
|
|
52
|
+
"aliases": {**_OIL_ALIASES, "B": "8", "S": "5", "T": "P", "N": "M", "W": "V"},
|
|
53
|
+
},
|
|
54
|
+
"heavy": {
|
|
55
|
+
"profileId": "baseh-heavy",
|
|
56
|
+
"bodyAlphabet": "0123456789ABCEFHJKMPQRVXYZ",
|
|
57
|
+
"checksumAlphabet": "234679ACEFHJKMPQRUVXY",
|
|
58
|
+
"checksumLength": 1,
|
|
59
|
+
"separator": "",
|
|
60
|
+
"grouping": [],
|
|
61
|
+
"aliases": {
|
|
62
|
+
**_OIL_ALIASES,
|
|
63
|
+
"D": "B",
|
|
64
|
+
"T": "P",
|
|
65
|
+
"N": "M",
|
|
66
|
+
"W": "V",
|
|
67
|
+
"S": "F",
|
|
68
|
+
"G": "C",
|
|
69
|
+
},
|
|
70
|
+
},
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _tier(name: str, permutation: dict, p_suffix: bool) -> dict:
|
|
75
|
+
shape = _TIERS[name]
|
|
76
|
+
return {
|
|
77
|
+
"profileId": shape["profileId"] + ("-p" if p_suffix else "") + "-v1",
|
|
78
|
+
"bodyAlphabet": shape["bodyAlphabet"],
|
|
79
|
+
"bodyLength": 6,
|
|
80
|
+
"checksumAlphabet": shape["checksumAlphabet"],
|
|
81
|
+
"checksumLength": shape["checksumLength"],
|
|
82
|
+
"caseSensitive": False,
|
|
83
|
+
"separator": shape["separator"],
|
|
84
|
+
"grouping": list(shape["grouping"]),
|
|
85
|
+
"aliases": dict(shape["aliases"]),
|
|
86
|
+
"permutation": permutation,
|
|
87
|
+
"profanity": {"mode": "blocklist"},
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _keyed_permutation(key_bytes: bytes, key_id: str, rounds: int) -> dict:
|
|
92
|
+
return {
|
|
93
|
+
"enabled": True,
|
|
94
|
+
"algorithm": "feistel-v1",
|
|
95
|
+
"keyId": key_id,
|
|
96
|
+
"keyBytes": bytes(key_bytes),
|
|
97
|
+
"rounds": rounds,
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def baseh_minimum_v1() -> dict:
|
|
102
|
+
"""Alphanumeric, no safety strips, no checksum, hyphen-delimited XXX-XXX."""
|
|
103
|
+
return _tier("minimum", {"enabled": False}, False)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def baseh_minimum_p_v1(
|
|
107
|
+
key_bytes: bytes, key_id: str = "default", rounds: int = 8
|
|
108
|
+
) -> dict:
|
|
109
|
+
"""baseh-minimum with feistel-v1 permutation."""
|
|
110
|
+
return _tier("minimum", _keyed_permutation(key_bytes, key_id, rounds), True)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def baseh_light_v1() -> dict:
|
|
114
|
+
"""Visual light plus spoken light, one checksum symbol."""
|
|
115
|
+
return _tier("light", {"enabled": False}, False)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def baseh_light_p_v1(
|
|
119
|
+
key_bytes: bytes, key_id: str = "default", rounds: int = 8
|
|
120
|
+
) -> dict:
|
|
121
|
+
"""baseh-light with feistel-v1 permutation."""
|
|
122
|
+
return _tier("light", _keyed_permutation(key_bytes, key_id, rounds), True)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def baseh_medium_v1() -> dict:
|
|
126
|
+
"""Visual medium plus spoken medium, one checksum symbol. The default."""
|
|
127
|
+
return _tier("medium", {"enabled": False}, False)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def baseh_medium_p_v1(
|
|
131
|
+
key_bytes: bytes, key_id: str = "default", rounds: int = 8
|
|
132
|
+
) -> dict:
|
|
133
|
+
"""baseh-medium with feistel-v1 permutation."""
|
|
134
|
+
return _tier("medium", _keyed_permutation(key_bytes, key_id, rounds), True)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def baseh_heavy_v1() -> dict:
|
|
138
|
+
"""Conservative alphabet plus spoken heavy, one checksum symbol."""
|
|
139
|
+
return _tier("heavy", {"enabled": False}, False)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def baseh_heavy_p_v1(
|
|
143
|
+
key_bytes: bytes, key_id: str = "default", rounds: int = 8
|
|
144
|
+
) -> dict:
|
|
145
|
+
"""baseh-heavy with feistel-v1 permutation."""
|
|
146
|
+
return _tier("heavy", _keyed_permutation(key_bytes, key_id, rounds), True)
|
baseh/zero.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Zero-config pair over the frozen baseh-medium-v1 profile.
|
|
2
|
+
|
|
3
|
+
No profile object, no key: just the two functions an application needs
|
|
4
|
+
when it does not want to think about configuration.
|
|
5
|
+
|
|
6
|
+
to_code(id) -> "7KM4Q2H"
|
|
7
|
+
from_code(code) -> id
|
|
8
|
+
|
|
9
|
+
to_code accepts an int or a decimal string of digits. from_code strips
|
|
10
|
+
every whitespace character (edges and internal), accepts lowercase and
|
|
11
|
+
the typed aliases (O, I, L) and returns the id as an int. Any invalid
|
|
12
|
+
input raises BasehError, including the rare BLOCKED_CODE identifiers
|
|
13
|
+
that spell a blocklisted word; no correction attempts are ever made.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
from .codec import Baseh
|
|
21
|
+
from .profiles import baseh_medium_v1
|
|
22
|
+
|
|
23
|
+
_DECIMAL = re.compile(r"^[0-9]+$")
|
|
24
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
25
|
+
|
|
26
|
+
_ZERO = Baseh(baseh_medium_v1())
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _to_int(id: object) -> int:
|
|
30
|
+
if isinstance(id, bool):
|
|
31
|
+
raise TypeError(
|
|
32
|
+
"to_code expects a non-negative int or a decimal string"
|
|
33
|
+
)
|
|
34
|
+
if isinstance(id, int):
|
|
35
|
+
if id < 0:
|
|
36
|
+
raise ValueError("to_code expects a non-negative id")
|
|
37
|
+
return id
|
|
38
|
+
if isinstance(id, str) and _DECIMAL.match(id):
|
|
39
|
+
return int(id)
|
|
40
|
+
raise TypeError("to_code expects a non-negative int or a decimal string")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def to_code(id: object) -> str:
|
|
44
|
+
"""Encode an identifier with the zero-config Medium profile."""
|
|
45
|
+
return _ZERO.encode(_to_int(id))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def from_code(code: str) -> int:
|
|
49
|
+
"""Decode a code from the zero-config Medium profile back to its id."""
|
|
50
|
+
if not isinstance(code, str):
|
|
51
|
+
raise TypeError("from_code expects a string")
|
|
52
|
+
return _ZERO.decode(_WHITESPACE.sub("", code)).id
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: baseh
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: baseH codec: checksumed, optionally permuted human-readable identifiers
|
|
5
|
+
License: AGPL-3.0-only
|
|
6
|
+
Requires-Python: >=3.9
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
|
|
9
|
+
# baseh
|
|
10
|
+
|
|
11
|
+
Python implementation of the baseH codec. Encodes an internal integer ID as
|
|
12
|
+
a checksummed, optionally permuted human-readable reference code. Implements
|
|
13
|
+
the normative spec in `../spec/IMPLEMENTATION_CODEC.md` and passes the
|
|
14
|
+
frozen cross-language vectors in `../vectors/`.
|
|
15
|
+
|
|
16
|
+
Zero runtime dependencies. HMAC-SHA-256 comes from the standard library.
|
|
17
|
+
|
|
18
|
+
## Install
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
pip install ./python
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Or run in place without installing:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
PYTHONPATH=python/src python3 -c "import baseh"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Usage
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from baseh import Baseh, BasehError, baseh_medium_v1
|
|
34
|
+
|
|
35
|
+
codec = Baseh(baseh_medium_v1())
|
|
36
|
+
|
|
37
|
+
code = codec.encode(123456789)
|
|
38
|
+
print(code) # fixed-length, checksummed code
|
|
39
|
+
|
|
40
|
+
result = codec.decode(code.lower()) # case-insensitive
|
|
41
|
+
print(result.id) # 123456789
|
|
42
|
+
print(result.canonical_code) # canonical rendering
|
|
43
|
+
print(result.corrected) # False (input differed only in case)
|
|
44
|
+
|
|
45
|
+
# Assisted correction over spoken-confusion pairs (B/D, P/T, ...):
|
|
46
|
+
fixed = codec.decode(
|
|
47
|
+
code,
|
|
48
|
+
try_correction=True,
|
|
49
|
+
confusion_profile="light",
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# Non-throwing validation for user input:
|
|
53
|
+
check = codec.validate("0000000")
|
|
54
|
+
print(check) # {"valid": False, "reason": "INVALID_CHECKSUM"}
|
|
55
|
+
|
|
56
|
+
print(codec.capacity()) # 481890304
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Errors raise `BasehError` with a `.code` attribute, one of:
|
|
60
|
+
`INVALID_PROFILE`, `OUT_OF_RANGE`, `PERMUTATION_FAILURE`, `INVALID_LENGTH`,
|
|
61
|
+
`INVALID_CHARACTER`, `INVALID_CHECKSUM`, `AMBIGUOUS_INPUT`,
|
|
62
|
+
`TOO_MANY_CANDIDATES`, `BLOCKED_CODE`.
|
|
63
|
+
|
|
64
|
+
## Frozen tiers
|
|
65
|
+
|
|
66
|
+
Four frozen profiles, each 6 body symbols and case-insensitive, built from
|
|
67
|
+
the full alphanumeric set with cumulative visual and spoken strips:
|
|
68
|
+
|
|
69
|
+
| Tier | Helper | Symbols | Checksum | Capacity |
|
|
70
|
+
|------|--------|---------|----------|----------|
|
|
71
|
+
| Minimum | `baseh_minimum_v1()` | 36 | none | 2,176,782,336 |
|
|
72
|
+
| Light | `baseh_light_v1()` | 31 | 1 | 887,503,681 |
|
|
73
|
+
| Medium | `baseh_medium_v1()` | 28 | 1 | 481,890,304 |
|
|
74
|
+
| Heavy | `baseh_heavy_v1()` | 26 | 1 | 308,915,776 |
|
|
75
|
+
|
|
76
|
+
Medium is the default. All four keep the typed O/I/L aliases where possible
|
|
77
|
+
and run the default profanity blocklist. Minimum also uses a hyphen
|
|
78
|
+
delimiter (`XXX-XXX`); the rest have none.
|
|
79
|
+
|
|
80
|
+
Each helper returns a freshly-built mutable profile dict on every call, so
|
|
81
|
+
callers can load a default and modify it:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
profile = baseh_medium_v1()
|
|
85
|
+
profile["checksumLength"] = 2
|
|
86
|
+
codec = Baseh(profile)
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
### Permuted variants
|
|
90
|
+
|
|
91
|
+
The `_p` variants are identical to their tier but enable feistel-v1
|
|
92
|
+
permutation and require caller-supplied key material:
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from baseh import Baseh, baseh_medium_p_v1
|
|
96
|
+
|
|
97
|
+
key = bytes.fromhex("746573742d6f6e6c792d6b65792d6d6174657269616c2d30303031")
|
|
98
|
+
codec = Baseh(baseh_medium_p_v1(key, key_id="my-app-01")) # rounds defaults to 8
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Available as `baseh_minimum_p_v1`, `baseh_light_p_v1`, `baseh_medium_p_v1`
|
|
102
|
+
and `baseh_heavy_p_v1`.
|
|
103
|
+
|
|
104
|
+
## Profanity safety (spec 18)
|
|
105
|
+
|
|
106
|
+
All frozen tiers run the default blocklist. Profiles accept a `profanity`
|
|
107
|
+
field to change that:
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
# Block specific words (substrings of the raw code) at encode time:
|
|
111
|
+
profile = baseh_medium_v1()
|
|
112
|
+
profile["profanity"] = {"mode": "blocklist", "extraWords": ["QQQQ"]}
|
|
113
|
+
codec = Baseh(profile)
|
|
114
|
+
# codec.encode(id) raises BasehError(code="BLOCKED_CODE") on a match.
|
|
115
|
+
|
|
116
|
+
# Or remove vowels from both alphabets entirely:
|
|
117
|
+
profile["profanity"] = {"mode": "no-vowels"}
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
## Tests
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
cd python
|
|
124
|
+
PYTHONPATH=src python3 -m unittest discover -s tests -v
|
|
125
|
+
```
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
baseh/__init__.py,sha256=xCvcyAFBCJheYW4EaB1CDxpgp19P8X7RRDQz7b-BuSk,1450
|
|
2
|
+
baseh/basen.py,sha256=C7Uy9zy-nUu7DndzbN4-1BDOQn7d8fVTYWFq8hA5O_E,1152
|
|
3
|
+
baseh/blocklist.py,sha256=10-PPRNseYj_8LCbnyN4Ol4k9nP-oHr3rEb8G2c1xBk,1373
|
|
4
|
+
baseh/checksum.py,sha256=fU2zlu1eRTRGVFRs-dF_vK5Geby1y4Q6U8-Mrd54e64,1324
|
|
5
|
+
baseh/codec.py,sha256=n1_ym6NQzLsMQtvXMWRstuyHmsYYAGzZ9k0xt7kZJbM,9299
|
|
6
|
+
baseh/errors.py,sha256=qCeUAj3qGD5c05M2-kib4LJvK5H8mraO51i_dwPxKl0,1070
|
|
7
|
+
baseh/feistel.py,sha256=2z2-0p7QFFGJ2Pc66Wr1F6--7MJ314qhS4wp9GK149M,3365
|
|
8
|
+
baseh/profile.py,sha256=WQkLOB5MGVfbLbdNdWg2ng4mkrTmiLS-4czrEhLHydk,7961
|
|
9
|
+
baseh/profiles.py,sha256=hYYX3FWycKwMzwPIoUp0rcX07Z0sS3qX3vCnFU-OuJY,4827
|
|
10
|
+
baseh/zero.py,sha256=7D2yoWoHL-vCJjF4OdxOwHpgCShr2bpyZgusd3e4vm4,1655
|
|
11
|
+
baseh-1.1.0.dist-info/METADATA,sha256=fUS2eFLmP2avQLSzj-POcLljP4DfkjjLowLQxlv7zXg,3700
|
|
12
|
+
baseh-1.1.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
13
|
+
baseh-1.1.0.dist-info/top_level.txt,sha256=heWKFtH7f_7c9iEqeekm2zkwa3fsVdSIPzAIX1uyarI,6
|
|
14
|
+
baseh-1.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
baseh
|