jev-secrets 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jev_secrets/__init__.py +18 -0
- jev_secrets/_catalogue.py +44 -0
- jev_secrets/_config.py +290 -0
- jev_secrets/_dates.py +333 -0
- jev_secrets/_detection.py +345 -0
- jev_secrets/_generation.py +366 -0
- jev_secrets/_pseudonymiser.py +489 -0
- jev_secrets/_secrets.py +51 -0
- jev_secrets/_text.py +87 -0
- jev_secrets/_vault.py +150 -0
- jev_secrets/catalogue/alphabets.json +130 -0
- jev_secrets/catalogue/patterns.json +129 -0
- jev_secrets/errors.py +31 -0
- jev_secrets/py.typed +0 -0
- jev_secrets/typesafe.py +105 -0
- jev_secrets-1.0.0.dist-info/METADATA +342 -0
- jev_secrets-1.0.0.dist-info/RECORD +19 -0
- jev_secrets-1.0.0.dist-info/WHEEL +4 -0
- jev_secrets-1.0.0.dist-info/licenses/LICENSE +21 -0
jev_secrets/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Pseudonymises names, keys and other secrets in Jev queries before they are sent, and restores them in the answers."""
|
|
2
|
+
|
|
3
|
+
from ._config import Config
|
|
4
|
+
from ._pseudonymiser import Pseudonymised
|
|
5
|
+
from ._secrets import Secrets
|
|
6
|
+
from ._vault import Vault
|
|
7
|
+
from .errors import CollisionError, ConfigError, LeakError, SecretsError
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
'CollisionError',
|
|
11
|
+
'Config',
|
|
12
|
+
'ConfigError',
|
|
13
|
+
'LeakError',
|
|
14
|
+
'Pseudonymised',
|
|
15
|
+
'Secrets',
|
|
16
|
+
'SecretsError',
|
|
17
|
+
'Vault',
|
|
18
|
+
]
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The data files in catalogue/, which the PHP, JavaScript and Python implementations all read.
|
|
3
|
+
|
|
4
|
+
A wheel carries them inside the package; a clone has them at the repository root.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
from functools import cache
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def directory() -> Path:
|
|
16
|
+
here = Path(__file__).resolve().parent
|
|
17
|
+
bundled = here / 'catalogue'
|
|
18
|
+
|
|
19
|
+
return bundled if bundled.is_dir() else here.parents[2] / 'catalogue'
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@cache
|
|
23
|
+
def _load(name: str) -> dict[str, Any]:
|
|
24
|
+
path = directory() / name
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
data = json.loads(path.read_text(encoding='utf-8'))
|
|
28
|
+
except OSError as error:
|
|
29
|
+
raise RuntimeError(f'Cannot read {path}.') from error
|
|
30
|
+
except json.JSONDecodeError as error:
|
|
31
|
+
raise RuntimeError(f'{path} is not valid JSON: {error}') from error
|
|
32
|
+
|
|
33
|
+
if not isinstance(data, dict):
|
|
34
|
+
raise RuntimeError(f'{path} does not hold an object.')
|
|
35
|
+
|
|
36
|
+
return data
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def patterns() -> dict[str, Any]:
|
|
40
|
+
return _load('patterns.json')
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def alphabets() -> dict[str, Any]:
|
|
44
|
+
return _load('alphabets.json')
|
jev_secrets/_config.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
"""
|
|
2
|
+
What to pseudonymise and how. spec/config.schema.json describes the same shape, which the PHP
|
|
3
|
+
and JavaScript implementations also accept, refusing the same mistakes with the same messages.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from collections.abc import Mapping, Sequence
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Literal
|
|
12
|
+
|
|
13
|
+
from . import _catalogue
|
|
14
|
+
from ._detection import Pattern
|
|
15
|
+
from ._text import has_word_char, nfc, trim
|
|
16
|
+
from .errors import ConfigError
|
|
17
|
+
|
|
18
|
+
OPTIONS = (
|
|
19
|
+
'terms', 'patterns', 'fields', 'allow', 'defaults', 'disable', 'secretKeys', 'scope',
|
|
20
|
+
'letters', 'digits', 'dates', 'minLength', 'key',
|
|
21
|
+
)
|
|
22
|
+
DEFAULTS: dict[str, Any] = {'defaults': True, 'secretKeys': True, 'letters': 'shape', 'digits': 'lower', 'minLength': 2}
|
|
23
|
+
SCOPE_DEFAULTS = {'state': True, 'questions': True, 'keys': False}
|
|
24
|
+
DATES_DEFAULTS: dict[str, Any] = {'mode': 'shift', 'order': 'dmy', 'detect': False}
|
|
25
|
+
|
|
26
|
+
KEY_ENV = 'JEV_SECRETS_KEY'
|
|
27
|
+
"""The environment variable read for the key when the config names none."""
|
|
28
|
+
|
|
29
|
+
# Options whose object value an override merges into, one entry at a time, instead of replacing.
|
|
30
|
+
_NESTED = ('scope', 'dates')
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class Config:
|
|
35
|
+
terms: list[str]
|
|
36
|
+
patterns: list[Pattern]
|
|
37
|
+
defaults: list[Pattern]
|
|
38
|
+
fields: list[list[str]]
|
|
39
|
+
allow: list[str]
|
|
40
|
+
secret_keys: frozenset[str]
|
|
41
|
+
scope_state: bool
|
|
42
|
+
scope_questions: bool
|
|
43
|
+
scope_keys: bool
|
|
44
|
+
letters: str
|
|
45
|
+
digits: str
|
|
46
|
+
dates_mode: str
|
|
47
|
+
dates_order: str
|
|
48
|
+
dates_detect: bool
|
|
49
|
+
min_length: int
|
|
50
|
+
key_setting: str | Literal[False] | None
|
|
51
|
+
"""A key, False for a random key every call, or None to read KEY_ENV."""
|
|
52
|
+
raw: Mapping[str, Any] = field(repr=False)
|
|
53
|
+
"""The options as given, which with_() merges overrides into."""
|
|
54
|
+
|
|
55
|
+
def with_(self, overrides: Mapping[str, Any] | None) -> Config:
|
|
56
|
+
"""This config with some options replaced for one call. `scope` and `dates` merge entry
|
|
57
|
+
by entry; every other option given replaces the configured one."""
|
|
58
|
+
if not overrides:
|
|
59
|
+
return self
|
|
60
|
+
|
|
61
|
+
merged = dict(self.raw)
|
|
62
|
+
|
|
63
|
+
for name, value in overrides.items():
|
|
64
|
+
if name in _NESTED and isinstance(value, Mapping) and isinstance(merged.get(name), Mapping):
|
|
65
|
+
merged[name] = {**merged[name], **value}
|
|
66
|
+
else:
|
|
67
|
+
merged[name] = value
|
|
68
|
+
|
|
69
|
+
return Config.from_mapping(merged)
|
|
70
|
+
|
|
71
|
+
def key(self) -> str | None:
|
|
72
|
+
"""The key for a call: the configured one, or the environment's when none is configured,
|
|
73
|
+
or None for a random one. Read on every call, so a rotated environment key takes effect
|
|
74
|
+
without rebuilding the config."""
|
|
75
|
+
if isinstance(self.key_setting, str):
|
|
76
|
+
return self.key_setting
|
|
77
|
+
|
|
78
|
+
if self.key_setting is False:
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
return os.environ.get(KEY_ENV) or None
|
|
82
|
+
|
|
83
|
+
@classmethod
|
|
84
|
+
def from_mapping(cls, config: Mapping[str, Any]) -> Config:
|
|
85
|
+
raw = dict(config)
|
|
86
|
+
|
|
87
|
+
for name in config:
|
|
88
|
+
if name not in OPTIONS:
|
|
89
|
+
raise ConfigError(f'Unknown config option "{name}". Known options: {", ".join(OPTIONS)}.')
|
|
90
|
+
|
|
91
|
+
config = {**DEFAULTS, **config}
|
|
92
|
+
|
|
93
|
+
if not isinstance(config['defaults'], bool):
|
|
94
|
+
raise ConfigError('defaults must be true or false.')
|
|
95
|
+
|
|
96
|
+
min_length = config['minLength']
|
|
97
|
+
|
|
98
|
+
if not isinstance(min_length, int) or isinstance(min_length, bool) or min_length < 1:
|
|
99
|
+
raise ConfigError('minLength must be a positive integer.')
|
|
100
|
+
|
|
101
|
+
letters, digits = config['letters'], config['digits']
|
|
102
|
+
|
|
103
|
+
if letters not in ('shape', 'strict'):
|
|
104
|
+
raise ConfigError('letters must be "shape" or "strict".')
|
|
105
|
+
|
|
106
|
+
if digits not in ('lower', 'any'):
|
|
107
|
+
raise ConfigError('digits must be "lower" or "any".')
|
|
108
|
+
|
|
109
|
+
key = config.get('key')
|
|
110
|
+
|
|
111
|
+
if key is not None and key is not False and (not isinstance(key, str) or key == ''):
|
|
112
|
+
raise ConfigError('key must be a non-empty string, or false for a random key on every call.')
|
|
113
|
+
|
|
114
|
+
dates = {**DATES_DEFAULTS, **_map(config, 'dates')}
|
|
115
|
+
|
|
116
|
+
for name in dates:
|
|
117
|
+
if name not in DATES_DEFAULTS:
|
|
118
|
+
raise ConfigError(f'Unknown dates option "{name}". Known options: mode, order, detect.')
|
|
119
|
+
|
|
120
|
+
if dates['mode'] not in ('shift', 'digits'):
|
|
121
|
+
raise ConfigError('dates.mode must be "shift" or "digits".')
|
|
122
|
+
|
|
123
|
+
if dates['order'] not in ('dmy', 'mdy'):
|
|
124
|
+
raise ConfigError('dates.order must be "dmy" or "mdy".')
|
|
125
|
+
|
|
126
|
+
if not isinstance(dates['detect'], bool):
|
|
127
|
+
raise ConfigError('dates.detect must be true or false.')
|
|
128
|
+
|
|
129
|
+
scope = _map(config, 'scope')
|
|
130
|
+
|
|
131
|
+
for name in scope:
|
|
132
|
+
if name not in SCOPE_DEFAULTS:
|
|
133
|
+
raise ConfigError(f'Unknown scope "{name}". Known scopes: state, questions, keys.')
|
|
134
|
+
|
|
135
|
+
catalogue = _catalogue.patterns()
|
|
136
|
+
disable = _strings(config, 'disable')
|
|
137
|
+
ids = [entry['id'] for entry in catalogue['patterns']]
|
|
138
|
+
|
|
139
|
+
for pattern_id in disable:
|
|
140
|
+
if pattern_id not in ids:
|
|
141
|
+
raise ConfigError(
|
|
142
|
+
f'disable names "{pattern_id}", which is not a default pattern. Default patterns: {", ".join(ids)}.'
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
defaults = [
|
|
146
|
+
Pattern(entry['id'], entry['pattern'], entry.get('flags', ''), entry.get('validator'))
|
|
147
|
+
for entry in catalogue['patterns']
|
|
148
|
+
if config['defaults'] and entry['id'] not in disable
|
|
149
|
+
]
|
|
150
|
+
|
|
151
|
+
patterns = []
|
|
152
|
+
|
|
153
|
+
for i, entry in enumerate(_list(config, 'patterns')):
|
|
154
|
+
if not isinstance(entry, Mapping) or not isinstance(entry.get('pattern'), str):
|
|
155
|
+
raise ConfigError(f'patterns[{i}] must be an object with a string "pattern".')
|
|
156
|
+
|
|
157
|
+
flags = entry.get('flags') if entry.get('flags') is not None else ''
|
|
158
|
+
validator = entry.get('validator')
|
|
159
|
+
pattern_id = entry.get('id') if entry.get('id') is not None else f'patterns[{i}]'
|
|
160
|
+
|
|
161
|
+
valid = isinstance(flags, str) and isinstance(pattern_id, str) and isinstance(validator, (str, type(None)))
|
|
162
|
+
|
|
163
|
+
if not valid:
|
|
164
|
+
raise ConfigError(f'patterns[{i}] has an id, flags or validator that is not a string.')
|
|
165
|
+
|
|
166
|
+
patterns.append(Pattern(pattern_id, entry['pattern'], flags, validator))
|
|
167
|
+
|
|
168
|
+
secret_keys = _secret_keys(config['secretKeys'], catalogue['secretKeys'])
|
|
169
|
+
|
|
170
|
+
return cls(
|
|
171
|
+
terms=normalise_values(_strings(config, 'terms'), 'terms', min_length),
|
|
172
|
+
patterns=patterns,
|
|
173
|
+
defaults=defaults,
|
|
174
|
+
fields=[_path(p) for p in _strings(config, 'fields')],
|
|
175
|
+
allow=[s for s in (trim(nfc(s)) for s in _strings(config, 'allow')) if s != ''],
|
|
176
|
+
secret_keys=secret_keys,
|
|
177
|
+
scope_state=_flag(scope, 'state'),
|
|
178
|
+
scope_questions=_flag(scope, 'questions'),
|
|
179
|
+
scope_keys=_flag(scope, 'keys'),
|
|
180
|
+
letters=letters,
|
|
181
|
+
digits=digits,
|
|
182
|
+
dates_mode=dates['mode'],
|
|
183
|
+
dates_order=dates['order'],
|
|
184
|
+
dates_detect=dates['detect'],
|
|
185
|
+
min_length=min_length,
|
|
186
|
+
key_setting=key,
|
|
187
|
+
raw=raw,
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _secret_keys(setting: Any, catalogue: list[str]) -> frozenset[str]:
|
|
192
|
+
"""True for the catalogue's names, False for none, a list for exactly those names, or
|
|
193
|
+
{add, remove} to change the catalogue's list."""
|
|
194
|
+
if setting is True:
|
|
195
|
+
names = catalogue
|
|
196
|
+
elif setting is False:
|
|
197
|
+
names = []
|
|
198
|
+
elif isinstance(setting, (list, tuple)):
|
|
199
|
+
names = _strings({'secretKeys': setting}, 'secretKeys')
|
|
200
|
+
elif isinstance(setting, Mapping) and set(setting) <= {'add', 'remove'}:
|
|
201
|
+
remove = [normalise_key(name) for name in _strings(setting, 'remove')]
|
|
202
|
+
names = [name for name in [*catalogue, *map(normalise_key, _strings(setting, 'add'))] if name not in remove]
|
|
203
|
+
else:
|
|
204
|
+
raise ConfigError(
|
|
205
|
+
'secretKeys must be true, false, a list of key names, or an object with add and remove lists.'
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
return frozenset(normalise_key(name) for name in names)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def normalise_values(values: Sequence[str], name: str, min_length: int) -> list[str]:
|
|
212
|
+
"""Normalises values passed to a call or listed as terms, and rejects the ones that cannot
|
|
213
|
+
be matched safely."""
|
|
214
|
+
out: list[str] = []
|
|
215
|
+
|
|
216
|
+
for i, value in enumerate(values):
|
|
217
|
+
if not isinstance(value, str):
|
|
218
|
+
raise ConfigError(f'{name}[{i}] must be a string.')
|
|
219
|
+
|
|
220
|
+
value = trim(nfc(value))
|
|
221
|
+
|
|
222
|
+
if not has_word_char(value):
|
|
223
|
+
raise ConfigError(f'{name}[{i}] has no letter or digit, so there is nothing to match or replace.')
|
|
224
|
+
|
|
225
|
+
if len(value) < min_length:
|
|
226
|
+
raise ConfigError(f'{name}[{i}] is shorter than minLength ({min_length}). Lower minLength to allow it.')
|
|
227
|
+
|
|
228
|
+
if value not in out:
|
|
229
|
+
out.append(value)
|
|
230
|
+
|
|
231
|
+
return out
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def normalise_key(key: str) -> str:
|
|
235
|
+
"""Normalises a key the way the catalogue's secretKeys are written: lowercase, with spaces,
|
|
236
|
+
hyphens and underscores removed."""
|
|
237
|
+
return key.lower().replace(' ', '').replace('-', '').replace('_', '')
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _path(path: str) -> list[str]:
|
|
241
|
+
segments = path.split('.')
|
|
242
|
+
|
|
243
|
+
if path == '' or '' in segments or segments[0] not in ('state', 'questions', '*'):
|
|
244
|
+
raise ConfigError(f'The field path "{path}" must start with state or questions and have no empty segments.')
|
|
245
|
+
|
|
246
|
+
return segments
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _strings(config: Mapping[str, Any], name: str) -> list[str]:
|
|
250
|
+
items = _list(config, name)
|
|
251
|
+
|
|
252
|
+
for i, item in enumerate(items):
|
|
253
|
+
if not isinstance(item, str):
|
|
254
|
+
raise ConfigError(f'{name}[{i}] must be a string.')
|
|
255
|
+
|
|
256
|
+
return items
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _list(config: Mapping[str, Any], name: str) -> list[Any]:
|
|
260
|
+
value = config.get(name)
|
|
261
|
+
|
|
262
|
+
if value is None:
|
|
263
|
+
return []
|
|
264
|
+
|
|
265
|
+
if not isinstance(value, (list, tuple)):
|
|
266
|
+
raise ConfigError(f'{name} must be a list.')
|
|
267
|
+
|
|
268
|
+
return list(value)
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _map(config: Mapping[str, Any], name: str) -> Mapping[str, Any]:
|
|
272
|
+
value = config.get(name)
|
|
273
|
+
|
|
274
|
+
if value is None or value == []:
|
|
275
|
+
return {}
|
|
276
|
+
|
|
277
|
+
if not isinstance(value, Mapping):
|
|
278
|
+
raise ConfigError(f'{name} must be an object.')
|
|
279
|
+
|
|
280
|
+
return value
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _flag(scope: Mapping[str, Any], name: str) -> bool:
|
|
284
|
+
value = scope.get(name)
|
|
285
|
+
value = SCOPE_DEFAULTS[name] if value is None else value
|
|
286
|
+
|
|
287
|
+
if not isinstance(value, bool):
|
|
288
|
+
raise ConfigError(f'scope.{name} must be true or false.')
|
|
289
|
+
|
|
290
|
+
return value
|
jev_secrets/_dates.py
ADDED
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Dates, times and datetimes: finding them in text and moving every one in a request by the
|
|
3
|
+
same offset, so their order and the intervals between them survive.
|
|
4
|
+
|
|
5
|
+
Dates move by a whole number of weeks, so a weekday written beside a date stays right. Times
|
|
6
|
+
move by a number of minutes chosen so no time in the request crosses midnight, and the move
|
|
7
|
+
never carries into the date, so a date written alone and the same date in a datetime stay
|
|
8
|
+
equal. docs/dates.md lists the formats.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
from ._generation import ATTEMPTS, Stream
|
|
17
|
+
|
|
18
|
+
MONTHS = (
|
|
19
|
+
'january', 'february', 'march', 'april', 'may', 'june',
|
|
20
|
+
'july', 'august', 'september', 'october', 'november', 'december',
|
|
21
|
+
)
|
|
22
|
+
ABBREVIATIONS = ('jan', 'feb', 'mar', 'apr', 'may', 'jun', 'jul', 'aug', 'sep', 'oct', 'nov', 'dec')
|
|
23
|
+
|
|
24
|
+
WEEKS = 52
|
|
25
|
+
"""Weeks a date can move, either way."""
|
|
26
|
+
|
|
27
|
+
_BEFORE = '(?<![0-9A-Za-z])'
|
|
28
|
+
_AFTER = '(?![0-9A-Za-z])'
|
|
29
|
+
_MONTH_NAMES = (
|
|
30
|
+
'january|february|march|april|may|june|july|august|september|october|november|december'
|
|
31
|
+
'|sept|jan|feb|mar|apr|jun|jul|aug|sep|oct|nov|dec'
|
|
32
|
+
)
|
|
33
|
+
_TIME = (
|
|
34
|
+
r'(?P<h>[0-9]{1,2}):(?P<mi>[0-9]{2})(?::(?P<sec>[0-9]{2})(?P<frac>\.[0-9]{1,9})?)?'
|
|
35
|
+
r'(?:(?P<aps> ?)(?P<ap>[AaPp]\.?[Mm]\.?))?'
|
|
36
|
+
)
|
|
37
|
+
_DATES = {
|
|
38
|
+
'ymd': r'(?P<y>[0-9]{4})(?P<s1>[-/.])(?P<m>[0-9]{1,2})(?P=s1)(?P<d>[0-9]{1,2})',
|
|
39
|
+
'numeric': r'(?P<a>[0-9]{1,2})(?P<s1>[-/.])(?P<b>[0-9]{1,2})(?P=s1)(?P<y>[0-9]{4}|[0-9]{2})',
|
|
40
|
+
'day-month': (
|
|
41
|
+
r'(?P<d>[0-9]{1,2})(?P<o>st|nd|rd|th|ST|ND|RD|TH)?(?P<s1> +|-)(?P<mon>(?i:' + _MONTH_NAMES + r'))'
|
|
42
|
+
r'(?P<dot>\.)?(?P<s2>,? +|-)(?P<y>[0-9]{4})'
|
|
43
|
+
),
|
|
44
|
+
'month-day': (
|
|
45
|
+
r'(?P<mon>(?i:' + _MONTH_NAMES + r'))(?P<dot>\.)?(?P<s1> +)(?P<d>[0-9]{1,2})'
|
|
46
|
+
r'(?P<o>st|nd|rd|th|ST|ND|RD|TH)?(?P<s2>,? +)(?P<y>[0-9]{4})'
|
|
47
|
+
),
|
|
48
|
+
}
|
|
49
|
+
_DATE_REGEXES = {
|
|
50
|
+
name: re.compile(
|
|
51
|
+
_BEFORE + date + r'(?:(?P<j>T| |, | at )' + _TIME + r'(?P<tz>Z|[+-][0-9]{2}:?[0-9]{2})?)?' + _AFTER
|
|
52
|
+
)
|
|
53
|
+
for name, date in _DATES.items()
|
|
54
|
+
}
|
|
55
|
+
_TIME_REGEX = re.compile(_BEFORE + _TIME + _AFTER)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _intdiv(a: int, b: int) -> int:
|
|
59
|
+
"""PHP's intdiv, which truncates toward zero where // floors."""
|
|
60
|
+
quotient = abs(a) // abs(b)
|
|
61
|
+
return quotient if (a >= 0) == (b > 0) else -quotient
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def to_days(year: int, month: int, day: int) -> int:
|
|
65
|
+
"""Days from 1970-01-01 in the proleptic Gregorian calendar, by Howard Hinnant's days_from_civil."""
|
|
66
|
+
y = year - 1 if month <= 2 else year
|
|
67
|
+
era = _intdiv(y if y >= 0 else y - 399, 400)
|
|
68
|
+
yoe = y - era * 400
|
|
69
|
+
doy = _intdiv(153 * (month + (-3 if month > 2 else 9)) + 2, 5) + day - 1
|
|
70
|
+
doe = yoe * 365 + _intdiv(yoe, 4) - _intdiv(yoe, 100) + doy
|
|
71
|
+
|
|
72
|
+
return era * 146097 + doe - 719468
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def from_days(days: int) -> tuple[int, int, int]:
|
|
76
|
+
"""Year, month and day, by Howard Hinnant's civil_from_days."""
|
|
77
|
+
z = days + 719468
|
|
78
|
+
era = _intdiv(z if z >= 0 else z - 146096, 146097)
|
|
79
|
+
doe = z - era * 146097
|
|
80
|
+
yoe = _intdiv(doe - _intdiv(doe, 1460) + _intdiv(doe, 36524) - _intdiv(doe, 146096), 365)
|
|
81
|
+
doy = doe - (365 * yoe + _intdiv(yoe, 4) - _intdiv(yoe, 100))
|
|
82
|
+
mp = _intdiv(5 * doy + 2, 153)
|
|
83
|
+
day = doy - _intdiv(153 * mp + 2, 5) + 1
|
|
84
|
+
month = mp + 3 if mp < 10 else mp - 9
|
|
85
|
+
|
|
86
|
+
return yoe + era * 400 + (1 if month <= 2 else 0), month, day
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def days_in_month(year: int, month: int) -> int:
|
|
90
|
+
if month == 2:
|
|
91
|
+
return 29 if (year % 4 == 0 and year % 100 != 0) or year % 400 == 0 else 28
|
|
92
|
+
|
|
93
|
+
return 30 if month in (4, 6, 9, 11) else 31
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class Shifter:
|
|
97
|
+
def __init__(self, order: str) -> None:
|
|
98
|
+
"""`order` is how a numeric date with the year last is read: `dmy` or `mdy`."""
|
|
99
|
+
self._order = order
|
|
100
|
+
self._days = 0
|
|
101
|
+
self._minutes = 0
|
|
102
|
+
self._original_days: set[int] = set()
|
|
103
|
+
self._original_minutes: set[int] = set()
|
|
104
|
+
|
|
105
|
+
def find(self, text: str) -> list[dict[str, Any]]:
|
|
106
|
+
"""Every date, time and datetime in the text, longest first where two overlap."""
|
|
107
|
+
found: list[dict[str, Any]] = []
|
|
108
|
+
|
|
109
|
+
for fmt, regex in (*_DATE_REGEXES.items(), ('time', _TIME_REGEX)):
|
|
110
|
+
for m in regex.finditer(text):
|
|
111
|
+
match = {name: value for name, value in m.groupdict().items() if value is not None}
|
|
112
|
+
parsed = self._parse(fmt, match)
|
|
113
|
+
|
|
114
|
+
if parsed is not None:
|
|
115
|
+
found.append({'start': m.start(), 'end': m.end(), 'match': match, 'format': fmt, **parsed})
|
|
116
|
+
|
|
117
|
+
found.sort(key=lambda f: (-(f['end'] - f['start']), f['start']))
|
|
118
|
+
kept: list[dict[str, Any]] = []
|
|
119
|
+
|
|
120
|
+
for candidate in found:
|
|
121
|
+
if not any(candidate['start'] < taken['end'] and taken['start'] < candidate['end'] for taken in kept):
|
|
122
|
+
kept.append(candidate)
|
|
123
|
+
|
|
124
|
+
return sorted(kept, key=lambda f: f['start'])
|
|
125
|
+
|
|
126
|
+
def prepare(self, key: bytes, texts: list[str]) -> None:
|
|
127
|
+
"""
|
|
128
|
+
Picks this request's offsets from every date and time it holds.
|
|
129
|
+
|
|
130
|
+
The minute offset keeps every time between 00:00 and 23:59. Offsets are drawn again, up
|
|
131
|
+
to ATTEMPTS times, until no moved date or time equals one already in the request; after
|
|
132
|
+
that the draw with the fewest such collisions is kept, and shift() returns None for a
|
|
133
|
+
colliding value, which the caller leaves to the character generator.
|
|
134
|
+
"""
|
|
135
|
+
# Sets suffice: only membership and the earliest and latest minute are read.
|
|
136
|
+
for text in texts:
|
|
137
|
+
for found in self.find(text):
|
|
138
|
+
if found['day'] is not None:
|
|
139
|
+
self._original_days.add(found['day'])
|
|
140
|
+
|
|
141
|
+
if found['minute'] is not None:
|
|
142
|
+
self._original_minutes.add(found['minute'])
|
|
143
|
+
|
|
144
|
+
low = -min(self._original_minutes) if self._original_minutes else 0
|
|
145
|
+
high = 1439 - max(self._original_minutes) if self._original_minutes else 0
|
|
146
|
+
best: tuple[int, int, int] | None = None
|
|
147
|
+
|
|
148
|
+
for attempt in range(ATTEMPTS):
|
|
149
|
+
stream = Stream(key, f'jev-secrets/v1\0dates\0{attempt}'.encode())
|
|
150
|
+
week = stream.below(2 * WEEKS)
|
|
151
|
+
days = 7 * (week - WEEKS if week < WEEKS else week - WEEKS + 1)
|
|
152
|
+
minutes = 0
|
|
153
|
+
|
|
154
|
+
if high - low >= 1:
|
|
155
|
+
minutes = low + stream.below(high - low)
|
|
156
|
+
minutes += 1 if minutes >= 0 else 0
|
|
157
|
+
|
|
158
|
+
collisions = sum(1 for day in self._original_days if day + days in self._original_days)
|
|
159
|
+
|
|
160
|
+
if minutes != 0:
|
|
161
|
+
collisions += sum(1 for minute in self._original_minutes if minute + minutes in self._original_minutes)
|
|
162
|
+
|
|
163
|
+
if best is None or collisions < best[0]:
|
|
164
|
+
best = (collisions, days, minutes)
|
|
165
|
+
|
|
166
|
+
if collisions == 0:
|
|
167
|
+
break
|
|
168
|
+
|
|
169
|
+
assert best is not None
|
|
170
|
+
_, self._days, self._minutes = best
|
|
171
|
+
|
|
172
|
+
def shift(self, found: dict[str, Any]) -> str | None:
|
|
173
|
+
"""The moved text of a match from find(), or None when it cannot move: its time would
|
|
174
|
+
leave the day, or it would land on a date or time already in the request."""
|
|
175
|
+
m = found['match']
|
|
176
|
+
out = ''
|
|
177
|
+
|
|
178
|
+
if found['day'] is not None:
|
|
179
|
+
day = found['day'] + self._days
|
|
180
|
+
|
|
181
|
+
if day in self._original_days:
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
out = self._format_date(found['format'], m, *from_days(day))
|
|
185
|
+
|
|
186
|
+
if found['minute'] is not None:
|
|
187
|
+
minute = found['minute'] + self._minutes
|
|
188
|
+
|
|
189
|
+
if minute < 0 or minute > 1439 or (self._minutes != 0 and minute in self._original_minutes):
|
|
190
|
+
return None
|
|
191
|
+
|
|
192
|
+
out += m.get('j', '') + _format_time(m, minute) + m.get('tz', '')
|
|
193
|
+
|
|
194
|
+
return out
|
|
195
|
+
|
|
196
|
+
def _parse(self, fmt: str, m: dict[str, str]) -> dict[str, int | None] | None:
|
|
197
|
+
minute = None
|
|
198
|
+
|
|
199
|
+
if 'h' in m:
|
|
200
|
+
minute = _minute(m)
|
|
201
|
+
|
|
202
|
+
if minute is None:
|
|
203
|
+
return None
|
|
204
|
+
|
|
205
|
+
if fmt == 'time':
|
|
206
|
+
return {'day': None, 'minute': minute}
|
|
207
|
+
|
|
208
|
+
year = int(m['y'])
|
|
209
|
+
|
|
210
|
+
if len(m['y']) == 2:
|
|
211
|
+
year += 2000 if year < 70 else 1900
|
|
212
|
+
|
|
213
|
+
if fmt == 'numeric':
|
|
214
|
+
month, day = (int(m['b']), int(m['a'])) if self._order == 'dmy' else (int(m['a']), int(m['b']))
|
|
215
|
+
elif fmt in ('day-month', 'month-day'):
|
|
216
|
+
month, day = _month_number(m['mon']), int(m['d'])
|
|
217
|
+
else:
|
|
218
|
+
month, day = int(m['m']), int(m['d'])
|
|
219
|
+
|
|
220
|
+
if month < 1 or month > 12 or day < 1 or day > days_in_month(year, month):
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
return {'day': to_days(year, month, day), 'minute': minute}
|
|
224
|
+
|
|
225
|
+
def _format_date(self, fmt: str, m: dict[str, str], year: int, month: int, day: int) -> str:
|
|
226
|
+
def pad(value: int, like: str) -> str:
|
|
227
|
+
return f'{value:02d}' if len(like) >= 2 else str(value)
|
|
228
|
+
|
|
229
|
+
y = f'{year % 100:02d}' if len(m['y']) == 2 else _php_04d(year)
|
|
230
|
+
|
|
231
|
+
if fmt == 'ymd':
|
|
232
|
+
return y + m['s1'] + pad(month, m['m']) + m['s1'] + pad(day, m['d'])
|
|
233
|
+
|
|
234
|
+
if fmt == 'numeric':
|
|
235
|
+
if self._order == 'dmy':
|
|
236
|
+
return pad(day, m['a']) + m['s1'] + pad(month, m['b']) + m['s1'] + y
|
|
237
|
+
|
|
238
|
+
return pad(month, m['a']) + m['s1'] + pad(day, m['b']) + m['s1'] + y
|
|
239
|
+
|
|
240
|
+
if fmt == 'day-month':
|
|
241
|
+
return (
|
|
242
|
+
pad(day, m['d']) + _ordinal(day, m.get('o')) + m['s1'] + _month_name(month, m['mon'])
|
|
243
|
+
+ m.get('dot', '') + m['s2'] + y
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
return (
|
|
247
|
+
_month_name(month, m['mon']) + m.get('dot', '') + m['s1'] + pad(day, m['d'])
|
|
248
|
+
+ _ordinal(day, m.get('o')) + m['s2'] + y
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _php_04d(value: int) -> str:
|
|
253
|
+
"""sprintf('%04d') as PHP writes it: a negative year pads to four characters including the sign."""
|
|
254
|
+
return f'{value:04d}' if value >= 0 else '-' + f'{-value:03d}'
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _minute(m: dict[str, str]) -> int | None:
|
|
258
|
+
hour = int(m['h'])
|
|
259
|
+
minute = int(m['mi'])
|
|
260
|
+
|
|
261
|
+
if minute > 59 or ('sec' in m and int(m['sec']) > 59):
|
|
262
|
+
return None
|
|
263
|
+
|
|
264
|
+
if 'ap' in m:
|
|
265
|
+
if hour < 1 or hour > 12:
|
|
266
|
+
return None
|
|
267
|
+
|
|
268
|
+
hour = hour % 12 + (12 if m['ap'][0].lower() == 'p' else 0)
|
|
269
|
+
elif hour > 23:
|
|
270
|
+
return None
|
|
271
|
+
|
|
272
|
+
return hour * 60 + minute
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _format_time(m: dict[str, str], minute: int) -> str:
|
|
276
|
+
hour = minute // 60
|
|
277
|
+
out = f'{minute % 60:02d}'
|
|
278
|
+
|
|
279
|
+
if 'sec' in m:
|
|
280
|
+
out += ':' + m['sec'] + m.get('frac', '')
|
|
281
|
+
|
|
282
|
+
if 'ap' in m:
|
|
283
|
+
twelve = 12 if hour % 12 == 0 else hour % 12
|
|
284
|
+
letter = 'a' if hour < 12 else 'p'
|
|
285
|
+
ap = (letter.upper() if m['ap'][0].isupper() else letter) + m['ap'][1:]
|
|
286
|
+
hours = f'{twelve:02d}' if len(m['h']) >= 2 else str(twelve)
|
|
287
|
+
|
|
288
|
+
return hours + ':' + out + m.get('aps', '') + ap
|
|
289
|
+
|
|
290
|
+
return (f'{hour:02d}' if len(m['h']) >= 2 else str(hour)) + ':' + out
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _month_number(name: str) -> int:
|
|
294
|
+
lower = name.lower()
|
|
295
|
+
lower = 'sep' if lower == 'sept' else lower
|
|
296
|
+
|
|
297
|
+
if lower in ABBREVIATIONS:
|
|
298
|
+
return ABBREVIATIONS.index(lower) + 1
|
|
299
|
+
|
|
300
|
+
return MONTHS.index(lower) + 1 if lower in MONTHS else 0
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _month_name(month: int, like: str) -> str:
|
|
304
|
+
"""The month's name written the way the original was: full or abbreviated, and in the same
|
|
305
|
+
case. "Sept" stays four letters when the month lands on September."""
|
|
306
|
+
lower = like.lower()
|
|
307
|
+
full = lower in MONTHS and lower not in ABBREVIATIONS
|
|
308
|
+
|
|
309
|
+
# "May" is both a full name and an abbreviation; a date written with it gets the moved
|
|
310
|
+
# month's full name.
|
|
311
|
+
if full or lower == 'may':
|
|
312
|
+
name = MONTHS[month - 1]
|
|
313
|
+
elif lower == 'sept' and month == 9:
|
|
314
|
+
name = 'sept'
|
|
315
|
+
else:
|
|
316
|
+
name = ABBREVIATIONS[month - 1]
|
|
317
|
+
|
|
318
|
+
if like.upper() == like:
|
|
319
|
+
return name.upper()
|
|
320
|
+
|
|
321
|
+
if lower[:1].upper() + lower[1:] == like:
|
|
322
|
+
return name[:1].upper() + name[1:]
|
|
323
|
+
|
|
324
|
+
return name
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _ordinal(day: int, like: str | None) -> str:
|
|
328
|
+
if like is None:
|
|
329
|
+
return ''
|
|
330
|
+
|
|
331
|
+
suffix = 'th' if 11 <= day % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(day % 10, 'th')
|
|
332
|
+
|
|
333
|
+
return suffix.upper() if like.isupper() else suffix
|