mscs 2.3.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mscs-2.3.0 → mscs-2.4.0}/PKG-INFO +4 -3
- {mscs-2.3.0 → mscs-2.4.0}/README.md +3 -2
- {mscs-2.3.0 → mscs-2.4.0}/pyproject.toml +1 -1
- {mscs-2.3.0 → mscs-2.4.0}/src/mscs/_core.py +97 -33
- {mscs-2.3.0 → mscs-2.4.0}/.gitignore +0 -0
- {mscs-2.3.0 → mscs-2.4.0}/LICENSE +0 -0
- {mscs-2.3.0 → mscs-2.4.0}/src/mscs/__init__.py +0 -0
- {mscs-2.3.0 → mscs-2.4.0}/src/mscs/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mscs
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Safe, fast serialization for Python — a secure replacement for pickle with HMAC authentication, native numpy/PyTorch support.
|
|
5
5
|
Project-URL: Homepage, https://github.com/ElEscribanoSilente/MSC-Serial
|
|
6
6
|
Project-URL: Repository, https://github.com/ElEscribanoSilente/MSC-Serial
|
|
@@ -41,7 +41,7 @@ Description-Content-Type: text/markdown
|
|
|
41
41
|
|
|
42
42
|
# MSCS — Safe Serialization for Python
|
|
43
43
|
|
|
44
|
-
**v2.
|
|
44
|
+
**v2.4.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
|
|
45
45
|
|
|
46
46
|
> **Status: Beta** — API is stable but the format may evolve. Not yet battle-tested in large-scale production.
|
|
47
47
|
|
|
@@ -208,7 +208,7 @@ mscs.loads(unsigned_data, hmac_key=key) # MSCSecurityError: anti-downgrade prot
|
|
|
208
208
|
|------|-------|
|
|
209
209
|
| `None`, `bool`, `int`, `float`, `complex` | Ints up to 8192 bytes (~19,700 digits) |
|
|
210
210
|
| `str`, `bytes`, `bytearray` | UTF-8, ref-tracked |
|
|
211
|
-
| `list`, `tuple`, `dict`, `set`, `frozenset` | Circular refs supported |
|
|
211
|
+
| `list`, `tuple`, `dict`, `set`, `frozenset`, `deque` | Circular refs supported; `deque` preserves `maxlen` |
|
|
212
212
|
| `datetime`, `date`, `time`, `timedelta` | ISO 8601 |
|
|
213
213
|
| `Decimal`, `UUID`, `Path` | Lossless |
|
|
214
214
|
| `Enum` | Must be registered |
|
|
@@ -303,6 +303,7 @@ CRC32 and HMAC are mutually exclusive (HMAC is strictly superior).
|
|
|
303
303
|
| 0x17 | Path | `<I>` str len + UTF-8 path string |
|
|
304
304
|
| 0x18 | Tensor | str(meta) + `<I>` data size + raw buffer |
|
|
305
305
|
| 0x19 | timedelta2 | `<iiI>` days, seconds, microseconds |
|
|
306
|
+
| 0x1A | deque | `<i>` maxlen (-1 if None) + `<I>` item count + items |
|
|
306
307
|
|
|
307
308
|
**ndarray meta**: `"{dtype}|{shape}"` where shape is `"dim0xdim1x..."` (e.g., `"float32|100x100"`).
|
|
308
309
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# MSCS — Safe Serialization for Python
|
|
2
2
|
|
|
3
|
-
**v2.
|
|
3
|
+
**v2.4.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
|
|
4
4
|
|
|
5
5
|
> **Status: Beta** — API is stable but the format may evolve. Not yet battle-tested in large-scale production.
|
|
6
6
|
|
|
@@ -167,7 +167,7 @@ mscs.loads(unsigned_data, hmac_key=key) # MSCSecurityError: anti-downgrade prot
|
|
|
167
167
|
|------|-------|
|
|
168
168
|
| `None`, `bool`, `int`, `float`, `complex` | Ints up to 8192 bytes (~19,700 digits) |
|
|
169
169
|
| `str`, `bytes`, `bytearray` | UTF-8, ref-tracked |
|
|
170
|
-
| `list`, `tuple`, `dict`, `set`, `frozenset` | Circular refs supported |
|
|
170
|
+
| `list`, `tuple`, `dict`, `set`, `frozenset`, `deque` | Circular refs supported; `deque` preserves `maxlen` |
|
|
171
171
|
| `datetime`, `date`, `time`, `timedelta` | ISO 8601 |
|
|
172
172
|
| `Decimal`, `UUID`, `Path` | Lossless |
|
|
173
173
|
| `Enum` | Must be registered |
|
|
@@ -262,6 +262,7 @@ CRC32 and HMAC are mutually exclusive (HMAC is strictly superior).
|
|
|
262
262
|
| 0x17 | Path | `<I>` str len + UTF-8 path string |
|
|
263
263
|
| 0x18 | Tensor | str(meta) + `<I>` data size + raw buffer |
|
|
264
264
|
| 0x19 | timedelta2 | `<iiI>` days, seconds, microseconds |
|
|
265
|
+
| 0x1A | deque | `<i>` maxlen (-1 if None) + `<I>` item count + items |
|
|
265
266
|
|
|
266
267
|
**ndarray meta**: `"{dtype}|{shape}"` where shape is `"dim0xdim1x..."` (e.g., `"float32|100x100"`).
|
|
267
268
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "mscs"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.4.0"
|
|
8
8
|
description = "Safe, fast serialization for Python — a secure replacement for pickle with HMAC authentication, native numpy/PyTorch support."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
"""
|
|
2
|
-
MSC Serial v2.
|
|
2
|
+
MSC Serial v2.4
|
|
3
3
|
===============
|
|
4
4
|
Reemplazo personal y seguro de pickle.
|
|
5
5
|
|
|
6
|
-
Soporta: dict, list, tuple, set, frozenset, str, int, float,
|
|
7
|
-
bool, None, bytes, bytearray, datetime, date, time,
|
|
8
|
-
Decimal, UUID, Path, Enum, numpy arrays, torch.Tensor,
|
|
6
|
+
Soporta: dict, list, tuple, set, frozenset, deque, str, int, float,
|
|
7
|
+
complex, bool, None, bytes, bytearray, datetime, date, time,
|
|
8
|
+
timedelta, Decimal, UUID, Path, Enum, numpy arrays, torch.Tensor,
|
|
9
9
|
dataclasses, objetos con __slots__, objetos custom registrados,
|
|
10
10
|
referencias circulares.
|
|
11
11
|
|
|
@@ -39,6 +39,16 @@ Seguridad:
|
|
|
39
39
|
todos los objetos serializados, los IDs no se reutilizan durante
|
|
40
40
|
una sola llamada a encode().
|
|
41
41
|
|
|
42
|
+
Changelog v2.4.0:
|
|
43
|
+
- FIX: __getstate__/__setstate__ ahora tiene prioridad sobre dataclass
|
|
44
|
+
field walking — antes, dataclasses que definían __getstate__ eran
|
|
45
|
+
serializados recorriendo fields directamente, lo que causaba
|
|
46
|
+
MSCEncodeError si algún field contenía tipos no soportados (ej. deque).
|
|
47
|
+
La prioridad ahora es: __getstate__/__setstate__ > dataclass fields > __slots__ > __dict__
|
|
48
|
+
- ADD: Soporte nativo collections.deque (tag 0x1A) — preserva maxlen
|
|
49
|
+
y soporta referencias circulares
|
|
50
|
+
- Retrocompatible con payloads v2.3, v2.2, v2.1, v2.0 y v1.0
|
|
51
|
+
|
|
42
52
|
Changelog v2.3.0:
|
|
43
53
|
- ADD: HMAC-SHA256 autenticación criptográfica (hmac_key= en dumps/loads)
|
|
44
54
|
- ADD: Protección anti-downgrade (payload sin HMAC + clave = rechazado)
|
|
@@ -99,6 +109,7 @@ import hashlib
|
|
|
99
109
|
import threading
|
|
100
110
|
import inspect as _inspect_mod
|
|
101
111
|
import dataclasses
|
|
112
|
+
from collections import deque
|
|
102
113
|
from datetime import datetime, date, time, timedelta
|
|
103
114
|
from decimal import Decimal
|
|
104
115
|
from enum import Enum
|
|
@@ -118,7 +129,7 @@ try:
|
|
|
118
129
|
except ImportError:
|
|
119
130
|
_torch = None
|
|
120
131
|
|
|
121
|
-
__version__ = "2.
|
|
132
|
+
__version__ = "2.4.0"
|
|
122
133
|
__all__ = [
|
|
123
134
|
"dump", "load", "dumps", "loads",
|
|
124
135
|
"dump_compressed", "load_compressed",
|
|
@@ -170,6 +181,7 @@ _UUID = b'\x16'
|
|
|
170
181
|
_PATH = b'\x17'
|
|
171
182
|
_TENSOR = b'\x18'
|
|
172
183
|
_TIMEDELTA2 = b'\x19' # v2.2: timedelta sin ambiguedad
|
|
184
|
+
_DEQUE = b'\x1A' # v2.4: collections.deque nativo
|
|
173
185
|
|
|
174
186
|
_TAG_NAMES: Dict[int, str] = {
|
|
175
187
|
0x00: 'None', 0x01: 'bool', 0x02: 'int', 0x03: 'float',
|
|
@@ -178,7 +190,7 @@ _TAG_NAMES: Dict[int, str] = {
|
|
|
178
190
|
0x0C: 'complex', 0x0D: 'frozenset', 0x0E: 'datetime', 0x0F: 'date',
|
|
179
191
|
0x10: 'time', 0x11: 'timedelta', 0x12: 'Decimal', 0x13: 'Enum',
|
|
180
192
|
0x14: 'bytearray', 0x15: 'ref', 0x16: 'UUID', 0x17: 'Path',
|
|
181
|
-
0x18: 'tensor', 0x19: 'timedelta2',
|
|
193
|
+
0x18: 'tensor', 0x19: 'timedelta2', 0x1A: 'deque',
|
|
182
194
|
}
|
|
183
195
|
|
|
184
196
|
MAGIC = b'MSCS'
|
|
@@ -213,31 +225,48 @@ _SAFE_NUMPY_DTYPES: Set[str] = {
|
|
|
213
225
|
|
|
214
226
|
|
|
215
227
|
import re as _re
|
|
216
|
-
|
|
228
|
+
# Solo dtypes numéricos seguros: f(loat), i(nt), u(int), b(ool), c(omplex)
|
|
229
|
+
# NO incluir S(tring), U(nicode), V(oid) — estos permiten datos arbitrarios.
|
|
230
|
+
_RE_DTYPE_SHORT = _re.compile(r'[fiubc]\d+')
|
|
217
231
|
_RE_DTYPE_LONG = _re.compile(r'(int|uint|float|complex|bool)\d*_?')
|
|
218
232
|
|
|
219
233
|
|
|
234
|
+
# Dtypes no-numéricos: S(tring), U(nicode), V(oid) en notación corta.
|
|
235
|
+
# Solo las versiones MAYÚSCULAS y 'v' minúscula son peligrosas.
|
|
236
|
+
# 'u' minúscula es uint (u1=uint8, u2=uint16, etc.) — SEGURO.
|
|
237
|
+
# 's' minúscula no existe como dtype válido en numpy, pero no es peligroso.
|
|
238
|
+
_RE_UNSAFE_SHORTHAND = _re.compile(r'[<>=|!]?[SUV]\d+')
|
|
239
|
+
_RE_UNSAFE_VOID_LOW = _re.compile(r'[<>=|!]?v\d+')
|
|
240
|
+
|
|
241
|
+
|
|
220
242
|
def _is_safe_dtype(dtype_str: str) -> bool:
|
|
221
|
-
"""Valida que un dtype string sea seguro (no structured/object/void)."""
|
|
243
|
+
"""Valida que un dtype string sea seguro (no structured/object/void/string)."""
|
|
222
244
|
clean = dtype_str.strip()
|
|
223
245
|
# Rechazar explícitamente tipos peligrosos (case-insensitive)
|
|
224
|
-
|
|
246
|
+
low = clean.lower()
|
|
247
|
+
if low in ('object', 'o', 'void', 'v', 's', 'u'):
|
|
248
|
+
return False
|
|
249
|
+
# Rechazar string/unicode/void shorthand: S<n>, U<n>, V<n>
|
|
250
|
+
# (con o sin prefijo byteorder: <U8, >S16, |V32, etc.)
|
|
251
|
+
# Esto previene que un payload con dtype='U8' sea aceptado como
|
|
252
|
+
# 'u8' (uint64) tras lowercase pero interpretado como Unicode por numpy.
|
|
253
|
+
if _RE_UNSAFE_SHORTHAND.fullmatch(clean):
|
|
254
|
+
return False
|
|
255
|
+
# Rechazar void minúscula: v<n> (ej: v8)
|
|
256
|
+
if _RE_UNSAFE_VOID_LOW.fullmatch(clean):
|
|
225
257
|
return False
|
|
226
|
-
clean = clean.lower()
|
|
227
|
-
# Tipos simples directos
|
|
228
|
-
if clean in _SAFE_NUMPY_DTYPES:
|
|
229
|
-
return True
|
|
230
258
|
# Con prefijo de byteorder: <f4, >i8, =f8, |b1, etc.
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
259
|
+
stripped = low
|
|
260
|
+
if len(stripped) > 1 and stripped[0] in '<>=|!':
|
|
261
|
+
stripped = stripped[1:]
|
|
262
|
+
# Tipos simples directos (en minúsculas)
|
|
263
|
+
if low in _SAFE_NUMPY_DTYPES:
|
|
264
|
+
return True
|
|
265
|
+
# Numpy shorthand numérico: f4, f8, i4, i8, u2, b1, c8, c16
|
|
266
|
+
if _RE_DTYPE_SHORT.fullmatch(stripped):
|
|
238
267
|
return True
|
|
239
268
|
# Nombre completo con bitsize: float32, int64, etc.
|
|
240
|
-
if _RE_DTYPE_LONG.fullmatch(
|
|
269
|
+
if _RE_DTYPE_LONG.fullmatch(stripped):
|
|
241
270
|
return True
|
|
242
271
|
return False
|
|
243
272
|
|
|
@@ -490,6 +519,17 @@ class _Encoder:
|
|
|
490
519
|
|
|
491
520
|
# ── Colecciones (con ref tracking) ──
|
|
492
521
|
|
|
522
|
+
if isinstance(obj, deque):
|
|
523
|
+
if self._assign_ref(obj):
|
|
524
|
+
return
|
|
525
|
+
buf.write(_DEQUE)
|
|
526
|
+
maxlen = obj.maxlen
|
|
527
|
+
buf.write(struct.pack('<i', -1 if maxlen is None else maxlen))
|
|
528
|
+
self._write_length(len(obj))
|
|
529
|
+
for item in obj:
|
|
530
|
+
self.encode(item)
|
|
531
|
+
return
|
|
532
|
+
|
|
493
533
|
if isinstance(obj, list):
|
|
494
534
|
if self._assign_ref(obj):
|
|
495
535
|
return
|
|
@@ -589,7 +629,12 @@ class _Encoder:
|
|
|
589
629
|
cls_path = _class_key(type(obj))
|
|
590
630
|
self._encode_str(cls_path)
|
|
591
631
|
|
|
592
|
-
if
|
|
632
|
+
if '__getstate__' in type(obj).__dict__ or any(
|
|
633
|
+
'__getstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
|
|
634
|
+
if c is not object
|
|
635
|
+
):
|
|
636
|
+
state = obj.__getstate__()
|
|
637
|
+
elif dataclasses.is_dataclass(obj) and not isinstance(obj, type):
|
|
593
638
|
state = {f.name: getattr(obj, f.name) for f in dataclasses.fields(obj)}
|
|
594
639
|
elif hasattr(obj, '__slots__') and not hasattr(obj, '__dict__'):
|
|
595
640
|
state = {}
|
|
@@ -597,11 +642,6 @@ class _Encoder:
|
|
|
597
642
|
for s in getattr(cls, '__slots__', ()):
|
|
598
643
|
if hasattr(obj, s) and s not in state:
|
|
599
644
|
state[s] = getattr(obj, s)
|
|
600
|
-
elif '__getstate__' in type(obj).__dict__ or any(
|
|
601
|
-
'__getstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
|
|
602
|
-
if c is not object
|
|
603
|
-
):
|
|
604
|
-
state = obj.__getstate__()
|
|
605
645
|
elif hasattr(obj, '__dict__'):
|
|
606
646
|
state = obj.__dict__
|
|
607
647
|
else:
|
|
@@ -871,6 +911,30 @@ class _Decoder:
|
|
|
871
911
|
t = t.requires_grad_(True)
|
|
872
912
|
return self._store_ref(t)
|
|
873
913
|
|
|
914
|
+
if tag == _DEQUE:
|
|
915
|
+
maxlen_raw = struct.unpack('<i', self._read(4))[0]
|
|
916
|
+
if maxlen_raw < -1:
|
|
917
|
+
raise MSCDecodeError(
|
|
918
|
+
f"maxlen inválido para deque: {maxlen_raw} en {self._path_str()}"
|
|
919
|
+
)
|
|
920
|
+
maxlen = None if maxlen_raw == -1 else maxlen_raw
|
|
921
|
+
n = self._read_length()
|
|
922
|
+
# Prevenir CPU exhaustion: no decodear más items de los que
|
|
923
|
+
# el deque puede retener. Un payload con maxlen=1, count=10M
|
|
924
|
+
# forzaría decodear 10M items descartando 9,999,999.
|
|
925
|
+
if maxlen is not None and n > maxlen:
|
|
926
|
+
raise MSCDecodeError(
|
|
927
|
+
f"Deque count ({n:,}) excede maxlen ({maxlen:,}) "
|
|
928
|
+
f"en {self._path_str()} — posible payload adversarial"
|
|
929
|
+
)
|
|
930
|
+
result = deque(maxlen=maxlen)
|
|
931
|
+
self._store_ref(result)
|
|
932
|
+
for i in range(n):
|
|
933
|
+
self.path.append(f'[{i}]')
|
|
934
|
+
result.append(self.decode())
|
|
935
|
+
self.path.pop()
|
|
936
|
+
return result
|
|
937
|
+
|
|
874
938
|
if tag == _OBJ:
|
|
875
939
|
# Reserve ref slot BEFORE decoding children (matches encoder order)
|
|
876
940
|
ref_id = len(self.refs)
|
|
@@ -891,17 +955,17 @@ class _Decoder:
|
|
|
891
955
|
return fallback
|
|
892
956
|
|
|
893
957
|
obj = cls.__new__(cls)
|
|
894
|
-
if
|
|
958
|
+
if '__setstate__' in type(obj).__dict__ or any(
|
|
959
|
+
'__setstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
|
|
960
|
+
if c is not object
|
|
961
|
+
):
|
|
962
|
+
obj.__setstate__(state)
|
|
963
|
+
elif dataclasses.is_dataclass(cls):
|
|
895
964
|
for k, v in state.items():
|
|
896
965
|
setattr(obj, k, v)
|
|
897
966
|
elif hasattr(obj, '__slots__') and not hasattr(obj, '__dict__'):
|
|
898
967
|
for k, v in state.items():
|
|
899
968
|
setattr(obj, k, v)
|
|
900
|
-
elif '__setstate__' in type(obj).__dict__ or any(
|
|
901
|
-
'__setstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
|
|
902
|
-
if c is not object
|
|
903
|
-
):
|
|
904
|
-
obj.__setstate__(state)
|
|
905
969
|
elif hasattr(obj, '__dict__'):
|
|
906
970
|
obj.__dict__.update(state)
|
|
907
971
|
else:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|