mscs 2.3.0__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mscs
3
- Version: 2.3.0
3
+ Version: 2.4.0
4
4
  Summary: Safe, fast serialization for Python — a secure replacement for pickle with HMAC authentication, native numpy/PyTorch support.
5
5
  Project-URL: Homepage, https://github.com/ElEscribanoSilente/MSC-Serial
6
6
  Project-URL: Repository, https://github.com/ElEscribanoSilente/MSC-Serial
@@ -41,7 +41,7 @@ Description-Content-Type: text/markdown
41
41
 
42
42
  # MSCS — Safe Serialization for Python
43
43
 
44
- **v2.3.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
44
+ **v2.4.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
45
45
 
46
46
  > **Status: Beta** — API is stable but the format may evolve. Not yet battle-tested in large-scale production.
47
47
 
@@ -208,7 +208,7 @@ mscs.loads(unsigned_data, hmac_key=key) # MSCSecurityError: anti-downgrade prot
208
208
  |------|-------|
209
209
  | `None`, `bool`, `int`, `float`, `complex` | Ints up to 8192 bytes (~19,700 digits) |
210
210
  | `str`, `bytes`, `bytearray` | UTF-8, ref-tracked |
211
- | `list`, `tuple`, `dict`, `set`, `frozenset` | Circular refs supported |
211
+ | `list`, `tuple`, `dict`, `set`, `frozenset`, `deque` | Circular refs supported; `deque` preserves `maxlen` |
212
212
  | `datetime`, `date`, `time`, `timedelta` | ISO 8601 |
213
213
  | `Decimal`, `UUID`, `Path` | Lossless |
214
214
  | `Enum` | Must be registered |
@@ -303,6 +303,7 @@ CRC32 and HMAC are mutually exclusive (HMAC is strictly superior).
303
303
  | 0x17 | Path | `<I>` str len + UTF-8 path string |
304
304
  | 0x18 | Tensor | str(meta) + `<I>` data size + raw buffer |
305
305
  | 0x19 | timedelta2 | `<iiI>` days, seconds, microseconds |
306
+ | 0x1A | deque | `<i>` maxlen (-1 if None) + `<I>` item count + items |
306
307
 
307
308
  **ndarray meta**: `"{dtype}|{shape}"` where shape is `"dim0xdim1x..."` (e.g., `"float32|100x100"`).
308
309
 
@@ -1,6 +1,6 @@
1
1
  # MSCS — Safe Serialization for Python
2
2
 
3
- **v2.3.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
3
+ **v2.4.0** | [Changelog](CHANGELOG.md) | [PyPI](https://pypi.org/project/mscs/)
4
4
 
5
5
  > **Status: Beta** — API is stable but the format may evolve. Not yet battle-tested in large-scale production.
6
6
 
@@ -167,7 +167,7 @@ mscs.loads(unsigned_data, hmac_key=key) # MSCSecurityError: anti-downgrade prot
167
167
  |------|-------|
168
168
  | `None`, `bool`, `int`, `float`, `complex` | Ints up to 8192 bytes (~19,700 digits) |
169
169
  | `str`, `bytes`, `bytearray` | UTF-8, ref-tracked |
170
- | `list`, `tuple`, `dict`, `set`, `frozenset` | Circular refs supported |
170
+ | `list`, `tuple`, `dict`, `set`, `frozenset`, `deque` | Circular refs supported; `deque` preserves `maxlen` |
171
171
  | `datetime`, `date`, `time`, `timedelta` | ISO 8601 |
172
172
  | `Decimal`, `UUID`, `Path` | Lossless |
173
173
  | `Enum` | Must be registered |
@@ -262,6 +262,7 @@ CRC32 and HMAC are mutually exclusive (HMAC is strictly superior).
262
262
  | 0x17 | Path | `<I>` str len + UTF-8 path string |
263
263
  | 0x18 | Tensor | str(meta) + `<I>` data size + raw buffer |
264
264
  | 0x19 | timedelta2 | `<iiI>` days, seconds, microseconds |
265
+ | 0x1A | deque | `<i>` maxlen (-1 if None) + `<I>` item count + items |
265
266
 
266
267
  **ndarray meta**: `"{dtype}|{shape}"` where shape is `"dim0xdim1x..."` (e.g., `"float32|100x100"`).
267
268
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "mscs"
7
- version = "2.3.0"
7
+ version = "2.4.0"
8
8
  description = "Safe, fast serialization for Python — a secure replacement for pickle with HMAC authentication, native numpy/PyTorch support."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -1,11 +1,11 @@
1
1
  """
2
- MSC Serial v2.2
2
+ MSC Serial v2.4
3
3
  ===============
4
4
  Reemplazo personal y seguro de pickle.
5
5
 
6
- Soporta: dict, list, tuple, set, frozenset, str, int, float, complex,
7
- bool, None, bytes, bytearray, datetime, date, time, timedelta,
8
- Decimal, UUID, Path, Enum, numpy arrays, torch.Tensor,
6
+ Soporta: dict, list, tuple, set, frozenset, deque, str, int, float,
7
+ complex, bool, None, bytes, bytearray, datetime, date, time,
8
+ timedelta, Decimal, UUID, Path, Enum, numpy arrays, torch.Tensor,
9
9
  dataclasses, objetos con __slots__, objetos custom registrados,
10
10
  referencias circulares.
11
11
 
@@ -39,6 +39,16 @@ Seguridad:
39
39
  todos los objetos serializados, los IDs no se reutilizan durante
40
40
  una sola llamada a encode().
41
41
 
42
+ Changelog v2.4.0:
43
+ - FIX: __getstate__/__setstate__ ahora tiene prioridad sobre dataclass
44
+ field walking — antes, dataclasses que definían __getstate__ eran
45
+ serializados recorriendo fields directamente, lo que causaba
46
+ MSCEncodeError si algún field contenía tipos no soportados (ej. deque).
47
+ La prioridad ahora es: __getstate__/__setstate__ > dataclass fields > __slots__ > __dict__
48
+ - ADD: Soporte nativo collections.deque (tag 0x1A) — preserva maxlen
49
+ y soporta referencias circulares
50
+ - Retrocompatible con payloads v2.3, v2.2, v2.1, v2.0 y v1.0
51
+
42
52
  Changelog v2.3.0:
43
53
  - ADD: HMAC-SHA256 autenticación criptográfica (hmac_key= en dumps/loads)
44
54
  - ADD: Protección anti-downgrade (payload sin HMAC + clave = rechazado)
@@ -99,6 +109,7 @@ import hashlib
99
109
  import threading
100
110
  import inspect as _inspect_mod
101
111
  import dataclasses
112
+ from collections import deque
102
113
  from datetime import datetime, date, time, timedelta
103
114
  from decimal import Decimal
104
115
  from enum import Enum
@@ -118,7 +129,7 @@ try:
118
129
  except ImportError:
119
130
  _torch = None
120
131
 
121
- __version__ = "2.3.0"
132
+ __version__ = "2.4.0"
122
133
  __all__ = [
123
134
  "dump", "load", "dumps", "loads",
124
135
  "dump_compressed", "load_compressed",
@@ -170,6 +181,7 @@ _UUID = b'\x16'
170
181
  _PATH = b'\x17'
171
182
  _TENSOR = b'\x18'
172
183
  _TIMEDELTA2 = b'\x19' # v2.2: timedelta sin ambiguedad
184
+ _DEQUE = b'\x1A' # v2.4: collections.deque nativo
173
185
 
174
186
  _TAG_NAMES: Dict[int, str] = {
175
187
  0x00: 'None', 0x01: 'bool', 0x02: 'int', 0x03: 'float',
@@ -178,7 +190,7 @@ _TAG_NAMES: Dict[int, str] = {
178
190
  0x0C: 'complex', 0x0D: 'frozenset', 0x0E: 'datetime', 0x0F: 'date',
179
191
  0x10: 'time', 0x11: 'timedelta', 0x12: 'Decimal', 0x13: 'Enum',
180
192
  0x14: 'bytearray', 0x15: 'ref', 0x16: 'UUID', 0x17: 'Path',
181
- 0x18: 'tensor', 0x19: 'timedelta2',
193
+ 0x18: 'tensor', 0x19: 'timedelta2', 0x1A: 'deque',
182
194
  }
183
195
 
184
196
  MAGIC = b'MSCS'
@@ -213,31 +225,48 @@ _SAFE_NUMPY_DTYPES: Set[str] = {
213
225
 
214
226
 
215
227
  import re as _re
216
- _RE_DTYPE_SHORT = _re.compile(r'[fiubcUSV]\d+')
228
+ # Solo dtypes numéricos seguros: f(loat), i(nt), u(int), b(ool), c(omplex)
229
+ # NO incluir S(tring), U(nicode), V(oid) — estos permiten datos arbitrarios.
230
+ _RE_DTYPE_SHORT = _re.compile(r'[fiubc]\d+')
217
231
  _RE_DTYPE_LONG = _re.compile(r'(int|uint|float|complex|bool)\d*_?')
218
232
 
219
233
 
234
+ # Dtypes no-numéricos: S(tring), U(nicode), V(oid) en notación corta.
235
+ # Solo las versiones MAYÚSCULAS y 'v' minúscula son peligrosas.
236
+ # 'u' minúscula es uint (u1=uint8, u2=uint16, etc.) — SEGURO.
237
+ # 's' minúscula no existe como dtype válido en numpy, pero no es peligroso.
238
+ _RE_UNSAFE_SHORTHAND = _re.compile(r'[<>=|!]?[SUV]\d+')
239
+ _RE_UNSAFE_VOID_LOW = _re.compile(r'[<>=|!]?v\d+')
240
+
241
+
220
242
  def _is_safe_dtype(dtype_str: str) -> bool:
221
- """Valida que un dtype string sea seguro (no structured/object/void)."""
243
+ """Valida que un dtype string sea seguro (no structured/object/void/string)."""
222
244
  clean = dtype_str.strip()
223
245
  # Rechazar explícitamente tipos peligrosos (case-insensitive)
224
- if clean.lower() in ('object', 'o', 'void', 'v'):
246
+ low = clean.lower()
247
+ if low in ('object', 'o', 'void', 'v', 's', 'u'):
248
+ return False
249
+ # Rechazar string/unicode/void shorthand: S<n>, U<n>, V<n>
250
+ # (con o sin prefijo byteorder: <U8, >S16, |V32, etc.)
251
+ # Esto previene que un payload con dtype='U8' sea aceptado como
252
+ # 'u8' (uint64) tras lowercase pero interpretado como Unicode por numpy.
253
+ if _RE_UNSAFE_SHORTHAND.fullmatch(clean):
254
+ return False
255
+ # Rechazar void minúscula: v<n> (ej: v8)
256
+ if _RE_UNSAFE_VOID_LOW.fullmatch(clean):
225
257
  return False
226
- clean = clean.lower()
227
- # Tipos simples directos
228
- if clean in _SAFE_NUMPY_DTYPES:
229
- return True
230
258
  # Con prefijo de byteorder: <f4, >i8, =f8, |b1, etc.
231
- if len(clean) > 1 and clean[0] in '<>=|!':
232
- clean = clean[1:]
233
- # Numpy shorthand: f4, f8, i4, i8, u2, b1, c8, c16, etc.
234
- if _RE_DTYPE_SHORT.fullmatch(clean):
235
- # Rechazar V (void) — ya cubierto arriba
236
- if clean[0] == 'V':
237
- return False
259
+ stripped = low
260
+ if len(stripped) > 1 and stripped[0] in '<>=|!':
261
+ stripped = stripped[1:]
262
+ # Tipos simples directos (en minúsculas)
263
+ if low in _SAFE_NUMPY_DTYPES:
264
+ return True
265
+ # Numpy shorthand numérico: f4, f8, i4, i8, u2, b1, c8, c16
266
+ if _RE_DTYPE_SHORT.fullmatch(stripped):
238
267
  return True
239
268
  # Nombre completo con bitsize: float32, int64, etc.
240
- if _RE_DTYPE_LONG.fullmatch(clean):
269
+ if _RE_DTYPE_LONG.fullmatch(stripped):
241
270
  return True
242
271
  return False
243
272
 
@@ -490,6 +519,17 @@ class _Encoder:
490
519
 
491
520
  # ── Colecciones (con ref tracking) ──
492
521
 
522
+ if isinstance(obj, deque):
523
+ if self._assign_ref(obj):
524
+ return
525
+ buf.write(_DEQUE)
526
+ maxlen = obj.maxlen
527
+ buf.write(struct.pack('<i', -1 if maxlen is None else maxlen))
528
+ self._write_length(len(obj))
529
+ for item in obj:
530
+ self.encode(item)
531
+ return
532
+
493
533
  if isinstance(obj, list):
494
534
  if self._assign_ref(obj):
495
535
  return
@@ -589,7 +629,12 @@ class _Encoder:
589
629
  cls_path = _class_key(type(obj))
590
630
  self._encode_str(cls_path)
591
631
 
592
- if dataclasses.is_dataclass(obj) and not isinstance(obj, type):
632
+ if '__getstate__' in type(obj).__dict__ or any(
633
+ '__getstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
634
+ if c is not object
635
+ ):
636
+ state = obj.__getstate__()
637
+ elif dataclasses.is_dataclass(obj) and not isinstance(obj, type):
593
638
  state = {f.name: getattr(obj, f.name) for f in dataclasses.fields(obj)}
594
639
  elif hasattr(obj, '__slots__') and not hasattr(obj, '__dict__'):
595
640
  state = {}
@@ -597,11 +642,6 @@ class _Encoder:
597
642
  for s in getattr(cls, '__slots__', ()):
598
643
  if hasattr(obj, s) and s not in state:
599
644
  state[s] = getattr(obj, s)
600
- elif '__getstate__' in type(obj).__dict__ or any(
601
- '__getstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
602
- if c is not object
603
- ):
604
- state = obj.__getstate__()
605
645
  elif hasattr(obj, '__dict__'):
606
646
  state = obj.__dict__
607
647
  else:
@@ -871,6 +911,30 @@ class _Decoder:
871
911
  t = t.requires_grad_(True)
872
912
  return self._store_ref(t)
873
913
 
914
+ if tag == _DEQUE:
915
+ maxlen_raw = struct.unpack('<i', self._read(4))[0]
916
+ if maxlen_raw < -1:
917
+ raise MSCDecodeError(
918
+ f"maxlen inválido para deque: {maxlen_raw} en {self._path_str()}"
919
+ )
920
+ maxlen = None if maxlen_raw == -1 else maxlen_raw
921
+ n = self._read_length()
922
+ # Prevenir CPU exhaustion: no decodear más items de los que
923
+ # el deque puede retener. Un payload con maxlen=1, count=10M
924
+ # forzaría decodear 10M items descartando 9,999,999.
925
+ if maxlen is not None and n > maxlen:
926
+ raise MSCDecodeError(
927
+ f"Deque count ({n:,}) excede maxlen ({maxlen:,}) "
928
+ f"en {self._path_str()} — posible payload adversarial"
929
+ )
930
+ result = deque(maxlen=maxlen)
931
+ self._store_ref(result)
932
+ for i in range(n):
933
+ self.path.append(f'[{i}]')
934
+ result.append(self.decode())
935
+ self.path.pop()
936
+ return result
937
+
874
938
  if tag == _OBJ:
875
939
  # Reserve ref slot BEFORE decoding children (matches encoder order)
876
940
  ref_id = len(self.refs)
@@ -891,17 +955,17 @@ class _Decoder:
891
955
  return fallback
892
956
 
893
957
  obj = cls.__new__(cls)
894
- if dataclasses.is_dataclass(cls):
958
+ if '__setstate__' in type(obj).__dict__ or any(
959
+ '__setstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
960
+ if c is not object
961
+ ):
962
+ obj.__setstate__(state)
963
+ elif dataclasses.is_dataclass(cls):
895
964
  for k, v in state.items():
896
965
  setattr(obj, k, v)
897
966
  elif hasattr(obj, '__slots__') and not hasattr(obj, '__dict__'):
898
967
  for k, v in state.items():
899
968
  setattr(obj, k, v)
900
- elif '__setstate__' in type(obj).__dict__ or any(
901
- '__setstate__' in c.__dict__ for c in type(obj).__mro__[:-1]
902
- if c is not object
903
- ):
904
- obj.__setstate__(state)
905
969
  elif hasattr(obj, '__dict__'):
906
970
  obj.__dict__.update(state)
907
971
  else:
File without changes
File without changes
File without changes
File without changes