simscope 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. simscope/__init__.py +6 -0
  2. simscope/__main__.py +8 -0
  3. simscope/_assets/simscope-app.css +2 -0
  4. simscope/_assets/simscope-app.js +4311 -0
  5. simscope/_assets/simscope-player.js +4325 -0
  6. simscope/_assets/simscope-web.LICENSES.txt +407 -0
  7. simscope/_icon.py +22 -0
  8. simscope/_mjviser.py +203 -0
  9. simscope/annotations.py +1132 -0
  10. simscope/cli.py +482 -0
  11. simscope/core.py +257 -0
  12. simscope/derived.py +697 -0
  13. simscope/export.py +799 -0
  14. simscope/highlights.py +947 -0
  15. simscope/importers.py +874 -0
  16. simscope/index.py +579 -0
  17. simscope/io/__init__.py +45 -0
  18. simscope/io/blockfile.py +938 -0
  19. simscope/io/cas.py +294 -0
  20. simscope/io/codecs.py +566 -0
  21. simscope/io/errors.py +9 -0
  22. simscope/io/manifest.py +358 -0
  23. simscope/io/pack.py +563 -0
  24. simscope/io/scene.py +239 -0
  25. simscope/isaaclab.py +1460 -0
  26. simscope/library.py +705 -0
  27. simscope/mujoco.py +578 -0
  28. simscope/py.typed +0 -0
  29. simscope/recorder.py +784 -0
  30. simscope/server/__init__.py +9 -0
  31. simscope/server/app.py +149 -0
  32. simscope/server/blocks.py +191 -0
  33. simscope/server/jobs.py +166 -0
  34. simscope/server/routes.py +707 -0
  35. simscope/server/security.py +218 -0
  36. simscope/server/state.py +751 -0
  37. simscope/server/static.py +84 -0
  38. simscope/transforms.py +147 -0
  39. simscope-0.1.1.dist-info/METADATA +132 -0
  40. simscope-0.1.1.dist-info/RECORD +45 -0
  41. simscope-0.1.1.dist-info/WHEEL +4 -0
  42. simscope-0.1.1.dist-info/entry_points.txt +3 -0
  43. simscope-0.1.1.dist-info/licenses/LICENSE.md +201 -0
  44. simscope-0.1.1.dist-info/licenses/THIRD_PARTY_NOTICES.md +267 -0
  45. simscope-0.1.1.dist-info/licenses/src/simscope/_assets/simscope-web.LICENSES.txt +407 -0
@@ -0,0 +1,938 @@
1
+ """Block files (``*.blk``): writer, reader and crash recovery."""
2
+
3
+ import collections
4
+ import dataclasses
5
+ import math
6
+ import mmap
7
+ import os
8
+ import pathlib
9
+ import struct
10
+ import threading
11
+ import zlib
12
+ from collections.abc import Sequence
13
+ from types import TracebackType
14
+ from typing import BinaryIO
15
+
16
+ import numpy as np
17
+ import numpy.typing as npt
18
+
19
+ from simscope import core, transforms
20
+ from simscope.io import codecs, errors
21
+
22
+ FILE_MAGIC = b"SSBK"
23
+ BLOCK_MAGIC = b"SSBB"
24
+ MAJOR = 1
25
+ MINOR = 0
26
+ HEADER = struct.Struct("<4sHHII4IIIIIQII")
27
+ BLOCK_HEADER = struct.Struct("<4sB3xIIIIII")
28
+ DIR_DTYPE = np.dtype(
29
+ [
30
+ ("offset", "<u8"),
31
+ ("env", "<u4"),
32
+ ("t0", "<u4"),
33
+ ("n", "<u4"),
34
+ ("clen", "<u4"),
35
+ ("ulen", "<u4"),
36
+ ("codec", "u1"),
37
+ ("reserved", "V3"),
38
+ ]
39
+ )
40
+ _DIR_FIELDS = ("offset", "env", "t0", "n", "clen", "ulen", "codec")
41
+ HEADER_SIZE = 64
42
+ DEFAULT_BLOCK_FRAMES = 100
43
+ assert HEADER.size == HEADER_SIZE
44
+ assert BLOCK_HEADER.size == 32
45
+ assert DIR_DTYPE.itemsize == 32
46
+
47
+ _MAX_NDIM = 4
48
+ _PAD = bytes(8)
49
+
50
+
51
+ def _align8(offset: int) -> int:
52
+ """Rounds ``offset`` up to a multiple of 8."""
53
+ return (offset + 7) & ~7
54
+
55
+
56
+ @dataclasses.dataclass(frozen=True)
57
+ class Header:
58
+ """Parsed block-file header.
59
+
60
+ Attributes:
61
+ item_shape: Shape of one env-frame item.
62
+ n_envs: Number of envs.
63
+ n_frames: Total frames, 0 while recording.
64
+ block_frames: Nominal frames per block.
65
+ n_blocks: Number of blocks, 0 while recording.
66
+ dir_offset: Directory offset, 0 while recording.
67
+ dir_length: Directory length in bytes.
68
+ """
69
+
70
+ item_shape: tuple[int, ...]
71
+ n_envs: int
72
+ n_frames: int
73
+ block_frames: int
74
+ n_blocks: int
75
+ dir_offset: int
76
+ dir_length: int
77
+
78
+ @property
79
+ def k(self) -> int:
80
+ """Floats per env per frame."""
81
+ return math.prod(self.item_shape)
82
+
83
+ def pack(self) -> bytes:
84
+ """Serializes the header, including its CRC."""
85
+ shape = [*self.item_shape, 0, 0, 0, 0][:_MAX_NDIM]
86
+ head = HEADER.pack(
87
+ FILE_MAGIC,
88
+ MAJOR,
89
+ MINOR,
90
+ HEADER_SIZE,
91
+ len(self.item_shape),
92
+ *shape,
93
+ self.n_envs,
94
+ self.n_frames,
95
+ self.block_frames,
96
+ self.n_blocks,
97
+ self.dir_offset,
98
+ self.dir_length,
99
+ 0,
100
+ )
101
+ return head[:60] + struct.pack("<I", zlib.crc32(head[:60]))
102
+
103
+ @classmethod
104
+ def parse(cls, buf: bytes | memoryview) -> "Header":
105
+ """Parses and validates a header.
106
+
107
+ Args:
108
+ buf: At least 64 bytes from the start of a block file.
109
+
110
+ Returns:
111
+ The header.
112
+
113
+ Raises:
114
+ errors.FormatError: On a short file, bad magic, unknown major
115
+ version, CRC mismatch, or inconsistent fields.
116
+ """
117
+ if len(buf) < HEADER_SIZE:
118
+ raise errors.FormatError("block file shorter than its header")
119
+ fields = HEADER.unpack_from(buf, 0)
120
+ magic, major = fields[0], fields[1]
121
+ if magic != FILE_MAGIC:
122
+ raise errors.FormatError(f"bad block file magic {magic!r}")
123
+ if major != MAJOR:
124
+ raise errors.FormatError(
125
+ f"unknown block file major version {major}"
126
+ )
127
+ if zlib.crc32(buf[:60]) != fields[-1]:
128
+ raise errors.FormatError("block file header CRC mismatch")
129
+ header_size, ndim = fields[3], fields[4]
130
+ if header_size != HEADER_SIZE or ndim > _MAX_NDIM:
131
+ raise errors.FormatError("invalid block file header fields")
132
+ shape = tuple(fields[5 : 5 + ndim])
133
+ n_envs, n_frames, block_frames, n_blocks = fields[9:13]
134
+ if 0 in shape or n_envs < 1 or block_frames < 1:
135
+ raise errors.FormatError("invalid block file header fields")
136
+ return cls(
137
+ shape,
138
+ n_envs,
139
+ n_frames,
140
+ block_frames,
141
+ n_blocks,
142
+ fields[13],
143
+ fields[14],
144
+ )
145
+
146
+
147
+ @dataclasses.dataclass(frozen=True)
148
+ class EncodedBlock:
149
+ """One encoded block, ready to be written.
150
+
151
+ Attributes:
152
+ env: Env index.
153
+ n: Frames in the block.
154
+ codec: Codec id actually used.
155
+ ulen: Uncompressed payload length.
156
+ payload: Compressed payload.
157
+ """
158
+
159
+ env: int
160
+ n: int
161
+ codec: int
162
+ ulen: int
163
+ payload: bytes
164
+
165
+
166
+ def encode_window(
167
+ window: npt.NDArray[np.float32], codec: str | int = "f32s"
168
+ ) -> list[EncodedBlock]:
169
+ """Encodes one time window into one block per env (pure, no file I/O).
170
+
171
+ Args:
172
+ window: Float32 array ``[n, n_envs, K]`` for frames ``t0..t0+n-1``.
173
+ codec: Requested codec. q16d falls back to f32s per block when the
174
+ block has non-finite values.
175
+
176
+ Returns:
177
+ Encoded blocks in env order.
178
+ """
179
+ n, n_envs, k = window.shape
180
+ blocks = []
181
+ for env in range(n_envs):
182
+ cid, payload = codecs.encode_block_auto(window[:, env, :], codec)
183
+ blocks.append(
184
+ EncodedBlock(env, n, cid, codecs.payload_ulen(cid, n, k), payload)
185
+ )
186
+ return blocks
187
+
188
+
189
+ DirEntry = tuple[int, int, int, int, int, int, int]
190
+
191
+
192
+ def pack_window(
193
+ t0: int, blocks: Sequence[EncodedBlock], offset: int
194
+ ) -> tuple[bytes, list[DirEntry]]:
195
+ """Serializes encoded blocks with headers and 8-byte alignment.
196
+
197
+ Args:
198
+ t0: First frame index of the window.
199
+ blocks: Encoded blocks of the window, in env order.
200
+ offset: File offset where the first block will start (aligned to 8).
201
+
202
+ Returns:
203
+ ``(data, entries)``: the bytes to write at ``offset`` (each block
204
+ padded to 8 bytes) and the directory entries
205
+ ``(offset, env, t0, n, clen, ulen, codec)`` of the blocks.
206
+ """
207
+ parts = []
208
+ entries: list[DirEntry] = []
209
+ pos = offset
210
+ for b in blocks:
211
+ clen = len(b.payload)
212
+ pad = -(32 + clen) % 8
213
+ head = BLOCK_HEADER.pack(
214
+ BLOCK_MAGIC,
215
+ b.codec,
216
+ b.env,
217
+ t0,
218
+ b.n,
219
+ clen,
220
+ b.ulen,
221
+ zlib.crc32(b.payload),
222
+ )
223
+ parts += [head, b.payload, _PAD[:pad]]
224
+ entries.append((pos, b.env, t0, b.n, clen, b.ulen, b.codec))
225
+ pos += 32 + clen + pad
226
+ return b"".join(parts), entries
227
+
228
+
229
+ def _dir_bytes(entries: Sequence[DirEntry]) -> bytes:
230
+ """Packs directory entries into the 32-byte on-disk records."""
231
+ arr = np.zeros(len(entries), DIR_DTYPE)
232
+ if entries:
233
+ cols = np.array(entries, dtype=np.uint64).T
234
+ for name, col in zip(_DIR_FIELDS, cols, strict=True):
235
+ arr[name] = col
236
+ return arr.tobytes()
237
+
238
+
239
+ class BlockWriter:
240
+ """Writes one stream as a block file.
241
+
242
+ ``append`` copies frames into a preallocated ``[block_frames, n_envs, K]``
243
+ buffer. Each time it fills, the window is encoded (one block per env) and
244
+ written. Leaving the ``with`` block normally calls :meth:`finalize`; on an
245
+ exception the file is closed unfinished so :func:`recover` can salvage it.
246
+
247
+ Attributes:
248
+ path: Destination path.
249
+ item_shape: Shape of one env-frame item.
250
+ n_envs: Number of envs.
251
+ kind: Stream kind (``"pose"`` streams get sign continuity).
252
+ codec: Requested codec name.
253
+ block_frames: Frames per block.
254
+ """
255
+
256
+ def __init__(
257
+ self,
258
+ path: os.PathLike[str] | str,
259
+ *,
260
+ item_shape: Sequence[int],
261
+ n_envs: int,
262
+ kind: core.StreamKind,
263
+ codec: str = "f32s",
264
+ block_frames: int = DEFAULT_BLOCK_FRAMES,
265
+ ) -> None:
266
+ """Creates the file and writes a provisional header.
267
+
268
+ Args:
269
+ path: Destination path (overwritten).
270
+ item_shape: Shape of one env-frame item, at most 4 dims, all > 0.
271
+ n_envs: Number of envs (at least 1).
272
+ kind: Stream kind. ``"pose"`` requires a last dim of 7.
273
+ codec: ``"f32s"`` (lossless) or ``"q16d"``.
274
+ block_frames: Frames per block (at least 1).
275
+
276
+ Raises:
277
+ ValueError: If an argument is invalid.
278
+ errors.FormatError: If the codec is unknown.
279
+ """
280
+ self.path = pathlib.Path(path)
281
+ self.item_shape = tuple(int(d) for d in item_shape)
282
+ self.n_envs = int(n_envs)
283
+ self.kind = kind
284
+ self.block_frames = int(block_frames)
285
+ self._codec_id = codecs.codec_id(codec)
286
+ self.codec = codecs.BLOCK_CODEC_NAMES[self._codec_id]
287
+ if len(self.item_shape) > _MAX_NDIM or 0 in self.item_shape:
288
+ raise ValueError(f"invalid item_shape {self.item_shape}")
289
+ if self.n_envs < 1 or self.block_frames < 1:
290
+ raise ValueError("n_envs and block_frames must be at least 1")
291
+ if kind == "pose" and self.item_shape[-1:] != (core.POSE_DIM,):
292
+ raise ValueError("pose streams need a last item dim of 7")
293
+ self._k = math.prod(self.item_shape)
294
+ self._buf = np.empty(
295
+ (self.block_frames, self.n_envs, self._k), np.float32
296
+ )
297
+ self._fill = 0
298
+ self._n_frames = 0
299
+ self._prev_quat: npt.NDArray[np.float32] | None = None
300
+ self._entries: list[DirEntry] = []
301
+ self._offset = HEADER_SIZE
302
+ self._finalized = False
303
+ self._file: BinaryIO | None = open(self.path, "wb") # noqa: SIM115
304
+ self._file.write(self._header(0, 0, 0, 0).pack())
305
+ self._file.flush() # tailing readers see the header at once
306
+
307
+ def _header(
308
+ self, n_frames: int, n_blocks: int, dir_offset: int, dir_len: int
309
+ ) -> Header:
310
+ """Builds a header for the current stream parameters."""
311
+ return Header(
312
+ self.item_shape,
313
+ self.n_envs,
314
+ n_frames,
315
+ self.block_frames,
316
+ n_blocks,
317
+ dir_offset,
318
+ dir_len,
319
+ )
320
+
321
+ @property
322
+ def n_frames(self) -> int:
323
+ """Frames appended so far, including the unflushed window."""
324
+ return self._n_frames + self._fill
325
+
326
+ def append(self, frames: npt.ArrayLike) -> None:
327
+ """Appends frames to the stream.
328
+
329
+ Args:
330
+ frames: Array ``[n, n_envs, *item_shape]`` with n at least 1.
331
+ It is copied; the caller may reuse it.
332
+
333
+ Raises:
334
+ ValueError: If the shape is wrong or the writer is finalized.
335
+ """
336
+ if self._file is None or self._finalized:
337
+ raise ValueError("writer is closed")
338
+ a = np.asarray(frames, dtype=np.float32)
339
+ want = (self.n_envs, *self.item_shape)
340
+ if a.ndim != len(want) + 1 or a.shape[1:] != want or a.shape[0] < 1:
341
+ raise ValueError(
342
+ f"expected frames of shape [n>=1, {', '.join(map(str, want))}]"
343
+ f", got {a.shape}"
344
+ )
345
+ a = a.reshape(a.shape[0], self.n_envs, self._k)
346
+ pos = 0
347
+ while pos < a.shape[0]:
348
+ take = min(a.shape[0] - pos, self.block_frames - self._fill)
349
+ dst = self._buf[self._fill : self._fill + take]
350
+ dst[...] = a[pos : pos + take]
351
+ if self.kind == "pose":
352
+ self._fix_signs(dst)
353
+ self._fill += take
354
+ pos += take
355
+ if self._fill == self.block_frames:
356
+ self._flush()
357
+
358
+ def _fix_signs(self, chunk: npt.NDArray[np.float32]) -> None:
359
+ """Enforces quaternion sign continuity on a buffer chunk in place."""
360
+ n = chunk.shape[0]
361
+ q = chunk.reshape(n, self.n_envs, -1, core.POSE_DIM)[..., 3:]
362
+ fixed = transforms.enforce_sign_continuity(q, self._prev_quat)
363
+ q[...] = fixed
364
+ self._prev_quat = fixed[-1].copy()
365
+
366
+ def _flush(self) -> None:
367
+ """Encodes and writes the buffered window."""
368
+ if self._fill == 0 or self._file is None:
369
+ return
370
+ blocks = encode_window(self._buf[: self._fill], self._codec_id)
371
+ self.write_encoded(self._n_frames, blocks) # advances _n_frames
372
+ self._fill = 0
373
+
374
+ def write_encoded(self, t0: int, blocks: Sequence[EncodedBlock]) -> None:
375
+ """Writes already-encoded blocks of one window.
376
+
377
+ Advances the frame count, so ``finalize`` records it.
378
+
379
+ Args:
380
+ t0: First frame index of the window.
381
+ blocks: One encoded block per env, in env order.
382
+ """
383
+ assert self._file is not None
384
+ data, entries = pack_window(t0, blocks, self._offset)
385
+ self._file.write(data)
386
+ self._file.flush() # a tailing reader sees the window right away
387
+ self._entries += entries
388
+ self._offset += len(data)
389
+ if blocks:
390
+ self._n_frames = max(self._n_frames, t0 + blocks[0].n)
391
+
392
+ def finalize(self) -> None:
393
+ """Flushes the last window, writes the directory and the header."""
394
+ if self._finalized or self._file is None:
395
+ return
396
+ self._flush()
397
+ directory = _dir_bytes(self._entries)
398
+ self._file.write(directory)
399
+ header = self._header(
400
+ self._n_frames, len(self._entries), self._offset, len(directory)
401
+ )
402
+ self._file.seek(0)
403
+ self._file.write(header.pack())
404
+ self._file.close()
405
+ self._file = None
406
+ self._finalized = True
407
+
408
+ def close(self) -> None:
409
+ """Closes the file without finalizing (the file stays recoverable)."""
410
+ if self._file is not None:
411
+ self._file.close()
412
+ self._file = None
413
+
414
+ def __enter__(self) -> "BlockWriter":
415
+ """Returns the writer."""
416
+ return self
417
+
418
+ def __exit__(
419
+ self,
420
+ exc_type: type[BaseException] | None,
421
+ exc: BaseException | None,
422
+ tb: TracebackType | None,
423
+ ) -> None:
424
+ """Finalizes on success; on error closes and leaves it unfinished."""
425
+ if exc_type is None:
426
+ self.finalize()
427
+ else:
428
+ self.close()
429
+
430
+
431
+ def _renormalize_poses(block: npt.NDArray[np.float32]) -> None:
432
+ """Renormalizes the quaternion of every pose in a ``[n, K]`` block."""
433
+ q = block.reshape(block.shape[0], -1, core.POSE_DIM)[..., 3:]
434
+ norm = np.sqrt(np.sum(q * q, axis=-1, keepdims=True))
435
+ np.divide(q, norm, out=q, where=norm > 0)
436
+
437
+
438
+ class BlockReader:
439
+ """Random-access reader for a block file.
440
+
441
+ The file is memory-mapped. ``read`` decodes only the blocks that overlap
442
+ the requested window and keeps the most recent decoded blocks in an LRU
443
+ cache. Instances are safe to share between threads.
444
+
445
+ With ``partial=True`` the reader can also tail a file that a
446
+ :class:`BlockWriter` is still writing: it builds the
447
+ directory by scanning blocks and :meth:`refresh` picks up windows
448
+ appended later. It never writes to the file, and only complete windows
449
+ whose CRCs check out are visible, so a window the writer is in the middle
450
+ of writing is never returned.
451
+
452
+ Attributes:
453
+ n_frames: Total frames (the complete windows found so far, for a
454
+ file that is still being written).
455
+ n_envs: Number of envs.
456
+ item_shape: Shape of one env-frame item.
457
+ block_frames: Nominal frames per block.
458
+ n_blocks: Number of blocks.
459
+ directory: Structured array of directory entries (``DIR_DTYPE``).
460
+ """
461
+
462
+ def __init__(
463
+ self,
464
+ source: os.PathLike[str] | str | bytes | bytearray | memoryview,
465
+ *,
466
+ cache_blocks: int = 64,
467
+ kind: core.StreamKind | None = None,
468
+ verify: bool = True,
469
+ partial: bool = False,
470
+ ) -> None:
471
+ """Opens a block file or an in-memory block file.
472
+
473
+ Args:
474
+ source: A path, or the file bytes (for example a pack entry).
475
+ cache_blocks: Decoded blocks to keep in the LRU cache.
476
+ kind: The stream kind from the manifest. For ``"pose"`` streams,
477
+ quaternions of q16d blocks are renormalized.
478
+ verify: Check each block's payload CRC when it is decoded.
479
+ partial: Accept an unfinished file (``dir_offset == 0``) by
480
+ scanning its blocks. A finished file opens as usual.
481
+
482
+ Raises:
483
+ errors.FormatError: If the file is unfinished and ``partial`` is
484
+ false (use :func:`recover`), or is corrupt, or uses an
485
+ unknown version or codec.
486
+ """
487
+ self._mm: mmap.mmap | None = None
488
+ self._view: memoryview | None = None
489
+ self._path: pathlib.Path | None = None
490
+ self._verify = verify
491
+ self._kind = kind
492
+ self._cache_blocks = max(0, int(cache_blocks))
493
+ self._cache: collections.OrderedDict[int, npt.NDArray[np.float32]] = (
494
+ collections.OrderedDict()
495
+ )
496
+ self._lock = threading.Lock()
497
+ self._refresh_lock = threading.Lock()
498
+ self._scan: _ScanState | None = None
499
+ self._head: Header | None = None
500
+ self._dir_buf = np.empty(0, DIR_DTYPE)
501
+ if isinstance(source, str | os.PathLike):
502
+ self._path = pathlib.Path(source)
503
+ with open(source, "rb") as f:
504
+ if os.fstat(f.fileno()).st_size < HEADER_SIZE:
505
+ raise errors.FormatError(f"{source}: shorter than header")
506
+ self._mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
507
+ self._view = memoryview(self._mm)
508
+ else:
509
+ self._view = memoryview(source).cast("B")
510
+ try:
511
+ self._init_from_view(self._view, partial)
512
+ except BaseException:
513
+ self.close()
514
+ raise
515
+
516
+ @property
517
+ def finished(self) -> bool:
518
+ """True if the file has its directory (it will not grow)."""
519
+ return self._scan is None
520
+
521
+ def _init_from_view(self, view: memoryview, partial: bool) -> None:
522
+ """Parses the header and directory and validates them."""
523
+ h = Header.parse(view)
524
+ self.item_shape = h.item_shape
525
+ self.n_envs = h.n_envs
526
+ self.block_frames = h.block_frames
527
+ self._k = h.k
528
+ if h.dir_offset == 0:
529
+ if not partial:
530
+ raise errors.FormatError(
531
+ "block file is unfinished (dir_offset == 0); "
532
+ "run simscope.io.blockfile.recover() first"
533
+ )
534
+ self._head = h
535
+ self._scan = _ScanState()
536
+ self.n_frames = 0
537
+ self.n_blocks = 0
538
+ self.directory = self._dir_buf
539
+ self._scan_more(view)
540
+ return
541
+ self._install_finished(view, h)
542
+
543
+ def _install_finished(self, view: memoryview, h: Header) -> None:
544
+ """Validates the directory of a finished file and adopts it.
545
+
546
+ The directory is published before the counts, so a concurrent
547
+ ``read`` never sees a count that its directory does not cover.
548
+ """
549
+ end = h.dir_offset + h.dir_length
550
+ if h.dir_length != 32 * h.n_blocks or end > len(view):
551
+ raise errors.FormatError("block directory is truncated or invalid")
552
+ d = np.frombuffer(view, DIR_DTYPE, h.n_blocks, h.dir_offset).copy()
553
+ self._check_directory(d, len(view), h.n_frames, h.n_blocks)
554
+ self.directory = d
555
+ self.n_blocks = h.n_blocks
556
+ self.n_frames = h.n_frames
557
+ self._scan = None
558
+ self._head = None
559
+ self._dir_buf = np.empty(0, DIR_DTYPE)
560
+
561
+ def _scan_more(self, view: memoryview) -> None:
562
+ """Scans blocks appended since the last scan and publishes them."""
563
+ st, h = self._scan, self._head
564
+ assert st is not None and h is not None
565
+ _advance(view, h, st)
566
+ if not st.kept:
567
+ return
568
+ new = np.frombuffer(_dir_bytes(st.kept), DIR_DTYPE)
569
+ st.kept = []
570
+ n = self.n_blocks
571
+ if n + len(new) > len(self._dir_buf):
572
+ grown = np.empty(max(64, 2 * (n + len(new))), DIR_DTYPE)
573
+ grown[:n] = self._dir_buf[:n]
574
+ self._dir_buf = grown
575
+ self._dir_buf[n : n + len(new)] = new
576
+ self.directory = self._dir_buf[: n + len(new)]
577
+ self.n_blocks = n + len(new)
578
+ self.n_frames = st.frames
579
+
580
+ def refresh(self) -> int:
581
+ """Picks up windows appended to a file that is still being written.
582
+
583
+ Re-reads the header and scans only the bytes after the last known
584
+ good window, so the cost is proportional to the new data. When the
585
+ writer has finalized the file, the reader switches to the finished
586
+ directory. A finished file, or an in-memory source, is left as is.
587
+
588
+ Returns:
589
+ The number of frames now readable.
590
+
591
+ Raises:
592
+ errors.FormatError: If the file shrank, or a finished file has
593
+ an invalid directory.
594
+ ValueError: If the reader is closed.
595
+ """
596
+ if self._view is None:
597
+ raise ValueError("reader is closed")
598
+ if self._scan is None or self._path is None:
599
+ return self.n_frames
600
+ with self._refresh_lock:
601
+ if self._scan is None:
602
+ return self.n_frames
603
+ head, size = self._read_tail_state(self._path)
604
+ if size < self._scan.end:
605
+ raise errors.FormatError(f"{self._path}: file shrank")
606
+ if size != len(self._view):
607
+ # Also when it shrank (a concurrent recover): never touch
608
+ # mapped pages past the end of the file.
609
+ self._remap()
610
+ view = self._view
611
+ assert view is not None
612
+ if head is not None and head.dir_offset != 0:
613
+ self._install_finished(view, head)
614
+ else:
615
+ self._scan_more(view)
616
+ return self.n_frames
617
+
618
+ def _read_tail_state(self, path: pathlib.Path) -> tuple[Header | None, int]:
619
+ """Reads the header, then the size (in that order, see refresh).
620
+
621
+ Returns:
622
+ ``(header, size)``. The header is ``None`` if it cannot be
623
+ parsed yet (a torn write); the next refresh retries.
624
+ """
625
+ with open(path, "rb") as f:
626
+ raw = f.read(HEADER_SIZE)
627
+ size = os.fstat(f.fileno()).st_size
628
+ try:
629
+ return Header.parse(raw), size
630
+ except errors.FormatError:
631
+ return None, size
632
+
633
+ def _remap(self) -> None:
634
+ """Maps the file again after its size changed.
635
+
636
+ The old map is not closed: a concurrent ``read`` may still hold a
637
+ slice of it, and it is unmapped when the last one goes away.
638
+ """
639
+ assert self._path is not None
640
+ with open(self._path, "rb") as f:
641
+ mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
642
+ self._view, self._mm = memoryview(mm), mm
643
+
644
+ def _check_directory(
645
+ self, d: npt.NDArray, size: int, n_frames: int, n_blocks: int
646
+ ) -> None:
647
+ """Validates the directory against the header, vectorized."""
648
+ e, bf = self.n_envs, self.block_frames
649
+ n_windows = -(-n_frames // bf)
650
+ if n_blocks != n_windows * e:
651
+ raise errors.FormatError("block count does not match n_frames")
652
+ if not np.isin(d["codec"], list(codecs.BLOCK_CODEC_NAMES)).all():
653
+ bad = sorted(
654
+ set(d["codec"].tolist()) - set(codecs.BLOCK_CODEC_NAMES)
655
+ )
656
+ raise errors.FormatError(f"unknown block codec id {bad[0]}")
657
+ idx = np.arange(n_blocks)
658
+ t0 = (idx // e) * bf
659
+ n = np.minimum(bf, n_frames - t0)
660
+ k = self._k
661
+ ulen = np.where(
662
+ d["codec"] == codecs.CODEC_F32S, 4 * k * n, 8 * k + 2 * k * n
663
+ )
664
+ ok = (
665
+ (d["env"] == idx % e).all()
666
+ and (d["t0"] == t0).all()
667
+ and (d["n"] == n).all()
668
+ and (d["ulen"] == ulen).all()
669
+ and (d["offset"] % 8 == 0).all()
670
+ and (d["offset"] + 32 + d["clen"].astype(np.uint64) <= size).all()
671
+ )
672
+ if not ok:
673
+ raise errors.FormatError("block directory is inconsistent")
674
+
675
+ def _block(self, idx: int) -> npt.NDArray[np.float32]:
676
+ """Returns the decoded block ``idx`` from the cache or the file."""
677
+ with self._lock:
678
+ hit = self._cache.get(idx)
679
+ if hit is not None:
680
+ self._cache.move_to_end(idx)
681
+ return hit
682
+ assert self._view is not None
683
+ ent = self.directory[idx]
684
+ off, clen, n = int(ent["offset"]), int(ent["clen"]), int(ent["n"])
685
+ head = BLOCK_HEADER.unpack_from(self._view, off)
686
+ if head[0] != BLOCK_MAGIC:
687
+ raise errors.FormatError(f"bad block magic at offset {off}")
688
+ codec = int(ent["codec"])
689
+ if (head[1], head[2], head[3], head[4], head[5]) != (
690
+ codec,
691
+ ent["env"],
692
+ ent["t0"],
693
+ n,
694
+ clen,
695
+ ):
696
+ raise errors.FormatError(
697
+ f"block header at {off} disagrees with directory"
698
+ )
699
+ payload = self._view[off + 32 : off + 32 + clen]
700
+ try:
701
+ if self._verify and zlib.crc32(payload) != head[7]:
702
+ raise errors.FormatError(f"block CRC mismatch at offset {off}")
703
+ out = np.empty((n, self._k), np.float32)
704
+ codecs.decode_block_into(payload, codec, out)
705
+ finally:
706
+ payload.release()
707
+ if self._kind == "pose" and codec == codecs.CODEC_Q16D:
708
+ _renormalize_poses(out)
709
+ out.flags.writeable = False
710
+ with self._lock:
711
+ if self._cache_blocks:
712
+ self._cache[idx] = out
713
+ while len(self._cache) > self._cache_blocks:
714
+ self._cache.popitem(last=False)
715
+ return out
716
+
717
+ def read(
718
+ self,
719
+ t0: int,
720
+ t1: int,
721
+ envs: Sequence[int] | npt.NDArray[np.integer] | None = None,
722
+ ) -> npt.NDArray[np.float32]:
723
+ """Reads frames ``t0 <= t < t1``.
724
+
725
+ Args:
726
+ t0: First frame.
727
+ t1: One past the last frame.
728
+ envs: Env indices to read, or ``None`` for all envs.
729
+
730
+ Returns:
731
+ A new float32 array ``[t1 - t0, len(envs), *item_shape]``.
732
+
733
+ Raises:
734
+ IndexError: If the frame range or an env index is out of range.
735
+ ValueError: If the reader is closed.
736
+ """
737
+ if self._view is None:
738
+ raise ValueError("reader is closed")
739
+ if not 0 <= t0 <= t1 <= self.n_frames:
740
+ raise IndexError(
741
+ f"frame range [{t0}, {t1}) outside [0, {self.n_frames}]"
742
+ )
743
+ env_ids = (
744
+ np.arange(self.n_envs)
745
+ if envs is None
746
+ else np.asarray(envs, dtype=np.int64).reshape(-1)
747
+ )
748
+ if env_ids.size and (env_ids.min() < 0 or env_ids.max() >= self.n_envs):
749
+ raise IndexError(f"env index outside [0, {self.n_envs})")
750
+ out = np.empty((t1 - t0, env_ids.size, self._k), np.float32)
751
+ bf = self.block_frames
752
+ for w in range(t0 // bf, -(-t1 // bf)):
753
+ lo, hi = max(t0, w * bf), min(t1, (w + 1) * bf)
754
+ for j, env in enumerate(env_ids.tolist()):
755
+ blk = self._block(w * self.n_envs + env)
756
+ out[lo - t0 : hi - t0, j] = blk[lo - w * bf : hi - w * bf]
757
+ return out.reshape(t1 - t0, env_ids.size, *self.item_shape)
758
+
759
+ def close(self) -> None:
760
+ """Releases the memory map. Safe to call more than once."""
761
+ self._cache.clear()
762
+ if self._view is not None:
763
+ self._view.release()
764
+ self._view = None
765
+ if self._mm is not None:
766
+ self._mm.close()
767
+ self._mm = None
768
+
769
+ def __enter__(self) -> "BlockReader":
770
+ """Returns the reader."""
771
+ return self
772
+
773
+ def __exit__(
774
+ self,
775
+ exc_type: type[BaseException] | None,
776
+ exc: BaseException | None,
777
+ tb: TracebackType | None,
778
+ ) -> None:
779
+ """Closes the reader."""
780
+ self.close()
781
+
782
+
783
+ @dataclasses.dataclass(frozen=True)
784
+ class RecoverResult:
785
+ """Outcome of :func:`recover`.
786
+
787
+ Attributes:
788
+ n_frames: Frames kept.
789
+ n_blocks: Blocks kept.
790
+ dropped_bytes: Bytes removed from the end of the file (partial
791
+ blocks and incomplete windows).
792
+ already_finished: True if the file already had a directory and was
793
+ left untouched.
794
+ """
795
+
796
+ n_frames: int
797
+ n_blocks: int
798
+ dropped_bytes: int
799
+ already_finished: bool = False
800
+
801
+
802
+ @dataclasses.dataclass
803
+ class _ScanState:
804
+ """Resumable state of a walk over the blocks of an unfinished file.
805
+
806
+ Attributes:
807
+ off: Offset where the next block should start. It stays put at a
808
+ block that fails validation, so a later scan retries it.
809
+ end: Offset just after the last complete window.
810
+ frames: Frames in the complete windows.
811
+ env: Env of the next block within the current window.
812
+ win_n: Frames in the current window (set by its first block).
813
+ closed: True once a short window was kept: it can only be the last.
814
+ kept: Entries of complete windows not yet taken by the caller.
815
+ pending: Entries of the incomplete window being read.
816
+ """
817
+
818
+ off: int = HEADER_SIZE
819
+ end: int = HEADER_SIZE
820
+ frames: int = 0
821
+ env: int = 0
822
+ win_n: int = 0
823
+ closed: bool = False
824
+ kept: list[DirEntry] = dataclasses.field(default_factory=list)
825
+ pending: list[DirEntry] = dataclasses.field(default_factory=list)
826
+
827
+
828
+ def _advance(view: memoryview, h: Header, st: _ScanState) -> None:
829
+ """Continues a block walk, keeping only complete valid windows.
830
+
831
+ Checks each block's magic, lengths and CRC, and stops at the first one
832
+ that fails or is not fully present yet. Only the bytes after ``st.off``
833
+ are read.
834
+
835
+ Args:
836
+ view: The whole file as currently visible.
837
+ h: The parsed header.
838
+ st: Scan state, advanced in place.
839
+ """
840
+ off, env, win_n, frames = st.off, st.env, st.win_n, st.frames
841
+ size = len(view)
842
+ k = h.k
843
+ while not st.closed and off + 32 <= size:
844
+ (magic, codec, b_env, b_t0, n, clen, ulen, crc) = (
845
+ BLOCK_HEADER.unpack_from(view, off)
846
+ )
847
+ stop = (
848
+ magic != BLOCK_MAGIC
849
+ or codec not in codecs.BLOCK_CODEC_NAMES
850
+ or b_env != env
851
+ or b_t0 != frames
852
+ or not 1 <= n <= h.block_frames
853
+ or (env > 0 and n != win_n)
854
+ or ulen != codecs.payload_ulen(codec, n, k)
855
+ or off + 32 + clen > size
856
+ )
857
+ if stop:
858
+ break
859
+ with view[off + 32 : off + 32 + clen] as payload:
860
+ if zlib.crc32(payload) != crc:
861
+ break
862
+ st.pending.append((off, env, b_t0, n, clen, ulen, codec))
863
+ win_n = n
864
+ off = _align8(off + 32 + clen)
865
+ if env == h.n_envs - 1:
866
+ st.kept += st.pending
867
+ st.pending = []
868
+ frames += n
869
+ st.end = off
870
+ env = 0
871
+ if n < h.block_frames:
872
+ st.closed = True # a short window can only be the last one
873
+ else:
874
+ env += 1
875
+ st.off, st.env, st.win_n, st.frames = off, env, win_n, frames
876
+
877
+
878
+ def _scan_blocks(
879
+ view: memoryview, h: Header
880
+ ) -> tuple[list[DirEntry], int, int]:
881
+ """Walks blocks from offset 64 and keeps only complete valid windows.
882
+
883
+ Args:
884
+ view: The whole file.
885
+ h: The parsed header.
886
+
887
+ Returns:
888
+ ``(entries, n_frames, end_offset)`` for the complete windows. A
889
+ window is complete when all envs have a valid block.
890
+ """
891
+ st = _ScanState()
892
+ _advance(view, h, st)
893
+ return st.kept, st.frames, st.end
894
+
895
+
896
+ def recover(path: os.PathLike[str] | str) -> RecoverResult:
897
+ """Rebuilds the directory of an unfinished block file.
898
+
899
+ Walks blocks from offset 64, keeps every complete time window (a block
900
+ for every env) whose header and CRC check out, truncates the rest, then
901
+ appends the directory and rewrites the header. A finished file is left
902
+ untouched.
903
+
904
+ Args:
905
+ path: The block file.
906
+
907
+ Returns:
908
+ What was kept and dropped.
909
+
910
+ Raises:
911
+ errors.FormatError: If the file header is missing or corrupt.
912
+ """
913
+ path = pathlib.Path(path)
914
+ with open(path, "r+b") as f:
915
+ size = os.fstat(f.fileno()).st_size
916
+ if size < HEADER_SIZE:
917
+ raise errors.FormatError(f"{path}: shorter than header")
918
+ with mmap.mmap(f.fileno(), 0) as mm, memoryview(mm) as view:
919
+ h = Header.parse(view)
920
+ if h.dir_offset != 0:
921
+ blocks = h.n_blocks
922
+ return RecoverResult(h.n_frames, blocks, 0, True)
923
+ entries, frames, end = _scan_blocks(view, h)
924
+ dropped = max(0, size - end)
925
+ f.truncate(end)
926
+ directory = _dir_bytes(entries)
927
+ f.seek(end)
928
+ f.write(directory)
929
+ header = dataclasses.replace(
930
+ h,
931
+ n_frames=frames,
932
+ n_blocks=len(entries),
933
+ dir_offset=end,
934
+ dir_length=len(directory),
935
+ )
936
+ f.seek(0)
937
+ f.write(header.pack())
938
+ return RecoverResult(frames, len(entries), dropped)