simscope 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simscope/__init__.py +6 -0
- simscope/__main__.py +8 -0
- simscope/_assets/simscope-app.css +2 -0
- simscope/_assets/simscope-app.js +4311 -0
- simscope/_assets/simscope-player.js +4325 -0
- simscope/_assets/simscope-web.LICENSES.txt +407 -0
- simscope/_icon.py +22 -0
- simscope/_mjviser.py +203 -0
- simscope/annotations.py +1132 -0
- simscope/cli.py +482 -0
- simscope/core.py +257 -0
- simscope/derived.py +697 -0
- simscope/export.py +799 -0
- simscope/highlights.py +947 -0
- simscope/importers.py +874 -0
- simscope/index.py +579 -0
- simscope/io/__init__.py +45 -0
- simscope/io/blockfile.py +938 -0
- simscope/io/cas.py +294 -0
- simscope/io/codecs.py +566 -0
- simscope/io/errors.py +9 -0
- simscope/io/manifest.py +358 -0
- simscope/io/pack.py +563 -0
- simscope/io/scene.py +239 -0
- simscope/isaaclab.py +1460 -0
- simscope/library.py +705 -0
- simscope/mujoco.py +578 -0
- simscope/py.typed +0 -0
- simscope/recorder.py +784 -0
- simscope/server/__init__.py +9 -0
- simscope/server/app.py +149 -0
- simscope/server/blocks.py +191 -0
- simscope/server/jobs.py +166 -0
- simscope/server/routes.py +707 -0
- simscope/server/security.py +218 -0
- simscope/server/state.py +751 -0
- simscope/server/static.py +84 -0
- simscope/transforms.py +147 -0
- simscope-0.1.1.dist-info/METADATA +132 -0
- simscope-0.1.1.dist-info/RECORD +45 -0
- simscope-0.1.1.dist-info/WHEEL +4 -0
- simscope-0.1.1.dist-info/entry_points.txt +3 -0
- simscope-0.1.1.dist-info/licenses/LICENSE.md +201 -0
- simscope-0.1.1.dist-info/licenses/THIRD_PARTY_NOTICES.md +267 -0
- simscope-0.1.1.dist-info/licenses/src/simscope/_assets/simscope-web.LICENSES.txt +407 -0
simscope/io/blockfile.py
ADDED
|
@@ -0,0 +1,938 @@
|
|
|
1
|
+
"""Block files (``*.blk``): writer, reader and crash recovery."""
|
|
2
|
+
|
|
3
|
+
import collections
|
|
4
|
+
import dataclasses
|
|
5
|
+
import math
|
|
6
|
+
import mmap
|
|
7
|
+
import os
|
|
8
|
+
import pathlib
|
|
9
|
+
import struct
|
|
10
|
+
import threading
|
|
11
|
+
import zlib
|
|
12
|
+
from collections.abc import Sequence
|
|
13
|
+
from types import TracebackType
|
|
14
|
+
from typing import BinaryIO
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
import numpy.typing as npt
|
|
18
|
+
|
|
19
|
+
from simscope import core, transforms
|
|
20
|
+
from simscope.io import codecs, errors
|
|
21
|
+
|
|
22
|
+
FILE_MAGIC = b"SSBK"
|
|
23
|
+
BLOCK_MAGIC = b"SSBB"
|
|
24
|
+
MAJOR = 1
|
|
25
|
+
MINOR = 0
|
|
26
|
+
HEADER = struct.Struct("<4sHHII4IIIIIQII")
|
|
27
|
+
BLOCK_HEADER = struct.Struct("<4sB3xIIIIII")
|
|
28
|
+
DIR_DTYPE = np.dtype(
|
|
29
|
+
[
|
|
30
|
+
("offset", "<u8"),
|
|
31
|
+
("env", "<u4"),
|
|
32
|
+
("t0", "<u4"),
|
|
33
|
+
("n", "<u4"),
|
|
34
|
+
("clen", "<u4"),
|
|
35
|
+
("ulen", "<u4"),
|
|
36
|
+
("codec", "u1"),
|
|
37
|
+
("reserved", "V3"),
|
|
38
|
+
]
|
|
39
|
+
)
|
|
40
|
+
_DIR_FIELDS = ("offset", "env", "t0", "n", "clen", "ulen", "codec")
|
|
41
|
+
HEADER_SIZE = 64
|
|
42
|
+
DEFAULT_BLOCK_FRAMES = 100
|
|
43
|
+
assert HEADER.size == HEADER_SIZE
|
|
44
|
+
assert BLOCK_HEADER.size == 32
|
|
45
|
+
assert DIR_DTYPE.itemsize == 32
|
|
46
|
+
|
|
47
|
+
_MAX_NDIM = 4
|
|
48
|
+
_PAD = bytes(8)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _align8(offset: int) -> int:
|
|
52
|
+
"""Rounds ``offset`` up to a multiple of 8."""
|
|
53
|
+
return (offset + 7) & ~7
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclasses.dataclass(frozen=True)
|
|
57
|
+
class Header:
|
|
58
|
+
"""Parsed block-file header.
|
|
59
|
+
|
|
60
|
+
Attributes:
|
|
61
|
+
item_shape: Shape of one env-frame item.
|
|
62
|
+
n_envs: Number of envs.
|
|
63
|
+
n_frames: Total frames, 0 while recording.
|
|
64
|
+
block_frames: Nominal frames per block.
|
|
65
|
+
n_blocks: Number of blocks, 0 while recording.
|
|
66
|
+
dir_offset: Directory offset, 0 while recording.
|
|
67
|
+
dir_length: Directory length in bytes.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
item_shape: tuple[int, ...]
|
|
71
|
+
n_envs: int
|
|
72
|
+
n_frames: int
|
|
73
|
+
block_frames: int
|
|
74
|
+
n_blocks: int
|
|
75
|
+
dir_offset: int
|
|
76
|
+
dir_length: int
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def k(self) -> int:
|
|
80
|
+
"""Floats per env per frame."""
|
|
81
|
+
return math.prod(self.item_shape)
|
|
82
|
+
|
|
83
|
+
def pack(self) -> bytes:
|
|
84
|
+
"""Serializes the header, including its CRC."""
|
|
85
|
+
shape = [*self.item_shape, 0, 0, 0, 0][:_MAX_NDIM]
|
|
86
|
+
head = HEADER.pack(
|
|
87
|
+
FILE_MAGIC,
|
|
88
|
+
MAJOR,
|
|
89
|
+
MINOR,
|
|
90
|
+
HEADER_SIZE,
|
|
91
|
+
len(self.item_shape),
|
|
92
|
+
*shape,
|
|
93
|
+
self.n_envs,
|
|
94
|
+
self.n_frames,
|
|
95
|
+
self.block_frames,
|
|
96
|
+
self.n_blocks,
|
|
97
|
+
self.dir_offset,
|
|
98
|
+
self.dir_length,
|
|
99
|
+
0,
|
|
100
|
+
)
|
|
101
|
+
return head[:60] + struct.pack("<I", zlib.crc32(head[:60]))
|
|
102
|
+
|
|
103
|
+
@classmethod
|
|
104
|
+
def parse(cls, buf: bytes | memoryview) -> "Header":
|
|
105
|
+
"""Parses and validates a header.
|
|
106
|
+
|
|
107
|
+
Args:
|
|
108
|
+
buf: At least 64 bytes from the start of a block file.
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
The header.
|
|
112
|
+
|
|
113
|
+
Raises:
|
|
114
|
+
errors.FormatError: On a short file, bad magic, unknown major
|
|
115
|
+
version, CRC mismatch, or inconsistent fields.
|
|
116
|
+
"""
|
|
117
|
+
if len(buf) < HEADER_SIZE:
|
|
118
|
+
raise errors.FormatError("block file shorter than its header")
|
|
119
|
+
fields = HEADER.unpack_from(buf, 0)
|
|
120
|
+
magic, major = fields[0], fields[1]
|
|
121
|
+
if magic != FILE_MAGIC:
|
|
122
|
+
raise errors.FormatError(f"bad block file magic {magic!r}")
|
|
123
|
+
if major != MAJOR:
|
|
124
|
+
raise errors.FormatError(
|
|
125
|
+
f"unknown block file major version {major}"
|
|
126
|
+
)
|
|
127
|
+
if zlib.crc32(buf[:60]) != fields[-1]:
|
|
128
|
+
raise errors.FormatError("block file header CRC mismatch")
|
|
129
|
+
header_size, ndim = fields[3], fields[4]
|
|
130
|
+
if header_size != HEADER_SIZE or ndim > _MAX_NDIM:
|
|
131
|
+
raise errors.FormatError("invalid block file header fields")
|
|
132
|
+
shape = tuple(fields[5 : 5 + ndim])
|
|
133
|
+
n_envs, n_frames, block_frames, n_blocks = fields[9:13]
|
|
134
|
+
if 0 in shape or n_envs < 1 or block_frames < 1:
|
|
135
|
+
raise errors.FormatError("invalid block file header fields")
|
|
136
|
+
return cls(
|
|
137
|
+
shape,
|
|
138
|
+
n_envs,
|
|
139
|
+
n_frames,
|
|
140
|
+
block_frames,
|
|
141
|
+
n_blocks,
|
|
142
|
+
fields[13],
|
|
143
|
+
fields[14],
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@dataclasses.dataclass(frozen=True)
|
|
148
|
+
class EncodedBlock:
|
|
149
|
+
"""One encoded block, ready to be written.
|
|
150
|
+
|
|
151
|
+
Attributes:
|
|
152
|
+
env: Env index.
|
|
153
|
+
n: Frames in the block.
|
|
154
|
+
codec: Codec id actually used.
|
|
155
|
+
ulen: Uncompressed payload length.
|
|
156
|
+
payload: Compressed payload.
|
|
157
|
+
"""
|
|
158
|
+
|
|
159
|
+
env: int
|
|
160
|
+
n: int
|
|
161
|
+
codec: int
|
|
162
|
+
ulen: int
|
|
163
|
+
payload: bytes
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def encode_window(
|
|
167
|
+
window: npt.NDArray[np.float32], codec: str | int = "f32s"
|
|
168
|
+
) -> list[EncodedBlock]:
|
|
169
|
+
"""Encodes one time window into one block per env (pure, no file I/O).
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
window: Float32 array ``[n, n_envs, K]`` for frames ``t0..t0+n-1``.
|
|
173
|
+
codec: Requested codec. q16d falls back to f32s per block when the
|
|
174
|
+
block has non-finite values.
|
|
175
|
+
|
|
176
|
+
Returns:
|
|
177
|
+
Encoded blocks in env order.
|
|
178
|
+
"""
|
|
179
|
+
n, n_envs, k = window.shape
|
|
180
|
+
blocks = []
|
|
181
|
+
for env in range(n_envs):
|
|
182
|
+
cid, payload = codecs.encode_block_auto(window[:, env, :], codec)
|
|
183
|
+
blocks.append(
|
|
184
|
+
EncodedBlock(env, n, cid, codecs.payload_ulen(cid, n, k), payload)
|
|
185
|
+
)
|
|
186
|
+
return blocks
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
DirEntry = tuple[int, int, int, int, int, int, int]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def pack_window(
|
|
193
|
+
t0: int, blocks: Sequence[EncodedBlock], offset: int
|
|
194
|
+
) -> tuple[bytes, list[DirEntry]]:
|
|
195
|
+
"""Serializes encoded blocks with headers and 8-byte alignment.
|
|
196
|
+
|
|
197
|
+
Args:
|
|
198
|
+
t0: First frame index of the window.
|
|
199
|
+
blocks: Encoded blocks of the window, in env order.
|
|
200
|
+
offset: File offset where the first block will start (aligned to 8).
|
|
201
|
+
|
|
202
|
+
Returns:
|
|
203
|
+
``(data, entries)``: the bytes to write at ``offset`` (each block
|
|
204
|
+
padded to 8 bytes) and the directory entries
|
|
205
|
+
``(offset, env, t0, n, clen, ulen, codec)`` of the blocks.
|
|
206
|
+
"""
|
|
207
|
+
parts = []
|
|
208
|
+
entries: list[DirEntry] = []
|
|
209
|
+
pos = offset
|
|
210
|
+
for b in blocks:
|
|
211
|
+
clen = len(b.payload)
|
|
212
|
+
pad = -(32 + clen) % 8
|
|
213
|
+
head = BLOCK_HEADER.pack(
|
|
214
|
+
BLOCK_MAGIC,
|
|
215
|
+
b.codec,
|
|
216
|
+
b.env,
|
|
217
|
+
t0,
|
|
218
|
+
b.n,
|
|
219
|
+
clen,
|
|
220
|
+
b.ulen,
|
|
221
|
+
zlib.crc32(b.payload),
|
|
222
|
+
)
|
|
223
|
+
parts += [head, b.payload, _PAD[:pad]]
|
|
224
|
+
entries.append((pos, b.env, t0, b.n, clen, b.ulen, b.codec))
|
|
225
|
+
pos += 32 + clen + pad
|
|
226
|
+
return b"".join(parts), entries
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _dir_bytes(entries: Sequence[DirEntry]) -> bytes:
|
|
230
|
+
"""Packs directory entries into the 32-byte on-disk records."""
|
|
231
|
+
arr = np.zeros(len(entries), DIR_DTYPE)
|
|
232
|
+
if entries:
|
|
233
|
+
cols = np.array(entries, dtype=np.uint64).T
|
|
234
|
+
for name, col in zip(_DIR_FIELDS, cols, strict=True):
|
|
235
|
+
arr[name] = col
|
|
236
|
+
return arr.tobytes()
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
class BlockWriter:
|
|
240
|
+
"""Writes one stream as a block file.
|
|
241
|
+
|
|
242
|
+
``append`` copies frames into a preallocated ``[block_frames, n_envs, K]``
|
|
243
|
+
buffer. Each time it fills, the window is encoded (one block per env) and
|
|
244
|
+
written. Leaving the ``with`` block normally calls :meth:`finalize`; on an
|
|
245
|
+
exception the file is closed unfinished so :func:`recover` can salvage it.
|
|
246
|
+
|
|
247
|
+
Attributes:
|
|
248
|
+
path: Destination path.
|
|
249
|
+
item_shape: Shape of one env-frame item.
|
|
250
|
+
n_envs: Number of envs.
|
|
251
|
+
kind: Stream kind (``"pose"`` streams get sign continuity).
|
|
252
|
+
codec: Requested codec name.
|
|
253
|
+
block_frames: Frames per block.
|
|
254
|
+
"""
|
|
255
|
+
|
|
256
|
+
def __init__(
|
|
257
|
+
self,
|
|
258
|
+
path: os.PathLike[str] | str,
|
|
259
|
+
*,
|
|
260
|
+
item_shape: Sequence[int],
|
|
261
|
+
n_envs: int,
|
|
262
|
+
kind: core.StreamKind,
|
|
263
|
+
codec: str = "f32s",
|
|
264
|
+
block_frames: int = DEFAULT_BLOCK_FRAMES,
|
|
265
|
+
) -> None:
|
|
266
|
+
"""Creates the file and writes a provisional header.
|
|
267
|
+
|
|
268
|
+
Args:
|
|
269
|
+
path: Destination path (overwritten).
|
|
270
|
+
item_shape: Shape of one env-frame item, at most 4 dims, all > 0.
|
|
271
|
+
n_envs: Number of envs (at least 1).
|
|
272
|
+
kind: Stream kind. ``"pose"`` requires a last dim of 7.
|
|
273
|
+
codec: ``"f32s"`` (lossless) or ``"q16d"``.
|
|
274
|
+
block_frames: Frames per block (at least 1).
|
|
275
|
+
|
|
276
|
+
Raises:
|
|
277
|
+
ValueError: If an argument is invalid.
|
|
278
|
+
errors.FormatError: If the codec is unknown.
|
|
279
|
+
"""
|
|
280
|
+
self.path = pathlib.Path(path)
|
|
281
|
+
self.item_shape = tuple(int(d) for d in item_shape)
|
|
282
|
+
self.n_envs = int(n_envs)
|
|
283
|
+
self.kind = kind
|
|
284
|
+
self.block_frames = int(block_frames)
|
|
285
|
+
self._codec_id = codecs.codec_id(codec)
|
|
286
|
+
self.codec = codecs.BLOCK_CODEC_NAMES[self._codec_id]
|
|
287
|
+
if len(self.item_shape) > _MAX_NDIM or 0 in self.item_shape:
|
|
288
|
+
raise ValueError(f"invalid item_shape {self.item_shape}")
|
|
289
|
+
if self.n_envs < 1 or self.block_frames < 1:
|
|
290
|
+
raise ValueError("n_envs and block_frames must be at least 1")
|
|
291
|
+
if kind == "pose" and self.item_shape[-1:] != (core.POSE_DIM,):
|
|
292
|
+
raise ValueError("pose streams need a last item dim of 7")
|
|
293
|
+
self._k = math.prod(self.item_shape)
|
|
294
|
+
self._buf = np.empty(
|
|
295
|
+
(self.block_frames, self.n_envs, self._k), np.float32
|
|
296
|
+
)
|
|
297
|
+
self._fill = 0
|
|
298
|
+
self._n_frames = 0
|
|
299
|
+
self._prev_quat: npt.NDArray[np.float32] | None = None
|
|
300
|
+
self._entries: list[DirEntry] = []
|
|
301
|
+
self._offset = HEADER_SIZE
|
|
302
|
+
self._finalized = False
|
|
303
|
+
self._file: BinaryIO | None = open(self.path, "wb") # noqa: SIM115
|
|
304
|
+
self._file.write(self._header(0, 0, 0, 0).pack())
|
|
305
|
+
self._file.flush() # tailing readers see the header at once
|
|
306
|
+
|
|
307
|
+
def _header(
|
|
308
|
+
self, n_frames: int, n_blocks: int, dir_offset: int, dir_len: int
|
|
309
|
+
) -> Header:
|
|
310
|
+
"""Builds a header for the current stream parameters."""
|
|
311
|
+
return Header(
|
|
312
|
+
self.item_shape,
|
|
313
|
+
self.n_envs,
|
|
314
|
+
n_frames,
|
|
315
|
+
self.block_frames,
|
|
316
|
+
n_blocks,
|
|
317
|
+
dir_offset,
|
|
318
|
+
dir_len,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
@property
|
|
322
|
+
def n_frames(self) -> int:
|
|
323
|
+
"""Frames appended so far, including the unflushed window."""
|
|
324
|
+
return self._n_frames + self._fill
|
|
325
|
+
|
|
326
|
+
def append(self, frames: npt.ArrayLike) -> None:
|
|
327
|
+
"""Appends frames to the stream.
|
|
328
|
+
|
|
329
|
+
Args:
|
|
330
|
+
frames: Array ``[n, n_envs, *item_shape]`` with n at least 1.
|
|
331
|
+
It is copied; the caller may reuse it.
|
|
332
|
+
|
|
333
|
+
Raises:
|
|
334
|
+
ValueError: If the shape is wrong or the writer is finalized.
|
|
335
|
+
"""
|
|
336
|
+
if self._file is None or self._finalized:
|
|
337
|
+
raise ValueError("writer is closed")
|
|
338
|
+
a = np.asarray(frames, dtype=np.float32)
|
|
339
|
+
want = (self.n_envs, *self.item_shape)
|
|
340
|
+
if a.ndim != len(want) + 1 or a.shape[1:] != want or a.shape[0] < 1:
|
|
341
|
+
raise ValueError(
|
|
342
|
+
f"expected frames of shape [n>=1, {', '.join(map(str, want))}]"
|
|
343
|
+
f", got {a.shape}"
|
|
344
|
+
)
|
|
345
|
+
a = a.reshape(a.shape[0], self.n_envs, self._k)
|
|
346
|
+
pos = 0
|
|
347
|
+
while pos < a.shape[0]:
|
|
348
|
+
take = min(a.shape[0] - pos, self.block_frames - self._fill)
|
|
349
|
+
dst = self._buf[self._fill : self._fill + take]
|
|
350
|
+
dst[...] = a[pos : pos + take]
|
|
351
|
+
if self.kind == "pose":
|
|
352
|
+
self._fix_signs(dst)
|
|
353
|
+
self._fill += take
|
|
354
|
+
pos += take
|
|
355
|
+
if self._fill == self.block_frames:
|
|
356
|
+
self._flush()
|
|
357
|
+
|
|
358
|
+
def _fix_signs(self, chunk: npt.NDArray[np.float32]) -> None:
|
|
359
|
+
"""Enforces quaternion sign continuity on a buffer chunk in place."""
|
|
360
|
+
n = chunk.shape[0]
|
|
361
|
+
q = chunk.reshape(n, self.n_envs, -1, core.POSE_DIM)[..., 3:]
|
|
362
|
+
fixed = transforms.enforce_sign_continuity(q, self._prev_quat)
|
|
363
|
+
q[...] = fixed
|
|
364
|
+
self._prev_quat = fixed[-1].copy()
|
|
365
|
+
|
|
366
|
+
def _flush(self) -> None:
|
|
367
|
+
"""Encodes and writes the buffered window."""
|
|
368
|
+
if self._fill == 0 or self._file is None:
|
|
369
|
+
return
|
|
370
|
+
blocks = encode_window(self._buf[: self._fill], self._codec_id)
|
|
371
|
+
self.write_encoded(self._n_frames, blocks) # advances _n_frames
|
|
372
|
+
self._fill = 0
|
|
373
|
+
|
|
374
|
+
def write_encoded(self, t0: int, blocks: Sequence[EncodedBlock]) -> None:
|
|
375
|
+
"""Writes already-encoded blocks of one window.
|
|
376
|
+
|
|
377
|
+
Advances the frame count, so ``finalize`` records it.
|
|
378
|
+
|
|
379
|
+
Args:
|
|
380
|
+
t0: First frame index of the window.
|
|
381
|
+
blocks: One encoded block per env, in env order.
|
|
382
|
+
"""
|
|
383
|
+
assert self._file is not None
|
|
384
|
+
data, entries = pack_window(t0, blocks, self._offset)
|
|
385
|
+
self._file.write(data)
|
|
386
|
+
self._file.flush() # a tailing reader sees the window right away
|
|
387
|
+
self._entries += entries
|
|
388
|
+
self._offset += len(data)
|
|
389
|
+
if blocks:
|
|
390
|
+
self._n_frames = max(self._n_frames, t0 + blocks[0].n)
|
|
391
|
+
|
|
392
|
+
def finalize(self) -> None:
|
|
393
|
+
"""Flushes the last window, writes the directory and the header."""
|
|
394
|
+
if self._finalized or self._file is None:
|
|
395
|
+
return
|
|
396
|
+
self._flush()
|
|
397
|
+
directory = _dir_bytes(self._entries)
|
|
398
|
+
self._file.write(directory)
|
|
399
|
+
header = self._header(
|
|
400
|
+
self._n_frames, len(self._entries), self._offset, len(directory)
|
|
401
|
+
)
|
|
402
|
+
self._file.seek(0)
|
|
403
|
+
self._file.write(header.pack())
|
|
404
|
+
self._file.close()
|
|
405
|
+
self._file = None
|
|
406
|
+
self._finalized = True
|
|
407
|
+
|
|
408
|
+
def close(self) -> None:
|
|
409
|
+
"""Closes the file without finalizing (the file stays recoverable)."""
|
|
410
|
+
if self._file is not None:
|
|
411
|
+
self._file.close()
|
|
412
|
+
self._file = None
|
|
413
|
+
|
|
414
|
+
def __enter__(self) -> "BlockWriter":
|
|
415
|
+
"""Returns the writer."""
|
|
416
|
+
return self
|
|
417
|
+
|
|
418
|
+
def __exit__(
|
|
419
|
+
self,
|
|
420
|
+
exc_type: type[BaseException] | None,
|
|
421
|
+
exc: BaseException | None,
|
|
422
|
+
tb: TracebackType | None,
|
|
423
|
+
) -> None:
|
|
424
|
+
"""Finalizes on success; on error closes and leaves it unfinished."""
|
|
425
|
+
if exc_type is None:
|
|
426
|
+
self.finalize()
|
|
427
|
+
else:
|
|
428
|
+
self.close()
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _renormalize_poses(block: npt.NDArray[np.float32]) -> None:
|
|
432
|
+
"""Renormalizes the quaternion of every pose in a ``[n, K]`` block."""
|
|
433
|
+
q = block.reshape(block.shape[0], -1, core.POSE_DIM)[..., 3:]
|
|
434
|
+
norm = np.sqrt(np.sum(q * q, axis=-1, keepdims=True))
|
|
435
|
+
np.divide(q, norm, out=q, where=norm > 0)
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
class BlockReader:
|
|
439
|
+
"""Random-access reader for a block file.
|
|
440
|
+
|
|
441
|
+
The file is memory-mapped. ``read`` decodes only the blocks that overlap
|
|
442
|
+
the requested window and keeps the most recent decoded blocks in an LRU
|
|
443
|
+
cache. Instances are safe to share between threads.
|
|
444
|
+
|
|
445
|
+
With ``partial=True`` the reader can also tail a file that a
|
|
446
|
+
:class:`BlockWriter` is still writing: it builds the
|
|
447
|
+
directory by scanning blocks and :meth:`refresh` picks up windows
|
|
448
|
+
appended later. It never writes to the file, and only complete windows
|
|
449
|
+
whose CRCs check out are visible, so a window the writer is in the middle
|
|
450
|
+
of writing is never returned.
|
|
451
|
+
|
|
452
|
+
Attributes:
|
|
453
|
+
n_frames: Total frames (the complete windows found so far, for a
|
|
454
|
+
file that is still being written).
|
|
455
|
+
n_envs: Number of envs.
|
|
456
|
+
item_shape: Shape of one env-frame item.
|
|
457
|
+
block_frames: Nominal frames per block.
|
|
458
|
+
n_blocks: Number of blocks.
|
|
459
|
+
directory: Structured array of directory entries (``DIR_DTYPE``).
|
|
460
|
+
"""
|
|
461
|
+
|
|
462
|
+
def __init__(
|
|
463
|
+
self,
|
|
464
|
+
source: os.PathLike[str] | str | bytes | bytearray | memoryview,
|
|
465
|
+
*,
|
|
466
|
+
cache_blocks: int = 64,
|
|
467
|
+
kind: core.StreamKind | None = None,
|
|
468
|
+
verify: bool = True,
|
|
469
|
+
partial: bool = False,
|
|
470
|
+
) -> None:
|
|
471
|
+
"""Opens a block file or an in-memory block file.
|
|
472
|
+
|
|
473
|
+
Args:
|
|
474
|
+
source: A path, or the file bytes (for example a pack entry).
|
|
475
|
+
cache_blocks: Decoded blocks to keep in the LRU cache.
|
|
476
|
+
kind: The stream kind from the manifest. For ``"pose"`` streams,
|
|
477
|
+
quaternions of q16d blocks are renormalized.
|
|
478
|
+
verify: Check each block's payload CRC when it is decoded.
|
|
479
|
+
partial: Accept an unfinished file (``dir_offset == 0``) by
|
|
480
|
+
scanning its blocks. A finished file opens as usual.
|
|
481
|
+
|
|
482
|
+
Raises:
|
|
483
|
+
errors.FormatError: If the file is unfinished and ``partial`` is
|
|
484
|
+
false (use :func:`recover`), or is corrupt, or uses an
|
|
485
|
+
unknown version or codec.
|
|
486
|
+
"""
|
|
487
|
+
self._mm: mmap.mmap | None = None
|
|
488
|
+
self._view: memoryview | None = None
|
|
489
|
+
self._path: pathlib.Path | None = None
|
|
490
|
+
self._verify = verify
|
|
491
|
+
self._kind = kind
|
|
492
|
+
self._cache_blocks = max(0, int(cache_blocks))
|
|
493
|
+
self._cache: collections.OrderedDict[int, npt.NDArray[np.float32]] = (
|
|
494
|
+
collections.OrderedDict()
|
|
495
|
+
)
|
|
496
|
+
self._lock = threading.Lock()
|
|
497
|
+
self._refresh_lock = threading.Lock()
|
|
498
|
+
self._scan: _ScanState | None = None
|
|
499
|
+
self._head: Header | None = None
|
|
500
|
+
self._dir_buf = np.empty(0, DIR_DTYPE)
|
|
501
|
+
if isinstance(source, str | os.PathLike):
|
|
502
|
+
self._path = pathlib.Path(source)
|
|
503
|
+
with open(source, "rb") as f:
|
|
504
|
+
if os.fstat(f.fileno()).st_size < HEADER_SIZE:
|
|
505
|
+
raise errors.FormatError(f"{source}: shorter than header")
|
|
506
|
+
self._mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
|
|
507
|
+
self._view = memoryview(self._mm)
|
|
508
|
+
else:
|
|
509
|
+
self._view = memoryview(source).cast("B")
|
|
510
|
+
try:
|
|
511
|
+
self._init_from_view(self._view, partial)
|
|
512
|
+
except BaseException:
|
|
513
|
+
self.close()
|
|
514
|
+
raise
|
|
515
|
+
|
|
516
|
+
@property
|
|
517
|
+
def finished(self) -> bool:
|
|
518
|
+
"""True if the file has its directory (it will not grow)."""
|
|
519
|
+
return self._scan is None
|
|
520
|
+
|
|
521
|
+
def _init_from_view(self, view: memoryview, partial: bool) -> None:
|
|
522
|
+
"""Parses the header and directory and validates them."""
|
|
523
|
+
h = Header.parse(view)
|
|
524
|
+
self.item_shape = h.item_shape
|
|
525
|
+
self.n_envs = h.n_envs
|
|
526
|
+
self.block_frames = h.block_frames
|
|
527
|
+
self._k = h.k
|
|
528
|
+
if h.dir_offset == 0:
|
|
529
|
+
if not partial:
|
|
530
|
+
raise errors.FormatError(
|
|
531
|
+
"block file is unfinished (dir_offset == 0); "
|
|
532
|
+
"run simscope.io.blockfile.recover() first"
|
|
533
|
+
)
|
|
534
|
+
self._head = h
|
|
535
|
+
self._scan = _ScanState()
|
|
536
|
+
self.n_frames = 0
|
|
537
|
+
self.n_blocks = 0
|
|
538
|
+
self.directory = self._dir_buf
|
|
539
|
+
self._scan_more(view)
|
|
540
|
+
return
|
|
541
|
+
self._install_finished(view, h)
|
|
542
|
+
|
|
543
|
+
def _install_finished(self, view: memoryview, h: Header) -> None:
|
|
544
|
+
"""Validates the directory of a finished file and adopts it.
|
|
545
|
+
|
|
546
|
+
The directory is published before the counts, so a concurrent
|
|
547
|
+
``read`` never sees a count that its directory does not cover.
|
|
548
|
+
"""
|
|
549
|
+
end = h.dir_offset + h.dir_length
|
|
550
|
+
if h.dir_length != 32 * h.n_blocks or end > len(view):
|
|
551
|
+
raise errors.FormatError("block directory is truncated or invalid")
|
|
552
|
+
d = np.frombuffer(view, DIR_DTYPE, h.n_blocks, h.dir_offset).copy()
|
|
553
|
+
self._check_directory(d, len(view), h.n_frames, h.n_blocks)
|
|
554
|
+
self.directory = d
|
|
555
|
+
self.n_blocks = h.n_blocks
|
|
556
|
+
self.n_frames = h.n_frames
|
|
557
|
+
self._scan = None
|
|
558
|
+
self._head = None
|
|
559
|
+
self._dir_buf = np.empty(0, DIR_DTYPE)
|
|
560
|
+
|
|
561
|
+
def _scan_more(self, view: memoryview) -> None:
|
|
562
|
+
"""Scans blocks appended since the last scan and publishes them."""
|
|
563
|
+
st, h = self._scan, self._head
|
|
564
|
+
assert st is not None and h is not None
|
|
565
|
+
_advance(view, h, st)
|
|
566
|
+
if not st.kept:
|
|
567
|
+
return
|
|
568
|
+
new = np.frombuffer(_dir_bytes(st.kept), DIR_DTYPE)
|
|
569
|
+
st.kept = []
|
|
570
|
+
n = self.n_blocks
|
|
571
|
+
if n + len(new) > len(self._dir_buf):
|
|
572
|
+
grown = np.empty(max(64, 2 * (n + len(new))), DIR_DTYPE)
|
|
573
|
+
grown[:n] = self._dir_buf[:n]
|
|
574
|
+
self._dir_buf = grown
|
|
575
|
+
self._dir_buf[n : n + len(new)] = new
|
|
576
|
+
self.directory = self._dir_buf[: n + len(new)]
|
|
577
|
+
self.n_blocks = n + len(new)
|
|
578
|
+
self.n_frames = st.frames
|
|
579
|
+
|
|
580
|
+
def refresh(self) -> int:
|
|
581
|
+
"""Picks up windows appended to a file that is still being written.
|
|
582
|
+
|
|
583
|
+
Re-reads the header and scans only the bytes after the last known
|
|
584
|
+
good window, so the cost is proportional to the new data. When the
|
|
585
|
+
writer has finalized the file, the reader switches to the finished
|
|
586
|
+
directory. A finished file, or an in-memory source, is left as is.
|
|
587
|
+
|
|
588
|
+
Returns:
|
|
589
|
+
The number of frames now readable.
|
|
590
|
+
|
|
591
|
+
Raises:
|
|
592
|
+
errors.FormatError: If the file shrank, or a finished file has
|
|
593
|
+
an invalid directory.
|
|
594
|
+
ValueError: If the reader is closed.
|
|
595
|
+
"""
|
|
596
|
+
if self._view is None:
|
|
597
|
+
raise ValueError("reader is closed")
|
|
598
|
+
if self._scan is None or self._path is None:
|
|
599
|
+
return self.n_frames
|
|
600
|
+
with self._refresh_lock:
|
|
601
|
+
if self._scan is None:
|
|
602
|
+
return self.n_frames
|
|
603
|
+
head, size = self._read_tail_state(self._path)
|
|
604
|
+
if size < self._scan.end:
|
|
605
|
+
raise errors.FormatError(f"{self._path}: file shrank")
|
|
606
|
+
if size != len(self._view):
|
|
607
|
+
# Also when it shrank (a concurrent recover): never touch
|
|
608
|
+
# mapped pages past the end of the file.
|
|
609
|
+
self._remap()
|
|
610
|
+
view = self._view
|
|
611
|
+
assert view is not None
|
|
612
|
+
if head is not None and head.dir_offset != 0:
|
|
613
|
+
self._install_finished(view, head)
|
|
614
|
+
else:
|
|
615
|
+
self._scan_more(view)
|
|
616
|
+
return self.n_frames
|
|
617
|
+
|
|
618
|
+
def _read_tail_state(self, path: pathlib.Path) -> tuple[Header | None, int]:
|
|
619
|
+
"""Reads the header, then the size (in that order, see refresh).
|
|
620
|
+
|
|
621
|
+
Returns:
|
|
622
|
+
``(header, size)``. The header is ``None`` if it cannot be
|
|
623
|
+
parsed yet (a torn write); the next refresh retries.
|
|
624
|
+
"""
|
|
625
|
+
with open(path, "rb") as f:
|
|
626
|
+
raw = f.read(HEADER_SIZE)
|
|
627
|
+
size = os.fstat(f.fileno()).st_size
|
|
628
|
+
try:
|
|
629
|
+
return Header.parse(raw), size
|
|
630
|
+
except errors.FormatError:
|
|
631
|
+
return None, size
|
|
632
|
+
|
|
633
|
+
def _remap(self) -> None:
|
|
634
|
+
"""Maps the file again after its size changed.
|
|
635
|
+
|
|
636
|
+
The old map is not closed: a concurrent ``read`` may still hold a
|
|
637
|
+
slice of it, and it is unmapped when the last one goes away.
|
|
638
|
+
"""
|
|
639
|
+
assert self._path is not None
|
|
640
|
+
with open(self._path, "rb") as f:
|
|
641
|
+
mm = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
|
|
642
|
+
self._view, self._mm = memoryview(mm), mm
|
|
643
|
+
|
|
644
|
+
def _check_directory(
|
|
645
|
+
self, d: npt.NDArray, size: int, n_frames: int, n_blocks: int
|
|
646
|
+
) -> None:
|
|
647
|
+
"""Validates the directory against the header, vectorized."""
|
|
648
|
+
e, bf = self.n_envs, self.block_frames
|
|
649
|
+
n_windows = -(-n_frames // bf)
|
|
650
|
+
if n_blocks != n_windows * e:
|
|
651
|
+
raise errors.FormatError("block count does not match n_frames")
|
|
652
|
+
if not np.isin(d["codec"], list(codecs.BLOCK_CODEC_NAMES)).all():
|
|
653
|
+
bad = sorted(
|
|
654
|
+
set(d["codec"].tolist()) - set(codecs.BLOCK_CODEC_NAMES)
|
|
655
|
+
)
|
|
656
|
+
raise errors.FormatError(f"unknown block codec id {bad[0]}")
|
|
657
|
+
idx = np.arange(n_blocks)
|
|
658
|
+
t0 = (idx // e) * bf
|
|
659
|
+
n = np.minimum(bf, n_frames - t0)
|
|
660
|
+
k = self._k
|
|
661
|
+
ulen = np.where(
|
|
662
|
+
d["codec"] == codecs.CODEC_F32S, 4 * k * n, 8 * k + 2 * k * n
|
|
663
|
+
)
|
|
664
|
+
ok = (
|
|
665
|
+
(d["env"] == idx % e).all()
|
|
666
|
+
and (d["t0"] == t0).all()
|
|
667
|
+
and (d["n"] == n).all()
|
|
668
|
+
and (d["ulen"] == ulen).all()
|
|
669
|
+
and (d["offset"] % 8 == 0).all()
|
|
670
|
+
and (d["offset"] + 32 + d["clen"].astype(np.uint64) <= size).all()
|
|
671
|
+
)
|
|
672
|
+
if not ok:
|
|
673
|
+
raise errors.FormatError("block directory is inconsistent")
|
|
674
|
+
|
|
675
|
+
def _block(self, idx: int) -> npt.NDArray[np.float32]:
|
|
676
|
+
"""Returns the decoded block ``idx`` from the cache or the file."""
|
|
677
|
+
with self._lock:
|
|
678
|
+
hit = self._cache.get(idx)
|
|
679
|
+
if hit is not None:
|
|
680
|
+
self._cache.move_to_end(idx)
|
|
681
|
+
return hit
|
|
682
|
+
assert self._view is not None
|
|
683
|
+
ent = self.directory[idx]
|
|
684
|
+
off, clen, n = int(ent["offset"]), int(ent["clen"]), int(ent["n"])
|
|
685
|
+
head = BLOCK_HEADER.unpack_from(self._view, off)
|
|
686
|
+
if head[0] != BLOCK_MAGIC:
|
|
687
|
+
raise errors.FormatError(f"bad block magic at offset {off}")
|
|
688
|
+
codec = int(ent["codec"])
|
|
689
|
+
if (head[1], head[2], head[3], head[4], head[5]) != (
|
|
690
|
+
codec,
|
|
691
|
+
ent["env"],
|
|
692
|
+
ent["t0"],
|
|
693
|
+
n,
|
|
694
|
+
clen,
|
|
695
|
+
):
|
|
696
|
+
raise errors.FormatError(
|
|
697
|
+
f"block header at {off} disagrees with directory"
|
|
698
|
+
)
|
|
699
|
+
payload = self._view[off + 32 : off + 32 + clen]
|
|
700
|
+
try:
|
|
701
|
+
if self._verify and zlib.crc32(payload) != head[7]:
|
|
702
|
+
raise errors.FormatError(f"block CRC mismatch at offset {off}")
|
|
703
|
+
out = np.empty((n, self._k), np.float32)
|
|
704
|
+
codecs.decode_block_into(payload, codec, out)
|
|
705
|
+
finally:
|
|
706
|
+
payload.release()
|
|
707
|
+
if self._kind == "pose" and codec == codecs.CODEC_Q16D:
|
|
708
|
+
_renormalize_poses(out)
|
|
709
|
+
out.flags.writeable = False
|
|
710
|
+
with self._lock:
|
|
711
|
+
if self._cache_blocks:
|
|
712
|
+
self._cache[idx] = out
|
|
713
|
+
while len(self._cache) > self._cache_blocks:
|
|
714
|
+
self._cache.popitem(last=False)
|
|
715
|
+
return out
|
|
716
|
+
|
|
717
|
+
def read(
|
|
718
|
+
self,
|
|
719
|
+
t0: int,
|
|
720
|
+
t1: int,
|
|
721
|
+
envs: Sequence[int] | npt.NDArray[np.integer] | None = None,
|
|
722
|
+
) -> npt.NDArray[np.float32]:
|
|
723
|
+
"""Reads frames ``t0 <= t < t1``.
|
|
724
|
+
|
|
725
|
+
Args:
|
|
726
|
+
t0: First frame.
|
|
727
|
+
t1: One past the last frame.
|
|
728
|
+
envs: Env indices to read, or ``None`` for all envs.
|
|
729
|
+
|
|
730
|
+
Returns:
|
|
731
|
+
A new float32 array ``[t1 - t0, len(envs), *item_shape]``.
|
|
732
|
+
|
|
733
|
+
Raises:
|
|
734
|
+
IndexError: If the frame range or an env index is out of range.
|
|
735
|
+
ValueError: If the reader is closed.
|
|
736
|
+
"""
|
|
737
|
+
if self._view is None:
|
|
738
|
+
raise ValueError("reader is closed")
|
|
739
|
+
if not 0 <= t0 <= t1 <= self.n_frames:
|
|
740
|
+
raise IndexError(
|
|
741
|
+
f"frame range [{t0}, {t1}) outside [0, {self.n_frames}]"
|
|
742
|
+
)
|
|
743
|
+
env_ids = (
|
|
744
|
+
np.arange(self.n_envs)
|
|
745
|
+
if envs is None
|
|
746
|
+
else np.asarray(envs, dtype=np.int64).reshape(-1)
|
|
747
|
+
)
|
|
748
|
+
if env_ids.size and (env_ids.min() < 0 or env_ids.max() >= self.n_envs):
|
|
749
|
+
raise IndexError(f"env index outside [0, {self.n_envs})")
|
|
750
|
+
out = np.empty((t1 - t0, env_ids.size, self._k), np.float32)
|
|
751
|
+
bf = self.block_frames
|
|
752
|
+
for w in range(t0 // bf, -(-t1 // bf)):
|
|
753
|
+
lo, hi = max(t0, w * bf), min(t1, (w + 1) * bf)
|
|
754
|
+
for j, env in enumerate(env_ids.tolist()):
|
|
755
|
+
blk = self._block(w * self.n_envs + env)
|
|
756
|
+
out[lo - t0 : hi - t0, j] = blk[lo - w * bf : hi - w * bf]
|
|
757
|
+
return out.reshape(t1 - t0, env_ids.size, *self.item_shape)
|
|
758
|
+
|
|
759
|
+
def close(self) -> None:
|
|
760
|
+
"""Releases the memory map. Safe to call more than once."""
|
|
761
|
+
self._cache.clear()
|
|
762
|
+
if self._view is not None:
|
|
763
|
+
self._view.release()
|
|
764
|
+
self._view = None
|
|
765
|
+
if self._mm is not None:
|
|
766
|
+
self._mm.close()
|
|
767
|
+
self._mm = None
|
|
768
|
+
|
|
769
|
+
def __enter__(self) -> "BlockReader":
|
|
770
|
+
"""Returns the reader."""
|
|
771
|
+
return self
|
|
772
|
+
|
|
773
|
+
def __exit__(
|
|
774
|
+
self,
|
|
775
|
+
exc_type: type[BaseException] | None,
|
|
776
|
+
exc: BaseException | None,
|
|
777
|
+
tb: TracebackType | None,
|
|
778
|
+
) -> None:
|
|
779
|
+
"""Closes the reader."""
|
|
780
|
+
self.close()
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
@dataclasses.dataclass(frozen=True)
|
|
784
|
+
class RecoverResult:
|
|
785
|
+
"""Outcome of :func:`recover`.
|
|
786
|
+
|
|
787
|
+
Attributes:
|
|
788
|
+
n_frames: Frames kept.
|
|
789
|
+
n_blocks: Blocks kept.
|
|
790
|
+
dropped_bytes: Bytes removed from the end of the file (partial
|
|
791
|
+
blocks and incomplete windows).
|
|
792
|
+
already_finished: True if the file already had a directory and was
|
|
793
|
+
left untouched.
|
|
794
|
+
"""
|
|
795
|
+
|
|
796
|
+
n_frames: int
|
|
797
|
+
n_blocks: int
|
|
798
|
+
dropped_bytes: int
|
|
799
|
+
already_finished: bool = False
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
@dataclasses.dataclass
|
|
803
|
+
class _ScanState:
|
|
804
|
+
"""Resumable state of a walk over the blocks of an unfinished file.
|
|
805
|
+
|
|
806
|
+
Attributes:
|
|
807
|
+
off: Offset where the next block should start. It stays put at a
|
|
808
|
+
block that fails validation, so a later scan retries it.
|
|
809
|
+
end: Offset just after the last complete window.
|
|
810
|
+
frames: Frames in the complete windows.
|
|
811
|
+
env: Env of the next block within the current window.
|
|
812
|
+
win_n: Frames in the current window (set by its first block).
|
|
813
|
+
closed: True once a short window was kept: it can only be the last.
|
|
814
|
+
kept: Entries of complete windows not yet taken by the caller.
|
|
815
|
+
pending: Entries of the incomplete window being read.
|
|
816
|
+
"""
|
|
817
|
+
|
|
818
|
+
off: int = HEADER_SIZE
|
|
819
|
+
end: int = HEADER_SIZE
|
|
820
|
+
frames: int = 0
|
|
821
|
+
env: int = 0
|
|
822
|
+
win_n: int = 0
|
|
823
|
+
closed: bool = False
|
|
824
|
+
kept: list[DirEntry] = dataclasses.field(default_factory=list)
|
|
825
|
+
pending: list[DirEntry] = dataclasses.field(default_factory=list)
|
|
826
|
+
|
|
827
|
+
|
|
828
|
+
def _advance(view: memoryview, h: Header, st: _ScanState) -> None:
|
|
829
|
+
"""Continues a block walk, keeping only complete valid windows.
|
|
830
|
+
|
|
831
|
+
Checks each block's magic, lengths and CRC, and stops at the first one
|
|
832
|
+
that fails or is not fully present yet. Only the bytes after ``st.off``
|
|
833
|
+
are read.
|
|
834
|
+
|
|
835
|
+
Args:
|
|
836
|
+
view: The whole file as currently visible.
|
|
837
|
+
h: The parsed header.
|
|
838
|
+
st: Scan state, advanced in place.
|
|
839
|
+
"""
|
|
840
|
+
off, env, win_n, frames = st.off, st.env, st.win_n, st.frames
|
|
841
|
+
size = len(view)
|
|
842
|
+
k = h.k
|
|
843
|
+
while not st.closed and off + 32 <= size:
|
|
844
|
+
(magic, codec, b_env, b_t0, n, clen, ulen, crc) = (
|
|
845
|
+
BLOCK_HEADER.unpack_from(view, off)
|
|
846
|
+
)
|
|
847
|
+
stop = (
|
|
848
|
+
magic != BLOCK_MAGIC
|
|
849
|
+
or codec not in codecs.BLOCK_CODEC_NAMES
|
|
850
|
+
or b_env != env
|
|
851
|
+
or b_t0 != frames
|
|
852
|
+
or not 1 <= n <= h.block_frames
|
|
853
|
+
or (env > 0 and n != win_n)
|
|
854
|
+
or ulen != codecs.payload_ulen(codec, n, k)
|
|
855
|
+
or off + 32 + clen > size
|
|
856
|
+
)
|
|
857
|
+
if stop:
|
|
858
|
+
break
|
|
859
|
+
with view[off + 32 : off + 32 + clen] as payload:
|
|
860
|
+
if zlib.crc32(payload) != crc:
|
|
861
|
+
break
|
|
862
|
+
st.pending.append((off, env, b_t0, n, clen, ulen, codec))
|
|
863
|
+
win_n = n
|
|
864
|
+
off = _align8(off + 32 + clen)
|
|
865
|
+
if env == h.n_envs - 1:
|
|
866
|
+
st.kept += st.pending
|
|
867
|
+
st.pending = []
|
|
868
|
+
frames += n
|
|
869
|
+
st.end = off
|
|
870
|
+
env = 0
|
|
871
|
+
if n < h.block_frames:
|
|
872
|
+
st.closed = True # a short window can only be the last one
|
|
873
|
+
else:
|
|
874
|
+
env += 1
|
|
875
|
+
st.off, st.env, st.win_n, st.frames = off, env, win_n, frames
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
def _scan_blocks(
|
|
879
|
+
view: memoryview, h: Header
|
|
880
|
+
) -> tuple[list[DirEntry], int, int]:
|
|
881
|
+
"""Walks blocks from offset 64 and keeps only complete valid windows.
|
|
882
|
+
|
|
883
|
+
Args:
|
|
884
|
+
view: The whole file.
|
|
885
|
+
h: The parsed header.
|
|
886
|
+
|
|
887
|
+
Returns:
|
|
888
|
+
``(entries, n_frames, end_offset)`` for the complete windows. A
|
|
889
|
+
window is complete when all envs have a valid block.
|
|
890
|
+
"""
|
|
891
|
+
st = _ScanState()
|
|
892
|
+
_advance(view, h, st)
|
|
893
|
+
return st.kept, st.frames, st.end
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
def recover(path: os.PathLike[str] | str) -> RecoverResult:
|
|
897
|
+
"""Rebuilds the directory of an unfinished block file.
|
|
898
|
+
|
|
899
|
+
Walks blocks from offset 64, keeps every complete time window (a block
|
|
900
|
+
for every env) whose header and CRC check out, truncates the rest, then
|
|
901
|
+
appends the directory and rewrites the header. A finished file is left
|
|
902
|
+
untouched.
|
|
903
|
+
|
|
904
|
+
Args:
|
|
905
|
+
path: The block file.
|
|
906
|
+
|
|
907
|
+
Returns:
|
|
908
|
+
What was kept and dropped.
|
|
909
|
+
|
|
910
|
+
Raises:
|
|
911
|
+
errors.FormatError: If the file header is missing or corrupt.
|
|
912
|
+
"""
|
|
913
|
+
path = pathlib.Path(path)
|
|
914
|
+
with open(path, "r+b") as f:
|
|
915
|
+
size = os.fstat(f.fileno()).st_size
|
|
916
|
+
if size < HEADER_SIZE:
|
|
917
|
+
raise errors.FormatError(f"{path}: shorter than header")
|
|
918
|
+
with mmap.mmap(f.fileno(), 0) as mm, memoryview(mm) as view:
|
|
919
|
+
h = Header.parse(view)
|
|
920
|
+
if h.dir_offset != 0:
|
|
921
|
+
blocks = h.n_blocks
|
|
922
|
+
return RecoverResult(h.n_frames, blocks, 0, True)
|
|
923
|
+
entries, frames, end = _scan_blocks(view, h)
|
|
924
|
+
dropped = max(0, size - end)
|
|
925
|
+
f.truncate(end)
|
|
926
|
+
directory = _dir_bytes(entries)
|
|
927
|
+
f.seek(end)
|
|
928
|
+
f.write(directory)
|
|
929
|
+
header = dataclasses.replace(
|
|
930
|
+
h,
|
|
931
|
+
n_frames=frames,
|
|
932
|
+
n_blocks=len(entries),
|
|
933
|
+
dir_offset=end,
|
|
934
|
+
dir_length=len(directory),
|
|
935
|
+
)
|
|
936
|
+
f.seek(0)
|
|
937
|
+
f.write(header.pack())
|
|
938
|
+
return RecoverResult(frames, len(entries), dropped)
|