hbkit 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of hbkit might be problematic. Click here for more details.
- {hbkit-0.2.0 → hbkit-0.2.1}/PKG-INFO +28 -4
- {hbkit-0.2.0 → hbkit-0.2.1}/README.md +27 -3
- {hbkit-0.2.0 → hbkit-0.2.1}/pyproject.toml +1 -1
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/__init__.py +1 -1
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/archive.py +61 -17
- {hbkit-0.2.0 → hbkit-0.2.1}/tests/test_hbkit.py +56 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/.gitignore +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/FORMAT.md +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/LICENSE +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/cli.py +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/crypto.py +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/doctor.py +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/index.py +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/runner.py +0 -0
- {hbkit-0.2.0 → hbkit-0.2.1}/src/hbkit/tui.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hbkit
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Recover files from Synology Hyper Backup (.hbk) archives without Synology software
|
|
5
5
|
Project-URL: Homepage, https://github.com/YordiLorenzo/hbkit
|
|
6
6
|
Project-URL: Source, https://github.com/YordiLorenzo/hbkit
|
|
@@ -35,9 +35,9 @@ Description-Content-Type: text/markdown
|
|
|
35
35
|
|
|
36
36
|
Recover files from **Synology Hyper Backup (`.hbk`)** archives without any Synology software.
|
|
37
37
|
|
|
38
|
-
Point it at a backup
|
|
39
|
-
|
|
40
|
-
|
|
38
|
+
Point it at a backup on a local disk, an external drive, or a network mount, browse it as
|
|
39
|
+
a tree, and pull out what you want. Works headless on Linux and macOS, including Apple Silicon, where Synology's
|
|
40
|
+
own Hyper Backup Explorer is awkward or unavailable.
|
|
41
41
|
|
|
42
42
|
```sh
|
|
43
43
|
brew install lz4 # or: sudo apt install liblz4-1
|
|
@@ -141,6 +141,30 @@ readable by rebuilding a random sample of real files with full checksum verifica
|
|
|
141
141
|
- **Index cached** per archive in `~/.cache/hbkit`, rebuilt automatically when the archive
|
|
142
142
|
changes. Browsing 1.1M files is instant after the first open.
|
|
143
143
|
|
|
144
|
+
## Network mounts (rclone / S3 / R2)
|
|
145
|
+
|
|
146
|
+
Opening an archive no longer measures every index shard. Shards are a fixed 8 MiB except
|
|
147
|
+
the last, so offsets are computed instead — one directory listing and a single stat per
|
|
148
|
+
index family, and `file_chunk<N>.index` families are opened only if a file references them.
|
|
149
|
+
On a 3 TB archive that removed roughly **2,600 network round-trips** from startup.
|
|
150
|
+
|
|
151
|
+
If you do mount a bucket, the flags matter more than anything hbkit does:
|
|
152
|
+
|
|
153
|
+
```sh
|
|
154
|
+
rclone mount r2:bucket ~/mnt/r2 --read-only \
|
|
155
|
+
--vfs-cache-mode off \ # range requests; 'full' downloads whole files
|
|
156
|
+
--dir-cache-time 72h # first listing of ~2000 shards is slow, then cached
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
**Do not use `--vfs-cache-mode full`.** hbkit reads a 32-byte index record and a ~5 KB
|
|
160
|
+
chunk at a time; in `full` mode each of those pulls an entire file, so a 10 KB extraction
|
|
161
|
+
downloaded ~82 MB (an 8 MiB index shard plus a ~50 MB bucket) and took ten minutes.
|
|
162
|
+
|
|
163
|
+
Even configured well, a network mount is dramatically slower than local storage — the
|
|
164
|
+
access pattern is thousands of small scattered reads. **If you can, copy the archive to a
|
|
165
|
+
local disk first.** Treat mounted-bucket recovery as workable for pulling out a handful of
|
|
166
|
+
files, not for restoring terabytes.
|
|
167
|
+
|
|
144
168
|
## Performance
|
|
145
169
|
|
|
146
170
|
Use `-j` to set worker processes (default 8). Threads do not help — extraction is
|
|
@@ -6,9 +6,9 @@
|
|
|
6
6
|
|
|
7
7
|
Recover files from **Synology Hyper Backup (`.hbk`)** archives without any Synology software.
|
|
8
8
|
|
|
9
|
-
Point it at a backup
|
|
10
|
-
|
|
11
|
-
|
|
9
|
+
Point it at a backup on a local disk, an external drive, or a network mount, browse it as
|
|
10
|
+
a tree, and pull out what you want. Works headless on Linux and macOS, including Apple Silicon, where Synology's
|
|
11
|
+
own Hyper Backup Explorer is awkward or unavailable.
|
|
12
12
|
|
|
13
13
|
```sh
|
|
14
14
|
brew install lz4 # or: sudo apt install liblz4-1
|
|
@@ -112,6 +112,30 @@ readable by rebuilding a random sample of real files with full checksum verifica
|
|
|
112
112
|
- **Index cached** per archive in `~/.cache/hbkit`, rebuilt automatically when the archive
|
|
113
113
|
changes. Browsing 1.1M files is instant after the first open.
|
|
114
114
|
|
|
115
|
+
## Network mounts (rclone / S3 / R2)
|
|
116
|
+
|
|
117
|
+
Opening an archive no longer measures every index shard. Shards are a fixed 8 MiB except
|
|
118
|
+
the last, so offsets are computed instead — one directory listing and a single stat per
|
|
119
|
+
index family, and `file_chunk<N>.index` families are opened only if a file references them.
|
|
120
|
+
On a 3 TB archive that removed roughly **2,600 network round-trips** from startup.
|
|
121
|
+
|
|
122
|
+
If you do mount a bucket, the flags matter more than anything hbkit does:
|
|
123
|
+
|
|
124
|
+
```sh
|
|
125
|
+
rclone mount r2:bucket ~/mnt/r2 --read-only \
|
|
126
|
+
--vfs-cache-mode off \ # range requests; 'full' downloads whole files
|
|
127
|
+
--dir-cache-time 72h # first listing of ~2000 shards is slow, then cached
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
**Do not use `--vfs-cache-mode full`.** hbkit reads a 32-byte index record and a ~5 KB
|
|
131
|
+
chunk at a time; in `full` mode each of those pulls an entire file, so a 10 KB extraction
|
|
132
|
+
downloaded ~82 MB (an 8 MiB index shard plus a ~50 MB bucket) and took ten minutes.
|
|
133
|
+
|
|
134
|
+
Even configured well, a network mount is dramatically slower than local storage — the
|
|
135
|
+
access pattern is thousands of small scattered reads. **If you can, copy the archive to a
|
|
136
|
+
local disk first.** Treat mounted-bucket recovery as workable for pulling out a handful of
|
|
137
|
+
files, not for restoring terabytes.
|
|
138
|
+
|
|
115
139
|
## Performance
|
|
116
140
|
|
|
117
141
|
Use `-j` to set worker processes (default 8). Threads do not help — extraction is
|
|
@@ -45,6 +45,7 @@ BUCKET_BUF = int(os.environ.get("HBK_BUCKET_BUF", 1 << 22)) # read-ahead; plat
|
|
|
45
45
|
INDEX_BUF = int(os.environ.get("HBK_INDEX_BUF", 1 << 16)) # chunk keys are mostly sequential;
|
|
46
46
|
# unbuffered = one syscall per 29 B
|
|
47
47
|
MAGIC = 0x7053A86E
|
|
48
|
+
SHARD_SIZE = 8 << 20 # every index shard is exactly 8 MiB except the last
|
|
48
49
|
|
|
49
50
|
_LZ4_CANDIDATES = [
|
|
50
51
|
os.environ.get("HBK_LZ4"),
|
|
@@ -112,7 +113,14 @@ def resolve(d: str, base: str) -> str:
|
|
|
112
113
|
|
|
113
114
|
|
|
114
115
|
class Cat:
|
|
115
|
-
"""Numbered <N>.idx[.gen] shards presented as a single logical byte stream.
|
|
116
|
+
"""Numbered <N>.idx[.gen] shards presented as a single logical byte stream.
|
|
117
|
+
|
|
118
|
+
Shards are a fixed SHARD_SIZE except the last, so offsets are computed rather than
|
|
119
|
+
measured. That matters enormously on network mounts: stat-ing every shard cost ~2,600
|
|
120
|
+
round-trips on a 3 TB archive (minutes before reading a byte). We now do one directory
|
|
121
|
+
listing plus a single stat of the final shard. If the assumption were ever wrong, a
|
|
122
|
+
short read surfaces as a chunk MD5 failure - loudly - never as silent corruption.
|
|
123
|
+
"""
|
|
116
124
|
|
|
117
125
|
def __init__(self, d: str):
|
|
118
126
|
self.d = d
|
|
@@ -129,12 +137,11 @@ class Cat:
|
|
|
129
137
|
if not found:
|
|
130
138
|
raise FileNotFoundError(f"no .idx shards in {d}")
|
|
131
139
|
self.files = [found[k][1] for k in sorted(found)]
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
self.total = t
|
|
140
|
+
n = len(self.files)
|
|
141
|
+
last = os.path.getsize(os.path.join(d, self.files[-1]))
|
|
142
|
+
self.sizes = [SHARD_SIZE] * (n - 1) + [last]
|
|
143
|
+
self.starts = [i * SHARD_SIZE for i in range(n)]
|
|
144
|
+
self.total = (n - 1) * SHARD_SIZE + last
|
|
138
145
|
self._fh: dict[int, object] = {}
|
|
139
146
|
# Shard 0 carries a 64-byte header that declares the record size, so readers do
|
|
140
147
|
# not have to hardcode per-DSM-version layouts.
|
|
@@ -150,16 +157,26 @@ class Cat:
|
|
|
150
157
|
self.declared_total = (w[5] << 32) | w[6]
|
|
151
158
|
|
|
152
159
|
def read(self, off: int, n: int) -> bytes:
|
|
160
|
+
"""Read n bytes at logical offset off, spanning shards as needed."""
|
|
153
161
|
out = b""
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
162
|
+
i = off // SHARD_SIZE
|
|
163
|
+
while n > 0 and i < len(self.files):
|
|
164
|
+
st = self.starts[i]
|
|
165
|
+
take = min(n, st + self.sizes[i] - off)
|
|
166
|
+
if take <= 0:
|
|
167
|
+
break
|
|
158
168
|
fh = self._fh.get(i)
|
|
159
169
|
if fh is None:
|
|
160
|
-
fh = self._fh[i] = open(os.path.join(self.d,
|
|
161
|
-
|
|
162
|
-
|
|
170
|
+
fh = self._fh[i] = open(os.path.join(self.d, self.files[i]), "rb",
|
|
171
|
+
buffering=INDEX_BUF)
|
|
172
|
+
fh.seek(off - st)
|
|
173
|
+
got = fh.read(take)
|
|
174
|
+
out += got
|
|
175
|
+
if len(got) < take: # short shard: stop rather than mis-align
|
|
176
|
+
break
|
|
177
|
+
off += take
|
|
178
|
+
n -= take
|
|
179
|
+
i += 1
|
|
163
180
|
return out
|
|
164
181
|
|
|
165
182
|
def close(self):
|
|
@@ -168,6 +185,32 @@ class Cat:
|
|
|
168
185
|
self._fh.clear()
|
|
169
186
|
|
|
170
187
|
|
|
188
|
+
class _LazyCats(dict):
|
|
189
|
+
"""file_chunk<N>.index families, opened only when a file actually references one."""
|
|
190
|
+
|
|
191
|
+
def __init__(self, dirs: dict[int, str]):
|
|
192
|
+
super().__init__()
|
|
193
|
+
self._dirs = dirs
|
|
194
|
+
|
|
195
|
+
def get(self, k, default=None):
|
|
196
|
+
if k in self:
|
|
197
|
+
return self[k]
|
|
198
|
+
d = self._dirs.get(k)
|
|
199
|
+
if d is None:
|
|
200
|
+
return default
|
|
201
|
+
self[k] = c = Cat(d)
|
|
202
|
+
return c
|
|
203
|
+
|
|
204
|
+
def __missing__(self, k):
|
|
205
|
+
c = self.get(k)
|
|
206
|
+
if c is None:
|
|
207
|
+
raise KeyError(k)
|
|
208
|
+
return c
|
|
209
|
+
|
|
210
|
+
def values(self):
|
|
211
|
+
return [v for v in dict.values(self)]
|
|
212
|
+
|
|
213
|
+
|
|
171
214
|
def is_archive(d: str) -> bool:
|
|
172
215
|
return os.path.isdir(os.path.join(d, "Pool")) and os.path.isdir(os.path.join(d, "Config"))
|
|
173
216
|
|
|
@@ -198,13 +241,14 @@ class Archive:
|
|
|
198
241
|
c, p = f"{self.root}/Config", f"{self.root}/Pool"
|
|
199
242
|
self.vf = Cat(f"{c}/virtual_file.index")
|
|
200
243
|
self.ci = Cat(f"{p}/chunk_index")
|
|
201
|
-
self.
|
|
244
|
+
self._fc_dirs: dict[int, str] = {}
|
|
202
245
|
for e in os.listdir(c):
|
|
203
246
|
m = re.match(r"^file_chunk(\d+)\.index$", e)
|
|
204
247
|
if m and os.path.isdir(os.path.join(c, e)):
|
|
205
|
-
self.
|
|
206
|
-
if not self.
|
|
248
|
+
self._fc_dirs[int(m.group(1))] = os.path.join(c, e)
|
|
249
|
+
if not self._fc_dirs:
|
|
207
250
|
raise FileNotFoundError(f"no file_chunk<N>.index directories in {c}")
|
|
251
|
+
self.fc: dict[int, Cat] = _LazyCats(self._fc_dirs)
|
|
208
252
|
pools = [e for e in os.listdir(p) if e.isdigit() and os.path.isdir(os.path.join(p, e))]
|
|
209
253
|
if not pools:
|
|
210
254
|
raise FileNotFoundError(f"no numbered pool directory under {p}")
|
|
@@ -229,3 +229,59 @@ def test_encrypted_names_and_content():
|
|
|
229
229
|
data = arc.extract(ovf, size, verify=True) # verify=True checks every chunk MD5
|
|
230
230
|
assert len(data) == size, path
|
|
231
231
|
arc.close()
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# ------------------------------------------------- shard maths (no archive needed)
|
|
235
|
+
|
|
236
|
+
def _make_shards(tmp_path, sizes, shard_size, monkeypatch):
|
|
237
|
+
"""Write synthetic <N>.idx.2 shards. Shard 0 carries a valid 64-byte header, as real
|
|
238
|
+
archives do; the header is part of the logical stream so offsets stay comparable."""
|
|
239
|
+
import struct
|
|
240
|
+
|
|
241
|
+
from hbkit import archive as A
|
|
242
|
+
monkeypatch.setattr(A, "SHARD_SIZE", shard_size)
|
|
243
|
+
total = sum(sizes)
|
|
244
|
+
header = struct.pack(">16I", A.MAGIC, 0, 0, 0, 32, total >> 32, total & 0xFFFFFFFF,
|
|
245
|
+
*([0] * 9))
|
|
246
|
+
blob = b""
|
|
247
|
+
for i, n in enumerate(sizes):
|
|
248
|
+
if i == 0:
|
|
249
|
+
part = header + bytes((64 + j) % 251 for j in range(n - 64))
|
|
250
|
+
else:
|
|
251
|
+
part = bytes((len(blob) + j) % 251 for j in range(n))
|
|
252
|
+
(tmp_path / f"{i}.idx.2").write_bytes(part)
|
|
253
|
+
blob += part
|
|
254
|
+
return blob
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def test_cat_computes_offsets_without_stating_every_shard(tmp_path, monkeypatch):
|
|
258
|
+
"""Sizes/offsets must be derived from the fixed shard size + one final stat."""
|
|
259
|
+
from hbkit import archive as A
|
|
260
|
+
blob = _make_shards(tmp_path, [128, 128, 10], 128, monkeypatch)
|
|
261
|
+
c = A.Cat(str(tmp_path))
|
|
262
|
+
assert len(c.files) == 3
|
|
263
|
+
assert c.total == len(blob) == 266
|
|
264
|
+
assert c.starts == [0, 128, 256]
|
|
265
|
+
assert c.sizes == [128, 128, 10]
|
|
266
|
+
assert c.record_size == 32 # read from the header, not guessed
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def test_cat_reads_across_shard_boundaries(tmp_path, monkeypatch):
|
|
270
|
+
"""The multi-shard read path: every offset/length combination must match one blob."""
|
|
271
|
+
from hbkit import archive as A
|
|
272
|
+
blob = _make_shards(tmp_path, [128, 128, 10], 128, monkeypatch)
|
|
273
|
+
c = A.Cat(str(tmp_path))
|
|
274
|
+
for off in range(0, len(blob), 7):
|
|
275
|
+
for n in (1, 5, 127, 128, 129, 260):
|
|
276
|
+
assert c.read(off, n) == blob[off:off + n], f"off={off} n={n}"
|
|
277
|
+
assert c.read(0, len(blob)) == blob # whole stream in one read
|
|
278
|
+
assert c.read(len(blob), 10) == b"" # past the end
|
|
279
|
+
c.close()
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def test_cat_single_shard_still_works(tmp_path, monkeypatch):
|
|
283
|
+
from hbkit import archive as A
|
|
284
|
+
blob = _make_shards(tmp_path, [100], 128, monkeypatch)
|
|
285
|
+
c = A.Cat(str(tmp_path))
|
|
286
|
+
assert c.total == 100 and c.read(70, 20) == blob[70:90]
|
|
287
|
+
c.close()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|