hbkit 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of hbkit might be problematic. Click here for more details.
- {hbkit-0.2.0 → hbkit-0.3.0}/PKG-INFO +70 -4
- {hbkit-0.2.0 → hbkit-0.3.0}/README.md +69 -3
- {hbkit-0.2.0 → hbkit-0.3.0}/pyproject.toml +1 -1
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/__init__.py +1 -1
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/archive.py +61 -17
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/cli.py +46 -1
- hbkit-0.3.0/src/hbkit/mount.py +174 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/tests/test_hbkit.py +109 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/.gitignore +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/FORMAT.md +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/LICENSE +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/crypto.py +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/doctor.py +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/index.py +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/runner.py +0 -0
- {hbkit-0.2.0 → hbkit-0.3.0}/src/hbkit/tui.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hbkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Recover files from Synology Hyper Backup (.hbk) archives without Synology software
|
|
5
5
|
Project-URL: Homepage, https://github.com/YordiLorenzo/hbkit
|
|
6
6
|
Project-URL: Source, https://github.com/YordiLorenzo/hbkit
|
|
@@ -35,9 +35,9 @@ Description-Content-Type: text/markdown
|
|
|
35
35
|
|
|
36
36
|
Recover files from **Synology Hyper Backup (`.hbk`)** archives without any Synology software.
|
|
37
37
|
|
|
38
|
-
Point it at a backup
|
|
39
|
-
|
|
40
|
-
|
|
38
|
+
Point it at a backup on a local disk, an external drive, or a network mount, browse it as
|
|
39
|
+
a tree, and pull out what you want. Works headless on Linux and macOS, including Apple Silicon, where Synology's
|
|
40
|
+
own Hyper Backup Explorer is awkward or unavailable.
|
|
41
41
|
|
|
42
42
|
```sh
|
|
43
43
|
brew install lz4 # or: sudo apt install liblz4-1
|
|
@@ -141,6 +141,72 @@ readable by rebuilding a random sample of real files with full checksum verifica
|
|
|
141
141
|
- **Index cached** per archive in `~/.cache/hbkit`, rebuilt automatically when the archive
|
|
142
142
|
changes. Browsing 1.1M files is instant after the first open.
|
|
143
143
|
|
|
144
|
+
## Network mounts (rclone / S3 / R2)
|
|
145
|
+
|
|
146
|
+
You can point hbkit at a bucket mounted with rclone instead of copying the archive down
|
|
147
|
+
first. Opening an archive no longer measures every index shard — shards are a fixed 8 MiB
|
|
148
|
+
except the last, so offsets are computed: one directory listing and a single stat per index
|
|
149
|
+
family, and `file_chunk<N>.index` families open only if a file references them. On a 3 TB
|
|
150
|
+
archive that removed roughly **2,600 network round-trips** from startup.
|
|
151
|
+
|
|
152
|
+
### Setting up the mount
|
|
153
|
+
|
|
154
|
+
hbkit can drive `rclone` for you. It stores no credentials and talks to no cloud API —
|
|
155
|
+
`rclone config` still owns all of that — it just picks flags that are easy to get wrong.
|
|
156
|
+
|
|
157
|
+
```sh
|
|
158
|
+
rclone config # add an s3 remote; for R2 pick "Cloudflare"
|
|
159
|
+
hbk remotes # what's configured
|
|
160
|
+
hbk mount r2:mybucket ~/mnt/backup --for browse # mount + pre-warm
|
|
161
|
+
hbk ~/mnt/backup/target.hbk doctor
|
|
162
|
+
hbk unmount ~/mnt/backup
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
`--for browse` (default) uses range requests; `--for restore` uses whole-file fetching as
|
|
166
|
+
read-ahead with the cache capped at 50G (`--cache-size 10G` to change it). `--no-warm`
|
|
167
|
+
skips pre-warming, `--dry-run` just prints the rclone command.
|
|
168
|
+
|
|
169
|
+
Measured on a 3 TB archive in Cloudflare R2:
|
|
170
|
+
|
|
171
|
+
| | before tuning | with `hbk mount` |
|
|
172
|
+
|---|---|---|
|
|
173
|
+
| cold mount + directory pre-warm | ~12 min | **16 s** |
|
|
174
|
+
| opening the archive | 9 min 21 s | **1.8 s** |
|
|
175
|
+
| extracting a 10 KB file | 584 s | **11 s** |
|
|
176
|
+
|
|
177
|
+
Three things get you there, and you can apply them by hand if you'd rather not use
|
|
178
|
+
`hbk mount`:
|
|
179
|
+
|
|
180
|
+
- **`--no-modtime`** — the big one. Without it rclone issues a HEAD per object just to fill
|
|
181
|
+
in modification times for a listing: ~0.27 s per entry, so the 2,021-shard `chunk_index`
|
|
182
|
+
directory took **546.8 s**. With it, **4.07 s** — 134× faster. hbkit never uses the
|
|
183
|
+
modtimes of an archive's internal files (restored mtimes come from the archive's own
|
|
184
|
+
metadata), so this costs nothing.
|
|
185
|
+
- **`--dir-cache-time 72h`** — the first listing of a large shard directory is the expensive
|
|
186
|
+
one; cache it and every later open is instant.
|
|
187
|
+
- **pre-warming** — `hbk mount` walks the index directories once up front, so the wait
|
|
188
|
+
happens visibly at mount time instead of looking like a hang on your first command. Run
|
|
189
|
+
it separately with `hbk warm ~/mnt/backup`.
|
|
190
|
+
|
|
191
|
+
### Choosing `--vfs-cache-mode` — it depends on what you're doing
|
|
192
|
+
|
|
193
|
+
`hbk mount --for` picks this for you, but if you mount by hand: rclone's `full` mode
|
|
194
|
+
downloads **whole files** while `off` issues **range requests**.
|
|
195
|
+
|
|
196
|
+
| What you're doing | Mode | Why |
|
|
197
|
+
|---|---|---|
|
|
198
|
+
| First index build | `full` | Reads one share database end to end (356 MB on a 1.1M-file archive). Sequential — what whole-file fetching is good at. One time; cached in `~/.cache/hbkit` after. |
|
|
199
|
+
| Browsing, `list`, the TUI tree | either | Served from the local index, no network at all. |
|
|
200
|
+
| Pulling out a few files | `off` | hbkit reads a 32-byte index record and a ~5 KB chunk at a time. In `full` mode each pulls a whole file: a measured 10 KB extraction fetched ~82 MB and took ten minutes. |
|
|
201
|
+
| Bulk restoring a folder | `full` | You touch most of each ~50 MB bucket anyway, so whole-file fetching becomes read-ahead. **Cap it** with `--vfs-cache-max-size`, or it will fill your disk. |
|
|
202
|
+
|
|
203
|
+
### Expectations
|
|
204
|
+
|
|
205
|
+
Even configured well, a mounted bucket is dramatically slower than local storage — the
|
|
206
|
+
access pattern is thousands of small scattered reads, and each one crosses the network.
|
|
207
|
+
**If you can, copy the archive to a local disk first.** Start with one small folder and
|
|
208
|
+
watch the throughput before committing to a large restore.
|
|
209
|
+
|
|
144
210
|
## Performance
|
|
145
211
|
|
|
146
212
|
Use `-j` to set worker processes (default 8). Threads do not help — extraction is
|
|
@@ -6,9 +6,9 @@
|
|
|
6
6
|
|
|
7
7
|
Recover files from **Synology Hyper Backup (`.hbk`)** archives without any Synology software.
|
|
8
8
|
|
|
9
|
-
Point it at a backup
|
|
10
|
-
|
|
11
|
-
|
|
9
|
+
Point it at a backup on a local disk, an external drive, or a network mount, browse it as
|
|
10
|
+
a tree, and pull out what you want. Works headless on Linux and macOS, including Apple Silicon, where Synology's
|
|
11
|
+
own Hyper Backup Explorer is awkward or unavailable.
|
|
12
12
|
|
|
13
13
|
```sh
|
|
14
14
|
brew install lz4 # or: sudo apt install liblz4-1
|
|
@@ -112,6 +112,72 @@ readable by rebuilding a random sample of real files with full checksum verifica
|
|
|
112
112
|
- **Index cached** per archive in `~/.cache/hbkit`, rebuilt automatically when the archive
|
|
113
113
|
changes. Browsing 1.1M files is instant after the first open.
|
|
114
114
|
|
|
115
|
+
## Network mounts (rclone / S3 / R2)
|
|
116
|
+
|
|
117
|
+
You can point hbkit at a bucket mounted with rclone instead of copying the archive down
|
|
118
|
+
first. Opening an archive no longer measures every index shard — shards are a fixed 8 MiB
|
|
119
|
+
except the last, so offsets are computed: one directory listing and a single stat per index
|
|
120
|
+
family, and `file_chunk<N>.index` families open only if a file references them. On a 3 TB
|
|
121
|
+
archive that removed roughly **2,600 network round-trips** from startup.
|
|
122
|
+
|
|
123
|
+
### Setting up the mount
|
|
124
|
+
|
|
125
|
+
hbkit can drive `rclone` for you. It stores no credentials and talks to no cloud API —
|
|
126
|
+
`rclone config` still owns all of that — it just picks flags that are easy to get wrong.
|
|
127
|
+
|
|
128
|
+
```sh
|
|
129
|
+
rclone config # add an s3 remote; for R2 pick "Cloudflare"
|
|
130
|
+
hbk remotes # what's configured
|
|
131
|
+
hbk mount r2:mybucket ~/mnt/backup --for browse # mount + pre-warm
|
|
132
|
+
hbk ~/mnt/backup/target.hbk doctor
|
|
133
|
+
hbk unmount ~/mnt/backup
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`--for browse` (default) uses range requests; `--for restore` uses whole-file fetching as
|
|
137
|
+
read-ahead with the cache capped at 50G (`--cache-size 10G` to change it). `--no-warm`
|
|
138
|
+
skips pre-warming, `--dry-run` just prints the rclone command.
|
|
139
|
+
|
|
140
|
+
Measured on a 3 TB archive in Cloudflare R2:
|
|
141
|
+
|
|
142
|
+
| | before tuning | with `hbk mount` |
|
|
143
|
+
|---|---|---|
|
|
144
|
+
| cold mount + directory pre-warm | ~12 min | **16 s** |
|
|
145
|
+
| opening the archive | 9 min 21 s | **1.8 s** |
|
|
146
|
+
| extracting a 10 KB file | 584 s | **11 s** |
|
|
147
|
+
|
|
148
|
+
Three things get you there, and you can apply them by hand if you'd rather not use
|
|
149
|
+
`hbk mount`:
|
|
150
|
+
|
|
151
|
+
- **`--no-modtime`** — the big one. Without it rclone issues a HEAD per object just to fill
|
|
152
|
+
in modification times for a listing: ~0.27 s per entry, so the 2,021-shard `chunk_index`
|
|
153
|
+
directory took **546.8 s**. With it, **4.07 s** — 134× faster. hbkit never uses the
|
|
154
|
+
modtimes of an archive's internal files (restored mtimes come from the archive's own
|
|
155
|
+
metadata), so this costs nothing.
|
|
156
|
+
- **`--dir-cache-time 72h`** — the first listing of a large shard directory is the expensive
|
|
157
|
+
one; cache it and every later open is instant.
|
|
158
|
+
- **pre-warming** — `hbk mount` walks the index directories once up front, so the wait
|
|
159
|
+
happens visibly at mount time instead of looking like a hang on your first command. Run
|
|
160
|
+
it separately with `hbk warm ~/mnt/backup`.
|
|
161
|
+
|
|
162
|
+
### Choosing `--vfs-cache-mode` — it depends on what you're doing
|
|
163
|
+
|
|
164
|
+
`hbk mount --for` picks this for you, but if you mount by hand: rclone's `full` mode
|
|
165
|
+
downloads **whole files** while `off` issues **range requests**.
|
|
166
|
+
|
|
167
|
+
| What you're doing | Mode | Why |
|
|
168
|
+
|---|---|---|
|
|
169
|
+
| First index build | `full` | Reads one share database end to end (356 MB on a 1.1M-file archive). Sequential — what whole-file fetching is good at. One time; cached in `~/.cache/hbkit` after. |
|
|
170
|
+
| Browsing, `list`, the TUI tree | either | Served from the local index, no network at all. |
|
|
171
|
+
| Pulling out a few files | `off` | hbkit reads a 32-byte index record and a ~5 KB chunk at a time. In `full` mode each pulls a whole file: a measured 10 KB extraction fetched ~82 MB and took ten minutes. |
|
|
172
|
+
| Bulk restoring a folder | `full` | You touch most of each ~50 MB bucket anyway, so whole-file fetching becomes read-ahead. **Cap it** with `--vfs-cache-max-size`, or it will fill your disk. |
|
|
173
|
+
|
|
174
|
+
### Expectations
|
|
175
|
+
|
|
176
|
+
Even configured well, a mounted bucket is dramatically slower than local storage — the
|
|
177
|
+
access pattern is thousands of small scattered reads, and each one crosses the network.
|
|
178
|
+
**If you can, copy the archive to a local disk first.** Start with one small folder and
|
|
179
|
+
watch the throughput before committing to a large restore.
|
|
180
|
+
|
|
115
181
|
## Performance
|
|
116
182
|
|
|
117
183
|
Use `-j` to set worker processes (default 8). Threads do not help — extraction is
|
|
@@ -45,6 +45,7 @@ BUCKET_BUF = int(os.environ.get("HBK_BUCKET_BUF", 1 << 22)) # read-ahead; plat
|
|
|
45
45
|
INDEX_BUF = int(os.environ.get("HBK_INDEX_BUF", 1 << 16)) # chunk keys are mostly sequential;
|
|
46
46
|
# unbuffered = one syscall per 29 B
|
|
47
47
|
MAGIC = 0x7053A86E
|
|
48
|
+
SHARD_SIZE = 8 << 20 # every index shard is exactly 8 MiB except the last
|
|
48
49
|
|
|
49
50
|
_LZ4_CANDIDATES = [
|
|
50
51
|
os.environ.get("HBK_LZ4"),
|
|
@@ -112,7 +113,14 @@ def resolve(d: str, base: str) -> str:
|
|
|
112
113
|
|
|
113
114
|
|
|
114
115
|
class Cat:
|
|
115
|
-
"""Numbered <N>.idx[.gen] shards presented as a single logical byte stream.
|
|
116
|
+
"""Numbered <N>.idx[.gen] shards presented as a single logical byte stream.
|
|
117
|
+
|
|
118
|
+
Shards are a fixed SHARD_SIZE except the last, so offsets are computed rather than
|
|
119
|
+
measured. That matters enormously on network mounts: stat-ing every shard cost ~2,600
|
|
120
|
+
round-trips on a 3 TB archive (minutes before reading a byte). We now do one directory
|
|
121
|
+
listing plus a single stat of the final shard. If the assumption were ever wrong, a
|
|
122
|
+
short read surfaces as a chunk MD5 failure - loudly - never as silent corruption.
|
|
123
|
+
"""
|
|
116
124
|
|
|
117
125
|
def __init__(self, d: str):
|
|
118
126
|
self.d = d
|
|
@@ -129,12 +137,11 @@ class Cat:
|
|
|
129
137
|
if not found:
|
|
130
138
|
raise FileNotFoundError(f"no .idx shards in {d}")
|
|
131
139
|
self.files = [found[k][1] for k in sorted(found)]
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
self.total = t
|
|
140
|
+
n = len(self.files)
|
|
141
|
+
last = os.path.getsize(os.path.join(d, self.files[-1]))
|
|
142
|
+
self.sizes = [SHARD_SIZE] * (n - 1) + [last]
|
|
143
|
+
self.starts = [i * SHARD_SIZE for i in range(n)]
|
|
144
|
+
self.total = (n - 1) * SHARD_SIZE + last
|
|
138
145
|
self._fh: dict[int, object] = {}
|
|
139
146
|
# Shard 0 carries a 64-byte header that declares the record size, so readers do
|
|
140
147
|
# not have to hardcode per-DSM-version layouts.
|
|
@@ -150,16 +157,26 @@ class Cat:
|
|
|
150
157
|
self.declared_total = (w[5] << 32) | w[6]
|
|
151
158
|
|
|
152
159
|
def read(self, off: int, n: int) -> bytes:
|
|
160
|
+
"""Read n bytes at logical offset off, spanning shards as needed."""
|
|
153
161
|
out = b""
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
162
|
+
i = off // SHARD_SIZE
|
|
163
|
+
while n > 0 and i < len(self.files):
|
|
164
|
+
st = self.starts[i]
|
|
165
|
+
take = min(n, st + self.sizes[i] - off)
|
|
166
|
+
if take <= 0:
|
|
167
|
+
break
|
|
158
168
|
fh = self._fh.get(i)
|
|
159
169
|
if fh is None:
|
|
160
|
-
fh = self._fh[i] = open(os.path.join(self.d,
|
|
161
|
-
|
|
162
|
-
|
|
170
|
+
fh = self._fh[i] = open(os.path.join(self.d, self.files[i]), "rb",
|
|
171
|
+
buffering=INDEX_BUF)
|
|
172
|
+
fh.seek(off - st)
|
|
173
|
+
got = fh.read(take)
|
|
174
|
+
out += got
|
|
175
|
+
if len(got) < take: # short shard: stop rather than mis-align
|
|
176
|
+
break
|
|
177
|
+
off += take
|
|
178
|
+
n -= take
|
|
179
|
+
i += 1
|
|
163
180
|
return out
|
|
164
181
|
|
|
165
182
|
def close(self):
|
|
@@ -168,6 +185,32 @@ class Cat:
|
|
|
168
185
|
self._fh.clear()
|
|
169
186
|
|
|
170
187
|
|
|
188
|
+
class _LazyCats(dict):
|
|
189
|
+
"""file_chunk<N>.index families, opened only when a file actually references one."""
|
|
190
|
+
|
|
191
|
+
def __init__(self, dirs: dict[int, str]):
|
|
192
|
+
super().__init__()
|
|
193
|
+
self._dirs = dirs
|
|
194
|
+
|
|
195
|
+
def get(self, k, default=None):
|
|
196
|
+
if k in self:
|
|
197
|
+
return self[k]
|
|
198
|
+
d = self._dirs.get(k)
|
|
199
|
+
if d is None:
|
|
200
|
+
return default
|
|
201
|
+
self[k] = c = Cat(d)
|
|
202
|
+
return c
|
|
203
|
+
|
|
204
|
+
def __missing__(self, k):
|
|
205
|
+
c = self.get(k)
|
|
206
|
+
if c is None:
|
|
207
|
+
raise KeyError(k)
|
|
208
|
+
return c
|
|
209
|
+
|
|
210
|
+
def values(self):
|
|
211
|
+
return [v for v in dict.values(self)]
|
|
212
|
+
|
|
213
|
+
|
|
171
214
|
def is_archive(d: str) -> bool:
|
|
172
215
|
return os.path.isdir(os.path.join(d, "Pool")) and os.path.isdir(os.path.join(d, "Config"))
|
|
173
216
|
|
|
@@ -198,13 +241,14 @@ class Archive:
|
|
|
198
241
|
c, p = f"{self.root}/Config", f"{self.root}/Pool"
|
|
199
242
|
self.vf = Cat(f"{c}/virtual_file.index")
|
|
200
243
|
self.ci = Cat(f"{p}/chunk_index")
|
|
201
|
-
self.
|
|
244
|
+
self._fc_dirs: dict[int, str] = {}
|
|
202
245
|
for e in os.listdir(c):
|
|
203
246
|
m = re.match(r"^file_chunk(\d+)\.index$", e)
|
|
204
247
|
if m and os.path.isdir(os.path.join(c, e)):
|
|
205
|
-
self.
|
|
206
|
-
if not self.
|
|
248
|
+
self._fc_dirs[int(m.group(1))] = os.path.join(c, e)
|
|
249
|
+
if not self._fc_dirs:
|
|
207
250
|
raise FileNotFoundError(f"no file_chunk<N>.index directories in {c}")
|
|
251
|
+
self.fc: dict[int, Cat] = _LazyCats(self._fc_dirs)
|
|
208
252
|
pools = [e for e in os.listdir(p) if e.isdigit() and os.path.isdir(os.path.join(p, e))]
|
|
209
253
|
if not pools:
|
|
210
254
|
raise FileNotFoundError(f"no numbered pool directory under {p}")
|
|
@@ -8,12 +8,19 @@
|
|
|
8
8
|
hbk <archive> verify <glob> [-j N] integrity-check, write nothing
|
|
9
9
|
hbk <archive> tui browse in the full-screen UI
|
|
10
10
|
|
|
11
|
+
hbk mount <remote:bucket> <dir> [--for browse|restore] [--no-warm]
|
|
12
|
+
hbk warm <dir> pre-cache dir listings
|
|
13
|
+
hbk unmount <dir> unmount it
|
|
14
|
+
hbk remotes list configured rclone remotes
|
|
15
|
+
|
|
11
16
|
Encrypted archives: add -p/--password, or set HBK_PASSWORD, or you'll be prompted.
|
|
12
17
|
|
|
13
18
|
<archive> is a .hbk directory, or any drive/folder containing one.
|
|
14
19
|
Globs match the full archive path, which starts with the share name.
|
|
15
20
|
Extraction is resumable: correctly-sized files are skipped.
|
|
16
21
|
|
|
22
|
+
hbk mount r2:mybucket ~/mnt/backup --for browse
|
|
23
|
+
hbk ~/mnt/backup/target.hbk doctor
|
|
17
24
|
hbk /Volumes/Backup doctor
|
|
18
25
|
hbk /Volumes/Backup list holiday
|
|
19
26
|
hbk /Volumes/Backup get "/Photos/2019/*" ~/restore
|
|
@@ -149,9 +156,47 @@ def main() -> int:
|
|
|
149
156
|
i = a.index(flag)
|
|
150
157
|
password = a[i + 1]
|
|
151
158
|
del a[i:i + 2]
|
|
152
|
-
if len(a) < 2:
|
|
159
|
+
if len(a) < 2 and (not a or a[0] != "remotes"):
|
|
153
160
|
print(__doc__)
|
|
154
161
|
return 1
|
|
162
|
+
if a[0] in ("mount", "unmount", "remotes", "warm"):
|
|
163
|
+
from . import mount as m
|
|
164
|
+
try:
|
|
165
|
+
if a[0] == "remotes":
|
|
166
|
+
rs = m.remotes()
|
|
167
|
+
print("\n".join(f" {r}" for r in rs) if rs
|
|
168
|
+
else " no rclone remotes configured - run: rclone config")
|
|
169
|
+
return 0
|
|
170
|
+
if a[0] == "warm":
|
|
171
|
+
if len(a) < 2:
|
|
172
|
+
print("usage: hbk warm <dir>")
|
|
173
|
+
return 1
|
|
174
|
+
return m.warm(a[1])
|
|
175
|
+
if a[0] == "unmount":
|
|
176
|
+
if len(a) < 2:
|
|
177
|
+
print("usage: hbk unmount <dir>")
|
|
178
|
+
return 1
|
|
179
|
+
return m.unmount(a[1])
|
|
180
|
+
if len(a) < 3:
|
|
181
|
+
print("usage: hbk mount <remote:bucket> <dir> [--for browse|restore]")
|
|
182
|
+
return 1
|
|
183
|
+
prof, rest = "browse", a[3:]
|
|
184
|
+
if "--for" in rest:
|
|
185
|
+
i = rest.index("--for")
|
|
186
|
+
prof = rest[i + 1]
|
|
187
|
+
cache = None
|
|
188
|
+
if "--cache-size" in rest:
|
|
189
|
+
cache = rest[rest.index("--cache-size") + 1]
|
|
190
|
+
rc = m.mount(a[1], a[2], prof, cache, dry_run="--dry-run" in rest)
|
|
191
|
+
if rc == 0 and "--dry-run" not in rest and "--no-warm" not in rest:
|
|
192
|
+
print("\npre-warming directory cache (first open over a network mount "
|
|
193
|
+
"is slow; doing it here so later commands are instant)")
|
|
194
|
+
m.warm(a[2])
|
|
195
|
+
return rc
|
|
196
|
+
except (m.RcloneMissing, ValueError) as e:
|
|
197
|
+
print(f"{e}", file=sys.stderr)
|
|
198
|
+
return 2
|
|
199
|
+
|
|
155
200
|
archive, cmd = a[0], a[1]
|
|
156
201
|
|
|
157
202
|
if cmd == "doctor":
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""Convenience wrapper around `rclone mount` for archives kept in object storage.
|
|
2
|
+
|
|
3
|
+
This stores no credentials and talks to no cloud API - `rclone config` still owns all of
|
|
4
|
+
that. All it does is pick the flags, because the flags are easy to get badly wrong:
|
|
5
|
+
rclone's `--vfs-cache-mode full` fetches *whole files*, and hbkit reads a 32-byte index
|
|
6
|
+
record and a ~5 KB chunk at a time, so a 10 KB extraction can pull ~82 MB. Meanwhile
|
|
7
|
+
`off` issues range requests, which is wrong for the one-time index build (a single
|
|
8
|
+
sequential read of a share database that can be hundreds of MB).
|
|
9
|
+
|
|
10
|
+
So the mode depends on the job, and that is what `--for` selects:
|
|
11
|
+
|
|
12
|
+
browse -> off, range requests; cheap metadata, good for cherry-picking files
|
|
13
|
+
restore -> full, whole-file fetch acts as read-ahead when pulling a whole folder,
|
|
14
|
+
capped so it cannot fill the disk
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
import shutil
|
|
20
|
+
import subprocess
|
|
21
|
+
import sys
|
|
22
|
+
|
|
23
|
+
PROFILES = {
|
|
24
|
+
# name: (vfs-cache-mode, extra flags, one-line rationale)
|
|
25
|
+
"browse": ("off", [],
|
|
26
|
+
"range requests - best for opening an archive and pulling a few files"),
|
|
27
|
+
"restore": ("full", ["--vfs-cache-max-size", "50G"],
|
|
28
|
+
"whole-file fetch as read-ahead for bulk restores, cache capped at 50G"),
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
# --no-modtime is the single biggest win. Without it rclone issues a HEAD per object just
|
|
32
|
+
# to fill in modification times for a directory listing, which is ~0.27s per entry: the
|
|
33
|
+
# 2,021-shard chunk_index directory took 546.8s cold. With it, 4.07s - a 134x speedup.
|
|
34
|
+
# hbkit never uses the modtimes of an archive's internal files (restored file mtimes come
|
|
35
|
+
# from the archive's own SQLite metadata), so dropping them costs nothing.
|
|
36
|
+
COMMON = ["--read-only", "--no-modtime", "--dir-cache-time", "72h", "--daemon"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class RcloneMissing(Exception):
|
|
40
|
+
pass
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def require_rclone() -> str:
|
|
44
|
+
p = shutil.which("rclone")
|
|
45
|
+
if not p:
|
|
46
|
+
raise RcloneMissing(
|
|
47
|
+
"rclone is not installed. Install it (brew install rclone / "
|
|
48
|
+
"apt install rclone), then run `rclone config` to add your S3/R2 remote.")
|
|
49
|
+
return p
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def remotes() -> list[str]:
|
|
53
|
+
"""Configured rclone remotes, e.g. ['r2:', 's3:']."""
|
|
54
|
+
try:
|
|
55
|
+
out = subprocess.run([require_rclone(), "listremotes"],
|
|
56
|
+
capture_output=True, text=True, timeout=15)
|
|
57
|
+
return [x for x in out.stdout.split() if x]
|
|
58
|
+
except (RcloneMissing, subprocess.SubprocessError):
|
|
59
|
+
return []
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def is_mounted(path: str) -> bool:
|
|
63
|
+
path = os.path.abspath(os.path.expanduser(path))
|
|
64
|
+
try:
|
|
65
|
+
out = subprocess.run(["mount"], capture_output=True, text=True, timeout=10).stdout
|
|
66
|
+
except subprocess.SubprocessError:
|
|
67
|
+
return False
|
|
68
|
+
return any(f" {path} " in ln or ln.endswith(f" {path}") for ln in out.splitlines())
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def build_command(remote: str, mountpoint: str, profile: str = "browse",
|
|
72
|
+
cache_size: str | None = None) -> list[str]:
|
|
73
|
+
if profile not in PROFILES:
|
|
74
|
+
raise ValueError(f"unknown profile {profile!r}; choose from {', '.join(PROFILES)}")
|
|
75
|
+
mode, extra, _ = PROFILES[profile]
|
|
76
|
+
extra = list(extra)
|
|
77
|
+
if cache_size and mode == "full":
|
|
78
|
+
extra = ["--vfs-cache-max-size", cache_size]
|
|
79
|
+
return ([require_rclone(), "mount", remote, os.path.abspath(os.path.expanduser(mountpoint)),
|
|
80
|
+
"--vfs-cache-mode", mode] + extra + COMMON)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def mount(remote: str, mountpoint: str, profile: str = "browse",
|
|
84
|
+
cache_size: str | None = None, dry_run: bool = False) -> int:
|
|
85
|
+
mp = os.path.abspath(os.path.expanduser(mountpoint))
|
|
86
|
+
if is_mounted(mp):
|
|
87
|
+
print(f"already mounted: {mp}")
|
|
88
|
+
return 0
|
|
89
|
+
cmd = build_command(remote, mp, profile, cache_size)
|
|
90
|
+
mode, _, why = PROFILES[profile]
|
|
91
|
+
print(f"profile {profile}: --vfs-cache-mode {mode} ({why})")
|
|
92
|
+
print(" " + " ".join(cmd))
|
|
93
|
+
if dry_run:
|
|
94
|
+
return 0
|
|
95
|
+
os.makedirs(mp, exist_ok=True)
|
|
96
|
+
r = subprocess.run(cmd, capture_output=True, text=True)
|
|
97
|
+
if r.returncode != 0:
|
|
98
|
+
print((r.stderr or r.stdout).strip(), file=sys.stderr)
|
|
99
|
+
return r.returncode
|
|
100
|
+
# --daemon returns immediately; wait for the mountpoint to become live
|
|
101
|
+
import time
|
|
102
|
+
for _ in range(60):
|
|
103
|
+
if is_mounted(mp) and os.listdir(mp):
|
|
104
|
+
break
|
|
105
|
+
time.sleep(0.5)
|
|
106
|
+
if not is_mounted(mp):
|
|
107
|
+
print(f"rclone returned success but {mp} is not mounted", file=sys.stderr)
|
|
108
|
+
return 1
|
|
109
|
+
print(f"mounted {remote} at {mp}")
|
|
110
|
+
for e in sorted(os.listdir(mp))[:10]:
|
|
111
|
+
print(f" {e}")
|
|
112
|
+
return 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# Directories hbkit lists when opening an archive. rclone caches a listing for
|
|
116
|
+
# --dir-cache-time, so walking these once up front turns a ~9 minute "did it hang?" on
|
|
117
|
+
# the user's first command into a visible, attributable wait here. Measured on a 3 TB
|
|
118
|
+
# archive over R2: 9m21s cold, then 1.6s, then 0.07s.
|
|
119
|
+
def warm(mountpoint: str, quiet: bool = False) -> int:
|
|
120
|
+
"""Populate rclone's directory cache for every .hbk archive under `mountpoint`."""
|
|
121
|
+
import time
|
|
122
|
+
mp = os.path.abspath(os.path.expanduser(mountpoint))
|
|
123
|
+
archives = []
|
|
124
|
+
for e in sorted(os.listdir(mp)):
|
|
125
|
+
d = os.path.join(mp, e)
|
|
126
|
+
if os.path.isdir(d) and os.path.isdir(os.path.join(d, "Config")):
|
|
127
|
+
archives.append(d)
|
|
128
|
+
if not archives:
|
|
129
|
+
return 0
|
|
130
|
+
n = 0
|
|
131
|
+
t0 = time.time()
|
|
132
|
+
for arc in archives:
|
|
133
|
+
targets = [arc, os.path.join(arc, "Config"), os.path.join(arc, "Pool")]
|
|
134
|
+
for sub in ("Config/virtual_file.index", "Pool/chunk_index", "Config/@Share"):
|
|
135
|
+
targets.append(os.path.join(arc, sub))
|
|
136
|
+
cfg = os.path.join(arc, "Config")
|
|
137
|
+
try:
|
|
138
|
+
targets += [os.path.join(cfg, x) for x in os.listdir(cfg)
|
|
139
|
+
if x.startswith("file_chunk") and os.path.isdir(os.path.join(cfg, x))]
|
|
140
|
+
except OSError:
|
|
141
|
+
pass
|
|
142
|
+
share = os.path.join(arc, "Config", "@Share")
|
|
143
|
+
if os.path.isdir(share):
|
|
144
|
+
targets += [os.path.join(share, x) for x in os.listdir(share)]
|
|
145
|
+
for t in targets:
|
|
146
|
+
if not os.path.isdir(t):
|
|
147
|
+
continue
|
|
148
|
+
if not quiet:
|
|
149
|
+
print(f" warming {os.path.relpath(t, mp)} ...", end="", flush=True)
|
|
150
|
+
s0 = time.time()
|
|
151
|
+
try:
|
|
152
|
+
c = len(os.listdir(t))
|
|
153
|
+
except OSError as e:
|
|
154
|
+
c = -1
|
|
155
|
+
if not quiet:
|
|
156
|
+
print(f" {c} entries, {time.time()-s0:.1f}s")
|
|
157
|
+
n += 1
|
|
158
|
+
if not quiet:
|
|
159
|
+
print(f"warmed {n} directories in {time.time()-t0:.0f}s "
|
|
160
|
+
f"(cached for --dir-cache-time; later opens are instant)")
|
|
161
|
+
return 0
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def unmount(mountpoint: str) -> int:
|
|
165
|
+
mp = os.path.abspath(os.path.expanduser(mountpoint))
|
|
166
|
+
if not is_mounted(mp):
|
|
167
|
+
print(f"not mounted: {mp}")
|
|
168
|
+
return 0
|
|
169
|
+
for cmd in (["umount", mp], ["fusermount", "-u", mp], ["diskutil", "unmount", "force", mp]):
|
|
170
|
+
if shutil.which(cmd[0]) and subprocess.run(cmd, capture_output=True).returncode == 0:
|
|
171
|
+
print(f"unmounted {mp}")
|
|
172
|
+
return 0
|
|
173
|
+
print(f"could not unmount {mp}", file=sys.stderr)
|
|
174
|
+
return 1
|
|
@@ -229,3 +229,112 @@ def test_encrypted_names_and_content():
|
|
|
229
229
|
data = arc.extract(ovf, size, verify=True) # verify=True checks every chunk MD5
|
|
230
230
|
assert len(data) == size, path
|
|
231
231
|
arc.close()
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# ------------------------------------------------- shard maths (no archive needed)
|
|
235
|
+
|
|
236
|
+
def _make_shards(tmp_path, sizes, shard_size, monkeypatch):
|
|
237
|
+
"""Write synthetic <N>.idx.2 shards. Shard 0 carries a valid 64-byte header, as real
|
|
238
|
+
archives do; the header is part of the logical stream so offsets stay comparable."""
|
|
239
|
+
import struct
|
|
240
|
+
|
|
241
|
+
from hbkit import archive as A
|
|
242
|
+
monkeypatch.setattr(A, "SHARD_SIZE", shard_size)
|
|
243
|
+
total = sum(sizes)
|
|
244
|
+
header = struct.pack(">16I", A.MAGIC, 0, 0, 0, 32, total >> 32, total & 0xFFFFFFFF,
|
|
245
|
+
*([0] * 9))
|
|
246
|
+
blob = b""
|
|
247
|
+
for i, n in enumerate(sizes):
|
|
248
|
+
if i == 0:
|
|
249
|
+
part = header + bytes((64 + j) % 251 for j in range(n - 64))
|
|
250
|
+
else:
|
|
251
|
+
part = bytes((len(blob) + j) % 251 for j in range(n))
|
|
252
|
+
(tmp_path / f"{i}.idx.2").write_bytes(part)
|
|
253
|
+
blob += part
|
|
254
|
+
return blob
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def test_cat_computes_offsets_without_stating_every_shard(tmp_path, monkeypatch):
|
|
258
|
+
"""Sizes/offsets must be derived from the fixed shard size + one final stat."""
|
|
259
|
+
from hbkit import archive as A
|
|
260
|
+
blob = _make_shards(tmp_path, [128, 128, 10], 128, monkeypatch)
|
|
261
|
+
c = A.Cat(str(tmp_path))
|
|
262
|
+
assert len(c.files) == 3
|
|
263
|
+
assert c.total == len(blob) == 266
|
|
264
|
+
assert c.starts == [0, 128, 256]
|
|
265
|
+
assert c.sizes == [128, 128, 10]
|
|
266
|
+
assert c.record_size == 32 # read from the header, not guessed
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def test_cat_reads_across_shard_boundaries(tmp_path, monkeypatch):
|
|
270
|
+
"""The multi-shard read path: every offset/length combination must match one blob."""
|
|
271
|
+
from hbkit import archive as A
|
|
272
|
+
blob = _make_shards(tmp_path, [128, 128, 10], 128, monkeypatch)
|
|
273
|
+
c = A.Cat(str(tmp_path))
|
|
274
|
+
for off in range(0, len(blob), 7):
|
|
275
|
+
for n in (1, 5, 127, 128, 129, 260):
|
|
276
|
+
assert c.read(off, n) == blob[off:off + n], f"off={off} n={n}"
|
|
277
|
+
assert c.read(0, len(blob)) == blob # whole stream in one read
|
|
278
|
+
assert c.read(len(blob), 10) == b"" # past the end
|
|
279
|
+
c.close()
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def test_cat_single_shard_still_works(tmp_path, monkeypatch):
|
|
283
|
+
from hbkit import archive as A
|
|
284
|
+
blob = _make_shards(tmp_path, [100], 128, monkeypatch)
|
|
285
|
+
c = A.Cat(str(tmp_path))
|
|
286
|
+
assert c.total == 100 and c.read(70, 20) == blob[70:90]
|
|
287
|
+
c.close()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
# --------------------------------------------------- rclone mount helper (no network)
|
|
291
|
+
|
|
292
|
+
def test_mount_profiles_pick_the_right_cache_mode(monkeypatch):
|
|
293
|
+
from hbkit import mount as m
|
|
294
|
+
monkeypatch.setattr(m, "require_rclone", lambda: "/usr/bin/rclone")
|
|
295
|
+
browse = m.build_command("r2:b", "/tmp/mp", "browse")
|
|
296
|
+
restore = m.build_command("r2:b", "/tmp/mp", "restore")
|
|
297
|
+
assert "off" == browse[browse.index("--vfs-cache-mode") + 1]
|
|
298
|
+
assert "full" == restore[restore.index("--vfs-cache-mode") + 1]
|
|
299
|
+
# bulk restores cache whole ~50MB buckets, so the cache MUST be capped
|
|
300
|
+
assert "--vfs-cache-max-size" in restore
|
|
301
|
+
assert "--vfs-cache-max-size" not in browse
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def test_mount_command_is_always_read_only_and_skips_modtimes(monkeypatch):
|
|
305
|
+
"""--no-modtime avoids a HEAD per object: 2021-entry listing 546.8s -> 4.07s."""
|
|
306
|
+
from hbkit import mount as m
|
|
307
|
+
monkeypatch.setattr(m, "require_rclone", lambda: "/usr/bin/rclone")
|
|
308
|
+
for profile in ("browse", "restore"):
|
|
309
|
+
cmd = m.build_command("r2:b", "/tmp/mp", profile)
|
|
310
|
+
assert "--read-only" in cmd, profile
|
|
311
|
+
assert "--no-modtime" in cmd, profile
|
|
312
|
+
assert "--dir-cache-time" in cmd, profile
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def test_mount_rejects_unknown_profile(monkeypatch):
|
|
316
|
+
from hbkit import mount as m
|
|
317
|
+
monkeypatch.setattr(m, "require_rclone", lambda: "/usr/bin/rclone")
|
|
318
|
+
with pytest.raises(ValueError):
|
|
319
|
+
m.build_command("r2:b", "/tmp/mp", "nonsense")
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def test_mount_cache_size_override(monkeypatch):
|
|
323
|
+
from hbkit import mount as m
|
|
324
|
+
monkeypatch.setattr(m, "require_rclone", lambda: "/usr/bin/rclone")
|
|
325
|
+
cmd = m.build_command("r2:b", "/tmp/mp", "restore", cache_size="10G")
|
|
326
|
+
assert cmd[cmd.index("--vfs-cache-max-size") + 1] == "10G"
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def test_missing_rclone_is_a_clear_error(monkeypatch):
|
|
330
|
+
from hbkit import mount as m
|
|
331
|
+
monkeypatch.setattr(m.shutil, "which", lambda _: None)
|
|
332
|
+
with pytest.raises(m.RcloneMissing) as e:
|
|
333
|
+
m.require_rclone()
|
|
334
|
+
assert "rclone config" in str(e.value)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def test_warm_ignores_a_directory_with_no_archives(tmp_path):
|
|
338
|
+
from hbkit import mount as m
|
|
339
|
+
(tmp_path / "not-an-archive").mkdir()
|
|
340
|
+
assert m.warm(str(tmp_path), quiet=True) == 0
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|