telstore 0.1.9 → 0.1.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,407 @@
1
+ import { promises as fs } from 'node:fs'
2
+ import path from 'node:path'
3
+
4
+ import { describeChat } from '../chat.js'
5
+ import {
6
+ closeQuietly,
7
+ connect as realConnect,
8
+ findManifestMessage,
9
+ readMessageBytes as realReadMessageBytes,
10
+ } from '../client.js'
11
+ import { configFile, defaultConfigDir, loadConfig } from '../config.js'
12
+ import { parseManifest } from '../manifest.js'
13
+ import { createProgress, formatBytes, plural } from '../progress.js'
14
+ import { assertLoggedIn } from '../session.js'
15
+ import { requireChat, resolveSettings } from '../settings.js'
16
+ import { shellArg } from '../shell.js'
17
+ import { spawnProducer } from '../spawn.js'
18
+ import { tempDirFor } from '../state.js'
19
+ import { DestinationGoneError, discardChunkFile, writeChunkTo } from '../stream.js'
20
+ import { isGzipName } from '../tar.js'
21
+ import { createOnRetry, realDownloadChunk, realGetMessage } from './restore.js'
22
+
23
+ // `telstore restore <id> -- tar xzf -`: the backup's bytes go to a command's stdin instead of
24
+ // into a file.
25
+ //
26
+ // This is runRestore with the file taken away, and the file was carrying more than the bytes:
27
+ // no .partial, no resume record, no scan of what an earlier run left, no rename, and no stat
28
+ // at the end to compare against the manifest. What replaces all of it is the order of two
29
+ // things — verify the chunk, then hand it over — and the command's exit code.
30
+ export async function runRestoreStream(backupId, childArgv, options = {}, deps = {}) {
31
+ const {
32
+ connect = realConnect,
33
+ disconnect = (client) => client.destroy(),
34
+ configDir = defaultConfigDir(),
35
+ searchManifest = findManifestMessage,
36
+ readMessageBytes = realReadMessageBytes,
37
+ getMessage = realGetMessage,
38
+ downloadChunk = realDownloadChunk,
39
+ spawn = spawnProducer,
40
+ retryOptions = {},
41
+ writeErr = (line) => process.stderr.write(line),
42
+ log: writeLog = (line) => console.log(line),
43
+ silent = false,
44
+ onBackupId = () => {},
45
+ onTempChunk = () => {},
46
+ // How a Ctrl-C that will not wait still stops the command. Unlike the upload direction
47
+ // there is nothing in the chat to unwind, so this run does not ask to be waited for — but
48
+ // leaving without stopping tar lets it go on writing files after telstore is gone.
49
+ onChild = () => {},
50
+ // tarx only. The alias promises gzip, so it checks the claim before spending a gigabyte
51
+ // finding out; `restore <id> -- tar xf -` promises nothing and is asked nothing.
52
+ requireGzipName = false,
53
+ } = deps
54
+
55
+ const config = await loadConfig(configDir)
56
+
57
+ assertLoggedIn(config)
58
+
59
+ const { values: settings } = resolveSettings(options, config, { file: configFile(configDir) })
60
+ const chat = requireChat(settings)
61
+ const log = silent ? () => {} : writeLog
62
+ const warn = silent ? () => {} : writeErr
63
+ const onRetry = createOnRetry(warn)
64
+
65
+ onBackupId(backupId)
66
+
67
+ const client = await connect(config, { verbose: settings.verbose })
68
+ let child = null
69
+ let ended = false
70
+ let written = 0
71
+
72
+ // Bytes written into the command, counted as they go rather than added up from the sizes the
73
+ // manifest claims: what the end of this function compares against manifest.size is then a
74
+ // measurement of what went into the pipe rather than a restatement of what was expected to.
75
+ // It is also the only way a run that stopped in the middle of a chunk can say how much it had
76
+ // written, which is in the message it fails with — and it is written, never read: whether the
77
+ // command on the far end took those bytes off the pipe is what its exit code answers.
78
+ const handed = (bytes) => {
79
+ written += bytes
80
+ }
81
+
82
+ try {
83
+ const manifestMessage = await searchManifest(client, chat, backupId)
84
+
85
+ if (!manifestMessage) {
86
+ throw new Error(
87
+ `No manifest found for ${backupId} in ${chat}. ` +
88
+ 'Check the backup id, or use --chat to point at the right chat.',
89
+ )
90
+ }
91
+
92
+ // The whole-backup arithmetic is in here and stays there: parseManifest adds the chunk
93
+ // sizes up and compares them with manifest.size, refuses a layout where any chunk is not
94
+ // where restore would look for it, and counts the list against the size. A manifest that
95
+ // disagrees with itself therefore never reaches the line below, which is why this command
96
+ // does not add the sum again — two copies of one piece of arithmetic is how they start
97
+ // disagreeing, and the copy that is never reached is the one that would be wrong.
98
+ const manifest = parseManifest(await readMessageBytes(client, manifestMessage))
99
+
100
+ // Before the download, which is the only reason this check is worth making at all: tar
101
+ // would say "not in gzip format" by itself, but only after every byte had arrived.
102
+ if (requireGzipName && !isGzipName(manifest.name)) {
103
+ throw new Error(
104
+ `${backupId} is called ${manifest.name}, which does not claim to be gzipped, and tarx ` +
105
+ 'always extracts with "tar xzf -". Run ' +
106
+ `"npx telstore restore ${shellArg(backupId)} -- tar xf -" if it is a plain tar ` +
107
+ 'archive. Refused now rather than after the whole backup has been downloaded, which ' +
108
+ 'is when tar would find out.',
109
+ )
110
+ }
111
+
112
+ log(`Backup ${backupId}`)
113
+ log(`Name ${manifest.name} (${plural(manifest.chunks.length, 'chunk')}, ${formatBytes(manifest.size)})`)
114
+ log(`From ${describeChat(chat)}`)
115
+ log(`Into ${childArgv.join(' ')}\n`)
116
+
117
+ const tmp = tempDirFor(configDir)
118
+ await fs.mkdir(tmp, { recursive: true })
119
+
120
+ // Before the first download. A command that cannot start costs nothing here and a whole
121
+ // chunk if it is started later.
122
+ child = spawn(childArgv, { stdio: ['pipe', 'inherit', 'inherit'] })
123
+ onChild(child.kill)
124
+
125
+ // Raced against every step below rather than checked between them: a command that dies
126
+ // two minutes into an 1800MB download leaves that download with nowhere to go, and
127
+ // finishing it first would be eight minutes spent on bytes nobody will read. Always
128
+ // throws, so it can never be the value a race resolves with. The no-op catch is for the
129
+ // window before the first race attaches a handler, exactly as `spawnProducer` does it.
130
+ const gone = child.exited.then(
131
+ ({ code, signal }) => {
132
+ throw stoppedReading(childArgv, written, manifest.size, { code, signal })
133
+ },
134
+ (err) => {
135
+ throw err
136
+ },
137
+ )
138
+ gone.catch(() => {})
139
+
140
+ // An 'error' event with nothing listening for it is an uncaught exception, and stdin's own
141
+ // failures do not all arrive inside a window `writeChunkTo` is watching: `write()` returning
142
+ // true means the pipe took the bytes into its buffer, so the EPIPE belonging to the last
143
+ // write of a chunk can surface after that call has returned and its listeners have come off.
144
+ // This used to be absorbed by accident — the `pipeline` `writeChunkTo` was built on left its
145
+ // handlers on this stream for the life of the run — and absorbing it on purpose is not
146
+ // enough either: a pipe that failed means the command did not get what was written into it,
147
+ // which is the difference between a restore and a plausible-looking one. So it is latched,
148
+ // and read in two places: before every chunk after the first, and once the child's exit is
149
+ // in hand.
150
+ let pipeFailure = null
151
+ child.stdin.on('error', (err) => {
152
+ pipeFailure ??= err
153
+ })
154
+
155
+ const progress = createProgress({
156
+ total: manifest.size,
157
+ label: `Chunk 1/${manifest.chunks.length}`,
158
+ write: warn,
159
+ })
160
+
161
+ try {
162
+ for (const chunk of manifest.chunks) {
163
+ // Asked before the next chunk rather than only at the end, because by then the run has
164
+ // known for a whole download: a pipe that has gone will not take this chunk either, and
165
+ // fetching 1800MB to push into it is eight minutes spent on bytes nobody will read. It is
166
+ // also the only thing that ends a run whose command closed its stdin at a chunk boundary
167
+ // and did not exit — `gone` never settles for a command that is still running, and
168
+ // nothing here puts a deadline on waiting for one.
169
+ //
170
+ // `pipeFailure` alone is not enough here: it is latched from an 'error' event, and a
171
+ // command that closes its end of the pipe quietly between two chunks — without ever
172
+ // writing into it again to provoke one — leaves `pipeFailure` null. `writeChunkTo`'s own
173
+ // probe would still catch that, but only once it is pumping — after this chunk has
174
+ // already been downloaded for nothing. Checking the flags directly is what the probe
175
+ // itself checks, asked a chunk earlier.
176
+ if (pipeFailure !== null || child.stdin.destroyed || child.stdin.writableEnded) {
177
+ // A real 'error' means some of what telstore already wrote may never have arrived.
178
+ // A destroyed-or-ended pipe with no error behind it means the opposite: everything
179
+ // written so far was taken cleanly, and the command simply stopped reading before the
180
+ // backup ended — the same ending `stoppedReading` already says correctly.
181
+ throw pipeFailure !== null
182
+ ? pipeFailed(childArgv, written, pipeFailure)
183
+ : stoppedReading(childArgv, written, manifest.size)
184
+ }
185
+
186
+ const file = path.join(tmp, `${backupId}-${chunk.i}.chunk`)
187
+
188
+ // Said before the open, for the reason the upload direction says it before its own:
189
+ // the handler has to hold the name for the whole window in which the file can exist.
190
+ onTempChunk(file)
191
+
192
+ const handle = await fs.open(file, 'w+')
193
+
194
+ try {
195
+ progress.setLabel(`Chunk ${chunk.i + 1}/${manifest.chunks.length}`)
196
+
197
+ // getMessage and downloadChunk go over the network, and either can reject on its own
198
+ // — a stall timeout, a FLOOD_WAIT that outlived its retries, any teleproto error —
199
+ // which is a different ending from `gone` losing the race: `gone` only ever fires
200
+ // once the child has already exited, and its own throw already says so. A rejection
201
+ // from the network call itself used to propagate raw, saying nothing about the
202
+ // prefix already handed to the command — the same omission `networkFailed` closes for
203
+ // both calls, through the one `received()` wording rather than a fifth variant of it.
204
+ const message = await Promise.race([
205
+ getMessage(client, chat, chunk.msgId).catch((err) => {
206
+ throw networkFailed(err, childArgv, written)
207
+ }),
208
+ gone,
209
+ ])
210
+
211
+ if (!message) {
212
+ throw new Error(
213
+ `Missing chunk ${chunk.i + 1}/${manifest.chunks.length}: message ${chunk.msgId} ` +
214
+ `is no longer in ${chat}. This backup cannot be restored. ` +
215
+ `${received(childArgv, written)}`,
216
+ )
217
+ }
218
+
219
+ const { sha256, size } = await Promise.race([
220
+ downloadChunk(client, message, handle, 0, progress.advance, {
221
+ retryOptions: { ...retryOptions, onRetry },
222
+ concurrency: settings.downloadConcurrency,
223
+ }).catch((err) => {
224
+ throw networkFailed(err, childArgv, written)
225
+ }),
226
+ gone,
227
+ ])
228
+
229
+ if (size !== chunk.size) {
230
+ throw new Error(
231
+ `Chunk ${chunk.i + 1} arrived with ${size} bytes and the manifest records ` +
232
+ `${chunk.size} — mismatch. ${received(childArgv, written)}`,
233
+ )
234
+ }
235
+
236
+ if (sha256 !== chunk.sha256) {
237
+ throw new Error(
238
+ `Chunk ${chunk.i + 1} has a sha256 that does not match the manifest. ` +
239
+ `${received(childArgv, written)}`,
240
+ )
241
+ }
242
+
243
+ // Only now, and this line is the guarantee: everything above it is what makes the
244
+ // difference between handing a command the backup and handing it whatever arrived.
245
+ //
246
+ // The write's own failure is worded here rather than reported as it arrived, because
247
+ // a dead pipe is not a sentence anybody can act on: "write EPIPE" names the symptom,
248
+ // and what happened is what `stoppedReading` says — minus the exit code, which this
249
+ // path has not got. Usually it is not needed: measured on node 22, a command that
250
+ // dies mid-chunk loses this race to `gone`, which is one microtask behind the child's
251
+ // exit while the write has a read to finish and a latch to notice. What is left for
252
+ // this catch is the ending where no exit status is coming at all — a command that
253
+ // closes its end of the pipe and goes on working, `head -c 10` inside a shell that
254
+ // has more to do — and even there the write is only a mechanism that *can* report it,
255
+ // not one that always does: once `head` has read its ten bytes and stopped, a
256
+ // remainder small enough to sit entirely in the OS pipe buffer is accepted by
257
+ // `write()` without complaint — the kernel took it, nobody will ever read it, and this
258
+ // catch never fires. Measured 2026-09-10: a 19.8KB remainder behind that same `head -c
259
+ // 10` reports a restore that never happened; a 391KB one is refused correctly.
260
+ // `docs/design/data-integrity.md` has both numbers and what binds the guarantee.
261
+ //
262
+ // Which failure this was is asked of writeChunkTo, which says so by type:
263
+ // DestinationGoneError is the far end and nothing else is. Asking the pipe instead —
264
+ // `child.stdin.destroyed` — would have been one inference too many: a read error on
265
+ // the chunk file can leave the same trace, and telstore would then blame tar for a
266
+ // fault of its own. A code on the error is no better, because a far end that has gone
267
+ // reports itself as EPIPE, as ECONNRESET or as a premature close depending on when
268
+ // it went, and a list of spellings is how such a check quietly stops matching.
269
+ const pumping = writeChunkTo(child.stdin, handle, size, { onProgress: handed }).catch(
270
+ (err) => {
271
+ if (err instanceof DestinationGoneError) {
272
+ throw stoppedReading(childArgv, written, manifest.size)
273
+ }
274
+
275
+ throw err
276
+ },
277
+ )
278
+
279
+ await Promise.race([pumping, gone])
280
+ } finally {
281
+ await discardChunkFile(handle, file, { writeErr, chunkSize: manifest.chunkSize })
282
+
283
+ // Unsaid whether the removal worked or not: if it did there is nothing left to
284
+ // remove, and if it did not, discardChunkFile has already named the file on stderr.
285
+ onTempChunk(null)
286
+ }
287
+ }
288
+ } finally {
289
+ // Ended here rather than after the loop, so a failure mid-chunk still leaves the cursor
290
+ // on a fresh line and "Error: ..." does not land on top of the bar.
291
+ progress.finish()
292
+ }
293
+
294
+ // Unreachable if every chunk verified and every write moved the length it was given, which
295
+ // is why it is worth keeping: it is the one check that does not trust the ones above it,
296
+ // and it is counted from the bytes that went into the pipe rather than from the manifest.
297
+ if (written !== manifest.size) {
298
+ throw new Error(
299
+ `telstore wrote ${written} bytes into ${childArgv[0]} and the manifest records ` +
300
+ `${manifest.size}. Refusing to report a restore it cannot account for.`,
301
+ )
302
+ }
303
+
304
+ // The command has had every byte the manifest names; EOF is how it is told so.
305
+ child.stdin.end()
306
+
307
+ const { code, signal } = await child.exited
308
+ ended = true
309
+
310
+ // Before the exit code is read for what it says, because it cannot answer this: a command
311
+ // that exited 0 with the tail of the backup still sitting in a pipe it had stopped reading
312
+ // exited 0 all the same. The cost of checking is a restore refused in the one case where a
313
+ // command read every byte and left before telstore closed the pipe behind it; the cost of
314
+ // not checking is a backup reported as restored that the command never finished receiving,
315
+ // and this project pays the first to avoid the second.
316
+ if (pipeFailure !== null) {
317
+ throw pipeFailed(childArgv, written, pipeFailure, { code, signal })
318
+ }
319
+
320
+ if (code !== 0 || signal !== null) {
321
+ throw new Error(
322
+ signal !== null
323
+ ? `${childArgv[0]} was killed by ${signal} after receiving all ` +
324
+ `${formatBytes(written)}. telstore is not reporting a restore on its behalf.`
325
+ : `${childArgv[0]} exited ${code} after receiving all ${formatBytes(written)}. ` +
326
+ 'telstore is not reporting a restore on its behalf: every byte was correct and ' +
327
+ 'the command still did not finish.',
328
+ )
329
+ }
330
+
331
+ log(`\nDone. telstore wrote ${formatBytes(written)} into ${childArgv[0]}, which exited 0.`)
332
+
333
+ return { id: backupId, size: written, chunks: manifest.chunks.length }
334
+ } finally {
335
+ // A run that fell over mid-chunk leaves the command alive and blocked on a pipe nothing is
336
+ // going to write to again. Destroying the pipe is what turns that wait into something it
337
+ // can act on; the kill is for a command that is not waiting on stdin at all.
338
+ if (child && !ended) {
339
+ child.stdin.destroy()
340
+ child.kill()
341
+ }
342
+
343
+ onChild(null)
344
+
345
+ await closeQuietly(client, disconnect, (err) =>
346
+ warn(`\nWarning: could not close the Telegram connection: ${err.message}\n`),
347
+ )
348
+ }
349
+ }
350
+
351
+ // What the command got before this went wrong, in every message about a chunk that did not.
352
+ // A person deciding what to do next needs to know the difference between "it has half of my
353
+ // archive" and "it has nothing".
354
+ function received(childArgv, written) {
355
+ return written === 0
356
+ ? `${childArgv[0]} was given nothing.`
357
+ : `${childArgv[0]} had already been given ${formatBytes(written)}, which was correct but ` +
358
+ 'is not the whole backup — whatever it did with that is incomplete.'
359
+ }
360
+
361
+ // getMessage or downloadChunk rejecting is the same class of failure as the two chunk-mismatch
362
+ // branches above — the run stops mid-loop, not at a boundary the command already knows about —
363
+ // so it gets the same treatment rather than propagating whatever teleproto's error happened to
364
+ // say. `err.message` alone was measured reading "FAIL TIMEOUT: no response from Telegram after
365
+ // 3 attempts" while the command had already been handed a correct 195KB prefix and written it
366
+ // to disk; not one word about that in the raw error.
367
+ function networkFailed(err, childArgv, written) {
368
+ return new Error(`${err.message}. ${received(childArgv, written)}`)
369
+ }
370
+
371
+ // A pipe that failed, which is the one ending the command's own exit code cannot speak for, and
372
+ // one sentence for both the places that report it: the chunk boundary, where there is no exit
373
+ // status yet and may never be one, and after the child has exited.
374
+ function pipeFailed(childArgv, written, err, exit = null) {
375
+ const what =
376
+ exit === null
377
+ ? `${childArgv[0]}'s stdin failed`
378
+ : `${childArgv[0]} exited ${exit.signal !== null ? `on ${exit.signal}` : String(exit.code)}` +
379
+ ', but its stdin failed first'
380
+
381
+ return new Error(
382
+ `${what} (${err.message}) — so some of the ${formatBytes(written)} telstore wrote into it ` +
383
+ 'never arrived. Not reporting a restore on that. Nothing in the chat changed.',
384
+ )
385
+ }
386
+
387
+ // A command that is gone while there are bytes left. Exit 0 is included on purpose: `head -c
388
+ // 10` exits 0 having read ten bytes of a gigabyte, and reporting that as a restore would be
389
+ // the confident wrong answer. Measured, not assumed — see the probe table in the spec.
390
+ //
391
+ // `exit` is null on the one ending where no exit status is coming: the command closed its end
392
+ // of the pipe and went on working. What happened is the same thing either way, so it is the
393
+ // same sentence, and only the clause naming the exit code is missing — waiting for one there
394
+ // would mean waiting without a deadline on a command that may never exit at all.
395
+ function stoppedReading(childArgv, written, total, exit = null) {
396
+ const how =
397
+ exit === null
398
+ ? 'stopped reading'
399
+ : exit.signal !== null
400
+ ? `was killed by ${exit.signal}`
401
+ : `exited ${exit.code}`
402
+
403
+ return new Error(
404
+ `${childArgv[0]} ${how} with ${formatBytes(written)} of ${formatBytes(total)} written into ` +
405
+ 'it, so it did not receive the backup. Nothing in the chat changed.',
406
+ )
407
+ }
@@ -25,7 +25,32 @@ const LONG_WAIT_MS = 60_000
25
25
  // by then the trouble has outlived two backoffs and is worth saying out loud.
26
26
  const ANNOUNCE_AFTER_ATTEMPT = 3
27
27
 
28
- async function realGetMessage(client, peer, msgId) {
28
+ // Lifted out of runRestore so the streaming restore uses this one rather than a second copy
29
+ // that drifts. A retry nobody is told about is indistinguishable from a hung transfer, because
30
+ // the progress bar simply stops moving while the wait runs.
31
+ export function createOnRetry(warn) {
32
+ return function onRetry(err, attempt, delayMs, elapsedMs = 0) {
33
+ if (delayMs > LONG_WAIT_MS) {
34
+ warn(
35
+ `\nTelegram wants ${formatDuration(delayMs / 1000)} of waiting before the next part ` +
36
+ `(${err.message}). telstore is waiting and will carry on by itself, leave it running.\n`,
37
+ )
38
+ return
39
+ }
40
+
41
+ // The exception to staying quiet: an attempt that took a minute to fail spent that
42
+ // minute with the bar frozen, which is exactly what a hang looks like. Those are worth
43
+ // a line the first time, whatever the attempt number.
44
+ if (attempt < ANNOUNCE_AFTER_ATTEMPT && elapsedMs < LONG_WAIT_MS) return
45
+
46
+ warn(
47
+ `\nTemporary error (${err.message}), retry ${attempt} in ` +
48
+ `${formatDuration(delayMs / 1000)}.\n`,
49
+ )
50
+ }
51
+ }
52
+
53
+ export async function realGetMessage(client, peer, msgId) {
29
54
  const [message] = await client.getMessages(peer, { ids: [msgId] })
30
55
  return message ?? null
31
56
  }
@@ -100,25 +125,7 @@ export async function runRestore(backupId, options = {}, deps = {}) {
100
125
  // A restore keeps no progress file, so a part that comes back -503 is retried rather than
101
126
  // thrown away — and a retry nobody is told about is indistinguishable from a hung transfer,
102
127
  // because the progress bar simply stops moving while the wait runs.
103
- function onRetry(err, attempt, delayMs, elapsedMs = 0) {
104
- if (delayMs > LONG_WAIT_MS) {
105
- warn(
106
- `\nTelegram wants ${formatDuration(delayMs / 1000)} of waiting before the next part ` +
107
- `(${err.message}). telstore is waiting and will carry on by itself, leave it running.\n`,
108
- )
109
- return
110
- }
111
-
112
- // The exception to staying quiet: an attempt that took a minute to fail spent that
113
- // minute with the bar frozen, which is exactly what a hang looks like. Those are worth
114
- // a line the first time, whatever the attempt number.
115
- if (attempt < ANNOUNCE_AFTER_ATTEMPT && elapsedMs < LONG_WAIT_MS) return
116
-
117
- warn(
118
- `\nTemporary error (${err.message}), retry ${attempt} in ` +
119
- `${formatDuration(delayMs / 1000)}.\n`,
120
- )
121
- }
128
+ const onRetry = createOnRetry(warn)
122
129
 
123
130
  const client = await connect(config, { verbose: settings.verbose })
124
131