@magland/mochi 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +108 -0
- package/dist/ansi.js +174 -0
- package/dist/api/admin.js +416 -0
- package/dist/api/auth.js +166 -0
- package/dist/api/backup.js +598 -0
- package/dist/api/ci.js +336 -0
- package/dist/api/contents.js +339 -0
- package/dist/api/issues.js +165 -0
- package/dist/api/pulls.js +244 -0
- package/dist/api/releases.js +83 -0
- package/dist/api/repos.js +156 -0
- package/dist/api/write.js +518 -0
- package/dist/api.js +326 -0
- package/dist/assets.js +29 -0
- package/dist/atom.js +32 -0
- package/dist/atomic.js +171 -0
- package/dist/avatar.js +81 -0
- package/dist/browse.js +630 -0
- package/dist/build-info.json +4 -0
- package/dist/ci/actionref.js +86 -0
- package/dist/ci/api.js +829 -0
- package/dist/ci/artifacts.js +201 -0
- package/dist/ci/dispatch.js +30 -0
- package/dist/ci/engine.js +1321 -0
- package/dist/ci/expr.js +526 -0
- package/dist/ci/manual.js +199 -0
- package/dist/ci/present.js +82 -0
- package/dist/ci/protocol.js +6 -0
- package/dist/ci/runners.js +256 -0
- package/dist/ci/runs.js +208 -0
- package/dist/ci/trigger.js +28 -0
- package/dist/ci/views.js +441 -0
- package/dist/ci/wake.js +194 -0
- package/dist/ci/web.js +617 -0
- package/dist/ci/workflow.js +436 -0
- package/dist/cli/admin-cmd.js +324 -0
- package/dist/cli/api-cmd.js +128 -0
- package/dist/cli/backup-cmd.js +1500 -0
- package/dist/cli/exit.js +69 -0
- package/dist/cli/input.js +64 -0
- package/dist/cli/issue-cmd.js +243 -0
- package/dist/cli/output.js +93 -0
- package/dist/cli/parse.js +317 -0
- package/dist/cli/pr-cmd.js +289 -0
- package/dist/cli/release-cmd.js +171 -0
- package/dist/cli/repo-cmd.js +763 -0
- package/dist/cli/repo.js +101 -0
- package/dist/cli/run-cmd.js +438 -0
- package/dist/cli/target.js +54 -0
- package/dist/cli-api.js +84 -0
- package/dist/compare.js +111 -0
- package/dist/config.js +212 -0
- package/dist/credentials.js +235 -0
- package/dist/deploy-cli.js +859 -0
- package/dist/deploy-runner-cli.js +592 -0
- package/dist/diff.js +171 -0
- package/dist/discussion.js +253 -0
- package/dist/egress.js +559 -0
- package/dist/filecache.js +68 -0
- package/dist/find.js +162 -0
- package/dist/forms.js +737 -0
- package/dist/git.js +547 -0
- package/dist/githttp.js +428 -0
- package/dist/html.js +87 -0
- package/dist/icons.js +101 -0
- package/dist/import-cli.js +316 -0
- package/dist/index.js +752 -0
- package/dist/issues.js +308 -0
- package/dist/issueweb.js +447 -0
- package/dist/job-cli.js +197 -0
- package/dist/jobtoken.js +96 -0
- package/dist/languages.js +383 -0
- package/dist/layout.js +100 -0
- package/dist/lfs.js +438 -0
- package/dist/lfsstore.js +425 -0
- package/dist/limit.js +259 -0
- package/dist/logo.js +61 -0
- package/dist/markdown.js +382 -0
- package/dist/migrate.js +334 -0
- package/dist/multipart.js +90 -0
- package/dist/ops.js +869 -0
- package/dist/pagescript.js +465 -0
- package/dist/perms.js +370 -0
- package/dist/pointer.js +55 -0
- package/dist/profile.js +106 -0
- package/dist/pulls.js +320 -0
- package/dist/pullweb.js +461 -0
- package/dist/redirects.js +455 -0
- package/dist/releases.js +435 -0
- package/dist/render.js +233 -0
- package/dist/runner/actions.js +448 -0
- package/dist/runner/client.js +428 -0
- package/dist/runner/context.js +247 -0
- package/dist/runner/docker.js +197 -0
- package/dist/runner/externals.js +175 -0
- package/dist/runner/job.js +290 -0
- package/dist/runner/manual-run.js +272 -0
- package/dist/runner/overrides.js +554 -0
- package/dist/runner/steps.js +571 -0
- package/dist/runner/wake.js +84 -0
- package/dist/runner-cli.js +405 -0
- package/dist/scan.js +231 -0
- package/dist/server.js +424 -0
- package/dist/session.js +267 -0
- package/dist/site.js +259 -0
- package/dist/siteshost.js +94 -0
- package/dist/source.js +90 -0
- package/dist/style.js +1295 -0
- package/dist/themes.js +369 -0
- package/dist/vault.js +442 -0
- package/dist/version.js +88 -0
- package/dist/views.js +1007 -0
- package/dist/web.js +182 -0
- package/dist/webops.js +1402 -0
- package/package.json +71 -0
|
@@ -0,0 +1,1500 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
+
if (k2 === undefined) k2 = k;
|
|
4
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
+
}
|
|
8
|
+
Object.defineProperty(o, k2, desc);
|
|
9
|
+
}) : (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
o[k2] = m[k];
|
|
12
|
+
}));
|
|
13
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
14
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
15
|
+
}) : function(o, v) {
|
|
16
|
+
o["default"] = v;
|
|
17
|
+
});
|
|
18
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
19
|
+
var ownKeys = function(o) {
|
|
20
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
21
|
+
var ar = [];
|
|
22
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
23
|
+
return ar;
|
|
24
|
+
};
|
|
25
|
+
return ownKeys(o);
|
|
26
|
+
};
|
|
27
|
+
return function (mod) {
|
|
28
|
+
if (mod && mod.__esModule) return mod;
|
|
29
|
+
var result = {};
|
|
30
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
31
|
+
__setModuleDefault(result, mod);
|
|
32
|
+
return result;
|
|
33
|
+
};
|
|
34
|
+
})();
|
|
35
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
exports.backupCommands = void 0;
|
|
37
|
+
exports.backupsIndexPath = backupsIndexPath;
|
|
38
|
+
exports.knownBackups = knownBackups;
|
|
39
|
+
exports.backupLineFor = backupLineFor;
|
|
40
|
+
const child_process_1 = require("child_process");
|
|
41
|
+
const crypto = __importStar(require("crypto"));
|
|
42
|
+
const fs = __importStar(require("fs"));
|
|
43
|
+
const os = __importStar(require("os"));
|
|
44
|
+
const path = __importStar(require("path"));
|
|
45
|
+
const atomic_1 = require("../atomic");
|
|
46
|
+
const scan_1 = require("../scan");
|
|
47
|
+
const exit_1 = require("./exit");
|
|
48
|
+
const output_1 = require("./output");
|
|
49
|
+
const target_1 = require("./target");
|
|
50
|
+
// `mochi backup <dir>`: an incremental copy of a whole vault onto a disk of
|
|
51
|
+
// your own, over HTTP.
|
|
52
|
+
//
|
|
53
|
+
// The documentation is entitled to say that backing up a vault is `cp -a`, and
|
|
54
|
+
// that is true of a vault on a machine you have a shell on. It is not true of
|
|
55
|
+
// the deployment the CLI recommends, where the vault is on a Fly volume with no
|
|
56
|
+
// shell in the ordinary sense and no rsync at the far end. This command is the
|
|
57
|
+
// answer for that case, and works identically against a VPS, a Docker
|
|
58
|
+
// deployment, and 127.0.0.1:3000.
|
|
59
|
+
//
|
|
60
|
+
// Two things shape everything here.
|
|
61
|
+
//
|
|
62
|
+
// The backup directory is itself a vault, so restoring is `mochi serve
|
|
63
|
+
// <dir>/current` rather than a program that only gets exercised during a
|
|
64
|
+
// disaster. Each mirror is a bare repository like any other, so the recovery
|
|
65
|
+
// procedure is one line and can be rehearsed at any time.
|
|
66
|
+
//
|
|
67
|
+
// Nothing in the backup is ever modified in place. Git rewrites refs and
|
|
68
|
+
// packfiles by rename, mochi writes its state files by rename, this file
|
|
69
|
+
// writes by rename, and reflogs - the one thing git appends to - are turned off
|
|
70
|
+
// on the mirrors. That is what makes a snapshot a directory of hardlinks
|
|
71
|
+
// costing inodes rather than bytes. Any future code here that opens a file
|
|
72
|
+
// under current/ for appending breaks every existing snapshot.
|
|
73
|
+
//
|
|
74
|
+
// See docs/backup.md, and src/api/backup.ts for the two routes this speaks to.
|
|
75
|
+
const STATE_FILE = 'backup.json';
|
|
76
|
+
const LOCK_FILE = '.lock';
|
|
77
|
+
const CURRENT = 'current';
|
|
78
|
+
const SNAPSHOTS = 'snapshots';
|
|
79
|
+
/** How many runs of history backup.json keeps. Enough to see a pattern, not a log file. */
|
|
80
|
+
const KEEP_RUNS = 20;
|
|
81
|
+
/** The server's own caps, which the client chunks to fit. */
|
|
82
|
+
const MAX_FETCH_PATHS = 2000;
|
|
83
|
+
const MAX_FETCH_BYTES = 64 * 1024 * 1024;
|
|
84
|
+
const DEFAULT_RETENTION = { daily: 7, weekly: 4, monthly: 6 };
|
|
85
|
+
function emptyState() {
|
|
86
|
+
return {
|
|
87
|
+
version: 1,
|
|
88
|
+
host: '',
|
|
89
|
+
lfs: 'volume',
|
|
90
|
+
excluded: [],
|
|
91
|
+
retention: { ...DEFAULT_RETENTION },
|
|
92
|
+
repos: {},
|
|
93
|
+
files: {},
|
|
94
|
+
runs: [],
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
function statePath(dir) {
|
|
98
|
+
return path.join(dir, STATE_FILE);
|
|
99
|
+
}
|
|
100
|
+
function loadState(dir) {
|
|
101
|
+
let parsed;
|
|
102
|
+
try {
|
|
103
|
+
parsed = JSON.parse(fs.readFileSync(statePath(dir), 'utf8'));
|
|
104
|
+
}
|
|
105
|
+
catch {
|
|
106
|
+
return emptyState();
|
|
107
|
+
}
|
|
108
|
+
const state = emptyState();
|
|
109
|
+
if (typeof parsed.host === 'string')
|
|
110
|
+
state.host = parsed.host;
|
|
111
|
+
if (typeof parsed.lfs === 'string')
|
|
112
|
+
state.lfs = parsed.lfs;
|
|
113
|
+
if (Array.isArray(parsed.excluded))
|
|
114
|
+
state.excluded = parsed.excluded.filter((x) => typeof x === 'string');
|
|
115
|
+
if (typeof parsed.retention === 'object' && parsed.retention !== null) {
|
|
116
|
+
const r = parsed.retention;
|
|
117
|
+
for (const k of ['daily', 'weekly', 'monthly']) {
|
|
118
|
+
if (typeof r[k] === 'number' && r[k] >= 0)
|
|
119
|
+
state.retention[k] = Math.floor(r[k]);
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
if (typeof parsed.repos === 'object' && parsed.repos !== null)
|
|
123
|
+
state.repos = parsed.repos;
|
|
124
|
+
if (typeof parsed.files === 'object' && parsed.files !== null)
|
|
125
|
+
state.files = parsed.files;
|
|
126
|
+
if (Array.isArray(parsed.runs))
|
|
127
|
+
state.runs = parsed.runs;
|
|
128
|
+
return state;
|
|
129
|
+
}
|
|
130
|
+
function saveState(dir, state) {
|
|
131
|
+
(0, atomic_1.writeFileAtomic)(statePath(dir), JSON.stringify(state, null, 2) + '\n');
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* One run at a time per backup directory. Two runs interleaving would fetch
|
|
135
|
+
* against each other's half-written files and produce a backup of no particular
|
|
136
|
+
* moment at all, so the second exits 5, the code a caller already reads as "run
|
|
137
|
+
* this again later" rather than "this is broken".
|
|
138
|
+
*
|
|
139
|
+
* A lock whose holder is gone is broken rather than honoured: the common way to
|
|
140
|
+
* leave one behind is a machine that lost power mid-run, and a backup that
|
|
141
|
+
* stops running until someone notices a stale file is a backup that stops
|
|
142
|
+
* running. Only a lock taken on this same machine can be checked that way, so a
|
|
143
|
+
* lock from elsewhere - a backup directory on a network share - is honoured
|
|
144
|
+
* whatever its age.
|
|
145
|
+
*
|
|
146
|
+
* Not `withFileLock` from src/atomic.ts, which guards a read-modify-write of one
|
|
147
|
+
* state file: it waits for the lock and breaks one older than ten seconds, both
|
|
148
|
+
* of which are right for a critical section measured in milliseconds and wrong
|
|
149
|
+
* here. A backup of a large vault holds this for many minutes, so age says
|
|
150
|
+
* nothing about whether the holder is alive, and a second run should be told to
|
|
151
|
+
* come back rather than made to wait for a transfer it cannot know the length of.
|
|
152
|
+
*/
|
|
153
|
+
function takeLock(dir, quiet) {
|
|
154
|
+
const file = path.join(dir, LOCK_FILE);
|
|
155
|
+
const mine = JSON.stringify({ pid: process.pid, host: os.hostname(), started: new Date().toISOString() }) + '\n';
|
|
156
|
+
const attempt = () => {
|
|
157
|
+
try {
|
|
158
|
+
return fs.openSync(file, 'wx');
|
|
159
|
+
}
|
|
160
|
+
catch (e) {
|
|
161
|
+
if (e.code === 'EEXIST')
|
|
162
|
+
return null;
|
|
163
|
+
throw e;
|
|
164
|
+
}
|
|
165
|
+
};
|
|
166
|
+
let fd = attempt();
|
|
167
|
+
if (fd === null) {
|
|
168
|
+
let held = {};
|
|
169
|
+
try {
|
|
170
|
+
held = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
171
|
+
}
|
|
172
|
+
catch {
|
|
173
|
+
held = {};
|
|
174
|
+
}
|
|
175
|
+
const sameMachine = held.host === os.hostname();
|
|
176
|
+
const pid = typeof held.pid === 'number' ? held.pid : null;
|
|
177
|
+
let alive = true;
|
|
178
|
+
if (sameMachine && pid !== null) {
|
|
179
|
+
try {
|
|
180
|
+
process.kill(pid, 0);
|
|
181
|
+
}
|
|
182
|
+
catch (e) {
|
|
183
|
+
alive = e.code === 'EPERM';
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
if (alive) {
|
|
187
|
+
throw new exit_1.CliError(`Another backup is running in ${dir} (${LOCK_FILE} held by pid ${pid ?? '?'} on ${String(held.host ?? '?')}` +
|
|
188
|
+
`${held.started ? `, since ${String(held.started)}` : ''}). Nothing was changed.`, exit_1.EXIT_CONFLICT);
|
|
189
|
+
}
|
|
190
|
+
if (!quiet) {
|
|
191
|
+
console.error(`Warning: breaking a stale lock in ${dir}: pid ${pid} on this machine is gone. ` +
|
|
192
|
+
'A previous run did not finish, so this one may have more to do than usual.');
|
|
193
|
+
}
|
|
194
|
+
fs.rmSync(file, { force: true });
|
|
195
|
+
fd = attempt();
|
|
196
|
+
if (fd === null)
|
|
197
|
+
throw new exit_1.CliError(`Could not take the lock in ${dir}.`, exit_1.EXIT_CONFLICT);
|
|
198
|
+
}
|
|
199
|
+
fs.writeFileSync(fd, mine);
|
|
200
|
+
fs.closeSync(fd);
|
|
201
|
+
let released = false;
|
|
202
|
+
const release = () => {
|
|
203
|
+
if (released)
|
|
204
|
+
return;
|
|
205
|
+
released = true;
|
|
206
|
+
try {
|
|
207
|
+
fs.rmSync(file, { force: true });
|
|
208
|
+
}
|
|
209
|
+
catch {
|
|
210
|
+
// A lock that cannot be removed is a warning at worst; the next run
|
|
211
|
+
// breaks it.
|
|
212
|
+
}
|
|
213
|
+
};
|
|
214
|
+
// A run interrupted at the terminal should not leave a lock behind either.
|
|
215
|
+
// Note that having a listener at all takes away the signal's default, so this
|
|
216
|
+
// has to end the process itself: 130 is what a shell reports for a command
|
|
217
|
+
// stopped by Ctrl-C, and the next run then finds no lock to break.
|
|
218
|
+
const onSignal = () => {
|
|
219
|
+
release();
|
|
220
|
+
process.exit(130);
|
|
221
|
+
};
|
|
222
|
+
process.once('SIGINT', onSignal);
|
|
223
|
+
process.once('SIGTERM', onSignal);
|
|
224
|
+
return { release };
|
|
225
|
+
}
|
|
226
|
+
// ---- git ----
|
|
227
|
+
/**
|
|
228
|
+
* Run git, with its output captured. Captured rather than inherited so that
|
|
229
|
+
* `--json` really does put one JSON value on stdout and nothing else; the
|
|
230
|
+
* output is printed only when git failed, which is when it is worth reading.
|
|
231
|
+
*/
|
|
232
|
+
function git(args, cwd) {
|
|
233
|
+
return new Promise((resolve, reject) => {
|
|
234
|
+
const child = (0, child_process_1.spawn)('git', args, {
|
|
235
|
+
cwd,
|
|
236
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
237
|
+
env: { ...process.env, GIT_TERMINAL_PROMPT: '0' },
|
|
238
|
+
});
|
|
239
|
+
let out = '';
|
|
240
|
+
child.stdout.on('data', (d) => (out += d));
|
|
241
|
+
child.stderr.on('data', (d) => (out += d));
|
|
242
|
+
child.on('error', (e) => reject(new exit_1.CliError(e.code === 'ENOENT' ? 'git is not on PATH' : String(e.message))));
|
|
243
|
+
child.on('close', (code) => resolve({ code: code ?? 1, out }));
|
|
244
|
+
});
|
|
245
|
+
}
|
|
246
|
+
async function gitOrFail(args, what, cwd) {
|
|
247
|
+
const r = await git(args, cwd);
|
|
248
|
+
if (r.code !== 0)
|
|
249
|
+
throw new exit_1.CliError(`${what}:\n${r.out.trim()}`);
|
|
250
|
+
return r.out;
|
|
251
|
+
}
|
|
252
|
+
// ---- the manifest ----
|
|
253
|
+
/**
|
|
254
|
+
* Whether a path the vault named is one this command may write under the
|
|
255
|
+
* backup directory.
|
|
256
|
+
*
|
|
257
|
+
* Every path in a manifest becomes a path on the operator's machine, by way of
|
|
258
|
+
* path.join with the backup directory, so the manifest is input from somewhere
|
|
259
|
+
* else and is checked as such. The vault applies the same rule to the paths a
|
|
260
|
+
* fetch asks for (see vaultPath in src/api/backup.ts); without the mirror of it
|
|
261
|
+
* here, a vault that answered a manifest with `../../.bashrc` would be writing
|
|
262
|
+
* a file outside the backup, and one that answered with a repository at such a
|
|
263
|
+
* path would have this command delete a directory outside it.
|
|
264
|
+
*
|
|
265
|
+
* Refused rather than normalized, and the run stops rather than skipping the
|
|
266
|
+
* line: a vault sending one of these is not a vault whose other answers are
|
|
267
|
+
* worth acting on.
|
|
268
|
+
*/
|
|
269
|
+
function isVaultRelative(p) {
|
|
270
|
+
if (typeof p !== 'string' || p === '' || p.length > 1024)
|
|
271
|
+
return false;
|
|
272
|
+
if (p.includes('\\') || p.includes('\0') || p.startsWith('/'))
|
|
273
|
+
return false;
|
|
274
|
+
return !p.split('/').some((s) => s === '' || s === '.' || s === '..');
|
|
275
|
+
}
|
|
276
|
+
/** How a refused path is reported, in one place since two kinds of line carry one. */
|
|
277
|
+
function refusedPath(p) {
|
|
278
|
+
return (`The vault named a path this backup will not write: ${JSON.stringify(p)}. ` +
|
|
279
|
+
'A manifest path must be relative to the vault and must not climb out of it, so nothing was copied.');
|
|
280
|
+
}
|
|
281
|
+
async function fetchManifest(target, exclude, hash) {
|
|
282
|
+
const query = [];
|
|
283
|
+
if (exclude.length)
|
|
284
|
+
query.push(`exclude=${encodeURIComponent(exclude.join(','))}`);
|
|
285
|
+
if (hash)
|
|
286
|
+
query.push('hash=1');
|
|
287
|
+
const url = `${target.host}/api/backup/manifest${query.length ? `?${query.join('&')}` : ''}`;
|
|
288
|
+
let resp;
|
|
289
|
+
try {
|
|
290
|
+
resp = await fetch(url, { headers: { authorization: `Bearer ${target.token}` } });
|
|
291
|
+
}
|
|
292
|
+
catch (e) {
|
|
293
|
+
throw new exit_1.CliError(`Could not reach ${target.host}: ${e instanceof Error ? e.message : e}`);
|
|
294
|
+
}
|
|
295
|
+
if (!resp.ok) {
|
|
296
|
+
let message = `HTTP ${resp.status} from ${url}`;
|
|
297
|
+
try {
|
|
298
|
+
const data = JSON.parse(await resp.text());
|
|
299
|
+
if (data.error)
|
|
300
|
+
message = String(data.error);
|
|
301
|
+
}
|
|
302
|
+
catch {
|
|
303
|
+
// not JSON; the status is all there is to say
|
|
304
|
+
}
|
|
305
|
+
// A vault that answers other routes and not this one does not have these
|
|
306
|
+
// routes, which means it is older than this client. Worth saying, because a
|
|
307
|
+
// bare 404 here reads as "no such vault" and sends the reader looking at the
|
|
308
|
+
// URL and the token, neither of which is the problem.
|
|
309
|
+
if (resp.status === 404) {
|
|
310
|
+
message =
|
|
311
|
+
`${target.host} has no /api/backup/manifest route, so it is running a mochi older than this ` +
|
|
312
|
+
'command. Deploy the vault again from a version that has it, then run this.';
|
|
313
|
+
}
|
|
314
|
+
// The same status-to-code mapping every other command uses, so that a
|
|
315
|
+
// caller branching on the exit code does not have to learn a second table.
|
|
316
|
+
throw new exit_1.CliError(message, (0, exit_1.exitCodeForStatus)(resp.status));
|
|
317
|
+
}
|
|
318
|
+
const manifest = {
|
|
319
|
+
lfs: 'volume',
|
|
320
|
+
excluded: [],
|
|
321
|
+
files: new Map(),
|
|
322
|
+
repos: [],
|
|
323
|
+
counts: { files: 0, bytes: 0, repos: 0 },
|
|
324
|
+
};
|
|
325
|
+
let ended = false;
|
|
326
|
+
for await (const line of ndjson(resp)) {
|
|
327
|
+
const kind = line.kind;
|
|
328
|
+
if (kind === 'vault') {
|
|
329
|
+
const v = line;
|
|
330
|
+
if (typeof v.lfs === 'string')
|
|
331
|
+
manifest.lfs = v.lfs;
|
|
332
|
+
if (Array.isArray(v.excluded))
|
|
333
|
+
manifest.excluded = v.excluded;
|
|
334
|
+
}
|
|
335
|
+
else if (kind === 'file') {
|
|
336
|
+
const f = line;
|
|
337
|
+
if (!isVaultRelative(f.path))
|
|
338
|
+
throw new exit_1.CliError(refusedPath(f.path));
|
|
339
|
+
manifest.files.set(f.path, f);
|
|
340
|
+
}
|
|
341
|
+
else if (kind === 'repo') {
|
|
342
|
+
const r = line;
|
|
343
|
+
if (!isVaultRelative(r.path))
|
|
344
|
+
throw new exit_1.CliError(refusedPath(r.path));
|
|
345
|
+
manifest.repos.push(r);
|
|
346
|
+
}
|
|
347
|
+
else if (kind === 'end') {
|
|
348
|
+
const e = line;
|
|
349
|
+
manifest.counts = { files: e.files, bytes: e.bytes, repos: e.repos };
|
|
350
|
+
ended = true;
|
|
351
|
+
}
|
|
352
|
+
else if (kind === 'error') {
|
|
353
|
+
throw new exit_1.CliError(`The vault could not finish the manifest: ${String(line.error)}`);
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
// The end line is what says the walk completed. Acting on a truncated
|
|
357
|
+
// manifest would delete every path the vault did not get around to listing.
|
|
358
|
+
if (!ended) {
|
|
359
|
+
throw new exit_1.CliError('The manifest ended early, so what the vault holds is not fully known. Nothing was deleted; try again.');
|
|
360
|
+
}
|
|
361
|
+
return manifest;
|
|
362
|
+
}
|
|
363
|
+
/** Each line of an NDJSON response, parsed. */
|
|
364
|
+
async function* ndjson(resp) {
|
|
365
|
+
const body = resp.body;
|
|
366
|
+
if (!body)
|
|
367
|
+
return;
|
|
368
|
+
let buf = '';
|
|
369
|
+
for await (const chunk of body) {
|
|
370
|
+
buf += Buffer.from(chunk).toString('utf8');
|
|
371
|
+
let nl;
|
|
372
|
+
while ((nl = buf.indexOf('\n')) !== -1) {
|
|
373
|
+
const line = buf.slice(0, nl);
|
|
374
|
+
buf = buf.slice(nl + 1);
|
|
375
|
+
if (line.trim() === '')
|
|
376
|
+
continue;
|
|
377
|
+
yield JSON.parse(line);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
if (buf.trim() !== '')
|
|
381
|
+
yield JSON.parse(buf);
|
|
382
|
+
}
|
|
383
|
+
// ---- the length-prefixed fetch stream ----
|
|
384
|
+
/**
|
|
385
|
+
* Reads the framed response of POST /api/backup/fetch: a JSON line, then
|
|
386
|
+
* exactly the bytes it declared, repeated, then a line saying what was missing.
|
|
387
|
+
* A byte reader rather than a line reader, because the payloads are arbitrary
|
|
388
|
+
* bytes and a newline inside one means nothing.
|
|
389
|
+
*/
|
|
390
|
+
class FrameReader {
|
|
391
|
+
buf = Buffer.alloc(0);
|
|
392
|
+
done = false;
|
|
393
|
+
it;
|
|
394
|
+
constructor(body) {
|
|
395
|
+
this.it = body[Symbol.asyncIterator]();
|
|
396
|
+
}
|
|
397
|
+
async more() {
|
|
398
|
+
if (this.done)
|
|
399
|
+
return false;
|
|
400
|
+
const next = await this.it.next();
|
|
401
|
+
if (next.done) {
|
|
402
|
+
this.done = true;
|
|
403
|
+
return false;
|
|
404
|
+
}
|
|
405
|
+
this.buf = Buffer.concat([this.buf, Buffer.from(next.value)]);
|
|
406
|
+
return true;
|
|
407
|
+
}
|
|
408
|
+
/** The next line, without its newline, or null at the end of the stream. */
|
|
409
|
+
async line() {
|
|
410
|
+
for (;;) {
|
|
411
|
+
const nl = this.buf.indexOf(0x0a);
|
|
412
|
+
if (nl !== -1) {
|
|
413
|
+
const line = this.buf.subarray(0, nl).toString('utf8');
|
|
414
|
+
this.buf = this.buf.subarray(nl + 1);
|
|
415
|
+
return line;
|
|
416
|
+
}
|
|
417
|
+
if (!(await this.more())) {
|
|
418
|
+
if (this.buf.length === 0)
|
|
419
|
+
return null;
|
|
420
|
+
const line = this.buf.toString('utf8');
|
|
421
|
+
this.buf = Buffer.alloc(0);
|
|
422
|
+
return line;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
/** Exactly n bytes, handed to the sink as they arrive. */
|
|
427
|
+
async bytes(n, sink) {
|
|
428
|
+
let left = n;
|
|
429
|
+
while (left > 0) {
|
|
430
|
+
if (this.buf.length === 0 && !(await this.more())) {
|
|
431
|
+
throw new exit_1.CliError('The vault closed the connection part way through a file. Nothing was left half-written.');
|
|
432
|
+
}
|
|
433
|
+
const take = Math.min(left, this.buf.length);
|
|
434
|
+
sink(this.buf.subarray(0, take));
|
|
435
|
+
this.buf = this.buf.subarray(take);
|
|
436
|
+
left -= take;
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
// ---- writing into current/ ----
|
|
441
|
+
/**
|
|
442
|
+
* Write one file into the backup, by a temporary file in the same directory and
|
|
443
|
+
* a rename. Required by the snapshots: a hardlinked snapshot shares the inode,
|
|
444
|
+
* so writing in place would edit every snapshot that ever linked this file.
|
|
445
|
+
*/
|
|
446
|
+
function writeVia(dest, mode, fill) {
|
|
447
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
448
|
+
const tmp = `${dest}.tmp-${process.pid}`;
|
|
449
|
+
const fd = fs.openSync(tmp, 'w', mode & 0o777 ? mode & 0o777 : 0o644);
|
|
450
|
+
return fill((b) => {
|
|
451
|
+
fs.writeSync(fd, b);
|
|
452
|
+
})
|
|
453
|
+
.then(() => {
|
|
454
|
+
fs.closeSync(fd);
|
|
455
|
+
fs.renameSync(tmp, dest);
|
|
456
|
+
})
|
|
457
|
+
.catch((e) => {
|
|
458
|
+
try {
|
|
459
|
+
fs.closeSync(fd);
|
|
460
|
+
}
|
|
461
|
+
catch {
|
|
462
|
+
// already closed
|
|
463
|
+
}
|
|
464
|
+
fs.rmSync(tmp, { force: true });
|
|
465
|
+
throw e;
|
|
466
|
+
});
|
|
467
|
+
}
|
|
468
|
+
/** The same chunked hash the vault computes, so a --checksum run of a large vault
|
|
469
|
+
* costs a fixed amount of memory on this end too. */
|
|
470
|
+
function sha256Of(file) {
|
|
471
|
+
const h = crypto.createHash('sha256');
|
|
472
|
+
const fd = fs.openSync(file, 'r');
|
|
473
|
+
try {
|
|
474
|
+
const buf = Buffer.allocUnsafe(1 << 16);
|
|
475
|
+
for (;;) {
|
|
476
|
+
const n = fs.readSync(fd, buf, 0, buf.length, null);
|
|
477
|
+
if (n === 0)
|
|
478
|
+
break;
|
|
479
|
+
h.update(buf.subarray(0, n));
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
finally {
|
|
483
|
+
fs.closeSync(fd);
|
|
484
|
+
}
|
|
485
|
+
return h.digest('hex');
|
|
486
|
+
}
|
|
487
|
+
// ---- snapshots ----
|
|
488
|
+
/** The UTC stamp a snapshot directory is named by: 2026-08-19T140311Z. */
|
|
489
|
+
function stampNow() {
|
|
490
|
+
const iso = new Date().toISOString();
|
|
491
|
+
return `${iso.slice(0, 10)}T${iso.slice(11, 13)}${iso.slice(14, 16)}${iso.slice(17, 19)}Z`;
|
|
492
|
+
}
|
|
493
|
+
const STAMP_RE = /^(\d{4})-(\d{2})-(\d{2})T(\d{2})(\d{2})(\d{2})Z$/;
|
|
494
|
+
function stampDate(name) {
|
|
495
|
+
const m = name.match(STAMP_RE);
|
|
496
|
+
if (!m)
|
|
497
|
+
return null;
|
|
498
|
+
const [, y, mo, d, h, mi, s] = m;
|
|
499
|
+
return new Date(Date.UTC(+y, +mo - 1, +d, +h, +mi, +s));
|
|
500
|
+
}
|
|
501
|
+
function listSnapshots(dir) {
|
|
502
|
+
const base = path.join(dir, SNAPSHOTS);
|
|
503
|
+
let names;
|
|
504
|
+
try {
|
|
505
|
+
names = fs.readdirSync(base);
|
|
506
|
+
}
|
|
507
|
+
catch {
|
|
508
|
+
return [];
|
|
509
|
+
}
|
|
510
|
+
return names
|
|
511
|
+
.map((name) => ({ name, at: stampDate(name) }))
|
|
512
|
+
.filter((s) => s.at !== null)
|
|
513
|
+
.sort((a, b) => b.at.getTime() - a.at.getTime());
|
|
514
|
+
}
|
|
515
|
+
/** Every file under dir, as paths relative to it. Directories are not listed. */
|
|
516
|
+
function walkFiles(dir, rel = '', out = []) {
|
|
517
|
+
let entries;
|
|
518
|
+
try {
|
|
519
|
+
entries = fs.readdirSync(path.join(dir, rel), { withFileTypes: true });
|
|
520
|
+
}
|
|
521
|
+
catch {
|
|
522
|
+
return out;
|
|
523
|
+
}
|
|
524
|
+
for (const e of entries) {
|
|
525
|
+
const child = rel ? `${rel}/${e.name}` : e.name;
|
|
526
|
+
if (e.isDirectory())
|
|
527
|
+
walkFiles(dir, child, out);
|
|
528
|
+
else if (e.isFile())
|
|
529
|
+
out.push(child);
|
|
530
|
+
}
|
|
531
|
+
return out;
|
|
532
|
+
}
|
|
533
|
+
/**
|
|
534
|
+
* A snapshot of current/ as a directory of hardlinks: one inode per file and no
|
|
535
|
+
* data, and still a servable vault.
|
|
536
|
+
*
|
|
537
|
+
* The link count is checked as it goes. A file in current/ can legitimately be
|
|
538
|
+
* linked once per existing snapshot plus once for current/ itself; more links
|
|
539
|
+
* than that means something outside this backup shares the inode, and hardlinking
|
|
540
|
+
* it would tie the snapshot to a file this program does not control. That is
|
|
541
|
+
* refused rather than recorded, because the failure it guards against - somebody
|
|
542
|
+
* appending to a file under current/ - silently corrupts every snapshot at once.
|
|
543
|
+
*/
|
|
544
|
+
function takeSnapshot(dir, quiet) {
|
|
545
|
+
const from = path.join(dir, CURRENT);
|
|
546
|
+
const before = listSnapshots(dir).length;
|
|
547
|
+
const name = stampNow();
|
|
548
|
+
const to = path.join(dir, SNAPSHOTS, name);
|
|
549
|
+
if (fs.existsSync(to)) {
|
|
550
|
+
throw new exit_1.CliError(`A snapshot named ${name} is already there, so this second is left alone.`, exit_1.EXIT_CONFLICT);
|
|
551
|
+
}
|
|
552
|
+
const files = walkFiles(from);
|
|
553
|
+
const maxLinks = before + 1;
|
|
554
|
+
fs.mkdirSync(to, { recursive: true });
|
|
555
|
+
let linked = 0;
|
|
556
|
+
try {
|
|
557
|
+
for (const rel of files) {
|
|
558
|
+
const src = path.join(from, rel);
|
|
559
|
+
const st = fs.lstatSync(src);
|
|
560
|
+
if (st.nlink > maxLinks) {
|
|
561
|
+
throw new exit_1.CliError(`${rel} in the backup has ${st.nlink} hard links, more than the ${maxLinks} this backup can account for. ` +
|
|
562
|
+
'Something outside the backup shares the file, so snapshotting it is refused. ' +
|
|
563
|
+
'See the note on hardlinks in docs/backup.md.');
|
|
564
|
+
}
|
|
565
|
+
const dest = path.join(to, rel);
|
|
566
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
567
|
+
fs.linkSync(src, dest);
|
|
568
|
+
linked++;
|
|
569
|
+
}
|
|
570
|
+
}
|
|
571
|
+
catch (e) {
|
|
572
|
+
// A half-built snapshot is worse than none: it would be pruned as if it
|
|
573
|
+
// were a copy of the vault at some moment, and it is not.
|
|
574
|
+
fs.rmSync(to, { recursive: true, force: true });
|
|
575
|
+
throw e;
|
|
576
|
+
}
|
|
577
|
+
if (!quiet)
|
|
578
|
+
console.error(`Snapshot ${name}: ${linked} files hardlinked`);
|
|
579
|
+
return { name, files: linked };
|
|
580
|
+
}
|
|
581
|
+
function isoWeekKey(d) {
|
|
582
|
+
// ISO weeks, in UTC: Thursday decides the year, so shift to that week's
|
|
583
|
+
// Thursday and count weeks from the first one.
|
|
584
|
+
const t = new Date(Date.UTC(d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate()));
|
|
585
|
+
const day = (t.getUTCDay() + 6) % 7; // Monday = 0
|
|
586
|
+
t.setUTCDate(t.getUTCDate() - day + 3);
|
|
587
|
+
const firstThursday = new Date(Date.UTC(t.getUTCFullYear(), 0, 4));
|
|
588
|
+
const firstDay = (firstThursday.getUTCDay() + 6) % 7;
|
|
589
|
+
firstThursday.setUTCDate(firstThursday.getUTCDate() - firstDay + 3);
|
|
590
|
+
const week = 1 + Math.round((t.getTime() - firstThursday.getTime()) / (7 * 86400000));
|
|
591
|
+
return `${t.getUTCFullYear()}-W${String(week).padStart(2, '0')}`;
|
|
592
|
+
}
|
|
593
|
+
/**
|
|
594
|
+
* Grandfather-father-son: the newest snapshot of each of the last N days, weeks,
|
|
595
|
+
* and months is kept, everything else is dropped. Evaluated in UTC, so the
|
|
596
|
+
* decision does not move with the machine's timezone or with daylight saving.
|
|
597
|
+
*/
|
|
598
|
+
function snapshotsToKeep(snapshots, retention) {
|
|
599
|
+
const keep = new Set();
|
|
600
|
+
const byPeriod = (key, n) => {
|
|
601
|
+
if (n <= 0)
|
|
602
|
+
return;
|
|
603
|
+
const seen = new Set();
|
|
604
|
+
// Newest first, so the first snapshot of a period is the newest in it.
|
|
605
|
+
for (const s of snapshots) {
|
|
606
|
+
const k = key(s.at);
|
|
607
|
+
if (seen.has(k))
|
|
608
|
+
continue;
|
|
609
|
+
seen.add(k);
|
|
610
|
+
if (seen.size > n)
|
|
611
|
+
break;
|
|
612
|
+
keep.add(s.name);
|
|
613
|
+
}
|
|
614
|
+
};
|
|
615
|
+
byPeriod((d) => d.toISOString().slice(0, 10), retention.daily);
|
|
616
|
+
byPeriod(isoWeekKey, retention.weekly);
|
|
617
|
+
byPeriod((d) => d.toISOString().slice(0, 7), retention.monthly);
|
|
618
|
+
// The newest is always kept, whatever the policy says: a retention of zeroes
|
|
619
|
+
// is a policy for how long to keep history, not permission to leave none.
|
|
620
|
+
if (snapshots.length)
|
|
621
|
+
keep.add(snapshots[0].name);
|
|
622
|
+
return keep;
|
|
623
|
+
}
|
|
624
|
+
function pruneSnapshots(dir, retention, quiet) {
|
|
625
|
+
const snapshots = listSnapshots(dir);
|
|
626
|
+
const keep = snapshotsToKeep(snapshots, retention);
|
|
627
|
+
const dropped = [];
|
|
628
|
+
for (const s of snapshots) {
|
|
629
|
+
if (keep.has(s.name))
|
|
630
|
+
continue;
|
|
631
|
+
fs.rmSync(path.join(dir, SNAPSHOTS, s.name), { recursive: true, force: true });
|
|
632
|
+
dropped.push(s.name);
|
|
633
|
+
}
|
|
634
|
+
if (dropped.length && !quiet) {
|
|
635
|
+
console.error(`Pruned ${dropped.length} snapshot${dropped.length === 1 ? '' : 's'}: ${dropped.join(', ')}`);
|
|
636
|
+
}
|
|
637
|
+
return dropped;
|
|
638
|
+
}
|
|
639
|
+
/** Apparent size: what the files say, before hardlinks are taken into account. */
|
|
640
|
+
function apparentSize(dir) {
|
|
641
|
+
let total = 0;
|
|
642
|
+
for (const rel of walkFiles(dir)) {
|
|
643
|
+
try {
|
|
644
|
+
total += fs.statSync(path.join(dir, rel)).size;
|
|
645
|
+
}
|
|
646
|
+
catch {
|
|
647
|
+
// vanished under us; nothing to add
|
|
648
|
+
}
|
|
649
|
+
}
|
|
650
|
+
return total;
|
|
651
|
+
}
|
|
652
|
+
function human(bytes) {
|
|
653
|
+
const units = ['B', 'kB', 'MB', 'GB', 'TB'];
|
|
654
|
+
let n = bytes;
|
|
655
|
+
let i = 0;
|
|
656
|
+
while (n >= 1000 && i < units.length - 1) {
|
|
657
|
+
n /= 1000;
|
|
658
|
+
i++;
|
|
659
|
+
}
|
|
660
|
+
return `${i === 0 ? n : n.toFixed(n < 10 ? 1 : 0)} ${units[i]}`;
|
|
661
|
+
}
|
|
662
|
+
// ---- options ----
|
|
663
|
+
const RETENTION_OPTIONS = [
|
|
664
|
+
{ name: 'keep-daily', type: 'int', value: '<n>', summary: `Daily snapshots to keep (default ${DEFAULT_RETENTION.daily})` },
|
|
665
|
+
{ name: 'keep-weekly', type: 'int', value: '<n>', summary: `Weekly snapshots to keep (default ${DEFAULT_RETENTION.weekly})` },
|
|
666
|
+
{
|
|
667
|
+
name: 'keep-monthly',
|
|
668
|
+
type: 'int',
|
|
669
|
+
value: '<n>',
|
|
670
|
+
summary: `Monthly snapshots to keep (default ${DEFAULT_RETENTION.monthly})`,
|
|
671
|
+
},
|
|
672
|
+
];
|
|
673
|
+
const EXCLUDE_OPTIONS = [
|
|
674
|
+
{ name: 'no-runs', type: 'boolean', summary: 'Leave out workflow run history (<repo>.runs)' },
|
|
675
|
+
{ name: 'no-sites', type: 'boolean', summary: 'Leave out published sites (<repo>.site)' },
|
|
676
|
+
{ name: 'no-lfs', type: 'boolean', summary: 'Leave out LFS objects on the volume (<repo>.lfs)' },
|
|
677
|
+
{ name: 'no-secrets', type: 'boolean', summary: 'Leave out vault.json, runners.json, and .secret' },
|
|
678
|
+
];
|
|
679
|
+
const QUIET_OPTION = { name: 'quiet', type: 'boolean', summary: 'Say nothing on success' };
|
|
680
|
+
/** The backup directory a command was given, made if it is not there yet. */
|
|
681
|
+
function backupDirectory(inv, create) {
|
|
682
|
+
const given = inv.args[0];
|
|
683
|
+
if (!given)
|
|
684
|
+
throw new exit_1.CliError('Which directory? Usage: mochi backup <dir>', exit_1.EXIT_USAGE);
|
|
685
|
+
const dir = path.resolve(given);
|
|
686
|
+
if (!fs.existsSync(dir)) {
|
|
687
|
+
if (!create)
|
|
688
|
+
throw new exit_1.CliError(`No backup directory at ${dir}.`, exit_1.EXIT_USAGE);
|
|
689
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
690
|
+
}
|
|
691
|
+
else if (!fs.statSync(dir).isDirectory()) {
|
|
692
|
+
throw new exit_1.CliError(`${dir} is not a directory.`, exit_1.EXIT_USAGE);
|
|
693
|
+
}
|
|
694
|
+
return dir;
|
|
695
|
+
}
|
|
696
|
+
/** An existing backup directory, refusing one that has never been synced. */
|
|
697
|
+
function existingBackup(inv) {
|
|
698
|
+
const dir = backupDirectory(inv, false);
|
|
699
|
+
if (!fs.existsSync(statePath(dir))) {
|
|
700
|
+
throw new exit_1.CliError(`${dir} holds no backup (no ${STATE_FILE}). Make one first: mochi backup ${inv.args[0]}`, exit_1.EXIT_USAGE);
|
|
701
|
+
}
|
|
702
|
+
return { dir, state: loadState(dir) };
|
|
703
|
+
}
|
|
704
|
+
/**
|
|
705
|
+
* The exclusions in force. Sticky, because a cron entry is the command and a
|
|
706
|
+
* directory and should not have to repeat them: naming none keeps whatever the
|
|
707
|
+
* last run used, and naming any at all replaces the set, which is how a
|
|
708
|
+
* category can be put back.
|
|
709
|
+
*/
|
|
710
|
+
function exclusionsFor(inv, state) {
|
|
711
|
+
const given = [];
|
|
712
|
+
if (inv.bool('no-runs'))
|
|
713
|
+
given.push('runs');
|
|
714
|
+
if (inv.bool('no-sites'))
|
|
715
|
+
given.push('sites');
|
|
716
|
+
if (inv.bool('no-lfs'))
|
|
717
|
+
given.push('lfs');
|
|
718
|
+
if (inv.bool('no-secrets'))
|
|
719
|
+
given.push('secrets');
|
|
720
|
+
return given.length ? given : state.excluded;
|
|
721
|
+
}
|
|
722
|
+
function retentionFor(inv, state) {
|
|
723
|
+
return {
|
|
724
|
+
daily: inv.int('keep-daily') ?? state.retention.daily,
|
|
725
|
+
weekly: inv.int('keep-weekly') ?? state.retention.weekly,
|
|
726
|
+
monthly: inv.int('keep-monthly') ?? state.retention.monthly,
|
|
727
|
+
};
|
|
728
|
+
}
|
|
729
|
+
/**
|
|
730
|
+
* Which vault, and with what token. targetFrom does the work, including the
|
|
731
|
+
* --token-stdin rules and the exit codes docs/cli.md promises for them; the URL
|
|
732
|
+
* recorded in backup.json is handed to it as the fallback that outranks the last
|
|
733
|
+
* login, so a backup directory keeps pointing at the vault it is a backup of.
|
|
734
|
+
*/
|
|
735
|
+
async function targetForBackup(inv, state) {
|
|
736
|
+
return await (0, target_1.targetFrom)(inv, { host: state.host || null });
|
|
737
|
+
}
|
|
738
|
+
/**
|
|
739
|
+
* Credentials for a git call against the vault, as a per-invocation config.
|
|
740
|
+
* The backup token is a site admin's, so it may read every repository the
|
|
741
|
+
* manifest names, private ones included; git alone would clone anonymously
|
|
742
|
+
* and be told a private repository is not there. The username half of the
|
|
743
|
+
* Basic pair is a placeholder: the server identifies a token by its value.
|
|
744
|
+
*/
|
|
745
|
+
function gitAuth(target) {
|
|
746
|
+
const basic = Buffer.from(`x-token:${target.token}`, 'utf8').toString('base64');
|
|
747
|
+
return ['-c', `http.extraHeader=Authorization: Basic ${basic}`];
|
|
748
|
+
}
|
|
749
|
+
async function syncRepo(target, current, entry, quiet) {
|
|
750
|
+
const dest = path.join(current, entry.path);
|
|
751
|
+
const url = `${target.host}/${encodeURIComponent(entry.collection)}/${encodeURIComponent(entry.repo)}`;
|
|
752
|
+
if (!(0, scan_1.isBareRepo)(dest)) {
|
|
753
|
+
// A directory that is there but is not a repository is a previous run that
|
|
754
|
+
// died during its clone. Nothing in it is worth keeping.
|
|
755
|
+
fs.rmSync(dest, { recursive: true, force: true });
|
|
756
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
757
|
+
if (!quiet)
|
|
758
|
+
console.error(`Cloning ${entry.collection}/${entry.repo}`);
|
|
759
|
+
await gitOrFail([...gitAuth(target), 'clone', '--mirror', '--', url, dest], `could not clone ${url}`);
|
|
760
|
+
// Reflogs are the one thing git appends to in place, and an appended file
|
|
761
|
+
// corrupts every snapshot that has already hardlinked it. A mirror has no
|
|
762
|
+
// use for them anyway: it holds no work of its own to recover.
|
|
763
|
+
await gitOrFail(['-C', dest, 'config', 'core.logAllRefUpdates', 'false'], 'could not configure the mirror');
|
|
764
|
+
return 'cloned';
|
|
765
|
+
}
|
|
766
|
+
if (!quiet)
|
|
767
|
+
console.error(`Fetching ${entry.collection}/${entry.repo}`);
|
|
768
|
+
// The refspec is written out, and forced, so that a ref the vault rewound or
|
|
769
|
+
// force-pushed is followed rather than refused: this is a copy of the vault,
|
|
770
|
+
// not a branch with opinions of its own. --prune is what carries a deletion
|
|
771
|
+
// across.
|
|
772
|
+
await gitOrFail([...gitAuth(target), '-C', dest, 'fetch', '--prune', '--', url, '+refs/*:refs/*'], `could not fetch ${url}`);
|
|
773
|
+
return 'fetched';
|
|
774
|
+
}
|
|
775
|
+
/**
|
|
776
|
+
* Point the mirror's HEAD where the vault's points. A fetch does not carry HEAD,
|
|
777
|
+
* so without this a repository whose default branch was changed would come back
|
|
778
|
+
* from the backup still naming the old one.
|
|
779
|
+
*/
|
|
780
|
+
async function syncHead(target, current, entry) {
|
|
781
|
+
const dest = path.join(current, entry.path);
|
|
782
|
+
const url = `${target.host}/${encodeURIComponent(entry.collection)}/${encodeURIComponent(entry.repo)}`;
|
|
783
|
+
const r = await git([...gitAuth(target), 'ls-remote', '--symref', '--', url, 'HEAD']);
|
|
784
|
+
if (r.code !== 0)
|
|
785
|
+
return;
|
|
786
|
+
const m = r.out.match(/^ref:\s+(\S+)\s+HEAD$/m);
|
|
787
|
+
if (!m)
|
|
788
|
+
return;
|
|
789
|
+
const want = m[1];
|
|
790
|
+
const have = await git(['-C', dest, 'symbolic-ref', '--quiet', 'HEAD']);
|
|
791
|
+
if (have.code === 0 && have.out.trim() === want)
|
|
792
|
+
return;
|
|
793
|
+
await git(['-C', dest, 'symbolic-ref', 'HEAD', want]);
|
|
794
|
+
}
|
|
795
|
+
/**
|
|
796
|
+
* Whether this file has to be fetched, which is the question the whole
|
|
797
|
+
* incremental half turns on.
|
|
798
|
+
*
|
|
799
|
+
* Size and modification time, as rsync does by default, and against both ends:
|
|
800
|
+
* the vault's timestamp must match the one the last run recorded, and the copy
|
|
801
|
+
* on disk must still be the copy that run wrote. The second half is what
|
|
802
|
+
* repairs a backup that was damaged locally; a comparison against the recorded
|
|
803
|
+
* state alone would call a truncated or edited copy up to date forever.
|
|
804
|
+
*
|
|
805
|
+
* `--checksum` asks the vault for hashes and compares those against the bytes
|
|
806
|
+
* on disk, which is slower, reads everything, and is the mode to reach for when
|
|
807
|
+
* something is suspected rather than the one to run nightly.
|
|
808
|
+
*/
|
|
809
|
+
function needsFetch(f, dest, known, checksum) {
|
|
810
|
+
let st;
|
|
811
|
+
try {
|
|
812
|
+
st = fs.statSync(dest);
|
|
813
|
+
}
|
|
814
|
+
catch {
|
|
815
|
+
return true;
|
|
816
|
+
}
|
|
817
|
+
if (checksum)
|
|
818
|
+
return !f.sha256 || sha256Of(dest) !== f.sha256;
|
|
819
|
+
if (!known)
|
|
820
|
+
return true;
|
|
821
|
+
if (st.size !== f.size || known.size !== f.size)
|
|
822
|
+
return true;
|
|
823
|
+
if (known.mtime !== f.mtime)
|
|
824
|
+
return true;
|
|
825
|
+
return st.mtimeMs !== known.local;
|
|
826
|
+
}
|
|
827
|
+
/** The paths to fetch, in chunks that fit the server's caps. */
|
|
828
|
+
function chunkPaths(files) {
|
|
829
|
+
const chunks = [];
|
|
830
|
+
let chunk = [];
|
|
831
|
+
let bytes = 0;
|
|
832
|
+
for (const f of files) {
|
|
833
|
+
// A file over the whole per-request cap travels alone, which the server
|
|
834
|
+
// allows precisely so that a large one is fetchable at all.
|
|
835
|
+
if (f.size > MAX_FETCH_BYTES) {
|
|
836
|
+
if (chunk.length) {
|
|
837
|
+
chunks.push(chunk);
|
|
838
|
+
chunk = [];
|
|
839
|
+
bytes = 0;
|
|
840
|
+
}
|
|
841
|
+
chunks.push([f]);
|
|
842
|
+
continue;
|
|
843
|
+
}
|
|
844
|
+
if (chunk.length >= MAX_FETCH_PATHS || bytes + f.size > MAX_FETCH_BYTES) {
|
|
845
|
+
chunks.push(chunk);
|
|
846
|
+
chunk = [];
|
|
847
|
+
bytes = 0;
|
|
848
|
+
}
|
|
849
|
+
chunk.push(f);
|
|
850
|
+
bytes += f.size;
|
|
851
|
+
}
|
|
852
|
+
if (chunk.length)
|
|
853
|
+
chunks.push(chunk);
|
|
854
|
+
return chunks;
|
|
855
|
+
}
|
|
856
|
+
/**
|
|
857
|
+
* Fetch one chunk of files into current/, returning what the vault said was
|
|
858
|
+
* missing. A missing file is not a failure: run history is trimmed while a
|
|
859
|
+
* backup of it is in flight, and the manifest was a walk of a live tree.
|
|
860
|
+
*/
|
|
861
|
+
async function fetchChunk(target, current, chunk, state, wantHashes) {
|
|
862
|
+
const byPath = new Map(chunk.map((f) => [f.path, f]));
|
|
863
|
+
let resp;
|
|
864
|
+
try {
|
|
865
|
+
resp = await fetch(`${target.host}/api/backup/fetch`, {
|
|
866
|
+
method: 'POST',
|
|
867
|
+
headers: { authorization: `Bearer ${target.token}`, 'content-type': 'application/json' },
|
|
868
|
+
body: JSON.stringify({ paths: chunk.map((f) => f.path) }),
|
|
869
|
+
});
|
|
870
|
+
}
|
|
871
|
+
catch (e) {
|
|
872
|
+
throw new exit_1.CliError(`Could not reach ${target.host}: ${e instanceof Error ? e.message : e}`);
|
|
873
|
+
}
|
|
874
|
+
if (!resp.ok) {
|
|
875
|
+
let message = `HTTP ${resp.status} from ${target.host}/api/backup/fetch`;
|
|
876
|
+
try {
|
|
877
|
+
const data = JSON.parse(await resp.text());
|
|
878
|
+
if (data.error)
|
|
879
|
+
message = String(data.error);
|
|
880
|
+
}
|
|
881
|
+
catch {
|
|
882
|
+
// not JSON
|
|
883
|
+
}
|
|
884
|
+
throw new exit_1.CliError(message, (0, exit_1.exitCodeForStatus)(resp.status));
|
|
885
|
+
}
|
|
886
|
+
if (!resp.body)
|
|
887
|
+
throw new exit_1.CliError('The vault answered a fetch with no body.');
|
|
888
|
+
const reader = new FrameReader(resp.body);
|
|
889
|
+
let bytes = 0;
|
|
890
|
+
let missing = [];
|
|
891
|
+
for (;;) {
|
|
892
|
+
const line = await reader.line();
|
|
893
|
+
if (line === null)
|
|
894
|
+
throw new exit_1.CliError('The vault ended a fetch without saying it had finished.');
|
|
895
|
+
const frame = JSON.parse(line);
|
|
896
|
+
if (frame.end) {
|
|
897
|
+
missing = frame.missing ?? [];
|
|
898
|
+
break;
|
|
899
|
+
}
|
|
900
|
+
const rel = frame.path;
|
|
901
|
+
const size = frame.size;
|
|
902
|
+
if (typeof rel !== 'string' || typeof size !== 'number' || !byPath.has(rel)) {
|
|
903
|
+
throw new exit_1.CliError(`The vault sent a file this run did not ask for: ${String(rel)}`);
|
|
904
|
+
}
|
|
905
|
+
const wanted = byPath.get(rel);
|
|
906
|
+
const dest = path.join(current, ...rel.split('/'));
|
|
907
|
+
const hash = wantHashes ? crypto.createHash('sha256') : null;
|
|
908
|
+
await writeVia(dest, wanted.mode, async (write) => {
|
|
909
|
+
await reader.bytes(size, (b) => {
|
|
910
|
+
if (hash)
|
|
911
|
+
hash.update(b);
|
|
912
|
+
write(b);
|
|
913
|
+
});
|
|
914
|
+
});
|
|
915
|
+
// The vault's own timestamp, so that the next run's size-and-mtime
|
|
916
|
+
// comparison is against what the vault has rather than against when this
|
|
917
|
+
// run happened.
|
|
918
|
+
const mtime = wanted.mtime / 1000;
|
|
919
|
+
try {
|
|
920
|
+
fs.utimesSync(dest, mtime, mtime);
|
|
921
|
+
}
|
|
922
|
+
catch {
|
|
923
|
+
// A filesystem that will not take a timestamp costs this backup a
|
|
924
|
+
// re-fetch of the file next time, and nothing else.
|
|
925
|
+
}
|
|
926
|
+
let local = wanted.mtime;
|
|
927
|
+
try {
|
|
928
|
+
local = fs.statSync(dest).mtimeMs;
|
|
929
|
+
}
|
|
930
|
+
catch {
|
|
931
|
+
// Written a moment ago; if it cannot be stat'ed the next run fetches it.
|
|
932
|
+
}
|
|
933
|
+
const record = { size, mtime: wanted.mtime, local, mode: wanted.mode };
|
|
934
|
+
if (hash)
|
|
935
|
+
record.sha256 = hash.digest('hex');
|
|
936
|
+
state.files[rel] = record;
|
|
937
|
+
bytes += size;
|
|
938
|
+
}
|
|
939
|
+
return { bytes, missing };
|
|
940
|
+
}
|
|
941
|
+
/**
|
|
942
|
+
* A mirror's config is one of the files copied from the vault, so after a fetch
|
|
943
|
+
* the mirror's settings are the vault's own, `mochi.forkedFrom` and the
|
|
944
|
+
* `receive.*` protections included. It is copied byte for byte and not edited
|
|
945
|
+
* afterwards, because an edited copy would differ from the vault for good: every
|
|
946
|
+
* later run would fetch it again and `verify` would report it as wrong.
|
|
947
|
+
*
|
|
948
|
+
* That leaves one thing to check rather than to set. A bare repository keeps no
|
|
949
|
+
* reflogs by default, and the snapshots depend on that: a reflog is the one file
|
|
950
|
+
* git appends to in place, and appending to a file under current/ corrupts every
|
|
951
|
+
* snapshot that has hardlinked it. A vault whose config asks for reflogs is
|
|
952
|
+
* therefore reported, and the mirror is left without them.
|
|
953
|
+
*/
|
|
954
|
+
async function settleMirrorConfigs(current, fetched, quiet) {
|
|
955
|
+
const mirrors = new Set(fetched.filter((f) => f.path.endsWith('/config')).map((f) => f.path.slice(0, -'/config'.length)));
|
|
956
|
+
for (const rel of mirrors) {
|
|
957
|
+
const dir = path.join(current, ...rel.split('/'));
|
|
958
|
+
if (!(0, scan_1.isBareRepo)(dir))
|
|
959
|
+
continue;
|
|
960
|
+
const r = await git(['-C', dir, 'config', '--bool', '--get', 'core.logAllRefUpdates']);
|
|
961
|
+
// Nothing set, which for a bare repository means off: the usual case, and
|
|
962
|
+
// the reason this is a check and not a write.
|
|
963
|
+
if (r.code !== 0 || r.out.trim() !== 'true')
|
|
964
|
+
continue;
|
|
965
|
+
await git(['-C', dir, 'config', 'core.logAllRefUpdates', 'false']);
|
|
966
|
+
if (!quiet) {
|
|
967
|
+
console.error(`Warning: ${rel} asks for reflogs in the vault, which snapshots of this backup cannot allow, so the ` +
|
|
968
|
+
'mirror keeps none. Its config will differ from the vault by that one setting.');
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
/** Directories under dir that hold no files at all, deepest first. */
|
|
973
|
+
function emptyDirs(dir, rel = '', out = []) {
|
|
974
|
+
let entries;
|
|
975
|
+
try {
|
|
976
|
+
entries = fs.readdirSync(path.join(dir, rel), { withFileTypes: true });
|
|
977
|
+
}
|
|
978
|
+
catch {
|
|
979
|
+
return false;
|
|
980
|
+
}
|
|
981
|
+
let empty = true;
|
|
982
|
+
for (const e of entries) {
|
|
983
|
+
const child = rel ? `${rel}/${e.name}` : e.name;
|
|
984
|
+
if (e.isDirectory()) {
|
|
985
|
+
if (!emptyDirs(dir, child, out))
|
|
986
|
+
empty = false;
|
|
987
|
+
else
|
|
988
|
+
out.push(child);
|
|
989
|
+
}
|
|
990
|
+
else {
|
|
991
|
+
empty = false;
|
|
992
|
+
}
|
|
993
|
+
}
|
|
994
|
+
return empty;
|
|
995
|
+
}
|
|
996
|
+
async function syncCmd(inv) {
|
|
997
|
+
const json = (0, output_1.jsonMode)(inv);
|
|
998
|
+
// Two kinds of silence, and they are not the same. --json puts one JSON value
|
|
999
|
+
// on stdout, so the running commentary has to go, but a warning is a
|
|
1000
|
+
// diagnostic on stderr and a caller parsing stdout still wants to see it.
|
|
1001
|
+
// --quiet means nothing on success, warnings included.
|
|
1002
|
+
const quiet = inv.bool('quiet') || json.enabled;
|
|
1003
|
+
const warn = !inv.bool('quiet');
|
|
1004
|
+
const dir = backupDirectory(inv, true);
|
|
1005
|
+
const state = loadState(dir);
|
|
1006
|
+
const target = await targetForBackup(inv, state);
|
|
1007
|
+
const exclude = exclusionsFor(inv, state);
|
|
1008
|
+
const retention = retentionFor(inv, state);
|
|
1009
|
+
const checksum = inv.bool('checksum');
|
|
1010
|
+
const current = path.join(dir, CURRENT);
|
|
1011
|
+
fs.mkdirSync(current, { recursive: true });
|
|
1012
|
+
const lock = takeLock(dir, !warn);
|
|
1013
|
+
const started = new Date().toISOString();
|
|
1014
|
+
const summary = {
|
|
1015
|
+
host: target.host,
|
|
1016
|
+
dir,
|
|
1017
|
+
repos: { total: 0, cloned: 0, fetched: 0, skipped: 0, removed: 0 },
|
|
1018
|
+
files: { total: 0, fetched: 0, removed: 0, bytes: 0 },
|
|
1019
|
+
lfs: 'volume',
|
|
1020
|
+
excluded: exclude,
|
|
1021
|
+
snapshot: null,
|
|
1022
|
+
pruned: [],
|
|
1023
|
+
};
|
|
1024
|
+
try {
|
|
1025
|
+
if (!quiet) {
|
|
1026
|
+
console.error(`Backing up ${target.host} to ${dir}`);
|
|
1027
|
+
// A Fly machine with min_machines_running = 0 is stopped between
|
|
1028
|
+
// requests, so the first one waits for it to boot. Said plainly, because
|
|
1029
|
+
// a client that looks hung is a client someone interrupts.
|
|
1030
|
+
console.error('Asking for the manifest (a machine that sleeps when idle takes a few seconds to wake)');
|
|
1031
|
+
}
|
|
1032
|
+
const manifest = await fetchManifest(target, exclude, checksum);
|
|
1033
|
+
summary.lfs = manifest.lfs;
|
|
1034
|
+
summary.repos.total = manifest.repos.length;
|
|
1035
|
+
summary.files.total = manifest.files.size;
|
|
1036
|
+
state.host = target.host;
|
|
1037
|
+
state.lfs = manifest.lfs;
|
|
1038
|
+
state.excluded = exclude;
|
|
1039
|
+
state.retention = retention;
|
|
1040
|
+
if (manifest.lfs === 'bucket' && warn) {
|
|
1041
|
+
console.error('Warning: this vault keeps its Git LFS objects in a bucket, so they are not in the vault and not in ' +
|
|
1042
|
+
'this backup. Back the bucket up alongside it, e.g. rclone sync; see docs/backup.md.');
|
|
1043
|
+
}
|
|
1044
|
+
// Repositories, by mirror. A repository whose refs digest matches the one
|
|
1045
|
+
// the last run recorded is skipped entirely: no handshake, no request, and
|
|
1046
|
+
// nothing for a sleeping machine to wake up for.
|
|
1047
|
+
for (const entry of manifest.repos) {
|
|
1048
|
+
const known = state.repos[entry.path];
|
|
1049
|
+
const dest = path.join(current, entry.path);
|
|
1050
|
+
if (known && known.refs === entry.refs && (0, scan_1.isBareRepo)(dest)) {
|
|
1051
|
+
summary.repos.skipped++;
|
|
1052
|
+
continue;
|
|
1053
|
+
}
|
|
1054
|
+
const how = await syncRepo(target, current, entry, quiet);
|
|
1055
|
+
await syncHead(target, current, entry);
|
|
1056
|
+
if (how === 'cloned')
|
|
1057
|
+
summary.repos.cloned++;
|
|
1058
|
+
else
|
|
1059
|
+
summary.repos.fetched++;
|
|
1060
|
+
state.repos[entry.path] = { refs: entry.refs };
|
|
1061
|
+
}
|
|
1062
|
+
// A mirror whose repository is gone from the vault goes too. It survives in
|
|
1063
|
+
// whatever snapshots hold it, which is the whole reason retention is worth
|
|
1064
|
+
// configuring.
|
|
1065
|
+
const live = new Set(manifest.repos.map((r) => r.path));
|
|
1066
|
+
for (const known of Object.keys(state.repos)) {
|
|
1067
|
+
if (live.has(known))
|
|
1068
|
+
continue;
|
|
1069
|
+
// The recorded paths come from manifests this command already refused to
|
|
1070
|
+
// accept a climbing path from, so this holds for anything written by a
|
|
1071
|
+
// version that had that check. A state file from before it, or one edited
|
|
1072
|
+
// by hand, is the case worth refusing to delete through.
|
|
1073
|
+
if (!isVaultRelative(known)) {
|
|
1074
|
+
delete state.repos[known];
|
|
1075
|
+
continue;
|
|
1076
|
+
}
|
|
1077
|
+
fs.rmSync(path.join(current, known), { recursive: true, force: true });
|
|
1078
|
+
delete state.repos[known];
|
|
1079
|
+
summary.repos.removed++;
|
|
1080
|
+
if (!quiet)
|
|
1081
|
+
console.error(`Removed ${known}, which the vault no longer holds`);
|
|
1082
|
+
}
|
|
1083
|
+
// Files.
|
|
1084
|
+
const changed = [];
|
|
1085
|
+
for (const f of manifest.files.values()) {
|
|
1086
|
+
const dest = path.join(current, ...f.path.split('/'));
|
|
1087
|
+
if (needsFetch(f, dest, state.files[f.path], checksum))
|
|
1088
|
+
changed.push(f);
|
|
1089
|
+
}
|
|
1090
|
+
for (const chunk of chunkPaths(changed)) {
|
|
1091
|
+
const r = await fetchChunk(target, current, chunk, state, checksum);
|
|
1092
|
+
summary.files.fetched += chunk.length - r.missing.length;
|
|
1093
|
+
summary.files.bytes += r.bytes;
|
|
1094
|
+
for (const gone of r.missing) {
|
|
1095
|
+
delete state.files[gone];
|
|
1096
|
+
if (!quiet)
|
|
1097
|
+
console.error(`${gone} vanished from the vault while this run was reading it`);
|
|
1098
|
+
}
|
|
1099
|
+
}
|
|
1100
|
+
// A mirror's config is one of the files the manifest names, so the copy just
|
|
1101
|
+
// written is the vault's, which says nothing about reflogs. A bare
|
|
1102
|
+
// repository keeps none by default, but the snapshots depend on that being
|
|
1103
|
+
// true rather than merely usual, so it is stated on the mirror itself.
|
|
1104
|
+
await settleMirrorConfigs(current, changed, !warn);
|
|
1105
|
+
// The manifest is authoritative for deletions: a file in current/ that it
|
|
1106
|
+
// does not list is gone from the vault, or is a category this run excludes.
|
|
1107
|
+
// Either way it does not belong in a copy of what the vault holds now.
|
|
1108
|
+
const mirrors = [...live].map((p) => p.split('/').join(path.sep));
|
|
1109
|
+
const insideMirror = (rel) => mirrors.some((m) => rel === m || rel.startsWith(m + path.sep));
|
|
1110
|
+
for (const rel of walkFiles(current)) {
|
|
1111
|
+
const asPath = rel.split('/').join(path.sep);
|
|
1112
|
+
if (insideMirror(asPath))
|
|
1113
|
+
continue;
|
|
1114
|
+
if (manifest.files.has(rel))
|
|
1115
|
+
continue;
|
|
1116
|
+
fs.rmSync(path.join(current, asPath), { force: true });
|
|
1117
|
+
delete state.files[rel];
|
|
1118
|
+
summary.files.removed++;
|
|
1119
|
+
}
|
|
1120
|
+
// Records for files the manifest no longer lists, whether or not a copy was
|
|
1121
|
+
// still on disk to remove.
|
|
1122
|
+
for (const rel of Object.keys(state.files)) {
|
|
1123
|
+
if (!manifest.files.has(rel))
|
|
1124
|
+
delete state.files[rel];
|
|
1125
|
+
}
|
|
1126
|
+
const empties = [];
|
|
1127
|
+
emptyDirs(current, '', empties);
|
|
1128
|
+
for (const rel of empties.sort((a, b) => b.length - a.length)) {
|
|
1129
|
+
if (insideMirror(rel.split('/').join(path.sep)))
|
|
1130
|
+
continue;
|
|
1131
|
+
try {
|
|
1132
|
+
fs.rmdirSync(path.join(current, ...rel.split('/')));
|
|
1133
|
+
}
|
|
1134
|
+
catch {
|
|
1135
|
+
// not empty after all, or already gone
|
|
1136
|
+
}
|
|
1137
|
+
}
|
|
1138
|
+
// One cheap mitigation for the fact that a backup is a walk of a live tree:
|
|
1139
|
+
// ask again, and re-fetch anything whose size or timestamp moved while this
|
|
1140
|
+
// run was working. It closes the window for everything except a file
|
|
1141
|
+
// written twice during the same run.
|
|
1142
|
+
const again = await fetchManifest(target, exclude, checksum);
|
|
1143
|
+
const moved = [];
|
|
1144
|
+
for (const f of again.files.values()) {
|
|
1145
|
+
const dest = path.join(current, ...f.path.split('/'));
|
|
1146
|
+
if (needsFetch(f, dest, state.files[f.path], checksum))
|
|
1147
|
+
moved.push(f);
|
|
1148
|
+
}
|
|
1149
|
+
if (moved.length) {
|
|
1150
|
+
if (!quiet)
|
|
1151
|
+
console.error(`${moved.length} file(s) changed during the run; fetching them again`);
|
|
1152
|
+
for (const chunk of chunkPaths(moved)) {
|
|
1153
|
+
const r = await fetchChunk(target, current, chunk, state, checksum);
|
|
1154
|
+
summary.files.fetched += chunk.length - r.missing.length;
|
|
1155
|
+
summary.files.bytes += r.bytes;
|
|
1156
|
+
for (const gone of r.missing)
|
|
1157
|
+
delete state.files[gone];
|
|
1158
|
+
}
|
|
1159
|
+
await settleMirrorConfigs(current, moved, !warn);
|
|
1160
|
+
}
|
|
1161
|
+
state.runs.push({
|
|
1162
|
+
started,
|
|
1163
|
+
finished: new Date().toISOString(),
|
|
1164
|
+
files: summary.files.fetched,
|
|
1165
|
+
bytes: summary.files.bytes,
|
|
1166
|
+
repos: summary.repos.cloned + summary.repos.fetched,
|
|
1167
|
+
deleted: summary.files.removed + summary.repos.removed,
|
|
1168
|
+
});
|
|
1169
|
+
state.runs = state.runs.slice(-KEEP_RUNS);
|
|
1170
|
+
saveState(dir, state);
|
|
1171
|
+
if (inv.bool('snapshot')) {
|
|
1172
|
+
summary.snapshot = takeSnapshot(dir, quiet).name;
|
|
1173
|
+
summary.pruned = pruneSnapshots(dir, retention, quiet);
|
|
1174
|
+
}
|
|
1175
|
+
}
|
|
1176
|
+
catch (e) {
|
|
1177
|
+
// A failed run is recorded too. "It has been failing since Tuesday" is the
|
|
1178
|
+
// thing a backup most needs to be able to say.
|
|
1179
|
+
state.runs.push({
|
|
1180
|
+
started,
|
|
1181
|
+
finished: new Date().toISOString(),
|
|
1182
|
+
files: summary.files.fetched,
|
|
1183
|
+
bytes: summary.files.bytes,
|
|
1184
|
+
repos: summary.repos.cloned + summary.repos.fetched,
|
|
1185
|
+
deleted: summary.files.removed + summary.repos.removed,
|
|
1186
|
+
error: e instanceof Error ? e.message : String(e),
|
|
1187
|
+
});
|
|
1188
|
+
state.runs = state.runs.slice(-KEEP_RUNS);
|
|
1189
|
+
try {
|
|
1190
|
+
saveState(dir, state);
|
|
1191
|
+
}
|
|
1192
|
+
catch {
|
|
1193
|
+
// Reporting the original failure matters more than recording it.
|
|
1194
|
+
}
|
|
1195
|
+
throw e;
|
|
1196
|
+
}
|
|
1197
|
+
finally {
|
|
1198
|
+
lock.release();
|
|
1199
|
+
}
|
|
1200
|
+
if (json.enabled) {
|
|
1201
|
+
(0, output_1.printJson)((0, output_1.pickObject)(summary, json.fields));
|
|
1202
|
+
return;
|
|
1203
|
+
}
|
|
1204
|
+
if (quiet)
|
|
1205
|
+
return;
|
|
1206
|
+
const r = summary.repos;
|
|
1207
|
+
console.log(`${r.total} repositories: ${r.cloned} cloned, ${r.fetched} fetched, ${r.skipped} unchanged` +
|
|
1208
|
+
(r.removed ? `, ${r.removed} removed` : ''));
|
|
1209
|
+
console.log(`${summary.files.total} files: ${summary.files.fetched} fetched (${human(summary.files.bytes)})` +
|
|
1210
|
+
(summary.files.removed ? `, ${summary.files.removed} removed` : ''));
|
|
1211
|
+
if (summary.snapshot)
|
|
1212
|
+
console.log(`Snapshot ${summary.snapshot}`);
|
|
1213
|
+
console.log('');
|
|
1214
|
+
console.log(`Serve this backup to look at it, or to stand the vault back up:`);
|
|
1215
|
+
console.log(` mochi serve ${path.join(dir, CURRENT)}`);
|
|
1216
|
+
}
|
|
1217
|
+
// ---- list, prune, verify ----
|
|
1218
|
+
function listCmd(inv) {
|
|
1219
|
+
const { dir, state } = existingBackup(inv);
|
|
1220
|
+
const snapshots = listSnapshots(dir).map((s) => ({
|
|
1221
|
+
name: s.name,
|
|
1222
|
+
at: s.at.toISOString(),
|
|
1223
|
+
bytes: apparentSize(path.join(dir, SNAPSHOTS, s.name)),
|
|
1224
|
+
}));
|
|
1225
|
+
const json = (0, output_1.jsonMode)(inv);
|
|
1226
|
+
if (json.enabled) {
|
|
1227
|
+
(0, output_1.printJson)((0, output_1.pickObject)({
|
|
1228
|
+
host: state.host,
|
|
1229
|
+
dir,
|
|
1230
|
+
excluded: state.excluded,
|
|
1231
|
+
retention: state.retention,
|
|
1232
|
+
current: { bytes: apparentSize(path.join(dir, CURRENT)) },
|
|
1233
|
+
snapshots,
|
|
1234
|
+
runs: state.runs,
|
|
1235
|
+
}, json.fields));
|
|
1236
|
+
return;
|
|
1237
|
+
}
|
|
1238
|
+
console.log(`${dir}`);
|
|
1239
|
+
console.log(` vault ${state.host || '(unknown)'}`);
|
|
1240
|
+
console.log(` current ${human(apparentSize(path.join(dir, CURRENT)))} apparent`);
|
|
1241
|
+
console.log(` excluded ${state.excluded.length ? state.excluded.join(', ') : 'nothing'}`);
|
|
1242
|
+
console.log(` retention ${state.retention.daily} daily, ${state.retention.weekly} weekly, ${state.retention.monthly} monthly`);
|
|
1243
|
+
const last = state.runs[state.runs.length - 1];
|
|
1244
|
+
if (last) {
|
|
1245
|
+
console.log(` last run ${last.finished}${last.error ? ` FAILED: ${last.error}` : ` (${human(last.bytes)} in ${last.files} files)`}`);
|
|
1246
|
+
}
|
|
1247
|
+
console.log('');
|
|
1248
|
+
if (snapshots.length === 0) {
|
|
1249
|
+
console.log('No snapshots. `mochi backup <dir> --snapshot` takes one after a sync.');
|
|
1250
|
+
return;
|
|
1251
|
+
}
|
|
1252
|
+
// Apparent size rather than disk use: a snapshot is hardlinked, so what it
|
|
1253
|
+
// costs on disk is close to nothing and what it holds is this.
|
|
1254
|
+
(0, output_1.printTable)([['SNAPSHOT', 'TAKEN', 'APPARENT'], ...snapshots.map((s) => [s.name, s.at, human(s.bytes)])]);
|
|
1255
|
+
}
|
|
1256
|
+
function pruneCmd(inv) {
|
|
1257
|
+
const { dir, state } = existingBackup(inv);
|
|
1258
|
+
const retention = retentionFor(inv, state);
|
|
1259
|
+
const json = (0, output_1.jsonMode)(inv);
|
|
1260
|
+
const lock = takeLock(dir, inv.bool('quiet') || json.enabled);
|
|
1261
|
+
let dropped;
|
|
1262
|
+
try {
|
|
1263
|
+
dropped = pruneSnapshots(dir, retention, true);
|
|
1264
|
+
state.retention = retention;
|
|
1265
|
+
saveState(dir, state);
|
|
1266
|
+
}
|
|
1267
|
+
finally {
|
|
1268
|
+
lock.release();
|
|
1269
|
+
}
|
|
1270
|
+
const kept = listSnapshots(dir).map((s) => s.name);
|
|
1271
|
+
if (json.enabled) {
|
|
1272
|
+
(0, output_1.printJson)((0, output_1.pickObject)({ dir, pruned: dropped, kept, retention }, json.fields));
|
|
1273
|
+
return;
|
|
1274
|
+
}
|
|
1275
|
+
if (inv.bool('quiet'))
|
|
1276
|
+
return;
|
|
1277
|
+
console.log(dropped.length ? `Pruned ${dropped.length}: ${dropped.join(', ')}` : 'Nothing to prune under this retention.');
|
|
1278
|
+
console.log(`${kept.length} snapshot${kept.length === 1 ? '' : 's'} kept.`);
|
|
1279
|
+
}
|
|
1280
|
+
async function verifyCmd(inv) {
|
|
1281
|
+
const { dir, state } = existingBackup(inv);
|
|
1282
|
+
const json = (0, output_1.jsonMode)(inv);
|
|
1283
|
+
const quiet = inv.bool('quiet') || json.enabled;
|
|
1284
|
+
const current = path.join(dir, CURRENT);
|
|
1285
|
+
const target = await targetForBackup(inv, state);
|
|
1286
|
+
const problems = [];
|
|
1287
|
+
// Each mirror is a real repository, so git can be asked the question rather
|
|
1288
|
+
// than reimplemented. --connectivity-only skips re-hashing every blob, which
|
|
1289
|
+
// turns an hour into a minute and still catches the failure that matters: an
|
|
1290
|
+
// object the history refers to and the backup does not have.
|
|
1291
|
+
const repos = Object.keys(state.repos).sort();
|
|
1292
|
+
for (const rel of repos) {
|
|
1293
|
+
const repoDir = path.join(current, rel);
|
|
1294
|
+
if (!(0, scan_1.isBareRepo)(repoDir)) {
|
|
1295
|
+
problems.push({ path: rel, problem: 'not a repository' });
|
|
1296
|
+
continue;
|
|
1297
|
+
}
|
|
1298
|
+
if (!quiet)
|
|
1299
|
+
console.error(`Checking ${rel}`);
|
|
1300
|
+
const r = await git(['-C', repoDir, 'fsck', '--connectivity-only', '--no-progress']);
|
|
1301
|
+
if (r.code !== 0)
|
|
1302
|
+
problems.push({ path: rel, problem: `git fsck: ${r.out.trim().split('\n')[0]}` });
|
|
1303
|
+
}
|
|
1304
|
+
// The files, against hashes the vault computes now. This is the part a
|
|
1305
|
+
// size-and-mtime sync cannot check on its own.
|
|
1306
|
+
if (!quiet)
|
|
1307
|
+
console.error('Asking the vault for hashes');
|
|
1308
|
+
const manifest = await fetchManifest(target, state.excluded, true);
|
|
1309
|
+
for (const f of manifest.files.values()) {
|
|
1310
|
+
const dest = path.join(current, ...f.path.split('/'));
|
|
1311
|
+
let st;
|
|
1312
|
+
try {
|
|
1313
|
+
st = fs.statSync(dest);
|
|
1314
|
+
}
|
|
1315
|
+
catch {
|
|
1316
|
+
problems.push({ path: f.path, problem: 'missing from the backup' });
|
|
1317
|
+
continue;
|
|
1318
|
+
}
|
|
1319
|
+
if (st.size !== f.size) {
|
|
1320
|
+
problems.push({ path: f.path, problem: `size ${st.size}, the vault has ${f.size}` });
|
|
1321
|
+
continue;
|
|
1322
|
+
}
|
|
1323
|
+
if (f.sha256 && sha256Of(dest) !== f.sha256) {
|
|
1324
|
+
problems.push({ path: f.path, problem: 'contents differ from the vault' });
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
const mirrors = repos.map((p) => p.split('/').join(path.sep));
|
|
1328
|
+
const insideMirror = (rel) => mirrors.some((m) => rel === m || rel.startsWith(m + path.sep));
|
|
1329
|
+
for (const rel of walkFiles(current)) {
|
|
1330
|
+
const asPath = rel.split('/').join(path.sep);
|
|
1331
|
+
if (insideMirror(asPath))
|
|
1332
|
+
continue;
|
|
1333
|
+
if (!manifest.files.has(rel))
|
|
1334
|
+
problems.push({ path: rel, problem: 'in the backup, not in the vault' });
|
|
1335
|
+
}
|
|
1336
|
+
for (const entry of manifest.repos) {
|
|
1337
|
+
if (!state.repos[entry.path])
|
|
1338
|
+
problems.push({ path: entry.path, problem: 'in the vault, not in the backup' });
|
|
1339
|
+
}
|
|
1340
|
+
// The hardlink invariant the snapshots rest on. A file with more links than
|
|
1341
|
+
// current/ plus the snapshots can account for is shared with something
|
|
1342
|
+
// outside this backup, which means a snapshot could be changed from outside.
|
|
1343
|
+
const maxLinks = listSnapshots(dir).length + 1;
|
|
1344
|
+
for (const rel of walkFiles(current)) {
|
|
1345
|
+
try {
|
|
1346
|
+
const st = fs.lstatSync(path.join(current, ...rel.split('/')));
|
|
1347
|
+
if (st.nlink > maxLinks)
|
|
1348
|
+
problems.push({ path: rel, problem: `${st.nlink} hard links, more than ${maxLinks}` });
|
|
1349
|
+
}
|
|
1350
|
+
catch {
|
|
1351
|
+
// walked a moment ago and gone now; the file checks above already say so
|
|
1352
|
+
}
|
|
1353
|
+
}
|
|
1354
|
+
if (json.enabled) {
|
|
1355
|
+
(0, output_1.printJson)((0, output_1.pickObject)({ dir, host: target.host, repos: repos.length, files: manifest.files.size, problems }, json.fields));
|
|
1356
|
+
}
|
|
1357
|
+
else if (problems.length === 0) {
|
|
1358
|
+
if (!inv.bool('quiet')) {
|
|
1359
|
+
console.log(`${repos.length} mirrors and ${manifest.files.size} files check out against ${target.host}.`);
|
|
1360
|
+
}
|
|
1361
|
+
}
|
|
1362
|
+
else {
|
|
1363
|
+
for (const p of problems)
|
|
1364
|
+
console.log(`${p.path}: ${p.problem}`);
|
|
1365
|
+
console.log('');
|
|
1366
|
+
console.log(`${problems.length} problem${problems.length === 1 ? '' : 's'}.`);
|
|
1367
|
+
}
|
|
1368
|
+
if (problems.length) {
|
|
1369
|
+
throw new exit_1.CliError(`${problems.length} problem${problems.length === 1 ? '' : 's'} in ${dir}.`, exit_1.EXIT_FAIL);
|
|
1370
|
+
}
|
|
1371
|
+
}
|
|
1372
|
+
// ---- what a machine knows about its backups ----
|
|
1373
|
+
/**
|
|
1374
|
+
* Where this machine's backup directories are remembered, so that `mochi
|
|
1375
|
+
* deploy fly show` can say whether the app it is describing has one. It is a
|
|
1376
|
+
* convenience and nothing depends on it: the backup itself is entirely
|
|
1377
|
+
* described by its own backup.json.
|
|
1378
|
+
*/
|
|
1379
|
+
function backupsIndexPath() {
|
|
1380
|
+
const base = process.env.XDG_CONFIG_HOME ?? path.join(os.homedir(), '.config');
|
|
1381
|
+
return path.join(base, 'mochi', 'backups.json');
|
|
1382
|
+
}
|
|
1383
|
+
function knownBackups() {
|
|
1384
|
+
try {
|
|
1385
|
+
const parsed = JSON.parse(fs.readFileSync(backupsIndexPath(), 'utf8'));
|
|
1386
|
+
if (!Array.isArray(parsed.backups))
|
|
1387
|
+
return [];
|
|
1388
|
+
return parsed.backups.filter((b) => typeof b === 'object' && b !== null && typeof b.dir === 'string');
|
|
1389
|
+
}
|
|
1390
|
+
catch {
|
|
1391
|
+
return [];
|
|
1392
|
+
}
|
|
1393
|
+
}
|
|
1394
|
+
function rememberBackup(dir, host) {
|
|
1395
|
+
try {
|
|
1396
|
+
const others = knownBackups().filter((b) => b.dir !== dir && fs.existsSync(path.join(b.dir, STATE_FILE)));
|
|
1397
|
+
const file = backupsIndexPath();
|
|
1398
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
1399
|
+
(0, atomic_1.writeFileAtomic)(file, JSON.stringify({ backups: [...others, { dir, host }] }, null, 2) + '\n');
|
|
1400
|
+
}
|
|
1401
|
+
catch {
|
|
1402
|
+
// A machine with no writable configuration directory still has a working
|
|
1403
|
+
// backup; only `deploy fly show` is any the wiser.
|
|
1404
|
+
}
|
|
1405
|
+
}
|
|
1406
|
+
/**
|
|
1407
|
+
* What to say about a vault's backup on this machine, if it has one. Several
|
|
1408
|
+
* hosts may name one vault, since a Fly app with a domain of its own answers on
|
|
1409
|
+
* both, and the backup records whichever name it was given.
|
|
1410
|
+
*/
|
|
1411
|
+
function backupLineFor(hosts) {
|
|
1412
|
+
const wanted = new Set((Array.isArray(hosts) ? hosts : [hosts]).map((h) => h.replace(/\/+$/, '')));
|
|
1413
|
+
for (const b of knownBackups()) {
|
|
1414
|
+
if (!wanted.has(b.host.replace(/\/+$/, '')))
|
|
1415
|
+
continue;
|
|
1416
|
+
const state = loadState(b.dir);
|
|
1417
|
+
const last = state.runs[state.runs.length - 1];
|
|
1418
|
+
if (!last)
|
|
1419
|
+
return `${b.dir} (never run)`;
|
|
1420
|
+
const when = last.finished.slice(0, 10);
|
|
1421
|
+
return last.error ? `${b.dir} (last run ${when} FAILED)` : `${b.dir} (last run ${when})`;
|
|
1422
|
+
}
|
|
1423
|
+
return null;
|
|
1424
|
+
}
|
|
1425
|
+
// ---- the commands ----
|
|
1426
|
+
const COMMON = [output_1.JSON_OPTION, QUIET_OPTION, ...target_1.TARGET_OPTIONS];
|
|
1427
|
+
exports.backupCommands = [
|
|
1428
|
+
{
|
|
1429
|
+
path: ['backup'],
|
|
1430
|
+
summary: 'Copy a whole vault to a directory on this machine, incrementally',
|
|
1431
|
+
description: `A vault is a directory, so a backup of one is a directory too, and this makes it
|
|
1432
|
+
over HTTP: it needs no shell on the server, no flyctl, and no rsync at the far
|
|
1433
|
+
end, so it works the same against a Fly app, a VPS, a Docker deployment, and
|
|
1434
|
+
127.0.0.1:3000.
|
|
1435
|
+
|
|
1436
|
+
<dir>/current a servable vault. Restoring is: mochi serve <dir>/current
|
|
1437
|
+
<dir>/snapshots hardlinked copies, each one also a servable vault
|
|
1438
|
+
<dir>/backup.json which vault, what is left out, and how each run went
|
|
1439
|
+
|
|
1440
|
+
Repositories come across as mirrors, so a second run moves only the objects it
|
|
1441
|
+
does not have and skips a repository nothing was pushed to. Everything beside
|
|
1442
|
+
them - issues, pull requests, releases, sites, run history, LFS objects on the
|
|
1443
|
+
volume, and the vault's state files - is compared by size and modification time
|
|
1444
|
+
and fetched only where it differs.
|
|
1445
|
+
|
|
1446
|
+
The token needs to belong to a site admin, because the copy includes
|
|
1447
|
+
vault.json. The vault URL, the exclusions, and the retention policy are recorded
|
|
1448
|
+
in backup.json, so a cron entry is this command and a directory.
|
|
1449
|
+
|
|
1450
|
+
There is no vault-wide point-in-time image: the server holds no lock a client
|
|
1451
|
+
could take, so a run is a walk of a live tree and can catch a mixed vintage.
|
|
1452
|
+
Every individual file in a backup is one that really existed. See docs/backup.md.
|
|
1453
|
+
|
|
1454
|
+
Related: mochi backup list, verify, prune.`,
|
|
1455
|
+
args: [{ name: 'dir', required: true }],
|
|
1456
|
+
options: [
|
|
1457
|
+
{ name: 'snapshot', type: 'boolean', summary: 'Take a snapshot after a successful sync, then prune' },
|
|
1458
|
+
...RETENTION_OPTIONS,
|
|
1459
|
+
...EXCLUDE_OPTIONS,
|
|
1460
|
+
{ name: 'checksum', type: 'boolean', summary: 'Compare hashes rather than size and modification time' },
|
|
1461
|
+
...COMMON,
|
|
1462
|
+
],
|
|
1463
|
+
async run(inv) {
|
|
1464
|
+
await syncCmd(inv);
|
|
1465
|
+
const dir = path.resolve(inv.args[0]);
|
|
1466
|
+
rememberBackup(dir, loadState(dir).host);
|
|
1467
|
+
},
|
|
1468
|
+
},
|
|
1469
|
+
{
|
|
1470
|
+
path: ['backup', 'list'],
|
|
1471
|
+
summary: "Show a backup's snapshots, and how the last run went",
|
|
1472
|
+
args: [{ name: 'dir', required: true }],
|
|
1473
|
+
options: [output_1.JSON_OPTION],
|
|
1474
|
+
run: listCmd,
|
|
1475
|
+
},
|
|
1476
|
+
{
|
|
1477
|
+
path: ['backup', 'verify'],
|
|
1478
|
+
summary: 'Check a backup against the vault, and its mirrors against git',
|
|
1479
|
+
description: `Runs git fsck --connectivity-only over every mirror, asks the vault for hashes,
|
|
1480
|
+
and reports anything missing, extra, or different. Exits non-zero when there is
|
|
1481
|
+
something to report, so it can be run from cron.`,
|
|
1482
|
+
args: [{ name: 'dir', required: true }],
|
|
1483
|
+
options: [...COMMON],
|
|
1484
|
+
run: verifyCmd,
|
|
1485
|
+
},
|
|
1486
|
+
{
|
|
1487
|
+
path: ['backup', 'prune'],
|
|
1488
|
+
summary: 'Apply the retention policy to the snapshots, without syncing',
|
|
1489
|
+
description: `Grandfather-father-son: the newest snapshot of each of the last N days, weeks,
|
|
1490
|
+
and months is kept and the rest are removed, evaluated in UTC. The newest
|
|
1491
|
+
snapshot is always kept.
|
|
1492
|
+
|
|
1493
|
+
A snapshot pins the packfiles that were current when it was taken, so a repack
|
|
1494
|
+
in a busy repository leaves the old pack on disk until the last snapshot
|
|
1495
|
+
referring to it is pruned. This is what reclaims that space.`,
|
|
1496
|
+
args: [{ name: 'dir', required: true }],
|
|
1497
|
+
options: [...RETENTION_OPTIONS, output_1.JSON_OPTION, QUIET_OPTION],
|
|
1498
|
+
run: pruneCmd,
|
|
1499
|
+
},
|
|
1500
|
+
];
|