agmsg-cloud 0.0.1 → 0.1.0-rc.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -2
- package/dist/src/api.js +517 -0
- package/dist/src/authenticated-digest.js +234 -0
- package/dist/src/browser.js +241 -0
- package/dist/src/ceremony.js +181 -0
- package/dist/src/commands/approve.js +392 -0
- package/dist/src/commands/connect.js +273 -0
- package/dist/src/commands/fetch.js +249 -0
- package/dist/src/commands/login.js +334 -0
- package/dist/src/commands/logout.js +74 -0
- package/dist/src/commands/pull.js +80 -0
- package/dist/src/commands/request.js +371 -0
- package/dist/src/commands/sync.js +138 -0
- package/dist/src/commands/vault.js +478 -0
- package/dist/src/commands/watch.js +47 -0
- package/dist/src/config.js +34 -0
- package/dist/src/credentials.js +374 -0
- package/dist/src/device-slot.js +148 -0
- package/dist/src/filelock.js +167 -0
- package/dist/src/index.js +242 -0
- package/dist/src/ledger.js +296 -0
- package/dist/src/machine-name.js +90 -0
- package/dist/src/oss-env.js +49 -0
- package/dist/src/oss.js +289 -0
- package/dist/src/paths.js +8 -0
- package/dist/src/pending.js +330 -0
- package/dist/src/pick-request.js +56 -0
- package/dist/src/preflight.js +257 -0
- package/dist/src/recovery-key.js +386 -0
- package/dist/src/sas.js +18 -0
- package/dist/src/secure-store.js +176 -0
- package/dist/src/shell-arg.js +18 -0
- package/dist/src/slot-advice.js +74 -0
- package/dist/src/vault-container.js +115 -0
- package/dist/src/vault-crypto.js +190 -0
- package/dist/src/vault-protocol.js +358 -0
- package/dist/src/version.js +57 -0
- package/node_modules/@agmsg-cloud/sas-core/dist/src/bech32.d.ts +17 -0
- package/node_modules/@agmsg-cloud/sas-core/dist/src/bech32.js +103 -0
- package/node_modules/@agmsg-cloud/sas-core/dist/src/index.d.ts +17 -0
- package/node_modules/@agmsg-cloud/sas-core/dist/src/index.js +147 -0
- package/node_modules/@agmsg-cloud/sas-core/package.json +30 -0
- package/package.json +50 -7
- package/bin/agmsg-cloud.js +0 -4
|
@@ -0,0 +1,478 @@
|
|
|
1
|
+
import { mkdtempSync, readFileSync, rmSync } from 'node:fs';
|
|
2
|
+
import { tmpdir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { CourierClient } from '../api.js';
|
|
5
|
+
import { connectedTeams, keyHandoff, remoteBinding, unlockAuthenticatedBundle } from '../oss.js';
|
|
6
|
+
import { NEEDS, ensurePreflight, preflight } from '../preflight.js';
|
|
7
|
+
import { generateRecoveryKey, normalizeRecoveryKey, promptRecoveryKey, showRecoveryKey, } from '../recovery-key.js';
|
|
8
|
+
import { appendVaultVersionWithVdk, createVault, openVaultWithVdk, readAccountVault, vdkFromRecoveryKey, } from '../vault-protocol.js';
|
|
9
|
+
import { openDeviceSlot, saveDeviceSlot } from '../device-slot.js';
|
|
10
|
+
import { adviseOnSlot, renderSlotAdvice } from '../slot-advice.js';
|
|
11
|
+
import { emptyContainer, parseContainer, selectTeam, serializeContainer, upsertTeam, } from '../vault-container.js';
|
|
12
|
+
// The two recovery commands: resolve the team, obtain the bundle from the OSS
|
|
13
|
+
// side, get the recovery key from the terminal, and hand off to
|
|
14
|
+
// vault-protocol.ts for everything that talks to the server.
|
|
15
|
+
//
|
|
16
|
+
// The file, the functions and the storage are still named "vault" — that is
|
|
17
|
+
// what the thing IS, and the spec calls it that. Only the two commands were
|
|
18
|
+
// renamed, because `vault put` read as filing a copy away rather than as
|
|
19
|
+
// preparing to recover. Every heading below names the CURRENT command.
|
|
20
|
+
// Where this machine's key slot lives for a vault: the deployment, the account,
|
|
21
|
+
// the vault, and the generation. All four, because a slot must not be carried to
|
|
22
|
+
// another deployment, another account, another vault, or forward across a
|
|
23
|
+
// re-issuance.
|
|
24
|
+
//
|
|
25
|
+
// The generation is read from the version the server just returned, with no
|
|
26
|
+
// default anywhere on this side. A local fallback of 1 would let a server that
|
|
27
|
+
// omits the field name generation 1 by omission, and — worse — a slot saved
|
|
28
|
+
// under a made-up 1 would be indistinguishable from a legitimate generation-1
|
|
29
|
+
// slot once a re-issuance produced one.
|
|
30
|
+
function slotAddress(identity, vaultId, generation) {
|
|
31
|
+
return {
|
|
32
|
+
vaultServiceId: identity.vaultServiceId,
|
|
33
|
+
accountId: identity.accountId,
|
|
34
|
+
vaultId,
|
|
35
|
+
recoveryGeneration: generation,
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Get the vault's data key without asking for the recovery key if this machine
|
|
40
|
+
* can avoid it.
|
|
41
|
+
*
|
|
42
|
+
* This is the whole point of the device slot, and the owner's requirement: the
|
|
43
|
+
* master key should not come out unless a machine is lost. A machine that has
|
|
44
|
+
* backed up before opens its own slot and never prompts. Every other machine —
|
|
45
|
+
* and every machine on a platform with no secure store — falls back to the key,
|
|
46
|
+
* which is what every machine did before this existed.
|
|
47
|
+
*
|
|
48
|
+
* Nothing here is a dead end. Each way of failing to open a slot continues to
|
|
49
|
+
* the same place; what differs is what the person is told, and that judgement
|
|
50
|
+
* lives in slot-advice.ts rather than here.
|
|
51
|
+
*/
|
|
52
|
+
async function vdkForVault(identity, version, team) {
|
|
53
|
+
const address = slotAddress(identity, version.vault_id, version.recovery_generation);
|
|
54
|
+
const opened = await openDeviceSlot(address);
|
|
55
|
+
if (opened.ok) {
|
|
56
|
+
// Proving the slot's key is THIS vault's key is not done here — the callers
|
|
57
|
+
// both open or append with it, and both of those fail closed on a key that
|
|
58
|
+
// does not fit. A check here would be a third place that has to agree.
|
|
59
|
+
return { vdk: opened.vdk, usedRecoveryKey: false };
|
|
60
|
+
}
|
|
61
|
+
const advice = renderSlotAdvice(adviseOnSlot(opened, { team }));
|
|
62
|
+
if (advice)
|
|
63
|
+
process.stdout.write(advice);
|
|
64
|
+
const recoveryKey = normalizeRecoveryKey(await promptRecoveryKey('Recovery key: '));
|
|
65
|
+
return { vdk: await vdkFromRecoveryKey(version, recoveryKey), usedRecoveryKey: true };
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Keep this machine's slot, so the next run does not ask for the key.
|
|
69
|
+
*
|
|
70
|
+
* Called AFTER the vault write, not before. The design's crash order puts the
|
|
71
|
+
* device-wrap durable write first (recovery-vault-v1.md:533), and that order
|
|
72
|
+
* exists for the end state where the recovery key is shown LAST and can be
|
|
73
|
+
* re-derived from this slot. Today the key is shown before anything is stored,
|
|
74
|
+
* so no vault can exist that nobody can open, and writing the slot first would
|
|
75
|
+
* instead leave a slot addressed to a vault that was never created — one per
|
|
76
|
+
* failed attempt, since the address carries the vault id. WHEN THE KEY MOVES TO
|
|
77
|
+
* BEING SHOWN LAST, THIS ORDER MUST FLIP.
|
|
78
|
+
*
|
|
79
|
+
* A failure here never fails the command. The backup landed; the slot is what
|
|
80
|
+
* makes the next one keyless, and saying "the backup failed" because a keychain
|
|
81
|
+
* refused would be false.
|
|
82
|
+
*/
|
|
83
|
+
async function keepSlot(address, vdk) {
|
|
84
|
+
const saved = await saveDeviceSlot(address, vdk);
|
|
85
|
+
if (saved.ok) {
|
|
86
|
+
process.stdout.write('this machine now holds a key slot for this vault; later backups will not ask ' +
|
|
87
|
+
'for the recovery key\n');
|
|
88
|
+
return;
|
|
89
|
+
}
|
|
90
|
+
// Named, not swallowed. Someone who was told the key would stop being needed,
|
|
91
|
+
// and is then asked for it next time, is owed the reason — and there is no
|
|
92
|
+
// plaintext fallback here that would make it "work" instead.
|
|
93
|
+
process.stdout.write(`note: this machine could not keep a key slot (${saved.reason}), so the recovery ` +
|
|
94
|
+
'key will be needed again next time.\n');
|
|
95
|
+
}
|
|
96
|
+
// Whether a team's material is the same material, judged on the key epochs the
|
|
97
|
+
// bundle declares rather than on its bytes.
|
|
98
|
+
//
|
|
99
|
+
// The ciphertext differs every run (nonce), and whether the bundle itself is
|
|
100
|
+
// byte-stable is not something this side has established — so neither can carry
|
|
101
|
+
// the comparison. `key_ids` are the epochs, they are already stored per entry,
|
|
102
|
+
// and they are exactly what "the key situation has not changed" means. Compared
|
|
103
|
+
// as a set: the order the exporter lists them in is not part of the fact.
|
|
104
|
+
/**
|
|
105
|
+
* The key that decides whether two records name the same team.
|
|
106
|
+
*
|
|
107
|
+
* One function, so both sides of every comparison are built the same way — a
|
|
108
|
+
* separator chosen twice is a separator that eventually differs. Framed rather
|
|
109
|
+
* than joined: `('a|b','c')` and `('a','b|c')` are different pairs and must not
|
|
110
|
+
* produce one key, which is the collision the request digest had to be fixed
|
|
111
|
+
* for. These two values are uuids today, so no pair can collide — the framing
|
|
112
|
+
* is here so that stays true if either ever stops being one.
|
|
113
|
+
*
|
|
114
|
+
* Written after the first version put a literal separator inside a template and
|
|
115
|
+
* a NUL character ended up in the source, which made the whole file binary to
|
|
116
|
+
* `grep`. Both sides still agreed, so nothing failed; the file simply stopped
|
|
117
|
+
* answering searches.
|
|
118
|
+
*/
|
|
119
|
+
function vaultTeamKey(serverInstanceId, teamId) {
|
|
120
|
+
return JSON.stringify([serverInstanceId, teamId]);
|
|
121
|
+
}
|
|
122
|
+
function sameEpochs(a, b) {
|
|
123
|
+
if (a.length !== b.length)
|
|
124
|
+
return false;
|
|
125
|
+
const left = [...a].sort();
|
|
126
|
+
const right = [...b].sort();
|
|
127
|
+
return left.every((v, i) => v === right[i]);
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* `recovery setup` — put the keys of every team THIS MACHINE reports into the
|
|
131
|
+
* account's vault.
|
|
132
|
+
*
|
|
133
|
+
* NO ARGUMENT IS THE ORDINARY FORM. The vault holds one recovery key for the
|
|
134
|
+
* whole account, so "set up recovery" is an account-level act; naming a team
|
|
135
|
+
* made it something to repeat every time a team was added, which is the folding
|
|
136
|
+
* this design exists to do. A team may still be named, for adding one on its
|
|
137
|
+
* own.
|
|
138
|
+
*
|
|
139
|
+
* NOT "every team on the account", and the difference is load-bearing until
|
|
140
|
+
* fujibee/agmsg#650 lands. The set comes from `remote.sh status --json`, which
|
|
141
|
+
* drops a team it could not read and exits 0 — so an active team that has never
|
|
142
|
+
* been backed up can be missed with nothing said. Every sentence describing this
|
|
143
|
+
* command, here and in `--help` and in the README, stops at what the machine
|
|
144
|
+
* reported for exactly that reason. Widening them is the change that closes
|
|
145
|
+
* #182, and it comes after #650, not before.
|
|
146
|
+
*
|
|
147
|
+
* IDEMPOTENT. A run where no team's epochs have moved writes nothing and leaves
|
|
148
|
+
* the revision where it was. `setup` that appended an identical version every
|
|
149
|
+
* time was a name promising one thing and a behaviour doing another — and the
|
|
150
|
+
* cost is not storage, it is that a version history exists to show when
|
|
151
|
+
* something changed, and identical versions make it unreadable.
|
|
152
|
+
*/
|
|
153
|
+
export async function cmdVaultPut(config, args) {
|
|
154
|
+
// Same check connect gets. Without it a machine with no agmsg install runs
|
|
155
|
+
// straight into `bash exited 127: …/remote.sh: No such file or directory` —
|
|
156
|
+
// the same absence connect reports as a named checklist.
|
|
157
|
+
ensurePreflight(preflight(config.scriptsDir, NEEDS.vaultPut));
|
|
158
|
+
const client = new CourierClient(config);
|
|
159
|
+
// Two questions, and they are not the same one.
|
|
160
|
+
//
|
|
161
|
+
// observed every team whose status this machine could READ, whatever it
|
|
162
|
+
// said. This is what makes "the vault holds a team that did not
|
|
163
|
+
// appear" mean something: without it, a team disconnected here
|
|
164
|
+
// is indistinguishable from one whose status could not be read.
|
|
165
|
+
// targets those with an ACTIVE binding. Only these may be filed — a
|
|
166
|
+
// disconnected team's ids are a leftover, and a vault entry
|
|
167
|
+
// under one records a backup against a remote this machine is no
|
|
168
|
+
// longer bound to.
|
|
169
|
+
const observed = args.team === undefined ? await connectedTeams(config.scriptsDir) : [];
|
|
170
|
+
const targets = args.team === undefined
|
|
171
|
+
? observed.filter((t) => t.state === 'active')
|
|
172
|
+
: [{ team: args.team, ...(await remoteBinding(config.scriptsDir, args.team)) }];
|
|
173
|
+
if (targets.length === 0) {
|
|
174
|
+
throw new Error('no team on this machine is connected to the hosted service, so there are no keys to back up.\n' +
|
|
175
|
+
'Run `agmsg-cloud connect <team>` first, or name a team if one is connected under a different name.');
|
|
176
|
+
}
|
|
177
|
+
// The account's vault, not this team's: one vault, one recovery key. A second
|
|
178
|
+
// team finds the vault that already exists and is added to it.
|
|
179
|
+
const { identity, version } = await readAccountVault(client);
|
|
180
|
+
let recoveryKey = null;
|
|
181
|
+
let vdk = null;
|
|
182
|
+
// Whether this run had to reach the key. It decides whether a slot is worth
|
|
183
|
+
// keeping afterwards: a run the slot already answered has one, and saving
|
|
184
|
+
// again would mint a fresh KEK and replace a working slot for nothing.
|
|
185
|
+
let askedForTheKey = true;
|
|
186
|
+
let container;
|
|
187
|
+
// Whether a key was put on this person's screen during this run. It decides
|
|
188
|
+
// what a later failure has to tell them: a key they wrote down that never
|
|
189
|
+
// came into use is worse than no key, because they will keep it.
|
|
190
|
+
if (version) {
|
|
191
|
+
// This machine's slot first. The recovery key is only asked for when the
|
|
192
|
+
// slot cannot answer — which is what makes "shown once, then filed away" a
|
|
193
|
+
// livable instruction rather than a thing nobody follows.
|
|
194
|
+
//
|
|
195
|
+
// The prompt, when it happens, is for the ACCOUNT's key. Asking "for team
|
|
196
|
+
// X" was true when a team had its own vault; saying it now would teach
|
|
197
|
+
// someone to expect a different key per team, which is the belief the
|
|
198
|
+
// account vault exists to remove.
|
|
199
|
+
const obtained = await vdkForVault(identity, version, args.team ?? targets[0].team);
|
|
200
|
+
vdk = obtained.vdk;
|
|
201
|
+
askedForTheKey = obtained.usedRecoveryKey;
|
|
202
|
+
// Open before writing. The container has to be read to add a team to it,
|
|
203
|
+
// and opening it is also what proves this key is the vault's key — a
|
|
204
|
+
// mistyped key derives a valid but different VDK, and a version stored
|
|
205
|
+
// under it is unopenable at exactly the moment there is no way back.
|
|
206
|
+
const opened = openVaultWithVdk(identity, version, vdk);
|
|
207
|
+
container = parseContainer(opened.content);
|
|
208
|
+
}
|
|
209
|
+
else {
|
|
210
|
+
// The one moment the recovery key exists in the clear.
|
|
211
|
+
//
|
|
212
|
+
// Generated ONCE per execution and never regenerated inside it. The
|
|
213
|
+
// previous version showed the key, asked for it back, and minted a fresh
|
|
214
|
+
// one if the retype missed — so the sentence "write this down, it is shown
|
|
215
|
+
// once" was followed, in the same run, by making what they wrote down
|
|
216
|
+
// useless. Anything that fails after this point fails with THIS key, and
|
|
217
|
+
// the failure says so.
|
|
218
|
+
const minted = generateRecoveryKey();
|
|
219
|
+
recoveryKey = normalizeRecoveryKey(minted);
|
|
220
|
+
// Shown BEFORE anything is stored, and this order is load-bearing.
|
|
221
|
+
//
|
|
222
|
+
// Showing it after the upload reads better — the vault exists, so the key
|
|
223
|
+
// opens something — but it adds a failure this one does not have: a crash
|
|
224
|
+
// between the upload and the screen leaves a vault encrypted under a key
|
|
225
|
+
// nobody has ever seen, and no recovery key can be produced for it again.
|
|
226
|
+
// A key on paper for a vault that was never created is recoverable by
|
|
227
|
+
// throwing the paper away; that is not.
|
|
228
|
+
//
|
|
229
|
+
// The device slot does NOT change this, which is worth stating because it
|
|
230
|
+
// sounds like it should. The slot is what makes later runs keyless, and it
|
|
231
|
+
// is written after the vault exists — so at THIS point in a first run there
|
|
232
|
+
// is nothing on this machine the key could be re-derived from. Showing last
|
|
233
|
+
// becomes safe only when the slot is durable before the vault is created,
|
|
234
|
+
// and that ordering is a separate change with its own failure to think
|
|
235
|
+
// through (a slot addressed to a vault that was never made).
|
|
236
|
+
await showRecoveryKey(minted, args.team);
|
|
237
|
+
container = emptyContainer();
|
|
238
|
+
}
|
|
239
|
+
// THE ENUMERATION IS NOT FAIL-CLOSED, AND THIS DOES NOT PRETEND TO FIX THAT.
|
|
240
|
+
//
|
|
241
|
+
// `remote.sh status --json` with no team DROPS a team it could not read and
|
|
242
|
+
// still exits 0, and the single-team form cannot tell "never connected" from
|
|
243
|
+
// "could not be read" either — same message, same exit code. Measured, and
|
|
244
|
+
// filed as fujibee/agmsg#650.
|
|
245
|
+
//
|
|
246
|
+
// The first version of this REFUSED when the vault held a team the run did
|
|
247
|
+
// not see. That was wrong, and the reason is worth keeping: a team
|
|
248
|
+
// deliberately disconnected here disappears from the enumeration too, and the
|
|
249
|
+
// vault keeps its entry forever by design — so the refusal fired every run,
|
|
250
|
+
// for the one reason that is not a problem, with no way out. Refusing on a
|
|
251
|
+
// distinction the system cannot draw treats "unknown" as "wrong".
|
|
252
|
+
//
|
|
253
|
+
// So it reports instead. What was not seen is put on the screen by name and
|
|
254
|
+
// the judgement goes back to the operator, who is the only party that knows
|
|
255
|
+
// whether they disconnected it. That leaves a real gap — nobody who does not
|
|
256
|
+
// read the output is protected — and the gap is smaller than a command that
|
|
257
|
+
// cannot be run.
|
|
258
|
+
//
|
|
259
|
+
// When #650 lands this can become a refusal again, because by then a team
|
|
260
|
+
// that could not be read will say so.
|
|
261
|
+
const seenThisRun = new Set(observed.map((t) => vaultTeamKey(t.serverInstanceId, t.teamId)));
|
|
262
|
+
const unseen = args.team === undefined
|
|
263
|
+
? container.teams.filter((e) => !seenThisRun.has(vaultTeamKey(e.server_instance_id, e.team_id)))
|
|
264
|
+
: [];
|
|
265
|
+
// Seen, and deliberately not filed. Kept apart from `unseen` because they are
|
|
266
|
+
// different facts: this machine HEARD about these, and the ones above it did
|
|
267
|
+
// not hear about at all.
|
|
268
|
+
const disconnected = observed.filter((t) => t.state === 'disconnected');
|
|
269
|
+
// WHAT THIS RUN COVERED, BY NAME. Written once, so both exits below say the
|
|
270
|
+
// same thing — a sentence written twice is one that eventually disagrees with
|
|
271
|
+
// itself, and these two paths differ in everything else.
|
|
272
|
+
//
|
|
273
|
+
// NO COUNT WITHOUT ITS MEMBERS. What the operator saw on the production
|
|
274
|
+
// walkthrough was `stored revision 3 — team 'rc3walk' backed up; 2 teams`,
|
|
275
|
+
// and the "2 teams" never said which two. A count reads as "the account is
|
|
276
|
+
// backed up", which is wider than what was checked: the set came from what
|
|
277
|
+
// the local store reported, and that store can come back short without
|
|
278
|
+
// saying so (agmsg#650).
|
|
279
|
+
//
|
|
280
|
+
// The first version of this got it wrong in its own way — it printed the
|
|
281
|
+
// heading "Covered" and then listed only the DISCONNECTED teams, because the
|
|
282
|
+
// backed-up ones are not known until the loop below has run. Hence the
|
|
283
|
+
// parameters: the groups are passed in rather than closed over, so the
|
|
284
|
+
// heading cannot outrun what is under it.
|
|
285
|
+
const coverage = (written, unchanged) => {
|
|
286
|
+
const group = (title, members) => members.length === 0 ? [] : [`\n ${title}\n`, ...members.map((m) => ` ${m}\n`)];
|
|
287
|
+
// The first two hold for either form — a run that names one team still has
|
|
288
|
+
// to say what it did with it. The rest are about the ENUMERATION, so they
|
|
289
|
+
// only mean anything when the enumeration is what chose the set.
|
|
290
|
+
const enumerated = args.team === undefined;
|
|
291
|
+
const lines = [
|
|
292
|
+
...group('Backed up in this revision:', written),
|
|
293
|
+
...group('Already current, nothing to write:', unchanged),
|
|
294
|
+
...(!enumerated
|
|
295
|
+
? []
|
|
296
|
+
: group('Disconnected on this machine, left as they are:', disconnected.map((t) => t.team))),
|
|
297
|
+
// Ids, because a team that never reported has no local name here to
|
|
298
|
+
// print — the same absence that put it on this list. Said plainly rather
|
|
299
|
+
// than dressed up as an instruction: this machine does not know whether
|
|
300
|
+
// these were disconnected on purpose, and the operator does.
|
|
301
|
+
...(!enumerated
|
|
302
|
+
? []
|
|
303
|
+
: group('In the vault, but not reported by this machine on this run:', unseen.map((e) => `${e.team_id} (server ${e.server_instance_id})`))),
|
|
304
|
+
];
|
|
305
|
+
if (unseen.length > 0) {
|
|
306
|
+
lines.push('\n Those entries are untouched. If you disconnected them here, nothing is\n', ' wrong. If you did not, this machine could not read their state and they\n', ' were not re-checked — that is agmsg#650, and it cannot be told apart\n', ' from here.\n');
|
|
307
|
+
}
|
|
308
|
+
// Said only by the form that DERIVED its set. A run given a team covered
|
|
309
|
+
// what it was told to; claiming it covered what the store reported would
|
|
310
|
+
// be a statement about an enumeration that never ran.
|
|
311
|
+
if (enumerated) {
|
|
312
|
+
lines.push("\n That is every team this machine's store reported. A team connected only\n", ' on another machine is backed up by running this there.\n');
|
|
313
|
+
}
|
|
314
|
+
return lines.join('');
|
|
315
|
+
};
|
|
316
|
+
const scratch = mkdtempSync(join(tmpdir(), 'agmsg-cloud-'));
|
|
317
|
+
try {
|
|
318
|
+
// One pass over every target. A team whose declared epochs already match
|
|
319
|
+
// what the vault holds is left exactly as it is — not re-sealed under a
|
|
320
|
+
// fresh nonce, which would be a new version saying nothing new.
|
|
321
|
+
let next = container;
|
|
322
|
+
const written = [];
|
|
323
|
+
const unchanged = [];
|
|
324
|
+
for (const [i, target] of targets.entries()) {
|
|
325
|
+
const bundleFile = join(scratch, `handoff-${i}.bundle`);
|
|
326
|
+
await keyHandoff(config.scriptsDir, target.team, bundleFile);
|
|
327
|
+
const bundle = readFileSync(bundleFile);
|
|
328
|
+
const keyIds = bundleKeyIds(bundle);
|
|
329
|
+
const already = next.teams.find((e) => e.server_instance_id === target.serverInstanceId && e.team_id === target.teamId);
|
|
330
|
+
if (already && sameEpochs(already.key_ids, keyIds)) {
|
|
331
|
+
unchanged.push(target.team);
|
|
332
|
+
continue;
|
|
333
|
+
}
|
|
334
|
+
next = upsertTeam(next, {
|
|
335
|
+
server_instance_id: target.serverInstanceId,
|
|
336
|
+
team_id: target.teamId,
|
|
337
|
+
key_ids: keyIds,
|
|
338
|
+
bundle: bundle.toString('base64'),
|
|
339
|
+
});
|
|
340
|
+
written.push(target.team);
|
|
341
|
+
}
|
|
342
|
+
// Nothing moved, and the vault already exists: there is no version to
|
|
343
|
+
// write. Saying "stored revision N+1" here would be true of the server and
|
|
344
|
+
// false about the account — the point of a revision is that something is
|
|
345
|
+
// different in it.
|
|
346
|
+
if (written.length === 0 && version && vdk) {
|
|
347
|
+
// The headline carries no bare count: what "up to date" covers is the
|
|
348
|
+
// list below it, and a number on its own is the thing this was reported
|
|
349
|
+
// for.
|
|
350
|
+
process.stdout.write(`already up to date — nothing has new keys. Revision stays at ${version.revision}.\n`);
|
|
351
|
+
process.stdout.write(coverage(written, unchanged));
|
|
352
|
+
// A run that had to type the key still earns a slot: the key is in hand,
|
|
353
|
+
// and the next run should not ask again just because this one had nothing
|
|
354
|
+
// to store.
|
|
355
|
+
if (askedForTheKey) {
|
|
356
|
+
await keepSlot(slotAddress(identity, version.vault_id, version.recovery_generation), vdk);
|
|
357
|
+
}
|
|
358
|
+
return;
|
|
359
|
+
}
|
|
360
|
+
const content = serializeContainer(next);
|
|
361
|
+
// The address this machine's slot is kept at, decided by which write
|
|
362
|
+
// happened: an append addresses the vault that already exists, a create
|
|
363
|
+
// addresses the one it just made — and takes the generation from the create
|
|
364
|
+
// rather than restating that a new vault is generation 1.
|
|
365
|
+
let address;
|
|
366
|
+
let revision;
|
|
367
|
+
if (version && vdk) {
|
|
368
|
+
const result = await appendVaultVersionWithVdk(client, identity, version, vdk, content);
|
|
369
|
+
revision = result.revision;
|
|
370
|
+
address = slotAddress(identity, version.vault_id, version.recovery_generation);
|
|
371
|
+
}
|
|
372
|
+
else {
|
|
373
|
+
if (recoveryKey === null)
|
|
374
|
+
throw new Error('unreachable: a create without a recovery key');
|
|
375
|
+
const created = await createVault(client, identity, recoveryKey, content);
|
|
376
|
+
revision = created.revision;
|
|
377
|
+
vdk = created.vdk;
|
|
378
|
+
address = slotAddress(identity, created.vaultId, created.recoveryGeneration);
|
|
379
|
+
}
|
|
380
|
+
// `; 2 teams in this account's vault` used to end this line. It is the
|
|
381
|
+
// count the walkthrough operator read and could not act on — it never said
|
|
382
|
+
// WHICH two, and a bare number here reads as a statement about the account
|
|
383
|
+
// rather than about what this run saw. The groups below say both.
|
|
384
|
+
process.stdout.write(`stored revision ${revision}${version ? '' : ' (vault created)'}\n`);
|
|
385
|
+
// WHAT WAS COVERED IS WHAT THIS MACHINE REPORTED, and that is what the line
|
|
386
|
+
// says. It used to end at the count above, which reads as "the account is
|
|
387
|
+
// backed up" — a claim wider than the thing that was checked, since the
|
|
388
|
+
// enumeration this ran on can come back short without saying so
|
|
389
|
+
// (agmsg#650). Teams already in the vault are covered by the refusal
|
|
390
|
+
// higher up; a team that has never been in it cannot be seen from here at
|
|
391
|
+
// all, so the sentence stops at the machine rather than the account.
|
|
392
|
+
process.stdout.write(coverage(written, unchanged));
|
|
393
|
+
// After the write, and only when this run had to reach the key. A run the
|
|
394
|
+
// slot already answered has a working slot at this address; saving again
|
|
395
|
+
// would mint a fresh KEK, replace it, and touch the keychain on every
|
|
396
|
+
// routine backup — for a machine that is already in the state the slot
|
|
397
|
+
// exists to produce. The same condition governs the restore path below,
|
|
398
|
+
// and it has to be the same condition in both or one of them is wrong.
|
|
399
|
+
if (askedForTheKey)
|
|
400
|
+
await keepSlot(address, vdk);
|
|
401
|
+
}
|
|
402
|
+
catch (err) {
|
|
403
|
+
if (!version) {
|
|
404
|
+
// The key is already on their screen, and the vault it was minted for
|
|
405
|
+
// does not exist. Said plainly, because the alternative is someone
|
|
406
|
+
// filing a key that opens nothing and trusting it for years. A re-run
|
|
407
|
+
// mints a new one, which is correct — and only correct if they know to
|
|
408
|
+
// discard this one.
|
|
409
|
+
process.stdout.write('\nThe backup did NOT complete, so the recovery key above was never used\n' +
|
|
410
|
+
'and opens nothing. Discard it. Running this command again will show you\n' +
|
|
411
|
+
'a new key.\n\n');
|
|
412
|
+
}
|
|
413
|
+
throw err;
|
|
414
|
+
}
|
|
415
|
+
finally {
|
|
416
|
+
rmSync(scratch, { recursive: true, force: true });
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
// The key epochs the bundle declares, for the entry's binding. This is the only
|
|
420
|
+
// part of the bundle the cloud side reads: everything else is carried opaquely
|
|
421
|
+
// and handed back to the OSS unlock path, so the bundle's format stays theirs.
|
|
422
|
+
function bundleKeyIds(bundle) {
|
|
423
|
+
try {
|
|
424
|
+
const parsed = JSON.parse(bundle.toString('utf8'));
|
|
425
|
+
const ids = (parsed.identities ?? [])
|
|
426
|
+
.map((i) => i.key_id)
|
|
427
|
+
.filter((k) => typeof k === 'string' && k !== '');
|
|
428
|
+
if (ids.length === 0)
|
|
429
|
+
throw new Error('no key ids');
|
|
430
|
+
return ids;
|
|
431
|
+
}
|
|
432
|
+
catch {
|
|
433
|
+
// Refuse rather than record an entry that binds nothing. A backup whose
|
|
434
|
+
// entry cannot say which epochs it holds is a backup nobody can reason
|
|
435
|
+
// about at restore time.
|
|
436
|
+
throw new Error('the handoff bundle does not declare any key epochs; refusing to file it ' +
|
|
437
|
+
'in the vault without them');
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
// `agmsg-cloud recovery restore <team>` — pull the current version, open it with the
|
|
441
|
+
// recovery key, and hand the bundle to the OSS unlock path.
|
|
442
|
+
export async function cmdVaultRestore(config, args) {
|
|
443
|
+
ensurePreflight(preflight(config.scriptsDir, NEEDS.vaultRestore));
|
|
444
|
+
const client = new CourierClient(config);
|
|
445
|
+
const binding = await remoteBinding(config.scriptsDir, args.team);
|
|
446
|
+
const { identity, version } = await readAccountVault(client);
|
|
447
|
+
if (!version) {
|
|
448
|
+
throw new Error('this account has no recovery vault yet — nothing to restore from');
|
|
449
|
+
}
|
|
450
|
+
// The slot first here too. Restore is the disaster path, so most of the time
|
|
451
|
+
// it runs on a machine that has no slot and the key is typed — but a machine
|
|
452
|
+
// that HAS one is a machine that has already proven it holds this vault's key,
|
|
453
|
+
// and asking it for the recovery key anyway would be asking for the one secret
|
|
454
|
+
// this design exists to keep filed away.
|
|
455
|
+
const { vdk, usedRecoveryKey } = await vdkForVault(identity, version, args.team);
|
|
456
|
+
const opened = openVaultWithVdk(identity, version, vdk);
|
|
457
|
+
// One vault, one key, whichever team was asked for: the container is opened
|
|
458
|
+
// as a whole and the team is selected from it.
|
|
459
|
+
const entry = selectTeam(parseContainer(opened.content), {
|
|
460
|
+
serverInstanceId: binding.serverInstanceId,
|
|
461
|
+
teamId: binding.teamId,
|
|
462
|
+
});
|
|
463
|
+
// Straight from the AEAD output into the OSS import, with no file in between.
|
|
464
|
+
// Writing it out first — even 0600 — would mean handing over a path, and a path
|
|
465
|
+
// can be made to point at other bytes after we vouched for these ones.
|
|
466
|
+
await unlockAuthenticatedBundle(config.scriptsDir, args.team, Buffer.from(entry.bundle, 'base64'));
|
|
467
|
+
process.stdout.write(`unlocked team '${args.team}' from the account vault, revision ${opened.revision}\n`);
|
|
468
|
+
// A restore that typed the key is the moment this machine first holds the
|
|
469
|
+
// vault's data key. Keeping a slot now is what stops the NEXT command on this
|
|
470
|
+
// replacement machine from asking for the key again — which is the whole
|
|
471
|
+
// complaint the slot exists to answer.
|
|
472
|
+
//
|
|
473
|
+
// Only when the key was typed: if the slot answered, one already exists at
|
|
474
|
+
// this address and saving would mint a second KEK for no reason.
|
|
475
|
+
if (usedRecoveryKey) {
|
|
476
|
+
await keepSlot(slotAddress(identity, version.vault_id, version.recovery_generation), vdk);
|
|
477
|
+
}
|
|
478
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import { CourierClient } from '../api.js';
|
|
2
|
+
import { shellArg } from '../shell-arg.js';
|
|
3
|
+
// (a) Bridge waiting enrollment requests to the operator's attention: poll the
|
|
4
|
+
// approver-facing list and print one line per NEW request, so it can be surfaced
|
|
5
|
+
// through the agmsg monitor. Runs until interrupted; only the first sighting of a
|
|
6
|
+
// request id is emitted.
|
|
7
|
+
//
|
|
8
|
+
// It deliberately does NOT print a code. At this point there is no code to
|
|
9
|
+
// print: the requester has published only a digest, and the value it hides is
|
|
10
|
+
// not revealed until an approver has committed. Anything shown here would have
|
|
11
|
+
// to be invented, and a code that appears before the ceremony has run is exactly
|
|
12
|
+
// the thing a user must never learn to accept.
|
|
13
|
+
export async function cmdWatch(config, args, deps = {}) {
|
|
14
|
+
const client = new CourierClient(config);
|
|
15
|
+
const interval = args.intervalMs ?? 15_000;
|
|
16
|
+
const sleep = deps.sleep ?? ((ms) => new Promise((r) => setTimeout(r, ms)));
|
|
17
|
+
const seen = new Set();
|
|
18
|
+
while (!deps.signal?.aborted) {
|
|
19
|
+
let live = [];
|
|
20
|
+
try {
|
|
21
|
+
live = await client.listEnrollments();
|
|
22
|
+
}
|
|
23
|
+
catch (err) {
|
|
24
|
+
process.stderr.write(`poll failed: ${err.message}\n`);
|
|
25
|
+
}
|
|
26
|
+
for (const req of live) {
|
|
27
|
+
// Only requests still waiting for an approver to commit. One already in
|
|
28
|
+
// flight belongs to whoever is running `approve`, and announcing it again
|
|
29
|
+
// would invite a second key holder to start a competing ceremony.
|
|
30
|
+
if (req.status !== 'requester_committed')
|
|
31
|
+
continue;
|
|
32
|
+
if (seen.has(req.id))
|
|
33
|
+
continue;
|
|
34
|
+
seen.add(req.id);
|
|
35
|
+
process.stdout.write(
|
|
36
|
+
// The id is the server's, not ours, and it lands in an argument
|
|
37
|
+
// position of a line written to be pasted. Quoted for that reason
|
|
38
|
+
// rather than because a UUID needs it: what makes a value safe here is
|
|
39
|
+
// where it goes, not what it happens to contain today.
|
|
40
|
+
`enrollment request ${req.id} "${req.label}" — run \`agmsg-cloud approve <team> ${shellArg(req.id)}\` ` +
|
|
41
|
+
`to compare codes with them (expires ${req.expires_at})\n`);
|
|
42
|
+
}
|
|
43
|
+
if (deps.signal?.aborted)
|
|
44
|
+
break;
|
|
45
|
+
await sleep(interval);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import { homedir } from 'node:os';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
import { readCredential } from './credentials.js';
|
|
4
|
+
// The two environment values are ALL-OR-NOTHING. Accepting one and filling the
|
|
5
|
+
// other from the store would combine a stored secret with an endpoint chosen
|
|
6
|
+
// elsewhere — that is not an override, it is sending this machine's capability
|
|
7
|
+
// secret to a host it was never minted for. Mixing is refused; replacing wholesale
|
|
8
|
+
// is not (CI and headless runs are exactly that).
|
|
9
|
+
function fromEnv(env) {
|
|
10
|
+
const baseUrl = env.AGMSG_CLOUD_ENDPOINT;
|
|
11
|
+
const secret = env.AGMSG_CLOUD_SECRET;
|
|
12
|
+
if (baseUrl && secret)
|
|
13
|
+
return { endpoint: baseUrl, secret };
|
|
14
|
+
if (baseUrl || secret) {
|
|
15
|
+
const given = baseUrl ? 'AGMSG_CLOUD_ENDPOINT' : 'AGMSG_CLOUD_SECRET';
|
|
16
|
+
const missing = baseUrl ? 'AGMSG_CLOUD_SECRET' : 'AGMSG_CLOUD_ENDPOINT';
|
|
17
|
+
throw new Error(`${given} is set but ${missing} is not. Set both or neither: a stored secret is only ever sent to the endpoint it was minted for.`);
|
|
18
|
+
}
|
|
19
|
+
return null;
|
|
20
|
+
}
|
|
21
|
+
// Where the OSS scripts live, resolved WITHOUT a credential.
|
|
22
|
+
//
|
|
23
|
+
// `connect --preflight` is a dry run whose whole purpose is to be usable before
|
|
24
|
+
// anything is set up, so it must not be refused for not being signed in.
|
|
25
|
+
export function resolveScriptsDir(env = process.env) {
|
|
26
|
+
return env.AGMSG_SCRIPTS_DIR ?? join(homedir(), '.agents', 'skills', 'agmsg', 'scripts');
|
|
27
|
+
}
|
|
28
|
+
export function loadConfig(env = process.env) {
|
|
29
|
+
const resolved = fromEnv(env) ?? readCredential(null, env);
|
|
30
|
+
if (!resolved) {
|
|
31
|
+
throw new Error('not signed in on this machine — run `agmsg-cloud login --endpoint <url>` first');
|
|
32
|
+
}
|
|
33
|
+
return { baseUrl: resolved.endpoint, secret: resolved.secret, scriptsDir: resolveScriptsDir(env) };
|
|
34
|
+
}
|