agmsg-cloud 0.1.3 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -78,6 +78,35 @@ export function writeAll(fd, bytes, what, write = writeSync) {
78
78
  * temp file, fsync, rename, fsync the directory — because a half-written
79
79
  * record is one `fetch` would either reject a good bundle over or, worse, read
80
80
  * a truncated digest from.
81
+ *
82
+ * Also clears any OTHER record already held for this same (origin, scope)
83
+ * (#523). A machine that stalls after this ceremony — the bundle never
84
+ * fetched — and then connects again performs a SECOND ceremony for the same
85
+ * credential and device, and without this the old approval sat beside the
86
+ * new one forever: `fetch` finds two records for one scope and, correctly,
87
+ * refuses to guess which bundle either belongs to. There is exactly one live
88
+ * approval per (credential, device) at a time, so the new one replacing the
89
+ * old is not a loss — the old one was already superseded the moment this
90
+ * ceremony ran, since it authenticates the account's current snapshot, not a
91
+ * fixed past one.
92
+ *
93
+ * Two ceremonies for the SAME credential and device, running at once, only
94
+ * clear records OLDER than the one each just wrote (raised in review) —
95
+ * never the newest, and never each other. Without that qualifier, two
96
+ * concurrent calls could each write their own record and then each delete
97
+ * the OTHER's, leaving zero records for a scope that just ran two
98
+ * successful ceremonies — worse than the pile-up this exists to fix. This
99
+ * guarantees the scope is never left with zero records, however two
100
+ * genuinely concurrent calls interleave; it does not by itself guarantee
101
+ * they converge to exactly one (a pair racing closely enough can both
102
+ * survive, since each only knows what predates it, not what is still being
103
+ * written elsewhere). Running two ceremonies for one scope at once is out
104
+ * of scope to serialize against, deliberately — nobody starts a second
105
+ * `request` on top of one already in flight. What this fix exists to
106
+ * guarantee, and does, is the case that actually happens: from a stuck
107
+ * two-record state, one ordinary retry always recovers to exactly one,
108
+ * because that retry's own record is newer than both existing ones and
109
+ * clears them both in the same pass.
81
110
  */
82
111
  export function recordAuthenticatedDigest(input, env = process.env) {
83
112
  if (!HEX64.test(input.handoffDigest)) {
@@ -94,10 +123,11 @@ export function recordAuthenticatedDigest(input, env = process.env) {
94
123
  // from the first one — the same reason credentials.ts does it (raised in review).
95
124
  enforceMode(dir, 0o700);
96
125
  const path = pathFor(dir, input.serverOrigin, input.requestId);
126
+ const scope = fingerprintScope(input.secret, input.devicePubkey);
97
127
  const record = {
98
128
  requestId: input.requestId,
99
129
  handoffDigest: input.handoffDigest,
100
- scopeFingerprint: fingerprintScope(input.secret, input.devicePubkey),
130
+ scopeFingerprint: scope,
101
131
  recordedAt: new Date().toISOString(),
102
132
  };
103
133
  const tmp = `${path}.${randomBytes(8).toString('hex')}.tmp`;
@@ -143,6 +173,116 @@ export function recordAuthenticatedDigest(input, env = process.env) {
143
173
  // reason it is now probed rather than assumed (#512). Fixing only the other
144
174
  // site would have moved the EPERM here instead of removing it.
145
175
  syncDirectoryEntryIfSupported(dirname(path));
176
+ // AFTER the new record is durably in place, never before: deleting the old
177
+ // one first would leave a window with zero records for this scope if the
178
+ // write above failed partway. Best-effort from here — see
179
+ // removeOlderRecordsForScope — because the new record above is already
180
+ // durable, and nothing below may throw back to a caller who would only
181
+ // retry and pile up yet another new record (raised in review).
182
+ removeOlderRecordsForScope(dir, input.serverOrigin, scope, input.requestId, record.recordedAt);
183
+ }
184
+ /**
185
+ * Best-effort cleanup of every OTHER, OLDER record this (origin, scope)
186
+ * still holds — older BY `recordedAt`, strictly, never equal or newer
187
+ * (raised in review). That qualifier is what keeps this safe under two
188
+ * concurrent ceremonies for the same scope: each of the two only deletes
189
+ * what it can already tell predates it, so neither ever deletes the other
190
+ * when both are genuinely new, and whichever of the two IS the newest is
191
+ * never deleted by the other. What this does NOT guarantee: running two
192
+ * ceremonies for one scope AT THE SAME TIME can still leave both of their
193
+ * records behind — deliberately out of scope, since nobody runs a second
194
+ * `request` on top of one already in flight, and this never claims a
195
+ * concurrent pair converges to one on its own. What it does guarantee is
196
+ * the case that matters: from that stuck two-record state, ONE ordinary
197
+ * retry always recovers it, because the retry's own record is newer than
198
+ * both and this cleanup removes them both in that single pass. Never zero,
199
+ * either way, is unconditional.
200
+ *
201
+ * Best-effort, not authoritative, in two more ways. A neighbour this cannot
202
+ * parse, or whose `recordedAt` this cannot confidently compare, is left
203
+ * alone rather than guessed at, because a wrongly-deleted approval is a
204
+ * fetch nobody can complete, while a wrongly-kept one is only a state
205
+ * `fetch`'s own multiple-record refusal already handles. And a deletion that
206
+ * FAILS is caught rather than thrown, for the same reason: this runs after
207
+ * the new record is already durable, so throwing here would report failure
208
+ * for a call that mostly succeeded, and a caller who retries on that error
209
+ * would only record ANOTHER new approval on top (raised in review). Either
210
+ * way, `fetch`'s refusal stays the backstop for whatever this pass cannot
211
+ * confidently remove.
212
+ */
213
+ function removeOlderRecordsForScope(dir, origin, scope, keepRequestId, keepRecordedAt) {
214
+ const prefix = `${Buffer.from(origin, 'utf8').toString('base64url')}.`;
215
+ let names;
216
+ try {
217
+ names = readdirSync(dir);
218
+ }
219
+ catch {
220
+ // Same reasoning as a failed deletion below: the new record is already
221
+ // durable, so a directory this cannot even list must not fail the call.
222
+ return;
223
+ }
224
+ let removedAny = false;
225
+ for (const name of names) {
226
+ if (!name.startsWith(prefix) || !name.endsWith('.json'))
227
+ continue;
228
+ const candidate = join(dir, name);
229
+ let parsed;
230
+ try {
231
+ const fd = openSync(candidate, constants.O_RDONLY | constants.O_NOFOLLOW);
232
+ try {
233
+ parsed = JSON.parse(readFileSync(fd, 'utf8'));
234
+ }
235
+ finally {
236
+ closeSync(fd);
237
+ }
238
+ }
239
+ catch {
240
+ continue;
241
+ }
242
+ const r = parsed;
243
+ if (typeof r?.['requestId'] !== 'string' || r['requestId'] === keepRequestId)
244
+ continue;
245
+ if (r['scopeFingerprint'] !== scope)
246
+ continue;
247
+ // Strictly older, by ISO-8601 string comparison — valid because
248
+ // `recordedAt` is always written by `toISOString()` here, a fixed-width
249
+ // format lexical order agrees with chronological order on. Equal or
250
+ // unreadable is NOT older: a missing/malformed timestamp cannot be
251
+ // confidently placed before this one, and a tie (two ceremonies stamped
252
+ // in the same millisecond) must not delete either — that is what keeps
253
+ // a concurrent pair from being able to empty the scope between them.
254
+ if (typeof r['recordedAt'] !== 'string' || !(r['recordedAt'] < keepRecordedAt))
255
+ continue;
256
+ try {
257
+ rmSync(candidate, { force: true });
258
+ removedAny = true;
259
+ }
260
+ catch {
261
+ continue;
262
+ }
263
+ }
264
+ // Once, after every deletion in this pass, not per file: one flush covers
265
+ // everything this call removed. Without it, a crash right after could roll
266
+ // a removal back on disk while this pass has already moved on — putting the
267
+ // old record back and reintroducing the exact stuck state this cleanup
268
+ // exists to prevent (raised in review).
269
+ //
270
+ // Caught, not thrown (raised in review): this whole function runs after
271
+ // the caller's own new record is already durable, so a failure HERE must
272
+ // not be reported back as the call having failed — that would tell a
273
+ // caller whose write actually succeeded to retry, and a retry only
274
+ // records yet another new approval on top.
275
+ if (removedAny) {
276
+ try {
277
+ syncDirectoryEntryIfSupported(dir);
278
+ }
279
+ catch {
280
+ // Best-effort, like the deletions themselves: the files are gone from
281
+ // the directory listing either way, and `fetch`'s own multiple-record
282
+ // refusal is what protects a machine if this particular flush is what
283
+ // a crash rolls back.
284
+ }
285
+ }
146
286
  }
147
287
  /**
148
288
  * Every settled ceremony recorded for this origin.
@@ -4,6 +4,7 @@ import { join } from 'node:path';
4
4
  import { CourierClient } from '../api.js';
5
5
  import { clearAuthenticatedDigest, readAuthenticatedDigests } from '../authenticated-digest.js';
6
6
  import { originOf } from '../credentials.js';
7
+ import { shellArg } from '../shell-arg.js';
7
8
  import { decryptWithIdentity, localTeamLookup, publicKeyOf, unlockBundle, verifyHandoffDigest, } from '../oss.js';
8
9
  import { deviceIdentityPath } from '../paths.js';
9
10
  import { NEEDS, ensurePreflight, preflight } from '../preflight.js';
@@ -107,7 +108,11 @@ env = process.env) {
107
108
  // not a weaker check — it is no check, in the shape of one.
108
109
  const ids = authenticated.map((a) => ` ${a.requestId} (${a.recordedAt})`).join('\n');
109
110
  throw new Error(`${authenticated.length} approved enrollments are recorded for this service:\n\n${ids}\n\n` +
110
- 'Which one this bundle belongs to cannot be decided from here.');
111
+ 'Which one this bundle belongs to cannot be decided from here.\n\n' +
112
+ `Run \`agmsg-cloud sync ${shellArg(args.team)}\` again and have it approved: a\n` +
113
+ 'fresh approval replaces every other one recorded here for this machine, leaving\n' +
114
+ 'exactly one. (Run `agmsg-cloud request <label>` instead if this machine got here\n' +
115
+ 'through `request` on its own rather than through `sync`.)');
111
116
  }
112
117
  const consumed = authenticated[0];
113
118
  const expected = consumed.handoffDigest;
@@ -53,8 +53,16 @@ export async function cmdPull(config, opts) {
53
53
  // What it adds is the route, which the OSS script no longer names because
54
54
  // its route is not ours. Phrased as a condition rather than a claim, so it
55
55
  // is true after either line above.
56
- process.stdout.write(`\n"${opts.team}" is on this machine.\n`);
56
+ //
57
+ // BOTH lines withheld under `nextStepsFromCaller`, not just the route
58
+ // (#523). `sync` runs `fetch` immediately after this, and "is on this
59
+ // machine" reads as this step's own success — which it is — right before a
60
+ // failing `fetch` throws with nothing to say the two are different steps.
61
+ // The caller owns saying what happened once every one of its steps has
62
+ // run; this side saying its own half early is what let the two read as one
63
+ // outcome.
57
64
  if (opts.nextStepsFromCaller !== true) {
65
+ process.stdout.write(`\n"${opts.team}" is on this machine.\n`);
58
66
  process.stdout.write(`If it is still locked, \`agmsg-cloud sync ${shellArg(opts.team)}\` completes the key handoff:\n` +
59
67
  'it asks a machine that already has the team, and the two of you compare eight digits.\n');
60
68
  }
@@ -216,7 +216,19 @@ export async function cmdSync(config, opts) {
216
216
  // server says the ceremony was approved — `request` waits for that, and
217
217
  // throws otherwise. The bundle is checked against the snapshot those digits
218
218
  // authenticated, by machine.
219
- await fetch(config, { team: opts.team });
219
+ //
220
+ // `pull` no longer says anything on its own when it is a step of this
221
+ // command (#523), so if this throws, nothing above has claimed success yet
222
+ // — the operator has only been told the ceremony finished. Said here,
223
+ // before the error propagates, so a failure at this LAST step still reads
224
+ // as a failure and not as a silent stop after what looked like the finish.
225
+ try {
226
+ await fetch(config, { team: opts.team });
227
+ }
228
+ catch (err) {
229
+ out(`\n"${opts.team}" is on this machine, but the key has not arrived — it cannot be used yet.\n`);
230
+ throw err;
231
+ }
220
232
  // The last thing said, because "what now" is the question the screen leaves
221
233
  // otherwise: someone who has just run five steps and watched a code
222
234
  // comparison has every reason to expect a sixth.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "agmsg-cloud",
3
- "version": "0.1.3",
3
+ "version": "0.1.4",
4
4
  "description": "Companion CLI for the agmsg cloud service: connect a team, join from another machine, and back up its keys.",
5
5
  "license": "UNLICENSED",
6
6
  "type": "module",