@ran-sh/dsh-crew 1.10.0 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "1.10.0",
3
+ "version": "1.10.2",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "1.10.0",
3
+ "version": "1.10.2",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -11,6 +11,7 @@ import {
11
11
  existsSync,
12
12
  lstatSync,
13
13
  mkdirSync,
14
+ readdirSync,
14
15
  readFileSync,
15
16
  realpathSync,
16
17
  renameSync,
@@ -528,9 +529,21 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
528
529
  return recovery;
529
530
  }
530
531
  }
531
- try { rmSync(liveRoot, { recursive: true, force: true }); } catch {}
532
532
  try {
533
- if (existsSync(prevRoot)) { rename(prevRoot, liveRoot); recovery.restore = true; }
533
+ rmSync(liveRoot, { recursive: true, force: true });
534
+ } catch (error) {
535
+ // A failed removal must stop the recovery. Swallowing it left the window
536
+ // open: the restore was attempted over whatever remained, and
537
+ // `startOwned()` ran regardless — so a tree that could not be removed was
538
+ // then asked to start again, which on Windows is exactly the case where
539
+ // live handles blocked the removal.
540
+ recovery.removeError = String(error?.message ?? error);
541
+ recovery.ok = false;
542
+ return recovery;
543
+ }
544
+ try {
545
+ if (!existsSync(prevRoot)) recovery.restoreError = `previous runtime is missing at ${prevRoot}`;
546
+ else { rename(prevRoot, liveRoot); recovery.restore = true; }
534
547
  } catch (error) {
535
548
  recovery.restoreError = String(error?.message ?? error);
536
549
  }
@@ -538,6 +551,14 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
538
551
  recovery.ok = recovery.restore === true;
539
552
  return recovery;
540
553
  }
554
+ // Restart only on positive proof that the prior runtime is the tree now at the
555
+ // live root. Starting without it would run the failed candidate again, or
556
+ // nothing at all, and report it as a recovery.
557
+ if (recovery.restore !== true || !existsSync(liveRoot)) {
558
+ recovery.restartSkipped = 'the previous runtime was not restored';
559
+ recovery.ok = false;
560
+ return recovery;
561
+ }
541
562
  try {
542
563
  const restarted = await startOwned();
543
564
  recovery.restart = restarted?.ok === true;
@@ -551,6 +572,8 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
551
572
 
552
573
  // ---- retained runtime cohorts -------------------------------------------------
553
574
 
575
+ const DISPLACED_RUNTIME_PREFIX = 'runtime-displaced-';
576
+
554
577
  function retainedRuntimesRoot({ home }) {
555
578
  return join(crewDshHome({ home }), RETAINED_RUNTIMES_DIRNAME);
556
579
  }
@@ -572,12 +595,22 @@ function runtimeTreeVersion(root, read = readFileSync) {
572
595
  // Best-effort retention of the swapped-out prior runtime tree. Never throws
573
596
  // and never fails the caller: retention is an optimization for offline
574
597
  // rollback, not a correctness requirement of the migration itself.
575
- function retainPriorRuntime({ home, prevRoot, rename = renameSync }) {
598
+ /**
599
+ * Retain a parked runtime tree under the version it ships.
600
+ *
601
+ * `removeUnreadable` exists for one caller: the rollback rotation, where the
602
+ * parked tree is the ONLY copy of the cohort just displaced. There, failing to
603
+ * read its version is not a reason to delete it — that would destroy the very
604
+ * copy the rotation exists to keep — so it stays parked and is found later by
605
+ * `findRetainedRuntime`. The forward migration path keeps the default, where an
606
+ * unreadable tree really is junk.
607
+ */
608
+ function retainPriorRuntime({ home, prevRoot, rename = renameSync, removeUnreadable = true }) {
576
609
  try {
577
610
  if (!existsSync(prevRoot)) return { ok: true, retained: false };
578
611
  const version = runtimeTreeVersion(prevRoot);
579
612
  if (!version) {
580
- // No version to key retention on; the tree is unreadable junk.
613
+ if (!removeUnreadable) return { ok: false, retained: false, prevRoot, error: 'parked cohort version unreadable; left in place' };
581
614
  try { rmSync(prevRoot, { recursive: true, force: true }); } catch {}
582
615
  return { ok: true, retained: false, reason: 'prior tree version unreadable; removed' };
583
616
  }
@@ -601,9 +634,24 @@ function retainPriorRuntime({ home, prevRoot, rename = renameSync }) {
601
634
  export function findRetainedRuntime({ home = homedir(), version, exists = existsSync, read = readFileSync } = {}) {
602
635
  if (typeof version !== 'string' || version.length === 0) return null;
603
636
  const dir = retainedRuntimeDir({ home, version });
604
- if (!exists(dir)) return null;
605
- if (runtimeTreeVersion(dir, read) !== version) return null;
606
- return dir;
637
+ if (exists(dir) && runtimeTreeVersion(dir, read) === version) return dir;
638
+ // A rotation whose retain step failed leaves the displaced cohort parked in the
639
+ // harness home. It is a complete, valid copy of that cohort, so it is found
640
+ // here rather than being written off: otherwise a later offline rollback to
641
+ // that version reports RETAINED_MISSING while the tree sits on disk.
642
+ return findDisplacedRuntime({ home, version, exists, read });
643
+ }
644
+
645
+ function findDisplacedRuntime({ home, version, exists = existsSync, read = readFileSync }) {
646
+ let names;
647
+ try { names = readdirSync(crewDshHome({ home })); } catch { return null; }
648
+ for (const name of names) {
649
+ if (!name.startsWith(DISPLACED_RUNTIME_PREFIX)) continue;
650
+ const dir = join(crewDshHome({ home }), name);
651
+ if (!exists(dir)) continue;
652
+ if (runtimeTreeVersion(dir, read) === version) return dir;
653
+ }
654
+ return null;
607
655
  }
608
656
 
609
657
  // Offline cohort restore for cross-cohort rollback: move the retained tree
@@ -642,14 +690,34 @@ export async function restoreRetainedRuntime({
642
690
  if (!stop.ok) {
643
691
  return { ok: false, code: stop.code ?? 'DSH_RUNTIME_STOP_FAILED', error: stop.error ?? 'could not stop owned 3210' };
644
692
  }
693
+ // Park the displaced cohort instead of deleting it. Deleting it was what made a
694
+ // second rollback impossible: after B→A the B tree was gone, so a later A→B
695
+ // found no retained B. Retention has to rotate, like the forward path does.
696
+ const parked = join(crewDshHome({ home }), `${DISPLACED_RUNTIME_PREFIX}${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`);
697
+ let parkedLive = false;
698
+ try {
699
+ if (existsSync(liveRoot)) { rename(liveRoot, parked); parkedLive = true; }
700
+ } catch (error) {
701
+ return { ok: false, code: 'DSH_RUNTIME_PARK_FAILED', error: String(error?.message ?? error) };
702
+ }
645
703
  try {
646
- if (existsSync(liveRoot)) rmSync(liveRoot, { recursive: true, force: true });
647
704
  rename(retained, liveRoot);
648
705
  } catch (error) {
649
- // Restore the pre-existing live tree is impossible (it was replaced only
650
- // on success above); report and let the caller reconcile.
706
+ // Put the displaced tree back: leaving the live root empty because a rename
707
+ // failed would take a working runtime down for nothing.
708
+ try { if (parkedLive && !existsSync(liveRoot)) rename(parked, liveRoot); } catch { /* reported below */ }
651
709
  return { ok: false, code: 'DSH_RUNTIME_RESTORE_SWAP_FAILED', error: String(error?.message ?? error) };
652
710
  }
711
+ // The cohort just displaced becomes the retained one, so the version this
712
+ // restore moved away from can still be rolled back to. This parked tree is the
713
+ // only copy of that cohort, so an unreadable version keeps it rather than
714
+ // deleting it, and a failure is surfaced instead of passing as a clean restore.
715
+ const rotated = parkedLive
716
+ ? retainPriorRuntime({ home, prevRoot: parked, rename, removeUnreadable: false })
717
+ : { ok: true, retained: false };
718
+ if (parkedLive && rotated.ok === false) {
719
+ log(`! could not retain the displaced runtime cohort; it remains at ${parked}: ${rotated.error ?? ''}`);
720
+ }
653
721
  if (prepareOnly) {
654
722
  log(`- runtime tree prepared offline from retained tree (@${version}); process not started`);
655
723
  return { ok: true, version, liveRoot, prepared: true };
@@ -8,7 +8,12 @@ import { readHistoryState, writeHistoryState, historyPending, publicHistoryState
8
8
  import { readSessionOrigins } from '../session-origins.mjs';
9
9
  import { defaultWorktreeRoot } from '../workspace-isolation.mjs';
10
10
 
11
- export function createHistoryService({ crewRoot, agents, persistence, runtimeId, launch, now = Date.now }) {
11
+ // `readOrigins` is injectable because the provenance ledger it reads by default
12
+ // is machine-global, and the plan revision hashes the whole set: anything that
13
+ // appends to that file between a preview and its execute invalidates the plan.
14
+ // That is correct in production, where the ledger is this machine's own record,
15
+ // but a caller that cannot control the ledger cannot control its own revisions.
16
+ export function createHistoryService({ crewRoot, agents, persistence, runtimeId, launch, now = Date.now, readOrigins = readSessionOrigins }) {
12
17
  const gate = installHistoryAdmissionGate(agents, () => historyPending(crewRoot));
13
18
  const plans = new Map();
14
19
  let entering = false;
@@ -67,7 +72,7 @@ export function createHistoryService({ crewRoot, agents, persistence, runtimeId,
67
72
  });
68
73
  // A retained child keeps its ancestor chain; do not leave a newer fork orphaned.
69
74
  const workspaces = Object.entries(store.tables.workspaces).map(([id, row]) => ({ id, ...row }));
70
- const crewSessionIds = [...readSessionOrigins()];
75
+ const crewSessionIds = [...readOrigins()];
71
76
  const crewSet = new Set(crewSessionIds);
72
77
  for (const row of sessions) if (crewSet.has(row.id)) row.crew = true;
73
78
  // The isolated-workspace marker: the hub stamps every worktree session with a
package/src/hub/entry.mjs CHANGED
@@ -10,7 +10,7 @@
10
10
  import { apply as applyHub, inject, name, WorkerRegistry } from './index.mjs';
11
11
  import { getHubRuntimeIdentity } from '../runtime-identity.mjs';
12
12
  import { getProcessAdaptiveHealthStore } from '../adaptive-routing.mjs';
13
- import { claimReleaseInUse } from '../release-in-use.mjs';
13
+ import { claimReleaseInUse, clearReleaseClaim } from '../release-in-use.mjs';
14
14
 
15
15
  const RUNTIME_PATH = '/_dsh/dsh-crew/runtime';
16
16
  const ADAPTIVE_OBSERVER_INSTALLED = Symbol.for('@ran-sh/dsh-crew/adaptive-observer-installed');
@@ -97,8 +97,37 @@ export async function apply(ctx) {
97
97
  // modules lazily with a cache-busting query, so deleting the release under a
98
98
  // running process would break those routes with no way to recover but a
99
99
  // restart. Retention reads these claims and leaves a live release alone.
100
- claimReleaseInUse();
101
- registerRuntimeEndpoint(ctx);
102
- installAdaptiveHealthObserver();
103
- return applyHub(ctx);
100
+ //
101
+ // A claim that could not be written is worth saying out loud: retention cannot
102
+ // see an unclaimed release, so this process is then the one that a later update
103
+ // may delete from under itself.
104
+ const claimFile = claimReleaseInUse();
105
+ if (!claimFile) {
106
+ ctx.logger?.warn?.('dsh-crew: could not record this release as in use; a later update may prune it while this Hub is running');
107
+ }
108
+ try {
109
+ registerRuntimeEndpoint(ctx);
110
+ installAdaptiveHealthObserver();
111
+ const disposeHub = await applyHub(ctx);
112
+ // Release the claim on disposal, and only once teardown has actually
113
+ // finished: clearing it first would say "this release is unused" while the
114
+ // Hub is still running. Nothing else may remove a claim file — the reader
115
+ // deliberately never unlinks one, because it cannot tell its object from a
116
+ // successor published at the same name — but the process that owns a claim
117
+ // may remove its own, and it removes *this* claim rather than any other this
118
+ // process happens to hold.
119
+ return async () => {
120
+ try { if (typeof disposeHub === 'function') await disposeHub(); }
121
+ // Only a claim this mount actually holds. A null handle means the claim was
122
+ // never acquired, and clearing something anyway could remove a sibling
123
+ // mount's protection.
124
+ finally { if (claimFile) { try { clearReleaseClaim({ file: claimFile }); } catch { /* best effort */ } } }
125
+ };
126
+ } catch (error) {
127
+ // No disposer will ever be returned for a failed mount, so the claim has to
128
+ // be released here or the release stays pinned by a process that never
129
+ // provided anything.
130
+ if (claimFile) { try { clearReleaseClaim({ file: claimFile }); } catch { /* best effort */ } }
131
+ throw error;
132
+ }
104
133
  }
@@ -42,7 +42,7 @@ import { homedir } from 'node:os';
42
42
  import * as realInstaller from './install.mjs';
43
43
  import { samePayloadContent, capturePayloadContent } from './payload-content.mjs';
44
44
  import { crewDshHome, crewProfileDir } from './install.mjs';
45
- import { liveReleaseClaims } from '../release-in-use.mjs';
45
+ import { releaseClaimsState } from '../release-in-use.mjs';
46
46
  import { ensureCrewDshRuntime, ensureCrewPluginRegistration, removeCrewPluginRegistration, migrateCrewDshRuntime, installDshInto, restoreRetainedRuntime, crewDshRuntimeRoot, payloadDshVersion, TARGET_DSH_VERSION } from '../dsh-cli-runtime.mjs';
47
47
  import {
48
48
  ensureOfficialWebIntegration,
@@ -161,21 +161,62 @@ export function readCurrentPointerState({ home = homedir() } = {}) {
161
161
  return { status: 'malformed', file, code: 'POINTER_FIELDS_INVALID' };
162
162
  }
163
163
  if (!isAbsolute(raw.path)) return { status: 'malformed', file, code: 'POINTER_PATH_NOT_ABSOLUTE' };
164
- return { status: 'valid', file, pointer: raw };
164
+ // The pointer names the release that is live, and recovery acts on it. A path
165
+ // outside the managed releases directory is a corrupt or tampered pointer, not
166
+ // a release, so it fails closed exactly like the other malformed cases — and
167
+ // the canonical value is what the caller gets, so every later use acts on the
168
+ // path that was checked.
169
+ const canonical = managedReleasePath({ home, value: raw.path });
170
+ if (canonical === null) {
171
+ return { status: 'malformed', file, code: 'POINTER_PATH_OUTSIDE_RELEASES' };
172
+ }
173
+ return { status: 'valid', file, pointer: { ...raw, path: canonical } };
174
+ }
175
+
176
+ /**
177
+ * Resolve a path that a journal, the pointer, or recovery is about to act on, and
178
+ * refuse it unless it is inside the Crew-owned releases directory.
179
+ *
180
+ * An absolute path is not enough. Recovery recursively deletes the journal's
181
+ * candidate stage directory and moves the pointer's release, so a syntactically
182
+ * valid but corrupt or tampered journal naming any absolute path would grant
183
+ * delete authority over it — including the official `~/.dsh` tree this plugin is
184
+ * required never to touch. Containment is therefore checked before any read,
185
+ * write, delete or activation, and a path that fails it is treated exactly like
186
+ * a malformed journal: nothing is touched and the state is retained.
187
+ */
188
+ export function managedReleasePath({ home = homedir(), value } = {}) {
189
+ if (typeof value !== 'string' || value.length === 0 || !isAbsolute(value)) return null;
190
+ const releases = realpathOr(resolve(crewReleasesDir({ home })));
191
+ const candidate = realpathOr(resolve(value));
192
+ // Resolve both sides before comparing: `..` segments and symlinks must not be
193
+ // able to leave the managed directory after the check.
194
+ const relative = relativeTo(releases, candidate);
195
+ if (relative === null || relative === '' || relative.startsWith('..') || isAbsolute(relative)) return null;
196
+ return candidate;
165
197
  }
166
198
 
167
- function validJournalRelease(value) {
199
+ function realpathOr(path) {
200
+ try { return realpathSync(path); } catch { return path; }
201
+ }
202
+
203
+ /** `path` expressed relative to `from`, or null when they are on different roots. */
204
+ function relativeTo(from, path) {
205
+ return relative(from, path);
206
+ }
207
+
208
+ function validJournalRelease({ home, value }) {
168
209
  return !!value && typeof value === 'object'
169
210
  && typeof value.name === 'string' && value.name.length > 0
170
211
  && typeof value.version === 'string' && value.version.length > 0
171
- && typeof value.path === 'string' && isAbsolute(value.path);
212
+ && managedReleasePath({ home, value: value.path }) !== null;
172
213
  }
173
214
 
174
- function validJournalCandidate(value) {
215
+ function validJournalCandidate({ home, value }) {
175
216
  return !!value && typeof value === 'object'
176
217
  && typeof value.name === 'string' && value.name.length > 0
177
218
  && typeof value.version === 'string' && value.version.length > 0
178
- && typeof value.stageDir === 'string' && isAbsolute(value.stageDir);
219
+ && managedReleasePath({ home, value: value.stageDir }) !== null;
179
220
  }
180
221
 
181
222
  function writeFileAtomic(file, content) {
@@ -359,13 +400,23 @@ function readUpdateJournal({ home = homedir() } = {}) {
359
400
  return { malformed: true, file };
360
401
  }
361
402
  // Full schema check: a semantically broken journal (null candidate,
362
- // incomplete prior) must fail closed, never enter recovery.
363
- if (!validJournalCandidate(raw.candidate)) {
403
+ // incomplete prior) must fail closed, never enter recovery. The paths are
404
+ // checked for containment too, because recovery deletes and moves them, and
405
+ // they are replaced with the canonical value that was checked so that every
406
+ // later use acts on the path that was validated rather than the string that
407
+ // happened to be written.
408
+ const canonicalCandidate = managedReleasePath({ home, value: raw.candidate?.stageDir });
409
+ if (!validJournalCandidate({ home, value: raw.candidate }) || canonicalCandidate === null) {
364
410
  return { malformed: true, file, code: 'JOURNAL_CANDIDATE_SCHEMA_INVALID' };
365
411
  }
366
- if (raw.prior !== null && raw.prior !== undefined && !validJournalRelease(raw.prior)) {
367
- return { malformed: true, file, code: 'JOURNAL_PRIOR_SCHEMA_INVALID' };
412
+ if (raw.prior !== null && raw.prior !== undefined) {
413
+ const canonicalPrior = managedReleasePath({ home, value: raw.prior?.path });
414
+ if (!validJournalRelease({ home, value: raw.prior }) || canonicalPrior === null) {
415
+ return { malformed: true, file, code: 'JOURNAL_PRIOR_SCHEMA_INVALID' };
416
+ }
417
+ raw = { ...raw, prior: { ...raw.prior, path: canonicalPrior } };
368
418
  }
419
+ raw = { ...raw, candidate: { ...raw.candidate, stageDir: canonicalCandidate } };
369
420
  return raw;
370
421
  }
371
422
 
@@ -554,7 +605,9 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
554
605
  return { ok: false, code: 'POINTER_MALFORMED', file: pointerState.file, error: pointerState.code ?? 'pointer unreadable; refusing recovery' };
555
606
  }
556
607
  const pointer = pointerState.status === 'valid' ? pointerState.pointer : null;
557
- const candidateDir = journal.candidate?.stageDir ?? null;
608
+ // Act on the canonical path that was validated, not the raw string that was
609
+ // written: the delete below is only as safe as the path it is handed.
610
+ const candidateDir = managedReleasePath({ home, value: journal.candidate?.stageDir ?? null });
558
611
  const candidateManifest = candidateDir && existsSync(candidateDir) ? readManifest(candidateDir) : null;
559
612
 
560
613
  // A coordinated-update journal MUST carry a runtime segment; recovery may
@@ -1084,14 +1137,26 @@ const STALE_INCOMPLETE_MS = 24 * 60 * 60 * 1000;
1084
1137
  function gcOldReleases({ home, keep = KEEP_RELEASES, protect = null }) {
1085
1138
  const pointer = readCurrentPointer({ home });
1086
1139
  const releasesDir = crewReleasesDir({ home });
1087
- if (!existsSync(releasesDir)) return;
1140
+ if (!existsSync(releasesDir)) return [];
1088
1141
  const removed = [];
1089
1142
  // A running Hub keeps executing the release it started from and re-reads
1090
1143
  // several modules from disk on every request, so removing that release breaks
1091
1144
  // those routes with no recovery but a restart. Protect what live processes
1092
1145
  // claim, not just what the pointer names.
1093
- const claimed = liveReleaseClaims({ home }).map((dir) => resolve(dir));
1094
- const live = new Set([...(protect ? [resolve(protect)] : []), ...claimed]);
1146
+ //
1147
+ // Liveness has to be known, not merely unrefuted: if the claim directory
1148
+ // cannot be read, "no live claims" is an absence of evidence, and deleting on
1149
+ // it reproduces the outage this protection exists to prevent. Pruning is
1150
+ // optional and a skipped pass costs disk; guessing wrong costs a broken Hub.
1151
+ const claims = releaseClaimsState({ home });
1152
+ if (!claims.reliable) {
1153
+ // Say so rather than passing over it in silence: nothing is pruned until the
1154
+ // claim directory can be read again, and an operator who never hears about
1155
+ // that finds out when the disk fills.
1156
+ process.emitWarning(`dsh-crew: release liveness could not be read (${claims.error ?? 'unknown cause'}); skipping release pruning this pass`);
1157
+ return removed;
1158
+ }
1159
+ const live = new Set([...(protect ? [resolve(protect)] : []), ...claims.live.map((dir) => resolve(dir))]);
1095
1160
  const dirs = readdirSync(releasesDir)
1096
1161
  .map((name) => join(releasesDir, name))
1097
1162
  .filter((dir) => (!pointer || dir !== pointer.path) && !live.has(resolve(dir)));
@@ -74,12 +74,23 @@ export function capturePayloadContent(root, suppliedManifest) {
74
74
  } catch { return null; }
75
75
  }
76
76
 
77
- /** This digest covers first-party shipped files, not dependency tamper attestation. */
77
+ /**
78
+ * This digest covers first-party shipped files, not dependency tamper attestation.
79
+ *
80
+ * The permission bits are part of it: the copy below writes each file with the
81
+ * mode captured from the source, so a payload whose `0644` became `0755` with
82
+ * identical bytes is not the same payload — it is the case where an intended
83
+ * permission repair would otherwise be skipped as already done.
84
+ */
78
85
  export function payloadContentDigest(root) {
79
86
  const snapshot = capturePayloadContent(root);
80
87
  if (!snapshot) return null;
81
- const hashes = [...snapshot.files].map(([name, bytes]) => [name, createHash('sha256').update(bytes).digest('hex')]);
82
- return createHash('sha256').update(JSON.stringify(hashes.sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0))).digest('hex');
88
+ const entries = [...snapshot.files].map(([name, bytes]) => [
89
+ name,
90
+ snapshot.modes.get(name) ?? null,
91
+ createHash('sha256').update(bytes).digest('hex'),
92
+ ]);
93
+ return createHash('sha256').update(JSON.stringify(entries.sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0))).digest('hex');
83
94
  }
84
95
 
85
96
  export function samePayloadContent(sourceRoot, installedRoot) {
@@ -16,13 +16,21 @@
16
16
  // release with a live claim. Liveness is decided by the pid, not by the file, so
17
17
  // a process that dies without cleaning up cannot pin a release forever.
18
18
 
19
- import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
19
+ import { randomBytes } from 'node:crypto';
20
+ import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync } from 'node:fs';
20
21
  import { homedir } from 'node:os';
21
22
  import { dirname, join, resolve } from 'node:path';
22
23
  import { fileURLToPath } from 'node:url';
23
24
 
24
25
  export const IN_USE_DIRNAME = 'in-use';
25
26
 
27
+ // A claim file is `<pid>.json` or `<pid>-<nonce>.json`. The PID alone is not
28
+ // enough to identify a claim: two mounts inside one process would publish to the
29
+ // same pathname, and whichever disposed first would remove the other's
30
+ // protection. A claim is owned by the mount that wrote it, so it needs a name
31
+ // that says which one that was.
32
+ const CLAIM_NAME_RE = /^(\d+)(?:-[0-9a-f]+)?\.json$/;
33
+
26
34
  export function releaseInUseDir({ home = homedir() } = {}) {
27
35
  return join(home, '.config', 'dsh-crew', 'app', IN_USE_DIRNAME);
28
36
  }
@@ -55,8 +63,13 @@ function isAlive(pid) {
55
63
 
56
64
  /**
57
65
  * Record that this process is running `releasePath`. Best effort: failing to
58
- * claim must never stop the Hub from starting, because the cost of a missed
59
- * claim is a possible retention of one extra release.
66
+ * claim must never stop the Hub from starting.
67
+ *
68
+ * A null return is a real degradation, not a footnote. Retention cannot see a
69
+ * release that was never claimed, so an unclaimable release is one that a later
70
+ * update may delete from under this process — the 500-answering routes this
71
+ * module exists to prevent. Callers should surface it; `releaseClaimsState`
72
+ * gives retention the matching fail-closed signal.
60
73
  */
61
74
  export function claimReleaseInUse({ moduleUrl = import.meta.url, releasePath, home = homedir(), pid = process.pid, now = Date.now() } = {}) {
62
75
  try {
@@ -64,47 +77,122 @@ export function claimReleaseInUse({ moduleUrl = import.meta.url, releasePath, ho
64
77
  if (!target) return null;
65
78
  const dir = releaseInUseDir({ home });
66
79
  mkdirSync(dir, { recursive: true });
67
- const file = join(dir, `${pid}.json`);
68
- writeFileSync(file, JSON.stringify({ pid, release: resolve(target), claimed_at: now }) + '\n');
80
+ // The nonce makes this claim the property of this call. Clearing by PID would
81
+ // let one mount remove a sibling mount's only protection.
82
+ const nonce = randomBytes(8).toString('hex');
83
+ const file = join(dir, `${pid}-${nonce}.json`);
84
+ // Write then rename: a reader that lists the directory can otherwise see this
85
+ // filename while its JSON is still empty or half-written, and an unparseable
86
+ // claim is indistinguishable from a dead one.
87
+ const pending = `${file}.tmp`;
88
+ writeFileSync(pending, JSON.stringify({ pid, nonce, release: resolve(target), claimed_at: now }) + '\n');
89
+ renameSync(pending, file);
69
90
  return file;
70
91
  } catch { return null; }
71
92
  }
72
93
 
94
+ /** The unbranded per-PID claim path, as written by releases before claims were per-mount. */
73
95
  export function releaseClaimFile({ home = homedir(), pid = process.pid } = {}) {
74
96
  return join(releaseInUseDir({ home }), `${pid}.json`);
75
97
  }
76
98
 
77
99
  export function releaseClaimInUse({ home = homedir(), pid = process.pid } = {}) {
78
- const file = releaseClaimFile({ home, pid });
79
- return existsSync(file) || false;
100
+ let names;
101
+ try { names = readdirSync(releaseInUseDir({ home })); } catch { return false; }
102
+ return names.some((name) => CLAIM_NAME_RE.exec(name)?.[1] === String(pid));
103
+ }
104
+
105
+ /**
106
+ * Remove the claim whose handle `claimReleaseInUse` returned.
107
+ *
108
+ * A missing handle is not permission to remove something: without one there is
109
+ * no way to know which claim this caller owns, and guessing by PID can remove a
110
+ * sibling mount's protection. Legacy unbranded claims are removed by
111
+ * `clearLegacyReleaseClaim`, which says so in its name.
112
+ */
113
+ export function clearReleaseClaim({ file = null } = {}) {
114
+ if (!file) return false;
115
+ try { rmSync(file, { force: true }); return true; } catch { return false; }
80
116
  }
81
117
 
82
- export function clearReleaseClaim({ home = homedir(), pid = process.pid } = {}) {
118
+ /**
119
+ * Remove the unbranded `<pid>.json` claim written by releases before claims were
120
+ * per-mount. Separate from `clearReleaseClaim` because it can take a claim this
121
+ * caller did not write, and that should never happen by falling through a
122
+ * missing argument.
123
+ */
124
+ export function clearLegacyReleaseClaim({ home = homedir(), pid = process.pid } = {}) {
83
125
  try { rmSync(releaseClaimFile({ home, pid }), { force: true }); return true; } catch { return false; }
84
126
  }
85
127
 
86
128
  /**
87
- * Releases currently held by a live process. Dead claims are removed as they are
88
- * encountered, so the directory cannot accumulate stale files.
129
+ * Releases currently held by a live process, plus whether that answer is
130
+ * trustworthy.
131
+ *
132
+ * `reliable: false` means liveness could not be determined, so an empty `live`
133
+ * list is not evidence that nothing is running. A caller that deletes releases
134
+ * must treat unknown as "do not delete": reading an unreadable claim directory
135
+ * as "no claims" is exactly how a release disappears from under a running Hub,
136
+ * and the damage is to the process, not to the files.
137
+ *
138
+ * Nothing is deleted here. A claim is only ever a protection, so a reaper that
139
+ * unlinks one has to be certain the object it inspected is still the object it
140
+ * is removing — and it cannot be, because the pathname outlives the claim and a
141
+ * process restarting with a reused PID publishes a new file at the same one.
142
+ * Checking liveness and then unlinking is a check-then-act over a name, and
143
+ * losing that race deletes a live process's protection. Stale files therefore
144
+ * stay; a Hub clears its own claim on a clean shutdown, and one file per
145
+ * hard-killed process is not worth a race over a safety mechanism.
89
146
  */
90
- export function liveReleaseClaims({ home = homedir(), alive = isAlive } = {}) {
147
+ export function releaseClaimsState({ home = homedir(), alive = isAlive } = {}) {
91
148
  const dir = releaseInUseDir({ home });
92
149
  let names;
93
- // A state directory that is unreadable — or replaced by a file — must read as
94
- // "no claims" rather than throwing: this runs inside install, and failing it
95
- // would block an update over a problem that only affects pruning.
96
- try { names = readdirSync(dir); } catch { return []; }
150
+ try { names = readdirSync(dir); } catch (error) {
151
+ // A directory that does not exist yet reliably holds no claims; any other
152
+ // failure — permissions, a file where the directory should be — does not.
153
+ if (error?.code === 'ENOENT') return { live: [], reliable: true };
154
+ return { live: [], reliable: false, error: String(error?.message ?? error) };
155
+ }
97
156
  const live = [];
157
+ let unknown = null;
98
158
  for (const name of names) {
99
159
  if (!name.endsWith('.json')) continue;
100
160
  const file = join(dir, name);
101
- let record;
102
- try { record = JSON.parse(readFileSync(file, 'utf8')); } catch { record = null; }
103
- if (record && Number.isInteger(record.pid) && typeof record.release === 'string' && alive(record.pid)) {
104
- live.push(resolve(record.release));
161
+ // The filename carries the PID, and the name is complete before any content
162
+ // exists — so a torn write still says whose claim it was, which is what
163
+ // keeps one old unusable file from disabling pruning forever.
164
+ const pidFromName = CLAIM_NAME_RE.exec(name) ? Number.parseInt(name, 10) : null;
165
+
166
+ let parsed = null;
167
+ try { parsed = JSON.parse(readFileSync(file, 'utf8')); } catch { parsed = null; }
168
+
169
+ const wellFormed = parsed && typeof parsed === 'object'
170
+ && Number.isInteger(parsed.pid) && parsed.pid > 0
171
+ && typeof parsed.release === 'string' && parsed.release !== ''
172
+ && parsed.pid === pidFromName;
173
+
174
+ if (wellFormed) {
175
+ // A well-formed claim for a process that is gone simply protects nothing;
176
+ // it is inert, not a reason to distrust the rest.
177
+ if (alive(parsed.pid)) live.push(resolve(parsed.release));
105
178
  continue;
106
179
  }
107
- try { rmSync(file, { force: true }); } catch { /* best effort */ }
180
+
181
+ // Anything else is not evidence of a dead process. Syntax that happens to
182
+ // parse is not a schema: `{}` and `{"pid":123}` say nothing about liveness.
183
+ // The filename is the only identity left, and a claim whose named process is
184
+ // positively gone is inert for the same reason.
185
+ if (pidFromName !== null && !alive(pidFromName)) continue;
186
+ // No usable PID, or the PID may still be running. A reused PID is
187
+ // indistinguishable from the original here, so this stays unknown rather
188
+ // than guessing; that is a fail-closed condition, not a resolved one.
189
+ unknown ??= `unusable claim: ${file}`;
108
190
  }
109
- return [...new Set(live)];
191
+ if (unknown) return { live: [...new Set(live)], reliable: false, error: unknown };
192
+ return { live: [...new Set(live)], reliable: true };
193
+ }
194
+
195
+ /** Releases currently held by a live process. See `releaseClaimsState`. */
196
+ export function liveReleaseClaims(opts = {}) {
197
+ return releaseClaimsState(opts).live;
110
198
  }
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '1.10.0';
39
+ export const RUNTIME_VERSION = '1.10.2';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([