@ran-sh/dsh-crew 1.10.1 → 1.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "1.10.1",
3
+ "version": "1.10.3",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "1.10.1",
3
+ "version": "1.10.3",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -11,6 +11,7 @@ import {
11
11
  existsSync,
12
12
  lstatSync,
13
13
  mkdirSync,
14
+ readdirSync,
14
15
  readFileSync,
15
16
  realpathSync,
16
17
  renameSync,
@@ -528,9 +529,21 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
528
529
  return recovery;
529
530
  }
530
531
  }
531
- try { rmSync(liveRoot, { recursive: true, force: true }); } catch {}
532
532
  try {
533
- if (existsSync(prevRoot)) { rename(prevRoot, liveRoot); recovery.restore = true; }
533
+ rmSync(liveRoot, { recursive: true, force: true });
534
+ } catch (error) {
535
+ // A failed removal must stop the recovery. Swallowing it left the window
536
+ // open: the restore was attempted over whatever remained, and
537
+ // `startOwned()` ran regardless — so a tree that could not be removed was
538
+ // then asked to start again, which on Windows is exactly the case where
539
+ // live handles blocked the removal.
540
+ recovery.removeError = String(error?.message ?? error);
541
+ recovery.ok = false;
542
+ return recovery;
543
+ }
544
+ try {
545
+ if (!existsSync(prevRoot)) recovery.restoreError = `previous runtime is missing at ${prevRoot}`;
546
+ else { rename(prevRoot, liveRoot); recovery.restore = true; }
534
547
  } catch (error) {
535
548
  recovery.restoreError = String(error?.message ?? error);
536
549
  }
@@ -538,6 +551,14 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
538
551
  recovery.ok = recovery.restore === true;
539
552
  return recovery;
540
553
  }
554
+ // Restart only on positive proof that the prior runtime is the tree now at the
555
+ // live root. Starting without it would run the failed candidate again, or
556
+ // nothing at all, and report it as a recovery.
557
+ if (recovery.restore !== true || !existsSync(liveRoot)) {
558
+ recovery.restartSkipped = 'the previous runtime was not restored';
559
+ recovery.ok = false;
560
+ return recovery;
561
+ }
541
562
  try {
542
563
  const restarted = await startOwned();
543
564
  recovery.restart = restarted?.ok === true;
@@ -551,6 +572,8 @@ async function rollbackRuntimeSwap({ liveRoot, prevRoot, stopOwned, startOwned,
551
572
 
552
573
  // ---- retained runtime cohorts -------------------------------------------------
553
574
 
575
+ const DISPLACED_RUNTIME_PREFIX = 'runtime-displaced-';
576
+
554
577
  function retainedRuntimesRoot({ home }) {
555
578
  return join(crewDshHome({ home }), RETAINED_RUNTIMES_DIRNAME);
556
579
  }
@@ -572,12 +595,22 @@ function runtimeTreeVersion(root, read = readFileSync) {
572
595
  // Best-effort retention of the swapped-out prior runtime tree. Never throws
573
596
  // and never fails the caller: retention is an optimization for offline
574
597
  // rollback, not a correctness requirement of the migration itself.
575
- function retainPriorRuntime({ home, prevRoot, rename = renameSync }) {
598
+ /**
599
+ * Retain a parked runtime tree under the version it ships.
600
+ *
601
+ * `removeUnreadable` exists for one caller: the rollback rotation, where the
602
+ * parked tree is the ONLY copy of the cohort just displaced. There, failing to
603
+ * read its version is not a reason to delete it — that would destroy the very
604
+ * copy the rotation exists to keep — so it stays parked and is found later by
605
+ * `findRetainedRuntime`. The forward migration path keeps the default, where an
606
+ * unreadable tree really is junk.
607
+ */
608
+ function retainPriorRuntime({ home, prevRoot, rename = renameSync, removeUnreadable = true }) {
576
609
  try {
577
610
  if (!existsSync(prevRoot)) return { ok: true, retained: false };
578
611
  const version = runtimeTreeVersion(prevRoot);
579
612
  if (!version) {
580
- // No version to key retention on; the tree is unreadable junk.
613
+ if (!removeUnreadable) return { ok: false, retained: false, prevRoot, error: 'parked cohort version unreadable; left in place' };
581
614
  try { rmSync(prevRoot, { recursive: true, force: true }); } catch {}
582
615
  return { ok: true, retained: false, reason: 'prior tree version unreadable; removed' };
583
616
  }
@@ -601,9 +634,24 @@ function retainPriorRuntime({ home, prevRoot, rename = renameSync }) {
601
634
  export function findRetainedRuntime({ home = homedir(), version, exists = existsSync, read = readFileSync } = {}) {
602
635
  if (typeof version !== 'string' || version.length === 0) return null;
603
636
  const dir = retainedRuntimeDir({ home, version });
604
- if (!exists(dir)) return null;
605
- if (runtimeTreeVersion(dir, read) !== version) return null;
606
- return dir;
637
+ if (exists(dir) && runtimeTreeVersion(dir, read) === version) return dir;
638
+ // A rotation whose retain step failed leaves the displaced cohort parked in the
639
+ // harness home. It is a complete, valid copy of that cohort, so it is found
640
+ // here rather than being written off: otherwise a later offline rollback to
641
+ // that version reports RETAINED_MISSING while the tree sits on disk.
642
+ return findDisplacedRuntime({ home, version, exists, read });
643
+ }
644
+
645
+ function findDisplacedRuntime({ home, version, exists = existsSync, read = readFileSync }) {
646
+ let names;
647
+ try { names = readdirSync(crewDshHome({ home })); } catch { return null; }
648
+ for (const name of names) {
649
+ if (!name.startsWith(DISPLACED_RUNTIME_PREFIX)) continue;
650
+ const dir = join(crewDshHome({ home }), name);
651
+ if (!exists(dir)) continue;
652
+ if (runtimeTreeVersion(dir, read) === version) return dir;
653
+ }
654
+ return null;
607
655
  }
608
656
 
609
657
  // Offline cohort restore for cross-cohort rollback: move the retained tree
@@ -642,14 +690,34 @@ export async function restoreRetainedRuntime({
642
690
  if (!stop.ok) {
643
691
  return { ok: false, code: stop.code ?? 'DSH_RUNTIME_STOP_FAILED', error: stop.error ?? 'could not stop owned 3210' };
644
692
  }
693
+ // Park the displaced cohort instead of deleting it. Deleting it was what made a
694
+ // second rollback impossible: after B→A the B tree was gone, so a later A→B
695
+ // found no retained B. Retention has to rotate, like the forward path does.
696
+ const parked = join(crewDshHome({ home }), `${DISPLACED_RUNTIME_PREFIX}${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`);
697
+ let parkedLive = false;
698
+ try {
699
+ if (existsSync(liveRoot)) { rename(liveRoot, parked); parkedLive = true; }
700
+ } catch (error) {
701
+ return { ok: false, code: 'DSH_RUNTIME_PARK_FAILED', error: String(error?.message ?? error) };
702
+ }
645
703
  try {
646
- if (existsSync(liveRoot)) rmSync(liveRoot, { recursive: true, force: true });
647
704
  rename(retained, liveRoot);
648
705
  } catch (error) {
649
- // Restore the pre-existing live tree is impossible (it was replaced only
650
- // on success above); report and let the caller reconcile.
706
+ // Put the displaced tree back: leaving the live root empty because a rename
707
+ // failed would take a working runtime down for nothing.
708
+ try { if (parkedLive && !existsSync(liveRoot)) rename(parked, liveRoot); } catch { /* reported below */ }
651
709
  return { ok: false, code: 'DSH_RUNTIME_RESTORE_SWAP_FAILED', error: String(error?.message ?? error) };
652
710
  }
711
+ // The cohort just displaced becomes the retained one, so the version this
712
+ // restore moved away from can still be rolled back to. This parked tree is the
713
+ // only copy of that cohort, so an unreadable version keeps it rather than
714
+ // deleting it, and a failure is surfaced instead of passing as a clean restore.
715
+ const rotated = parkedLive
716
+ ? retainPriorRuntime({ home, prevRoot: parked, rename, removeUnreadable: false })
717
+ : { ok: true, retained: false };
718
+ if (parkedLive && rotated.ok === false) {
719
+ log(`! could not retain the displaced runtime cohort; it remains at ${parked}: ${rotated.error ?? ''}`);
720
+ }
653
721
  if (prepareOnly) {
654
722
  log(`- runtime tree prepared offline from retained tree (@${version}); process not started`);
655
723
  return { ok: true, version, liveRoot, prepared: true };
@@ -161,21 +161,62 @@ export function readCurrentPointerState({ home = homedir() } = {}) {
161
161
  return { status: 'malformed', file, code: 'POINTER_FIELDS_INVALID' };
162
162
  }
163
163
  if (!isAbsolute(raw.path)) return { status: 'malformed', file, code: 'POINTER_PATH_NOT_ABSOLUTE' };
164
- return { status: 'valid', file, pointer: raw };
164
+ // The pointer names the release that is live, and recovery acts on it. A path
165
+ // outside the managed releases directory is a corrupt or tampered pointer, not
166
+ // a release, so it fails closed exactly like the other malformed cases — and
167
+ // the canonical value is what the caller gets, so every later use acts on the
168
+ // path that was checked.
169
+ const canonical = managedReleasePath({ home, value: raw.path });
170
+ if (canonical === null) {
171
+ return { status: 'malformed', file, code: 'POINTER_PATH_OUTSIDE_RELEASES' };
172
+ }
173
+ return { status: 'valid', file, pointer: { ...raw, path: canonical } };
174
+ }
175
+
176
+ /**
177
+ * Resolve a path that a journal, the pointer, or recovery is about to act on, and
178
+ * refuse it unless it is inside the Crew-owned releases directory.
179
+ *
180
+ * An absolute path is not enough. Recovery recursively deletes the journal's
181
+ * candidate stage directory and moves the pointer's release, so a syntactically
182
+ * valid but corrupt or tampered journal naming any absolute path would grant
183
+ * delete authority over it — including the official `~/.dsh` tree this plugin is
184
+ * required never to touch. Containment is therefore checked before any read,
185
+ * write, delete or activation, and a path that fails it is treated exactly like
186
+ * a malformed journal: nothing is touched and the state is retained.
187
+ */
188
+ export function managedReleasePath({ home = homedir(), value } = {}) {
189
+ if (typeof value !== 'string' || value.length === 0 || !isAbsolute(value)) return null;
190
+ const releases = realpathOr(resolve(crewReleasesDir({ home })));
191
+ const candidate = realpathOr(resolve(value));
192
+ // Resolve both sides before comparing: `..` segments and symlinks must not be
193
+ // able to leave the managed directory after the check.
194
+ const relative = relativeTo(releases, candidate);
195
+ if (relative === null || relative === '' || relative.startsWith('..') || isAbsolute(relative)) return null;
196
+ return candidate;
197
+ }
198
+
199
+ function realpathOr(path) {
200
+ try { return realpathSync(path); } catch { return path; }
201
+ }
202
+
203
+ /** `path` expressed relative to `from`, or null when they are on different roots. */
204
+ function relativeTo(from, path) {
205
+ return relative(from, path);
165
206
  }
166
207
 
167
- function validJournalRelease(value) {
208
+ function validJournalRelease({ home, value }) {
168
209
  return !!value && typeof value === 'object'
169
210
  && typeof value.name === 'string' && value.name.length > 0
170
211
  && typeof value.version === 'string' && value.version.length > 0
171
- && typeof value.path === 'string' && isAbsolute(value.path);
212
+ && managedReleasePath({ home, value: value.path }) !== null;
172
213
  }
173
214
 
174
- function validJournalCandidate(value) {
215
+ function validJournalCandidate({ home, value }) {
175
216
  return !!value && typeof value === 'object'
176
217
  && typeof value.name === 'string' && value.name.length > 0
177
218
  && typeof value.version === 'string' && value.version.length > 0
178
- && typeof value.stageDir === 'string' && isAbsolute(value.stageDir);
219
+ && managedReleasePath({ home, value: value.stageDir }) !== null;
179
220
  }
180
221
 
181
222
  function writeFileAtomic(file, content) {
@@ -359,13 +400,23 @@ function readUpdateJournal({ home = homedir() } = {}) {
359
400
  return { malformed: true, file };
360
401
  }
361
402
  // Full schema check: a semantically broken journal (null candidate,
362
- // incomplete prior) must fail closed, never enter recovery.
363
- if (!validJournalCandidate(raw.candidate)) {
403
+ // incomplete prior) must fail closed, never enter recovery. The paths are
404
+ // checked for containment too, because recovery deletes and moves them, and
405
+ // they are replaced with the canonical value that was checked so that every
406
+ // later use acts on the path that was validated rather than the string that
407
+ // happened to be written.
408
+ const canonicalCandidate = managedReleasePath({ home, value: raw.candidate?.stageDir });
409
+ if (!validJournalCandidate({ home, value: raw.candidate }) || canonicalCandidate === null) {
364
410
  return { malformed: true, file, code: 'JOURNAL_CANDIDATE_SCHEMA_INVALID' };
365
411
  }
366
- if (raw.prior !== null && raw.prior !== undefined && !validJournalRelease(raw.prior)) {
367
- return { malformed: true, file, code: 'JOURNAL_PRIOR_SCHEMA_INVALID' };
412
+ if (raw.prior !== null && raw.prior !== undefined) {
413
+ const canonicalPrior = managedReleasePath({ home, value: raw.prior?.path });
414
+ if (!validJournalRelease({ home, value: raw.prior }) || canonicalPrior === null) {
415
+ return { malformed: true, file, code: 'JOURNAL_PRIOR_SCHEMA_INVALID' };
416
+ }
417
+ raw = { ...raw, prior: { ...raw.prior, path: canonicalPrior } };
368
418
  }
419
+ raw = { ...raw, candidate: { ...raw.candidate, stageDir: canonicalCandidate } };
369
420
  return raw;
370
421
  }
371
422
 
@@ -536,12 +587,90 @@ function validateJournalRuntimeSegment({ home, rt }) {
536
587
  return { ok: true };
537
588
  }
538
589
 
590
+ /**
591
+ * The runtime window a coordinated update left open, or null.
592
+ *
593
+ * The durable maintenance session is the source of truth, not the journal. The
594
+ * supervisor persists the session as part of stopping 3210, so it exists from the
595
+ * instant the stop lands — whereas the journal is written a few instructions
596
+ * later. Reading the session closes that gap: whatever stop happened, recovery
597
+ * finds it. The journal's copy is kept as a fallback for a session that has since
598
+ * been cleared.
599
+ */
600
+ async function openMaintenanceWindow({ home, journal }) {
601
+ const appRoot = join(home, '.config', 'dsh-crew');
602
+ try {
603
+ const { readMaintenanceSession } = await import('../supervisor/restart-request.mjs');
604
+ const durable = readMaintenanceSession(appRoot);
605
+ if (durable.ok && durable.state === 'present') {
606
+ return { lease: durable.session.lease, runtime_id: durable.session.runtime_id, source: 'session' };
607
+ }
608
+ // A session that exists but cannot be read is not "no session": recovery
609
+ // must not clear the journal and report success while a stopped runtime it
610
+ // cannot account for is sitting there.
611
+ if (!durable.ok && durable.state === 'malformed') {
612
+ return { malformed: true, error: durable.code ?? 'MAINTENANCE_SESSION_MALFORMED' };
613
+ }
614
+ } catch { /* fall through to the journal's copy */ }
615
+ const recorded = journal?.runtime?.maintenance ?? null;
616
+ if (recorded?.lease && recorded?.runtime_id) return { lease: recorded.lease, runtime_id: recorded.runtime_id, source: 'journal' };
617
+ return null;
618
+ }
619
+
620
+ /**
621
+ * Restart a runtime that a coordinated update left deliberately stopped.
622
+ *
623
+ * The window is the lease and runtime id the supervisor issued when it stopped
624
+ * 3210. Passing them back resumes that exact stopped window rather than asking
625
+ * for a new stop, which is the only thing the durable maintenance session
626
+ * accepts — a second independent stop is refused while the first is in force.
627
+ */
628
+ async function closeMaintenanceWindow({ home, window, supervisorFactory = crewSupervisor, log = () => {} }) {
629
+ try {
630
+ const supervisor = supervisorFactory({ home });
631
+ const started = await supervisor.startOwnedBackend({ lease: window.lease, runtimeId: window.runtime_id });
632
+ if (started?.ok !== true) {
633
+ return { ok: false, code: started?.code ?? 'MAINTENANCE_START_FAILED', error: started?.error ?? null };
634
+ }
635
+ log('- recovery restarted the runtime that the interrupted update had stopped');
636
+ return { ok: true };
637
+ } catch (error) {
638
+ return { ok: false, code: 'MAINTENANCE_START_FAILED', error: String(error?.message ?? error) };
639
+ }
640
+ }
641
+
642
+ /**
643
+ * Attempt to stop the owned runtime, and report whether the stop actually
644
+ * happened.
645
+ *
646
+ * Only a successful stop counts. `MAINTENANCE_IDENTITY_UNAVAILABLE` means the
647
+ * supervisor could not obtain the live runtime's identity over its HTTP endpoint,
648
+ * which is equally consistent with a live runtime whose request timed out — so it
649
+ * is not evidence that nothing is running and must never be read as any. An
650
+ * earlier revision of this helper did read it that way and would have replaced a
651
+ * live runtime's tree, which is the exact hazard it exists to prevent.
652
+ */
653
+ async function ensureRuntimeStopped({ home, supervisorFactory = crewSupervisor, log = () => {} }) {
654
+ try {
655
+ const supervisor = supervisorFactory({ home });
656
+ if (typeof supervisor?.stopOwnedBackend !== 'function') return { ok: false, code: 'MAINTENANCE_UNAVAILABLE' };
657
+ const stopped = await supervisor.stopOwnedBackend();
658
+ if (stopped?.ok === true) {
659
+ log('- recovery stopped the runtime before replacing its tree');
660
+ return { ok: true, stopped: true };
661
+ }
662
+ return { ok: false, code: stopped?.code ?? 'MAINTENANCE_STOP_FAILED', error: stopped?.error ?? null };
663
+ } catch (error) {
664
+ return { ok: false, code: 'MAINTENANCE_STOP_FAILED', error: String(error?.message ?? error) };
665
+ }
666
+ }
667
+
539
668
  // Reconcile a leftover journal from a crashed update/install. The single
540
669
  // commit point is the pointer write: pointer == candidate means committed
541
670
  // (finalize, do NOT roll back); pointer == prior/absent means pre-commit
542
671
  // (restore activation surfaces, drop candidate). A malformed journal fails
543
672
  // closed and is retained for operator inspection.
544
- export function reconcileUpdateJournal({ home = homedir(), log = () => {}, installer = realInstaller } = {}) {
673
+ export async function reconcileUpdateJournal({ home = homedir(), log = () => {}, installer = realInstaller, supervisorFactory = crewSupervisor } = {}) {
545
674
  const journal = readUpdateJournal({ home });
546
675
  if (!journal) return { ok: true, reconciled: false };
547
676
  if (journal.malformed) {
@@ -554,7 +683,9 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
554
683
  return { ok: false, code: 'POINTER_MALFORMED', file: pointerState.file, error: pointerState.code ?? 'pointer unreadable; refusing recovery' };
555
684
  }
556
685
  const pointer = pointerState.status === 'valid' ? pointerState.pointer : null;
557
- const candidateDir = journal.candidate?.stageDir ?? null;
686
+ // Act on the canonical path that was validated, not the raw string that was
687
+ // written: the delete below is only as safe as the path it is handed.
688
+ const candidateDir = managedReleasePath({ home, value: journal.candidate?.stageDir ?? null });
558
689
  const candidateManifest = candidateDir && existsSync(candidateDir) ? readManifest(candidateDir) : null;
559
690
 
560
691
  // A coordinated-update journal MUST carry a runtime segment; recovery may
@@ -633,6 +764,36 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
633
764
  return { ok: false, code: 'JOURNAL_POINTER_DIVERGED', stage: journal.stage, error: `pointer references unexpected release ${pointer?.path ?? 'unknown'}; refusing recovery` };
634
765
  }
635
766
 
767
+ // A verified coordinated journal means the candidate runtime was started and
768
+ // its identity checked, and the pointer write happens after that — so this
769
+ // state means the pointer write was lost and the candidate is what is running.
770
+ // Completing the commit is the recovery; rolling back would swap the tree and
771
+ // the payload out from under a live process. This runs only after the
772
+ // divergence check above, so "the pointer has not moved" means it still names
773
+ // the prior release or nothing at all — a third release has already failed
774
+ // closed, and its release identity can never be overwritten by a stale journal.
775
+ if (journal.stage === 'coordinated-update' && journal.verified === true && !pointerMatchesCandidate) {
776
+ if (!candidateIdent?.name || !candidateIdent?.version || !candidateDir) {
777
+ return { ok: false, code: 'JOURNAL_CANDIDATE_INVALID', stage: journal.stage };
778
+ }
779
+ const rt = journal.runtime ?? null;
780
+ if (rt?.candidateVersion && rt?.liveRoot && readRuntimeTreeVersionSync(rt.liveRoot) !== rt.candidateVersion) {
781
+ return { ok: false, code: 'JOURNAL_COORDINATED_RUNTIME_MISMATCH', stage: journal.stage, error: 'a verified coordinated journal expects the candidate runtime to be live; refusing to commit' };
782
+ }
783
+ // The same check the committed path applies, so a roll-forward cannot install
784
+ // a pointer to a payload that would have been rejected had it arrived the
785
+ // ordinary way.
786
+ const validated = validateInstalledPayload(candidateDir, { expectedName: candidateIdent.name, expectedVersion: candidateIdent.version });
787
+ if (!validated.ok) {
788
+ return { ok: false, code: 'JOURNAL_CANDIDATE_INVALID', stage: journal.stage, error: (validated.errors ?? []).join('; ') };
789
+ }
790
+ writeCurrentPointer({ home, name: candidateIdent.name, version: candidateIdent.version, path: candidateDir });
791
+ clearUpdateJournal({ home });
792
+ gcOldReleases({ home, protect: journal.prior?.path ?? null });
793
+ log(`- recovered update journal at stage ${journal.stage}: candidate ${candidateIdent.version} was verified and started, so the commit was completed`);
794
+ return { ok: true, reconciled: true, stage: journal.stage, committed: true };
795
+ }
796
+
636
797
  // Pre-commit side: prior stays authoritative. Re-point live activation
637
798
  // surfaces back at prior (a crash between activation and pointer write
638
799
  // leaves them on the candidate), then drop the candidate. A rollback
@@ -650,6 +811,29 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
650
811
  if (rt?.priorVersion && rt?.liveRoot) {
651
812
  const liveVersion = readRuntimeTreeVersionSync(rt.liveRoot);
652
813
  if (liveVersion !== rt.priorVersion) {
814
+ // The tree has to be replaced, and replacing it under a running process is
815
+ // the damage. The journal says whether a start can have happened: it is
816
+ // set to `starting` before the candidate is started and to `verified`
817
+ // after it checks out, so anything else means no runtime was ever started
818
+ // from the candidate cohort and the swap is safe.
819
+ //
820
+ // `starting` cannot be resolved from the journal alone — the candidate may
821
+ // be live and unverified. It must not be a dead end either, so recovery
822
+ // stops the runtime and rolls back on a stop that actually happened, or on
823
+ // a durable STOPPED session that already proves one did. Anything less is
824
+ // not proof: a runtime that cannot be shown to be stopped keeps its tree.
825
+ if (rt.state === 'starting') {
826
+ const window = await openMaintenanceWindow({ home, journal });
827
+ if (window?.malformed) {
828
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_SESSION_MALFORMED', stage: journal.stage, error: `${window.error}; refusing to replace a runtime tree while a stopped runtime cannot be accounted for` };
829
+ }
830
+ if (!window) {
831
+ const stopped = await ensureRuntimeStopped({ home, supervisorFactory, log });
832
+ if (!stopped.ok) {
833
+ return { ok: false, code: 'JOURNAL_RUNTIME_STOP_UNPROVEN', stage: journal.stage, error: `an interrupted update may have started the candidate runtime and could not be shown to be stopped (${stopped.code ?? 'unknown'}), so its tree was left alone; stop the owned runtime through the Crew supervisor, then re-run` };
834
+ }
835
+ }
836
+ }
653
837
  // Move the parked prior tree back onto liveRoot. This is a pure
654
838
  // directory swap: no network, no registry, safe under a stale lock.
655
839
  const parked = rt.priorRoot && existsSync(rt.priorRoot) ? rt.priorRoot : null;
@@ -694,29 +878,46 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
694
878
  // and may be removed. (rollback candidates are retained releases.)
695
879
  try { rmSync(candidateDir, { recursive: true, force: true }); } catch {}
696
880
  }
881
+ // The prior release is back in place, so a runtime the interrupted update
882
+ // stopped has to be started again — otherwise recovery reports success and
883
+ // leaves 3210 down. The journal is kept until the window is closed, so a
884
+ // crash between the two retries with the same recorded window instead of
885
+ // losing track of a stopped runtime.
886
+ const openWindow = journal.stage === 'coordinated-update'
887
+ ? await openMaintenanceWindow({ home, journal })
888
+ : null;
889
+ if (openWindow?.malformed) {
890
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_SESSION_MALFORMED', stage: journal.stage, error: `${openWindow.error}; refusing to report recovery while a stopped runtime cannot be accounted for` };
891
+ }
892
+ if (openWindow) {
893
+ const closed = await closeMaintenanceWindow({ home, window: openWindow, supervisorFactory, log });
894
+ if (!closed.ok) {
895
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_WINDOW_OPEN', stage: journal.stage, error: `the runtime was left stopped and could not be restarted (${closed.code ?? 'unknown'}); journal retained` };
896
+ }
897
+ }
697
898
  clearUpdateJournal({ home });
698
899
  log(`- recovered update journal at stage ${journal.stage}: restored prior release ${journal.prior.version}`);
699
900
  return { ok: true, reconciled: true, stage: journal.stage, committed: false };
700
901
  }
701
902
 
702
- // First-install pre-commit: no prior exists. Undo the candidate's
703
- // activation surfaces FIRST (a crash between activation and pointer
704
- // write leaves the profile link on the candidate), then remove the
705
- // orphan candidate pointer + dir. Journal clears only after successful
706
- // compensation.
707
- const undone = undoCandidateActivationSync({ home, candidateDir, candidateName: journal.candidate?.name ?? null });
708
- if (!undone.ok) {
709
- return { ok: false, code: 'JOURNAL_UNDO_FAILED', stage: journal.stage, error: undone.error ?? undone.code };
710
- }
711
- if (candidateDir && existsSync(candidateDir)) {
712
- try { rmSync(candidateDir, { recursive: true, force: true }); } catch {}
713
- }
714
- if (pointer && candidateDir && pointer.path === candidateDir) {
715
- try { rmSync(currentPointerFile({ home }), { force: true }); } catch {}
716
- }
717
- clearUpdateJournal({ home });
718
- log(`- recovered first-install journal at stage ${journal.stage}: removed orphan candidate`);
719
- return { ok: true, reconciled: true, stage: journal.stage, committed: false };
903
+ // First-install pre-commit: no prior release exists, and there is nothing this
904
+ // branch can undo that it can prove is its own. Removing the loader link would
905
+ // break every host integration that names it; removing the integration records
906
+ // would mean deleting files in the operator's home on the strength of a
907
+ // filename, because those uninstallers take only `home` and cannot tell this
908
+ // failed install's records from ones the operator wrote, or from a manual
909
+ // installation that predates Crew. So nothing is changed: the candidate stays,
910
+ // the link still resolves to it, whatever names the link still resolves, and
911
+ // the journal is retained so an operator decides. A recovery that cannot
912
+ // attribute what it would delete does not delete it.
913
+ return {
914
+ ok: false,
915
+ code: 'JOURNAL_FIRST_INSTALL_NEEDS_OPERATOR',
916
+ stage: journal.stage,
917
+ candidate: candidateDir,
918
+ journal: updateJournalFile({ home }),
919
+ error: `a first install did not complete; nothing was changed, because Crew cannot prove which host records are its own. To finish by hand: repoint or remove the Crew profile registration first (its loader link is ${join(crewProfileDir({ home }), 'node_modules', ...'@ran-sh/dsh-crew'.split('/'))}), remove the host integrations that name it, then delete ${updateJournalFile({ home })} and ${candidateDir ?? 'the staged candidate'}`,
920
+ };
720
921
  }
721
922
 
722
923
  // ---- dependency tree materialization ----------------------------------------
@@ -1797,7 +1998,7 @@ async function npxRollbackInner({ home, version: targetVersion, log, installer,
1797
1998
  // Journal-aware entry: refuse to overwrite a journal left by a crashed
1798
1999
  // transaction. Reconcile it first (or fail closed) instead of silently
1799
2000
  // replacing it with the rollback intent.
1800
- const pending = reconcileUpdateJournal({ home, log });
2001
+ const pending = await reconcileUpdateJournal({ home, log, supervisorFactory });
1801
2002
  if (!pending.ok) return { ok: false, error: `refusing rollback with unreconciled journal (${pending.code ?? 'unknown'})` };
1802
2003
  const current = readCurrentPointer({ home });
1803
2004
  if (!current?.path || !existsSync(current.path)) return { ok: false, error: 'no active Crew payload to roll back' };
@@ -2125,7 +2326,7 @@ export async function performCoordinatedCohortUpdate({
2125
2326
  // Journal the full coordinated intent BEFORE any destructive step. The
2126
2327
  // journal carries prior+candidate dshVersion plus runtime roots so a crash
2127
2328
  // at any point is recoverable by reconcileUpdateJournal.
2128
- writeUpdateJournal({
2329
+ const journalBase = {
2129
2330
  home,
2130
2331
  stage: 'coordinated-update',
2131
2332
  prior: { name: prior.name, version: prior.version, path: prior.path, dshVersion: priorDshVersion },
@@ -2138,7 +2339,8 @@ export async function performCoordinatedCohortUpdate({
2138
2339
  candidateVersion: candidateDshVersion,
2139
2340
  retainedRoot,
2140
2341
  },
2141
- });
2342
+ };
2343
+ writeUpdateJournal(journalBase);
2142
2344
 
2143
2345
  const switchPointer = (release) => writeCurrentPointer({ home, name: release.name, version: release.version, path: release.path });
2144
2346
  const priorRelease = { name: prior.name, version: prior.version, path: prior.path };
@@ -2193,6 +2395,16 @@ export async function performCoordinatedCohortUpdate({
2193
2395
  clearUpdateJournal({ home });
2194
2396
  return { ok: false, code: stopped.code ?? 'COORDINATED_STOP_FAILED', error: stopped.error ?? 'could not stop owned 3210' };
2195
2397
  }
2398
+ // The runtime is now deliberately stopped inside a durable maintenance
2399
+ // window, and everything from here on mutates its tree. Record the window in
2400
+ // the journal before that mutation: a crash in the next few steps used to
2401
+ // leave 3210 down with no record of how to resume it, and the surviving
2402
+ // session then refused the next ordinary stop, so the machine could not
2403
+ // recover on its own. `startOwnedBackend` accepts this exact lease and
2404
+ // runtime id, which is what recovery needs to close the window.
2405
+ if (stopped.lease && stopped.runtime_id) {
2406
+ writeUpdateJournal({ ...journalBase, runtime: { ...journalBase.runtime, maintenance: { lease: stopped.lease, runtime_id: stopped.runtime_id } } });
2407
+ }
2196
2408
  let liveMoved = false;
2197
2409
  try {
2198
2410
  // Move live runtime aside, retaining its tree for offline rollback.
@@ -2229,6 +2441,12 @@ export async function performCoordinatedCohortUpdate({
2229
2441
  const comp = await compensate();
2230
2442
  return finalizeCompensationFailure({ home, code: null, error: 'candidate activation failed', comp });
2231
2443
  }
2444
+ // Record that a start may be about to happen BEFORE it happens. Recovery
2445
+ // decides whether the runtime tree can be replaced by reading this, and a
2446
+ // crash between the start and the verification would otherwise look exactly
2447
+ // like "the candidate was never started" — which is how a rollback ends up
2448
+ // swapping the tree out from under a running process.
2449
+ writeUpdateJournal({ ...journalBase, runtime: { ...journalBase.runtime, state: 'starting' } });
2232
2450
  const started = await startFn();
2233
2451
  if (started?.ok !== true) {
2234
2452
  const comp = await compensate();
@@ -2254,9 +2472,12 @@ export async function performCoordinatedCohortUpdate({
2254
2472
  }
2255
2473
  }
2256
2474
 
2257
- // Mark the journal verified BEFORE the pointer write. Crash recovery
2258
- // then has a clean WAL relation: pointer==prior means not committed,
2259
- // pointer==candidate with verified journal means committed.
2475
+ // Mark the journal verified BEFORE the pointer write, which makes a crash
2476
+ // between the two distinguishable — but note what "verified" means here: the
2477
+ // candidate runtime has already been started and its identity checked, so the
2478
+ // candidate is what is running. A pointer still naming prior with a verified
2479
+ // journal therefore means the pointer write was lost, and recovery completes
2480
+ // the commit rather than rolling the tree out from under the live process.
2260
2481
  const marked = markJournalVerified({
2261
2482
  home,
2262
2483
  stage: 'coordinated-update',
@@ -2360,7 +2581,16 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2360
2581
  }
2361
2582
  log(`✓ Harness plugin registered (dedicated dsh-crew profile → ${releaseDir})`);
2362
2583
 
2363
- const codex = installer.installCodex({ home, root: releaseDir });
2584
+ // The host integrations are pointed at the profile's loader link, not at the
2585
+ // release directory. That link is what registration just re-pointed, so it
2586
+ // always resolves to whichever release is live: an upgrade re-points it and
2587
+ // the integrations keep working, and removing an old release can no longer
2588
+ // leave four configurations naming a directory that is gone. Writing the
2589
+ // release path directly is what made crash recovery unable to delete a
2590
+ // candidate without breaking Codex, ZCode and Claude Code.
2591
+ const integrationRoot = registration.linkPath ?? releaseDir;
2592
+
2593
+ const codex = installer.installCodex({ home, root: integrationRoot });
2364
2594
  if (codex.ok === false) {
2365
2595
  log(`✗ Codex Desktop integration failed: ${(codex.actions ?? []).join('; ')}`);
2366
2596
  return false;
@@ -2368,7 +2598,7 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2368
2598
  log('✓ Codex Desktop integration');
2369
2599
 
2370
2600
  if (installer.installZCode) {
2371
- const zcode = installer.installZCode({ home, root: releaseDir });
2601
+ const zcode = installer.installZCode({ home, root: integrationRoot });
2372
2602
  if (zcode.ok === false) {
2373
2603
  log(`✗ ZCode integration failed (${zcode.code ?? 'unknown'})`);
2374
2604
  return false;
@@ -2383,7 +2613,7 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2383
2613
  }
2384
2614
  if (startup?.supported) log('✓ Windows login startup');
2385
2615
 
2386
- const claude = await installer.installClaudeCode({ home, root: releaseDir });
2616
+ const claude = await installer.installClaudeCode({ home, root: integrationRoot });
2387
2617
  if (claude.ok === false) {
2388
2618
  log(`✗ Claude Code integration failed`);
2389
2619
  return false;
@@ -2449,7 +2679,7 @@ export async function npxInstall({
2449
2679
  }
2450
2680
 
2451
2681
  async function npxInstallInner({ home, log, sourceRoot, installer, ensureRuntime, npmInstaller }) {
2452
- const reconciled = reconcileUpdateJournal({ home, log });
2682
+ const reconciled = await reconcileUpdateJournal({ home, log });
2453
2683
  if (!reconciled.ok) return { ok: false, error: `update journal recovery failed (${reconciled.code ?? 'unknown'})` };
2454
2684
  const candidateRoot = sourceRoot ?? runningPackageRoot();
2455
2685
  const manifest = readManifest(candidateRoot);
@@ -2666,7 +2896,7 @@ export async function npxUpdate({
2666
2896
  }
2667
2897
 
2668
2898
  async function npxUpdateInner({ home, log, sourceRoot, candidate, spec, installer, ensureRuntime, npmInstaller, runner = spawnSync }) {
2669
- const journalRecovery = reconcileUpdateJournal({ home, log });
2899
+ const journalRecovery = await reconcileUpdateJournal({ home, log });
2670
2900
  if (!journalRecovery.ok) return { ok: false, error: `update journal recovery failed (${journalRecovery.code ?? 'unknown'})` };
2671
2901
  // Candidate resolution: explicit path/dir override > a newer validated
2672
2902
  // running launcher > configured npm registry (@latest). This makes the
@@ -74,12 +74,23 @@ export function capturePayloadContent(root, suppliedManifest) {
74
74
  } catch { return null; }
75
75
  }
76
76
 
77
- /** This digest covers first-party shipped files, not dependency tamper attestation. */
77
+ /**
78
+ * This digest covers first-party shipped files, not dependency tamper attestation.
79
+ *
80
+ * The permission bits are part of it: the copy below writes each file with the
81
+ * mode captured from the source, so a payload whose `0644` became `0755` with
82
+ * identical bytes is not the same payload — it is the case where an intended
83
+ * permission repair would otherwise be skipped as already done.
84
+ */
78
85
  export function payloadContentDigest(root) {
79
86
  const snapshot = capturePayloadContent(root);
80
87
  if (!snapshot) return null;
81
- const hashes = [...snapshot.files].map(([name, bytes]) => [name, createHash('sha256').update(bytes).digest('hex')]);
82
- return createHash('sha256').update(JSON.stringify(hashes.sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0))).digest('hex');
88
+ const entries = [...snapshot.files].map(([name, bytes]) => [
89
+ name,
90
+ snapshot.modes.get(name) ?? null,
91
+ createHash('sha256').update(bytes).digest('hex'),
92
+ ]);
93
+ return createHash('sha256').update(JSON.stringify(entries.sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0))).digest('hex');
83
94
  }
84
95
 
85
96
  export function samePayloadContent(sourceRoot, installedRoot) {
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '1.10.1';
39
+ export const RUNTIME_VERSION = '1.10.3';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([