@ran-sh/dsh-crew 1.10.2 → 1.10.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "1.10.2",
3
+ "version": "1.10.4",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "1.10.2",
3
+ "version": "1.10.4",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -587,12 +587,90 @@ function validateJournalRuntimeSegment({ home, rt }) {
587
587
  return { ok: true };
588
588
  }
589
589
 
590
+ /**
591
+ * The runtime window a coordinated update left open, or null.
592
+ *
593
+ * The durable maintenance session is the source of truth, not the journal. The
594
+ * supervisor persists the session as part of stopping 3210, so it exists from the
595
+ * instant the stop lands — whereas the journal is written a few instructions
596
+ * later. Reading the session closes that gap: whatever stop happened, recovery
597
+ * finds it. The journal's copy is kept as a fallback for a session that has since
598
+ * been cleared.
599
+ */
600
+ async function openMaintenanceWindow({ home, journal }) {
601
+ const appRoot = join(home, '.config', 'dsh-crew');
602
+ try {
603
+ const { readMaintenanceSession } = await import('../supervisor/restart-request.mjs');
604
+ const durable = readMaintenanceSession(appRoot);
605
+ if (durable.ok && durable.state === 'present') {
606
+ return { lease: durable.session.lease, runtime_id: durable.session.runtime_id, source: 'session' };
607
+ }
608
+ // A session that exists but cannot be read is not "no session": recovery
609
+ // must not clear the journal and report success while a stopped runtime it
610
+ // cannot account for is sitting there.
611
+ if (!durable.ok && durable.state === 'malformed') {
612
+ return { malformed: true, error: durable.code ?? 'MAINTENANCE_SESSION_MALFORMED' };
613
+ }
614
+ } catch { /* fall through to the journal's copy */ }
615
+ const recorded = journal?.runtime?.maintenance ?? null;
616
+ if (recorded?.lease && recorded?.runtime_id) return { lease: recorded.lease, runtime_id: recorded.runtime_id, source: 'journal' };
617
+ return null;
618
+ }
619
+
620
+ /**
621
+ * Restart a runtime that a coordinated update left deliberately stopped.
622
+ *
623
+ * The window is the lease and runtime id the supervisor issued when it stopped
624
+ * 3210. Passing them back resumes that exact stopped window rather than asking
625
+ * for a new stop, which is the only thing the durable maintenance session
626
+ * accepts — a second independent stop is refused while the first is in force.
627
+ */
628
+ async function closeMaintenanceWindow({ home, window, supervisorFactory = crewSupervisor, log = () => {} }) {
629
+ try {
630
+ const supervisor = supervisorFactory({ home });
631
+ const started = await supervisor.startOwnedBackend({ lease: window.lease, runtimeId: window.runtime_id });
632
+ if (started?.ok !== true) {
633
+ return { ok: false, code: started?.code ?? 'MAINTENANCE_START_FAILED', error: started?.error ?? null };
634
+ }
635
+ log('- recovery restarted the runtime that the interrupted update had stopped');
636
+ return { ok: true };
637
+ } catch (error) {
638
+ return { ok: false, code: 'MAINTENANCE_START_FAILED', error: String(error?.message ?? error) };
639
+ }
640
+ }
641
+
642
+ /**
643
+ * Attempt to stop the owned runtime, and report whether the stop actually
644
+ * happened.
645
+ *
646
+ * Only a successful stop counts. `MAINTENANCE_IDENTITY_UNAVAILABLE` means the
647
+ * supervisor could not obtain the live runtime's identity over its HTTP endpoint,
648
+ * which is equally consistent with a live runtime whose request timed out — so it
649
+ * is not evidence that nothing is running and must never be read as any. An
650
+ * earlier revision of this helper did read it that way and would have replaced a
651
+ * live runtime's tree, which is the exact hazard it exists to prevent.
652
+ */
653
+ async function ensureRuntimeStopped({ home, supervisorFactory = crewSupervisor, log = () => {} }) {
654
+ try {
655
+ const supervisor = supervisorFactory({ home });
656
+ if (typeof supervisor?.stopOwnedBackend !== 'function') return { ok: false, code: 'MAINTENANCE_UNAVAILABLE' };
657
+ const stopped = await supervisor.stopOwnedBackend();
658
+ if (stopped?.ok === true) {
659
+ log('- recovery stopped the runtime before replacing its tree');
660
+ return { ok: true, stopped: true };
661
+ }
662
+ return { ok: false, code: stopped?.code ?? 'MAINTENANCE_STOP_FAILED', error: stopped?.error ?? null };
663
+ } catch (error) {
664
+ return { ok: false, code: 'MAINTENANCE_STOP_FAILED', error: String(error?.message ?? error) };
665
+ }
666
+ }
667
+
590
668
  // Reconcile a leftover journal from a crashed update/install. The single
591
669
  // commit point is the pointer write: pointer == candidate means committed
592
670
  // (finalize, do NOT roll back); pointer == prior/absent means pre-commit
593
671
  // (restore activation surfaces, drop candidate). A malformed journal fails
594
672
  // closed and is retained for operator inspection.
595
- export function reconcileUpdateJournal({ home = homedir(), log = () => {}, installer = realInstaller } = {}) {
673
+ export async function reconcileUpdateJournal({ home = homedir(), log = () => {}, installer = realInstaller, supervisorFactory = crewSupervisor } = {}) {
596
674
  const journal = readUpdateJournal({ home });
597
675
  if (!journal) return { ok: true, reconciled: false };
598
676
  if (journal.malformed) {
@@ -686,6 +764,36 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
686
764
  return { ok: false, code: 'JOURNAL_POINTER_DIVERGED', stage: journal.stage, error: `pointer references unexpected release ${pointer?.path ?? 'unknown'}; refusing recovery` };
687
765
  }
688
766
 
767
+ // A verified coordinated journal means the candidate runtime was started and
768
+ // its identity checked, and the pointer write happens after that — so this
769
+ // state means the pointer write was lost and the candidate is what is running.
770
+ // Completing the commit is the recovery; rolling back would swap the tree and
771
+ // the payload out from under a live process. This runs only after the
772
+ // divergence check above, so "the pointer has not moved" means it still names
773
+ // the prior release or nothing at all — a third release has already failed
774
+ // closed, and its release identity can never be overwritten by a stale journal.
775
+ if (journal.stage === 'coordinated-update' && journal.verified === true && !pointerMatchesCandidate) {
776
+ if (!candidateIdent?.name || !candidateIdent?.version || !candidateDir) {
777
+ return { ok: false, code: 'JOURNAL_CANDIDATE_INVALID', stage: journal.stage };
778
+ }
779
+ const rt = journal.runtime ?? null;
780
+ if (rt?.candidateVersion && rt?.liveRoot && readRuntimeTreeVersionSync(rt.liveRoot) !== rt.candidateVersion) {
781
+ return { ok: false, code: 'JOURNAL_COORDINATED_RUNTIME_MISMATCH', stage: journal.stage, error: 'a verified coordinated journal expects the candidate runtime to be live; refusing to commit' };
782
+ }
783
+ // The same check the committed path applies, so a roll-forward cannot install
784
+ // a pointer to a payload that would have been rejected had it arrived the
785
+ // ordinary way.
786
+ const validated = validateInstalledPayload(candidateDir, { expectedName: candidateIdent.name, expectedVersion: candidateIdent.version });
787
+ if (!validated.ok) {
788
+ return { ok: false, code: 'JOURNAL_CANDIDATE_INVALID', stage: journal.stage, error: (validated.errors ?? []).join('; ') };
789
+ }
790
+ writeCurrentPointer({ home, name: candidateIdent.name, version: candidateIdent.version, path: candidateDir });
791
+ clearUpdateJournal({ home });
792
+ gcOldReleases({ home, protect: journal.prior?.path ?? null });
793
+ log(`- recovered update journal at stage ${journal.stage}: candidate ${candidateIdent.version} was verified and started, so the commit was completed`);
794
+ return { ok: true, reconciled: true, stage: journal.stage, committed: true };
795
+ }
796
+
689
797
  // Pre-commit side: prior stays authoritative. Re-point live activation
690
798
  // surfaces back at prior (a crash between activation and pointer write
691
799
  // leaves them on the candidate), then drop the candidate. A rollback
@@ -703,6 +811,29 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
703
811
  if (rt?.priorVersion && rt?.liveRoot) {
704
812
  const liveVersion = readRuntimeTreeVersionSync(rt.liveRoot);
705
813
  if (liveVersion !== rt.priorVersion) {
814
+ // The tree has to be replaced, and replacing it under a running process is
815
+ // the damage. The journal says whether a start can have happened: it is
816
+ // set to `starting` before the candidate is started and to `verified`
817
+ // after it checks out, so anything else means no runtime was ever started
818
+ // from the candidate cohort and the swap is safe.
819
+ //
820
+ // `starting` cannot be resolved from the journal alone — the candidate may
821
+ // be live and unverified. It must not be a dead end either, so recovery
822
+ // stops the runtime and rolls back on a stop that actually happened, or on
823
+ // a durable STOPPED session that already proves one did. Anything less is
824
+ // not proof: a runtime that cannot be shown to be stopped keeps its tree.
825
+ if (rt.state === 'starting') {
826
+ const window = await openMaintenanceWindow({ home, journal });
827
+ if (window?.malformed) {
828
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_SESSION_MALFORMED', stage: journal.stage, error: `${window.error}; refusing to replace a runtime tree while a stopped runtime cannot be accounted for` };
829
+ }
830
+ if (!window) {
831
+ const stopped = await ensureRuntimeStopped({ home, supervisorFactory, log });
832
+ if (!stopped.ok) {
833
+ return { ok: false, code: 'JOURNAL_RUNTIME_STOP_UNPROVEN', stage: journal.stage, error: `an interrupted update may have started the candidate runtime and could not be shown to be stopped (${stopped.code ?? 'unknown'}), so its tree was left alone; stop the owned runtime through the Crew supervisor, then re-run` };
834
+ }
835
+ }
836
+ }
706
837
  // Move the parked prior tree back onto liveRoot. This is a pure
707
838
  // directory swap: no network, no registry, safe under a stale lock.
708
839
  const parked = rt.priorRoot && existsSync(rt.priorRoot) ? rt.priorRoot : null;
@@ -747,29 +878,46 @@ export function reconcileUpdateJournal({ home = homedir(), log = () => {}, insta
747
878
  // and may be removed. (rollback candidates are retained releases.)
748
879
  try { rmSync(candidateDir, { recursive: true, force: true }); } catch {}
749
880
  }
881
+ // The prior release is back in place, so a runtime the interrupted update
882
+ // stopped has to be started again — otherwise recovery reports success and
883
+ // leaves 3210 down. The journal is kept until the window is closed, so a
884
+ // crash between the two retries with the same recorded window instead of
885
+ // losing track of a stopped runtime.
886
+ const openWindow = journal.stage === 'coordinated-update'
887
+ ? await openMaintenanceWindow({ home, journal })
888
+ : null;
889
+ if (openWindow?.malformed) {
890
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_SESSION_MALFORMED', stage: journal.stage, error: `${openWindow.error}; refusing to report recovery while a stopped runtime cannot be accounted for` };
891
+ }
892
+ if (openWindow) {
893
+ const closed = await closeMaintenanceWindow({ home, window: openWindow, supervisorFactory, log });
894
+ if (!closed.ok) {
895
+ return { ok: false, code: 'JOURNAL_MAINTENANCE_WINDOW_OPEN', stage: journal.stage, error: `the runtime was left stopped and could not be restarted (${closed.code ?? 'unknown'}); journal retained` };
896
+ }
897
+ }
750
898
  clearUpdateJournal({ home });
751
899
  log(`- recovered update journal at stage ${journal.stage}: restored prior release ${journal.prior.version}`);
752
900
  return { ok: true, reconciled: true, stage: journal.stage, committed: false };
753
901
  }
754
902
 
755
- // First-install pre-commit: no prior exists. Undo the candidate's
756
- // activation surfaces FIRST (a crash between activation and pointer
757
- // write leaves the profile link on the candidate), then remove the
758
- // orphan candidate pointer + dir. Journal clears only after successful
759
- // compensation.
760
- const undone = undoCandidateActivationSync({ home, candidateDir, candidateName: journal.candidate?.name ?? null });
761
- if (!undone.ok) {
762
- return { ok: false, code: 'JOURNAL_UNDO_FAILED', stage: journal.stage, error: undone.error ?? undone.code };
763
- }
764
- if (candidateDir && existsSync(candidateDir)) {
765
- try { rmSync(candidateDir, { recursive: true, force: true }); } catch {}
766
- }
767
- if (pointer && candidateDir && pointer.path === candidateDir) {
768
- try { rmSync(currentPointerFile({ home }), { force: true }); } catch {}
769
- }
770
- clearUpdateJournal({ home });
771
- log(`- recovered first-install journal at stage ${journal.stage}: removed orphan candidate`);
772
- return { ok: true, reconciled: true, stage: journal.stage, committed: false };
903
+ // First-install pre-commit: no prior release exists, and there is nothing this
904
+ // branch can undo that it can prove is its own. Removing the loader link would
905
+ // break every host integration that names it; removing the integration records
906
+ // would mean deleting files in the operator's home on the strength of a
907
+ // filename, because those uninstallers take only `home` and cannot tell this
908
+ // failed install's records from ones the operator wrote, or from a manual
909
+ // installation that predates Crew. So nothing is changed: the candidate stays,
910
+ // the link still resolves to it, whatever names the link still resolves, and
911
+ // the journal is retained so an operator decides. A recovery that cannot
912
+ // attribute what it would delete does not delete it.
913
+ return {
914
+ ok: false,
915
+ code: 'JOURNAL_FIRST_INSTALL_NEEDS_OPERATOR',
916
+ stage: journal.stage,
917
+ candidate: candidateDir,
918
+ journal: updateJournalFile({ home }),
919
+ error: `a first install did not complete; nothing was changed, because Crew cannot prove which host records are its own. To finish by hand: repoint or remove the Crew profile registration first (its loader link is ${join(crewProfileDir({ home }), 'node_modules', ...'@ran-sh/dsh-crew'.split('/'))}), remove the host integrations that name it, then delete ${updateJournalFile({ home })} and ${candidateDir ?? 'the staged candidate'}`,
920
+ };
773
921
  }
774
922
 
775
923
  // ---- dependency tree materialization ----------------------------------------
@@ -1850,7 +1998,7 @@ async function npxRollbackInner({ home, version: targetVersion, log, installer,
1850
1998
  // Journal-aware entry: refuse to overwrite a journal left by a crashed
1851
1999
  // transaction. Reconcile it first (or fail closed) instead of silently
1852
2000
  // replacing it with the rollback intent.
1853
- const pending = reconcileUpdateJournal({ home, log });
2001
+ const pending = await reconcileUpdateJournal({ home, log, supervisorFactory });
1854
2002
  if (!pending.ok) return { ok: false, error: `refusing rollback with unreconciled journal (${pending.code ?? 'unknown'})` };
1855
2003
  const current = readCurrentPointer({ home });
1856
2004
  if (!current?.path || !existsSync(current.path)) return { ok: false, error: 'no active Crew payload to roll back' };
@@ -2178,7 +2326,7 @@ export async function performCoordinatedCohortUpdate({
2178
2326
  // Journal the full coordinated intent BEFORE any destructive step. The
2179
2327
  // journal carries prior+candidate dshVersion plus runtime roots so a crash
2180
2328
  // at any point is recoverable by reconcileUpdateJournal.
2181
- writeUpdateJournal({
2329
+ const journalBase = {
2182
2330
  home,
2183
2331
  stage: 'coordinated-update',
2184
2332
  prior: { name: prior.name, version: prior.version, path: prior.path, dshVersion: priorDshVersion },
@@ -2191,7 +2339,8 @@ export async function performCoordinatedCohortUpdate({
2191
2339
  candidateVersion: candidateDshVersion,
2192
2340
  retainedRoot,
2193
2341
  },
2194
- });
2342
+ };
2343
+ writeUpdateJournal(journalBase);
2195
2344
 
2196
2345
  const switchPointer = (release) => writeCurrentPointer({ home, name: release.name, version: release.version, path: release.path });
2197
2346
  const priorRelease = { name: prior.name, version: prior.version, path: prior.path };
@@ -2246,6 +2395,16 @@ export async function performCoordinatedCohortUpdate({
2246
2395
  clearUpdateJournal({ home });
2247
2396
  return { ok: false, code: stopped.code ?? 'COORDINATED_STOP_FAILED', error: stopped.error ?? 'could not stop owned 3210' };
2248
2397
  }
2398
+ // The runtime is now deliberately stopped inside a durable maintenance
2399
+ // window, and everything from here on mutates its tree. Record the window in
2400
+ // the journal before that mutation: a crash in the next few steps used to
2401
+ // leave 3210 down with no record of how to resume it, and the surviving
2402
+ // session then refused the next ordinary stop, so the machine could not
2403
+ // recover on its own. `startOwnedBackend` accepts this exact lease and
2404
+ // runtime id, which is what recovery needs to close the window.
2405
+ if (stopped.lease && stopped.runtime_id) {
2406
+ writeUpdateJournal({ ...journalBase, runtime: { ...journalBase.runtime, maintenance: { lease: stopped.lease, runtime_id: stopped.runtime_id } } });
2407
+ }
2249
2408
  let liveMoved = false;
2250
2409
  try {
2251
2410
  // Move live runtime aside, retaining its tree for offline rollback.
@@ -2282,6 +2441,12 @@ export async function performCoordinatedCohortUpdate({
2282
2441
  const comp = await compensate();
2283
2442
  return finalizeCompensationFailure({ home, code: null, error: 'candidate activation failed', comp });
2284
2443
  }
2444
+ // Record that a start may be about to happen BEFORE it happens. Recovery
2445
+ // decides whether the runtime tree can be replaced by reading this, and a
2446
+ // crash between the start and the verification would otherwise look exactly
2447
+ // like "the candidate was never started" — which is how a rollback ends up
2448
+ // swapping the tree out from under a running process.
2449
+ writeUpdateJournal({ ...journalBase, runtime: { ...journalBase.runtime, state: 'starting' } });
2285
2450
  const started = await startFn();
2286
2451
  if (started?.ok !== true) {
2287
2452
  const comp = await compensate();
@@ -2307,9 +2472,12 @@ export async function performCoordinatedCohortUpdate({
2307
2472
  }
2308
2473
  }
2309
2474
 
2310
- // Mark the journal verified BEFORE the pointer write. Crash recovery
2311
- // then has a clean WAL relation: pointer==prior means not committed,
2312
- // pointer==candidate with verified journal means committed.
2475
+ // Mark the journal verified BEFORE the pointer write, which makes a crash
2476
+ // between the two distinguishable — but note what "verified" means here: the
2477
+ // candidate runtime has already been started and its identity checked, so the
2478
+ // candidate is what is running. A pointer still naming prior with a verified
2479
+ // journal therefore means the pointer write was lost, and recovery completes
2480
+ // the commit rather than rolling the tree out from under the live process.
2313
2481
  const marked = markJournalVerified({
2314
2482
  home,
2315
2483
  stage: 'coordinated-update',
@@ -2413,7 +2581,16 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2413
2581
  }
2414
2582
  log(`✓ Harness plugin registered (dedicated dsh-crew profile → ${releaseDir})`);
2415
2583
 
2416
- const codex = installer.installCodex({ home, root: releaseDir });
2584
+ // The host integrations are pointed at the profile's loader link, not at the
2585
+ // release directory. That link is what registration just re-pointed, so it
2586
+ // always resolves to whichever release is live: an upgrade re-points it and
2587
+ // the integrations keep working, and removing an old release can no longer
2588
+ // leave four configurations naming a directory that is gone. Writing the
2589
+ // release path directly is what made crash recovery unable to delete a
2590
+ // candidate without breaking Codex, ZCode and Claude Code.
2591
+ const integrationRoot = registration.linkPath ?? releaseDir;
2592
+
2593
+ const codex = installer.installCodex({ home, root: integrationRoot });
2417
2594
  if (codex.ok === false) {
2418
2595
  log(`✗ Codex Desktop integration failed: ${(codex.actions ?? []).join('; ')}`);
2419
2596
  return false;
@@ -2421,7 +2598,7 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2421
2598
  log('✓ Codex Desktop integration');
2422
2599
 
2423
2600
  if (installer.installZCode) {
2424
- const zcode = installer.installZCode({ home, root: releaseDir });
2601
+ const zcode = installer.installZCode({ home, root: integrationRoot });
2425
2602
  if (zcode.ok === false) {
2426
2603
  log(`✗ ZCode integration failed (${zcode.code ?? 'unknown'})`);
2427
2604
  return false;
@@ -2436,7 +2613,7 @@ async function activateRelease({ home, releaseDir, manifest, log, installer, sup
2436
2613
  }
2437
2614
  if (startup?.supported) log('✓ Windows login startup');
2438
2615
 
2439
- const claude = await installer.installClaudeCode({ home, root: releaseDir });
2616
+ const claude = await installer.installClaudeCode({ home, root: integrationRoot });
2440
2617
  if (claude.ok === false) {
2441
2618
  log(`✗ Claude Code integration failed`);
2442
2619
  return false;
@@ -2502,7 +2679,7 @@ export async function npxInstall({
2502
2679
  }
2503
2680
 
2504
2681
  async function npxInstallInner({ home, log, sourceRoot, installer, ensureRuntime, npmInstaller }) {
2505
- const reconciled = reconcileUpdateJournal({ home, log });
2682
+ const reconciled = await reconcileUpdateJournal({ home, log });
2506
2683
  if (!reconciled.ok) return { ok: false, error: `update journal recovery failed (${reconciled.code ?? 'unknown'})` };
2507
2684
  const candidateRoot = sourceRoot ?? runningPackageRoot();
2508
2685
  const manifest = readManifest(candidateRoot);
@@ -2719,7 +2896,7 @@ export async function npxUpdate({
2719
2896
  }
2720
2897
 
2721
2898
  async function npxUpdateInner({ home, log, sourceRoot, candidate, spec, installer, ensureRuntime, npmInstaller, runner = spawnSync }) {
2722
- const journalRecovery = reconcileUpdateJournal({ home, log });
2899
+ const journalRecovery = await reconcileUpdateJournal({ home, log });
2723
2900
  if (!journalRecovery.ok) return { ok: false, error: `update journal recovery failed (${journalRecovery.code ?? 'unknown'})` };
2724
2901
  // Candidate resolution: explicit path/dir override > a newer validated
2725
2902
  // running launcher > configured npm registry (@latest). This makes the
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '1.10.2';
39
+ export const RUNTIME_VERSION = '1.10.4';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -25,6 +25,8 @@ export const NOT_GIT_REPOSITORY = 'NOT_GIT_REPOSITORY';
25
25
  export const GIT_NOT_FOUND = 'GIT_NOT_FOUND';
26
26
  export const GIT_TIMEOUT = 'GIT_TIMEOUT';
27
27
  export const GIT_ERROR = 'GIT_ERROR';
28
+ /** A valid repository whose HEAD does not resolve because nothing is committed. */
29
+ export const REPOSITORY_HAS_NO_COMMITS = 'REPOSITORY_HAS_NO_COMMITS';
28
30
  export const WORKTREE_LOCKED = 'WORKTREE_LOCKED';
29
31
  export const WORKTREE_RESERVE_FAILED = 'WORKTREE_RESERVE_FAILED';
30
32
  export const CANDIDATE_CAPTURE_FAILED = 'CANDIDATE_CAPTURE_FAILED';
@@ -269,7 +271,22 @@ export async function inspectRepository({ cwd, git, runner } = {}) {
269
271
  runGit(run, ['status', '--porcelain', '-uall'], { cwd }),
270
272
  ]);
271
273
  if (!root.ok) return { ok: false, reason: root.reason, error: root.error };
272
- if (!head.ok) return { ok: false, reason: head.reason, error: head.error };
274
+ if (!head.ok) {
275
+ // A repository with no commits yet is a valid repository whose HEAD simply
276
+ // does not resolve, and it cannot be told apart from a broken one by that
277
+ // command alone. It is worth telling apart: `git init && <ask Crew to do
278
+ // something>` is how a new project starts, and reporting it as a generic git
279
+ // error leaves the operator with nothing to act on.
280
+ const verified = await runGit(run, ['rev-parse', '--verify', '--quiet', 'HEAD'], { cwd });
281
+ if (verified.ok === false && (verified.code === 1 || verified.code === 128) && !verified.stdout?.trim()) {
282
+ return {
283
+ ok: false,
284
+ reason: REPOSITORY_HAS_NO_COMMITS,
285
+ error: 'this repository has no commits yet, so there is no revision for an isolated job to start from; make an initial commit, or set execution.isolation to "shared" to run in the working tree',
286
+ };
287
+ }
288
+ return { ok: false, reason: head.reason, error: head.error };
289
+ }
273
290
  return {
274
291
  ok: true,
275
292
  repoRoot: resolve(root.stdout.trim()),