@ran-sh/dsh-crew 2.1.0 → 2.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.1.0",
3
+ "version": "2.1.2",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.1.0",
3
+ "version": "2.1.2",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -1263,6 +1263,25 @@ export function validateInstalledPayload(dir, { expectedName, expectedVersion, a
1263
1263
  // pointer write LAST -> clear journal -> GC (prior protected until commit).
1264
1264
  // A crash at any point before the pointer write leaves the prior release
1265
1265
  // authoritative and the journal behind for reconcileUpdateJournal.
1266
+ /**
1267
+ * Drop the stage marker a copied candidate carries from the moment it is written.
1268
+ *
1269
+ * Ordered for callers that have already journaled the candidate, never before:
1270
+ * a crash between the journal and this must leave a journaled candidate rather
1271
+ * than a marker-free orphan that reads as complete but was never in a
1272
+ * transaction.
1273
+ */
1274
+ function clearStageMarker(stageDir) {
1275
+ try {
1276
+ rmSync(join(stageDir, INCOMPLETE_MARKER), { force: true });
1277
+ } catch (error) {
1278
+ throw Object.assign(new Error(`cannot clear stage marker: ${error?.message ?? error}`), { code: 'STAGE_MARKER_REMOVE_FAILED' });
1279
+ }
1280
+ if (existsSync(join(stageDir, INCOMPLETE_MARKER))) {
1281
+ throw Object.assign(new Error('stage marker still present after removal'), { code: 'STAGE_MARKER_REMOVE_FAILED' });
1282
+ }
1283
+ }
1284
+
1266
1285
  export function beginReleaseActivation({ stageDir, manifest, home, prior = null }) {
1267
1286
  // Survives pointer commit and updater crashes; versions alone cannot prove
1268
1287
  // that a same-version code replacement has reached the running process.
@@ -1277,14 +1296,7 @@ export function beginReleaseActivation({ stageDir, manifest, home, prior = null
1277
1296
  prior: prior ? { name: prior.name, version: prior.version, path: prior.path } : null,
1278
1297
  candidate: { name: manifest.name, version: manifest.version, stageDir },
1279
1298
  });
1280
- try {
1281
- rmSync(join(stageDir, INCOMPLETE_MARKER), { force: true });
1282
- } catch (error) {
1283
- throw Object.assign(new Error(`cannot clear stage marker: ${error?.message ?? error}`), { code: 'STAGE_MARKER_REMOVE_FAILED' });
1284
- }
1285
- if (existsSync(join(stageDir, INCOMPLETE_MARKER))) {
1286
- throw Object.assign(new Error('stage marker still present after removal'), { code: 'STAGE_MARKER_REMOVE_FAILED' });
1287
- }
1299
+ clearStageMarker(stageDir);
1288
1300
  return stageDir;
1289
1301
  }
1290
1302
 
@@ -2533,6 +2545,19 @@ export async function performCoordinatedCohortUpdate({
2533
2545
  return finalizeCompensationFailure({ home, code: marked.code ?? 'JOURNAL_MARK_FAILED', error: 'coordinated journal mark-verified failed', comp });
2534
2546
  }
2535
2547
 
2548
+ // This path commits the pointer itself and never goes through
2549
+ // beginReleaseActivation, so the stage marker the candidate was copied with
2550
+ // has to be cleared here — at the same point in the order, after the journal
2551
+ // and before the commit. Left behind, it is permanent: every health check
2552
+ // then reads a running payload as incomplete, which is what `dsh-crew status`
2553
+ // reports as "unverifiable/damaged".
2554
+ try {
2555
+ clearStageMarker(stageDir);
2556
+ } catch (error) {
2557
+ const comp = await compensate().catch(() => ({ ok: false }));
2558
+ return finalizeCompensationFailure({ home, code: error?.code ?? 'STAGE_MARKER_REMOVE_FAILED', error: error?.message ?? 'cannot clear stage marker', comp });
2559
+ }
2560
+
2536
2561
  // COMMIT POINT / LAST: pointer write after the verified journal.
2537
2562
  switchPointer(candidateRelease);
2538
2563
 
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.1.0';
39
+ export const RUNTIME_VERSION = '2.1.2';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -35,13 +35,15 @@ if not exist "%LAUNCH_HELPER%" (
35
35
  >>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
36
36
  echo ERROR: DSH Crew launcher helper is missing.
37
37
  echo Repair it with: dsh-crew update
38
- if /i "%LAUNCH_MODE%"=="open" pause
38
+ if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
39
39
  exit /b 1
40
40
  )
41
41
 
42
42
  powershell.exe -NoLogo -NoProfile -NonInteractive -ExecutionPolicy Bypass -File "%LAUNCH_HELPER%" -Mode "%LAUNCH_MODE%"
43
43
  set "LAUNCH_EXIT=%ERRORLEVEL%"
44
- if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" pause
44
+ rem DSH_CREW_LAUNCHER_NO_PAUSE: a wrapper that reports the failure itself asks
45
+ rem for the pause to be skipped here, so the operator presses a key once.
46
+ if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
45
47
  exit /b %LAUNCH_EXIT%
46
48
 
47
49
  :invalid_argument
@@ -51,7 +53,8 @@ exit /b 64
51
53
 
52
54
  :help
53
55
  echo Usage: %~nx0 [--open ^| --background ^| --watch]
54
- echo --open Open official Harness on 3080, then start Crew silently on 3210.
56
+ echo --open Open official Harness on 3080 and return once Crew is supervised;
57
+ echo 3210 keeps starting in the background.
55
58
  echo --background Start the Crew-owned 3210 service silently.
56
59
  echo --watch Keep the Crew-owned 3210 service healthy.
57
60
  exit /b 0
@@ -645,18 +645,50 @@ function Restore-OwnedServiceRecord {
645
645
  return $true
646
646
  }
647
647
 
648
- function Ensure-CrewSupervisorRunning {
649
- param([int] $TimeoutSeconds = 90)
650
- $watcher = $null
648
+ # Spawns the persistent watcher when no live one is present, and returns its
649
+ # heartbeat record when one already is. Shared so that the interactive and the
650
+ # blocking entries agree on what counts as "a supervisor is already running" —
651
+ # two answers to that question would let one entry spawn a duplicate that the
652
+ # mutex immediately kills.
653
+ function Start-CrewSupervisorProcess {
651
654
  $observed = Get-SupervisorHeartbeatRecord
652
655
  if ($observed -and $observed.State -eq 'legacy-v1') {
653
656
  throw 'CREW_SUPERVISOR_UPGRADE_REQUIRED: a legacy watcher is active and must be handed off before interactive launch.'
654
657
  }
655
- if (-not $observed) {
656
- $arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
657
- $watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
658
- Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
658
+ if ($observed) { return $observed }
659
+ $arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
660
+ $watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
661
+ Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
662
+ return $null
663
+ }
664
+
665
+ # Waits only for the watcher to exist, not for 3210 to answer. The watcher
666
+ # publishes its heartbeat before it first touches the port, so this returns as
667
+ # soon as Crew is being supervised. Used by the interactive entry: 3080 is
668
+ # already serving by then, and making the operator's window wait for a first
669
+ # 3210 boot (measured 18-78s on this machine, longer under load) delays nothing
670
+ # they can see — the watcher performs that same wait either way.
671
+ function Wait-CrewSupervisorStarted {
672
+ param([int] $TimeoutSeconds = 30)
673
+ $observed = Start-CrewSupervisorProcess
674
+ if ($observed) {
675
+ Write-LaunchLog ('Persistent Crew supervisor already running; PID={0}.' -f $observed.Record.pid)
676
+ return $true
659
677
  }
678
+ $deadline = (Get-Date).AddSeconds($TimeoutSeconds)
679
+ do {
680
+ $record = Get-SupervisorHeartbeatRecord
681
+ if ($record) {
682
+ Write-LaunchLog ('Persistent Crew supervisor started; PID={0}; state={1}.' -f $record.Record.pid, $record.State)
683
+ return $true
684
+ }
685
+ Start-Sleep -Milliseconds 250
686
+ } while ((Get-Date) -lt $deadline)
687
+ return $false
688
+ }
689
+
690
+ function Ensure-CrewSupervisorRunning { param([int] $TimeoutSeconds = 90)
691
+ $null = Start-CrewSupervisorProcess
660
692
 
661
693
  $crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
662
694
  if (-not $crew) { throw 'Crew-owned 3210 service definition is missing.' }
@@ -1211,7 +1243,10 @@ function Ensure-CrewServices {
1211
1243
  # matching maintenance-start owns the launch right.
1212
1244
  if ($service.CrewOwned -and (Test-MaintenanceSessionActive)) {
1213
1245
  $service.State = 'maintenance'
1214
- $service.LastError = $null
1246
+ # Say why, rather than clearing the field: an empty reason reaches the
1247
+ # startup wait as "deadline exceeded: dsh-crew:3210 ()", which reads like a
1248
+ # fault when the fence is the supervisor doing exactly as it was told.
1249
+ $service.LastError = 'a maintenance session holds the launch right (an update is mid-handoff); auto-start deferred'
1215
1250
  continue
1216
1251
  }
1217
1252
  $health = Get-HealthState $service
@@ -1393,10 +1428,35 @@ try {
1393
1428
  exit 0
1394
1429
  }
1395
1430
 
1396
- Ensure-CrewSupervisorRunning
1397
-
1398
1431
  if ($Mode -eq 'open') {
1399
- Write-LaunchLog 'Official frontend is on 3080; Crew is running silently on 3210.'
1432
+ # A desktop launch promises the frontend on 3080, and that is up in about a
1433
+ # second. The supervisor owns 3210 from the moment it starts — it publishes
1434
+ # its heartbeat before it first touches the port — so this waits for the
1435
+ # watcher to be running, reports where 3210 actually is, and returns. Holding
1436
+ # the operator's window for 3210 readiness bought nothing: the watcher is
1437
+ # doing that wait anyway, and under load a first boot here has taken 78s.
1438
+ if (-not (Wait-CrewSupervisorStarted)) {
1439
+ throw 'No Crew supervisor started within 30s; 3210 has nothing watching it.'
1440
+ }
1441
+ $crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
1442
+ $health = if ($crew) { Get-HealthState $crew } else { $null }
1443
+ if ($health -and $health.Ready) {
1444
+ Write-LaunchLog 'Official frontend is on 3080; Crew is ready on 3210.'
1445
+ } else {
1446
+ Write-LaunchLog ('Official frontend is on 3080; the supervisor is bringing 3210 up in the background. Last health: {0}' -f $health.Error) 'WARN'
1447
+ }
1448
+ # Operator-facing summary rather than a log line: clicking Crew before 3210
1449
+ # answers looks like a broken feature, so say that it is still coming up.
1450
+ Write-Host ''
1451
+ Write-Host 'DSH Crew: the frontend is on http://127.0.0.1:3080.' -ForegroundColor Green
1452
+ if ($health -and $health.Ready) {
1453
+ Write-Host 'Backend 3210 is ready.' -ForegroundColor Green
1454
+ } else {
1455
+ Write-Host 'Backend 3210 is still starting; Crew features appear once it answers.' -ForegroundColor Yellow
1456
+ }
1457
+ Write-Host ("Diagnostic log: {0}" -f $launcherLog) -ForegroundColor DarkGray
1458
+ } else {
1459
+ Ensure-CrewSupervisorRunning
1400
1460
  }
1401
1461
  Write-LaunchLog ('Launcher completed successfully in {0:n1}s.' -f ((Get-Date) - $startedAt).TotalSeconds)
1402
1462
  exit 0