@ran-sh/dsh-crew 2.1.0 → 2.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -1263,6 +1263,25 @@ export function validateInstalledPayload(dir, { expectedName, expectedVersion, a
|
|
|
1263
1263
|
// pointer write LAST -> clear journal -> GC (prior protected until commit).
|
|
1264
1264
|
// A crash at any point before the pointer write leaves the prior release
|
|
1265
1265
|
// authoritative and the journal behind for reconcileUpdateJournal.
|
|
1266
|
+
/**
|
|
1267
|
+
* Drop the stage marker a copied candidate carries from the moment it is written.
|
|
1268
|
+
*
|
|
1269
|
+
* Ordered for callers that have already journaled the candidate, never before:
|
|
1270
|
+
* a crash between the journal and this must leave a journaled candidate rather
|
|
1271
|
+
* than a marker-free orphan that reads as complete but was never in a
|
|
1272
|
+
* transaction.
|
|
1273
|
+
*/
|
|
1274
|
+
function clearStageMarker(stageDir) {
|
|
1275
|
+
try {
|
|
1276
|
+
rmSync(join(stageDir, INCOMPLETE_MARKER), { force: true });
|
|
1277
|
+
} catch (error) {
|
|
1278
|
+
throw Object.assign(new Error(`cannot clear stage marker: ${error?.message ?? error}`), { code: 'STAGE_MARKER_REMOVE_FAILED' });
|
|
1279
|
+
}
|
|
1280
|
+
if (existsSync(join(stageDir, INCOMPLETE_MARKER))) {
|
|
1281
|
+
throw Object.assign(new Error('stage marker still present after removal'), { code: 'STAGE_MARKER_REMOVE_FAILED' });
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1284
|
+
|
|
1266
1285
|
export function beginReleaseActivation({ stageDir, manifest, home, prior = null }) {
|
|
1267
1286
|
// Survives pointer commit and updater crashes; versions alone cannot prove
|
|
1268
1287
|
// that a same-version code replacement has reached the running process.
|
|
@@ -1277,14 +1296,7 @@ export function beginReleaseActivation({ stageDir, manifest, home, prior = null
|
|
|
1277
1296
|
prior: prior ? { name: prior.name, version: prior.version, path: prior.path } : null,
|
|
1278
1297
|
candidate: { name: manifest.name, version: manifest.version, stageDir },
|
|
1279
1298
|
});
|
|
1280
|
-
|
|
1281
|
-
rmSync(join(stageDir, INCOMPLETE_MARKER), { force: true });
|
|
1282
|
-
} catch (error) {
|
|
1283
|
-
throw Object.assign(new Error(`cannot clear stage marker: ${error?.message ?? error}`), { code: 'STAGE_MARKER_REMOVE_FAILED' });
|
|
1284
|
-
}
|
|
1285
|
-
if (existsSync(join(stageDir, INCOMPLETE_MARKER))) {
|
|
1286
|
-
throw Object.assign(new Error('stage marker still present after removal'), { code: 'STAGE_MARKER_REMOVE_FAILED' });
|
|
1287
|
-
}
|
|
1299
|
+
clearStageMarker(stageDir);
|
|
1288
1300
|
return stageDir;
|
|
1289
1301
|
}
|
|
1290
1302
|
|
|
@@ -2533,6 +2545,19 @@ export async function performCoordinatedCohortUpdate({
|
|
|
2533
2545
|
return finalizeCompensationFailure({ home, code: marked.code ?? 'JOURNAL_MARK_FAILED', error: 'coordinated journal mark-verified failed', comp });
|
|
2534
2546
|
}
|
|
2535
2547
|
|
|
2548
|
+
// This path commits the pointer itself and never goes through
|
|
2549
|
+
// beginReleaseActivation, so the stage marker the candidate was copied with
|
|
2550
|
+
// has to be cleared here — at the same point in the order, after the journal
|
|
2551
|
+
// and before the commit. Left behind, it is permanent: every health check
|
|
2552
|
+
// then reads a running payload as incomplete, which is what `dsh-crew status`
|
|
2553
|
+
// reports as "unverifiable/damaged".
|
|
2554
|
+
try {
|
|
2555
|
+
clearStageMarker(stageDir);
|
|
2556
|
+
} catch (error) {
|
|
2557
|
+
const comp = await compensate().catch(() => ({ ok: false }));
|
|
2558
|
+
return finalizeCompensationFailure({ home, code: error?.code ?? 'STAGE_MARKER_REMOVE_FAILED', error: error?.message ?? 'cannot clear stage marker', comp });
|
|
2559
|
+
}
|
|
2560
|
+
|
|
2536
2561
|
// COMMIT POINT / LAST: pointer write after the verified journal.
|
|
2537
2562
|
switchPointer(candidateRelease);
|
|
2538
2563
|
|
package/src/runtime-identity.mjs
CHANGED
|
@@ -36,7 +36,7 @@ export {
|
|
|
36
36
|
// included in the identity contract.
|
|
37
37
|
const RUNTIME_ID = randomUUID();
|
|
38
38
|
|
|
39
|
-
export const RUNTIME_VERSION = '2.1.
|
|
39
|
+
export const RUNTIME_VERSION = '2.1.2';
|
|
40
40
|
export const HUB_PROTOCOL_VERSION = 1;
|
|
41
41
|
|
|
42
42
|
export const HUB_CAPABILITIES = Object.freeze([
|
|
@@ -35,13 +35,15 @@ if not exist "%LAUNCH_HELPER%" (
|
|
|
35
35
|
>>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
|
|
36
36
|
echo ERROR: DSH Crew launcher helper is missing.
|
|
37
37
|
echo Repair it with: dsh-crew update
|
|
38
|
-
if /i "%LAUNCH_MODE%"=="open" pause
|
|
38
|
+
if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
|
|
39
39
|
exit /b 1
|
|
40
40
|
)
|
|
41
41
|
|
|
42
42
|
powershell.exe -NoLogo -NoProfile -NonInteractive -ExecutionPolicy Bypass -File "%LAUNCH_HELPER%" -Mode "%LAUNCH_MODE%"
|
|
43
43
|
set "LAUNCH_EXIT=%ERRORLEVEL%"
|
|
44
|
-
|
|
44
|
+
rem DSH_CREW_LAUNCHER_NO_PAUSE: a wrapper that reports the failure itself asks
|
|
45
|
+
rem for the pause to be skipped here, so the operator presses a key once.
|
|
46
|
+
if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
|
|
45
47
|
exit /b %LAUNCH_EXIT%
|
|
46
48
|
|
|
47
49
|
:invalid_argument
|
|
@@ -51,7 +53,8 @@ exit /b 64
|
|
|
51
53
|
|
|
52
54
|
:help
|
|
53
55
|
echo Usage: %~nx0 [--open ^| --background ^| --watch]
|
|
54
|
-
echo --open Open official Harness on 3080
|
|
56
|
+
echo --open Open official Harness on 3080 and return once Crew is supervised;
|
|
57
|
+
echo 3210 keeps starting in the background.
|
|
55
58
|
echo --background Start the Crew-owned 3210 service silently.
|
|
56
59
|
echo --watch Keep the Crew-owned 3210 service healthy.
|
|
57
60
|
exit /b 0
|
|
@@ -645,18 +645,50 @@ function Restore-OwnedServiceRecord {
|
|
|
645
645
|
return $true
|
|
646
646
|
}
|
|
647
647
|
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
648
|
+
# Spawns the persistent watcher when no live one is present, and returns its
|
|
649
|
+
# heartbeat record when one already is. Shared so that the interactive and the
|
|
650
|
+
# blocking entries agree on what counts as "a supervisor is already running" —
|
|
651
|
+
# two answers to that question would let one entry spawn a duplicate that the
|
|
652
|
+
# mutex immediately kills.
|
|
653
|
+
function Start-CrewSupervisorProcess {
|
|
651
654
|
$observed = Get-SupervisorHeartbeatRecord
|
|
652
655
|
if ($observed -and $observed.State -eq 'legacy-v1') {
|
|
653
656
|
throw 'CREW_SUPERVISOR_UPGRADE_REQUIRED: a legacy watcher is active and must be handed off before interactive launch.'
|
|
654
657
|
}
|
|
655
|
-
if (
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
658
|
+
if ($observed) { return $observed }
|
|
659
|
+
$arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
|
|
660
|
+
$watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
|
|
661
|
+
Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
|
|
662
|
+
return $null
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
# Waits only for the watcher to exist, not for 3210 to answer. The watcher
|
|
666
|
+
# publishes its heartbeat before it first touches the port, so this returns as
|
|
667
|
+
# soon as Crew is being supervised. Used by the interactive entry: 3080 is
|
|
668
|
+
# already serving by then, and making the operator's window wait for a first
|
|
669
|
+
# 3210 boot (measured 18-78s on this machine, longer under load) delays nothing
|
|
670
|
+
# they can see — the watcher performs that same wait either way.
|
|
671
|
+
function Wait-CrewSupervisorStarted {
|
|
672
|
+
param([int] $TimeoutSeconds = 30)
|
|
673
|
+
$observed = Start-CrewSupervisorProcess
|
|
674
|
+
if ($observed) {
|
|
675
|
+
Write-LaunchLog ('Persistent Crew supervisor already running; PID={0}.' -f $observed.Record.pid)
|
|
676
|
+
return $true
|
|
659
677
|
}
|
|
678
|
+
$deadline = (Get-Date).AddSeconds($TimeoutSeconds)
|
|
679
|
+
do {
|
|
680
|
+
$record = Get-SupervisorHeartbeatRecord
|
|
681
|
+
if ($record) {
|
|
682
|
+
Write-LaunchLog ('Persistent Crew supervisor started; PID={0}; state={1}.' -f $record.Record.pid, $record.State)
|
|
683
|
+
return $true
|
|
684
|
+
}
|
|
685
|
+
Start-Sleep -Milliseconds 250
|
|
686
|
+
} while ((Get-Date) -lt $deadline)
|
|
687
|
+
return $false
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
function Ensure-CrewSupervisorRunning { param([int] $TimeoutSeconds = 90)
|
|
691
|
+
$null = Start-CrewSupervisorProcess
|
|
660
692
|
|
|
661
693
|
$crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
|
|
662
694
|
if (-not $crew) { throw 'Crew-owned 3210 service definition is missing.' }
|
|
@@ -1211,7 +1243,10 @@ function Ensure-CrewServices {
|
|
|
1211
1243
|
# matching maintenance-start owns the launch right.
|
|
1212
1244
|
if ($service.CrewOwned -and (Test-MaintenanceSessionActive)) {
|
|
1213
1245
|
$service.State = 'maintenance'
|
|
1214
|
-
|
|
1246
|
+
# Say why, rather than clearing the field: an empty reason reaches the
|
|
1247
|
+
# startup wait as "deadline exceeded: dsh-crew:3210 ()", which reads like a
|
|
1248
|
+
# fault when the fence is the supervisor doing exactly as it was told.
|
|
1249
|
+
$service.LastError = 'a maintenance session holds the launch right (an update is mid-handoff); auto-start deferred'
|
|
1215
1250
|
continue
|
|
1216
1251
|
}
|
|
1217
1252
|
$health = Get-HealthState $service
|
|
@@ -1393,10 +1428,35 @@ try {
|
|
|
1393
1428
|
exit 0
|
|
1394
1429
|
}
|
|
1395
1430
|
|
|
1396
|
-
Ensure-CrewSupervisorRunning
|
|
1397
|
-
|
|
1398
1431
|
if ($Mode -eq 'open') {
|
|
1399
|
-
|
|
1432
|
+
# A desktop launch promises the frontend on 3080, and that is up in about a
|
|
1433
|
+
# second. The supervisor owns 3210 from the moment it starts — it publishes
|
|
1434
|
+
# its heartbeat before it first touches the port — so this waits for the
|
|
1435
|
+
# watcher to be running, reports where 3210 actually is, and returns. Holding
|
|
1436
|
+
# the operator's window for 3210 readiness bought nothing: the watcher is
|
|
1437
|
+
# doing that wait anyway, and under load a first boot here has taken 78s.
|
|
1438
|
+
if (-not (Wait-CrewSupervisorStarted)) {
|
|
1439
|
+
throw 'No Crew supervisor started within 30s; 3210 has nothing watching it.'
|
|
1440
|
+
}
|
|
1441
|
+
$crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
|
|
1442
|
+
$health = if ($crew) { Get-HealthState $crew } else { $null }
|
|
1443
|
+
if ($health -and $health.Ready) {
|
|
1444
|
+
Write-LaunchLog 'Official frontend is on 3080; Crew is ready on 3210.'
|
|
1445
|
+
} else {
|
|
1446
|
+
Write-LaunchLog ('Official frontend is on 3080; the supervisor is bringing 3210 up in the background. Last health: {0}' -f $health.Error) 'WARN'
|
|
1447
|
+
}
|
|
1448
|
+
# Operator-facing summary rather than a log line: clicking Crew before 3210
|
|
1449
|
+
# answers looks like a broken feature, so say that it is still coming up.
|
|
1450
|
+
Write-Host ''
|
|
1451
|
+
Write-Host 'DSH Crew: the frontend is on http://127.0.0.1:3080.' -ForegroundColor Green
|
|
1452
|
+
if ($health -and $health.Ready) {
|
|
1453
|
+
Write-Host 'Backend 3210 is ready.' -ForegroundColor Green
|
|
1454
|
+
} else {
|
|
1455
|
+
Write-Host 'Backend 3210 is still starting; Crew features appear once it answers.' -ForegroundColor Yellow
|
|
1456
|
+
}
|
|
1457
|
+
Write-Host ("Diagnostic log: {0}" -f $launcherLog) -ForegroundColor DarkGray
|
|
1458
|
+
} else {
|
|
1459
|
+
Ensure-CrewSupervisorRunning
|
|
1400
1460
|
}
|
|
1401
1461
|
Write-LaunchLog ('Launcher completed successfully in {0:n1}s.' -f ((Get-Date) - $startedAt).TotalSeconds)
|
|
1402
1462
|
exit 0
|