@ran-sh/dsh-crew 0.5.6 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -7
- package/README.zh.md +12 -7
- package/docs/installation.md +1 -1
- package/docs/ui-surfaces.md +65 -46
- package/lib/client.js +345 -22
- package/official-web-bridge/lib/client.js +345 -22
- package/package.json +1 -1
- package/src/client/collapsible-sections.mjs +3 -2
- package/src/client/index.tsx +227 -51
- package/src/config-readiness.mjs +56 -5
- package/src/extension-contract.mjs +9 -5
- package/src/hub/index.mjs +854 -445
- package/src/hub-client.mjs +24 -4
- package/src/hub-compatibility.mjs +23 -7
- package/src/install/install.mjs +4 -4
- package/src/install/npx-lifecycle.mjs +263 -118
- package/src/job-contracts.mjs +19 -3
- package/src/jobs.mjs +2 -4
- package/src/local-request-guard.mjs +60 -0
- package/src/mcp-runtime.mjs +7 -5
- package/src/model-routing.mjs +141 -49
- package/src/multimodal.mjs +0 -0
- package/src/official-web-bridge.mjs +242 -84
- package/src/policy.mjs +62 -31
- package/src/provider-config-scrub.mjs +71 -0
- package/src/provider-delete-adapters.mjs +424 -0
- package/src/provider-health.mjs +124 -0
- package/src/provider-inventory.mjs +132 -0
- package/src/provider-lifecycle-state.mjs +81 -0
- package/src/provider-lifecycle.mjs +214 -0
- package/src/provider-profile-store.mjs +171 -0
- package/src/readiness-matrix.mjs +11 -5
- package/src/removable-waiter.mjs +29 -0
- package/src/runtime-identity.mjs +43 -13
- package/src/server.mjs +34 -16
- package/src/status-shard.mjs +13 -2
- package/src/workflow-runtime.mjs +44 -10
- package/windows/start-dsh-crew.cmd +3 -3
- package/windows/start-dsh-crew.ps1 +49 -5
package/src/server.mjs
CHANGED
|
@@ -56,7 +56,9 @@ const constraintsSchema = z.object({
|
|
|
56
56
|
// Initial values come from the global config (~/.config/dsh-crew/config.json,
|
|
57
57
|
// edited on the DSH settings page); dsh_worker_config overrides per session.
|
|
58
58
|
// All dispatch decisions (dsh_run_worker AND dsh_spawn_worker, hub AND
|
|
59
|
-
//
|
|
59
|
+
// production dispatches go through the 3210-only policy resolver in
|
|
60
|
+
// src/policy.mjs; the historical standalone path is retained only for
|
|
61
|
+
// read/migration compatibility and is never selected for execution.
|
|
60
62
|
import { readGlobalConfig } from './install/install.mjs';
|
|
61
63
|
const initialGlobalConfig = normalizeGlobalConfig(readGlobalConfig());
|
|
62
64
|
const legacyDefaults = deriveLegacyConfig(initialGlobalConfig);
|
|
@@ -65,7 +67,7 @@ const sessionConfig = {
|
|
|
65
67
|
enabled: true,
|
|
66
68
|
default_tier: legacyDefaults.default_tier,
|
|
67
69
|
default_effort: legacyDefaults.default_effort,
|
|
68
|
-
mode: legacyDefaults.mode, // auto | hub
|
|
70
|
+
mode: legacyDefaults.mode === 'standalone' ? 'hub' : legacyDefaults.mode, // auto | hub; legacy standalone is migrated to hub
|
|
69
71
|
default_timeout_seconds: legacyDefaults.default_timeout_seconds,
|
|
70
72
|
tier_policy: undefined,
|
|
71
73
|
escalate_on_failure: legacyDefaults.escalate_on_failure,
|
|
@@ -83,7 +85,7 @@ function resetSessionConfig() {
|
|
|
83
85
|
enabled: true,
|
|
84
86
|
default_tier: legacyDefaults.default_tier,
|
|
85
87
|
default_effort: legacyDefaults.default_effort,
|
|
86
|
-
mode: legacyDefaults.mode,
|
|
88
|
+
mode: legacyDefaults.mode === 'standalone' ? 'hub' : legacyDefaults.mode,
|
|
87
89
|
default_timeout_seconds: legacyDefaults.default_timeout_seconds,
|
|
88
90
|
tier_policy: undefined,
|
|
89
91
|
escalate_on_failure: legacyDefaults.escalate_on_failure,
|
|
@@ -142,9 +144,9 @@ function dispatchDisabled() {
|
|
|
142
144
|
});
|
|
143
145
|
}
|
|
144
146
|
|
|
145
|
-
async function resolveMode() {
|
|
146
|
-
const status = await hubStatus();
|
|
147
|
-
const decision = resolveHubExecutionMode(sessionConfig.mode, status);
|
|
147
|
+
async function resolveMode() {
|
|
148
|
+
const status = await hubStatus();
|
|
149
|
+
const decision = resolveHubExecutionMode(sessionConfig.mode, status, { productionOnly: true });
|
|
148
150
|
if (!decision.ok) {
|
|
149
151
|
throw Object.assign(new Error(decision.error), {
|
|
150
152
|
code: decision.code,
|
|
@@ -274,7 +276,7 @@ server.registerTool('dsh_worker_config', {
|
|
|
274
276
|
enabled: z.boolean().optional().describe('false = refuse all worker dispatch this session'),
|
|
275
277
|
default_tier: z.enum(['flash', 'pro']).optional(),
|
|
276
278
|
default_effort: z.enum(['off', 'high', 'max']).optional(),
|
|
277
|
-
mode: z.enum(['auto', 'hub'
|
|
279
|
+
mode: z.enum(['auto', 'hub']).optional().describe('Execution mode. Production dispatch always uses the isolated 3210 Crew Harness; standalone is retained only as a legacy read/migration value.'),
|
|
278
280
|
default_timeout_seconds: z.number().int().positive().max(7200).optional(),
|
|
279
281
|
tier_policy: z.enum(['auto', 'flash-only', 'pro-only']).optional().describe('session hard clamp: flash-only / pro-only pin every dispatch to one tier'),
|
|
280
282
|
escalate_on_failure: z.boolean().optional().describe('allow an unverified worker attempt to retry through the stronger model policy (applies to run and spawn)'),
|
|
@@ -325,9 +327,11 @@ async function buildConfigReport() {
|
|
|
325
327
|
let effectiveWorkerProvider = null;
|
|
326
328
|
let effectiveWorkerSelection = { flash: null, pro: null };
|
|
327
329
|
let providerResolutionError;
|
|
328
|
-
let providerCatalogChecked = false;
|
|
329
|
-
let providerCatalogBody = null;
|
|
330
|
-
let
|
|
330
|
+
let providerCatalogChecked = false;
|
|
331
|
+
let providerCatalogBody = null;
|
|
332
|
+
let providerInventoryChecked = false;
|
|
333
|
+
let providerInventoryBody = null;
|
|
334
|
+
let hubJobsChecked = false;
|
|
331
335
|
let hubJobsBody = null;
|
|
332
336
|
const workerProviderMode = globalConfig.worker_provider_mode ?? 'deepseek-official';
|
|
333
337
|
if (workerProviderMode === 'deepseek-official') {
|
|
@@ -362,9 +366,21 @@ async function buildConfigReport() {
|
|
|
362
366
|
} else if (hubCompatibility.reachable) {
|
|
363
367
|
providerResolutionError = hubCompatibilityMessage(hubCompatibility);
|
|
364
368
|
}
|
|
365
|
-
if (hubCompatibility.compatible) {
|
|
366
|
-
try {
|
|
367
|
-
|
|
369
|
+
if (hubCompatibility.compatible) {
|
|
370
|
+
try {
|
|
371
|
+
providerInventoryChecked = true;
|
|
372
|
+
const inventoryRes = await fetch(`${globalConfig.hub_url}/_dsh/dsh-crew/providers`, { signal: AbortSignal.timeout(800) });
|
|
373
|
+
providerInventoryBody = inventoryRes.ok
|
|
374
|
+
? await inventoryRes.json()
|
|
375
|
+
: { ok: false, code: 'PROVIDER_INVENTORY_UNAVAILABLE' };
|
|
376
|
+
} catch {
|
|
377
|
+
providerInventoryChecked = true;
|
|
378
|
+
providerInventoryBody = { ok: false, code: 'PROVIDER_INVENTORY_UNAVAILABLE' };
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
if (hubCompatibility.compatible) {
|
|
382
|
+
try {
|
|
383
|
+
hubJobsChecked = true;
|
|
368
384
|
const jobsRes = await fetch(`${globalConfig.hub_url}/_dsh/dsh-crew/jobs`, { signal: AbortSignal.timeout(800) });
|
|
369
385
|
hubJobsBody = await jobsRes.json();
|
|
370
386
|
} catch {
|
|
@@ -375,9 +391,11 @@ async function buildConfigReport() {
|
|
|
375
391
|
const readinessMatrix = buildConfigReadinessMatrix({
|
|
376
392
|
hubCompatibility,
|
|
377
393
|
workerProviderMode,
|
|
378
|
-
providerCatalogChecked,
|
|
379
|
-
providerCatalogBody,
|
|
380
|
-
|
|
394
|
+
providerCatalogChecked,
|
|
395
|
+
providerCatalogBody,
|
|
396
|
+
providerInventoryChecked,
|
|
397
|
+
providerInventoryBody,
|
|
398
|
+
hubJobsChecked,
|
|
381
399
|
hubJobsBody,
|
|
382
400
|
});
|
|
383
401
|
const roleProfiles = loadRoleProfiles();
|
package/src/status-shard.mjs
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
// and readers merge all fresh shards. Kills the last-writer-wins race that a
|
|
4
4
|
// single shared status.json had with multiple concurrent writers.
|
|
5
5
|
|
|
6
|
-
import { writeFileSync, mkdirSync, rmSync, readdirSync, readFileSync } from 'node:fs';
|
|
6
|
+
import { writeFileSync, mkdirSync, rmSync, readdirSync, readFileSync, renameSync } from 'node:fs';
|
|
7
7
|
import { join } from 'node:path';
|
|
8
8
|
import { homedir } from 'node:os';
|
|
9
9
|
|
|
@@ -21,9 +21,20 @@ export function createShardWriter(kind) {
|
|
|
21
21
|
return {
|
|
22
22
|
writer,
|
|
23
23
|
publish(jobs) {
|
|
24
|
+
// Same durability pattern as the profile/workspace registries: write a
|
|
25
|
+
// same-directory temp file as 0600 and rename it over the shard, so a
|
|
26
|
+
// reader never observes a partially written shard (parse failures are
|
|
27
|
+
// silently ignored, which would make the writer look vanished).
|
|
24
28
|
try {
|
|
25
29
|
mkdirSync(SHARD_DIR, { recursive: true });
|
|
26
|
-
|
|
30
|
+
const temp = `${file}.tmp-${Date.now()}`;
|
|
31
|
+
try {
|
|
32
|
+
writeFileSync(temp, JSON.stringify({ updatedAt: new Date().toISOString(), writer, jobs }, null, 2), { encoding: 'utf8', mode: 0o600 });
|
|
33
|
+
renameSync(temp, file);
|
|
34
|
+
} catch (error) {
|
|
35
|
+
try { rmSync(temp, { force: true }); } catch {}
|
|
36
|
+
throw error;
|
|
37
|
+
}
|
|
27
38
|
} catch {}
|
|
28
39
|
},
|
|
29
40
|
dispose: cleanup,
|
package/src/workflow-runtime.mjs
CHANGED
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
// appear here.
|
|
16
16
|
|
|
17
17
|
import { resolveModelPolicy, shouldAutoReview, getRoleState } from './policy.mjs';
|
|
18
|
+
import { raceWaiters } from './removable-waiter.mjs';
|
|
18
19
|
import { applyWorkspaceEvidence, buildOutcome, decideNextStep, JOB_PHASES, canTransition } from './workflow.mjs';
|
|
19
20
|
import { parseDeliveryReport } from './delivery.mjs';
|
|
20
21
|
import { classifyFailure } from './failure-classification.mjs';
|
|
@@ -298,7 +299,8 @@ export function createWorkflowRuntime(adapters, {
|
|
|
298
299
|
let config = {};
|
|
299
300
|
|
|
300
301
|
try {
|
|
301
|
-
config = adapters.getConfig?.() ?? {};
|
|
302
|
+
config = adapters.getConfig?.() ?? {};
|
|
303
|
+
job.review_gate = ['required', 'optional', 'off'].includes(config?.review?.gate) ? config.review.gate : 'required';
|
|
302
304
|
const alloc = await adapters.allocateWorkspace?.(job);
|
|
303
305
|
if (alloc && alloc.ok === false) {
|
|
304
306
|
failJob(job, Object.assign(new Error(alloc.error ?? alloc.reason), { code: alloc.reason }));
|
|
@@ -324,6 +326,19 @@ export function createWorkflowRuntime(adapters, {
|
|
|
324
326
|
const review = await runReviewerAttempt(job, reviewTask, config, before);
|
|
325
327
|
if (job.cancelling) { await cancelWorkflow(job); return; }
|
|
326
328
|
job.review = review;
|
|
329
|
+
const reviewApproved = review.status === 'done'
|
|
330
|
+
&& review.delivery_complete === true
|
|
331
|
+
&& review.verdict === 'approve';
|
|
332
|
+
if (!reviewApproved) {
|
|
333
|
+
const code = review.verdict === 'request_changes'
|
|
334
|
+
? 'REVIEW_CHANGES_REQUESTED'
|
|
335
|
+
: 'REVIEW_INCONCLUSIVE';
|
|
336
|
+
failJob(job, Object.assign(
|
|
337
|
+
new Error(`explicit review blocked acceptance: verdict=${review.verdict}, reviewer status=${review.status}, review delivery_complete=${review.delivery_complete === true}`),
|
|
338
|
+
{ code },
|
|
339
|
+
));
|
|
340
|
+
return;
|
|
341
|
+
}
|
|
327
342
|
if (job.review) transition(job, JOB_PHASES.READY, 'review complete');
|
|
328
343
|
finalize(job);
|
|
329
344
|
return;
|
|
@@ -453,6 +468,24 @@ export function createWorkflowRuntime(adapters, {
|
|
|
453
468
|
const review = await runReviewerAttempt(job, reviewTask, config, before, job.execution_cwd, job.base_revision);
|
|
454
469
|
if (job.cancelling) { await cancelWorkflow(job); return; }
|
|
455
470
|
job.review = review;
|
|
471
|
+
// The review verdict is an acceptance gate, not a footnote: only a
|
|
472
|
+
// reviewer that ran to completion, delivered its own auditable
|
|
473
|
+
// contract, and approved may finalize as COMPLETED. Anything else
|
|
474
|
+
// fails closed through the existing FAILED machinery so the MCP
|
|
475
|
+
// surface reports a failed workflow instead of a false "done".
|
|
476
|
+
const reviewApproved = review.status === 'done'
|
|
477
|
+
&& review.delivery_complete === true
|
|
478
|
+
&& review.verdict === 'approve';
|
|
479
|
+
if (!reviewApproved && job.review_gate === 'required') {
|
|
480
|
+
const code = review.verdict === 'request_changes'
|
|
481
|
+
? 'REVIEW_CHANGES_REQUESTED'
|
|
482
|
+
: 'REVIEW_INCONCLUSIVE';
|
|
483
|
+
failJob(job, Object.assign(
|
|
484
|
+
new Error(`automatic review blocked acceptance: verdict=${review.verdict}, reviewer status=${review.status}, review delivery_complete=${review.delivery_complete === true}`),
|
|
485
|
+
{ code },
|
|
486
|
+
));
|
|
487
|
+
return;
|
|
488
|
+
}
|
|
456
489
|
transition(job, JOB_PHASES.READY, decision.reason);
|
|
457
490
|
finalize(job);
|
|
458
491
|
return;
|
|
@@ -535,8 +568,9 @@ export function createWorkflowRuntime(adapters, {
|
|
|
535
568
|
id: ar.id,
|
|
536
569
|
role: ar.role ?? 'worker',
|
|
537
570
|
attempt,
|
|
538
|
-
provider: ar.provider ?? null,
|
|
539
|
-
model: ar.model ?? null,
|
|
571
|
+
provider: ar.provider ?? null,
|
|
572
|
+
model: ar.model ?? null,
|
|
573
|
+
execution_context: ar.execution_context ?? null,
|
|
540
574
|
selection_source: ar.selection_source ?? null,
|
|
541
575
|
selection_trace: ar.selection_trace ?? null,
|
|
542
576
|
status: ar.status ?? 'failed',
|
|
@@ -594,7 +628,8 @@ export function createWorkflowRuntime(adapters, {
|
|
|
594
628
|
phase: job.phase,
|
|
595
629
|
status: job.status,
|
|
596
630
|
attempt: job.attempts.length,
|
|
597
|
-
current_model: job.attempts[job.attempts.length - 1]?.model ?? null,
|
|
631
|
+
current_model: job.attempts[job.attempts.length - 1]?.model ?? null,
|
|
632
|
+
execution_context: job.attempts[job.attempts.length - 1]?.execution_context ?? null,
|
|
598
633
|
model_class_hint: job.model_class_hint,
|
|
599
634
|
source: job.source,
|
|
600
635
|
profile_id: job.profile_id,
|
|
@@ -619,12 +654,14 @@ export function createWorkflowRuntime(adapters, {
|
|
|
619
654
|
workspace_retained: job.workspace_retained === true,
|
|
620
655
|
decision: job.decision,
|
|
621
656
|
candidate_available: !!job.candidate,
|
|
622
|
-
review_status: job.review ? job.review.status ?? null : null,
|
|
657
|
+
review_status: job.review ? job.review.status ?? null : null,
|
|
658
|
+
review_gate: job.review_gate ?? 'required',
|
|
623
659
|
event_cursor: job.canonical_events.at(-1)?.sequence ?? 0,
|
|
624
660
|
};
|
|
625
661
|
if (withResult) {
|
|
626
662
|
v.child_attempts = job.attempts.map((a) => ({
|
|
627
|
-
id: a.id, role: a.role, attempt: a.attempt, provider: a.provider, model: a.model,
|
|
663
|
+
id: a.id, role: a.role, attempt: a.attempt, provider: a.provider, model: a.model,
|
|
664
|
+
execution_context: a.execution_context ?? null,
|
|
628
665
|
selection_source: a.selection_source, selection_trace: a.selection_trace ?? null,
|
|
629
666
|
status: a.status, stopReason: a.stopReason,
|
|
630
667
|
outcome_task: a.outcome?.task_status ?? null,
|
|
@@ -661,10 +698,7 @@ export function createWorkflowRuntime(adapters, {
|
|
|
661
698
|
const job = jobs.get(id);
|
|
662
699
|
if (!job) return undefined;
|
|
663
700
|
if (job.status !== 'running') return job;
|
|
664
|
-
await
|
|
665
|
-
new Promise((res) => job.waiters.push(res)),
|
|
666
|
-
timeoutMs > 0 ? new Promise((res) => setTimeout(res, timeoutMs)) : new Promise(() => {}),
|
|
667
|
-
]);
|
|
701
|
+
await raceWaiters(job.waiters, { timeoutMs });
|
|
668
702
|
return job;
|
|
669
703
|
}
|
|
670
704
|
|
|
@@ -51,7 +51,7 @@ exit /b 64
|
|
|
51
51
|
|
|
52
52
|
:help
|
|
53
53
|
echo Usage: %~nx0 [--open ^| --background ^| --watch]
|
|
54
|
-
echo --open Start
|
|
55
|
-
echo --background Start
|
|
56
|
-
echo --watch Keep
|
|
54
|
+
echo --open Start the 3080 console and its supervised 3210 backend, then open the console.
|
|
55
|
+
echo --background Start the 3080 console and its supervised 3210 backend silently.
|
|
56
|
+
echo --watch Keep the 3080 console and supervised 3210 backend healthy.
|
|
57
57
|
exit /b 0
|
|
@@ -15,8 +15,8 @@ $logRoot = if ($env:TEMP) { $env:TEMP } else { [System.IO.Path]::GetTempPath() }
|
|
|
15
15
|
$launcherLog = Join-Path $logRoot 'dsh-crew-launcher.log'
|
|
16
16
|
$startedAt = Get-Date
|
|
17
17
|
$services = @(
|
|
18
|
-
[pscustomobject]@{ Name = '
|
|
19
|
-
[pscustomobject]@{ Name = '
|
|
18
|
+
[pscustomobject]@{ Name = 'Official UI'; Profile = 'web'; Home = $officialHome; Port = 3080; Url = 'http://127.0.0.1:3080'; ManagedByBridge = $false; State = 'pending'; Process = $null; RootPid = $null; RootStartedAtUtcTicks = $null; ListenerPid = $null; ListenerStartedAtUtcTicks = $null; ConsecutiveFailures = 0; LastError = $null },
|
|
19
|
+
[pscustomobject]@{ Name = 'Crew backend'; Profile = 'dsh-crew'; Home = $crewHome; Port = 3210; Url = 'http://127.0.0.1:3210'; ManagedByBridge = $true; State = 'pending'; Process = $null; RootPid = $null; RootStartedAtUtcTicks = $null; ListenerPid = $null; ListenerStartedAtUtcTicks = $null; ConsecutiveFailures = 0; LastError = $null }
|
|
20
20
|
)
|
|
21
21
|
|
|
22
22
|
function Write-LaunchLog {
|
|
@@ -44,6 +44,18 @@ function Get-HealthState {
|
|
|
44
44
|
}
|
|
45
45
|
}
|
|
46
46
|
|
|
47
|
+
function Start-BridgedCrewService {
|
|
48
|
+
try {
|
|
49
|
+
$response = Invoke-WebRequest -UseBasicParsing -Uri 'http://127.0.0.1:3080/_dsh/dsh-crew/ping' -TimeoutSec 5
|
|
50
|
+
if ($response.StatusCode -ge 200 -and $response.StatusCode -lt 300) {
|
|
51
|
+
return [pscustomobject]@{ Ready = $true; Error = $null }
|
|
52
|
+
}
|
|
53
|
+
return [pscustomobject]@{ Ready = $false; Error = ('3080 bridge returned HTTP {0}.' -f $response.StatusCode) }
|
|
54
|
+
} catch {
|
|
55
|
+
return [pscustomobject]@{ Ready = $false; Error = ('3080 bridge could not start the supervised 3210 backend: {0}' -f $_.Exception.Message) }
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
47
59
|
function Get-PortState {
|
|
48
60
|
param([int] $Port)
|
|
49
61
|
try {
|
|
@@ -175,7 +187,7 @@ function Wait-CrewServices {
|
|
|
175
187
|
if ($health.Ready) {
|
|
176
188
|
$service.State = 'ready'
|
|
177
189
|
$service.ConsecutiveFailures = 0
|
|
178
|
-
[void] (Set-TrackedListenerIdentity -Service $service)
|
|
190
|
+
if (-not $service.ManagedByBridge) { [void] (Set-TrackedListenerIdentity -Service $service) }
|
|
179
191
|
Write-LaunchLog ('{0} ready on {1}; runtime={2}' -f $service.Name, $service.Port, $health.Version)
|
|
180
192
|
} elseif ($service.Process -and $service.Process.HasExited) {
|
|
181
193
|
throw ('{0} exited before becoming ready; PID={1}; exit={2}; last health error: {3}' -f $service.Name, $service.Process.Id, $service.Process.ExitCode, $health.Error)
|
|
@@ -184,7 +196,9 @@ function Wait-CrewServices {
|
|
|
184
196
|
if (@($services | Where-Object State -eq 'starting').Count -gt 0) { Start-Sleep -Milliseconds 500 }
|
|
185
197
|
}
|
|
186
198
|
|
|
187
|
-
$notReady = @($services | Where-Object
|
|
199
|
+
$notReady = @($services | Where-Object {
|
|
200
|
+
$_.State -ne 'ready' -and -not ($_.ManagedByBridge -and $_.State -eq 'pending')
|
|
201
|
+
})
|
|
188
202
|
if ($notReady.Count -gt 0) {
|
|
189
203
|
$details = ($notReady | ForEach-Object { '{0}:{1} ({2})' -f $_.Profile, $_.Port, $_.LastError }) -join '; '
|
|
190
204
|
throw "Startup health deadline exceeded: $details"
|
|
@@ -193,21 +207,36 @@ function Wait-CrewServices {
|
|
|
193
207
|
|
|
194
208
|
function Ensure-CrewServices {
|
|
195
209
|
param([switch] $QuietHealthy)
|
|
210
|
+
$officialRestarted = $false
|
|
196
211
|
|
|
197
212
|
foreach ($service in $services) {
|
|
198
213
|
$health = Get-HealthState $service
|
|
214
|
+
if ($service.ManagedByBridge -and $officialRestarted) {
|
|
215
|
+
# A restarted 3080 process must re-establish ownership of 3210 through
|
|
216
|
+
# its own sidecar, even if an old detached listener still answers.
|
|
217
|
+
$service.State = 'pending'
|
|
218
|
+
$service.LastError = 'Waiting for the restarted 3080 bridge to claim 3210.'
|
|
219
|
+
continue
|
|
220
|
+
}
|
|
199
221
|
if ($health.Ready) {
|
|
200
222
|
$wasReady = $service.State -eq 'ready'
|
|
201
223
|
$service.State = 'ready'
|
|
202
224
|
$service.ConsecutiveFailures = 0
|
|
203
225
|
$service.LastError = $null
|
|
204
|
-
if (-not $service.ListenerPid) { [void] (Set-TrackedListenerIdentity -Service $service) }
|
|
226
|
+
if (-not $service.ManagedByBridge -and -not $service.ListenerPid) { [void] (Set-TrackedListenerIdentity -Service $service) }
|
|
205
227
|
if (-not $QuietHealthy -or -not $wasReady) {
|
|
206
228
|
Write-LaunchLog ('{0} already ready on {1}; runtime={2}' -f $service.Name, $service.Port, $health.Version)
|
|
207
229
|
}
|
|
208
230
|
continue
|
|
209
231
|
}
|
|
210
232
|
|
|
233
|
+
if ($service.ManagedByBridge) {
|
|
234
|
+
$service.LastError = $health.Error
|
|
235
|
+
$service.ConsecutiveFailures += 1
|
|
236
|
+
$service.State = 'pending'
|
|
237
|
+
continue
|
|
238
|
+
}
|
|
239
|
+
|
|
211
240
|
$service.LastError = $health.Error
|
|
212
241
|
$service.ConsecutiveFailures += 1
|
|
213
242
|
if ($service.Process -and $service.Process.HasExited) {
|
|
@@ -242,9 +271,24 @@ function Ensure-CrewServices {
|
|
|
242
271
|
Write-LaunchLog ('Supervisor detected {0} unavailable on {1}; restarting it.' -f $service.Name, $service.Port) 'WARN'
|
|
243
272
|
}
|
|
244
273
|
Start-CrewService $service
|
|
274
|
+
if (-not $service.ManagedByBridge) { $officialRestarted = $true }
|
|
245
275
|
}
|
|
246
276
|
|
|
247
277
|
Wait-CrewServices
|
|
278
|
+
|
|
279
|
+
# The 3080 official bridge is the sole owner of the 3210 child. Trigger it
|
|
280
|
+
# only after 3080 is healthy, then wait on the direct 3210 runtime contract.
|
|
281
|
+
foreach ($service in ($services | Where-Object ManagedByBridge -eq $true | Where-Object State -ne 'ready')) {
|
|
282
|
+
$bridge = Start-BridgedCrewService
|
|
283
|
+
if (-not $bridge.Ready) {
|
|
284
|
+
$service.LastError = $bridge.Error
|
|
285
|
+
throw $bridge.Error
|
|
286
|
+
}
|
|
287
|
+
$service.State = 'starting'
|
|
288
|
+
$service.ConsecutiveFailures = 0
|
|
289
|
+
$service.LastError = $null
|
|
290
|
+
}
|
|
291
|
+
if (@($services | Where-Object State -eq 'starting').Count -gt 0) { Wait-CrewServices }
|
|
248
292
|
}
|
|
249
293
|
|
|
250
294
|
function Start-ServiceSupervisor {
|