@ran-sh/dsh-crew 2.1.1 → 2.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +4 -2
- package/package.json +1 -1
- package/src/history/archive-store.mjs +46 -6
- package/src/history/cleanup-plan.mjs +40 -8
- package/src/history/operation.mjs +31 -1
- package/src/history/runner.mjs +22 -6
- package/src/history/service.mjs +12 -5
- package/src/install/npx-lifecycle.mjs +6 -2
- package/src/install/windows-supervisor-adapter.mjs +5 -2
- package/src/runtime-identity.mjs +1 -1
- package/windows/start-dsh-crew.cmd +6 -3
- package/windows/start-dsh-crew.ps1 +169 -19
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-crew",
|
|
3
|
-
"version": "2.1.
|
|
3
|
+
"version": "2.1.3",
|
|
4
4
|
"description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "ZSeven-W"
|
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
"mcpServers": {
|
|
9
9
|
"dsh-crew": {
|
|
10
10
|
"command": "node",
|
|
11
|
-
"args": [
|
|
11
|
+
"args": [
|
|
12
|
+
"${CLAUDE_PLUGIN_ROOT}/src/server.mjs"
|
|
13
|
+
]
|
|
12
14
|
}
|
|
13
15
|
}
|
|
14
16
|
}
|
package/package.json
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { createHash, randomUUID } from 'node:crypto';
|
|
2
|
-
import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
2
|
+
import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
4
4
|
|
|
5
5
|
const WORKSPACE = 'harness/storages/workspace.json';
|
|
@@ -121,16 +121,52 @@ async function requireStopped(assertStopped) {
|
|
|
121
121
|
if (typeof assertStopped !== 'function' || await assertStopped() !== true) fail('BACKEND_NOT_STOPPED');
|
|
122
122
|
}
|
|
123
123
|
|
|
124
|
+
/**
|
|
125
|
+
* Every session id that still has an artifact under `harness/sessions`.
|
|
126
|
+
*
|
|
127
|
+
* A request may name sessions as "already gone" so that a workspace they belong
|
|
128
|
+
* to stops being unreachable. That claim decides whether a record is deleted, so
|
|
129
|
+
* it is proven here rather than trusted: a live session named as absent would
|
|
130
|
+
* drop its workspace while leaving the artifact behind.
|
|
131
|
+
*/
|
|
132
|
+
function presentSessionIds(root) {
|
|
133
|
+
const sessionsRoot = pathInside(root, 'harness/sessions');
|
|
134
|
+
let projects;
|
|
135
|
+
try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
|
|
136
|
+
if (projects.length > 20000) fail('INVALID_FILE');
|
|
137
|
+
const present = new Set();
|
|
138
|
+
for (const project of projects) {
|
|
139
|
+
if (!project.isDirectory() || project.isSymbolicLink()) continue;
|
|
140
|
+
let entries;
|
|
141
|
+
try { entries = readdirSync(join(sessionsRoot, project.name), { withFileTypes: true }); } catch { continue; }
|
|
142
|
+
for (const entry of entries) {
|
|
143
|
+
if (!entry.isDirectory() || !validId(entry.name)) continue;
|
|
144
|
+
const directory = join(sessionsRoot, project.name, entry.name);
|
|
145
|
+
if (entry.isSymbolicLink()) { present.add(entry.name); continue; }
|
|
146
|
+
let files = [];
|
|
147
|
+
try { files = readdirSync(directory); } catch { present.add(entry.name); continue; }
|
|
148
|
+
if (files.some(name => /^session(?:\.v[1-9][0-9]*)?\.jsonl(?:\.zstd)?$/.test(name))) present.add(entry.name);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return present;
|
|
152
|
+
}
|
|
153
|
+
|
|
124
154
|
/** Internal offline primitive: caller must own the maintenance lease + update lock. */
|
|
125
155
|
export async function archiveHistory({ crewRoot, request, assertStopped, archiveId = randomUUID() }) {
|
|
126
156
|
await requireStopped(assertStopped);
|
|
127
157
|
if (!request || !['archive', 'delete'].includes(request.operation) || !Array.isArray(request.artifacts)
|
|
128
158
|
|| !Array.isArray(request.sessionIds) || !Array.isArray(request.workspaceIds)
|
|
129
|
-
||
|
|
159
|
+
|| !Array.isArray(request.absentSessionIds ?? [])
|
|
160
|
+
|| request.artifacts.length > 10000 || request.workspaceIds.length > 10000
|
|
161
|
+
|| (request.absentSessionIds ?? []).length > 10000) fail('INVALID_REQUEST');
|
|
130
162
|
const selectedSessions = new Set(request.sessionIds);
|
|
131
163
|
const selectedWorkspaces = new Set(request.workspaceIds);
|
|
132
|
-
|
|
164
|
+
const absentSessions = new Set(request.absentSessionIds ?? []);
|
|
165
|
+
const removedSessions = new Set([...selectedSessions, ...absentSessions]);
|
|
166
|
+
if ([...removedSessions, ...selectedWorkspaces].some(id => !validId(id))
|
|
133
167
|
|| selectedSessions.size !== request.sessionIds.length || selectedWorkspaces.size !== request.workspaceIds.length
|
|
168
|
+
|| absentSessions.size !== (request.absentSessionIds ?? []).length
|
|
169
|
+
|| absentSessions.size !== removedSessions.size - selectedSessions.size
|
|
134
170
|
|| request.artifacts.length !== selectedSessions.size
|
|
135
171
|
|| new Set(request.artifacts.map(f => f.sessionId)).size !== selectedSessions.size
|
|
136
172
|
|| request.artifacts.some(f => !selectedSessions.has(f.sessionId))) fail('INVALID_SELECTION');
|
|
@@ -140,12 +176,16 @@ export async function archiveHistory({ crewRoot, request, assertStopped, archive
|
|
|
140
176
|
const after = structuredClone(before);
|
|
141
177
|
for (const id of selectedWorkspaces) {
|
|
142
178
|
const record = before.tables.workspaces[id];
|
|
143
|
-
if (!record || record.sessionIds.some(sid => !
|
|
179
|
+
if (!record || record.sessionIds.some(sid => !removedSessions.has(sid))) fail('PREVIEW_CHANGED');
|
|
144
180
|
delete after.tables.workspaces[id];
|
|
145
181
|
}
|
|
182
|
+
if (absentSessions.size > 0) {
|
|
183
|
+
const live = presentSessionIds(crewRoot);
|
|
184
|
+
for (const id of absentSessions) if (live.has(id)) fail('PREVIEW_CHANGED');
|
|
185
|
+
}
|
|
146
186
|
after.global.workspaceIds = after.global.workspaceIds.filter(id => !selectedWorkspaces.has(id));
|
|
147
|
-
after.global.archivedSessionIds = after.global.archivedSessionIds.filter(id => !
|
|
148
|
-
for (const record of Object.values(after.tables.workspaces)) record.sessionIds = record.sessionIds.filter(id => !
|
|
187
|
+
after.global.archivedSessionIds = after.global.archivedSessionIds.filter(id => !removedSessions.has(id));
|
|
188
|
+
for (const record of Object.values(after.tables.workspaces)) record.sessionIds = record.sessionIds.filter(id => !removedSessions.has(id));
|
|
149
189
|
let size = 0;
|
|
150
190
|
const files = request.artifacts.map(file => {
|
|
151
191
|
const bytes = readBounded(artifactPath(crewRoot, file));
|
|
@@ -53,31 +53,63 @@ export function planHistoryCleanup(snapshot, { operation = 'archive', scope = 'c
|
|
|
53
53
|
};
|
|
54
54
|
const withinTime = (row) => cutoff === null
|
|
55
55
|
|| (instant(row.createdAt) !== null && instant(row.createdAt) < cutoff);
|
|
56
|
+
const present = new Set(sessions.map(row => row.id));
|
|
57
|
+
const worktreeSessions = new Set(sessions.filter(row => row.worktree === true).map(row => row.id));
|
|
56
58
|
const selected = new Set(sessions.filter(row => admitted(row) && withinTime(row)).map(row => row.id));
|
|
57
59
|
// A workspace carries no provenance of its own: it follows its sessions, and
|
|
58
60
|
// only when every one of them is selected. Checking `admitted(row)` here would
|
|
59
61
|
// test a workspace id against a session ledger and always fail. A provenance
|
|
60
|
-
// scope additionally requires at least one
|
|
61
|
-
//
|
|
62
|
+
// scope additionally requires at least one child — an empty workspace holds no
|
|
63
|
+
// Crew work, so it is not Crew's to remove; `all` keeps its historical
|
|
62
64
|
// behaviour of following an empty workspace.
|
|
65
|
+
//
|
|
66
|
+
// A child whose artifact is already gone cannot be "selected": there is
|
|
67
|
+
// nothing left to select. Such a child still counts as covered when the ledger
|
|
68
|
+
// recorded Crew creating it, the same evidence that makes a live one Crew's.
|
|
69
|
+
// Without that, a workspace whose sessions were removed by an earlier cleanup
|
|
70
|
+
// became permanently unreachable — every scope demanded its children, and its
|
|
71
|
+
// children no longer existed — which is exactly how 48 dead rows stayed in the
|
|
72
|
+
// store while the operator could only see them.
|
|
63
73
|
const requiresOwnedChild = scope === 'crew' || scope === 'worktree';
|
|
64
|
-
|
|
74
|
+
// One decision per workspace, from that row alone: what a row is allowed to
|
|
75
|
+
// remove must never depend on which rows were visited before it.
|
|
76
|
+
const decideWorkspace = (row) => {
|
|
65
77
|
const children = ids(row.sessionIds);
|
|
66
|
-
|
|
67
|
-
|
|
78
|
+
if (children.length === 0) return { ok: !requiresOwnedChild, absent: [] };
|
|
79
|
+
// Worktree scope marks a live child by its session header. A gone child has
|
|
80
|
+
// no header left, so the workspace's own path under the worktree root is the
|
|
81
|
+
// only marker there is.
|
|
82
|
+
if (scope === 'worktree' && row.worktree !== true && !children.every(id => worktreeSessions.has(id))) return { ok: false, absent: [] };
|
|
83
|
+
const gone = [];
|
|
84
|
+
for (const id of children) {
|
|
85
|
+
if (selected.has(id)) continue;
|
|
86
|
+
// Kept back by the scope, or gone with no record that Crew made it.
|
|
87
|
+
if (present.has(id) || !origin.has(id)) return { ok: false, absent: [] };
|
|
88
|
+
gone.push(id);
|
|
89
|
+
}
|
|
90
|
+
return { ok: true, absent: gone };
|
|
91
|
+
};
|
|
92
|
+
const absent = new Set();
|
|
93
|
+
const workspaceIds = workspaces.filter(row => {
|
|
94
|
+
if (!withinTime(row)) return false;
|
|
95
|
+
const decision = decideWorkspace(row);
|
|
96
|
+
if (!decision.ok) return false;
|
|
97
|
+
for (const id of decision.absent) absent.add(id);
|
|
98
|
+
return true;
|
|
68
99
|
}).map(row => row.id).sort();
|
|
100
|
+
const absentSessionIds = [...absent].sort();
|
|
69
101
|
const sessionIds = [...selected].sort();
|
|
70
102
|
const signature = {
|
|
71
103
|
operation, scope, cutoff,
|
|
72
|
-
workspaces: workspaces.map(row => ({ id: row.id, createdAt: row.createdAt, updatedAt: row.updatedAt, sessionIds: ids(row.sessionIds) })),
|
|
73
|
-
sessions: sessions.map(row => ({ id: row.id, createdAt: row.createdAt, revision: row.revision })),
|
|
104
|
+
workspaces: workspaces.map(row => ({ id: row.id, createdAt: row.createdAt, updatedAt: row.updatedAt, sessionIds: ids(row.sessionIds), worktree: row.worktree === true })),
|
|
105
|
+
sessions: sessions.map(row => ({ id: row.id, createdAt: row.createdAt, revision: row.revision, worktree: row.worktree === true })),
|
|
74
106
|
active,
|
|
75
107
|
crewSessionIds: [...origin].sort(),
|
|
76
108
|
};
|
|
77
109
|
return {
|
|
78
110
|
schemaVersion: 1, operation, scope,
|
|
79
111
|
before: cutoff === null ? null : new Date(cutoff).toISOString(),
|
|
80
|
-
timeBasis: 'createdAt', workspaceIds, sessionIds,
|
|
112
|
+
timeBasis: 'createdAt', workspaceIds, sessionIds, absentSessionIds,
|
|
81
113
|
counts: { workspaces: workspaceIds.length, sessions: sessionIds.length },
|
|
82
114
|
protectedCounts: { workspaces: workspaces.length - workspaceIds.length, sessions: sessions.length - sessionIds.length },
|
|
83
115
|
executable: active.length === 0 && (workspaceIds.length > 0 || sessionIds.length > 0),
|
|
@@ -19,6 +19,17 @@ function verifyDisk(root, manifest) {
|
|
|
19
19
|
}
|
|
20
20
|
}
|
|
21
21
|
|
|
22
|
+
function removedWorkspaceIds(root, archiveId) {
|
|
23
|
+
const manifest = readHistoryManifest(root, archiveId);
|
|
24
|
+
return Object.keys(manifest.before.tables.workspaces).filter(id => !Object.hasOwn(manifest.after.tables.workspaces, id));
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function workspacesStillPresent(root, ids) {
|
|
28
|
+
if (ids.length === 0) return [];
|
|
29
|
+
const current = decodeWorkspaceStore(readHistoryBytes(historyPath(root, 'harness/storages/workspace.json')));
|
|
30
|
+
return ids.filter(id => Object.hasOwn(current.tables.workspaces, id));
|
|
31
|
+
}
|
|
32
|
+
|
|
22
33
|
/** Detached executor core; injected boundaries make the real transaction testable. */
|
|
23
34
|
export async function runHistoryOperation({ crewRoot, id, acquire, release, supervisor, checkFence,
|
|
24
35
|
assertStopped, verifyRunning, recover = false }) {
|
|
@@ -46,7 +57,11 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
|
|
|
46
57
|
if (!alreadyStarted) {
|
|
47
58
|
if (!recover || await assertStopped(state) !== true) await checkFence(state);
|
|
48
59
|
save('STOPPING');
|
|
49
|
-
|
|
60
|
+
// `refresh_frontend`: this operation rewrites `storages/workspace.json`,
|
|
61
|
+
// which the Crew-managed frontend on 3080 shares, so the launcher stops
|
|
62
|
+
// that server for the same window and starts it again afterwards. The npx
|
|
63
|
+
// lifecycle stops only 3210, because a tree swap leaves that file alone.
|
|
64
|
+
const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId, refresh_frontend: true });
|
|
50
65
|
if (!stopped?.ok || await assertStopped(state) !== true) throw Error('HISTORY_STOP_NOT_VERIFIED');
|
|
51
66
|
const archiveId = state.operation === 'restore' ? state.archiveId : state.id;
|
|
52
67
|
const manifestFile = historyPath(crewRoot, `history/transactions/${archiveId}/manifest.json`);
|
|
@@ -73,6 +88,21 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
|
|
|
73
88
|
}
|
|
74
89
|
save('VERIFYING');
|
|
75
90
|
if (await verifyRunning(state) !== true) throw Error('HISTORY_RESTART_NOT_VERIFIED');
|
|
91
|
+
// The store this operation rewrote is shared with every other DSH server on
|
|
92
|
+
// the same home, and each of them writes the whole file back from memory. A
|
|
93
|
+
// process that stayed live across the window puts back exactly what was
|
|
94
|
+
// removed, and from here that is indistinguishable from success: the
|
|
95
|
+
// manifest is applied and the runtime is up. Ask the disk instead, and say
|
|
96
|
+
// so rather than report a cleanup that did not happen.
|
|
97
|
+
if (!state.rolledBack && state.operation !== 'restore') {
|
|
98
|
+
const returned = workspacesStillPresent(crewRoot, removedWorkspaceIds(crewRoot, state.archiveId ?? state.id));
|
|
99
|
+
if (returned.length > 0) {
|
|
100
|
+
state = { ...state, phase: 'FAILED', code: 'HISTORY_STORE_CHANGED_AFTER_APPLY',
|
|
101
|
+
counts: { ...(state.counts ?? {}), returned: returned.length } };
|
|
102
|
+
writeHistoryState(crewRoot, state);
|
|
103
|
+
return state;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
76
106
|
if (state.operation === 'delete' && !state.rolledBack) {
|
|
77
107
|
await finalizeHistoryDeletion({ crewRoot, archiveId: state.archiveId ?? state.id, assertRestarted: () => verifyRunning(state) });
|
|
78
108
|
}
|
package/src/history/runner.mjs
CHANGED
|
@@ -8,9 +8,9 @@ import { runHistoryOperation } from './operation.mjs';
|
|
|
8
8
|
import { readHistoryState } from './state.mjs';
|
|
9
9
|
import { TARGET_DSH_VERSION } from '../dsh-cohort.mjs';
|
|
10
10
|
|
|
11
|
-
async function portFree() {
|
|
11
|
+
async function portFree(port = 3210) {
|
|
12
12
|
return new Promise(resolve => {
|
|
13
|
-
const socket = createConnection({ host: '127.0.0.1', port
|
|
13
|
+
const socket = createConnection({ host: '127.0.0.1', port });
|
|
14
14
|
socket.setTimeout(2000);
|
|
15
15
|
socket.once('connect', () => { socket.destroy(); resolve(false); });
|
|
16
16
|
socket.once('timeout', () => { socket.destroy(); resolve(false); });
|
|
@@ -18,6 +18,23 @@ async function portFree() {
|
|
|
18
18
|
});
|
|
19
19
|
}
|
|
20
20
|
|
|
21
|
+
/**
|
|
22
|
+
* The stopped window, proven from outside the launcher that reported it.
|
|
23
|
+
*
|
|
24
|
+
* `storages/workspace.json` is rewritten by this operation and is held in memory
|
|
25
|
+
* by every DSH server on the home, so the window has to cover more than the hub:
|
|
26
|
+
* the Crew-managed frontend on 3080 shares that home and is stopped for the same
|
|
27
|
+
* window (`refresh_frontend`). Both ports are re-probed here rather than taken
|
|
28
|
+
* on the launcher's word, and the session must be the one this transaction
|
|
29
|
+
* stopped — a lease alone never means the servers are gone.
|
|
30
|
+
*/
|
|
31
|
+
export async function stoppedWindowIsClean({ session, lease, runtimeId, probe = portFree }) {
|
|
32
|
+
if (!session?.ok || session.state !== 'present') return false;
|
|
33
|
+
if (session.session?.lease !== lease || session.session?.runtime_id !== runtimeId) return false;
|
|
34
|
+
if (!await probe(3210)) return false;
|
|
35
|
+
return session.session?.frontend_stopped === true ? await probe(3080) : true;
|
|
36
|
+
}
|
|
37
|
+
|
|
21
38
|
export async function runProductionHistory({ id, recover = false } = {}) {
|
|
22
39
|
if (process.platform !== 'win32') throw Error('HISTORY_PLATFORM_UNSUPPORTED');
|
|
23
40
|
const home = homedir(); const crewRoot = join(home, '.config', 'dsh-crew');
|
|
@@ -31,10 +48,9 @@ export async function runProductionHistory({ id, recover = false } = {}) {
|
|
|
31
48
|
const response = await fetch('http://127.0.0.1:3210/_dsh/dsh-crew/history/fenced-check', { method: 'POST', headers: { 'content-type': 'application/json' }, body: '{}', signal: AbortSignal.timeout(30000) });
|
|
32
49
|
if (!response.ok || (await response.json()).ok !== true) throw Error('HISTORY_FENCE_NOT_IDLE');
|
|
33
50
|
},
|
|
34
|
-
assertStopped: async s => {
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
},
|
|
51
|
+
assertStopped: async s => stoppedWindowIsClean({
|
|
52
|
+
session: readMaintenanceSession(crewRoot), lease: s.lease, runtimeId: s.runtimeId,
|
|
53
|
+
}),
|
|
38
54
|
verifyRunning: async s => {
|
|
39
55
|
try {
|
|
40
56
|
const response = await fetch('http://127.0.0.1:3210/_dsh/dsh-crew/runtime', { signal: AbortSignal.timeout(3000) });
|
package/src/history/service.mjs
CHANGED
|
@@ -77,12 +77,15 @@ export function createHistoryService({ crewRoot, agents, persistence, runtimeId,
|
|
|
77
77
|
for (const row of sessions) if (crewSet.has(row.id)) row.crew = true;
|
|
78
78
|
// The isolated-workspace marker: the hub stamps every worktree session with a
|
|
79
79
|
// cwd under the Crew worktree root, so that subset is scopeable on its own.
|
|
80
|
+
// A workspace carries the same marker from its own path, which is all that is
|
|
81
|
+
// left of a workspace whose sessions were already removed.
|
|
80
82
|
const worktreeRoot = defaultWorktreeRoot().replaceAll('\\', '/').toLowerCase();
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
83
|
+
const underWorktreeRoot = value => typeof value === 'string' && value.replaceAll('\\', '/').toLowerCase().startsWith(worktreeRoot);
|
|
84
|
+
for (const row of sessions) if (row.crew === true && underWorktreeRoot(row.cwd)) row.worktree = true;
|
|
85
|
+
for (const row of workspaces) if (underWorktreeRoot(row.path)) row.worktree = true;
|
|
84
86
|
const plan = planHistoryCleanup({ workspaces, sessions, crewSessionIds, activeSessionIds: gate.idle() ? [] : ['active-agent'] }, options);
|
|
85
87
|
const selected = new Set(plan.sessionIds);
|
|
88
|
+
const alreadyGone = new Set(plan.absentSessionIds ?? []);
|
|
86
89
|
const byId = new Map(sessions.map(row => [row.id, row]));
|
|
87
90
|
const queue = sessions.filter(row => !selected.has(row.id));
|
|
88
91
|
for (let i = 0; i < queue.length; i++) {
|
|
@@ -90,12 +93,16 @@ export function createHistoryService({ crewRoot, agents, persistence, runtimeId,
|
|
|
90
93
|
if (parent && selected.delete(parent.id)) queue.push(parent);
|
|
91
94
|
}
|
|
92
95
|
plan.sessionIds = plan.sessionIds.filter(id => selected.has(id));
|
|
93
|
-
|
|
96
|
+
// Same rule the plan used, re-applied after the ancestor closure dropped
|
|
97
|
+
// sessions from the selection: a workspace follows its children, and a child
|
|
98
|
+
// the plan proved Crew's own but already removed still counts.
|
|
99
|
+
plan.workspaceIds = plan.workspaceIds.filter(id => store.tables.workspaces[id].sessionIds.every(sid => selected.has(sid) || alreadyGone.has(sid)));
|
|
94
100
|
plan.counts = { workspaces: plan.workspaceIds.length, sessions: plan.sessionIds.length };
|
|
95
101
|
plan.executable = gate.idle() && plan.counts.workspaces + plan.counts.sessions > 0;
|
|
96
102
|
if (!plan.executable && !plan.blockedReason) plan.blockedReason = 'EMPTY_SELECTION';
|
|
97
103
|
const request = { operation: plan.operation, workspaceHash: historyHash(bytes), workspaceIds: plan.workspaceIds,
|
|
98
|
-
sessionIds: plan.sessionIds,
|
|
104
|
+
sessionIds: plan.sessionIds, absentSessionIds: plan.absentSessionIds ?? [],
|
|
105
|
+
artifacts: sessions.filter(row => selected.has(row.id)).map(row => row.artifact) };
|
|
99
106
|
const revision = historyHash(JSON.stringify([runtimeId, plan.revision, request]));
|
|
100
107
|
return { plan: { ...plan, revision, items: workspaces.filter(row => plan.workspaceIds.includes(row.id)).slice(0, 100).map(row => ({ id: row.id, title: row.title })) }, request };
|
|
101
108
|
}
|
|
@@ -1639,7 +1639,7 @@ export function createCrewSupervisor({
|
|
|
1639
1639
|
}
|
|
1640
1640
|
}
|
|
1641
1641
|
return {
|
|
1642
|
-
stopOwnedBackend: async ({ lease = null, runtimeId = null } = {}) => {
|
|
1642
|
+
stopOwnedBackend: async ({ lease = null, runtimeId = null, refreshFrontend = false } = {}) => {
|
|
1643
1643
|
const { readMaintenanceSession } = await import('../supervisor/restart-request.mjs');
|
|
1644
1644
|
const durable = readMaintenanceSession(appRoot);
|
|
1645
1645
|
if (!durable.ok) {
|
|
@@ -1675,7 +1675,11 @@ export function createCrewSupervisor({
|
|
|
1675
1675
|
lease: lease ?? `txn-${now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`,
|
|
1676
1676
|
runtime_id: identity.runtime_id,
|
|
1677
1677
|
};
|
|
1678
|
-
const result = await writeMaintenance({ operation: 'maintenance-stop', lease: transaction.lease, identity
|
|
1678
|
+
const result = await writeMaintenance({ operation: 'maintenance-stop', lease: transaction.lease, identity,
|
|
1679
|
+
// A history maintenance rewrites the workspace store the managed 3080
|
|
1680
|
+
// frontend shares, so it asks for that server too. A runtime-tree swap
|
|
1681
|
+
// does not, and leaves the flag unset.
|
|
1682
|
+
extra: refreshFrontend ? { refresh_frontend: true } : null });
|
|
1679
1683
|
if (!result.ok) {
|
|
1680
1684
|
// A failed stop must not leave a usable lease behind: clear it so a
|
|
1681
1685
|
// later startOwnedBackend() fails closed instead of presenting a
|
|
@@ -281,8 +281,11 @@ export function createWindowsSupervisorHandoffHooks({
|
|
|
281
281
|
verifyExactWatcher: async ({ expected, role }) => inspect(expected, {
|
|
282
282
|
allowHelperDrift: role === 'old' && !sameHash(expected?.helper_hash, target.helper_hash),
|
|
283
283
|
}),
|
|
284
|
-
maintenanceStop: async ({ lease, runtime_id: runtimeId }) => maintenanceClient?.stopOwnedBackend?.({
|
|
285
|
-
|
|
284
|
+
maintenanceStop: async ({ lease, runtime_id: runtimeId, refresh_frontend: refreshFrontend = false }) => maintenanceClient?.stopOwnedBackend?.({
|
|
285
|
+
lease,
|
|
286
|
+
runtimeId,
|
|
287
|
+
...(refreshFrontend ? { refreshFrontend: true } : {}),
|
|
288
|
+
}) ?? { ok: false, code: 'SUPERVISOR_MAINTENANCE_UNAVAILABLE' },
|
|
286
289
|
maintenanceStatus: async ({ lease, runtime_id: runtimeId }) => {
|
|
287
290
|
const durable = readMaintenanceSession(appRoot);
|
|
288
291
|
if (!durable.ok) return durable;
|
package/src/runtime-identity.mjs
CHANGED
|
@@ -36,7 +36,7 @@ export {
|
|
|
36
36
|
// included in the identity contract.
|
|
37
37
|
const RUNTIME_ID = randomUUID();
|
|
38
38
|
|
|
39
|
-
export const RUNTIME_VERSION = '2.1.
|
|
39
|
+
export const RUNTIME_VERSION = '2.1.3';
|
|
40
40
|
export const HUB_PROTOCOL_VERSION = 1;
|
|
41
41
|
|
|
42
42
|
export const HUB_CAPABILITIES = Object.freeze([
|
|
@@ -35,13 +35,15 @@ if not exist "%LAUNCH_HELPER%" (
|
|
|
35
35
|
>>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
|
|
36
36
|
echo ERROR: DSH Crew launcher helper is missing.
|
|
37
37
|
echo Repair it with: dsh-crew update
|
|
38
|
-
if /i "%LAUNCH_MODE%"=="open" pause
|
|
38
|
+
if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
|
|
39
39
|
exit /b 1
|
|
40
40
|
)
|
|
41
41
|
|
|
42
42
|
powershell.exe -NoLogo -NoProfile -NonInteractive -ExecutionPolicy Bypass -File "%LAUNCH_HELPER%" -Mode "%LAUNCH_MODE%"
|
|
43
43
|
set "LAUNCH_EXIT=%ERRORLEVEL%"
|
|
44
|
-
|
|
44
|
+
rem DSH_CREW_LAUNCHER_NO_PAUSE: a wrapper that reports the failure itself asks
|
|
45
|
+
rem for the pause to be skipped here, so the operator presses a key once.
|
|
46
|
+
if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
|
|
45
47
|
exit /b %LAUNCH_EXIT%
|
|
46
48
|
|
|
47
49
|
:invalid_argument
|
|
@@ -51,7 +53,8 @@ exit /b 64
|
|
|
51
53
|
|
|
52
54
|
:help
|
|
53
55
|
echo Usage: %~nx0 [--open ^| --background ^| --watch]
|
|
54
|
-
echo --open Open official Harness on 3080
|
|
56
|
+
echo --open Open official Harness on 3080 and return once Crew is supervised;
|
|
57
|
+
echo 3210 keeps starting in the background.
|
|
55
58
|
echo --background Start the Crew-owned 3210 service silently.
|
|
56
59
|
echo --watch Keep the Crew-owned 3210 service healthy.
|
|
57
60
|
exit /b 0
|
|
@@ -227,7 +227,7 @@ function Test-CrewWebSessionUrl {
|
|
|
227
227
|
}
|
|
228
228
|
|
|
229
229
|
function Open-CrewManagedFrontend {
|
|
230
|
-
param([int] $TimeoutSeconds = 90)
|
|
230
|
+
param([int] $TimeoutSeconds = 90, [switch] $Quiet)
|
|
231
231
|
# The 3080 frontend boots from the same Crew-managed Harness entry as 3210,
|
|
232
232
|
# against that entry's own DSH_HOME (npm runtime -> Crew home, source cohort
|
|
233
233
|
# -> its own tree). No official ~/.dsh state is read or written here.
|
|
@@ -253,7 +253,7 @@ function Open-CrewManagedFrontend {
|
|
|
253
253
|
if (-not (Test-OfficialWebReady)) { throw 'The Crew-managed 3080 frontend is not ready; its process was left running.' }
|
|
254
254
|
$existingUrl = Get-CrewWebSessionUrl
|
|
255
255
|
if ($existingUrl -and (Test-CrewWebSessionUrl -Url $existingUrl)) {
|
|
256
|
-
Open-CrewBrowserUrl -Url $existingUrl
|
|
256
|
+
if (-not $Quiet) { Open-CrewBrowserUrl -Url $existingUrl }
|
|
257
257
|
Write-LaunchLog 'Opened the existing Crew-managed Harness frontend on 3080 with its session URL.'
|
|
258
258
|
return $true
|
|
259
259
|
}
|
|
@@ -289,7 +289,10 @@ function Open-CrewManagedFrontend {
|
|
|
289
289
|
}
|
|
290
290
|
if (Test-OfficialWebReady) {
|
|
291
291
|
$startedUrl = Get-CrewWebSessionUrl -OutputLog $stdout
|
|
292
|
-
if ($startedUrl) {
|
|
292
|
+
if ($startedUrl) {
|
|
293
|
+
if (-not $Quiet) { Open-CrewBrowserUrl -Url $startedUrl }
|
|
294
|
+
return $true
|
|
295
|
+
}
|
|
293
296
|
}
|
|
294
297
|
}
|
|
295
298
|
if ($process.HasExited) { throw ('Crew-managed Harness exited before readiness. Diagnostic log: {0}' -f $stderr) }
|
|
@@ -302,8 +305,7 @@ function Open-CrewManagedFrontend {
|
|
|
302
305
|
}
|
|
303
306
|
}
|
|
304
307
|
|
|
305
|
-
function Test-OfficialWebReady {
|
|
306
|
-
try {
|
|
308
|
+
function Test-OfficialWebReady { try {
|
|
307
309
|
$response = Invoke-WebRequest -UseBasicParsing -Uri 'http://127.0.0.1:3080/' -TimeoutSec 2
|
|
308
310
|
return $response.StatusCode -ge 200 -and $response.StatusCode -lt 300
|
|
309
311
|
} catch {
|
|
@@ -313,6 +315,63 @@ function Test-OfficialWebReady {
|
|
|
313
315
|
}
|
|
314
316
|
}
|
|
315
317
|
|
|
318
|
+
# ---- The other server on this DSH home -------------------------------------
|
|
319
|
+
# The Crew-managed frontend on 3080 boots from the same Crew-managed entry as the
|
|
320
|
+
# hub, so it runs on the same DSH home: same sessions, same settings, and the
|
|
321
|
+
# same `storages/workspace.json`. DSH's JSON storage replaces that whole file and
|
|
322
|
+
# lets the last writer win, so an external rewrite of it only sticks while no
|
|
323
|
+
# server holds the file in memory. A history maintenance is exactly such an
|
|
324
|
+
# external rewrite, which is why it asks for this server to be stopped too —
|
|
325
|
+
# stopping the hub alone is what let a cleanup be undone minutes later.
|
|
326
|
+
function Stop-CrewManagedFrontend {
|
|
327
|
+
param([int] $TimeoutSeconds = 15)
|
|
328
|
+
$port = Get-PortState -Port 3080
|
|
329
|
+
if ($port.State -eq 'free') { return $true }
|
|
330
|
+
# An unenumerable listener is not provably free: fail closed rather than write
|
|
331
|
+
# under a server that might hold the very file being rewritten.
|
|
332
|
+
if ($port.State -ne 'occupied' -or -not $port.Pid) { return $false }
|
|
333
|
+
$ours = $false
|
|
334
|
+
if ($dshCliIsNodeEntry) {
|
|
335
|
+
$official = [pscustomobject]@{ NodePath = $dshCommand; Entry = $dshCli; Profile = 'web' }
|
|
336
|
+
$ours = Test-OfficialHarnessListener -OwnerPid ([int] $port.Pid) -Official $official -Profile 'web'
|
|
337
|
+
}
|
|
338
|
+
if (-not $ours) {
|
|
339
|
+
# Crew did not start this listener, so it cannot be proven to share the
|
|
340
|
+
# workspace store. A Crew-patched one is on a Crew home either way and is
|
|
341
|
+
# refused; the legacy official frontend and anything unrelated keep running,
|
|
342
|
+
# exactly as the start path leaves a foreign 3080 alone.
|
|
343
|
+
$patched = $false
|
|
344
|
+
try {
|
|
345
|
+
$probe = Invoke-RestMethod -Uri 'http://127.0.0.1:3080/_dsh/dsh-crew/bridge-status' -TimeoutSec 2
|
|
346
|
+
$patched = $probe.surface -eq 'official-bridge'
|
|
347
|
+
} catch { $patched = $false }
|
|
348
|
+
if ($patched) { return $false }
|
|
349
|
+
Write-LaunchLog 'A 3080 listener that is not the Crew-managed frontend was left running; it does not serve this DSH home.' 'WARN'
|
|
350
|
+
return $true
|
|
351
|
+
}
|
|
352
|
+
try { Stop-Process -Id ([int] $port.Pid) -Force -ErrorAction Stop } catch { return $false }
|
|
353
|
+
$deadline = (Get-Date).AddSeconds($TimeoutSeconds)
|
|
354
|
+
do {
|
|
355
|
+
$port = Get-PortState -Port 3080
|
|
356
|
+
if ($port.State -eq 'free') { return $true }
|
|
357
|
+
Start-Sleep -Milliseconds 250
|
|
358
|
+
} while ((Get-Date) -lt $deadline)
|
|
359
|
+
Write-LaunchLog ('The Crew-managed frontend on 3080 did not release the port within {0}s.' -f $TimeoutSeconds) 'WARN'
|
|
360
|
+
return $false
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
function Start-CrewManagedFrontendQuietly {
|
|
364
|
+
# Give back what a maintenance window took. Never fatal: the next desktop
|
|
365
|
+
# launch starts the frontend anyway, and that is where the operator reloads
|
|
366
|
+
# from, because a restarted frontend answers on a new session URL.
|
|
367
|
+
try {
|
|
368
|
+
if (Open-CrewManagedFrontend -Quiet) { Write-LaunchLog 'Crew-managed frontend on 3080 is serving again after maintenance.' }
|
|
369
|
+
else { Write-LaunchLog 'Crew-managed frontend on 3080 was not restarted after maintenance; the next desktop launch will start it.' 'WARN' }
|
|
370
|
+
} catch {
|
|
371
|
+
Write-LaunchLog ('Crew-managed frontend on 3080 could not be restarted after maintenance: {0}' -f $_.Exception.Message) 'WARN'
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
|
|
316
375
|
function Get-OfficialFrontendOverlay {
|
|
317
376
|
$frontendRoot = Join-Path $env:USERPROFILE '.config\dsh-crew\frontend'
|
|
318
377
|
$path = Join-Path $frontendRoot 'official-web.patch.json'
|
|
@@ -645,18 +704,50 @@ function Restore-OwnedServiceRecord {
|
|
|
645
704
|
return $true
|
|
646
705
|
}
|
|
647
706
|
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
707
|
+
# Spawns the persistent watcher when no live one is present, and returns its
|
|
708
|
+
# heartbeat record when one already is. Shared so that the interactive and the
|
|
709
|
+
# blocking entries agree on what counts as "a supervisor is already running" —
|
|
710
|
+
# two answers to that question would let one entry spawn a duplicate that the
|
|
711
|
+
# mutex immediately kills.
|
|
712
|
+
function Start-CrewSupervisorProcess {
|
|
651
713
|
$observed = Get-SupervisorHeartbeatRecord
|
|
652
714
|
if ($observed -and $observed.State -eq 'legacy-v1') {
|
|
653
715
|
throw 'CREW_SUPERVISOR_UPGRADE_REQUIRED: a legacy watcher is active and must be handed off before interactive launch.'
|
|
654
716
|
}
|
|
655
|
-
if (
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
717
|
+
if ($observed) { return $observed }
|
|
718
|
+
$arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
|
|
719
|
+
$watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
|
|
720
|
+
Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
|
|
721
|
+
return $null
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
# Waits only for the watcher to exist, not for 3210 to answer. The watcher
|
|
725
|
+
# publishes its heartbeat before it first touches the port, so this returns as
|
|
726
|
+
# soon as Crew is being supervised. Used by the interactive entry: 3080 is
|
|
727
|
+
# already serving by then, and making the operator's window wait for a first
|
|
728
|
+
# 3210 boot (measured 18-78s on this machine, longer under load) delays nothing
|
|
729
|
+
# they can see — the watcher performs that same wait either way.
|
|
730
|
+
function Wait-CrewSupervisorStarted {
|
|
731
|
+
param([int] $TimeoutSeconds = 30)
|
|
732
|
+
$observed = Start-CrewSupervisorProcess
|
|
733
|
+
if ($observed) {
|
|
734
|
+
Write-LaunchLog ('Persistent Crew supervisor already running; PID={0}.' -f $observed.Record.pid)
|
|
735
|
+
return $true
|
|
659
736
|
}
|
|
737
|
+
$deadline = (Get-Date).AddSeconds($TimeoutSeconds)
|
|
738
|
+
do {
|
|
739
|
+
$record = Get-SupervisorHeartbeatRecord
|
|
740
|
+
if ($record) {
|
|
741
|
+
Write-LaunchLog ('Persistent Crew supervisor started; PID={0}; state={1}.' -f $record.Record.pid, $record.State)
|
|
742
|
+
return $true
|
|
743
|
+
}
|
|
744
|
+
Start-Sleep -Milliseconds 250
|
|
745
|
+
} while ((Get-Date) -lt $deadline)
|
|
746
|
+
return $false
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
function Ensure-CrewSupervisorRunning { param([int] $TimeoutSeconds = 90)
|
|
750
|
+
$null = Start-CrewSupervisorProcess
|
|
660
751
|
|
|
661
752
|
$crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
|
|
662
753
|
if (-not $crew) { throw 'Crew-owned 3210 service definition is missing.' }
|
|
@@ -792,7 +883,7 @@ function Test-MaintenanceSessionActive {
|
|
|
792
883
|
}
|
|
793
884
|
|
|
794
885
|
function Set-MaintenanceSession {
|
|
795
|
-
param([object] $Request)
|
|
886
|
+
param([object] $Request, [bool] $FrontendStopped = $false)
|
|
796
887
|
# Strong-failure semantics: the STOPPED result may only be published after
|
|
797
888
|
# the session is durably written AND read back with exact identity. Any
|
|
798
889
|
# failure returns $false and the caller must NOT claim the stopped window.
|
|
@@ -804,6 +895,9 @@ function Set-MaintenanceSession {
|
|
|
804
895
|
runtime_id = $Request.runtime_id
|
|
805
896
|
stopped_at = [DateTimeOffset]::UtcNow.ToUnixTimeMilliseconds()
|
|
806
897
|
request_id = $Request.request_id
|
|
898
|
+
# Recorded so the matching start restores what this window stopped: the
|
|
899
|
+
# 3080 frontend shares the workspace store the maintenance is rewriting.
|
|
900
|
+
frontend_stopped = $FrontendStopped
|
|
807
901
|
} | ConvertTo-Json -Compress
|
|
808
902
|
$temp = Join-Path $crewSupervisorRoot ("maintenance-session.{0}.tmp" -f $PID)
|
|
809
903
|
Write-Utf8NoBom -Path $temp -Content $session
|
|
@@ -909,8 +1003,30 @@ function Invoke-CrewMaintenanceRequests {
|
|
|
909
1003
|
Write-LaunchLog ('Maintenance-stop {0} rejected: live runtime_id mismatch.' -f $request.request_id) 'WARN'
|
|
910
1004
|
continue
|
|
911
1005
|
}
|
|
1006
|
+
# `refresh_frontend` is how a history maintenance asks for the whole DSH
|
|
1007
|
+
# home to be quiet, not just 3210: the managed frontend shares the
|
|
1008
|
+
# workspace store this operation rewrites. The npx lifecycle does not ask,
|
|
1009
|
+
# because a runtime-tree swap does not touch that store.
|
|
1010
|
+
$refreshFrontend = $false
|
|
1011
|
+
if ($request.extra) {
|
|
1012
|
+
$refreshFlag = $request.extra.PSObject.Properties['refresh_frontend']
|
|
1013
|
+
$refreshFrontend = $null -ne $refreshFlag -and $refreshFlag.Value -eq $true
|
|
1014
|
+
}
|
|
1015
|
+
$frontendStopped = $false
|
|
1016
|
+
if ($refreshFrontend) {
|
|
1017
|
+
# Stopped BEFORE the backend: a failure here must abort with everything
|
|
1018
|
+
# still running. Aborting after 3210 is down would leave no session
|
|
1019
|
+
# published, and ordinary supervision would restart it mid-transaction.
|
|
1020
|
+
if (-not (Stop-CrewManagedFrontend)) {
|
|
1021
|
+
Write-MaintenanceResult $request 'SUPERVISOR_FRONTEND_STOP_FAILED'
|
|
1022
|
+
Write-LaunchLog ('Maintenance-stop {0} could not stop the Crew-managed frontend on 3080; nothing was stopped.' -f $request.request_id) 'ERROR'
|
|
1023
|
+
continue
|
|
1024
|
+
}
|
|
1025
|
+
$frontendStopped = $true
|
|
1026
|
+
}
|
|
912
1027
|
$port = Get-PortState $crew.Port
|
|
913
1028
|
if ($port.State -ne 'occupied' -or -not $port.Pid) {
|
|
1029
|
+
if ($frontendStopped) { Start-CrewManagedFrontendQuietly }
|
|
914
1030
|
Write-MaintenanceResult $request 'SUPERVISOR_STOP_FAILED'
|
|
915
1031
|
continue
|
|
916
1032
|
}
|
|
@@ -923,15 +1039,16 @@ function Invoke-CrewMaintenanceRequests {
|
|
|
923
1039
|
# durably written AND read back with exact identity. A session write
|
|
924
1040
|
# failure must never tell npx it owns a stopped window it cannot
|
|
925
1041
|
# later prove (that race auto-restarts 3210 mid tree-swap).
|
|
926
|
-
$sessionDurable = Set-MaintenanceSession $request
|
|
1042
|
+
$sessionDurable = Set-MaintenanceSession $request $frontendStopped
|
|
927
1043
|
if ($sessionDurable) {
|
|
928
|
-
Write-MaintenanceResult $request 'STOPPED' @{ lease = $request.lease; stopped_runtime_id = $request.runtime_id }
|
|
1044
|
+
Write-MaintenanceResult $request 'STOPPED' @{ lease = $request.lease; stopped_runtime_id = $request.runtime_id; frontend_stopped = $frontendStopped }
|
|
929
1045
|
Write-LaunchLog ('Maintenance-stop {0} executed; lease issued.' -f $request.request_id)
|
|
930
1046
|
} else {
|
|
931
1047
|
Write-MaintenanceResult $request 'SUPERVISOR_SESSION_PERSIST_FAILED'
|
|
932
1048
|
Write-LaunchLog ('Maintenance-stop {0} stopped the process but the STOPPED session could not be persisted; NOT publishing STOPPED.' -f $request.request_id) 'ERROR'
|
|
933
1049
|
}
|
|
934
1050
|
} else {
|
|
1051
|
+
if ($frontendStopped) { Start-CrewManagedFrontendQuietly }
|
|
935
1052
|
Write-MaintenanceResult $request 'SUPERVISOR_STOP_FAILED'
|
|
936
1053
|
}
|
|
937
1054
|
} elseif ($request.operation -eq 'maintenance-start') {
|
|
@@ -953,6 +1070,8 @@ function Invoke-CrewMaintenanceRequests {
|
|
|
953
1070
|
Write-LaunchLog ('Maintenance-start {0} rejected: no matching STOPPED session (lease/identity mismatch).' -f $request.request_id) 'WARN'
|
|
954
1071
|
continue
|
|
955
1072
|
}
|
|
1073
|
+
$frontendProperty = $session.PSObject.Properties['frontend_stopped']
|
|
1074
|
+
$restoreFrontend = $null -ne $frontendProperty -and $frontendProperty.Value -eq $true
|
|
956
1075
|
$livePort = Get-PortState $crew.Port
|
|
957
1076
|
if ($livePort.State -ne 'free') {
|
|
958
1077
|
# occupied AND unknown both fail closed: the stopped window is not
|
|
@@ -994,6 +1113,9 @@ function Invoke-CrewMaintenanceRequests {
|
|
|
994
1113
|
Write-MaintenanceResult $request 'VERIFY_FAILED' @{ lease = $lease; runtime_id = $failedRuntimeId }
|
|
995
1114
|
Write-LaunchLog ('Maintenance-start {0} verification failed.' -f $request.request_id) 'ERROR'
|
|
996
1115
|
}
|
|
1116
|
+
# After the result, so a slow frontend boot never delays the transaction
|
|
1117
|
+
# this start is here to close.
|
|
1118
|
+
if ($restoreFrontend) { Start-CrewManagedFrontendQuietly }
|
|
997
1119
|
} else {
|
|
998
1120
|
# Unknown maintenance op: remove and report.
|
|
999
1121
|
Write-MaintenanceResult $request 'MAINTENANCE_UNKNOWN_OP'
|
|
@@ -1211,7 +1333,10 @@ function Ensure-CrewServices {
|
|
|
1211
1333
|
# matching maintenance-start owns the launch right.
|
|
1212
1334
|
if ($service.CrewOwned -and (Test-MaintenanceSessionActive)) {
|
|
1213
1335
|
$service.State = 'maintenance'
|
|
1214
|
-
|
|
1336
|
+
# Say why, rather than clearing the field: an empty reason reaches the
|
|
1337
|
+
# startup wait as "deadline exceeded: dsh-crew:3210 ()", which reads like a
|
|
1338
|
+
# fault when the fence is the supervisor doing exactly as it was told.
|
|
1339
|
+
$service.LastError = 'a maintenance session holds the launch right (an update is mid-handoff); auto-start deferred'
|
|
1215
1340
|
continue
|
|
1216
1341
|
}
|
|
1217
1342
|
$health = Get-HealthState $service
|
|
@@ -1393,10 +1518,35 @@ try {
|
|
|
1393
1518
|
exit 0
|
|
1394
1519
|
}
|
|
1395
1520
|
|
|
1396
|
-
Ensure-CrewSupervisorRunning
|
|
1397
|
-
|
|
1398
1521
|
if ($Mode -eq 'open') {
|
|
1399
|
-
|
|
1522
|
+
# A desktop launch promises the frontend on 3080, and that is up in about a
|
|
1523
|
+
# second. The supervisor owns 3210 from the moment it starts — it publishes
|
|
1524
|
+
# its heartbeat before it first touches the port — so this waits for the
|
|
1525
|
+
# watcher to be running, reports where 3210 actually is, and returns. Holding
|
|
1526
|
+
# the operator's window for 3210 readiness bought nothing: the watcher is
|
|
1527
|
+
# doing that wait anyway, and under load a first boot here has taken 78s.
|
|
1528
|
+
if (-not (Wait-CrewSupervisorStarted)) {
|
|
1529
|
+
throw 'No Crew supervisor started within 30s; 3210 has nothing watching it.'
|
|
1530
|
+
}
|
|
1531
|
+
$crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
|
|
1532
|
+
$health = if ($crew) { Get-HealthState $crew } else { $null }
|
|
1533
|
+
if ($health -and $health.Ready) {
|
|
1534
|
+
Write-LaunchLog 'Official frontend is on 3080; Crew is ready on 3210.'
|
|
1535
|
+
} else {
|
|
1536
|
+
Write-LaunchLog ('Official frontend is on 3080; the supervisor is bringing 3210 up in the background. Last health: {0}' -f $health.Error) 'WARN'
|
|
1537
|
+
}
|
|
1538
|
+
# Operator-facing summary rather than a log line: clicking Crew before 3210
|
|
1539
|
+
# answers looks like a broken feature, so say that it is still coming up.
|
|
1540
|
+
Write-Host ''
|
|
1541
|
+
Write-Host 'DSH Crew: the frontend is on http://127.0.0.1:3080.' -ForegroundColor Green
|
|
1542
|
+
if ($health -and $health.Ready) {
|
|
1543
|
+
Write-Host 'Backend 3210 is ready.' -ForegroundColor Green
|
|
1544
|
+
} else {
|
|
1545
|
+
Write-Host 'Backend 3210 is still starting; Crew features appear once it answers.' -ForegroundColor Yellow
|
|
1546
|
+
}
|
|
1547
|
+
Write-Host ("Diagnostic log: {0}" -f $launcherLog) -ForegroundColor DarkGray
|
|
1548
|
+
} else {
|
|
1549
|
+
Ensure-CrewSupervisorRunning
|
|
1400
1550
|
}
|
|
1401
1551
|
Write-LaunchLog ('Launcher completed successfully in {0:n1}s.' -f ((Get-Date) - $startedAt).TotalSeconds)
|
|
1402
1552
|
exit 0
|