@ran-sh/dsh-crew 2.1.1 → 2.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.1.1",
3
+ "version": "2.1.3",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
@@ -8,7 +8,9 @@
8
8
  "mcpServers": {
9
9
  "dsh-crew": {
10
10
  "command": "node",
11
- "args": ["${CLAUDE_PLUGIN_ROOT}/src/server.mjs"]
11
+ "args": [
12
+ "${CLAUDE_PLUGIN_ROOT}/src/server.mjs"
13
+ ]
12
14
  }
13
15
  }
14
16
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.1.1",
3
+ "version": "2.1.3",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -1,5 +1,5 @@
1
1
  import { createHash, randomUUID } from 'node:crypto';
2
- import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
2
+ import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
3
3
  import { dirname, isAbsolute, join, resolve } from 'node:path';
4
4
 
5
5
  const WORKSPACE = 'harness/storages/workspace.json';
@@ -121,16 +121,52 @@ async function requireStopped(assertStopped) {
121
121
  if (typeof assertStopped !== 'function' || await assertStopped() !== true) fail('BACKEND_NOT_STOPPED');
122
122
  }
123
123
 
124
+ /**
125
+ * Every session id that still has an artifact under `harness/sessions`.
126
+ *
127
+ * A request may name sessions as "already gone" so that a workspace they belong
128
+ * to stops being unreachable. That claim decides whether a record is deleted, so
129
+ * it is proven here rather than trusted: a live session named as absent would
130
+ * drop its workspace while leaving the artifact behind.
131
+ */
132
+ function presentSessionIds(root) {
133
+ const sessionsRoot = pathInside(root, 'harness/sessions');
134
+ let projects;
135
+ try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
136
+ if (projects.length > 20000) fail('INVALID_FILE');
137
+ const present = new Set();
138
+ for (const project of projects) {
139
+ if (!project.isDirectory() || project.isSymbolicLink()) continue;
140
+ let entries;
141
+ try { entries = readdirSync(join(sessionsRoot, project.name), { withFileTypes: true }); } catch { continue; }
142
+ for (const entry of entries) {
143
+ if (!entry.isDirectory() || !validId(entry.name)) continue;
144
+ const directory = join(sessionsRoot, project.name, entry.name);
145
+ if (entry.isSymbolicLink()) { present.add(entry.name); continue; }
146
+ let files = [];
147
+ try { files = readdirSync(directory); } catch { present.add(entry.name); continue; }
148
+ if (files.some(name => /^session(?:\.v[1-9][0-9]*)?\.jsonl(?:\.zstd)?$/.test(name))) present.add(entry.name);
149
+ }
150
+ }
151
+ return present;
152
+ }
153
+
124
154
  /** Internal offline primitive: caller must own the maintenance lease + update lock. */
125
155
  export async function archiveHistory({ crewRoot, request, assertStopped, archiveId = randomUUID() }) {
126
156
  await requireStopped(assertStopped);
127
157
  if (!request || !['archive', 'delete'].includes(request.operation) || !Array.isArray(request.artifacts)
128
158
  || !Array.isArray(request.sessionIds) || !Array.isArray(request.workspaceIds)
129
- || request.artifacts.length > 10000 || request.workspaceIds.length > 10000) fail('INVALID_REQUEST');
159
+ || !Array.isArray(request.absentSessionIds ?? [])
160
+ || request.artifacts.length > 10000 || request.workspaceIds.length > 10000
161
+ || (request.absentSessionIds ?? []).length > 10000) fail('INVALID_REQUEST');
130
162
  const selectedSessions = new Set(request.sessionIds);
131
163
  const selectedWorkspaces = new Set(request.workspaceIds);
132
- if ([...selectedSessions, ...selectedWorkspaces].some(id => !validId(id))
164
+ const absentSessions = new Set(request.absentSessionIds ?? []);
165
+ const removedSessions = new Set([...selectedSessions, ...absentSessions]);
166
+ if ([...removedSessions, ...selectedWorkspaces].some(id => !validId(id))
133
167
  || selectedSessions.size !== request.sessionIds.length || selectedWorkspaces.size !== request.workspaceIds.length
168
+ || absentSessions.size !== (request.absentSessionIds ?? []).length
169
+ || absentSessions.size !== removedSessions.size - selectedSessions.size
134
170
  || request.artifacts.length !== selectedSessions.size
135
171
  || new Set(request.artifacts.map(f => f.sessionId)).size !== selectedSessions.size
136
172
  || request.artifacts.some(f => !selectedSessions.has(f.sessionId))) fail('INVALID_SELECTION');
@@ -140,12 +176,16 @@ export async function archiveHistory({ crewRoot, request, assertStopped, archive
140
176
  const after = structuredClone(before);
141
177
  for (const id of selectedWorkspaces) {
142
178
  const record = before.tables.workspaces[id];
143
- if (!record || record.sessionIds.some(sid => !selectedSessions.has(sid))) fail('PREVIEW_CHANGED');
179
+ if (!record || record.sessionIds.some(sid => !removedSessions.has(sid))) fail('PREVIEW_CHANGED');
144
180
  delete after.tables.workspaces[id];
145
181
  }
182
+ if (absentSessions.size > 0) {
183
+ const live = presentSessionIds(crewRoot);
184
+ for (const id of absentSessions) if (live.has(id)) fail('PREVIEW_CHANGED');
185
+ }
146
186
  after.global.workspaceIds = after.global.workspaceIds.filter(id => !selectedWorkspaces.has(id));
147
- after.global.archivedSessionIds = after.global.archivedSessionIds.filter(id => !selectedSessions.has(id));
148
- for (const record of Object.values(after.tables.workspaces)) record.sessionIds = record.sessionIds.filter(id => !selectedSessions.has(id));
187
+ after.global.archivedSessionIds = after.global.archivedSessionIds.filter(id => !removedSessions.has(id));
188
+ for (const record of Object.values(after.tables.workspaces)) record.sessionIds = record.sessionIds.filter(id => !removedSessions.has(id));
149
189
  let size = 0;
150
190
  const files = request.artifacts.map(file => {
151
191
  const bytes = readBounded(artifactPath(crewRoot, file));
@@ -53,31 +53,63 @@ export function planHistoryCleanup(snapshot, { operation = 'archive', scope = 'c
53
53
  };
54
54
  const withinTime = (row) => cutoff === null
55
55
  || (instant(row.createdAt) !== null && instant(row.createdAt) < cutoff);
56
+ const present = new Set(sessions.map(row => row.id));
57
+ const worktreeSessions = new Set(sessions.filter(row => row.worktree === true).map(row => row.id));
56
58
  const selected = new Set(sessions.filter(row => admitted(row) && withinTime(row)).map(row => row.id));
57
59
  // A workspace carries no provenance of its own: it follows its sessions, and
58
60
  // only when every one of them is selected. Checking `admitted(row)` here would
59
61
  // test a workspace id against a session ledger and always fail. A provenance
60
- // scope additionally requires at least one selected child — an empty workspace
61
- // holds no Crew work, so it is not Crew's to remove; `all` keeps its historical
62
+ // scope additionally requires at least one child — an empty workspace holds no
63
+ // Crew work, so it is not Crew's to remove; `all` keeps its historical
62
64
  // behaviour of following an empty workspace.
65
+ //
66
+ // A child whose artifact is already gone cannot be "selected": there is
67
+ // nothing left to select. Such a child still counts as covered when the ledger
68
+ // recorded Crew creating it, the same evidence that makes a live one Crew's.
69
+ // Without that, a workspace whose sessions were removed by an earlier cleanup
70
+ // became permanently unreachable — every scope demanded its children, and its
71
+ // children no longer existed — which is exactly how 48 dead rows stayed in the
72
+ // store while the operator could only see them.
63
73
  const requiresOwnedChild = scope === 'crew' || scope === 'worktree';
64
- const workspaceIds = workspaces.filter(row => {
74
+ // One decision per workspace, from that row alone: what a row is allowed to
75
+ // remove must never depend on which rows were visited before it.
76
+ const decideWorkspace = (row) => {
65
77
  const children = ids(row.sessionIds);
66
- return withinTime(row) && (!requiresOwnedChild || children.length > 0)
67
- && children.every(id => selected.has(id));
78
+ if (children.length === 0) return { ok: !requiresOwnedChild, absent: [] };
79
+ // Worktree scope marks a live child by its session header. A gone child has
80
+ // no header left, so the workspace's own path under the worktree root is the
81
+ // only marker there is.
82
+ if (scope === 'worktree' && row.worktree !== true && !children.every(id => worktreeSessions.has(id))) return { ok: false, absent: [] };
83
+ const gone = [];
84
+ for (const id of children) {
85
+ if (selected.has(id)) continue;
86
+ // Kept back by the scope, or gone with no record that Crew made it.
87
+ if (present.has(id) || !origin.has(id)) return { ok: false, absent: [] };
88
+ gone.push(id);
89
+ }
90
+ return { ok: true, absent: gone };
91
+ };
92
+ const absent = new Set();
93
+ const workspaceIds = workspaces.filter(row => {
94
+ if (!withinTime(row)) return false;
95
+ const decision = decideWorkspace(row);
96
+ if (!decision.ok) return false;
97
+ for (const id of decision.absent) absent.add(id);
98
+ return true;
68
99
  }).map(row => row.id).sort();
100
+ const absentSessionIds = [...absent].sort();
69
101
  const sessionIds = [...selected].sort();
70
102
  const signature = {
71
103
  operation, scope, cutoff,
72
- workspaces: workspaces.map(row => ({ id: row.id, createdAt: row.createdAt, updatedAt: row.updatedAt, sessionIds: ids(row.sessionIds) })),
73
- sessions: sessions.map(row => ({ id: row.id, createdAt: row.createdAt, revision: row.revision })),
104
+ workspaces: workspaces.map(row => ({ id: row.id, createdAt: row.createdAt, updatedAt: row.updatedAt, sessionIds: ids(row.sessionIds), worktree: row.worktree === true })),
105
+ sessions: sessions.map(row => ({ id: row.id, createdAt: row.createdAt, revision: row.revision, worktree: row.worktree === true })),
74
106
  active,
75
107
  crewSessionIds: [...origin].sort(),
76
108
  };
77
109
  return {
78
110
  schemaVersion: 1, operation, scope,
79
111
  before: cutoff === null ? null : new Date(cutoff).toISOString(),
80
- timeBasis: 'createdAt', workspaceIds, sessionIds,
112
+ timeBasis: 'createdAt', workspaceIds, sessionIds, absentSessionIds,
81
113
  counts: { workspaces: workspaceIds.length, sessions: sessionIds.length },
82
114
  protectedCounts: { workspaces: workspaces.length - workspaceIds.length, sessions: sessions.length - sessionIds.length },
83
115
  executable: active.length === 0 && (workspaceIds.length > 0 || sessionIds.length > 0),
@@ -19,6 +19,17 @@ function verifyDisk(root, manifest) {
19
19
  }
20
20
  }
21
21
 
22
+ function removedWorkspaceIds(root, archiveId) {
23
+ const manifest = readHistoryManifest(root, archiveId);
24
+ return Object.keys(manifest.before.tables.workspaces).filter(id => !Object.hasOwn(manifest.after.tables.workspaces, id));
25
+ }
26
+
27
+ function workspacesStillPresent(root, ids) {
28
+ if (ids.length === 0) return [];
29
+ const current = decodeWorkspaceStore(readHistoryBytes(historyPath(root, 'harness/storages/workspace.json')));
30
+ return ids.filter(id => Object.hasOwn(current.tables.workspaces, id));
31
+ }
32
+
22
33
  /** Detached executor core; injected boundaries make the real transaction testable. */
23
34
  export async function runHistoryOperation({ crewRoot, id, acquire, release, supervisor, checkFence,
24
35
  assertStopped, verifyRunning, recover = false }) {
@@ -46,7 +57,11 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
46
57
  if (!alreadyStarted) {
47
58
  if (!recover || await assertStopped(state) !== true) await checkFence(state);
48
59
  save('STOPPING');
49
- const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId });
60
+ // `refresh_frontend`: this operation rewrites `storages/workspace.json`,
61
+ // which the Crew-managed frontend on 3080 shares, so the launcher stops
62
+ // that server for the same window and starts it again afterwards. The npx
63
+ // lifecycle stops only 3210, because a tree swap leaves that file alone.
64
+ const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId, refresh_frontend: true });
50
65
  if (!stopped?.ok || await assertStopped(state) !== true) throw Error('HISTORY_STOP_NOT_VERIFIED');
51
66
  const archiveId = state.operation === 'restore' ? state.archiveId : state.id;
52
67
  const manifestFile = historyPath(crewRoot, `history/transactions/${archiveId}/manifest.json`);
@@ -73,6 +88,21 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
73
88
  }
74
89
  save('VERIFYING');
75
90
  if (await verifyRunning(state) !== true) throw Error('HISTORY_RESTART_NOT_VERIFIED');
91
+ // The store this operation rewrote is shared with every other DSH server on
92
+ // the same home, and each of them writes the whole file back from memory. A
93
+ // process that stayed live across the window puts back exactly what was
94
+ // removed, and from here that is indistinguishable from success: the
95
+ // manifest is applied and the runtime is up. Ask the disk instead, and say
96
+ // so rather than report a cleanup that did not happen.
97
+ if (!state.rolledBack && state.operation !== 'restore') {
98
+ const returned = workspacesStillPresent(crewRoot, removedWorkspaceIds(crewRoot, state.archiveId ?? state.id));
99
+ if (returned.length > 0) {
100
+ state = { ...state, phase: 'FAILED', code: 'HISTORY_STORE_CHANGED_AFTER_APPLY',
101
+ counts: { ...(state.counts ?? {}), returned: returned.length } };
102
+ writeHistoryState(crewRoot, state);
103
+ return state;
104
+ }
105
+ }
76
106
  if (state.operation === 'delete' && !state.rolledBack) {
77
107
  await finalizeHistoryDeletion({ crewRoot, archiveId: state.archiveId ?? state.id, assertRestarted: () => verifyRunning(state) });
78
108
  }
@@ -8,9 +8,9 @@ import { runHistoryOperation } from './operation.mjs';
8
8
  import { readHistoryState } from './state.mjs';
9
9
  import { TARGET_DSH_VERSION } from '../dsh-cohort.mjs';
10
10
 
11
- async function portFree() {
11
+ async function portFree(port = 3210) {
12
12
  return new Promise(resolve => {
13
- const socket = createConnection({ host: '127.0.0.1', port: 3210 });
13
+ const socket = createConnection({ host: '127.0.0.1', port });
14
14
  socket.setTimeout(2000);
15
15
  socket.once('connect', () => { socket.destroy(); resolve(false); });
16
16
  socket.once('timeout', () => { socket.destroy(); resolve(false); });
@@ -18,6 +18,23 @@ async function portFree() {
18
18
  });
19
19
  }
20
20
 
21
+ /**
22
+ * The stopped window, proven from outside the launcher that reported it.
23
+ *
24
+ * `storages/workspace.json` is rewritten by this operation and is held in memory
25
+ * by every DSH server on the home, so the window has to cover more than the hub:
26
+ * the Crew-managed frontend on 3080 shares that home and is stopped for the same
27
+ * window (`refresh_frontend`). Both ports are re-probed here rather than taken
28
+ * on the launcher's word, and the session must be the one this transaction
29
+ * stopped — a lease alone never means the servers are gone.
30
+ */
31
+ export async function stoppedWindowIsClean({ session, lease, runtimeId, probe = portFree }) {
32
+ if (!session?.ok || session.state !== 'present') return false;
33
+ if (session.session?.lease !== lease || session.session?.runtime_id !== runtimeId) return false;
34
+ if (!await probe(3210)) return false;
35
+ return session.session?.frontend_stopped === true ? await probe(3080) : true;
36
+ }
37
+
21
38
  export async function runProductionHistory({ id, recover = false } = {}) {
22
39
  if (process.platform !== 'win32') throw Error('HISTORY_PLATFORM_UNSUPPORTED');
23
40
  const home = homedir(); const crewRoot = join(home, '.config', 'dsh-crew');
@@ -31,10 +48,9 @@ export async function runProductionHistory({ id, recover = false } = {}) {
31
48
  const response = await fetch('http://127.0.0.1:3210/_dsh/dsh-crew/history/fenced-check', { method: 'POST', headers: { 'content-type': 'application/json' }, body: '{}', signal: AbortSignal.timeout(30000) });
32
49
  if (!response.ok || (await response.json()).ok !== true) throw Error('HISTORY_FENCE_NOT_IDLE');
33
50
  },
34
- assertStopped: async s => {
35
- const lease = readMaintenanceSession(crewRoot);
36
- return lease.ok && lease.state === 'present' && lease.session.lease === s.lease && lease.session.runtime_id === s.runtimeId && await portFree();
37
- },
51
+ assertStopped: async s => stoppedWindowIsClean({
52
+ session: readMaintenanceSession(crewRoot), lease: s.lease, runtimeId: s.runtimeId,
53
+ }),
38
54
  verifyRunning: async s => {
39
55
  try {
40
56
  const response = await fetch('http://127.0.0.1:3210/_dsh/dsh-crew/runtime', { signal: AbortSignal.timeout(3000) });
@@ -77,12 +77,15 @@ export function createHistoryService({ crewRoot, agents, persistence, runtimeId,
77
77
  for (const row of sessions) if (crewSet.has(row.id)) row.crew = true;
78
78
  // The isolated-workspace marker: the hub stamps every worktree session with a
79
79
  // cwd under the Crew worktree root, so that subset is scopeable on its own.
80
+ // A workspace carries the same marker from its own path, which is all that is
81
+ // left of a workspace whose sessions were already removed.
80
82
  const worktreeRoot = defaultWorktreeRoot().replaceAll('\\', '/').toLowerCase();
81
- for (const row of sessions) {
82
- if (row.crew === true && typeof row.cwd === 'string' && row.cwd.replaceAll('\\', '/').toLowerCase().startsWith(worktreeRoot)) row.worktree = true;
83
- }
83
+ const underWorktreeRoot = value => typeof value === 'string' && value.replaceAll('\\', '/').toLowerCase().startsWith(worktreeRoot);
84
+ for (const row of sessions) if (row.crew === true && underWorktreeRoot(row.cwd)) row.worktree = true;
85
+ for (const row of workspaces) if (underWorktreeRoot(row.path)) row.worktree = true;
84
86
  const plan = planHistoryCleanup({ workspaces, sessions, crewSessionIds, activeSessionIds: gate.idle() ? [] : ['active-agent'] }, options);
85
87
  const selected = new Set(plan.sessionIds);
88
+ const alreadyGone = new Set(plan.absentSessionIds ?? []);
86
89
  const byId = new Map(sessions.map(row => [row.id, row]));
87
90
  const queue = sessions.filter(row => !selected.has(row.id));
88
91
  for (let i = 0; i < queue.length; i++) {
@@ -90,12 +93,16 @@ export function createHistoryService({ crewRoot, agents, persistence, runtimeId,
90
93
  if (parent && selected.delete(parent.id)) queue.push(parent);
91
94
  }
92
95
  plan.sessionIds = plan.sessionIds.filter(id => selected.has(id));
93
- plan.workspaceIds = plan.workspaceIds.filter(id => store.tables.workspaces[id].sessionIds.every(sid => selected.has(sid)));
96
+ // Same rule the plan used, re-applied after the ancestor closure dropped
97
+ // sessions from the selection: a workspace follows its children, and a child
98
+ // the plan proved Crew's own but already removed still counts.
99
+ plan.workspaceIds = plan.workspaceIds.filter(id => store.tables.workspaces[id].sessionIds.every(sid => selected.has(sid) || alreadyGone.has(sid)));
94
100
  plan.counts = { workspaces: plan.workspaceIds.length, sessions: plan.sessionIds.length };
95
101
  plan.executable = gate.idle() && plan.counts.workspaces + plan.counts.sessions > 0;
96
102
  if (!plan.executable && !plan.blockedReason) plan.blockedReason = 'EMPTY_SELECTION';
97
103
  const request = { operation: plan.operation, workspaceHash: historyHash(bytes), workspaceIds: plan.workspaceIds,
98
- sessionIds: plan.sessionIds, artifacts: sessions.filter(row => selected.has(row.id)).map(row => row.artifact) };
104
+ sessionIds: plan.sessionIds, absentSessionIds: plan.absentSessionIds ?? [],
105
+ artifacts: sessions.filter(row => selected.has(row.id)).map(row => row.artifact) };
99
106
  const revision = historyHash(JSON.stringify([runtimeId, plan.revision, request]));
100
107
  return { plan: { ...plan, revision, items: workspaces.filter(row => plan.workspaceIds.includes(row.id)).slice(0, 100).map(row => ({ id: row.id, title: row.title })) }, request };
101
108
  }
@@ -1639,7 +1639,7 @@ export function createCrewSupervisor({
1639
1639
  }
1640
1640
  }
1641
1641
  return {
1642
- stopOwnedBackend: async ({ lease = null, runtimeId = null } = {}) => {
1642
+ stopOwnedBackend: async ({ lease = null, runtimeId = null, refreshFrontend = false } = {}) => {
1643
1643
  const { readMaintenanceSession } = await import('../supervisor/restart-request.mjs');
1644
1644
  const durable = readMaintenanceSession(appRoot);
1645
1645
  if (!durable.ok) {
@@ -1675,7 +1675,11 @@ export function createCrewSupervisor({
1675
1675
  lease: lease ?? `txn-${now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`,
1676
1676
  runtime_id: identity.runtime_id,
1677
1677
  };
1678
- const result = await writeMaintenance({ operation: 'maintenance-stop', lease: transaction.lease, identity });
1678
+ const result = await writeMaintenance({ operation: 'maintenance-stop', lease: transaction.lease, identity,
1679
+ // A history maintenance rewrites the workspace store the managed 3080
1680
+ // frontend shares, so it asks for that server too. A runtime-tree swap
1681
+ // does not, and leaves the flag unset.
1682
+ extra: refreshFrontend ? { refresh_frontend: true } : null });
1679
1683
  if (!result.ok) {
1680
1684
  // A failed stop must not leave a usable lease behind: clear it so a
1681
1685
  // later startOwnedBackend() fails closed instead of presenting a
@@ -281,8 +281,11 @@ export function createWindowsSupervisorHandoffHooks({
281
281
  verifyExactWatcher: async ({ expected, role }) => inspect(expected, {
282
282
  allowHelperDrift: role === 'old' && !sameHash(expected?.helper_hash, target.helper_hash),
283
283
  }),
284
- maintenanceStop: async ({ lease, runtime_id: runtimeId }) => maintenanceClient?.stopOwnedBackend?.({ lease, runtimeId })
285
- ?? { ok: false, code: 'SUPERVISOR_MAINTENANCE_UNAVAILABLE' },
284
+ maintenanceStop: async ({ lease, runtime_id: runtimeId, refresh_frontend: refreshFrontend = false }) => maintenanceClient?.stopOwnedBackend?.({
285
+ lease,
286
+ runtimeId,
287
+ ...(refreshFrontend ? { refreshFrontend: true } : {}),
288
+ }) ?? { ok: false, code: 'SUPERVISOR_MAINTENANCE_UNAVAILABLE' },
286
289
  maintenanceStatus: async ({ lease, runtime_id: runtimeId }) => {
287
290
  const durable = readMaintenanceSession(appRoot);
288
291
  if (!durable.ok) return durable;
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.1.1';
39
+ export const RUNTIME_VERSION = '2.1.3';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -35,13 +35,15 @@ if not exist "%LAUNCH_HELPER%" (
35
35
  >>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
36
36
  echo ERROR: DSH Crew launcher helper is missing.
37
37
  echo Repair it with: dsh-crew update
38
- if /i "%LAUNCH_MODE%"=="open" pause
38
+ if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
39
39
  exit /b 1
40
40
  )
41
41
 
42
42
  powershell.exe -NoLogo -NoProfile -NonInteractive -ExecutionPolicy Bypass -File "%LAUNCH_HELPER%" -Mode "%LAUNCH_MODE%"
43
43
  set "LAUNCH_EXIT=%ERRORLEVEL%"
44
- if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" pause
44
+ rem DSH_CREW_LAUNCHER_NO_PAUSE: a wrapper that reports the failure itself asks
45
+ rem for the pause to be skipped here, so the operator presses a key once.
46
+ if not "%LAUNCH_EXIT%"=="0" if /i "%LAUNCH_MODE%"=="open" if not "%DSH_CREW_LAUNCHER_NO_PAUSE%"=="1" pause
45
47
  exit /b %LAUNCH_EXIT%
46
48
 
47
49
  :invalid_argument
@@ -51,7 +53,8 @@ exit /b 64
51
53
 
52
54
  :help
53
55
  echo Usage: %~nx0 [--open ^| --background ^| --watch]
54
- echo --open Open official Harness on 3080, then start Crew silently on 3210.
56
+ echo --open Open official Harness on 3080 and return once Crew is supervised;
57
+ echo 3210 keeps starting in the background.
55
58
  echo --background Start the Crew-owned 3210 service silently.
56
59
  echo --watch Keep the Crew-owned 3210 service healthy.
57
60
  exit /b 0
@@ -227,7 +227,7 @@ function Test-CrewWebSessionUrl {
227
227
  }
228
228
 
229
229
  function Open-CrewManagedFrontend {
230
- param([int] $TimeoutSeconds = 90)
230
+ param([int] $TimeoutSeconds = 90, [switch] $Quiet)
231
231
  # The 3080 frontend boots from the same Crew-managed Harness entry as 3210,
232
232
  # against that entry's own DSH_HOME (npm runtime -> Crew home, source cohort
233
233
  # -> its own tree). No official ~/.dsh state is read or written here.
@@ -253,7 +253,7 @@ function Open-CrewManagedFrontend {
253
253
  if (-not (Test-OfficialWebReady)) { throw 'The Crew-managed 3080 frontend is not ready; its process was left running.' }
254
254
  $existingUrl = Get-CrewWebSessionUrl
255
255
  if ($existingUrl -and (Test-CrewWebSessionUrl -Url $existingUrl)) {
256
- Open-CrewBrowserUrl -Url $existingUrl
256
+ if (-not $Quiet) { Open-CrewBrowserUrl -Url $existingUrl }
257
257
  Write-LaunchLog 'Opened the existing Crew-managed Harness frontend on 3080 with its session URL.'
258
258
  return $true
259
259
  }
@@ -289,7 +289,10 @@ function Open-CrewManagedFrontend {
289
289
  }
290
290
  if (Test-OfficialWebReady) {
291
291
  $startedUrl = Get-CrewWebSessionUrl -OutputLog $stdout
292
- if ($startedUrl) { Open-CrewBrowserUrl -Url $startedUrl; return $true }
292
+ if ($startedUrl) {
293
+ if (-not $Quiet) { Open-CrewBrowserUrl -Url $startedUrl }
294
+ return $true
295
+ }
293
296
  }
294
297
  }
295
298
  if ($process.HasExited) { throw ('Crew-managed Harness exited before readiness. Diagnostic log: {0}' -f $stderr) }
@@ -302,8 +305,7 @@ function Open-CrewManagedFrontend {
302
305
  }
303
306
  }
304
307
 
305
- function Test-OfficialWebReady {
306
- try {
308
+ function Test-OfficialWebReady { try {
307
309
  $response = Invoke-WebRequest -UseBasicParsing -Uri 'http://127.0.0.1:3080/' -TimeoutSec 2
308
310
  return $response.StatusCode -ge 200 -and $response.StatusCode -lt 300
309
311
  } catch {
@@ -313,6 +315,63 @@ function Test-OfficialWebReady {
313
315
  }
314
316
  }
315
317
 
318
+ # ---- The other server on this DSH home -------------------------------------
319
+ # The Crew-managed frontend on 3080 boots from the same Crew-managed entry as the
320
+ # hub, so it runs on the same DSH home: same sessions, same settings, and the
321
+ # same `storages/workspace.json`. DSH's JSON storage replaces that whole file and
322
+ # lets the last writer win, so an external rewrite of it only sticks while no
323
+ # server holds the file in memory. A history maintenance is exactly such an
324
+ # external rewrite, which is why it asks for this server to be stopped too —
325
+ # stopping the hub alone is what let a cleanup be undone minutes later.
326
+ function Stop-CrewManagedFrontend {
327
+ param([int] $TimeoutSeconds = 15)
328
+ $port = Get-PortState -Port 3080
329
+ if ($port.State -eq 'free') { return $true }
330
+ # An unenumerable listener is not provably free: fail closed rather than write
331
+ # under a server that might hold the very file being rewritten.
332
+ if ($port.State -ne 'occupied' -or -not $port.Pid) { return $false }
333
+ $ours = $false
334
+ if ($dshCliIsNodeEntry) {
335
+ $official = [pscustomobject]@{ NodePath = $dshCommand; Entry = $dshCli; Profile = 'web' }
336
+ $ours = Test-OfficialHarnessListener -OwnerPid ([int] $port.Pid) -Official $official -Profile 'web'
337
+ }
338
+ if (-not $ours) {
339
+ # Crew did not start this listener, so it cannot be proven to share the
340
+ # workspace store. A Crew-patched one is on a Crew home either way and is
341
+ # refused; the legacy official frontend and anything unrelated keep running,
342
+ # exactly as the start path leaves a foreign 3080 alone.
343
+ $patched = $false
344
+ try {
345
+ $probe = Invoke-RestMethod -Uri 'http://127.0.0.1:3080/_dsh/dsh-crew/bridge-status' -TimeoutSec 2
346
+ $patched = $probe.surface -eq 'official-bridge'
347
+ } catch { $patched = $false }
348
+ if ($patched) { return $false }
349
+ Write-LaunchLog 'A 3080 listener that is not the Crew-managed frontend was left running; it does not serve this DSH home.' 'WARN'
350
+ return $true
351
+ }
352
+ try { Stop-Process -Id ([int] $port.Pid) -Force -ErrorAction Stop } catch { return $false }
353
+ $deadline = (Get-Date).AddSeconds($TimeoutSeconds)
354
+ do {
355
+ $port = Get-PortState -Port 3080
356
+ if ($port.State -eq 'free') { return $true }
357
+ Start-Sleep -Milliseconds 250
358
+ } while ((Get-Date) -lt $deadline)
359
+ Write-LaunchLog ('The Crew-managed frontend on 3080 did not release the port within {0}s.' -f $TimeoutSeconds) 'WARN'
360
+ return $false
361
+ }
362
+
363
+ function Start-CrewManagedFrontendQuietly {
364
+ # Give back what a maintenance window took. Never fatal: the next desktop
365
+ # launch starts the frontend anyway, and that is where the operator reloads
366
+ # from, because a restarted frontend answers on a new session URL.
367
+ try {
368
+ if (Open-CrewManagedFrontend -Quiet) { Write-LaunchLog 'Crew-managed frontend on 3080 is serving again after maintenance.' }
369
+ else { Write-LaunchLog 'Crew-managed frontend on 3080 was not restarted after maintenance; the next desktop launch will start it.' 'WARN' }
370
+ } catch {
371
+ Write-LaunchLog ('Crew-managed frontend on 3080 could not be restarted after maintenance: {0}' -f $_.Exception.Message) 'WARN'
372
+ }
373
+ }
374
+
316
375
  function Get-OfficialFrontendOverlay {
317
376
  $frontendRoot = Join-Path $env:USERPROFILE '.config\dsh-crew\frontend'
318
377
  $path = Join-Path $frontendRoot 'official-web.patch.json'
@@ -645,18 +704,50 @@ function Restore-OwnedServiceRecord {
645
704
  return $true
646
705
  }
647
706
 
648
- function Ensure-CrewSupervisorRunning {
649
- param([int] $TimeoutSeconds = 90)
650
- $watcher = $null
707
+ # Spawns the persistent watcher when no live one is present, and returns its
708
+ # heartbeat record when one already is. Shared so that the interactive and the
709
+ # blocking entries agree on what counts as "a supervisor is already running" —
710
+ # two answers to that question would let one entry spawn a duplicate that the
711
+ # mutex immediately kills.
712
+ function Start-CrewSupervisorProcess {
651
713
  $observed = Get-SupervisorHeartbeatRecord
652
714
  if ($observed -and $observed.State -eq 'legacy-v1') {
653
715
  throw 'CREW_SUPERVISOR_UPGRADE_REQUIRED: a legacy watcher is active and must be handed off before interactive launch.'
654
716
  }
655
- if (-not $observed) {
656
- $arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
657
- $watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
658
- Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
717
+ if ($observed) { return $observed }
718
+ $arguments = @(Get-SupervisorLaunchArguments -ScriptPath $PSCommandPath)
719
+ $watcher = Start-Process -FilePath 'powershell.exe' -ArgumentList $arguments -WindowStyle Hidden -PassThru
720
+ Write-LaunchLog ('Started persistent Crew supervisor; PID={0}.' -f $watcher.Id)
721
+ return $null
722
+ }
723
+
724
+ # Waits only for the watcher to exist, not for 3210 to answer. The watcher
725
+ # publishes its heartbeat before it first touches the port, so this returns as
726
+ # soon as Crew is being supervised. Used by the interactive entry: 3080 is
727
+ # already serving by then, and making the operator's window wait for a first
728
+ # 3210 boot (measured 18-78s on this machine, longer under load) delays nothing
729
+ # they can see — the watcher performs that same wait either way.
730
+ function Wait-CrewSupervisorStarted {
731
+ param([int] $TimeoutSeconds = 30)
732
+ $observed = Start-CrewSupervisorProcess
733
+ if ($observed) {
734
+ Write-LaunchLog ('Persistent Crew supervisor already running; PID={0}.' -f $observed.Record.pid)
735
+ return $true
659
736
  }
737
+ $deadline = (Get-Date).AddSeconds($TimeoutSeconds)
738
+ do {
739
+ $record = Get-SupervisorHeartbeatRecord
740
+ if ($record) {
741
+ Write-LaunchLog ('Persistent Crew supervisor started; PID={0}; state={1}.' -f $record.Record.pid, $record.State)
742
+ return $true
743
+ }
744
+ Start-Sleep -Milliseconds 250
745
+ } while ((Get-Date) -lt $deadline)
746
+ return $false
747
+ }
748
+
749
+ function Ensure-CrewSupervisorRunning { param([int] $TimeoutSeconds = 90)
750
+ $null = Start-CrewSupervisorProcess
660
751
 
661
752
  $crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
662
753
  if (-not $crew) { throw 'Crew-owned 3210 service definition is missing.' }
@@ -792,7 +883,7 @@ function Test-MaintenanceSessionActive {
792
883
  }
793
884
 
794
885
  function Set-MaintenanceSession {
795
- param([object] $Request)
886
+ param([object] $Request, [bool] $FrontendStopped = $false)
796
887
  # Strong-failure semantics: the STOPPED result may only be published after
797
888
  # the session is durably written AND read back with exact identity. Any
798
889
  # failure returns $false and the caller must NOT claim the stopped window.
@@ -804,6 +895,9 @@ function Set-MaintenanceSession {
804
895
  runtime_id = $Request.runtime_id
805
896
  stopped_at = [DateTimeOffset]::UtcNow.ToUnixTimeMilliseconds()
806
897
  request_id = $Request.request_id
898
+ # Recorded so the matching start restores what this window stopped: the
899
+ # 3080 frontend shares the workspace store the maintenance is rewriting.
900
+ frontend_stopped = $FrontendStopped
807
901
  } | ConvertTo-Json -Compress
808
902
  $temp = Join-Path $crewSupervisorRoot ("maintenance-session.{0}.tmp" -f $PID)
809
903
  Write-Utf8NoBom -Path $temp -Content $session
@@ -909,8 +1003,30 @@ function Invoke-CrewMaintenanceRequests {
909
1003
  Write-LaunchLog ('Maintenance-stop {0} rejected: live runtime_id mismatch.' -f $request.request_id) 'WARN'
910
1004
  continue
911
1005
  }
1006
+ # `refresh_frontend` is how a history maintenance asks for the whole DSH
1007
+ # home to be quiet, not just 3210: the managed frontend shares the
1008
+ # workspace store this operation rewrites. The npx lifecycle does not ask,
1009
+ # because a runtime-tree swap does not touch that store.
1010
+ $refreshFrontend = $false
1011
+ if ($request.extra) {
1012
+ $refreshFlag = $request.extra.PSObject.Properties['refresh_frontend']
1013
+ $refreshFrontend = $null -ne $refreshFlag -and $refreshFlag.Value -eq $true
1014
+ }
1015
+ $frontendStopped = $false
1016
+ if ($refreshFrontend) {
1017
+ # Stopped BEFORE the backend: a failure here must abort with everything
1018
+ # still running. Aborting after 3210 is down would leave no session
1019
+ # published, and ordinary supervision would restart it mid-transaction.
1020
+ if (-not (Stop-CrewManagedFrontend)) {
1021
+ Write-MaintenanceResult $request 'SUPERVISOR_FRONTEND_STOP_FAILED'
1022
+ Write-LaunchLog ('Maintenance-stop {0} could not stop the Crew-managed frontend on 3080; nothing was stopped.' -f $request.request_id) 'ERROR'
1023
+ continue
1024
+ }
1025
+ $frontendStopped = $true
1026
+ }
912
1027
  $port = Get-PortState $crew.Port
913
1028
  if ($port.State -ne 'occupied' -or -not $port.Pid) {
1029
+ if ($frontendStopped) { Start-CrewManagedFrontendQuietly }
914
1030
  Write-MaintenanceResult $request 'SUPERVISOR_STOP_FAILED'
915
1031
  continue
916
1032
  }
@@ -923,15 +1039,16 @@ function Invoke-CrewMaintenanceRequests {
923
1039
  # durably written AND read back with exact identity. A session write
924
1040
  # failure must never tell npx it owns a stopped window it cannot
925
1041
  # later prove (that race auto-restarts 3210 mid tree-swap).
926
- $sessionDurable = Set-MaintenanceSession $request
1042
+ $sessionDurable = Set-MaintenanceSession $request $frontendStopped
927
1043
  if ($sessionDurable) {
928
- Write-MaintenanceResult $request 'STOPPED' @{ lease = $request.lease; stopped_runtime_id = $request.runtime_id }
1044
+ Write-MaintenanceResult $request 'STOPPED' @{ lease = $request.lease; stopped_runtime_id = $request.runtime_id; frontend_stopped = $frontendStopped }
929
1045
  Write-LaunchLog ('Maintenance-stop {0} executed; lease issued.' -f $request.request_id)
930
1046
  } else {
931
1047
  Write-MaintenanceResult $request 'SUPERVISOR_SESSION_PERSIST_FAILED'
932
1048
  Write-LaunchLog ('Maintenance-stop {0} stopped the process but the STOPPED session could not be persisted; NOT publishing STOPPED.' -f $request.request_id) 'ERROR'
933
1049
  }
934
1050
  } else {
1051
+ if ($frontendStopped) { Start-CrewManagedFrontendQuietly }
935
1052
  Write-MaintenanceResult $request 'SUPERVISOR_STOP_FAILED'
936
1053
  }
937
1054
  } elseif ($request.operation -eq 'maintenance-start') {
@@ -953,6 +1070,8 @@ function Invoke-CrewMaintenanceRequests {
953
1070
  Write-LaunchLog ('Maintenance-start {0} rejected: no matching STOPPED session (lease/identity mismatch).' -f $request.request_id) 'WARN'
954
1071
  continue
955
1072
  }
1073
+ $frontendProperty = $session.PSObject.Properties['frontend_stopped']
1074
+ $restoreFrontend = $null -ne $frontendProperty -and $frontendProperty.Value -eq $true
956
1075
  $livePort = Get-PortState $crew.Port
957
1076
  if ($livePort.State -ne 'free') {
958
1077
  # occupied AND unknown both fail closed: the stopped window is not
@@ -994,6 +1113,9 @@ function Invoke-CrewMaintenanceRequests {
994
1113
  Write-MaintenanceResult $request 'VERIFY_FAILED' @{ lease = $lease; runtime_id = $failedRuntimeId }
995
1114
  Write-LaunchLog ('Maintenance-start {0} verification failed.' -f $request.request_id) 'ERROR'
996
1115
  }
1116
+ # After the result, so a slow frontend boot never delays the transaction
1117
+ # this start is here to close.
1118
+ if ($restoreFrontend) { Start-CrewManagedFrontendQuietly }
997
1119
  } else {
998
1120
  # Unknown maintenance op: remove and report.
999
1121
  Write-MaintenanceResult $request 'MAINTENANCE_UNKNOWN_OP'
@@ -1211,7 +1333,10 @@ function Ensure-CrewServices {
1211
1333
  # matching maintenance-start owns the launch right.
1212
1334
  if ($service.CrewOwned -and (Test-MaintenanceSessionActive)) {
1213
1335
  $service.State = 'maintenance'
1214
- $service.LastError = $null
1336
+ # Say why, rather than clearing the field: an empty reason reaches the
1337
+ # startup wait as "deadline exceeded: dsh-crew:3210 ()", which reads like a
1338
+ # fault when the fence is the supervisor doing exactly as it was told.
1339
+ $service.LastError = 'a maintenance session holds the launch right (an update is mid-handoff); auto-start deferred'
1215
1340
  continue
1216
1341
  }
1217
1342
  $health = Get-HealthState $service
@@ -1393,10 +1518,35 @@ try {
1393
1518
  exit 0
1394
1519
  }
1395
1520
 
1396
- Ensure-CrewSupervisorRunning
1397
-
1398
1521
  if ($Mode -eq 'open') {
1399
- Write-LaunchLog 'Official frontend is on 3080; Crew is running silently on 3210.'
1522
+ # A desktop launch promises the frontend on 3080, and that is up in about a
1523
+ # second. The supervisor owns 3210 from the moment it starts — it publishes
1524
+ # its heartbeat before it first touches the port — so this waits for the
1525
+ # watcher to be running, reports where 3210 actually is, and returns. Holding
1526
+ # the operator's window for 3210 readiness bought nothing: the watcher is
1527
+ # doing that wait anyway, and under load a first boot here has taken 78s.
1528
+ if (-not (Wait-CrewSupervisorStarted)) {
1529
+ throw 'No Crew supervisor started within 30s; 3210 has nothing watching it.'
1530
+ }
1531
+ $crew = $services | Where-Object { $_.CrewOwned } | Select-Object -First 1
1532
+ $health = if ($crew) { Get-HealthState $crew } else { $null }
1533
+ if ($health -and $health.Ready) {
1534
+ Write-LaunchLog 'Official frontend is on 3080; Crew is ready on 3210.'
1535
+ } else {
1536
+ Write-LaunchLog ('Official frontend is on 3080; the supervisor is bringing 3210 up in the background. Last health: {0}' -f $health.Error) 'WARN'
1537
+ }
1538
+ # Operator-facing summary rather than a log line: clicking Crew before 3210
1539
+ # answers looks like a broken feature, so say that it is still coming up.
1540
+ Write-Host ''
1541
+ Write-Host 'DSH Crew: the frontend is on http://127.0.0.1:3080.' -ForegroundColor Green
1542
+ if ($health -and $health.Ready) {
1543
+ Write-Host 'Backend 3210 is ready.' -ForegroundColor Green
1544
+ } else {
1545
+ Write-Host 'Backend 3210 is still starting; Crew features appear once it answers.' -ForegroundColor Yellow
1546
+ }
1547
+ Write-Host ("Diagnostic log: {0}" -f $launcherLog) -ForegroundColor DarkGray
1548
+ } else {
1549
+ Ensure-CrewSupervisorRunning
1400
1550
  }
1401
1551
  Write-LaunchLog ('Launcher completed successfully in {0:n1}s.' -f ((Get-Date) - $startedAt).TotalSeconds)
1402
1552
  exit 0