@ran-sh/dsh-crew 2.2.6 → 2.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.2.6",
3
+ "version": "2.2.8",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
@@ -12,10 +12,29 @@ The matrix is emitted by `hubStatus()` and is therefore visible inside the exist
12
12
  - `FAIL` — a check actually ran and produced an incompatible or failed result.
13
13
  - `BLOCKED` — the check could not run because required infrastructure or authorization was unavailable.
14
14
  - `SKIP` — the row is intentionally not applicable for the active policy/path.
15
+ - `NOT_APPLICABLE` — the row asks about a route this machine does not use, and the
16
+ matrix says so instead of omitting it. The two built-in DeepSeek rows are the only
17
+ rows a policy may mark this way: they read `NOT_APPLICABLE` /
18
+ `WORKER_PROVIDER_FOLLOWS_DSH` whenever a known, other provider is selected, and keep
19
+ `NOT_RUN` when the selection is unknown.
15
20
  - `NOT_RUN` — no trusted evidence has been supplied for the row.
16
21
 
17
22
  `BLOCKED` and `SKIP` are not failures. `NOT_RUN` is not success.
18
23
 
24
+ Applicability is decided when the row is built and it is final for that row:
25
+
26
+ - Supplied evidence can neither create nor clear it. `NOT_APPLICABLE` is not part of
27
+ the evidence vocabulary (`PASS`, `FAIL`, `BLOCKED`, `SKIP`, `NOT_RUN`), so no
28
+ evidence source can make a required row disappear from readiness, and evidence
29
+ reported against a not-applicable row is kept as `reported_evidence` metadata
30
+ instead of flipping the status back.
31
+ - A required row that arrives as `NOT_APPLICABLE` anyway is a contradiction, not a
32
+ pass: consumers that project such a row into readiness report
33
+ `UNAVAILABLE` / `CHECK_NOT_APPLICABLE_ON_REQUIRED_ROW`, which cannot leave the
34
+ aggregate READY.
35
+ - The matrix declares `schema_version: 2`: a row may legally be `NOT_APPLICABLE` and
36
+ the summary carries a matching key, which is what a strict consumer needs to know.
37
+
19
38
  ## Evidence classes
20
39
 
21
40
  The matrix separates three kinds of evidence:
@@ -61,7 +80,7 @@ actually run" — `provider_health`, `reviewer_health`, `model_execution`,
61
80
  `worker_primary_callable`, `reviewer_primary_callable`, `reviewer_pipeline` — is
62
81
  filtered to whatever route is currently selected, so an operator on other
63
82
  providers is covered by those and loses nothing by the two DeepSeek rows
64
- reading `NOT_RUN`. A static row per provider was tried and replaced by these
83
+ reading `NOT_APPLICABLE` under `follow-dsh`. A static row per provider was tried and replaced by these
65
84
  dynamic signals; `opencode_go_mimo_qwen` was that attempt's last trace in this
66
85
  document.
67
86
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.2.6",
3
+ "version": "2.2.8",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
package/scripts/setup.mjs CHANGED
@@ -384,11 +384,19 @@ export async function setupStatus({ log = console.log, root = ROOT, home = homed
384
384
  } catch { dshPlugin = 'unknown'; }
385
385
  }
386
386
  log(`DSH plugin: ${dshPlugin} (dedicated dsh-crew profile; official web profile ignored)`);
387
+ // The frontend is on-demand by design: 3210 is the service, and 3080 is started
388
+ // for a browser session. Rendering it as anything else reads as a missing piece
389
+ // of the installation, which it is not.
390
+ const frontendConfig = (installer.readGlobalConfig ?? realInstaller.readGlobalConfig)({ configFile: join(home, '.config', 'dsh-crew', 'config.json') });
391
+ const frontend = frontendConfig?.frontend_autostart === true
392
+ ? 'configured to auto-start at login (dsh-crew open for a session URL)'
393
+ : 'available on demand (dsh-crew open)';
394
+ log(`Frontend (3080): ${frontend}`);
387
395
  log(`Codex Desktop integration: ${codex}`);
388
396
  log(`ZCode integration: ${zcode}`);
389
397
  log(`Claude Code integration: ${claude}`);
390
398
  log(`Windows login startup: ${windowsStartup}`);
391
- return { ok: true, dshPlugin, codex, zcode, claude, windowsStartup };
399
+ return { ok: true, dshPlugin, frontend, codex, zcode, claude, windowsStartup };
392
400
  }
393
401
 
394
402
  export async function runSetupCli({ argv = process.argv.slice(2), run: actions = {}, log = console.log } = {}) {
@@ -29,9 +29,19 @@ function workspaceComponent(workspace) {
29
29
  : { ...component('UNAVAILABLE', workspace?.code ?? 'WORKSPACE_NOT_CHECKED'), state: 'UNAVAILABLE' };
30
30
  }
31
31
 
32
- function readinessFromRow(entry, { pass = 'READY', notRun = 'DEGRADED' } = {}) {
32
+ function readinessFromRow(entry, { pass = 'READY', notRun = 'DEGRADED', allowNotApplicable = false } = {}) {
33
33
  if (!entry) return component('UNAVAILABLE', 'NO_EVIDENCE');
34
34
  if (entry.status === 'PASS') return component(pass, entry.reason_code ?? 'CHECK_PASSED');
35
+ // Every row this function is called with asks a question that applies to any Crew
36
+ // machine (a reachable hub, a consistent provider lifecycle, a runnable route), so
37
+ // "not applicable" is a contradiction here, not a pass: it is reported as missing
38
+ // evidence and it cannot leave the aggregate READY. Only a caller that names a row
39
+ // which may legitimately not apply opts in.
40
+ if (entry.status === 'NOT_APPLICABLE') {
41
+ return allowNotApplicable
42
+ ? component('NOT_APPLICABLE', entry.reason_code ?? 'CHECK_NOT_APPLICABLE')
43
+ : component('UNAVAILABLE', 'CHECK_NOT_APPLICABLE_ON_REQUIRED_ROW');
44
+ }
35
45
  if (entry.status === 'NOT_RUN' || entry.status === 'SKIP') return component(notRun, entry.reason_code ?? 'CHECK_NOT_RUN');
36
46
  return component('UNAVAILABLE', entry.reason_code ?? 'CHECK_FAILED');
37
47
  }
@@ -50,7 +60,12 @@ function modelReadinessFromSnapshot(readinessSnapshot, matrix, runtime, expected
50
60
  });
51
61
  if (validation.ok && validation.state === 'CALLABLE') return component('READY', 'CURRENT_MODEL_CALLABLE', { captured_at: callability.captured_at, expires_at: callability.expires_at, runtime_id: callability.current_runtime_id });
52
62
  if (validation.ok && validation.state === 'NOT_CALLABLE') return component('UNAVAILABLE', callability.roles.worker?.state === 'NOT_CALLABLE' ? callability.roles.worker.reason_code : callability.roles.reviewer.reason_code ?? 'CURRENT_MODEL_NOT_CALLABLE');
63
+ // Aging evidence is reported as aging, not as a fault, and "nothing has proven
64
+ // this route yet" is reported as unknown. The validator's own success code is
65
+ // `MODEL_CALLABILITY_VALID`, and rendering that next to DEGRADED read as a
66
+ // contradiction: the projection was valid, the evidence was simply not current.
53
67
  if (validation.ok && validation.state === 'STALE') return component('DEGRADED', 'MODEL_EVIDENCE_STALE');
68
+ if (validation.ok) return component('DEGRADED', 'MODEL_CALLABILITY_UNKNOWN');
54
69
  return component('DEGRADED', validation.reason_code ?? 'MODEL_CALLABILITY_UNKNOWN');
55
70
  }
56
71
  if (providerHealthEvidence?.status === 'FAIL') return readinessFromRow(providerHealthEvidence);
package/src/hub/index.mjs CHANGED
@@ -642,6 +642,10 @@ export class WorkerRegistry { constructor(ctx) {
642
642
  delivery_complete: !!job.delivery_complete,
643
643
  allow_no_changes: job.allow_no_changes === true,
644
644
  task_status: job.outcome?.task_status ?? null,
645
+ // Whether the worker's execution itself completed, kept beside the task-level
646
+ // verdict: a `partial` task still had a provider response, and that is what
647
+ // model callability asks about.
648
+ execution_status: job.outcome?.execution_status ?? null,
645
649
  workspace_evidence_ok: job.outcome?.workspace_evidence_ok ?? null,
646
650
  review_verdict: job.review?.verdict ?? null,
647
651
  workspace_diff_available: !!job.workspaceDiff && ['git', 'filesystem-empty'].includes(job.workspaceDiff.kind),
@@ -17,6 +17,14 @@ import { integrationRoot } from './crew-paths.mjs';
17
17
  const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..');
18
18
  const MARKETPLACE_NAME = 'dsh-crew';
19
19
  const PLUGIN_KEY = `dsh-crew@${MARKETPLACE_NAME}`;
20
+ // Every name this installer has ever registered Crew under in a Claude host, and
21
+ // the only names its uninstall may remove. `dsh-workers` is the pre-rename
22
+ // identity: an operator who upgraded keeps that marketplace, plugin entry and its
23
+ // permission rules until the same uninstall cleans them, and nothing else in the
24
+ // file is Crew's to touch.
25
+ const CREW_CLAUDE_MARKETPLACES = [MARKETPLACE_NAME, 'dsh-workers'];
26
+ const CREW_CLAUDE_PLUGIN_KEYS = [PLUGIN_KEY, 'dsh-workers@dsh-workers'];
27
+ const CREW_CLAUDE_PERMISSION_PREFIXES = ['mcp__plugin_dsh-crew_', 'mcp__plugin_dsh-workers_'];
20
28
  // A Claude Code plugin refresh is a real copy of the plugin tree, and its cost
21
29
  // tracks machine load: 163s measured idle, ~6 minutes measured during an
22
30
  // activation. The install is the step that copies, so it gets the ceiling that
@@ -645,7 +653,7 @@ export function uninstallCodex({ home = homedir(), env = process.env } = {}) {
645
653
  export function claudeIntegrationLine(result) {
646
654
  if (result?.ok === false) return '✗ Claude Code integration failed';
647
655
  if (result?.detected === false) {
648
- return '- Claude Code not detected; settings registered, CLI step skipped';
656
+ return '- Claude Code not installed (claude CLI not found); nothing written';
649
657
  }
650
658
  if (result?.degraded === true) {
651
659
  return `- Claude Code integration registered, but not loaded: ${result.reason ?? 'plugin snapshot not refreshed'}`;
@@ -906,9 +914,22 @@ export async function runClaudeStep(args, { timeoutMs = CLAUDE_STEP_TIMEOUT_MS,
906
914
  };
907
915
  }
908
916
 
909
- export async function installClaudeCode({ home = homedir(), statusline = false, root = ROOT } = {}) {
917
+ export async function installClaudeCode({ home = homedir(), statusline = false, root = ROOT, resolveClaude = resolveClaudeCommand } = {}) {
910
918
  const actions = [];
911
919
 
920
+ // This registration exists only to make Crew callable FROM Claude Code, so a
921
+ // host without the `claude` CLI has nothing for it to serve. Writing it anyway
922
+ // left Crew's marketplace, plugin entry and permission rules in a file Crew
923
+ // could never exercise, next to a footprint that read back as an integration
924
+ // wanting a CLI the machine does not have. Detect once, up front, and write
925
+ // nothing when the host cannot use it: `host_detected` then reports the truth
926
+ // with no stale registration to explain away.
927
+ const claudeCommand = resolveClaude();
928
+ if (!claudeCommand) {
929
+ actions.push('claude CLI not found; no Claude configuration written');
930
+ return { ok: true, detected: false, actions };
931
+ }
932
+
912
933
  // The repository carries its own marketplace manifest (.claude-plugin/
913
934
  // marketplace.json with source "."), so the marketplace root IS the plugin
914
935
  // checkout itself — no parent-directory layout assumption. This keeps the
@@ -999,7 +1020,8 @@ export async function installClaudeCode({ home = homedir(), statusline = false,
999
1020
  // Materialize the install through the claude CLI: registers the marketplace
1000
1021
  // in the plugin cache (settings alone leave a stale/absent cache entry) and
1001
1022
  // pulls the plugin so the next session loads it. Best-effort: settings are
1002
- // already correct, so a missing CLI just means one manual `plugin install`.
1023
+ // already correct, so a CLI that fails on its own terms leaves the operator one
1024
+ // manual `plugin install` short rather than an unregistered host.
1003
1025
  const registered = readJson(join(home, '.claude', 'plugins', 'known_marketplaces.json'), {})?.[MARKETPLACE_NAME];
1004
1026
  const installedRecord = readJson(join(home, '.claude', 'plugins', 'installed_plugins.json'), {})?.plugins?.[PLUGIN_KEY];
1005
1027
  const installedEntries = Array.isArray(installedRecord) ? installedRecord : [installedRecord];
@@ -1019,14 +1041,9 @@ export async function installClaudeCode({ home = homedir(), statusline = false,
1019
1041
  actions.push('cli: skipped (non-default home; test mode)');
1020
1042
  return { ok: true, actions };
1021
1043
  }
1022
- // The CLI is best-effort, but a host that does not have it must not be told to
1023
- // run it. The checkout entry already gates on this; doing it here too keeps the
1024
- // two entries describing the same machine the same way.
1025
- const claudeCommand = resolveClaudeCommand();
1026
- if (!claudeCommand) {
1027
- actions.push('cli: skipped (claude not found)');
1028
- return { ok: true, detected: false, actions };
1029
- }
1044
+ // `claudeCommand` was resolved before anything was written: reaching this line
1045
+ // already proves the CLI is there, so the step is best-effort rather than
1046
+ // possibly-pointless.
1030
1047
  return refreshClaudePlugin({ home, root, actions, claudeCommand });
1031
1048
  }
1032
1049
 
@@ -1230,25 +1247,94 @@ export function installHudSegment({ home = homedir() } = {}) {
1230
1247
  return { ok: true, actions: [...(bak ? [`backup: ${bak}`] : []), 'statusLine: claude-hud now runs worker-segment.sh via --extra-cmd'] };
1231
1248
  }
1232
1249
 
1250
+ // Crew-owned directories, identified by path SEGMENT rather than by substring: the
1251
+ // payload tree lives under `.../dsh-crew/...` and the pre-rename identity used
1252
+ // `dsh-workers`. A user path that merely contains those words (`my-dsh-crew-notes`)
1253
+ // is not Crew's, and nothing here may treat it as if it were.
1254
+ function crewOwnedDirectory(value) {
1255
+ if (typeof value !== 'string' || !value.trim()) return false;
1256
+ return value.split(/[\\/]+/).filter(Boolean).some((segment) => {
1257
+ const lower = segment.toLowerCase();
1258
+ return lower === 'dsh-crew' || lower === 'dsh-workers';
1259
+ });
1260
+ }
1261
+
1262
+ // Only the status line `--statusline` installs is Crew's to remove: it points at
1263
+ // Crew's own script, in Crew's own directory, in the exact shape the installer
1264
+ // writes (`bash <root>/statusline/statusline.sh`). A command that merely mentions
1265
+ // these words belongs to whoever wrote it.
1266
+ function crewOwnedStatusLine(value) {
1267
+ const command = typeof value?.command === 'string' ? value.command.trim() : '';
1268
+ const match = /^bash\s+(.+[\\/])statusline[\\/]statusline\.sh$/i.exec(command);
1269
+ return match ? crewOwnedDirectory(match[1]) : false;
1270
+ }
1271
+
1272
+ // The claude CLI records, per marketplace and plugin, what it materialized, in
1273
+ // files of its own. Unregistering the settings and leaving those records behind
1274
+ // hands the next install a snapshot with no owner, so an explicit uninstall
1275
+ // clears them too. Crew-scoped and best-effort: a file that is absent, unreadable
1276
+ // or shaped differently than expected is left exactly as it is found.
1277
+ function removeCrewClaudeCacheRecords({ home, actions }) {
1278
+ const dir = join(home, '.claude', 'plugins');
1279
+ const marketsFile = join(dir, 'known_marketplaces.json');
1280
+ const markets = readJson(marketsFile, null);
1281
+ if (markets && typeof markets === 'object' && !Array.isArray(markets)) {
1282
+ const owned = CREW_CLAUDE_MARKETPLACES.filter((name) => Object.hasOwn(markets, name));
1283
+ if (owned.length) {
1284
+ backup(marketsFile);
1285
+ for (const name of owned) delete markets[name];
1286
+ writeFileSync(marketsFile, JSON.stringify(markets, null, 2) + '\n');
1287
+ actions.push(`marketplace cache: removed ${owned.join(', ')}`);
1288
+ }
1289
+ }
1290
+ const pluginsFile = join(dir, 'installed_plugins.json');
1291
+ const installed = readJson(pluginsFile, null);
1292
+ if (installed && typeof installed === 'object' && !Array.isArray(installed)
1293
+ && installed.plugins && typeof installed.plugins === 'object' && !Array.isArray(installed.plugins)) {
1294
+ const owned = CREW_CLAUDE_PLUGIN_KEYS.filter((key) => Object.hasOwn(installed.plugins, key));
1295
+ if (owned.length) {
1296
+ backup(pluginsFile);
1297
+ for (const key of owned) delete installed.plugins[key];
1298
+ writeFileSync(pluginsFile, JSON.stringify(installed, null, 2) + '\n');
1299
+ actions.push(`plugin cache: removed ${owned.join(', ')}`);
1300
+ }
1301
+ }
1302
+ }
1303
+
1233
1304
  export function uninstallClaudeCode({ home = homedir() } = {}) {
1305
+ const actions = [];
1234
1306
  const settingsFile = join(home, '.claude', 'settings.json');
1235
1307
  const settings = readJson(settingsFile, null);
1236
- if (!settings) return { ok: true, actions: ['settings.json not found'] };
1237
- backup(settingsFile);
1238
- const mpDir = join(home, '.config', 'dsh-crew', 'marketplace');
1239
- if (Array.isArray(settings.extraKnownMarketplaces)) {
1240
- settings.extraKnownMarketplaces = settings.extraKnownMarketplaces.filter((m) => m?.path !== mpDir);
1241
- } else if (settings.extraKnownMarketplaces) {
1242
- delete settings.extraKnownMarketplaces[MARKETPLACE_NAME];
1243
- }
1244
- if (Array.isArray(settings.enabledPlugins)) {
1245
- settings.enabledPlugins = settings.enabledPlugins.filter((p) => p !== PLUGIN_KEY);
1246
- } else if (settings.enabledPlugins) {
1247
- delete settings.enabledPlugins[PLUGIN_KEY];
1248
- }
1249
- if (settings.permissions?.allow) {
1250
- settings.permissions.allow = settings.permissions.allow.filter((r) => !r.startsWith('mcp__plugin_dsh-crew_'));
1308
+ if (settings) {
1309
+ backup(settingsFile);
1310
+ const mpDir = join(home, '.config', 'dsh-crew', 'marketplace');
1311
+ if (Array.isArray(settings.extraKnownMarketplaces)) {
1312
+ // Legacy array shape: entries are `{ path }` records this installer wrote under
1313
+ // one of its own names, including the pre-rename one. Anything else in that
1314
+ // array belongs to another plugin and stays.
1315
+ settings.extraKnownMarketplaces = settings.extraKnownMarketplaces
1316
+ .filter((entry) => !crewOwnedDirectory(entry?.path) && entry?.path !== mpDir);
1317
+ } else if (settings.extraKnownMarketplaces) {
1318
+ for (const name of CREW_CLAUDE_MARKETPLACES) delete settings.extraKnownMarketplaces[name];
1319
+ }
1320
+ if (Array.isArray(settings.enabledPlugins)) {
1321
+ settings.enabledPlugins = settings.enabledPlugins.filter((p) => !CREW_CLAUDE_PLUGIN_KEYS.includes(p));
1322
+ } else if (settings.enabledPlugins) {
1323
+ for (const key of CREW_CLAUDE_PLUGIN_KEYS) delete settings.enabledPlugins[key];
1324
+ }
1325
+ if (settings.permissions?.allow) {
1326
+ settings.permissions.allow = settings.permissions.allow.filter((rule) => typeof rule !== 'string'
1327
+ || !CREW_CLAUDE_PERMISSION_PREFIXES.some((prefix) => rule.startsWith(prefix)));
1328
+ }
1329
+ if (crewOwnedStatusLine(settings.statusLine)) {
1330
+ delete settings.statusLine;
1331
+ actions.push('statusLine: removed (Crew-installed)');
1332
+ }
1333
+ writeFileSync(settingsFile, JSON.stringify(settings, null, 2) + '\n');
1334
+ actions.push('unregistered from settings.json (backup kept)');
1335
+ } else {
1336
+ actions.push('settings.json not found');
1251
1337
  }
1252
- writeFileSync(settingsFile, JSON.stringify(settings, null, 2) + '\n');
1253
- return { ok: true, actions: ['unregistered from settings.json (backup kept)'] };
1338
+ removeCrewClaudeCacheRecords({ home, actions });
1339
+ return { ok: true, actions };
1254
1340
  }
@@ -3327,6 +3327,15 @@ export function npxStatus({
3327
3327
  }
3328
3328
  log(`DSH plugin: ${dshPlugin} (dedicated dsh-crew profile on 3210)`);
3329
3329
  log(`Official 3080 UI bridge: ${officialWeb}`);
3330
+ // The frontend is not an install that can be missing: 3210 is the service, and
3331
+ // 3080 is started on demand. Saying so is the difference between "this machine is
3332
+ // incomplete" and "run `dsh-crew open` when you want the browser surface" — and
3333
+ // whether 3080 happens to be up right now does not change either answer.
3334
+ const frontendConfig = (installer.readGlobalConfig ?? realInstaller.readGlobalConfig)({ configFile: join(home, '.config', 'dsh-crew', 'config.json') });
3335
+ const frontend = frontendConfig?.frontend_autostart === true
3336
+ ? 'configured to auto-start at login (dsh-crew open for a session URL)'
3337
+ : 'available on demand (dsh-crew open)';
3338
+ log(`Frontend (3080): ${frontend}`);
3330
3339
  // The desktop app carries its own bridge entry, which pins a revision: after a
3331
3340
  // payload update that entry is stale until it is re-pointed, so say which of the
3332
3341
  // three states the machine is in instead of leaving the panel silently old.
@@ -3349,6 +3358,7 @@ export function npxStatus({
3349
3358
  installedPath: pointer?.path ?? null,
3350
3359
  dshPlugin,
3351
3360
  officialWeb,
3361
+ frontend,
3352
3362
  codex,
3353
3363
  zcode,
3354
3364
  claude,
@@ -19,9 +19,10 @@ function plan(home, root) {
19
19
  // `core.autocrlf=true` checkout hashes to a different revision than the npm
20
20
  // payload it installs, which made readiness look for a snapshot that does not
21
21
  // exist and reported "needs repair" on a correct install (and made every payload
22
- // update look like a stale desktop revision).
22
+ // update look like a stale desktop revision). Lone CR collapses too: one line
23
+ // break is one line break in these sources.
23
24
  const canonicalBytes = (bytes) => (bytes.includes(0) ? bytes
24
- : Buffer.from(bytes.toString('utf8').replace(/\r\n/g, '\n'), 'utf8'));
25
+ : Buffer.from(bytes.toString('utf8').replace(/\r\n?/g, '\n'), 'utf8'));
25
26
  for (const [name, bytes] of files) files.set(name, canonicalBytes(bytes));
26
27
  const hash = createHash('sha256').update(JSON.stringify(metadata));
27
28
  for (const [name, bytes] of files) hash.update(name).update(bytes);
@@ -43,10 +44,12 @@ function plan(home, root) {
43
44
  // snapshot was written from the npm payload (LF). The bridge sources are text that
44
45
  // either line ending serves, so a CRLF checkout must not read as a drifted snapshot;
45
46
  // bytes are still compared first, and the tolerant path only applies to text.
47
+ // Lone CR is normalized as well, so a file saved with classic-Mac endings reads as
48
+ // the same source rather than as content drift.
46
49
  function sameTextContent(installed, expected) {
47
50
  if (installed.equals(expected)) return true;
48
51
  if (installed.includes(0) || expected.includes(0)) return false;
49
- const normalize = (buffer) => buffer.toString('utf8').replace(/\r\n/g, '\n');
52
+ const normalize = (buffer) => buffer.toString('utf8').replace(/\r\n?/g, '\n');
50
53
  return normalize(installed) === normalize(expected);
51
54
  }
52
55
 
@@ -61,7 +64,7 @@ export function officialFrontendAssetsReady({ home = homedir(), root } = {}) {
61
64
  try {
62
65
  const p = plan(home, root);
63
66
  const overlay = readFileSync(p.overlayFile, 'utf8');
64
- return matches(p) && (overlay === p.overlay || overlay.replace(/\r\n/g, '\n') === p.overlay.replace(/\r\n/g, '\n'));
67
+ return matches(p) && (overlay === p.overlay || overlay.replace(/\r\n?/g, '\n') === p.overlay.replace(/\r\n?/g, '\n'));
65
68
  } catch { return false; }
66
69
  }
67
70
 
@@ -129,9 +129,12 @@ function sha256File(file) {
129
129
  // Git checks these scripts out as CRLF on a `core.autocrlf=true` machine and
130
130
  // PowerShell accepts either, so a line-ending difference is not a difference this
131
131
  // check may report: comparing raw bytes made every Windows dev checkout read as
132
- // "needs repair" while the npm payload (LF) was installed correctly.
132
+ // "needs repair" while the npm payload (LF) was installed correctly. Lone CR is
133
+ // collapsed too — an editor that saved one of these files with classic-Mac
134
+ // endings is still the same script, and only a real character difference should
135
+ // send the operator to a repair step.
133
136
  function normalizeEol(text) {
134
- return String(text).replace(/\r\n/g, '\n');
137
+ return String(text).replace(/\r\n?/g, '\n');
135
138
  }
136
139
 
137
140
  function samePath(left, right) {
package/src/log-prune.mjs CHANGED
@@ -10,14 +10,19 @@ import { readdirSync, rmSync, statSync } from 'node:fs';
10
10
  import { join } from 'node:path';
11
11
  import { tmpdir } from 'node:os';
12
12
 
13
- export const DEFAULT_KEEP_RUNS = 20;
13
+ export const DEFAULT_KEEP_RUNS = 10;
14
14
  export const DEFAULT_MAX_AGE_DAYS = 14;
15
15
  const RECENT_GUARD_MS = 5 * 60 * 1000;
16
16
 
17
+ // Per family. The hub-start pairs are what a crash-before-readiness leaves behind,
18
+ // and ten of them are the post-mortem window the policy names; the frontend families
19
+ // are launched far less often and keep the retention they already had, so one
20
+ // family's bound is not silently imposed on the others. An explicit `keepRuns`
21
+ // still applies to every family: that is an operator stating how many runs to keep.
17
22
  const LOG_FAMILIES = [
18
- { prefix: 'dsh-crew-dsh-crew-3210-', suffix: '.out.log' },
19
- { prefix: 'dsh-crew-web-', suffix: '.out.log' },
20
- { prefix: 'dsh-official-web-', suffix: '.out.log' },
23
+ { prefix: 'dsh-crew-dsh-crew-3210-', suffix: '.out.log', keep: 10 },
24
+ { prefix: 'dsh-crew-web-', suffix: '.out.log', keep: 20 },
25
+ { prefix: 'dsh-official-web-', suffix: '.out.log', keep: 20 },
21
26
  ];
22
27
 
23
28
  // `dsh-crew-<profile>-<port>-<stamp>.out.log` -> the stamp identifies one run, and
@@ -28,8 +33,11 @@ function runStamp(fileName, family) {
28
33
  return stamp || null;
29
34
  }
30
35
 
31
- export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = DEFAULT_KEEP_RUNS, maxAgeDays = DEFAULT_MAX_AGE_DAYS, now = Date.now() } = {}) {
32
- const keep = Number.isInteger(keepRuns) && keepRuns > 0 ? keepRuns : DEFAULT_KEEP_RUNS;
36
+ export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = null, maxAgeDays = DEFAULT_MAX_AGE_DAYS, now = Date.now() } = {}) {
37
+ // `keepRuns` is an explicit operator override for every family; without one each
38
+ // family uses its own bound, so the default is "nothing was asked for" rather than
39
+ // a number that would flatten them all to the hub's ten.
40
+ const keepOverride = Number.isInteger(keepRuns) && keepRuns > 0 ? keepRuns : null;
33
41
  const maxAgeMs = Number.isFinite(maxAgeDays) && maxAgeDays > 0 ? maxAgeDays * 24 * 60 * 60 * 1000 : DEFAULT_MAX_AGE_DAYS * 24 * 60 * 60 * 1000;
34
42
  let names;
35
43
  try { names = readdirSync(tempDir); } catch { return { ok: true, removed: [], kept: [], skipped: 'unreadable' }; }
@@ -50,10 +58,16 @@ export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = DEFAULT_KEEP_
50
58
  runs.set(stamp, run);
51
59
  }
52
60
  const ordered = [...runs.values()].sort((left, right) => right.newest - left.newest);
61
+ const keep = keepOverride ?? family.keep;
53
62
  ordered.forEach((run, index) => {
54
63
  const tooOld = now - run.newest > maxAgeMs;
55
64
  const tooMany = index >= keep;
56
65
  const tooRecentToTouch = now - run.newest < RECENT_GUARD_MS;
66
+ // Delete when EITHER bound is exceeded. The newest `keep` runs are what an
67
+ // operator reads, and past that a run beyond the age bound is gone regardless
68
+ // of its rank: keeping an eleventh run merely because it is only hours old
69
+ // would let a burst of starts grow without bound, which is the growth this
70
+ // bound exists to stop. A burst of eleven young runs still keeps ten.
57
71
  if ((tooOld || tooMany) && !tooRecentToTouch) {
58
72
  for (const name of run.files) removed.push(name);
59
73
  const errName = run.files[0].replace(/\.out\.log$/, '.err.log');
@@ -5,7 +5,18 @@
5
5
  // (Hub handshake / catalog read) are intentionally separate from execution
6
6
  // verification (real worker, reviewer, cancellation, timeout, etc.).
7
7
 
8
- const READINESS_STATUSES = Object.freeze(['PASS', 'FAIL', 'BLOCKED', 'SKIP', 'NOT_RUN']);
8
+ const READINESS_STATUSES = Object.freeze(['PASS', 'FAIL', 'BLOCKED', 'SKIP', 'NOT_APPLICABLE', 'NOT_RUN']);
9
+ // Statuses a trusted evidence source may report. `NOT_APPLICABLE` is deliberately
10
+ // absent: applicability is a fact about this machine's policy, decided when the row
11
+ // is built, and letting evidence declare a required row "not applicable" would be a
12
+ // way to make a check disappear while the matrix still reads healthy. Evidence can
13
+ // neither create nor clear it.
14
+ const EVIDENCE_STATUSES = Object.freeze(['PASS', 'FAIL', 'BLOCKED', 'SKIP', 'NOT_RUN']);
15
+ // The rows this release may decide are not applicable, and only when the reason is
16
+ // policy rather than observation: the two built-in DeepSeek routes exist to name a
17
+ // provider the operator may not be using. Every other row asks a question that is
18
+ // always applicable to a Crew machine.
19
+ const POLICY_NOT_APPLICABLE_ROWS = Object.freeze(['deepseek_flash', 'deepseek_pro']);
9
20
 
10
21
  export const READINESS_REASON_CODES = Object.freeze({
11
22
  LIVE_CHECK_PASSED: 'LIVE_CHECK_PASSED',
@@ -22,7 +33,7 @@ export const READINESS_REASON_CODES = Object.freeze({
22
33
  NO_CI_EVIDENCE: 'NO_CI_EVIDENCE',
23
34
  NO_EXECUTION_EVIDENCE: 'NO_EXECUTION_EVIDENCE',
24
35
  CREDENTIAL_STATUS_NOT_PROBED: 'CREDENTIAL_STATUS_NOT_PROBED',
25
- ROUTE_NOT_SELECTED: 'ROUTE_NOT_SELECTED',
36
+ WORKER_PROVIDER_FOLLOWS_DSH: 'WORKER_PROVIDER_FOLLOWS_DSH',
26
37
  EVIDENCE_REPORTED: 'EVIDENCE_REPORTED',
27
38
  });
28
39
 
@@ -59,7 +70,15 @@ function baseRow(id, category, status, reasonCode, evidenceSource = 'none', extr
59
70
 
60
71
  function normalizeEvidence(row, evidence) {
61
72
  if (!evidence || typeof evidence !== 'object') return row;
62
- if (!READINESS_STATUSES.includes(evidence.status)) return row;
73
+ if (!EVIDENCE_STATUSES.includes(evidence.status)) return row;
74
+ // A row the policy marked not applicable keeps that status: reported evidence says
75
+ // something happened, not that the current worker route changed, and letting it
76
+ // flip the row is how "this route is unused" would turn into "this route passed".
77
+ // What was reported is kept as metadata rather than thrown away.
78
+ if (row.status === 'NOT_APPLICABLE') {
79
+ if (typeof evidence.status !== 'string') return row;
80
+ return { ...row, reported_evidence: { status: evidence.status, ...(typeof evidence.reason_code === 'string' && evidence.reason_code.trim() ? { reason_code: evidence.reason_code.trim() } : {}), ...(typeof evidence.evidence_source === 'string' && evidence.evidence_source.trim() ? { evidence_source: evidence.evidence_source.trim() } : {}) } };
81
+ }
63
82
  const reason = typeof evidence.reason_code === 'string' && evidence.reason_code.trim()
64
83
  ? evidence.reason_code.trim()
65
84
  : READINESS_REASON_CODES.EVIDENCE_REPORTED;
@@ -166,15 +185,19 @@ export function buildReadinessMatrix({
166
185
  // The two built-in DeepSeek rows describe a route only when the operator runs
167
186
  // it. Under follow-dsh with a KNOWN, other selected provider they can never
168
187
  // gather evidence, and NOT_RUN reads like a check that should have run rather
169
- // than a route that was never chosen. An unknown selection stays NOT_RUN: this
170
- // matrix is conservative, and "we could not tell" must not read as "not
171
- // applicable". Reported evidence still wins, because it is applied below.
188
+ // than a route that was never chosen. They are reported NOT_APPLICABLE: the rows
189
+ // stay in the matrix (dropping them would hide that the question was asked and
190
+ // answered), and neither status can move the aggregate — only a FAIL, an
191
+ // UNPROVEN row or an UNAVAILABLE component does. An unknown selection stays
192
+ // NOT_RUN: this matrix is conservative, and "we could not tell" must not read as
193
+ // "not applicable". This policy decision is also final for the row: evidence
194
+ // applied below records what it saw without changing the status back.
172
195
  const selectionKnown = typeof workerSelection?.provider === 'string' && workerSelection.provider.length > 0;
173
196
  const deepseekRouteSelected = workerProviderMode === 'deepseek-official'
174
197
  || workerSelection?.provider === 'deepseek-official';
175
198
  if (workerProviderMode !== null && selectionKnown && !deepseekRouteSelected) {
176
- for (const id of ['deepseek_flash', 'deepseek_pro']) {
177
- rows[id] = baseRow(id, 'real-execution', 'SKIP', READINESS_REASON_CODES.ROUTE_NOT_SELECTED, 'runtime-policy');
199
+ for (const id of POLICY_NOT_APPLICABLE_ROWS) {
200
+ rows[id] = baseRow(id, 'real-execution', 'NOT_APPLICABLE', READINESS_REASON_CODES.WORKER_PROVIDER_FOLLOWS_DSH, 'runtime-policy');
178
201
  }
179
202
  }
180
203
 
@@ -189,7 +212,10 @@ export function buildReadinessMatrix({
189
212
  ]));
190
213
 
191
214
  return {
192
- schema_version: 1,
215
+ // A row can now legally be NOT_APPLICABLE and the summary carries a matching key,
216
+ // so the vocabulary a strict consumer must accept changed: that is what the
217
+ // version means here, not a change in the rules that produce PASS.
218
+ schema_version: 2,
193
219
  platform,
194
220
  conservative: true,
195
221
  summary: counts,
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.2.6';
39
+ export const RUNTIME_VERSION = '2.2.8';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -108,9 +108,15 @@ function projectModelRole({ role, selected, health, jobs, runtime, now, executio
108
108
  if (healthCurrent && matchingHealth.state === 'callable') {
109
109
  return { state: 'CALLABLE', reason_code: matchingHealth.reason_code ?? 'PROVIDER_CALLABLE', selected, observed_at: matchingHealth.observed_at ?? null, expires_at: expiresAt, source: 'provider_health', last_success: null };
110
110
  }
111
+ // What proves the route ran is a provider response, not a passing task contract:
112
+ // a job that reached `done` with its execution completed has demonstrably called
113
+ // the selected model even when the task itself came back `partial`. Failed and
114
+ // cancelled work never reaches this predicate, and neither does a job that errored
115
+ // before the model answered (its execution_status is `failed`, not `completed`).
116
+ const responded = (job) => job?.task_status === 'success' || job?.execution_status === 'completed';
111
117
  const completed = (Array.isArray(jobs) ? jobs : [])
112
118
  .filter((job) => job?.role === role && job?.provider === selected.provider && job?.model === selected.model
113
- && job?.status === 'done' && job?.task_status === 'success'
119
+ && job?.status === 'done' && responded(job)
114
120
  && sameCompleteRuntimeIdentity(job.execution_context, runtime))
115
121
  .map((job) => ({ job, endedAt: Date.parse(job.endedAt ?? '') }))
116
122
  .filter(({ endedAt }) => Number.isFinite(endedAt))
@@ -32,6 +32,7 @@ if not "%~1"=="" goto :invalid_argument
32
32
  set "LAUNCH_HELPER=%LAUNCH_DIR%start-dsh-crew.ps1"
33
33
  set "LAUNCH_LOG=%TEMP%\dsh-crew-launcher.log"
34
34
  if not exist "%LAUNCH_HELPER%" (
35
+ call :rotate_launcher_log
35
36
  >>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
36
37
  echo ERROR: DSH Crew launcher helper is missing.
37
38
  echo Repair it with: dsh-crew update
@@ -51,6 +52,21 @@ echo ERROR: Unsupported launcher arguments: %LAUNCH_REQUEST%
51
52
  echo Use --open, --background, or --watch.
52
53
  exit /b 64
53
54
 
55
+ :rotate_launcher_log
56
+ rem The PowerShell launcher bounds this log at 5 MiB before its first line, but the
57
+ rem emergency path that calls this is what runs when that helper is MISSING — so a
58
+ rem bounded log cannot depend on the helper whose absence is one of the failures the
59
+ rem log exists to diagnose. Same cap, same two generations, best-effort: a locked or
60
+ rem unreadable log is left alone rather than failing the launch.
61
+ set "LAUNCH_LOG_SIZE="
62
+ for %%A in ("%LAUNCH_LOG%") do set "LAUNCH_LOG_SIZE=%%~zA"
63
+ if not defined LAUNCH_LOG_SIZE goto :eof
64
+ if %LAUNCH_LOG_SIZE% LSS 5242880 goto :eof
65
+ if exist "%LAUNCH_LOG%.2" del /f /q "%LAUNCH_LOG%.2" >nul 2>&1
66
+ if exist "%LAUNCH_LOG%.1" move /y "%LAUNCH_LOG%.1" "%LAUNCH_LOG%.2" >nul 2>&1
67
+ move /y "%LAUNCH_LOG%" "%LAUNCH_LOG%.1" >nul 2>&1
68
+ goto :eof
69
+
54
70
  :help
55
71
  echo Usage: %~nx0 [--open ^| --background ^| --watch]
56
72
  echo --open Open official Harness on 3080 and return once Crew is supervised;
@@ -127,6 +127,24 @@ function Write-LaunchLog {
127
127
  }
128
128
  }
129
129
 
130
+ # Every launch appends to one file for the life of the machine, so it grows without
131
+ # bound while only its tail is ever read. Rotation runs once per launch before the
132
+ # first line is written and keeps two generations: a post-mortem needs the current
133
+ # run and the one before it. Deliberately best-effort — a locked, unreadable or
134
+ # huge log is a reason to leave it alone, never a reason to fail the launch.
135
+ $launcherLogMaxBytes = 5MB
136
+ function Rotate-LaunchLog {
137
+ try {
138
+ if (-not (Test-Path -LiteralPath $launcherLog -PathType Leaf)) { return }
139
+ if ((Get-Item -LiteralPath $launcherLog -ErrorAction Stop).Length -lt $launcherLogMaxBytes) { return }
140
+ $oldest = "$launcherLog.2"
141
+ $previous = "$launcherLog.1"
142
+ if (Test-Path -LiteralPath $oldest) { Remove-Item -LiteralPath $oldest -Force -ErrorAction Stop }
143
+ if (Test-Path -LiteralPath $previous) { Move-Item -LiteralPath $previous -Destination $oldest -Force -ErrorAction Stop }
144
+ Move-Item -LiteralPath $launcherLog -Destination $previous -Force -ErrorAction Stop
145
+ } catch { }
146
+ }
147
+
130
148
  function Resolve-OfficialHarnessCommand {
131
149
  $shim = Get-Command dsh.cmd -CommandType Application -ErrorAction Stop | Select-Object -First 1
132
150
  $root = Join-Path (Split-Path -Parent $shim.Source) 'node_modules\@deepseek-ai\dsh'
@@ -1528,6 +1546,7 @@ if ($env:DSH_CREW_LAUNCHER_TEST_IMPORT -eq '1') { return }
1528
1546
 
1529
1547
  try {
1530
1548
  New-Item -ItemType Directory -Path $logRoot -Force | Out-Null
1549
+ Rotate-LaunchLog
1531
1550
  Write-LaunchLog ('Launcher started; mode={0}; user={1}' -f $Mode, $env:USERNAME)
1532
1551
  if (-not (Test-Path -LiteralPath $dshCli -PathType Leaf)) {
1533
1552
  throw "DSH CLI was not found at $dshCli. Configure DSH_CREW_DSH_CLI to a Crew-owned official CLI entry, or run: npm install -g @ran-sh/dsh-crew@latest; dsh-crew update"