@ran-sh/dsh-crew 2.2.7 → 2.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.2.7",
3
+ "version": "2.2.8",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
@@ -13,15 +13,27 @@ The matrix is emitted by `hubStatus()` and is therefore visible inside the exist
13
13
  - `BLOCKED` — the check could not run because required infrastructure or authorization was unavailable.
14
14
  - `SKIP` — the row is intentionally not applicable for the active policy/path.
15
15
  - `NOT_APPLICABLE` — the row asks about a route this machine does not use, and the
16
- matrix says so instead of omitting it. The two built-in DeepSeek rows are the live
17
- case: they read `NOT_APPLICABLE` / `WORKER_PROVIDER_FOLLOWS_DSH` whenever a known,
18
- other provider is selected, and keep `NOT_RUN` when the selection is unknown.
16
+ matrix says so instead of omitting it. The two built-in DeepSeek rows are the only
17
+ rows a policy may mark this way: they read `NOT_APPLICABLE` /
18
+ `WORKER_PROVIDER_FOLLOWS_DSH` whenever a known, other provider is selected, and keep
19
+ `NOT_RUN` when the selection is unknown.
19
20
  - `NOT_RUN` — no trusted evidence has been supplied for the row.
20
21
 
21
- `BLOCKED` and `SKIP` are not failures. `NOT_RUN` is not success, and `NOT_APPLICABLE`
22
- is an answer: it is reported as itself on any component that consumes the row, and it
23
- cannot lower the readiness aggregate — only a `FAIL`, an unproven row or an
24
- `UNAVAILABLE` component does.
22
+ `BLOCKED` and `SKIP` are not failures. `NOT_RUN` is not success.
23
+
24
+ Applicability is decided when the row is built and it is final for that row:
25
+
26
+ - Supplied evidence can neither create nor clear it. `NOT_APPLICABLE` is not part of
27
+ the evidence vocabulary (`PASS`, `FAIL`, `BLOCKED`, `SKIP`, `NOT_RUN`), so no
28
+ evidence source can make a required row disappear from readiness, and evidence
29
+ reported against a not-applicable row is kept as `reported_evidence` metadata
30
+ instead of flipping the status back.
31
+ - A required row that arrives as `NOT_APPLICABLE` anyway is a contradiction, not a
32
+ pass: consumers that project such a row into readiness report
33
+ `UNAVAILABLE` / `CHECK_NOT_APPLICABLE_ON_REQUIRED_ROW`, which cannot leave the
34
+ aggregate READY.
35
+ - The matrix declares `schema_version: 2`: a row may legally be `NOT_APPLICABLE` and
36
+ the summary carries a matching key, which is what a strict consumer needs to know.
25
37
 
26
38
  ## Evidence classes
27
39
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.2.7",
3
+ "version": "2.2.8",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
package/scripts/setup.mjs CHANGED
@@ -389,7 +389,7 @@ export async function setupStatus({ log = console.log, root = ROOT, home = homed
389
389
  // of the installation, which it is not.
390
390
  const frontendConfig = (installer.readGlobalConfig ?? realInstaller.readGlobalConfig)({ configFile: join(home, '.config', 'dsh-crew', 'config.json') });
391
391
  const frontend = frontendConfig?.frontend_autostart === true
392
- ? 'auto-started at login (dsh-crew open for a session URL)'
392
+ ? 'configured to auto-start at login (dsh-crew open for a session URL)'
393
393
  : 'available on demand (dsh-crew open)';
394
394
  log(`Frontend (3080): ${frontend}`);
395
395
  log(`Codex Desktop integration: ${codex}`);
@@ -29,13 +29,19 @@ function workspaceComponent(workspace) {
29
29
  : { ...component('UNAVAILABLE', workspace?.code ?? 'WORKSPACE_NOT_CHECKED'), state: 'UNAVAILABLE' };
30
30
  }
31
31
 
32
- function readinessFromRow(entry, { pass = 'READY', notRun = 'DEGRADED' } = {}) {
32
+ function readinessFromRow(entry, { pass = 'READY', notRun = 'DEGRADED', allowNotApplicable = false } = {}) {
33
33
  if (!entry) return component('UNAVAILABLE', 'NO_EVIDENCE');
34
34
  if (entry.status === 'PASS') return component(pass, entry.reason_code ?? 'CHECK_PASSED');
35
- // A row that does not apply to this machine is reported as such, not as a check
36
- // that has not run: NOT_APPLICABLE says the question was answered, and it cannot
37
- // pull the aggregate down (see the readiness roll-up below).
38
- if (entry.status === 'NOT_APPLICABLE') return component('NOT_APPLICABLE', entry.reason_code ?? 'CHECK_NOT_APPLICABLE');
35
+ // Every row this function is called with asks a question that applies to any Crew
36
+ // machine (a reachable hub, a consistent provider lifecycle, a runnable route), so
37
+ // "not applicable" is a contradiction here, not a pass: it is reported as missing
38
+ // evidence and it cannot leave the aggregate READY. Only a caller that names a row
39
+ // which may legitimately not apply opts in.
40
+ if (entry.status === 'NOT_APPLICABLE') {
41
+ return allowNotApplicable
42
+ ? component('NOT_APPLICABLE', entry.reason_code ?? 'CHECK_NOT_APPLICABLE')
43
+ : component('UNAVAILABLE', 'CHECK_NOT_APPLICABLE_ON_REQUIRED_ROW');
44
+ }
39
45
  if (entry.status === 'NOT_RUN' || entry.status === 'SKIP') return component(notRun, entry.reason_code ?? 'CHECK_NOT_RUN');
40
46
  return component('UNAVAILABLE', entry.reason_code ?? 'CHECK_FAILED');
41
47
  }
package/src/hub/index.mjs CHANGED
@@ -642,6 +642,10 @@ export class WorkerRegistry { constructor(ctx) {
642
642
  delivery_complete: !!job.delivery_complete,
643
643
  allow_no_changes: job.allow_no_changes === true,
644
644
  task_status: job.outcome?.task_status ?? null,
645
+ // Whether the worker's execution itself completed, kept beside the task-level
646
+ // verdict: a `partial` task still had a provider response, and that is what
647
+ // model callability asks about.
648
+ execution_status: job.outcome?.execution_status ?? null,
645
649
  workspace_evidence_ok: job.outcome?.workspace_evidence_ok ?? null,
646
650
  review_verdict: job.review?.verdict ?? null,
647
651
  workspace_diff_available: !!job.workspaceDiff && ['git', 'filesystem-empty'].includes(job.workspaceDiff.kind),
@@ -1247,12 +1247,26 @@ export function installHudSegment({ home = homedir() } = {}) {
1247
1247
  return { ok: true, actions: [...(bak ? [`backup: ${bak}`] : []), 'statusLine: claude-hud now runs worker-segment.sh via --extra-cmd'] };
1248
1248
  }
1249
1249
 
1250
+ // Crew-owned directories, identified by path SEGMENT rather than by substring: the
1251
+ // payload tree lives under `.../dsh-crew/...` and the pre-rename identity used
1252
+ // `dsh-workers`. A user path that merely contains those words (`my-dsh-crew-notes`)
1253
+ // is not Crew's, and nothing here may treat it as if it were.
1254
+ function crewOwnedDirectory(value) {
1255
+ if (typeof value !== 'string' || !value.trim()) return false;
1256
+ return value.split(/[\\/]+/).filter(Boolean).some((segment) => {
1257
+ const lower = segment.toLowerCase();
1258
+ return lower === 'dsh-crew' || lower === 'dsh-workers';
1259
+ });
1260
+ }
1261
+
1250
1262
  // Only the status line `--statusline` installs is Crew's to remove: it points at
1251
- // Crew's own script. Any other status line in the file belongs to whoever
1252
- // configured it, even when it happens to name this checkout.
1263
+ // Crew's own script, in Crew's own directory, in the exact shape the installer
1264
+ // writes (`bash <root>/statusline/statusline.sh`). A command that merely mentions
1265
+ // these words belongs to whoever wrote it.
1253
1266
  function crewOwnedStatusLine(value) {
1254
- const command = typeof value?.command === 'string' ? value.command : '';
1255
- return command.includes('statusline.sh') && command.includes('dsh-crew');
1267
+ const command = typeof value?.command === 'string' ? value.command.trim() : '';
1268
+ const match = /^bash\s+(.+[\\/])statusline[\\/]statusline\.sh$/i.exec(command);
1269
+ return match ? crewOwnedDirectory(match[1]) : false;
1256
1270
  }
1257
1271
 
1258
1272
  // The claude CLI records, per marketplace and plugin, what it materialized, in
@@ -1295,7 +1309,11 @@ export function uninstallClaudeCode({ home = homedir() } = {}) {
1295
1309
  backup(settingsFile);
1296
1310
  const mpDir = join(home, '.config', 'dsh-crew', 'marketplace');
1297
1311
  if (Array.isArray(settings.extraKnownMarketplaces)) {
1298
- settings.extraKnownMarketplaces = settings.extraKnownMarketplaces.filter((m) => m?.path !== mpDir);
1312
+ // Legacy array shape: entries are `{ path }` records this installer wrote under
1313
+ // one of its own names, including the pre-rename one. Anything else in that
1314
+ // array belongs to another plugin and stays.
1315
+ settings.extraKnownMarketplaces = settings.extraKnownMarketplaces
1316
+ .filter((entry) => !crewOwnedDirectory(entry?.path) && entry?.path !== mpDir);
1299
1317
  } else if (settings.extraKnownMarketplaces) {
1300
1318
  for (const name of CREW_CLAUDE_MARKETPLACES) delete settings.extraKnownMarketplaces[name];
1301
1319
  }
@@ -3333,7 +3333,7 @@ export function npxStatus({
3333
3333
  // whether 3080 happens to be up right now does not change either answer.
3334
3334
  const frontendConfig = (installer.readGlobalConfig ?? realInstaller.readGlobalConfig)({ configFile: join(home, '.config', 'dsh-crew', 'config.json') });
3335
3335
  const frontend = frontendConfig?.frontend_autostart === true
3336
- ? 'auto-started at login (dsh-crew open for a session URL)'
3336
+ ? 'configured to auto-start at login (dsh-crew open for a session URL)'
3337
3337
  : 'available on demand (dsh-crew open)';
3338
3338
  log(`Frontend (3080): ${frontend}`);
3339
3339
  // The desktop app carries its own bridge entry, which pins a revision: after a
package/src/log-prune.mjs CHANGED
@@ -14,10 +14,15 @@ export const DEFAULT_KEEP_RUNS = 10;
14
14
  export const DEFAULT_MAX_AGE_DAYS = 14;
15
15
  const RECENT_GUARD_MS = 5 * 60 * 1000;
16
16
 
17
+ // Per family. The hub-start pairs are what a crash-before-readiness leaves behind,
18
+ // and ten of them are the post-mortem window the policy names; the frontend families
19
+ // are launched far less often and keep the retention they already had, so one
20
+ // family's bound is not silently imposed on the others. An explicit `keepRuns`
21
+ // still applies to every family: that is an operator stating how many runs to keep.
17
22
  const LOG_FAMILIES = [
18
- { prefix: 'dsh-crew-dsh-crew-3210-', suffix: '.out.log' },
19
- { prefix: 'dsh-crew-web-', suffix: '.out.log' },
20
- { prefix: 'dsh-official-web-', suffix: '.out.log' },
23
+ { prefix: 'dsh-crew-dsh-crew-3210-', suffix: '.out.log', keep: 10 },
24
+ { prefix: 'dsh-crew-web-', suffix: '.out.log', keep: 20 },
25
+ { prefix: 'dsh-official-web-', suffix: '.out.log', keep: 20 },
21
26
  ];
22
27
 
23
28
  // `dsh-crew-<profile>-<port>-<stamp>.out.log` -> the stamp identifies one run, and
@@ -28,8 +33,11 @@ function runStamp(fileName, family) {
28
33
  return stamp || null;
29
34
  }
30
35
 
31
- export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = DEFAULT_KEEP_RUNS, maxAgeDays = DEFAULT_MAX_AGE_DAYS, now = Date.now() } = {}) {
32
- const keep = Number.isInteger(keepRuns) && keepRuns > 0 ? keepRuns : DEFAULT_KEEP_RUNS;
36
+ export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = null, maxAgeDays = DEFAULT_MAX_AGE_DAYS, now = Date.now() } = {}) {
37
+ // `keepRuns` is an explicit operator override for every family; without one each
38
+ // family uses its own bound, so the default is "nothing was asked for" rather than
39
+ // a number that would flatten them all to the hub's ten.
40
+ const keepOverride = Number.isInteger(keepRuns) && keepRuns > 0 ? keepRuns : null;
33
41
  const maxAgeMs = Number.isFinite(maxAgeDays) && maxAgeDays > 0 ? maxAgeDays * 24 * 60 * 60 * 1000 : DEFAULT_MAX_AGE_DAYS * 24 * 60 * 60 * 1000;
34
42
  let names;
35
43
  try { names = readdirSync(tempDir); } catch { return { ok: true, removed: [], kept: [], skipped: 'unreadable' }; }
@@ -50,10 +58,16 @@ export function pruneCrewTempLogs({ tempDir = tmpdir(), keepRuns = DEFAULT_KEEP_
50
58
  runs.set(stamp, run);
51
59
  }
52
60
  const ordered = [...runs.values()].sort((left, right) => right.newest - left.newest);
61
+ const keep = keepOverride ?? family.keep;
53
62
  ordered.forEach((run, index) => {
54
63
  const tooOld = now - run.newest > maxAgeMs;
55
64
  const tooMany = index >= keep;
56
65
  const tooRecentToTouch = now - run.newest < RECENT_GUARD_MS;
66
+ // Delete when EITHER bound is exceeded. The newest `keep` runs are what an
67
+ // operator reads, and past that a run beyond the age bound is gone regardless
68
+ // of its rank: keeping an eleventh run merely because it is only hours old
69
+ // would let a burst of starts grow without bound, which is the growth this
70
+ // bound exists to stop. A burst of eleven young runs still keeps ten.
57
71
  if ((tooOld || tooMany) && !tooRecentToTouch) {
58
72
  for (const name of run.files) removed.push(name);
59
73
  const errName = run.files[0].replace(/\.out\.log$/, '.err.log');
@@ -6,6 +6,17 @@
6
6
  // verification (real worker, reviewer, cancellation, timeout, etc.).
7
7
 
8
8
  const READINESS_STATUSES = Object.freeze(['PASS', 'FAIL', 'BLOCKED', 'SKIP', 'NOT_APPLICABLE', 'NOT_RUN']);
9
+ // Statuses a trusted evidence source may report. `NOT_APPLICABLE` is deliberately
10
+ // absent: applicability is a fact about this machine's policy, decided when the row
11
+ // is built, and letting evidence declare a required row "not applicable" would be a
12
+ // way to make a check disappear while the matrix still reads healthy. Evidence can
13
+ // neither create nor clear it.
14
+ const EVIDENCE_STATUSES = Object.freeze(['PASS', 'FAIL', 'BLOCKED', 'SKIP', 'NOT_RUN']);
15
+ // The rows this release may decide are not applicable, and only when the reason is
16
+ // policy rather than observation: the two built-in DeepSeek routes exist to name a
17
+ // provider the operator may not be using. Every other row asks a question that is
18
+ // always applicable to a Crew machine.
19
+ const POLICY_NOT_APPLICABLE_ROWS = Object.freeze(['deepseek_flash', 'deepseek_pro']);
9
20
 
10
21
  export const READINESS_REASON_CODES = Object.freeze({
11
22
  LIVE_CHECK_PASSED: 'LIVE_CHECK_PASSED',
@@ -59,7 +70,15 @@ function baseRow(id, category, status, reasonCode, evidenceSource = 'none', extr
59
70
 
60
71
  function normalizeEvidence(row, evidence) {
61
72
  if (!evidence || typeof evidence !== 'object') return row;
62
- if (!READINESS_STATUSES.includes(evidence.status)) return row;
73
+ if (!EVIDENCE_STATUSES.includes(evidence.status)) return row;
74
+ // A row the policy marked not applicable keeps that status: reported evidence says
75
+ // something happened, not that the current worker route changed, and letting it
76
+ // flip the row is how "this route is unused" would turn into "this route passed".
77
+ // What was reported is kept as metadata rather than thrown away.
78
+ if (row.status === 'NOT_APPLICABLE') {
79
+ if (typeof evidence.status !== 'string') return row;
80
+ return { ...row, reported_evidence: { status: evidence.status, ...(typeof evidence.reason_code === 'string' && evidence.reason_code.trim() ? { reason_code: evidence.reason_code.trim() } : {}), ...(typeof evidence.evidence_source === 'string' && evidence.evidence_source.trim() ? { evidence_source: evidence.evidence_source.trim() } : {}) } };
81
+ }
63
82
  const reason = typeof evidence.reason_code === 'string' && evidence.reason_code.trim()
64
83
  ? evidence.reason_code.trim()
65
84
  : READINESS_REASON_CODES.EVIDENCE_REPORTED;
@@ -171,12 +190,13 @@ export function buildReadinessMatrix({
171
190
  // answered), and neither status can move the aggregate — only a FAIL, an
172
191
  // UNPROVEN row or an UNAVAILABLE component does. An unknown selection stays
173
192
  // NOT_RUN: this matrix is conservative, and "we could not tell" must not read as
174
- // "not applicable". Reported evidence still wins, because it is applied below.
193
+ // "not applicable". This policy decision is also final for the row: evidence
194
+ // applied below records what it saw without changing the status back.
175
195
  const selectionKnown = typeof workerSelection?.provider === 'string' && workerSelection.provider.length > 0;
176
196
  const deepseekRouteSelected = workerProviderMode === 'deepseek-official'
177
197
  || workerSelection?.provider === 'deepseek-official';
178
198
  if (workerProviderMode !== null && selectionKnown && !deepseekRouteSelected) {
179
- for (const id of ['deepseek_flash', 'deepseek_pro']) {
199
+ for (const id of POLICY_NOT_APPLICABLE_ROWS) {
180
200
  rows[id] = baseRow(id, 'real-execution', 'NOT_APPLICABLE', READINESS_REASON_CODES.WORKER_PROVIDER_FOLLOWS_DSH, 'runtime-policy');
181
201
  }
182
202
  }
@@ -192,7 +212,10 @@ export function buildReadinessMatrix({
192
212
  ]));
193
213
 
194
214
  return {
195
- schema_version: 1,
215
+ // A row can now legally be NOT_APPLICABLE and the summary carries a matching key,
216
+ // so the vocabulary a strict consumer must accept changed: that is what the
217
+ // version means here, not a change in the rules that produce PASS.
218
+ schema_version: 2,
196
219
  platform,
197
220
  conservative: true,
198
221
  summary: counts,
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.2.7';
39
+ export const RUNTIME_VERSION = '2.2.8';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -108,9 +108,15 @@ function projectModelRole({ role, selected, health, jobs, runtime, now, executio
108
108
  if (healthCurrent && matchingHealth.state === 'callable') {
109
109
  return { state: 'CALLABLE', reason_code: matchingHealth.reason_code ?? 'PROVIDER_CALLABLE', selected, observed_at: matchingHealth.observed_at ?? null, expires_at: expiresAt, source: 'provider_health', last_success: null };
110
110
  }
111
+ // What proves the route ran is a provider response, not a passing task contract:
112
+ // a job that reached `done` with its execution completed has demonstrably called
113
+ // the selected model even when the task itself came back `partial`. Failed and
114
+ // cancelled work never reaches this predicate, and neither does a job that errored
115
+ // before the model answered (its execution_status is `failed`, not `completed`).
116
+ const responded = (job) => job?.task_status === 'success' || job?.execution_status === 'completed';
111
117
  const completed = (Array.isArray(jobs) ? jobs : [])
112
118
  .filter((job) => job?.role === role && job?.provider === selected.provider && job?.model === selected.model
113
- && job?.status === 'done' && job?.task_status === 'success'
119
+ && job?.status === 'done' && responded(job)
114
120
  && sameCompleteRuntimeIdentity(job.execution_context, runtime))
115
121
  .map((job) => ({ job, endedAt: Date.parse(job.endedAt ?? '') }))
116
122
  .filter(({ endedAt }) => Number.isFinite(endedAt))
@@ -32,6 +32,7 @@ if not "%~1"=="" goto :invalid_argument
32
32
  set "LAUNCH_HELPER=%LAUNCH_DIR%start-dsh-crew.ps1"
33
33
  set "LAUNCH_LOG=%TEMP%\dsh-crew-launcher.log"
34
34
  if not exist "%LAUNCH_HELPER%" (
35
+ call :rotate_launcher_log
35
36
  >>"%LAUNCH_LOG%" echo [%date% %time%] ERROR Managed launcher helper is missing: %LAUNCH_HELPER%
36
37
  echo ERROR: DSH Crew launcher helper is missing.
37
38
  echo Repair it with: dsh-crew update
@@ -51,6 +52,21 @@ echo ERROR: Unsupported launcher arguments: %LAUNCH_REQUEST%
51
52
  echo Use --open, --background, or --watch.
52
53
  exit /b 64
53
54
 
55
+ :rotate_launcher_log
56
+ rem The PowerShell launcher bounds this log at 5 MiB before its first line, but the
57
+ rem emergency path that calls this is what runs when that helper is MISSING — so a
58
+ rem bounded log cannot depend on the helper whose absence is one of the failures the
59
+ rem log exists to diagnose. Same cap, same two generations, best-effort: a locked or
60
+ rem unreadable log is left alone rather than failing the launch.
61
+ set "LAUNCH_LOG_SIZE="
62
+ for %%A in ("%LAUNCH_LOG%") do set "LAUNCH_LOG_SIZE=%%~zA"
63
+ if not defined LAUNCH_LOG_SIZE goto :eof
64
+ if %LAUNCH_LOG_SIZE% LSS 5242880 goto :eof
65
+ if exist "%LAUNCH_LOG%.2" del /f /q "%LAUNCH_LOG%.2" >nul 2>&1
66
+ if exist "%LAUNCH_LOG%.1" move /y "%LAUNCH_LOG%.1" "%LAUNCH_LOG%.2" >nul 2>&1
67
+ move /y "%LAUNCH_LOG%" "%LAUNCH_LOG%.1" >nul 2>&1
68
+ goto :eof
69
+
54
70
  :help
55
71
  echo Usage: %~nx0 [--open ^| --background ^| --watch]
56
72
  echo --open Open official Harness on 3080 and return once Crew is supervised;