@ran-sh/dsh-crew 2.1.3 → 2.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.1.3",
3
+ "version": "2.1.5",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.1.3",
3
+ "version": "2.1.5",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -10,6 +10,7 @@ import { Readable } from 'node:stream';
10
10
  import { createHistoryService } from '../src/history/service.mjs';
11
11
  import { registerHistoryHttp } from '../src/history/http.mjs';
12
12
  import { runHistoryOperation } from '../src/history/operation.mjs';
13
+ import { CREW_SESSIONS_REL } from '../src/install/crew-paths.mjs';
13
14
 
14
15
  const require = createRequire(import.meta.url);
15
16
  const { chromium } = require(process.env.CREW_PLAYWRIGHT_MODULE || 'playwright');
@@ -17,8 +18,8 @@ const repo = dirname(dirname(fileURLToPath(import.meta.url)));
17
18
  const root = mkdtempSync(join(tmpdir(), 'crew-history-e2e-'));
18
19
  const output = mkdtempSync(join(tmpdir(), 'crew-history-ui-shots-'));
19
20
  mkdirSync(join(root, 'harness/storages'), { recursive: true });
20
- mkdirSync(join(root, 'harness/sessions/example/session-test'), { recursive: true });
21
- const file = join(root, 'harness/sessions/example/session-test/session.jsonl');
21
+ mkdirSync(join(root, CREW_SESSIONS_REL, 'example/session-test'), { recursive: true });
22
+ const file = join(root, CREW_SESSIONS_REL, 'example/session-test/session.jsonl');
22
23
  writeFileSync(file, 'DISPOSABLE TEST CONVERSATION');
23
24
  writeFileSync(join(root, 'harness/storages/workspace.json'), JSON.stringify({ unit: { name: 'workspace', version: 2 }, global: { initialized: true, workspaceIds: ['test-workspace'], archivedSessionIds: [] }, tables: { workspaces: { 'test-workspace': { path: '/example/project', title: 'Disposable workspace', createdAt: '2026-01-01T00:00:00Z', updatedAt: '2026-01-01T00:00:00Z', sessionIds: ['session-test'] } } } }));
24
25
  const errors = []; let stopped = false; let operationPromise;
@@ -2,6 +2,8 @@ import { createHash, randomUUID } from 'node:crypto';
2
2
  import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
3
3
  import { dirname, isAbsolute, join, resolve } from 'node:path';
4
4
 
5
+ import { CREW_SESSIONS_REL } from '../install/crew-paths.mjs';
6
+
5
7
  const WORKSPACE = 'harness/storages/workspace.json';
6
8
  const LIMIT = 512 * 1024 * 1024;
7
9
  const hash = bytes => createHash('sha256').update(bytes).digest('hex');
@@ -31,7 +33,8 @@ function pathInside(root, relativePath) {
31
33
  // `session.vN.jsonl` (N >= 1) after that, each optionally zstd-compressed.
32
34
  // Sharing the shape between the inventory and the archive guard keeps both in
33
35
  // step when the backend's naming changes.
34
- export const SESSION_ARTIFACT_PATTERN = /^harness\/sessions\/[^/]+\/[^/]+\/session(?:\.v[1-9][0-9]*)?\.jsonl(?:\.zstd)?$/;
36
+ const SESSION_ARTIFACT_DIR = CREW_SESSIONS_REL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
37
+ export const SESSION_ARTIFACT_PATTERN = new RegExp(`^${SESSION_ARTIFACT_DIR}/[^/]+/[^/]+/session(?:\\.v[1-9][0-9]*)?\\.jsonl(?:\\.zstd)?$`);
35
38
 
36
39
  export function isSessionArtifactPath(relativePath, sessionId) {
37
40
  return typeof relativePath === 'string'
@@ -130,7 +133,7 @@ async function requireStopped(assertStopped) {
130
133
  * drop its workspace while leaving the artifact behind.
131
134
  */
132
135
  function presentSessionIds(root) {
133
- const sessionsRoot = pathInside(root, 'harness/sessions');
136
+ const sessionsRoot = pathInside(root, CREW_SESSIONS_REL);
134
137
  let projects;
135
138
  try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
136
139
  if (projects.length > 20000) fail('INVALID_FILE');
@@ -57,11 +57,11 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
57
57
  if (!alreadyStarted) {
58
58
  if (!recover || await assertStopped(state) !== true) await checkFence(state);
59
59
  save('STOPPING');
60
- // `refresh_frontend`: this operation rewrites `storages/workspace.json`,
60
+ // `refreshFrontend`: this operation rewrites `storages/workspace.json`,
61
61
  // which the Crew-managed frontend on 3080 shares, so the launcher stops
62
62
  // that server for the same window and starts it again afterwards. The npx
63
63
  // lifecycle stops only 3210, because a tree swap leaves that file alone.
64
- const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId, refresh_frontend: true });
64
+ const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId, refreshFrontend: true });
65
65
  if (!stopped?.ok || await assertStopped(state) !== true) throw Error('HISTORY_STOP_NOT_VERIFIED');
66
66
  const archiveId = state.operation === 'restore' ? state.archiveId : state.id;
67
67
  const manifestFile = historyPath(crewRoot, `history/transactions/${archiveId}/manifest.json`);
@@ -26,6 +26,30 @@ function tests(values) {
26
26
  });
27
27
  }
28
28
 
29
+ /**
30
+ * The pointer to the reviewed attempt's persisted execution record.
31
+ *
32
+ * Only a pointer: the record itself is far too large to embed, and the reviewer
33
+ * is expected to open it in the workspace it already has. It is emitted only
34
+ * when the runtime could resolve both halves, so a transport that cannot name
35
+ * the store degrades to the previous capsule instead of printing a dead path.
36
+ */
37
+ function executionRecordLines(evidence) {
38
+ const sessionId = evidence?.sessionId;
39
+ const root = evidence?.root;
40
+ if (typeof sessionId !== 'string' || sessionId.trim() === '') return [];
41
+ if (typeof root !== 'string' || root.trim() === '') return [];
42
+ return [
43
+ '',
44
+ 'Persisted execution record:',
45
+ `session: ${sessionId}`,
46
+ `store: ${root}`,
47
+ `The record is the directory named "${sessionId}", one workspace level below that store. Read the session journal inside it — \`session.jsonl\` or a versioned \`session.v*.jsonl\`, possibly \`.zstd\`-compressed — and take your evidence from its \`tool/result\` entries: those hold the exact bytes a \`write\` produced (including a file that was later deleted) and the recorded output and exit code of every command that ran.`,
48
+ 'The journal is written as concatenated zstd frames, so a single-frame decompressor returns only the header; split it on the zstd magic 28 B5 2F FD and decompress each slice.',
49
+ 'This is the original execution, not a re-enactment. When the work was transient — created, run and removed before this review — this record is the only account of it, and a reproduction you run yourself cannot stand in for it: say so plainly and mark anything the record does not settle as NOT RUN rather than substituting an equivalent check.',
50
+ ];
51
+ }
52
+
29
53
  /** Build the automatic-review context capsule. */
30
54
  export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {}) {
31
55
  const outcome = view?.outcome ?? {};
@@ -60,6 +84,7 @@ export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {
60
84
  ...list(changedFiles, { count: 80, itemLimit: 500 }),
61
85
  '',
62
86
  'Inspect the candidate directly in the current isolated workspace. Use git diff against the base revision and open only the files needed for review. The worker\'s raw prose and full patch are intentionally not embedded in this hand-off.',
87
+ ...executionRecordLines(view?.evidence),
63
88
  '',
64
89
  'Report: 1) whether the implementation satisfies the objective, 2) concrete bugs/style/security risks, 3) suggested fixes. End with ## Review Findings / ## Evidence / ## Risks / ## Verdict (approved | needs changes | rejected).',
65
90
  ];
@@ -10,12 +10,26 @@
10
10
  // installer was right.
11
11
 
12
12
  import { existsSync, realpathSync } from 'node:fs';
13
- import { join } from 'node:path';
13
+ import { dirname, join } from 'node:path';
14
14
  import { homedir } from 'node:os';
15
15
 
16
16
  export const CREW_PROFILE_NAME = 'dsh-crew';
17
17
  export const CREW_HOME_REL = join('.config', 'dsh-crew', 'harness');
18
18
 
19
+ /**
20
+ * The Harness session store, as a POSIX path relative to the Crew root (the
21
+ * parent of the Harness home).
22
+ *
23
+ * Two very different readers need this fact in different shapes: `archive-store`
24
+ * walks it relative to the Crew root to prove a session artifact is really gone
25
+ * before history is deleted, and it also compiles the artifact path into a
26
+ * regular expression, while `crewHarnessSessionsDir` resolves it to an absolute
27
+ * path for the reviewer's evidence pointer. It is one literal here so those can
28
+ * not drift apart — a divergence would either mis-target a deletion or hand a
29
+ * reviewer a path that does not exist.
30
+ */
31
+ export const CREW_SESSIONS_REL = 'harness/sessions';
32
+
19
33
  export function crewDshHome({ home = homedir() } = {}) {
20
34
  return join(home, CREW_HOME_REL);
21
35
  }
@@ -24,6 +38,20 @@ export function crewProfileDir({ home = homedir() } = {}) {
24
38
  return join(crewDshHome({ home }), 'profiles', CREW_PROFILE_NAME);
25
39
  }
26
40
 
41
+ /**
42
+ * Where the Harness persists its session records (`harness/sessions`).
43
+ *
44
+ * This is the only durable record of what an attempt actually executed: the
45
+ * `tool/result` entries carry the exact file contents a `write` produced and the
46
+ * captured stdout, stderr and exit code of every command. A reviewer that has to
47
+ * judge a transient change — work that was created, run and then removed before
48
+ * the review started — has nothing left in the workspace to inspect, so this is
49
+ * the evidence it must read instead of trusting the worker's summary of it.
50
+ */
51
+ export function crewHarnessSessionsDir({ home = homedir() } = {}) {
52
+ return join(dirname(crewDshHome({ home })), CREW_SESSIONS_REL);
53
+ }
54
+
27
55
  // The profile's `node_modules/<name>` entry for the installed package. The
28
56
  // registration points this at the live release, so it is stable across
29
57
  // upgrades while the release path underneath it is not.
@@ -82,8 +82,13 @@ function evidenceStatus(view) {
82
82
  if (view?.status === 'failed' || view?.phase === 'failed' || view?.outcome?.execution_status === 'failed') return 'FAIL';
83
83
  if (view?.review?.verdict === 'request_changes' || view?.outcome?.task_status === 'partial') return 'PARTIAL';
84
84
  if (view?.outcome?.task_status === 'blocked') return 'BLOCKED';
85
- if (view?.role === 'reviewer' && view?.outcome == null) {
86
- return view.status === 'done' && view.review?.status === 'done'
85
+ // A reviewer is judged by its verdict, whether or not an outcome was built for
86
+ // it. Keying this branch on `outcome == null` meant the hub — which always
87
+ // builds one — took the generic path below, where a complete review contract
88
+ // was enough for PASS even when the verdict was `inconclusive`. The review
89
+ // verdict is the reviewer's whole product, so it is what has to be approving.
90
+ if (view?.role === 'reviewer' && view.review) {
91
+ return view.status === 'done' && view.review.status === 'done'
87
92
  && view.review.verdict === 'approve' && view.review.delivery_complete === true
88
93
  && view.review.mutated_candidate !== true ? 'PASS' : 'PARTIAL';
89
94
  }
@@ -18,6 +18,7 @@ import {
18
18
  } from './workspace-isolation.mjs';
19
19
  import { startJob, waitJob, jobView, cancelJob } from './jobs.mjs';
20
20
  import { hub } from './hub-client.mjs';
21
+ import { crewHarnessSessionsDir } from './install/crew-paths.mjs';
21
22
 
22
23
  const SESSION_CONFIG_KEYS = [
23
24
  'default_tier', 'default_effort', 'mode', 'default_timeout_seconds',
@@ -155,6 +156,10 @@ export function attemptFromView(view, spec) {
155
156
  provider: view?.provider ?? null,
156
157
  model: view?.model ?? null,
157
158
  execution_context: view?.execution_context ?? null,
159
+ // The Hub's own session id for this attempt. It is the key to the attempt's
160
+ // persisted execution record, which is what lets a later reviewer read the
161
+ // raw tool results instead of trusting the worker's summary of them.
162
+ session_id: view?.sessionId ?? null,
158
163
  selection_source: source,
159
164
  selection_trace: selectionTrace,
160
165
  status: view?.status ?? 'failed',
@@ -339,6 +344,10 @@ export function buildMcpWorkflowRuntime(deps) {
339
344
  captureCandidate,
340
345
  releaseWorkspace,
341
346
  buildReviewTask,
347
+ // Where the Harness writes session records. The automatic review needs it
348
+ // to point the reviewer at the worker's raw execution evidence; without it
349
+ // the reviewer can only re-read the worker's own summary of what ran.
350
+ evidenceRoot: () => crewHarnessSessionsDir(),
342
351
  getConfig,
343
352
  getRuntimeControls,
344
353
  },
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.1.3';
39
+ export const RUNTIME_VERSION = '2.1.5';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
@@ -508,7 +508,19 @@ export function createWorkflowRuntime(adapters, {
508
508
  if (decision.step === 'review') {
509
509
  transition(job, JOB_PHASES.REVIEWING, 'automatic review');
510
510
  const before = job.candidate;
511
- const reviewTask = adapters.buildReviewTask(job.original_task, { outcome, candidate: job.candidate ?? null }, { strictness: job.review_strictness ?? 'standard' });
511
+ // The reviewed attempt's own persisted execution record. The capsule
512
+ // already carries the worker's *summary* of what it ran; this is the
513
+ // handle to what it actually ran, which is the only thing that lets the
514
+ // reviewer check a transient workspace that no longer holds the work.
515
+ const reviewed = [...job.attempts].reverse().find((a) => a.role !== 'reviewer');
516
+ const reviewTask = adapters.buildReviewTask(job.original_task, {
517
+ outcome,
518
+ candidate: job.candidate ?? null,
519
+ evidence: {
520
+ sessionId: reviewed?.session_id ?? null,
521
+ root: adapters.evidenceRoot?.() ?? null,
522
+ },
523
+ }, { strictness: job.review_strictness ?? 'standard' });
512
524
  const review = await runReviewerAttempt(job, reviewTask, config, before, job.execution_cwd, job.base_revision);
513
525
  if (job.cancelling) { await cancelWorkflow(job); return; }
514
526
  job.review = review;
@@ -633,6 +645,7 @@ export function createWorkflowRuntime(adapters, {
633
645
  provider: ar.provider ?? null,
634
646
  model: ar.model ?? null,
635
647
  execution_context: ar.execution_context ?? null,
648
+ session_id: ar.session_id ?? null,
636
649
  selection_source: ar.selection_source ?? null,
637
650
  selection_trace: ar.selection_trace ?? null,
638
651
  status: ar.status ?? 'failed',
package/src/workflow.mjs CHANGED
@@ -172,8 +172,27 @@ export function buildOutcome({ result = '', deliveryMeta, executionStatus, stopR
172
172
  const parsed = parseDeliveryReport(result);
173
173
  // The aggregate status and the visible entries come from one parse, so they can
174
174
  // no longer disagree about whether the Tests section is evidence.
175
- const parsedTests = parseTestsSection(parsed.sections.Tests);
176
- const testsStatus = parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status;
175
+ //
176
+ // `parseDeliveryReport` already decided which contract this message answered —
177
+ // it reads the format off the headings that are present, and only a coding
178
+ // report has a Tests obligation. A review report has no Tests rule, so a
179
+ // reviewer that also lists what it checked under `## Tests` is describing its
180
+ // own coverage; reading those rows with the worker's rule made an honest
181
+ // `NOT RUN — <check> — <reason>`, which is exactly the answer the contract asks
182
+ // for when something cannot be verified, downgrade the reviewer to `partial`
183
+ // and surface `TESTS_NOT_RUN`. The role was penalised for the disclosure the
184
+ // contract requires, so the coding Tests rule now applies only to coding
185
+ // reports. The format decision stays in one place: this consumes it.
186
+ const reviewReport = parsed.format === 'review';
187
+ const parsedTests = reviewReport
188
+ ? { valid: false, status: undefined, tests: [] }
189
+ : parseTestsSection(parsed.sections.Tests);
190
+ // `deliveryMeta` is the same report read a second way, so for a review it can
191
+ // only reintroduce the verdict the gate above just dropped. It is a fallback
192
+ // for coding reports, not a way around the format gate.
193
+ const testsStatus = reviewReport
194
+ ? undefined
195
+ : (parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status);
177
196
  const tests = parsedTests.tests;
178
197
  const execStatus = executionStatus ?? (stopReason === 'completed' ? 'completed' : 'failed');
179
198
  return {