@ran-sh/dsh-crew 2.1.3 → 2.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/package.json +1 -1
- package/scripts/verify-history-ui.mjs +3 -2
- package/src/history/archive-store.mjs +5 -2
- package/src/history/operation.mjs +2 -2
- package/src/information-flow.mjs +25 -0
- package/src/install/crew-paths.mjs +29 -1
- package/src/job-contracts.mjs +7 -2
- package/src/mcp-runtime.mjs +9 -0
- package/src/runtime-identity.mjs +1 -1
- package/src/workflow-runtime.mjs +14 -1
- package/src/workflow.mjs +21 -2
package/package.json
CHANGED
|
@@ -10,6 +10,7 @@ import { Readable } from 'node:stream';
|
|
|
10
10
|
import { createHistoryService } from '../src/history/service.mjs';
|
|
11
11
|
import { registerHistoryHttp } from '../src/history/http.mjs';
|
|
12
12
|
import { runHistoryOperation } from '../src/history/operation.mjs';
|
|
13
|
+
import { CREW_SESSIONS_REL } from '../src/install/crew-paths.mjs';
|
|
13
14
|
|
|
14
15
|
const require = createRequire(import.meta.url);
|
|
15
16
|
const { chromium } = require(process.env.CREW_PLAYWRIGHT_MODULE || 'playwright');
|
|
@@ -17,8 +18,8 @@ const repo = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
|
17
18
|
const root = mkdtempSync(join(tmpdir(), 'crew-history-e2e-'));
|
|
18
19
|
const output = mkdtempSync(join(tmpdir(), 'crew-history-ui-shots-'));
|
|
19
20
|
mkdirSync(join(root, 'harness/storages'), { recursive: true });
|
|
20
|
-
mkdirSync(join(root, '
|
|
21
|
-
const file = join(root, '
|
|
21
|
+
mkdirSync(join(root, CREW_SESSIONS_REL, 'example/session-test'), { recursive: true });
|
|
22
|
+
const file = join(root, CREW_SESSIONS_REL, 'example/session-test/session.jsonl');
|
|
22
23
|
writeFileSync(file, 'DISPOSABLE TEST CONVERSATION');
|
|
23
24
|
writeFileSync(join(root, 'harness/storages/workspace.json'), JSON.stringify({ unit: { name: 'workspace', version: 2 }, global: { initialized: true, workspaceIds: ['test-workspace'], archivedSessionIds: [] }, tables: { workspaces: { 'test-workspace': { path: '/example/project', title: 'Disposable workspace', createdAt: '2026-01-01T00:00:00Z', updatedAt: '2026-01-01T00:00:00Z', sessionIds: ['session-test'] } } } }));
|
|
24
25
|
const errors = []; let stopped = false; let operationPromise;
|
|
@@ -2,6 +2,8 @@ import { createHash, randomUUID } from 'node:crypto';
|
|
|
2
2
|
import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
4
4
|
|
|
5
|
+
import { CREW_SESSIONS_REL } from '../install/crew-paths.mjs';
|
|
6
|
+
|
|
5
7
|
const WORKSPACE = 'harness/storages/workspace.json';
|
|
6
8
|
const LIMIT = 512 * 1024 * 1024;
|
|
7
9
|
const hash = bytes => createHash('sha256').update(bytes).digest('hex');
|
|
@@ -31,7 +33,8 @@ function pathInside(root, relativePath) {
|
|
|
31
33
|
// `session.vN.jsonl` (N >= 1) after that, each optionally zstd-compressed.
|
|
32
34
|
// Sharing the shape between the inventory and the archive guard keeps both in
|
|
33
35
|
// step when the backend's naming changes.
|
|
34
|
-
|
|
36
|
+
const SESSION_ARTIFACT_DIR = CREW_SESSIONS_REL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
37
|
+
export const SESSION_ARTIFACT_PATTERN = new RegExp(`^${SESSION_ARTIFACT_DIR}/[^/]+/[^/]+/session(?:\\.v[1-9][0-9]*)?\\.jsonl(?:\\.zstd)?$`);
|
|
35
38
|
|
|
36
39
|
export function isSessionArtifactPath(relativePath, sessionId) {
|
|
37
40
|
return typeof relativePath === 'string'
|
|
@@ -130,7 +133,7 @@ async function requireStopped(assertStopped) {
|
|
|
130
133
|
* drop its workspace while leaving the artifact behind.
|
|
131
134
|
*/
|
|
132
135
|
function presentSessionIds(root) {
|
|
133
|
-
const sessionsRoot = pathInside(root,
|
|
136
|
+
const sessionsRoot = pathInside(root, CREW_SESSIONS_REL);
|
|
134
137
|
let projects;
|
|
135
138
|
try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
|
|
136
139
|
if (projects.length > 20000) fail('INVALID_FILE');
|
|
@@ -57,11 +57,11 @@ export async function runHistoryOperation({ crewRoot, id, acquire, release, supe
|
|
|
57
57
|
if (!alreadyStarted) {
|
|
58
58
|
if (!recover || await assertStopped(state) !== true) await checkFence(state);
|
|
59
59
|
save('STOPPING');
|
|
60
|
-
// `
|
|
60
|
+
// `refreshFrontend`: this operation rewrites `storages/workspace.json`,
|
|
61
61
|
// which the Crew-managed frontend on 3080 shares, so the launcher stops
|
|
62
62
|
// that server for the same window and starts it again afterwards. The npx
|
|
63
63
|
// lifecycle stops only 3210, because a tree swap leaves that file alone.
|
|
64
|
-
const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId,
|
|
64
|
+
const stopped = await supervisor.stopOwnedBackend({ lease: state.lease, runtimeId: state.runtimeId, refreshFrontend: true });
|
|
65
65
|
if (!stopped?.ok || await assertStopped(state) !== true) throw Error('HISTORY_STOP_NOT_VERIFIED');
|
|
66
66
|
const archiveId = state.operation === 'restore' ? state.archiveId : state.id;
|
|
67
67
|
const manifestFile = historyPath(crewRoot, `history/transactions/${archiveId}/manifest.json`);
|
package/src/information-flow.mjs
CHANGED
|
@@ -26,6 +26,30 @@ function tests(values) {
|
|
|
26
26
|
});
|
|
27
27
|
}
|
|
28
28
|
|
|
29
|
+
/**
|
|
30
|
+
* The pointer to the reviewed attempt's persisted execution record.
|
|
31
|
+
*
|
|
32
|
+
* Only a pointer: the record itself is far too large to embed, and the reviewer
|
|
33
|
+
* is expected to open it in the workspace it already has. It is emitted only
|
|
34
|
+
* when the runtime could resolve both halves, so a transport that cannot name
|
|
35
|
+
* the store degrades to the previous capsule instead of printing a dead path.
|
|
36
|
+
*/
|
|
37
|
+
function executionRecordLines(evidence) {
|
|
38
|
+
const sessionId = evidence?.sessionId;
|
|
39
|
+
const root = evidence?.root;
|
|
40
|
+
if (typeof sessionId !== 'string' || sessionId.trim() === '') return [];
|
|
41
|
+
if (typeof root !== 'string' || root.trim() === '') return [];
|
|
42
|
+
return [
|
|
43
|
+
'',
|
|
44
|
+
'Persisted execution record:',
|
|
45
|
+
`session: ${sessionId}`,
|
|
46
|
+
`store: ${root}`,
|
|
47
|
+
`The record is the directory named "${sessionId}", one workspace level below that store. Read the session journal inside it — \`session.jsonl\` or a versioned \`session.v*.jsonl\`, possibly \`.zstd\`-compressed — and take your evidence from its \`tool/result\` entries: those hold the exact bytes a \`write\` produced (including a file that was later deleted) and the recorded output and exit code of every command that ran.`,
|
|
48
|
+
'The journal is written as concatenated zstd frames, so a single-frame decompressor returns only the header; split it on the zstd magic 28 B5 2F FD and decompress each slice.',
|
|
49
|
+
'This is the original execution, not a re-enactment. When the work was transient — created, run and removed before this review — this record is the only account of it, and a reproduction you run yourself cannot stand in for it: say so plainly and mark anything the record does not settle as NOT RUN rather than substituting an equivalent check.',
|
|
50
|
+
];
|
|
51
|
+
}
|
|
52
|
+
|
|
29
53
|
/** Build the automatic-review context capsule. */
|
|
30
54
|
export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {}) {
|
|
31
55
|
const outcome = view?.outcome ?? {};
|
|
@@ -60,6 +84,7 @@ export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {
|
|
|
60
84
|
...list(changedFiles, { count: 80, itemLimit: 500 }),
|
|
61
85
|
'',
|
|
62
86
|
'Inspect the candidate directly in the current isolated workspace. Use git diff against the base revision and open only the files needed for review. The worker\'s raw prose and full patch are intentionally not embedded in this hand-off.',
|
|
87
|
+
...executionRecordLines(view?.evidence),
|
|
63
88
|
'',
|
|
64
89
|
'Report: 1) whether the implementation satisfies the objective, 2) concrete bugs/style/security risks, 3) suggested fixes. End with ## Review Findings / ## Evidence / ## Risks / ## Verdict (approved | needs changes | rejected).',
|
|
65
90
|
];
|
|
@@ -10,12 +10,26 @@
|
|
|
10
10
|
// installer was right.
|
|
11
11
|
|
|
12
12
|
import { existsSync, realpathSync } from 'node:fs';
|
|
13
|
-
import { join } from 'node:path';
|
|
13
|
+
import { dirname, join } from 'node:path';
|
|
14
14
|
import { homedir } from 'node:os';
|
|
15
15
|
|
|
16
16
|
export const CREW_PROFILE_NAME = 'dsh-crew';
|
|
17
17
|
export const CREW_HOME_REL = join('.config', 'dsh-crew', 'harness');
|
|
18
18
|
|
|
19
|
+
/**
|
|
20
|
+
* The Harness session store, as a POSIX path relative to the Crew root (the
|
|
21
|
+
* parent of the Harness home).
|
|
22
|
+
*
|
|
23
|
+
* Two very different readers need this fact in different shapes: `archive-store`
|
|
24
|
+
* walks it relative to the Crew root to prove a session artifact is really gone
|
|
25
|
+
* before history is deleted, and it also compiles the artifact path into a
|
|
26
|
+
* regular expression, while `crewHarnessSessionsDir` resolves it to an absolute
|
|
27
|
+
* path for the reviewer's evidence pointer. It is one literal here so those can
|
|
28
|
+
* not drift apart — a divergence would either mis-target a deletion or hand a
|
|
29
|
+
* reviewer a path that does not exist.
|
|
30
|
+
*/
|
|
31
|
+
export const CREW_SESSIONS_REL = 'harness/sessions';
|
|
32
|
+
|
|
19
33
|
export function crewDshHome({ home = homedir() } = {}) {
|
|
20
34
|
return join(home, CREW_HOME_REL);
|
|
21
35
|
}
|
|
@@ -24,6 +38,20 @@ export function crewProfileDir({ home = homedir() } = {}) {
|
|
|
24
38
|
return join(crewDshHome({ home }), 'profiles', CREW_PROFILE_NAME);
|
|
25
39
|
}
|
|
26
40
|
|
|
41
|
+
/**
|
|
42
|
+
* Where the Harness persists its session records (`harness/sessions`).
|
|
43
|
+
*
|
|
44
|
+
* This is the only durable record of what an attempt actually executed: the
|
|
45
|
+
* `tool/result` entries carry the exact file contents a `write` produced and the
|
|
46
|
+
* captured stdout, stderr and exit code of every command. A reviewer that has to
|
|
47
|
+
* judge a transient change — work that was created, run and then removed before
|
|
48
|
+
* the review started — has nothing left in the workspace to inspect, so this is
|
|
49
|
+
* the evidence it must read instead of trusting the worker's summary of it.
|
|
50
|
+
*/
|
|
51
|
+
export function crewHarnessSessionsDir({ home = homedir() } = {}) {
|
|
52
|
+
return join(dirname(crewDshHome({ home })), CREW_SESSIONS_REL);
|
|
53
|
+
}
|
|
54
|
+
|
|
27
55
|
// The profile's `node_modules/<name>` entry for the installed package. The
|
|
28
56
|
// registration points this at the live release, so it is stable across
|
|
29
57
|
// upgrades while the release path underneath it is not.
|
package/src/job-contracts.mjs
CHANGED
|
@@ -82,8 +82,13 @@ function evidenceStatus(view) {
|
|
|
82
82
|
if (view?.status === 'failed' || view?.phase === 'failed' || view?.outcome?.execution_status === 'failed') return 'FAIL';
|
|
83
83
|
if (view?.review?.verdict === 'request_changes' || view?.outcome?.task_status === 'partial') return 'PARTIAL';
|
|
84
84
|
if (view?.outcome?.task_status === 'blocked') return 'BLOCKED';
|
|
85
|
-
|
|
86
|
-
|
|
85
|
+
// A reviewer is judged by its verdict, whether or not an outcome was built for
|
|
86
|
+
// it. Keying this branch on `outcome == null` meant the hub — which always
|
|
87
|
+
// builds one — took the generic path below, where a complete review contract
|
|
88
|
+
// was enough for PASS even when the verdict was `inconclusive`. The review
|
|
89
|
+
// verdict is the reviewer's whole product, so it is what has to be approving.
|
|
90
|
+
if (view?.role === 'reviewer' && view.review) {
|
|
91
|
+
return view.status === 'done' && view.review.status === 'done'
|
|
87
92
|
&& view.review.verdict === 'approve' && view.review.delivery_complete === true
|
|
88
93
|
&& view.review.mutated_candidate !== true ? 'PASS' : 'PARTIAL';
|
|
89
94
|
}
|
package/src/mcp-runtime.mjs
CHANGED
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
} from './workspace-isolation.mjs';
|
|
19
19
|
import { startJob, waitJob, jobView, cancelJob } from './jobs.mjs';
|
|
20
20
|
import { hub } from './hub-client.mjs';
|
|
21
|
+
import { crewHarnessSessionsDir } from './install/crew-paths.mjs';
|
|
21
22
|
|
|
22
23
|
const SESSION_CONFIG_KEYS = [
|
|
23
24
|
'default_tier', 'default_effort', 'mode', 'default_timeout_seconds',
|
|
@@ -155,6 +156,10 @@ export function attemptFromView(view, spec) {
|
|
|
155
156
|
provider: view?.provider ?? null,
|
|
156
157
|
model: view?.model ?? null,
|
|
157
158
|
execution_context: view?.execution_context ?? null,
|
|
159
|
+
// The Hub's own session id for this attempt. It is the key to the attempt's
|
|
160
|
+
// persisted execution record, which is what lets a later reviewer read the
|
|
161
|
+
// raw tool results instead of trusting the worker's summary of them.
|
|
162
|
+
session_id: view?.sessionId ?? null,
|
|
158
163
|
selection_source: source,
|
|
159
164
|
selection_trace: selectionTrace,
|
|
160
165
|
status: view?.status ?? 'failed',
|
|
@@ -339,6 +344,10 @@ export function buildMcpWorkflowRuntime(deps) {
|
|
|
339
344
|
captureCandidate,
|
|
340
345
|
releaseWorkspace,
|
|
341
346
|
buildReviewTask,
|
|
347
|
+
// Where the Harness writes session records. The automatic review needs it
|
|
348
|
+
// to point the reviewer at the worker's raw execution evidence; without it
|
|
349
|
+
// the reviewer can only re-read the worker's own summary of what ran.
|
|
350
|
+
evidenceRoot: () => crewHarnessSessionsDir(),
|
|
342
351
|
getConfig,
|
|
343
352
|
getRuntimeControls,
|
|
344
353
|
},
|
package/src/runtime-identity.mjs
CHANGED
|
@@ -36,7 +36,7 @@ export {
|
|
|
36
36
|
// included in the identity contract.
|
|
37
37
|
const RUNTIME_ID = randomUUID();
|
|
38
38
|
|
|
39
|
-
export const RUNTIME_VERSION = '2.1.
|
|
39
|
+
export const RUNTIME_VERSION = '2.1.5';
|
|
40
40
|
export const HUB_PROTOCOL_VERSION = 1;
|
|
41
41
|
|
|
42
42
|
export const HUB_CAPABILITIES = Object.freeze([
|
package/src/workflow-runtime.mjs
CHANGED
|
@@ -508,7 +508,19 @@ export function createWorkflowRuntime(adapters, {
|
|
|
508
508
|
if (decision.step === 'review') {
|
|
509
509
|
transition(job, JOB_PHASES.REVIEWING, 'automatic review');
|
|
510
510
|
const before = job.candidate;
|
|
511
|
-
|
|
511
|
+
// The reviewed attempt's own persisted execution record. The capsule
|
|
512
|
+
// already carries the worker's *summary* of what it ran; this is the
|
|
513
|
+
// handle to what it actually ran, which is the only thing that lets the
|
|
514
|
+
// reviewer check a transient workspace that no longer holds the work.
|
|
515
|
+
const reviewed = [...job.attempts].reverse().find((a) => a.role !== 'reviewer');
|
|
516
|
+
const reviewTask = adapters.buildReviewTask(job.original_task, {
|
|
517
|
+
outcome,
|
|
518
|
+
candidate: job.candidate ?? null,
|
|
519
|
+
evidence: {
|
|
520
|
+
sessionId: reviewed?.session_id ?? null,
|
|
521
|
+
root: adapters.evidenceRoot?.() ?? null,
|
|
522
|
+
},
|
|
523
|
+
}, { strictness: job.review_strictness ?? 'standard' });
|
|
512
524
|
const review = await runReviewerAttempt(job, reviewTask, config, before, job.execution_cwd, job.base_revision);
|
|
513
525
|
if (job.cancelling) { await cancelWorkflow(job); return; }
|
|
514
526
|
job.review = review;
|
|
@@ -633,6 +645,7 @@ export function createWorkflowRuntime(adapters, {
|
|
|
633
645
|
provider: ar.provider ?? null,
|
|
634
646
|
model: ar.model ?? null,
|
|
635
647
|
execution_context: ar.execution_context ?? null,
|
|
648
|
+
session_id: ar.session_id ?? null,
|
|
636
649
|
selection_source: ar.selection_source ?? null,
|
|
637
650
|
selection_trace: ar.selection_trace ?? null,
|
|
638
651
|
status: ar.status ?? 'failed',
|
package/src/workflow.mjs
CHANGED
|
@@ -172,8 +172,27 @@ export function buildOutcome({ result = '', deliveryMeta, executionStatus, stopR
|
|
|
172
172
|
const parsed = parseDeliveryReport(result);
|
|
173
173
|
// The aggregate status and the visible entries come from one parse, so they can
|
|
174
174
|
// no longer disagree about whether the Tests section is evidence.
|
|
175
|
-
|
|
176
|
-
|
|
175
|
+
//
|
|
176
|
+
// `parseDeliveryReport` already decided which contract this message answered —
|
|
177
|
+
// it reads the format off the headings that are present, and only a coding
|
|
178
|
+
// report has a Tests obligation. A review report has no Tests rule, so a
|
|
179
|
+
// reviewer that also lists what it checked under `## Tests` is describing its
|
|
180
|
+
// own coverage; reading those rows with the worker's rule made an honest
|
|
181
|
+
// `NOT RUN — <check> — <reason>`, which is exactly the answer the contract asks
|
|
182
|
+
// for when something cannot be verified, downgrade the reviewer to `partial`
|
|
183
|
+
// and surface `TESTS_NOT_RUN`. The role was penalised for the disclosure the
|
|
184
|
+
// contract requires, so the coding Tests rule now applies only to coding
|
|
185
|
+
// reports. The format decision stays in one place: this consumes it.
|
|
186
|
+
const reviewReport = parsed.format === 'review';
|
|
187
|
+
const parsedTests = reviewReport
|
|
188
|
+
? { valid: false, status: undefined, tests: [] }
|
|
189
|
+
: parseTestsSection(parsed.sections.Tests);
|
|
190
|
+
// `deliveryMeta` is the same report read a second way, so for a review it can
|
|
191
|
+
// only reintroduce the verdict the gate above just dropped. It is a fallback
|
|
192
|
+
// for coding reports, not a way around the format gate.
|
|
193
|
+
const testsStatus = reviewReport
|
|
194
|
+
? undefined
|
|
195
|
+
: (parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status);
|
|
177
196
|
const tests = parsedTests.tests;
|
|
178
197
|
const execStatus = executionStatus ?? (stopReason === 'completed' ? 'completed' : 'failed');
|
|
179
198
|
return {
|