@ran-sh/dsh-crew 2.1.4 → 2.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-crew",
3
- "version": "2.1.4",
3
+ "version": "2.1.6",
4
4
  "description": "Dispatch subtasks to DeepSeek Harness (DSH) agents as native subagents with live progress",
5
5
  "author": {
6
6
  "name": "ZSeven-W"
@@ -22,6 +22,16 @@ changes/tests/risks, changed-file names, base revision, and candidate
22
22
  fingerprint. It opens the relevant files and runs `git diff` in the isolated
23
23
  workspace when deeper inspection is needed.
24
24
 
25
+ The capsule also carries a pointer to the reviewed attempt's persisted execution
26
+ record — the Hub session id plus the Crew harness session store — because a
27
+ transient change leaves nothing in the workspace to inspect once it has been
28
+ created, run and removed. That record is the only account of what actually ran:
29
+ its `tool/result` entries hold the exact bytes a `write` produced, and the
30
+ captured output and exit code of every command. What travels is the pointer,
31
+ never the record, and it is omitted entirely when either half is unknown rather
32
+ than naming a path that would not hold the attempt. A reviewer's own
33
+ reproduction is not a substitute for reading it.
34
+
25
35
  The Hub keeps only the latest assistant message needed as the final Delivery
26
36
  Report. It does not retain an ever-growing list of intermediate assistant
27
37
  messages.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ran-sh/dsh-crew",
3
- "version": "2.1.4",
3
+ "version": "2.1.6",
4
4
  "type": "module",
5
5
  "main": "./src/hub/entry.mjs",
6
6
  "bin": {
@@ -10,6 +10,7 @@ import { Readable } from 'node:stream';
10
10
  import { createHistoryService } from '../src/history/service.mjs';
11
11
  import { registerHistoryHttp } from '../src/history/http.mjs';
12
12
  import { runHistoryOperation } from '../src/history/operation.mjs';
13
+ import { CREW_SESSIONS_REL } from '../src/install/crew-paths.mjs';
13
14
 
14
15
  const require = createRequire(import.meta.url);
15
16
  const { chromium } = require(process.env.CREW_PLAYWRIGHT_MODULE || 'playwright');
@@ -17,8 +18,8 @@ const repo = dirname(dirname(fileURLToPath(import.meta.url)));
17
18
  const root = mkdtempSync(join(tmpdir(), 'crew-history-e2e-'));
18
19
  const output = mkdtempSync(join(tmpdir(), 'crew-history-ui-shots-'));
19
20
  mkdirSync(join(root, 'harness/storages'), { recursive: true });
20
- mkdirSync(join(root, 'harness/sessions/example/session-test'), { recursive: true });
21
- const file = join(root, 'harness/sessions/example/session-test/session.jsonl');
21
+ mkdirSync(join(root, CREW_SESSIONS_REL, 'example/session-test'), { recursive: true });
22
+ const file = join(root, CREW_SESSIONS_REL, 'example/session-test/session.jsonl');
22
23
  writeFileSync(file, 'DISPOSABLE TEST CONVERSATION');
23
24
  writeFileSync(join(root, 'harness/storages/workspace.json'), JSON.stringify({ unit: { name: 'workspace', version: 2 }, global: { initialized: true, workspaceIds: ['test-workspace'], archivedSessionIds: [] }, tables: { workspaces: { 'test-workspace': { path: '/example/project', title: 'Disposable workspace', createdAt: '2026-01-01T00:00:00Z', updatedAt: '2026-01-01T00:00:00Z', sessionIds: ['session-test'] } } } }));
24
25
  const errors = []; let stopped = false; let operationPromise;
@@ -0,0 +1,162 @@
1
+ // Platform-validation evidence for the readiness matrix, loaded from the
2
+ // project's own CI runs.
3
+ //
4
+ // `readiness-matrix.mjs` is deliberately inert — it never reads files, GitHub or
5
+ // the network — and `docs/readiness-matrix.md` says loading and authenticating an
6
+ // evidence source is the responsibility of the higher layer that calls the
7
+ // builder. This is that layer, and it is the only thing here that touches the
8
+ // network.
9
+ //
10
+ // It fails closed in every direction. A version with no tag, a tag whose commit
11
+ // has no run, a run that did not succeed, a missing platform job, a timeout, an
12
+ // HTTP error — all of them return no evidence at all, which leaves the CI rows
13
+ // NOT_RUN. Promoting a row on anything less than a green run at the exact commit
14
+ // being validated is the one failure this module exists to avoid.
15
+ //
16
+ // Authentication is optional and never required: the repository is public, so
17
+ // the anonymous API answers. A token is used only when the caller already has
18
+ // one; nothing here reads credentials from disk, and the value is never logged,
19
+ // returned, or included in an evidence record.
20
+
21
+ export const CI_EVIDENCE_REPO = 'Ran-sh/dsh-crew';
22
+
23
+ // The row each CI job validates. A platform with no job in the workflow cannot
24
+ // be evidenced by any run, so `macos_smoke` is deliberately absent rather than
25
+ // listed and always missing.
26
+ const PLATFORM_JOBS = [
27
+ { row: 'linux_deterministic', job: 'deterministic', runner: 'ubuntu' },
28
+ { row: 'windows_regressions', job: 'windows-paths', runner: 'windows' },
29
+ { row: 'macos_smoke', job: 'macos-smoke', runner: 'macos' },
30
+ ];
31
+
32
+ const CACHE_TTL_MS = 10 * 60 * 1000;
33
+ const cache = new Map();
34
+
35
+ /** Test seam: drop memoized evidence between cases. */
36
+ export function clearCiEvidenceCache() {
37
+ cache.clear();
38
+ }
39
+
40
+ function apiHeaders(token) {
41
+ const headers = {
42
+ accept: 'application/vnd.github+json',
43
+ 'user-agent': 'dsh-crew-readiness',
44
+ };
45
+ if (typeof token === 'string' && token.trim() !== '') headers.authorization = `Bearer ${token.trim()}`;
46
+ return headers;
47
+ }
48
+
49
+ // One deadline covers the whole chain, not one per request: this runs on the
50
+ // readiness route, and three sequential 5s timeouts would be a 15s stall on a
51
+ // cold cache.
52
+ //
53
+ // The deadline is enforced twice on purpose. The abort signal is the polite
54
+ // half — it lets a well-behaved transport drop the socket — but the readiness
55
+ // route must not depend on the transport cooperating, so the request is also
56
+ // raced against the clock. A fetch that ignores its signal then returns nothing
57
+ // at the deadline instead of holding the route open.
58
+ async function getJson(fetchImpl, url, { token, deadline }) {
59
+ const remaining = deadline - Date.now();
60
+ if (remaining <= 0) return null;
61
+ const controller = new AbortController();
62
+ let timer;
63
+ try {
64
+ const attempt = fetchImpl(url, { headers: apiHeaders(token), signal: controller.signal })
65
+ .then((response) => (response?.status === 200 ? response.json() : null))
66
+ .catch(() => null);
67
+ const expiry = new Promise((resolve) => {
68
+ timer = setTimeout(() => { controller.abort(); resolve(null); }, remaining);
69
+ });
70
+ return await Promise.race([attempt, expiry]);
71
+ } catch {
72
+ return null;
73
+ } finally {
74
+ clearTimeout(timer);
75
+ }
76
+ }
77
+
78
+ /**
79
+ * The commit a released version was cut from.
80
+ *
81
+ * `v<version>` may be a lightweight tag (the ref names the commit) or annotated
82
+ * (the ref names a tag object that names the commit), so both are tried. A
83
+ * version with no tag resolves to null and no evidence is produced: the running
84
+ * code is then not the code CI validated, and saying so is the honest answer.
85
+ */
86
+ async function resolveVersionCommit(fetchImpl, { repo, version, token, deadline }) {
87
+ const tag = `v${version}`;
88
+ const ref = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/git/ref/tags/${encodeURIComponent(tag)}`, { token, deadline });
89
+ const object = ref?.object;
90
+ if (!object) return null;
91
+ if (object.type === 'commit' && typeof object.sha === 'string') return object.sha;
92
+ if (object.type === 'tag' && typeof object.sha === 'string') {
93
+ const annotated = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/git/tags/${object.sha}`, { token, deadline });
94
+ const sha = annotated?.object?.sha;
95
+ return typeof sha === 'string' ? sha : null;
96
+ }
97
+ return null;
98
+ }
99
+
100
+ function evidenceFor(jobs, { runId, sha }) {
101
+ const evidence = {};
102
+ for (const { row, job, runner } of PLATFORM_JOBS) {
103
+ const match = jobs.find((entry) => entry?.name === job);
104
+ if (!match) continue;
105
+ if (String(match.conclusion ?? '').toLowerCase() !== 'success') continue;
106
+ const labels = Array.isArray(match.labels) ? match.labels.join(',') : '';
107
+ evidence[row] = {
108
+ status: 'PASS',
109
+ reason_code: 'CI_GREEN',
110
+ evidence_source: 'github-actions',
111
+ // The commit is part of the reference on purpose: version -> tag -> commit
112
+ // is a mapping, and the reader has to be able to audit which commit the
113
+ // green run actually covered.
114
+ evidence_ref: `run-${runId}/${job}/${runner}@${String(sha).slice(0, 12)}${labels ? ` (${labels})` : ''}`,
115
+ };
116
+ }
117
+ return evidence;
118
+ }
119
+
120
+ /**
121
+ * Load CI evidence for a released version. Returns `{}` when nothing can be
122
+ * proven, so the caller can merge the result unconditionally.
123
+ */
124
+ export async function loadCiEvidence({
125
+ repo = CI_EVIDENCE_REPO,
126
+ version,
127
+ token,
128
+ fetchImpl = globalThis.fetch,
129
+ timeoutMs = 5000,
130
+ now = Date.now,
131
+ useCache = true,
132
+ } = {}) {
133
+ if (typeof version !== 'string' || version.trim() === '') return {};
134
+ if (typeof fetchImpl !== 'function') return {};
135
+
136
+ const key = `${repo}@${version.trim()}`;
137
+ const hit = cache.get(key);
138
+ if (useCache && hit && now() - hit.at < CACHE_TTL_MS) return hit.evidence;
139
+
140
+ const deadline = now() + timeoutMs;
141
+ const evidence = await loadUncached({ repo, version: version.trim(), token, fetchImpl, deadline });
142
+ cache.set(key, { at: now(), evidence });
143
+ return evidence;
144
+ }
145
+
146
+ async function loadUncached({ repo, version, token, fetchImpl, deadline }) {
147
+ const sha = await resolveVersionCommit(fetchImpl, { repo, version, token, deadline });
148
+ if (!sha) return {};
149
+
150
+ const runs = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/actions/runs?head_sha=${sha}&per_page=20`, { token, deadline });
151
+ const list = Array.isArray(runs?.workflow_runs) ? runs.workflow_runs : [];
152
+ // The newest CI run for this commit, and each row judged by its own job. The
153
+ // run's overall conclusion is deliberately not a gate: one platform's red job
154
+ // must not withdraw another platform's evidence, and each row asks only
155
+ // whether *its* validation ran and passed.
156
+ const ciRun = list.find((run) => run?.name === 'CI');
157
+ if (!ciRun) return {};
158
+
159
+ const jobsBody = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/actions/runs/${ciRun.id}/jobs?per_page=50`, { token, deadline });
160
+ const jobs = Array.isArray(jobsBody?.jobs) ? jobsBody.jobs : [];
161
+ return evidenceFor(jobs, { runId: ciRun.id, sha });
162
+ }
@@ -191,6 +191,7 @@ export function buildConfigReadinessMatrix({
191
191
  hubJobsChecked = false,
192
192
  hubJobsBody = null,
193
193
  currentSelections = null,
194
+ ciEvidence = null,
194
195
  } = {}) {
195
196
  const warnings = warningCodes(providerCatalogBody);
196
197
  const catalogResponseOk = !!providerCatalogBody
@@ -198,7 +199,13 @@ export function buildConfigReadinessMatrix({
198
199
  && providerCatalogBody.ok !== false;
199
200
  const catalogOk = providerCatalogChecked && catalogResponseOk && warnings.length === 0;
200
201
 
201
- const evidence = {};
202
+ // Platform validation a higher layer actually loaded. `buildReadinessMatrix`
203
+ // drops any record whose status is outside the vocabulary, so this is merged
204
+ // as-is: an empty or malformed map leaves those rows NOT_RUN rather than
205
+ // widening what a row can mean.
206
+ const evidence = ciEvidence && typeof ciEvidence === 'object' && !Array.isArray(ciEvidence)
207
+ ? { ...ciEvidence }
208
+ : {};
202
209
  const hubJobs = hubCompatibility?.compatible === true
203
210
  && hubJobsChecked
204
211
  && hubJobsBody?.ok !== false
@@ -2,6 +2,8 @@ import { createHash, randomUUID } from 'node:crypto';
2
2
  import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
3
3
  import { dirname, isAbsolute, join, resolve } from 'node:path';
4
4
 
5
+ import { CREW_SESSIONS_REL } from '../install/crew-paths.mjs';
6
+
5
7
  const WORKSPACE = 'harness/storages/workspace.json';
6
8
  const LIMIT = 512 * 1024 * 1024;
7
9
  const hash = bytes => createHash('sha256').update(bytes).digest('hex');
@@ -31,7 +33,8 @@ function pathInside(root, relativePath) {
31
33
  // `session.vN.jsonl` (N >= 1) after that, each optionally zstd-compressed.
32
34
  // Sharing the shape between the inventory and the archive guard keeps both in
33
35
  // step when the backend's naming changes.
34
- export const SESSION_ARTIFACT_PATTERN = /^harness\/sessions\/[^/]+\/[^/]+\/session(?:\.v[1-9][0-9]*)?\.jsonl(?:\.zstd)?$/;
36
+ const SESSION_ARTIFACT_DIR = CREW_SESSIONS_REL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
37
+ export const SESSION_ARTIFACT_PATTERN = new RegExp(`^${SESSION_ARTIFACT_DIR}/[^/]+/[^/]+/session(?:\\.v[1-9][0-9]*)?\\.jsonl(?:\\.zstd)?$`);
35
38
 
36
39
  export function isSessionArtifactPath(relativePath, sessionId) {
37
40
  return typeof relativePath === 'string'
@@ -130,7 +133,7 @@ async function requireStopped(assertStopped) {
130
133
  * drop its workspace while leaving the artifact behind.
131
134
  */
132
135
  function presentSessionIds(root) {
133
- const sessionsRoot = pathInside(root, 'harness/sessions');
136
+ const sessionsRoot = pathInside(root, CREW_SESSIONS_REL);
134
137
  let projects;
135
138
  try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
136
139
  if (projects.length > 20000) fail('INVALID_FILE');
package/src/hub/index.mjs CHANGED
@@ -26,6 +26,7 @@ import { normalizeReviewVerdict } from '../workflow-runtime.mjs';
26
26
  import { boundedMachineCodeFromError } from '../structured-error-code.mjs';
27
27
  import { createCanonicalJobEvent, projectWorkflowView } from '../job-contracts.mjs';
28
28
  import { getHubRuntimeIdentity } from '../runtime-identity.mjs';
29
+ import { loadCiEvidence } from '../ci-evidence.mjs';
29
30
  import { loadRoleProfiles, resolveRoleProfile, saveRoleProfiles } from '../role-profiles.mjs';
30
31
  import { addContextReferences, buildWorkspaceTask, isSafeBranchName, loadWorkspaceContexts, resolveWorkspaceContext, saveWorkspaceContexts } from '../workspace-context.mjs';
31
32
  import { buildExtensionContract } from '../extension-contract.mjs';
@@ -1619,6 +1620,10 @@ export async function apply(ctx) {
1619
1620
  currentSelections,
1620
1621
  hubJobsChecked: true,
1621
1622
  hubJobsBody: { ok: true, jobs: boundedJobs },
1623
+ // Platform validation this release actually passed. Resolved from the
1624
+ // running version's tag commit, so a row can only go green for a
1625
+ // commit CI really validated; anything unresolved stays NOT_RUN.
1626
+ ciEvidence: await loadCiEvidence({ version: runtime.runtime_version }),
1622
1627
  });
1623
1628
  const readinessSnapshot = buildRuntimeReadinessSnapshot({
1624
1629
  runtime,
@@ -26,6 +26,30 @@ function tests(values) {
26
26
  });
27
27
  }
28
28
 
29
+ /**
30
+ * The pointer to the reviewed attempt's persisted execution record.
31
+ *
32
+ * Only a pointer: the record itself is far too large to embed, and the reviewer
33
+ * is expected to open it in the workspace it already has. It is emitted only
34
+ * when the runtime could resolve both halves, so a transport that cannot name
35
+ * the store degrades to the previous capsule instead of printing a dead path.
36
+ */
37
+ function executionRecordLines(evidence) {
38
+ const sessionId = evidence?.sessionId;
39
+ const root = evidence?.root;
40
+ if (typeof sessionId !== 'string' || sessionId.trim() === '') return [];
41
+ if (typeof root !== 'string' || root.trim() === '') return [];
42
+ return [
43
+ '',
44
+ 'Persisted execution record:',
45
+ `session: ${sessionId}`,
46
+ `store: ${root}`,
47
+ `The record is the directory named "${sessionId}", one workspace level below that store. Read the session journal inside it — \`session.jsonl\` or a versioned \`session.v*.jsonl\`, possibly \`.zstd\`-compressed — and take your evidence from its \`tool/result\` entries: those hold the exact bytes a \`write\` produced (including a file that was later deleted) and the recorded output and exit code of every command that ran.`,
48
+ 'The journal is written as concatenated zstd frames, so a single-frame decompressor returns only the header; split it on the zstd magic 28 B5 2F FD and decompress each slice.',
49
+ 'This is the original execution, not a re-enactment. When the work was transient — created, run and removed before this review — this record is the only account of it, and a reproduction you run yourself cannot stand in for it: say so plainly and mark anything the record does not settle as NOT RUN rather than substituting an equivalent check.',
50
+ ];
51
+ }
52
+
29
53
  /** Build the automatic-review context capsule. */
30
54
  export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {}) {
31
55
  const outcome = view?.outcome ?? {};
@@ -60,6 +84,7 @@ export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {
60
84
  ...list(changedFiles, { count: 80, itemLimit: 500 }),
61
85
  '',
62
86
  'Inspect the candidate directly in the current isolated workspace. Use git diff against the base revision and open only the files needed for review. The worker\'s raw prose and full patch are intentionally not embedded in this hand-off.',
87
+ ...executionRecordLines(view?.evidence),
63
88
  '',
64
89
  'Report: 1) whether the implementation satisfies the objective, 2) concrete bugs/style/security risks, 3) suggested fixes. End with ## Review Findings / ## Evidence / ## Risks / ## Verdict (approved | needs changes | rejected).',
65
90
  ];
@@ -10,12 +10,26 @@
10
10
  // installer was right.
11
11
 
12
12
  import { existsSync, realpathSync } from 'node:fs';
13
- import { join } from 'node:path';
13
+ import { dirname, join } from 'node:path';
14
14
  import { homedir } from 'node:os';
15
15
 
16
16
  export const CREW_PROFILE_NAME = 'dsh-crew';
17
17
  export const CREW_HOME_REL = join('.config', 'dsh-crew', 'harness');
18
18
 
19
+ /**
20
+ * The Harness session store, as a POSIX path relative to the Crew root (the
21
+ * parent of the Harness home).
22
+ *
23
+ * Two very different readers need this fact in different shapes: `archive-store`
24
+ * walks it relative to the Crew root to prove a session artifact is really gone
25
+ * before history is deleted, and it also compiles the artifact path into a
26
+ * regular expression, while `crewHarnessSessionsDir` resolves it to an absolute
27
+ * path for the reviewer's evidence pointer. It is one literal here so those can
28
+ * not drift apart — a divergence would either mis-target a deletion or hand a
29
+ * reviewer a path that does not exist.
30
+ */
31
+ export const CREW_SESSIONS_REL = 'harness/sessions';
32
+
19
33
  export function crewDshHome({ home = homedir() } = {}) {
20
34
  return join(home, CREW_HOME_REL);
21
35
  }
@@ -24,6 +38,20 @@ export function crewProfileDir({ home = homedir() } = {}) {
24
38
  return join(crewDshHome({ home }), 'profiles', CREW_PROFILE_NAME);
25
39
  }
26
40
 
41
+ /**
42
+ * Where the Harness persists its session records (`harness/sessions`).
43
+ *
44
+ * This is the only durable record of what an attempt actually executed: the
45
+ * `tool/result` entries carry the exact file contents a `write` produced and the
46
+ * captured stdout, stderr and exit code of every command. A reviewer that has to
47
+ * judge a transient change — work that was created, run and then removed before
48
+ * the review started — has nothing left in the workspace to inspect, so this is
49
+ * the evidence it must read instead of trusting the worker's summary of it.
50
+ */
51
+ export function crewHarnessSessionsDir({ home = homedir() } = {}) {
52
+ return join(dirname(crewDshHome({ home })), CREW_SESSIONS_REL);
53
+ }
54
+
27
55
  // The profile's `node_modules/<name>` entry for the installed package. The
28
56
  // registration points this at the live release, so it is stable across
29
57
  // upgrades while the release path underneath it is not.
@@ -82,8 +82,13 @@ function evidenceStatus(view) {
82
82
  if (view?.status === 'failed' || view?.phase === 'failed' || view?.outcome?.execution_status === 'failed') return 'FAIL';
83
83
  if (view?.review?.verdict === 'request_changes' || view?.outcome?.task_status === 'partial') return 'PARTIAL';
84
84
  if (view?.outcome?.task_status === 'blocked') return 'BLOCKED';
85
- if (view?.role === 'reviewer' && view?.outcome == null) {
86
- return view.status === 'done' && view.review?.status === 'done'
85
+ // A reviewer is judged by its verdict, whether or not an outcome was built for
86
+ // it. Keying this branch on `outcome == null` meant the hub — which always
87
+ // builds one — took the generic path below, where a complete review contract
88
+ // was enough for PASS even when the verdict was `inconclusive`. The review
89
+ // verdict is the reviewer's whole product, so it is what has to be approving.
90
+ if (view?.role === 'reviewer' && view.review) {
91
+ return view.status === 'done' && view.review.status === 'done'
87
92
  && view.review.verdict === 'approve' && view.review.delivery_complete === true
88
93
  && view.review.mutated_candidate !== true ? 'PASS' : 'PARTIAL';
89
94
  }
@@ -18,6 +18,7 @@ import {
18
18
  } from './workspace-isolation.mjs';
19
19
  import { startJob, waitJob, jobView, cancelJob } from './jobs.mjs';
20
20
  import { hub } from './hub-client.mjs';
21
+ import { crewHarnessSessionsDir } from './install/crew-paths.mjs';
21
22
 
22
23
  const SESSION_CONFIG_KEYS = [
23
24
  'default_tier', 'default_effort', 'mode', 'default_timeout_seconds',
@@ -155,6 +156,10 @@ export function attemptFromView(view, spec) {
155
156
  provider: view?.provider ?? null,
156
157
  model: view?.model ?? null,
157
158
  execution_context: view?.execution_context ?? null,
159
+ // The Hub's own session id for this attempt. It is the key to the attempt's
160
+ // persisted execution record, which is what lets a later reviewer read the
161
+ // raw tool results instead of trusting the worker's summary of them.
162
+ session_id: view?.sessionId ?? null,
158
163
  selection_source: source,
159
164
  selection_trace: selectionTrace,
160
165
  status: view?.status ?? 'failed',
@@ -339,6 +344,10 @@ export function buildMcpWorkflowRuntime(deps) {
339
344
  captureCandidate,
340
345
  releaseWorkspace,
341
346
  buildReviewTask,
347
+ // Where the Harness writes session records. The automatic review needs it
348
+ // to point the reviewer at the worker's raw execution evidence; without it
349
+ // the reviewer can only re-read the worker's own summary of what ran.
350
+ evidenceRoot: () => crewHarnessSessionsDir(),
342
351
  getConfig,
343
352
  getRuntimeControls,
344
353
  },
@@ -36,7 +36,7 @@ export {
36
36
  // included in the identity contract.
37
37
  const RUNTIME_ID = randomUUID();
38
38
 
39
- export const RUNTIME_VERSION = '2.1.4';
39
+ export const RUNTIME_VERSION = '2.1.6';
40
40
  export const HUB_PROTOCOL_VERSION = 1;
41
41
 
42
42
  export const HUB_CAPABILITIES = Object.freeze([
package/src/server.mjs CHANGED
@@ -10,6 +10,7 @@ import { RUNTIME_VERSION, getHubRuntimeIdentity } from './runtime-identity.mjs';
10
10
  import { resolveWorkerModel } from './model-routing.mjs';
11
11
  import { runtimeActivationMetadata } from './runtime-controls.mjs';
12
12
  import { buildConfigReadinessMatrix } from './config-readiness.mjs';
13
+ import { loadCiEvidence } from './ci-evidence.mjs';
13
14
  import { buildRuntimeReadinessSnapshot, reprojectRuntimeModelCallability } from './runtime-readiness-snapshot.mjs';
14
15
  import { classifyFailure, classifyFailureCode } from './failure-classification.mjs';
15
16
  import {
@@ -402,6 +403,9 @@ async function buildConfigReport() {
402
403
  workerProviderMode,
403
404
  providerCatalogChecked,
404
405
  providerCatalogBody,
406
+ // The same platform evidence the hub route merges, so the MCP surface does
407
+ // not report the CI rows differently when it has to build the matrix itself.
408
+ ciEvidence: await loadCiEvidence({ version: RUNTIME_VERSION }),
405
409
  });
406
410
  const readinessMatrix = hubReadinessSnapshot?.readiness_matrix ?? fallbackReadinessMatrix;
407
411
  const roleProfiles = loadRoleProfiles();
@@ -508,7 +508,19 @@ export function createWorkflowRuntime(adapters, {
508
508
  if (decision.step === 'review') {
509
509
  transition(job, JOB_PHASES.REVIEWING, 'automatic review');
510
510
  const before = job.candidate;
511
- const reviewTask = adapters.buildReviewTask(job.original_task, { outcome, candidate: job.candidate ?? null }, { strictness: job.review_strictness ?? 'standard' });
511
+ // The reviewed attempt's own persisted execution record. The capsule
512
+ // already carries the worker's *summary* of what it ran; this is the
513
+ // handle to what it actually ran, which is the only thing that lets the
514
+ // reviewer check a transient workspace that no longer holds the work.
515
+ const reviewed = [...job.attempts].reverse().find((a) => a.role !== 'reviewer');
516
+ const reviewTask = adapters.buildReviewTask(job.original_task, {
517
+ outcome,
518
+ candidate: job.candidate ?? null,
519
+ evidence: {
520
+ sessionId: reviewed?.session_id ?? null,
521
+ root: adapters.evidenceRoot?.() ?? null,
522
+ },
523
+ }, { strictness: job.review_strictness ?? 'standard' });
512
524
  const review = await runReviewerAttempt(job, reviewTask, config, before, job.execution_cwd, job.base_revision);
513
525
  if (job.cancelling) { await cancelWorkflow(job); return; }
514
526
  job.review = review;
@@ -633,6 +645,7 @@ export function createWorkflowRuntime(adapters, {
633
645
  provider: ar.provider ?? null,
634
646
  model: ar.model ?? null,
635
647
  execution_context: ar.execution_context ?? null,
648
+ session_id: ar.session_id ?? null,
636
649
  selection_source: ar.selection_source ?? null,
637
650
  selection_trace: ar.selection_trace ?? null,
638
651
  status: ar.status ?? 'failed',
package/src/workflow.mjs CHANGED
@@ -172,8 +172,27 @@ export function buildOutcome({ result = '', deliveryMeta, executionStatus, stopR
172
172
  const parsed = parseDeliveryReport(result);
173
173
  // The aggregate status and the visible entries come from one parse, so they can
174
174
  // no longer disagree about whether the Tests section is evidence.
175
- const parsedTests = parseTestsSection(parsed.sections.Tests);
176
- const testsStatus = parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status;
175
+ //
176
+ // `parseDeliveryReport` already decided which contract this message answered —
177
+ // it reads the format off the headings that are present, and only a coding
178
+ // report has a Tests obligation. A review report has no Tests rule, so a
179
+ // reviewer that also lists what it checked under `## Tests` is describing its
180
+ // own coverage; reading those rows with the worker's rule made an honest
181
+ // `NOT RUN — <check> — <reason>`, which is exactly the answer the contract asks
182
+ // for when something cannot be verified, downgrade the reviewer to `partial`
183
+ // and surface `TESTS_NOT_RUN`. The role was penalised for the disclosure the
184
+ // contract requires, so the coding Tests rule now applies only to coding
185
+ // reports. The format decision stays in one place: this consumes it.
186
+ const reviewReport = parsed.format === 'review';
187
+ const parsedTests = reviewReport
188
+ ? { valid: false, status: undefined, tests: [] }
189
+ : parseTestsSection(parsed.sections.Tests);
190
+ // `deliveryMeta` is the same report read a second way, so for a review it can
191
+ // only reintroduce the verdict the gate above just dropped. It is a fallback
192
+ // for coding reports, not a way around the format gate.
193
+ const testsStatus = reviewReport
194
+ ? undefined
195
+ : (parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status);
177
196
  const tests = parsedTests.tests;
178
197
  const execStatus = executionStatus ?? (stopReason === 'completed' ? 'completed' : 'failed');
179
198
  return {