@ran-sh/dsh-crew 2.1.4 → 2.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/docs/job-contracts.md +10 -0
- package/package.json +1 -1
- package/scripts/verify-history-ui.mjs +3 -2
- package/src/ci-evidence.mjs +162 -0
- package/src/config-readiness.mjs +8 -1
- package/src/history/archive-store.mjs +5 -2
- package/src/hub/index.mjs +5 -0
- package/src/information-flow.mjs +25 -0
- package/src/install/crew-paths.mjs +29 -1
- package/src/job-contracts.mjs +7 -2
- package/src/mcp-runtime.mjs +9 -0
- package/src/runtime-identity.mjs +1 -1
- package/src/server.mjs +4 -0
- package/src/workflow-runtime.mjs +14 -1
- package/src/workflow.mjs +21 -2
package/docs/job-contracts.md
CHANGED
|
@@ -22,6 +22,16 @@ changes/tests/risks, changed-file names, base revision, and candidate
|
|
|
22
22
|
fingerprint. It opens the relevant files and runs `git diff` in the isolated
|
|
23
23
|
workspace when deeper inspection is needed.
|
|
24
24
|
|
|
25
|
+
The capsule also carries a pointer to the reviewed attempt's persisted execution
|
|
26
|
+
record — the Hub session id plus the Crew harness session store — because a
|
|
27
|
+
transient change leaves nothing in the workspace to inspect once it has been
|
|
28
|
+
created, run and removed. That record is the only account of what actually ran:
|
|
29
|
+
its `tool/result` entries hold the exact bytes a `write` produced, and the
|
|
30
|
+
captured output and exit code of every command. What travels is the pointer,
|
|
31
|
+
never the record, and it is omitted entirely when either half is unknown rather
|
|
32
|
+
than naming a path that would not hold the attempt. A reviewer's own
|
|
33
|
+
reproduction is not a substitute for reading it.
|
|
34
|
+
|
|
25
35
|
The Hub keeps only the latest assistant message needed as the final Delivery
|
|
26
36
|
Report. It does not retain an ever-growing list of intermediate assistant
|
|
27
37
|
messages.
|
package/package.json
CHANGED
|
@@ -10,6 +10,7 @@ import { Readable } from 'node:stream';
|
|
|
10
10
|
import { createHistoryService } from '../src/history/service.mjs';
|
|
11
11
|
import { registerHistoryHttp } from '../src/history/http.mjs';
|
|
12
12
|
import { runHistoryOperation } from '../src/history/operation.mjs';
|
|
13
|
+
import { CREW_SESSIONS_REL } from '../src/install/crew-paths.mjs';
|
|
13
14
|
|
|
14
15
|
const require = createRequire(import.meta.url);
|
|
15
16
|
const { chromium } = require(process.env.CREW_PLAYWRIGHT_MODULE || 'playwright');
|
|
@@ -17,8 +18,8 @@ const repo = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
|
17
18
|
const root = mkdtempSync(join(tmpdir(), 'crew-history-e2e-'));
|
|
18
19
|
const output = mkdtempSync(join(tmpdir(), 'crew-history-ui-shots-'));
|
|
19
20
|
mkdirSync(join(root, 'harness/storages'), { recursive: true });
|
|
20
|
-
mkdirSync(join(root, '
|
|
21
|
-
const file = join(root, '
|
|
21
|
+
mkdirSync(join(root, CREW_SESSIONS_REL, 'example/session-test'), { recursive: true });
|
|
22
|
+
const file = join(root, CREW_SESSIONS_REL, 'example/session-test/session.jsonl');
|
|
22
23
|
writeFileSync(file, 'DISPOSABLE TEST CONVERSATION');
|
|
23
24
|
writeFileSync(join(root, 'harness/storages/workspace.json'), JSON.stringify({ unit: { name: 'workspace', version: 2 }, global: { initialized: true, workspaceIds: ['test-workspace'], archivedSessionIds: [] }, tables: { workspaces: { 'test-workspace': { path: '/example/project', title: 'Disposable workspace', createdAt: '2026-01-01T00:00:00Z', updatedAt: '2026-01-01T00:00:00Z', sessionIds: ['session-test'] } } } }));
|
|
24
25
|
const errors = []; let stopped = false; let operationPromise;
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
// Platform-validation evidence for the readiness matrix, loaded from the
|
|
2
|
+
// project's own CI runs.
|
|
3
|
+
//
|
|
4
|
+
// `readiness-matrix.mjs` is deliberately inert — it never reads files, GitHub or
|
|
5
|
+
// the network — and `docs/readiness-matrix.md` says loading and authenticating an
|
|
6
|
+
// evidence source is the responsibility of the higher layer that calls the
|
|
7
|
+
// builder. This is that layer, and it is the only thing here that touches the
|
|
8
|
+
// network.
|
|
9
|
+
//
|
|
10
|
+
// It fails closed in every direction. A version with no tag, a tag whose commit
|
|
11
|
+
// has no run, a run that did not succeed, a missing platform job, a timeout, an
|
|
12
|
+
// HTTP error — all of them return no evidence at all, which leaves the CI rows
|
|
13
|
+
// NOT_RUN. Promoting a row on anything less than a green run at the exact commit
|
|
14
|
+
// being validated is the one failure this module exists to avoid.
|
|
15
|
+
//
|
|
16
|
+
// Authentication is optional and never required: the repository is public, so
|
|
17
|
+
// the anonymous API answers. A token is used only when the caller already has
|
|
18
|
+
// one; nothing here reads credentials from disk, and the value is never logged,
|
|
19
|
+
// returned, or included in an evidence record.
|
|
20
|
+
|
|
21
|
+
export const CI_EVIDENCE_REPO = 'Ran-sh/dsh-crew';
|
|
22
|
+
|
|
23
|
+
// The row each CI job validates. A platform with no job in the workflow cannot
|
|
24
|
+
// be evidenced by any run, so `macos_smoke` is deliberately absent rather than
|
|
25
|
+
// listed and always missing.
|
|
26
|
+
const PLATFORM_JOBS = [
|
|
27
|
+
{ row: 'linux_deterministic', job: 'deterministic', runner: 'ubuntu' },
|
|
28
|
+
{ row: 'windows_regressions', job: 'windows-paths', runner: 'windows' },
|
|
29
|
+
{ row: 'macos_smoke', job: 'macos-smoke', runner: 'macos' },
|
|
30
|
+
];
|
|
31
|
+
|
|
32
|
+
const CACHE_TTL_MS = 10 * 60 * 1000;
|
|
33
|
+
const cache = new Map();
|
|
34
|
+
|
|
35
|
+
/** Test seam: drop memoized evidence between cases. */
|
|
36
|
+
export function clearCiEvidenceCache() {
|
|
37
|
+
cache.clear();
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function apiHeaders(token) {
|
|
41
|
+
const headers = {
|
|
42
|
+
accept: 'application/vnd.github+json',
|
|
43
|
+
'user-agent': 'dsh-crew-readiness',
|
|
44
|
+
};
|
|
45
|
+
if (typeof token === 'string' && token.trim() !== '') headers.authorization = `Bearer ${token.trim()}`;
|
|
46
|
+
return headers;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// One deadline covers the whole chain, not one per request: this runs on the
|
|
50
|
+
// readiness route, and three sequential 5s timeouts would be a 15s stall on a
|
|
51
|
+
// cold cache.
|
|
52
|
+
//
|
|
53
|
+
// The deadline is enforced twice on purpose. The abort signal is the polite
|
|
54
|
+
// half — it lets a well-behaved transport drop the socket — but the readiness
|
|
55
|
+
// route must not depend on the transport cooperating, so the request is also
|
|
56
|
+
// raced against the clock. A fetch that ignores its signal then returns nothing
|
|
57
|
+
// at the deadline instead of holding the route open.
|
|
58
|
+
async function getJson(fetchImpl, url, { token, deadline }) {
|
|
59
|
+
const remaining = deadline - Date.now();
|
|
60
|
+
if (remaining <= 0) return null;
|
|
61
|
+
const controller = new AbortController();
|
|
62
|
+
let timer;
|
|
63
|
+
try {
|
|
64
|
+
const attempt = fetchImpl(url, { headers: apiHeaders(token), signal: controller.signal })
|
|
65
|
+
.then((response) => (response?.status === 200 ? response.json() : null))
|
|
66
|
+
.catch(() => null);
|
|
67
|
+
const expiry = new Promise((resolve) => {
|
|
68
|
+
timer = setTimeout(() => { controller.abort(); resolve(null); }, remaining);
|
|
69
|
+
});
|
|
70
|
+
return await Promise.race([attempt, expiry]);
|
|
71
|
+
} catch {
|
|
72
|
+
return null;
|
|
73
|
+
} finally {
|
|
74
|
+
clearTimeout(timer);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* The commit a released version was cut from.
|
|
80
|
+
*
|
|
81
|
+
* `v<version>` may be a lightweight tag (the ref names the commit) or annotated
|
|
82
|
+
* (the ref names a tag object that names the commit), so both are tried. A
|
|
83
|
+
* version with no tag resolves to null and no evidence is produced: the running
|
|
84
|
+
* code is then not the code CI validated, and saying so is the honest answer.
|
|
85
|
+
*/
|
|
86
|
+
async function resolveVersionCommit(fetchImpl, { repo, version, token, deadline }) {
|
|
87
|
+
const tag = `v${version}`;
|
|
88
|
+
const ref = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/git/ref/tags/${encodeURIComponent(tag)}`, { token, deadline });
|
|
89
|
+
const object = ref?.object;
|
|
90
|
+
if (!object) return null;
|
|
91
|
+
if (object.type === 'commit' && typeof object.sha === 'string') return object.sha;
|
|
92
|
+
if (object.type === 'tag' && typeof object.sha === 'string') {
|
|
93
|
+
const annotated = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/git/tags/${object.sha}`, { token, deadline });
|
|
94
|
+
const sha = annotated?.object?.sha;
|
|
95
|
+
return typeof sha === 'string' ? sha : null;
|
|
96
|
+
}
|
|
97
|
+
return null;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
function evidenceFor(jobs, { runId, sha }) {
|
|
101
|
+
const evidence = {};
|
|
102
|
+
for (const { row, job, runner } of PLATFORM_JOBS) {
|
|
103
|
+
const match = jobs.find((entry) => entry?.name === job);
|
|
104
|
+
if (!match) continue;
|
|
105
|
+
if (String(match.conclusion ?? '').toLowerCase() !== 'success') continue;
|
|
106
|
+
const labels = Array.isArray(match.labels) ? match.labels.join(',') : '';
|
|
107
|
+
evidence[row] = {
|
|
108
|
+
status: 'PASS',
|
|
109
|
+
reason_code: 'CI_GREEN',
|
|
110
|
+
evidence_source: 'github-actions',
|
|
111
|
+
// The commit is part of the reference on purpose: version -> tag -> commit
|
|
112
|
+
// is a mapping, and the reader has to be able to audit which commit the
|
|
113
|
+
// green run actually covered.
|
|
114
|
+
evidence_ref: `run-${runId}/${job}/${runner}@${String(sha).slice(0, 12)}${labels ? ` (${labels})` : ''}`,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
return evidence;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Load CI evidence for a released version. Returns `{}` when nothing can be
|
|
122
|
+
* proven, so the caller can merge the result unconditionally.
|
|
123
|
+
*/
|
|
124
|
+
export async function loadCiEvidence({
|
|
125
|
+
repo = CI_EVIDENCE_REPO,
|
|
126
|
+
version,
|
|
127
|
+
token,
|
|
128
|
+
fetchImpl = globalThis.fetch,
|
|
129
|
+
timeoutMs = 5000,
|
|
130
|
+
now = Date.now,
|
|
131
|
+
useCache = true,
|
|
132
|
+
} = {}) {
|
|
133
|
+
if (typeof version !== 'string' || version.trim() === '') return {};
|
|
134
|
+
if (typeof fetchImpl !== 'function') return {};
|
|
135
|
+
|
|
136
|
+
const key = `${repo}@${version.trim()}`;
|
|
137
|
+
const hit = cache.get(key);
|
|
138
|
+
if (useCache && hit && now() - hit.at < CACHE_TTL_MS) return hit.evidence;
|
|
139
|
+
|
|
140
|
+
const deadline = now() + timeoutMs;
|
|
141
|
+
const evidence = await loadUncached({ repo, version: version.trim(), token, fetchImpl, deadline });
|
|
142
|
+
cache.set(key, { at: now(), evidence });
|
|
143
|
+
return evidence;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
async function loadUncached({ repo, version, token, fetchImpl, deadline }) {
|
|
147
|
+
const sha = await resolveVersionCommit(fetchImpl, { repo, version, token, deadline });
|
|
148
|
+
if (!sha) return {};
|
|
149
|
+
|
|
150
|
+
const runs = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/actions/runs?head_sha=${sha}&per_page=20`, { token, deadline });
|
|
151
|
+
const list = Array.isArray(runs?.workflow_runs) ? runs.workflow_runs : [];
|
|
152
|
+
// The newest CI run for this commit, and each row judged by its own job. The
|
|
153
|
+
// run's overall conclusion is deliberately not a gate: one platform's red job
|
|
154
|
+
// must not withdraw another platform's evidence, and each row asks only
|
|
155
|
+
// whether *its* validation ran and passed.
|
|
156
|
+
const ciRun = list.find((run) => run?.name === 'CI');
|
|
157
|
+
if (!ciRun) return {};
|
|
158
|
+
|
|
159
|
+
const jobsBody = await getJson(fetchImpl, `https://api.github.com/repos/${repo}/actions/runs/${ciRun.id}/jobs?per_page=50`, { token, deadline });
|
|
160
|
+
const jobs = Array.isArray(jobsBody?.jobs) ? jobsBody.jobs : [];
|
|
161
|
+
return evidenceFor(jobs, { runId: ciRun.id, sha });
|
|
162
|
+
}
|
package/src/config-readiness.mjs
CHANGED
|
@@ -191,6 +191,7 @@ export function buildConfigReadinessMatrix({
|
|
|
191
191
|
hubJobsChecked = false,
|
|
192
192
|
hubJobsBody = null,
|
|
193
193
|
currentSelections = null,
|
|
194
|
+
ciEvidence = null,
|
|
194
195
|
} = {}) {
|
|
195
196
|
const warnings = warningCodes(providerCatalogBody);
|
|
196
197
|
const catalogResponseOk = !!providerCatalogBody
|
|
@@ -198,7 +199,13 @@ export function buildConfigReadinessMatrix({
|
|
|
198
199
|
&& providerCatalogBody.ok !== false;
|
|
199
200
|
const catalogOk = providerCatalogChecked && catalogResponseOk && warnings.length === 0;
|
|
200
201
|
|
|
201
|
-
|
|
202
|
+
// Platform validation a higher layer actually loaded. `buildReadinessMatrix`
|
|
203
|
+
// drops any record whose status is outside the vocabulary, so this is merged
|
|
204
|
+
// as-is: an empty or malformed map leaves those rows NOT_RUN rather than
|
|
205
|
+
// widening what a row can mean.
|
|
206
|
+
const evidence = ciEvidence && typeof ciEvidence === 'object' && !Array.isArray(ciEvidence)
|
|
207
|
+
? { ...ciEvidence }
|
|
208
|
+
: {};
|
|
202
209
|
const hubJobs = hubCompatibility?.compatible === true
|
|
203
210
|
&& hubJobsChecked
|
|
204
211
|
&& hubJobsBody?.ok !== false
|
|
@@ -2,6 +2,8 @@ import { createHash, randomUUID } from 'node:crypto';
|
|
|
2
2
|
import { closeSync, existsSync, fsyncSync, linkSync, lstatSync, mkdirSync, openSync, readdirSync, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from 'node:fs';
|
|
3
3
|
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
4
4
|
|
|
5
|
+
import { CREW_SESSIONS_REL } from '../install/crew-paths.mjs';
|
|
6
|
+
|
|
5
7
|
const WORKSPACE = 'harness/storages/workspace.json';
|
|
6
8
|
const LIMIT = 512 * 1024 * 1024;
|
|
7
9
|
const hash = bytes => createHash('sha256').update(bytes).digest('hex');
|
|
@@ -31,7 +33,8 @@ function pathInside(root, relativePath) {
|
|
|
31
33
|
// `session.vN.jsonl` (N >= 1) after that, each optionally zstd-compressed.
|
|
32
34
|
// Sharing the shape between the inventory and the archive guard keeps both in
|
|
33
35
|
// step when the backend's naming changes.
|
|
34
|
-
|
|
36
|
+
const SESSION_ARTIFACT_DIR = CREW_SESSIONS_REL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
37
|
+
export const SESSION_ARTIFACT_PATTERN = new RegExp(`^${SESSION_ARTIFACT_DIR}/[^/]+/[^/]+/session(?:\\.v[1-9][0-9]*)?\\.jsonl(?:\\.zstd)?$`);
|
|
35
38
|
|
|
36
39
|
export function isSessionArtifactPath(relativePath, sessionId) {
|
|
37
40
|
return typeof relativePath === 'string'
|
|
@@ -130,7 +133,7 @@ async function requireStopped(assertStopped) {
|
|
|
130
133
|
* drop its workspace while leaving the artifact behind.
|
|
131
134
|
*/
|
|
132
135
|
function presentSessionIds(root) {
|
|
133
|
-
const sessionsRoot = pathInside(root,
|
|
136
|
+
const sessionsRoot = pathInside(root, CREW_SESSIONS_REL);
|
|
134
137
|
let projects;
|
|
135
138
|
try { projects = readdirSync(sessionsRoot, { withFileTypes: true }); } catch { return new Set(); }
|
|
136
139
|
if (projects.length > 20000) fail('INVALID_FILE');
|
package/src/hub/index.mjs
CHANGED
|
@@ -26,6 +26,7 @@ import { normalizeReviewVerdict } from '../workflow-runtime.mjs';
|
|
|
26
26
|
import { boundedMachineCodeFromError } from '../structured-error-code.mjs';
|
|
27
27
|
import { createCanonicalJobEvent, projectWorkflowView } from '../job-contracts.mjs';
|
|
28
28
|
import { getHubRuntimeIdentity } from '../runtime-identity.mjs';
|
|
29
|
+
import { loadCiEvidence } from '../ci-evidence.mjs';
|
|
29
30
|
import { loadRoleProfiles, resolveRoleProfile, saveRoleProfiles } from '../role-profiles.mjs';
|
|
30
31
|
import { addContextReferences, buildWorkspaceTask, isSafeBranchName, loadWorkspaceContexts, resolveWorkspaceContext, saveWorkspaceContexts } from '../workspace-context.mjs';
|
|
31
32
|
import { buildExtensionContract } from '../extension-contract.mjs';
|
|
@@ -1619,6 +1620,10 @@ export async function apply(ctx) {
|
|
|
1619
1620
|
currentSelections,
|
|
1620
1621
|
hubJobsChecked: true,
|
|
1621
1622
|
hubJobsBody: { ok: true, jobs: boundedJobs },
|
|
1623
|
+
// Platform validation this release actually passed. Resolved from the
|
|
1624
|
+
// running version's tag commit, so a row can only go green for a
|
|
1625
|
+
// commit CI really validated; anything unresolved stays NOT_RUN.
|
|
1626
|
+
ciEvidence: await loadCiEvidence({ version: runtime.runtime_version }),
|
|
1622
1627
|
});
|
|
1623
1628
|
const readinessSnapshot = buildRuntimeReadinessSnapshot({
|
|
1624
1629
|
runtime,
|
package/src/information-flow.mjs
CHANGED
|
@@ -26,6 +26,30 @@ function tests(values) {
|
|
|
26
26
|
});
|
|
27
27
|
}
|
|
28
28
|
|
|
29
|
+
/**
|
|
30
|
+
* The pointer to the reviewed attempt's persisted execution record.
|
|
31
|
+
*
|
|
32
|
+
* Only a pointer: the record itself is far too large to embed, and the reviewer
|
|
33
|
+
* is expected to open it in the workspace it already has. It is emitted only
|
|
34
|
+
* when the runtime could resolve both halves, so a transport that cannot name
|
|
35
|
+
* the store degrades to the previous capsule instead of printing a dead path.
|
|
36
|
+
*/
|
|
37
|
+
function executionRecordLines(evidence) {
|
|
38
|
+
const sessionId = evidence?.sessionId;
|
|
39
|
+
const root = evidence?.root;
|
|
40
|
+
if (typeof sessionId !== 'string' || sessionId.trim() === '') return [];
|
|
41
|
+
if (typeof root !== 'string' || root.trim() === '') return [];
|
|
42
|
+
return [
|
|
43
|
+
'',
|
|
44
|
+
'Persisted execution record:',
|
|
45
|
+
`session: ${sessionId}`,
|
|
46
|
+
`store: ${root}`,
|
|
47
|
+
`The record is the directory named "${sessionId}", one workspace level below that store. Read the session journal inside it — \`session.jsonl\` or a versioned \`session.v*.jsonl\`, possibly \`.zstd\`-compressed — and take your evidence from its \`tool/result\` entries: those hold the exact bytes a \`write\` produced (including a file that was later deleted) and the recorded output and exit code of every command that ran.`,
|
|
48
|
+
'The journal is written as concatenated zstd frames, so a single-frame decompressor returns only the header; split it on the zstd magic 28 B5 2F FD and decompress each slice.',
|
|
49
|
+
'This is the original execution, not a re-enactment. When the work was transient — created, run and removed before this review — this record is the only account of it, and a reproduction you run yourself cannot stand in for it: say so plainly and mark anything the record does not settle as NOT RUN rather than substituting an equivalent check.',
|
|
50
|
+
];
|
|
51
|
+
}
|
|
52
|
+
|
|
29
53
|
/** Build the automatic-review context capsule. */
|
|
30
54
|
export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {}) {
|
|
31
55
|
const outcome = view?.outcome ?? {};
|
|
@@ -60,6 +84,7 @@ export function buildReviewTask(task, view = {}, { strictness = 'standard' } = {
|
|
|
60
84
|
...list(changedFiles, { count: 80, itemLimit: 500 }),
|
|
61
85
|
'',
|
|
62
86
|
'Inspect the candidate directly in the current isolated workspace. Use git diff against the base revision and open only the files needed for review. The worker\'s raw prose and full patch are intentionally not embedded in this hand-off.',
|
|
87
|
+
...executionRecordLines(view?.evidence),
|
|
63
88
|
'',
|
|
64
89
|
'Report: 1) whether the implementation satisfies the objective, 2) concrete bugs/style/security risks, 3) suggested fixes. End with ## Review Findings / ## Evidence / ## Risks / ## Verdict (approved | needs changes | rejected).',
|
|
65
90
|
];
|
|
@@ -10,12 +10,26 @@
|
|
|
10
10
|
// installer was right.
|
|
11
11
|
|
|
12
12
|
import { existsSync, realpathSync } from 'node:fs';
|
|
13
|
-
import { join } from 'node:path';
|
|
13
|
+
import { dirname, join } from 'node:path';
|
|
14
14
|
import { homedir } from 'node:os';
|
|
15
15
|
|
|
16
16
|
export const CREW_PROFILE_NAME = 'dsh-crew';
|
|
17
17
|
export const CREW_HOME_REL = join('.config', 'dsh-crew', 'harness');
|
|
18
18
|
|
|
19
|
+
/**
|
|
20
|
+
* The Harness session store, as a POSIX path relative to the Crew root (the
|
|
21
|
+
* parent of the Harness home).
|
|
22
|
+
*
|
|
23
|
+
* Two very different readers need this fact in different shapes: `archive-store`
|
|
24
|
+
* walks it relative to the Crew root to prove a session artifact is really gone
|
|
25
|
+
* before history is deleted, and it also compiles the artifact path into a
|
|
26
|
+
* regular expression, while `crewHarnessSessionsDir` resolves it to an absolute
|
|
27
|
+
* path for the reviewer's evidence pointer. It is one literal here so those can
|
|
28
|
+
* not drift apart — a divergence would either mis-target a deletion or hand a
|
|
29
|
+
* reviewer a path that does not exist.
|
|
30
|
+
*/
|
|
31
|
+
export const CREW_SESSIONS_REL = 'harness/sessions';
|
|
32
|
+
|
|
19
33
|
export function crewDshHome({ home = homedir() } = {}) {
|
|
20
34
|
return join(home, CREW_HOME_REL);
|
|
21
35
|
}
|
|
@@ -24,6 +38,20 @@ export function crewProfileDir({ home = homedir() } = {}) {
|
|
|
24
38
|
return join(crewDshHome({ home }), 'profiles', CREW_PROFILE_NAME);
|
|
25
39
|
}
|
|
26
40
|
|
|
41
|
+
/**
|
|
42
|
+
* Where the Harness persists its session records (`harness/sessions`).
|
|
43
|
+
*
|
|
44
|
+
* This is the only durable record of what an attempt actually executed: the
|
|
45
|
+
* `tool/result` entries carry the exact file contents a `write` produced and the
|
|
46
|
+
* captured stdout, stderr and exit code of every command. A reviewer that has to
|
|
47
|
+
* judge a transient change — work that was created, run and then removed before
|
|
48
|
+
* the review started — has nothing left in the workspace to inspect, so this is
|
|
49
|
+
* the evidence it must read instead of trusting the worker's summary of it.
|
|
50
|
+
*/
|
|
51
|
+
export function crewHarnessSessionsDir({ home = homedir() } = {}) {
|
|
52
|
+
return join(dirname(crewDshHome({ home })), CREW_SESSIONS_REL);
|
|
53
|
+
}
|
|
54
|
+
|
|
27
55
|
// The profile's `node_modules/<name>` entry for the installed package. The
|
|
28
56
|
// registration points this at the live release, so it is stable across
|
|
29
57
|
// upgrades while the release path underneath it is not.
|
package/src/job-contracts.mjs
CHANGED
|
@@ -82,8 +82,13 @@ function evidenceStatus(view) {
|
|
|
82
82
|
if (view?.status === 'failed' || view?.phase === 'failed' || view?.outcome?.execution_status === 'failed') return 'FAIL';
|
|
83
83
|
if (view?.review?.verdict === 'request_changes' || view?.outcome?.task_status === 'partial') return 'PARTIAL';
|
|
84
84
|
if (view?.outcome?.task_status === 'blocked') return 'BLOCKED';
|
|
85
|
-
|
|
86
|
-
|
|
85
|
+
// A reviewer is judged by its verdict, whether or not an outcome was built for
|
|
86
|
+
// it. Keying this branch on `outcome == null` meant the hub — which always
|
|
87
|
+
// builds one — took the generic path below, where a complete review contract
|
|
88
|
+
// was enough for PASS even when the verdict was `inconclusive`. The review
|
|
89
|
+
// verdict is the reviewer's whole product, so it is what has to be approving.
|
|
90
|
+
if (view?.role === 'reviewer' && view.review) {
|
|
91
|
+
return view.status === 'done' && view.review.status === 'done'
|
|
87
92
|
&& view.review.verdict === 'approve' && view.review.delivery_complete === true
|
|
88
93
|
&& view.review.mutated_candidate !== true ? 'PASS' : 'PARTIAL';
|
|
89
94
|
}
|
package/src/mcp-runtime.mjs
CHANGED
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
} from './workspace-isolation.mjs';
|
|
19
19
|
import { startJob, waitJob, jobView, cancelJob } from './jobs.mjs';
|
|
20
20
|
import { hub } from './hub-client.mjs';
|
|
21
|
+
import { crewHarnessSessionsDir } from './install/crew-paths.mjs';
|
|
21
22
|
|
|
22
23
|
const SESSION_CONFIG_KEYS = [
|
|
23
24
|
'default_tier', 'default_effort', 'mode', 'default_timeout_seconds',
|
|
@@ -155,6 +156,10 @@ export function attemptFromView(view, spec) {
|
|
|
155
156
|
provider: view?.provider ?? null,
|
|
156
157
|
model: view?.model ?? null,
|
|
157
158
|
execution_context: view?.execution_context ?? null,
|
|
159
|
+
// The Hub's own session id for this attempt. It is the key to the attempt's
|
|
160
|
+
// persisted execution record, which is what lets a later reviewer read the
|
|
161
|
+
// raw tool results instead of trusting the worker's summary of them.
|
|
162
|
+
session_id: view?.sessionId ?? null,
|
|
158
163
|
selection_source: source,
|
|
159
164
|
selection_trace: selectionTrace,
|
|
160
165
|
status: view?.status ?? 'failed',
|
|
@@ -339,6 +344,10 @@ export function buildMcpWorkflowRuntime(deps) {
|
|
|
339
344
|
captureCandidate,
|
|
340
345
|
releaseWorkspace,
|
|
341
346
|
buildReviewTask,
|
|
347
|
+
// Where the Harness writes session records. The automatic review needs it
|
|
348
|
+
// to point the reviewer at the worker's raw execution evidence; without it
|
|
349
|
+
// the reviewer can only re-read the worker's own summary of what ran.
|
|
350
|
+
evidenceRoot: () => crewHarnessSessionsDir(),
|
|
342
351
|
getConfig,
|
|
343
352
|
getRuntimeControls,
|
|
344
353
|
},
|
package/src/runtime-identity.mjs
CHANGED
|
@@ -36,7 +36,7 @@ export {
|
|
|
36
36
|
// included in the identity contract.
|
|
37
37
|
const RUNTIME_ID = randomUUID();
|
|
38
38
|
|
|
39
|
-
export const RUNTIME_VERSION = '2.1.
|
|
39
|
+
export const RUNTIME_VERSION = '2.1.6';
|
|
40
40
|
export const HUB_PROTOCOL_VERSION = 1;
|
|
41
41
|
|
|
42
42
|
export const HUB_CAPABILITIES = Object.freeze([
|
package/src/server.mjs
CHANGED
|
@@ -10,6 +10,7 @@ import { RUNTIME_VERSION, getHubRuntimeIdentity } from './runtime-identity.mjs';
|
|
|
10
10
|
import { resolveWorkerModel } from './model-routing.mjs';
|
|
11
11
|
import { runtimeActivationMetadata } from './runtime-controls.mjs';
|
|
12
12
|
import { buildConfigReadinessMatrix } from './config-readiness.mjs';
|
|
13
|
+
import { loadCiEvidence } from './ci-evidence.mjs';
|
|
13
14
|
import { buildRuntimeReadinessSnapshot, reprojectRuntimeModelCallability } from './runtime-readiness-snapshot.mjs';
|
|
14
15
|
import { classifyFailure, classifyFailureCode } from './failure-classification.mjs';
|
|
15
16
|
import {
|
|
@@ -402,6 +403,9 @@ async function buildConfigReport() {
|
|
|
402
403
|
workerProviderMode,
|
|
403
404
|
providerCatalogChecked,
|
|
404
405
|
providerCatalogBody,
|
|
406
|
+
// The same platform evidence the hub route merges, so the MCP surface does
|
|
407
|
+
// not report the CI rows differently when it has to build the matrix itself.
|
|
408
|
+
ciEvidence: await loadCiEvidence({ version: RUNTIME_VERSION }),
|
|
405
409
|
});
|
|
406
410
|
const readinessMatrix = hubReadinessSnapshot?.readiness_matrix ?? fallbackReadinessMatrix;
|
|
407
411
|
const roleProfiles = loadRoleProfiles();
|
package/src/workflow-runtime.mjs
CHANGED
|
@@ -508,7 +508,19 @@ export function createWorkflowRuntime(adapters, {
|
|
|
508
508
|
if (decision.step === 'review') {
|
|
509
509
|
transition(job, JOB_PHASES.REVIEWING, 'automatic review');
|
|
510
510
|
const before = job.candidate;
|
|
511
|
-
|
|
511
|
+
// The reviewed attempt's own persisted execution record. The capsule
|
|
512
|
+
// already carries the worker's *summary* of what it ran; this is the
|
|
513
|
+
// handle to what it actually ran, which is the only thing that lets the
|
|
514
|
+
// reviewer check a transient workspace that no longer holds the work.
|
|
515
|
+
const reviewed = [...job.attempts].reverse().find((a) => a.role !== 'reviewer');
|
|
516
|
+
const reviewTask = adapters.buildReviewTask(job.original_task, {
|
|
517
|
+
outcome,
|
|
518
|
+
candidate: job.candidate ?? null,
|
|
519
|
+
evidence: {
|
|
520
|
+
sessionId: reviewed?.session_id ?? null,
|
|
521
|
+
root: adapters.evidenceRoot?.() ?? null,
|
|
522
|
+
},
|
|
523
|
+
}, { strictness: job.review_strictness ?? 'standard' });
|
|
512
524
|
const review = await runReviewerAttempt(job, reviewTask, config, before, job.execution_cwd, job.base_revision);
|
|
513
525
|
if (job.cancelling) { await cancelWorkflow(job); return; }
|
|
514
526
|
job.review = review;
|
|
@@ -633,6 +645,7 @@ export function createWorkflowRuntime(adapters, {
|
|
|
633
645
|
provider: ar.provider ?? null,
|
|
634
646
|
model: ar.model ?? null,
|
|
635
647
|
execution_context: ar.execution_context ?? null,
|
|
648
|
+
session_id: ar.session_id ?? null,
|
|
636
649
|
selection_source: ar.selection_source ?? null,
|
|
637
650
|
selection_trace: ar.selection_trace ?? null,
|
|
638
651
|
status: ar.status ?? 'failed',
|
package/src/workflow.mjs
CHANGED
|
@@ -172,8 +172,27 @@ export function buildOutcome({ result = '', deliveryMeta, executionStatus, stopR
|
|
|
172
172
|
const parsed = parseDeliveryReport(result);
|
|
173
173
|
// The aggregate status and the visible entries come from one parse, so they can
|
|
174
174
|
// no longer disagree about whether the Tests section is evidence.
|
|
175
|
-
|
|
176
|
-
|
|
175
|
+
//
|
|
176
|
+
// `parseDeliveryReport` already decided which contract this message answered —
|
|
177
|
+
// it reads the format off the headings that are present, and only a coding
|
|
178
|
+
// report has a Tests obligation. A review report has no Tests rule, so a
|
|
179
|
+
// reviewer that also lists what it checked under `## Tests` is describing its
|
|
180
|
+
// own coverage; reading those rows with the worker's rule made an honest
|
|
181
|
+
// `NOT RUN — <check> — <reason>`, which is exactly the answer the contract asks
|
|
182
|
+
// for when something cannot be verified, downgrade the reviewer to `partial`
|
|
183
|
+
// and surface `TESTS_NOT_RUN`. The role was penalised for the disclosure the
|
|
184
|
+
// contract requires, so the coding Tests rule now applies only to coding
|
|
185
|
+
// reports. The format decision stays in one place: this consumes it.
|
|
186
|
+
const reviewReport = parsed.format === 'review';
|
|
187
|
+
const parsedTests = reviewReport
|
|
188
|
+
? { valid: false, status: undefined, tests: [] }
|
|
189
|
+
: parseTestsSection(parsed.sections.Tests);
|
|
190
|
+
// `deliveryMeta` is the same report read a second way, so for a review it can
|
|
191
|
+
// only reintroduce the verdict the gate above just dropped. It is a fallback
|
|
192
|
+
// for coding reports, not a way around the format gate.
|
|
193
|
+
const testsStatus = reviewReport
|
|
194
|
+
? undefined
|
|
195
|
+
: (parsedTests.status ?? parsed.tests_status ?? deliveryMeta?.tests_status);
|
|
177
196
|
const tests = parsedTests.tests;
|
|
178
197
|
const execStatus = executionStatus ?? (stopReason === 'completed' ? 'completed' : 'failed');
|
|
179
198
|
return {
|