nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -480
- package/bin/nomarmy.mjs +1081 -185
- package/docker/Dockerfile +2 -2
- package/docker/Dockerfile.go +6 -4
- package/docker/Dockerfile.rust +17 -2
- package/harnesses/_template/README.md +27 -0
- package/harnesses/_template/harness.yml +26 -0
- package/harnesses/browser-playwright/README.md +35 -0
- package/harnesses/browser-playwright/fixture/package.json +1 -0
- package/harnesses/browser-playwright/fixture/page.html +1 -0
- package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
- package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
- package/harnesses/browser-playwright/harness.yml +18 -0
- package/harnesses/go/README.md +45 -0
- package/harnesses/go/harness.yml +14 -0
- package/harnesses/mock-oidc/README.md +31 -0
- package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
- package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
- package/harnesses/mock-oidc/harness.yml +19 -0
- package/harnesses/node/README.md +53 -0
- package/harnesses/node/harness.yml +18 -0
- package/harnesses/python/README.md +46 -0
- package/harnesses/python/harness.yml +16 -0
- package/harnesses/rust/README.md +45 -0
- package/harnesses/rust/harness.yml +13 -0
- package/install.sh +29 -9
- package/lib/admission.mjs +178 -30
- package/lib/agents.mjs +8 -6
- package/lib/army.mjs +25 -10
- package/lib/codex-link.mjs +37 -0
- package/lib/config.mjs +15 -0
- package/lib/connect.mjs +232 -19
- package/lib/continue-from.mjs +103 -0
- package/lib/coordinator-instructions.mjs +5 -1
- package/lib/diff-checks.mjs +114 -0
- package/lib/dispatch-schema.mjs +14 -12
- package/lib/doctor.mjs +98 -9
- package/lib/egress-proxy.mjs +116 -0
- package/lib/execute.mjs +241 -33
- package/lib/git-record.mjs +27 -3
- package/lib/harness-schema.mjs +61 -0
- package/lib/harnesses.mjs +99 -0
- package/lib/health.mjs +162 -18
- package/lib/install-freshness.mjs +114 -0
- package/lib/jev-checks.mjs +110 -0
- package/lib/job-format.mjs +54 -0
- package/lib/judge.mjs +130 -0
- package/lib/limits.mjs +77 -0
- package/lib/model-probe.mjs +61 -0
- package/lib/mutation.mjs +159 -0
- package/lib/notify.mjs +30 -3
- package/lib/openclaw-install.mjs +122 -0
- package/lib/openclaw-path.mjs +28 -0
- package/lib/openclaw-run.mjs +74 -12
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/outcome.mjs +21 -2
- package/lib/outcomes.mjs +6 -0
- package/lib/path-utils.mjs +4 -0
- package/lib/podman-health.mjs +41 -0
- package/lib/process.mjs +4 -1
- package/lib/propose.mjs +10 -11
- package/lib/refusal-retry.mjs +16 -0
- package/lib/registry-python.mjs +98 -0
- package/lib/registry-secrets.mjs +140 -0
- package/lib/repo-query.mjs +13 -7
- package/lib/runs.mjs +7 -1
- package/lib/same-path.mjs +14 -0
- package/lib/sandbox-images.mjs +499 -83
- package/lib/sandbox-vm.mjs +32 -0
- package/lib/scan.mjs +5 -1
- package/lib/schema.mjs +20 -11
- package/lib/scout.mjs +21 -3
- package/lib/server-context.mjs +21 -1
- package/lib/setup-steps.mjs +55 -0
- package/lib/share.mjs +82 -0
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +315 -0
- package/lib/statusline.mjs +32 -6
- package/lib/subscription-setup.mjs +13 -0
- package/lib/suggestions.mjs +153 -0
- package/lib/thinking.mjs +23 -0
- package/lib/transcript.mjs +30 -5
- package/lib/usage-limits.mjs +329 -0
- package/lib/user-config.mjs +106 -0
- package/lib/validators.mjs +220 -0
- package/lib/verification-artifacts.mjs +46 -0
- package/lib/verification-flow.mjs +52 -7
- package/lib/verification-network.mjs +66 -0
- package/lib/verify.mjs +338 -85
- package/lib/worker-prompt.mjs +5 -2
- package/lib/wsl-cli.mjs +152 -0
- package/lib/wsl.mjs +230 -0
- package/lib/zod-issues.mjs +15 -0
- package/mcp/server.mjs +165 -34
- package/package.json +7 -5
- package/playbooks/feature.md +8 -5
- package/scripts/configure-openclaw.sh +4 -2
- package/scripts/generate-harness-docs.mjs +42 -0
- package/scripts/install-openclaw.mjs +23 -0
- package/scripts/lib.sh +9 -2
- package/scripts/select-model.mjs +12 -5
- package/scripts/start-inference.sh +2 -2
package/lib/execute.mjs
CHANGED
|
@@ -1,13 +1,18 @@
|
|
|
1
|
-
import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery } from "./openclaw-run.mjs";
|
|
1
|
+
import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery, readStopRequest } from "./openclaw-run.mjs";
|
|
2
2
|
import fs from "node:fs";
|
|
3
3
|
import path from "node:path";
|
|
4
4
|
import { fileURLToPath } from "node:url";
|
|
5
|
+
import { vmRestartIssue } from "./podman-health.mjs";
|
|
5
6
|
import { measureReads } from "./job-budgets.mjs";
|
|
6
7
|
import { coordinatorCommitMessage, worktreePointerState } from "./git-record.mjs";
|
|
7
8
|
import { parseScoutReport, verifyCitations, resolveScoutOutcome, renderScoutReport, isScoutReportUnusable, scoutReportRecoveryPrompt } from "./scout.mjs";
|
|
8
9
|
import { parseDecomposeReport, buildDecomposeFindings, resolveDecomposeOutcome, checkDecompositionOverlap, renderDecomposeReport } from "./decompose.mjs";
|
|
9
10
|
import { deriveTimeBudget } from "./budget.mjs";
|
|
10
|
-
import {
|
|
11
|
+
import { continuationProblem, continuationBase, snapshotRetainedWork, applyRetainedWork, continuationNote } from "./continue-from.mjs";
|
|
12
|
+
import { checkScoutCitations, checkReportClaims } from "./jev-checks.mjs";
|
|
13
|
+
import { runJudge } from "./judge.mjs";
|
|
14
|
+
import { pickMutants, runMutants, describeSurvivors } from "./mutation.mjs";
|
|
15
|
+
import { estimateDisplacement, readOpenClawTranscript } from "./transcript.mjs";
|
|
11
16
|
import { outlineFile, findReferences } from "./repo-query.mjs";
|
|
12
17
|
import { loadConfig } from "./config.mjs";
|
|
13
18
|
import { linkNodePackages, nodeModulesState, repairHostInstalls } from "./sandbox-images.mjs";
|
|
@@ -15,8 +20,8 @@ import { detectTestSabotage, addedLinesOf, loadDependencyNames } from "./sabotag
|
|
|
15
20
|
import { describeRecoveryChanges, reportRecoveryPrompt } from "./worker-prompt.mjs";
|
|
16
21
|
import { parseWorkerReport } from "./report.mjs";
|
|
17
22
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "./outcomes.mjs";
|
|
18
|
-
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy } from "./outcome.mjs";
|
|
19
|
-
import { isTestPath,
|
|
23
|
+
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy, HIGH_STAKES_NOTE } from "./outcome.mjs";
|
|
24
|
+
import { parseAddedLineNumbers, isTestPath, planRegressionProductionFiles, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
20
25
|
|
|
21
26
|
// ---------------------------------------------------------------------------
|
|
22
27
|
// Job status for polling. `status.json` is written at every phase transition
|
|
@@ -34,13 +39,23 @@ export function createExecutor(deps) {
|
|
|
34
39
|
resolveReasoningApplied, recordedBudgets, runOpenClaw, sweepStaleSandboxContainers,
|
|
35
40
|
verificationFlow, normalizeVerification, runIndependentVerification,
|
|
36
41
|
runRegressionCheck, repoPolicy } = deps;
|
|
42
|
+
// Optional Jev checks (lib/validators.mjs): the server wires the settings; without them (tests), none run.
|
|
43
|
+
const jevSettingsFor = (check) => { try { const s = deps.jevSettings?.(); return s?.checks?.includes(check) ? s : null; } catch { return null; } };
|
|
37
44
|
|
|
38
|
-
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, jobId: presetJobId = null }) {
|
|
45
|
+
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, stakes = null, reviews = null, jobId: presetJobId = null }) {
|
|
39
46
|
await assertRepo();
|
|
47
|
+
// A continuation starts from the retained job's own base commit.
|
|
48
|
+
let continuation = null;
|
|
49
|
+
if (continueFrom) {
|
|
50
|
+
const checked = continuationProblem({ continueFrom, mode, baseRef, jobsRoot, projectDir });
|
|
51
|
+
if (checked.problem) throw new Error(checked.problem);
|
|
52
|
+
continuation = { jobId: continueFrom, record: checked.record };
|
|
53
|
+
baseRef = continuationBase(checked.record);
|
|
54
|
+
}
|
|
40
55
|
ensureJobsRoot();
|
|
41
56
|
// Fire-and-forget: sweeps whatever this or any other nomArmy install left
|
|
42
57
|
// behind, without adding container-CLI round-trip latency to this job's own start.
|
|
43
|
-
sweepStaleSandboxContainers().catch(() => {});
|
|
58
|
+
if (mode !== "verify") sweepStaleSandboxContainers().catch(() => {});
|
|
44
59
|
const jobStartedMs = Date.now();
|
|
45
60
|
const base = await resolveBase(baseRef), jobId = presetJobId || slug(workerId || (mode === "scout" ? "scout" : mode === "decompose" ? "decompose" : "worker")), jobDir = path.join(jobsRoot, jobId), runtimeDir = path.join(jobDir, "runtime");
|
|
46
61
|
fs.mkdirSync(runtimeDir, { recursive: true });
|
|
@@ -48,20 +63,65 @@ export function createExecutor(deps) {
|
|
|
48
63
|
jobId, workerId: workerId || jobId, mode, phase, state: phase === "finished" ? "finished" : "running",
|
|
49
64
|
serverPid: process.pid, baseSha: base.sha, timeoutSeconds, ...extra
|
|
50
65
|
});
|
|
51
|
-
progress("starting", { startedAt: new Date().toISOString(), agent: pool ?? subscriptionWorker ?? "local", model: model ?? null });
|
|
66
|
+
progress("starting", { startedAt: new Date().toISOString(), agent: mode === "verify" ? null : pool ?? subscriptionWorker ?? "local", model: model ?? null });
|
|
52
67
|
const common = { task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, workerId, progress, jobStartedMs };
|
|
53
|
-
if (mode === "
|
|
68
|
+
if (mode === "verify") return executeVerify({ ...common, verification });
|
|
69
|
+
if (mode === "scout") return executeScout({ ...common, reviews });
|
|
54
70
|
if (mode === "decompose") return executeDecompose(common);
|
|
55
|
-
return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor });
|
|
71
|
+
return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor, continuation, stakes });
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
async function executeVerify({ task, verification: profile, base, jobId, jobDir, workerId, progress, jobStartedMs }) {
|
|
75
|
+
const worktree = path.join(jobDir, "worktree");
|
|
76
|
+
let verification, record = null, error = null;
|
|
77
|
+
try {
|
|
78
|
+
progress("worktree");
|
|
79
|
+
await run("git", ["worktree", "add", "--detach", worktree, base.sha], { cwd: projectDir });
|
|
80
|
+
record = await collectGitRecord({ cwd: worktree, baseSha: base.sha, baseRef: base.ref, branch: null, jobId });
|
|
81
|
+
progress("verification");
|
|
82
|
+
verification = await runIndependentVerification({ profile, cwd: worktree, jobId, baseSha: base.sha, branch: null, mode: "verify", record, logFile: path.join(jobDir, "verification.log") });
|
|
83
|
+
} catch (err) {
|
|
84
|
+
error = err.message;
|
|
85
|
+
verification = normalizeVerification({ status: "not_run", reason: error }, profile);
|
|
86
|
+
} finally {
|
|
87
|
+
try { await run("git", ["worktree", "remove", "--force", worktree], { cwd: projectDir }); }
|
|
88
|
+
catch (err) { error = [error, err.message].filter(Boolean).join("\n"); }
|
|
89
|
+
}
|
|
90
|
+
const retained = fs.existsSync(worktree);
|
|
91
|
+
const outcome = retained ? OUTCOMES.VERIFICATION_NOT_RUN : {
|
|
92
|
+
pass: OUTCOMES.VERIFIED, fail: OUTCOMES.VERIFICATION_FAILED, not_run: OUTCOMES.VERIFICATION_NOT_RUN
|
|
93
|
+
}[verification.status];
|
|
94
|
+
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode: "verify", task,
|
|
95
|
+
baseRef: base.ref, baseSha: base.sha, branch: null, worktree: retained ? worktree : null, worktreeRetained: retained,
|
|
96
|
+
startedAt: new Date(jobStartedMs).toISOString(), finishedAt: new Date().toISOString(),
|
|
97
|
+
outcome, coordinatorStatus: COORDINATOR_STATUS_BY_OUTCOME[outcome], verification, git: record, error,
|
|
98
|
+
// Network access (step 9) and skipped registry credentials (step 8) both surface as issues.
|
|
99
|
+
issues: [...new Set([...(verification.network ? verification.issues ?? [] : []), ...(record?.issues ?? [])])],
|
|
100
|
+
metrics: { total_elapsed: Date.now() - jobStartedMs, worker_cost_usd: 0, worker_tokens_total: 0, model_calls: 0 } };
|
|
101
|
+
fs.writeFileSync(path.join(jobDir, "metadata.json"), JSON.stringify(manifest, null, 2));
|
|
102
|
+
progress("finished", { outcome, coordinatorStatus: manifest.coordinatorStatus });
|
|
103
|
+
return { ok: outcome === OUTCOMES.VERIFIED, manifest, jobDir, report: "" };
|
|
56
104
|
}
|
|
57
105
|
|
|
58
|
-
async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, progress, jobStartedMs }) {
|
|
106
|
+
async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, continuation = null, stakes = null, progress, jobStartedMs }) {
|
|
59
107
|
const mode = "implement";
|
|
60
108
|
let branch = `agent/${jobId}`, worktree = path.join(jobDir, "worktree");
|
|
61
109
|
try {
|
|
62
110
|
progress("worktree");
|
|
63
111
|
await run("git", ["worktree", "add", "-b", branch, worktree, base.sha], { cwd: projectDir });
|
|
64
112
|
const cwd = worktree;
|
|
113
|
+
// continue_from: lay the retained job's unfinished work into this
|
|
114
|
+
// worktree as uncommitted changes, so this job's diff (and so its
|
|
115
|
+
// verification and revert check) covers that work too.
|
|
116
|
+
let continuedFrom = null;
|
|
117
|
+
if (continuation) {
|
|
118
|
+
const git = async (args, opts = {}) => (await run("git", args, { trim: false, ...opts })).stdout;
|
|
119
|
+
const snapshot = await snapshotRetainedWork({ worktree: continuation.record.worktree, baseSha: base.sha, jobId: continuation.jobId, git });
|
|
120
|
+
await applyRetainedWork({ worktree, baseSha: base.sha, commit: snapshot.commit, git });
|
|
121
|
+
continuedFrom = { jobId: continuation.jobId, snapshot: snapshot.commit, files: snapshot.files };
|
|
122
|
+
evidence = [continuationNote({ continueFrom: continuation.jobId, record: continuation.record, files: snapshot.files }), evidence].filter(Boolean).join("\n\n");
|
|
123
|
+
fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} continuing job ${continuation.jobId}: ${snapshot.files.length} file(s) of its unfinished work carried into this worktree (snapshot ${snapshot.commit})\n`);
|
|
124
|
+
}
|
|
65
125
|
// Each npm package below the root reaches its install in the sandbox image.
|
|
66
126
|
const nodeConfig = (() => { try { return loadConfig(projectDir)?.config ?? null; } catch { return null; } })();
|
|
67
127
|
try { linkNodePackages(worktree, nodeConfig); } catch { /* verification reports what's missing */ }
|
|
@@ -79,6 +139,7 @@ export function createExecutor(deps) {
|
|
|
79
139
|
const timeBudget = deriveTimeBudget({ timeoutSeconds });
|
|
80
140
|
let result = null, attempted = null, workerFailed = false, workerTimedOut = false, workerStopReason = null, workerError = null;
|
|
81
141
|
const workerStartedMs = Date.now();
|
|
142
|
+
const vmStartedBefore = deps.podmanVmStartedAt?.() ?? null;
|
|
82
143
|
progress("worker");
|
|
83
144
|
try {
|
|
84
145
|
result = await runOpenClaw({
|
|
@@ -177,7 +238,7 @@ export function createExecutor(deps) {
|
|
|
177
238
|
const recoveryResult = await runOpenClaw({
|
|
178
239
|
task, acceptance, verification, mode, cwd, baseRef: base.ref, baseSha: base.sha,
|
|
179
240
|
timeoutSeconds: timeBudget.reportReserveSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId,
|
|
180
|
-
overridePrompt: reportRecoveryPrompt({ report: budgetState.budgets.report.implement, changes }), logSuffix: "-recovery",
|
|
241
|
+
overridePrompt: reportRecoveryPrompt({ report: budgetState.budgets.report.implement, changes, task }), logSuffix: "-recovery",
|
|
181
242
|
});
|
|
182
243
|
const recoveryText = finalText(recoveryResult);
|
|
183
244
|
const recoveryValidation = parseWorkerReport(recoveryText);
|
|
@@ -214,23 +275,26 @@ export function createExecutor(deps) {
|
|
|
214
275
|
// failed job that changed nothing was recorded "pass" (a Senti run),
|
|
215
276
|
// which reads as evidence about work that never happened.
|
|
216
277
|
independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the worker changed nothing, so there was none of its work to verify" }, verification ?? null);
|
|
278
|
+
} else if (workerStopReason === "stopped") {
|
|
279
|
+
// Stopped on request: end now, without spending time on tests.
|
|
280
|
+
independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the job was stopped on request" }, verification ?? null);
|
|
217
281
|
} else if (verificationFlow.verificationRunner || !reportValidation.valid) {
|
|
218
|
-
independentVerification = await runIndependentVerification({ profile: verification ?? null, cwd, jobId, baseSha: base.sha, branch, mode, record: preCommit });
|
|
282
|
+
independentVerification = await runIndependentVerification({ profile: verification ?? null, cwd, jobId, baseSha: base.sha, branch, mode, record: preCommit, logFile: path.join(jobDir, "verification.log") });
|
|
219
283
|
}
|
|
220
284
|
|
|
221
285
|
// verify_regression: on by default whenever there's a verification
|
|
222
286
|
// profile (resolveVerifyRegression). It doubles verification wall-clock,
|
|
223
287
|
// so it runs only when there's something to re-check: a passing
|
|
224
288
|
// first-pass verification on a diff that touched production code.
|
|
225
|
-
// Documentation
|
|
226
|
-
//
|
|
227
|
-
const codeFilesChanged = preCommit.testChanges.production_files_changed
|
|
289
|
+
// Documentation and CI configuration cannot be proven by local tests,
|
|
290
|
+
// so neither is reverted or used as a reason to run this check.
|
|
291
|
+
const codeFilesChanged = planRegressionProductionFiles(preCommit.testChanges.production_files_changed);
|
|
228
292
|
let regressionCheck = null, regressionCheckFatal = false, regressionCheckElapsedMs = null;
|
|
229
293
|
if (verifyRegression && independentVerification.status === "pass" && codeFilesChanged.length > 0) {
|
|
230
294
|
const regressionStartedMs = Date.now();
|
|
231
295
|
try {
|
|
232
296
|
regressionCheck = await runRegressionCheck({
|
|
233
|
-
cwd, jobId, productionFiles: codeFilesChanged,
|
|
297
|
+
cwd, jobId, productionFiles: codeFilesChanged, verificationResult: independentVerification,
|
|
234
298
|
nameStatus: preCommit.nameStatus, profile: verification, baseSha: base.sha, branch, mode,
|
|
235
299
|
});
|
|
236
300
|
} catch (error) {
|
|
@@ -244,6 +308,36 @@ export function createExecutor(deps) {
|
|
|
244
308
|
if (regressionCheck.status === "restore_failed") regressionCheckFatal = true;
|
|
245
309
|
}
|
|
246
310
|
|
|
311
|
+
// Mutation testing (lib/mutation.mjs), when this repo opts in: small
|
|
312
|
+
// mistakes planted one at a time in the changed lines must each fail
|
|
313
|
+
// the same profile. Survivors raise review; a failed restore is fatal.
|
|
314
|
+
let mutation = null, mutationElapsedMs = null;
|
|
315
|
+
const mutationConfig = (() => { try { return loadConfig(projectDir)?.config?.mutation ?? null; } catch { return null; } })();
|
|
316
|
+
if (mutationConfig && verification && independentVerification.status === "pass" && codeFilesChanged.length > 0 && !regressionCheckFatal) {
|
|
317
|
+
const mutationStartedMs = Date.now();
|
|
318
|
+
try {
|
|
319
|
+
const untracked = new Set((preCommit.nameStatus ?? []).filter((e) => e.untracked).map((e) => e.path));
|
|
320
|
+
const deleted = new Set((preCommit.nameStatus ?? []).filter((e) => /^D/.test(e.status)).map((e) => e.path));
|
|
321
|
+
const files = [];
|
|
322
|
+
for (const file of codeFilesChanged.filter((f) => !deleted.has(f))) {
|
|
323
|
+
const full = path.join(cwd, file);
|
|
324
|
+
let lines;
|
|
325
|
+
if (untracked.has(file)) { try { lines = fs.readFileSync(full, "utf8").split("\n").map((_, i) => i + 1); } catch { continue; } }
|
|
326
|
+
else lines = parseAddedLineNumbers(await gitRaw(["diff", "-U0", base.sha, "--", file], cwd));
|
|
327
|
+
if (lines.length) files.push({ path: file, full, lines });
|
|
328
|
+
}
|
|
329
|
+
const mutants = pickMutants(files, mutationConfig.mutants);
|
|
330
|
+
let n = 0;
|
|
331
|
+
mutation = await runMutants({ mutants, deadlineMs: mutationStartedMs + mutationConfig.max_seconds * 1000,
|
|
332
|
+
verify: () => runIndependentVerification({ profile: verification, cwd, jobId: `${jobId}-mutant-${++n}`, baseSha: base.sha, branch, mode, record: preCommit }) });
|
|
333
|
+
mutation.planned = mutants.length;
|
|
334
|
+
} catch (error) {
|
|
335
|
+
mutation = { status: "not_run", killed: 0, survived: [], inconclusive: 0, tried: 0, reason: `mutation testing failed to run: ${error.message}` };
|
|
336
|
+
}
|
|
337
|
+
mutationElapsedMs = Date.now() - mutationStartedMs;
|
|
338
|
+
fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} mutation testing: ${mutation.killed ?? 0} killed, ${mutation.survived?.length ?? 0} survived, ${mutation.inconclusive ?? 0} inconclusive of ${mutation.tried ?? 0} tried (${Math.round(mutationElapsedMs / 1000)}s)\n`);
|
|
339
|
+
}
|
|
340
|
+
|
|
247
341
|
// resolveOutcome's own contract only ever sees pass/fail/not_run for
|
|
248
342
|
// regressionCheck -- a restore_failed status is substituted to not_run
|
|
249
343
|
// here so resolveOutcome never needs a fourth value; the hard override
|
|
@@ -264,12 +358,19 @@ export function createExecutor(deps) {
|
|
|
264
358
|
// Cheap, always-on, additive: never changes commitAllowed/commitBlockedReason
|
|
265
359
|
// on its own (unlike the regression-check override above), only flags for
|
|
266
360
|
// review -- see detectScopedTestSelectionRisk's own doc comment for why.
|
|
267
|
-
let selectionRisk = null;
|
|
361
|
+
let selectionRisk = null, verificationInputs = null;
|
|
268
362
|
if (mode === "implement" && verification) {
|
|
269
363
|
try {
|
|
270
364
|
const loaded = loadConfig(projectDir); // the operator's contract; see registerVerificationRunner's call
|
|
271
365
|
const profileCommands = loaded.found ? (loaded.config?.verification?.[verification]?.commands ?? []) : [];
|
|
272
366
|
selectionRisk = detectScopedTestSelectionRisk({ commands: profileCommands, testChanges: preCommit.testChanges });
|
|
367
|
+
// A diff that changes what those commands run (see
|
|
368
|
+
// detectVerificationInputChanges): applied below, after the others.
|
|
369
|
+
verificationInputs = await detectVerificationInputChanges({
|
|
370
|
+
commands: profileCommands, changedFiles: preCommit.changedFiles ?? [],
|
|
371
|
+
readBase: (file) => gitRaw(["show", `${base.sha}:${file}`], cwd).catch(() => null),
|
|
372
|
+
readHead: (file) => { try { return fs.readFileSync(path.join(cwd, file), "utf8"); } catch { return null; } },
|
|
373
|
+
});
|
|
273
374
|
} catch { /* a config load failure here is the verification runner's own problem to report, not this check's */ }
|
|
274
375
|
}
|
|
275
376
|
const afterSelectionRisk = selectionRisk
|
|
@@ -364,18 +465,90 @@ export function createExecutor(deps) {
|
|
|
364
465
|
commitBlockedReason: `possible secret detected: ${possibleSecrets.reason}`,
|
|
365
466
|
reasons: [...afterHostInstalls.reasons, `POSSIBLE SECRET DETECTED: ${possibleSecrets.reason}`] }
|
|
366
467
|
: afterHostInstalls;
|
|
367
|
-
|
|
468
|
+
// Jev: does the worker's report match its diff? Found live: a note said
|
|
469
|
+
// "restored check.js to base commit" while the diff rewrote check.js.
|
|
470
|
+
// Only raises review; it never blocks or allows a commit.
|
|
471
|
+
let jevClaims = null, judged = null;
|
|
472
|
+
const jevImplement = jevSettingsFor("report-claims");
|
|
473
|
+
const judge = (() => { try { return deps.judgeSettings?.() ?? null; } catch { return null; } })();
|
|
474
|
+
// The whole change, new files included, for whichever validators run.
|
|
475
|
+
const jobDiff = async () => {
|
|
476
|
+
let diff = await gitRaw(["diff", base.sha, "--"], cwd);
|
|
477
|
+
for (const entry of (preCommit.nameStatus ?? []).filter((e) => e.untracked)) {
|
|
478
|
+
let text = "";
|
|
479
|
+
try { text = fs.readFileSync(path.join(cwd, entry.path), "utf8"); } catch { continue; }
|
|
480
|
+
diff += `\ndiff --git a/${entry.path} b/${entry.path}\nnew file\n--- /dev/null\n+++ b/${entry.path}\n${text.split("\n").map((l) => `+${l}`).join("\n")}\n`;
|
|
481
|
+
}
|
|
482
|
+
return diff;
|
|
483
|
+
};
|
|
484
|
+
let diffText = null;
|
|
485
|
+
if (mode === "implement" && reportValidation?.valid && repositoryChanged && (jevImplement || (judge && !judge.problem))) {
|
|
486
|
+
try { diffText = await jobDiff(); } catch { diffText = null; }
|
|
487
|
+
}
|
|
488
|
+
if (jevImplement && diffText != null) {
|
|
489
|
+
try { jevClaims = await checkReportClaims({ report: reportValidation, diff: diffText, settings: jevImplement }); }
|
|
490
|
+
catch (error) { jevClaims = { flag: null, verdict: null, error: error.message, usage: 0 }; }
|
|
491
|
+
}
|
|
492
|
+
// The model judge (lib/judge.mjs): acceptance criteria, the report and
|
|
493
|
+
// changed tests. Only raises review, never blocks or allows a commit.
|
|
494
|
+
if (judge?.problem) judged = { flags: [], answer: null, error: `judge not run: ${judge.problem}`, skipped: true };
|
|
495
|
+
else if (judge && diffText != null) {
|
|
496
|
+
const modifiedTests = preCommit.testChanges?.existing_tests_modified ?? [];
|
|
497
|
+
let testDiff = "";
|
|
498
|
+
if (modifiedTests.length) { try { testDiff = await gitRaw(["diff", base.sha, "--", ...modifiedTests], cwd); } catch { testDiff = ""; } }
|
|
499
|
+
judged = await runJudge({ settings: judge, task, acceptance: acceptance ?? [], report: reportValidation, diff: diffText, testDiff, stateRoot: path.join(jobsRoot, "..") });
|
|
500
|
+
}
|
|
501
|
+
const afterJevOnly = jevClaims?.flag
|
|
502
|
+
? { ...afterSecrets, reviewRequired: true, reasons: [...afterSecrets.reasons, `REPORT MAY NOT MATCH THE DIFF (Jev, ${jevClaims.flag.probability.toFixed(2)}): the report says "${String(reportValidation.note ?? "").slice(0, 200)}", and the diff may show otherwise. Read the diff before accepting.`] }
|
|
503
|
+
: afterSecrets;
|
|
504
|
+
const afterJev = judged?.flags?.length
|
|
505
|
+
? { ...afterJevOnly, reviewRequired: true, reasons: [...afterJevOnly.reasons, `JUDGE (${judge.agent}/${judge.model}): ${judged.flags.join("; ")}. Read the diff before accepting.`] }
|
|
506
|
+
: afterJevOnly;
|
|
507
|
+
|
|
508
|
+
// A worker must not be judged by a check it rewrote: a changed script,
|
|
509
|
+
// Makefile or package.json script that a verification command runs
|
|
510
|
+
// blocks the commit; changed test-runner config only asks for review.
|
|
511
|
+
const blockedInputs = verificationInputs?.blocked ?? [], flaggedInputs = verificationInputs?.flagged ?? [];
|
|
512
|
+
const inputLine = blockedInputs.map((b) => `${b.file} (${b.why})`).join("; ");
|
|
513
|
+
const afterInputs = blockedInputs.length
|
|
514
|
+
? { ...afterJev, outcome: afterJev.commitAllowed || afterJev.outcome === OUTCOMES.WORKER_DONE ? OUTCOMES.NEEDS_REVIEW : afterJev.outcome,
|
|
515
|
+
reviewRequired: true, commitAllowed: false,
|
|
516
|
+
commitBlockedReason: afterJev.commitAllowed ? `the diff changes what verification runs: ${inputLine}` : afterJev.commitBlockedReason,
|
|
517
|
+
reasons: [...afterJev.reasons, `VERIFICATION INPUT CHANGED: the diff changes what profile '${verification}' runs, so its result can't be trusted: ${inputLine}`] }
|
|
518
|
+
: afterJev;
|
|
519
|
+
const afterConfig = flaggedInputs.length
|
|
520
|
+
? { ...afterInputs, reviewRequired: true, reasons: [...afterInputs.reasons, `TEST CONFIG CHANGED: ${flaggedInputs.map((c) => `${c.file} (${c.why})`).join("; ")}`] }
|
|
521
|
+
: afterInputs;
|
|
522
|
+
const afterMutation = mutation?.status === "restore_failed"
|
|
523
|
+
? { ...afterConfig, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true, commitAllowed: false,
|
|
524
|
+
commitBlockedReason: `mutation testing could not restore the worker's file: ${mutation.reason}`, reasons: [...afterConfig.reasons, `MUTATION RESTORE FAILED: ${mutation.reason}`] }
|
|
525
|
+
: mutation?.status === "survivors"
|
|
526
|
+
? { ...afterConfig, reviewRequired: true, reasons: [...afterConfig.reasons, describeSurvivors(mutation, verification)] }
|
|
527
|
+
: afterConfig;
|
|
528
|
+
const afterStakes = stakes === "high" ? { ...afterMutation, reviewRequired: true, reasons: [...afterMutation.reasons, HIGH_STAKES_NOTE] } : afterMutation;
|
|
529
|
+
const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterStakes, independentVerification.status, repoPolicy()),
|
|
368
530
|
{ refactor, verificationStatus: independentVerification.status, testChanges: preCommit.testChanges });
|
|
369
531
|
|
|
370
532
|
progress("commit");
|
|
371
533
|
const commit = await createCoordinatorCommit({ cwd, jobId, outcome: finalOutcome,
|
|
372
|
-
message: coordinatorCommitMessage({ task, subject: commitSubject, note: reportValidation?.note ?? null, jobId, workerId, recovered: Boolean(finalOutcome.recovered), provider: (result ?? attempted)?.provider ?? null, model: (result ?? attempted)?.model ?? null }) });
|
|
534
|
+
message: coordinatorCommitMessage({ task: continuation?.record?.objective ?? task, subject: commitSubject, note: reportValidation?.note ?? null, jobId, workerId, recovered: Boolean(finalOutcome.recovered), provider: (result ?? attempted)?.provider ?? null, model: (result ?? attempted)?.model ?? null, continuedFrom: continuedFrom?.jobId ?? null }) });
|
|
373
535
|
progress("record");
|
|
374
536
|
const record = await collectGitRecord({ cwd, baseSha: base.sha, branch, baseRef: base.ref, jobId }), worker = workerMetadata(result ?? attempted);
|
|
375
537
|
|
|
376
538
|
let coordinatorStatus = COORDINATOR_STATUS_BY_OUTCOME[finalOutcome.outcome] ?? "incomplete";
|
|
377
|
-
const issues = [...finalOutcome.reasons];
|
|
378
|
-
if (
|
|
539
|
+
const issues = [...finalOutcome.reasons, ...(preCommit.issues ?? [])];
|
|
540
|
+
if (workerStopReason === "stopped") {
|
|
541
|
+
const request = readStopRequest(jobDir);
|
|
542
|
+
issues.push(`stopped on request${request?.reason ? `: ${request.reason}` : ""}; the worktree is kept, so continue_from can pick the work up (on another model too)`);
|
|
543
|
+
} else if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
|
|
544
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
545
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
546
|
+
// stats, out of the issues the General reviews.
|
|
547
|
+
const runnerNotes = [];
|
|
548
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
|
|
549
|
+
if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
|
|
550
|
+
if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
|
|
551
|
+
if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
|
|
379
552
|
if (repositoryChanged && !commit.created) {
|
|
380
553
|
if (coordinatorStatus === "complete") coordinatorStatus = "incomplete";
|
|
381
554
|
// A timed-out or crashed worker can still leave real, salvageable work
|
|
@@ -385,7 +558,7 @@ export function createExecutor(deps) {
|
|
|
385
558
|
// the diffstat right in the issue a caller actually reads -- not just
|
|
386
559
|
// buried in the full manifest's git record -- is what makes "go look at
|
|
387
560
|
// the worktree" worth doing instead of discarding the job.
|
|
388
|
-
issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
|
|
561
|
+
if (record.filesChanged > 0) issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
|
|
389
562
|
}
|
|
390
563
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`worker recorded ${failures} tool failure(s)`);
|
|
391
564
|
if (record.ignoredRuntimeJunk.length) issues.push(`runtime junk ignored: ${record.ignoredRuntimeJunk.join(", ")}`);
|
|
@@ -403,11 +576,13 @@ export function createExecutor(deps) {
|
|
|
403
576
|
|
|
404
577
|
const metrics = buildMetrics({ result: result ?? attempted, record, reportValidation, outcome: finalOutcome, workerElapsedMs, totalElapsedMs: Date.now() - jobStartedMs, regressionCheckElapsedMs, transientAbortRetried });
|
|
405
578
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree, branch, startedAt, finishedAt,
|
|
406
|
-
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null,
|
|
579
|
+
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}), ...(stakes ? { stakes } : {}),
|
|
580
|
+
...(mutation ? { mutation: { ...mutation, elapsedSeconds: Math.round((mutationElapsedMs ?? 0) / 1000) } } : {}),
|
|
581
|
+
...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, provider: judge?.provider ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
|
|
407
582
|
outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
|
|
408
583
|
reportRecoveryAttempted, reportRecovered,
|
|
409
584
|
reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
|
|
410
|
-
coordinatorStatus, issues, reportValidation, independentVerification,
|
|
585
|
+
coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], runnerNotes, reportValidation, independentVerification,
|
|
411
586
|
// Original, unsubstituted regressionCheck (real "restore_failed" status
|
|
412
587
|
// visible here even though resolveOutcome above only ever saw a
|
|
413
588
|
// not_run-substituted view) -- full transparency for the caller.
|
|
@@ -446,7 +621,7 @@ export function createExecutor(deps) {
|
|
|
446
621
|
// the worktree, so a scout that wrote to its snapshot cannot forge evidence.
|
|
447
622
|
// A clean scout worktree holds no work and is removed; a dirty one is retained
|
|
448
623
|
// because a scout that wrote is a scout that misbehaved, and that is worth a look.
|
|
449
|
-
async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs }) {
|
|
624
|
+
async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs, reviews = null }) {
|
|
450
625
|
const mode = "scout", worktree = path.join(jobDir, "worktree");
|
|
451
626
|
let worktreeRetained = false;
|
|
452
627
|
try {
|
|
@@ -469,7 +644,10 @@ export function createExecutor(deps) {
|
|
|
469
644
|
const workerStartedMs = Date.now();
|
|
470
645
|
progress("worker");
|
|
471
646
|
try {
|
|
472
|
-
|
|
647
|
+
// The run gets the work share; the reserve stays for report recovery
|
|
648
|
+
// below. Scouts had none, so a timed-out scout left about 0 seconds
|
|
649
|
+
// and its findings were never recovered.
|
|
650
|
+
result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds: deriveTimeBudget({ timeoutSeconds }).workTimeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
|
|
473
651
|
} catch (error) {
|
|
474
652
|
workerFailed = true;
|
|
475
653
|
// error.timedOut is set only by our own spawn timer (run(), above) --
|
|
@@ -499,18 +677,26 @@ export function createExecutor(deps) {
|
|
|
499
677
|
|
|
500
678
|
// See shouldAttemptScoutRecovery's own doc comment: this only fires when
|
|
501
679
|
// the report is genuinely unusable, gated by whatever time is actually
|
|
502
|
-
// left against the caller's original timeout (
|
|
503
|
-
//
|
|
680
|
+
// left against the caller's original timeout (the reserve held back
|
|
681
|
+
// from the run above).
|
|
504
682
|
let reportRecoveryAttempted = false, reportRecovered = false;
|
|
505
|
-
|
|
683
|
+
// At least the reserve, as implement's recovery gets: nomArmy's own kill
|
|
684
|
+
// lands 30s after OpenClaw's timer, which would otherwise eat it.
|
|
685
|
+
const remainingSeconds = Math.max(deriveTimeBudget({ timeoutSeconds }).reportReserveSeconds, timeoutSeconds - Math.round(workerElapsedMs / 1000));
|
|
506
686
|
if (shouldAttemptScoutRecovery({ workerFailed, workerTimedOut, report, remainingSeconds })) {
|
|
507
687
|
reportRecoveryAttempted = true;
|
|
688
|
+
// What the first run left, in case the follow-up can't see its session.
|
|
689
|
+
let filesRead = [], earlierReply = reportText;
|
|
690
|
+
try {
|
|
691
|
+
const first = await readOpenClawTranscript(path.join(runtimeDir, "state"));
|
|
692
|
+
if (first.available) { filesRead = first.filesRead ?? []; earlierReply = earlierReply || first.lastAssistantText || ""; }
|
|
693
|
+
} catch { /* the reply alone still helps */ }
|
|
508
694
|
try {
|
|
509
695
|
const recoveryResult = await runOpenClaw({
|
|
510
696
|
task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha,
|
|
511
697
|
timeoutSeconds: remainingSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId,
|
|
512
698
|
evidenceTool: evidencePlaced ? evidenceTool : null,
|
|
513
|
-
overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout }), logSuffix: "-recovery",
|
|
699
|
+
overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout, question: task, acceptance, earlierReply, filesRead }), logSuffix: "-recovery",
|
|
514
700
|
});
|
|
515
701
|
const recoveryReport = parseScoutReport(finalText(recoveryResult), (recoveryResult?.budgetsUsed ?? used).scout);
|
|
516
702
|
if (!isScoutReportUnusable(recoveryReport)) {
|
|
@@ -530,7 +716,16 @@ export function createExecutor(deps) {
|
|
|
530
716
|
const dirty = record.repoStatusFiles.length > 0;
|
|
531
717
|
const readFile = async p => { try { return await gitRaw(["show", `${base.sha}:${p}`], projectDir); } catch { return null; } };
|
|
532
718
|
const verified = await verifyCitations(report.findings, { readFile, limits: used.scout });
|
|
719
|
+
// Jev: do the cited lines support each finding? Only adds flags.
|
|
720
|
+
let jevCitations = null;
|
|
721
|
+
const jevScout = jevSettingsFor("scout-citations");
|
|
722
|
+
if (jevScout && verified?.findings?.length) {
|
|
723
|
+
try { jevCitations = await checkScoutCitations({ findings: verified.findings, settings: jevScout, readFile }); }
|
|
724
|
+
catch (error) { jevCitations = { flags: [], checked: 0, errors: [error.message], usage: 0, verdicts: [] }; }
|
|
725
|
+
for (const f of jevCitations.flags) verified.findings[f.index].jev = { verdict: f.verdict, probability: f.probability };
|
|
726
|
+
}
|
|
533
727
|
const outcome = resolveScoutOutcome({ report, verified, workerFailed, workerTimedOut, dirty });
|
|
728
|
+
if (jevCitations?.flags.length) outcome.reviewRequired = true;
|
|
534
729
|
|
|
535
730
|
progress("record");
|
|
536
731
|
if (outcome.retainWorktree) worktreeRetained = true;
|
|
@@ -539,8 +734,15 @@ export function createExecutor(deps) {
|
|
|
539
734
|
const worker = workerMetadata(result ?? attempted);
|
|
540
735
|
const issues = [...outcome.reasons];
|
|
541
736
|
if (workerError) issues.push(`scout error: ${String(workerError).split("\n")[0]}`);
|
|
737
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
738
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
739
|
+
// stats, out of the issues the General reviews.
|
|
740
|
+
const runnerNotes = [];
|
|
741
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
|
|
542
742
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
|
|
543
743
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
744
|
+
if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
|
|
745
|
+
if (jevCitations?.errors.length) issues.push(`Jev citation check skipped or incomplete (${jevCitations.errors.join("; ")}); this job's result doesn't depend on it`);
|
|
544
746
|
if (reportRecoveryAttempted) {
|
|
545
747
|
issues.push(reportRecovered
|
|
546
748
|
? "scout report recovered via a follow-up call after the first reply was cut off"
|
|
@@ -577,8 +779,9 @@ export function createExecutor(deps) {
|
|
|
577
779
|
displacement_verdict: displacement.verdict
|
|
578
780
|
};
|
|
579
781
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
580
|
-
objective: task, mustCover: acceptance ?? [],
|
|
581
|
-
|
|
782
|
+
objective: task, mustCover: acceptance ?? [], ...(reviews ? { reviews } : {}),
|
|
783
|
+
...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
|
|
784
|
+
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
|
|
582
785
|
scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
|
|
583
786
|
findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|
|
584
787
|
excerptLinesUsed: verified.excerptLinesUsed, excerptTruncated: verified.excerptTruncated,
|
|
@@ -670,6 +873,11 @@ export function createExecutor(deps) {
|
|
|
670
873
|
const worker = workerMetadata(result ?? attempted);
|
|
671
874
|
const issues = [...outcome.reasons];
|
|
672
875
|
if (workerError) issues.push(`decompose error: ${String(workerError).split("\n")[0]}`);
|
|
876
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
877
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
878
|
+
// stats, out of the issues the General reviews.
|
|
879
|
+
const runnerNotes = [];
|
|
880
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the decomposer's report was recovered from the run's transcript`);
|
|
673
881
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`decomposer recorded ${failures} tool failure(s)`);
|
|
674
882
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
675
883
|
if (overlaps.length) issues.push(`${overlaps.length} subtask pair(s) claim overlapping files; not safe to dispatch as independent jobs as proposed`);
|
|
@@ -699,7 +907,7 @@ export function createExecutor(deps) {
|
|
|
699
907
|
};
|
|
700
908
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
701
909
|
objective: task, constraints: acceptance ?? [],
|
|
702
|
-
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
910
|
+
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
|
|
703
911
|
decompose: { objective: report.objective, confidence: report.confidence, notSplittable: report.notSplittable,
|
|
704
912
|
subtasks: report.subtasks.map((s, i) => ({ task: s.task, acceptance: s.acceptance, citations: verified.findings[i]?.citations ?? [], supported: verified.findings[i]?.supported ?? false, weak: verified.findings[i]?.weak ?? false })),
|
|
705
913
|
overlaps, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|
package/lib/git-record.mjs
CHANGED
|
@@ -65,7 +65,7 @@ export function isRuntimeJunk(file) {
|
|
|
65
65
|
* gave one, else the task's first sentence (the army role header and an
|
|
66
66
|
* "OBJECTIVE:" label dropped). The job id stays, as a trailer.
|
|
67
67
|
*/
|
|
68
|
-
export function coordinatorCommitMessage({ task = "", subject = null, note = null, jobId, workerId = null, recovered = false, provider = null, model = null }) {
|
|
68
|
+
export function coordinatorCommitMessage({ task = "", subject = null, note = null, jobId, workerId = null, recovered = false, provider = null, model = null, continuedFrom = null }) {
|
|
69
69
|
const oneLine = (t) => String(t ?? "").replace(/\s+/g, " ").trim();
|
|
70
70
|
let body = String(task ?? "");
|
|
71
71
|
if (/^\[nomArmy role:/.test(body)) body = body.includes("\n\n") ? body.slice(body.indexOf("\n\n") + 2) : "";
|
|
@@ -78,6 +78,7 @@ export function coordinatorCommitMessage({ task = "", subject = null, note = nul
|
|
|
78
78
|
const cleanNote = oneLine(note);
|
|
79
79
|
if (cleanNote) lines.push("", ...wrapText(cleanNote, 72));
|
|
80
80
|
lines.push("", `nomArmy-Job: ${jobId}`);
|
|
81
|
+
if (continuedFrom) lines.push(`nomArmy-Continues: ${continuedFrom}`);
|
|
81
82
|
if (provider || model) lines.push(`nomArmy-Worker: ${[provider, model].filter(Boolean).join("/")}`);
|
|
82
83
|
return lines.join("\n");
|
|
83
84
|
}
|
|
@@ -138,9 +139,26 @@ export function createGitRecord({ run, git, gitRaw }) {
|
|
|
138
139
|
// idleMinElapsedMs of the work phase has passed (an early snapshot mid-first-
|
|
139
140
|
// edit looks identical to no edit at all). A worktree read failing mid-write
|
|
140
141
|
// is expected, not an error; it just means "nothing to report this tick."
|
|
141
|
-
|
|
142
|
-
|
|
142
|
+
//
|
|
143
|
+
// An unchanged worktree isn't enough on its own: a frontier worker running
|
|
144
|
+
// a suite of thousands of tests, or reading before its next edit, changes
|
|
145
|
+
// no file for minutes. Three Senti jobs were cut off at 7 to 13 minutes
|
|
146
|
+
// while still working. With `activity` (the worker's transcript: event
|
|
147
|
+
// count and whether a tool call is still running), the breaker waits while
|
|
148
|
+
// the worker is active, and stops only when it has also gone quiet for
|
|
149
|
+
// idleMs, or when nothing has changed for ACTIVE_CAP times idleMs however
|
|
150
|
+
// busy it looks (a worker looping without progress).
|
|
151
|
+
function makeIdleDiffTick(cwd, { idleMs, minElapsedMs, activity = null }) {
|
|
152
|
+
const ACTIVE_CAP = 3;
|
|
153
|
+
let lastHash = null, lastChangeAtMs = 0, sawChange = false, lastEvents = null, lastActivityAtMs = 0, inFlight = false;
|
|
143
154
|
return async elapsedMs => {
|
|
155
|
+
if (activity) {
|
|
156
|
+
try {
|
|
157
|
+
const a = await activity();
|
|
158
|
+
if (a && a.events !== lastEvents) { lastEvents = a.events; lastActivityAtMs = elapsedMs; }
|
|
159
|
+
inFlight = Boolean(a?.toolInFlight);
|
|
160
|
+
} catch { /* a failed read just means no activity signal this tick */ }
|
|
161
|
+
}
|
|
144
162
|
let statusOut;
|
|
145
163
|
try { statusOut = await gitRaw(["status", "--porcelain=v1", "-z", "--untracked-files=all"], cwd); }
|
|
146
164
|
catch { return { stop: false }; }
|
|
@@ -181,6 +199,12 @@ export function createGitRecord({ run, git, gitRaw }) {
|
|
|
181
199
|
if (!sawChange || elapsedMs < minElapsedMs) return { stop: false };
|
|
182
200
|
const idleForMs = elapsedMs - lastChangeAtMs;
|
|
183
201
|
if (idleForMs < idleMs) return { stop: false };
|
|
202
|
+
if (activity) {
|
|
203
|
+
const quietForMs = elapsedMs - lastActivityAtMs;
|
|
204
|
+
const active = inFlight || quietForMs < idleMs;
|
|
205
|
+
if (active && idleForMs < idleMs * ACTIVE_CAP) return { stop: false };
|
|
206
|
+
if (active) return { stop: true, reason: "idle_diff", detail: `worktree unchanged for ${Math.round(idleForMs / 1000)}s while the worker kept working (${ACTIVE_CAP}x the idle limit)` };
|
|
207
|
+
}
|
|
184
208
|
return { stop: true, reason: "idle_diff", detail: `worktree unchanged for ${Math.round(idleForMs / 1000)}s` };
|
|
185
209
|
};
|
|
186
210
|
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
|
|
3
|
+
const text = z.string().trim().min(1, "must not be empty");
|
|
4
|
+
const name = z.string().regex(/^[a-z][a-z0-9]*(?:-[a-z0-9]+)*$/, "must be kebab-case");
|
|
5
|
+
const relativePath = text.refine((value) => !value.startsWith("/") && !value.includes("\\") && !value.includes(":") && !value.split("/").includes(".."), "must be a repository-relative path");
|
|
6
|
+
const detect = z.union([
|
|
7
|
+
z.object({ file: relativePath }).strict(),
|
|
8
|
+
z.object({ package: text }).strict(),
|
|
9
|
+
z.object({ lockfile: relativePath }).strict(),
|
|
10
|
+
]);
|
|
11
|
+
|
|
12
|
+
const env = z.record(z.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/), z.string());
|
|
13
|
+
const pinnedImage = text.refine((value) => {
|
|
14
|
+
if (/\s/.test(value) || value.startsWith("-")) return false;
|
|
15
|
+
const reference = value.split("@")[0];
|
|
16
|
+
if (reference.split("/").at(-1).endsWith(":latest")) return false;
|
|
17
|
+
return /^[a-z0-9][a-z0-9._:/-]*@sha256:[a-f0-9]{64}$/.test(value)
|
|
18
|
+
|| /^[a-z0-9][a-z0-9._:/-]*:[A-Za-z0-9_][A-Za-z0-9_.-]*$/.test(value)
|
|
19
|
+
&& value.split("/").at(-1).includes(":");
|
|
20
|
+
}, "image must have an explicit non-latest tag or sha256 digest");
|
|
21
|
+
const service = z.object({
|
|
22
|
+
name, image: pinnedImage, port: z.number().int().min(1).max(65535),
|
|
23
|
+
env: env.optional(),
|
|
24
|
+
health: z.string().regex(/^\/(?!\/)[^\s\\#]*$/, "health must be an HTTP path").optional(),
|
|
25
|
+
}).strict();
|
|
26
|
+
|
|
27
|
+
export const harnessSchema = z.object({
|
|
28
|
+
name,
|
|
29
|
+
summary: text,
|
|
30
|
+
detect: z.array(detect),
|
|
31
|
+
after: z.array(name).default([]),
|
|
32
|
+
image: z.union([
|
|
33
|
+
z.object({ builtin: name }).strict(),
|
|
34
|
+
z.object({ apt: z.array(text), run: z.array(text) }).strict(),
|
|
35
|
+
]),
|
|
36
|
+
verification: z.record(text, z.union([text, z.array(text).min(1)])).default({}),
|
|
37
|
+
artifacts: z.array(relativePath).default([]),
|
|
38
|
+
requires: z.object({
|
|
39
|
+
memoryMb: z.number().int().positive().optional(),
|
|
40
|
+
shmMb: z.number().int().positive().optional(),
|
|
41
|
+
kvm: z.boolean().optional(),
|
|
42
|
+
}).strict().default({}),
|
|
43
|
+
network: z.enum(["none", "services", "allowlist"]).default("none"),
|
|
44
|
+
services: z.array(service).optional(),
|
|
45
|
+
env: env.optional(),
|
|
46
|
+
suggestedRole: z.object({ name, description: text }).strict().optional(),
|
|
47
|
+
docs: relativePath.default("README.md"),
|
|
48
|
+
}).strict().superRefine((spec, ctx) => {
|
|
49
|
+
if (new Set(spec.services?.map((service) => service.name)).size !== (spec.services?.length ?? 0)) {
|
|
50
|
+
ctx.addIssue({ code: z.ZodIssueCode.custom, path: ["services"], message: "service names must be unique" });
|
|
51
|
+
}
|
|
52
|
+
if (spec.services !== undefined && spec.network === "none") {
|
|
53
|
+
ctx.addIssue({ code: z.ZodIssueCode.custom, path: ["services"], message: "services requires network services or allowlist" });
|
|
54
|
+
}
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
export function harnessSchemaFor(folderName) {
|
|
58
|
+
return harnessSchema.refine((spec) => spec.name === folderName, {
|
|
59
|
+
path: ["name"], message: "name must equal its folder name",
|
|
60
|
+
});
|
|
61
|
+
}
|