nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +86 -480
  2. package/bin/nomarmy.mjs +1081 -185
  3. package/docker/Dockerfile +2 -2
  4. package/docker/Dockerfile.go +6 -4
  5. package/docker/Dockerfile.rust +17 -2
  6. package/harnesses/_template/README.md +27 -0
  7. package/harnesses/_template/harness.yml +26 -0
  8. package/harnesses/browser-playwright/README.md +35 -0
  9. package/harnesses/browser-playwright/fixture/package.json +1 -0
  10. package/harnesses/browser-playwright/fixture/page.html +1 -0
  11. package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
  12. package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
  13. package/harnesses/browser-playwright/harness.yml +18 -0
  14. package/harnesses/go/README.md +45 -0
  15. package/harnesses/go/harness.yml +14 -0
  16. package/harnesses/mock-oidc/README.md +31 -0
  17. package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
  18. package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
  19. package/harnesses/mock-oidc/harness.yml +19 -0
  20. package/harnesses/node/README.md +53 -0
  21. package/harnesses/node/harness.yml +18 -0
  22. package/harnesses/python/README.md +46 -0
  23. package/harnesses/python/harness.yml +16 -0
  24. package/harnesses/rust/README.md +45 -0
  25. package/harnesses/rust/harness.yml +13 -0
  26. package/install.sh +29 -9
  27. package/lib/admission.mjs +178 -30
  28. package/lib/agents.mjs +8 -6
  29. package/lib/army.mjs +25 -10
  30. package/lib/codex-link.mjs +37 -0
  31. package/lib/config.mjs +15 -0
  32. package/lib/connect.mjs +232 -19
  33. package/lib/continue-from.mjs +103 -0
  34. package/lib/coordinator-instructions.mjs +5 -1
  35. package/lib/diff-checks.mjs +114 -0
  36. package/lib/dispatch-schema.mjs +14 -12
  37. package/lib/doctor.mjs +98 -9
  38. package/lib/egress-proxy.mjs +116 -0
  39. package/lib/execute.mjs +241 -33
  40. package/lib/git-record.mjs +27 -3
  41. package/lib/harness-schema.mjs +61 -0
  42. package/lib/harnesses.mjs +99 -0
  43. package/lib/health.mjs +162 -18
  44. package/lib/install-freshness.mjs +114 -0
  45. package/lib/jev-checks.mjs +110 -0
  46. package/lib/job-format.mjs +54 -0
  47. package/lib/judge.mjs +130 -0
  48. package/lib/limits.mjs +77 -0
  49. package/lib/model-probe.mjs +61 -0
  50. package/lib/mutation.mjs +159 -0
  51. package/lib/notify.mjs +30 -3
  52. package/lib/openclaw-install.mjs +122 -0
  53. package/lib/openclaw-path.mjs +28 -0
  54. package/lib/openclaw-run.mjs +74 -12
  55. package/lib/openclaw-runtime-health.mjs +56 -0
  56. package/lib/outcome.mjs +21 -2
  57. package/lib/outcomes.mjs +6 -0
  58. package/lib/path-utils.mjs +4 -0
  59. package/lib/podman-health.mjs +41 -0
  60. package/lib/process.mjs +4 -1
  61. package/lib/propose.mjs +10 -11
  62. package/lib/refusal-retry.mjs +16 -0
  63. package/lib/registry-python.mjs +98 -0
  64. package/lib/registry-secrets.mjs +140 -0
  65. package/lib/repo-query.mjs +13 -7
  66. package/lib/runs.mjs +7 -1
  67. package/lib/same-path.mjs +14 -0
  68. package/lib/sandbox-images.mjs +499 -83
  69. package/lib/sandbox-vm.mjs +32 -0
  70. package/lib/scan.mjs +5 -1
  71. package/lib/schema.mjs +20 -11
  72. package/lib/scout.mjs +21 -3
  73. package/lib/server-context.mjs +21 -1
  74. package/lib/setup-steps.mjs +55 -0
  75. package/lib/share.mjs +82 -0
  76. package/lib/stale-sessions.mjs +60 -0
  77. package/lib/stats.mjs +315 -0
  78. package/lib/statusline.mjs +32 -6
  79. package/lib/subscription-setup.mjs +13 -0
  80. package/lib/suggestions.mjs +153 -0
  81. package/lib/thinking.mjs +23 -0
  82. package/lib/transcript.mjs +30 -5
  83. package/lib/usage-limits.mjs +329 -0
  84. package/lib/user-config.mjs +106 -0
  85. package/lib/validators.mjs +220 -0
  86. package/lib/verification-artifacts.mjs +46 -0
  87. package/lib/verification-flow.mjs +52 -7
  88. package/lib/verification-network.mjs +66 -0
  89. package/lib/verify.mjs +338 -85
  90. package/lib/worker-prompt.mjs +5 -2
  91. package/lib/wsl-cli.mjs +152 -0
  92. package/lib/wsl.mjs +230 -0
  93. package/lib/zod-issues.mjs +15 -0
  94. package/mcp/server.mjs +165 -34
  95. package/package.json +7 -5
  96. package/playbooks/feature.md +8 -5
  97. package/scripts/configure-openclaw.sh +4 -2
  98. package/scripts/generate-harness-docs.mjs +42 -0
  99. package/scripts/install-openclaw.mjs +23 -0
  100. package/scripts/lib.sh +9 -2
  101. package/scripts/select-model.mjs +12 -5
  102. package/scripts/start-inference.sh +2 -2
package/lib/execute.mjs CHANGED
@@ -1,13 +1,18 @@
1
- import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery } from "./openclaw-run.mjs";
1
+ import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery, readStopRequest } from "./openclaw-run.mjs";
2
2
  import fs from "node:fs";
3
3
  import path from "node:path";
4
4
  import { fileURLToPath } from "node:url";
5
+ import { vmRestartIssue } from "./podman-health.mjs";
5
6
  import { measureReads } from "./job-budgets.mjs";
6
7
  import { coordinatorCommitMessage, worktreePointerState } from "./git-record.mjs";
7
8
  import { parseScoutReport, verifyCitations, resolveScoutOutcome, renderScoutReport, isScoutReportUnusable, scoutReportRecoveryPrompt } from "./scout.mjs";
8
9
  import { parseDecomposeReport, buildDecomposeFindings, resolveDecomposeOutcome, checkDecompositionOverlap, renderDecomposeReport } from "./decompose.mjs";
9
10
  import { deriveTimeBudget } from "./budget.mjs";
10
- import { estimateDisplacement } from "./transcript.mjs";
11
+ import { continuationProblem, continuationBase, snapshotRetainedWork, applyRetainedWork, continuationNote } from "./continue-from.mjs";
12
+ import { checkScoutCitations, checkReportClaims } from "./jev-checks.mjs";
13
+ import { runJudge } from "./judge.mjs";
14
+ import { pickMutants, runMutants, describeSurvivors } from "./mutation.mjs";
15
+ import { estimateDisplacement, readOpenClawTranscript } from "./transcript.mjs";
11
16
  import { outlineFile, findReferences } from "./repo-query.mjs";
12
17
  import { loadConfig } from "./config.mjs";
13
18
  import { linkNodePackages, nodeModulesState, repairHostInstalls } from "./sandbox-images.mjs";
@@ -15,8 +20,8 @@ import { detectTestSabotage, addedLinesOf, loadDependencyNames } from "./sabotag
15
20
  import { describeRecoveryChanges, reportRecoveryPrompt } from "./worker-prompt.mjs";
16
21
  import { parseWorkerReport } from "./report.mjs";
17
22
  import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "./outcomes.mjs";
18
- import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy } from "./outcome.mjs";
19
- import { isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets } from "./diff-checks.mjs";
23
+ import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy, HIGH_STAKES_NOTE } from "./outcome.mjs";
24
+ import { parseAddedLineNumbers, isTestPath, planRegressionProductionFiles, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
20
25
 
21
26
  // ---------------------------------------------------------------------------
22
27
  // Job status for polling. `status.json` is written at every phase transition
@@ -34,13 +39,23 @@ export function createExecutor(deps) {
34
39
  resolveReasoningApplied, recordedBudgets, runOpenClaw, sweepStaleSandboxContainers,
35
40
  verificationFlow, normalizeVerification, runIndependentVerification,
36
41
  runRegressionCheck, repoPolicy } = deps;
42
+ // Optional Jev checks (lib/validators.mjs): the server wires the settings; without them (tests), none run.
43
+ const jevSettingsFor = (check) => { try { const s = deps.jevSettings?.(); return s?.checks?.includes(check) ? s : null; } catch { return null; } };
37
44
 
38
- async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, jobId: presetJobId = null }) {
45
+ async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, stakes = null, reviews = null, jobId: presetJobId = null }) {
39
46
  await assertRepo();
47
+ // A continuation starts from the retained job's own base commit.
48
+ let continuation = null;
49
+ if (continueFrom) {
50
+ const checked = continuationProblem({ continueFrom, mode, baseRef, jobsRoot, projectDir });
51
+ if (checked.problem) throw new Error(checked.problem);
52
+ continuation = { jobId: continueFrom, record: checked.record };
53
+ baseRef = continuationBase(checked.record);
54
+ }
40
55
  ensureJobsRoot();
41
56
  // Fire-and-forget: sweeps whatever this or any other nomArmy install left
42
57
  // behind, without adding container-CLI round-trip latency to this job's own start.
43
- sweepStaleSandboxContainers().catch(() => {});
58
+ if (mode !== "verify") sweepStaleSandboxContainers().catch(() => {});
44
59
  const jobStartedMs = Date.now();
45
60
  const base = await resolveBase(baseRef), jobId = presetJobId || slug(workerId || (mode === "scout" ? "scout" : mode === "decompose" ? "decompose" : "worker")), jobDir = path.join(jobsRoot, jobId), runtimeDir = path.join(jobDir, "runtime");
46
61
  fs.mkdirSync(runtimeDir, { recursive: true });
@@ -48,20 +63,65 @@ export function createExecutor(deps) {
48
63
  jobId, workerId: workerId || jobId, mode, phase, state: phase === "finished" ? "finished" : "running",
49
64
  serverPid: process.pid, baseSha: base.sha, timeoutSeconds, ...extra
50
65
  });
51
- progress("starting", { startedAt: new Date().toISOString(), agent: pool ?? subscriptionWorker ?? "local", model: model ?? null });
66
+ progress("starting", { startedAt: new Date().toISOString(), agent: mode === "verify" ? null : pool ?? subscriptionWorker ?? "local", model: model ?? null });
52
67
  const common = { task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, workerId, progress, jobStartedMs };
53
- if (mode === "scout") return executeScout(common);
68
+ if (mode === "verify") return executeVerify({ ...common, verification });
69
+ if (mode === "scout") return executeScout({ ...common, reviews });
54
70
  if (mode === "decompose") return executeDecompose(common);
55
- return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor });
71
+ return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor, continuation, stakes });
72
+ }
73
+
74
+ async function executeVerify({ task, verification: profile, base, jobId, jobDir, workerId, progress, jobStartedMs }) {
75
+ const worktree = path.join(jobDir, "worktree");
76
+ let verification, record = null, error = null;
77
+ try {
78
+ progress("worktree");
79
+ await run("git", ["worktree", "add", "--detach", worktree, base.sha], { cwd: projectDir });
80
+ record = await collectGitRecord({ cwd: worktree, baseSha: base.sha, baseRef: base.ref, branch: null, jobId });
81
+ progress("verification");
82
+ verification = await runIndependentVerification({ profile, cwd: worktree, jobId, baseSha: base.sha, branch: null, mode: "verify", record, logFile: path.join(jobDir, "verification.log") });
83
+ } catch (err) {
84
+ error = err.message;
85
+ verification = normalizeVerification({ status: "not_run", reason: error }, profile);
86
+ } finally {
87
+ try { await run("git", ["worktree", "remove", "--force", worktree], { cwd: projectDir }); }
88
+ catch (err) { error = [error, err.message].filter(Boolean).join("\n"); }
89
+ }
90
+ const retained = fs.existsSync(worktree);
91
+ const outcome = retained ? OUTCOMES.VERIFICATION_NOT_RUN : {
92
+ pass: OUTCOMES.VERIFIED, fail: OUTCOMES.VERIFICATION_FAILED, not_run: OUTCOMES.VERIFICATION_NOT_RUN
93
+ }[verification.status];
94
+ const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode: "verify", task,
95
+ baseRef: base.ref, baseSha: base.sha, branch: null, worktree: retained ? worktree : null, worktreeRetained: retained,
96
+ startedAt: new Date(jobStartedMs).toISOString(), finishedAt: new Date().toISOString(),
97
+ outcome, coordinatorStatus: COORDINATOR_STATUS_BY_OUTCOME[outcome], verification, git: record, error,
98
+ // Network access (step 9) and skipped registry credentials (step 8) both surface as issues.
99
+ issues: [...new Set([...(verification.network ? verification.issues ?? [] : []), ...(record?.issues ?? [])])],
100
+ metrics: { total_elapsed: Date.now() - jobStartedMs, worker_cost_usd: 0, worker_tokens_total: 0, model_calls: 0 } };
101
+ fs.writeFileSync(path.join(jobDir, "metadata.json"), JSON.stringify(manifest, null, 2));
102
+ progress("finished", { outcome, coordinatorStatus: manifest.coordinatorStatus });
103
+ return { ok: outcome === OUTCOMES.VERIFIED, manifest, jobDir, report: "" };
56
104
  }
57
105
 
58
- async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, progress, jobStartedMs }) {
106
+ async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, continuation = null, stakes = null, progress, jobStartedMs }) {
59
107
  const mode = "implement";
60
108
  let branch = `agent/${jobId}`, worktree = path.join(jobDir, "worktree");
61
109
  try {
62
110
  progress("worktree");
63
111
  await run("git", ["worktree", "add", "-b", branch, worktree, base.sha], { cwd: projectDir });
64
112
  const cwd = worktree;
113
+ // continue_from: lay the retained job's unfinished work into this
114
+ // worktree as uncommitted changes, so this job's diff (and so its
115
+ // verification and revert check) covers that work too.
116
+ let continuedFrom = null;
117
+ if (continuation) {
118
+ const git = async (args, opts = {}) => (await run("git", args, { trim: false, ...opts })).stdout;
119
+ const snapshot = await snapshotRetainedWork({ worktree: continuation.record.worktree, baseSha: base.sha, jobId: continuation.jobId, git });
120
+ await applyRetainedWork({ worktree, baseSha: base.sha, commit: snapshot.commit, git });
121
+ continuedFrom = { jobId: continuation.jobId, snapshot: snapshot.commit, files: snapshot.files };
122
+ evidence = [continuationNote({ continueFrom: continuation.jobId, record: continuation.record, files: snapshot.files }), evidence].filter(Boolean).join("\n\n");
123
+ fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} continuing job ${continuation.jobId}: ${snapshot.files.length} file(s) of its unfinished work carried into this worktree (snapshot ${snapshot.commit})\n`);
124
+ }
65
125
  // Each npm package below the root reaches its install in the sandbox image.
66
126
  const nodeConfig = (() => { try { return loadConfig(projectDir)?.config ?? null; } catch { return null; } })();
67
127
  try { linkNodePackages(worktree, nodeConfig); } catch { /* verification reports what's missing */ }
@@ -79,6 +139,7 @@ export function createExecutor(deps) {
79
139
  const timeBudget = deriveTimeBudget({ timeoutSeconds });
80
140
  let result = null, attempted = null, workerFailed = false, workerTimedOut = false, workerStopReason = null, workerError = null;
81
141
  const workerStartedMs = Date.now();
142
+ const vmStartedBefore = deps.podmanVmStartedAt?.() ?? null;
82
143
  progress("worker");
83
144
  try {
84
145
  result = await runOpenClaw({
@@ -177,7 +238,7 @@ export function createExecutor(deps) {
177
238
  const recoveryResult = await runOpenClaw({
178
239
  task, acceptance, verification, mode, cwd, baseRef: base.ref, baseSha: base.sha,
179
240
  timeoutSeconds: timeBudget.reportReserveSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId,
180
- overridePrompt: reportRecoveryPrompt({ report: budgetState.budgets.report.implement, changes }), logSuffix: "-recovery",
241
+ overridePrompt: reportRecoveryPrompt({ report: budgetState.budgets.report.implement, changes, task }), logSuffix: "-recovery",
181
242
  });
182
243
  const recoveryText = finalText(recoveryResult);
183
244
  const recoveryValidation = parseWorkerReport(recoveryText);
@@ -214,23 +275,26 @@ export function createExecutor(deps) {
214
275
  // failed job that changed nothing was recorded "pass" (a Senti run),
215
276
  // which reads as evidence about work that never happened.
216
277
  independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the worker changed nothing, so there was none of its work to verify" }, verification ?? null);
278
+ } else if (workerStopReason === "stopped") {
279
+ // Stopped on request: end now, without spending time on tests.
280
+ independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the job was stopped on request" }, verification ?? null);
217
281
  } else if (verificationFlow.verificationRunner || !reportValidation.valid) {
218
- independentVerification = await runIndependentVerification({ profile: verification ?? null, cwd, jobId, baseSha: base.sha, branch, mode, record: preCommit });
282
+ independentVerification = await runIndependentVerification({ profile: verification ?? null, cwd, jobId, baseSha: base.sha, branch, mode, record: preCommit, logFile: path.join(jobDir, "verification.log") });
219
283
  }
220
284
 
221
285
  // verify_regression: on by default whenever there's a verification
222
286
  // profile (resolveVerifyRegression). It doubles verification wall-clock,
223
287
  // so it runs only when there's something to re-check: a passing
224
288
  // first-pass verification on a diff that touched production code.
225
- // Documentation isn't code a test can prove, so it's neither reverted
226
- // nor a reason to run the check (isDocumentationPath).
227
- const codeFilesChanged = preCommit.testChanges.production_files_changed.filter((f) => !isDocumentationPath(f));
289
+ // Documentation and CI configuration cannot be proven by local tests,
290
+ // so neither is reverted or used as a reason to run this check.
291
+ const codeFilesChanged = planRegressionProductionFiles(preCommit.testChanges.production_files_changed);
228
292
  let regressionCheck = null, regressionCheckFatal = false, regressionCheckElapsedMs = null;
229
293
  if (verifyRegression && independentVerification.status === "pass" && codeFilesChanged.length > 0) {
230
294
  const regressionStartedMs = Date.now();
231
295
  try {
232
296
  regressionCheck = await runRegressionCheck({
233
- cwd, jobId, productionFiles: codeFilesChanged,
297
+ cwd, jobId, productionFiles: codeFilesChanged, verificationResult: independentVerification,
234
298
  nameStatus: preCommit.nameStatus, profile: verification, baseSha: base.sha, branch, mode,
235
299
  });
236
300
  } catch (error) {
@@ -244,6 +308,36 @@ export function createExecutor(deps) {
244
308
  if (regressionCheck.status === "restore_failed") regressionCheckFatal = true;
245
309
  }
246
310
 
311
+ // Mutation testing (lib/mutation.mjs), when this repo opts in: small
312
+ // mistakes planted one at a time in the changed lines must each fail
313
+ // the same profile. Survivors raise review; a failed restore is fatal.
314
+ let mutation = null, mutationElapsedMs = null;
315
+ const mutationConfig = (() => { try { return loadConfig(projectDir)?.config?.mutation ?? null; } catch { return null; } })();
316
+ if (mutationConfig && verification && independentVerification.status === "pass" && codeFilesChanged.length > 0 && !regressionCheckFatal) {
317
+ const mutationStartedMs = Date.now();
318
+ try {
319
+ const untracked = new Set((preCommit.nameStatus ?? []).filter((e) => e.untracked).map((e) => e.path));
320
+ const deleted = new Set((preCommit.nameStatus ?? []).filter((e) => /^D/.test(e.status)).map((e) => e.path));
321
+ const files = [];
322
+ for (const file of codeFilesChanged.filter((f) => !deleted.has(f))) {
323
+ const full = path.join(cwd, file);
324
+ let lines;
325
+ if (untracked.has(file)) { try { lines = fs.readFileSync(full, "utf8").split("\n").map((_, i) => i + 1); } catch { continue; } }
326
+ else lines = parseAddedLineNumbers(await gitRaw(["diff", "-U0", base.sha, "--", file], cwd));
327
+ if (lines.length) files.push({ path: file, full, lines });
328
+ }
329
+ const mutants = pickMutants(files, mutationConfig.mutants);
330
+ let n = 0;
331
+ mutation = await runMutants({ mutants, deadlineMs: mutationStartedMs + mutationConfig.max_seconds * 1000,
332
+ verify: () => runIndependentVerification({ profile: verification, cwd, jobId: `${jobId}-mutant-${++n}`, baseSha: base.sha, branch, mode, record: preCommit }) });
333
+ mutation.planned = mutants.length;
334
+ } catch (error) {
335
+ mutation = { status: "not_run", killed: 0, survived: [], inconclusive: 0, tried: 0, reason: `mutation testing failed to run: ${error.message}` };
336
+ }
337
+ mutationElapsedMs = Date.now() - mutationStartedMs;
338
+ fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} mutation testing: ${mutation.killed ?? 0} killed, ${mutation.survived?.length ?? 0} survived, ${mutation.inconclusive ?? 0} inconclusive of ${mutation.tried ?? 0} tried (${Math.round(mutationElapsedMs / 1000)}s)\n`);
339
+ }
340
+
247
341
  // resolveOutcome's own contract only ever sees pass/fail/not_run for
248
342
  // regressionCheck -- a restore_failed status is substituted to not_run
249
343
  // here so resolveOutcome never needs a fourth value; the hard override
@@ -264,12 +358,19 @@ export function createExecutor(deps) {
264
358
  // Cheap, always-on, additive: never changes commitAllowed/commitBlockedReason
265
359
  // on its own (unlike the regression-check override above), only flags for
266
360
  // review -- see detectScopedTestSelectionRisk's own doc comment for why.
267
- let selectionRisk = null;
361
+ let selectionRisk = null, verificationInputs = null;
268
362
  if (mode === "implement" && verification) {
269
363
  try {
270
364
  const loaded = loadConfig(projectDir); // the operator's contract; see registerVerificationRunner's call
271
365
  const profileCommands = loaded.found ? (loaded.config?.verification?.[verification]?.commands ?? []) : [];
272
366
  selectionRisk = detectScopedTestSelectionRisk({ commands: profileCommands, testChanges: preCommit.testChanges });
367
+ // A diff that changes what those commands run (see
368
+ // detectVerificationInputChanges): applied below, after the others.
369
+ verificationInputs = await detectVerificationInputChanges({
370
+ commands: profileCommands, changedFiles: preCommit.changedFiles ?? [],
371
+ readBase: (file) => gitRaw(["show", `${base.sha}:${file}`], cwd).catch(() => null),
372
+ readHead: (file) => { try { return fs.readFileSync(path.join(cwd, file), "utf8"); } catch { return null; } },
373
+ });
273
374
  } catch { /* a config load failure here is the verification runner's own problem to report, not this check's */ }
274
375
  }
275
376
  const afterSelectionRisk = selectionRisk
@@ -364,18 +465,90 @@ export function createExecutor(deps) {
364
465
  commitBlockedReason: `possible secret detected: ${possibleSecrets.reason}`,
365
466
  reasons: [...afterHostInstalls.reasons, `POSSIBLE SECRET DETECTED: ${possibleSecrets.reason}`] }
366
467
  : afterHostInstalls;
367
- const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterSecrets, independentVerification.status, repoPolicy()),
468
+ // Jev: does the worker's report match its diff? Found live: a note said
469
+ // "restored check.js to base commit" while the diff rewrote check.js.
470
+ // Only raises review; it never blocks or allows a commit.
471
+ let jevClaims = null, judged = null;
472
+ const jevImplement = jevSettingsFor("report-claims");
473
+ const judge = (() => { try { return deps.judgeSettings?.() ?? null; } catch { return null; } })();
474
+ // The whole change, new files included, for whichever validators run.
475
+ const jobDiff = async () => {
476
+ let diff = await gitRaw(["diff", base.sha, "--"], cwd);
477
+ for (const entry of (preCommit.nameStatus ?? []).filter((e) => e.untracked)) {
478
+ let text = "";
479
+ try { text = fs.readFileSync(path.join(cwd, entry.path), "utf8"); } catch { continue; }
480
+ diff += `\ndiff --git a/${entry.path} b/${entry.path}\nnew file\n--- /dev/null\n+++ b/${entry.path}\n${text.split("\n").map((l) => `+${l}`).join("\n")}\n`;
481
+ }
482
+ return diff;
483
+ };
484
+ let diffText = null;
485
+ if (mode === "implement" && reportValidation?.valid && repositoryChanged && (jevImplement || (judge && !judge.problem))) {
486
+ try { diffText = await jobDiff(); } catch { diffText = null; }
487
+ }
488
+ if (jevImplement && diffText != null) {
489
+ try { jevClaims = await checkReportClaims({ report: reportValidation, diff: diffText, settings: jevImplement }); }
490
+ catch (error) { jevClaims = { flag: null, verdict: null, error: error.message, usage: 0 }; }
491
+ }
492
+ // The model judge (lib/judge.mjs): acceptance criteria, the report and
493
+ // changed tests. Only raises review, never blocks or allows a commit.
494
+ if (judge?.problem) judged = { flags: [], answer: null, error: `judge not run: ${judge.problem}`, skipped: true };
495
+ else if (judge && diffText != null) {
496
+ const modifiedTests = preCommit.testChanges?.existing_tests_modified ?? [];
497
+ let testDiff = "";
498
+ if (modifiedTests.length) { try { testDiff = await gitRaw(["diff", base.sha, "--", ...modifiedTests], cwd); } catch { testDiff = ""; } }
499
+ judged = await runJudge({ settings: judge, task, acceptance: acceptance ?? [], report: reportValidation, diff: diffText, testDiff, stateRoot: path.join(jobsRoot, "..") });
500
+ }
501
+ const afterJevOnly = jevClaims?.flag
502
+ ? { ...afterSecrets, reviewRequired: true, reasons: [...afterSecrets.reasons, `REPORT MAY NOT MATCH THE DIFF (Jev, ${jevClaims.flag.probability.toFixed(2)}): the report says "${String(reportValidation.note ?? "").slice(0, 200)}", and the diff may show otherwise. Read the diff before accepting.`] }
503
+ : afterSecrets;
504
+ const afterJev = judged?.flags?.length
505
+ ? { ...afterJevOnly, reviewRequired: true, reasons: [...afterJevOnly.reasons, `JUDGE (${judge.agent}/${judge.model}): ${judged.flags.join("; ")}. Read the diff before accepting.`] }
506
+ : afterJevOnly;
507
+
508
+ // A worker must not be judged by a check it rewrote: a changed script,
509
+ // Makefile or package.json script that a verification command runs
510
+ // blocks the commit; changed test-runner config only asks for review.
511
+ const blockedInputs = verificationInputs?.blocked ?? [], flaggedInputs = verificationInputs?.flagged ?? [];
512
+ const inputLine = blockedInputs.map((b) => `${b.file} (${b.why})`).join("; ");
513
+ const afterInputs = blockedInputs.length
514
+ ? { ...afterJev, outcome: afterJev.commitAllowed || afterJev.outcome === OUTCOMES.WORKER_DONE ? OUTCOMES.NEEDS_REVIEW : afterJev.outcome,
515
+ reviewRequired: true, commitAllowed: false,
516
+ commitBlockedReason: afterJev.commitAllowed ? `the diff changes what verification runs: ${inputLine}` : afterJev.commitBlockedReason,
517
+ reasons: [...afterJev.reasons, `VERIFICATION INPUT CHANGED: the diff changes what profile '${verification}' runs, so its result can't be trusted: ${inputLine}`] }
518
+ : afterJev;
519
+ const afterConfig = flaggedInputs.length
520
+ ? { ...afterInputs, reviewRequired: true, reasons: [...afterInputs.reasons, `TEST CONFIG CHANGED: ${flaggedInputs.map((c) => `${c.file} (${c.why})`).join("; ")}`] }
521
+ : afterInputs;
522
+ const afterMutation = mutation?.status === "restore_failed"
523
+ ? { ...afterConfig, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true, commitAllowed: false,
524
+ commitBlockedReason: `mutation testing could not restore the worker's file: ${mutation.reason}`, reasons: [...afterConfig.reasons, `MUTATION RESTORE FAILED: ${mutation.reason}`] }
525
+ : mutation?.status === "survivors"
526
+ ? { ...afterConfig, reviewRequired: true, reasons: [...afterConfig.reasons, describeSurvivors(mutation, verification)] }
527
+ : afterConfig;
528
+ const afterStakes = stakes === "high" ? { ...afterMutation, reviewRequired: true, reasons: [...afterMutation.reasons, HIGH_STAKES_NOTE] } : afterMutation;
529
+ const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterStakes, independentVerification.status, repoPolicy()),
368
530
  { refactor, verificationStatus: independentVerification.status, testChanges: preCommit.testChanges });
369
531
 
370
532
  progress("commit");
371
533
  const commit = await createCoordinatorCommit({ cwd, jobId, outcome: finalOutcome,
372
- message: coordinatorCommitMessage({ task, subject: commitSubject, note: reportValidation?.note ?? null, jobId, workerId, recovered: Boolean(finalOutcome.recovered), provider: (result ?? attempted)?.provider ?? null, model: (result ?? attempted)?.model ?? null }) });
534
+ message: coordinatorCommitMessage({ task: continuation?.record?.objective ?? task, subject: commitSubject, note: reportValidation?.note ?? null, jobId, workerId, recovered: Boolean(finalOutcome.recovered), provider: (result ?? attempted)?.provider ?? null, model: (result ?? attempted)?.model ?? null, continuedFrom: continuedFrom?.jobId ?? null }) });
373
535
  progress("record");
374
536
  const record = await collectGitRecord({ cwd, baseSha: base.sha, branch, baseRef: base.ref, jobId }), worker = workerMetadata(result ?? attempted);
375
537
 
376
538
  let coordinatorStatus = COORDINATOR_STATUS_BY_OUTCOME[finalOutcome.outcome] ?? "incomplete";
377
- const issues = [...finalOutcome.reasons];
378
- if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
539
+ const issues = [...finalOutcome.reasons, ...(preCommit.issues ?? [])];
540
+ if (workerStopReason === "stopped") {
541
+ const request = readStopRequest(jobDir);
542
+ issues.push(`stopped on request${request?.reason ? `: ${request.reason}` : ""}; the worktree is kept, so continue_from can pick the work up (on another model too)`);
543
+ } else if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
544
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
545
+ // is recovered, and it happens on most Codex jobs: kept on the record for
546
+ // stats, out of the issues the General reviews.
547
+ const runnerNotes = [];
548
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
549
+ if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
550
+ if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
551
+ if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
379
552
  if (repositoryChanged && !commit.created) {
380
553
  if (coordinatorStatus === "complete") coordinatorStatus = "incomplete";
381
554
  // A timed-out or crashed worker can still leave real, salvageable work
@@ -385,7 +558,7 @@ export function createExecutor(deps) {
385
558
  // the diffstat right in the issue a caller actually reads -- not just
386
559
  // buried in the full manifest's git record -- is what makes "go look at
387
560
  // the worktree" worth doing instead of discarding the job.
388
- issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
561
+ if (record.filesChanged > 0) issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
389
562
  }
390
563
  const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`worker recorded ${failures} tool failure(s)`);
391
564
  if (record.ignoredRuntimeJunk.length) issues.push(`runtime junk ignored: ${record.ignoredRuntimeJunk.join(", ")}`);
@@ -403,11 +576,13 @@ export function createExecutor(deps) {
403
576
 
404
577
  const metrics = buildMetrics({ result: result ?? attempted, record, reportValidation, outcome: finalOutcome, workerElapsedMs, totalElapsedMs: Date.now() - jobStartedMs, regressionCheckElapsedMs, transientAbortRetried });
405
578
  const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree, branch, startedAt, finishedAt,
406
- objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null,
579
+ objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}), ...(stakes ? { stakes } : {}),
580
+ ...(mutation ? { mutation: { ...mutation, elapsedSeconds: Math.round((mutationElapsedMs ?? 0) / 1000) } } : {}),
581
+ ...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, provider: judge?.provider ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
407
582
  outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
408
583
  reportRecoveryAttempted, reportRecovered,
409
584
  reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
410
- coordinatorStatus, issues, reportValidation, independentVerification,
585
+ coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], runnerNotes, reportValidation, independentVerification,
411
586
  // Original, unsubstituted regressionCheck (real "restore_failed" status
412
587
  // visible here even though resolveOutcome above only ever saw a
413
588
  // not_run-substituted view) -- full transparency for the caller.
@@ -446,7 +621,7 @@ export function createExecutor(deps) {
446
621
  // the worktree, so a scout that wrote to its snapshot cannot forge evidence.
447
622
  // A clean scout worktree holds no work and is removed; a dirty one is retained
448
623
  // because a scout that wrote is a scout that misbehaved, and that is worth a look.
449
- async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs }) {
624
+ async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs, reviews = null }) {
450
625
  const mode = "scout", worktree = path.join(jobDir, "worktree");
451
626
  let worktreeRetained = false;
452
627
  try {
@@ -469,7 +644,10 @@ export function createExecutor(deps) {
469
644
  const workerStartedMs = Date.now();
470
645
  progress("worker");
471
646
  try {
472
- result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
647
+ // The run gets the work share; the reserve stays for report recovery
648
+ // below. Scouts had none, so a timed-out scout left about 0 seconds
649
+ // and its findings were never recovered.
650
+ result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds: deriveTimeBudget({ timeoutSeconds }).workTimeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
473
651
  } catch (error) {
474
652
  workerFailed = true;
475
653
  // error.timedOut is set only by our own spawn timer (run(), above) --
@@ -499,18 +677,26 @@ export function createExecutor(deps) {
499
677
 
500
678
  // See shouldAttemptScoutRecovery's own doc comment: this only fires when
501
679
  // the report is genuinely unusable, gated by whatever time is actually
502
- // left against the caller's original timeout (scout has no reserved
503
- // report-phase budget the way implement does).
680
+ // left against the caller's original timeout (the reserve held back
681
+ // from the run above).
504
682
  let reportRecoveryAttempted = false, reportRecovered = false;
505
- const remainingSeconds = timeoutSeconds - Math.round(workerElapsedMs / 1000);
683
+ // At least the reserve, as implement's recovery gets: nomArmy's own kill
684
+ // lands 30s after OpenClaw's timer, which would otherwise eat it.
685
+ const remainingSeconds = Math.max(deriveTimeBudget({ timeoutSeconds }).reportReserveSeconds, timeoutSeconds - Math.round(workerElapsedMs / 1000));
506
686
  if (shouldAttemptScoutRecovery({ workerFailed, workerTimedOut, report, remainingSeconds })) {
507
687
  reportRecoveryAttempted = true;
688
+ // What the first run left, in case the follow-up can't see its session.
689
+ let filesRead = [], earlierReply = reportText;
690
+ try {
691
+ const first = await readOpenClawTranscript(path.join(runtimeDir, "state"));
692
+ if (first.available) { filesRead = first.filesRead ?? []; earlierReply = earlierReply || first.lastAssistantText || ""; }
693
+ } catch { /* the reply alone still helps */ }
508
694
  try {
509
695
  const recoveryResult = await runOpenClaw({
510
696
  task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha,
511
697
  timeoutSeconds: remainingSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId,
512
698
  evidenceTool: evidencePlaced ? evidenceTool : null,
513
- overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout }), logSuffix: "-recovery",
699
+ overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout, question: task, acceptance, earlierReply, filesRead }), logSuffix: "-recovery",
514
700
  });
515
701
  const recoveryReport = parseScoutReport(finalText(recoveryResult), (recoveryResult?.budgetsUsed ?? used).scout);
516
702
  if (!isScoutReportUnusable(recoveryReport)) {
@@ -530,7 +716,16 @@ export function createExecutor(deps) {
530
716
  const dirty = record.repoStatusFiles.length > 0;
531
717
  const readFile = async p => { try { return await gitRaw(["show", `${base.sha}:${p}`], projectDir); } catch { return null; } };
532
718
  const verified = await verifyCitations(report.findings, { readFile, limits: used.scout });
719
+ // Jev: do the cited lines support each finding? Only adds flags.
720
+ let jevCitations = null;
721
+ const jevScout = jevSettingsFor("scout-citations");
722
+ if (jevScout && verified?.findings?.length) {
723
+ try { jevCitations = await checkScoutCitations({ findings: verified.findings, settings: jevScout, readFile }); }
724
+ catch (error) { jevCitations = { flags: [], checked: 0, errors: [error.message], usage: 0, verdicts: [] }; }
725
+ for (const f of jevCitations.flags) verified.findings[f.index].jev = { verdict: f.verdict, probability: f.probability };
726
+ }
533
727
  const outcome = resolveScoutOutcome({ report, verified, workerFailed, workerTimedOut, dirty });
728
+ if (jevCitations?.flags.length) outcome.reviewRequired = true;
534
729
 
535
730
  progress("record");
536
731
  if (outcome.retainWorktree) worktreeRetained = true;
@@ -539,8 +734,15 @@ export function createExecutor(deps) {
539
734
  const worker = workerMetadata(result ?? attempted);
540
735
  const issues = [...outcome.reasons];
541
736
  if (workerError) issues.push(`scout error: ${String(workerError).split("\n")[0]}`);
737
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
738
+ // is recovered, and it happens on most Codex jobs: kept on the record for
739
+ // stats, out of the issues the General reviews.
740
+ const runnerNotes = [];
741
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
542
742
  const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
543
743
  if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
744
+ if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
745
+ if (jevCitations?.errors.length) issues.push(`Jev citation check skipped or incomplete (${jevCitations.errors.join("; ")}); this job's result doesn't depend on it`);
544
746
  if (reportRecoveryAttempted) {
545
747
  issues.push(reportRecovered
546
748
  ? "scout report recovered via a follow-up call after the first reply was cut off"
@@ -577,8 +779,9 @@ export function createExecutor(deps) {
577
779
  displacement_verdict: displacement.verdict
578
780
  };
579
781
  const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
580
- objective: task, mustCover: acceptance ?? [],
581
- outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
782
+ objective: task, mustCover: acceptance ?? [], ...(reviews ? { reviews } : {}),
783
+ ...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
784
+ outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
582
785
  scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
583
786
  findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
584
787
  excerptLinesUsed: verified.excerptLinesUsed, excerptTruncated: verified.excerptTruncated,
@@ -670,6 +873,11 @@ export function createExecutor(deps) {
670
873
  const worker = workerMetadata(result ?? attempted);
671
874
  const issues = [...outcome.reasons];
672
875
  if (workerError) issues.push(`decompose error: ${String(workerError).split("\n")[0]}`);
876
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
877
+ // is recovered, and it happens on most Codex jobs: kept on the record for
878
+ // stats, out of the issues the General reviews.
879
+ const runnerNotes = [];
880
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the decomposer's report was recovered from the run's transcript`);
673
881
  const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`decomposer recorded ${failures} tool failure(s)`);
674
882
  if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
675
883
  if (overlaps.length) issues.push(`${overlaps.length} subtask pair(s) claim overlapping files; not safe to dispatch as independent jobs as proposed`);
@@ -699,7 +907,7 @@ export function createExecutor(deps) {
699
907
  };
700
908
  const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
701
909
  objective: task, constraints: acceptance ?? [],
702
- outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
910
+ outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
703
911
  decompose: { objective: report.objective, confidence: report.confidence, notSplittable: report.notSplittable,
704
912
  subtasks: report.subtasks.map((s, i) => ({ task: s.task, acceptance: s.acceptance, citations: verified.findings[i]?.citations ?? [], supported: verified.findings[i]?.supported ?? false, weak: verified.findings[i]?.weak ?? false })),
705
913
  overlaps, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
@@ -65,7 +65,7 @@ export function isRuntimeJunk(file) {
65
65
  * gave one, else the task's first sentence (the army role header and an
66
66
  * "OBJECTIVE:" label dropped). The job id stays, as a trailer.
67
67
  */
68
- export function coordinatorCommitMessage({ task = "", subject = null, note = null, jobId, workerId = null, recovered = false, provider = null, model = null }) {
68
+ export function coordinatorCommitMessage({ task = "", subject = null, note = null, jobId, workerId = null, recovered = false, provider = null, model = null, continuedFrom = null }) {
69
69
  const oneLine = (t) => String(t ?? "").replace(/\s+/g, " ").trim();
70
70
  let body = String(task ?? "");
71
71
  if (/^\[nomArmy role:/.test(body)) body = body.includes("\n\n") ? body.slice(body.indexOf("\n\n") + 2) : "";
@@ -78,6 +78,7 @@ export function coordinatorCommitMessage({ task = "", subject = null, note = nul
78
78
  const cleanNote = oneLine(note);
79
79
  if (cleanNote) lines.push("", ...wrapText(cleanNote, 72));
80
80
  lines.push("", `nomArmy-Job: ${jobId}`);
81
+ if (continuedFrom) lines.push(`nomArmy-Continues: ${continuedFrom}`);
81
82
  if (provider || model) lines.push(`nomArmy-Worker: ${[provider, model].filter(Boolean).join("/")}`);
82
83
  return lines.join("\n");
83
84
  }
@@ -138,9 +139,26 @@ export function createGitRecord({ run, git, gitRaw }) {
138
139
  // idleMinElapsedMs of the work phase has passed (an early snapshot mid-first-
139
140
  // edit looks identical to no edit at all). A worktree read failing mid-write
140
141
  // is expected, not an error; it just means "nothing to report this tick."
141
- function makeIdleDiffTick(cwd, { idleMs, minElapsedMs }) {
142
- let lastHash = null, lastChangeAtMs = 0, sawChange = false;
142
+ //
143
+ // An unchanged worktree isn't enough on its own: a frontier worker running
144
+ // a suite of thousands of tests, or reading before its next edit, changes
145
+ // no file for minutes. Three Senti jobs were cut off at 7 to 13 minutes
146
+ // while still working. With `activity` (the worker's transcript: event
147
+ // count and whether a tool call is still running), the breaker waits while
148
+ // the worker is active, and stops only when it has also gone quiet for
149
+ // idleMs, or when nothing has changed for ACTIVE_CAP times idleMs however
150
+ // busy it looks (a worker looping without progress).
151
+ function makeIdleDiffTick(cwd, { idleMs, minElapsedMs, activity = null }) {
152
+ const ACTIVE_CAP = 3;
153
+ let lastHash = null, lastChangeAtMs = 0, sawChange = false, lastEvents = null, lastActivityAtMs = 0, inFlight = false;
143
154
  return async elapsedMs => {
155
+ if (activity) {
156
+ try {
157
+ const a = await activity();
158
+ if (a && a.events !== lastEvents) { lastEvents = a.events; lastActivityAtMs = elapsedMs; }
159
+ inFlight = Boolean(a?.toolInFlight);
160
+ } catch { /* a failed read just means no activity signal this tick */ }
161
+ }
144
162
  let statusOut;
145
163
  try { statusOut = await gitRaw(["status", "--porcelain=v1", "-z", "--untracked-files=all"], cwd); }
146
164
  catch { return { stop: false }; }
@@ -181,6 +199,12 @@ export function createGitRecord({ run, git, gitRaw }) {
181
199
  if (!sawChange || elapsedMs < minElapsedMs) return { stop: false };
182
200
  const idleForMs = elapsedMs - lastChangeAtMs;
183
201
  if (idleForMs < idleMs) return { stop: false };
202
+ if (activity) {
203
+ const quietForMs = elapsedMs - lastActivityAtMs;
204
+ const active = inFlight || quietForMs < idleMs;
205
+ if (active && idleForMs < idleMs * ACTIVE_CAP) return { stop: false };
206
+ if (active) return { stop: true, reason: "idle_diff", detail: `worktree unchanged for ${Math.round(idleForMs / 1000)}s while the worker kept working (${ACTIVE_CAP}x the idle limit)` };
207
+ }
184
208
  return { stop: true, reason: "idle_diff", detail: `worktree unchanged for ${Math.round(idleForMs / 1000)}s` };
185
209
  };
186
210
  }
@@ -0,0 +1,61 @@
1
+ import { z } from "zod";
2
+
3
+ const text = z.string().trim().min(1, "must not be empty");
4
+ const name = z.string().regex(/^[a-z][a-z0-9]*(?:-[a-z0-9]+)*$/, "must be kebab-case");
5
+ const relativePath = text.refine((value) => !value.startsWith("/") && !value.includes("\\") && !value.includes(":") && !value.split("/").includes(".."), "must be a repository-relative path");
6
+ const detect = z.union([
7
+ z.object({ file: relativePath }).strict(),
8
+ z.object({ package: text }).strict(),
9
+ z.object({ lockfile: relativePath }).strict(),
10
+ ]);
11
+
12
+ const env = z.record(z.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/), z.string());
13
+ const pinnedImage = text.refine((value) => {
14
+ if (/\s/.test(value) || value.startsWith("-")) return false;
15
+ const reference = value.split("@")[0];
16
+ if (reference.split("/").at(-1).endsWith(":latest")) return false;
17
+ return /^[a-z0-9][a-z0-9._:/-]*@sha256:[a-f0-9]{64}$/.test(value)
18
+ || /^[a-z0-9][a-z0-9._:/-]*:[A-Za-z0-9_][A-Za-z0-9_.-]*$/.test(value)
19
+ && value.split("/").at(-1).includes(":");
20
+ }, "image must have an explicit non-latest tag or sha256 digest");
21
+ const service = z.object({
22
+ name, image: pinnedImage, port: z.number().int().min(1).max(65535),
23
+ env: env.optional(),
24
+ health: z.string().regex(/^\/(?!\/)[^\s\\#]*$/, "health must be an HTTP path").optional(),
25
+ }).strict();
26
+
27
+ export const harnessSchema = z.object({
28
+ name,
29
+ summary: text,
30
+ detect: z.array(detect),
31
+ after: z.array(name).default([]),
32
+ image: z.union([
33
+ z.object({ builtin: name }).strict(),
34
+ z.object({ apt: z.array(text), run: z.array(text) }).strict(),
35
+ ]),
36
+ verification: z.record(text, z.union([text, z.array(text).min(1)])).default({}),
37
+ artifacts: z.array(relativePath).default([]),
38
+ requires: z.object({
39
+ memoryMb: z.number().int().positive().optional(),
40
+ shmMb: z.number().int().positive().optional(),
41
+ kvm: z.boolean().optional(),
42
+ }).strict().default({}),
43
+ network: z.enum(["none", "services", "allowlist"]).default("none"),
44
+ services: z.array(service).optional(),
45
+ env: env.optional(),
46
+ suggestedRole: z.object({ name, description: text }).strict().optional(),
47
+ docs: relativePath.default("README.md"),
48
+ }).strict().superRefine((spec, ctx) => {
49
+ if (new Set(spec.services?.map((service) => service.name)).size !== (spec.services?.length ?? 0)) {
50
+ ctx.addIssue({ code: z.ZodIssueCode.custom, path: ["services"], message: "service names must be unique" });
51
+ }
52
+ if (spec.services !== undefined && spec.network === "none") {
53
+ ctx.addIssue({ code: z.ZodIssueCode.custom, path: ["services"], message: "services requires network services or allowlist" });
54
+ }
55
+ });
56
+
57
+ export function harnessSchemaFor(folderName) {
58
+ return harnessSchema.refine((spec) => spec.name === folderName, {
59
+ path: ["name"], message: "name must equal its folder name",
60
+ });
61
+ }