nomarmy 0.1.0-alpha.11 → 0.1.0-alpha.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -0
- package/bin/nomarmy.mjs +192 -6
- package/lib/admission.mjs +14 -1
- package/lib/coordinator-instructions.mjs +2 -2
- package/lib/execute.mjs +120 -11
- package/lib/git-record.mjs +25 -2
- package/lib/jev-checks.mjs +110 -0
- package/lib/judge.mjs +130 -0
- package/lib/mutation.mjs +159 -0
- package/lib/openclaw-run.mjs +31 -1
- package/lib/process.mjs +4 -1
- package/lib/schema.mjs +10 -0
- package/lib/scout.mjs +17 -3
- package/lib/stats.mjs +213 -0
- package/lib/transcript.mjs +3 -0
- package/lib/validators.mjs +179 -0
- package/mcp/server.mjs +30 -0
- package/package.json +1 -1
- package/playbooks/feature.md +1 -1
package/README.md
CHANGED
|
@@ -52,6 +52,8 @@ Failing verification stays failed, unconditionally. A malformed report isn't aut
|
|
|
52
52
|
|
|
53
53
|
**Checking without building** costs nothing: `mode: verify` runs a verification profile against any branch, with no worker and no model tokens.
|
|
54
54
|
|
|
55
|
+
**Want deeper checks?** Three optional [validators](https://github.com/rayson-tech/nomarmy/blob/main/docs/validators.md) go further, each only adding review flags: mutation testing (do the tests pin down the changed lines?), Jev (do a scout's citations support its findings, does a report match its diff?) and a model judge (acceptance criteria, weakened tests).
|
|
56
|
+
|
|
55
57
|
## Where the work runs
|
|
56
58
|
|
|
57
59
|
- **Agents** say where a job can run: an API key, your own ChatGPT or Muse Code subscription, or a local model on llama.cpp.
|
|
@@ -74,10 +76,12 @@ Developed and maintained by Rayson Technologies. This is an alpha (`0.1.0-alpha`
|
|
|
74
76
|
| [Agents and the army](https://github.com/rayson-tech/nomarmy/blob/main/docs/agents-and-army.md) | Where a job can run, who does what, usage limits, picking an agent |
|
|
75
77
|
| [`/feature` runs](https://github.com/rayson-tech/nomarmy/blob/main/docs/feature-runs.md) | A feature end to end, and watching what nomArmy is doing |
|
|
76
78
|
| [Your repository](https://github.com/rayson-tech/nomarmy/blob/main/docs/your-repo.md) | `.nomarmy.yml`, verification, dependencies, private registries, what nomArmy checks |
|
|
79
|
+
| [Validators](https://github.com/rayson-tech/nomarmy/blob/main/docs/validators.md) | Optional deeper checks: mutation testing, Jev, a model judge |
|
|
77
80
|
| [Harnesses](https://github.com/rayson-tech/nomarmy/blob/main/docs/harnesses.md) | Ecosystem registry, detection, network levels, and requirements |
|
|
78
81
|
| [Configuration](https://github.com/rayson-tech/nomarmy/blob/main/docs/configuration.md) | Settings, swapping the local model, sizing, admission |
|
|
79
82
|
| [Reference](https://github.com/rayson-tech/nomarmy/blob/main/docs/reference.md) | Every CLI command and MCP tool |
|
|
80
83
|
| [Security posture](https://github.com/rayson-tech/nomarmy/blob/main/docs/security.md) | What the sandbox holds back, and the one exception |
|
|
84
|
+
| [FAQ](https://github.com/rayson-tech/nomarmy/blob/main/docs/faq.md) | Which model for which role, switching models, usage limits |
|
|
81
85
|
| [Troubleshooting](https://github.com/rayson-tech/nomarmy/blob/main/docs/troubleshooting.md) | Symptoms and fixes |
|
|
82
86
|
|
|
83
87
|
## Security
|
|
@@ -97,6 +101,8 @@ A nom gets a writable git worktree inside a Podman sandbox and nothing else: no
|
|
|
97
101
|
| The army and `/feature` | Driven by a real Claude Code General across three runs, about 18 implement jobs |
|
|
98
102
|
| Harnesses: Go, Rust, Python, Node and mixed repos; Playwright; fake services | Live-verified offline |
|
|
99
103
|
| Private registries and a verification-only network allowlist | Live-verified; each passed an independent security review |
|
|
104
|
+
| Validators: mutation testing, Jev, a model judge | Unit and live tested; Jev and the judge evaluated on real job records |
|
|
105
|
+
| `nomarmy stats` | Checked against a hand-built report on real job records |
|
|
100
106
|
|
|
101
107
|
What we've learned from real runs, including where delegating pays and where it doesn't, is in [docs/findings.md](https://github.com/rayson-tech/nomarmy/blob/main/docs/findings.md).
|
|
102
108
|
|
package/bin/nomarmy.mjs
CHANGED
|
@@ -20,6 +20,10 @@ import { readGGUFMetadata, resolveModelPath, totalSplitBytes } from "../lib/gguf
|
|
|
20
20
|
import { recommend, customRecommendation, evaluateConfig, bytesPerKvElementForCacheTypes, MIN_CONTEXT_PER_NOM } from "../lib/sizing.mjs";
|
|
21
21
|
import { connectClaude, connectCodex, connectCursor, cursorAlreadyConnected, deriveWorkerModelEnv, defaultInstallDir, installMcpCopy, SCOPES, claudeUserScoped, portableServerLaunch } from "../lib/connect.mjs";
|
|
22
22
|
import { compareVersions, readPackageVersion, readInstallVersions, copyIsStale } from "../lib/install-freshness.mjs";
|
|
23
|
+
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo } from "../lib/stats.mjs";
|
|
24
|
+
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
25
|
+
import { loadValidators, saveJevKey, removeJev, jevSettings, askJev, validatorsPath, JEV_CHECKS, saveJudge, removeJudge, judgeSettings } from "../lib/validators.mjs";
|
|
26
|
+
import { probeModel } from "../lib/model-probe.mjs";
|
|
23
27
|
import { ID_RE, AUTH_ENV_NAME_RE, OPENCLAW_PROVIDER_ID_RE, openclawProviderId, isNativeProviderType } from "../lib/dispatch-schema.mjs";
|
|
24
28
|
import { loadAgents, readAgentsFile, writeAgentsFile, agentsConfigPath, apiAgentAsPoolEntry, describeAgent as describeAgentLabel, agentRunsToolsOnHost, agentProviderId, AGENT_KINDS, API_PROVIDER_TYPES, RESERVED_AGENT_NAMES, BUILTIN_LOCAL_AGENT } from "../lib/agents.mjs";
|
|
25
29
|
import { loadArmy, mergeArmy, describeArmy, readArmyFile, updateArmyInFile, assignRoleInFile, parseTargetSpec, armyLayerPath, globalConfigDir, DEFAULT_ARMY, ARMY_PHASES, LOCAL_CONFIG_FILENAME } from "../lib/army.mjs";
|
|
@@ -195,17 +199,25 @@ Usage: nomarmy <command> [options]
|
|
|
195
199
|
which agent the General is, in --global
|
|
196
200
|
(default) or --local
|
|
197
201
|
config paths where agents.yml and the three army layers live
|
|
198
|
-
jobs [--watch|--events|--prune|--wait <jobId>] [--interval N] [--older-than DAYS]
|
|
202
|
+
jobs [--watch|--events [--until-done]|--prune|--wait <jobId>|--stop <jobId> [--reason <text>]] [--interval N] [--older-than DAYS]
|
|
199
203
|
what's running across every session (agent, model, phase,
|
|
200
204
|
last tool call, files changed, heartbeat) and what just
|
|
201
205
|
finished; --watch redraws every N seconds (default 3);
|
|
202
206
|
--events prints one line per start, phase change and
|
|
203
|
-
finish (for
|
|
204
|
-
|
|
207
|
+
finish (--json for JSON lines). It's a stream: read it
|
|
208
|
+
with a monitor that wakes on each line. A background
|
|
209
|
+
command is only reported when it exits, so there use
|
|
210
|
+
--events --until-done, which exits once every job it saw
|
|
211
|
+
running has finished (or --wait for one job). The plain
|
|
212
|
+
stream ends on its own after 30 minutes with nothing
|
|
213
|
+
running (--idle-minutes N); --prune removes the bulky runtime data
|
|
205
214
|
from finished jobs older than DAYS (default 2), keeping
|
|
206
215
|
their records, reports and any retained worktree;
|
|
207
216
|
--wait <jobId> [--timeout <seconds>] blocks for one job
|
|
208
|
-
to finish (default timeout 1800; --json is supported)
|
|
217
|
+
to finish (default timeout 1800; --json is supported);
|
|
218
|
+
--stop <jobId> stops a running job's worker (no report
|
|
219
|
+
recovery, no verification), keeping its worktree for
|
|
220
|
+
continue_from
|
|
209
221
|
health check what's likely to break a run before it does:
|
|
210
222
|
expiring logins, an outdated OpenClaw or plugin, roles
|
|
211
223
|
that can't be dispatched, an unloadable agents.yml,
|
|
@@ -222,6 +234,24 @@ Usage: nomarmy <command> [options]
|
|
|
222
234
|
project for this repository, committed for the team
|
|
223
235
|
(.mcp.json or .cursor/mcp.json, running \`nomarmy mcp\`).
|
|
224
236
|
Codex has only the user scope.
|
|
237
|
+
stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>]
|
|
238
|
+
[--repo <path|name>] [--all-repos] [--json]
|
|
239
|
+
What nomArmy's job records show for this repository (or
|
|
240
|
+
all): volume by role and model, code committed, time,
|
|
241
|
+
tokens and spend, how often a "done" report failed
|
|
242
|
+
independent verification, what didn't finish, reviewers,
|
|
243
|
+
and review flags. From verified records, never reports.
|
|
244
|
+
validators <list|add jev|test jev|remove jev|add judge|test judge|remove judge>
|
|
245
|
+
Optional semantic checks from a model you configure with
|
|
246
|
+
your own key. Today: Jev (TypeSafe). \`add jev\` asks for the
|
|
247
|
+
key without echoing it (or reads --key-stdin), saves it
|
|
248
|
+
where only you can read it, and makes one test call. Its
|
|
249
|
+
answers only add review flags, and it sends excerpts of
|
|
250
|
+
your code to TypeSafe. \`add judge --agent <name> --model
|
|
251
|
+
<model>\` makes one of your agents a model judge: does the
|
|
252
|
+
diff meet each acceptance criterion, match the report, keep
|
|
253
|
+
its tests as strong? An agent whose tools run on this
|
|
254
|
+
machine needs --host-tools.
|
|
225
255
|
mcp Start nomArmy's MCP server on stdio with this machine's
|
|
226
256
|
settings. What a --scope project registration runs.
|
|
227
257
|
sandbox The Podman VM every sandbox shares (macOS, Windows): its
|
|
@@ -2497,6 +2527,12 @@ function renderJobs({ running, recent }) {
|
|
|
2497
2527
|
*/
|
|
2498
2528
|
async function streamJobEvents() {
|
|
2499
2529
|
const interval = Math.max(1, Number(value("interval", "3")) || 3) * 1000;
|
|
2530
|
+
// A stream nobody reads must still end: a General that ran this as a
|
|
2531
|
+
// background command (reported only on exit) was never told jobs had
|
|
2532
|
+
// finished, and eight of these streams were left running for days.
|
|
2533
|
+
const untilDone = flag("until-done");
|
|
2534
|
+
const idleLimitMs = Math.max(1, Number(value("idle-minutes", "30")) || 30) * 60000;
|
|
2535
|
+
let idleSinceMs = Date.now(), sawRunning = false;
|
|
2500
2536
|
const seen = new Map();
|
|
2501
2537
|
const emit = (event, job, detail = "") => {
|
|
2502
2538
|
if (json) console.log(JSON.stringify({ at: new Date().toISOString(), event, jobId: job.jobId, agent: job.agent, model: job.model, phase: job.phase, detail }));
|
|
@@ -2519,10 +2555,23 @@ async function streamJobEvents() {
|
|
|
2519
2555
|
seen.clear();
|
|
2520
2556
|
for (const [id, j] of now) seen.set(id, j);
|
|
2521
2557
|
first = false;
|
|
2558
|
+
if (running.length) { sawRunning = true; idleSinceMs = Date.now(); }
|
|
2559
|
+
else if (untilDone && sawRunning) {
|
|
2560
|
+
if (json) console.log(JSON.stringify({ at: new Date().toISOString(), event: "done", detail: "every job seen running has finished" }));
|
|
2561
|
+
else console.log(`${new Date().toLocaleTimeString()} done every job seen running has finished`);
|
|
2562
|
+
return;
|
|
2563
|
+
} else if (Date.now() - idleSinceMs >= (untilDone ? Math.min(idleLimitMs, 120000) : idleLimitMs)) {
|
|
2564
|
+
const why = untilDone ? "no job was running to wait for" : `nothing has run for ${Math.round(idleLimitMs / 60000)} minutes`;
|
|
2565
|
+
if (json) console.log(JSON.stringify({ at: new Date().toISOString(), event: "idle", detail: why }));
|
|
2566
|
+
else console.log(`${new Date().toLocaleTimeString()} idle ${why}; exiting`);
|
|
2567
|
+
return;
|
|
2568
|
+
}
|
|
2522
2569
|
await new Promise((r) => setTimeout(r, interval));
|
|
2523
2570
|
}
|
|
2524
2571
|
}
|
|
2525
2572
|
|
|
2573
|
+
const commitSha = (commit) => (typeof commit === "string" ? commit : typeof commit?.sha === "string" ? commit.sha : null);
|
|
2574
|
+
|
|
2526
2575
|
/** Wait for one job in the shared, cross-session state directory. */
|
|
2527
2576
|
async function waitForJobCli() {
|
|
2528
2577
|
const requested = value("wait");
|
|
@@ -2559,7 +2608,8 @@ async function waitForJobCli() {
|
|
|
2559
2608
|
outcome: meta.outcome ?? status.outcome ?? null,
|
|
2560
2609
|
coordinatorStatus: meta.coordinatorStatus ?? status.coordinatorStatus ?? null,
|
|
2561
2610
|
branch: meta.branch ?? status.branch ?? null,
|
|
2562
|
-
commit:
|
|
2611
|
+
// A job that made no commit has commit: { created: false, sha: null }; only a sha is a commit.
|
|
2612
|
+
commit: commitSha(meta.commit) ?? commitSha(status.commit),
|
|
2563
2613
|
issues,
|
|
2564
2614
|
};
|
|
2565
2615
|
if (json) out(result);
|
|
@@ -2600,6 +2650,13 @@ function pruneJobRuntimeCli() {
|
|
|
2600
2650
|
|
|
2601
2651
|
async function cmdJobs() {
|
|
2602
2652
|
if (flag("wait")) return waitForJobCli();
|
|
2653
|
+
if (flag("stop")) {
|
|
2654
|
+
const r = requestJobStop({ jobsRoot: jobsRootDir(), jobId: value("stop"), reason: value("reason") });
|
|
2655
|
+
if (json) return out(r);
|
|
2656
|
+
console.log(r.ok ? c.green(`✓ ${r.message}`) : c.red(`✗ ${r.message}`));
|
|
2657
|
+
if (!r.ok) process.exitCode = 1;
|
|
2658
|
+
return;
|
|
2659
|
+
}
|
|
2603
2660
|
if (flag("events")) return streamJobEvents();
|
|
2604
2661
|
if (flag("prune")) return pruneJobRuntimeCli();
|
|
2605
2662
|
if (json) return out(collectJobs());
|
|
@@ -2646,6 +2703,135 @@ async function cmdStatusline() {
|
|
|
2646
2703
|
process.stdout.write(`${statusLineText({ session })}\n`);
|
|
2647
2704
|
}
|
|
2648
2705
|
|
|
2706
|
+
function cmdStats() {
|
|
2707
|
+
const stateRoot = process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents");
|
|
2708
|
+
const records = loadJobRecords(path.join(stateRoot, "jobs"));
|
|
2709
|
+
let repo = null;
|
|
2710
|
+
if (value("repo")) repo = resolveRepo(records, value("repo"));
|
|
2711
|
+
else if (!flag("all-repos")) {
|
|
2712
|
+
try { repo = execFileSync("git", ["rev-parse", "--show-toplevel"], { cwd: repoDir, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }).trim(); }
|
|
2713
|
+
catch { throw new Error(`${repoDir} isn't inside a git repository; run nomarmy stats from one, or pass --repo <name> or --all-repos`); }
|
|
2714
|
+
}
|
|
2715
|
+
const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model") });
|
|
2716
|
+
if (json) return out(stats);
|
|
2717
|
+
console.log(formatStats(stats));
|
|
2718
|
+
}
|
|
2719
|
+
|
|
2720
|
+
// Read one line without echoing it: stty -echo around the read, restored
|
|
2721
|
+
// even if the read fails. Windows has no stty, so it says the input shows.
|
|
2722
|
+
async function readHiddenLine(prompt) {
|
|
2723
|
+
const hide = process.stdin.isTTY && process.platform !== "win32";
|
|
2724
|
+
if (!hide) console.log(c.yellow("(your input will be visible as you type)"));
|
|
2725
|
+
const rl = createInterface({ input, output });
|
|
2726
|
+
try {
|
|
2727
|
+
if (hide) spawnSync("stty", ["-echo"], { stdio: ["inherit", "ignore", "ignore"] });
|
|
2728
|
+
return (await rl.question(prompt)).trim();
|
|
2729
|
+
} finally {
|
|
2730
|
+
if (hide) { spawnSync("stty", ["echo"], { stdio: ["inherit", "ignore", "ignore"] }); process.stdout.write("\n"); }
|
|
2731
|
+
rl.close();
|
|
2732
|
+
}
|
|
2733
|
+
}
|
|
2734
|
+
|
|
2735
|
+
async function readStdin() {
|
|
2736
|
+
const chunks = [];
|
|
2737
|
+
for await (const chunk of process.stdin) chunks.push(chunk);
|
|
2738
|
+
return Buffer.concat(chunks).toString("utf8").trim();
|
|
2739
|
+
}
|
|
2740
|
+
|
|
2741
|
+
// One tiny System One request, to prove the key and the route work.
|
|
2742
|
+
async function testJev(settings) {
|
|
2743
|
+
const { answers } = await askJev({ key: settings.key, model: settings.model, state: { text: "The build finished and all 12 tests passed." },
|
|
2744
|
+
questions: { passed: { type: "noul", instructions: "Does the text say the tests passed?", criteria: { true: "It says the tests passed", false: "It doesn't" } } } });
|
|
2745
|
+
return typeof answers.passed?.noul === "number";
|
|
2746
|
+
}
|
|
2747
|
+
|
|
2748
|
+
async function cmdValidators() {
|
|
2749
|
+
const [sub = "list", name] = argv.slice(1).filter((a) => !a.startsWith("--"));
|
|
2750
|
+
if (sub === "list") {
|
|
2751
|
+
let config = {};
|
|
2752
|
+
try { config = loadValidators(); } catch (error) { if (json) return out({ error: error.message }); console.log(c.red(error.message)); process.exitCode = 1; return; }
|
|
2753
|
+
const judge = config.judge ? { enabled: config.judge.enabled, agent: config.judge.agent, model: config.judge.model, checks: config.judge.checks, hostTools: config.judge.host_tools } : null;
|
|
2754
|
+
const jev = config.jev ? { enabled: config.jev.enabled, checks: config.jev.checks, model: config.jev.model, key: config.jev.key_env ? `env ${config.jev.key_env}` : config.jev.key_file, keyReadable: Boolean(jevSettings()) } : null;
|
|
2755
|
+
if (json) return out({ path: validatorsPath(), jev, judge });
|
|
2756
|
+
if (!jev && !judge) { console.log("No validators configured. Add one with: nomarmy validators add jev, or nomarmy validators add judge --agent <name> --model <model>"); return; }
|
|
2757
|
+
if (jev) console.log(`Jev: ${jev.enabled ? c.green("on") : "off"} (${jev.model}); checks: ${jev.checks.join(", ")}; key: ${jev.key}${jev.keyReadable ? "" : c.red(" (not readable)")}`);
|
|
2758
|
+
if (judge) console.log(`Judge: ${judge.enabled ? c.green("on") : "off"} (${judge.agent}/${judge.model}); checks: ${judge.checks.join(", ")}${judge.hostTools ? c.yellow("; host tools allowed") : ""}`);
|
|
2759
|
+
return;
|
|
2760
|
+
}
|
|
2761
|
+
if (name === "judge") return cmdValidatorsJudge(sub);
|
|
2762
|
+
if (name !== "jev") throw new Error("Usage: nomarmy validators <list|add jev|test jev|remove jev|add judge|test judge|remove judge>");
|
|
2763
|
+
if (sub === "add") {
|
|
2764
|
+
if (!json) {
|
|
2765
|
+
console.log(c.bold("🍪 Jev (TypeSafe) for nomArmy's semantic checks\n"));
|
|
2766
|
+
console.log("It checks that a scout's cited lines support its finding, and that a worker's report matches its diff.");
|
|
2767
|
+
console.log("Its answers only add review flags; they never pass a check or allow a commit.");
|
|
2768
|
+
console.log(c.yellow("It sends excerpts of your code (findings, cited lines, diffs, worker reports) to TypeSafe.\n"));
|
|
2769
|
+
}
|
|
2770
|
+
const key = flag("key-stdin") ? await readStdin() : await readHiddenLine("TypeSafe API key (not shown): ");
|
|
2771
|
+
const saved = saveJevKey(key);
|
|
2772
|
+
let ok = false, why = null;
|
|
2773
|
+
try { ok = await testJev(jevSettings()); } catch (error) { why = error.message; }
|
|
2774
|
+
if (json) return out({ saved: true, keyFile: saved.keyFile, configPath: saved.configPath, test: ok ? "pass" : "fail", reason: why });
|
|
2775
|
+
console.log(c.green(`✓ Saved the key to ${saved.keyFile} (readable only by you) and turned Jev on in ${saved.configPath}.`));
|
|
2776
|
+
console.log(ok ? c.green("✓ Test call answered. New jobs use it; restart open coordinator sessions to pick it up.") : c.red(`✗ Test call failed: ${why ?? "no answer"}. Check the key, then: nomarmy validators test jev`));
|
|
2777
|
+
if (!ok) process.exitCode = 1;
|
|
2778
|
+
return;
|
|
2779
|
+
}
|
|
2780
|
+
if (sub === "test") {
|
|
2781
|
+
const settings = jevSettings();
|
|
2782
|
+
if (!settings) throw new Error("Jev isn't configured, or its key isn't readable. Add it with: nomarmy validators add jev");
|
|
2783
|
+
let ok = false, why = null;
|
|
2784
|
+
try { ok = await testJev(settings); } catch (error) { why = error.message; }
|
|
2785
|
+
if (json) return out({ test: ok ? "pass" : "fail", reason: why });
|
|
2786
|
+
console.log(ok ? c.green("✓ Jev answered.") : c.red(`✗ Jev test call failed: ${why ?? "no answer"}`));
|
|
2787
|
+
if (!ok) process.exitCode = 1;
|
|
2788
|
+
return;
|
|
2789
|
+
}
|
|
2790
|
+
if (sub === "remove") {
|
|
2791
|
+
const result = removeJev();
|
|
2792
|
+
if (json) return out(result);
|
|
2793
|
+
console.log(c.green(`✓ Jev is off${result.removedKey ? ", and its saved key is deleted" : ""}.`));
|
|
2794
|
+
return;
|
|
2795
|
+
}
|
|
2796
|
+
throw new Error("Usage: nomarmy validators <list|add jev|test jev|remove jev>");
|
|
2797
|
+
}
|
|
2798
|
+
|
|
2799
|
+
async function cmdValidatorsJudge(sub) {
|
|
2800
|
+
const agents = loadAgents(globalConfigDir()).agents;
|
|
2801
|
+
const resolve = () => judgeSettings({ agents, providerOf: agentProviderId, runsOnHost: agentRunsToolsOnHost });
|
|
2802
|
+
const probe = async (settings) => probeModel({ provider: settings.provider, model: settings.model, stateRoot: process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents") });
|
|
2803
|
+
if (sub === "add") {
|
|
2804
|
+
const agent = value("agent"), model = value("model");
|
|
2805
|
+
if (!agent || !model) throw new Error("Usage: nomarmy validators add judge --agent <name> --model <model> [--host-tools]");
|
|
2806
|
+
if (!agents[agent]) throw new Error(`"${agent}" isn't an agent in agents.yml. Agents: ${Object.keys(agents).join(", ") || "(none)"}`);
|
|
2807
|
+
if (agentRunsToolsOnHost(agents[agent]) && !flag("host-tools")) throw new Error(`agent "${agent}" runs its tools on this machine, and a judge reads text the worker wrote. Pass --host-tools to accept that, or pick a sandboxed agent (an api key, Codex, Muse).`);
|
|
2808
|
+
const saved = saveJudge({ agent, model, hostTools: flag("host-tools") });
|
|
2809
|
+
const settings = resolve();
|
|
2810
|
+
if (settings?.problem) throw new Error(settings.problem);
|
|
2811
|
+
const test = await probe(settings);
|
|
2812
|
+
if (json) return out({ saved: true, configPath: saved.configPath, test: test.ok ? "pass" : test.refused ? "refused" : "inconclusive", reason: test.reason });
|
|
2813
|
+
console.log(c.green(`✓ The judge is ${agent}/${model}, in ${saved.configPath}.`));
|
|
2814
|
+
console.log(test.ok ? c.green("✓ Test call answered. New implement jobs use it; restart open coordinator sessions to pick it up.") : c.red(`✗ Test call ${test.refused ? "refused" : "didn't answer"}: ${test.reason ?? "no answer"}`));
|
|
2815
|
+
if (!test.ok) process.exitCode = 1;
|
|
2816
|
+
return;
|
|
2817
|
+
}
|
|
2818
|
+
if (sub === "test") {
|
|
2819
|
+
const settings = resolve();
|
|
2820
|
+
if (!settings) throw new Error("No judge configured. Add one with: nomarmy validators add judge --agent <name> --model <model>");
|
|
2821
|
+
if (settings.problem) throw new Error(settings.problem);
|
|
2822
|
+
const test = await probe(settings);
|
|
2823
|
+
if (json) return out({ test: test.ok ? "pass" : "fail", reason: test.reason });
|
|
2824
|
+
console.log(test.ok ? c.green(`✓ ${settings.agent}/${settings.model} answered.`) : c.red(`✗ ${test.reason ?? "no answer"}`));
|
|
2825
|
+
if (!test.ok) process.exitCode = 1;
|
|
2826
|
+
return;
|
|
2827
|
+
}
|
|
2828
|
+
if (sub === "remove") {
|
|
2829
|
+
removeJudge();
|
|
2830
|
+
return json ? out({ removed: true }) : console.log(c.green("✓ The judge is off."));
|
|
2831
|
+
}
|
|
2832
|
+
throw new Error("Usage: nomarmy validators <add judge --agent <name> --model <model> [--host-tools]|test judge|remove judge>");
|
|
2833
|
+
}
|
|
2834
|
+
|
|
2649
2835
|
// `nomarmy mcp`: what a --scope project registration runs. Nothing goes to
|
|
2650
2836
|
// stdout but the server's own protocol.
|
|
2651
2837
|
function cmdMcp() {
|
|
@@ -2655,7 +2841,7 @@ function cmdMcp() {
|
|
|
2655
2841
|
child.on("exit", (code, signal) => { if (signal) process.kill(process.pid, signal); else process.exit(code ?? 1); });
|
|
2656
2842
|
}
|
|
2657
2843
|
|
|
2658
|
-
const commands = { mcp: cmdMcp, scan: cmdScan, validate: cmdValidate, sizing: cmdSizing, init: cmdInit, setup: cmdSetup, install: cmdInstall, model: cmdModel, agents: cmdAgents, army: cmdArmy, jobs: cmdJobs, statusline: cmdStatusline, health: cmdHealth, config: cmdConfig, update: cmdUpdate, connect: cmdConnect, sandbox: cmdSandbox, start: cmdStart, stop: cmdStop, uninstall: cmdUninstall, help: () => usage(0) };
|
|
2844
|
+
const commands = { stats: cmdStats, validators: cmdValidators, mcp: cmdMcp, scan: cmdScan, validate: cmdValidate, sizing: cmdSizing, init: cmdInit, setup: cmdSetup, install: cmdInstall, model: cmdModel, agents: cmdAgents, army: cmdArmy, jobs: cmdJobs, statusline: cmdStatusline, health: cmdHealth, config: cmdConfig, update: cmdUpdate, connect: cmdConnect, sandbox: cmdSandbox, start: cmdStart, stop: cmdStop, uninstall: cmdUninstall, help: () => usage(0) };
|
|
2659
2845
|
// doctor command
|
|
2660
2846
|
async function cmdDoctor() {
|
|
2661
2847
|
// Import lazily to avoid circular dependencies
|
package/lib/admission.mjs
CHANGED
|
@@ -126,6 +126,19 @@ export function createJobRuntime(deps) {
|
|
|
126
126
|
try { return await fn(); } finally { slot.release(); }
|
|
127
127
|
})();
|
|
128
128
|
}
|
|
129
|
+
// Who ran the job and for whom, onto its finished record, so `nomarmy
|
|
130
|
+
// stats` can count by role and agent, and by repo for verify runs too
|
|
131
|
+
// (their records didn't carry the repo). Best-effort; never fails the job.
|
|
132
|
+
function stampJobRecord(entry, result) {
|
|
133
|
+
const file = path.join(result?.jobDir ?? path.join(jobsRoot, entry.jobId), "metadata.json");
|
|
134
|
+
try {
|
|
135
|
+
const record = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
136
|
+
record.labels = { role: entry.role ?? null, agent: entry.agent ?? null, label: entry.label ?? null, runId: entry.runId ?? null };
|
|
137
|
+
record.projectDir = record.projectDir ?? projectDir;
|
|
138
|
+
fs.writeFileSync(`${file}.tmp`, JSON.stringify(record, null, 2));
|
|
139
|
+
fs.renameSync(`${file}.tmp`, file);
|
|
140
|
+
} catch { /* no record (a refused or crashed job): nothing to label */ }
|
|
141
|
+
}
|
|
129
142
|
function track(jobId, meta, promise) {
|
|
130
143
|
const entry = { ...meta, jobId, startedAt: new Date().toISOString(), settled: false, result: null, error: null, promise: null };
|
|
131
144
|
// A machine-wide lease for as long as the job runs, so every session's
|
|
@@ -134,7 +147,7 @@ export function createJobRuntime(deps) {
|
|
|
134
147
|
if (meta.lane) writeLease(leasesRoot, jobId, { lane: meta.lane, agent: meta.agent ?? null, runId: meta.runId ?? null, role: meta.role ?? null, model: meta.model ?? null, repo: projectDir });
|
|
135
148
|
const release = () => removeLease(leasesRoot, jobId);
|
|
136
149
|
entry.promise = promise.then(
|
|
137
|
-
r => { entry.settled = true; entry.result = r; release(); notifyJobFinished(entry, r, null); return r; },
|
|
150
|
+
r => { entry.settled = true; entry.result = r; release(); stampJobRecord(entry, r); notifyJobFinished(entry, r, null); return r; },
|
|
138
151
|
e => { entry.settled = true; entry.error = e; release(); notifyJobFinished(entry, null, e); throw e; });
|
|
139
152
|
entry.promise.catch(() => {});
|
|
140
153
|
activeJobs.set(jobId, entry);
|
|
@@ -15,12 +15,12 @@ Before dispatching:
|
|
|
15
15
|
- A Claude subscription agent (claude-cli) runs its tools on this machine, outside the sandbox: use it for scout and review work. nomArmy refuses implement jobs on it unless the operator set allow_host_tools; send build work to a sandboxed agent.
|
|
16
16
|
- To run tests without changing anything, use mode: verify; it costs no model usage.
|
|
17
17
|
- Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
|
|
18
|
-
- Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator; otherwise poll local_worker_status with wait_seconds. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
18
|
+
- Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
19
19
|
|
|
20
20
|
Trust boundary:
|
|
21
21
|
- A worker's four-line report is a claim; nomArmy's verified git record and independent verification are the evidence. A job isn't complete if its report is missing or malformed, its STATUS is partial or blocked, STATUS done lacks VERIFICATION pass, or its changes aren't committed by nomArmy.
|
|
22
22
|
- Read the diff of anything material before integrating it. nomArmy commits on the worker's branch and never merges into yours: integration, conflicts and pushes are yours.
|
|
23
|
-
- To finish a job that came back partial, blocked
|
|
23
|
+
- A job on the wrong track can be stopped with local_worker_stop; it keeps its worktree. To finish a job that came back partial, blocked, failing verification or stopped, dispatch the correction with continue_from: <that job id> (a cheaper model is fine). The new job starts with its unfinished work in place and verifies the whole. Don't fix and commit a worker's files yourself: that lands them unverified. If you must, commit them on a branch, run mode: verify on it before building on it, and say so in your report.
|
|
24
24
|
- Failed and incomplete worktrees are kept for review; clean up with local_worker_cleanup or local_worker_sweep once you've decided.
|
|
25
25
|
|
|
26
26
|
Never delegate deployments, production access, cloud or SSH credentials, secrets, Terraform state or kubectl contexts to a worker.`;
|
package/lib/execute.mjs
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery } from "./openclaw-run.mjs";
|
|
1
|
+
import { writeStatus, shouldRetryTransientAbort, shouldAttemptScoutRecovery, readStopRequest } from "./openclaw-run.mjs";
|
|
2
2
|
import fs from "node:fs";
|
|
3
3
|
import path from "node:path";
|
|
4
4
|
import { fileURLToPath } from "node:url";
|
|
@@ -9,7 +9,10 @@ import { parseScoutReport, verifyCitations, resolveScoutOutcome, renderScoutRepo
|
|
|
9
9
|
import { parseDecomposeReport, buildDecomposeFindings, resolveDecomposeOutcome, checkDecompositionOverlap, renderDecomposeReport } from "./decompose.mjs";
|
|
10
10
|
import { deriveTimeBudget } from "./budget.mjs";
|
|
11
11
|
import { continuationProblem, continuationBase, snapshotRetainedWork, applyRetainedWork, continuationNote } from "./continue-from.mjs";
|
|
12
|
-
import {
|
|
12
|
+
import { checkScoutCitations, checkReportClaims } from "./jev-checks.mjs";
|
|
13
|
+
import { runJudge } from "./judge.mjs";
|
|
14
|
+
import { pickMutants, runMutants, describeSurvivors } from "./mutation.mjs";
|
|
15
|
+
import { estimateDisplacement, readOpenClawTranscript } from "./transcript.mjs";
|
|
13
16
|
import { outlineFile, findReferences } from "./repo-query.mjs";
|
|
14
17
|
import { loadConfig } from "./config.mjs";
|
|
15
18
|
import { linkNodePackages, nodeModulesState, repairHostInstalls } from "./sandbox-images.mjs";
|
|
@@ -18,7 +21,7 @@ import { describeRecoveryChanges, reportRecoveryPrompt } from "./worker-prompt.m
|
|
|
18
21
|
import { parseWorkerReport } from "./report.mjs";
|
|
19
22
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "./outcomes.mjs";
|
|
20
23
|
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy } from "./outcome.mjs";
|
|
21
|
-
import { isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
24
|
+
import { parseAddedLineNumbers, isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
22
25
|
|
|
23
26
|
// ---------------------------------------------------------------------------
|
|
24
27
|
// Job status for polling. `status.json` is written at every phase transition
|
|
@@ -36,6 +39,8 @@ export function createExecutor(deps) {
|
|
|
36
39
|
resolveReasoningApplied, recordedBudgets, runOpenClaw, sweepStaleSandboxContainers,
|
|
37
40
|
verificationFlow, normalizeVerification, runIndependentVerification,
|
|
38
41
|
runRegressionCheck, repoPolicy } = deps;
|
|
42
|
+
// Optional Jev checks (lib/validators.mjs): the server wires the settings; without them (tests), none run.
|
|
43
|
+
const jevSettingsFor = (check) => { try { const s = deps.jevSettings?.(); return s?.checks?.includes(check) ? s : null; } catch { return null; } };
|
|
39
44
|
|
|
40
45
|
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, jobId: presetJobId = null }) {
|
|
41
46
|
await assertRepo();
|
|
@@ -270,6 +275,9 @@ export function createExecutor(deps) {
|
|
|
270
275
|
// failed job that changed nothing was recorded "pass" (a Senti run),
|
|
271
276
|
// which reads as evidence about work that never happened.
|
|
272
277
|
independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the worker changed nothing, so there was none of its work to verify" }, verification ?? null);
|
|
278
|
+
} else if (workerStopReason === "stopped") {
|
|
279
|
+
// Stopped on request: end now, without spending time on tests.
|
|
280
|
+
independentVerification = normalizeVerification({ status: "not_run", basis: "not-applicable", reason: "the job was stopped on request" }, verification ?? null);
|
|
273
281
|
} else if (verificationFlow.verificationRunner || !reportValidation.valid) {
|
|
274
282
|
independentVerification = await runIndependentVerification({ profile: verification ?? null, cwd, jobId, baseSha: base.sha, branch, mode, record: preCommit, logFile: path.join(jobDir, "verification.log") });
|
|
275
283
|
}
|
|
@@ -300,6 +308,36 @@ export function createExecutor(deps) {
|
|
|
300
308
|
if (regressionCheck.status === "restore_failed") regressionCheckFatal = true;
|
|
301
309
|
}
|
|
302
310
|
|
|
311
|
+
// Mutation testing (lib/mutation.mjs), when this repo opts in: small
|
|
312
|
+
// mistakes planted one at a time in the changed lines must each fail
|
|
313
|
+
// the same profile. Survivors raise review; a failed restore is fatal.
|
|
314
|
+
let mutation = null, mutationElapsedMs = null;
|
|
315
|
+
const mutationConfig = (() => { try { return loadConfig(projectDir)?.config?.mutation ?? null; } catch { return null; } })();
|
|
316
|
+
if (mutationConfig && verification && independentVerification.status === "pass" && codeFilesChanged.length > 0 && !regressionCheckFatal) {
|
|
317
|
+
const mutationStartedMs = Date.now();
|
|
318
|
+
try {
|
|
319
|
+
const untracked = new Set((preCommit.nameStatus ?? []).filter((e) => e.untracked).map((e) => e.path));
|
|
320
|
+
const deleted = new Set((preCommit.nameStatus ?? []).filter((e) => /^D/.test(e.status)).map((e) => e.path));
|
|
321
|
+
const files = [];
|
|
322
|
+
for (const file of codeFilesChanged.filter((f) => !deleted.has(f))) {
|
|
323
|
+
const full = path.join(cwd, file);
|
|
324
|
+
let lines;
|
|
325
|
+
if (untracked.has(file)) { try { lines = fs.readFileSync(full, "utf8").split("\n").map((_, i) => i + 1); } catch { continue; } }
|
|
326
|
+
else lines = parseAddedLineNumbers(await gitRaw(["diff", "-U0", base.sha, "--", file], cwd));
|
|
327
|
+
if (lines.length) files.push({ path: file, full, lines });
|
|
328
|
+
}
|
|
329
|
+
const mutants = pickMutants(files, mutationConfig.mutants);
|
|
330
|
+
let n = 0;
|
|
331
|
+
mutation = await runMutants({ mutants, deadlineMs: mutationStartedMs + mutationConfig.max_seconds * 1000,
|
|
332
|
+
verify: () => runIndependentVerification({ profile: verification, cwd, jobId: `${jobId}-mutant-${++n}`, baseSha: base.sha, branch, mode, record: preCommit }) });
|
|
333
|
+
mutation.planned = mutants.length;
|
|
334
|
+
} catch (error) {
|
|
335
|
+
mutation = { status: "not_run", killed: 0, survived: [], inconclusive: 0, tried: 0, reason: `mutation testing failed to run: ${error.message}` };
|
|
336
|
+
}
|
|
337
|
+
mutationElapsedMs = Date.now() - mutationStartedMs;
|
|
338
|
+
fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} mutation testing: ${mutation.killed ?? 0} killed, ${mutation.survived?.length ?? 0} survived, ${mutation.inconclusive ?? 0} inconclusive of ${mutation.tried ?? 0} tried (${Math.round(mutationElapsedMs / 1000)}s)\n`);
|
|
339
|
+
}
|
|
340
|
+
|
|
303
341
|
// resolveOutcome's own contract only ever sees pass/fail/not_run for
|
|
304
342
|
// regressionCheck -- a restore_failed status is substituted to not_run
|
|
305
343
|
// here so resolveOutcome never needs a fourth value; the hard override
|
|
@@ -427,21 +465,67 @@ export function createExecutor(deps) {
|
|
|
427
465
|
commitBlockedReason: `possible secret detected: ${possibleSecrets.reason}`,
|
|
428
466
|
reasons: [...afterHostInstalls.reasons, `POSSIBLE SECRET DETECTED: ${possibleSecrets.reason}`] }
|
|
429
467
|
: afterHostInstalls;
|
|
468
|
+
// Jev: does the worker's report match its diff? Found live: a note said
|
|
469
|
+
// "restored check.js to base commit" while the diff rewrote check.js.
|
|
470
|
+
// Only raises review; it never blocks or allows a commit.
|
|
471
|
+
let jevClaims = null, judged = null;
|
|
472
|
+
const jevImplement = jevSettingsFor("report-claims");
|
|
473
|
+
const judge = (() => { try { return deps.judgeSettings?.() ?? null; } catch { return null; } })();
|
|
474
|
+
// The whole change, new files included, for whichever validators run.
|
|
475
|
+
const jobDiff = async () => {
|
|
476
|
+
let diff = await gitRaw(["diff", base.sha, "--"], cwd);
|
|
477
|
+
for (const entry of (preCommit.nameStatus ?? []).filter((e) => e.untracked)) {
|
|
478
|
+
let text = "";
|
|
479
|
+
try { text = fs.readFileSync(path.join(cwd, entry.path), "utf8"); } catch { continue; }
|
|
480
|
+
diff += `\ndiff --git a/${entry.path} b/${entry.path}\nnew file\n--- /dev/null\n+++ b/${entry.path}\n${text.split("\n").map((l) => `+${l}`).join("\n")}\n`;
|
|
481
|
+
}
|
|
482
|
+
return diff;
|
|
483
|
+
};
|
|
484
|
+
let diffText = null;
|
|
485
|
+
if (mode === "implement" && reportValidation?.valid && repositoryChanged && (jevImplement || (judge && !judge.problem))) {
|
|
486
|
+
try { diffText = await jobDiff(); } catch { diffText = null; }
|
|
487
|
+
}
|
|
488
|
+
if (jevImplement && diffText != null) {
|
|
489
|
+
try { jevClaims = await checkReportClaims({ report: reportValidation, diff: diffText, settings: jevImplement }); }
|
|
490
|
+
catch (error) { jevClaims = { flag: null, verdict: null, error: error.message, usage: 0 }; }
|
|
491
|
+
}
|
|
492
|
+
// The model judge (lib/judge.mjs): acceptance criteria, the report and
|
|
493
|
+
// changed tests. Only raises review, never blocks or allows a commit.
|
|
494
|
+
if (judge?.problem) judged = { flags: [], answer: null, error: `judge not run: ${judge.problem}`, skipped: true };
|
|
495
|
+
else if (judge && diffText != null) {
|
|
496
|
+
const modifiedTests = preCommit.testChanges?.existing_tests_modified ?? [];
|
|
497
|
+
let testDiff = "";
|
|
498
|
+
if (modifiedTests.length) { try { testDiff = await gitRaw(["diff", base.sha, "--", ...modifiedTests], cwd); } catch { testDiff = ""; } }
|
|
499
|
+
judged = await runJudge({ settings: judge, task, acceptance: acceptance ?? [], report: reportValidation, diff: diffText, testDiff, stateRoot: path.join(jobsRoot, "..") });
|
|
500
|
+
}
|
|
501
|
+
const afterJevOnly = jevClaims?.flag
|
|
502
|
+
? { ...afterSecrets, reviewRequired: true, reasons: [...afterSecrets.reasons, `REPORT MAY NOT MATCH THE DIFF (Jev, ${jevClaims.flag.probability.toFixed(2)}): the report says "${String(reportValidation.note ?? "").slice(0, 200)}", and the diff may show otherwise. Read the diff before accepting.`] }
|
|
503
|
+
: afterSecrets;
|
|
504
|
+
const afterJev = judged?.flags?.length
|
|
505
|
+
? { ...afterJevOnly, reviewRequired: true, reasons: [...afterJevOnly.reasons, `JUDGE (${judge.agent}/${judge.model}): ${judged.flags.join("; ")}. Read the diff before accepting.`] }
|
|
506
|
+
: afterJevOnly;
|
|
507
|
+
|
|
430
508
|
// A worker must not be judged by a check it rewrote: a changed script,
|
|
431
509
|
// Makefile or package.json script that a verification command runs
|
|
432
510
|
// blocks the commit; changed test-runner config only asks for review.
|
|
433
511
|
const blockedInputs = verificationInputs?.blocked ?? [], flaggedInputs = verificationInputs?.flagged ?? [];
|
|
434
512
|
const inputLine = blockedInputs.map((b) => `${b.file} (${b.why})`).join("; ");
|
|
435
513
|
const afterInputs = blockedInputs.length
|
|
436
|
-
? { ...
|
|
514
|
+
? { ...afterJev, outcome: afterJev.commitAllowed || afterJev.outcome === OUTCOMES.WORKER_DONE ? OUTCOMES.NEEDS_REVIEW : afterJev.outcome,
|
|
437
515
|
reviewRequired: true, commitAllowed: false,
|
|
438
|
-
commitBlockedReason:
|
|
439
|
-
reasons: [...
|
|
440
|
-
:
|
|
516
|
+
commitBlockedReason: afterJev.commitAllowed ? `the diff changes what verification runs: ${inputLine}` : afterJev.commitBlockedReason,
|
|
517
|
+
reasons: [...afterJev.reasons, `VERIFICATION INPUT CHANGED: the diff changes what profile '${verification}' runs, so its result can't be trusted: ${inputLine}`] }
|
|
518
|
+
: afterJev;
|
|
441
519
|
const afterConfig = flaggedInputs.length
|
|
442
520
|
? { ...afterInputs, reviewRequired: true, reasons: [...afterInputs.reasons, `TEST CONFIG CHANGED: ${flaggedInputs.map((c) => `${c.file} (${c.why})`).join("; ")}`] }
|
|
443
521
|
: afterInputs;
|
|
444
|
-
const
|
|
522
|
+
const afterMutation = mutation?.status === "restore_failed"
|
|
523
|
+
? { ...afterConfig, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true, commitAllowed: false,
|
|
524
|
+
commitBlockedReason: `mutation testing could not restore the worker's file: ${mutation.reason}`, reasons: [...afterConfig.reasons, `MUTATION RESTORE FAILED: ${mutation.reason}`] }
|
|
525
|
+
: mutation?.status === "survivors"
|
|
526
|
+
? { ...afterConfig, reviewRequired: true, reasons: [...afterConfig.reasons, describeSurvivors(mutation, verification)] }
|
|
527
|
+
: afterConfig;
|
|
528
|
+
const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterMutation, independentVerification.status, repoPolicy()),
|
|
445
529
|
{ refactor, verificationStatus: independentVerification.status, testChanges: preCommit.testChanges });
|
|
446
530
|
|
|
447
531
|
progress("commit");
|
|
@@ -452,8 +536,13 @@ export function createExecutor(deps) {
|
|
|
452
536
|
|
|
453
537
|
let coordinatorStatus = COORDINATOR_STATUS_BY_OUTCOME[finalOutcome.outcome] ?? "incomplete";
|
|
454
538
|
const issues = [...finalOutcome.reasons, ...(preCommit.issues ?? [])];
|
|
455
|
-
if (
|
|
539
|
+
if (workerStopReason === "stopped") {
|
|
540
|
+
const request = readStopRequest(jobDir);
|
|
541
|
+
issues.push(`stopped on request${request?.reason ? `: ${request.reason}` : ""}; the worktree is kept, so continue_from can pick the work up (on another model too)`);
|
|
542
|
+
} else if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
|
|
456
543
|
if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
|
|
544
|
+
if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
|
|
545
|
+
if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
|
|
457
546
|
if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
|
|
458
547
|
if (repositoryChanged && !commit.created) {
|
|
459
548
|
if (coordinatorStatus === "complete") coordinatorStatus = "incomplete";
|
|
@@ -464,7 +553,7 @@ export function createExecutor(deps) {
|
|
|
464
553
|
// the diffstat right in the issue a caller actually reads -- not just
|
|
465
554
|
// buried in the full manifest's git record -- is what makes "go look at
|
|
466
555
|
// the worktree" worth doing instead of discarding the job.
|
|
467
|
-
issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
|
|
556
|
+
if (record.filesChanged > 0) issues.push(`repository changes remain uncommitted (${record.filesChanged} file(s), +${record.additions}/-${record.deletions}): ${commit.reason}`);
|
|
468
557
|
}
|
|
469
558
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`worker recorded ${failures} tool failure(s)`);
|
|
470
559
|
if (record.ignoredRuntimeJunk.length) issues.push(`runtime junk ignored: ${record.ignoredRuntimeJunk.join(", ")}`);
|
|
@@ -483,6 +572,8 @@ export function createExecutor(deps) {
|
|
|
483
572
|
const metrics = buildMetrics({ result: result ?? attempted, record, reportValidation, outcome: finalOutcome, workerElapsedMs, totalElapsedMs: Date.now() - jobStartedMs, regressionCheckElapsedMs, transientAbortRetried });
|
|
484
573
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree, branch, startedAt, finishedAt,
|
|
485
574
|
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}),
|
|
575
|
+
...(mutation ? { mutation: { ...mutation, elapsedSeconds: Math.round((mutationElapsedMs ?? 0) / 1000) } } : {}),
|
|
576
|
+
...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
|
|
486
577
|
outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
|
|
487
578
|
reportRecoveryAttempted, reportRecovered,
|
|
488
579
|
reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
|
|
@@ -584,12 +675,18 @@ export function createExecutor(deps) {
|
|
|
584
675
|
const remainingSeconds = timeoutSeconds - Math.round(workerElapsedMs / 1000);
|
|
585
676
|
if (shouldAttemptScoutRecovery({ workerFailed, workerTimedOut, report, remainingSeconds })) {
|
|
586
677
|
reportRecoveryAttempted = true;
|
|
678
|
+
// What the first run left, in case the follow-up can't see its session.
|
|
679
|
+
let filesRead = [], earlierReply = reportText;
|
|
680
|
+
try {
|
|
681
|
+
const first = await readOpenClawTranscript(path.join(runtimeDir, "state"));
|
|
682
|
+
if (first.available) { filesRead = first.filesRead ?? []; earlierReply = earlierReply || first.lastAssistantText || ""; }
|
|
683
|
+
} catch { /* the reply alone still helps */ }
|
|
587
684
|
try {
|
|
588
685
|
const recoveryResult = await runOpenClaw({
|
|
589
686
|
task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha,
|
|
590
687
|
timeoutSeconds: remainingSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId,
|
|
591
688
|
evidenceTool: evidencePlaced ? evidenceTool : null,
|
|
592
|
-
overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout, question: task, acceptance }), logSuffix: "-recovery",
|
|
689
|
+
overridePrompt: scoutReportRecoveryPrompt({ report: used.report.scout, question: task, acceptance, earlierReply, filesRead }), logSuffix: "-recovery",
|
|
593
690
|
});
|
|
594
691
|
const recoveryReport = parseScoutReport(finalText(recoveryResult), (recoveryResult?.budgetsUsed ?? used).scout);
|
|
595
692
|
if (!isScoutReportUnusable(recoveryReport)) {
|
|
@@ -609,7 +706,16 @@ export function createExecutor(deps) {
|
|
|
609
706
|
const dirty = record.repoStatusFiles.length > 0;
|
|
610
707
|
const readFile = async p => { try { return await gitRaw(["show", `${base.sha}:${p}`], projectDir); } catch { return null; } };
|
|
611
708
|
const verified = await verifyCitations(report.findings, { readFile, limits: used.scout });
|
|
709
|
+
// Jev: do the cited lines support each finding? Only adds flags.
|
|
710
|
+
let jevCitations = null;
|
|
711
|
+
const jevScout = jevSettingsFor("scout-citations");
|
|
712
|
+
if (jevScout && verified?.findings?.length) {
|
|
713
|
+
try { jevCitations = await checkScoutCitations({ findings: verified.findings, settings: jevScout, readFile }); }
|
|
714
|
+
catch (error) { jevCitations = { flags: [], checked: 0, errors: [error.message], usage: 0, verdicts: [] }; }
|
|
715
|
+
for (const f of jevCitations.flags) verified.findings[f.index].jev = { verdict: f.verdict, probability: f.probability };
|
|
716
|
+
}
|
|
612
717
|
const outcome = resolveScoutOutcome({ report, verified, workerFailed, workerTimedOut, dirty });
|
|
718
|
+
if (jevCitations?.flags.length) outcome.reviewRequired = true;
|
|
613
719
|
|
|
614
720
|
progress("record");
|
|
615
721
|
if (outcome.retainWorktree) worktreeRetained = true;
|
|
@@ -621,6 +727,8 @@ export function createExecutor(deps) {
|
|
|
621
727
|
if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
|
|
622
728
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
|
|
623
729
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
730
|
+
if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
|
|
731
|
+
if (jevCitations?.errors.length) issues.push(`Jev citation check skipped or incomplete (${jevCitations.errors.join("; ")}); this job's result doesn't depend on it`);
|
|
624
732
|
if (reportRecoveryAttempted) {
|
|
625
733
|
issues.push(reportRecovered
|
|
626
734
|
? "scout report recovered via a follow-up call after the first reply was cut off"
|
|
@@ -658,6 +766,7 @@ export function createExecutor(deps) {
|
|
|
658
766
|
};
|
|
659
767
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
660
768
|
objective: task, mustCover: acceptance ?? [],
|
|
769
|
+
...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
|
|
661
770
|
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
662
771
|
scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
|
|
663
772
|
findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|