nomarmy 0.1.0-alpha.11 → 0.1.0-alpha.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nomarmy.mjs +151 -1
- package/lib/admission.mjs +14 -1
- package/lib/execute.mjs +103 -6
- package/lib/jev-checks.mjs +110 -0
- package/lib/judge.mjs +130 -0
- package/lib/mutation.mjs +159 -0
- package/lib/schema.mjs +10 -0
- package/lib/scout.mjs +2 -1
- package/lib/stats.mjs +213 -0
- package/lib/validators.mjs +179 -0
- package/mcp/server.mjs +21 -0
- package/package.json +1 -1
package/bin/nomarmy.mjs
CHANGED
|
@@ -20,6 +20,9 @@ import { readGGUFMetadata, resolveModelPath, totalSplitBytes } from "../lib/gguf
|
|
|
20
20
|
import { recommend, customRecommendation, evaluateConfig, bytesPerKvElementForCacheTypes, MIN_CONTEXT_PER_NOM } from "../lib/sizing.mjs";
|
|
21
21
|
import { connectClaude, connectCodex, connectCursor, cursorAlreadyConnected, deriveWorkerModelEnv, defaultInstallDir, installMcpCopy, SCOPES, claudeUserScoped, portableServerLaunch } from "../lib/connect.mjs";
|
|
22
22
|
import { compareVersions, readPackageVersion, readInstallVersions, copyIsStale } from "../lib/install-freshness.mjs";
|
|
23
|
+
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo } from "../lib/stats.mjs";
|
|
24
|
+
import { loadValidators, saveJevKey, removeJev, jevSettings, askJev, validatorsPath, JEV_CHECKS, saveJudge, removeJudge, judgeSettings } from "../lib/validators.mjs";
|
|
25
|
+
import { probeModel } from "../lib/model-probe.mjs";
|
|
23
26
|
import { ID_RE, AUTH_ENV_NAME_RE, OPENCLAW_PROVIDER_ID_RE, openclawProviderId, isNativeProviderType } from "../lib/dispatch-schema.mjs";
|
|
24
27
|
import { loadAgents, readAgentsFile, writeAgentsFile, agentsConfigPath, apiAgentAsPoolEntry, describeAgent as describeAgentLabel, agentRunsToolsOnHost, agentProviderId, AGENT_KINDS, API_PROVIDER_TYPES, RESERVED_AGENT_NAMES, BUILTIN_LOCAL_AGENT } from "../lib/agents.mjs";
|
|
25
28
|
import { loadArmy, mergeArmy, describeArmy, readArmyFile, updateArmyInFile, assignRoleInFile, parseTargetSpec, armyLayerPath, globalConfigDir, DEFAULT_ARMY, ARMY_PHASES, LOCAL_CONFIG_FILENAME } from "../lib/army.mjs";
|
|
@@ -222,6 +225,24 @@ Usage: nomarmy <command> [options]
|
|
|
222
225
|
project for this repository, committed for the team
|
|
223
226
|
(.mcp.json or .cursor/mcp.json, running \`nomarmy mcp\`).
|
|
224
227
|
Codex has only the user scope.
|
|
228
|
+
stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>]
|
|
229
|
+
[--repo <path|name>] [--all-repos] [--json]
|
|
230
|
+
What nomArmy's job records show for this repository (or
|
|
231
|
+
all): volume by role and model, code committed, time,
|
|
232
|
+
tokens and spend, how often a "done" report failed
|
|
233
|
+
independent verification, what didn't finish, reviewers,
|
|
234
|
+
and review flags. From verified records, never reports.
|
|
235
|
+
validators <list|add jev|test jev|remove jev|add judge|test judge|remove judge>
|
|
236
|
+
Optional semantic checks from a model you configure with
|
|
237
|
+
your own key. Today: Jev (TypeSafe). \`add jev\` asks for the
|
|
238
|
+
key without echoing it (or reads --key-stdin), saves it
|
|
239
|
+
where only you can read it, and makes one test call. Its
|
|
240
|
+
answers only add review flags, and it sends excerpts of
|
|
241
|
+
your code to TypeSafe. \`add judge --agent <name> --model
|
|
242
|
+
<model>\` makes one of your agents a model judge: does the
|
|
243
|
+
diff meet each acceptance criterion, match the report, keep
|
|
244
|
+
its tests as strong? An agent whose tools run on this
|
|
245
|
+
machine needs --host-tools.
|
|
225
246
|
mcp Start nomArmy's MCP server on stdio with this machine's
|
|
226
247
|
settings. What a --scope project registration runs.
|
|
227
248
|
sandbox The Podman VM every sandbox shares (macOS, Windows): its
|
|
@@ -2646,6 +2667,135 @@ async function cmdStatusline() {
|
|
|
2646
2667
|
process.stdout.write(`${statusLineText({ session })}\n`);
|
|
2647
2668
|
}
|
|
2648
2669
|
|
|
2670
|
+
function cmdStats() {
|
|
2671
|
+
const stateRoot = process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents");
|
|
2672
|
+
const records = loadJobRecords(path.join(stateRoot, "jobs"));
|
|
2673
|
+
let repo = null;
|
|
2674
|
+
if (value("repo")) repo = resolveRepo(records, value("repo"));
|
|
2675
|
+
else if (!flag("all-repos")) {
|
|
2676
|
+
try { repo = execFileSync("git", ["rev-parse", "--show-toplevel"], { cwd: repoDir, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }).trim(); }
|
|
2677
|
+
catch { throw new Error(`${repoDir} isn't inside a git repository; run nomarmy stats from one, or pass --repo <name> or --all-repos`); }
|
|
2678
|
+
}
|
|
2679
|
+
const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model") });
|
|
2680
|
+
if (json) return out(stats);
|
|
2681
|
+
console.log(formatStats(stats));
|
|
2682
|
+
}
|
|
2683
|
+
|
|
2684
|
+
// Read one line without echoing it: stty -echo around the read, restored
|
|
2685
|
+
// even if the read fails. Windows has no stty, so it says the input shows.
|
|
2686
|
+
async function readHiddenLine(prompt) {
|
|
2687
|
+
const hide = process.stdin.isTTY && process.platform !== "win32";
|
|
2688
|
+
if (!hide) console.log(c.yellow("(your input will be visible as you type)"));
|
|
2689
|
+
const rl = createInterface({ input, output });
|
|
2690
|
+
try {
|
|
2691
|
+
if (hide) spawnSync("stty", ["-echo"], { stdio: ["inherit", "ignore", "ignore"] });
|
|
2692
|
+
return (await rl.question(prompt)).trim();
|
|
2693
|
+
} finally {
|
|
2694
|
+
if (hide) { spawnSync("stty", ["echo"], { stdio: ["inherit", "ignore", "ignore"] }); process.stdout.write("\n"); }
|
|
2695
|
+
rl.close();
|
|
2696
|
+
}
|
|
2697
|
+
}
|
|
2698
|
+
|
|
2699
|
+
async function readStdin() {
|
|
2700
|
+
const chunks = [];
|
|
2701
|
+
for await (const chunk of process.stdin) chunks.push(chunk);
|
|
2702
|
+
return Buffer.concat(chunks).toString("utf8").trim();
|
|
2703
|
+
}
|
|
2704
|
+
|
|
2705
|
+
// One tiny System One request, to prove the key and the route work.
|
|
2706
|
+
async function testJev(settings) {
|
|
2707
|
+
const { answers } = await askJev({ key: settings.key, model: settings.model, state: { text: "The build finished and all 12 tests passed." },
|
|
2708
|
+
questions: { passed: { type: "noul", instructions: "Does the text say the tests passed?", criteria: { true: "It says the tests passed", false: "It doesn't" } } } });
|
|
2709
|
+
return typeof answers.passed?.noul === "number";
|
|
2710
|
+
}
|
|
2711
|
+
|
|
2712
|
+
async function cmdValidators() {
|
|
2713
|
+
const [sub = "list", name] = argv.slice(1).filter((a) => !a.startsWith("--"));
|
|
2714
|
+
if (sub === "list") {
|
|
2715
|
+
let config = {};
|
|
2716
|
+
try { config = loadValidators(); } catch (error) { if (json) return out({ error: error.message }); console.log(c.red(error.message)); process.exitCode = 1; return; }
|
|
2717
|
+
const judge = config.judge ? { enabled: config.judge.enabled, agent: config.judge.agent, model: config.judge.model, checks: config.judge.checks, hostTools: config.judge.host_tools } : null;
|
|
2718
|
+
const jev = config.jev ? { enabled: config.jev.enabled, checks: config.jev.checks, model: config.jev.model, key: config.jev.key_env ? `env ${config.jev.key_env}` : config.jev.key_file, keyReadable: Boolean(jevSettings()) } : null;
|
|
2719
|
+
if (json) return out({ path: validatorsPath(), jev, judge });
|
|
2720
|
+
if (!jev && !judge) { console.log("No validators configured. Add one with: nomarmy validators add jev, or nomarmy validators add judge --agent <name> --model <model>"); return; }
|
|
2721
|
+
if (jev) console.log(`Jev: ${jev.enabled ? c.green("on") : "off"} (${jev.model}); checks: ${jev.checks.join(", ")}; key: ${jev.key}${jev.keyReadable ? "" : c.red(" (not readable)")}`);
|
|
2722
|
+
if (judge) console.log(`Judge: ${judge.enabled ? c.green("on") : "off"} (${judge.agent}/${judge.model}); checks: ${judge.checks.join(", ")}${judge.hostTools ? c.yellow("; host tools allowed") : ""}`);
|
|
2723
|
+
return;
|
|
2724
|
+
}
|
|
2725
|
+
if (name === "judge") return cmdValidatorsJudge(sub);
|
|
2726
|
+
if (name !== "jev") throw new Error("Usage: nomarmy validators <list|add jev|test jev|remove jev|add judge|test judge|remove judge>");
|
|
2727
|
+
if (sub === "add") {
|
|
2728
|
+
if (!json) {
|
|
2729
|
+
console.log(c.bold("🍪 Jev (TypeSafe) for nomArmy's semantic checks\n"));
|
|
2730
|
+
console.log("It checks that a scout's cited lines support its finding, and that a worker's report matches its diff.");
|
|
2731
|
+
console.log("Its answers only add review flags; they never pass a check or allow a commit.");
|
|
2732
|
+
console.log(c.yellow("It sends excerpts of your code (findings, cited lines, diffs, worker reports) to TypeSafe.\n"));
|
|
2733
|
+
}
|
|
2734
|
+
const key = flag("key-stdin") ? await readStdin() : await readHiddenLine("TypeSafe API key (not shown): ");
|
|
2735
|
+
const saved = saveJevKey(key);
|
|
2736
|
+
let ok = false, why = null;
|
|
2737
|
+
try { ok = await testJev(jevSettings()); } catch (error) { why = error.message; }
|
|
2738
|
+
if (json) return out({ saved: true, keyFile: saved.keyFile, configPath: saved.configPath, test: ok ? "pass" : "fail", reason: why });
|
|
2739
|
+
console.log(c.green(`✓ Saved the key to ${saved.keyFile} (readable only by you) and turned Jev on in ${saved.configPath}.`));
|
|
2740
|
+
console.log(ok ? c.green("✓ Test call answered. New jobs use it; restart open coordinator sessions to pick it up.") : c.red(`✗ Test call failed: ${why ?? "no answer"}. Check the key, then: nomarmy validators test jev`));
|
|
2741
|
+
if (!ok) process.exitCode = 1;
|
|
2742
|
+
return;
|
|
2743
|
+
}
|
|
2744
|
+
if (sub === "test") {
|
|
2745
|
+
const settings = jevSettings();
|
|
2746
|
+
if (!settings) throw new Error("Jev isn't configured, or its key isn't readable. Add it with: nomarmy validators add jev");
|
|
2747
|
+
let ok = false, why = null;
|
|
2748
|
+
try { ok = await testJev(settings); } catch (error) { why = error.message; }
|
|
2749
|
+
if (json) return out({ test: ok ? "pass" : "fail", reason: why });
|
|
2750
|
+
console.log(ok ? c.green("✓ Jev answered.") : c.red(`✗ Jev test call failed: ${why ?? "no answer"}`));
|
|
2751
|
+
if (!ok) process.exitCode = 1;
|
|
2752
|
+
return;
|
|
2753
|
+
}
|
|
2754
|
+
if (sub === "remove") {
|
|
2755
|
+
const result = removeJev();
|
|
2756
|
+
if (json) return out(result);
|
|
2757
|
+
console.log(c.green(`✓ Jev is off${result.removedKey ? ", and its saved key is deleted" : ""}.`));
|
|
2758
|
+
return;
|
|
2759
|
+
}
|
|
2760
|
+
throw new Error("Usage: nomarmy validators <list|add jev|test jev|remove jev>");
|
|
2761
|
+
}
|
|
2762
|
+
|
|
2763
|
+
async function cmdValidatorsJudge(sub) {
|
|
2764
|
+
const agents = loadAgents(globalConfigDir()).agents;
|
|
2765
|
+
const resolve = () => judgeSettings({ agents, providerOf: agentProviderId, runsOnHost: agentRunsToolsOnHost });
|
|
2766
|
+
const probe = async (settings) => probeModel({ provider: settings.provider, model: settings.model, stateRoot: process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents") });
|
|
2767
|
+
if (sub === "add") {
|
|
2768
|
+
const agent = value("agent"), model = value("model");
|
|
2769
|
+
if (!agent || !model) throw new Error("Usage: nomarmy validators add judge --agent <name> --model <model> [--host-tools]");
|
|
2770
|
+
if (!agents[agent]) throw new Error(`"${agent}" isn't an agent in agents.yml. Agents: ${Object.keys(agents).join(", ") || "(none)"}`);
|
|
2771
|
+
if (agentRunsToolsOnHost(agents[agent]) && !flag("host-tools")) throw new Error(`agent "${agent}" runs its tools on this machine, and a judge reads text the worker wrote. Pass --host-tools to accept that, or pick a sandboxed agent (an api key, Codex, Muse).`);
|
|
2772
|
+
const saved = saveJudge({ agent, model, hostTools: flag("host-tools") });
|
|
2773
|
+
const settings = resolve();
|
|
2774
|
+
if (settings?.problem) throw new Error(settings.problem);
|
|
2775
|
+
const test = await probe(settings);
|
|
2776
|
+
if (json) return out({ saved: true, configPath: saved.configPath, test: test.ok ? "pass" : test.refused ? "refused" : "inconclusive", reason: test.reason });
|
|
2777
|
+
console.log(c.green(`✓ The judge is ${agent}/${model}, in ${saved.configPath}.`));
|
|
2778
|
+
console.log(test.ok ? c.green("✓ Test call answered. New implement jobs use it; restart open coordinator sessions to pick it up.") : c.red(`✗ Test call ${test.refused ? "refused" : "didn't answer"}: ${test.reason ?? "no answer"}`));
|
|
2779
|
+
if (!test.ok) process.exitCode = 1;
|
|
2780
|
+
return;
|
|
2781
|
+
}
|
|
2782
|
+
if (sub === "test") {
|
|
2783
|
+
const settings = resolve();
|
|
2784
|
+
if (!settings) throw new Error("No judge configured. Add one with: nomarmy validators add judge --agent <name> --model <model>");
|
|
2785
|
+
if (settings.problem) throw new Error(settings.problem);
|
|
2786
|
+
const test = await probe(settings);
|
|
2787
|
+
if (json) return out({ test: test.ok ? "pass" : "fail", reason: test.reason });
|
|
2788
|
+
console.log(test.ok ? c.green(`✓ ${settings.agent}/${settings.model} answered.`) : c.red(`✗ ${test.reason ?? "no answer"}`));
|
|
2789
|
+
if (!test.ok) process.exitCode = 1;
|
|
2790
|
+
return;
|
|
2791
|
+
}
|
|
2792
|
+
if (sub === "remove") {
|
|
2793
|
+
removeJudge();
|
|
2794
|
+
return json ? out({ removed: true }) : console.log(c.green("✓ The judge is off."));
|
|
2795
|
+
}
|
|
2796
|
+
throw new Error("Usage: nomarmy validators <add judge --agent <name> --model <model> [--host-tools]|test judge|remove judge>");
|
|
2797
|
+
}
|
|
2798
|
+
|
|
2649
2799
|
// `nomarmy mcp`: what a --scope project registration runs. Nothing goes to
|
|
2650
2800
|
// stdout but the server's own protocol.
|
|
2651
2801
|
function cmdMcp() {
|
|
@@ -2655,7 +2805,7 @@ function cmdMcp() {
|
|
|
2655
2805
|
child.on("exit", (code, signal) => { if (signal) process.kill(process.pid, signal); else process.exit(code ?? 1); });
|
|
2656
2806
|
}
|
|
2657
2807
|
|
|
2658
|
-
const commands = { mcp: cmdMcp, scan: cmdScan, validate: cmdValidate, sizing: cmdSizing, init: cmdInit, setup: cmdSetup, install: cmdInstall, model: cmdModel, agents: cmdAgents, army: cmdArmy, jobs: cmdJobs, statusline: cmdStatusline, health: cmdHealth, config: cmdConfig, update: cmdUpdate, connect: cmdConnect, sandbox: cmdSandbox, start: cmdStart, stop: cmdStop, uninstall: cmdUninstall, help: () => usage(0) };
|
|
2808
|
+
const commands = { stats: cmdStats, validators: cmdValidators, mcp: cmdMcp, scan: cmdScan, validate: cmdValidate, sizing: cmdSizing, init: cmdInit, setup: cmdSetup, install: cmdInstall, model: cmdModel, agents: cmdAgents, army: cmdArmy, jobs: cmdJobs, statusline: cmdStatusline, health: cmdHealth, config: cmdConfig, update: cmdUpdate, connect: cmdConnect, sandbox: cmdSandbox, start: cmdStart, stop: cmdStop, uninstall: cmdUninstall, help: () => usage(0) };
|
|
2659
2809
|
// doctor command
|
|
2660
2810
|
async function cmdDoctor() {
|
|
2661
2811
|
// Import lazily to avoid circular dependencies
|
package/lib/admission.mjs
CHANGED
|
@@ -126,6 +126,19 @@ export function createJobRuntime(deps) {
|
|
|
126
126
|
try { return await fn(); } finally { slot.release(); }
|
|
127
127
|
})();
|
|
128
128
|
}
|
|
129
|
+
// Who ran the job and for whom, onto its finished record, so `nomarmy
|
|
130
|
+
// stats` can count by role and agent, and by repo for verify runs too
|
|
131
|
+
// (their records didn't carry the repo). Best-effort; never fails the job.
|
|
132
|
+
function stampJobRecord(entry, result) {
|
|
133
|
+
const file = path.join(result?.jobDir ?? path.join(jobsRoot, entry.jobId), "metadata.json");
|
|
134
|
+
try {
|
|
135
|
+
const record = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
136
|
+
record.labels = { role: entry.role ?? null, agent: entry.agent ?? null, label: entry.label ?? null, runId: entry.runId ?? null };
|
|
137
|
+
record.projectDir = record.projectDir ?? projectDir;
|
|
138
|
+
fs.writeFileSync(`${file}.tmp`, JSON.stringify(record, null, 2));
|
|
139
|
+
fs.renameSync(`${file}.tmp`, file);
|
|
140
|
+
} catch { /* no record (a refused or crashed job): nothing to label */ }
|
|
141
|
+
}
|
|
129
142
|
function track(jobId, meta, promise) {
|
|
130
143
|
const entry = { ...meta, jobId, startedAt: new Date().toISOString(), settled: false, result: null, error: null, promise: null };
|
|
131
144
|
// A machine-wide lease for as long as the job runs, so every session's
|
|
@@ -134,7 +147,7 @@ export function createJobRuntime(deps) {
|
|
|
134
147
|
if (meta.lane) writeLease(leasesRoot, jobId, { lane: meta.lane, agent: meta.agent ?? null, runId: meta.runId ?? null, role: meta.role ?? null, model: meta.model ?? null, repo: projectDir });
|
|
135
148
|
const release = () => removeLease(leasesRoot, jobId);
|
|
136
149
|
entry.promise = promise.then(
|
|
137
|
-
r => { entry.settled = true; entry.result = r; release(); notifyJobFinished(entry, r, null); return r; },
|
|
150
|
+
r => { entry.settled = true; entry.result = r; release(); stampJobRecord(entry, r); notifyJobFinished(entry, r, null); return r; },
|
|
138
151
|
e => { entry.settled = true; entry.error = e; release(); notifyJobFinished(entry, null, e); throw e; });
|
|
139
152
|
entry.promise.catch(() => {});
|
|
140
153
|
activeJobs.set(jobId, entry);
|
package/lib/execute.mjs
CHANGED
|
@@ -9,6 +9,9 @@ import { parseScoutReport, verifyCitations, resolveScoutOutcome, renderScoutRepo
|
|
|
9
9
|
import { parseDecomposeReport, buildDecomposeFindings, resolveDecomposeOutcome, checkDecompositionOverlap, renderDecomposeReport } from "./decompose.mjs";
|
|
10
10
|
import { deriveTimeBudget } from "./budget.mjs";
|
|
11
11
|
import { continuationProblem, continuationBase, snapshotRetainedWork, applyRetainedWork, continuationNote } from "./continue-from.mjs";
|
|
12
|
+
import { checkScoutCitations, checkReportClaims } from "./jev-checks.mjs";
|
|
13
|
+
import { runJudge } from "./judge.mjs";
|
|
14
|
+
import { pickMutants, runMutants, describeSurvivors } from "./mutation.mjs";
|
|
12
15
|
import { estimateDisplacement } from "./transcript.mjs";
|
|
13
16
|
import { outlineFile, findReferences } from "./repo-query.mjs";
|
|
14
17
|
import { loadConfig } from "./config.mjs";
|
|
@@ -18,7 +21,7 @@ import { describeRecoveryChanges, reportRecoveryPrompt } from "./worker-prompt.m
|
|
|
18
21
|
import { parseWorkerReport } from "./report.mjs";
|
|
19
22
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "./outcomes.mjs";
|
|
20
23
|
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy } from "./outcome.mjs";
|
|
21
|
-
import { isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
24
|
+
import { parseAddedLineNumbers, isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
22
25
|
|
|
23
26
|
// ---------------------------------------------------------------------------
|
|
24
27
|
// Job status for polling. `status.json` is written at every phase transition
|
|
@@ -36,6 +39,8 @@ export function createExecutor(deps) {
|
|
|
36
39
|
resolveReasoningApplied, recordedBudgets, runOpenClaw, sweepStaleSandboxContainers,
|
|
37
40
|
verificationFlow, normalizeVerification, runIndependentVerification,
|
|
38
41
|
runRegressionCheck, repoPolicy } = deps;
|
|
42
|
+
// Optional Jev checks (lib/validators.mjs): the server wires the settings; without them (tests), none run.
|
|
43
|
+
const jevSettingsFor = (check) => { try { const s = deps.jevSettings?.(); return s?.checks?.includes(check) ? s : null; } catch { return null; } };
|
|
39
44
|
|
|
40
45
|
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, jobId: presetJobId = null }) {
|
|
41
46
|
await assertRepo();
|
|
@@ -300,6 +305,36 @@ export function createExecutor(deps) {
|
|
|
300
305
|
if (regressionCheck.status === "restore_failed") regressionCheckFatal = true;
|
|
301
306
|
}
|
|
302
307
|
|
|
308
|
+
// Mutation testing (lib/mutation.mjs), when this repo opts in: small
|
|
309
|
+
// mistakes planted one at a time in the changed lines must each fail
|
|
310
|
+
// the same profile. Survivors raise review; a failed restore is fatal.
|
|
311
|
+
let mutation = null, mutationElapsedMs = null;
|
|
312
|
+
const mutationConfig = (() => { try { return loadConfig(projectDir)?.config?.mutation ?? null; } catch { return null; } })();
|
|
313
|
+
if (mutationConfig && verification && independentVerification.status === "pass" && codeFilesChanged.length > 0 && !regressionCheckFatal) {
|
|
314
|
+
const mutationStartedMs = Date.now();
|
|
315
|
+
try {
|
|
316
|
+
const untracked = new Set((preCommit.nameStatus ?? []).filter((e) => e.untracked).map((e) => e.path));
|
|
317
|
+
const deleted = new Set((preCommit.nameStatus ?? []).filter((e) => /^D/.test(e.status)).map((e) => e.path));
|
|
318
|
+
const files = [];
|
|
319
|
+
for (const file of codeFilesChanged.filter((f) => !deleted.has(f))) {
|
|
320
|
+
const full = path.join(cwd, file);
|
|
321
|
+
let lines;
|
|
322
|
+
if (untracked.has(file)) { try { lines = fs.readFileSync(full, "utf8").split("\n").map((_, i) => i + 1); } catch { continue; } }
|
|
323
|
+
else lines = parseAddedLineNumbers(await gitRaw(["diff", "-U0", base.sha, "--", file], cwd));
|
|
324
|
+
if (lines.length) files.push({ path: file, full, lines });
|
|
325
|
+
}
|
|
326
|
+
const mutants = pickMutants(files, mutationConfig.mutants);
|
|
327
|
+
let n = 0;
|
|
328
|
+
mutation = await runMutants({ mutants, deadlineMs: mutationStartedMs + mutationConfig.max_seconds * 1000,
|
|
329
|
+
verify: () => runIndependentVerification({ profile: verification, cwd, jobId: `${jobId}-mutant-${++n}`, baseSha: base.sha, branch, mode, record: preCommit }) });
|
|
330
|
+
mutation.planned = mutants.length;
|
|
331
|
+
} catch (error) {
|
|
332
|
+
mutation = { status: "not_run", killed: 0, survived: [], inconclusive: 0, tried: 0, reason: `mutation testing failed to run: ${error.message}` };
|
|
333
|
+
}
|
|
334
|
+
mutationElapsedMs = Date.now() - mutationStartedMs;
|
|
335
|
+
fs.appendFileSync(path.join(jobDir, "coordinator.log"), `${new Date().toISOString()} mutation testing: ${mutation.killed ?? 0} killed, ${mutation.survived?.length ?? 0} survived, ${mutation.inconclusive ?? 0} inconclusive of ${mutation.tried ?? 0} tried (${Math.round(mutationElapsedMs / 1000)}s)\n`);
|
|
336
|
+
}
|
|
337
|
+
|
|
303
338
|
// resolveOutcome's own contract only ever sees pass/fail/not_run for
|
|
304
339
|
// regressionCheck -- a restore_failed status is substituted to not_run
|
|
305
340
|
// here so resolveOutcome never needs a fourth value; the hard override
|
|
@@ -427,21 +462,67 @@ export function createExecutor(deps) {
|
|
|
427
462
|
commitBlockedReason: `possible secret detected: ${possibleSecrets.reason}`,
|
|
428
463
|
reasons: [...afterHostInstalls.reasons, `POSSIBLE SECRET DETECTED: ${possibleSecrets.reason}`] }
|
|
429
464
|
: afterHostInstalls;
|
|
465
|
+
// Jev: does the worker's report match its diff? Found live: a note said
|
|
466
|
+
// "restored check.js to base commit" while the diff rewrote check.js.
|
|
467
|
+
// Only raises review; it never blocks or allows a commit.
|
|
468
|
+
let jevClaims = null, judged = null;
|
|
469
|
+
const jevImplement = jevSettingsFor("report-claims");
|
|
470
|
+
const judge = (() => { try { return deps.judgeSettings?.() ?? null; } catch { return null; } })();
|
|
471
|
+
// The whole change, new files included, for whichever validators run.
|
|
472
|
+
const jobDiff = async () => {
|
|
473
|
+
let diff = await gitRaw(["diff", base.sha, "--"], cwd);
|
|
474
|
+
for (const entry of (preCommit.nameStatus ?? []).filter((e) => e.untracked)) {
|
|
475
|
+
let text = "";
|
|
476
|
+
try { text = fs.readFileSync(path.join(cwd, entry.path), "utf8"); } catch { continue; }
|
|
477
|
+
diff += `\ndiff --git a/${entry.path} b/${entry.path}\nnew file\n--- /dev/null\n+++ b/${entry.path}\n${text.split("\n").map((l) => `+${l}`).join("\n")}\n`;
|
|
478
|
+
}
|
|
479
|
+
return diff;
|
|
480
|
+
};
|
|
481
|
+
let diffText = null;
|
|
482
|
+
if (mode === "implement" && reportValidation?.valid && repositoryChanged && (jevImplement || (judge && !judge.problem))) {
|
|
483
|
+
try { diffText = await jobDiff(); } catch { diffText = null; }
|
|
484
|
+
}
|
|
485
|
+
if (jevImplement && diffText != null) {
|
|
486
|
+
try { jevClaims = await checkReportClaims({ report: reportValidation, diff: diffText, settings: jevImplement }); }
|
|
487
|
+
catch (error) { jevClaims = { flag: null, verdict: null, error: error.message, usage: 0 }; }
|
|
488
|
+
}
|
|
489
|
+
// The model judge (lib/judge.mjs): acceptance criteria, the report and
|
|
490
|
+
// changed tests. Only raises review, never blocks or allows a commit.
|
|
491
|
+
if (judge?.problem) judged = { flags: [], answer: null, error: `judge not run: ${judge.problem}`, skipped: true };
|
|
492
|
+
else if (judge && diffText != null) {
|
|
493
|
+
const modifiedTests = preCommit.testChanges?.existing_tests_modified ?? [];
|
|
494
|
+
let testDiff = "";
|
|
495
|
+
if (modifiedTests.length) { try { testDiff = await gitRaw(["diff", base.sha, "--", ...modifiedTests], cwd); } catch { testDiff = ""; } }
|
|
496
|
+
judged = await runJudge({ settings: judge, task, acceptance: acceptance ?? [], report: reportValidation, diff: diffText, testDiff, stateRoot: path.join(jobsRoot, "..") });
|
|
497
|
+
}
|
|
498
|
+
const afterJevOnly = jevClaims?.flag
|
|
499
|
+
? { ...afterSecrets, reviewRequired: true, reasons: [...afterSecrets.reasons, `REPORT MAY NOT MATCH THE DIFF (Jev, ${jevClaims.flag.probability.toFixed(2)}): the report says "${String(reportValidation.note ?? "").slice(0, 200)}", and the diff may show otherwise. Read the diff before accepting.`] }
|
|
500
|
+
: afterSecrets;
|
|
501
|
+
const afterJev = judged?.flags?.length
|
|
502
|
+
? { ...afterJevOnly, reviewRequired: true, reasons: [...afterJevOnly.reasons, `JUDGE (${judge.agent}/${judge.model}): ${judged.flags.join("; ")}. Read the diff before accepting.`] }
|
|
503
|
+
: afterJevOnly;
|
|
504
|
+
|
|
430
505
|
// A worker must not be judged by a check it rewrote: a changed script,
|
|
431
506
|
// Makefile or package.json script that a verification command runs
|
|
432
507
|
// blocks the commit; changed test-runner config only asks for review.
|
|
433
508
|
const blockedInputs = verificationInputs?.blocked ?? [], flaggedInputs = verificationInputs?.flagged ?? [];
|
|
434
509
|
const inputLine = blockedInputs.map((b) => `${b.file} (${b.why})`).join("; ");
|
|
435
510
|
const afterInputs = blockedInputs.length
|
|
436
|
-
? { ...
|
|
511
|
+
? { ...afterJev, outcome: afterJev.commitAllowed || afterJev.outcome === OUTCOMES.WORKER_DONE ? OUTCOMES.NEEDS_REVIEW : afterJev.outcome,
|
|
437
512
|
reviewRequired: true, commitAllowed: false,
|
|
438
|
-
commitBlockedReason:
|
|
439
|
-
reasons: [...
|
|
440
|
-
:
|
|
513
|
+
commitBlockedReason: afterJev.commitAllowed ? `the diff changes what verification runs: ${inputLine}` : afterJev.commitBlockedReason,
|
|
514
|
+
reasons: [...afterJev.reasons, `VERIFICATION INPUT CHANGED: the diff changes what profile '${verification}' runs, so its result can't be trusted: ${inputLine}`] }
|
|
515
|
+
: afterJev;
|
|
441
516
|
const afterConfig = flaggedInputs.length
|
|
442
517
|
? { ...afterInputs, reviewRequired: true, reasons: [...afterInputs.reasons, `TEST CONFIG CHANGED: ${flaggedInputs.map((c) => `${c.file} (${c.why})`).join("; ")}`] }
|
|
443
518
|
: afterInputs;
|
|
444
|
-
const
|
|
519
|
+
const afterMutation = mutation?.status === "restore_failed"
|
|
520
|
+
? { ...afterConfig, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true, commitAllowed: false,
|
|
521
|
+
commitBlockedReason: `mutation testing could not restore the worker's file: ${mutation.reason}`, reasons: [...afterConfig.reasons, `MUTATION RESTORE FAILED: ${mutation.reason}`] }
|
|
522
|
+
: mutation?.status === "survivors"
|
|
523
|
+
? { ...afterConfig, reviewRequired: true, reasons: [...afterConfig.reasons, describeSurvivors(mutation, verification)] }
|
|
524
|
+
: afterConfig;
|
|
525
|
+
const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterMutation, independentVerification.status, repoPolicy()),
|
|
445
526
|
{ refactor, verificationStatus: independentVerification.status, testChanges: preCommit.testChanges });
|
|
446
527
|
|
|
447
528
|
progress("commit");
|
|
@@ -454,6 +535,8 @@ export function createExecutor(deps) {
|
|
|
454
535
|
const issues = [...finalOutcome.reasons, ...(preCommit.issues ?? [])];
|
|
455
536
|
if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
|
|
456
537
|
if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
|
|
538
|
+
if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
|
|
539
|
+
if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
|
|
457
540
|
if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
|
|
458
541
|
if (repositoryChanged && !commit.created) {
|
|
459
542
|
if (coordinatorStatus === "complete") coordinatorStatus = "incomplete";
|
|
@@ -483,6 +566,8 @@ export function createExecutor(deps) {
|
|
|
483
566
|
const metrics = buildMetrics({ result: result ?? attempted, record, reportValidation, outcome: finalOutcome, workerElapsedMs, totalElapsedMs: Date.now() - jobStartedMs, regressionCheckElapsedMs, transientAbortRetried });
|
|
484
567
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree, branch, startedAt, finishedAt,
|
|
485
568
|
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}),
|
|
569
|
+
...(mutation ? { mutation: { ...mutation, elapsedSeconds: Math.round((mutationElapsedMs ?? 0) / 1000) } } : {}),
|
|
570
|
+
...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
|
|
486
571
|
outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
|
|
487
572
|
reportRecoveryAttempted, reportRecovered,
|
|
488
573
|
reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
|
|
@@ -609,7 +694,16 @@ export function createExecutor(deps) {
|
|
|
609
694
|
const dirty = record.repoStatusFiles.length > 0;
|
|
610
695
|
const readFile = async p => { try { return await gitRaw(["show", `${base.sha}:${p}`], projectDir); } catch { return null; } };
|
|
611
696
|
const verified = await verifyCitations(report.findings, { readFile, limits: used.scout });
|
|
697
|
+
// Jev: do the cited lines support each finding? Only adds flags.
|
|
698
|
+
let jevCitations = null;
|
|
699
|
+
const jevScout = jevSettingsFor("scout-citations");
|
|
700
|
+
if (jevScout && verified?.findings?.length) {
|
|
701
|
+
try { jevCitations = await checkScoutCitations({ findings: verified.findings, settings: jevScout, readFile }); }
|
|
702
|
+
catch (error) { jevCitations = { flags: [], checked: 0, errors: [error.message], usage: 0, verdicts: [] }; }
|
|
703
|
+
for (const f of jevCitations.flags) verified.findings[f.index].jev = { verdict: f.verdict, probability: f.probability };
|
|
704
|
+
}
|
|
612
705
|
const outcome = resolveScoutOutcome({ report, verified, workerFailed, workerTimedOut, dirty });
|
|
706
|
+
if (jevCitations?.flags.length) outcome.reviewRequired = true;
|
|
613
707
|
|
|
614
708
|
progress("record");
|
|
615
709
|
if (outcome.retainWorktree) worktreeRetained = true;
|
|
@@ -621,6 +715,8 @@ export function createExecutor(deps) {
|
|
|
621
715
|
if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
|
|
622
716
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
|
|
623
717
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
718
|
+
if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
|
|
719
|
+
if (jevCitations?.errors.length) issues.push(`Jev citation check skipped or incomplete (${jevCitations.errors.join("; ")}); this job's result doesn't depend on it`);
|
|
624
720
|
if (reportRecoveryAttempted) {
|
|
625
721
|
issues.push(reportRecovered
|
|
626
722
|
? "scout report recovered via a follow-up call after the first reply was cut off"
|
|
@@ -658,6 +754,7 @@ export function createExecutor(deps) {
|
|
|
658
754
|
};
|
|
659
755
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
660
756
|
objective: task, mustCover: acceptance ?? [],
|
|
757
|
+
...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
|
|
661
758
|
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
662
759
|
scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
|
|
663
760
|
findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
// nomArmy's Jev checks (see lib/validators.mjs): narrow judgments on evidence
|
|
2
|
+
// nomArmy already has, each able only to add a review flag.
|
|
3
|
+
//
|
|
4
|
+
// scout-citations nomArmy verifies a scout's [path:line] exists and attaches
|
|
5
|
+
// the lines; this asks whether those lines support the
|
|
6
|
+
// finding (the shape of TypeSafe's citation-check cookbook).
|
|
7
|
+
// report-claims whether a worker's report matches its diff. Found live: a
|
|
8
|
+
// worker's note said "restored check.js to base commit"
|
|
9
|
+
// while its diff rewrote check.js.
|
|
10
|
+
|
|
11
|
+
import { askJev, jevBreaker, tripJevBreaker } from "./validators.mjs";
|
|
12
|
+
|
|
13
|
+
// A flag needs the model to lean clearly; the rest is left to the General.
|
|
14
|
+
export const FLAG_AT = 0.7;
|
|
15
|
+
const DIFF_CHARS = 60000; // well inside the 32k-token state budget
|
|
16
|
+
const MAX_FINDINGS = 24;
|
|
17
|
+
// The report's excerpts are capped at 12 lines for the General's context; Jev
|
|
18
|
+
// judges the whole cited range (to this cap), or a long range looks
|
|
19
|
+
// unrelated when the supporting line is past the excerpt.
|
|
20
|
+
const CITED_LINES = 80;
|
|
21
|
+
const CONCURRENCY = 4;
|
|
22
|
+
// However many findings, a job never waits longer than this on Jev.
|
|
23
|
+
const JOB_BUDGET_MS = 45000;
|
|
24
|
+
|
|
25
|
+
// The answer, or null with the breaker tripped: any failure means TypeSafe
|
|
26
|
+
// isn't answering well right now, so stop asking (lib/validators.mjs).
|
|
27
|
+
async function guarded(ask, request, errors) {
|
|
28
|
+
if (jevBreaker().open) { errors.push(`skipped: Jev failed recently (${jevBreaker().reason}); retrying after ${jevBreaker().retryAt}`); return null; }
|
|
29
|
+
try { return await ask(request); }
|
|
30
|
+
catch (error) { const why = error?.name === "AbortError" ? "timed out" : error.message; tripJevBreaker(why); errors.push(why); return null; }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
async function mapLimit(items, limit, fn) {
|
|
34
|
+
const out = new Array(items.length);
|
|
35
|
+
let next = 0;
|
|
36
|
+
await Promise.all(Array.from({ length: Math.min(limit, items.length) }, async () => {
|
|
37
|
+
while (next < items.length) { const i = next++; out[i] = await fn(items[i], i); }
|
|
38
|
+
}));
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const SUPPORT_QUESTION = {
|
|
43
|
+
type: "choice",
|
|
44
|
+
instructions: "A code researcher made the claim in `finding` and cited the source lines in `cited` as evidence. Judge only what the cited lines show, not whether the claim might be true elsewhere in the codebase. How do the cited lines relate to the claim?",
|
|
45
|
+
criteria: {
|
|
46
|
+
supports: "The cited lines directly show what the claim says.",
|
|
47
|
+
contradicts: "The cited lines show something different from, or opposite to, the claim.",
|
|
48
|
+
unrelated: "The cited lines don't address the claim: they're about something else, or too little of the claim is visible in them.",
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* For each finding with verified cited lines: does the evidence support it?
|
|
54
|
+
* @returns {Promise<{ flags: {index: number, verdict: string, probability: number}[], checked: number, errors: string[], usage: number }>}
|
|
55
|
+
*/
|
|
56
|
+
export async function checkScoutCitations({ findings = [], settings, ask = askJev, readFile = null }) {
|
|
57
|
+
const fileCache = new Map();
|
|
58
|
+
const fullRange = async (c) => {
|
|
59
|
+
if (!readFile) return null;
|
|
60
|
+
if (!fileCache.has(c.path)) fileCache.set(c.path, readFile(c.path).then((t) => (typeof t === "string" ? t.split("\n") : null)).catch(() => null));
|
|
61
|
+
const lines = await fileCache.get(c.path);
|
|
62
|
+
if (!lines) return null;
|
|
63
|
+
const end = Math.min(c.end, c.start + CITED_LINES - 1, lines.length);
|
|
64
|
+
return Array.from({ length: Math.max(0, end - c.start + 1) }, (_, k) => `${c.start + k}: ${lines[c.start + k - 1]}`).join("\n");
|
|
65
|
+
};
|
|
66
|
+
const candidates = findings.map((f, index) => ({ f, index }))
|
|
67
|
+
.filter(({ f }) => (f.citations ?? []).some((c) => c.status === "ok" && c.excerpt?.length)).slice(0, MAX_FINDINGS);
|
|
68
|
+
const errors = [];
|
|
69
|
+
let usage = 0;
|
|
70
|
+
const deadline = Date.now() + JOB_BUDGET_MS;
|
|
71
|
+
const verdicts = await mapLimit(candidates, CONCURRENCY, async ({ f, index }) => {
|
|
72
|
+
if (Date.now() > deadline) { errors.push("skipped the rest: over the job's time budget for Jev"); return null; }
|
|
73
|
+
const cited = await Promise.all(f.citations.filter((c) => c.status === "ok" && c.excerpt?.length)
|
|
74
|
+
.map(async (c) => ({ file: c.path, lines: `${c.start}-${c.end}`, text: (await fullRange(c)) ?? c.excerpt.map((l) => `${l.line}: ${l.text}`).join("\n") })));
|
|
75
|
+
const r = await guarded(ask, { key: settings.key, model: settings.model, state: { finding: f.text, cited }, questions: { support: SUPPORT_QUESTION } }, errors);
|
|
76
|
+
if (!r) return null;
|
|
77
|
+
usage += r.usage?.input_tokens ?? 0;
|
|
78
|
+
const a = r.answers?.support;
|
|
79
|
+
return a ? { index, verdict: a.choice, probability: a.probabilities?.[a.choice] ?? 0 } : null;
|
|
80
|
+
});
|
|
81
|
+
const flags = verdicts.filter((v) => v && v.verdict !== "supports" && v.probability >= FLAG_AT);
|
|
82
|
+
return { flags, checked: verdicts.filter(Boolean).length, errors: [...new Set(errors)].slice(0, 3), usage, verdicts: verdicts.filter(Boolean) };
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const CLAIMS_QUESTION = {
|
|
86
|
+
type: "choice",
|
|
87
|
+
instructions: "A coding worker wrote the report in `report` about the change it made. `diff` shows every change from the base commit it started from: a file that isn't in the diff is exactly as it was at the base commit, and a file restored to the base commit wouldn't appear. Judge only the concrete things the report says were done or changed. Does the diff match them?",
|
|
88
|
+
criteria: {
|
|
89
|
+
consistent: "The concrete things the report says were done are what the diff shows.",
|
|
90
|
+
contradicts: "The diff shows a concrete claim in the report is false: something said to be done, restored, removed or left alone wasn't, or was done differently.",
|
|
91
|
+
unclear: "The report makes no concrete claim the diff can confirm or refute, or the diff shown is too partial to tell.",
|
|
92
|
+
},
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Does the worker's report match its diff?
|
|
97
|
+
* @returns {Promise<{ flag: {verdict: string, probability: number}|null, verdict: object|null, error: string|null, usage: number, truncated: boolean }>}
|
|
98
|
+
*/
|
|
99
|
+
export async function checkReportClaims({ report, diff, settings, ask = askJev }) {
|
|
100
|
+
const note = [report?.note, report?.notDone && report.notDone !== "none" ? `Not done: ${report.notDone}` : null].filter(Boolean).join("\n");
|
|
101
|
+
if (!note.trim() || !String(diff ?? "").trim()) return { flag: null, verdict: null, error: null, usage: 0, truncated: false };
|
|
102
|
+
const truncated = diff.length > DIFF_CHARS;
|
|
103
|
+
const state = { report: { status: report.status ?? null, tests: report.tests ?? null, note }, diff: truncated ? `${diff.slice(0, DIFF_CHARS)}\n[diff truncated]` : diff };
|
|
104
|
+
const errors = [];
|
|
105
|
+
const r = await guarded(ask, { key: settings.key, model: settings.model, state, questions: { claims: CLAIMS_QUESTION } }, errors);
|
|
106
|
+
if (!r) return { flag: null, verdict: null, error: errors[0] ?? "no answer", usage: 0, truncated };
|
|
107
|
+
const a = r.answers?.claims;
|
|
108
|
+
const verdict = a ? { verdict: a.choice, probability: a.probabilities?.[a.choice] ?? 0 } : null;
|
|
109
|
+
return { flag: verdict && verdict.verdict === "contradicts" && verdict.probability >= FLAG_AT ? verdict : null, verdict, error: null, usage: r.usage?.input_tokens ?? 0, truncated };
|
|
110
|
+
}
|