nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -480
- package/bin/nomarmy.mjs +1081 -185
- package/docker/Dockerfile +2 -2
- package/docker/Dockerfile.go +6 -4
- package/docker/Dockerfile.rust +17 -2
- package/harnesses/_template/README.md +27 -0
- package/harnesses/_template/harness.yml +26 -0
- package/harnesses/browser-playwright/README.md +35 -0
- package/harnesses/browser-playwright/fixture/package.json +1 -0
- package/harnesses/browser-playwright/fixture/page.html +1 -0
- package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
- package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
- package/harnesses/browser-playwright/harness.yml +18 -0
- package/harnesses/go/README.md +45 -0
- package/harnesses/go/harness.yml +14 -0
- package/harnesses/mock-oidc/README.md +31 -0
- package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
- package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
- package/harnesses/mock-oidc/harness.yml +19 -0
- package/harnesses/node/README.md +53 -0
- package/harnesses/node/harness.yml +18 -0
- package/harnesses/python/README.md +46 -0
- package/harnesses/python/harness.yml +16 -0
- package/harnesses/rust/README.md +45 -0
- package/harnesses/rust/harness.yml +13 -0
- package/install.sh +29 -9
- package/lib/admission.mjs +178 -30
- package/lib/agents.mjs +8 -6
- package/lib/army.mjs +25 -10
- package/lib/codex-link.mjs +37 -0
- package/lib/config.mjs +15 -0
- package/lib/connect.mjs +232 -19
- package/lib/continue-from.mjs +103 -0
- package/lib/coordinator-instructions.mjs +5 -1
- package/lib/diff-checks.mjs +114 -0
- package/lib/dispatch-schema.mjs +14 -12
- package/lib/doctor.mjs +98 -9
- package/lib/egress-proxy.mjs +116 -0
- package/lib/execute.mjs +241 -33
- package/lib/git-record.mjs +27 -3
- package/lib/harness-schema.mjs +61 -0
- package/lib/harnesses.mjs +99 -0
- package/lib/health.mjs +162 -18
- package/lib/install-freshness.mjs +114 -0
- package/lib/jev-checks.mjs +110 -0
- package/lib/job-format.mjs +54 -0
- package/lib/judge.mjs +130 -0
- package/lib/limits.mjs +77 -0
- package/lib/model-probe.mjs +61 -0
- package/lib/mutation.mjs +159 -0
- package/lib/notify.mjs +30 -3
- package/lib/openclaw-install.mjs +122 -0
- package/lib/openclaw-path.mjs +28 -0
- package/lib/openclaw-run.mjs +74 -12
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/outcome.mjs +21 -2
- package/lib/outcomes.mjs +6 -0
- package/lib/path-utils.mjs +4 -0
- package/lib/podman-health.mjs +41 -0
- package/lib/process.mjs +4 -1
- package/lib/propose.mjs +10 -11
- package/lib/refusal-retry.mjs +16 -0
- package/lib/registry-python.mjs +98 -0
- package/lib/registry-secrets.mjs +140 -0
- package/lib/repo-query.mjs +13 -7
- package/lib/runs.mjs +7 -1
- package/lib/same-path.mjs +14 -0
- package/lib/sandbox-images.mjs +499 -83
- package/lib/sandbox-vm.mjs +32 -0
- package/lib/scan.mjs +5 -1
- package/lib/schema.mjs +20 -11
- package/lib/scout.mjs +21 -3
- package/lib/server-context.mjs +21 -1
- package/lib/setup-steps.mjs +55 -0
- package/lib/share.mjs +82 -0
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +315 -0
- package/lib/statusline.mjs +32 -6
- package/lib/subscription-setup.mjs +13 -0
- package/lib/suggestions.mjs +153 -0
- package/lib/thinking.mjs +23 -0
- package/lib/transcript.mjs +30 -5
- package/lib/usage-limits.mjs +329 -0
- package/lib/user-config.mjs +106 -0
- package/lib/validators.mjs +220 -0
- package/lib/verification-artifacts.mjs +46 -0
- package/lib/verification-flow.mjs +52 -7
- package/lib/verification-network.mjs +66 -0
- package/lib/verify.mjs +338 -85
- package/lib/worker-prompt.mjs +5 -2
- package/lib/wsl-cli.mjs +152 -0
- package/lib/wsl.mjs +230 -0
- package/lib/zod-issues.mjs +15 -0
- package/mcp/server.mjs +165 -34
- package/package.json +7 -5
- package/playbooks/feature.md +8 -5
- package/scripts/configure-openclaw.sh +4 -2
- package/scripts/generate-harness-docs.mjs +42 -0
- package/scripts/install-openclaw.mjs +23 -0
- package/scripts/lib.sh +9 -2
- package/scripts/select-model.mjs +12 -5
- package/scripts/start-inference.sh +2 -2
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { parse } from "yaml";
|
|
5
|
+
import { harnessSchemaFor } from "./harness-schema.mjs";
|
|
6
|
+
import { nodeDependencyFiles } from "./sandbox-images.mjs";
|
|
7
|
+
|
|
8
|
+
export const HARNESS_ROOT = fileURLToPath(new URL("../harnesses/", import.meta.url));
|
|
9
|
+
|
|
10
|
+
// Resolve the whole graph before publishing any specs, including dependents
|
|
11
|
+
// of broken harnesses. A partial layer stack is not a usable harness.
|
|
12
|
+
function orderedNames(harnesses) {
|
|
13
|
+
const order = [], state = new Map(), failures = new Map();
|
|
14
|
+
function visit(name, trail = []) {
|
|
15
|
+
if (state.get(name) === "done") return !failures.has(name);
|
|
16
|
+
if (state.get(name) === "visiting") {
|
|
17
|
+
const cycle = [...trail.slice(trail.indexOf(name)), name];
|
|
18
|
+
for (const member of cycle) failures.set(member, `after cycle: ${cycle.join(" -> ")}`);
|
|
19
|
+
return false;
|
|
20
|
+
}
|
|
21
|
+
state.set(name, "visiting");
|
|
22
|
+
for (const parent of harnesses[name].after) {
|
|
23
|
+
if (!Object.hasOwn(harnesses, parent)) failures.set(name, `unknown after harness: ${parent}`);
|
|
24
|
+
else if (!visit(parent, [...trail, name]) && !failures.has(name)) failures.set(name, `invalid after harness: ${parent}`);
|
|
25
|
+
}
|
|
26
|
+
state.set(name, "done");
|
|
27
|
+
if (!failures.has(name)) order.push(name);
|
|
28
|
+
return !failures.has(name);
|
|
29
|
+
}
|
|
30
|
+
for (const name of Object.keys(harnesses).sort()) visit(name);
|
|
31
|
+
return { order, failures };
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function loadHarnesses(root = HARNESS_ROOT) {
|
|
35
|
+
const harnesses = {}, problems = [];
|
|
36
|
+
let entries;
|
|
37
|
+
try { entries = fs.readdirSync(root, { withFileTypes: true }); }
|
|
38
|
+
catch (error) { return { harnesses, problems: [{ name: path.basename(root), reason: error.message }] }; }
|
|
39
|
+
for (const entry of entries.sort((a, b) => a.name.localeCompare(b.name))) {
|
|
40
|
+
if (!entry.isDirectory() || entry.name.startsWith("_")) continue;
|
|
41
|
+
try {
|
|
42
|
+
const raw = parse(fs.readFileSync(path.join(root, entry.name, "harness.yml"), "utf8"));
|
|
43
|
+
harnesses[entry.name] = harnessSchemaFor(entry.name).parse(raw);
|
|
44
|
+
} catch (error) {
|
|
45
|
+
problems.push({ name: entry.name, reason: error.message });
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
const { failures } = orderedNames(harnesses);
|
|
49
|
+
for (const [name, reason] of failures) {
|
|
50
|
+
delete harnesses[name];
|
|
51
|
+
problems.push({ name, reason });
|
|
52
|
+
}
|
|
53
|
+
return { harnesses, problems };
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export function validateEnabledHarnesses(enabled = [], harnesses = loadHarnesses().harnesses) {
|
|
57
|
+
for (const name of enabled) {
|
|
58
|
+
if (!Object.hasOwn(harnesses, name)) throw new Error(`unknown harness '${name}'; available harnesses: ${Object.keys(harnesses).sort().join(", ")}`);
|
|
59
|
+
if (harnesses[name].network === "allowlist") throw new Error(`harness '${name}' requires allowlist; .nomarmy.yml cannot enable it (operator-local opt-in required)`);
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export function matchHarnesses(repoDir, harnesses, enabled = []) {
|
|
64
|
+
validateEnabledHarnesses(enabled, harnesses);
|
|
65
|
+
let nodeDirs, packages;
|
|
66
|
+
function directories() {
|
|
67
|
+
if (!nodeDirs) {
|
|
68
|
+
const found = nodeDependencyFiles(repoDir);
|
|
69
|
+
nodeDirs = [...new Set([".", ...found.packages, ...(found.links ?? []), ...found.skipped.map(({ dir }) => dir)])];
|
|
70
|
+
}
|
|
71
|
+
return nodeDirs;
|
|
72
|
+
}
|
|
73
|
+
function dependencies() {
|
|
74
|
+
if (!packages) {
|
|
75
|
+
packages = new Set();
|
|
76
|
+
for (const dir of directories()) {
|
|
77
|
+
try {
|
|
78
|
+
const pkg = JSON.parse(fs.readFileSync(path.join(repoDir, dir, "package.json"), "utf8"));
|
|
79
|
+
for (const field of ["dependencies", "devDependencies", "optionalDependencies", "peerDependencies"]) {
|
|
80
|
+
for (const name of Object.keys(pkg[field] ?? {})) packages.add(name);
|
|
81
|
+
}
|
|
82
|
+
} catch { /* An unreadable manifest cannot establish a match. */ }
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
return packages;
|
|
86
|
+
}
|
|
87
|
+
const isFile = (relative) => {
|
|
88
|
+
try { return fs.statSync(path.join(repoDir, relative)).isFile(); } catch { return false; }
|
|
89
|
+
};
|
|
90
|
+
const matched = Object.keys(harnesses).filter((name) => enabled.includes(name) || harnesses[name].detect.some((rule) => {
|
|
91
|
+
if (rule.package) return dependencies().has(rule.package);
|
|
92
|
+
if (rule.file) return isFile(rule.file);
|
|
93
|
+
return isFile(rule.lockfile) || directories().some((dir) => isFile(path.join(dir, rule.lockfile)));
|
|
94
|
+
}));
|
|
95
|
+
// Unmatched harnesses must not reorder unrelated matches through their after edges.
|
|
96
|
+
return orderedNames(Object.fromEntries(matched.map((name) => [name, {
|
|
97
|
+
...harnesses[name], after: harnesses[name].after.filter((parent) => matched.includes(parent)),
|
|
98
|
+
}]))).order;
|
|
99
|
+
}
|
package/lib/health.mjs
CHANGED
|
@@ -4,13 +4,16 @@
|
|
|
4
4
|
// stall the server that runs it.
|
|
5
5
|
//
|
|
6
6
|
// login-expiry an OAuth login (the ChatGPT plan's Codex import) about
|
|
7
|
-
// to expire or already expired
|
|
7
|
+
// to expire or already expired. The import is a copy of
|
|
8
|
+
// Codex's own login; re-import it, don't run agents add.
|
|
8
9
|
// openclaw-update OpenClaw older than npm's latest (a stale 2026.9.5
|
|
9
10
|
// catalog made gpt-6-sol look unavailable)
|
|
10
11
|
// plugin-skew an OpenClaw plugin older than OpenClaw itself
|
|
11
12
|
// army a role or the General pointing at something unusable
|
|
12
13
|
// config agents.yml unloadable or unsafe
|
|
13
14
|
// leftovers retained worktrees, job storage, stale "running" jobs
|
|
15
|
+
// nomarmy-update a newer nomArmy on npm; nomarmy-copy: coordinators run
|
|
16
|
+
// an older copy than the installed CLI (install-freshness.mjs)
|
|
14
17
|
//
|
|
15
18
|
// Each issue: { id, severity: "error"|"warn"|"info", title, detail, fix,
|
|
16
19
|
// short }. `id` is stable across runs, so a notification goes out once per
|
|
@@ -19,37 +22,91 @@
|
|
|
19
22
|
import { execFile } from "node:child_process";
|
|
20
23
|
import fs from "node:fs";
|
|
21
24
|
import path from "node:path";
|
|
25
|
+
import { codexImportRecovery, codexImportIssues, modelPolicyIssues } from "./openclaw-runtime-health.mjs";
|
|
22
26
|
import { executionMode } from "./execution.mjs";
|
|
23
27
|
import { modelRejection } from "./openclaw-errors.mjs";
|
|
24
28
|
import { providerConfigured, readOpenclawConfig } from "./openclaw-config.mjs";
|
|
29
|
+
import { readUsageSnapshots, usageStatus, usageDisplayText, usageReadingIsStale, refreshStaleOverLimitReadings } from "./usage-limits.mjs";
|
|
30
|
+
import { freshnessIssues, readInstallVersions } from "./install-freshness.mjs";
|
|
31
|
+
|
|
32
|
+
import { PINNED_OPENCLAW_VERSION } from "./openclaw-install.mjs";
|
|
25
33
|
|
|
26
34
|
const DAY = 86400000;
|
|
27
35
|
|
|
28
36
|
/** Run a command asynchronously, bounded; resolves { ok, stdout }. Never rejects. */
|
|
29
|
-
export function runBounded(cmd, args, { timeoutMs = 20000 } = {}) {
|
|
37
|
+
export function runBounded(cmd, args, { timeoutMs = 20000, cwd } = {}) {
|
|
30
38
|
return new Promise((resolve) => {
|
|
31
|
-
execFile(cmd, args, { encoding: "utf8", timeout: timeoutMs, maxBuffer: 16 * 1024 * 1024 }, (error, stdout) => resolve({ ok: !error, stdout: stdout ?? "" }));
|
|
39
|
+
execFile(cmd, args, { encoding: "utf8", timeout: timeoutMs, cwd, maxBuffer: 16 * 1024 * 1024 }, (error, stdout) => resolve({ ok: !error, stdout: stdout ?? "" }));
|
|
32
40
|
});
|
|
33
41
|
}
|
|
34
42
|
|
|
35
43
|
const versionOf = (text) => /(\d+)\.(\d+)\.(\d+)/.exec(String(text ?? ""))?.slice(1, 4).map(Number) ?? null;
|
|
36
44
|
const older = (a, b) => { for (let i = 0; i < 3; i++) { if (a[i] !== b[i]) return a[i] < b[i]; } return false; };
|
|
37
45
|
|
|
46
|
+
/** Read OpenClaw's auth-list JSON, tolerating text before the JSON envelope. */
|
|
47
|
+
export function parseOpenclawAuthProfiles(stdout) {
|
|
48
|
+
try {
|
|
49
|
+
const profiles = JSON.parse(stdout.slice(stdout.indexOf("{"))).profiles;
|
|
50
|
+
return Array.isArray(profiles) ? profiles : null;
|
|
51
|
+
} catch { return null; }
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export { CODEX_IMPORT_RECOVERY, codexImportRecovery } from "./openclaw-runtime-health.mjs";
|
|
55
|
+
|
|
56
|
+
function profileUsable(profile, now) {
|
|
57
|
+
if (profile?.expiresAt == null || profile.expiresAt === "") return true;
|
|
58
|
+
const expires = Date.parse(profile.expiresAt);
|
|
59
|
+
return Number.isFinite(expires) && expires > now;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function profileName(profile) {
|
|
63
|
+
const id = String(profile.id ?? profile.provider ?? "unknown");
|
|
64
|
+
const raw = [profile.label, profile.name, profile.displayName].find((value) => typeof value === "string" && value.trim());
|
|
65
|
+
if (!raw) return id;
|
|
66
|
+
const label = raw.trim();
|
|
67
|
+
if (label === id || label.startsWith(`${id} `)) return label;
|
|
68
|
+
return label.startsWith("(") ? `${id} ${label}` : `${id} (${label})`;
|
|
69
|
+
}
|
|
70
|
+
|
|
38
71
|
/** OAuth profiles near or past expiry, from `openclaw models auth list --json`. */
|
|
39
72
|
export function loginExpiryIssues(authJson, { now = Date.now(), warnDays = 7 } = {}) {
|
|
73
|
+
const profiles = authJson?.profiles ?? [];
|
|
40
74
|
const issues = [];
|
|
41
|
-
for (const p of
|
|
75
|
+
for (const p of profiles) {
|
|
42
76
|
const at = Date.parse(p.expiresAt ?? "");
|
|
43
77
|
if (!Number.isFinite(at)) continue;
|
|
44
78
|
const days = Math.floor((at - now) / DAY);
|
|
45
79
|
const who = p.provider === "openai" ? "Codex (ChatGPT plan)" : p.provider;
|
|
46
|
-
const
|
|
47
|
-
|
|
48
|
-
|
|
80
|
+
const name = profileName(p);
|
|
81
|
+
const expired = at <= now;
|
|
82
|
+
const siblingValid = profiles.some((other) => other !== p && other.provider === p.provider && other.id !== p.id && profileUsable(other, now));
|
|
83
|
+
const fix = p.provider === "openai" ? codexImportRecovery({ profiles }) : `openclaw models auth login --provider ${p.provider}`;
|
|
84
|
+
if (expired) {
|
|
85
|
+
const detail = siblingValid
|
|
86
|
+
? `Auth profile ${name} expired ${new Date(at).toISOString().slice(0, 10)}. OpenClaw may still pick it.`
|
|
87
|
+
: `Auth profile ${name} expired ${new Date(at).toISOString().slice(0, 10)}; every job on it will fail.`;
|
|
88
|
+
issues.push({ id: `login-expired:${p.id}`, severity: "error", title: `${who} login has expired: ${name}`, detail, fix, short: `${p.provider} login expired` });
|
|
89
|
+
} else if (at - now <= warnDays * DAY) issues.push({ id: `login-expiring:${p.id}`, severity: "warn", title: `${who} login expires in ${days} day${days === 1 ? "" : "s"}: ${name}`, detail: `Auth profile ${name} expires ${new Date(at).toISOString().slice(0, 10)}.`, fix, short: `${p.provider} login ${days}d` });
|
|
49
90
|
}
|
|
50
91
|
return issues;
|
|
51
92
|
}
|
|
52
93
|
|
|
94
|
+
/** Shared runtime preconditions for doctor and health. */
|
|
95
|
+
export async function runtimeIssues({ run = runBounded, openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", openclawAgent = "main", agents = {}, armySummary = null, modelsInUse = null, now = Date.now() } = {}) {
|
|
96
|
+
const paths = [`agents.entries.${openclawAgent}.modelPolicy.allow`, "agents.defaults.modelPolicy.allow"];
|
|
97
|
+
const [auth, ...policies] = await Promise.all([
|
|
98
|
+
run(openclawCmd, ["models", "auth", "list", "--json"]),
|
|
99
|
+
...paths.map((p) => run(openclawCmd, ["config", "get", p, "--json"])),
|
|
100
|
+
]);
|
|
101
|
+
const profiles = auth.ok ? parseOpenclawAuthProfiles(auth.stdout) : null;
|
|
102
|
+
const issues = profiles ? loginExpiryIssues({ profiles }, { now }) : [];
|
|
103
|
+
if (Object.values(agents ?? {}).some((a) => a.kind === "subscription" && a.provider === "openai")) {
|
|
104
|
+
issues.push(...codexImportIssues(profiles ?? [], { now }));
|
|
105
|
+
}
|
|
106
|
+
issues.push(...modelPolicyIssues({ policies, paths, agents, armySummary, modelsInUse }));
|
|
107
|
+
return issues;
|
|
108
|
+
}
|
|
109
|
+
|
|
53
110
|
/** OpenClaw behind npm's latest, and plugins behind OpenClaw. */
|
|
54
111
|
export function versionIssues({ installed, latest, plugins = [] }) {
|
|
55
112
|
const issues = [];
|
|
@@ -57,7 +114,7 @@ export function versionIssues({ installed, latest, plugins = [] }) {
|
|
|
57
114
|
if (have && want && older(have, want)) {
|
|
58
115
|
issues.push({ id: `openclaw-update:${want.join(".")}`, severity: "info", title: `OpenClaw ${want.join(".")} is out (you have ${have.join(".")})`,
|
|
59
116
|
detail: "Its bundled model catalog lags new models; an old one made a working model look unavailable and budgeted new ones at the 32k fallback.",
|
|
60
|
-
fix:
|
|
117
|
+
fix: older(have, versionOf(PINNED_OPENCLAW_VERSION)) ? "nomarmy doctor --fix (installs the tested OpenClaw release, when no nomArmy jobs are running)" : "No automatic upgrade: nomArmy keeps the tested release or a newer installed version", short: "openclaw update" });
|
|
61
118
|
}
|
|
62
119
|
for (const p of plugins) {
|
|
63
120
|
const pv = versionOf(p.version);
|
|
@@ -93,16 +150,21 @@ export function armyIssues(summary, mode = "local") {
|
|
|
93
150
|
* from job records. A listed model isn't proof it runs: Muse was listed
|
|
94
151
|
* while every job on it failed this way (a real Senti review).
|
|
95
152
|
*/
|
|
153
|
+
/** The "<provider>/<model>" a job record says its vendor refused, or null. */
|
|
154
|
+
export function refusedModelIn(record) {
|
|
155
|
+
const text = `${record?.workerError ?? ""} ${record?.worker?.error ?? ""}`;
|
|
156
|
+
// A job's own model_not_found line first (mcp/server.mjs names the model
|
|
157
|
+
// it attempted), then any raw vendor wording in older records.
|
|
158
|
+
const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
|
|
159
|
+
return (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, "") ?? null; // the sentence's own trailing period isn't part of the name
|
|
160
|
+
}
|
|
161
|
+
|
|
96
162
|
export function unknownModelIssues(jobRecords, { now = Date.now(), inUse = null, probedOk = {} } = {}) {
|
|
97
163
|
const failed = new Map(), lastOk = new Map(Object.entries(probedOk));
|
|
98
164
|
for (const m of jobRecords) {
|
|
99
165
|
const at = Date.parse(m.finishedAt ?? "");
|
|
100
166
|
if (now - at > DAY) continue;
|
|
101
|
-
const
|
|
102
|
-
// A job's own model_not_found line first (mcp/server.mjs names the model
|
|
103
|
-
// it attempted), then any raw vendor wording in older records.
|
|
104
|
-
const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
|
|
105
|
-
const model = (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, ""); // the sentence's own trailing period isn't part of the name
|
|
167
|
+
const model = refusedModelIn(m);
|
|
106
168
|
if (model) { const f = failed.get(model) ?? { n: 0, last: 0 }; f.n++; f.last = Math.max(f.last, at); failed.set(model, f); }
|
|
107
169
|
else if (m.worker?.provider && m.worker?.model && !m.workerError) {
|
|
108
170
|
const ran = `${m.worker.provider}/${m.worker.model}`;
|
|
@@ -147,6 +209,43 @@ export function recordProbeSuccess(stateRoot, model, { now = Date.now() } = {})
|
|
|
147
209
|
try { seen = JSON.parse(fs.readFileSync(file, "utf8")); } catch { /* first */ }
|
|
148
210
|
seen[model] = new Date(now).toISOString();
|
|
149
211
|
try { fs.writeFileSync(file, JSON.stringify(seen, null, 2)); } catch { /* best-effort */ }
|
|
212
|
+
clearModelRefusal(stateRoot, model);
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// Models a vendor refused on a job, kept until a job or test call on them
|
|
216
|
+
// works. A plan's model limit doesn't expire overnight (gpt-6-luna was
|
|
217
|
+
// refused again days after the first time), so this isn't a 24-hour memory.
|
|
218
|
+
// <stateRoot>/model-refusals.json: { "<provider>/<model>": { at, reason, retriedAt? } }.
|
|
219
|
+
export const REFUSAL_RETRY_MS = 24 * 60 * 60 * 1000;
|
|
220
|
+
const REFUSALS_FILE = "model-refusals.json";
|
|
221
|
+
export function modelRefusals(stateRoot) {
|
|
222
|
+
try { const data = JSON.parse(fs.readFileSync(path.join(stateRoot, REFUSALS_FILE), "utf8")); return data && typeof data === "object" && !Array.isArray(data) ? data : {}; } catch { return {}; }
|
|
223
|
+
}
|
|
224
|
+
function writeRefusals(stateRoot, data) {
|
|
225
|
+
try { fs.mkdirSync(stateRoot, { recursive: true }); fs.writeFileSync(path.join(stateRoot, REFUSALS_FILE), JSON.stringify(data, null, 2)); } catch { /* best-effort */ }
|
|
226
|
+
}
|
|
227
|
+
export function recordModelRefusal(stateRoot, model, reason = null, { now = Date.now() } = {}) {
|
|
228
|
+
writeRefusals(stateRoot, { ...modelRefusals(stateRoot), [model]: { at: new Date(now).toISOString(), reason } });
|
|
229
|
+
}
|
|
230
|
+
/** Claim a stale refusal before a retry so overlapping calls do not re-test it. */
|
|
231
|
+
export function claimModelRefusalRetry(stateRoot, model, { now = Date.now() } = {}) {
|
|
232
|
+
const data = modelRefusals(stateRoot);
|
|
233
|
+
const entry = data[model];
|
|
234
|
+
if (!entry || now - Date.parse(entry.at) < REFUSAL_RETRY_MS ||
|
|
235
|
+
(entry.retriedAt && now - Date.parse(entry.retriedAt) < REFUSAL_RETRY_MS)) return false;
|
|
236
|
+
data[model] = { ...entry, retriedAt: new Date(now).toISOString() };
|
|
237
|
+
writeRefusals(stateRoot, data);
|
|
238
|
+
return true;
|
|
239
|
+
}
|
|
240
|
+
export function clearModelRefusal(stateRoot, model) {
|
|
241
|
+
const data = modelRefusals(stateRoot);
|
|
242
|
+
if (!(model in data)) return;
|
|
243
|
+
delete data[model];
|
|
244
|
+
writeRefusals(stateRoot, data);
|
|
245
|
+
}
|
|
246
|
+
/** Whether a test call or a finished job has worked on `model` ("provider/model"). */
|
|
247
|
+
export function modelProven(stateRoot, model) {
|
|
248
|
+
return Object.hasOwn(readProbeSuccesses(stateRoot), model);
|
|
150
249
|
}
|
|
151
250
|
function readProbeSuccesses(stateRoot) {
|
|
152
251
|
try { return Object.fromEntries(Object.entries(JSON.parse(fs.readFileSync(path.join(stateRoot, "probe-ok.json"), "utf8"))).map(([k, v]) => [k, Date.parse(v)])); } catch { return {}; }
|
|
@@ -159,6 +258,8 @@ function readProbeSuccesses(stateRoot) {
|
|
|
159
258
|
* ChatGPT plan failed job after job before health noticed). The issue, or null.
|
|
160
259
|
*/
|
|
161
260
|
export function recentModelRefusal(stateRoot, model, { now = Date.now() } = {}) {
|
|
261
|
+
const kept = modelRefusals(stateRoot)[model];
|
|
262
|
+
if (kept) return { id: `unknown-model:${model}`, severity: "warn", title: `${model} was refused on a job (${kept.at.slice(0, 10)}) and hasn't worked since`, detail: kept.reason ?? "The provider won't run it, though it may be listed in the catalog.", fix: "nomarmy army assign <role> <agent> <another model>", short: `${model.split("/").pop()} not running` };
|
|
162
263
|
const jobsRoot = path.join(stateRoot, "jobs");
|
|
163
264
|
const records = [];
|
|
164
265
|
let names = [];
|
|
@@ -183,19 +284,41 @@ export function leftoverIssues({ retainedWorktrees = 0, jobsBytes = 0, staleRunn
|
|
|
183
284
|
* Run every check. `env` supplies what each needs, with real defaults;
|
|
184
285
|
* tests pass their own.
|
|
185
286
|
*/
|
|
186
|
-
export async function runHealthChecks({ now = Date.now(), mode = "local", openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", run = runBounded, armySummary = null, agentsError = null, jobsRoot = null, pidAlive = () => true, agents = null, openclawConfig = null, vendors = {}, modelsInUse = null, autoPruned = null } = {}) {
|
|
287
|
+
export async function runHealthChecks({ now = Date.now(), mode = "local", openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", run = runBounded, armySummary = null, agentsError = null, jobsRoot = null, pidAlive = () => true, agents = null, openclawConfig = null, vendors = {}, modelsInUse = null, autoPruned = null, usageSnapshots = null, stateRoot = null, install = null } = {}) {
|
|
187
288
|
const issues = [];
|
|
188
289
|
if (autoPruned?.freedBytes) issues.push({ id: `auto-prune:${new Date(now).toISOString()}`, severity: "info",
|
|
189
290
|
title: `Freed ${(autoPruned.freedBytes / 1024 ** 3).toFixed(2)} GB: ${[autoPruned.pruned ? `runtime data of ${autoPruned.pruned} finished job${autoPruned.pruned === 1 ? "" : "s"} older than ${autoPruned.olderThanHours}h` : null, autoPruned.scratchCleared ? `OpenClaw scratch files of ${autoPruned.scratchCleared} more` : null].filter(Boolean).join(", ")}`,
|
|
190
291
|
detail: "Automatic; each job's record and report are kept. NOMARMY_AUTO_PRUNE_HOURS sets the age (0 turns it off).", fix: null, short: null });
|
|
191
292
|
if (agents) issues.push(...providerConfigIssues({ agents, openclawConfig, vendors }));
|
|
192
|
-
|
|
193
|
-
|
|
293
|
+
usageSnapshots ??= stateRoot ? readUsageSnapshots(stateRoot) : null;
|
|
294
|
+
const staleOver = usageSnapshots && Object.values(usageSnapshots).some((snapshot) => usageReadingIsStale(usageStatus(snapshot, now)));
|
|
295
|
+
const [runtime, version, latest, plugins, nomarmyLatest, refreshed] = await Promise.all([
|
|
296
|
+
runtimeIssues({ run, openclawCmd, agents, armySummary, modelsInUse, now }),
|
|
194
297
|
run(openclawCmd, ["--version"]),
|
|
195
298
|
run("npm", ["view", "openclaw", "version"], { timeoutMs: 15000 }),
|
|
196
299
|
run(openclawCmd, ["plugins", "inspect", "codex"]),
|
|
300
|
+
install ? run("npm", ["view", "nomarmy", "dist-tags.alpha"], { timeoutMs: 15000 }) : null,
|
|
301
|
+
staleOver ? refreshStaleOverLimitReadings(stateRoot, { run, now, snapshots: usageSnapshots, openclawCmd }) : null,
|
|
197
302
|
]);
|
|
198
|
-
|
|
303
|
+
const snapshots = refreshed?.snapshots ?? usageSnapshots;
|
|
304
|
+
const failedProviders = new Set(refreshed?.failedProviders ?? []);
|
|
305
|
+
if (snapshots) {
|
|
306
|
+
for (const [provider, snapshot] of Object.entries(snapshots)) {
|
|
307
|
+
const status = usageStatus(snapshot, now);
|
|
308
|
+
if (status.level === "ok") continue;
|
|
309
|
+
const short = `${provider} ${status.short}`;
|
|
310
|
+
const stale = usageReadingIsStale(status);
|
|
311
|
+
const detail = stale
|
|
312
|
+
? `${usageDisplayText(status)}.${failedProviders.has(provider) ? ` ${refreshed.error}.` : ""}`
|
|
313
|
+
: `${status.text} (reading ${status.ageMinutes} minutes old).`;
|
|
314
|
+
issues.push({ id: `usage:${provider}:${status.level}`, severity: "warn", title: status.level === "over" ? `${provider} is at its usage limit` : `${provider} usage is ${status.short}`,
|
|
315
|
+
detail,
|
|
316
|
+
fix: status.level === "over" ? "wait for the reset, or move its roles with nomarmy army assign" : "plan remaining work or move its roles with nomarmy army assign",
|
|
317
|
+
short });
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
if (install) issues.push(...freshnessIssues({ ...install, latestVersion: nomarmyLatest?.ok ? nomarmyLatest.stdout : null }));
|
|
321
|
+
issues.push(...runtime);
|
|
199
322
|
const pluginVersion = /Version:\s*(\S+)/.exec(plugins.stdout ?? "")?.[1];
|
|
200
323
|
if (version.ok) issues.push(...versionIssues({ installed: version.stdout, latest: latest.ok ? latest.stdout : null, plugins: pluginVersion ? [{ id: "codex", version: pluginVersion }] : [] }));
|
|
201
324
|
if (agentsError) issues.push({ id: `config:agents:${agentsError}`, severity: "error", title: "agents.yml can't be loaded", detail: agentsError, fix: "nomarmy agents list (shows the problem)", short: "agents.yml broken" });
|
|
@@ -245,6 +368,18 @@ export function recordHealth(file, result, { now = Date.now() } = {}) {
|
|
|
245
368
|
return toNotify;
|
|
246
369
|
}
|
|
247
370
|
|
|
371
|
+
/** Read versions and harnesses only from MCP registrations that launch nomArmy. */
|
|
372
|
+
export async function readRegisteredInstall({ projectDir, run = runBounded, installDir, cursorConfigPath, homeDir } = {}) {
|
|
373
|
+
const { registeredMcpCopies } = await import("./connect.mjs");
|
|
374
|
+
const { loadHarnesses } = await import("./harnesses.mjs");
|
|
375
|
+
const registered = await registeredMcpCopies({ projectDir, run, installDir, cursorConfigPath, homeDir });
|
|
376
|
+
if (!registered.length) return null;
|
|
377
|
+
return { registrations: registered.map(({ target, scope, serverPath }) => {
|
|
378
|
+
const copyDir = path.dirname(path.dirname(serverPath));
|
|
379
|
+
return { target, scope, ...readInstallVersions(copyDir), copyHarnesses: Object.keys(loadHarnesses(path.join(copyDir, "harnesses")).harnesses).length };
|
|
380
|
+
}) };
|
|
381
|
+
}
|
|
382
|
+
|
|
248
383
|
/**
|
|
249
384
|
* Everything a check run needs, gathered the same way for the server's
|
|
250
385
|
* periodic run and `nomarmy health`, then run and recorded to the shared
|
|
@@ -277,8 +412,17 @@ export async function checkAndRecordHealth({ projectDir, stateRoot, configDir, n
|
|
|
277
412
|
let autoPruned = null;
|
|
278
413
|
if (ageMs !== null) { try { autoPruned = { ...pruneJobRuntime({ stateRoot, olderThanMs: ageMs, now }), olderThanHours: ageMs / 3600000 }; } catch { /* best-effort */ } }
|
|
279
414
|
const mode = executionMode(env).mode;
|
|
415
|
+
const install = await readRegisteredInstall({ projectDir });
|
|
280
416
|
const result = await runHealthChecks({ now, mode, armySummary, agentsError, jobsRoot: path.join(stateRoot, "jobs"), pidAlive,
|
|
281
|
-
agents, openclawConfig: readOpenclawConfig(), vendors: SUBSCRIPTION_VENDORS, modelsInUse, autoPruned });
|
|
417
|
+
agents, openclawConfig: readOpenclawConfig(), vendors: SUBSCRIPTION_VENDORS, modelsInUse, autoPruned, usageSnapshots: readUsageSnapshots(stateRoot), stateRoot, install });
|
|
418
|
+
try {
|
|
419
|
+
const { loadJobRecords, agentLookup } = await import("./stats.mjs");
|
|
420
|
+
const { recentSuggestions } = await import("./suggestions.mjs");
|
|
421
|
+
for (const s of recentSuggestions(loadJobRecords(path.join(stateRoot, "jobs")), { projectDir, agentFor: agents ? agentLookup(agents, agentProviderId) : () => null, now })) {
|
|
422
|
+
if (s.level !== "warn" && s.level !== "act") continue;
|
|
423
|
+
result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: s.level === "act" ? "review needed" : "routing tip" });
|
|
424
|
+
}
|
|
425
|
+
} catch { /* suggestions never break a health check */ }
|
|
282
426
|
const toNotify = recordHealth(path.join(stateRoot, "health.json"), result, { now });
|
|
283
427
|
return { result, toNotify };
|
|
284
428
|
}
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
// Whether the nomArmy a coordinator runs is current. Claude Code, Codex and
|
|
2
|
+
// Cursor run a copy of the server that `nomarmy connect` puts in the install
|
|
3
|
+
// dir, and each session keeps the code it started with. So an upgrade can
|
|
4
|
+
// stall at three points, each checked here:
|
|
5
|
+
//
|
|
6
|
+
// nomarmy-update npm's alpha release is newer than the installed CLI
|
|
7
|
+
// nomarmy-copy the CLI is newer than the copy coordinators run
|
|
8
|
+
// (`nomarmy connect` wasn't re-run)
|
|
9
|
+
// restart the copy on disk changed after this server started
|
|
10
|
+
// (the session wasn't restarted); per session, so it is
|
|
11
|
+
// reported by that session's server, not in health.json
|
|
12
|
+
|
|
13
|
+
import { execFileSync } from "node:child_process";
|
|
14
|
+
import fs from "node:fs";
|
|
15
|
+
import path from "node:path";
|
|
16
|
+
|
|
17
|
+
/** Written into the install dir by `nomarmy connect`: where the copy came from. */
|
|
18
|
+
export const SOURCE_FILE = "source.json";
|
|
19
|
+
|
|
20
|
+
export function readPackageVersion(root) {
|
|
21
|
+
try { return JSON.parse(fs.readFileSync(path.join(root, "package.json"), "utf8")).version ?? null; }
|
|
22
|
+
catch { return null; }
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** The checkout's commit, for a git install; null for an npm install. */
|
|
26
|
+
export function readSourceCommit(root) {
|
|
27
|
+
if (!fs.existsSync(path.join(root, ".git"))) return null;
|
|
28
|
+
try { return execFileSync("git", ["rev-parse", "HEAD"], { cwd: root, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }).trim() || null; }
|
|
29
|
+
catch { return null; }
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export function recordCopySource(installDir, nomarmyRoot) {
|
|
33
|
+
const commit = readSourceCommit(nomarmyRoot);
|
|
34
|
+
fs.writeFileSync(path.join(installDir, SOURCE_FILE), JSON.stringify({ root: nomarmyRoot, version: readPackageVersion(nomarmyRoot), ...(commit ? { commit } : {}) }, null, 2) + "\n");
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Whether the copy in installDir is behind the checkout or package at
|
|
39
|
+
* nomarmyRoot: an older version, a different commit (a git install moves on
|
|
40
|
+
* without a version bump), or no record of its source at all.
|
|
41
|
+
*/
|
|
42
|
+
export function copyIsStale(installDir, nomarmyRoot) {
|
|
43
|
+
let source = null;
|
|
44
|
+
try { source = JSON.parse(fs.readFileSync(path.join(installDir, SOURCE_FILE), "utf8")); } catch { return true; }
|
|
45
|
+
const copyVersion = readPackageVersion(installDir), rootVersion = readPackageVersion(nomarmyRoot);
|
|
46
|
+
if (!copyVersion || !rootVersion || copyVersion !== rootVersion) return true;
|
|
47
|
+
const commit = readSourceCommit(nomarmyRoot);
|
|
48
|
+
return Boolean(commit && source.commit !== commit);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** The copy's version and, when connect recorded it, the version now at its source. */
|
|
52
|
+
export function readInstallVersions(installDir) {
|
|
53
|
+
let source = null;
|
|
54
|
+
try { source = JSON.parse(fs.readFileSync(path.join(installDir, SOURCE_FILE), "utf8")); } catch { /* connected before source.json existed */ }
|
|
55
|
+
return { copyVersion: readPackageVersion(installDir), sourceVersion: source?.root ? readPackageVersion(source.root) : null };
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Semver order, prerelease included (0.1.0-alpha.7 < 0.1.0-alpha.10 < 0.1.0). */
|
|
59
|
+
export function compareVersions(a, b) {
|
|
60
|
+
const split = (v) => { const [main, pre] = String(v).trim().replace(/^v/, "").split("-", 2); return { main: main.split(".").map(Number), pre: pre ? pre.split(".") : null }; };
|
|
61
|
+
const x = split(a), y = split(b);
|
|
62
|
+
for (let i = 0; i < 3; i++) if ((x.main[i] || 0) !== (y.main[i] || 0)) return (x.main[i] || 0) < (y.main[i] || 0) ? -1 : 1;
|
|
63
|
+
if (!x.pre || !y.pre) return x.pre === y.pre ? 0 : x.pre ? -1 : 1;
|
|
64
|
+
for (let i = 0; i < Math.max(x.pre.length, y.pre.length); i++) {
|
|
65
|
+
const p = x.pre[i], q = y.pre[i];
|
|
66
|
+
if (p === undefined || q === undefined) return p === undefined ? -1 : 1;
|
|
67
|
+
if (p === q) continue;
|
|
68
|
+
const pn = /^\d+$/.test(p), qn = /^\d+$/.test(q);
|
|
69
|
+
if (pn && qn) return Number(p) < Number(q) ? -1 : 1;
|
|
70
|
+
if (pn !== qn) return pn ? -1 : 1;
|
|
71
|
+
return p < q ? -1 : 1;
|
|
72
|
+
}
|
|
73
|
+
return 0;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const valid = (v) => typeof v === "string" && /^v?\d+\.\d+\.\d+/.test(v.trim());
|
|
77
|
+
|
|
78
|
+
/** Health issues for a stale CLI, a stale installed copy, or a copy missing its harnesses. */
|
|
79
|
+
export function freshnessIssues({ copyVersion = null, sourceVersion = null, latestVersion = null, copyHarnesses = null, registrations = null, target = "claude", scope = "user" }) {
|
|
80
|
+
if (registrations) return registrations.flatMap((copy, index) => freshnessIssues({ ...copy, latestVersion: index === 0 ? latestVersion : null }));
|
|
81
|
+
const reconnect = `nomarmy connect ${target}${scope === "user" ? "" : ` --scope ${scope}`}`;
|
|
82
|
+
const fix = registrations === null && target === "claude" && scope === "user" ? "nomarmy connect claude (and codex, cursor), then restart those sessions" : `${reconnect}, then restart that session`;
|
|
83
|
+
const issues = [];
|
|
84
|
+
if (valid(copyVersion) && copyHarnesses === 0) {
|
|
85
|
+
issues.push({ id: `nomarmy-harnesses:${copyVersion}`, severity: "error", title: "The nomArmy your coordinators run has no harnesses",
|
|
86
|
+
detail: "Jobs match no harness, so they run in the plain base image: no dependencies, fake services, browser tests or artifacts.",
|
|
87
|
+
fix, short: "no harnesses" });
|
|
88
|
+
}
|
|
89
|
+
if (valid(latestVersion) && valid(sourceVersion) && compareVersions(sourceVersion, latestVersion) < 0) {
|
|
90
|
+
const latest = latestVersion.trim();
|
|
91
|
+
issues.push({ id: `nomarmy-update:${latest}`, severity: "info", title: `nomArmy ${latest} is out (you have ${sourceVersion})`,
|
|
92
|
+
detail: "Updating installs it, reconnects your coordinators and tells you which sessions to restart.",
|
|
93
|
+
fix: "nomarmy update", short: "nomarmy update" });
|
|
94
|
+
}
|
|
95
|
+
if (valid(sourceVersion) && valid(copyVersion) && compareVersions(copyVersion, sourceVersion) < 0) {
|
|
96
|
+
issues.push({ id: `nomarmy-copy:${sourceVersion}`, severity: "warn", title: `Your coordinators run nomArmy ${copyVersion}, but ${sourceVersion} is installed`,
|
|
97
|
+
detail: "Claude Code, Codex and Cursor run a copy of nomArmy that only `nomarmy connect` refreshes.",
|
|
98
|
+
fix, short: "nomarmy reconnect" });
|
|
99
|
+
}
|
|
100
|
+
return issues;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* For the running server: a notice when its copy on disk changed after it
|
|
105
|
+
* started, so this session still runs the old code. Null when current.
|
|
106
|
+
*/
|
|
107
|
+
export function restartNotice({ serverFile, startedAtMs, runningVersion, stat = fs.statSync, readVersion = readPackageVersion }) {
|
|
108
|
+
let changedMs;
|
|
109
|
+
try { changedMs = stat(serverFile).mtimeMs; } catch { return null; }
|
|
110
|
+
if (!(changedMs > startedAtMs)) return null;
|
|
111
|
+
const onDisk = readVersion(path.join(path.dirname(serverFile), ".."));
|
|
112
|
+
const versions = onDisk && onDisk !== runningVersion ? ` (this session runs ${runningVersion}; ${onDisk} is installed)` : "";
|
|
113
|
+
return `nomArmy was updated after this session started${versions}. Restart this session to use the new version; until then it runs the old code.`;
|
|
114
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
// nomArmy's Jev checks (see lib/validators.mjs): narrow judgments on evidence
|
|
2
|
+
// nomArmy already has, each able only to add a review flag.
|
|
3
|
+
//
|
|
4
|
+
// scout-citations nomArmy verifies a scout's [path:line] exists and attaches
|
|
5
|
+
// the lines; this asks whether those lines support the
|
|
6
|
+
// finding (the shape of TypeSafe's citation-check cookbook).
|
|
7
|
+
// report-claims whether a worker's report matches its diff. Found live: a
|
|
8
|
+
// worker's note said "restored check.js to base commit"
|
|
9
|
+
// while its diff rewrote check.js.
|
|
10
|
+
|
|
11
|
+
import { askJev, jevBreaker, tripJevBreaker } from "./validators.mjs";
|
|
12
|
+
|
|
13
|
+
// A flag needs the model to lean clearly; the rest is left to the General.
|
|
14
|
+
export const FLAG_AT = 0.7;
|
|
15
|
+
const DIFF_CHARS = 60000; // well inside the 32k-token state budget
|
|
16
|
+
const MAX_FINDINGS = 24;
|
|
17
|
+
// The report's excerpts are capped at 12 lines for the General's context; Jev
|
|
18
|
+
// judges the whole cited range (to this cap), or a long range looks
|
|
19
|
+
// unrelated when the supporting line is past the excerpt.
|
|
20
|
+
const CITED_LINES = 80;
|
|
21
|
+
const CONCURRENCY = 4;
|
|
22
|
+
// However many findings, a job never waits longer than this on Jev.
|
|
23
|
+
const JOB_BUDGET_MS = 45000;
|
|
24
|
+
|
|
25
|
+
// The answer, or null with the breaker tripped: any failure means TypeSafe
|
|
26
|
+
// isn't answering well right now, so stop asking (lib/validators.mjs).
|
|
27
|
+
async function guarded(ask, request, errors) {
|
|
28
|
+
if (jevBreaker().open) { errors.push(`skipped: Jev failed recently (${jevBreaker().reason}); retrying after ${jevBreaker().retryAt}`); return null; }
|
|
29
|
+
try { return await ask(request); }
|
|
30
|
+
catch (error) { const why = error?.name === "AbortError" ? "timed out" : error.message; tripJevBreaker(why); errors.push(why); return null; }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
async function mapLimit(items, limit, fn) {
|
|
34
|
+
const out = new Array(items.length);
|
|
35
|
+
let next = 0;
|
|
36
|
+
await Promise.all(Array.from({ length: Math.min(limit, items.length) }, async () => {
|
|
37
|
+
while (next < items.length) { const i = next++; out[i] = await fn(items[i], i); }
|
|
38
|
+
}));
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const SUPPORT_QUESTION = {
|
|
43
|
+
type: "choice",
|
|
44
|
+
instructions: "A code researcher made the claim in `finding` and cited the source lines in `cited` as evidence. Judge only what the cited lines show, not whether the claim might be true elsewhere in the codebase. How do the cited lines relate to the claim?",
|
|
45
|
+
criteria: {
|
|
46
|
+
supports: "The cited lines directly show what the claim says.",
|
|
47
|
+
contradicts: "The cited lines show something different from, or opposite to, the claim.",
|
|
48
|
+
unrelated: "The cited lines don't address the claim: they're about something else, or too little of the claim is visible in them.",
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* For each finding with verified cited lines: does the evidence support it?
|
|
54
|
+
* @returns {Promise<{ flags: {index: number, verdict: string, probability: number}[], checked: number, errors: string[], usage: number }>}
|
|
55
|
+
*/
|
|
56
|
+
export async function checkScoutCitations({ findings = [], settings, ask = askJev, readFile = null }) {
|
|
57
|
+
const fileCache = new Map();
|
|
58
|
+
const fullRange = async (c) => {
|
|
59
|
+
if (!readFile) return null;
|
|
60
|
+
if (!fileCache.has(c.path)) fileCache.set(c.path, readFile(c.path).then((t) => (typeof t === "string" ? t.split("\n") : null)).catch(() => null));
|
|
61
|
+
const lines = await fileCache.get(c.path);
|
|
62
|
+
if (!lines) return null;
|
|
63
|
+
const end = Math.min(c.end, c.start + CITED_LINES - 1, lines.length);
|
|
64
|
+
return Array.from({ length: Math.max(0, end - c.start + 1) }, (_, k) => `${c.start + k}: ${lines[c.start + k - 1]}`).join("\n");
|
|
65
|
+
};
|
|
66
|
+
const candidates = findings.map((f, index) => ({ f, index }))
|
|
67
|
+
.filter(({ f }) => (f.citations ?? []).some((c) => c.status === "ok" && c.excerpt?.length)).slice(0, MAX_FINDINGS);
|
|
68
|
+
const errors = [];
|
|
69
|
+
let usage = 0;
|
|
70
|
+
const deadline = Date.now() + JOB_BUDGET_MS;
|
|
71
|
+
const verdicts = await mapLimit(candidates, CONCURRENCY, async ({ f, index }) => {
|
|
72
|
+
if (Date.now() > deadline) { errors.push("skipped the rest: over the job's time budget for Jev"); return null; }
|
|
73
|
+
const cited = await Promise.all(f.citations.filter((c) => c.status === "ok" && c.excerpt?.length)
|
|
74
|
+
.map(async (c) => ({ file: c.path, lines: `${c.start}-${c.end}`, text: (await fullRange(c)) ?? c.excerpt.map((l) => `${l.line}: ${l.text}`).join("\n") })));
|
|
75
|
+
const r = await guarded(ask, { key: settings.key, model: settings.model, state: { finding: f.text, cited }, questions: { support: SUPPORT_QUESTION } }, errors);
|
|
76
|
+
if (!r) return null;
|
|
77
|
+
usage += r.usage?.input_tokens ?? 0;
|
|
78
|
+
const a = r.answers?.support;
|
|
79
|
+
return a ? { index, verdict: a.choice, probability: a.probabilities?.[a.choice] ?? 0 } : null;
|
|
80
|
+
});
|
|
81
|
+
const flags = verdicts.filter((v) => v && v.verdict !== "supports" && v.probability >= FLAG_AT);
|
|
82
|
+
return { flags, checked: verdicts.filter(Boolean).length, errors: [...new Set(errors)].slice(0, 3), usage, verdicts: verdicts.filter(Boolean) };
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const CLAIMS_QUESTION = {
|
|
86
|
+
type: "choice",
|
|
87
|
+
instructions: "A coding worker wrote the report in `report` about the change it made. `diff` shows every change from the base commit it started from: a file that isn't in the diff is exactly as it was at the base commit, and a file restored to the base commit wouldn't appear. Judge only the concrete things the report says were done or changed. Does the diff match them?",
|
|
88
|
+
criteria: {
|
|
89
|
+
consistent: "The concrete things the report says were done are what the diff shows.",
|
|
90
|
+
contradicts: "The diff shows a concrete claim in the report is false: something said to be done, restored, removed or left alone wasn't, or was done differently.",
|
|
91
|
+
unclear: "The report makes no concrete claim the diff can confirm or refute, or the diff shown is too partial to tell.",
|
|
92
|
+
},
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Does the worker's report match its diff?
|
|
97
|
+
* @returns {Promise<{ flag: {verdict: string, probability: number}|null, verdict: object|null, error: string|null, usage: number, truncated: boolean }>}
|
|
98
|
+
*/
|
|
99
|
+
export async function checkReportClaims({ report, diff, settings, ask = askJev }) {
|
|
100
|
+
const note = [report?.note, report?.notDone && report.notDone !== "none" ? `Not done: ${report.notDone}` : null].filter(Boolean).join("\n");
|
|
101
|
+
if (!note.trim() || !String(diff ?? "").trim()) return { flag: null, verdict: null, error: null, usage: 0, truncated: false };
|
|
102
|
+
const truncated = diff.length > DIFF_CHARS;
|
|
103
|
+
const state = { report: { status: report.status ?? null, tests: report.tests ?? null, note }, diff: truncated ? `${diff.slice(0, DIFF_CHARS)}\n[diff truncated]` : diff };
|
|
104
|
+
const errors = [];
|
|
105
|
+
const r = await guarded(ask, { key: settings.key, model: settings.model, state, questions: { claims: CLAIMS_QUESTION } }, errors);
|
|
106
|
+
if (!r) return { flag: null, verdict: null, error: errors[0] ?? "no answer", usage: 0, truncated };
|
|
107
|
+
const a = r.answers?.claims;
|
|
108
|
+
const verdict = a ? { verdict: a.choice, probability: a.probabilities?.[a.choice] ?? 0 } : null;
|
|
109
|
+
return { flag: verdict && verdict.verdict === "contradicts" && verdict.probability >= FLAG_AT ? verdict : null, verdict, error: null, usage: r.usage?.input_tokens ?? 0, truncated };
|
|
110
|
+
}
|