nomarmy 0.1.0-alpha.16 ā 0.1.0-alpha.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nomarmy.mjs +27 -6
- package/lib/health.mjs +2 -2
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +78 -5
- package/lib/suggestions.mjs +33 -13
- package/mcp/server.mjs +4 -3
- package/package.json +1 -1
package/bin/nomarmy.mjs
CHANGED
|
@@ -20,7 +20,7 @@ import { readGGUFMetadata, resolveModelPath, totalSplitBytes } from "../lib/gguf
|
|
|
20
20
|
import { recommend, customRecommendation, evaluateConfig, bytesPerKvElementForCacheTypes, MIN_CONTEXT_PER_NOM } from "../lib/sizing.mjs";
|
|
21
21
|
import { connectClaude, connectCodex, connectCursor, cursorAlreadyConnected, deriveWorkerModelEnv, defaultInstallDir, installMcpCopy, SCOPES, claudeUserScoped, portableServerLaunch } from "../lib/connect.mjs";
|
|
22
22
|
import { compareVersions, readPackageVersion, readInstallVersions, copyIsStale } from "../lib/install-freshness.mjs";
|
|
23
|
-
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
23
|
+
import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
24
24
|
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
25
25
|
import { loadValidators, saveJevKey, removeJev, jevSettings, askJev, validatorsPath, JEV_CHECKS, saveJudge, removeJudge, judgeSettings } from "../lib/validators.mjs";
|
|
26
26
|
import { probeModel } from "../lib/model-probe.mjs";
|
|
@@ -31,6 +31,7 @@ import { parseLlamaUrl } from "../lib/execution.mjs";
|
|
|
31
31
|
import { setupSteps, formatSetupSteps, runSetupPlaybook } from "../lib/setup-steps.mjs";
|
|
32
32
|
import { readUsageSnapshots } from "../lib/usage-limits.mjs";
|
|
33
33
|
import { pickMachine, planResize } from "../lib/sandbox-vm.mjs";
|
|
34
|
+
import { listProcesses, staleSessions, formatStaleSessions } from "../lib/stale-sessions.mjs";
|
|
34
35
|
import { MIN_PODMAN_VM_MB } from "../lib/doctor.mjs";
|
|
35
36
|
import { liveLeases } from "../lib/slots.mjs";
|
|
36
37
|
import { ensureProviderConfig } from "../lib/openclaw-config.mjs";
|
|
@@ -238,7 +239,7 @@ Usage: nomarmy <command> [options]
|
|
|
238
239
|
project for this repository, committed for the team
|
|
239
240
|
(.mcp.json or .cursor/mcp.json, running \`nomarmy mcp\`).
|
|
240
241
|
Codex has only the user scope.
|
|
241
|
-
stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>]
|
|
242
|
+
stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>] [--details] [--all-suggestions]
|
|
242
243
|
[--repo <path|name>] [--all-repos] [--json]
|
|
243
244
|
What nomArmy's job records show for this repository (or
|
|
244
245
|
all): volume by role and model, code committed, time,
|
|
@@ -1714,13 +1715,14 @@ async function cmdUpdate() {
|
|
|
1714
1715
|
if (!copyIsStale(defaultInstallDir(), nomarmyRoot)) {
|
|
1715
1716
|
if (json) return out({ updated: false, reason: "already up to date" });
|
|
1716
1717
|
console.log(c.green("ā Already up to date, and your coordinators run this checkout."));
|
|
1718
|
+
printSessionRestarts({ quietWhenNone: true });
|
|
1717
1719
|
return;
|
|
1718
1720
|
}
|
|
1719
1721
|
say(c.bold("šŖ nomArmy update\n"));
|
|
1720
1722
|
say("Nothing to pull, but your coordinators run an older copy of this checkout.");
|
|
1721
1723
|
const resynced = reconnectCoordinators();
|
|
1722
1724
|
if (json) return out({ updated: false, resynced, sha: local });
|
|
1723
|
-
|
|
1725
|
+
printSessionRestarts();
|
|
1724
1726
|
return;
|
|
1725
1727
|
}
|
|
1726
1728
|
if (base !== local) {
|
|
@@ -1747,6 +1749,25 @@ async function cmdUpdate() {
|
|
|
1747
1749
|
// Reconnect every connected coordinator through a child process, so it runs
|
|
1748
1750
|
// the code now on disk (just pulled or installed) rather than the old code
|
|
1749
1751
|
// this process loaded. Returns the targets reconnected.
|
|
1752
|
+
// Each open session keeps the nomArmy it started with: name the ones that
|
|
1753
|
+
// started before the installed copy, rather than a blanket "restart".
|
|
1754
|
+
function printSessionRestarts({ quietWhenNone = false } = {}) {
|
|
1755
|
+
let list = null;
|
|
1756
|
+
try {
|
|
1757
|
+
const installedAt = fs.statSync(path.join(defaultInstallDir(), "source.json")).mtimeMs;
|
|
1758
|
+
const procs = listProcesses();
|
|
1759
|
+
if (procs) list = staleSessions(procs, { installedAt });
|
|
1760
|
+
} catch { /* no installed copy yet, or ps unavailable */ }
|
|
1761
|
+
if (list === null) {
|
|
1762
|
+
if (!quietWhenNone) console.log(c.yellow("\nRestart every open Claude Code, Codex and Cursor session: each keeps the code it started with until then."));
|
|
1763
|
+
return;
|
|
1764
|
+
}
|
|
1765
|
+
if (!list.length) { if (!quietWhenNone) console.log(c.green("\nā No open session runs an older nomArmy.")); return; }
|
|
1766
|
+
console.log(c.yellow(`\n${list.length} open session(s) still run an older nomArmy, until each is restarted:`));
|
|
1767
|
+
for (const line of formatStaleSessions(list)) console.log(line);
|
|
1768
|
+
console.log(c.dim("In Claude Code: /exit, then claude --resume (or /mcp ā nomarmy-local-worker ā Reconnect). Close any you no longer use."));
|
|
1769
|
+
}
|
|
1770
|
+
|
|
1750
1771
|
function reconnectCoordinators() {
|
|
1751
1772
|
const targets = connectedTargets();
|
|
1752
1773
|
// Per-repo registrations run the installed copy (or `nomarmy mcp`), so a
|
|
@@ -1791,7 +1812,7 @@ async function updateFromNpm() {
|
|
|
1791
1812
|
}
|
|
1792
1813
|
const targets = reconnectCoordinators();
|
|
1793
1814
|
if (json) return out({ updated: upgrade, from: current, version: upgrade ? latest : current, resynced: targets });
|
|
1794
|
-
|
|
1815
|
+
printSessionRestarts();
|
|
1795
1816
|
}
|
|
1796
1817
|
|
|
1797
1818
|
function commandExists(cmd) {
|
|
@@ -2747,9 +2768,9 @@ function cmdStats() {
|
|
|
2747
2768
|
}
|
|
2748
2769
|
let agentFor = () => null;
|
|
2749
2770
|
try { agentFor = agentLookup(loadAgents(globalConfigDir()).agents, agentProviderId); } catch { /* no agents.yml: commands name <agent> */ }
|
|
2750
|
-
const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model"), agentFor });
|
|
2771
|
+
const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model"), agentFor, allSuggestions: flag("all-suggestions") });
|
|
2751
2772
|
if (json) return out(stats);
|
|
2752
|
-
console.log(formatStats(stats));
|
|
2773
|
+
console.log(flag("details") ? formatStats(stats) : formatStatsSummary(stats, { c }));
|
|
2753
2774
|
}
|
|
2754
2775
|
|
|
2755
2776
|
// Read one line without echoing it: stty -echo around the read, restored
|
package/lib/health.mjs
CHANGED
|
@@ -337,8 +337,8 @@ export async function checkAndRecordHealth({ projectDir, stateRoot, configDir, n
|
|
|
337
337
|
const { loadJobRecords, agentLookup } = await import("./stats.mjs");
|
|
338
338
|
const { recentSuggestions } = await import("./suggestions.mjs");
|
|
339
339
|
for (const s of recentSuggestions(loadJobRecords(path.join(stateRoot, "jobs")), { projectDir, agentFor: agents ? agentLookup(agents, agentProviderId) : () => null, now })) {
|
|
340
|
-
if (s.level !== "warn") continue;
|
|
341
|
-
result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: "routing tip" });
|
|
340
|
+
if (s.level !== "warn" && s.level !== "act") continue;
|
|
341
|
+
result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: s.level === "act" ? "review needed" : "routing tip" });
|
|
342
342
|
}
|
|
343
343
|
} catch { /* suggestions never break a health check */ }
|
|
344
344
|
const toNotify = recordHealth(path.join(stateRoot, "health.json"), result, { now });
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
// Which open coordinator sessions still run an older nomArmy. Each Claude
|
|
2
|
+
// Code, Codex or Cursor session starts its own nomArmy server and keeps the
|
|
3
|
+
// code it started with, so after an update "restart your sessions" wasn't
|
|
4
|
+
// enough: one real afternoon had a session on alpha.14 for hours and four
|
|
5
|
+
// more, days old, on alpha.6 to alpha.12. This names each one.
|
|
6
|
+
|
|
7
|
+
import { spawnSync } from "node:child_process";
|
|
8
|
+
import path from "node:path";
|
|
9
|
+
|
|
10
|
+
// The installed copy's server, or the portable `nomarmy mcp` launcher.
|
|
11
|
+
const SERVER_RE = /nomarmy-local-worker[\\/]mcp[\\/]server\.mjs|\bnomarmy(?:\.mjs)?\s+mcp\b/;
|
|
12
|
+
|
|
13
|
+
/** `ps -A -o pid=,ppid=,tty=,lstart=,args=` lines. lstart reads "Sun Sep 27 21:33:11 2026" on macOS and Linux alike. */
|
|
14
|
+
export function parsePs(text) {
|
|
15
|
+
const out = [];
|
|
16
|
+
for (const line of String(text ?? "").split("\n")) {
|
|
17
|
+
const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+\w{3}\s+(\w{3})\s+(\d{1,2})\s+(\d{2}:\d{2}:\d{2})\s+(\d{4})\s+(.*)$/.exec(line);
|
|
18
|
+
if (!m) continue;
|
|
19
|
+
const startedAt = new Date(`${m[4]} ${m[5]} ${m[7]} ${m[6]}`).getTime();
|
|
20
|
+
if (!Number.isFinite(startedAt)) continue;
|
|
21
|
+
out.push({ pid: Number(m[1]), ppid: Number(m[2]), tty: /^\?+$/.test(m[3]) ? null : m[3], startedAt, args: m[8].trim() });
|
|
22
|
+
}
|
|
23
|
+
return out;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** The app a server belongs to, from its parent's command line. */
|
|
27
|
+
export function appName(args) {
|
|
28
|
+
const first = String(args ?? "").split(/\s+/)[0] ?? "";
|
|
29
|
+
const base = path.basename(first).toLowerCase();
|
|
30
|
+
if (base === "claude" || /claude/.test(base)) return "Claude Code";
|
|
31
|
+
if (base === "codex" || /codex/.test(base)) return "Codex";
|
|
32
|
+
if (/cursor/i.test(first)) return "Cursor";
|
|
33
|
+
return base || "a session";
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Servers that started before the copy they'd now load was installed. */
|
|
37
|
+
export function staleSessions(procs, { installedAt }) {
|
|
38
|
+
if (!Number.isFinite(installedAt)) return [];
|
|
39
|
+
const byPid = new Map(procs.map((p) => [p.pid, p]));
|
|
40
|
+
return procs
|
|
41
|
+
.filter((p) => SERVER_RE.test(p.args) && p.startedAt < installedAt)
|
|
42
|
+
.map((p) => {
|
|
43
|
+
const parent = byPid.get(p.ppid);
|
|
44
|
+
return { pid: p.pid, appPid: parent?.pid ?? p.ppid, app: appName(parent?.args), tty: p.tty ?? parent?.tty ?? null, startedAt: p.startedAt };
|
|
45
|
+
})
|
|
46
|
+
.sort((a, b) => a.startedAt - b.startedAt);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Running processes, or null where `ps` isn't available (Windows). */
|
|
50
|
+
export function listProcesses({ run = spawnSync, platform = process.platform } = {}) {
|
|
51
|
+
if (platform === "win32") return null;
|
|
52
|
+
const res = run("ps", ["-A", "-o", "pid=,ppid=,tty=,lstart=,args="], { encoding: "utf8", timeout: 10000 });
|
|
53
|
+
return res.status === 0 ? parsePs(res.stdout) : null;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export function formatStaleSessions(list, { now = Date.now() } = {}) {
|
|
57
|
+
const when = (ms) => new Date(ms).toLocaleString("en-US", { weekday: "short", hour: "numeric", minute: "2-digit" });
|
|
58
|
+
const age = (ms) => { const h = Math.round((now - ms) / 3600000); return h < 24 ? `${h}h ago` : `${Math.round(h / 24)}d ago`; };
|
|
59
|
+
return list.map((s) => ` ${s.app}${s.tty ? ` on ${s.tty}` : ""}, started ${when(s.startedAt)} (${age(s.startedAt)}), pid ${s.appPid}`);
|
|
60
|
+
}
|
package/lib/stats.mjs
CHANGED
|
@@ -99,7 +99,7 @@ export function resolveRepo(records, value) {
|
|
|
99
99
|
* @param {object[]} records
|
|
100
100
|
* @param {{ repo?: string|null, sinceMs?: number|null, untilMs?: number|null, role?: string|null, model?: string|null }} filter
|
|
101
101
|
*/
|
|
102
|
-
export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, agentFor = () => null, now = Date.now() } = {}) {
|
|
102
|
+
export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, agentFor = () => null, now = Date.now(), allSuggestions = false } = {}) {
|
|
103
103
|
const inRange = records.filter((r) => {
|
|
104
104
|
const at = Date.parse(r.startedAt ?? r.finishedAt ?? "");
|
|
105
105
|
if (sinceMs != null && !(at >= sinceMs)) return false;
|
|
@@ -136,6 +136,8 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
136
136
|
const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
|
|
137
137
|
const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
|
|
138
138
|
const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
|
|
139
|
+
// New tests the revert check showed would catch their change going away.
|
|
140
|
+
const provenTestFiles = committed.filter((r) => r.regressionCheck?.status === "pass").reduce((n, r) => n + (r.testChanges?.new_tests_added?.length ?? r.metrics?.new_tests_added ?? 0), 0);
|
|
139
141
|
|
|
140
142
|
const signals = {};
|
|
141
143
|
for (const [name, re] of SIGNALS) {
|
|
@@ -181,16 +183,18 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
181
183
|
passedBoth: passedBoth.length,
|
|
182
184
|
changedNothing: changedNothing.length,
|
|
183
185
|
flaggedAfterPassing: flaggedAfterPassing.length,
|
|
186
|
+
provenTestFiles,
|
|
184
187
|
},
|
|
185
188
|
notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
|
|
186
189
|
reviewers,
|
|
187
190
|
signals: sortDesc(signals),
|
|
188
191
|
highStakes: (() => {
|
|
189
|
-
|
|
192
|
+
// Work that landed: an uncommitted partial isn't accepted work.
|
|
193
|
+
const high = implement.filter((r) => r.stakes === "high" && r.commit?.created);
|
|
190
194
|
return { jobs: high.length, reviewed: high.filter((r) => reviewOf(r, records)).length };
|
|
191
195
|
})(),
|
|
192
196
|
// How you're set up now: the last 14 days unless a period was asked for.
|
|
193
|
-
suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor }),
|
|
197
|
+
suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor, now, includeStale: allSuggestions }),
|
|
194
198
|
suggestionWindow: sinceMs == null ? "the last 14 days" : "this period",
|
|
195
199
|
};
|
|
196
200
|
}
|
|
@@ -200,14 +204,83 @@ const list = (obj) => Object.entries(obj).map(([k, v]) => `${k} ${v}`).join(" Ā·
|
|
|
200
204
|
const mins = (m) => (m == null ? "n/a" : `${m.toFixed(1)} min`);
|
|
201
205
|
const big = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${(n / 1e3).toFixed(1)}k` : String(n));
|
|
202
206
|
|
|
207
|
+
/** The headline: claims that didn't hold up, and tests shown to catch their change. */
|
|
208
|
+
function caughtLines(c) {
|
|
209
|
+
if (!c.claimedDone) return [" no implement job reported \"done, tests pass\" in this period"];
|
|
210
|
+
const wrong = c.verificationFailed + c.revertStillPassed;
|
|
211
|
+
const parts = [c.verificationFailed && `${c.verificationFailed} failed when nomArmy ran the tests itself`, c.revertStillPassed && `${c.revertStillPassed} had tests that still pass with the change reverted`].filter(Boolean);
|
|
212
|
+
return [
|
|
213
|
+
wrong ? ` ${wrong} of ${c.claimedDone} "done, tests pass" claims didn't hold up: ${parts.join(", ")}` : ` all ${c.claimedDone} "done, tests pass" claims held up when nomArmy checked them`,
|
|
214
|
+
...(c.flaggedAfterPassing ? [` ${c.flaggedAfterPassing} more passed both but were flagged (mutants, Jev, judge, a rewritten check)`] : []),
|
|
215
|
+
...(c.provenTestFiles ? [` ${c.provenTestFiles} new test file(s) shown to fail without their change`] : []),
|
|
216
|
+
];
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* The default view: one screen. What nomArmy caught, what needs you, the top
|
|
221
|
+
* tips, and the totals. `--details` prints formatStats. `c` paints (the CLI
|
|
222
|
+
* passes its colors, plain when not a terminal); left out, it's plain text.
|
|
223
|
+
*/
|
|
224
|
+
const PLAIN = { bold: String, dim: String, red: String, green: String, yellow: String, cyan: String };
|
|
225
|
+
export function formatStatsSummary(s, { c = PLAIN, width = 28 } = {}) {
|
|
226
|
+
const cv = s.claimVsEvidence;
|
|
227
|
+
const where = s.repo ? path.basename(s.repo) : "all repositories";
|
|
228
|
+
const month = (iso) => new Date(iso).toLocaleString("en-US", { month: "short", day: "numeric", timeZone: "UTC" });
|
|
229
|
+
const when = s.period.from ? `${month(s.period.from)} to ${month(s.period.to)}` : "no jobs yet";
|
|
230
|
+
const act = (s.suggestions ?? []).filter((x) => x.level === "act");
|
|
231
|
+
// The spend share is on the totals line already.
|
|
232
|
+
const tips = (s.suggestions ?? []).filter((x) => x.level !== "act" && x.key !== "stale" && !x.key.startsWith("spend:"));
|
|
233
|
+
const hidden = (s.suggestions ?? []).find((x) => x.key === "stale");
|
|
234
|
+
const shown = tips.slice(0, 3), more = tips.length - shown.length;
|
|
235
|
+
const label = (t) => c.bold(t.padEnd(9));
|
|
236
|
+
const lines = [`${c.bold("nomArmy stats")} ${c.dim(`${where} Ā· ${when} Ā· ${s.volume.jobs} jobs Ā· ${s.code.committedJobs} committed`)}`, ""];
|
|
237
|
+
|
|
238
|
+
// What the checks caught: the reason to run nomArmy, first.
|
|
239
|
+
if (cv.claimedDone) {
|
|
240
|
+
const wrong = cv.verificationFailed + cv.revertStillPassed, held = cv.claimedDone - wrong;
|
|
241
|
+
const bad = wrong ? Math.max(1, Math.round((width * wrong) / cv.claimedDone)) : 0;
|
|
242
|
+
lines.push(`${label("CAUGHT")}${c.green("ā".repeat(width - bad))}${c.red("ā".repeat(bad))} ${held} of ${cv.claimedDone} "done, tests pass" claims held up${wrong ? c.red(` Ā· ${wrong} didn't`) : ""}`);
|
|
243
|
+
const why = [cv.verificationFailed && `${cv.verificationFailed} failed when nomArmy ran the tests itself`, cv.revertStillPassed && `${cv.revertStillPassed} had tests that pass with the change reverted`, cv.flaggedAfterPassing && `${cv.flaggedAfterPassing} passed but were flagged`].filter(Boolean);
|
|
244
|
+
if (why.length) lines.push(`${" ".repeat(9)}${c.dim(why.join(" Ā· "))}`);
|
|
245
|
+
} else lines.push(`${label("CAUGHT")}${c.dim('no job reported "done, tests pass" in this period')}`);
|
|
246
|
+
if (cv.provenTestFiles) lines.push(`${label("PROVEN")}${c.green("ā")} ${cv.provenTestFiles} new test files fail without their change`);
|
|
247
|
+
|
|
248
|
+
for (const x of act) {
|
|
249
|
+
const ids = /: (.+)$/.exec(x.title)?.[1]?.split(", ") ?? [];
|
|
250
|
+
const head = x.title.replace(/: .+$/, "");
|
|
251
|
+
lines.push("", `${c.red(c.bold("ā REVIEW BEFORE MERGING"))} ${head}`);
|
|
252
|
+
for (let i = 0; i < ids.length; i += 2) lines.push(` ${ids.slice(i, i + 2).map((id) => id.padEnd(32)).join("")}`.trimEnd());
|
|
253
|
+
lines.push(` ${c.cyan("ā")} a scout on another vendor with ${c.cyan("reviews: <job id>")} (army_role security-analyst); a failed review doesn't count`);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
lines.push("");
|
|
257
|
+
if (!shown.length) lines.push(`${label("TIPS")}${c.dim("none: nothing in the records suggests a routing change")}`);
|
|
258
|
+
shown.forEach((t, i) => {
|
|
259
|
+
lines.push(`${i ? " ".repeat(9) : label("TIPS")}${t.level === "warn" ? c.yellow("ā²") : c.dim("Ā·")} ${t.title}`);
|
|
260
|
+
if (t.command) lines.push(`${" ".repeat(11)}${c.cyan(`ā ${t.command}`)}`);
|
|
261
|
+
});
|
|
262
|
+
const staleCount = hidden ? Number(/^\d+/.exec(hidden.title)?.[0] ?? 0) : 0;
|
|
263
|
+
const notes = [more > 0 && `${more} more (--details)`, staleCount && `${staleCount} about pairings unused for 3+ days (--all-suggestions)`].filter(Boolean);
|
|
264
|
+
if (notes.length) lines.push(`${" ".repeat(11)}${c.dim(notes.join(" Ā· "))}`);
|
|
265
|
+
|
|
266
|
+
const top = Object.entries(s.spendUsd.byModel)[0];
|
|
267
|
+
lines.push("", `${label("SPEND")}$${s.spendUsd.total.toFixed(2)} API${top ? c.dim(` (${top[0]} ${Math.round((100 * top[1]) / (s.spendUsd.total || 1))}%)`) : ""} Ā· ${Math.round(s.jobMinutes.total)} min of jobs Ā· ${big(s.tokens.total)} tokens`);
|
|
268
|
+
lines.push("", c.dim("Everything else (volume, reviewers, flags, what didn't finish): nomarmy stats --details"));
|
|
269
|
+
return lines.join("\n");
|
|
270
|
+
}
|
|
271
|
+
|
|
203
272
|
/** The terminal report. */
|
|
204
273
|
export function formatStats(s) {
|
|
205
274
|
const c = s.claimVsEvidence;
|
|
206
275
|
const lines = [
|
|
207
276
|
`nomArmy stats${s.repo ? ` for ${s.repo}` : " (all repositories)"}${s.role ? `, role ${s.role}` : ""}${s.model ? `, model ${s.model}` : ""}, ${s.period.from ? `${s.period.from.slice(0, 10)} to ${s.period.to.slice(0, 10)}` : "no jobs"}`,
|
|
208
277
|
"",
|
|
278
|
+
"WHAT NOMARMY CAUGHT",
|
|
279
|
+
...caughtLines(c),
|
|
280
|
+
...(() => { const act = (s.suggestions ?? []).filter((x) => x.level === "act"); return act.length ? ["", "NEEDS YOUR ATTENTION", ...formatSuggestions(act)] : []; })(),
|
|
281
|
+
"",
|
|
209
282
|
`SUGGESTIONS (from ${s.suggestionWindow ?? "this period"}; never applied for you)`,
|
|
210
|
-
...formatSuggestions(s.suggestions ?? []),
|
|
283
|
+
...formatSuggestions((s.suggestions ?? []).filter((x) => x.level !== "act")),
|
|
211
284
|
"",
|
|
212
285
|
"VOLUME",
|
|
213
286
|
` Jobs ${s.volume.jobs}: ${list(s.volume.byMode)}${s.unplacedVerifyRuns ? ` (plus ${s.unplacedVerifyRuns} older verify run(s) that don't record their repository)` : ""}`,
|
|
@@ -225,7 +298,7 @@ export function formatStats(s) {
|
|
|
225
298
|
` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
|
|
226
299
|
` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
|
|
227
300
|
...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
|
|
228
|
-
` High-stakes jobs
|
|
301
|
+
` High-stakes jobs committed ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with a finished independent review`,
|
|
229
302
|
" Defects the General found at integration aren't in the records; count them in your own review.",
|
|
230
303
|
"",
|
|
231
304
|
"DIDN'T COMPLETE",
|
package/lib/suggestions.mjs
CHANGED
|
@@ -19,10 +19,15 @@ const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.work
|
|
|
19
19
|
export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
|
|
20
20
|
const pct = (n, of) => Math.round((100 * n) / of);
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
const REVIEW_FINISHED = /^(SCOUT_DONE|SCOUT_NOT_FOUND)$/;
|
|
23
|
+
// How recently a pairing must have run for a suggestion about it to still be
|
|
24
|
+
// about how you work now: a week-old local-model experiment led Senti's list.
|
|
25
|
+
export const CURRENT_DAYS = 3;
|
|
26
|
+
|
|
27
|
+
/** Whether a high-stakes job has had an independent review: a finished scout, or a judge, on another vendor. A review that timed out or failed isn't one. */
|
|
23
28
|
export function reviewOf(job, records) {
|
|
24
29
|
const workerProvider = provider(job);
|
|
25
|
-
const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && provider(r) && provider(r) !== workerProvider);
|
|
30
|
+
const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && REVIEW_FINISHED.test(r.outcome ?? "") && provider(r) && provider(r) !== workerProvider);
|
|
26
31
|
if (scout) return { by: "scout", jobId: scout.jobId, provider: provider(scout) };
|
|
27
32
|
const judge = job.validators?.judge;
|
|
28
33
|
if (judge?.answer && judge.provider && judge.provider !== workerProvider) return { by: "judge", provider: judge.provider };
|
|
@@ -33,16 +38,18 @@ export function reviewOf(job, records) {
|
|
|
33
38
|
* @param {object[]} records this repo's records, already filtered to a period
|
|
34
39
|
* @returns {{ level: "warn"|"info", key: string, title: string, evidence: string, command: string|null }[]}
|
|
35
40
|
*/
|
|
36
|
-
export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null } = {}) {
|
|
41
|
+
export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null, now = Date.now(), includeStale = false } = {}) {
|
|
37
42
|
const out = [];
|
|
43
|
+
let stale = 0;
|
|
38
44
|
const work = records.filter((r) => r.mode === "implement" || r.mode === "scout");
|
|
39
45
|
|
|
40
46
|
// Per role and model.
|
|
41
47
|
const groups = new Map();
|
|
42
48
|
for (const r of work) {
|
|
43
49
|
const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
|
|
44
|
-
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
|
|
50
|
+
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0, lastAt: 0 };
|
|
45
51
|
g.jobs++;
|
|
52
|
+
g.lastAt = Math.max(g.lastAt, Date.parse(r.startedAt ?? "") || 0);
|
|
46
53
|
if (runnerFailed(r)) g.runner++; else g.rated++;
|
|
47
54
|
if (OK.test(r.outcome ?? "")) g.ok++;
|
|
48
55
|
if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
|
|
@@ -53,8 +60,15 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
53
60
|
}
|
|
54
61
|
const assign = (g, to = "<another model>") => (g.role ? `nomarmy army assign ${g.role} ${g.agent ?? "<agent>"} ${to}` : null);
|
|
55
62
|
|
|
63
|
+
const current = (g) => includeStale || g.lastAt >= now - CURRENT_DAYS * 86400000;
|
|
56
64
|
for (const g of groups.values()) {
|
|
57
65
|
if (!g.model) continue;
|
|
66
|
+
// A stale pairing's suggestions are worked out, then only counted.
|
|
67
|
+
const before = out.length;
|
|
68
|
+
groupSuggestions(g);
|
|
69
|
+
if (!current(g)) stale += out.splice(before).length;
|
|
70
|
+
}
|
|
71
|
+
function groupSuggestions(g) {
|
|
58
72
|
const kind = g.mode === "scout" ? "scouts" : "implement jobs";
|
|
59
73
|
const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
|
|
60
74
|
// The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
|
|
@@ -66,9 +80,9 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
66
80
|
if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
|
|
67
81
|
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
|
|
68
82
|
evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
|
|
69
|
-
|
|
83
|
+
return;
|
|
70
84
|
}
|
|
71
|
-
if (g.rated < minJobs)
|
|
85
|
+
if (g.rated < minJobs) return;
|
|
72
86
|
const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
|
|
73
87
|
// A pairing that rarely finishes.
|
|
74
88
|
if (g.ok / g.rated < 0.5) {
|
|
@@ -87,7 +101,7 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
87
101
|
|
|
88
102
|
// A lighter model doing as well on the same role's implement work, for much less. Only
|
|
89
103
|
// within one role: different roles do different work, so across roles the numbers don't compare.
|
|
90
|
-
const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
|
|
104
|
+
const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs && current(g));
|
|
91
105
|
for (const heavy of impl) {
|
|
92
106
|
for (const light of impl) {
|
|
93
107
|
if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
|
|
@@ -112,22 +126,28 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
112
126
|
evidence: "Worth knowing rather than changing if it's catching real problems; check its reviews' findings before moving it.", command: null });
|
|
113
127
|
}
|
|
114
128
|
|
|
115
|
-
//
|
|
116
|
-
|
|
129
|
+
// Committed high-stakes work without an independent review. Only work that
|
|
130
|
+
// landed: a partial never committed isn't accepted work, and one finished by
|
|
131
|
+
// a later job is that job's to review.
|
|
132
|
+
const unreviewed = records.filter((r) => r.mode === "implement" && r.stakes === "high" && r.commit?.created && !reviewOf(r, records));
|
|
117
133
|
if (unreviewed.length) {
|
|
118
|
-
|
|
119
|
-
|
|
134
|
+
const ids = unreviewed.map((r) => r.jobId).sort();
|
|
135
|
+
out.unshift({ level: "act", key: `unreviewed:${ids.join(",")}`, title: `${ids.length} high-stakes job(s) committed without an independent review: ${ids.join(", ")}`,
|
|
136
|
+
evidence: "Review each before merging: a scout on another vendor (army_role security-analyst, say) with reviews: <job id>. A review that failed or timed out doesn't count.", command: null });
|
|
120
137
|
}
|
|
138
|
+
const rank = { act: 0, warn: 1, info: 2 };
|
|
139
|
+
out.sort((a, b) => rank[a.level] - rank[b.level]);
|
|
140
|
+
if (stale) out.push({ level: "info", key: "stale", title: `${stale} more about role and model pairings you haven't used in ${CURRENT_DAYS} days, hidden (nomarmy stats --all-suggestions)`, evidence: null, command: null });
|
|
121
141
|
return out;
|
|
122
142
|
}
|
|
123
143
|
|
|
124
144
|
/** This repository's suggestions from its last 14 days of jobs. */
|
|
125
145
|
export function recentSuggestions(records, { projectDir, agentFor = () => null, now = Date.now(), days = 14 } = {}) {
|
|
126
146
|
const since = now - days * 86400000;
|
|
127
|
-
return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor });
|
|
147
|
+
return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor, now });
|
|
128
148
|
}
|
|
129
149
|
|
|
130
150
|
export function formatSuggestions(list) {
|
|
131
151
|
if (!list.length) return [" none: nothing in the records suggests a routing change"];
|
|
132
|
-
return list.flatMap((s) => [` ${
|
|
152
|
+
return list.flatMap((s) => [` ${{ act: "!!", warn: "!", info: "-" }[s.level] ?? "-"} ${s.title}`, ...(s.evidence ? [` ${s.evidence}`] : []), ...(s.command ? [` ${s.command}`] : [])]);
|
|
133
153
|
}
|
package/mcp/server.mjs
CHANGED
|
@@ -37,7 +37,7 @@ import { modelRefusals } from "../lib/health.mjs";
|
|
|
37
37
|
import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
|
|
38
38
|
import { restartNotice } from "../lib/install-freshness.mjs";
|
|
39
39
|
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
40
|
-
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
40
|
+
import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
41
41
|
import { recentSuggestions } from "../lib/suggestions.mjs";
|
|
42
42
|
import { probeModel } from "../lib/model-probe.mjs";
|
|
43
43
|
import { jevSettings, judgeSettings } from "../lib/validators.mjs";
|
|
@@ -453,13 +453,14 @@ server.tool("stats", "What nomArmy's own job records show for this repository (o
|
|
|
453
453
|
role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Only jobs dispatched as this army role (e.g. sr-dev)."),
|
|
454
454
|
model: z.string().regex(/^\S{1,200}$/).optional().describe("Only jobs that ran on this model (e.g. grok-4.7)."),
|
|
455
455
|
format: z.enum(["text", "json"]).optional().describe("text (default) is the report; json is the raw numbers."),
|
|
456
|
-
|
|
456
|
+
details: z.boolean().optional().describe("The full report (volume, reviewers, flags, what didn't finish). Default is the one-screen summary: what nomArmy caught, high-stakes work needing review, the top routing tips, spend."),
|
|
457
|
+
}, async ({ since, until, all_repos, repo, role, model, format, details }) => {
|
|
457
458
|
try {
|
|
458
459
|
const records = loadJobRecords(jobsRoot);
|
|
459
460
|
let agentFor = () => null;
|
|
460
461
|
try { agentFor = agentLookup(agentsConfig().agents, agentProviderId); } catch { /* commands name <agent> */ }
|
|
461
462
|
const stats = computeStats(records, { repo: repo ? resolveRepo(records, repo) : all_repos ? null : projectDir, sinceMs: parseSince(since), untilMs: parseSince(until), role: role ?? null, model: model ?? null, agentFor });
|
|
462
|
-
return toolText(format === "json" ? JSON.stringify(stats, null, 2) : formatStats(stats));
|
|
463
|
+
return toolText(format === "json" ? JSON.stringify(stats, null, 2) : details ? formatStats(stats) : formatStatsSummary(stats));
|
|
463
464
|
} catch (error) { return toolText(error.message, true); }
|
|
464
465
|
});
|
|
465
466
|
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.17",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|