nomarmy 0.1.0-alpha.16 → 0.1.0-alpha.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/nomarmy.mjs CHANGED
@@ -20,7 +20,7 @@ import { readGGUFMetadata, resolveModelPath, totalSplitBytes } from "../lib/gguf
20
20
  import { recommend, customRecommendation, evaluateConfig, bytesPerKvElementForCacheTypes, MIN_CONTEXT_PER_NOM } from "../lib/sizing.mjs";
21
21
  import { connectClaude, connectCodex, connectCursor, cursorAlreadyConnected, deriveWorkerModelEnv, defaultInstallDir, installMcpCopy, SCOPES, claudeUserScoped, portableServerLaunch } from "../lib/connect.mjs";
22
22
  import { compareVersions, readPackageVersion, readInstallVersions, copyIsStale } from "../lib/install-freshness.mjs";
23
- import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
23
+ import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
24
24
  import { requestJobStop } from "../lib/openclaw-run.mjs";
25
25
  import { loadValidators, saveJevKey, removeJev, jevSettings, askJev, validatorsPath, JEV_CHECKS, saveJudge, removeJudge, judgeSettings } from "../lib/validators.mjs";
26
26
  import { probeModel } from "../lib/model-probe.mjs";
@@ -31,6 +31,7 @@ import { parseLlamaUrl } from "../lib/execution.mjs";
31
31
  import { setupSteps, formatSetupSteps, runSetupPlaybook } from "../lib/setup-steps.mjs";
32
32
  import { readUsageSnapshots } from "../lib/usage-limits.mjs";
33
33
  import { pickMachine, planResize } from "../lib/sandbox-vm.mjs";
34
+ import { listProcesses, staleSessions, formatStaleSessions } from "../lib/stale-sessions.mjs";
34
35
  import { MIN_PODMAN_VM_MB } from "../lib/doctor.mjs";
35
36
  import { liveLeases } from "../lib/slots.mjs";
36
37
  import { ensureProviderConfig } from "../lib/openclaw-config.mjs";
@@ -238,7 +239,7 @@ Usage: nomarmy <command> [options]
238
239
  project for this repository, committed for the team
239
240
  (.mcp.json or .cursor/mcp.json, running \`nomarmy mcp\`).
240
241
  Codex has only the user scope.
241
- stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>]
242
+ stats [--since 7d|<date>] [--until <date>] [--role <role>] [--model <model>] [--details] [--all-suggestions]
242
243
  [--repo <path|name>] [--all-repos] [--json]
243
244
  What nomArmy's job records show for this repository (or
244
245
  all): volume by role and model, code committed, time,
@@ -1714,13 +1715,14 @@ async function cmdUpdate() {
1714
1715
  if (!copyIsStale(defaultInstallDir(), nomarmyRoot)) {
1715
1716
  if (json) return out({ updated: false, reason: "already up to date" });
1716
1717
  console.log(c.green("āœ“ Already up to date, and your coordinators run this checkout."));
1718
+ printSessionRestarts({ quietWhenNone: true });
1717
1719
  return;
1718
1720
  }
1719
1721
  say(c.bold("šŸŖ nomArmy update\n"));
1720
1722
  say("Nothing to pull, but your coordinators run an older copy of this checkout.");
1721
1723
  const resynced = reconnectCoordinators();
1722
1724
  if (json) return out({ updated: false, resynced, sha: local });
1723
- console.log(c.yellow("\nRestart every open Claude Code, Codex and Cursor session: each keeps the code it started with until then."));
1725
+ printSessionRestarts();
1724
1726
  return;
1725
1727
  }
1726
1728
  if (base !== local) {
@@ -1747,6 +1749,25 @@ async function cmdUpdate() {
1747
1749
  // Reconnect every connected coordinator through a child process, so it runs
1748
1750
  // the code now on disk (just pulled or installed) rather than the old code
1749
1751
  // this process loaded. Returns the targets reconnected.
1752
+ // Each open session keeps the nomArmy it started with: name the ones that
1753
+ // started before the installed copy, rather than a blanket "restart".
1754
+ function printSessionRestarts({ quietWhenNone = false } = {}) {
1755
+ let list = null;
1756
+ try {
1757
+ const installedAt = fs.statSync(path.join(defaultInstallDir(), "source.json")).mtimeMs;
1758
+ const procs = listProcesses();
1759
+ if (procs) list = staleSessions(procs, { installedAt });
1760
+ } catch { /* no installed copy yet, or ps unavailable */ }
1761
+ if (list === null) {
1762
+ if (!quietWhenNone) console.log(c.yellow("\nRestart every open Claude Code, Codex and Cursor session: each keeps the code it started with until then."));
1763
+ return;
1764
+ }
1765
+ if (!list.length) { if (!quietWhenNone) console.log(c.green("\nāœ“ No open session runs an older nomArmy.")); return; }
1766
+ console.log(c.yellow(`\n${list.length} open session(s) still run an older nomArmy, until each is restarted:`));
1767
+ for (const line of formatStaleSessions(list)) console.log(line);
1768
+ console.log(c.dim("In Claude Code: /exit, then claude --resume (or /mcp → nomarmy-local-worker → Reconnect). Close any you no longer use."));
1769
+ }
1770
+
1750
1771
  function reconnectCoordinators() {
1751
1772
  const targets = connectedTargets();
1752
1773
  // Per-repo registrations run the installed copy (or `nomarmy mcp`), so a
@@ -1791,7 +1812,7 @@ async function updateFromNpm() {
1791
1812
  }
1792
1813
  const targets = reconnectCoordinators();
1793
1814
  if (json) return out({ updated: upgrade, from: current, version: upgrade ? latest : current, resynced: targets });
1794
- console.log(c.yellow("\nRestart every open Claude Code, Codex and Cursor session: each keeps the code it started with until then."));
1815
+ printSessionRestarts();
1795
1816
  }
1796
1817
 
1797
1818
  function commandExists(cmd) {
@@ -2747,9 +2768,9 @@ function cmdStats() {
2747
2768
  }
2748
2769
  let agentFor = () => null;
2749
2770
  try { agentFor = agentLookup(loadAgents(globalConfigDir()).agents, agentProviderId); } catch { /* no agents.yml: commands name <agent> */ }
2750
- const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model"), agentFor });
2771
+ const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model"), agentFor, allSuggestions: flag("all-suggestions") });
2751
2772
  if (json) return out(stats);
2752
- console.log(formatStats(stats));
2773
+ console.log(flag("details") ? formatStats(stats) : formatStatsSummary(stats, { c }));
2753
2774
  }
2754
2775
 
2755
2776
  // Read one line without echoing it: stty -echo around the read, restored
package/lib/health.mjs CHANGED
@@ -337,8 +337,8 @@ export async function checkAndRecordHealth({ projectDir, stateRoot, configDir, n
337
337
  const { loadJobRecords, agentLookup } = await import("./stats.mjs");
338
338
  const { recentSuggestions } = await import("./suggestions.mjs");
339
339
  for (const s of recentSuggestions(loadJobRecords(path.join(stateRoot, "jobs")), { projectDir, agentFor: agents ? agentLookup(agents, agentProviderId) : () => null, now })) {
340
- if (s.level !== "warn") continue;
341
- result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: "routing tip" });
340
+ if (s.level !== "warn" && s.level !== "act") continue;
341
+ result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: s.level === "act" ? "review needed" : "routing tip" });
342
342
  }
343
343
  } catch { /* suggestions never break a health check */ }
344
344
  const toNotify = recordHealth(path.join(stateRoot, "health.json"), result, { now });
@@ -0,0 +1,60 @@
1
+ // Which open coordinator sessions still run an older nomArmy. Each Claude
2
+ // Code, Codex or Cursor session starts its own nomArmy server and keeps the
3
+ // code it started with, so after an update "restart your sessions" wasn't
4
+ // enough: one real afternoon had a session on alpha.14 for hours and four
5
+ // more, days old, on alpha.6 to alpha.12. This names each one.
6
+
7
+ import { spawnSync } from "node:child_process";
8
+ import path from "node:path";
9
+
10
+ // The installed copy's server, or the portable `nomarmy mcp` launcher.
11
+ const SERVER_RE = /nomarmy-local-worker[\\/]mcp[\\/]server\.mjs|\bnomarmy(?:\.mjs)?\s+mcp\b/;
12
+
13
+ /** `ps -A -o pid=,ppid=,tty=,lstart=,args=` lines. lstart reads "Sun Sep 27 21:33:11 2026" on macOS and Linux alike. */
14
+ export function parsePs(text) {
15
+ const out = [];
16
+ for (const line of String(text ?? "").split("\n")) {
17
+ const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+\w{3}\s+(\w{3})\s+(\d{1,2})\s+(\d{2}:\d{2}:\d{2})\s+(\d{4})\s+(.*)$/.exec(line);
18
+ if (!m) continue;
19
+ const startedAt = new Date(`${m[4]} ${m[5]} ${m[7]} ${m[6]}`).getTime();
20
+ if (!Number.isFinite(startedAt)) continue;
21
+ out.push({ pid: Number(m[1]), ppid: Number(m[2]), tty: /^\?+$/.test(m[3]) ? null : m[3], startedAt, args: m[8].trim() });
22
+ }
23
+ return out;
24
+ }
25
+
26
+ /** The app a server belongs to, from its parent's command line. */
27
+ export function appName(args) {
28
+ const first = String(args ?? "").split(/\s+/)[0] ?? "";
29
+ const base = path.basename(first).toLowerCase();
30
+ if (base === "claude" || /claude/.test(base)) return "Claude Code";
31
+ if (base === "codex" || /codex/.test(base)) return "Codex";
32
+ if (/cursor/i.test(first)) return "Cursor";
33
+ return base || "a session";
34
+ }
35
+
36
+ /** Servers that started before the copy they'd now load was installed. */
37
+ export function staleSessions(procs, { installedAt }) {
38
+ if (!Number.isFinite(installedAt)) return [];
39
+ const byPid = new Map(procs.map((p) => [p.pid, p]));
40
+ return procs
41
+ .filter((p) => SERVER_RE.test(p.args) && p.startedAt < installedAt)
42
+ .map((p) => {
43
+ const parent = byPid.get(p.ppid);
44
+ return { pid: p.pid, appPid: parent?.pid ?? p.ppid, app: appName(parent?.args), tty: p.tty ?? parent?.tty ?? null, startedAt: p.startedAt };
45
+ })
46
+ .sort((a, b) => a.startedAt - b.startedAt);
47
+ }
48
+
49
+ /** Running processes, or null where `ps` isn't available (Windows). */
50
+ export function listProcesses({ run = spawnSync, platform = process.platform } = {}) {
51
+ if (platform === "win32") return null;
52
+ const res = run("ps", ["-A", "-o", "pid=,ppid=,tty=,lstart=,args="], { encoding: "utf8", timeout: 10000 });
53
+ return res.status === 0 ? parsePs(res.stdout) : null;
54
+ }
55
+
56
+ export function formatStaleSessions(list, { now = Date.now() } = {}) {
57
+ const when = (ms) => new Date(ms).toLocaleString("en-US", { weekday: "short", hour: "numeric", minute: "2-digit" });
58
+ const age = (ms) => { const h = Math.round((now - ms) / 3600000); return h < 24 ? `${h}h ago` : `${Math.round(h / 24)}d ago`; };
59
+ return list.map((s) => ` ${s.app}${s.tty ? ` on ${s.tty}` : ""}, started ${when(s.startedAt)} (${age(s.startedAt)}), pid ${s.appPid}`);
60
+ }
package/lib/stats.mjs CHANGED
@@ -99,7 +99,7 @@ export function resolveRepo(records, value) {
99
99
  * @param {object[]} records
100
100
  * @param {{ repo?: string|null, sinceMs?: number|null, untilMs?: number|null, role?: string|null, model?: string|null }} filter
101
101
  */
102
- export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, agentFor = () => null, now = Date.now() } = {}) {
102
+ export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, agentFor = () => null, now = Date.now(), allSuggestions = false } = {}) {
103
103
  const inRange = records.filter((r) => {
104
104
  const at = Date.parse(r.startedAt ?? r.finishedAt ?? "");
105
105
  if (sinceMs != null && !(at >= sinceMs)) return false;
@@ -136,6 +136,8 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
136
136
  const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
137
137
  const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
138
138
  const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
139
+ // New tests the revert check showed would catch their change going away.
140
+ const provenTestFiles = committed.filter((r) => r.regressionCheck?.status === "pass").reduce((n, r) => n + (r.testChanges?.new_tests_added?.length ?? r.metrics?.new_tests_added ?? 0), 0);
139
141
 
140
142
  const signals = {};
141
143
  for (const [name, re] of SIGNALS) {
@@ -181,16 +183,18 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
181
183
  passedBoth: passedBoth.length,
182
184
  changedNothing: changedNothing.length,
183
185
  flaggedAfterPassing: flaggedAfterPassing.length,
186
+ provenTestFiles,
184
187
  },
185
188
  notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
186
189
  reviewers,
187
190
  signals: sortDesc(signals),
188
191
  highStakes: (() => {
189
- const high = implement.filter((r) => r.stakes === "high");
192
+ // Work that landed: an uncommitted partial isn't accepted work.
193
+ const high = implement.filter((r) => r.stakes === "high" && r.commit?.created);
190
194
  return { jobs: high.length, reviewed: high.filter((r) => reviewOf(r, records)).length };
191
195
  })(),
192
196
  // How you're set up now: the last 14 days unless a period was asked for.
193
- suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor }),
197
+ suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor, now, includeStale: allSuggestions }),
194
198
  suggestionWindow: sinceMs == null ? "the last 14 days" : "this period",
195
199
  };
196
200
  }
@@ -200,14 +204,83 @@ const list = (obj) => Object.entries(obj).map(([k, v]) => `${k} ${v}`).join(" Ā·
200
204
  const mins = (m) => (m == null ? "n/a" : `${m.toFixed(1)} min`);
201
205
  const big = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${(n / 1e3).toFixed(1)}k` : String(n));
202
206
 
207
+ /** The headline: claims that didn't hold up, and tests shown to catch their change. */
208
+ function caughtLines(c) {
209
+ if (!c.claimedDone) return [" no implement job reported \"done, tests pass\" in this period"];
210
+ const wrong = c.verificationFailed + c.revertStillPassed;
211
+ const parts = [c.verificationFailed && `${c.verificationFailed} failed when nomArmy ran the tests itself`, c.revertStillPassed && `${c.revertStillPassed} had tests that still pass with the change reverted`].filter(Boolean);
212
+ return [
213
+ wrong ? ` ${wrong} of ${c.claimedDone} "done, tests pass" claims didn't hold up: ${parts.join(", ")}` : ` all ${c.claimedDone} "done, tests pass" claims held up when nomArmy checked them`,
214
+ ...(c.flaggedAfterPassing ? [` ${c.flaggedAfterPassing} more passed both but were flagged (mutants, Jev, judge, a rewritten check)`] : []),
215
+ ...(c.provenTestFiles ? [` ${c.provenTestFiles} new test file(s) shown to fail without their change`] : []),
216
+ ];
217
+ }
218
+
219
+ /**
220
+ * The default view: one screen. What nomArmy caught, what needs you, the top
221
+ * tips, and the totals. `--details` prints formatStats. `c` paints (the CLI
222
+ * passes its colors, plain when not a terminal); left out, it's plain text.
223
+ */
224
+ const PLAIN = { bold: String, dim: String, red: String, green: String, yellow: String, cyan: String };
225
+ export function formatStatsSummary(s, { c = PLAIN, width = 28 } = {}) {
226
+ const cv = s.claimVsEvidence;
227
+ const where = s.repo ? path.basename(s.repo) : "all repositories";
228
+ const month = (iso) => new Date(iso).toLocaleString("en-US", { month: "short", day: "numeric", timeZone: "UTC" });
229
+ const when = s.period.from ? `${month(s.period.from)} to ${month(s.period.to)}` : "no jobs yet";
230
+ const act = (s.suggestions ?? []).filter((x) => x.level === "act");
231
+ // The spend share is on the totals line already.
232
+ const tips = (s.suggestions ?? []).filter((x) => x.level !== "act" && x.key !== "stale" && !x.key.startsWith("spend:"));
233
+ const hidden = (s.suggestions ?? []).find((x) => x.key === "stale");
234
+ const shown = tips.slice(0, 3), more = tips.length - shown.length;
235
+ const label = (t) => c.bold(t.padEnd(9));
236
+ const lines = [`${c.bold("nomArmy stats")} ${c.dim(`${where} Ā· ${when} Ā· ${s.volume.jobs} jobs Ā· ${s.code.committedJobs} committed`)}`, ""];
237
+
238
+ // What the checks caught: the reason to run nomArmy, first.
239
+ if (cv.claimedDone) {
240
+ const wrong = cv.verificationFailed + cv.revertStillPassed, held = cv.claimedDone - wrong;
241
+ const bad = wrong ? Math.max(1, Math.round((width * wrong) / cv.claimedDone)) : 0;
242
+ lines.push(`${label("CAUGHT")}${c.green("ā–ˆ".repeat(width - bad))}${c.red("ā–‘".repeat(bad))} ${held} of ${cv.claimedDone} "done, tests pass" claims held up${wrong ? c.red(` Ā· ${wrong} didn't`) : ""}`);
243
+ const why = [cv.verificationFailed && `${cv.verificationFailed} failed when nomArmy ran the tests itself`, cv.revertStillPassed && `${cv.revertStillPassed} had tests that pass with the change reverted`, cv.flaggedAfterPassing && `${cv.flaggedAfterPassing} passed but were flagged`].filter(Boolean);
244
+ if (why.length) lines.push(`${" ".repeat(9)}${c.dim(why.join(" Ā· "))}`);
245
+ } else lines.push(`${label("CAUGHT")}${c.dim('no job reported "done, tests pass" in this period')}`);
246
+ if (cv.provenTestFiles) lines.push(`${label("PROVEN")}${c.green("āœ“")} ${cv.provenTestFiles} new test files fail without their change`);
247
+
248
+ for (const x of act) {
249
+ const ids = /: (.+)$/.exec(x.title)?.[1]?.split(", ") ?? [];
250
+ const head = x.title.replace(/: .+$/, "");
251
+ lines.push("", `${c.red(c.bold("⚠ REVIEW BEFORE MERGING"))} ${head}`);
252
+ for (let i = 0; i < ids.length; i += 2) lines.push(` ${ids.slice(i, i + 2).map((id) => id.padEnd(32)).join("")}`.trimEnd());
253
+ lines.push(` ${c.cyan("→")} a scout on another vendor with ${c.cyan("reviews: <job id>")} (army_role security-analyst); a failed review doesn't count`);
254
+ }
255
+
256
+ lines.push("");
257
+ if (!shown.length) lines.push(`${label("TIPS")}${c.dim("none: nothing in the records suggests a routing change")}`);
258
+ shown.forEach((t, i) => {
259
+ lines.push(`${i ? " ".repeat(9) : label("TIPS")}${t.level === "warn" ? c.yellow("ā–²") : c.dim("Ā·")} ${t.title}`);
260
+ if (t.command) lines.push(`${" ".repeat(11)}${c.cyan(`→ ${t.command}`)}`);
261
+ });
262
+ const staleCount = hidden ? Number(/^\d+/.exec(hidden.title)?.[0] ?? 0) : 0;
263
+ const notes = [more > 0 && `${more} more (--details)`, staleCount && `${staleCount} about pairings unused for 3+ days (--all-suggestions)`].filter(Boolean);
264
+ if (notes.length) lines.push(`${" ".repeat(11)}${c.dim(notes.join(" Ā· "))}`);
265
+
266
+ const top = Object.entries(s.spendUsd.byModel)[0];
267
+ lines.push("", `${label("SPEND")}$${s.spendUsd.total.toFixed(2)} API${top ? c.dim(` (${top[0]} ${Math.round((100 * top[1]) / (s.spendUsd.total || 1))}%)`) : ""} Ā· ${Math.round(s.jobMinutes.total)} min of jobs Ā· ${big(s.tokens.total)} tokens`);
268
+ lines.push("", c.dim("Everything else (volume, reviewers, flags, what didn't finish): nomarmy stats --details"));
269
+ return lines.join("\n");
270
+ }
271
+
203
272
  /** The terminal report. */
204
273
  export function formatStats(s) {
205
274
  const c = s.claimVsEvidence;
206
275
  const lines = [
207
276
  `nomArmy stats${s.repo ? ` for ${s.repo}` : " (all repositories)"}${s.role ? `, role ${s.role}` : ""}${s.model ? `, model ${s.model}` : ""}, ${s.period.from ? `${s.period.from.slice(0, 10)} to ${s.period.to.slice(0, 10)}` : "no jobs"}`,
208
277
  "",
278
+ "WHAT NOMARMY CAUGHT",
279
+ ...caughtLines(c),
280
+ ...(() => { const act = (s.suggestions ?? []).filter((x) => x.level === "act"); return act.length ? ["", "NEEDS YOUR ATTENTION", ...formatSuggestions(act)] : []; })(),
281
+ "",
209
282
  `SUGGESTIONS (from ${s.suggestionWindow ?? "this period"}; never applied for you)`,
210
- ...formatSuggestions(s.suggestions ?? []),
283
+ ...formatSuggestions((s.suggestions ?? []).filter((x) => x.level !== "act")),
211
284
  "",
212
285
  "VOLUME",
213
286
  ` Jobs ${s.volume.jobs}: ${list(s.volume.byMode)}${s.unplacedVerifyRuns ? ` (plus ${s.unplacedVerifyRuns} older verify run(s) that don't record their repository)` : ""}`,
@@ -225,7 +298,7 @@ export function formatStats(s) {
225
298
  ` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
226
299
  ` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
227
300
  ...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
228
- ` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
301
+ ` High-stakes jobs committed ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with a finished independent review`,
229
302
  " Defects the General found at integration aren't in the records; count them in your own review.",
230
303
  "",
231
304
  "DIDN'T COMPLETE",
@@ -19,10 +19,15 @@ const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.work
19
19
  export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
20
20
  const pct = (n, of) => Math.round((100 * n) / of);
21
21
 
22
- /** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
22
+ const REVIEW_FINISHED = /^(SCOUT_DONE|SCOUT_NOT_FOUND)$/;
23
+ // How recently a pairing must have run for a suggestion about it to still be
24
+ // about how you work now: a week-old local-model experiment led Senti's list.
25
+ export const CURRENT_DAYS = 3;
26
+
27
+ /** Whether a high-stakes job has had an independent review: a finished scout, or a judge, on another vendor. A review that timed out or failed isn't one. */
23
28
  export function reviewOf(job, records) {
24
29
  const workerProvider = provider(job);
25
- const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && provider(r) && provider(r) !== workerProvider);
30
+ const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && REVIEW_FINISHED.test(r.outcome ?? "") && provider(r) && provider(r) !== workerProvider);
26
31
  if (scout) return { by: "scout", jobId: scout.jobId, provider: provider(scout) };
27
32
  const judge = job.validators?.judge;
28
33
  if (judge?.answer && judge.provider && judge.provider !== workerProvider) return { by: "judge", provider: judge.provider };
@@ -33,16 +38,18 @@ export function reviewOf(job, records) {
33
38
  * @param {object[]} records this repo's records, already filtered to a period
34
39
  * @returns {{ level: "warn"|"info", key: string, title: string, evidence: string, command: string|null }[]}
35
40
  */
36
- export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null } = {}) {
41
+ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null, now = Date.now(), includeStale = false } = {}) {
37
42
  const out = [];
43
+ let stale = 0;
38
44
  const work = records.filter((r) => r.mode === "implement" || r.mode === "scout");
39
45
 
40
46
  // Per role and model.
41
47
  const groups = new Map();
42
48
  for (const r of work) {
43
49
  const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
44
- const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
50
+ const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0, lastAt: 0 };
45
51
  g.jobs++;
52
+ g.lastAt = Math.max(g.lastAt, Date.parse(r.startedAt ?? "") || 0);
46
53
  if (runnerFailed(r)) g.runner++; else g.rated++;
47
54
  if (OK.test(r.outcome ?? "")) g.ok++;
48
55
  if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
@@ -53,8 +60,15 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
53
60
  }
54
61
  const assign = (g, to = "<another model>") => (g.role ? `nomarmy army assign ${g.role} ${g.agent ?? "<agent>"} ${to}` : null);
55
62
 
63
+ const current = (g) => includeStale || g.lastAt >= now - CURRENT_DAYS * 86400000;
56
64
  for (const g of groups.values()) {
57
65
  if (!g.model) continue;
66
+ // A stale pairing's suggestions are worked out, then only counted.
67
+ const before = out.length;
68
+ groupSuggestions(g);
69
+ if (!current(g)) stale += out.splice(before).length;
70
+ }
71
+ function groupSuggestions(g) {
58
72
  const kind = g.mode === "scout" ? "scouts" : "implement jobs";
59
73
  const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
60
74
  // The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
@@ -66,9 +80,9 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
66
80
  if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
67
81
  out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
68
82
  evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
69
- continue;
83
+ return;
70
84
  }
71
- if (g.rated < minJobs) continue;
85
+ if (g.rated < minJobs) return;
72
86
  const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
73
87
  // A pairing that rarely finishes.
74
88
  if (g.ok / g.rated < 0.5) {
@@ -87,7 +101,7 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
87
101
 
88
102
  // A lighter model doing as well on the same role's implement work, for much less. Only
89
103
  // within one role: different roles do different work, so across roles the numbers don't compare.
90
- const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
104
+ const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs && current(g));
91
105
  for (const heavy of impl) {
92
106
  for (const light of impl) {
93
107
  if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
@@ -112,22 +126,28 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
112
126
  evidence: "Worth knowing rather than changing if it's catching real problems; check its reviews' findings before moving it.", command: null });
113
127
  }
114
128
 
115
- // High-stakes work without an independent review.
116
- const unreviewed = records.filter((r) => r.mode === "implement" && r.stakes === "high" && !reviewOf(r, records));
129
+ // Committed high-stakes work without an independent review. Only work that
130
+ // landed: a partial never committed isn't accepted work, and one finished by
131
+ // a later job is that job's to review.
132
+ const unreviewed = records.filter((r) => r.mode === "implement" && r.stakes === "high" && r.commit?.created && !reviewOf(r, records));
117
133
  if (unreviewed.length) {
118
- out.push({ level: "warn", key: `unreviewed:${unreviewed.map((r) => r.jobId).sort().join(",")}`, title: `${unreviewed.length} high-stakes job(s) without an independent review: ${unreviewed.slice(0, 5).map((r) => r.jobId).join(", ")}`,
119
- evidence: "Send a scout on another vendor with reviews: <job id> before accepting them (or configure a judge on another vendor).", command: null });
134
+ const ids = unreviewed.map((r) => r.jobId).sort();
135
+ out.unshift({ level: "act", key: `unreviewed:${ids.join(",")}`, title: `${ids.length} high-stakes job(s) committed without an independent review: ${ids.join(", ")}`,
136
+ evidence: "Review each before merging: a scout on another vendor (army_role security-analyst, say) with reviews: <job id>. A review that failed or timed out doesn't count.", command: null });
120
137
  }
138
+ const rank = { act: 0, warn: 1, info: 2 };
139
+ out.sort((a, b) => rank[a.level] - rank[b.level]);
140
+ if (stale) out.push({ level: "info", key: "stale", title: `${stale} more about role and model pairings you haven't used in ${CURRENT_DAYS} days, hidden (nomarmy stats --all-suggestions)`, evidence: null, command: null });
121
141
  return out;
122
142
  }
123
143
 
124
144
  /** This repository's suggestions from its last 14 days of jobs. */
125
145
  export function recentSuggestions(records, { projectDir, agentFor = () => null, now = Date.now(), days = 14 } = {}) {
126
146
  const since = now - days * 86400000;
127
- return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor });
147
+ return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor, now });
128
148
  }
129
149
 
130
150
  export function formatSuggestions(list) {
131
151
  if (!list.length) return [" none: nothing in the records suggests a routing change"];
132
- return list.flatMap((s) => [` ${s.level === "warn" ? "!" : "-"} ${s.title}`, ` ${s.evidence}`, ...(s.command ? [` ${s.command}`] : [])]);
152
+ return list.flatMap((s) => [` ${{ act: "!!", warn: "!", info: "-" }[s.level] ?? "-"} ${s.title}`, ...(s.evidence ? [` ${s.evidence}`] : []), ...(s.command ? [` ${s.command}`] : [])]);
133
153
  }
package/mcp/server.mjs CHANGED
@@ -37,7 +37,7 @@ import { modelRefusals } from "../lib/health.mjs";
37
37
  import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
38
38
  import { restartNotice } from "../lib/install-freshness.mjs";
39
39
  import { requestJobStop } from "../lib/openclaw-run.mjs";
40
- import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
40
+ import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
41
41
  import { recentSuggestions } from "../lib/suggestions.mjs";
42
42
  import { probeModel } from "../lib/model-probe.mjs";
43
43
  import { jevSettings, judgeSettings } from "../lib/validators.mjs";
@@ -453,13 +453,14 @@ server.tool("stats", "What nomArmy's own job records show for this repository (o
453
453
  role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Only jobs dispatched as this army role (e.g. sr-dev)."),
454
454
  model: z.string().regex(/^\S{1,200}$/).optional().describe("Only jobs that ran on this model (e.g. grok-4.7)."),
455
455
  format: z.enum(["text", "json"]).optional().describe("text (default) is the report; json is the raw numbers."),
456
- }, async ({ since, until, all_repos, repo, role, model, format }) => {
456
+ details: z.boolean().optional().describe("The full report (volume, reviewers, flags, what didn't finish). Default is the one-screen summary: what nomArmy caught, high-stakes work needing review, the top routing tips, spend."),
457
+ }, async ({ since, until, all_repos, repo, role, model, format, details }) => {
457
458
  try {
458
459
  const records = loadJobRecords(jobsRoot);
459
460
  let agentFor = () => null;
460
461
  try { agentFor = agentLookup(agentsConfig().agents, agentProviderId); } catch { /* commands name <agent> */ }
461
462
  const stats = computeStats(records, { repo: repo ? resolveRepo(records, repo) : all_repos ? null : projectDir, sinceMs: parseSince(since), untilMs: parseSince(until), role: role ?? null, model: model ?? null, agentFor });
462
- return toolText(format === "json" ? JSON.stringify(stats, null, 2) : formatStats(stats));
463
+ return toolText(format === "json" ? JSON.stringify(stats, null, 2) : details ? formatStats(stats) : formatStatsSummary(stats));
463
464
  } catch (error) { return toolText(error.message, true); }
464
465
  });
465
466
 
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
4
4
  "author": "Rayson Technologies",
5
5
  "license": "Apache-2.0",
6
- "version": "0.1.0-alpha.16",
6
+ "version": "0.1.0-alpha.17",
7
7
  "private": false,
8
8
  "type": "module",
9
9
  "engines": {