nomarmy 0.1.0-alpha.14 → 0.1.0-alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/stats.mjs CHANGED
@@ -118,12 +118,13 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
118
118
 
119
119
  const workerMinutes = implement.map((r) => r.metrics?.worker_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
120
120
  const jobMinutes = implement.map((r) => r.metrics?.total_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
121
- const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
121
+ const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, untracked: 0 };
122
122
  const spend = {};
123
123
  for (const r of jobs) {
124
124
  const m = r.metrics ?? {};
125
125
  tokens.total += m.worker_tokens_total ?? 0; tokens.input += m.worker_tokens_in ?? 0; tokens.output += m.worker_tokens_out ?? 0;
126
126
  tokens.cacheRead += m.worker_tokens_cache_read ?? 0; tokens.cacheWrite += m.worker_tokens_cache_write ?? 0;
127
+ if (r.mode !== "verify" && !(m.worker_tokens_total > 0)) tokens.untracked++;
127
128
  if (Number.isFinite(m.worker_cost_usd) && m.worker_cost_usd > 0) spend[jobModel(r) ?? "unknown"] = (spend[jobModel(r) ?? "unknown"] ?? 0) + m.worker_cost_usd;
128
129
  }
129
130
 
@@ -131,6 +132,8 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
131
132
  const claimedDone = implement.filter((r) => r.reportValidation?.status === "done" && r.reportValidation?.tests === "pass");
132
133
  const verificationFailed = claimedDone.filter((r) => r.independentVerification?.status === "fail");
133
134
  const revertStillPassed = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status === "fail");
135
+ // Claimed success with an empty diff: there was nothing to verify.
136
+ const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
134
137
  const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
135
138
  const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
136
139
 
@@ -176,6 +179,7 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
176
179
  verificationFailed: verificationFailed.length,
177
180
  revertStillPassed: revertStillPassed.length,
178
181
  passedBoth: passedBoth.length,
182
+ changedNothing: changedNothing.length,
179
183
  flaggedAfterPassing: flaggedAfterPassing.length,
180
184
  },
181
185
  notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
@@ -212,7 +216,7 @@ export function formatStats(s) {
212
216
  ` Committed ${s.code.committedJobs} job(s) · +${s.code.linesAdded} / -${s.code.linesRemoved} lines · ${s.code.files} files · ${s.code.newTestFiles} new test files`,
213
217
  ` Worker time median ${mins(s.workerMinutes.median)} per implement job, p90 ${mins(s.workerMinutes.p90)}, total ${Math.round(s.workerMinutes.total)} min`,
214
218
  ` Job time median ${mins(s.jobMinutes.median)}, p90 ${mins(s.jobMinutes.p90)}, total ${Math.round(s.jobMinutes.total)} min (with verification and checks)`,
215
- ` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)`,
219
+ ` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)${s.tokens.untracked ? `; ${s.tokens.untracked} job(s) recorded no token counts` : ""}`,
216
220
  ` API spend $${s.spendUsd.total.toFixed(2)}${Object.keys(s.spendUsd.byModel).length ? ` (${Object.entries(s.spendUsd.byModel).map(([k, v]) => `${k} $${v.toFixed(2)}`).join(", ")})` : ""}; subscriptions aren't billed per call`,
217
221
  "",
218
222
  `CLAIM VS EVIDENCE (implement jobs that reported "done, tests pass": ${c.claimedDone})`,
@@ -220,6 +224,7 @@ export function formatStats(s) {
220
224
  ` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
221
225
  ` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
222
226
  ` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
227
+ ...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
223
228
  ` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
224
229
  " Defects the General found at integration aren't in the records; count them in your own review.",
225
230
  "",
@@ -11,7 +11,12 @@ const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE
11
11
 
12
12
  const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
13
13
  const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
14
- const agent = (r) => r.labels?.agent ?? null;
14
+ // The local model is the built-in "local" agent, whatever model is loaded.
15
+ const agent = (r) => r.labels?.agent ?? (provider(r) === "llama-cpp" ? "local" : null);
16
+ // New tokens only: cache reads are most of an api job's total and cost a fraction.
17
+ const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.worker_tokens_out ?? 0);
18
+ /** The runner exited before a report: an OpenClaw, sandbox or provider failure, not the model's work. */
19
+ export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
15
20
  const pct = (n, of) => Math.round((100 * n) / of);
16
21
 
17
22
  /** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
@@ -36,11 +41,13 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
36
41
  const groups = new Map();
37
42
  for (const r of work) {
38
43
  const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
39
- const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, ok: 0, timeout: 0, unsupported: 0, tokens: 0 };
40
- g.jobs++; if (OK.test(r.outcome ?? "")) g.ok++;
44
+ const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
45
+ g.jobs++;
46
+ if (runnerFailed(r)) g.runner++; else g.rated++;
47
+ if (OK.test(r.outcome ?? "")) g.ok++;
41
48
  if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
42
49
  if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
43
- g.tokens += r.metrics?.worker_tokens_total ?? 0;
50
+ if (freshTokens(r) > 0) { g.tokens += freshTokens(r); g.tokenJobs++; }
44
51
  g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
45
52
  groups.set(key, g);
46
53
  }
@@ -50,39 +57,47 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
50
57
  if (!g.model) continue;
51
58
  const kind = g.mode === "scout" ? "scouts" : "implement jobs";
52
59
  const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
60
+ // The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
61
+ if (g.runner >= 3 && g.runner / g.jobs >= 0.4) {
62
+ out.push({ level: "warn", key: `runner-failed:${g.role}:${g.model}:${g.mode}`, title: `${who}: the runner failed on ${g.runner} of ${g.jobs} ${kind} before any report`,
63
+ evidence: "OpenClaw, the sandbox or the provider exited early, so these say nothing about the model's work. Check `nomarmy health` and one job's log (`nomarmy jobs <id>`); a model its vendor refuses fails this way too.", command: null });
64
+ }
53
65
  // Scouts that come back empty.
54
- if (g.mode === "scout" && g.jobs >= 3 && g.unsupported / g.jobs >= 0.4) {
55
- out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.jobs} scouts came back unsupported`,
66
+ if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
67
+ out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
56
68
  evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
57
69
  continue;
58
70
  }
59
- if (g.jobs < minJobs) continue;
71
+ if (g.rated < minJobs) continue;
72
+ const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
60
73
  // A pairing that rarely finishes.
61
- if (g.ok / g.jobs < 0.5) {
62
- const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.jobs >= minJobs && o.model && o.ok / o.jobs >= g.ok / g.jobs + 0.2)
63
- .sort((a, b) => b.ok / b.jobs - a.ok / a.jobs)[0];
64
- out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.jobs} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.jobs)}%)`,
65
- evidence: (better ? `${better.model} finished ${pct(better.ok, better.jobs)}% of its ${better.jobs} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
74
+ if (g.ok / g.rated < 0.5) {
75
+ const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.rated >= minJobs && o.model && o.ok / o.rated >= g.ok / g.rated + 0.2)
76
+ .sort((a, b) => b.ok / b.rated - a.ok / a.rated)[0];
77
+ out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.rated} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.rated)}%)${runnerNote}`,
78
+ evidence: (better ? `${better.model} finished ${pct(better.ok, better.rated)}% of its ${better.rated} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
66
79
  command: g.role ? assign(g, better?.model ?? "<another model>") : null });
67
80
  }
68
81
  // Timeouts.
69
- if (g.timeout >= 3 && g.timeout / g.jobs >= 0.25) {
70
- out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.jobs} jobs`,
82
+ if (g.timeout >= 3 && g.timeout / g.rated >= 0.25) {
83
+ out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.rated} jobs`,
71
84
  evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
72
85
  }
73
86
  }
74
87
 
75
- // A lighter model doing as well on implement work, for much less.
76
- const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.model && g.jobs >= minJobs && g.tokens > 0);
88
+ // A lighter model doing as well on the same role's implement work, for much less. Only
89
+ // within one role: different roles do different work, so across roles the numbers don't compare.
90
+ const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
77
91
  for (const heavy of impl) {
78
92
  for (const light of impl) {
79
- if (light === heavy || light.model === heavy.model) continue;
80
- const lightRate = light.ok / light.jobs, heavyRate = heavy.ok / heavy.jobs;
81
- if (lightRate >= heavyRate - 0.05 && light.tokens / light.jobs <= 0.5 * (heavy.tokens / heavy.jobs)) {
93
+ if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
94
+ const lightRate = light.ok / light.rated, heavyRate = heavy.ok / heavy.rated;
95
+ const lightPerJob = light.tokens / light.tokenJobs, heavyPerJob = heavy.tokens / heavy.tokenJobs;
96
+ if (lightRate >= heavyRate - 0.05 && lightPerJob <= 0.5 * heavyPerJob) {
82
97
  out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
83
- title: `${light.model} finished ${pct(light.ok, light.jobs)}% of its jobs${light.role ? ` (as ${light.role})` : ""} on ${Math.round((light.tokens / light.jobs) / 1000)}k tokens a job; ${heavy.model}${heavy.role ? ` (as ${heavy.role})` : ""} finished ${pct(heavy.ok, heavy.jobs)}% on ${Math.round((heavy.tokens / heavy.jobs) / 1000)}k`,
84
- evidence: "They did different work, so try it rather than switch outright: put the role on auto so the General picks per job, or send its routine pieces to the lighter model.",
85
- command: heavy.role ? `nomarmy army assign ${heavy.role} ${heavy.agent ?? "<agent>"} auto` : null });
98
+ title: `${heavy.role}: ${light.model} finished ${pct(light.ok, light.rated)}% of its jobs on ${Math.round(lightPerJob / 1000)}k new tokens a job; ${heavy.model} finished ${pct(heavy.ok, heavy.rated)}% on ${Math.round(heavyPerJob / 1000)}k`,
99
+ evidence: `Same role, so similar work. Moving ${heavy.role} to ${light.model} would cost less; keep ${heavy.model} for the harder pieces with model on the job.${light.agent === "local" ? ` The local agent runs whichever model is loaded; these ran on ${light.model}.` : ""}`,
100
+ command: light.agent ? `nomarmy army assign ${heavy.role} ${light.agent}${light.agent === "local" ? "" : ` ${light.model}`}` : null });
86
101
  }
87
102
  }
88
103
  }
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
4
4
  "author": "Rayson Technologies",
5
5
  "license": "Apache-2.0",
6
- "version": "0.1.0-alpha.14",
6
+ "version": "0.1.0-alpha.15",
7
7
  "private": false,
8
8
  "type": "module",
9
9
  "engines": {