nomarmy 0.1.0-alpha.14 → 0.1.0-alpha.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/stats.mjs +7 -2
- package/lib/suggestions.mjs +37 -22
- package/package.json +1 -1
package/lib/stats.mjs
CHANGED
|
@@ -118,12 +118,13 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
118
118
|
|
|
119
119
|
const workerMinutes = implement.map((r) => r.metrics?.worker_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
|
|
120
120
|
const jobMinutes = implement.map((r) => r.metrics?.total_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
|
|
121
|
-
const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
|
|
121
|
+
const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, untracked: 0 };
|
|
122
122
|
const spend = {};
|
|
123
123
|
for (const r of jobs) {
|
|
124
124
|
const m = r.metrics ?? {};
|
|
125
125
|
tokens.total += m.worker_tokens_total ?? 0; tokens.input += m.worker_tokens_in ?? 0; tokens.output += m.worker_tokens_out ?? 0;
|
|
126
126
|
tokens.cacheRead += m.worker_tokens_cache_read ?? 0; tokens.cacheWrite += m.worker_tokens_cache_write ?? 0;
|
|
127
|
+
if (r.mode !== "verify" && !(m.worker_tokens_total > 0)) tokens.untracked++;
|
|
127
128
|
if (Number.isFinite(m.worker_cost_usd) && m.worker_cost_usd > 0) spend[jobModel(r) ?? "unknown"] = (spend[jobModel(r) ?? "unknown"] ?? 0) + m.worker_cost_usd;
|
|
128
129
|
}
|
|
129
130
|
|
|
@@ -131,6 +132,8 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
131
132
|
const claimedDone = implement.filter((r) => r.reportValidation?.status === "done" && r.reportValidation?.tests === "pass");
|
|
132
133
|
const verificationFailed = claimedDone.filter((r) => r.independentVerification?.status === "fail");
|
|
133
134
|
const revertStillPassed = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status === "fail");
|
|
135
|
+
// Claimed success with an empty diff: there was nothing to verify.
|
|
136
|
+
const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
|
|
134
137
|
const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
|
|
135
138
|
const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
|
|
136
139
|
|
|
@@ -176,6 +179,7 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
176
179
|
verificationFailed: verificationFailed.length,
|
|
177
180
|
revertStillPassed: revertStillPassed.length,
|
|
178
181
|
passedBoth: passedBoth.length,
|
|
182
|
+
changedNothing: changedNothing.length,
|
|
179
183
|
flaggedAfterPassing: flaggedAfterPassing.length,
|
|
180
184
|
},
|
|
181
185
|
notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
|
|
@@ -212,7 +216,7 @@ export function formatStats(s) {
|
|
|
212
216
|
` Committed ${s.code.committedJobs} job(s) · +${s.code.linesAdded} / -${s.code.linesRemoved} lines · ${s.code.files} files · ${s.code.newTestFiles} new test files`,
|
|
213
217
|
` Worker time median ${mins(s.workerMinutes.median)} per implement job, p90 ${mins(s.workerMinutes.p90)}, total ${Math.round(s.workerMinutes.total)} min`,
|
|
214
218
|
` Job time median ${mins(s.jobMinutes.median)}, p90 ${mins(s.jobMinutes.p90)}, total ${Math.round(s.jobMinutes.total)} min (with verification and checks)`,
|
|
215
|
-
` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)`,
|
|
219
|
+
` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)${s.tokens.untracked ? `; ${s.tokens.untracked} job(s) recorded no token counts` : ""}`,
|
|
216
220
|
` API spend $${s.spendUsd.total.toFixed(2)}${Object.keys(s.spendUsd.byModel).length ? ` (${Object.entries(s.spendUsd.byModel).map(([k, v]) => `${k} $${v.toFixed(2)}`).join(", ")})` : ""}; subscriptions aren't billed per call`,
|
|
217
221
|
"",
|
|
218
222
|
`CLAIM VS EVIDENCE (implement jobs that reported "done, tests pass": ${c.claimedDone})`,
|
|
@@ -220,6 +224,7 @@ export function formatStats(s) {
|
|
|
220
224
|
` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
|
|
221
225
|
` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
|
|
222
226
|
` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
|
|
227
|
+
...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
|
|
223
228
|
` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
|
|
224
229
|
" Defects the General found at integration aren't in the records; count them in your own review.",
|
|
225
230
|
"",
|
package/lib/suggestions.mjs
CHANGED
|
@@ -11,7 +11,12 @@ const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE
|
|
|
11
11
|
|
|
12
12
|
const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
|
|
13
13
|
const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
|
|
14
|
-
|
|
14
|
+
// The local model is the built-in "local" agent, whatever model is loaded.
|
|
15
|
+
const agent = (r) => r.labels?.agent ?? (provider(r) === "llama-cpp" ? "local" : null);
|
|
16
|
+
// New tokens only: cache reads are most of an api job's total and cost a fraction.
|
|
17
|
+
const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.worker_tokens_out ?? 0);
|
|
18
|
+
/** The runner exited before a report: an OpenClaw, sandbox or provider failure, not the model's work. */
|
|
19
|
+
export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
|
|
15
20
|
const pct = (n, of) => Math.round((100 * n) / of);
|
|
16
21
|
|
|
17
22
|
/** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
|
|
@@ -36,11 +41,13 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
36
41
|
const groups = new Map();
|
|
37
42
|
for (const r of work) {
|
|
38
43
|
const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
|
|
39
|
-
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, ok: 0, timeout: 0, unsupported: 0, tokens: 0 };
|
|
40
|
-
g.jobs++;
|
|
44
|
+
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
|
|
45
|
+
g.jobs++;
|
|
46
|
+
if (runnerFailed(r)) g.runner++; else g.rated++;
|
|
47
|
+
if (OK.test(r.outcome ?? "")) g.ok++;
|
|
41
48
|
if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
|
|
42
49
|
if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
|
|
43
|
-
g.tokens += r.
|
|
50
|
+
if (freshTokens(r) > 0) { g.tokens += freshTokens(r); g.tokenJobs++; }
|
|
44
51
|
g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
|
|
45
52
|
groups.set(key, g);
|
|
46
53
|
}
|
|
@@ -50,39 +57,47 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
50
57
|
if (!g.model) continue;
|
|
51
58
|
const kind = g.mode === "scout" ? "scouts" : "implement jobs";
|
|
52
59
|
const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
|
|
60
|
+
// The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
|
|
61
|
+
if (g.runner >= 3 && g.runner / g.jobs >= 0.4) {
|
|
62
|
+
out.push({ level: "warn", key: `runner-failed:${g.role}:${g.model}:${g.mode}`, title: `${who}: the runner failed on ${g.runner} of ${g.jobs} ${kind} before any report`,
|
|
63
|
+
evidence: "OpenClaw, the sandbox or the provider exited early, so these say nothing about the model's work. Check `nomarmy health` and one job's log (`nomarmy jobs <id>`); a model its vendor refuses fails this way too.", command: null });
|
|
64
|
+
}
|
|
53
65
|
// Scouts that come back empty.
|
|
54
|
-
if (g.mode === "scout" && g.
|
|
55
|
-
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.
|
|
66
|
+
if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
|
|
67
|
+
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
|
|
56
68
|
evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
|
|
57
69
|
continue;
|
|
58
70
|
}
|
|
59
|
-
if (g.
|
|
71
|
+
if (g.rated < minJobs) continue;
|
|
72
|
+
const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
|
|
60
73
|
// A pairing that rarely finishes.
|
|
61
|
-
if (g.ok / g.
|
|
62
|
-
const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.
|
|
63
|
-
.sort((a, b) => b.ok / b.
|
|
64
|
-
out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.
|
|
65
|
-
evidence: (better ? `${better.model} finished ${pct(better.ok, better.
|
|
74
|
+
if (g.ok / g.rated < 0.5) {
|
|
75
|
+
const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.rated >= minJobs && o.model && o.ok / o.rated >= g.ok / g.rated + 0.2)
|
|
76
|
+
.sort((a, b) => b.ok / b.rated - a.ok / a.rated)[0];
|
|
77
|
+
out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.rated} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.rated)}%)${runnerNote}`,
|
|
78
|
+
evidence: (better ? `${better.model} finished ${pct(better.ok, better.rated)}% of its ${better.rated} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
|
|
66
79
|
command: g.role ? assign(g, better?.model ?? "<another model>") : null });
|
|
67
80
|
}
|
|
68
81
|
// Timeouts.
|
|
69
|
-
if (g.timeout >= 3 && g.timeout / g.
|
|
70
|
-
out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.
|
|
82
|
+
if (g.timeout >= 3 && g.timeout / g.rated >= 0.25) {
|
|
83
|
+
out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.rated} jobs`,
|
|
71
84
|
evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
|
|
72
85
|
}
|
|
73
86
|
}
|
|
74
87
|
|
|
75
|
-
// A lighter model doing as well on implement work, for much less.
|
|
76
|
-
|
|
88
|
+
// A lighter model doing as well on the same role's implement work, for much less. Only
|
|
89
|
+
// within one role: different roles do different work, so across roles the numbers don't compare.
|
|
90
|
+
const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
|
|
77
91
|
for (const heavy of impl) {
|
|
78
92
|
for (const light of impl) {
|
|
79
|
-
if (light === heavy || light.model === heavy.model) continue;
|
|
80
|
-
const lightRate = light.ok / light.
|
|
81
|
-
|
|
93
|
+
if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
|
|
94
|
+
const lightRate = light.ok / light.rated, heavyRate = heavy.ok / heavy.rated;
|
|
95
|
+
const lightPerJob = light.tokens / light.tokenJobs, heavyPerJob = heavy.tokens / heavy.tokenJobs;
|
|
96
|
+
if (lightRate >= heavyRate - 0.05 && lightPerJob <= 0.5 * heavyPerJob) {
|
|
82
97
|
out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
|
|
83
|
-
title: `${light.model} finished ${pct(light.ok, light.
|
|
84
|
-
evidence:
|
|
85
|
-
command:
|
|
98
|
+
title: `${heavy.role}: ${light.model} finished ${pct(light.ok, light.rated)}% of its jobs on ${Math.round(lightPerJob / 1000)}k new tokens a job; ${heavy.model} finished ${pct(heavy.ok, heavy.rated)}% on ${Math.round(heavyPerJob / 1000)}k`,
|
|
99
|
+
evidence: `Same role, so similar work. Moving ${heavy.role} to ${light.model} would cost less; keep ${heavy.model} for the harder pieces with model on the job.${light.agent === "local" ? ` The local agent runs whichever model is loaded; these ran on ${light.model}.` : ""}`,
|
|
100
|
+
command: light.agent ? `nomarmy army assign ${heavy.role} ${light.agent}${light.agent === "local" ? "" : ` ${light.model}`}` : null });
|
|
86
101
|
}
|
|
87
102
|
}
|
|
88
103
|
}
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.15",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|