nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +86 -480
  2. package/bin/nomarmy.mjs +1081 -185
  3. package/docker/Dockerfile +2 -2
  4. package/docker/Dockerfile.go +6 -4
  5. package/docker/Dockerfile.rust +17 -2
  6. package/harnesses/_template/README.md +27 -0
  7. package/harnesses/_template/harness.yml +26 -0
  8. package/harnesses/browser-playwright/README.md +35 -0
  9. package/harnesses/browser-playwright/fixture/package.json +1 -0
  10. package/harnesses/browser-playwright/fixture/page.html +1 -0
  11. package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
  12. package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
  13. package/harnesses/browser-playwright/harness.yml +18 -0
  14. package/harnesses/go/README.md +45 -0
  15. package/harnesses/go/harness.yml +14 -0
  16. package/harnesses/mock-oidc/README.md +31 -0
  17. package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
  18. package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
  19. package/harnesses/mock-oidc/harness.yml +19 -0
  20. package/harnesses/node/README.md +53 -0
  21. package/harnesses/node/harness.yml +18 -0
  22. package/harnesses/python/README.md +46 -0
  23. package/harnesses/python/harness.yml +16 -0
  24. package/harnesses/rust/README.md +45 -0
  25. package/harnesses/rust/harness.yml +13 -0
  26. package/install.sh +29 -9
  27. package/lib/admission.mjs +178 -30
  28. package/lib/agents.mjs +8 -6
  29. package/lib/army.mjs +25 -10
  30. package/lib/codex-link.mjs +37 -0
  31. package/lib/config.mjs +15 -0
  32. package/lib/connect.mjs +232 -19
  33. package/lib/continue-from.mjs +103 -0
  34. package/lib/coordinator-instructions.mjs +5 -1
  35. package/lib/diff-checks.mjs +114 -0
  36. package/lib/dispatch-schema.mjs +14 -12
  37. package/lib/doctor.mjs +98 -9
  38. package/lib/egress-proxy.mjs +116 -0
  39. package/lib/execute.mjs +241 -33
  40. package/lib/git-record.mjs +27 -3
  41. package/lib/harness-schema.mjs +61 -0
  42. package/lib/harnesses.mjs +99 -0
  43. package/lib/health.mjs +162 -18
  44. package/lib/install-freshness.mjs +114 -0
  45. package/lib/jev-checks.mjs +110 -0
  46. package/lib/job-format.mjs +54 -0
  47. package/lib/judge.mjs +130 -0
  48. package/lib/limits.mjs +77 -0
  49. package/lib/model-probe.mjs +61 -0
  50. package/lib/mutation.mjs +159 -0
  51. package/lib/notify.mjs +30 -3
  52. package/lib/openclaw-install.mjs +122 -0
  53. package/lib/openclaw-path.mjs +28 -0
  54. package/lib/openclaw-run.mjs +74 -12
  55. package/lib/openclaw-runtime-health.mjs +56 -0
  56. package/lib/outcome.mjs +21 -2
  57. package/lib/outcomes.mjs +6 -0
  58. package/lib/path-utils.mjs +4 -0
  59. package/lib/podman-health.mjs +41 -0
  60. package/lib/process.mjs +4 -1
  61. package/lib/propose.mjs +10 -11
  62. package/lib/refusal-retry.mjs +16 -0
  63. package/lib/registry-python.mjs +98 -0
  64. package/lib/registry-secrets.mjs +140 -0
  65. package/lib/repo-query.mjs +13 -7
  66. package/lib/runs.mjs +7 -1
  67. package/lib/same-path.mjs +14 -0
  68. package/lib/sandbox-images.mjs +499 -83
  69. package/lib/sandbox-vm.mjs +32 -0
  70. package/lib/scan.mjs +5 -1
  71. package/lib/schema.mjs +20 -11
  72. package/lib/scout.mjs +21 -3
  73. package/lib/server-context.mjs +21 -1
  74. package/lib/setup-steps.mjs +55 -0
  75. package/lib/share.mjs +82 -0
  76. package/lib/stale-sessions.mjs +60 -0
  77. package/lib/stats.mjs +315 -0
  78. package/lib/statusline.mjs +32 -6
  79. package/lib/subscription-setup.mjs +13 -0
  80. package/lib/suggestions.mjs +153 -0
  81. package/lib/thinking.mjs +23 -0
  82. package/lib/transcript.mjs +30 -5
  83. package/lib/usage-limits.mjs +329 -0
  84. package/lib/user-config.mjs +106 -0
  85. package/lib/validators.mjs +220 -0
  86. package/lib/verification-artifacts.mjs +46 -0
  87. package/lib/verification-flow.mjs +52 -7
  88. package/lib/verification-network.mjs +66 -0
  89. package/lib/verify.mjs +338 -85
  90. package/lib/worker-prompt.mjs +5 -2
  91. package/lib/wsl-cli.mjs +152 -0
  92. package/lib/wsl.mjs +230 -0
  93. package/lib/zod-issues.mjs +15 -0
  94. package/mcp/server.mjs +165 -34
  95. package/package.json +7 -5
  96. package/playbooks/feature.md +8 -5
  97. package/scripts/configure-openclaw.sh +4 -2
  98. package/scripts/generate-harness-docs.mjs +42 -0
  99. package/scripts/install-openclaw.mjs +23 -0
  100. package/scripts/lib.sh +9 -2
  101. package/scripts/select-model.mjs +12 -5
  102. package/scripts/start-inference.sh +2 -2
package/lib/stats.mjs ADDED
@@ -0,0 +1,315 @@
1
+ // nomarmy stats: what nomArmy's own job records show, for one repo or all,
2
+ // over a period. Every number comes from a job's verified record
3
+ // (metadata.json), never from a worker's report. What the records can't
4
+ // show (defects the General found at integration) is said, not guessed.
5
+
6
+ import fs from "node:fs";
7
+ import path from "node:path";
8
+ import { computeSuggestions, formatSuggestions, reviewOf } from "./suggestions.mjs";
9
+
10
+ const SUGGESTION_WINDOW_MS = 14 * 86400000;
11
+
12
+ /** Every readable job record under jobsRoot. */
13
+ export function loadJobRecords(jobsRoot) {
14
+ let names = [];
15
+ try { names = fs.readdirSync(jobsRoot); } catch { return []; }
16
+ const records = [];
17
+ for (const name of names) {
18
+ try { records.push(JSON.parse(fs.readFileSync(path.join(jobsRoot, name, "metadata.json"), "utf8"))); } catch { /* unfinished or unreadable */ }
19
+ }
20
+ return records;
21
+ }
22
+
23
+ /** An agent name from agents.yml for a provider id, when exactly one agent uses it. */
24
+ export function agentLookup(agents = {}, providerOf) {
25
+ return (provider) => {
26
+ if (!provider) return null;
27
+ const names = Object.entries(agents).filter(([, a]) => { try { return providerOf(a) === provider; } catch { return false; } }).map(([name]) => name);
28
+ return names.length === 1 ? names[0] : null;
29
+ };
30
+ }
31
+
32
+ /** "7d", "24h", or a date; returns epoch ms or null. */
33
+ export function parseSince(value, now = Date.now()) {
34
+ if (!value) return null;
35
+ const rel = /^(\d+)\s*([dh])$/i.exec(String(value).trim());
36
+ if (rel) return now - Number(rel[1]) * (rel[2].toLowerCase() === "d" ? 86400000 : 3600000);
37
+ const ms = Date.parse(value);
38
+ if (!Number.isFinite(ms)) throw new Error(`--since must be a date (2026-09-25) or an age (7d, 24h), got "${value}"`);
39
+ return ms;
40
+ }
41
+
42
+ /** The army role a job ran as: stamped on newer records, read from the brief on older ones. */
43
+ export function jobRole(record) {
44
+ if (record.labels?.role) return record.labels.role;
45
+ const m = /^\[nomArmy role: ([a-z][a-z0-9-]*)/.exec(String(record.objective ?? record.task ?? ""));
46
+ return m ? m[1] : null;
47
+ }
48
+
49
+ export function jobModel(record) {
50
+ const provider = record.metrics?.worker_provider ?? record.worker?.provider ?? null;
51
+ const model = record.metrics?.worker_model ?? record.worker?.model ?? null;
52
+ if (record.mode === "verify") return null;
53
+ return model ? (provider && provider !== "llama-cpp" ? `${model}` : `${model} (local)`) : null;
54
+ }
55
+
56
+ const count = (items, key) => items.reduce((m, x) => { const k = key(x) ?? "unassigned"; m[k] = (m[k] ?? 0) + 1; return m; }, {});
57
+ const sortDesc = (obj) => Object.fromEntries(Object.entries(obj).sort((a, b) => b[1] - a[1]));
58
+ function percentile(values, p) {
59
+ if (!values.length) return null;
60
+ const sorted = [...values].sort((a, b) => a - b);
61
+ return sorted[Math.min(sorted.length - 1, Math.ceil((p / 100) * sorted.length) - 1)];
62
+ }
63
+
64
+ // Review flags and harness notes, counted by the issue line nomArmy wrote.
65
+ const SIGNALS = [
66
+ ["runner cleanup crash (report recovered)", /runner cleanup failed after the run|OpenClaw's cleanup failed/i],
67
+ ["scoped test selection risk", /^SCOPED TEST SELECTION RISK/],
68
+ ["existing tests modified (review)", /^TEST CHANGE REVIEW/],
69
+ ["code wired to nothing", /^UNWIRED NEW DEFINITION/],
70
+ ["check rewritten to pass", /^VERIFICATION INPUT CHANGED/],
71
+ ["mutants survived", /^MUTANTS SURVIVED/],
72
+ ["report may not match diff (Jev)", /^REPORT MAY NOT MATCH THE DIFF/],
73
+ ["citations may not support findings (Jev)", /^CITATIONS MAY NOT SUPPORT/],
74
+ ["judge flag", /^JUDGE \(/],
75
+ ["possible secret", /^POSSIBLE SECRET/],
76
+ ["tools outside the sandbox", /^TOOLS OUTSIDE THE SANDBOX/],
77
+ ["Podman VM restarted mid-job", /Podman (VM|machine).*(restarted|stopped) during this job/i],
78
+ ];
79
+
80
+ /**
81
+ * A repository by path, or by folder name: an exact folder name wins, then
82
+ * one ending in it (senti → rayson-senti, not rayson-senti-shared-services),
83
+ * then one containing it. Throws when it matches none or several.
84
+ */
85
+ export function resolveRepo(records, value) {
86
+ if (!value) return null;
87
+ if (fs.existsSync(value)) return path.resolve(value);
88
+ const repos = [...new Set(records.map((r) => r.projectDir).filter(Boolean).map((p) => path.resolve(p)))];
89
+ const exact = repos.filter((p) => path.basename(p) === value);
90
+ const lower = String(value).toLowerCase();
91
+ const ending = repos.filter((p) => path.basename(p).toLowerCase().endsWith(`-${lower}`) || path.basename(p).toLowerCase().endsWith(`_${lower}`));
92
+ const matches = exact.length ? exact : ending.length ? ending : repos.filter((p) => path.basename(p).toLowerCase().includes(lower));
93
+ if (matches.length === 1) return matches[0];
94
+ if (!matches.length) throw new Error(`no repository with jobs matches "${value}"; repositories: ${repos.map((p) => path.basename(p)).join(", ") || "(none)"}`);
95
+ throw new Error(`"${value}" matches several repositories: ${matches.map((p) => path.basename(p)).join(", ")}; use more of the name or a path`);
96
+ }
97
+
98
+ /**
99
+ * @param {object[]} records
100
+ * @param {{ repo?: string|null, sinceMs?: number|null, untilMs?: number|null, role?: string|null, model?: string|null }} filter
101
+ */
102
+ export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, runId = null, agentFor = () => null, now = Date.now(), allSuggestions = false } = {}) {
103
+ const inRange = records.filter((r) => {
104
+ const at = Date.parse(r.startedAt ?? r.finishedAt ?? "");
105
+ if (sinceMs != null && !(at >= sinceMs)) return false;
106
+ if (untilMs != null && !(at <= untilMs)) return false;
107
+ if (repo && r.projectDir && path.resolve(r.projectDir) !== path.resolve(repo)) return false;
108
+ if (role && jobRole(r) !== role) return false;
109
+ if (model && (r.metrics?.worker_model ?? r.worker?.model) !== model) return false;
110
+ if (runId && r.labels?.runId !== runId) return false;
111
+ return true;
112
+ });
113
+ // Verify runs from before records carried the repo can't be placed in one.
114
+ const unplaced = repo ? inRange.filter((r) => !r.projectDir).length : 0;
115
+ const jobs = repo ? inRange.filter((r) => r.projectDir) : inRange;
116
+ const implement = jobs.filter((r) => r.mode === "implement");
117
+ const scouts = jobs.filter((r) => r.mode === "scout");
118
+ const committed = implement.filter((r) => r.commit?.created || r.commit?.sha);
119
+
120
+ const workerMinutes = implement.map((r) => r.metrics?.worker_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
121
+ const jobMinutes = implement.map((r) => r.metrics?.total_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
122
+ const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, untracked: 0 };
123
+ const spend = {};
124
+ for (const r of jobs) {
125
+ const m = r.metrics ?? {};
126
+ tokens.total += m.worker_tokens_total ?? 0; tokens.input += m.worker_tokens_in ?? 0; tokens.output += m.worker_tokens_out ?? 0;
127
+ tokens.cacheRead += m.worker_tokens_cache_read ?? 0; tokens.cacheWrite += m.worker_tokens_cache_write ?? 0;
128
+ if (r.mode !== "verify" && !(m.worker_tokens_total > 0)) tokens.untracked++;
129
+ if (Number.isFinite(m.worker_cost_usd) && m.worker_cost_usd > 0) spend[jobModel(r) ?? "unknown"] = (spend[jobModel(r) ?? "unknown"] ?? 0) + m.worker_cost_usd;
130
+ }
131
+
132
+ // Claim vs evidence: implement jobs whose worker reported done with tests passing.
133
+ const claimedDone = implement.filter((r) => r.reportValidation?.status === "done" && r.reportValidation?.tests === "pass");
134
+ const verificationFailed = claimedDone.filter((r) => r.independentVerification?.status === "fail");
135
+ const revertStillPassed = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status === "fail");
136
+ // Claimed success with an empty diff: there was nothing to verify.
137
+ const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
138
+ const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
139
+ const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
140
+ // New tests the revert check showed would catch their change going away.
141
+ const provenTestFiles = committed.filter((r) => r.regressionCheck?.status === "pass").reduce((n, r) => n + (r.testChanges?.new_tests_added?.length ?? r.metrics?.new_tests_added ?? 0), 0);
142
+
143
+ const signals = {};
144
+ for (const [name, re] of SIGNALS) {
145
+ const n = jobs.filter((r) => [...(r.issues ?? []), ...(r.runnerNotes ?? [])].some((i) => re.test(i))).length;
146
+ if (n) signals[name] = n;
147
+ }
148
+
149
+ const reviewers = {};
150
+ for (const r of scouts) {
151
+ const role = jobRole(r) ?? "unassigned";
152
+ const entry = reviewers[role] ??= { runs: 0, outcomes: {}, findings: 0 };
153
+ entry.runs++;
154
+ entry.outcomes[r.outcome ?? "unknown"] = (entry.outcomes[r.outcome ?? "unknown"] ?? 0) + 1;
155
+ entry.findings += r.scout?.findings?.length ?? 0;
156
+ }
157
+
158
+ const times = jobs.map((r) => Date.parse(r.startedAt ?? "")).filter(Number.isFinite);
159
+ return {
160
+ period: { from: times.length ? new Date(Math.min(...times)).toISOString() : null, to: times.length ? new Date(Math.max(...times)).toISOString() : null },
161
+ repo, role, model, unplacedVerifyRuns: role || model ? 0 : unplaced,
162
+ volume: {
163
+ jobs: jobs.length,
164
+ byMode: sortDesc(count(jobs, (r) => r.mode)),
165
+ byRole: sortDesc(count(jobs.filter((r) => r.mode !== "verify"), jobRole)),
166
+ byModel: sortDesc(count(jobs.filter((r) => r.mode !== "verify"), jobModel)),
167
+ byOutcome: sortDesc(count(jobs, (r) => r.outcome)),
168
+ },
169
+ code: {
170
+ committedJobs: committed.length,
171
+ linesAdded: committed.reduce((s, r) => s + (r.git?.additions ?? r.metrics?.lines_added ?? 0), 0),
172
+ linesRemoved: committed.reduce((s, r) => s + (r.git?.deletions ?? r.metrics?.lines_removed ?? 0), 0),
173
+ files: new Set(committed.flatMap((r) => r.git?.changedFiles ?? [])).size,
174
+ newTestFiles: committed.reduce((s, r) => s + (r.testChanges?.new_tests_added?.length ?? r.metrics?.new_tests_added ?? 0), 0),
175
+ },
176
+ workerMinutes: { median: percentile(workerMinutes, 50), p90: percentile(workerMinutes, 90), total: workerMinutes.reduce((a, b) => a + b, 0) },
177
+ jobMinutes: { median: percentile(jobMinutes, 50), p90: percentile(jobMinutes, 90), total: jobMinutes.reduce((a, b) => a + b, 0) },
178
+ tokens,
179
+ spendUsd: { total: Object.values(spend).reduce((a, b) => a + b, 0), byModel: sortDesc(spend) },
180
+ claimVsEvidence: {
181
+ claimedDone: claimedDone.length,
182
+ verificationFailed: verificationFailed.length,
183
+ revertStillPassed: revertStillPassed.length,
184
+ passedBoth: passedBoth.length,
185
+ changedNothing: changedNothing.length,
186
+ flaggedAfterPassing: flaggedAfterPassing.length,
187
+ provenTestFiles,
188
+ },
189
+ notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
190
+ reviewers,
191
+ signals: sortDesc(signals),
192
+ highStakes: (() => {
193
+ // Work that landed: an uncommitted partial isn't accepted work.
194
+ const high = implement.filter((r) => r.stakes === "high" && r.commit?.created);
195
+ return { jobs: high.length, reviewed: high.filter((r) => reviewOf(r, records)).length };
196
+ })(),
197
+ // How you're set up now: the last 14 days unless a period was asked for.
198
+ suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor, now, includeStale: allSuggestions }),
199
+ suggestionWindow: sinceMs == null ? "the last 14 days" : "this period",
200
+ };
201
+ }
202
+
203
+ const pct = (n, of) => (of ? ` (${Math.round((100 * n) / of)}%)` : "");
204
+ const list = (obj) => Object.entries(obj).map(([k, v]) => `${k} ${v}`).join(" · ") || "none";
205
+ const mins = (m) => (m == null ? "n/a" : `${m.toFixed(1)} min`);
206
+ const big = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${(n / 1e3).toFixed(1)}k` : String(n));
207
+
208
+ /** The headline: claims that didn't hold up, and tests shown to catch their change. */
209
+ function caughtLines(c) {
210
+ if (!c.claimedDone) return [" no implement job reported \"done, tests pass\" in this period"];
211
+ const wrong = c.verificationFailed + c.revertStillPassed;
212
+ const parts = [c.verificationFailed && `${c.verificationFailed} failed when nomArmy ran the tests itself`, c.revertStillPassed && `${c.revertStillPassed} had tests that still pass with the change reverted`].filter(Boolean);
213
+ return [
214
+ wrong ? ` ${wrong} of ${c.claimedDone} "done, tests pass" claims didn't hold up: ${parts.join(", ")}` : ` all ${c.claimedDone} "done, tests pass" claims held up when nomArmy checked them`,
215
+ ...(c.flaggedAfterPassing ? [` ${c.flaggedAfterPassing} more passed both but were flagged (mutants, Jev, judge, a rewritten check)`] : []),
216
+ ...(c.provenTestFiles ? [` ${c.provenTestFiles} new test file(s) shown to fail without their change`] : []),
217
+ ];
218
+ }
219
+
220
+ /**
221
+ * The default view: one screen. What nomArmy caught, what needs you, the top
222
+ * tips, and the totals. `--details` prints formatStats. `c` paints (the CLI
223
+ * passes its colors, plain when not a terminal); left out, it's plain text.
224
+ */
225
+ const PLAIN = { bold: String, dim: String, red: String, green: String, yellow: String, cyan: String };
226
+ export function formatStatsSummary(s, { c = PLAIN, width = 28 } = {}) {
227
+ const cv = s.claimVsEvidence;
228
+ const where = s.repo ? path.basename(s.repo) : "all repositories";
229
+ const month = (iso) => new Date(iso).toLocaleString("en-US", { month: "short", day: "numeric", timeZone: "UTC" });
230
+ const when = s.period.from ? `${month(s.period.from)} to ${month(s.period.to)}` : "no jobs yet";
231
+ const act = (s.suggestions ?? []).filter((x) => x.level === "act");
232
+ // The spend share is on the totals line already.
233
+ const tips = (s.suggestions ?? []).filter((x) => x.level !== "act" && x.key !== "stale" && !x.key.startsWith("spend:"));
234
+ const hidden = (s.suggestions ?? []).find((x) => x.key === "stale");
235
+ const shown = tips.slice(0, 3), more = tips.length - shown.length;
236
+ const label = (t) => c.bold(t.padEnd(9));
237
+ const lines = [`${c.bold("nomArmy stats")} ${c.dim(`${where} · ${when} · ${s.volume.jobs} jobs · ${s.code.committedJobs} committed`)}`, ""];
238
+
239
+ // What the checks caught: the reason to run nomArmy, first.
240
+ if (cv.claimedDone) {
241
+ const wrong = cv.verificationFailed + cv.revertStillPassed, held = cv.claimedDone - wrong;
242
+ const bad = wrong ? Math.max(1, Math.round((width * wrong) / cv.claimedDone)) : 0;
243
+ lines.push(`${label("CAUGHT")}${c.green("█".repeat(width - bad))}${c.red("░".repeat(bad))} ${held} of ${cv.claimedDone} "done, tests pass" claims held up${wrong ? c.red(` · ${wrong} didn't`) : ""}`);
244
+ const why = [cv.verificationFailed && `${cv.verificationFailed} failed when nomArmy ran the tests itself`, cv.revertStillPassed && `${cv.revertStillPassed} had tests that pass with the change reverted`, cv.flaggedAfterPassing && `${cv.flaggedAfterPassing} passed but were flagged`].filter(Boolean);
245
+ if (why.length) lines.push(`${" ".repeat(9)}${c.dim(why.join(" · "))}`);
246
+ } else lines.push(`${label("CAUGHT")}${c.dim('no job reported "done, tests pass" in this period')}`);
247
+ if (cv.provenTestFiles) lines.push(`${label("PROVEN")}${c.green("✓")} ${cv.provenTestFiles} new test files fail without their change`);
248
+
249
+ for (const x of act) {
250
+ const ids = /: (.+)$/.exec(x.title)?.[1]?.split(", ") ?? [];
251
+ const head = x.title.replace(/: .+$/, "");
252
+ lines.push("", `${c.red(c.bold("⚠ REVIEW BEFORE MERGING"))} ${head}`);
253
+ for (let i = 0; i < ids.length; i += 2) lines.push(` ${ids.slice(i, i + 2).map((id) => id.padEnd(32)).join("")}`.trimEnd());
254
+ lines.push(` ${c.cyan("→")} a scout on another vendor with ${c.cyan("reviews: <job id>")} (army_role security-analyst); a failed review doesn't count`);
255
+ }
256
+
257
+ lines.push("");
258
+ if (!shown.length) lines.push(`${label("TIPS")}${c.dim("none: nothing in the records suggests a routing change")}`);
259
+ shown.forEach((t, i) => {
260
+ lines.push(`${i ? " ".repeat(9) : label("TIPS")}${t.level === "warn" ? c.yellow("▲") : c.dim("·")} ${t.title}`);
261
+ if (t.command) lines.push(`${" ".repeat(11)}${c.cyan(`→ ${t.command}`)}`);
262
+ });
263
+ const staleCount = hidden ? Number(/^\d+/.exec(hidden.title)?.[0] ?? 0) : 0;
264
+ const notes = [more > 0 && `${more} more (--details)`, staleCount && `${staleCount} about pairings unused for 3+ days (--all-suggestions)`].filter(Boolean);
265
+ if (notes.length) lines.push(`${" ".repeat(11)}${c.dim(notes.join(" · "))}`);
266
+
267
+ const top = Object.entries(s.spendUsd.byModel)[0];
268
+ lines.push("", `${label("SPEND")}$${s.spendUsd.total.toFixed(2)} API${top ? c.dim(` (${top[0]} ${Math.round((100 * top[1]) / (s.spendUsd.total || 1))}%)`) : ""} · ${Math.round(s.jobMinutes.total)} min of jobs · ${big(s.tokens.total)} tokens`);
269
+ lines.push("", c.dim("Everything else (volume, reviewers, flags, what didn't finish): nomarmy stats --details"));
270
+ return lines.join("\n");
271
+ }
272
+
273
+ /** The terminal report. */
274
+ export function formatStats(s) {
275
+ const c = s.claimVsEvidence;
276
+ const lines = [
277
+ `nomArmy stats${s.repo ? ` for ${s.repo}` : " (all repositories)"}${s.role ? `, role ${s.role}` : ""}${s.model ? `, model ${s.model}` : ""}, ${s.period.from ? `${s.period.from.slice(0, 10)} to ${s.period.to.slice(0, 10)}` : "no jobs"}`,
278
+ "",
279
+ "WHAT NOMARMY CAUGHT",
280
+ ...caughtLines(c),
281
+ ...(() => { const act = (s.suggestions ?? []).filter((x) => x.level === "act"); return act.length ? ["", "NEEDS YOUR ATTENTION", ...formatSuggestions(act)] : []; })(),
282
+ "",
283
+ `SUGGESTIONS (from ${s.suggestionWindow ?? "this period"}; never applied for you)`,
284
+ ...formatSuggestions((s.suggestions ?? []).filter((x) => x.level !== "act")),
285
+ "",
286
+ "VOLUME",
287
+ ` Jobs ${s.volume.jobs}: ${list(s.volume.byMode)}${s.unplacedVerifyRuns ? ` (plus ${s.unplacedVerifyRuns} older verify run(s) that don't record their repository)` : ""}`,
288
+ ` By role ${list(s.volume.byRole)}`,
289
+ ` By model ${list(s.volume.byModel)}`,
290
+ ` Committed ${s.code.committedJobs} job(s) · +${s.code.linesAdded} / -${s.code.linesRemoved} lines · ${s.code.files} files · ${s.code.newTestFiles} new test files`,
291
+ ` Worker time median ${mins(s.workerMinutes.median)} per implement job, p90 ${mins(s.workerMinutes.p90)}, total ${Math.round(s.workerMinutes.total)} min`,
292
+ ` Job time median ${mins(s.jobMinutes.median)}, p90 ${mins(s.jobMinutes.p90)}, total ${Math.round(s.jobMinutes.total)} min (with verification and checks)`,
293
+ ` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)${s.tokens.untracked ? `; ${s.tokens.untracked} job(s) recorded no token counts` : ""}`,
294
+ ` API spend $${s.spendUsd.total.toFixed(2)}${Object.keys(s.spendUsd.byModel).length ? ` (${Object.entries(s.spendUsd.byModel).map(([k, v]) => `${k} $${v.toFixed(2)}`).join(", ")})` : ""}; subscriptions aren't billed per call`,
295
+ "",
296
+ `CLAIM VS EVIDENCE (implement jobs that reported "done, tests pass": ${c.claimedDone})`,
297
+ ` Independent verification failed ${c.verificationFailed}${pct(c.verificationFailed, c.claimedDone)}`,
298
+ ` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
299
+ ` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
300
+ ` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
301
+ ...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
302
+ ` High-stakes jobs committed ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with a finished independent review`,
303
+ " Defects the General found at integration aren't in the records; count them in your own review.",
304
+ "",
305
+ "DIDN'T COMPLETE",
306
+ ...(Object.keys(s.notCompleted).length ? Object.entries(s.notCompleted).map(([k, v]) => ` ${k.padEnd(28)} ${v}`) : [" none"]),
307
+ "",
308
+ "REVIEWERS (scouts)",
309
+ ...(Object.keys(s.reviewers).length ? Object.entries(s.reviewers).map(([role, r]) => ` ${role.padEnd(18)} ${r.runs} run(s), ${r.findings} finding(s); ${list(r.outcomes)}`) : [" none"]),
310
+ "",
311
+ "REVIEW FLAGS AND HARNESS SIGNALS (jobs)",
312
+ ...(Object.keys(s.signals).length ? Object.entries(s.signals).map(([k, v]) => ` ${k.padEnd(42)} ${v}`) : [" none"]),
313
+ ];
314
+ return lines.join("\n");
315
+ }
@@ -27,6 +27,7 @@ import fs from "node:fs";
27
27
  import os from "node:os";
28
28
  import path from "node:path";
29
29
  import { fileURLToPath } from "node:url";
30
+ import { mergeUsageSnapshot, normalizeClaudeRateLimits, readUsageSnapshots, recordUsageSnapshot, usageStatus } from "./usage-limits.mjs";
30
31
 
31
32
  function readJson(file) { try { return JSON.parse(fs.readFileSync(file, "utf8")); } catch { return null; } }
32
33
  function pidAlive(pid) {
@@ -47,6 +48,17 @@ const minutes = (ms) => (ms < 60000 ? `${Math.max(1, Math.round(ms / 1000))}s` :
47
48
  */
48
49
  export function statusLineText({ session = {}, stateRoot, now = Date.now(), maxLength = Number(process.env.NOMARMY_STATUSLINE_MAX) || 90 } = {}) {
49
50
  const root = stateRoot ?? (process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents"));
51
+ // Every Claude Code window redraws this line with its own last-seen
52
+ // limits, idle ones included, so a reading is merged rather than trusted
53
+ // (mergeUsageSnapshot). Unchanged observations refresh on a bounded interval.
54
+ try {
55
+ const snapshot = normalizeClaudeRateLimits(session.rate_limits);
56
+ if (snapshot) {
57
+ const previous = readUsageSnapshots(root)["claude-cli"];
58
+ const merged = mergeUsageSnapshot(previous, { ...snapshot, sourceId: session.session_id, observedAt: now }, now);
59
+ if (merged !== previous) recordUsageSnapshot(root, "claude-cli", merged);
60
+ }
61
+ } catch { /* usage capture must never break the status line */ }
50
62
  // workspace.project_dir is where Claude Code was launched -- the repo that
51
63
  // session's nomArmy server works in; current_dir can be a subdirectory.
52
64
  const cwd = session.workspace?.project_dir ?? session.workspace?.current_dir ?? session.cwd ?? process.cwd();
@@ -99,23 +111,37 @@ export function statusLineText({ session = {}, stateRoot, now = Date.now(), maxL
99
111
  try {
100
112
  const health = readJson(path.join(root, "health.json"));
101
113
  if (health && now - Date.parse(health.checkedAt) < 2 * 86400000) {
102
- const top = (health.issues ?? []).find((i) => i.severity !== "info" && i.short);
114
+ // Usage warnings get their own part below, so they aren't repeated here.
115
+ const top = (health.issues ?? []).find((i) => i.severity !== "info" && i.short && !String(i.id).startsWith("usage:"));
103
116
  if (top) healthPart = ` │ ⚠ ${top.short}`;
104
117
  }
105
118
  } catch { /* no health yet */ }
106
119
  runPart += healthPart;
107
- // As many jobs as fit, then "+N": the run summary is never what gets cut.
120
+ let usageParts = [];
121
+ try {
122
+ const warnings = Object.entries(readUsageSnapshots(root)).map(([provider, snapshot]) => {
123
+ const status = usageStatus(snapshot, now);
124
+ if (status.level === "ok") return null;
125
+ return { level: status.level, text: ` │ ${status.level === "over" ? "⛔" : "⚠"} ${provider} ${status.short}` };
126
+ }).filter(Boolean).sort((a, b) => (a.level === "over" ? 0 : 1) - (b.level === "over" ? 0 : 1));
127
+ usageParts = warnings.map((warning) => warning.text);
128
+ } catch { /* usage display is best-effort */ }
129
+ // As many jobs as fit, then "+N". Only after every job is collapsed may
130
+ // lower-priority usage warnings be omitted to honor the hard cap.
108
131
  const prefix = `${head ? `${head} │ ` : ""}🍪 `;
109
132
  const count = jobs.length > 1 ? `${jobs.length}: ` : "";
110
- let shown = jobs.length, army;
133
+ let shown = jobs.length, army, suffix;
111
134
  for (;;) {
112
135
  const rest = jobs.length - shown;
113
136
  army = (!jobs.length ? "idle" : `${count}${jobs.slice(0, shown).join(" · ")}${rest ? `${shown ? " " : ""}+${rest}` : ""}`)
114
137
  + (elsewhere ? ` · ${elsewhere} in other repo${elsewhere === 1 ? "" : "s"}` : "");
115
- if (shown === 0 || [...`${prefix}${army}${runPart}`].length <= maxLength) break;
116
- shown--;
138
+ suffix = runPart + usageParts.join("");
139
+ if ([...`${prefix}${army}${suffix}`].length <= maxLength) break;
140
+ if (shown > 0) { shown--; continue; }
141
+ if (usageParts.length) { usageParts.pop(); continue; }
142
+ break;
117
143
  }
118
- return `${prefix}${army}${runPart}`;
144
+ return `${prefix}${army}${suffix}`;
119
145
  }
120
146
 
121
147
  const isMain = (() => { try { return path.resolve(process.argv[1] ?? "") === fileURLToPath(import.meta.url); } catch { return false; } })();
@@ -187,6 +187,19 @@ export function probeOutcome({ stdout = "", stderr = "" } = {}) {
187
187
  return { ok: false, reason: message ? message.replace(/\\"/g, '"').slice(0, 300) : null };
188
188
  }
189
189
 
190
+ /**
191
+ * A probe answer that means the login itself failed, not the model.
192
+ * 401, a missing bearer, no usable profiles, or an unavailable selected
193
+ * profile are sign-in failures. A refusal, rate limit, or timeout is not.
194
+ */
195
+ export function openclawSignInFailure(text) {
196
+ const s = String(text ?? "");
197
+ return /\b401\b/.test(s)
198
+ || /missing bearer/i.test(s)
199
+ || /no usable profiles/i.test(s)
200
+ || /selected auth profile\b[^]{0,240}?\bis unavailable/i.test(s);
201
+ }
202
+
190
203
  /**
191
204
  * Muse Code's non-secret login descriptor (~/.config/muse/auth.json) ->
192
205
  * { loggedIn, email }. Reads only descriptor fields; the credential itself
@@ -0,0 +1,153 @@
1
+ // Routing suggestions from the job records: which role and model pairings
2
+ // are working, which aren't, and what a change would be. Evidence first,
3
+ // with minimum sample sizes; roles do different work, so a comparison across
4
+ // roles is worded as something to try, not a verdict. Never applied: the
5
+ // operator or the General decides, and each suggestion carries the command.
6
+
7
+ import { jobRole } from "./stats.mjs";
8
+
9
+ export const MIN_JOBS = 5;
10
+ const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE_DONE|VERIFIED)$/;
11
+
12
+ const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
13
+ const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
14
+ // The local model is the built-in "local" agent, whatever model is loaded.
15
+ const agent = (r) => r.labels?.agent ?? (provider(r) === "llama-cpp" ? "local" : null);
16
+ // New tokens only: cache reads are most of an api job's total and cost a fraction.
17
+ const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.worker_tokens_out ?? 0);
18
+ /** The runner exited before a report: an OpenClaw, sandbox or provider failure, not the model's work. */
19
+ export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
20
+ const pct = (n, of) => Math.round((100 * n) / of);
21
+
22
+ const REVIEW_FINISHED = /^(SCOUT_DONE|SCOUT_NOT_FOUND)$/;
23
+ // How recently a pairing must have run for a suggestion about it to still be
24
+ // about how you work now: a week-old local-model experiment led Senti's list.
25
+ export const CURRENT_DAYS = 3;
26
+
27
+ /** Whether a high-stakes job has had an independent review: a finished scout, or a judge, on another vendor. A review that timed out or failed isn't one. */
28
+ export function reviewOf(job, records) {
29
+ const workerProvider = provider(job);
30
+ const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && REVIEW_FINISHED.test(r.outcome ?? "") && provider(r) && provider(r) !== workerProvider);
31
+ if (scout) return { by: "scout", jobId: scout.jobId, provider: provider(scout) };
32
+ const judge = job.validators?.judge;
33
+ if (judge?.answer && judge.provider && judge.provider !== workerProvider) return { by: "judge", provider: judge.provider };
34
+ return null;
35
+ }
36
+
37
+ /**
38
+ * @param {object[]} records this repo's records, already filtered to a period
39
+ * @returns {{ level: "warn"|"info", key: string, title: string, evidence: string, command: string|null }[]}
40
+ */
41
+ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null, now = Date.now(), includeStale = false } = {}) {
42
+ const out = [];
43
+ let stale = 0;
44
+ const work = records.filter((r) => r.mode === "implement" || r.mode === "scout");
45
+
46
+ // Per role and model.
47
+ const groups = new Map();
48
+ for (const r of work) {
49
+ const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
50
+ const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0, lastAt: 0 };
51
+ g.jobs++;
52
+ g.lastAt = Math.max(g.lastAt, Date.parse(r.startedAt ?? "") || 0);
53
+ if (runnerFailed(r)) g.runner++; else g.rated++;
54
+ if (OK.test(r.outcome ?? "")) g.ok++;
55
+ if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
56
+ if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
57
+ if (freshTokens(r) > 0) { g.tokens += freshTokens(r); g.tokenJobs++; }
58
+ g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
59
+ groups.set(key, g);
60
+ }
61
+ const assign = (g, to = "<another model>") => (g.role ? `nomarmy army assign ${g.role} ${g.agent ?? "<agent>"} ${to}` : null);
62
+
63
+ const current = (g) => includeStale || g.lastAt >= now - CURRENT_DAYS * 86400000;
64
+ for (const g of groups.values()) {
65
+ if (!g.model) continue;
66
+ // A stale pairing's suggestions are worked out, then only counted.
67
+ const before = out.length;
68
+ groupSuggestions(g);
69
+ if (!current(g)) stale += out.splice(before).length;
70
+ }
71
+ function groupSuggestions(g) {
72
+ const kind = g.mode === "scout" ? "scouts" : "implement jobs";
73
+ const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
74
+ // The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
75
+ if (g.runner >= 3 && g.runner / g.jobs >= 0.4) {
76
+ out.push({ level: "warn", key: `runner-failed:${g.role}:${g.model}:${g.mode}`, title: `${who}: the runner failed on ${g.runner} of ${g.jobs} ${kind} before any report`,
77
+ evidence: "OpenClaw, the sandbox or the provider exited early, so these say nothing about the model's work. Check `nomarmy health` and one job's log (`nomarmy jobs <id>`); a model its vendor refuses fails this way too.", command: null });
78
+ }
79
+ // Scouts that come back empty.
80
+ if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
81
+ out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
82
+ evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
83
+ return;
84
+ }
85
+ if (g.rated < minJobs) return;
86
+ const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
87
+ // A pairing that rarely finishes.
88
+ if (g.ok / g.rated < 0.5) {
89
+ const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.rated >= minJobs && o.model && o.ok / o.rated >= g.ok / g.rated + 0.2)
90
+ .sort((a, b) => b.ok / b.rated - a.ok / a.rated)[0];
91
+ out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.rated} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.rated)}%)${runnerNote}`,
92
+ evidence: (better ? `${better.model} finished ${pct(better.ok, better.rated)}% of its ${better.rated} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
93
+ command: g.role ? assign(g, better?.model ?? "<another model>") : null });
94
+ }
95
+ // Timeouts.
96
+ if (g.timeout >= 3 && g.timeout / g.rated >= 0.25) {
97
+ out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.rated} jobs`,
98
+ evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
99
+ }
100
+ }
101
+
102
+ // A lighter model doing as well on the same role's implement work, for much less. Only
103
+ // within one role: different roles do different work, so across roles the numbers don't compare.
104
+ const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs && current(g));
105
+ for (const heavy of impl) {
106
+ for (const light of impl) {
107
+ if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
108
+ const lightRate = light.ok / light.rated, heavyRate = heavy.ok / heavy.rated;
109
+ const lightPerJob = light.tokens / light.tokenJobs, heavyPerJob = heavy.tokens / heavy.tokenJobs;
110
+ if (lightRate >= heavyRate - 0.05 && lightPerJob <= 0.5 * heavyPerJob) {
111
+ out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
112
+ title: `${heavy.role}: ${light.model} finished ${pct(light.ok, light.rated)}% of its jobs on ${Math.round(lightPerJob / 1000)}k new tokens a job; ${heavy.model} finished ${pct(heavy.ok, heavy.rated)}% on ${Math.round(heavyPerJob / 1000)}k`,
113
+ evidence: `Same role, so similar work. Moving ${heavy.role} to ${light.model} would cost less; keep ${heavy.model} for the harder pieces with model on the job.${light.agent === "local" ? ` The local agent runs whichever model is loaded; these ran on ${light.model}.` : ""}`,
114
+ command: light.agent ? `nomarmy army assign ${heavy.role} ${light.agent}${light.agent === "local" ? "" : ` ${light.model}`}` : null });
115
+ }
116
+ }
117
+ }
118
+
119
+ // Where the money goes.
120
+ const spend = new Map();
121
+ for (const r of work) if (Number.isFinite(r.metrics?.worker_cost_usd) && r.metrics.worker_cost_usd > 0) spend.set(model(r), (spend.get(model(r)) ?? 0) + r.metrics.worker_cost_usd);
122
+ const total = [...spend.values()].reduce((a, b) => a + b, 0);
123
+ const [top, topUsd] = [...spend.entries()].sort((a, b) => b[1] - a[1])[0] ?? [];
124
+ if (top && topUsd >= 5 && topUsd / total >= 0.5) {
125
+ out.push({ level: "info", key: `spend:${top}`, title: `${top} is ${pct(topUsd, total)}% of API spend ($${topUsd.toFixed(2)} of $${total.toFixed(2)})`,
126
+ evidence: "Worth knowing rather than changing if it's catching real problems; check its reviews' findings before moving it.", command: null });
127
+ }
128
+
129
+ // Committed high-stakes work without an independent review. Only work that
130
+ // landed: a partial never committed isn't accepted work, and one finished by
131
+ // a later job is that job's to review.
132
+ const unreviewed = records.filter((r) => r.mode === "implement" && r.stakes === "high" && r.commit?.created && !reviewOf(r, records));
133
+ if (unreviewed.length) {
134
+ const ids = unreviewed.map((r) => r.jobId).sort();
135
+ out.unshift({ level: "act", key: `unreviewed:${ids.join(",")}`, title: `${ids.length} high-stakes job(s) committed without an independent review: ${ids.join(", ")}`,
136
+ evidence: "Review each before merging: a scout on another vendor (army_role security-analyst, say) with reviews: <job id>. A review that failed or timed out doesn't count.", command: null });
137
+ }
138
+ const rank = { act: 0, warn: 1, info: 2 };
139
+ out.sort((a, b) => rank[a.level] - rank[b.level]);
140
+ if (stale) out.push({ level: "info", key: "stale", title: `${stale} more about role and model pairings you haven't used in ${CURRENT_DAYS} days, hidden (nomarmy stats --all-suggestions)`, evidence: null, command: null });
141
+ return out;
142
+ }
143
+
144
+ /** This repository's suggestions from its last 14 days of jobs. */
145
+ export function recentSuggestions(records, { projectDir, agentFor = () => null, now = Date.now(), days = 14 } = {}) {
146
+ const since = now - days * 86400000;
147
+ return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor, now });
148
+ }
149
+
150
+ export function formatSuggestions(list) {
151
+ if (!list.length) return [" none: nothing in the records suggests a routing change"];
152
+ return list.flatMap((s) => [` ${{ act: "!!", warn: "!", info: "-" }[s.level] ?? "-"} ${s.title}`, ...(s.evidence ? [` ${s.evidence}`] : []), ...(s.command ? [` ${s.command}`] : [])]);
153
+ }
@@ -0,0 +1,23 @@
1
+ // Thinking levels, as OpenClaw names them (`openclaw agent exec --thinking`).
2
+ // Which ones a model honors is the vendor's business: OpenClaw refuses a level
3
+ // a model lacks and names the ones it has, and nomArmy retries once with the
4
+ // nearest of those (lib/openclaw-run.mjs).
5
+
6
+ export const THINKING_LEVELS = Object.freeze(["off", "minimal", "low", "medium", "high", "xhigh", "adaptive", "max", "ultra"]);
7
+
8
+ // Strength order for picking a fallback; "adaptive" lets the model choose, so it sits mid-way.
9
+ const RANK = { off: 0, minimal: 1, low: 2, medium: 3, adaptive: 3.5, high: 4, xhigh: 5, max: 6, ultra: 7 };
10
+
11
+ /**
12
+ * The supported level closest to the one refused: the strongest at or below
13
+ * it, else the weakest above it. The old fallback took the first listed,
14
+ * often "off", so a refused xhigh would have run with no thinking at all.
15
+ */
16
+ export function nearestThinkingLevel(requested, supported) {
17
+ const known = supported.filter((l) => l in RANK);
18
+ if (!known.length) return supported[0] ?? null;
19
+ const want = RANK[requested] ?? RANK.medium;
20
+ const below = known.filter((l) => RANK[l] <= want).sort((a, b) => RANK[b] - RANK[a]);
21
+ if (below.length && !(below[0] === "off" && want > 0 && known.some((l) => RANK[l] > 0))) return below[0];
22
+ return known.filter((l) => RANK[l] > 0).sort((a, b) => RANK[a] - RANK[b])[0] ?? below[0];
23
+ }