nomarmy 0.1.0-alpha.14 → 0.1.0-alpha.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/nomarmy.mjs CHANGED
@@ -199,6 +199,10 @@ Usage: nomarmy <command> [options]
199
199
  which agent the General is, in --global
200
200
  (default) or --local
201
201
  config paths where agents.yml and the three army layers live
202
+ config max-jobs [n]
203
+ how many api and subscription jobs run at once, across
204
+ every session (default 4); with n, sets it in limits.yml.
205
+ Warns when the Podman VM is too small for that many.
202
206
  jobs [--watch|--events [--until-done]|--prune|--wait <jobId>|--stop <jobId> [--reason <text>]] [--interval N] [--older-than DAYS]
203
207
  what's running across every session (agent, model, phase,
204
208
  last tool call, files changed, heartbeat) and what just
@@ -1919,6 +1923,11 @@ async function cmdSandbox() {
1919
1923
  const low = machine.memoryMb && machine.memoryMb < MIN_PODMAN_VM_MB;
1920
1924
  console.log(`\nVM ${machine.name} (${machine.state}): ${machine.cpus} CPUs, ${low ? c.red(`${machine.memoryMb / 1024} GiB memory`) : `${machine.memoryMb / 1024} GiB memory`}, ${machine.diskGb} GB disk`);
1921
1925
  if (low) console.log(c.yellow(` Too small: worker commands get cut off below ${MIN_PODMAN_VM_MB / 1024} GiB. Fix: nomarmy sandbox --memory 8`));
1926
+ const { maxJobs, jobsThatFit, vmGibFor } = await import("../lib/limits.mjs");
1927
+ const limit = maxJobs().value, fit = jobsThatFit(machine.memoryMb);
1928
+ if (fit !== null) console.log(limit > fit
1929
+ ? c.yellow(` Fits about ${fit} sandboxes at once, but up to ${limit} api and subscription jobs may run. Fix: nomarmy sandbox --memory ${vmGibFor(limit)}, or nomarmy config max-jobs ${fit}`)
1930
+ : c.dim(` Fits about ${fit} sandboxes at once; up to ${limit} api and subscription jobs may run (nomarmy config max-jobs).`));
1922
1931
  }
1923
1932
  if (images) console.log(`Images: ${images.count}, ${images.size}${images.reclaimable ? `, ${images.reclaimable} reclaimable (nomarmy sandbox --prune)` : ""}`);
1924
1933
  console.log(c.dim(runningJobs ? `${runningJobs} nomArmy job(s) running.` : "No nomArmy jobs running."));
@@ -2451,14 +2460,38 @@ async function cmdConfigPaths() {
2451
2460
  if (json) return out({ globalDir: globalConfigDir(), agents: { path: agentsPath, exists: fs.existsSync(agentsPath) }, army });
2452
2461
  console.log(c.bold("nomArmy config") + c.dim(` (global dir: ${globalConfigDir()})`));
2453
2462
  console.log(` ${fs.existsSync(agentsPath) ? c.green("●") : c.dim("○")} ${"agents".padEnd(8)} ${c.dim(agentsPath)}`);
2463
+ const { limitsPath } = await import("../lib/limits.mjs");
2464
+ console.log(` ${fs.existsSync(limitsPath()) ? c.green("●") : c.dim("○")} ${"limits".padEnd(8)} ${c.dim(limitsPath())}`);
2454
2465
  console.log(c.bold("\nArmy layers"));
2455
2466
  for (const a of army) console.log(` ${a.exists ? c.green("●") : c.dim("○")} ${a.layer.padEnd(8)} ${c.dim(a.path)}`);
2456
2467
  }
2457
2468
 
2469
+ // `nomarmy config max-jobs [n]`: api and subscription jobs at once, machine-wide.
2470
+ async function cmdConfigMaxJobs() {
2471
+ const { maxJobs, setMaxJobs, jobsThatFit, vmGibFor } = await import("../lib/limits.mjs");
2472
+ const given = argv[2];
2473
+ if (given !== undefined) {
2474
+ if (!/^\d+$/.test(given)) throw new Error(`max-jobs must be a whole number, got "${given}"`);
2475
+ setMaxJobs(Number(given));
2476
+ }
2477
+ const limit = maxJobs();
2478
+ const machine = process.platform === "linux" ? null : pickMachine(spawnSync("podman", ["machine", "inspect"], { encoding: "utf8" }).stdout);
2479
+ const fit = jobsThatFit(machine?.memoryMb);
2480
+ if (json) return out({ maxJobs: limit.value, source: limit.source, path: limit.path, problem: limit.problem, podmanVmMemoryMb: machine?.memoryMb ?? null, jobsThatFit: fit });
2481
+ const from = { file: `set in ${limit.path}`, env: "from NOMARMY_MAX_POOL_WORKERS in this shell (limits.yml doesn't set it)", default: "the default" }[limit.source];
2482
+ if (limit.problem) console.log(c.yellow(`⚠ ${limit.problem}`));
2483
+ console.log(`${given !== undefined ? c.green("✓ ") : ""}Up to ${c.bold(String(limit.value))} api and subscription jobs at once, across every session (${from}).`);
2484
+ console.log(c.dim("Each agent's max_concurrent in agents.yml also applies, and local-model jobs have their own limit."));
2485
+ if (given !== undefined) console.log(c.dim("Applies to the next job in every session on this version, no restart."));
2486
+ if (fit !== null && limit.value > fit) console.log(c.yellow(`⚠ The Podman VM (${machine.memoryMb / 1024} GiB) fits about ${fit} sandboxes at once; more get refused for memory or cut off. Fix: nomarmy sandbox --memory ${vmGibFor(limit.value)}`));
2487
+ if (given === undefined) console.log(c.dim("Change it with `nomarmy config max-jobs <n>` (1 to 32)."));
2488
+ }
2489
+
2458
2490
  async function cmdConfig() {
2459
2491
  const sub = argv[1] ?? "paths";
2460
2492
  if (sub === "paths") return cmdConfigPaths();
2461
- throw new Error(`Unknown config subcommand "${sub}". Use: nomarmy config paths`);
2493
+ if (sub === "max-jobs") return cmdConfigMaxJobs();
2494
+ throw new Error(`Unknown config subcommand "${sub}". Use: nomarmy config <paths|max-jobs [n]>`);
2462
2495
  }
2463
2496
 
2464
2497
  // --- `nomarmy jobs [--watch]`: what's running, from any session -----------
package/lib/admission.mjs CHANGED
@@ -4,7 +4,7 @@ import { executionMode } from "./execution.mjs";
4
4
  import { jobLabel, jobElapsedSeconds } from "./job-format.mjs";
5
5
  import { continuationProblem } from "./continue-from.mjs";
6
6
  import { loadConfig } from "./config.mjs";
7
- import { clampInt } from "./budget-state.mjs";
7
+ import { maxJobs } from "./limits.mjs";
8
8
  import { checkBrief, assessAdmission, describeBudgets } from "./budget.mjs";
9
9
  import { parseStatusPorcelainZ, isRuntimeJunk } from "./git-record.mjs";
10
10
  import { readJson } from "./openclaw-run.mjs";
@@ -38,8 +38,9 @@ export function jobLane(job) {
38
38
  // benefit of spreading load across providers with their own separate rate
39
39
  // limits. Not rate-limit-aware (see config/providers.yml.example); read
40
40
  // fresh each call, matching currentMaxWorkers()'s own env-read pattern.
41
+ /** Api and subscription jobs at once, machine-wide: limits.yml's max_jobs (see lib/limits.mjs). */
41
42
  export function currentMaxPoolWorkers() {
42
- return clampInt(process.env.NOMARMY_MAX_POOL_WORKERS, 1, 32, 4);
43
+ return maxJobs().value;
43
44
  }
44
45
 
45
46
  // Pure partition of a batch's ORIGINAL indices by lane -- pulled out of
@@ -192,7 +193,7 @@ export function createJobRuntime(deps) {
192
193
  running: [...activeJobs.values()].filter(j => !j.settled).map(j => ({ jobId: j.jobId, workerId: j.workerId, mode: j.mode, lane: j.lane, startedAt: j.startedAt, phase: readJson(path.join(jobsRoot, j.jobId, "status.json"))?.phase ?? "starting" })),
193
194
  usageLimits: Object.fromEntries(Object.entries(readUsageSnapshots(stateRoot)).map(([provider, snapshot]) => [provider, usageStatus(snapshot)])),
194
195
  maxWorkers: currentMaxWorkers(),
195
- remote: { running: runningCount("remote"), maxWorkers: currentMaxPoolWorkers(), note: "api and subscription agents; each agent's own max_concurrent also applies" }
196
+ remote: (() => { const limit = maxJobs(); return { running: runningCount("remote"), maxWorkers: limit.value, setBy: limit.source === "file" ? limit.path : limit.source === "env" ? "NOMARMY_MAX_POOL_WORKERS" : "default", note: "api and subscription agents; each agent's own max_concurrent also applies. Change with `nomarmy config max-jobs <n>`." }; })()
196
197
  };
197
198
  }
198
199
  async function admit(jobs) {
@@ -352,7 +353,7 @@ export function createJobRuntime(deps) {
352
353
  if (jobs.some((j) => jobLane(j) === "remote")) {
353
354
  const remoteCeiling = currentMaxPoolWorkers(), runningRemote = runningCount("remote");
354
355
  if (runningRemote >= remoteCeiling) {
355
- problems.push(`not admitted (capacity): ${runningRemote} remote job(s) (api or subscription agents) already running, at NOMARMY_MAX_POOL_WORKERS=${remoteCeiling}`);
356
+ problems.push(`not admitted (capacity): ${runningRemote} remote job(s) (api or subscription agents) already running, the limit of ${remoteCeiling} at once (raise it with \`nomarmy config max-jobs <n>\`)`);
356
357
  }
357
358
  }
358
359
  return { problems, admission };
@@ -88,6 +88,16 @@ export async function applyRetainedWork({ worktree, baseSha, commit, git }) {
88
88
  /** The brief's note, so the worker builds on the work instead of redoing it. */
89
89
  export function continuationNote({ continueFrom, record, files }) {
90
90
  const listed = files.slice(0, 20).join(", ") + (files.length > 20 ? `, and ${files.length - 20} more` : "");
91
- const why = (record.issues ?? []).slice(0, 3).map((i) => `- ${String(i).slice(0, 300)}`).join("\n");
91
+ // What stopped the work, for the worker: its own NOT DONE, a failed
92
+ // verification, a timeout. Not the notes addressed to the General (review
93
+ // flags, "accept only after an independent review"): a worker handed those
94
+ // took the pending review as its own unfinished task and reported partial.
95
+ const reasons = [];
96
+ const notDone = record.reportValidation?.notDone;
97
+ if (notDone && !/^none\.?$/i.test(String(notDone).trim())) reasons.push(`its worker's NOT DONE: ${notDone}`);
98
+ if (record.independentVerification?.status === "fail") reasons.push(`its verification failed: ${record.independentVerification.detail ?? record.independentVerification.reason ?? "see the job record"}`);
99
+ if (record.outcome === "WORKER_TIMEOUT") reasons.push("it ran out of time");
100
+ if (!reasons.length && record.outcome) reasons.push(`it ended as ${record.outcome}`);
101
+ const why = reasons.map((r) => `- ${String(r).slice(0, 300)}`).join("\n");
92
102
  return `This worktree already holds the unfinished work of job ${continueFrom} (${files.length} file(s): ${listed}). Build on it; don't redo it. Your finished diff is verified as a whole, that work included.${why ? `\nThat job stopped because:\n${why}` : ""}`;
93
103
  }
@@ -16,7 +16,7 @@ Before dispatching:
16
16
  - To run tests without changing anything, use mode: verify; it costs no model usage.
17
17
  - Mark an implement job stakes: high when a mistake would be costly (security, access control, personal or tenant data, data loss, money, irreversible changes), however small it is. It then needs a verification profile, keeps the revert check, and needs an independent review before you accept it: a scout on another vendor with reviews: <job id>.
18
18
  - Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
19
- - Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
19
+ - Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. To read a finished job, use local_worker_status with report: true (the report, a scout's cited findings, the outcome, issues and commit); full: true is the whole record and rarely needed. Review scouts (reviews set, or a review-phase role) default to 20 minutes, other jobs to 10; pass timeout_seconds for more. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
20
20
 
21
21
  Trust boundary:
22
22
  - A worker's four-line report is a claim; nomArmy's verified git record and independent verification are the evidence. A job isn't complete if its report is missing or malformed, its STATUS is partial or blocked, STATUS done lacks VERIFICATION pass, or its changes aren't committed by nomArmy.
@@ -111,6 +111,12 @@ const TEST_SELECTION_FLAG_PATTERNS = Object.freeze([
111
111
  export function detectScopedTestSelectionRisk({ commands = [], testChanges = null } = {}) {
112
112
  const touchedTestFiles = [...(testChanges?.new_tests_added ?? []), ...(testChanges?.existing_tests_modified ?? [])];
113
113
  if (touchedTestFiles.length === 0) return null;
114
+ // A command that runs the changed test files by name, with no filter of its
115
+ // own, runs exactly the tests this diff touched: another command's -k can't
116
+ // exclude them. Senti's python profile does this and the warning fired on
117
+ // every job, which teaches people to ignore it.
118
+ const runsChangedTestsByName = commands.some((c) => /\$\{?NOMARMY_CHANGED_TEST_FILES\b/.test(String(c ?? "")) && !TEST_SELECTION_FLAG_PATTERNS.some((p) => p.re.test(String(c))));
119
+ if (runsChangedTestsByName) return null;
114
120
  const flagged = [];
115
121
  for (const command of commands) {
116
122
  const match = TEST_SELECTION_FLAG_PATTERNS.find((p) => p.re.test(String(command ?? "")));
package/lib/doctor.mjs CHANGED
@@ -10,6 +10,7 @@ import path from "node:path";
10
10
  import os from "node:os";
11
11
  import { spawn, spawnSync } from "node:child_process";
12
12
  import { executionMode } from "./execution.mjs";
13
+ import { maxJobs, jobsThatFit, vmGibFor } from "./limits.mjs";
13
14
 
14
15
  const MIN_NODE_MAJOR = 18;
15
16
  const DEFAULT_LLAMA_HOST = "127.0.0.1";
@@ -197,6 +198,20 @@ export function checkPodmanMachineMemory(facts) {
197
198
  };
198
199
  }
199
200
 
201
+ /** Whether the Podman VM fits the api and subscription jobs the operator allowed at once. */
202
+ export function checkPodmanVmFitsJobs(facts) {
203
+ const fit = jobsThatFit(facts.podmanMachineMemoryMb);
204
+ const limit = facts.maxJobs;
205
+ if (limit?.problem) return { ok: false, message: limit.problem, fix: "nomarmy config max-jobs <n>" };
206
+ if (fit === null || !limit) return { ok: true, message: "Jobs at once vs Podman VM: not applicable." };
207
+ if (limit.value <= fit) return { ok: true, message: `Up to ${limit.value} api and subscription jobs at once; the Podman VM fits about ${fit}.` };
208
+ return {
209
+ ok: false,
210
+ message: `Up to ${limit.value} api and subscription jobs may run at once, but the Podman VM (${(facts.podmanMachineMemoryMb / 1024).toFixed(0)} GiB) fits about ${fit} sandboxes; the rest get refused for memory or cut off.`,
211
+ fix: `nomarmy sandbox --memory ${vmGibFor(limit.value)} (or lower the limit: nomarmy config max-jobs ${fit})`,
212
+ };
213
+ }
214
+
200
215
  export function checkPodmanDaemon(facts) {
201
216
  if (!facts.podmanFound) {
202
217
  return {
@@ -496,6 +511,7 @@ export async function collectFacts(env = process.env) {
496
511
  podmanDaemonReachable: podmanDaemon.reachable,
497
512
  podmanDaemonError: podmanDaemon.error,
498
513
  podmanMachineMemoryMb: podmanPath && podmanDaemon.reachable ? probePodmanMachineMemory(podmanPath, platform) : null,
514
+ maxJobs: maxJobs({ env }),
499
515
  podmanIdMappings: podmanPath && podmanDaemon.reachable ? probePodmanIdMappings(podmanPath, platform) : null,
500
516
  execution,
501
517
  endpoint,
@@ -515,6 +531,7 @@ export function evaluateChecks(facts) {
515
531
  { id: "podman", ...checkPodmanPresent(facts) },
516
532
  { id: "podman-daemon", ...checkPodmanDaemon(facts) },
517
533
  { id: "podman-vm-memory", ...checkPodmanMachineMemory(facts) },
534
+ { id: "podman-vm-jobs", ...checkPodmanVmFitsJobs(facts) },
518
535
  { id: "podman-idmap", ...checkPodmanIdMappings(facts) },
519
536
  { id: "endpoint", ...checkEndpoint(facts) },
520
537
  ];
package/lib/execute.mjs CHANGED
@@ -541,7 +541,11 @@ export function createExecutor(deps) {
541
541
  const request = readStopRequest(jobDir);
542
542
  issues.push(`stopped on request${request?.reason ? `: ${request.reason}` : ""}; the worktree is kept, so continue_from can pick the work up (on another model too)`);
543
543
  } else if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
544
- if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
544
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
545
+ // is recovered, and it happens on most Codex jobs: kept on the record for
546
+ // stats, out of the issues the General reviews.
547
+ const runnerNotes = [];
548
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
545
549
  if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
546
550
  if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
547
551
  if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
@@ -578,7 +582,7 @@ export function createExecutor(deps) {
578
582
  outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
579
583
  reportRecoveryAttempted, reportRecovered,
580
584
  reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
581
- coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], reportValidation, independentVerification,
585
+ coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], runnerNotes, reportValidation, independentVerification,
582
586
  // Original, unsubstituted regressionCheck (real "restore_failed" status
583
587
  // visible here even though resolveOutcome above only ever saw a
584
588
  // not_run-substituted view) -- full transparency for the caller.
@@ -640,7 +644,10 @@ export function createExecutor(deps) {
640
644
  const workerStartedMs = Date.now();
641
645
  progress("worker");
642
646
  try {
643
- result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
647
+ // The run gets the work share; the reserve stays for report recovery
648
+ // below. Scouts had none, so a timed-out scout left about 0 seconds
649
+ // and its findings were never recovered.
650
+ result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds: deriveTimeBudget({ timeoutSeconds }).workTimeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
644
651
  } catch (error) {
645
652
  workerFailed = true;
646
653
  // error.timedOut is set only by our own spawn timer (run(), above) --
@@ -670,10 +677,12 @@ export function createExecutor(deps) {
670
677
 
671
678
  // See shouldAttemptScoutRecovery's own doc comment: this only fires when
672
679
  // the report is genuinely unusable, gated by whatever time is actually
673
- // left against the caller's original timeout (scout has no reserved
674
- // report-phase budget the way implement does).
680
+ // left against the caller's original timeout (the reserve held back
681
+ // from the run above).
675
682
  let reportRecoveryAttempted = false, reportRecovered = false;
676
- const remainingSeconds = timeoutSeconds - Math.round(workerElapsedMs / 1000);
683
+ // At least the reserve, as implement's recovery gets: nomArmy's own kill
684
+ // lands 30s after OpenClaw's timer, which would otherwise eat it.
685
+ const remainingSeconds = Math.max(deriveTimeBudget({ timeoutSeconds }).reportReserveSeconds, timeoutSeconds - Math.round(workerElapsedMs / 1000));
677
686
  if (shouldAttemptScoutRecovery({ workerFailed, workerTimedOut, report, remainingSeconds })) {
678
687
  reportRecoveryAttempted = true;
679
688
  // What the first run left, in case the follow-up can't see its session.
@@ -725,7 +734,11 @@ export function createExecutor(deps) {
725
734
  const worker = workerMetadata(result ?? attempted);
726
735
  const issues = [...outcome.reasons];
727
736
  if (workerError) issues.push(`scout error: ${String(workerError).split("\n")[0]}`);
728
- if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
737
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
738
+ // is recovered, and it happens on most Codex jobs: kept on the record for
739
+ // stats, out of the issues the General reviews.
740
+ const runnerNotes = [];
741
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
729
742
  const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
730
743
  if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
731
744
  if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
@@ -768,7 +781,7 @@ export function createExecutor(deps) {
768
781
  const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
769
782
  objective: task, mustCover: acceptance ?? [], ...(reviews ? { reviews } : {}),
770
783
  ...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
771
- outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
784
+ outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
772
785
  scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
773
786
  findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
774
787
  excerptLinesUsed: verified.excerptLinesUsed, excerptTruncated: verified.excerptTruncated,
@@ -860,7 +873,11 @@ export function createExecutor(deps) {
860
873
  const worker = workerMetadata(result ?? attempted);
861
874
  const issues = [...outcome.reasons];
862
875
  if (workerError) issues.push(`decompose error: ${String(workerError).split("\n")[0]}`);
863
- if ((result ?? attempted)?.salvaged) issues.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the decomposer's report was recovered from the run's transcript`);
876
+ // OpenClaw's own cleanup failing after a finished run is benign once the report
877
+ // is recovered, and it happens on most Codex jobs: kept on the record for
878
+ // stats, out of the issues the General reviews.
879
+ const runnerNotes = [];
880
+ if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the decomposer's report was recovered from the run's transcript`);
864
881
  const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`decomposer recorded ${failures} tool failure(s)`);
865
882
  if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
866
883
  if (overlaps.length) issues.push(`${overlaps.length} subtask pair(s) claim overlapping files; not safe to dispatch as independent jobs as proposed`);
@@ -890,7 +907,7 @@ export function createExecutor(deps) {
890
907
  };
891
908
  const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
892
909
  objective: task, constraints: acceptance ?? [],
893
- outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
910
+ outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
894
911
  decompose: { objective: report.objective, confidence: report.confidence, notSplittable: report.notSplittable,
895
912
  subtasks: report.subtasks.map((s, i) => ({ task: s.task, acceptance: s.acceptance, citations: verified.findings[i]?.citations ?? [], supported: verified.findings[i]?.supported ?? false, weak: verified.findings[i]?.weak ?? false })),
896
913
  overlaps, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
@@ -88,6 +88,28 @@ export function compactJobRecord(meta) {
88
88
  return meta.mode === "scout" ? compactScoutRecord(meta) : meta.mode === "decompose" ? compactDecomposeRecord(meta) : compactImplementRecord(meta);
89
89
  }
90
90
 
91
+ /**
92
+ * The report and what nomArmy decided about it, without the rest of the
93
+ * record: full=true put the whole execution record in the General's context
94
+ * when it only wanted the worker's answer. Scout findings keep their verified
95
+ * citations, since they are the report.
96
+ */
97
+ export function reportView(meta) {
98
+ const iv = meta.independentVerification;
99
+ const base = { jobId: meta.jobId, mode: meta.mode, outcome: meta.outcome, coordinatorStatus: meta.coordinatorStatus, reviewRequired: meta.reviewRequired ?? false, issues: meta.issues ?? [] };
100
+ if (meta.mode === "scout") {
101
+ const s = meta.scout ?? {};
102
+ return { ...base, question: s.question ?? meta.objective ?? null, confidence: s.confidence ?? null, notFound: s.notFound ?? null, findings: s.findings ?? [], unsupported: s.unsupported ?? [] };
103
+ }
104
+ return { ...base,
105
+ report: meta.reportValidation?.fields ?? null,
106
+ verification: iv ? { status: iv.status, detail: iv.status === "fail" ? String(iv.detail ?? iv.reason ?? "").slice(0, 600) || null : null } : null,
107
+ revertCheck: meta.regressionCheck?.status ?? null,
108
+ commit: meta.commit ? { created: Boolean(meta.commit.created), sha: meta.commit.sha ?? null, branch: meta.branch ?? null, reason: meta.commit.created ? null : meta.commit.reason ?? null } : null,
109
+ changedFiles: meta.git?.changedFiles ?? [], additions: meta.git?.additions ?? null, deletions: meta.git?.deletions ?? null,
110
+ ...(meta.mode === "decompose" ? { proposal: meta.decompose ?? meta.proposal ?? null } : {}) };
111
+ }
112
+
91
113
  // Evidence before claim, in the display order too: the record is what
92
114
  // nomArmy verified against Git, the worker's report is prose it wrote about
93
115
  // itself. Leading with the report buried the record below whatever the
package/lib/limits.mjs ADDED
@@ -0,0 +1,77 @@
1
+ // Machine-wide limits, in ~/.config/nomarmy/limits.yml. One place for every
2
+ // coordinator: an environment variable lives in each Claude Code, Codex and
3
+ // Cursor registration separately, so the sessions could disagree.
4
+ //
5
+ // Its own file, not a section of config.yml: config.yml is validated strictly,
6
+ // and a session still running an older copy refused the whole file (and with
7
+ // it every role) when a new key appeared there. Older copies never read this
8
+ // file, so writing it can't break them.
9
+
10
+ import fs from "node:fs";
11
+ import path from "node:path";
12
+ import YAML from "yaml";
13
+ import { z } from "zod";
14
+ import { globalConfigDir } from "./army.mjs";
15
+ import { RESERVES } from "./sizing.mjs";
16
+
17
+ export const LIMITS_FILENAME = "limits.yml";
18
+ export const DEFAULT_MAX_JOBS = 4;
19
+ export const MAX_MAX_JOBS = 32;
20
+
21
+ export const limitsSchema = z.object({
22
+ // Api and subscription jobs at once, across every session. Local-model jobs have their own limit.
23
+ max_jobs: z.number().int().min(1).max(MAX_MAX_JOBS).optional(),
24
+ }).passthrough();
25
+
26
+ export function limitsPath(env = process.env) {
27
+ return path.join(globalConfigDir(env), LIMITS_FILENAME);
28
+ }
29
+
30
+ /**
31
+ * How many api and subscription jobs may run at once, and where that came
32
+ * from: limits.yml, then NOMARMY_MAX_POOL_WORKERS (older setups set it in each
33
+ * registration), then the default. Read on every call, so a change applies to
34
+ * the next job in every session. A missing or unreadable file falls through
35
+ * rather than holding jobs up.
36
+ */
37
+ export function maxJobs({ env = process.env, filePath = limitsPath(env) } = {}) {
38
+ let problem = null;
39
+ if (fs.existsSync(filePath)) {
40
+ try {
41
+ const parsed = limitsSchema.safeParse(YAML.parse(fs.readFileSync(filePath, "utf8")) ?? {});
42
+ if (parsed.success && parsed.data.max_jobs) return { value: parsed.data.max_jobs, source: "file", path: filePath, problem };
43
+ if (!parsed.success) problem = `${filePath}: max_jobs must be a whole number from 1 to ${MAX_MAX_JOBS}; ignored. Fix: nomarmy config max-jobs <n>`;
44
+ } catch (error) { problem = `${filePath} is not valid YAML (${error.message.split("\n")[0]}); ignored.`; }
45
+ }
46
+ const declared = Number.parseInt(env.NOMARMY_MAX_POOL_WORKERS ?? "", 10);
47
+ if (Number.isFinite(declared)) return { value: Math.min(MAX_MAX_JOBS, Math.max(1, declared)), source: "env", path: null, problem };
48
+ return { value: DEFAULT_MAX_JOBS, source: "default", path: null, problem };
49
+ }
50
+
51
+ /** Write max_jobs, keeping the rest of limits.yml (comments included) as it was. */
52
+ export function setMaxJobs(n, { filePath = limitsPath() } = {}) {
53
+ if (!Number.isInteger(n) || n < 1 || n > MAX_MAX_JOBS) throw new Error(`max-jobs must be a whole number from 1 to ${MAX_MAX_JOBS}, got "${n}"`);
54
+ const text = fs.existsSync(filePath) ? fs.readFileSync(filePath, "utf8") : "# Machine-wide limits for nomArmy: `nomarmy config max-jobs <n>` sets max_jobs.\n";
55
+ const doc = YAML.parseDocument(text);
56
+ if (doc.errors.length) throw new Error(`${filePath} is not valid YAML: ${doc.errors[0].message}`);
57
+ if (doc.contents === null) doc.contents = doc.createNode({});
58
+ doc.set("max_jobs", n);
59
+ fs.mkdirSync(path.dirname(filePath), { recursive: true });
60
+ fs.writeFileSync(filePath, String(doc));
61
+ return n;
62
+ }
63
+
64
+ // The Podman VM's own system and page cache, before any sandbox.
65
+ const VM_BASE_BYTES = 2 * 1024 ** 3;
66
+
67
+ /** How many sandboxes fit in a Podman VM of this size, by the per-job reserve admission uses. */
68
+ export function jobsThatFit(vmMemoryMb) {
69
+ if (!Number.isFinite(vmMemoryMb)) return null;
70
+ return Math.max(0, Math.floor((vmMemoryMb * 1024 ** 2 - VM_BASE_BYTES) / RESERVES.sandboxPerNomBytes));
71
+ }
72
+
73
+ /** The VM size (GiB, rounded up to an even number) that fits `jobs` sandboxes. */
74
+ export function vmGibFor(jobs) {
75
+ const gib = Math.ceil((VM_BASE_BYTES + jobs * RESERVES.sandboxPerNomBytes) / 1024 ** 3);
76
+ return gib + (gib % 2);
77
+ }
@@ -157,12 +157,44 @@ export function parseUnsupportedThinkingError(errorMessage) {
157
157
  // worker that ran out of room, but has valid session state worth resuming)
158
158
  // it exists for. Checked against the real captured envelope from that
159
159
  // incident, not a synthesized shape.
160
- export function parseOpenClawInternalTimeout(stdout) {
160
+ export function parseOpenClawInternalTimeout(stdout, stderr = "") {
161
+ // OpenClaw's own timer can end a run in the middle of a tool call. The
162
+ // envelope then reports that call's failure ("Read failed", status
163
+ // "error"), not a timeout; its run log still says so. Seen live on a Grok
164
+ // security review cut off mid-read at 600s: recorded as a crash, so its
165
+ // findings were never recovered.
166
+ if (/embedded run timeout: /.test(String(stderr))) return true;
161
167
  let parsed;
162
168
  try { parsed = JSON.parse(stdout); } catch { return false; }
163
169
  return parsed?.ok === false && (parsed?.status === "timeout" || parsed?.error?.kind === "timeout");
164
170
  }
165
171
 
172
+ /** What a failed run's envelope still says it used, so a failed job's spend is recorded too. */
173
+ export function failedRunUsage(stdout) {
174
+ let parsed;
175
+ try { parsed = JSON.parse(stdout); } catch { return {}; }
176
+ if (!parsed || typeof parsed !== "object") return {};
177
+ const out = {};
178
+ if (parsed.usage && typeof parsed.usage === "object") out.usage = parsed.usage;
179
+ if (Number.isFinite(parsed.costUsd)) out.costUsd = parsed.costUsd;
180
+ if (parsed.toolSummary && typeof parsed.toolSummary === "object") out.toolSummary = parsed.toolSummary;
181
+ if (typeof parsed.sessionId === "string") out.sessionId = parsed.sessionId;
182
+ return out;
183
+ }
184
+
185
+ /**
186
+ * The time a run has, told to the worker. Without it a Grok scout read files
187
+ * for its whole 10 minutes and was cut off before writing a word of its report.
188
+ */
189
+ export function timeBudgetNote(timeoutSeconds, mode) {
190
+ const total = Number(timeoutSeconds);
191
+ if (!Number.isFinite(total) || total <= 0) return "";
192
+ const minutes = Math.max(1, Math.round(total / 60));
193
+ const wrapAt = Math.max(1, Math.floor((total * 0.8) / 60));
194
+ const what = mode === "scout" || mode === "decompose" ? "stop exploring and write your report from what you have" : "stop starting new work, finish verification and write your report";
195
+ return `\n\nTIME\nThis run has about ${minutes} minute(s), then it is cut off. By minute ${wrapAt}, ${what}. A report on part of the question, with what's left under NOT DONE, is far more useful than none.`;
196
+ }
197
+
166
198
  /**
167
199
  * A run whose work finished but whose exit failed: OpenClaw logged the run
168
200
  * ending normally (stopReason=stop) and then errored, e.g. "Codex one-shot
@@ -471,7 +503,7 @@ export function createOpenClawRunner(deps) {
471
503
  ? scoutPrompt({ question: task, mustCover: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.scout, report: jobBudgets.report.scout, evidenceTool })
472
504
  : mode === "decompose"
473
505
  ? decomposePrompt({ objective: task, constraints: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.decompose, report: jobBudgets.report.decompose, evidenceTool })
474
- : workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement }));
506
+ : workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement })) + (overridePrompt ? "" : timeBudgetNote(timeoutSeconds, mode));
475
507
  fs.writeFileSync(path.join(jobDir, `brief${logSuffix}.txt`), prompt + "\n");
476
508
  // --state-dir keeps OpenClaw's session state (its transcript database among
477
509
  // it) inside the job directory instead of a temp dir it deletes on exit.
@@ -586,7 +618,7 @@ export function createOpenClawRunner(deps) {
586
618
  // a graceful internal timeout, not an opaque crash -- relabel it so
587
619
  // executeImplement/executeScout's workerTimedOut check (and therefore
588
620
  // report recovery) sees it correctly.
589
- if (!error.timedOut && parseOpenClawInternalTimeout(error.stdout)) {
621
+ if (!error.timedOut && error.stopReason !== "stopped" && parseOpenClawInternalTimeout(error.stdout, error.stderr)) {
590
622
  error.timedOut = true;
591
623
  error.stopReason = error.stopReason ?? "openclaw_internal_timeout";
592
624
  }
@@ -611,7 +643,7 @@ export function createOpenClawRunner(deps) {
611
643
  }
612
644
  // What was attempted, for the job record: a failed job used to be
613
645
  // labeled with the local default model, whatever it really ran on.
614
- error.partialResult = { model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
646
+ error.partialResult = { ...failedRunUsage(error.stdout), model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
615
647
  throw error;
616
648
  } finally {
617
649
  // A cloned copy of the ambient OpenClaw config (which may carry a real
package/lib/stats.mjs CHANGED
@@ -118,12 +118,13 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
118
118
 
119
119
  const workerMinutes = implement.map((r) => r.metrics?.worker_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
120
120
  const jobMinutes = implement.map((r) => r.metrics?.total_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
121
- const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
121
+ const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, untracked: 0 };
122
122
  const spend = {};
123
123
  for (const r of jobs) {
124
124
  const m = r.metrics ?? {};
125
125
  tokens.total += m.worker_tokens_total ?? 0; tokens.input += m.worker_tokens_in ?? 0; tokens.output += m.worker_tokens_out ?? 0;
126
126
  tokens.cacheRead += m.worker_tokens_cache_read ?? 0; tokens.cacheWrite += m.worker_tokens_cache_write ?? 0;
127
+ if (r.mode !== "verify" && !(m.worker_tokens_total > 0)) tokens.untracked++;
127
128
  if (Number.isFinite(m.worker_cost_usd) && m.worker_cost_usd > 0) spend[jobModel(r) ?? "unknown"] = (spend[jobModel(r) ?? "unknown"] ?? 0) + m.worker_cost_usd;
128
129
  }
129
130
 
@@ -131,12 +132,14 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
131
132
  const claimedDone = implement.filter((r) => r.reportValidation?.status === "done" && r.reportValidation?.tests === "pass");
132
133
  const verificationFailed = claimedDone.filter((r) => r.independentVerification?.status === "fail");
133
134
  const revertStillPassed = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status === "fail");
135
+ // Claimed success with an empty diff: there was nothing to verify.
136
+ const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
134
137
  const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
135
138
  const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
136
139
 
137
140
  const signals = {};
138
141
  for (const [name, re] of SIGNALS) {
139
- const n = jobs.filter((r) => (r.issues ?? []).some((i) => re.test(i))).length;
142
+ const n = jobs.filter((r) => [...(r.issues ?? []), ...(r.runnerNotes ?? [])].some((i) => re.test(i))).length;
140
143
  if (n) signals[name] = n;
141
144
  }
142
145
 
@@ -176,6 +179,7 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
176
179
  verificationFailed: verificationFailed.length,
177
180
  revertStillPassed: revertStillPassed.length,
178
181
  passedBoth: passedBoth.length,
182
+ changedNothing: changedNothing.length,
179
183
  flaggedAfterPassing: flaggedAfterPassing.length,
180
184
  },
181
185
  notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
@@ -212,7 +216,7 @@ export function formatStats(s) {
212
216
  ` Committed ${s.code.committedJobs} job(s) · +${s.code.linesAdded} / -${s.code.linesRemoved} lines · ${s.code.files} files · ${s.code.newTestFiles} new test files`,
213
217
  ` Worker time median ${mins(s.workerMinutes.median)} per implement job, p90 ${mins(s.workerMinutes.p90)}, total ${Math.round(s.workerMinutes.total)} min`,
214
218
  ` Job time median ${mins(s.jobMinutes.median)}, p90 ${mins(s.jobMinutes.p90)}, total ${Math.round(s.jobMinutes.total)} min (with verification and checks)`,
215
- ` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)`,
219
+ ` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)${s.tokens.untracked ? `; ${s.tokens.untracked} job(s) recorded no token counts` : ""}`,
216
220
  ` API spend $${s.spendUsd.total.toFixed(2)}${Object.keys(s.spendUsd.byModel).length ? ` (${Object.entries(s.spendUsd.byModel).map(([k, v]) => `${k} $${v.toFixed(2)}`).join(", ")})` : ""}; subscriptions aren't billed per call`,
217
221
  "",
218
222
  `CLAIM VS EVIDENCE (implement jobs that reported "done, tests pass": ${c.claimedDone})`,
@@ -220,6 +224,7 @@ export function formatStats(s) {
220
224
  ` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
221
225
  ` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
222
226
  ` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
227
+ ...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
223
228
  ` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
224
229
  " Defects the General found at integration aren't in the records; count them in your own review.",
225
230
  "",
@@ -11,7 +11,12 @@ const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE
11
11
 
12
12
  const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
13
13
  const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
14
- const agent = (r) => r.labels?.agent ?? null;
14
+ // The local model is the built-in "local" agent, whatever model is loaded.
15
+ const agent = (r) => r.labels?.agent ?? (provider(r) === "llama-cpp" ? "local" : null);
16
+ // New tokens only: cache reads are most of an api job's total and cost a fraction.
17
+ const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.worker_tokens_out ?? 0);
18
+ /** The runner exited before a report: an OpenClaw, sandbox or provider failure, not the model's work. */
19
+ export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
15
20
  const pct = (n, of) => Math.round((100 * n) / of);
16
21
 
17
22
  /** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
@@ -36,11 +41,13 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
36
41
  const groups = new Map();
37
42
  for (const r of work) {
38
43
  const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
39
- const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, ok: 0, timeout: 0, unsupported: 0, tokens: 0 };
40
- g.jobs++; if (OK.test(r.outcome ?? "")) g.ok++;
44
+ const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
45
+ g.jobs++;
46
+ if (runnerFailed(r)) g.runner++; else g.rated++;
47
+ if (OK.test(r.outcome ?? "")) g.ok++;
41
48
  if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
42
49
  if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
43
- g.tokens += r.metrics?.worker_tokens_total ?? 0;
50
+ if (freshTokens(r) > 0) { g.tokens += freshTokens(r); g.tokenJobs++; }
44
51
  g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
45
52
  groups.set(key, g);
46
53
  }
@@ -50,39 +57,47 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
50
57
  if (!g.model) continue;
51
58
  const kind = g.mode === "scout" ? "scouts" : "implement jobs";
52
59
  const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
60
+ // The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
61
+ if (g.runner >= 3 && g.runner / g.jobs >= 0.4) {
62
+ out.push({ level: "warn", key: `runner-failed:${g.role}:${g.model}:${g.mode}`, title: `${who}: the runner failed on ${g.runner} of ${g.jobs} ${kind} before any report`,
63
+ evidence: "OpenClaw, the sandbox or the provider exited early, so these say nothing about the model's work. Check `nomarmy health` and one job's log (`nomarmy jobs <id>`); a model its vendor refuses fails this way too.", command: null });
64
+ }
53
65
  // Scouts that come back empty.
54
- if (g.mode === "scout" && g.jobs >= 3 && g.unsupported / g.jobs >= 0.4) {
55
- out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.jobs} scouts came back unsupported`,
66
+ if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
67
+ out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
56
68
  evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
57
69
  continue;
58
70
  }
59
- if (g.jobs < minJobs) continue;
71
+ if (g.rated < minJobs) continue;
72
+ const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
60
73
  // A pairing that rarely finishes.
61
- if (g.ok / g.jobs < 0.5) {
62
- const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.jobs >= minJobs && o.model && o.ok / o.jobs >= g.ok / g.jobs + 0.2)
63
- .sort((a, b) => b.ok / b.jobs - a.ok / a.jobs)[0];
64
- out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.jobs} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.jobs)}%)`,
65
- evidence: (better ? `${better.model} finished ${pct(better.ok, better.jobs)}% of its ${better.jobs} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
74
+ if (g.ok / g.rated < 0.5) {
75
+ const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.rated >= minJobs && o.model && o.ok / o.rated >= g.ok / g.rated + 0.2)
76
+ .sort((a, b) => b.ok / b.rated - a.ok / a.rated)[0];
77
+ out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.rated} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.rated)}%)${runnerNote}`,
78
+ evidence: (better ? `${better.model} finished ${pct(better.ok, better.rated)}% of its ${better.rated} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
66
79
  command: g.role ? assign(g, better?.model ?? "<another model>") : null });
67
80
  }
68
81
  // Timeouts.
69
- if (g.timeout >= 3 && g.timeout / g.jobs >= 0.25) {
70
- out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.jobs} jobs`,
82
+ if (g.timeout >= 3 && g.timeout / g.rated >= 0.25) {
83
+ out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.rated} jobs`,
71
84
  evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
72
85
  }
73
86
  }
74
87
 
75
- // A lighter model doing as well on implement work, for much less.
76
- const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.model && g.jobs >= minJobs && g.tokens > 0);
88
+ // A lighter model doing as well on the same role's implement work, for much less. Only
89
+ // within one role: different roles do different work, so across roles the numbers don't compare.
90
+ const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
77
91
  for (const heavy of impl) {
78
92
  for (const light of impl) {
79
- if (light === heavy || light.model === heavy.model) continue;
80
- const lightRate = light.ok / light.jobs, heavyRate = heavy.ok / heavy.jobs;
81
- if (lightRate >= heavyRate - 0.05 && light.tokens / light.jobs <= 0.5 * (heavy.tokens / heavy.jobs)) {
93
+ if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
94
+ const lightRate = light.ok / light.rated, heavyRate = heavy.ok / heavy.rated;
95
+ const lightPerJob = light.tokens / light.tokenJobs, heavyPerJob = heavy.tokens / heavy.tokenJobs;
96
+ if (lightRate >= heavyRate - 0.05 && lightPerJob <= 0.5 * heavyPerJob) {
82
97
  out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
83
- title: `${light.model} finished ${pct(light.ok, light.jobs)}% of its jobs${light.role ? ` (as ${light.role})` : ""} on ${Math.round((light.tokens / light.jobs) / 1000)}k tokens a job; ${heavy.model}${heavy.role ? ` (as ${heavy.role})` : ""} finished ${pct(heavy.ok, heavy.jobs)}% on ${Math.round((heavy.tokens / heavy.jobs) / 1000)}k`,
84
- evidence: "They did different work, so try it rather than switch outright: put the role on auto so the General picks per job, or send its routine pieces to the lighter model.",
85
- command: heavy.role ? `nomarmy army assign ${heavy.role} ${heavy.agent ?? "<agent>"} auto` : null });
98
+ title: `${heavy.role}: ${light.model} finished ${pct(light.ok, light.rated)}% of its jobs on ${Math.round(lightPerJob / 1000)}k new tokens a job; ${heavy.model} finished ${pct(heavy.ok, heavy.rated)}% on ${Math.round(heavyPerJob / 1000)}k`,
99
+ evidence: `Same role, so similar work. Moving ${heavy.role} to ${light.model} would cost less; keep ${heavy.model} for the harder pieces with model on the job.${light.agent === "local" ? ` The local agent runs whichever model is loaded; these ran on ${light.model}.` : ""}`,
100
+ command: light.agent ? `nomarmy army assign ${heavy.role} ${light.agent}${light.agent === "local" ? "" : ` ${light.model}`}` : null });
86
101
  }
87
102
  }
88
103
  }
@@ -14,6 +14,7 @@
14
14
  // Estimates are labeled as such and use the same 4-chars-per-token rule as
15
15
  // the budgets; the point is the sign and the order of magnitude.
16
16
  import fs from "node:fs";
17
+ import zlib from "node:zlib";
17
18
  import path from "node:path";
18
19
  import { CALIBRATED } from "./budget.mjs";
19
20
 
@@ -33,6 +34,27 @@ export function findTranscriptDb(stateDir) {
33
34
  return null;
34
35
  }
35
36
 
37
+ /**
38
+ * One transcript row's event. OpenClaw 2026.9.6 stores larger events
39
+ * zstd-compressed in event_zstd with event_json null; reading event_json alone
40
+ * missed 85 of a Grok scout's 140 events (the file reads and command output
41
+ * report recovery and the idle breaker rely on). Null when the row can't be
42
+ * read: a torn row, or a Node without zstd (before 22.15).
43
+ */
44
+ export function eventFromRow(row) {
45
+ try {
46
+ if (row.event_json != null) return JSON.parse(row.event_json);
47
+ if (row.event_zstd != null && typeof zlib.zstdDecompressSync === "function") return JSON.parse(zlib.zstdDecompressSync(Buffer.from(row.event_zstd)).toString("utf8"));
48
+ } catch { /* torn or unreadable */ }
49
+ return null;
50
+ }
51
+
52
+ // event_zstd arrived with OpenClaw 2026.9.6; older databases don't have it.
53
+ function eventColumns(db) {
54
+ try { return db.prepare("select name from pragma_table_info('transcript_events')").all().some((c) => c.name === "event_zstd") ? "event_json, event_zstd" : "event_json"; }
55
+ catch { return "event_json"; }
56
+ }
57
+
36
58
  /**
37
59
  * Reduce raw transcript events to what the coordinator cares about. Pure:
38
60
  * takes the parsed `event_json` objects in order.
@@ -119,9 +141,9 @@ export async function readOpenClawTranscript(stateDir) {
119
141
  try {
120
142
  const db = new DatabaseSync(dbPath, { readOnly: true });
121
143
  try {
122
- const rows = db.prepare("select event_json from transcript_events order by seq, rowid").all();
144
+ const rows = db.prepare(`select ${eventColumns(db)} from transcript_events order by seq, rowid`).all();
123
145
  const events = [];
124
- for (const r of rows) { try { events.push(JSON.parse(r.event_json)); } catch { /* skip a torn row */ } }
146
+ for (const r of rows) { const e = eventFromRow(r); if (e) events.push(e); }
125
147
  return { available: true, reason: null, dbPath, events: events.length, ...summarizeTranscriptEvents(events) };
126
148
  } finally { db.close(); }
127
149
  } catch (error) {
@@ -164,10 +186,10 @@ export async function readOpenClawTranscriptTail(stateDir, { limit = 40, sinceEv
164
186
  try {
165
187
  const total = Number(db.prepare("select count(*) as n from transcript_events").get()?.n ?? 0);
166
188
  const rows = sinceEvent !== null
167
- ? db.prepare("select event_json from transcript_events order by rowid limit -1 offset ?").all(sinceEvent)
168
- : limit > 0 ? db.prepare("select event_json from transcript_events order by rowid desc limit ?").all(limit).reverse() : [];
189
+ ? db.prepare(`select ${eventColumns(db)} from transcript_events order by rowid limit -1 offset ?`).all(sinceEvent)
190
+ : limit > 0 ? db.prepare(`select ${eventColumns(db)} from transcript_events order by rowid desc limit ?`).all(limit).reverse() : [];
169
191
  const events = [];
170
- for (const r of rows) { try { events.push(JSON.parse(r.event_json)); } catch { /* skip a torn row */ } }
192
+ for (const r of rows) { const e = eventFromRow(r); if (e) events.push(e); }
171
193
  return { available: true, reason: null, dbPath, events: total, ...summarizeTranscriptEvents(events) };
172
194
  } finally { db.close(); }
173
195
  } catch (error) {
package/mcp/server.mjs CHANGED
@@ -43,7 +43,7 @@ import { probeModel } from "../lib/model-probe.mjs";
43
43
  import { jevSettings, judgeSettings } from "../lib/validators.mjs";
44
44
  import { agentRunsToolsOnHost } from "../lib/dispatch-schema.mjs";
45
45
  import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
46
- import { jobLabel, compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
46
+ import { jobLabel, compactJobRecord, formatResult, reportView, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
47
47
 
48
48
  export { run, mapLimit };
49
49
  export { readsMeasurable, measureReads };
@@ -219,6 +219,17 @@ export function makeHeartbeatTick(jobDir) { return heartbeatTick(jobDir, livePro
219
219
  // Senti run none were tagged, so a 4-hour run went 8.46 hours unchecked.
220
220
  let activeRunId = null;
221
221
 
222
+ export const DEFAULT_TIMEOUT_SECONDS = 600;
223
+ export const REVIEW_SCOUT_TIMEOUT_SECONDS = 1200;
224
+ /** A review scout (reviews set, or a review-phase role) gets longer: reviews trace across the codebase. */
225
+ export function defaultTimeoutSeconds(job, getArmyFn) {
226
+ if (job.mode !== "scout") return DEFAULT_TIMEOUT_SECONDS;
227
+ if (job.reviews) return REVIEW_SCOUT_TIMEOUT_SECONDS;
228
+ if (!job.army_role) return DEFAULT_TIMEOUT_SECONDS;
229
+ try { return getArmyFn()?.roles?.[job.army_role]?.phase === "review" ? REVIEW_SCOUT_TIMEOUT_SECONDS : DEFAULT_TIMEOUT_SECONDS; }
230
+ catch { return DEFAULT_TIMEOUT_SECONDS; }
231
+ }
232
+
222
233
  export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agentsConfig().agents, getActiveRun = () => activeRunId, env = process.env } = {}) {
223
234
  const problems = [];
224
235
  let army = null, agents = null;
@@ -226,6 +237,7 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
226
237
  const expanded = jobs.map((job, i) => {
227
238
  try {
228
239
  let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
240
+ if (j.timeout_seconds == null) j = { ...j, timeout_seconds: defaultTimeoutSeconds(j, () => (army ??= getArmy().army)) };
229
241
  if (j.mode === "verify") {
230
242
  const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
231
243
  return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
@@ -297,7 +309,7 @@ export const jobSchema = z.object({
297
309
  ),
298
310
  mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
299
311
  base_ref: z.string().optional(),
300
- timeout_seconds: z.number().int().min(30).max(1800).default(600),
312
+ timeout_seconds: z.number().int().min(30).max(1800).optional().describe("Default 600; 1200 for a review scout (one with `reviews`, or an army role in the review phase), since a real security review read for the full 10 minutes and was cut off."),
301
313
  reasoning: z.enum(["low", "medium", "high"]).default("medium").describe("Thinking level passed to the worker model. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
302
314
  agent: z.string().regex(/^[A-Za-z0-9._-]{1,64}$/).optional().describe("Run on this agent from the operator's agents.yml, by name (e.g. \"codex\", \"grok\", \"local\"): the local model, a metered api key, or one person's subscription. Omit agent and army_role to use the local model. Refuses an unknown name, never falls back. Mutually exclusive with army_role. A subscription agent also requires on_behalf_of."),
303
315
  model: z.string().regex(/^\S{1,200}$/).optional().describe("The model to run on the job's agent (an api or subscription agent), e.g. \"gpt-6-sol\". Overrides the role's model and the agent's default. Required when the role's model is \"auto\" or the agent has no default. The `army` tool lists each agent's models. Refused on the local agent, whose model `nomarmy model` sets."),
@@ -401,9 +413,10 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
401
413
  // always crossed; 110s returns in-line with margin. Raise it only for a
402
414
  // client that neither backgrounds nor times out that early.
403
415
  export const MAX_STATUS_WAIT_SECONDS = Number.parseInt(process.env.NOMARMY_MAX_STATUS_WAIT_SECONDS ?? "", 10) || 110;
404
- server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). full=true returns the complete formatted result instead of a summary.`, {
405
- job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false)
406
- }, async ({ job_id, wait_seconds, full }) => {
416
+ server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). report=true returns just the worker's report (a scout's findings with their verified citations), the outcome, issues, verification and commit. full=true returns the complete execution record.`, {
417
+ job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false),
418
+ report: z.boolean().default(false).describe("Just the report and nomArmy's verdict on it, without the rest of the record. Prefer this to full."),
419
+ }, async ({ job_id, wait_seconds, full, report }) => {
407
420
  const jobId = path.basename(job_id), entry = activeJobs.get(jobId), jobDir = path.join(ensureJobsRoot(), jobId);
408
421
  if (entry && !entry.settled && wait_seconds > 0) await Promise.race([entry.promise.catch(() => {}), sleep(wait_seconds * 1000)]);
409
422
  const files = { status: readJson(path.join(jobDir, "status.json")), meta: readJson(path.join(jobDir, "metadata.json")), failure: readJson(path.join(jobDir, "failure.json")) };
@@ -416,9 +429,10 @@ server.tool("local_worker_status", `Status of one job started by this server: ph
416
429
  ]);
417
430
  if (summary.state === "running") return toolText(JSON.stringify({ ...summary, jobDir, hint: `poll again with wait_seconds up to ${MAX_STATUS_WAIT_SECONDS}; lastTool/filesChangedLive are best-effort and may be absent early in a run` }, null, 2));
418
431
  if (entry?.error) return toolText(JSON.stringify({ ...summary, jobDir }, null, 2), true);
432
+ if (report && files.meta) return toolText(JSON.stringify(reportView(files.meta), null, 2), summary.coordinatorStatus !== "complete");
419
433
  if (full && entry?.result) return toolText(formatResult(entry.result), !entry.result.ok);
420
434
  if (full && files.meta) return toolText(JSON.stringify(files.meta, null, 2), summary.coordinatorStatus !== "complete");
421
- return toolText(JSON.stringify({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with full=true for the complete report" : null }, null, 2), summary.state === "orphaned" || summary.state === "failed");
435
+ return toolText(JSON.stringify({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with report=true for the worker's report, or full=true for the complete record" : null }, null, 2), summary.state === "orphaned" || summary.state === "failed");
422
436
  });
423
437
  // Set when this session's copy of nomArmy changed on disk after it started
424
438
  // (nomarmy connect or update ran): shown first in army and capacity, and
@@ -582,7 +596,7 @@ server.tool("local_worker_config", "What this checkout's .nomarmy.yml defines --
582
596
  return toolText(JSON.stringify(summary, null, 2), summary.valid === false);
583
597
  });
584
598
  server.tool("local_workers", "Run independent jobs (implement or scout) with bounded parallelism and wait for all of them. Every implement job receives its own branch, worktree, sandbox session, logs, validation, and coordinator-owned commit. This tool never merges any branch into the developer's branch. With auto_union: true, implement jobs that reach a valid outcome and touch non-overlapping files are additionally merged (git merge --no-ff) into ONE new integration branch -- a review artifact alongside the untouched per-job branches, still not the developer's branch, still reviewed and integrated explicitly. Jobs that overlap or did not finish validly are excluded from the union and reported individually exactly as without auto_union. For long batches prefer local_worker_start per job and poll.", {
585
- jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (NOMARMY_MAX_POOL_WORKERS), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
599
+ jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (`nomarmy config max-jobs`, default 4), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
586
600
  auto_union: z.boolean().default(false).describe(
587
601
  "After all jobs finish, mechanically merge (git merge --no-ff) implement jobs that reached a valid outcome and touched non-overlapping files into ONE new integration branch for review -- never into the developer's branch. Overlapping or invalid-outcome jobs are excluded and still reported individually, unchanged. All jobs must share one base_ref (or omit it); it is resolved once, before any job starts, and forced onto every job so the union is provably rooted at a single base."
588
602
  ),
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
4
4
  "author": "Rayson Technologies",
5
5
  "license": "Apache-2.0",
6
- "version": "0.1.0-alpha.14",
6
+ "version": "0.1.0-alpha.16",
7
7
  "private": false,
8
8
  "type": "module",
9
9
  "engines": {