nomarmy 0.1.0-alpha.14 → 0.1.0-alpha.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nomarmy.mjs +34 -1
- package/lib/admission.mjs +5 -4
- package/lib/continue-from.mjs +11 -1
- package/lib/coordinator-instructions.mjs +1 -1
- package/lib/diff-checks.mjs +6 -0
- package/lib/doctor.mjs +17 -0
- package/lib/execute.mjs +27 -10
- package/lib/job-format.mjs +22 -0
- package/lib/limits.mjs +77 -0
- package/lib/openclaw-run.mjs +36 -4
- package/lib/stats.mjs +8 -3
- package/lib/suggestions.mjs +37 -22
- package/lib/transcript.mjs +27 -5
- package/mcp/server.mjs +21 -7
- package/package.json +1 -1
package/bin/nomarmy.mjs
CHANGED
|
@@ -199,6 +199,10 @@ Usage: nomarmy <command> [options]
|
|
|
199
199
|
which agent the General is, in --global
|
|
200
200
|
(default) or --local
|
|
201
201
|
config paths where agents.yml and the three army layers live
|
|
202
|
+
config max-jobs [n]
|
|
203
|
+
how many api and subscription jobs run at once, across
|
|
204
|
+
every session (default 4); with n, sets it in limits.yml.
|
|
205
|
+
Warns when the Podman VM is too small for that many.
|
|
202
206
|
jobs [--watch|--events [--until-done]|--prune|--wait <jobId>|--stop <jobId> [--reason <text>]] [--interval N] [--older-than DAYS]
|
|
203
207
|
what's running across every session (agent, model, phase,
|
|
204
208
|
last tool call, files changed, heartbeat) and what just
|
|
@@ -1919,6 +1923,11 @@ async function cmdSandbox() {
|
|
|
1919
1923
|
const low = machine.memoryMb && machine.memoryMb < MIN_PODMAN_VM_MB;
|
|
1920
1924
|
console.log(`\nVM ${machine.name} (${machine.state}): ${machine.cpus} CPUs, ${low ? c.red(`${machine.memoryMb / 1024} GiB memory`) : `${machine.memoryMb / 1024} GiB memory`}, ${machine.diskGb} GB disk`);
|
|
1921
1925
|
if (low) console.log(c.yellow(` Too small: worker commands get cut off below ${MIN_PODMAN_VM_MB / 1024} GiB. Fix: nomarmy sandbox --memory 8`));
|
|
1926
|
+
const { maxJobs, jobsThatFit, vmGibFor } = await import("../lib/limits.mjs");
|
|
1927
|
+
const limit = maxJobs().value, fit = jobsThatFit(machine.memoryMb);
|
|
1928
|
+
if (fit !== null) console.log(limit > fit
|
|
1929
|
+
? c.yellow(` Fits about ${fit} sandboxes at once, but up to ${limit} api and subscription jobs may run. Fix: nomarmy sandbox --memory ${vmGibFor(limit)}, or nomarmy config max-jobs ${fit}`)
|
|
1930
|
+
: c.dim(` Fits about ${fit} sandboxes at once; up to ${limit} api and subscription jobs may run (nomarmy config max-jobs).`));
|
|
1922
1931
|
}
|
|
1923
1932
|
if (images) console.log(`Images: ${images.count}, ${images.size}${images.reclaimable ? `, ${images.reclaimable} reclaimable (nomarmy sandbox --prune)` : ""}`);
|
|
1924
1933
|
console.log(c.dim(runningJobs ? `${runningJobs} nomArmy job(s) running.` : "No nomArmy jobs running."));
|
|
@@ -2451,14 +2460,38 @@ async function cmdConfigPaths() {
|
|
|
2451
2460
|
if (json) return out({ globalDir: globalConfigDir(), agents: { path: agentsPath, exists: fs.existsSync(agentsPath) }, army });
|
|
2452
2461
|
console.log(c.bold("nomArmy config") + c.dim(` (global dir: ${globalConfigDir()})`));
|
|
2453
2462
|
console.log(` ${fs.existsSync(agentsPath) ? c.green("●") : c.dim("○")} ${"agents".padEnd(8)} ${c.dim(agentsPath)}`);
|
|
2463
|
+
const { limitsPath } = await import("../lib/limits.mjs");
|
|
2464
|
+
console.log(` ${fs.existsSync(limitsPath()) ? c.green("●") : c.dim("○")} ${"limits".padEnd(8)} ${c.dim(limitsPath())}`);
|
|
2454
2465
|
console.log(c.bold("\nArmy layers"));
|
|
2455
2466
|
for (const a of army) console.log(` ${a.exists ? c.green("●") : c.dim("○")} ${a.layer.padEnd(8)} ${c.dim(a.path)}`);
|
|
2456
2467
|
}
|
|
2457
2468
|
|
|
2469
|
+
// `nomarmy config max-jobs [n]`: api and subscription jobs at once, machine-wide.
|
|
2470
|
+
async function cmdConfigMaxJobs() {
|
|
2471
|
+
const { maxJobs, setMaxJobs, jobsThatFit, vmGibFor } = await import("../lib/limits.mjs");
|
|
2472
|
+
const given = argv[2];
|
|
2473
|
+
if (given !== undefined) {
|
|
2474
|
+
if (!/^\d+$/.test(given)) throw new Error(`max-jobs must be a whole number, got "${given}"`);
|
|
2475
|
+
setMaxJobs(Number(given));
|
|
2476
|
+
}
|
|
2477
|
+
const limit = maxJobs();
|
|
2478
|
+
const machine = process.platform === "linux" ? null : pickMachine(spawnSync("podman", ["machine", "inspect"], { encoding: "utf8" }).stdout);
|
|
2479
|
+
const fit = jobsThatFit(machine?.memoryMb);
|
|
2480
|
+
if (json) return out({ maxJobs: limit.value, source: limit.source, path: limit.path, problem: limit.problem, podmanVmMemoryMb: machine?.memoryMb ?? null, jobsThatFit: fit });
|
|
2481
|
+
const from = { file: `set in ${limit.path}`, env: "from NOMARMY_MAX_POOL_WORKERS in this shell (limits.yml doesn't set it)", default: "the default" }[limit.source];
|
|
2482
|
+
if (limit.problem) console.log(c.yellow(`⚠ ${limit.problem}`));
|
|
2483
|
+
console.log(`${given !== undefined ? c.green("✓ ") : ""}Up to ${c.bold(String(limit.value))} api and subscription jobs at once, across every session (${from}).`);
|
|
2484
|
+
console.log(c.dim("Each agent's max_concurrent in agents.yml also applies, and local-model jobs have their own limit."));
|
|
2485
|
+
if (given !== undefined) console.log(c.dim("Applies to the next job in every session on this version, no restart."));
|
|
2486
|
+
if (fit !== null && limit.value > fit) console.log(c.yellow(`⚠ The Podman VM (${machine.memoryMb / 1024} GiB) fits about ${fit} sandboxes at once; more get refused for memory or cut off. Fix: nomarmy sandbox --memory ${vmGibFor(limit.value)}`));
|
|
2487
|
+
if (given === undefined) console.log(c.dim("Change it with `nomarmy config max-jobs <n>` (1 to 32)."));
|
|
2488
|
+
}
|
|
2489
|
+
|
|
2458
2490
|
async function cmdConfig() {
|
|
2459
2491
|
const sub = argv[1] ?? "paths";
|
|
2460
2492
|
if (sub === "paths") return cmdConfigPaths();
|
|
2461
|
-
|
|
2493
|
+
if (sub === "max-jobs") return cmdConfigMaxJobs();
|
|
2494
|
+
throw new Error(`Unknown config subcommand "${sub}". Use: nomarmy config <paths|max-jobs [n]>`);
|
|
2462
2495
|
}
|
|
2463
2496
|
|
|
2464
2497
|
// --- `nomarmy jobs [--watch]`: what's running, from any session -----------
|
package/lib/admission.mjs
CHANGED
|
@@ -4,7 +4,7 @@ import { executionMode } from "./execution.mjs";
|
|
|
4
4
|
import { jobLabel, jobElapsedSeconds } from "./job-format.mjs";
|
|
5
5
|
import { continuationProblem } from "./continue-from.mjs";
|
|
6
6
|
import { loadConfig } from "./config.mjs";
|
|
7
|
-
import {
|
|
7
|
+
import { maxJobs } from "./limits.mjs";
|
|
8
8
|
import { checkBrief, assessAdmission, describeBudgets } from "./budget.mjs";
|
|
9
9
|
import { parseStatusPorcelainZ, isRuntimeJunk } from "./git-record.mjs";
|
|
10
10
|
import { readJson } from "./openclaw-run.mjs";
|
|
@@ -38,8 +38,9 @@ export function jobLane(job) {
|
|
|
38
38
|
// benefit of spreading load across providers with their own separate rate
|
|
39
39
|
// limits. Not rate-limit-aware (see config/providers.yml.example); read
|
|
40
40
|
// fresh each call, matching currentMaxWorkers()'s own env-read pattern.
|
|
41
|
+
/** Api and subscription jobs at once, machine-wide: limits.yml's max_jobs (see lib/limits.mjs). */
|
|
41
42
|
export function currentMaxPoolWorkers() {
|
|
42
|
-
return
|
|
43
|
+
return maxJobs().value;
|
|
43
44
|
}
|
|
44
45
|
|
|
45
46
|
// Pure partition of a batch's ORIGINAL indices by lane -- pulled out of
|
|
@@ -192,7 +193,7 @@ export function createJobRuntime(deps) {
|
|
|
192
193
|
running: [...activeJobs.values()].filter(j => !j.settled).map(j => ({ jobId: j.jobId, workerId: j.workerId, mode: j.mode, lane: j.lane, startedAt: j.startedAt, phase: readJson(path.join(jobsRoot, j.jobId, "status.json"))?.phase ?? "starting" })),
|
|
193
194
|
usageLimits: Object.fromEntries(Object.entries(readUsageSnapshots(stateRoot)).map(([provider, snapshot]) => [provider, usageStatus(snapshot)])),
|
|
194
195
|
maxWorkers: currentMaxWorkers(),
|
|
195
|
-
remote: { running: runningCount("remote"), maxWorkers:
|
|
196
|
+
remote: (() => { const limit = maxJobs(); return { running: runningCount("remote"), maxWorkers: limit.value, setBy: limit.source === "file" ? limit.path : limit.source === "env" ? "NOMARMY_MAX_POOL_WORKERS" : "default", note: "api and subscription agents; each agent's own max_concurrent also applies. Change with `nomarmy config max-jobs <n>`." }; })()
|
|
196
197
|
};
|
|
197
198
|
}
|
|
198
199
|
async function admit(jobs) {
|
|
@@ -352,7 +353,7 @@ export function createJobRuntime(deps) {
|
|
|
352
353
|
if (jobs.some((j) => jobLane(j) === "remote")) {
|
|
353
354
|
const remoteCeiling = currentMaxPoolWorkers(), runningRemote = runningCount("remote");
|
|
354
355
|
if (runningRemote >= remoteCeiling) {
|
|
355
|
-
problems.push(`not admitted (capacity): ${runningRemote} remote job(s) (api or subscription agents) already running,
|
|
356
|
+
problems.push(`not admitted (capacity): ${runningRemote} remote job(s) (api or subscription agents) already running, the limit of ${remoteCeiling} at once (raise it with \`nomarmy config max-jobs <n>\`)`);
|
|
356
357
|
}
|
|
357
358
|
}
|
|
358
359
|
return { problems, admission };
|
package/lib/continue-from.mjs
CHANGED
|
@@ -88,6 +88,16 @@ export async function applyRetainedWork({ worktree, baseSha, commit, git }) {
|
|
|
88
88
|
/** The brief's note, so the worker builds on the work instead of redoing it. */
|
|
89
89
|
export function continuationNote({ continueFrom, record, files }) {
|
|
90
90
|
const listed = files.slice(0, 20).join(", ") + (files.length > 20 ? `, and ${files.length - 20} more` : "");
|
|
91
|
-
|
|
91
|
+
// What stopped the work, for the worker: its own NOT DONE, a failed
|
|
92
|
+
// verification, a timeout. Not the notes addressed to the General (review
|
|
93
|
+
// flags, "accept only after an independent review"): a worker handed those
|
|
94
|
+
// took the pending review as its own unfinished task and reported partial.
|
|
95
|
+
const reasons = [];
|
|
96
|
+
const notDone = record.reportValidation?.notDone;
|
|
97
|
+
if (notDone && !/^none\.?$/i.test(String(notDone).trim())) reasons.push(`its worker's NOT DONE: ${notDone}`);
|
|
98
|
+
if (record.independentVerification?.status === "fail") reasons.push(`its verification failed: ${record.independentVerification.detail ?? record.independentVerification.reason ?? "see the job record"}`);
|
|
99
|
+
if (record.outcome === "WORKER_TIMEOUT") reasons.push("it ran out of time");
|
|
100
|
+
if (!reasons.length && record.outcome) reasons.push(`it ended as ${record.outcome}`);
|
|
101
|
+
const why = reasons.map((r) => `- ${String(r).slice(0, 300)}`).join("\n");
|
|
92
102
|
return `This worktree already holds the unfinished work of job ${continueFrom} (${files.length} file(s): ${listed}). Build on it; don't redo it. Your finished diff is verified as a whole, that work included.${why ? `\nThat job stopped because:\n${why}` : ""}`;
|
|
93
103
|
}
|
|
@@ -16,7 +16,7 @@ Before dispatching:
|
|
|
16
16
|
- To run tests without changing anything, use mode: verify; it costs no model usage.
|
|
17
17
|
- Mark an implement job stakes: high when a mistake would be costly (security, access control, personal or tenant data, data loss, money, irreversible changes), however small it is. It then needs a verification profile, keeps the revert check, and needs an independent review before you accept it: a scout on another vendor with reviews: <job id>.
|
|
18
18
|
- Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
|
|
19
|
-
- Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
19
|
+
- Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. To read a finished job, use local_worker_status with report: true (the report, a scout's cited findings, the outcome, issues and commit); full: true is the whole record and rarely needed. Review scouts (reviews set, or a review-phase role) default to 20 minutes, other jobs to 10; pass timeout_seconds for more. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
20
20
|
|
|
21
21
|
Trust boundary:
|
|
22
22
|
- A worker's four-line report is a claim; nomArmy's verified git record and independent verification are the evidence. A job isn't complete if its report is missing or malformed, its STATUS is partial or blocked, STATUS done lacks VERIFICATION pass, or its changes aren't committed by nomArmy.
|
package/lib/diff-checks.mjs
CHANGED
|
@@ -111,6 +111,12 @@ const TEST_SELECTION_FLAG_PATTERNS = Object.freeze([
|
|
|
111
111
|
export function detectScopedTestSelectionRisk({ commands = [], testChanges = null } = {}) {
|
|
112
112
|
const touchedTestFiles = [...(testChanges?.new_tests_added ?? []), ...(testChanges?.existing_tests_modified ?? [])];
|
|
113
113
|
if (touchedTestFiles.length === 0) return null;
|
|
114
|
+
// A command that runs the changed test files by name, with no filter of its
|
|
115
|
+
// own, runs exactly the tests this diff touched: another command's -k can't
|
|
116
|
+
// exclude them. Senti's python profile does this and the warning fired on
|
|
117
|
+
// every job, which teaches people to ignore it.
|
|
118
|
+
const runsChangedTestsByName = commands.some((c) => /\$\{?NOMARMY_CHANGED_TEST_FILES\b/.test(String(c ?? "")) && !TEST_SELECTION_FLAG_PATTERNS.some((p) => p.re.test(String(c))));
|
|
119
|
+
if (runsChangedTestsByName) return null;
|
|
114
120
|
const flagged = [];
|
|
115
121
|
for (const command of commands) {
|
|
116
122
|
const match = TEST_SELECTION_FLAG_PATTERNS.find((p) => p.re.test(String(command ?? "")));
|
package/lib/doctor.mjs
CHANGED
|
@@ -10,6 +10,7 @@ import path from "node:path";
|
|
|
10
10
|
import os from "node:os";
|
|
11
11
|
import { spawn, spawnSync } from "node:child_process";
|
|
12
12
|
import { executionMode } from "./execution.mjs";
|
|
13
|
+
import { maxJobs, jobsThatFit, vmGibFor } from "./limits.mjs";
|
|
13
14
|
|
|
14
15
|
const MIN_NODE_MAJOR = 18;
|
|
15
16
|
const DEFAULT_LLAMA_HOST = "127.0.0.1";
|
|
@@ -197,6 +198,20 @@ export function checkPodmanMachineMemory(facts) {
|
|
|
197
198
|
};
|
|
198
199
|
}
|
|
199
200
|
|
|
201
|
+
/** Whether the Podman VM fits the api and subscription jobs the operator allowed at once. */
|
|
202
|
+
export function checkPodmanVmFitsJobs(facts) {
|
|
203
|
+
const fit = jobsThatFit(facts.podmanMachineMemoryMb);
|
|
204
|
+
const limit = facts.maxJobs;
|
|
205
|
+
if (limit?.problem) return { ok: false, message: limit.problem, fix: "nomarmy config max-jobs <n>" };
|
|
206
|
+
if (fit === null || !limit) return { ok: true, message: "Jobs at once vs Podman VM: not applicable." };
|
|
207
|
+
if (limit.value <= fit) return { ok: true, message: `Up to ${limit.value} api and subscription jobs at once; the Podman VM fits about ${fit}.` };
|
|
208
|
+
return {
|
|
209
|
+
ok: false,
|
|
210
|
+
message: `Up to ${limit.value} api and subscription jobs may run at once, but the Podman VM (${(facts.podmanMachineMemoryMb / 1024).toFixed(0)} GiB) fits about ${fit} sandboxes; the rest get refused for memory or cut off.`,
|
|
211
|
+
fix: `nomarmy sandbox --memory ${vmGibFor(limit.value)} (or lower the limit: nomarmy config max-jobs ${fit})`,
|
|
212
|
+
};
|
|
213
|
+
}
|
|
214
|
+
|
|
200
215
|
export function checkPodmanDaemon(facts) {
|
|
201
216
|
if (!facts.podmanFound) {
|
|
202
217
|
return {
|
|
@@ -496,6 +511,7 @@ export async function collectFacts(env = process.env) {
|
|
|
496
511
|
podmanDaemonReachable: podmanDaemon.reachable,
|
|
497
512
|
podmanDaemonError: podmanDaemon.error,
|
|
498
513
|
podmanMachineMemoryMb: podmanPath && podmanDaemon.reachable ? probePodmanMachineMemory(podmanPath, platform) : null,
|
|
514
|
+
maxJobs: maxJobs({ env }),
|
|
499
515
|
podmanIdMappings: podmanPath && podmanDaemon.reachable ? probePodmanIdMappings(podmanPath, platform) : null,
|
|
500
516
|
execution,
|
|
501
517
|
endpoint,
|
|
@@ -515,6 +531,7 @@ export function evaluateChecks(facts) {
|
|
|
515
531
|
{ id: "podman", ...checkPodmanPresent(facts) },
|
|
516
532
|
{ id: "podman-daemon", ...checkPodmanDaemon(facts) },
|
|
517
533
|
{ id: "podman-vm-memory", ...checkPodmanMachineMemory(facts) },
|
|
534
|
+
{ id: "podman-vm-jobs", ...checkPodmanVmFitsJobs(facts) },
|
|
518
535
|
{ id: "podman-idmap", ...checkPodmanIdMappings(facts) },
|
|
519
536
|
{ id: "endpoint", ...checkEndpoint(facts) },
|
|
520
537
|
];
|
package/lib/execute.mjs
CHANGED
|
@@ -541,7 +541,11 @@ export function createExecutor(deps) {
|
|
|
541
541
|
const request = readStopRequest(jobDir);
|
|
542
542
|
issues.push(`stopped on request${request?.reason ? `: ${request.reason}` : ""}; the worktree is kept, so continue_from can pick the work up (on another model too)`);
|
|
543
543
|
} else if (workerError) issues.push(`worker error: ${String(workerError).split("\n")[0]}`);
|
|
544
|
-
|
|
544
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
545
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
546
|
+
// stats, out of the issues the General reviews.
|
|
547
|
+
const runnerNotes = [];
|
|
548
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the worker's report was recovered from the run's transcript`);
|
|
545
549
|
if (jevClaims?.error) issues.push(`Jev report check skipped (${jevClaims.error}); this job's result doesn't depend on it`);
|
|
546
550
|
if (judged?.error) issues.push(`Judge ${judged.skipped ? "skipped" : "didn't answer"} (${judged.error}); this job's result doesn't depend on it`);
|
|
547
551
|
if (workerFailed || workerTimedOut) { const restarted = vmRestartIssue(vmStartedBefore, deps.podmanVmStartedAt?.() ?? null); if (restarted) issues.unshift(restarted); }
|
|
@@ -578,7 +582,7 @@ export function createExecutor(deps) {
|
|
|
578
582
|
outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
|
|
579
583
|
reportRecoveryAttempted, reportRecovered,
|
|
580
584
|
reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
|
|
581
|
-
coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], reportValidation, independentVerification,
|
|
585
|
+
coordinatorStatus, issues: [...issues, ...(independentVerification.issues ?? [])], runnerNotes, reportValidation, independentVerification,
|
|
582
586
|
// Original, unsubstituted regressionCheck (real "restore_failed" status
|
|
583
587
|
// visible here even though resolveOutcome above only ever saw a
|
|
584
588
|
// not_run-substituted view) -- full transparency for the caller.
|
|
@@ -640,7 +644,10 @@ export function createExecutor(deps) {
|
|
|
640
644
|
const workerStartedMs = Date.now();
|
|
641
645
|
progress("worker");
|
|
642
646
|
try {
|
|
643
|
-
|
|
647
|
+
// The run gets the work share; the reserve stays for report recovery
|
|
648
|
+
// below. Scouts had none, so a timed-out scout left about 0 seconds
|
|
649
|
+
// and its findings were never recovered.
|
|
650
|
+
result = await runOpenClaw({ task, acceptance, verification: null, mode, cwd: worktree, baseRef: base.ref, baseSha: base.sha, timeoutSeconds: deriveTimeBudget({ timeoutSeconds }).workTimeoutSeconds, runtimeDir, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, jobDir, workerId: workerId || jobId, evidenceTool: evidencePlaced ? evidenceTool : null });
|
|
644
651
|
} catch (error) {
|
|
645
652
|
workerFailed = true;
|
|
646
653
|
// error.timedOut is set only by our own spawn timer (run(), above) --
|
|
@@ -670,10 +677,12 @@ export function createExecutor(deps) {
|
|
|
670
677
|
|
|
671
678
|
// See shouldAttemptScoutRecovery's own doc comment: this only fires when
|
|
672
679
|
// the report is genuinely unusable, gated by whatever time is actually
|
|
673
|
-
// left against the caller's original timeout (
|
|
674
|
-
//
|
|
680
|
+
// left against the caller's original timeout (the reserve held back
|
|
681
|
+
// from the run above).
|
|
675
682
|
let reportRecoveryAttempted = false, reportRecovered = false;
|
|
676
|
-
|
|
683
|
+
// At least the reserve, as implement's recovery gets: nomArmy's own kill
|
|
684
|
+
// lands 30s after OpenClaw's timer, which would otherwise eat it.
|
|
685
|
+
const remainingSeconds = Math.max(deriveTimeBudget({ timeoutSeconds }).reportReserveSeconds, timeoutSeconds - Math.round(workerElapsedMs / 1000));
|
|
677
686
|
if (shouldAttemptScoutRecovery({ workerFailed, workerTimedOut, report, remainingSeconds })) {
|
|
678
687
|
reportRecoveryAttempted = true;
|
|
679
688
|
// What the first run left, in case the follow-up can't see its session.
|
|
@@ -725,7 +734,11 @@ export function createExecutor(deps) {
|
|
|
725
734
|
const worker = workerMetadata(result ?? attempted);
|
|
726
735
|
const issues = [...outcome.reasons];
|
|
727
736
|
if (workerError) issues.push(`scout error: ${String(workerError).split("\n")[0]}`);
|
|
728
|
-
|
|
737
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
738
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
739
|
+
// stats, out of the issues the General reviews.
|
|
740
|
+
const runnerNotes = [];
|
|
741
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the scout's report was recovered from the run's transcript`);
|
|
729
742
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`scout recorded ${failures} tool failure(s)`);
|
|
730
743
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
731
744
|
if (jevCitations?.flags.length) issues.push(`CITATIONS MAY NOT SUPPORT FINDINGS (Jev): ${jevCitations.flags.map((f) => `"${String(verified.findings[f.index].text).slice(0, 80)}${String(verified.findings[f.index].text).length > 80 ? "..." : ""}" (${f.verdict}, ${f.probability.toFixed(2)})`).join("; ")}. Read those cited lines before relying on them; they're marked [JEV] in the report.`);
|
|
@@ -768,7 +781,7 @@ export function createExecutor(deps) {
|
|
|
768
781
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
769
782
|
objective: task, mustCover: acceptance ?? [], ...(reviews ? { reviews } : {}),
|
|
770
783
|
...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
|
|
771
|
-
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
784
|
+
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
|
|
772
785
|
scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
|
|
773
786
|
findings: verified.findings, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|
|
774
787
|
excerptLinesUsed: verified.excerptLinesUsed, excerptTruncated: verified.excerptTruncated,
|
|
@@ -860,7 +873,11 @@ export function createExecutor(deps) {
|
|
|
860
873
|
const worker = workerMetadata(result ?? attempted);
|
|
861
874
|
const issues = [...outcome.reasons];
|
|
862
875
|
if (workerError) issues.push(`decompose error: ${String(workerError).split("\n")[0]}`);
|
|
863
|
-
|
|
876
|
+
// OpenClaw's own cleanup failing after a finished run is benign once the report
|
|
877
|
+
// is recovered, and it happens on most Codex jobs: kept on the record for
|
|
878
|
+
// stats, out of the issues the General reviews.
|
|
879
|
+
const runnerNotes = [];
|
|
880
|
+
if ((result ?? attempted)?.salvaged) runnerNotes.push(`runner cleanup failed after the run (${(result ?? attempted).salvagedFrom}); the decomposer's report was recovered from the run's transcript`);
|
|
864
881
|
const failures = worker.toolSummary?.failures ?? 0; if (failures > 0) issues.push(`decomposer recorded ${failures} tool failure(s)`);
|
|
865
882
|
if (dirty) issues.push(`snapshot changed: ${record.repoStatusFiles.join(", ")}`);
|
|
866
883
|
if (overlaps.length) issues.push(`${overlaps.length} subtask pair(s) claim overlapping files; not safe to dispatch as independent jobs as proposed`);
|
|
@@ -890,7 +907,7 @@ export function createExecutor(deps) {
|
|
|
890
907
|
};
|
|
891
908
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
892
909
|
objective: task, constraints: acceptance ?? [],
|
|
893
|
-
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
910
|
+
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues, runnerNotes,
|
|
894
911
|
decompose: { objective: report.objective, confidence: report.confidence, notSplittable: report.notSplittable,
|
|
895
912
|
subtasks: report.subtasks.map((s, i) => ({ task: s.task, acceptance: s.acceptance, citations: verified.findings[i]?.citations ?? [], supported: verified.findings[i]?.supported ?? false, weak: verified.findings[i]?.weak ?? false })),
|
|
896
913
|
overlaps, supported: verified.supported, unsupported: verified.unsupported, weak: verified.weak,
|
package/lib/job-format.mjs
CHANGED
|
@@ -88,6 +88,28 @@ export function compactJobRecord(meta) {
|
|
|
88
88
|
return meta.mode === "scout" ? compactScoutRecord(meta) : meta.mode === "decompose" ? compactDecomposeRecord(meta) : compactImplementRecord(meta);
|
|
89
89
|
}
|
|
90
90
|
|
|
91
|
+
/**
|
|
92
|
+
* The report and what nomArmy decided about it, without the rest of the
|
|
93
|
+
* record: full=true put the whole execution record in the General's context
|
|
94
|
+
* when it only wanted the worker's answer. Scout findings keep their verified
|
|
95
|
+
* citations, since they are the report.
|
|
96
|
+
*/
|
|
97
|
+
export function reportView(meta) {
|
|
98
|
+
const iv = meta.independentVerification;
|
|
99
|
+
const base = { jobId: meta.jobId, mode: meta.mode, outcome: meta.outcome, coordinatorStatus: meta.coordinatorStatus, reviewRequired: meta.reviewRequired ?? false, issues: meta.issues ?? [] };
|
|
100
|
+
if (meta.mode === "scout") {
|
|
101
|
+
const s = meta.scout ?? {};
|
|
102
|
+
return { ...base, question: s.question ?? meta.objective ?? null, confidence: s.confidence ?? null, notFound: s.notFound ?? null, findings: s.findings ?? [], unsupported: s.unsupported ?? [] };
|
|
103
|
+
}
|
|
104
|
+
return { ...base,
|
|
105
|
+
report: meta.reportValidation?.fields ?? null,
|
|
106
|
+
verification: iv ? { status: iv.status, detail: iv.status === "fail" ? String(iv.detail ?? iv.reason ?? "").slice(0, 600) || null : null } : null,
|
|
107
|
+
revertCheck: meta.regressionCheck?.status ?? null,
|
|
108
|
+
commit: meta.commit ? { created: Boolean(meta.commit.created), sha: meta.commit.sha ?? null, branch: meta.branch ?? null, reason: meta.commit.created ? null : meta.commit.reason ?? null } : null,
|
|
109
|
+
changedFiles: meta.git?.changedFiles ?? [], additions: meta.git?.additions ?? null, deletions: meta.git?.deletions ?? null,
|
|
110
|
+
...(meta.mode === "decompose" ? { proposal: meta.decompose ?? meta.proposal ?? null } : {}) };
|
|
111
|
+
}
|
|
112
|
+
|
|
91
113
|
// Evidence before claim, in the display order too: the record is what
|
|
92
114
|
// nomArmy verified against Git, the worker's report is prose it wrote about
|
|
93
115
|
// itself. Leading with the report buried the record below whatever the
|
package/lib/limits.mjs
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// Machine-wide limits, in ~/.config/nomarmy/limits.yml. One place for every
|
|
2
|
+
// coordinator: an environment variable lives in each Claude Code, Codex and
|
|
3
|
+
// Cursor registration separately, so the sessions could disagree.
|
|
4
|
+
//
|
|
5
|
+
// Its own file, not a section of config.yml: config.yml is validated strictly,
|
|
6
|
+
// and a session still running an older copy refused the whole file (and with
|
|
7
|
+
// it every role) when a new key appeared there. Older copies never read this
|
|
8
|
+
// file, so writing it can't break them.
|
|
9
|
+
|
|
10
|
+
import fs from "node:fs";
|
|
11
|
+
import path from "node:path";
|
|
12
|
+
import YAML from "yaml";
|
|
13
|
+
import { z } from "zod";
|
|
14
|
+
import { globalConfigDir } from "./army.mjs";
|
|
15
|
+
import { RESERVES } from "./sizing.mjs";
|
|
16
|
+
|
|
17
|
+
export const LIMITS_FILENAME = "limits.yml";
|
|
18
|
+
export const DEFAULT_MAX_JOBS = 4;
|
|
19
|
+
export const MAX_MAX_JOBS = 32;
|
|
20
|
+
|
|
21
|
+
export const limitsSchema = z.object({
|
|
22
|
+
// Api and subscription jobs at once, across every session. Local-model jobs have their own limit.
|
|
23
|
+
max_jobs: z.number().int().min(1).max(MAX_MAX_JOBS).optional(),
|
|
24
|
+
}).passthrough();
|
|
25
|
+
|
|
26
|
+
export function limitsPath(env = process.env) {
|
|
27
|
+
return path.join(globalConfigDir(env), LIMITS_FILENAME);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* How many api and subscription jobs may run at once, and where that came
|
|
32
|
+
* from: limits.yml, then NOMARMY_MAX_POOL_WORKERS (older setups set it in each
|
|
33
|
+
* registration), then the default. Read on every call, so a change applies to
|
|
34
|
+
* the next job in every session. A missing or unreadable file falls through
|
|
35
|
+
* rather than holding jobs up.
|
|
36
|
+
*/
|
|
37
|
+
export function maxJobs({ env = process.env, filePath = limitsPath(env) } = {}) {
|
|
38
|
+
let problem = null;
|
|
39
|
+
if (fs.existsSync(filePath)) {
|
|
40
|
+
try {
|
|
41
|
+
const parsed = limitsSchema.safeParse(YAML.parse(fs.readFileSync(filePath, "utf8")) ?? {});
|
|
42
|
+
if (parsed.success && parsed.data.max_jobs) return { value: parsed.data.max_jobs, source: "file", path: filePath, problem };
|
|
43
|
+
if (!parsed.success) problem = `${filePath}: max_jobs must be a whole number from 1 to ${MAX_MAX_JOBS}; ignored. Fix: nomarmy config max-jobs <n>`;
|
|
44
|
+
} catch (error) { problem = `${filePath} is not valid YAML (${error.message.split("\n")[0]}); ignored.`; }
|
|
45
|
+
}
|
|
46
|
+
const declared = Number.parseInt(env.NOMARMY_MAX_POOL_WORKERS ?? "", 10);
|
|
47
|
+
if (Number.isFinite(declared)) return { value: Math.min(MAX_MAX_JOBS, Math.max(1, declared)), source: "env", path: null, problem };
|
|
48
|
+
return { value: DEFAULT_MAX_JOBS, source: "default", path: null, problem };
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Write max_jobs, keeping the rest of limits.yml (comments included) as it was. */
|
|
52
|
+
export function setMaxJobs(n, { filePath = limitsPath() } = {}) {
|
|
53
|
+
if (!Number.isInteger(n) || n < 1 || n > MAX_MAX_JOBS) throw new Error(`max-jobs must be a whole number from 1 to ${MAX_MAX_JOBS}, got "${n}"`);
|
|
54
|
+
const text = fs.existsSync(filePath) ? fs.readFileSync(filePath, "utf8") : "# Machine-wide limits for nomArmy: `nomarmy config max-jobs <n>` sets max_jobs.\n";
|
|
55
|
+
const doc = YAML.parseDocument(text);
|
|
56
|
+
if (doc.errors.length) throw new Error(`${filePath} is not valid YAML: ${doc.errors[0].message}`);
|
|
57
|
+
if (doc.contents === null) doc.contents = doc.createNode({});
|
|
58
|
+
doc.set("max_jobs", n);
|
|
59
|
+
fs.mkdirSync(path.dirname(filePath), { recursive: true });
|
|
60
|
+
fs.writeFileSync(filePath, String(doc));
|
|
61
|
+
return n;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
// The Podman VM's own system and page cache, before any sandbox.
|
|
65
|
+
const VM_BASE_BYTES = 2 * 1024 ** 3;
|
|
66
|
+
|
|
67
|
+
/** How many sandboxes fit in a Podman VM of this size, by the per-job reserve admission uses. */
|
|
68
|
+
export function jobsThatFit(vmMemoryMb) {
|
|
69
|
+
if (!Number.isFinite(vmMemoryMb)) return null;
|
|
70
|
+
return Math.max(0, Math.floor((vmMemoryMb * 1024 ** 2 - VM_BASE_BYTES) / RESERVES.sandboxPerNomBytes));
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** The VM size (GiB, rounded up to an even number) that fits `jobs` sandboxes. */
|
|
74
|
+
export function vmGibFor(jobs) {
|
|
75
|
+
const gib = Math.ceil((VM_BASE_BYTES + jobs * RESERVES.sandboxPerNomBytes) / 1024 ** 3);
|
|
76
|
+
return gib + (gib % 2);
|
|
77
|
+
}
|
package/lib/openclaw-run.mjs
CHANGED
|
@@ -157,12 +157,44 @@ export function parseUnsupportedThinkingError(errorMessage) {
|
|
|
157
157
|
// worker that ran out of room, but has valid session state worth resuming)
|
|
158
158
|
// it exists for. Checked against the real captured envelope from that
|
|
159
159
|
// incident, not a synthesized shape.
|
|
160
|
-
export function parseOpenClawInternalTimeout(stdout) {
|
|
160
|
+
export function parseOpenClawInternalTimeout(stdout, stderr = "") {
|
|
161
|
+
// OpenClaw's own timer can end a run in the middle of a tool call. The
|
|
162
|
+
// envelope then reports that call's failure ("Read failed", status
|
|
163
|
+
// "error"), not a timeout; its run log still says so. Seen live on a Grok
|
|
164
|
+
// security review cut off mid-read at 600s: recorded as a crash, so its
|
|
165
|
+
// findings were never recovered.
|
|
166
|
+
if (/embedded run timeout: /.test(String(stderr))) return true;
|
|
161
167
|
let parsed;
|
|
162
168
|
try { parsed = JSON.parse(stdout); } catch { return false; }
|
|
163
169
|
return parsed?.ok === false && (parsed?.status === "timeout" || parsed?.error?.kind === "timeout");
|
|
164
170
|
}
|
|
165
171
|
|
|
172
|
+
/** What a failed run's envelope still says it used, so a failed job's spend is recorded too. */
|
|
173
|
+
export function failedRunUsage(stdout) {
|
|
174
|
+
let parsed;
|
|
175
|
+
try { parsed = JSON.parse(stdout); } catch { return {}; }
|
|
176
|
+
if (!parsed || typeof parsed !== "object") return {};
|
|
177
|
+
const out = {};
|
|
178
|
+
if (parsed.usage && typeof parsed.usage === "object") out.usage = parsed.usage;
|
|
179
|
+
if (Number.isFinite(parsed.costUsd)) out.costUsd = parsed.costUsd;
|
|
180
|
+
if (parsed.toolSummary && typeof parsed.toolSummary === "object") out.toolSummary = parsed.toolSummary;
|
|
181
|
+
if (typeof parsed.sessionId === "string") out.sessionId = parsed.sessionId;
|
|
182
|
+
return out;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* The time a run has, told to the worker. Without it a Grok scout read files
|
|
187
|
+
* for its whole 10 minutes and was cut off before writing a word of its report.
|
|
188
|
+
*/
|
|
189
|
+
export function timeBudgetNote(timeoutSeconds, mode) {
|
|
190
|
+
const total = Number(timeoutSeconds);
|
|
191
|
+
if (!Number.isFinite(total) || total <= 0) return "";
|
|
192
|
+
const minutes = Math.max(1, Math.round(total / 60));
|
|
193
|
+
const wrapAt = Math.max(1, Math.floor((total * 0.8) / 60));
|
|
194
|
+
const what = mode === "scout" || mode === "decompose" ? "stop exploring and write your report from what you have" : "stop starting new work, finish verification and write your report";
|
|
195
|
+
return `\n\nTIME\nThis run has about ${minutes} minute(s), then it is cut off. By minute ${wrapAt}, ${what}. A report on part of the question, with what's left under NOT DONE, is far more useful than none.`;
|
|
196
|
+
}
|
|
197
|
+
|
|
166
198
|
/**
|
|
167
199
|
* A run whose work finished but whose exit failed: OpenClaw logged the run
|
|
168
200
|
* ending normally (stopReason=stop) and then errored, e.g. "Codex one-shot
|
|
@@ -471,7 +503,7 @@ export function createOpenClawRunner(deps) {
|
|
|
471
503
|
? scoutPrompt({ question: task, mustCover: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.scout, report: jobBudgets.report.scout, evidenceTool })
|
|
472
504
|
: mode === "decompose"
|
|
473
505
|
? decomposePrompt({ objective: task, constraints: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.decompose, report: jobBudgets.report.decompose, evidenceTool })
|
|
474
|
-
: workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement }));
|
|
506
|
+
: workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement })) + (overridePrompt ? "" : timeBudgetNote(timeoutSeconds, mode));
|
|
475
507
|
fs.writeFileSync(path.join(jobDir, `brief${logSuffix}.txt`), prompt + "\n");
|
|
476
508
|
// --state-dir keeps OpenClaw's session state (its transcript database among
|
|
477
509
|
// it) inside the job directory instead of a temp dir it deletes on exit.
|
|
@@ -586,7 +618,7 @@ export function createOpenClawRunner(deps) {
|
|
|
586
618
|
// a graceful internal timeout, not an opaque crash -- relabel it so
|
|
587
619
|
// executeImplement/executeScout's workerTimedOut check (and therefore
|
|
588
620
|
// report recovery) sees it correctly.
|
|
589
|
-
if (!error.timedOut && parseOpenClawInternalTimeout(error.stdout)) {
|
|
621
|
+
if (!error.timedOut && error.stopReason !== "stopped" && parseOpenClawInternalTimeout(error.stdout, error.stderr)) {
|
|
590
622
|
error.timedOut = true;
|
|
591
623
|
error.stopReason = error.stopReason ?? "openclaw_internal_timeout";
|
|
592
624
|
}
|
|
@@ -611,7 +643,7 @@ export function createOpenClawRunner(deps) {
|
|
|
611
643
|
}
|
|
612
644
|
// What was attempted, for the job record: a failed job used to be
|
|
613
645
|
// labeled with the local default model, whatever it really ran on.
|
|
614
|
-
error.partialResult = { model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
|
|
646
|
+
error.partialResult = { ...failedRunUsage(error.stdout), model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
|
|
615
647
|
throw error;
|
|
616
648
|
} finally {
|
|
617
649
|
// A cloned copy of the ambient OpenClaw config (which may carry a real
|
package/lib/stats.mjs
CHANGED
|
@@ -118,12 +118,13 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
118
118
|
|
|
119
119
|
const workerMinutes = implement.map((r) => r.metrics?.worker_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
|
|
120
120
|
const jobMinutes = implement.map((r) => r.metrics?.total_elapsed).filter(Number.isFinite).map((ms) => ms / 60000);
|
|
121
|
-
const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
|
|
121
|
+
const tokens = { total: 0, input: 0, output: 0, cacheRead: 0, cacheWrite: 0, untracked: 0 };
|
|
122
122
|
const spend = {};
|
|
123
123
|
for (const r of jobs) {
|
|
124
124
|
const m = r.metrics ?? {};
|
|
125
125
|
tokens.total += m.worker_tokens_total ?? 0; tokens.input += m.worker_tokens_in ?? 0; tokens.output += m.worker_tokens_out ?? 0;
|
|
126
126
|
tokens.cacheRead += m.worker_tokens_cache_read ?? 0; tokens.cacheWrite += m.worker_tokens_cache_write ?? 0;
|
|
127
|
+
if (r.mode !== "verify" && !(m.worker_tokens_total > 0)) tokens.untracked++;
|
|
127
128
|
if (Number.isFinite(m.worker_cost_usd) && m.worker_cost_usd > 0) spend[jobModel(r) ?? "unknown"] = (spend[jobModel(r) ?? "unknown"] ?? 0) + m.worker_cost_usd;
|
|
128
129
|
}
|
|
129
130
|
|
|
@@ -131,12 +132,14 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
131
132
|
const claimedDone = implement.filter((r) => r.reportValidation?.status === "done" && r.reportValidation?.tests === "pass");
|
|
132
133
|
const verificationFailed = claimedDone.filter((r) => r.independentVerification?.status === "fail");
|
|
133
134
|
const revertStillPassed = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status === "fail");
|
|
135
|
+
// Claimed success with an empty diff: there was nothing to verify.
|
|
136
|
+
const changedNothing = claimedDone.filter((r) => r.independentVerification?.status === "not_run");
|
|
134
137
|
const passedBoth = claimedDone.filter((r) => r.independentVerification?.status === "pass" && r.regressionCheck?.status !== "fail");
|
|
135
138
|
const flaggedAfterPassing = passedBoth.filter((r) => (r.issues ?? []).some((i) => /^(MUTANTS SURVIVED|REPORT MAY NOT MATCH|JUDGE \(|VERIFICATION INPUT CHANGED)/.test(i)));
|
|
136
139
|
|
|
137
140
|
const signals = {};
|
|
138
141
|
for (const [name, re] of SIGNALS) {
|
|
139
|
-
const n = jobs.filter((r) => (r.issues ?? []).some((i) => re.test(i))).length;
|
|
142
|
+
const n = jobs.filter((r) => [...(r.issues ?? []), ...(r.runnerNotes ?? [])].some((i) => re.test(i))).length;
|
|
140
143
|
if (n) signals[name] = n;
|
|
141
144
|
}
|
|
142
145
|
|
|
@@ -176,6 +179,7 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
176
179
|
verificationFailed: verificationFailed.length,
|
|
177
180
|
revertStillPassed: revertStillPassed.length,
|
|
178
181
|
passedBoth: passedBoth.length,
|
|
182
|
+
changedNothing: changedNothing.length,
|
|
179
183
|
flaggedAfterPassing: flaggedAfterPassing.length,
|
|
180
184
|
},
|
|
181
185
|
notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
|
|
@@ -212,7 +216,7 @@ export function formatStats(s) {
|
|
|
212
216
|
` Committed ${s.code.committedJobs} job(s) · +${s.code.linesAdded} / -${s.code.linesRemoved} lines · ${s.code.files} files · ${s.code.newTestFiles} new test files`,
|
|
213
217
|
` Worker time median ${mins(s.workerMinutes.median)} per implement job, p90 ${mins(s.workerMinutes.p90)}, total ${Math.round(s.workerMinutes.total)} min`,
|
|
214
218
|
` Job time median ${mins(s.jobMinutes.median)}, p90 ${mins(s.jobMinutes.p90)}, total ${Math.round(s.jobMinutes.total)} min (with verification and checks)`,
|
|
215
|
-
` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)`,
|
|
219
|
+
` Tokens ${big(s.tokens.total)} total (${big(s.tokens.input)} in, ${big(s.tokens.output)} out, ${big(s.tokens.cacheRead)} cache read)${s.tokens.untracked ? `; ${s.tokens.untracked} job(s) recorded no token counts` : ""}`,
|
|
216
220
|
` API spend $${s.spendUsd.total.toFixed(2)}${Object.keys(s.spendUsd.byModel).length ? ` (${Object.entries(s.spendUsd.byModel).map(([k, v]) => `${k} $${v.toFixed(2)}`).join(", ")})` : ""}; subscriptions aren't billed per call`,
|
|
217
221
|
"",
|
|
218
222
|
`CLAIM VS EVIDENCE (implement jobs that reported "done, tests pass": ${c.claimedDone})`,
|
|
@@ -220,6 +224,7 @@ export function formatStats(s) {
|
|
|
220
224
|
` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
|
|
221
225
|
` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
|
|
222
226
|
` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
|
|
227
|
+
...(c.changedNothing ? [` Changed nothing, nothing to verify ${c.changedNothing}${pct(c.changedNothing, c.claimedDone)}`] : []),
|
|
223
228
|
` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
|
|
224
229
|
" Defects the General found at integration aren't in the records; count them in your own review.",
|
|
225
230
|
"",
|
package/lib/suggestions.mjs
CHANGED
|
@@ -11,7 +11,12 @@ const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE
|
|
|
11
11
|
|
|
12
12
|
const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
|
|
13
13
|
const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
|
|
14
|
-
|
|
14
|
+
// The local model is the built-in "local" agent, whatever model is loaded.
|
|
15
|
+
const agent = (r) => r.labels?.agent ?? (provider(r) === "llama-cpp" ? "local" : null);
|
|
16
|
+
// New tokens only: cache reads are most of an api job's total and cost a fraction.
|
|
17
|
+
const freshTokens = (r) => (r.metrics?.worker_tokens_in ?? 0) + (r.metrics?.worker_tokens_out ?? 0);
|
|
18
|
+
/** The runner exited before a report: an OpenClaw, sandbox or provider failure, not the model's work. */
|
|
19
|
+
export const runnerFailed = (r) => r.outcome === "WORKER_FAILED" && [...(r.issues ?? []), ...(r.reasons ?? [])].some((x) => /^(worker|scout|decomposer) process failed/.test(x));
|
|
15
20
|
const pct = (n, of) => Math.round((100 * n) / of);
|
|
16
21
|
|
|
17
22
|
/** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
|
|
@@ -36,11 +41,13 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
36
41
|
const groups = new Map();
|
|
37
42
|
for (const r of work) {
|
|
38
43
|
const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
|
|
39
|
-
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, ok: 0, timeout: 0, unsupported: 0, tokens: 0 };
|
|
40
|
-
g.jobs++;
|
|
44
|
+
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, rated: 0, ok: 0, runner: 0, timeout: 0, unsupported: 0, tokens: 0, tokenJobs: 0 };
|
|
45
|
+
g.jobs++;
|
|
46
|
+
if (runnerFailed(r)) g.runner++; else g.rated++;
|
|
47
|
+
if (OK.test(r.outcome ?? "")) g.ok++;
|
|
41
48
|
if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
|
|
42
49
|
if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
|
|
43
|
-
g.tokens += r.
|
|
50
|
+
if (freshTokens(r) > 0) { g.tokens += freshTokens(r); g.tokenJobs++; }
|
|
44
51
|
g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
|
|
45
52
|
groups.set(key, g);
|
|
46
53
|
}
|
|
@@ -50,39 +57,47 @@ export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = ()
|
|
|
50
57
|
if (!g.model) continue;
|
|
51
58
|
const kind = g.mode === "scout" ? "scouts" : "implement jobs";
|
|
52
59
|
const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
|
|
60
|
+
// The runner failing isn't the model doing poor work: say so apart, and leave those out of its rate.
|
|
61
|
+
if (g.runner >= 3 && g.runner / g.jobs >= 0.4) {
|
|
62
|
+
out.push({ level: "warn", key: `runner-failed:${g.role}:${g.model}:${g.mode}`, title: `${who}: the runner failed on ${g.runner} of ${g.jobs} ${kind} before any report`,
|
|
63
|
+
evidence: "OpenClaw, the sandbox or the provider exited early, so these say nothing about the model's work. Check `nomarmy health` and one job's log (`nomarmy jobs <id>`); a model its vendor refuses fails this way too.", command: null });
|
|
64
|
+
}
|
|
53
65
|
// Scouts that come back empty.
|
|
54
|
-
if (g.mode === "scout" && g.
|
|
55
|
-
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.
|
|
66
|
+
if (g.mode === "scout" && g.rated >= 3 && g.unsupported / g.rated >= 0.4) {
|
|
67
|
+
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.rated} scouts came back unsupported`,
|
|
56
68
|
evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
|
|
57
69
|
continue;
|
|
58
70
|
}
|
|
59
|
-
if (g.
|
|
71
|
+
if (g.rated < minJobs) continue;
|
|
72
|
+
const runnerNote = g.runner ? ` (plus ${g.runner} the runner failed on, not counted)` : "";
|
|
60
73
|
// A pairing that rarely finishes.
|
|
61
|
-
if (g.ok / g.
|
|
62
|
-
const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.
|
|
63
|
-
.sort((a, b) => b.ok / b.
|
|
64
|
-
out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.
|
|
65
|
-
evidence: (better ? `${better.model} finished ${pct(better.ok, better.
|
|
74
|
+
if (g.ok / g.rated < 0.5) {
|
|
75
|
+
const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.rated >= minJobs && o.model && o.ok / o.rated >= g.ok / g.rated + 0.2)
|
|
76
|
+
.sort((a, b) => b.ok / b.rated - a.ok / a.rated)[0];
|
|
77
|
+
out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.rated} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.rated)}%)${runnerNote}`,
|
|
78
|
+
evidence: (better ? `${better.model} finished ${pct(better.ok, better.rated)}% of its ${better.rated} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
|
|
66
79
|
command: g.role ? assign(g, better?.model ?? "<another model>") : null });
|
|
67
80
|
}
|
|
68
81
|
// Timeouts.
|
|
69
|
-
if (g.timeout >= 3 && g.timeout / g.
|
|
70
|
-
out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.
|
|
82
|
+
if (g.timeout >= 3 && g.timeout / g.rated >= 0.25) {
|
|
83
|
+
out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.rated} jobs`,
|
|
71
84
|
evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
|
|
72
85
|
}
|
|
73
86
|
}
|
|
74
87
|
|
|
75
|
-
// A lighter model doing as well on implement work, for much less.
|
|
76
|
-
|
|
88
|
+
// A lighter model doing as well on the same role's implement work, for much less. Only
|
|
89
|
+
// within one role: different roles do different work, so across roles the numbers don't compare.
|
|
90
|
+
const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.role && g.model && g.rated >= minJobs && g.tokenJobs >= minJobs);
|
|
77
91
|
for (const heavy of impl) {
|
|
78
92
|
for (const light of impl) {
|
|
79
|
-
if (light === heavy || light.model === heavy.model) continue;
|
|
80
|
-
const lightRate = light.ok / light.
|
|
81
|
-
|
|
93
|
+
if (light === heavy || light.role !== heavy.role || light.model === heavy.model) continue;
|
|
94
|
+
const lightRate = light.ok / light.rated, heavyRate = heavy.ok / heavy.rated;
|
|
95
|
+
const lightPerJob = light.tokens / light.tokenJobs, heavyPerJob = heavy.tokens / heavy.tokenJobs;
|
|
96
|
+
if (lightRate >= heavyRate - 0.05 && lightPerJob <= 0.5 * heavyPerJob) {
|
|
82
97
|
out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
|
|
83
|
-
title: `${light.model} finished ${pct(light.ok, light.
|
|
84
|
-
evidence:
|
|
85
|
-
command:
|
|
98
|
+
title: `${heavy.role}: ${light.model} finished ${pct(light.ok, light.rated)}% of its jobs on ${Math.round(lightPerJob / 1000)}k new tokens a job; ${heavy.model} finished ${pct(heavy.ok, heavy.rated)}% on ${Math.round(heavyPerJob / 1000)}k`,
|
|
99
|
+
evidence: `Same role, so similar work. Moving ${heavy.role} to ${light.model} would cost less; keep ${heavy.model} for the harder pieces with model on the job.${light.agent === "local" ? ` The local agent runs whichever model is loaded; these ran on ${light.model}.` : ""}`,
|
|
100
|
+
command: light.agent ? `nomarmy army assign ${heavy.role} ${light.agent}${light.agent === "local" ? "" : ` ${light.model}`}` : null });
|
|
86
101
|
}
|
|
87
102
|
}
|
|
88
103
|
}
|
package/lib/transcript.mjs
CHANGED
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
// Estimates are labeled as such and use the same 4-chars-per-token rule as
|
|
15
15
|
// the budgets; the point is the sign and the order of magnitude.
|
|
16
16
|
import fs from "node:fs";
|
|
17
|
+
import zlib from "node:zlib";
|
|
17
18
|
import path from "node:path";
|
|
18
19
|
import { CALIBRATED } from "./budget.mjs";
|
|
19
20
|
|
|
@@ -33,6 +34,27 @@ export function findTranscriptDb(stateDir) {
|
|
|
33
34
|
return null;
|
|
34
35
|
}
|
|
35
36
|
|
|
37
|
+
/**
|
|
38
|
+
* One transcript row's event. OpenClaw 2026.9.6 stores larger events
|
|
39
|
+
* zstd-compressed in event_zstd with event_json null; reading event_json alone
|
|
40
|
+
* missed 85 of a Grok scout's 140 events (the file reads and command output
|
|
41
|
+
* report recovery and the idle breaker rely on). Null when the row can't be
|
|
42
|
+
* read: a torn row, or a Node without zstd (before 22.15).
|
|
43
|
+
*/
|
|
44
|
+
export function eventFromRow(row) {
|
|
45
|
+
try {
|
|
46
|
+
if (row.event_json != null) return JSON.parse(row.event_json);
|
|
47
|
+
if (row.event_zstd != null && typeof zlib.zstdDecompressSync === "function") return JSON.parse(zlib.zstdDecompressSync(Buffer.from(row.event_zstd)).toString("utf8"));
|
|
48
|
+
} catch { /* torn or unreadable */ }
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// event_zstd arrived with OpenClaw 2026.9.6; older databases don't have it.
|
|
53
|
+
function eventColumns(db) {
|
|
54
|
+
try { return db.prepare("select name from pragma_table_info('transcript_events')").all().some((c) => c.name === "event_zstd") ? "event_json, event_zstd" : "event_json"; }
|
|
55
|
+
catch { return "event_json"; }
|
|
56
|
+
}
|
|
57
|
+
|
|
36
58
|
/**
|
|
37
59
|
* Reduce raw transcript events to what the coordinator cares about. Pure:
|
|
38
60
|
* takes the parsed `event_json` objects in order.
|
|
@@ -119,9 +141,9 @@ export async function readOpenClawTranscript(stateDir) {
|
|
|
119
141
|
try {
|
|
120
142
|
const db = new DatabaseSync(dbPath, { readOnly: true });
|
|
121
143
|
try {
|
|
122
|
-
const rows = db.prepare(
|
|
144
|
+
const rows = db.prepare(`select ${eventColumns(db)} from transcript_events order by seq, rowid`).all();
|
|
123
145
|
const events = [];
|
|
124
|
-
for (const r of rows) {
|
|
146
|
+
for (const r of rows) { const e = eventFromRow(r); if (e) events.push(e); }
|
|
125
147
|
return { available: true, reason: null, dbPath, events: events.length, ...summarizeTranscriptEvents(events) };
|
|
126
148
|
} finally { db.close(); }
|
|
127
149
|
} catch (error) {
|
|
@@ -164,10 +186,10 @@ export async function readOpenClawTranscriptTail(stateDir, { limit = 40, sinceEv
|
|
|
164
186
|
try {
|
|
165
187
|
const total = Number(db.prepare("select count(*) as n from transcript_events").get()?.n ?? 0);
|
|
166
188
|
const rows = sinceEvent !== null
|
|
167
|
-
? db.prepare(
|
|
168
|
-
: limit > 0 ? db.prepare(
|
|
189
|
+
? db.prepare(`select ${eventColumns(db)} from transcript_events order by rowid limit -1 offset ?`).all(sinceEvent)
|
|
190
|
+
: limit > 0 ? db.prepare(`select ${eventColumns(db)} from transcript_events order by rowid desc limit ?`).all(limit).reverse() : [];
|
|
169
191
|
const events = [];
|
|
170
|
-
for (const r of rows) {
|
|
192
|
+
for (const r of rows) { const e = eventFromRow(r); if (e) events.push(e); }
|
|
171
193
|
return { available: true, reason: null, dbPath, events: total, ...summarizeTranscriptEvents(events) };
|
|
172
194
|
} finally { db.close(); }
|
|
173
195
|
} catch (error) {
|
package/mcp/server.mjs
CHANGED
|
@@ -43,7 +43,7 @@ import { probeModel } from "../lib/model-probe.mjs";
|
|
|
43
43
|
import { jevSettings, judgeSettings } from "../lib/validators.mjs";
|
|
44
44
|
import { agentRunsToolsOnHost } from "../lib/dispatch-schema.mjs";
|
|
45
45
|
import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
|
|
46
|
-
import { jobLabel, compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
|
|
46
|
+
import { jobLabel, compactJobRecord, formatResult, reportView, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
|
|
47
47
|
|
|
48
48
|
export { run, mapLimit };
|
|
49
49
|
export { readsMeasurable, measureReads };
|
|
@@ -219,6 +219,17 @@ export function makeHeartbeatTick(jobDir) { return heartbeatTick(jobDir, livePro
|
|
|
219
219
|
// Senti run none were tagged, so a 4-hour run went 8.46 hours unchecked.
|
|
220
220
|
let activeRunId = null;
|
|
221
221
|
|
|
222
|
+
export const DEFAULT_TIMEOUT_SECONDS = 600;
|
|
223
|
+
export const REVIEW_SCOUT_TIMEOUT_SECONDS = 1200;
|
|
224
|
+
/** A review scout (reviews set, or a review-phase role) gets longer: reviews trace across the codebase. */
|
|
225
|
+
export function defaultTimeoutSeconds(job, getArmyFn) {
|
|
226
|
+
if (job.mode !== "scout") return DEFAULT_TIMEOUT_SECONDS;
|
|
227
|
+
if (job.reviews) return REVIEW_SCOUT_TIMEOUT_SECONDS;
|
|
228
|
+
if (!job.army_role) return DEFAULT_TIMEOUT_SECONDS;
|
|
229
|
+
try { return getArmyFn()?.roles?.[job.army_role]?.phase === "review" ? REVIEW_SCOUT_TIMEOUT_SECONDS : DEFAULT_TIMEOUT_SECONDS; }
|
|
230
|
+
catch { return DEFAULT_TIMEOUT_SECONDS; }
|
|
231
|
+
}
|
|
232
|
+
|
|
222
233
|
export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agentsConfig().agents, getActiveRun = () => activeRunId, env = process.env } = {}) {
|
|
223
234
|
const problems = [];
|
|
224
235
|
let army = null, agents = null;
|
|
@@ -226,6 +237,7 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
|
|
|
226
237
|
const expanded = jobs.map((job, i) => {
|
|
227
238
|
try {
|
|
228
239
|
let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
|
|
240
|
+
if (j.timeout_seconds == null) j = { ...j, timeout_seconds: defaultTimeoutSeconds(j, () => (army ??= getArmy().army)) };
|
|
229
241
|
if (j.mode === "verify") {
|
|
230
242
|
const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
|
|
231
243
|
return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
|
|
@@ -297,7 +309,7 @@ export const jobSchema = z.object({
|
|
|
297
309
|
),
|
|
298
310
|
mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
|
|
299
311
|
base_ref: z.string().optional(),
|
|
300
|
-
timeout_seconds: z.number().int().min(30).max(1800).
|
|
312
|
+
timeout_seconds: z.number().int().min(30).max(1800).optional().describe("Default 600; 1200 for a review scout (one with `reviews`, or an army role in the review phase), since a real security review read for the full 10 minutes and was cut off."),
|
|
301
313
|
reasoning: z.enum(["low", "medium", "high"]).default("medium").describe("Thinking level passed to the worker model. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
|
|
302
314
|
agent: z.string().regex(/^[A-Za-z0-9._-]{1,64}$/).optional().describe("Run on this agent from the operator's agents.yml, by name (e.g. \"codex\", \"grok\", \"local\"): the local model, a metered api key, or one person's subscription. Omit agent and army_role to use the local model. Refuses an unknown name, never falls back. Mutually exclusive with army_role. A subscription agent also requires on_behalf_of."),
|
|
303
315
|
model: z.string().regex(/^\S{1,200}$/).optional().describe("The model to run on the job's agent (an api or subscription agent), e.g. \"gpt-6-sol\". Overrides the role's model and the agent's default. Required when the role's model is \"auto\" or the agent has no default. The `army` tool lists each agent's models. Refused on the local agent, whose model `nomarmy model` sets."),
|
|
@@ -401,9 +413,10 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
|
|
|
401
413
|
// always crossed; 110s returns in-line with margin. Raise it only for a
|
|
402
414
|
// client that neither backgrounds nor times out that early.
|
|
403
415
|
export const MAX_STATUS_WAIT_SECONDS = Number.parseInt(process.env.NOMARMY_MAX_STATUS_WAIT_SECONDS ?? "", 10) || 110;
|
|
404
|
-
server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs).
|
|
405
|
-
job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false)
|
|
406
|
-
|
|
416
|
+
server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). report=true returns just the worker's report (a scout's findings with their verified citations), the outcome, issues, verification and commit. full=true returns the complete execution record.`, {
|
|
417
|
+
job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false),
|
|
418
|
+
report: z.boolean().default(false).describe("Just the report and nomArmy's verdict on it, without the rest of the record. Prefer this to full."),
|
|
419
|
+
}, async ({ job_id, wait_seconds, full, report }) => {
|
|
407
420
|
const jobId = path.basename(job_id), entry = activeJobs.get(jobId), jobDir = path.join(ensureJobsRoot(), jobId);
|
|
408
421
|
if (entry && !entry.settled && wait_seconds > 0) await Promise.race([entry.promise.catch(() => {}), sleep(wait_seconds * 1000)]);
|
|
409
422
|
const files = { status: readJson(path.join(jobDir, "status.json")), meta: readJson(path.join(jobDir, "metadata.json")), failure: readJson(path.join(jobDir, "failure.json")) };
|
|
@@ -416,9 +429,10 @@ server.tool("local_worker_status", `Status of one job started by this server: ph
|
|
|
416
429
|
]);
|
|
417
430
|
if (summary.state === "running") return toolText(JSON.stringify({ ...summary, jobDir, hint: `poll again with wait_seconds up to ${MAX_STATUS_WAIT_SECONDS}; lastTool/filesChangedLive are best-effort and may be absent early in a run` }, null, 2));
|
|
418
431
|
if (entry?.error) return toolText(JSON.stringify({ ...summary, jobDir }, null, 2), true);
|
|
432
|
+
if (report && files.meta) return toolText(JSON.stringify(reportView(files.meta), null, 2), summary.coordinatorStatus !== "complete");
|
|
419
433
|
if (full && entry?.result) return toolText(formatResult(entry.result), !entry.result.ok);
|
|
420
434
|
if (full && files.meta) return toolText(JSON.stringify(files.meta, null, 2), summary.coordinatorStatus !== "complete");
|
|
421
|
-
return toolText(JSON.stringify({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with full=true for the complete
|
|
435
|
+
return toolText(JSON.stringify({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with report=true for the worker's report, or full=true for the complete record" : null }, null, 2), summary.state === "orphaned" || summary.state === "failed");
|
|
422
436
|
});
|
|
423
437
|
// Set when this session's copy of nomArmy changed on disk after it started
|
|
424
438
|
// (nomarmy connect or update ran): shown first in army and capacity, and
|
|
@@ -582,7 +596,7 @@ server.tool("local_worker_config", "What this checkout's .nomarmy.yml defines --
|
|
|
582
596
|
return toolText(JSON.stringify(summary, null, 2), summary.valid === false);
|
|
583
597
|
});
|
|
584
598
|
server.tool("local_workers", "Run independent jobs (implement or scout) with bounded parallelism and wait for all of them. Every implement job receives its own branch, worktree, sandbox session, logs, validation, and coordinator-owned commit. This tool never merges any branch into the developer's branch. With auto_union: true, implement jobs that reach a valid outcome and touch non-overlapping files are additionally merged (git merge --no-ff) into ONE new integration branch -- a review artifact alongside the untouched per-job branches, still not the developer's branch, still reviewed and integrated explicitly. Jobs that overlap or did not finish validly are excluded from the union and reported individually exactly as without auto_union. For long batches prefer local_worker_start per job and poll.", {
|
|
585
|
-
jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (
|
|
599
|
+
jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (`nomarmy config max-jobs`, default 4), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
|
|
586
600
|
auto_union: z.boolean().default(false).describe(
|
|
587
601
|
"After all jobs finish, mechanically merge (git merge --no-ff) implement jobs that reached a valid outcome and touched non-overlapping files into ONE new integration branch for review -- never into the developer's branch. Overlapping or invalid-outcome jobs are excluded and still reported individually, unchanged. All jobs must share one base_ref (or omit it); it is resolved once, before any job starts, and forced onto every job so the union is provably rooted at a single base."
|
|
588
602
|
),
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.16",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|