nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +86 -480
  2. package/bin/nomarmy.mjs +1081 -185
  3. package/docker/Dockerfile +2 -2
  4. package/docker/Dockerfile.go +6 -4
  5. package/docker/Dockerfile.rust +17 -2
  6. package/harnesses/_template/README.md +27 -0
  7. package/harnesses/_template/harness.yml +26 -0
  8. package/harnesses/browser-playwright/README.md +35 -0
  9. package/harnesses/browser-playwright/fixture/package.json +1 -0
  10. package/harnesses/browser-playwright/fixture/page.html +1 -0
  11. package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
  12. package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
  13. package/harnesses/browser-playwright/harness.yml +18 -0
  14. package/harnesses/go/README.md +45 -0
  15. package/harnesses/go/harness.yml +14 -0
  16. package/harnesses/mock-oidc/README.md +31 -0
  17. package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
  18. package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
  19. package/harnesses/mock-oidc/harness.yml +19 -0
  20. package/harnesses/node/README.md +53 -0
  21. package/harnesses/node/harness.yml +18 -0
  22. package/harnesses/python/README.md +46 -0
  23. package/harnesses/python/harness.yml +16 -0
  24. package/harnesses/rust/README.md +45 -0
  25. package/harnesses/rust/harness.yml +13 -0
  26. package/install.sh +29 -9
  27. package/lib/admission.mjs +178 -30
  28. package/lib/agents.mjs +8 -6
  29. package/lib/army.mjs +25 -10
  30. package/lib/codex-link.mjs +37 -0
  31. package/lib/config.mjs +15 -0
  32. package/lib/connect.mjs +232 -19
  33. package/lib/continue-from.mjs +103 -0
  34. package/lib/coordinator-instructions.mjs +5 -1
  35. package/lib/diff-checks.mjs +114 -0
  36. package/lib/dispatch-schema.mjs +14 -12
  37. package/lib/doctor.mjs +98 -9
  38. package/lib/egress-proxy.mjs +116 -0
  39. package/lib/execute.mjs +241 -33
  40. package/lib/git-record.mjs +27 -3
  41. package/lib/harness-schema.mjs +61 -0
  42. package/lib/harnesses.mjs +99 -0
  43. package/lib/health.mjs +162 -18
  44. package/lib/install-freshness.mjs +114 -0
  45. package/lib/jev-checks.mjs +110 -0
  46. package/lib/job-format.mjs +54 -0
  47. package/lib/judge.mjs +130 -0
  48. package/lib/limits.mjs +77 -0
  49. package/lib/model-probe.mjs +61 -0
  50. package/lib/mutation.mjs +159 -0
  51. package/lib/notify.mjs +30 -3
  52. package/lib/openclaw-install.mjs +122 -0
  53. package/lib/openclaw-path.mjs +28 -0
  54. package/lib/openclaw-run.mjs +74 -12
  55. package/lib/openclaw-runtime-health.mjs +56 -0
  56. package/lib/outcome.mjs +21 -2
  57. package/lib/outcomes.mjs +6 -0
  58. package/lib/path-utils.mjs +4 -0
  59. package/lib/podman-health.mjs +41 -0
  60. package/lib/process.mjs +4 -1
  61. package/lib/propose.mjs +10 -11
  62. package/lib/refusal-retry.mjs +16 -0
  63. package/lib/registry-python.mjs +98 -0
  64. package/lib/registry-secrets.mjs +140 -0
  65. package/lib/repo-query.mjs +13 -7
  66. package/lib/runs.mjs +7 -1
  67. package/lib/same-path.mjs +14 -0
  68. package/lib/sandbox-images.mjs +499 -83
  69. package/lib/sandbox-vm.mjs +32 -0
  70. package/lib/scan.mjs +5 -1
  71. package/lib/schema.mjs +20 -11
  72. package/lib/scout.mjs +21 -3
  73. package/lib/server-context.mjs +21 -1
  74. package/lib/setup-steps.mjs +55 -0
  75. package/lib/share.mjs +82 -0
  76. package/lib/stale-sessions.mjs +60 -0
  77. package/lib/stats.mjs +315 -0
  78. package/lib/statusline.mjs +32 -6
  79. package/lib/subscription-setup.mjs +13 -0
  80. package/lib/suggestions.mjs +153 -0
  81. package/lib/thinking.mjs +23 -0
  82. package/lib/transcript.mjs +30 -5
  83. package/lib/usage-limits.mjs +329 -0
  84. package/lib/user-config.mjs +106 -0
  85. package/lib/validators.mjs +220 -0
  86. package/lib/verification-artifacts.mjs +46 -0
  87. package/lib/verification-flow.mjs +52 -7
  88. package/lib/verification-network.mjs +66 -0
  89. package/lib/verify.mjs +338 -85
  90. package/lib/worker-prompt.mjs +5 -2
  91. package/lib/wsl-cli.mjs +152 -0
  92. package/lib/wsl.mjs +230 -0
  93. package/lib/zod-issues.mjs +15 -0
  94. package/mcp/server.mjs +165 -34
  95. package/package.json +7 -5
  96. package/playbooks/feature.md +8 -5
  97. package/scripts/configure-openclaw.sh +4 -2
  98. package/scripts/generate-harness-docs.mjs +42 -0
  99. package/scripts/install-openclaw.mjs +23 -0
  100. package/scripts/lib.sh +9 -2
  101. package/scripts/select-model.mjs +12 -5
  102. package/scripts/start-inference.sh +2 -2
@@ -0,0 +1,122 @@
1
+ // One tested release for every nomArmy-managed OpenClaw installation.
2
+ // Never follow npm's dist-tag or downgrade a newer installation.
3
+ import { spawnSync } from "node:child_process";
4
+ import { SUBSCRIPTION_VENDORS, parseOpenclawVersion, versionAtLeast } from "./subscription-setup.mjs";
5
+
6
+ export const PINNED_OPENCLAW_VERSION = "2026.9.6";
7
+ export const OPENCLAW_INSTALL_ARGS = Object.freeze(["install", "-g", `openclaw@${PINNED_OPENCLAW_VERSION}`]);
8
+
9
+ export function runOpenclawCommand(command, args, { timeoutMs = command === "npm" ? 600000 : 60000 } = {}) {
10
+ const r = spawnSync(command, args, { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], timeout: timeoutMs });
11
+ const timedOut = r.error?.code === "ETIMEDOUT";
12
+ return { ok: !timedOut && r.status === 0, stdout: r.stdout ?? "",
13
+ stderr: timedOut ? `Command timed out after ${timeoutMs} ms.` : r.stderr ?? "", timedOut };
14
+ }
15
+
16
+ export function openclawInstallPlan(installed, { prefix = null } = {}) {
17
+ const have = parseOpenclawVersion(installed);
18
+ if (versionAtLeast(have, PINNED_OPENCLAW_VERSION)) return [];
19
+ return [{
20
+ description: `${have ? "Upgrade" : "Install"} OpenClaw to ${PINNED_OPENCLAW_VERSION}`,
21
+ command: "npm",
22
+ args: [...OPENCLAW_INSTALL_ARGS, ...(prefix ? ["--prefix", prefix] : [])],
23
+ }];
24
+ }
25
+
26
+ export function configuredSubscriptionVendors(agents = {}, extra = []) {
27
+ return [...new Set([...extra, ...Object.values(agents)
28
+ .filter((a) => a?.kind === "subscription")
29
+ .map((a) => Object.keys(SUBSCRIPTION_VENDORS).find((key) => SUBSCRIPTION_VENDORS[key].provider === a.provider))])].filter(Boolean);
30
+ }
31
+
32
+ /** Read-only postflight using structured status and plugin compatibility data. */
33
+ function parseCommandJson(stdout) {
34
+ const start = (stdout ?? "").indexOf("{");
35
+ if (start < 0) throw new Error("No JSON object in command output");
36
+ return JSON.parse(stdout.slice(start));
37
+ }
38
+
39
+ export function verifyOpenclaw({ run = runOpenclawCommand, command = "openclaw", vendors = [] } = {}) {
40
+ const checks = [];
41
+ const version = run(command, ["--version"]);
42
+ const have = version.ok ? parseOpenclawVersion(version.stdout) : null;
43
+ checks.push({
44
+ id: "openclaw", ok: versionAtLeast(have, PINNED_OPENCLAW_VERSION),
45
+ message: have ? `OpenClaw ${have.join(".")}${have.join(".") !== PINNED_OPENCLAW_VERSION && versionAtLeast(have, PINNED_OPENCLAW_VERSION) ? " is newer than the tested release; left unchanged" : ""}.` : version.timedOut ? "OpenClaw version check timed out." : "OpenClaw version could not be verified.",
46
+ fix: `npm ${OPENCLAW_INSTALL_ARGS.join(" ")}`,
47
+ });
48
+ if (!have) return checks;
49
+ for (const key of [...new Set(vendors)]) {
50
+ const vendor = SUBSCRIPTION_VENDORS[key];
51
+ if (!vendor) throw new Error(`Unknown subscription vendor: ${key}`);
52
+ if (!vendor.plugin) continue;
53
+ const { id, spec } = vendor.plugin;
54
+ const inspected = run(command, ["plugins", "inspect", id, "--json"]);
55
+ let plugin;
56
+ try { plugin = parseCommandJson(inspected.stdout).plugin; } catch { /* Unverified output fails below. */ }
57
+ const pluginVersion = plugin?.builtWithOpenClawVersion === undefined ? plugin?.version : plugin.builtWithOpenClawVersion;
58
+ const installedVersion = have.join(".");
59
+ const ok = inspected.ok && plugin?.enabled === true && pluginVersion !== undefined && versionAtLeast(have, pluginVersion);
60
+ checks.push({
61
+ id: `openclaw-plugin:${id}`, ok,
62
+ message: ok ? `OpenClaw plugin ${id} ${pluginVersion} is ready (built for ${pluginVersion}; OpenClaw is ${installedVersion}).` : `OpenClaw plugin ${id} ${inspected.timedOut ? "check timed out." : plugin?.enabled === true && pluginVersion === undefined ? "version could not be verified." : `is missing, disabled, unreadable, or built for a newer OpenClaw than ${installedVersion}.`}`,
63
+ fix: `openclaw plugins install ${spec}`,
64
+ });
65
+ }
66
+ const status = run(command, ["update", "status", "--json"]);
67
+ let warnings;
68
+ try {
69
+ if (!status.ok) throw new Error("Status command failed");
70
+ const data = parseCommandJson(status.stdout);
71
+ warnings = data.migrationWarnings === undefined ? [] : data.migrationWarnings;
72
+ if (!Array.isArray(warnings) || warnings.some((warning) => typeof warning !== "string")) throw new Error("Invalid migration warnings");
73
+ } catch {
74
+ checks.push({ id: "openclaw-migrations", ok: false, message: status.timedOut ? "OpenClaw migration check timed out." : "OpenClaw migrations could not be verified. Run openclaw update status --json by hand.", fix: "openclaw update status --json" });
75
+ return checks;
76
+ }
77
+ checks.push({
78
+ id: "openclaw-migrations", ok: warnings.length === 0,
79
+ message: warnings.length ? warnings.join("\n") : "No pending OpenClaw migrations.",
80
+ fix: "openclaw update repair",
81
+ });
82
+ return checks;
83
+ }
84
+
85
+ /** Print the complete mutation plan before a single default-no consent gate. */
86
+ export async function repairOpenclaw({
87
+ run = runOpenclawCommand, command = "openclaw", vendors = [], prefix = null,
88
+ yes = false, isTTY = false, ask = async () => "", print = console.log,
89
+ } = {}) {
90
+ const version = run(command, ["--version"]);
91
+ const actions = openclawInstallPlan(version.ok ? version.stdout : null, { prefix });
92
+ print("Planned changes:");
93
+ for (const action of actions) print(` ${action.description}: ${action.command} ${action.args.join(" ")}`);
94
+ if (!actions.length) print(" None. The installed OpenClaw is not older than the tested release.");
95
+ print("Postflight checks (read-only):");
96
+ print(` ${command} --version`);
97
+ for (const key of [...new Set(vendors)]) {
98
+ const vendor = SUBSCRIPTION_VENDORS[key];
99
+ if (!vendor) throw new Error(`Unknown subscription vendor: ${key}`);
100
+ if (vendor.plugin) print(` ${command} plugins inspect ${vendor.plugin.id} --json`);
101
+ }
102
+ print(` ${command} update status --json`);
103
+ print(" Migration repairs are not run automatically. If needed, run openclaw update repair --yes separately.");
104
+ const changed = [];
105
+ if (actions.length && !yes && (!isTTY || !/^(y|yes)$/i.test((await ask("Apply these changes? [y/N] ")).trim()))) {
106
+ print(isTTY ? "No changes made." : "No changes made. Non-interactive repair requires --yes.");
107
+ return { ok: false, actions, changed, checks: [] };
108
+ }
109
+ for (const action of actions) {
110
+ const result = run(action.command, action.args);
111
+ if (!result.ok) {
112
+ print(`Failed: ${action.description}. Fix: ${action.command} ${action.args.join(" ")}`);
113
+ print(`Completed changes: ${changed.length ? changed.join("; ") : "none"}. The failed install may have partially modified OpenClaw.`);
114
+ return { ok: false, actions, changed, checks: [] };
115
+ }
116
+ changed.push(action.description);
117
+ }
118
+ print(`Changed: ${changed.length ? changed.join("; ") : "nothing"}.`);
119
+ const checks = verifyOpenclaw({ run, command, vendors });
120
+ for (const check of checks) print(`${check.ok ? "OK" : "FAIL"}: ${check.message}${check.ok ? "" : ` Fix: ${check.fix}`}`);
121
+ return { ok: checks.every((check) => check.ok), actions, changed, checks };
122
+ }
@@ -0,0 +1,28 @@
1
+ // Find OpenClaw where install.sh may have put it. With no writable npm global
2
+ // folder (common on Linux), it installs into ~/.npm-global/bin, which isn't on
3
+ // most PATHs: a fresh-install practice run ended "OpenClaw not found", and
4
+ // every job would then fail. The CLI and the MCP server call this first, so
5
+ // they and everything they spawn find it.
6
+
7
+ import fs from "node:fs";
8
+ import os from "node:os";
9
+ import path from "node:path";
10
+
11
+ const isExecutable = (file) => { try { fs.accessSync(file, fs.constants.X_OK); return fs.statSync(file).isFile(); } catch { return false; } };
12
+
13
+ export function openclawFallbackDirs(home = os.homedir()) {
14
+ return [path.join(home, ".npm-global", "bin"), path.join(home, ".local", "bin")];
15
+ }
16
+
17
+ /** Adds the folder holding openclaw to env.PATH when it isn't already reachable. Returns the folder added, or null. */
18
+ export function ensureOpenClawOnPath(env = process.env, { home = os.homedir(), platform = process.platform } = {}) {
19
+ if (env.NOMARMY_OPENCLAW_CMD) return null;
20
+ const sep = platform === "win32" ? ";" : ":";
21
+ const names = platform === "win32" ? ["openclaw.cmd", "openclaw.exe", "openclaw"] : ["openclaw"];
22
+ const dirs = String(env.PATH ?? "").split(sep).filter(Boolean);
23
+ if (dirs.some((d) => names.some((n) => isExecutable(path.join(d, n))))) return null;
24
+ const found = openclawFallbackDirs(home).find((d) => names.some((n) => isExecutable(path.join(d, n))));
25
+ if (!found) return null;
26
+ env.PATH = [found, ...dirs].join(sep);
27
+ return found;
28
+ }
@@ -4,7 +4,7 @@ import path from "node:path";
4
4
  import { resolveExecutable } from "./process.mjs";
5
5
  import { readOpenClawTranscriptTail } from "./transcript.mjs";
6
6
  import { loadConfig } from "./config.mjs";
7
- import { resolveSandboxImage, detectPrimaryLanguage, EXEC_PATH_PREPEND, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
7
+ import { resolveSandboxImage, sandboxPathEntries, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
8
8
  import { DEFAULT_AGENT_IMAGE } from "./verify.mjs";
9
9
  import { scoutPrompt, isScoutReportUnusable } from "./scout.mjs";
10
10
  import { decomposePrompt } from "./decompose.mjs";
@@ -13,6 +13,7 @@ import { deriveBudgets } from "./budget.mjs";
13
13
  import { entryContextPerNom } from "./dispatch-config.mjs";
14
14
  import { readClaudeSessionUsage } from "./claude-transcript.mjs";
15
15
  import { modelRejection, modelRejectionLine } from "./openclaw-errors.mjs";
16
+ import { nearestThinkingLevel } from "./thinking.mjs";
16
17
 
17
18
  const sleep = ms => new Promise(resolve => setTimeout(resolve, ms));
18
19
 
@@ -61,6 +62,35 @@ export function makeAbandonedBackgroundProcessTick(stateDir, { idleMs, minElapse
61
62
  * tool call and files changed so far (liveProgress), and when. Never asks
62
63
  * to stop; a failed read just skips that beat.
63
64
  */
65
+ // local_worker_stop (or `nomarmy jobs --stop`) writes this into the job's
66
+ // folder, so any session can stop any job; the next tick ends the worker.
67
+ export const STOP_REQUEST_FILE = "stop-request.json";
68
+ export function makeStopRequestTick(jobDir) {
69
+ return async () => (fs.existsSync(path.join(jobDir, STOP_REQUEST_FILE)) ? { stop: true, reason: "stopped" } : { stop: false });
70
+ }
71
+ /**
72
+ * Ask a running job to stop. Refuses what isn't running; never kills
73
+ * anything itself (the job's own tick does, within one tick).
74
+ * @returns {{ ok: boolean, message: string }}
75
+ */
76
+ export function requestJobStop({ jobsRoot, jobId, reason = null, now = new Date() }) {
77
+ if (!/^[A-Za-z0-9._-]{1,120}$/.test(String(jobId ?? ""))) return { ok: false, message: "not a job id" };
78
+ const jobDir = path.join(jobsRoot, jobId);
79
+ let status = null;
80
+ try { status = JSON.parse(fs.readFileSync(path.join(jobDir, "status.json"), "utf8")); } catch { return { ok: false, message: `no job ${jobId} (see local_worker_jobs)` }; }
81
+ if (status.state !== "running") return { ok: false, message: `job ${jobId} isn't running (it's ${status.state ?? "unknown"})` };
82
+ if (status.phase && status.phase !== "worker" && status.phase !== "starting" && status.phase !== "worktree") {
83
+ return { ok: false, message: `job ${jobId} is past its worker (phase ${status.phase}); it's finishing on its own and spends no more model usage` };
84
+ }
85
+ if (fs.existsSync(path.join(jobDir, STOP_REQUEST_FILE))) return { ok: true, message: `a stop was already requested for ${jobId}` };
86
+ fs.writeFileSync(path.join(jobDir, STOP_REQUEST_FILE), JSON.stringify({ at: now.toISOString(), reason: reason ? String(reason).slice(0, 300) : null }));
87
+ return { ok: true, message: `stop requested for ${jobId}: its worker ends within about 15 seconds, verification is skipped, and the worktree is kept uncommitted, so continue_from can pick the work up (on another model too)` };
88
+ }
89
+
90
+ export function readStopRequest(jobDir) {
91
+ try { return JSON.parse(fs.readFileSync(path.join(jobDir, STOP_REQUEST_FILE), "utf8")); } catch { return null; }
92
+ }
93
+
64
94
  export function makeHeartbeatTick(jobDir, liveProgress) {
65
95
  // Never two beats at once: a slow beat used to overlap the next.
66
96
  let busy = false;
@@ -128,12 +158,44 @@ export function parseUnsupportedThinkingError(errorMessage) {
128
158
  // worker that ran out of room, but has valid session state worth resuming)
129
159
  // it exists for. Checked against the real captured envelope from that
130
160
  // incident, not a synthesized shape.
131
- export function parseOpenClawInternalTimeout(stdout) {
161
+ export function parseOpenClawInternalTimeout(stdout, stderr = "") {
162
+ // OpenClaw's own timer can end a run in the middle of a tool call. The
163
+ // envelope then reports that call's failure ("Read failed", status
164
+ // "error"), not a timeout; its run log still says so. Seen live on a Grok
165
+ // security review cut off mid-read at 600s: recorded as a crash, so its
166
+ // findings were never recovered.
167
+ if (/embedded run timeout: /.test(String(stderr))) return true;
132
168
  let parsed;
133
169
  try { parsed = JSON.parse(stdout); } catch { return false; }
134
170
  return parsed?.ok === false && (parsed?.status === "timeout" || parsed?.error?.kind === "timeout");
135
171
  }
136
172
 
173
+ /** What a failed run's envelope still says it used, so a failed job's spend is recorded too. */
174
+ export function failedRunUsage(stdout) {
175
+ let parsed;
176
+ try { parsed = JSON.parse(stdout); } catch { return {}; }
177
+ if (!parsed || typeof parsed !== "object") return {};
178
+ const out = {};
179
+ if (parsed.usage && typeof parsed.usage === "object") out.usage = parsed.usage;
180
+ if (Number.isFinite(parsed.costUsd)) out.costUsd = parsed.costUsd;
181
+ if (parsed.toolSummary && typeof parsed.toolSummary === "object") out.toolSummary = parsed.toolSummary;
182
+ if (typeof parsed.sessionId === "string") out.sessionId = parsed.sessionId;
183
+ return out;
184
+ }
185
+
186
+ /**
187
+ * The time a run has, told to the worker. Without it a Grok scout read files
188
+ * for its whole 10 minutes and was cut off before writing a word of its report.
189
+ */
190
+ export function timeBudgetNote(timeoutSeconds, mode) {
191
+ const total = Number(timeoutSeconds);
192
+ if (!Number.isFinite(total) || total <= 0) return "";
193
+ const minutes = Math.max(1, Math.round(total / 60));
194
+ const wrapAt = Math.max(1, Math.floor((total * 0.8) / 60));
195
+ const what = mode === "scout" || mode === "decompose" ? "stop exploring and write your report from what you have" : "stop starting new work, finish verification and write your report";
196
+ return `\n\nTIME\nThis run has about ${minutes} minute(s), then it is cut off. By minute ${wrapAt}, ${what}. A report on part of the question, with what's left under NOT DONE, is far more useful than none.`;
197
+ }
198
+
137
199
  /**
138
200
  * A run whose work finished but whose exit failed: OpenClaw logged the run
139
201
  * ending normally (stopReason=stop) and then errored, e.g. "Codex one-shot
@@ -337,7 +399,7 @@ export function createOpenClawRunner(deps) {
337
399
  // which is the one per-call lever that does reach the sandbox OpenClaw
338
400
  // starts for that run. This clones the ambient config, points
339
401
  // agents.defaults.sandbox.docker.image at the resolved image, and adds
340
- // EXEC_PATH_PREPEND's extra PATH entries -- verified live: OpenClaw's exec
402
+ // the matched harnesses' extra PATH entries -- verified live: OpenClaw's exec
341
403
  // tool does not inherit a sandbox image's own baked ENV PATH on its own (a
342
404
  // freshly built Go image's `go` resolved fine under a direct `podman exec`
343
405
  // but came back "not found" through `openclaw agent exec` until
@@ -360,7 +422,7 @@ export function createOpenClawRunner(deps) {
360
422
  // contract verification does.
361
423
  loadConfigFn = () => loadConfig(projectDir),
362
424
  resolveSandboxImageFn = resolveSandboxImage,
363
- detectPrimaryLanguageFn = detectPrimaryLanguage,
425
+ sandboxPathEntriesFn = sandboxPathEntries,
364
426
  ambientConfigPathFn = ambientOpenClawConfigPath,
365
427
  readAmbientConfig = (p) => JSON.parse(fs.readFileSync(p, "utf8")),
366
428
  } = {}) {
@@ -372,7 +434,7 @@ export function createOpenClawRunner(deps) {
372
434
 
373
435
  let image;
374
436
  try {
375
- image = resolveSandboxImageFn({ cwd, explicitImage: process.env.NOMARMY_AGENT_IMAGE || null, defaultImage: DEFAULT_AGENT_IMAGE, config });
437
+ image = resolveSandboxImageFn({ cwd, trustedDir: projectDir, explicitImage: process.env.NOMARMY_AGENT_IMAGE || null, defaultImage: DEFAULT_AGENT_IMAGE, config });
376
438
  } catch {
377
439
  // A lazy Go/Rust/Python image build failure here should not fail the
378
440
  // worker's turn -- it runs in the default image instead, same as before
@@ -399,8 +461,7 @@ export function createOpenClawRunner(deps) {
399
461
  // own verification runs: outside the worktree, and off.
400
462
  overridden.agents.defaults.sandbox.docker.env = { ...(overridden.agents.defaults.sandbox.docker.env ?? {}), ...SANDBOX_NPM_ENV };
401
463
 
402
- const lang = detectPrimaryLanguageFn(cwd, config);
403
- const pathPrepend = EXEC_PATH_PREPEND[lang] || [];
464
+ const pathPrepend = sandboxPathEntriesFn(cwd, config);
404
465
  if (pathPrepend.length) {
405
466
  overridden.tools ??= {};
406
467
  overridden.tools.exec ??= {};
@@ -443,7 +504,7 @@ export function createOpenClawRunner(deps) {
443
504
  ? scoutPrompt({ question: task, mustCover: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.scout, report: jobBudgets.report.scout, evidenceTool })
444
505
  : mode === "decompose"
445
506
  ? decomposePrompt({ objective: task, constraints: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.decompose, report: jobBudgets.report.decompose, evidenceTool })
446
- : workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement }));
507
+ : workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement })) + (overridePrompt ? "" : timeBudgetNote(timeoutSeconds, mode));
447
508
  fs.writeFileSync(path.join(jobDir, `brief${logSuffix}.txt`), prompt + "\n");
448
509
  // --state-dir keeps OpenClaw's session state (its transcript database among
449
510
  // it) inside the job directory instead of a temp dir it deletes on exit.
@@ -476,8 +537,9 @@ export function createOpenClawRunner(deps) {
476
537
  // The heartbeat runs for every job, so a running job is always watchable
477
538
  // (status.json used to keep its launch-time updatedAt until the end).
478
539
  const onTick = combineTicks([
540
+ makeStopRequestTick(jobDir),
479
541
  makeHeartbeatTick(jobDir),
480
- ...(idleDiff ? [makeIdleDiffTick(cwd, idleDiff), makeAbandonedBackgroundProcessTick(stateDir, idleDiff)] : []),
542
+ ...(idleDiff ? [makeIdleDiffTick(cwd, { ...idleDiff, activity: async () => { const t = await readOpenClawTranscriptTail(stateDir, { limit: 5 }); return t.available ? { events: t.events, toolInFlight: t.toolInFlight } : null; } }), makeAbandonedBackgroundProcessTick(stateDir, idleDiff)] : []),
481
543
  ]);
482
544
  const liveLogs = { stdout: path.join(jobDir, `openclaw${logSuffix}.stdout.log`), stderr: path.join(jobDir, `openclaw${logSuffix}.stderr.log`) };
483
545
  // Where this call's own transcript events will start (see salvageFinishedRun).
@@ -519,7 +581,7 @@ export function createOpenClawRunner(deps) {
519
581
  // -- never loop on the same level, and never mask a real, unrelated
520
582
  // failure as a thinking-level problem it isn't.
521
583
  if (unsupported && !unsupported.supported.includes(selected.thinking)) {
522
- const fallback = unsupported.supported[0];
584
+ const fallback = nearestThinkingLevel(selected.thinking, unsupported.supported);
523
585
  fs.appendFileSync(path.join(jobDir, "coordinator.log"),
524
586
  `${new Date().toISOString()} "${selected.model}" rejected thinking level "${selected.thinking}" (OpenClaw supports: ${unsupported.supported.join(", ")}) -- retrying once with "${fallback}"\n`);
525
587
  selected.thinking = fallback; // the manifest's requestedReasoning field should reflect what was ACTUALLY used, not the level that failed
@@ -557,7 +619,7 @@ export function createOpenClawRunner(deps) {
557
619
  // a graceful internal timeout, not an opaque crash -- relabel it so
558
620
  // executeImplement/executeScout's workerTimedOut check (and therefore
559
621
  // report recovery) sees it correctly.
560
- if (!error.timedOut && parseOpenClawInternalTimeout(error.stdout)) {
622
+ if (!error.timedOut && error.stopReason !== "stopped" && parseOpenClawInternalTimeout(error.stdout, error.stderr)) {
561
623
  error.timedOut = true;
562
624
  error.stopReason = error.stopReason ?? "openclaw_internal_timeout";
563
625
  }
@@ -582,7 +644,7 @@ export function createOpenClawRunner(deps) {
582
644
  }
583
645
  // What was attempted, for the job record: a failed job used to be
584
646
  // labeled with the local default model, whatever it really ran on.
585
- error.partialResult = { model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
647
+ error.partialResult = { ...failedRunUsage(error.stdout), model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
586
648
  throw error;
587
649
  } finally {
588
650
  // A cloned copy of the ambient OpenClaw config (which may carry a real
@@ -0,0 +1,56 @@
1
+ // Metadata-only checks for the runtime used by nomArmy's real test calls.
2
+ import { agentProviderId } from "./agents.mjs";
3
+ export const CODEX_IMPORT_RECOVERY = "openclaw migrate apply codex --from ~/.codex --agent main --include-secrets --item auth:openai --yes";
4
+ export function codexImportRecovery({ profiles = [] } = {}) {
5
+ return [...emailOpenaiProfiles(profiles).map((p) => `openclaw models auth logout ${shellId(p.id)}`), CODEX_IMPORT_RECOVERY].join(" && ");
6
+ }
7
+
8
+ export const CODEX_IMPORT_ARGS = ["migrate", "apply", "codex", "--from", "~/.codex", "--agent", "main", "--include-secrets", "--item", "auth:openai", "--yes"];
9
+ export function emailOpenaiProfiles(profiles) {
10
+ return profiles.filter((p) => /^openai:.*@/.test(p.id ?? ""));
11
+ }
12
+ export function shellId(id) {
13
+ return /^[\w@.+:-]+$/.test(id) ? id : "'" + id.replaceAll("'", "'\\''") + "'";
14
+ }
15
+ export function usableCodexImport(profiles, now = Date.now()) {
16
+ return profiles.some((p) => p.provider === "openai" && /^openai:account-/.test(p.id ?? "") &&
17
+ [p.label, p.name, p.displayName].some((s) => typeof s === "string" && s.includes("(Codex import)")) &&
18
+ (p.expiresAt == null || Date.parse(p.expiresAt) > now));
19
+ }
20
+ export function codexImportIssues(profiles, { now = Date.now() } = {}) {
21
+ const issues = [];
22
+ const emails = emailOpenaiProfiles(profiles);
23
+ const fix = codexImportRecovery({ profiles });
24
+ if (!usableCodexImport(profiles, now)) issues.push({
25
+ id: "codex-import:missing", severity: "error", title: "Codex has no unexpired account-ID (Codex import) profile",
26
+ detail: "The Codex app-server cannot use an email-keyed OpenAI login. Confirm Codex CLI login, then re-import it.",
27
+ fix, short: "Codex import missing",
28
+ });
29
+ if (emails.length) issues.push({
30
+ id: "codex-import:shadowed", severity: "error", title: "Email-keyed OpenAI profiles capture the Codex import",
31
+ detail: `Remove before importing: ${emails.map((p) => p.id).join(", ")}. A models status probe is not an app-server login test.`,
32
+ fix, short: "Codex import shadowed",
33
+ });
34
+ return issues;
35
+ }
36
+ export function modelPolicyIssues({ policies, paths, agents = {}, armySummary = null, modelsInUse = null }) {
37
+ const models = new Set(modelsInUse ?? []);
38
+ for (const a of Object.values(agents ?? {})) {
39
+ const provider = agentProviderId(a);
40
+ if (provider && a.model && a.model !== "auto") models.add(`${provider}/${a.model}`);
41
+ }
42
+ for (const role of Object.values(armySummary?.roles ?? {})) {
43
+ const provider = agentProviderId(agents?.[role.agent]);
44
+ if (provider && role.model && role.model !== "auto" && !role.modelIsAuto) models.add(`${provider}/${role.model}`);
45
+ }
46
+ return policies.flatMap((result, i) => {
47
+ let allow;
48
+ try { allow = result.ok ? JSON.parse(result.stdout) : null; } catch { return []; }
49
+ if (!Array.isArray(allow)) return [];
50
+ const missing = [...models].filter((m) => !allow.includes(m)).sort();
51
+ if (!missing.length) return [];
52
+ return [{ id: `model-policy:${paths[i]}`, severity: "error", title: `OpenClaw model policy excludes: ${missing.join(", ")}`,
53
+ detail: `${paths[i]} refuses models used by nomArmy agents or roles.`,
54
+ fix: `openclaw config unset ${paths[i].replace(/\.allow$/, "")} (or add the missing models to ${paths[i]})`, short: "models blocked" }];
55
+ });
56
+ }
package/lib/outcome.mjs CHANGED
@@ -7,6 +7,16 @@ import { parseWorkerReport } from "./report.mjs";
7
7
  // failed. Recovery exists so that a mangled REPORT cannot destroy correct WORK.
8
8
  // It does not exist to launder a failure into a success.
9
9
  // ---------------------------------------------------------------------------
10
+ // A failed verification's own detail (the command, its exit code, the end of
11
+ // its output), capped so an issue line stays readable; the full output is in
12
+ // the job's verification.log.
13
+ const FAILURE_DETAIL_CHARS = 600;
14
+ function failureDetail(independentVerification) {
15
+ const detail = String(independentVerification?.detail ?? independentVerification?.reason ?? "").trim();
16
+ if (!detail) return "";
17
+ return `: ${detail.length > FAILURE_DETAIL_CHARS ? `${detail.slice(0, FAILURE_DETAIL_CHARS)}...` : detail}`;
18
+ }
19
+
10
20
  export function resolveOutcome({ report, repositoryChanged = false, independentVerification = null, regressionCheck = null, workerFailed = false, workerTimedOut = false, mode = "implement" }) {
11
21
  const verification = independentVerification?.status ?? "not_run";
12
22
  const parsed = report ?? parseWorkerReport("");
@@ -34,7 +44,7 @@ export function resolveOutcome({ report, repositoryChanged = false, independentV
34
44
  if (verification === "fail") {
35
45
  return { ...base, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true,
36
46
  commitBlockedReason: "independent verification failed despite a clean done/pass report",
37
- reasons: ["worker claimed done/pass but independent verification failed"] };
47
+ reasons: [`worker claimed done/pass but independent verification failed${failureDetail(independentVerification)}`] };
38
48
  }
39
49
  // verify_regression: reverting just the production files and re-running
40
50
  // the SAME verification profile still passed (or came back genuinely
@@ -92,7 +102,7 @@ export function resolveOutcome({ report, repositoryChanged = false, independentV
92
102
  if (verification === "fail") {
93
103
  return { ...recovery, outcome: OUTCOMES.WORKER_REPORT_INVALID,
94
104
  commitBlockedReason: "independent verification failed; recovery cannot promote a failure",
95
- reasons: [...recovery.reasons, "independent verification FAILED"] };
105
+ reasons: [...recovery.reasons, `independent verification FAILED${failureDetail(independentVerification)}`] };
96
106
  }
97
107
  if (verification === "pass") {
98
108
  // A leniently recovered `done` plus a passing independent check is the
@@ -197,8 +207,17 @@ export function policyAdmissionProblems(job, policy) {
197
207
  // A refactor meets the regression requirement through its own contract
198
208
  // (applyRefactorContract), not the revert check.
199
209
  if (policy.require_regression_check && job.verify_regression === false && !job.refactor) problems.push("this repo requires the revert check (policy.require_regression_check in .nomarmy.yml): verify_regression can't be false (a behavior-preserving change can declare refactor: true instead)");
210
+ // stakes: high (security, data loss, irreversible): the checks a General
211
+ // could skip on routine work are mandatory, whatever the repo's policy.
212
+ if (job.stakes === "high") {
213
+ if (!job.verification && !job.refactor) problems.push("a stakes: high job needs a `verification` profile: its result is only as good as the tests that prove it");
214
+ if (job.verify_regression === false && !job.refactor) problems.push("a stakes: high job can't turn the revert check off (verify_regression: false): it's the proof a test catches the change");
215
+ }
200
216
  return problems;
201
217
  }
218
+
219
+ /** The review line every high-stakes job carries, whatever its outcome. */
220
+ export const HIGH_STAKES_NOTE = "HIGH STAKES: accept this only after an independent review: a scout on another vendor (army_role security-analyst or similar) with reviews: <this job id>, or a judge on another vendor. Passing its tests isn't enough on its own; most defects that pass every check are in security, data or deploy-only paths.";
202
221
  /**
203
222
  * A declared refactor commits only when verification passed and no test
204
223
  * file was added, changed or deleted. Mechanical, not the General's call:
package/lib/outcomes.mjs CHANGED
@@ -2,6 +2,9 @@ import { SCOUT_OUTCOMES, SCOUT_STATUS_BY_OUTCOME } from "./scout.mjs";
2
2
  import { DECOMPOSE_STATUS_BY_OUTCOME } from "./decompose.mjs";
3
3
 
4
4
  export const OUTCOMES = Object.freeze({
5
+ VERIFIED: "VERIFIED",
6
+ VERIFICATION_FAILED: "VERIFICATION_FAILED",
7
+ VERIFICATION_NOT_RUN: "VERIFICATION_NOT_RUN",
5
8
  WORKER_DONE: "WORKER_DONE",
6
9
  WORKER_PARTIAL: "WORKER_PARTIAL",
7
10
  WORKER_BLOCKED: "WORKER_BLOCKED",
@@ -14,6 +17,9 @@ export const OUTCOMES = Object.freeze({
14
17
  });
15
18
 
16
19
  export const COORDINATOR_STATUS_BY_OUTCOME = Object.freeze({
20
+ [OUTCOMES.VERIFIED]: "complete",
21
+ [OUTCOMES.VERIFICATION_FAILED]: "failed",
22
+ [OUTCOMES.VERIFICATION_NOT_RUN]: "incomplete",
17
23
  [OUTCOMES.WORKER_DONE]: "complete",
18
24
  [OUTCOMES.RECOVERED_SUCCESS]: "complete",
19
25
  [OUTCOMES.WORKER_BLOCKED]: "blocked",
@@ -0,0 +1,4 @@
1
+ /** Docker, Git, and repository metadata paths are POSIX paths, even on Windows. */
2
+ export function toPosix(value) {
3
+ return value.replaceAll("\\", "/");
4
+ }
@@ -0,0 +1,41 @@
1
+ // Is Podman answering, and when did its VM last start? A job dispatched into
2
+ // a stopped Podman, or running while its VM restarts (resizing it with
3
+ // `nomarmy sandbox --memory`, say), fails with nothing but "openclaw exited
4
+ // 2" unless nomArmy checks. Seen live: two jobs lost to a VM resize.
5
+
6
+ import { spawnSync } from "node:child_process";
7
+
8
+ // A loaded machine can take several seconds to answer; 15 seconds keeps a
9
+ // slow Podman from being reported as a stopped one (5 wasn't enough under a
10
+ // load average in the 40s to 80s).
11
+ export const PODMAN_ANSWER_TIMEOUT_MS = 15000;
12
+ const defaultRun = (args) => spawnSync("podman", args, { encoding: "utf8", timeout: PODMAN_ANSWER_TIMEOUT_MS });
13
+
14
+ /** Why jobs can't start their sandbox right now, or null. */
15
+ export function podmanProblem({ run = defaultRun } = {}) {
16
+ const r = run(["info", "--format", "{{.Host.Arch}}"]);
17
+ if (r.status === 0) return null;
18
+ const why = String(r.stderr || r.error?.message || "").split("\n").find((l) => l.trim()) ?? "no answer";
19
+ if (r.error?.code === "ETIMEDOUT") {
20
+ return `Podman didn't answer within ${PODMAN_ANSWER_TIMEOUT_MS / 1000} seconds, so no job was sent. It may be running but starved: check the machine's load (Docker, a local model, other jobs) and retry, or run \`nomarmy sandbox\`. Don't restart Podman while jobs are running; that kills their sandboxes.`;
21
+ }
22
+ return `Podman isn't answering (${why.trim().slice(0, 160)}), so no job was sent: its sandbox couldn't start. Start it with \`podman machine start\` (macOS, Windows), then check with \`nomarmy sandbox\`.`;
23
+ }
24
+
25
+ /** When the Podman VM last started (macOS, Windows), or null where there's no VM. */
26
+ export function podmanVmStartedAt({ run = defaultRun, platform = process.platform } = {}) {
27
+ if (!["darwin", "win32"].includes(platform)) return null;
28
+ try {
29
+ const machines = JSON.parse(run(["machine", "inspect"]).stdout || "[]");
30
+ const m = machines.find((x) => x.State === "running") ?? machines[0];
31
+ return m?.LastUp ?? null;
32
+ } catch { return null; }
33
+ }
34
+
35
+ /** A job issue when the VM restarted, or stopped, between two readings. */
36
+ export function vmRestartIssue(before, after) {
37
+ if (!before || before === after) return null;
38
+ return after
39
+ ? "the Podman VM restarted during this job, which removed its sandbox: the failure is most likely that, not the work. Re-dispatch once Podman is up (nomarmy sandbox)."
40
+ : "the Podman VM stopped during this job, which removed its sandbox: the failure is most likely that, not the work. Start it (podman machine start) and re-dispatch.";
41
+ }
package/lib/process.mjs CHANGED
@@ -73,7 +73,10 @@ export function createProcess(ctx) {
73
73
  if (ticker) clearInterval(ticker);
74
74
  child.kill("SIGTERM");
75
75
  const error = new Error(message);
76
- error.timedOut = true;
76
+ // A stop someone asked for (local_worker_stop) isn't a timeout: a
77
+ // timeout invites a report-recovery call, which would spend exactly
78
+ // the usage the stop was meant to save.
79
+ error.timedOut = stopReason !== "stopped";
77
80
  if (stopReason) error.stopReason = stopReason;
78
81
  reject(error);
79
82
  };
package/lib/propose.mjs CHANGED
@@ -50,21 +50,17 @@ export function buildConfigProposal(evidence) {
50
50
  }
51
51
 
52
52
  // --- environment.python.requirements -------------------------------------
53
- // lib/sandbox-images.mjs's ensurePythonImageBuilt only ever `pip install
54
- // -r <file>`s whatever this proposes -- it cannot install from
55
- // pyproject.toml/poetry/uv directly, so this stays scoped to the same
56
- // requirements.txt-shaped evidence handleRequirements records (tooling
57
- // name "pip"), never pyproject.toml evidence alone; proposing something
58
- // the sandbox builder can't actually act on would be worse than proposing
59
- // nothing. A repo whose real Python dependency source is poetry/uv with no
60
- // requirements.txt anywhere gets no proposal here -- a distinct, real gap
61
- // (the sandbox has no poetry/uv install path at all yet), not this one.
53
+ // Root project manifests are installed automatically by the Python harness.
54
+ // Do not override that manager selection with incidental requirements files.
55
+ const projectSources = new Set(["pyproject.toml", "uv.lock", "poetry.lock"]);
56
+ const pythonProject = (evidence?.files?.items ?? []).some((f) => projectSources.has(f.path) && !fixturePaths.has(f.path)) ||
57
+ (evidence?.tooling?.items ?? []).some((t) => projectSources.has(t.source) && !fixturePaths.has(t.source));
62
58
  const pipTooling = (evidence?.tooling?.items ?? []).filter((t) => t.name === "pip" && t.category === "package-manager" && t.source);
63
59
  for (const t of pipTooling) if (fixturePaths.has(t.source)) excludedFixturePaths.add(t.source);
64
60
  const requirementsFiles = [...new Set(pipTooling.filter((t) => !fixturePaths.has(t.source)).map((t) => t.source))].sort();
65
- if (requirementsFiles.length === 1) {
61
+ if (!pythonProject && requirementsFiles.length === 1) {
66
62
  proposal.environment = { ...proposal.environment, python: { requirements: requirementsFiles } };
67
- } else if (requirementsFiles.length > 1) {
63
+ } else if (!pythonProject && requirementsFiles.length > 1) {
68
64
  // The exact ambiguity this file's own header already treats as a human
69
65
  // judgment call for compose files: several requirements.txt-shaped files
70
66
  // (app, dev, a sub-package's own) with no single conventional
@@ -88,6 +84,9 @@ export function buildConfigProposal(evidence) {
88
84
  if (firstCommand) {
89
85
  notes.push(`verification.quick.commands was seeded from a command found in evidence (${firstCommand}) -- confirm this is actually the right check before trusting it.`);
90
86
  proposal.verification = { quick: { environment: "none", commands: [firstCommand] } };
87
+ } else if (pythonProject) {
88
+ notes.push("Python project detected: pytest is a proposed check using the sandbox venv; confirm the project declares pytest and this is the right test command.");
89
+ proposal.verification = { quick: { environment: "none", commands: ["pytest"] } };
91
90
  } else {
92
91
  notes.push("no command was found anywhere in evidence; verification.quick.commands is a fail-loud placeholder -- replace it before this profile is usable.");
93
92
  proposal.verification = { quick: { environment: "none", commands: [PLACEHOLDER_COMMAND] } };
@@ -0,0 +1,16 @@
1
+ import { claimModelRefusalRetry, modelRefusals, recordModelRefusal, recordProbeSuccess } from "./health.mjs";
2
+
3
+ /** Start eligible one-token probes without waiting for them in the caller. */
4
+ export function retryRefusedModelsInBackground(stateRoot, { probeModel, now = Date.now() } = {}) {
5
+ if (!probeModel) return;
6
+ for (const key of Object.keys(modelRefusals(stateRoot))) {
7
+ if (!claimModelRefusalRetry(stateRoot, key, { now })) continue;
8
+ const slash = key.indexOf("/");
9
+ if (slash < 1 || slash === key.length - 1) continue;
10
+ const provider = key.slice(0, slash), model = key.slice(slash + 1);
11
+ Promise.resolve().then(() => probeModel({ provider, model, stateRoot })).then((result) => {
12
+ if (result?.ok) recordProbeSuccess(stateRoot, key, { now });
13
+ else if (result?.refused) recordModelRefusal(stateRoot, key, result.reason, { now });
14
+ }).catch(() => { /* An inconclusive probe keeps the refusal and retry timestamp. */ });
15
+ }
16
+ }