nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +86 -480
  2. package/bin/nomarmy.mjs +1081 -185
  3. package/docker/Dockerfile +2 -2
  4. package/docker/Dockerfile.go +6 -4
  5. package/docker/Dockerfile.rust +17 -2
  6. package/harnesses/_template/README.md +27 -0
  7. package/harnesses/_template/harness.yml +26 -0
  8. package/harnesses/browser-playwright/README.md +35 -0
  9. package/harnesses/browser-playwright/fixture/package.json +1 -0
  10. package/harnesses/browser-playwright/fixture/page.html +1 -0
  11. package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
  12. package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
  13. package/harnesses/browser-playwright/harness.yml +18 -0
  14. package/harnesses/go/README.md +45 -0
  15. package/harnesses/go/harness.yml +14 -0
  16. package/harnesses/mock-oidc/README.md +31 -0
  17. package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
  18. package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
  19. package/harnesses/mock-oidc/harness.yml +19 -0
  20. package/harnesses/node/README.md +53 -0
  21. package/harnesses/node/harness.yml +18 -0
  22. package/harnesses/python/README.md +46 -0
  23. package/harnesses/python/harness.yml +16 -0
  24. package/harnesses/rust/README.md +45 -0
  25. package/harnesses/rust/harness.yml +13 -0
  26. package/install.sh +29 -9
  27. package/lib/admission.mjs +178 -30
  28. package/lib/agents.mjs +8 -6
  29. package/lib/army.mjs +25 -10
  30. package/lib/codex-link.mjs +37 -0
  31. package/lib/config.mjs +15 -0
  32. package/lib/connect.mjs +232 -19
  33. package/lib/continue-from.mjs +103 -0
  34. package/lib/coordinator-instructions.mjs +5 -1
  35. package/lib/diff-checks.mjs +114 -0
  36. package/lib/dispatch-schema.mjs +14 -12
  37. package/lib/doctor.mjs +98 -9
  38. package/lib/egress-proxy.mjs +116 -0
  39. package/lib/execute.mjs +241 -33
  40. package/lib/git-record.mjs +27 -3
  41. package/lib/harness-schema.mjs +61 -0
  42. package/lib/harnesses.mjs +99 -0
  43. package/lib/health.mjs +162 -18
  44. package/lib/install-freshness.mjs +114 -0
  45. package/lib/jev-checks.mjs +110 -0
  46. package/lib/job-format.mjs +54 -0
  47. package/lib/judge.mjs +130 -0
  48. package/lib/limits.mjs +77 -0
  49. package/lib/model-probe.mjs +61 -0
  50. package/lib/mutation.mjs +159 -0
  51. package/lib/notify.mjs +30 -3
  52. package/lib/openclaw-install.mjs +122 -0
  53. package/lib/openclaw-path.mjs +28 -0
  54. package/lib/openclaw-run.mjs +74 -12
  55. package/lib/openclaw-runtime-health.mjs +56 -0
  56. package/lib/outcome.mjs +21 -2
  57. package/lib/outcomes.mjs +6 -0
  58. package/lib/path-utils.mjs +4 -0
  59. package/lib/podman-health.mjs +41 -0
  60. package/lib/process.mjs +4 -1
  61. package/lib/propose.mjs +10 -11
  62. package/lib/refusal-retry.mjs +16 -0
  63. package/lib/registry-python.mjs +98 -0
  64. package/lib/registry-secrets.mjs +140 -0
  65. package/lib/repo-query.mjs +13 -7
  66. package/lib/runs.mjs +7 -1
  67. package/lib/same-path.mjs +14 -0
  68. package/lib/sandbox-images.mjs +499 -83
  69. package/lib/sandbox-vm.mjs +32 -0
  70. package/lib/scan.mjs +5 -1
  71. package/lib/schema.mjs +20 -11
  72. package/lib/scout.mjs +21 -3
  73. package/lib/server-context.mjs +21 -1
  74. package/lib/setup-steps.mjs +55 -0
  75. package/lib/share.mjs +82 -0
  76. package/lib/stale-sessions.mjs +60 -0
  77. package/lib/stats.mjs +315 -0
  78. package/lib/statusline.mjs +32 -6
  79. package/lib/subscription-setup.mjs +13 -0
  80. package/lib/suggestions.mjs +153 -0
  81. package/lib/thinking.mjs +23 -0
  82. package/lib/transcript.mjs +30 -5
  83. package/lib/usage-limits.mjs +329 -0
  84. package/lib/user-config.mjs +106 -0
  85. package/lib/validators.mjs +220 -0
  86. package/lib/verification-artifacts.mjs +46 -0
  87. package/lib/verification-flow.mjs +52 -7
  88. package/lib/verification-network.mjs +66 -0
  89. package/lib/verify.mjs +338 -85
  90. package/lib/worker-prompt.mjs +5 -2
  91. package/lib/wsl-cli.mjs +152 -0
  92. package/lib/wsl.mjs +230 -0
  93. package/lib/zod-issues.mjs +15 -0
  94. package/mcp/server.mjs +165 -34
  95. package/package.json +7 -5
  96. package/playbooks/feature.md +8 -5
  97. package/scripts/configure-openclaw.sh +4 -2
  98. package/scripts/generate-harness-docs.mjs +42 -0
  99. package/scripts/install-openclaw.mjs +23 -0
  100. package/scripts/lib.sh +9 -2
  101. package/scripts/select-model.mjs +12 -5
  102. package/scripts/start-inference.sh +2 -2
package/mcp/server.mjs CHANGED
@@ -32,8 +32,37 @@ import { liveLeases } from "../lib/slots.mjs";
32
32
  import { createRun, loadRun, runTotals, finishRun, resolveRunLimits, describeLoweredLimits } from "../lib/runs.mjs";
33
33
  import { agentDispatchFields, resolveAgentModel, agentProviderId, describeAgent } from "../lib/agents.mjs";
34
34
  import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "../lib/outcomes.mjs";
35
+ import { readUsageSnapshots, usageStatus, usageDisplayText, refreshStaleOverLimitReadings } from "../lib/usage-limits.mjs";
36
+ import { modelRefusals } from "../lib/health.mjs";
37
+ import { retryRefusedModelsInBackground } from "../lib/refusal-retry.mjs";
38
+ import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
39
+ import { restartNotice } from "../lib/install-freshness.mjs";
40
+ import { requestJobStop } from "../lib/openclaw-run.mjs";
41
+ import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
42
+ import { shareMarkdown } from "../lib/share.mjs";
43
+ import { recentSuggestions } from "../lib/suggestions.mjs";
44
+ import { probeModel } from "../lib/model-probe.mjs";
45
+ import { jevSettings, judgeSettings } from "../lib/validators.mjs";
46
+ import { agentRunsToolsOnHost } from "../lib/dispatch-schema.mjs";
35
47
  import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
36
- import { compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
48
+ import { jobLabel, compactJobRecord, formatResult, reportView, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
49
+ import { ensureOpenClawOnPath } from "../lib/openclaw-path.mjs";
50
+ import { THINKING_LEVELS } from "../lib/thinking.mjs";
51
+ import { samePath } from "../lib/same-path.mjs";
52
+ import { withWindowsPaths, dropWindowsPath } from "../lib/wsl.mjs";
53
+ export { withWindowsPaths };
54
+
55
+ function coordinatorJson(value) { return JSON.stringify(withWindowsPaths(value), null, 2); }
56
+ function coordinatorResult(result) {
57
+ const paths = withWindowsPaths({ jobDir: result.jobDir, worktree: result.manifest?.worktree });
58
+ const windows = ["jobDirWindows", "worktreeWindows"].filter(key => paths[key]).map(key => `${key}: ${paths[key]}`);
59
+ return formatResult(withWindowsPaths(result)) + (windows.length ? `\n\n${windows.join("\n")}` : "");
60
+ }
61
+
62
+ // Inside WSL, only the distro's own tools (see lib/wsl.mjs).
63
+ dropWindowsPath();
64
+ // OpenClaw in ~/.npm-global/bin (no writable npm prefix) is found without the operator editing PATH.
65
+ ensureOpenClawOnPath();
37
66
 
38
67
  export { run, mapLimit };
39
68
  export { readsMeasurable, measureReads };
@@ -54,6 +83,7 @@ export { TEST_PATH_PATTERNS, isTestPath, testPatternFor, classifyTestChanges, de
54
83
  // package.json to the same relative location next to the installed
55
84
  // mcp/server.mjs, so this resolves identically in a dev checkout or an
56
85
  // installed copy.
86
+ const SERVER_STARTED_MS = Date.now();
57
87
  const VERSION = JSON.parse(fs.readFileSync(path.join(path.dirname(fileURLToPath(import.meta.url)), "..", "package.json"), "utf8")).version;
58
88
  // Sent to every coordinator on connect, so no project needs a copied CLAUDE.md.
59
89
  const server = new McpServer({ name: "nomarmy-local-worker", version: VERSION }, { instructions: COORDINATOR_INSTRUCTIONS });
@@ -77,7 +107,8 @@ function slug(prefix = "local") {
77
107
  }
78
108
  async function assertRepo() {
79
109
  const root = await git(["rev-parse", "--show-toplevel"]);
80
- if (path.resolve(root) !== projectDir) throw new Error(`CLAUDE_PROJECT_DIR must be the Git root. Expected ${root}, got ${projectDir}`);
110
+ // Git may print a long, forward-slashed path while Windows supplies an 8.3 path.
111
+ if (!samePath(root, projectDir)) throw new Error(`CLAUDE_PROJECT_DIR must be the Git root. Expected ${root}, got ${projectDir}`);
81
112
  }
82
113
  async function resolveBase(baseRef) {
83
114
  const ref = baseRef || "HEAD";
@@ -208,6 +239,17 @@ export function makeHeartbeatTick(jobDir) { return heartbeatTick(jobDir, livePro
208
239
  // Senti run none were tagged, so a 4-hour run went 8.46 hours unchecked.
209
240
  let activeRunId = null;
210
241
 
242
+ export const DEFAULT_TIMEOUT_SECONDS = 600;
243
+ export const REVIEW_SCOUT_TIMEOUT_SECONDS = 1200;
244
+ /** A review scout (reviews set, or a review-phase role) gets longer: reviews trace across the codebase. */
245
+ export function defaultTimeoutSeconds(job, getArmyFn) {
246
+ if (job.mode !== "scout") return DEFAULT_TIMEOUT_SECONDS;
247
+ if (job.reviews) return REVIEW_SCOUT_TIMEOUT_SECONDS;
248
+ if (!job.army_role) return DEFAULT_TIMEOUT_SECONDS;
249
+ try { return getArmyFn()?.roles?.[job.army_role]?.phase === "review" ? REVIEW_SCOUT_TIMEOUT_SECONDS : DEFAULT_TIMEOUT_SECONDS; }
250
+ catch { return DEFAULT_TIMEOUT_SECONDS; }
251
+ }
252
+
211
253
  export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agentsConfig().agents, getActiveRun = () => activeRunId, env = process.env } = {}) {
212
254
  const problems = [];
213
255
  let army = null, agents = null;
@@ -215,6 +257,11 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
215
257
  const expanded = jobs.map((job, i) => {
216
258
  try {
217
259
  let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
260
+ if (j.timeout_seconds == null) j = { ...j, timeout_seconds: defaultTimeoutSeconds(j, () => (army ??= getArmy().army)) };
261
+ if (j.mode === "verify") {
262
+ const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
263
+ return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
264
+ }
218
265
  if (j.army_role) { army ??= getArmy().army; j = expandArmyRole(j, army); }
219
266
  const { agent, roleModel = null, ...rest } = j;
220
267
  if (!agent) {
@@ -228,6 +275,7 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
228
275
  const out = { ...rest, ...fields, agentName: agent };
229
276
  if (model) out.model = model; else delete out.model;
230
277
  if (!fields.subscription_worker) delete out.on_behalf_of;
278
+ if (out.mode === "scout" && out.report == null && (fields.pool || fields.subscription_worker)) out.report = "full";
231
279
  out.profile ??= "coder";
232
280
  return out;
233
281
  } catch (error) {
@@ -244,7 +292,12 @@ const verificationFlow = createVerificationFlow({
244
292
  const { registerVerificationRunner, normalizeVerification, runIndependentVerification, runRegressionCheck, selectUnionCandidates, buildUnionBranch } = verificationFlow;
245
293
  export { registerVerificationRunner, normalizeVerification, runRegressionCheck, selectUnionCandidates, buildUnionBranch };
246
294
 
295
+ // Podman checks, wired at startup below (a test importing this module gets none).
296
+ let podmanChecks = null;
247
297
  const { executeJob, executeImplement, executeScout, executeDecompose } = createExecutor({
298
+ podmanVmStartedAt: () => podmanChecks?.vmStartedAt() ?? null,
299
+ jevSettings: () => jevSettings(),
300
+ judgeSettings: () => judgeSettings({ agents: agentsConfig().agents, providerOf: agentProviderId, runsOnHost: agentRunsToolsOnHost }),
248
301
  VERSION, projectDir, jobsRoot, run, git, gitRaw,
249
302
  collectGitRecord, createCoordinatorCommit, ensureJobsRoot, slug, assertRepo,
250
303
  resolveBase, workerModelThinkingSupported, budgetState, execution, buildMetrics,
@@ -254,9 +307,11 @@ const { executeJob, executeImplement, executeScout, executeDecompose } = createE
254
307
  });
255
308
  export { executeJob };
256
309
 
257
- const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
310
+ const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, displayedCapacity, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
258
311
  projectDir, stateRoot, jobsRoot, runsRoot, leasesRoot, slotsRoot, run, currentMaxWorkers, slug, agentsConfig, modelCatalogReady, budgetsForJob, resolveSubscriptionSelection, executeJob, subscriptionJobFieldProblems, repoPolicy, jobArgs,
259
312
  env: process.env, budgetState, getActiveRunId: () => activeRunId,
313
+ sandboxProblem: () => podmanChecks?.problem() ?? null,
314
+ probeModel,
260
315
  });
261
316
  export { runningCount, track };
262
317
 
@@ -272,20 +327,24 @@ export const jobSchema = z.object({
272
327
  verify_regression: z.boolean().optional().describe(
273
328
  "implement only: after the diff passes `verification` and touches production files, temporarily revert just those production files, re-run the SAME verification profile (expected to fail without the fix), then restore them. A re-run that still PASSES proves no test would catch this regression, and the outcome is downgraded to NEEDS_REVIEW regardless of the worker's report -- never silently committed as done. This is the ONLY mechanism that catches a verification profile that passes for the wrong reason (a test-selection flag that accidentally excludes the changed file's own tests reports a real, honest, green run that never touched the diff -- exit-code checking alone cannot see the difference). Defaults to true whenever `verification` is set, since that gap is exactly what nomArmy's trust boundary claims to close; pass `false` explicitly to skip the doubled wall-clock cost (can matter on repos with thousands of tests) and accept the risk instead. No effect with no `verification` profile -- there is nothing to re-run. Ignored by scouts."
274
329
  ),
275
- mode: z.enum(["scout", "implement", "decompose"]).default("implement").describe("implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
330
+ mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
276
331
  base_ref: z.string().optional(),
277
- timeout_seconds: z.number().int().min(30).max(1800).default(600),
278
- reasoning: z.enum(["low", "medium", "high"]).default("medium").describe("Thinking level passed to the worker model. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
332
+ timeout_seconds: z.number().int().min(30).max(1800).optional().describe("Default 600; 1200 for a review scout (one with `reviews`, or an army role in the review phase), since a real security review read for the full 10 minutes and was cut off."),
333
+ reasoning: z.enum(THINKING_LEVELS).default("medium").describe("Thinking level passed to the worker model: off, minimal, low, medium, high, xhigh, adaptive, max or ultra (the higher ones only where the vendor offers them, e.g. xhigh on Codex models; a level the model lacks falls back once to its nearest supported level). Higher levels spend a subscription's usage limits faster. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
279
334
  agent: z.string().regex(/^[A-Za-z0-9._-]{1,64}$/).optional().describe("Run on this agent from the operator's agents.yml, by name (e.g. \"codex\", \"grok\", \"local\"): the local model, a metered api key, or one person's subscription. Omit agent and army_role to use the local model. Refuses an unknown name, never falls back. Mutually exclusive with army_role. A subscription agent also requires on_behalf_of."),
280
335
  model: z.string().regex(/^\S{1,200}$/).optional().describe("The model to run on the job's agent (an api or subscription agent), e.g. \"gpt-6-sol\". Overrides the role's model and the agent's default. Required when the role's model is \"auto\" or the agent has no default. The `army` tool lists each agent's models. Refused on the local agent, whose model `nomarmy model` sets."),
281
336
  run_id: z.string().regex(/^run-[a-z0-9-]{1,80}$/).optional().describe("The /feature run this job belongs to (from run_start). Admission then enforces the run's limits (jobs, api spend, hours) and refuses an agent the run has paused after a vendor usage-limit error; the finished job is recorded into the run."),
282
- report: z.enum(["brief", "standard", "full"]).optional().describe("How much the worker may report back, capped by its agent's tier: brief (today's local-sized report), standard (the default), full (the frontier ceiling: about 2k tokens for implement, 4k for a scout). The report lands in your own context and is re-read every later turn, so ask for full only when the job's findings are the point (a broad review). No effect on the local model, whose caps are calibrated."),
337
+ report: z.enum(["brief", "standard", "full"]).optional().describe("How much the worker may report back, capped by its agent's tier: brief (today's local-sized report), standard (the default), full (the frontier ceiling: about 2k tokens for implement, 4k for a scout). An api or subscription scout defaults to full because its findings are the point; other jobs default to standard. The report lands in your own context and is re-read every later turn. No effect on the local model, whose caps are calibrated."),
338
+ stakes: z.enum(["normal", "high"]).optional().describe("implement: how much a mistake would cost, separate from how hard the work is. high for anything touching security or access control, personal or tenant data, data loss, money, or changes that can't be undone: a verification profile is then required, the revert check can't be turned off, and the job always comes back needing review until an independent review (a scout on another vendor with reviews: <job id>, or a judge on another vendor) has looked at it. A one-line auth change is simple and high-stakes."),
339
+ reviews: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/).optional().describe("scout: the job id this scout independently reviews, so the review is recorded against that job (nomarmy stats shows high-stakes jobs with and without one). Use a different vendor than the job's worker."),
283
340
  commit_subject: z.string().max(200).optional().describe("implement: the subject line of the commit nomArmy makes on the worker branch, e.g. \"Keep held-back tables in the list_tables cache\". Defaults to the task's first sentence; the body is the worker's NOTE, and the job id is a trailer."),
284
341
  army_role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Dispatch by army role (e.g. \"sr-dev\", \"security-analyst\"): nomArmy runs it on the agent this repo assigns to that role and puts the role's description at the top of the brief. Call the `army` tool first to see this repo's roles. Mutually exclusive with agent. Add on_behalf_of in case the role's agent is a subscription; it's ignored otherwise."),
342
+ confirm_over_limit: z.boolean().optional().describe("Override a reached usage limit: the General must ask the operator before resubmitting with confirm_over_limit: true, or send the job to another agent. nomArmy never sets it itself."),
285
343
  on_behalf_of: z.string().min(1).max(254).optional().describe("Required when the job's agent is a subscription: must exactly match that agent's owner in agents.yml, or nomArmy refuses the job. A self-reported attestation, not an independently verified identity check -- nomArmy has no caller-identity boundary today, so what this guarantees is explicit, auditable intent and hard refusal on mismatch or omission, not cryptographic proof of who issued the call. Ignored for a local or api agent."),
286
344
  evidence: z.string().max(maxEvidenceChars,
287
345
  `Evidence exceeds the ${maxEvidenceChars}-character budget. This is for facts already resolved (e.g. with repo_evidence), not more description of the task -- if it needs more than this, resolve less per job or put the pointer (a path and line range) here instead of the material itself.`
288
346
  ).optional().describe("implement only: facts YOU already resolved (e.g. via repo_evidence) that the worker should trust and not re-derive -- exact signatures, call sites, line ranges, existing behavior. Cuts exploration that would otherwise burn the worker's own context budget on something you already know. Not a substitute for a clear objective and acceptance criteria."),
347
+ continue_from: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/).optional().describe("implement only: the job id of a retained, uncommitted implement job (partial, blocked, or failed verification) whose unfinished work this job should finish. The new worktree starts from that job's base with its changes in place, so brief only the correction; the finished whole, that work included, is verified and committed together. Use this instead of fixing a worker's files yourself, which would land them unverified. Leave base_ref out."),
289
348
  worker_id: z.string().regex(/^[A-Za-z0-9._-]+$/).optional()
290
349
  });
291
350
  // A plain function, not jobSchema.superRefine: server.tool(...) registers
@@ -329,17 +388,18 @@ function jobArgs(args, workerId) {
329
388
  return { task: args.task, acceptance: args.acceptance, verification: args.verification, mode: args.mode, baseRef: args.base_ref,
330
389
  timeoutSeconds: args.timeout_seconds, profile: args.profile, reasoning: args.reasoning, pool: args.pool,
331
390
  subscriptionWorker, onBehalfOf: args.on_behalf_of, model: args.model ?? null, reportSize: args.report ?? null, evidence: args.evidence,
332
- verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), workerId };
391
+ verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), continueFrom: args.continue_from ?? null, stakes: args.stakes ?? null, reviews: args.reviews ?? null, workerId };
333
392
  }
334
393
  server.tool("local_worker", "Run one isolated local worker and wait for it. mode=implement edits in its own worktree and the coordinator commits only on a valid done report (or a recovered job that passed independent verification); failed or incomplete worktrees are retained. mode=scout answers a question from a read-only snapshot with mandatory [path:line] citations that nomArmy verifies and expands. mode=decompose (also read-only) proposes 2+ independent subtasks for a broad objective instead of one worker turn trying to do too much; the proposal is never auto-dispatched, review it and make a separate call with the subtasks you choose. Refuses under memory pressure or over capacity; use local_worker_start + local_worker_status to avoid blocking.", jobSchema.shape,
335
394
  async rawArgs => {
336
395
  const expanded = expandJobs([rawArgs]);
337
396
  if (expanded.problems.length) return refusal(expanded.problems);
338
397
  const [args] = expanded.jobs;
339
- const { problems } = await admit([args]);
398
+ const { problems, admission } = await admit([args]);
340
399
  if (problems.length) return refusal(problems);
341
400
  const r = await launch(args).promise;
342
- return toolText(formatResult(r), !r.ok);
401
+ const refreshed = (admission.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
402
+ return toolText(refreshed.length ? `${coordinatorResult(r)}\n\n${refreshed.join("\n")}` : coordinatorResult(r), !r.ok);
343
403
  });
344
404
  server.tool("local_worker_start", "Start one worker or scout in the background and return immediately with a job_id. Poll it with local_worker_status (optionally long-polling with wait_seconds). Same admission rules as local_worker: refuses under memory pressure or when NOMARMY_MAX_WORKERS jobs are already running.", jobSchema.shape,
345
405
  async rawArgs => {
@@ -349,14 +409,15 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
349
409
  const { problems, admission } = await admit([args]);
350
410
  if (problems.length) return refusal(problems);
351
411
  const entry = launch(args);
352
- return toolText(JSON.stringify({ started: true, jobId: entry.jobId, workerId: entry.workerId, mode: entry.mode, state: "running",
412
+ return toolText(coordinatorJson({ started: true, jobId: entry.jobId, workerId: entry.workerId, mode: entry.mode, state: "running",
353
413
  jobDir: path.join(jobsRoot, entry.jobId), timeoutSeconds: args.timeout_seconds,
354
414
  poll: { tool: "local_worker_status", job_id: entry.jobId, wait_seconds: MAX_STATUS_WAIT_SECONDS },
415
+ wait: `nomarmy jobs --wait ${entry.jobId}`,
355
416
  // This job's own lane and budget: a subscription job used to be
356
417
  // reported with the local model's figures.
357
- lane: jobLane(args), agent: args.agentName ?? "local", model: args.model ?? null,
418
+ lane: jobLane(args), agent: args.mode === "verify" ? null : args.agentName ?? "local", model: args.model ?? null,
358
419
  ...(args.run_id ? { run: runBrief(args.run_id) } : {}),
359
- admission: { level: admission.level, notes: admission.reasons }, budgets: describeBudgets(budgetsForJob(args)) }, null, 2));
420
+ admission: { level: admission.level, notes: admission.reasons }, budgets: args.mode === "verify" ? null : describeBudgets(budgetsForJob(args)) }));
360
421
  });
361
422
  // A long poll must return inside the MCP client's own idle-timeout: it aborts
362
423
  // a tool call after N seconds with no response or progress notification,
@@ -373,9 +434,10 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
373
434
  // always crossed; 110s returns in-line with margin. Raise it only for a
374
435
  // client that neither backgrounds nor times out that early.
375
436
  export const MAX_STATUS_WAIT_SECONDS = Number.parseInt(process.env.NOMARMY_MAX_STATUS_WAIT_SECONDS ?? "", 10) || 110;
376
- server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). full=true returns the complete formatted result instead of a summary.`, {
377
- job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false)
378
- }, async ({ job_id, wait_seconds, full }) => {
437
+ server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). report=true returns just the worker's report (a scout's findings with their verified citations), the outcome, issues, verification and commit. full=true returns the complete execution record.`, {
438
+ job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false),
439
+ report: z.boolean().default(false).describe("Just the report and nomArmy's verdict on it, without the rest of the record. Prefer this to full."),
440
+ }, async ({ job_id, wait_seconds, full, report }) => {
379
441
  const jobId = path.basename(job_id), entry = activeJobs.get(jobId), jobDir = path.join(ensureJobsRoot(), jobId);
380
442
  if (entry && !entry.settled && wait_seconds > 0) await Promise.race([entry.promise.catch(() => {}), sleep(wait_seconds * 1000)]);
381
443
  const files = { status: readJson(path.join(jobDir, "status.json")), meta: readJson(path.join(jobDir, "metadata.json")), failure: readJson(path.join(jobDir, "failure.json")) };
@@ -386,15 +448,54 @@ server.tool("local_worker_status", `Status of one job started by this server: ph
386
448
  summarize(entry, files, jobDir),
387
449
  sleep(15000).then(() => summarize(entry, files, null)),
388
450
  ]);
389
- if (summary.state === "running") return toolText(JSON.stringify({ ...summary, jobDir, hint: `poll again with wait_seconds up to ${MAX_STATUS_WAIT_SECONDS}; lastTool/filesChangedLive are best-effort and may be absent early in a run` }, null, 2));
390
- if (entry?.error) return toolText(JSON.stringify({ ...summary, jobDir }, null, 2), true);
391
- if (full && entry?.result) return toolText(formatResult(entry.result), !entry.result.ok);
392
- if (full && files.meta) return toolText(JSON.stringify(files.meta, null, 2), summary.coordinatorStatus !== "complete");
393
- return toolText(JSON.stringify({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with full=true for the complete report" : null }, null, 2), summary.state === "orphaned" || summary.state === "failed");
451
+ if (summary.state === "running") return toolText(coordinatorJson({ ...summary, jobDir, hint: `poll again with wait_seconds up to ${MAX_STATUS_WAIT_SECONDS}; lastTool/filesChangedLive are best-effort and may be absent early in a run` }));
452
+ if (entry?.error) return toolText(coordinatorJson({ ...summary, jobDir }), true);
453
+ if (report && files.meta) return toolText(coordinatorJson(reportView(files.meta)), summary.coordinatorStatus !== "complete");
454
+ if (full && entry?.result) return toolText(coordinatorResult(entry.result), !entry.result.ok);
455
+ if (full && files.meta) return toolText(coordinatorJson(files.meta), summary.coordinatorStatus !== "complete");
456
+ return toolText(coordinatorJson({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with report=true for the worker's report, or full=true for the complete record" : null }), summary.state === "orphaned" || summary.state === "failed");
394
457
  });
458
+ // Set when this session's copy of nomArmy changed on disk after it started
459
+ // (nomarmy connect or update ran): shown first in army and capacity, and
460
+ // notified once, since only a restart of this session picks it up.
461
+ let restartNotified = false;
462
+ function currentRestartNotice() {
463
+ const notice = restartNotice({ serverFile: fileURLToPath(import.meta.url), startedAtMs: SERVER_STARTED_MS, runningVersion: VERSION });
464
+ if (notice && !restartNotified) { restartNotified = true; try { notify("nomArmy: restart this session", notice); } catch { /* best-effort */ } }
465
+ return notice;
466
+ }
467
+ const withRestartNotice = (value) => { const notice = currentRestartNotice(); return notice ? { restartNeeded: notice, ...value } : value; };
468
+
469
+ server.tool("stats", "What nomArmy's own job records show for this repository (or all repositories): jobs by mode, role and model; code committed; worker and job time; tokens and API spend; claim vs evidence (how often a \"done, tests pass\" report failed independent verification, or passed with tests that couldn't catch the change); what didn't complete; reviewer outcomes; and review flags. Every number comes from nomArmy's verified records, never a worker's report. Defects you find at integration aren't in the records: add your own count. Read-only.", {
470
+ since: z.string().max(40).optional().describe("Only jobs started since this: a date (2026-09-25) or an age (7d, 24h). Default: all."),
471
+ until: z.string().max(40).optional().describe("Only jobs started before this date."),
472
+ all_repos: z.boolean().optional().describe("Every repository this machine's nomArmy has run jobs for, not just this one."),
473
+ repo: z.string().max(400).optional().describe("Another repository, by path or folder name (\"senti\" matches rayson-senti if unambiguous)."),
474
+ role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Only jobs dispatched as this army role (e.g. sr-dev)."),
475
+ model: z.string().regex(/^\S{1,200}$/).optional().describe("Only jobs that ran on this model (e.g. grok-4.7)."),
476
+ format: z.enum(["text", "json"]).optional().describe("text (default) is the report; json is the raw numbers."),
477
+ details: z.boolean().optional().describe("The full report (volume, reviewers, flags, what didn't finish). Default is the one-screen summary: what nomArmy caught, high-stakes work needing review, the top routing tips, spend."),
478
+ }, async ({ since, until, all_repos, repo, role, model, format, details }) => {
479
+ try {
480
+ const records = loadJobRecords(jobsRoot);
481
+ let agentFor = () => null;
482
+ try { agentFor = agentLookup(agentsConfig().agents, agentProviderId); } catch { /* commands name <agent> */ }
483
+ const stats = computeStats(records, { repo: repo ? resolveRepo(records, repo) : all_repos ? null : projectDir, sinceMs: parseSince(since), untilMs: parseSince(until), role: role ?? null, model: model ?? null, agentFor });
484
+ return toolText(format === "json" ? JSON.stringify(stats, null, 2) : details ? formatStats(stats) : formatStatsSummary(stats));
485
+ } catch (error) { return toolText(error.message, true); }
486
+ });
487
+
488
+ server.tool("local_worker_stop", "Stop a running job's worker, for example one burning a frontier model's usage on the wrong track. Its worker ends within about 15 seconds, without the report-recovery call a timeout gets and without running verification; its worktree is kept uncommitted, so a new job with continue_from: <job_id> (and a cheaper model if you like) can finish the work. Works for a job started by any session. Refuses a job that isn't running or is already past its worker.", {
489
+ job_id: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/),
490
+ reason: z.string().max(300).optional().describe("Why it's being stopped; recorded in the job's issues."),
491
+ }, async ({ job_id, reason }) => {
492
+ const r = requestJobStop({ jobsRoot, jobId: job_id, reason: reason ?? null });
493
+ return toolText(r.message, !r.ok);
494
+ });
495
+
395
496
  server.tool("local_worker_capacity", "What this host can take right now: context per nom and the brief/report budgets derived from it, memory pressure and whether another job would be admitted, and the jobs currently running. Read-only.", {}, async () => {
396
497
  await budgetState.refresh();
397
- return toolText(JSON.stringify(capacitySnapshot(), null, 2));
498
+ return toolText(JSON.stringify(withRestartNotice(await displayedCapacity()), null, 2));
398
499
  });
399
500
  // The only way to know what `verification`/`union_verification`/
400
501
  // `verify_regression` profile names are actually valid for this repo used to
@@ -468,20 +569,37 @@ server.tool("run_finish", "Close a /feature run as complete or stopped, with a o
468
569
  try {
469
570
  const run = finishRun(runsRoot, run_id, { status, summary });
470
571
  if (activeRunId === run_id) activeRunId = null;
471
- return toolText(JSON.stringify({ id: run.id, status: run.status, ...runTotals(run) }, null, 2));
572
+ // What nomArmy verified in this run, ready for the pull request's description (lib/share.mjs).
573
+ let prBlock = null;
574
+ try { prBlock = shareMarkdown(computeStats(loadJobRecords(jobsRoot), { runId: run_id }), { scope: "this feature run" }); } catch { /* the totals stand without it */ }
575
+ return toolText(JSON.stringify({ id: run.id, status: run.status, ...runTotals(run), ...(prBlock ? { prBlock, prBlockNote: "Put prBlock in the pull request's description as it is: every number is from nomArmy's verified records." } : {}) }, null, 2));
472
576
  } catch (error) { return toolText(error.message, true); }
473
577
  });
474
578
  server.tool("army", "Who you, the General, are and who you call for what in this repository: your fixed charter and the agent you're defined as, the army's workflow, then each role's description, phase (build, review, acceptance), suggested mode, and the agent it runs on, with which config layer set each value (global, project .nomarmy.yml, local .nomarmy.local.yml). Flags roles with no usable agent, and roles that share your model or subscription (not an independent review). Dispatch a role with `army_role`, or an agent directly with `agent`. Read-only, re-read on every call.", {}, async () => {
475
579
  try {
476
580
  const agents = agentsConfig().agents;
477
- const summary = describeArmy(currentArmy(), { agents, describeAgent });
581
+ const usageRefresh = await refreshStaleOverLimitReadings(stateRoot);
582
+ const usageSnapshots = usageRefresh.snapshots;
583
+ const summary = describeArmy(currentArmy(), { agents, describeAgent, usageSnapshots, agentProviderId, usageRefreshError: usageRefresh.error, usageRefreshFailed: usageRefresh.failedProviders });
478
584
  // Each agent's models, from OpenClaw's catalog, so the General can pick
479
585
  // one for a role set to "auto". The catalog can lag a brand-new model.
480
586
  const catalog = await modelCatalogReady();
587
+ // A model its vendor refused on a job is listed apart, so a role on
588
+ // "auto" isn't sent to it (the catalog lists what a plan may refuse).
589
+ retryRefusedModelsInBackground(stateRoot, { probeModel });
590
+ const refusals = modelRefusals(stateRoot);
481
591
  summary.agents = Object.fromEntries(Object.entries(agents).map(([name, agent]) => {
482
592
  const provider = agentProviderId(agent);
483
- const models = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
484
- return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models }];
593
+ const listed = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
594
+ const models = listed.filter((m) => !refusals[`${provider}/${m}`]);
595
+ const refusedModels = listed.filter((m) => refusals[`${provider}/${m}`]);
596
+ const snapshot = usageSnapshots[provider];
597
+ const usage = snapshot ? (() => {
598
+ const status = usageStatus(snapshot);
599
+ const failed = usageRefresh.failedProviders.includes(provider);
600
+ return { level: status.level, text: `${usageDisplayText(status)}${failed ? `. ${usageRefresh.error}` : ""}` };
601
+ })() : null;
602
+ return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, ...(refusedModels.length ? { refusedModels } : {}), usage }];
485
603
  }));
486
604
  // A pinned model missing from the catalog isn't necessarily wrong:
487
605
  // `army assign` proves an unlisted model with a real test call, and the
@@ -493,7 +611,13 @@ server.tool("army", "Who you, the General, are and who you call for what in this
493
611
  role.modelNote = `${role.model} isn't in OpenClaw's catalog for ${role.agent}; \`army assign\` checked it with a real test call when it was set, and the catalog can lag new models. Use it as assigned; if a job reports "Unknown model", reassign.`;
494
612
  }
495
613
  }
496
- return toolText(JSON.stringify(summary, null, 2));
614
+ // Routing suggestions from this repo's recent jobs (lib/suggestions.mjs):
615
+ // tell the operator about them; never apply one without their say-so.
616
+ try {
617
+ const list = recentSuggestions(loadJobRecords(jobsRoot), { projectDir, agentFor: agentLookup(agents, agentProviderId) });
618
+ if (list.length) summary.suggestions = { note: "From this repo's last 14 days of jobs. Tell the operator; change routing only with their say-so (nomarmy army assign).", items: list.map(({ level, title, evidence, command }) => ({ level, title, evidence, command })) };
619
+ } catch { /* suggestions are a bonus; the army summary stands without them */ }
620
+ return toolText(JSON.stringify(withRestartNotice(summary), null, 2));
497
621
  } catch (error) {
498
622
  return toolText(error.message, true);
499
623
  }
@@ -503,7 +627,7 @@ server.tool("local_worker_config", "What this checkout's .nomarmy.yml defines --
503
627
  return toolText(JSON.stringify(summary, null, 2), summary.valid === false);
504
628
  });
505
629
  server.tool("local_workers", "Run independent jobs (implement or scout) with bounded parallelism and wait for all of them. Every implement job receives its own branch, worktree, sandbox session, logs, validation, and coordinator-owned commit. This tool never merges any branch into the developer's branch. With auto_union: true, implement jobs that reach a valid outcome and touch non-overlapping files are additionally merged (git merge --no-ff) into ONE new integration branch -- a review artifact alongside the untouched per-job branches, still not the developer's branch, still reviewed and integrated explicitly. Jobs that overlap or did not finish validly are excluded from the union and reported individually exactly as without auto_union. For long batches prefer local_worker_start per job and poll.", {
506
- jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (NOMARMY_MAX_POOL_WORKERS), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
630
+ jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (`nomarmy config max-jobs`, default 4), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
507
631
  auto_union: z.boolean().default(false).describe(
508
632
  "After all jobs finish, mechanically merge (git merge --no-ff) implement jobs that reached a valid outcome and touched non-overlapping files into ONE new integration branch for review -- never into the developer's branch. Overlapping or invalid-outcome jobs are excluded and still reported individually, unchanged. All jobs must share one base_ref (or omit it); it is resolved once, before any job starts, and forced onto every job so the union is provably rooted at a single base."
509
633
  ),
@@ -514,7 +638,7 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
514
638
  const expanded = expandJobs(rawJobs);
515
639
  if (expanded.problems.length) return refusal(expanded.problems);
516
640
  const { jobs } = expanded;
517
- const { problems } = await admit(jobs);
641
+ const { problems, admission } = await admit(jobs);
518
642
  let forcedBase = null;
519
643
  if (auto_union) {
520
644
  const refs = [...new Set(jobs.map(j => j.base_ref).filter(Boolean))];
@@ -547,7 +671,7 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
547
671
  // so it was invisible to both ceilings while it ran.
548
672
  // A batch job waits for its agent's slot (up to its own timeout) rather
549
673
  // than failing because an earlier job in the same batch holds it.
550
- return trackInRun(j, track(jobId, { mode: j.mode, workerId, lane: jobLane(j), agent: j.agentName ?? null, runId: j.run_id ?? null, role: j.armyRole ?? null, model: j.model ?? null },
674
+ return trackInRun(j, track(jobId, { mode: j.mode, workerId, lane: jobLane(j), agent: j.agentName ?? null, runId: j.run_id ?? null, role: j.armyRole ?? null, model: j.model ?? null, label: jobLabel(j) },
551
675
  withAgentSlot(j, jobId, () => executeJob({ ...jobArgs(effectiveJob, workerId), jobId }), { waitMs: (j.timeout_seconds ?? 600) * 1000 }))).promise;
552
676
  }, { staggerMs: WORKER_START_STAGGER_MS });
553
677
  indices.forEach((i, laneI) => { results[i] = laneResults[laneI]; });
@@ -578,8 +702,9 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
578
702
  reviewRequired: results.filter(r => r.manifest?.reviewRequired).length,
579
703
  jobs: results.map(r => ({ jobId: r.manifest.jobId, workerId: r.manifest.workerId, mode: r.manifest.mode, outcome: r.manifest.outcome || OUTCOMES.WORKER_FAILED, recovered: Boolean(r.manifest.recovered), status: r.manifest.coordinatorStatus || "failed", branch: r.manifest.branch, commit: r.manifest.commit?.sha || null, worktree: r.manifest.worktree, jobDir: r.jobDir })),
580
704
  ...(union ? { union } : {}) };
581
- const unionSection = union ? `UNION\n\n${formatUnion(union)}\n\n` : "";
582
- const text = `BATCH EXECUTION RECORD\n${JSON.stringify(summary, null, 2)}\n\n${unionSection}WORKER RESULTS\n\n${results.map((r, i) => `===== WORKER ${i + 1} =====\n${formatResult(r)}`).join("\n\n")}`;
705
+ const unionSection = union ? `UNION\n\n${formatUnion(withWindowsPaths(union))}\n\n` : "";
706
+ const refreshed = (admission?.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
707
+ const text = `BATCH EXECUTION RECORD\n${coordinatorJson(summary)}\n\n${unionSection}WORKER RESULTS\n\n${results.map((r, i) => `===== WORKER ${i + 1} =====\n${coordinatorResult(r)}`).join("\n\n")}${refreshed.length ? `\n\n${refreshed.join("\n")}` : ""}`;
583
708
  return toolText(text, results.some(r => !r.ok) || union?.status === "union_verification_failed" || union?.status === "union_error");
584
709
  });
585
710
  // No model, no sandbox, no tokens spent on a worker: the coordinator asks the
@@ -604,7 +729,7 @@ server.tool("local_worker_jobs", "List recent job records for review/recovery, i
604
729
  if (status) return summarize(activeJobs.get(name) ?? null, { status, meta: null, failure: null }, dir);
605
730
  return { jobId: name, state: "unknown" };
606
731
  }));
607
- return toolText(JSON.stringify(rows, null, 2));
732
+ return toolText(coordinatorJson(rows));
608
733
  });
609
734
  // The sandbox writes skill/guardrail files under .openclaw/ with permissions
610
735
  // meant to stop the SANDBOXED AGENT from deleting them. On macOS, the
@@ -736,7 +861,7 @@ server.tool("local_worker_sweep", "Bulk-reap job worktrees/branches that are PRO
736
861
  skipped.push({ jobId, reason: `removal failed: ${error.message}` });
737
862
  }
738
863
  }
739
- return toolText(JSON.stringify({ examined: dirs.length, reapedCount: reaped.length, skippedCount: skipped.length, dryRun: dry_run, reaped, skipped }, null, 2));
864
+ return toolText(coordinatorJson({ examined: dirs.length, reapedCount: reaped.length, skippedCount: skipped.length, dryRun: dry_run, reaped, skipped }));
740
865
  });
741
866
  server.tool("local_worker_cleanup", "Remove a retained worker worktree and optionally its agent branch after Claude has reviewed/integrated or deliberately discarded it. Refuses to delete the current branch. A branch whose commits were cherry-picked (not merged) into the current branch -- nomArmy's own integration model -- is recognized as integrated by comparing PATCH CONTENT (git cherry), not git's own ancestry-only check, so a genuinely-integrated job's cleanup does not need force: true. Reserve force for a branch you are actually discarding unintegrated work from.", {
742
867
  job_id: z.string().min(1), delete_branch: z.boolean().default(false), force: z.boolean().default(false)
@@ -791,6 +916,12 @@ if (isMain) {
791
916
  // sqlglot, so verification could never pass), and a worker could edit its
792
917
  // own worktree's copy to weaken the checks that judge it.
793
918
  registerVerificationRunner(createVerificationRunner({ hostProjectDir: projectDir, loadConfig: () => loadConfig(projectDir) }));
919
+ // A healthy answer is reused for 20 seconds, so a batch doesn't ask Podman per job.
920
+ let podmanOkUntil = 0;
921
+ podmanChecks = {
922
+ problem: () => { if (Date.now() < podmanOkUntil) return null; const p = podmanProblem(); if (!p) podmanOkUntil = Date.now() + 20000; return p; },
923
+ vmStartedAt: () => podmanVmStartedAt(),
924
+ };
794
925
  // Warm the budget from the profile or the running llama-server. Not awaited:
795
926
  // admission refreshes it anyway, and a slow hardware probe must not delay
796
927
  // the MCP handshake.
package/package.json CHANGED
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "name": "nomarmy",
3
- "description": "A harness for AI coding workers whose claims are never trusted: your coding assistant stays in charge while workers implement and test in sandboxes, on local models, API keys or your own subscriptions.",
3
+ "description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
4
4
  "author": "Rayson Technologies",
5
5
  "license": "Apache-2.0",
6
- "version": "0.1.0-alpha.2",
6
+ "version": "0.1.0-alpha.21",
7
7
  "private": false,
8
8
  "type": "module",
9
9
  "engines": {
@@ -21,14 +21,15 @@
21
21
  "tag": "alpha"
22
22
  },
23
23
  "dependencies": {
24
- "@modelcontextprotocol/sdk": "^1.0.0",
24
+ "@modelcontextprotocol/sdk": "^1.30.1",
25
25
  "@secretlint/node": "^13.0.5",
26
26
  "@secretlint/secretlint-rule-preset-recommend": "^13.0.5",
27
27
  "yaml": "^2.5.0",
28
- "zod": "^3.24.0"
28
+ "zod": "^4.6.5"
29
29
  },
30
30
  "scripts": {
31
- "test": "node --test tests/*.test.mjs"
31
+ "test": "node --test tests/*.test.mjs",
32
+ "docs:harnesses": "node scripts/generate-harness-docs.mjs"
32
33
  },
33
34
  "bin": {
34
35
  "nomarmy": "bin/nomarmy.mjs"
@@ -43,6 +44,7 @@
43
44
  "scripts",
44
45
  "policies",
45
46
  "docker",
47
+ "harnesses",
46
48
  "install.sh",
47
49
  "e2e.sh",
48
50
  "LICENSE",
@@ -12,10 +12,11 @@ You are the General. Build this feature end to end with nomArmy's army and come
12
12
 
13
13
  Follow the army's workflow, calling only the roles the work needs:
14
14
 
15
+ 0. **Routing.** The `army` tool's `suggestions` come from this repo's recent jobs (a role that keeps failing, a cheaper model doing as well, unreviewed high-stakes work). Tell the operator about any at the start and in your final summary; change routing only if they say so.
15
16
  1. **Plan.** Scout the repo as needed (`repo_evidence` first; a scout only for research that would pull many files into your context). Write the plan into the run log: the outcome, acceptance criteria, the pieces, and which role gets each.
16
- 2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work.
17
- 3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Send what they find back to the builders as new, bounded jobs.
18
- 4. **Acceptance.** PO and stakeholder test end to end. Fix what they find the same way.
17
+ 2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work. Mark a job `stakes: high` when a mistake would be costly (security or access control, personal or tenant data, data loss, money, anything irreversible), however small the change: that's separate from how hard it is.
18
+ 3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Every `stakes: high` build job gets an independent review before you accept it: a scout on a different vendor than its worker, with `reviews: <that job id>`. Send what they find back to the builders as new, bounded jobs. A build job that comes back partial, blocked or failing verification is finished with `continue_from: <its job id>` and a brief of just the correction, never by fixing its files yourself: only verified work lands.
19
+ 4. **Acceptance.** PO and stakeholder test end to end. Checks that only run existing tests use mode: verify; writing new e2e checks is still an implement job. Fix what they find the same way.
19
20
  5. **Integrate.** Review every diff against nomArmy's verified record -- a worker's report is a claim, not evidence -- and bring the accepted work together on one branch. **Never merge into the developer's branch, and never push.** The finished state is a branch ready for the operator to review and merge.
20
21
 
21
22
  ## Decisions along the way
@@ -24,9 +25,11 @@ When you hit a choice you'd normally ask the operator about, don't stop: pick th
24
25
 
25
26
  Stop and ask only for something irreversible or outside this repository: merging or pushing, deploying, anything needing cloud or production credentials, deleting data, or changing another repository.
26
27
 
28
+ A gate's result is not yours to reinterpret. When a check the plan or the operator set (an exact expected output, a count, a checksum, a contract) doesn't match, that's a failure, even when the difference looks cosmetic: whitespace from BSD versus GNU `wc`, line endings, ordering, a trailing newline. Don't judge it "equivalent" and pass it. Either fix the check so it compares what was meant (normalize both sides, and write that change into the run log as a decision), or send the mismatch back as a failed job. A gate the General can wave through is no gate.
29
+
27
30
  ## Watching jobs
28
31
 
29
- Don't poll in a loop: each status call costs your own usage. Where your coordinator can watch a background command (Claude Code's monitor), watch `nomarmy jobs --events` -- one line per job start, phase change and finish -- and act when a line arrives. Otherwise use `local_worker_status` with the longest `wait_seconds` it allows. `run_status` lists the run's running jobs as well as finished ones. The operator gets a desktop notification whenever a job finishes and whenever the run crosses a limit, so you don't need to relay each one.
32
+ Prefer `local_worker_start`. Right after starting a job, if your coordinator can run a background command, run `nomarmy jobs --wait <job_id>` in the background so you're told the moment it finishes and can tell the operator. Otherwise poll `local_worker_status` with the longest `wait_seconds` it allows. For several jobs at once, `nomarmy jobs --wait <id> <id> ...` exits when every one has finished. For a whole run, use `nomarmy jobs --events --until-done --run <run-id>`. Never use unscoped `--until-done` when other sessions may have jobs. The plain `nomarmy jobs --events` stream never exits on its own while jobs run, so don't run it as a background command (you'd only hear when it exits); use it only with a monitor that wakes on each line. `run_status` lists the run's running jobs as well as finished ones. The operator gets a desktop notification whenever a job finishes and whenever the run crosses a limit, so you don't need to relay each one.
30
33
 
31
34
  ## Limits
32
35
 
@@ -40,4 +43,4 @@ The `logPath` from `run_start`. Markdown, updated after every phase: the plan; e
40
43
 
41
44
  ## When it's done
42
45
 
43
- Call `run_finish` (`complete` or `stopped`), then report in one message: what was built and on which branch; what each role found and how it was resolved; the decisions made on the operator's behalf; test and verification results; and the run's cost from `run_status` (jobs and api spend per agent). If a push-notification tool is available, notify the operator that the run finished or stopped.
46
+ Call `run_finish` (`complete` or `stopped`). Its result has a `prBlock`: when you or the operator open a pull request for the run's branch, put it in the description as it is (every number is from nomArmy's verified records). Then report in one message: what was built and on which branch; what each role found and how it was resolved; the decisions made on the operator's behalf; test and verification results; and the run's cost from `run_status` (jobs and api spend per agent). If a push-notification tool is available, notify the operator that the run finished or stopped.
@@ -57,7 +57,9 @@ else
57
57
  if [[ -n "$REMOTE_MODEL" && "$REMOTE_MODEL" != "${NOMARMY_WORKER_MODEL:-}" ]]; then
58
58
  # Jobs ask for NOMARMY_WORKER_MODEL, so record the name this server
59
59
  # actually serves (`nomarmy connect`, run next by install.sh, reads it).
60
- COMMON="$ROOT/config/common.env"
60
+ # Your settings file, so an update can't reset it (lib/user-config.mjs).
61
+ COMMON="$(nomarmy_user_config_dir)/common.env"
62
+ mkdir -p "$(dirname "$COMMON")"; touch "$COMMON"
61
63
  for key in NOMARMY_MODEL_ALIAS NOMARMY_WORKER_MODEL; do
62
64
  if grep -q "^$key=" "$COMMON"; then
63
65
  KEY="$key" VALUE="$REMOTE_MODEL" node -e 'const fs=require("fs"),f=process.argv[1];fs.writeFileSync(f,fs.readFileSync(f,"utf8").replace(new RegExp(`^${process.env.KEY}=.*$`,"m"),`${process.env.KEY}=${process.env.VALUE}`))' "$COMMON"
@@ -66,7 +68,7 @@ else
66
68
  fi
67
69
  done
68
70
  export NOMARMY_MODEL_ALIAS="$REMOTE_MODEL" NOMARMY_WORKER_MODEL="$REMOTE_MODEL"
69
- echo "==> The server serves '$REMOTE_MODEL'; recorded it in config/common.env"
71
+ echo "==> The server serves '$REMOTE_MODEL'; recorded it in $COMMON"
70
72
  fi
71
73
  # llama-server reports the context of one slot, which is one nom's share.
72
74
  REMOTE_CTX="$(curl -fsS --max-time 5 "$SERVER/props" | node -e 'let s="";process.stdin.on("data",d=>s+=d).on("end",()=>{try{const n=JSON.parse(s).default_generation_settings?.n_ctx;if(Number.isInteger(n)&&n>0)process.stdout.write(String(n))}catch{}})' || true)"
@@ -0,0 +1,42 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+ import { HARNESS_ROOT, loadHarnesses } from "../lib/harnesses.mjs";
5
+
6
+ export const HARNESS_DOCS = fileURLToPath(new URL("../docs/harnesses.md", import.meta.url));
7
+ const cell = (value) => String(value).replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;").replaceAll("|", "&#124;").replaceAll("\n", " ").replaceAll("\r", " ");
8
+
9
+ export function generateHarnessDocs(root = HARNESS_ROOT) {
10
+ const { harnesses, problems } = loadHarnesses(root);
11
+ if (problems.length) throw new Error(problems.map(({ name, reason }) => `${name}: ${reason}`).join("\n"));
12
+ const rows = Object.values(harnesses).sort((a, b) => a.name.localeCompare(b.name)).map((spec) => {
13
+ const detects = spec.detect.map((rule) => Object.entries(rule).map(([kind, value]) => `${kind}: ${value}`).join(", ")).join("; ") || "none";
14
+ const requires = Object.entries(spec.requires).map(([key, value]) => `${key}: ${value}`).join(", ") || "none";
15
+ return `| [${spec.name}](https://github.com/rayson-tech/nomarmy/blob/main/harnesses/${spec.name}/README.md) | ${cell(spec.summary)} | ${cell(detects)} | ${spec.network} | ${cell(requires)} |`;
16
+ });
17
+ return `<!-- Generated by npm run docs:harnesses. Do not edit by hand. -->
18
+ # Harnesses
19
+
20
+ A harness is a data-only folder describing detection, image layers, and proposed verification profiles. Matched harnesses compose one cached image for workers and verification: toolchains first, then Python, Node, and declarative layers, respecting \`after\`. Tags hash the recipe and copied dependency files. Verification profiles remain proposals. Repositories can explicitly enable harnesses with \`harnesses: [names]\` in \`.nomarmy.yml\`; service networks are used only for nomArmy verification.
21
+
22
+ - **none**: offline verification (default).
23
+ - **services**: fake services on a private network with no route out.
24
+ - **allowlist**: verification-only access to operator-approved hosts, with dedicated test tenants and throwaway credentials, never production (planned).
25
+
26
+ Workers always remain offline. See the [plan](plans/2026-09-25-sandbox-dependencies.md) and [Adding a harness](../CONTRIBUTING.md#adding-a-harness).
27
+
28
+ | Harness | Summary | Detects | Network | Requires |
29
+ |---------|---------|---------|---------|----------|
30
+ ${rows.join("\n")}
31
+ `;
32
+ }
33
+
34
+ export function checkHarnessDocs(root = HARNESS_ROOT, output = HARNESS_DOCS) {
35
+ if (fs.readFileSync(output, "utf8") !== generateHarnessDocs(root)) {
36
+ throw new Error("Harness docs are stale. Run npm run docs:harnesses.");
37
+ }
38
+ }
39
+
40
+ if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
41
+ fs.writeFileSync(HARNESS_DOCS, generateHarnessDocs());
42
+ }
@@ -0,0 +1,23 @@
1
+ // install.sh has already installed nomArmy's dependencies. Nothing is ready
2
+ // until the configured vendors and read-only OpenClaw diagnostics pass.
3
+ import { loadAgents } from "../lib/agents.mjs";
4
+ import { globalConfigDir } from "../lib/army.mjs";
5
+ import { ensureOpenClawOnPath } from "../lib/openclaw-path.mjs";
6
+ import { configuredSubscriptionVendors, openclawInstallPlan, repairOpenclaw, runOpenclawCommand, PINNED_OPENCLAW_VERSION } from "../lib/openclaw-install.mjs";
7
+
8
+ ensureOpenClawOnPath();
9
+ const command = process.env.NOMARMY_OPENCLAW_CMD || "openclaw";
10
+ const version = runOpenclawCommand(command, ["--version"]);
11
+ const [major, minor] = process.versions.node.split(".").map(Number);
12
+ if (openclawInstallPlan(version.ok ? version.stdout : null).length &&
13
+ !((major === 24 && minor >= 16) || (major === 26 && minor >= 1) || major > 26)) {
14
+ console.error(`OpenClaw ${PINNED_OPENCLAW_VERSION} needs Node 24.16+ or 26.1+. Upgrade Node first.`);
15
+ process.exitCode = 1;
16
+ } else {
17
+ const prefixIndex = process.argv.indexOf("--prefix");
18
+ const result = await repairOpenclaw({
19
+ command, yes: true, prefix: prefixIndex < 0 ? null : process.argv[prefixIndex + 1],
20
+ vendors: configuredSubscriptionVendors(loadAgents(globalConfigDir()).agents),
21
+ });
22
+ process.exitCode = result.ok ? 0 : 1;
23
+ }