nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -480
- package/bin/nomarmy.mjs +1081 -185
- package/docker/Dockerfile +2 -2
- package/docker/Dockerfile.go +6 -4
- package/docker/Dockerfile.rust +17 -2
- package/harnesses/_template/README.md +27 -0
- package/harnesses/_template/harness.yml +26 -0
- package/harnesses/browser-playwright/README.md +35 -0
- package/harnesses/browser-playwright/fixture/package.json +1 -0
- package/harnesses/browser-playwright/fixture/page.html +1 -0
- package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
- package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
- package/harnesses/browser-playwright/harness.yml +18 -0
- package/harnesses/go/README.md +45 -0
- package/harnesses/go/harness.yml +14 -0
- package/harnesses/mock-oidc/README.md +31 -0
- package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
- package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
- package/harnesses/mock-oidc/harness.yml +19 -0
- package/harnesses/node/README.md +53 -0
- package/harnesses/node/harness.yml +18 -0
- package/harnesses/python/README.md +46 -0
- package/harnesses/python/harness.yml +16 -0
- package/harnesses/rust/README.md +45 -0
- package/harnesses/rust/harness.yml +13 -0
- package/install.sh +29 -9
- package/lib/admission.mjs +178 -30
- package/lib/agents.mjs +8 -6
- package/lib/army.mjs +25 -10
- package/lib/codex-link.mjs +37 -0
- package/lib/config.mjs +15 -0
- package/lib/connect.mjs +232 -19
- package/lib/continue-from.mjs +103 -0
- package/lib/coordinator-instructions.mjs +5 -1
- package/lib/diff-checks.mjs +114 -0
- package/lib/dispatch-schema.mjs +14 -12
- package/lib/doctor.mjs +98 -9
- package/lib/egress-proxy.mjs +116 -0
- package/lib/execute.mjs +241 -33
- package/lib/git-record.mjs +27 -3
- package/lib/harness-schema.mjs +61 -0
- package/lib/harnesses.mjs +99 -0
- package/lib/health.mjs +162 -18
- package/lib/install-freshness.mjs +114 -0
- package/lib/jev-checks.mjs +110 -0
- package/lib/job-format.mjs +54 -0
- package/lib/judge.mjs +130 -0
- package/lib/limits.mjs +77 -0
- package/lib/model-probe.mjs +61 -0
- package/lib/mutation.mjs +159 -0
- package/lib/notify.mjs +30 -3
- package/lib/openclaw-install.mjs +122 -0
- package/lib/openclaw-path.mjs +28 -0
- package/lib/openclaw-run.mjs +74 -12
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/outcome.mjs +21 -2
- package/lib/outcomes.mjs +6 -0
- package/lib/path-utils.mjs +4 -0
- package/lib/podman-health.mjs +41 -0
- package/lib/process.mjs +4 -1
- package/lib/propose.mjs +10 -11
- package/lib/refusal-retry.mjs +16 -0
- package/lib/registry-python.mjs +98 -0
- package/lib/registry-secrets.mjs +140 -0
- package/lib/repo-query.mjs +13 -7
- package/lib/runs.mjs +7 -1
- package/lib/same-path.mjs +14 -0
- package/lib/sandbox-images.mjs +499 -83
- package/lib/sandbox-vm.mjs +32 -0
- package/lib/scan.mjs +5 -1
- package/lib/schema.mjs +20 -11
- package/lib/scout.mjs +21 -3
- package/lib/server-context.mjs +21 -1
- package/lib/setup-steps.mjs +55 -0
- package/lib/share.mjs +82 -0
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +315 -0
- package/lib/statusline.mjs +32 -6
- package/lib/subscription-setup.mjs +13 -0
- package/lib/suggestions.mjs +153 -0
- package/lib/thinking.mjs +23 -0
- package/lib/transcript.mjs +30 -5
- package/lib/usage-limits.mjs +329 -0
- package/lib/user-config.mjs +106 -0
- package/lib/validators.mjs +220 -0
- package/lib/verification-artifacts.mjs +46 -0
- package/lib/verification-flow.mjs +52 -7
- package/lib/verification-network.mjs +66 -0
- package/lib/verify.mjs +338 -85
- package/lib/worker-prompt.mjs +5 -2
- package/lib/wsl-cli.mjs +152 -0
- package/lib/wsl.mjs +230 -0
- package/lib/zod-issues.mjs +15 -0
- package/mcp/server.mjs +165 -34
- package/package.json +7 -5
- package/playbooks/feature.md +8 -5
- package/scripts/configure-openclaw.sh +4 -2
- package/scripts/generate-harness-docs.mjs +42 -0
- package/scripts/install-openclaw.mjs +23 -0
- package/scripts/lib.sh +9 -2
- package/scripts/select-model.mjs +12 -5
- package/scripts/start-inference.sh +2 -2
package/mcp/server.mjs
CHANGED
|
@@ -32,8 +32,37 @@ import { liveLeases } from "../lib/slots.mjs";
|
|
|
32
32
|
import { createRun, loadRun, runTotals, finishRun, resolveRunLimits, describeLoweredLimits } from "../lib/runs.mjs";
|
|
33
33
|
import { agentDispatchFields, resolveAgentModel, agentProviderId, describeAgent } from "../lib/agents.mjs";
|
|
34
34
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "../lib/outcomes.mjs";
|
|
35
|
+
import { readUsageSnapshots, usageStatus, usageDisplayText, refreshStaleOverLimitReadings } from "../lib/usage-limits.mjs";
|
|
36
|
+
import { modelRefusals } from "../lib/health.mjs";
|
|
37
|
+
import { retryRefusedModelsInBackground } from "../lib/refusal-retry.mjs";
|
|
38
|
+
import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
|
|
39
|
+
import { restartNotice } from "../lib/install-freshness.mjs";
|
|
40
|
+
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
41
|
+
import { loadJobRecords, computeStats, formatStats, formatStatsSummary, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
42
|
+
import { shareMarkdown } from "../lib/share.mjs";
|
|
43
|
+
import { recentSuggestions } from "../lib/suggestions.mjs";
|
|
44
|
+
import { probeModel } from "../lib/model-probe.mjs";
|
|
45
|
+
import { jevSettings, judgeSettings } from "../lib/validators.mjs";
|
|
46
|
+
import { agentRunsToolsOnHost } from "../lib/dispatch-schema.mjs";
|
|
35
47
|
import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
|
|
36
|
-
import { compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
|
|
48
|
+
import { jobLabel, compactJobRecord, formatResult, reportView, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
|
|
49
|
+
import { ensureOpenClawOnPath } from "../lib/openclaw-path.mjs";
|
|
50
|
+
import { THINKING_LEVELS } from "../lib/thinking.mjs";
|
|
51
|
+
import { samePath } from "../lib/same-path.mjs";
|
|
52
|
+
import { withWindowsPaths, dropWindowsPath } from "../lib/wsl.mjs";
|
|
53
|
+
export { withWindowsPaths };
|
|
54
|
+
|
|
55
|
+
function coordinatorJson(value) { return JSON.stringify(withWindowsPaths(value), null, 2); }
|
|
56
|
+
function coordinatorResult(result) {
|
|
57
|
+
const paths = withWindowsPaths({ jobDir: result.jobDir, worktree: result.manifest?.worktree });
|
|
58
|
+
const windows = ["jobDirWindows", "worktreeWindows"].filter(key => paths[key]).map(key => `${key}: ${paths[key]}`);
|
|
59
|
+
return formatResult(withWindowsPaths(result)) + (windows.length ? `\n\n${windows.join("\n")}` : "");
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Inside WSL, only the distro's own tools (see lib/wsl.mjs).
|
|
63
|
+
dropWindowsPath();
|
|
64
|
+
// OpenClaw in ~/.npm-global/bin (no writable npm prefix) is found without the operator editing PATH.
|
|
65
|
+
ensureOpenClawOnPath();
|
|
37
66
|
|
|
38
67
|
export { run, mapLimit };
|
|
39
68
|
export { readsMeasurable, measureReads };
|
|
@@ -54,6 +83,7 @@ export { TEST_PATH_PATTERNS, isTestPath, testPatternFor, classifyTestChanges, de
|
|
|
54
83
|
// package.json to the same relative location next to the installed
|
|
55
84
|
// mcp/server.mjs, so this resolves identically in a dev checkout or an
|
|
56
85
|
// installed copy.
|
|
86
|
+
const SERVER_STARTED_MS = Date.now();
|
|
57
87
|
const VERSION = JSON.parse(fs.readFileSync(path.join(path.dirname(fileURLToPath(import.meta.url)), "..", "package.json"), "utf8")).version;
|
|
58
88
|
// Sent to every coordinator on connect, so no project needs a copied CLAUDE.md.
|
|
59
89
|
const server = new McpServer({ name: "nomarmy-local-worker", version: VERSION }, { instructions: COORDINATOR_INSTRUCTIONS });
|
|
@@ -77,7 +107,8 @@ function slug(prefix = "local") {
|
|
|
77
107
|
}
|
|
78
108
|
async function assertRepo() {
|
|
79
109
|
const root = await git(["rev-parse", "--show-toplevel"]);
|
|
80
|
-
|
|
110
|
+
// Git may print a long, forward-slashed path while Windows supplies an 8.3 path.
|
|
111
|
+
if (!samePath(root, projectDir)) throw new Error(`CLAUDE_PROJECT_DIR must be the Git root. Expected ${root}, got ${projectDir}`);
|
|
81
112
|
}
|
|
82
113
|
async function resolveBase(baseRef) {
|
|
83
114
|
const ref = baseRef || "HEAD";
|
|
@@ -208,6 +239,17 @@ export function makeHeartbeatTick(jobDir) { return heartbeatTick(jobDir, livePro
|
|
|
208
239
|
// Senti run none were tagged, so a 4-hour run went 8.46 hours unchecked.
|
|
209
240
|
let activeRunId = null;
|
|
210
241
|
|
|
242
|
+
export const DEFAULT_TIMEOUT_SECONDS = 600;
|
|
243
|
+
export const REVIEW_SCOUT_TIMEOUT_SECONDS = 1200;
|
|
244
|
+
/** A review scout (reviews set, or a review-phase role) gets longer: reviews trace across the codebase. */
|
|
245
|
+
export function defaultTimeoutSeconds(job, getArmyFn) {
|
|
246
|
+
if (job.mode !== "scout") return DEFAULT_TIMEOUT_SECONDS;
|
|
247
|
+
if (job.reviews) return REVIEW_SCOUT_TIMEOUT_SECONDS;
|
|
248
|
+
if (!job.army_role) return DEFAULT_TIMEOUT_SECONDS;
|
|
249
|
+
try { return getArmyFn()?.roles?.[job.army_role]?.phase === "review" ? REVIEW_SCOUT_TIMEOUT_SECONDS : DEFAULT_TIMEOUT_SECONDS; }
|
|
250
|
+
catch { return DEFAULT_TIMEOUT_SECONDS; }
|
|
251
|
+
}
|
|
252
|
+
|
|
211
253
|
export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agentsConfig().agents, getActiveRun = () => activeRunId, env = process.env } = {}) {
|
|
212
254
|
const problems = [];
|
|
213
255
|
let army = null, agents = null;
|
|
@@ -215,6 +257,11 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
|
|
|
215
257
|
const expanded = jobs.map((job, i) => {
|
|
216
258
|
try {
|
|
217
259
|
let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
|
|
260
|
+
if (j.timeout_seconds == null) j = { ...j, timeout_seconds: defaultTimeoutSeconds(j, () => (army ??= getArmy().army)) };
|
|
261
|
+
if (j.mode === "verify") {
|
|
262
|
+
const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
|
|
263
|
+
return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
|
|
264
|
+
}
|
|
218
265
|
if (j.army_role) { army ??= getArmy().army; j = expandArmyRole(j, army); }
|
|
219
266
|
const { agent, roleModel = null, ...rest } = j;
|
|
220
267
|
if (!agent) {
|
|
@@ -228,6 +275,7 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
|
|
|
228
275
|
const out = { ...rest, ...fields, agentName: agent };
|
|
229
276
|
if (model) out.model = model; else delete out.model;
|
|
230
277
|
if (!fields.subscription_worker) delete out.on_behalf_of;
|
|
278
|
+
if (out.mode === "scout" && out.report == null && (fields.pool || fields.subscription_worker)) out.report = "full";
|
|
231
279
|
out.profile ??= "coder";
|
|
232
280
|
return out;
|
|
233
281
|
} catch (error) {
|
|
@@ -244,7 +292,12 @@ const verificationFlow = createVerificationFlow({
|
|
|
244
292
|
const { registerVerificationRunner, normalizeVerification, runIndependentVerification, runRegressionCheck, selectUnionCandidates, buildUnionBranch } = verificationFlow;
|
|
245
293
|
export { registerVerificationRunner, normalizeVerification, runRegressionCheck, selectUnionCandidates, buildUnionBranch };
|
|
246
294
|
|
|
295
|
+
// Podman checks, wired at startup below (a test importing this module gets none).
|
|
296
|
+
let podmanChecks = null;
|
|
247
297
|
const { executeJob, executeImplement, executeScout, executeDecompose } = createExecutor({
|
|
298
|
+
podmanVmStartedAt: () => podmanChecks?.vmStartedAt() ?? null,
|
|
299
|
+
jevSettings: () => jevSettings(),
|
|
300
|
+
judgeSettings: () => judgeSettings({ agents: agentsConfig().agents, providerOf: agentProviderId, runsOnHost: agentRunsToolsOnHost }),
|
|
248
301
|
VERSION, projectDir, jobsRoot, run, git, gitRaw,
|
|
249
302
|
collectGitRecord, createCoordinatorCommit, ensureJobsRoot, slug, assertRepo,
|
|
250
303
|
resolveBase, workerModelThinkingSupported, budgetState, execution, buildMetrics,
|
|
@@ -254,9 +307,11 @@ const { executeJob, executeImplement, executeScout, executeDecompose } = createE
|
|
|
254
307
|
});
|
|
255
308
|
export { executeJob };
|
|
256
309
|
|
|
257
|
-
const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
|
|
310
|
+
const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, displayedCapacity, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
|
|
258
311
|
projectDir, stateRoot, jobsRoot, runsRoot, leasesRoot, slotsRoot, run, currentMaxWorkers, slug, agentsConfig, modelCatalogReady, budgetsForJob, resolveSubscriptionSelection, executeJob, subscriptionJobFieldProblems, repoPolicy, jobArgs,
|
|
259
312
|
env: process.env, budgetState, getActiveRunId: () => activeRunId,
|
|
313
|
+
sandboxProblem: () => podmanChecks?.problem() ?? null,
|
|
314
|
+
probeModel,
|
|
260
315
|
});
|
|
261
316
|
export { runningCount, track };
|
|
262
317
|
|
|
@@ -272,20 +327,24 @@ export const jobSchema = z.object({
|
|
|
272
327
|
verify_regression: z.boolean().optional().describe(
|
|
273
328
|
"implement only: after the diff passes `verification` and touches production files, temporarily revert just those production files, re-run the SAME verification profile (expected to fail without the fix), then restore them. A re-run that still PASSES proves no test would catch this regression, and the outcome is downgraded to NEEDS_REVIEW regardless of the worker's report -- never silently committed as done. This is the ONLY mechanism that catches a verification profile that passes for the wrong reason (a test-selection flag that accidentally excludes the changed file's own tests reports a real, honest, green run that never touched the diff -- exit-code checking alone cannot see the difference). Defaults to true whenever `verification` is set, since that gap is exactly what nomArmy's trust boundary claims to close; pass `false` explicitly to skip the doubled wall-clock cost (can matter on repos with thousands of tests) and accept the risk instead. No effect with no `verification` profile -- there is nothing to re-run. Ignored by scouts."
|
|
274
329
|
),
|
|
275
|
-
mode: z.enum(["scout", "implement", "decompose"]).default("implement").describe("implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
|
|
330
|
+
mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
|
|
276
331
|
base_ref: z.string().optional(),
|
|
277
|
-
timeout_seconds: z.number().int().min(30).max(1800).
|
|
278
|
-
reasoning: z.enum(
|
|
332
|
+
timeout_seconds: z.number().int().min(30).max(1800).optional().describe("Default 600; 1200 for a review scout (one with `reviews`, or an army role in the review phase), since a real security review read for the full 10 minutes and was cut off."),
|
|
333
|
+
reasoning: z.enum(THINKING_LEVELS).default("medium").describe("Thinking level passed to the worker model: off, minimal, low, medium, high, xhigh, adaptive, max or ultra (the higher ones only where the vendor offers them, e.g. xhigh on Codex models; a level the model lacks falls back once to its nearest supported level). Higher levels spend a subscription's usage limits faster. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
|
|
279
334
|
agent: z.string().regex(/^[A-Za-z0-9._-]{1,64}$/).optional().describe("Run on this agent from the operator's agents.yml, by name (e.g. \"codex\", \"grok\", \"local\"): the local model, a metered api key, or one person's subscription. Omit agent and army_role to use the local model. Refuses an unknown name, never falls back. Mutually exclusive with army_role. A subscription agent also requires on_behalf_of."),
|
|
280
335
|
model: z.string().regex(/^\S{1,200}$/).optional().describe("The model to run on the job's agent (an api or subscription agent), e.g. \"gpt-6-sol\". Overrides the role's model and the agent's default. Required when the role's model is \"auto\" or the agent has no default. The `army` tool lists each agent's models. Refused on the local agent, whose model `nomarmy model` sets."),
|
|
281
336
|
run_id: z.string().regex(/^run-[a-z0-9-]{1,80}$/).optional().describe("The /feature run this job belongs to (from run_start). Admission then enforces the run's limits (jobs, api spend, hours) and refuses an agent the run has paused after a vendor usage-limit error; the finished job is recorded into the run."),
|
|
282
|
-
report: z.enum(["brief", "standard", "full"]).optional().describe("How much the worker may report back, capped by its agent's tier: brief (today's local-sized report), standard (the default), full (the frontier ceiling: about 2k tokens for implement, 4k for a scout). The report lands in your own context and is re-read every later turn
|
|
337
|
+
report: z.enum(["brief", "standard", "full"]).optional().describe("How much the worker may report back, capped by its agent's tier: brief (today's local-sized report), standard (the default), full (the frontier ceiling: about 2k tokens for implement, 4k for a scout). An api or subscription scout defaults to full because its findings are the point; other jobs default to standard. The report lands in your own context and is re-read every later turn. No effect on the local model, whose caps are calibrated."),
|
|
338
|
+
stakes: z.enum(["normal", "high"]).optional().describe("implement: how much a mistake would cost, separate from how hard the work is. high for anything touching security or access control, personal or tenant data, data loss, money, or changes that can't be undone: a verification profile is then required, the revert check can't be turned off, and the job always comes back needing review until an independent review (a scout on another vendor with reviews: <job id>, or a judge on another vendor) has looked at it. A one-line auth change is simple and high-stakes."),
|
|
339
|
+
reviews: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/).optional().describe("scout: the job id this scout independently reviews, so the review is recorded against that job (nomarmy stats shows high-stakes jobs with and without one). Use a different vendor than the job's worker."),
|
|
283
340
|
commit_subject: z.string().max(200).optional().describe("implement: the subject line of the commit nomArmy makes on the worker branch, e.g. \"Keep held-back tables in the list_tables cache\". Defaults to the task's first sentence; the body is the worker's NOTE, and the job id is a trailer."),
|
|
284
341
|
army_role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Dispatch by army role (e.g. \"sr-dev\", \"security-analyst\"): nomArmy runs it on the agent this repo assigns to that role and puts the role's description at the top of the brief. Call the `army` tool first to see this repo's roles. Mutually exclusive with agent. Add on_behalf_of in case the role's agent is a subscription; it's ignored otherwise."),
|
|
342
|
+
confirm_over_limit: z.boolean().optional().describe("Override a reached usage limit: the General must ask the operator before resubmitting with confirm_over_limit: true, or send the job to another agent. nomArmy never sets it itself."),
|
|
285
343
|
on_behalf_of: z.string().min(1).max(254).optional().describe("Required when the job's agent is a subscription: must exactly match that agent's owner in agents.yml, or nomArmy refuses the job. A self-reported attestation, not an independently verified identity check -- nomArmy has no caller-identity boundary today, so what this guarantees is explicit, auditable intent and hard refusal on mismatch or omission, not cryptographic proof of who issued the call. Ignored for a local or api agent."),
|
|
286
344
|
evidence: z.string().max(maxEvidenceChars,
|
|
287
345
|
`Evidence exceeds the ${maxEvidenceChars}-character budget. This is for facts already resolved (e.g. with repo_evidence), not more description of the task -- if it needs more than this, resolve less per job or put the pointer (a path and line range) here instead of the material itself.`
|
|
288
346
|
).optional().describe("implement only: facts YOU already resolved (e.g. via repo_evidence) that the worker should trust and not re-derive -- exact signatures, call sites, line ranges, existing behavior. Cuts exploration that would otherwise burn the worker's own context budget on something you already know. Not a substitute for a clear objective and acceptance criteria."),
|
|
347
|
+
continue_from: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/).optional().describe("implement only: the job id of a retained, uncommitted implement job (partial, blocked, or failed verification) whose unfinished work this job should finish. The new worktree starts from that job's base with its changes in place, so brief only the correction; the finished whole, that work included, is verified and committed together. Use this instead of fixing a worker's files yourself, which would land them unverified. Leave base_ref out."),
|
|
289
348
|
worker_id: z.string().regex(/^[A-Za-z0-9._-]+$/).optional()
|
|
290
349
|
});
|
|
291
350
|
// A plain function, not jobSchema.superRefine: server.tool(...) registers
|
|
@@ -329,17 +388,18 @@ function jobArgs(args, workerId) {
|
|
|
329
388
|
return { task: args.task, acceptance: args.acceptance, verification: args.verification, mode: args.mode, baseRef: args.base_ref,
|
|
330
389
|
timeoutSeconds: args.timeout_seconds, profile: args.profile, reasoning: args.reasoning, pool: args.pool,
|
|
331
390
|
subscriptionWorker, onBehalfOf: args.on_behalf_of, model: args.model ?? null, reportSize: args.report ?? null, evidence: args.evidence,
|
|
332
|
-
verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), workerId };
|
|
391
|
+
verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), continueFrom: args.continue_from ?? null, stakes: args.stakes ?? null, reviews: args.reviews ?? null, workerId };
|
|
333
392
|
}
|
|
334
393
|
server.tool("local_worker", "Run one isolated local worker and wait for it. mode=implement edits in its own worktree and the coordinator commits only on a valid done report (or a recovered job that passed independent verification); failed or incomplete worktrees are retained. mode=scout answers a question from a read-only snapshot with mandatory [path:line] citations that nomArmy verifies and expands. mode=decompose (also read-only) proposes 2+ independent subtasks for a broad objective instead of one worker turn trying to do too much; the proposal is never auto-dispatched, review it and make a separate call with the subtasks you choose. Refuses under memory pressure or over capacity; use local_worker_start + local_worker_status to avoid blocking.", jobSchema.shape,
|
|
335
394
|
async rawArgs => {
|
|
336
395
|
const expanded = expandJobs([rawArgs]);
|
|
337
396
|
if (expanded.problems.length) return refusal(expanded.problems);
|
|
338
397
|
const [args] = expanded.jobs;
|
|
339
|
-
const { problems } = await admit([args]);
|
|
398
|
+
const { problems, admission } = await admit([args]);
|
|
340
399
|
if (problems.length) return refusal(problems);
|
|
341
400
|
const r = await launch(args).promise;
|
|
342
|
-
|
|
401
|
+
const refreshed = (admission.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
|
|
402
|
+
return toolText(refreshed.length ? `${coordinatorResult(r)}\n\n${refreshed.join("\n")}` : coordinatorResult(r), !r.ok);
|
|
343
403
|
});
|
|
344
404
|
server.tool("local_worker_start", "Start one worker or scout in the background and return immediately with a job_id. Poll it with local_worker_status (optionally long-polling with wait_seconds). Same admission rules as local_worker: refuses under memory pressure or when NOMARMY_MAX_WORKERS jobs are already running.", jobSchema.shape,
|
|
345
405
|
async rawArgs => {
|
|
@@ -349,14 +409,15 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
|
|
|
349
409
|
const { problems, admission } = await admit([args]);
|
|
350
410
|
if (problems.length) return refusal(problems);
|
|
351
411
|
const entry = launch(args);
|
|
352
|
-
return toolText(
|
|
412
|
+
return toolText(coordinatorJson({ started: true, jobId: entry.jobId, workerId: entry.workerId, mode: entry.mode, state: "running",
|
|
353
413
|
jobDir: path.join(jobsRoot, entry.jobId), timeoutSeconds: args.timeout_seconds,
|
|
354
414
|
poll: { tool: "local_worker_status", job_id: entry.jobId, wait_seconds: MAX_STATUS_WAIT_SECONDS },
|
|
415
|
+
wait: `nomarmy jobs --wait ${entry.jobId}`,
|
|
355
416
|
// This job's own lane and budget: a subscription job used to be
|
|
356
417
|
// reported with the local model's figures.
|
|
357
|
-
lane: jobLane(args), agent: args.agentName ?? "local", model: args.model ?? null,
|
|
418
|
+
lane: jobLane(args), agent: args.mode === "verify" ? null : args.agentName ?? "local", model: args.model ?? null,
|
|
358
419
|
...(args.run_id ? { run: runBrief(args.run_id) } : {}),
|
|
359
|
-
admission: { level: admission.level, notes: admission.reasons }, budgets: describeBudgets(budgetsForJob(args)) }
|
|
420
|
+
admission: { level: admission.level, notes: admission.reasons }, budgets: args.mode === "verify" ? null : describeBudgets(budgetsForJob(args)) }));
|
|
360
421
|
});
|
|
361
422
|
// A long poll must return inside the MCP client's own idle-timeout: it aborts
|
|
362
423
|
// a tool call after N seconds with no response or progress notification,
|
|
@@ -373,9 +434,10 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
|
|
|
373
434
|
// always crossed; 110s returns in-line with margin. Raise it only for a
|
|
374
435
|
// client that neither backgrounds nor times out that early.
|
|
375
436
|
export const MAX_STATUS_WAIT_SECONDS = Number.parseInt(process.env.NOMARMY_MAX_STATUS_WAIT_SECONDS ?? "", 10) || 110;
|
|
376
|
-
server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs).
|
|
377
|
-
job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false)
|
|
378
|
-
|
|
437
|
+
server.tool("local_worker_status", `Status of one job started by this server: phase (starting, worktree, worker, verification, commit, record, finished), elapsed time against its timeout, and the result once finished. wait_seconds long-polls up to that long for completion (max ${MAX_STATUS_WAIT_SECONDS}, to stay inside MCP client request timeouts; poll again for longer jobs). report=true returns just the worker's report (a scout's findings with their verified citations), the outcome, issues, verification and commit. full=true returns the complete execution record.`, {
|
|
438
|
+
job_id: z.string().min(1), wait_seconds: z.number().int().min(0).max(MAX_STATUS_WAIT_SECONDS).default(0), full: z.boolean().default(false),
|
|
439
|
+
report: z.boolean().default(false).describe("Just the report and nomArmy's verdict on it, without the rest of the record. Prefer this to full."),
|
|
440
|
+
}, async ({ job_id, wait_seconds, full, report }) => {
|
|
379
441
|
const jobId = path.basename(job_id), entry = activeJobs.get(jobId), jobDir = path.join(ensureJobsRoot(), jobId);
|
|
380
442
|
if (entry && !entry.settled && wait_seconds > 0) await Promise.race([entry.promise.catch(() => {}), sleep(wait_seconds * 1000)]);
|
|
381
443
|
const files = { status: readJson(path.join(jobDir, "status.json")), meta: readJson(path.join(jobDir, "metadata.json")), failure: readJson(path.join(jobDir, "failure.json")) };
|
|
@@ -386,15 +448,54 @@ server.tool("local_worker_status", `Status of one job started by this server: ph
|
|
|
386
448
|
summarize(entry, files, jobDir),
|
|
387
449
|
sleep(15000).then(() => summarize(entry, files, null)),
|
|
388
450
|
]);
|
|
389
|
-
if (summary.state === "running") return toolText(
|
|
390
|
-
if (entry?.error) return toolText(
|
|
391
|
-
if (
|
|
392
|
-
if (full &&
|
|
393
|
-
|
|
451
|
+
if (summary.state === "running") return toolText(coordinatorJson({ ...summary, jobDir, hint: `poll again with wait_seconds up to ${MAX_STATUS_WAIT_SECONDS}; lastTool/filesChangedLive are best-effort and may be absent early in a run` }));
|
|
452
|
+
if (entry?.error) return toolText(coordinatorJson({ ...summary, jobDir }), true);
|
|
453
|
+
if (report && files.meta) return toolText(coordinatorJson(reportView(files.meta)), summary.coordinatorStatus !== "complete");
|
|
454
|
+
if (full && entry?.result) return toolText(coordinatorResult(entry.result), !entry.result.ok);
|
|
455
|
+
if (full && files.meta) return toolText(coordinatorJson(files.meta), summary.coordinatorStatus !== "complete");
|
|
456
|
+
return toolText(coordinatorJson({ ...summary, jobDir, hint: entry?.result || files.meta ? "call again with report=true for the worker's report, or full=true for the complete record" : null }), summary.state === "orphaned" || summary.state === "failed");
|
|
394
457
|
});
|
|
458
|
+
// Set when this session's copy of nomArmy changed on disk after it started
|
|
459
|
+
// (nomarmy connect or update ran): shown first in army and capacity, and
|
|
460
|
+
// notified once, since only a restart of this session picks it up.
|
|
461
|
+
let restartNotified = false;
|
|
462
|
+
function currentRestartNotice() {
|
|
463
|
+
const notice = restartNotice({ serverFile: fileURLToPath(import.meta.url), startedAtMs: SERVER_STARTED_MS, runningVersion: VERSION });
|
|
464
|
+
if (notice && !restartNotified) { restartNotified = true; try { notify("nomArmy: restart this session", notice); } catch { /* best-effort */ } }
|
|
465
|
+
return notice;
|
|
466
|
+
}
|
|
467
|
+
const withRestartNotice = (value) => { const notice = currentRestartNotice(); return notice ? { restartNeeded: notice, ...value } : value; };
|
|
468
|
+
|
|
469
|
+
server.tool("stats", "What nomArmy's own job records show for this repository (or all repositories): jobs by mode, role and model; code committed; worker and job time; tokens and API spend; claim vs evidence (how often a \"done, tests pass\" report failed independent verification, or passed with tests that couldn't catch the change); what didn't complete; reviewer outcomes; and review flags. Every number comes from nomArmy's verified records, never a worker's report. Defects you find at integration aren't in the records: add your own count. Read-only.", {
|
|
470
|
+
since: z.string().max(40).optional().describe("Only jobs started since this: a date (2026-09-25) or an age (7d, 24h). Default: all."),
|
|
471
|
+
until: z.string().max(40).optional().describe("Only jobs started before this date."),
|
|
472
|
+
all_repos: z.boolean().optional().describe("Every repository this machine's nomArmy has run jobs for, not just this one."),
|
|
473
|
+
repo: z.string().max(400).optional().describe("Another repository, by path or folder name (\"senti\" matches rayson-senti if unambiguous)."),
|
|
474
|
+
role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Only jobs dispatched as this army role (e.g. sr-dev)."),
|
|
475
|
+
model: z.string().regex(/^\S{1,200}$/).optional().describe("Only jobs that ran on this model (e.g. grok-4.7)."),
|
|
476
|
+
format: z.enum(["text", "json"]).optional().describe("text (default) is the report; json is the raw numbers."),
|
|
477
|
+
details: z.boolean().optional().describe("The full report (volume, reviewers, flags, what didn't finish). Default is the one-screen summary: what nomArmy caught, high-stakes work needing review, the top routing tips, spend."),
|
|
478
|
+
}, async ({ since, until, all_repos, repo, role, model, format, details }) => {
|
|
479
|
+
try {
|
|
480
|
+
const records = loadJobRecords(jobsRoot);
|
|
481
|
+
let agentFor = () => null;
|
|
482
|
+
try { agentFor = agentLookup(agentsConfig().agents, agentProviderId); } catch { /* commands name <agent> */ }
|
|
483
|
+
const stats = computeStats(records, { repo: repo ? resolveRepo(records, repo) : all_repos ? null : projectDir, sinceMs: parseSince(since), untilMs: parseSince(until), role: role ?? null, model: model ?? null, agentFor });
|
|
484
|
+
return toolText(format === "json" ? JSON.stringify(stats, null, 2) : details ? formatStats(stats) : formatStatsSummary(stats));
|
|
485
|
+
} catch (error) { return toolText(error.message, true); }
|
|
486
|
+
});
|
|
487
|
+
|
|
488
|
+
server.tool("local_worker_stop", "Stop a running job's worker, for example one burning a frontier model's usage on the wrong track. Its worker ends within about 15 seconds, without the report-recovery call a timeout gets and without running verification; its worktree is kept uncommitted, so a new job with continue_from: <job_id> (and a cheaper model if you like) can finish the work. Works for a job started by any session. Refuses a job that isn't running or is already past its worker.", {
|
|
489
|
+
job_id: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/),
|
|
490
|
+
reason: z.string().max(300).optional().describe("Why it's being stopped; recorded in the job's issues."),
|
|
491
|
+
}, async ({ job_id, reason }) => {
|
|
492
|
+
const r = requestJobStop({ jobsRoot, jobId: job_id, reason: reason ?? null });
|
|
493
|
+
return toolText(r.message, !r.ok);
|
|
494
|
+
});
|
|
495
|
+
|
|
395
496
|
server.tool("local_worker_capacity", "What this host can take right now: context per nom and the brief/report budgets derived from it, memory pressure and whether another job would be admitted, and the jobs currently running. Read-only.", {}, async () => {
|
|
396
497
|
await budgetState.refresh();
|
|
397
|
-
return toolText(JSON.stringify(
|
|
498
|
+
return toolText(JSON.stringify(withRestartNotice(await displayedCapacity()), null, 2));
|
|
398
499
|
});
|
|
399
500
|
// The only way to know what `verification`/`union_verification`/
|
|
400
501
|
// `verify_regression` profile names are actually valid for this repo used to
|
|
@@ -468,20 +569,37 @@ server.tool("run_finish", "Close a /feature run as complete or stopped, with a o
|
|
|
468
569
|
try {
|
|
469
570
|
const run = finishRun(runsRoot, run_id, { status, summary });
|
|
470
571
|
if (activeRunId === run_id) activeRunId = null;
|
|
471
|
-
|
|
572
|
+
// What nomArmy verified in this run, ready for the pull request's description (lib/share.mjs).
|
|
573
|
+
let prBlock = null;
|
|
574
|
+
try { prBlock = shareMarkdown(computeStats(loadJobRecords(jobsRoot), { runId: run_id }), { scope: "this feature run" }); } catch { /* the totals stand without it */ }
|
|
575
|
+
return toolText(JSON.stringify({ id: run.id, status: run.status, ...runTotals(run), ...(prBlock ? { prBlock, prBlockNote: "Put prBlock in the pull request's description as it is: every number is from nomArmy's verified records." } : {}) }, null, 2));
|
|
472
576
|
} catch (error) { return toolText(error.message, true); }
|
|
473
577
|
});
|
|
474
578
|
server.tool("army", "Who you, the General, are and who you call for what in this repository: your fixed charter and the agent you're defined as, the army's workflow, then each role's description, phase (build, review, acceptance), suggested mode, and the agent it runs on, with which config layer set each value (global, project .nomarmy.yml, local .nomarmy.local.yml). Flags roles with no usable agent, and roles that share your model or subscription (not an independent review). Dispatch a role with `army_role`, or an agent directly with `agent`. Read-only, re-read on every call.", {}, async () => {
|
|
475
579
|
try {
|
|
476
580
|
const agents = agentsConfig().agents;
|
|
477
|
-
const
|
|
581
|
+
const usageRefresh = await refreshStaleOverLimitReadings(stateRoot);
|
|
582
|
+
const usageSnapshots = usageRefresh.snapshots;
|
|
583
|
+
const summary = describeArmy(currentArmy(), { agents, describeAgent, usageSnapshots, agentProviderId, usageRefreshError: usageRefresh.error, usageRefreshFailed: usageRefresh.failedProviders });
|
|
478
584
|
// Each agent's models, from OpenClaw's catalog, so the General can pick
|
|
479
585
|
// one for a role set to "auto". The catalog can lag a brand-new model.
|
|
480
586
|
const catalog = await modelCatalogReady();
|
|
587
|
+
// A model its vendor refused on a job is listed apart, so a role on
|
|
588
|
+
// "auto" isn't sent to it (the catalog lists what a plan may refuse).
|
|
589
|
+
retryRefusedModelsInBackground(stateRoot, { probeModel });
|
|
590
|
+
const refusals = modelRefusals(stateRoot);
|
|
481
591
|
summary.agents = Object.fromEntries(Object.entries(agents).map(([name, agent]) => {
|
|
482
592
|
const provider = agentProviderId(agent);
|
|
483
|
-
const
|
|
484
|
-
|
|
593
|
+
const listed = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
|
|
594
|
+
const models = listed.filter((m) => !refusals[`${provider}/${m}`]);
|
|
595
|
+
const refusedModels = listed.filter((m) => refusals[`${provider}/${m}`]);
|
|
596
|
+
const snapshot = usageSnapshots[provider];
|
|
597
|
+
const usage = snapshot ? (() => {
|
|
598
|
+
const status = usageStatus(snapshot);
|
|
599
|
+
const failed = usageRefresh.failedProviders.includes(provider);
|
|
600
|
+
return { level: status.level, text: `${usageDisplayText(status)}${failed ? `. ${usageRefresh.error}` : ""}` };
|
|
601
|
+
})() : null;
|
|
602
|
+
return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, ...(refusedModels.length ? { refusedModels } : {}), usage }];
|
|
485
603
|
}));
|
|
486
604
|
// A pinned model missing from the catalog isn't necessarily wrong:
|
|
487
605
|
// `army assign` proves an unlisted model with a real test call, and the
|
|
@@ -493,7 +611,13 @@ server.tool("army", "Who you, the General, are and who you call for what in this
|
|
|
493
611
|
role.modelNote = `${role.model} isn't in OpenClaw's catalog for ${role.agent}; \`army assign\` checked it with a real test call when it was set, and the catalog can lag new models. Use it as assigned; if a job reports "Unknown model", reassign.`;
|
|
494
612
|
}
|
|
495
613
|
}
|
|
496
|
-
|
|
614
|
+
// Routing suggestions from this repo's recent jobs (lib/suggestions.mjs):
|
|
615
|
+
// tell the operator about them; never apply one without their say-so.
|
|
616
|
+
try {
|
|
617
|
+
const list = recentSuggestions(loadJobRecords(jobsRoot), { projectDir, agentFor: agentLookup(agents, agentProviderId) });
|
|
618
|
+
if (list.length) summary.suggestions = { note: "From this repo's last 14 days of jobs. Tell the operator; change routing only with their say-so (nomarmy army assign).", items: list.map(({ level, title, evidence, command }) => ({ level, title, evidence, command })) };
|
|
619
|
+
} catch { /* suggestions are a bonus; the army summary stands without them */ }
|
|
620
|
+
return toolText(JSON.stringify(withRestartNotice(summary), null, 2));
|
|
497
621
|
} catch (error) {
|
|
498
622
|
return toolText(error.message, true);
|
|
499
623
|
}
|
|
@@ -503,7 +627,7 @@ server.tool("local_worker_config", "What this checkout's .nomarmy.yml defines --
|
|
|
503
627
|
return toolText(JSON.stringify(summary, null, 2), summary.valid === false);
|
|
504
628
|
});
|
|
505
629
|
server.tool("local_workers", "Run independent jobs (implement or scout) with bounded parallelism and wait for all of them. Every implement job receives its own branch, worktree, sandbox session, logs, validation, and coordinator-owned commit. This tool never merges any branch into the developer's branch. With auto_union: true, implement jobs that reach a valid outcome and touch non-overlapping files are additionally merged (git merge --no-ff) into ONE new integration branch -- a review artifact alongside the untouched per-job branches, still not the developer's branch, still reviewed and integrated explicitly. Jobs that overlap or did not finish validly are excluded from the union and reported individually exactly as without auto_union. For long batches prefer local_worker_start per job and poll.", {
|
|
506
|
-
jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (
|
|
630
|
+
jobs: z.array(jobSchema).min(1).max(8), max_parallel: z.number().int().min(1).max(8).optional().describe("A cap on how many of this batch run at once. Omit it: local jobs then use the local ceiling and api/subscription jobs theirs (`nomarmy config max-jobs`, default 4), with each agent's own max_concurrent on top. It used to default to the local ceiling, which ran an all-remote batch one job at a time."),
|
|
507
631
|
auto_union: z.boolean().default(false).describe(
|
|
508
632
|
"After all jobs finish, mechanically merge (git merge --no-ff) implement jobs that reached a valid outcome and touched non-overlapping files into ONE new integration branch for review -- never into the developer's branch. Overlapping or invalid-outcome jobs are excluded and still reported individually, unchanged. All jobs must share one base_ref (or omit it); it is resolved once, before any job starts, and forced onto every job so the union is provably rooted at a single base."
|
|
509
633
|
),
|
|
@@ -514,7 +638,7 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
|
|
|
514
638
|
const expanded = expandJobs(rawJobs);
|
|
515
639
|
if (expanded.problems.length) return refusal(expanded.problems);
|
|
516
640
|
const { jobs } = expanded;
|
|
517
|
-
const { problems } = await admit(jobs);
|
|
641
|
+
const { problems, admission } = await admit(jobs);
|
|
518
642
|
let forcedBase = null;
|
|
519
643
|
if (auto_union) {
|
|
520
644
|
const refs = [...new Set(jobs.map(j => j.base_ref).filter(Boolean))];
|
|
@@ -547,7 +671,7 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
|
|
|
547
671
|
// so it was invisible to both ceilings while it ran.
|
|
548
672
|
// A batch job waits for its agent's slot (up to its own timeout) rather
|
|
549
673
|
// than failing because an earlier job in the same batch holds it.
|
|
550
|
-
return trackInRun(j, track(jobId, { mode: j.mode, workerId, lane: jobLane(j), agent: j.agentName ?? null, runId: j.run_id ?? null, role: j.armyRole ?? null, model: j.model ?? null },
|
|
674
|
+
return trackInRun(j, track(jobId, { mode: j.mode, workerId, lane: jobLane(j), agent: j.agentName ?? null, runId: j.run_id ?? null, role: j.armyRole ?? null, model: j.model ?? null, label: jobLabel(j) },
|
|
551
675
|
withAgentSlot(j, jobId, () => executeJob({ ...jobArgs(effectiveJob, workerId), jobId }), { waitMs: (j.timeout_seconds ?? 600) * 1000 }))).promise;
|
|
552
676
|
}, { staggerMs: WORKER_START_STAGGER_MS });
|
|
553
677
|
indices.forEach((i, laneI) => { results[i] = laneResults[laneI]; });
|
|
@@ -578,8 +702,9 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
|
|
|
578
702
|
reviewRequired: results.filter(r => r.manifest?.reviewRequired).length,
|
|
579
703
|
jobs: results.map(r => ({ jobId: r.manifest.jobId, workerId: r.manifest.workerId, mode: r.manifest.mode, outcome: r.manifest.outcome || OUTCOMES.WORKER_FAILED, recovered: Boolean(r.manifest.recovered), status: r.manifest.coordinatorStatus || "failed", branch: r.manifest.branch, commit: r.manifest.commit?.sha || null, worktree: r.manifest.worktree, jobDir: r.jobDir })),
|
|
580
704
|
...(union ? { union } : {}) };
|
|
581
|
-
const unionSection = union ? `UNION\n\n${formatUnion(union)}\n\n` : "";
|
|
582
|
-
const
|
|
705
|
+
const unionSection = union ? `UNION\n\n${formatUnion(withWindowsPaths(union))}\n\n` : "";
|
|
706
|
+
const refreshed = (admission?.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
|
|
707
|
+
const text = `BATCH EXECUTION RECORD\n${coordinatorJson(summary)}\n\n${unionSection}WORKER RESULTS\n\n${results.map((r, i) => `===== WORKER ${i + 1} =====\n${coordinatorResult(r)}`).join("\n\n")}${refreshed.length ? `\n\n${refreshed.join("\n")}` : ""}`;
|
|
583
708
|
return toolText(text, results.some(r => !r.ok) || union?.status === "union_verification_failed" || union?.status === "union_error");
|
|
584
709
|
});
|
|
585
710
|
// No model, no sandbox, no tokens spent on a worker: the coordinator asks the
|
|
@@ -604,7 +729,7 @@ server.tool("local_worker_jobs", "List recent job records for review/recovery, i
|
|
|
604
729
|
if (status) return summarize(activeJobs.get(name) ?? null, { status, meta: null, failure: null }, dir);
|
|
605
730
|
return { jobId: name, state: "unknown" };
|
|
606
731
|
}));
|
|
607
|
-
return toolText(
|
|
732
|
+
return toolText(coordinatorJson(rows));
|
|
608
733
|
});
|
|
609
734
|
// The sandbox writes skill/guardrail files under .openclaw/ with permissions
|
|
610
735
|
// meant to stop the SANDBOXED AGENT from deleting them. On macOS, the
|
|
@@ -736,7 +861,7 @@ server.tool("local_worker_sweep", "Bulk-reap job worktrees/branches that are PRO
|
|
|
736
861
|
skipped.push({ jobId, reason: `removal failed: ${error.message}` });
|
|
737
862
|
}
|
|
738
863
|
}
|
|
739
|
-
return toolText(
|
|
864
|
+
return toolText(coordinatorJson({ examined: dirs.length, reapedCount: reaped.length, skippedCount: skipped.length, dryRun: dry_run, reaped, skipped }));
|
|
740
865
|
});
|
|
741
866
|
server.tool("local_worker_cleanup", "Remove a retained worker worktree and optionally its agent branch after Claude has reviewed/integrated or deliberately discarded it. Refuses to delete the current branch. A branch whose commits were cherry-picked (not merged) into the current branch -- nomArmy's own integration model -- is recognized as integrated by comparing PATCH CONTENT (git cherry), not git's own ancestry-only check, so a genuinely-integrated job's cleanup does not need force: true. Reserve force for a branch you are actually discarding unintegrated work from.", {
|
|
742
867
|
job_id: z.string().min(1), delete_branch: z.boolean().default(false), force: z.boolean().default(false)
|
|
@@ -791,6 +916,12 @@ if (isMain) {
|
|
|
791
916
|
// sqlglot, so verification could never pass), and a worker could edit its
|
|
792
917
|
// own worktree's copy to weaken the checks that judge it.
|
|
793
918
|
registerVerificationRunner(createVerificationRunner({ hostProjectDir: projectDir, loadConfig: () => loadConfig(projectDir) }));
|
|
919
|
+
// A healthy answer is reused for 20 seconds, so a batch doesn't ask Podman per job.
|
|
920
|
+
let podmanOkUntil = 0;
|
|
921
|
+
podmanChecks = {
|
|
922
|
+
problem: () => { if (Date.now() < podmanOkUntil) return null; const p = podmanProblem(); if (!p) podmanOkUntil = Date.now() + 20000; return p; },
|
|
923
|
+
vmStartedAt: () => podmanVmStartedAt(),
|
|
924
|
+
};
|
|
794
925
|
// Warm the budget from the profile or the running llama-server. Not awaited:
|
|
795
926
|
// admission refreshes it anyway, and a slow hardware probe must not delay
|
|
796
927
|
// the MCP handshake.
|
package/package.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "nomarmy",
|
|
3
|
-
"description": "
|
|
3
|
+
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.21",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|
|
@@ -21,14 +21,15 @@
|
|
|
21
21
|
"tag": "alpha"
|
|
22
22
|
},
|
|
23
23
|
"dependencies": {
|
|
24
|
-
"@modelcontextprotocol/sdk": "^1.
|
|
24
|
+
"@modelcontextprotocol/sdk": "^1.30.1",
|
|
25
25
|
"@secretlint/node": "^13.0.5",
|
|
26
26
|
"@secretlint/secretlint-rule-preset-recommend": "^13.0.5",
|
|
27
27
|
"yaml": "^2.5.0",
|
|
28
|
-
"zod": "^
|
|
28
|
+
"zod": "^4.6.5"
|
|
29
29
|
},
|
|
30
30
|
"scripts": {
|
|
31
|
-
"test": "node --test tests/*.test.mjs"
|
|
31
|
+
"test": "node --test tests/*.test.mjs",
|
|
32
|
+
"docs:harnesses": "node scripts/generate-harness-docs.mjs"
|
|
32
33
|
},
|
|
33
34
|
"bin": {
|
|
34
35
|
"nomarmy": "bin/nomarmy.mjs"
|
|
@@ -43,6 +44,7 @@
|
|
|
43
44
|
"scripts",
|
|
44
45
|
"policies",
|
|
45
46
|
"docker",
|
|
47
|
+
"harnesses",
|
|
46
48
|
"install.sh",
|
|
47
49
|
"e2e.sh",
|
|
48
50
|
"LICENSE",
|
package/playbooks/feature.md
CHANGED
|
@@ -12,10 +12,11 @@ You are the General. Build this feature end to end with nomArmy's army and come
|
|
|
12
12
|
|
|
13
13
|
Follow the army's workflow, calling only the roles the work needs:
|
|
14
14
|
|
|
15
|
+
0. **Routing.** The `army` tool's `suggestions` come from this repo's recent jobs (a role that keeps failing, a cheaper model doing as well, unreviewed high-stakes work). Tell the operator about any at the start and in your final summary; change routing only if they say so.
|
|
15
16
|
1. **Plan.** Scout the repo as needed (`repo_evidence` first; a scout only for research that would pull many files into your context). Write the plan into the run log: the outcome, acceptance criteria, the pieces, and which role gets each.
|
|
16
|
-
2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work.
|
|
17
|
-
3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Send what they find back to the builders as new, bounded jobs.
|
|
18
|
-
4. **Acceptance.** PO and stakeholder test end to end. Fix what they find the same way.
|
|
17
|
+
2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work. Mark a job `stakes: high` when a mistake would be costly (security or access control, personal or tenant data, data loss, money, anything irreversible), however small the change: that's separate from how hard it is.
|
|
18
|
+
3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Every `stakes: high` build job gets an independent review before you accept it: a scout on a different vendor than its worker, with `reviews: <that job id>`. Send what they find back to the builders as new, bounded jobs. A build job that comes back partial, blocked or failing verification is finished with `continue_from: <its job id>` and a brief of just the correction, never by fixing its files yourself: only verified work lands.
|
|
19
|
+
4. **Acceptance.** PO and stakeholder test end to end. Checks that only run existing tests use mode: verify; writing new e2e checks is still an implement job. Fix what they find the same way.
|
|
19
20
|
5. **Integrate.** Review every diff against nomArmy's verified record -- a worker's report is a claim, not evidence -- and bring the accepted work together on one branch. **Never merge into the developer's branch, and never push.** The finished state is a branch ready for the operator to review and merge.
|
|
20
21
|
|
|
21
22
|
## Decisions along the way
|
|
@@ -24,9 +25,11 @@ When you hit a choice you'd normally ask the operator about, don't stop: pick th
|
|
|
24
25
|
|
|
25
26
|
Stop and ask only for something irreversible or outside this repository: merging or pushing, deploying, anything needing cloud or production credentials, deleting data, or changing another repository.
|
|
26
27
|
|
|
28
|
+
A gate's result is not yours to reinterpret. When a check the plan or the operator set (an exact expected output, a count, a checksum, a contract) doesn't match, that's a failure, even when the difference looks cosmetic: whitespace from BSD versus GNU `wc`, line endings, ordering, a trailing newline. Don't judge it "equivalent" and pass it. Either fix the check so it compares what was meant (normalize both sides, and write that change into the run log as a decision), or send the mismatch back as a failed job. A gate the General can wave through is no gate.
|
|
29
|
+
|
|
27
30
|
## Watching jobs
|
|
28
31
|
|
|
29
|
-
|
|
32
|
+
Prefer `local_worker_start`. Right after starting a job, if your coordinator can run a background command, run `nomarmy jobs --wait <job_id>` in the background so you're told the moment it finishes and can tell the operator. Otherwise poll `local_worker_status` with the longest `wait_seconds` it allows. For several jobs at once, `nomarmy jobs --wait <id> <id> ...` exits when every one has finished. For a whole run, use `nomarmy jobs --events --until-done --run <run-id>`. Never use unscoped `--until-done` when other sessions may have jobs. The plain `nomarmy jobs --events` stream never exits on its own while jobs run, so don't run it as a background command (you'd only hear when it exits); use it only with a monitor that wakes on each line. `run_status` lists the run's running jobs as well as finished ones. The operator gets a desktop notification whenever a job finishes and whenever the run crosses a limit, so you don't need to relay each one.
|
|
30
33
|
|
|
31
34
|
## Limits
|
|
32
35
|
|
|
@@ -40,4 +43,4 @@ The `logPath` from `run_start`. Markdown, updated after every phase: the plan; e
|
|
|
40
43
|
|
|
41
44
|
## When it's done
|
|
42
45
|
|
|
43
|
-
Call `run_finish` (`complete` or `stopped`),
|
|
46
|
+
Call `run_finish` (`complete` or `stopped`). Its result has a `prBlock`: when you or the operator open a pull request for the run's branch, put it in the description as it is (every number is from nomArmy's verified records). Then report in one message: what was built and on which branch; what each role found and how it was resolved; the decisions made on the operator's behalf; test and verification results; and the run's cost from `run_status` (jobs and api spend per agent). If a push-notification tool is available, notify the operator that the run finished or stopped.
|
|
@@ -57,7 +57,9 @@ else
|
|
|
57
57
|
if [[ -n "$REMOTE_MODEL" && "$REMOTE_MODEL" != "${NOMARMY_WORKER_MODEL:-}" ]]; then
|
|
58
58
|
# Jobs ask for NOMARMY_WORKER_MODEL, so record the name this server
|
|
59
59
|
# actually serves (`nomarmy connect`, run next by install.sh, reads it).
|
|
60
|
-
|
|
60
|
+
# Your settings file, so an update can't reset it (lib/user-config.mjs).
|
|
61
|
+
COMMON="$(nomarmy_user_config_dir)/common.env"
|
|
62
|
+
mkdir -p "$(dirname "$COMMON")"; touch "$COMMON"
|
|
61
63
|
for key in NOMARMY_MODEL_ALIAS NOMARMY_WORKER_MODEL; do
|
|
62
64
|
if grep -q "^$key=" "$COMMON"; then
|
|
63
65
|
KEY="$key" VALUE="$REMOTE_MODEL" node -e 'const fs=require("fs"),f=process.argv[1];fs.writeFileSync(f,fs.readFileSync(f,"utf8").replace(new RegExp(`^${process.env.KEY}=.*$`,"m"),`${process.env.KEY}=${process.env.VALUE}`))' "$COMMON"
|
|
@@ -66,7 +68,7 @@ else
|
|
|
66
68
|
fi
|
|
67
69
|
done
|
|
68
70
|
export NOMARMY_MODEL_ALIAS="$REMOTE_MODEL" NOMARMY_WORKER_MODEL="$REMOTE_MODEL"
|
|
69
|
-
echo "==> The server serves '$REMOTE_MODEL'; recorded it in
|
|
71
|
+
echo "==> The server serves '$REMOTE_MODEL'; recorded it in $COMMON"
|
|
70
72
|
fi
|
|
71
73
|
# llama-server reports the context of one slot, which is one nom's share.
|
|
72
74
|
REMOTE_CTX="$(curl -fsS --max-time 5 "$SERVER/props" | node -e 'let s="";process.stdin.on("data",d=>s+=d).on("end",()=>{try{const n=JSON.parse(s).default_generation_settings?.n_ctx;if(Number.isInteger(n)&&n>0)process.stdout.write(String(n))}catch{}})' || true)"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { HARNESS_ROOT, loadHarnesses } from "../lib/harnesses.mjs";
|
|
5
|
+
|
|
6
|
+
export const HARNESS_DOCS = fileURLToPath(new URL("../docs/harnesses.md", import.meta.url));
|
|
7
|
+
const cell = (value) => String(value).replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll("|", "|").replaceAll("\n", " ").replaceAll("\r", " ");
|
|
8
|
+
|
|
9
|
+
export function generateHarnessDocs(root = HARNESS_ROOT) {
|
|
10
|
+
const { harnesses, problems } = loadHarnesses(root);
|
|
11
|
+
if (problems.length) throw new Error(problems.map(({ name, reason }) => `${name}: ${reason}`).join("\n"));
|
|
12
|
+
const rows = Object.values(harnesses).sort((a, b) => a.name.localeCompare(b.name)).map((spec) => {
|
|
13
|
+
const detects = spec.detect.map((rule) => Object.entries(rule).map(([kind, value]) => `${kind}: ${value}`).join(", ")).join("; ") || "none";
|
|
14
|
+
const requires = Object.entries(spec.requires).map(([key, value]) => `${key}: ${value}`).join(", ") || "none";
|
|
15
|
+
return `| [${spec.name}](https://github.com/rayson-tech/nomarmy/blob/main/harnesses/${spec.name}/README.md) | ${cell(spec.summary)} | ${cell(detects)} | ${spec.network} | ${cell(requires)} |`;
|
|
16
|
+
});
|
|
17
|
+
return `<!-- Generated by npm run docs:harnesses. Do not edit by hand. -->
|
|
18
|
+
# Harnesses
|
|
19
|
+
|
|
20
|
+
A harness is a data-only folder describing detection, image layers, and proposed verification profiles. Matched harnesses compose one cached image for workers and verification: toolchains first, then Python, Node, and declarative layers, respecting \`after\`. Tags hash the recipe and copied dependency files. Verification profiles remain proposals. Repositories can explicitly enable harnesses with \`harnesses: [names]\` in \`.nomarmy.yml\`; service networks are used only for nomArmy verification.
|
|
21
|
+
|
|
22
|
+
- **none**: offline verification (default).
|
|
23
|
+
- **services**: fake services on a private network with no route out.
|
|
24
|
+
- **allowlist**: verification-only access to operator-approved hosts, with dedicated test tenants and throwaway credentials, never production (planned).
|
|
25
|
+
|
|
26
|
+
Workers always remain offline. See the [plan](plans/2026-09-25-sandbox-dependencies.md) and [Adding a harness](../CONTRIBUTING.md#adding-a-harness).
|
|
27
|
+
|
|
28
|
+
| Harness | Summary | Detects | Network | Requires |
|
|
29
|
+
|---------|---------|---------|---------|----------|
|
|
30
|
+
${rows.join("\n")}
|
|
31
|
+
`;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function checkHarnessDocs(root = HARNESS_ROOT, output = HARNESS_DOCS) {
|
|
35
|
+
if (fs.readFileSync(output, "utf8") !== generateHarnessDocs(root)) {
|
|
36
|
+
throw new Error("Harness docs are stale. Run npm run docs:harnesses.");
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
41
|
+
fs.writeFileSync(HARNESS_DOCS, generateHarnessDocs());
|
|
42
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
// install.sh has already installed nomArmy's dependencies. Nothing is ready
|
|
2
|
+
// until the configured vendors and read-only OpenClaw diagnostics pass.
|
|
3
|
+
import { loadAgents } from "../lib/agents.mjs";
|
|
4
|
+
import { globalConfigDir } from "../lib/army.mjs";
|
|
5
|
+
import { ensureOpenClawOnPath } from "../lib/openclaw-path.mjs";
|
|
6
|
+
import { configuredSubscriptionVendors, openclawInstallPlan, repairOpenclaw, runOpenclawCommand, PINNED_OPENCLAW_VERSION } from "../lib/openclaw-install.mjs";
|
|
7
|
+
|
|
8
|
+
ensureOpenClawOnPath();
|
|
9
|
+
const command = process.env.NOMARMY_OPENCLAW_CMD || "openclaw";
|
|
10
|
+
const version = runOpenclawCommand(command, ["--version"]);
|
|
11
|
+
const [major, minor] = process.versions.node.split(".").map(Number);
|
|
12
|
+
if (openclawInstallPlan(version.ok ? version.stdout : null).length &&
|
|
13
|
+
!((major === 24 && minor >= 16) || (major === 26 && minor >= 1) || major > 26)) {
|
|
14
|
+
console.error(`OpenClaw ${PINNED_OPENCLAW_VERSION} needs Node 24.16+ or 26.1+. Upgrade Node first.`);
|
|
15
|
+
process.exitCode = 1;
|
|
16
|
+
} else {
|
|
17
|
+
const prefixIndex = process.argv.indexOf("--prefix");
|
|
18
|
+
const result = await repairOpenclaw({
|
|
19
|
+
command, yes: true, prefix: prefixIndex < 0 ? null : process.argv[prefixIndex + 1],
|
|
20
|
+
vendors: configuredSubscriptionVendors(loadAgents(globalConfigDir()).agents),
|
|
21
|
+
});
|
|
22
|
+
process.exitCode = result.ok ? 0 : 1;
|
|
23
|
+
}
|