nomarmy 0.1.0-alpha.4 → 0.1.0-alpha.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/admission.mjs +36 -17
- package/lib/army.mjs +1 -1
- package/lib/connect.mjs +31 -8
- package/lib/coordinator-instructions.mjs +1 -0
- package/lib/execute.mjs +33 -2
- package/lib/health.mjs +34 -5
- package/lib/job-format.mjs +8 -0
- package/lib/outcomes.mjs +6 -0
- package/lib/runs.mjs +7 -1
- package/lib/server-context.mjs +21 -1
- package/lib/verification-flow.mjs +9 -1
- package/lib/verify.mjs +2 -0
- package/mcp/server.mjs +15 -5
- package/package.json +1 -1
- package/playbooks/feature.md +1 -1
package/lib/admission.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import fs from "node:fs";
|
|
2
2
|
import path from "node:path";
|
|
3
3
|
import { executionMode } from "./execution.mjs";
|
|
4
|
+
import { loadConfig } from "./config.mjs";
|
|
4
5
|
import { clampInt } from "./budget-state.mjs";
|
|
5
6
|
import { checkBrief, assessAdmission, describeBudgets } from "./budget.mjs";
|
|
6
7
|
import { parseStatusPorcelainZ, isRuntimeJunk } from "./git-record.mjs";
|
|
@@ -9,7 +10,8 @@ import { readOpenClawTranscriptTail } from "./transcript.mjs";
|
|
|
9
10
|
import { readClaudeSessionTranscript } from "./claude-transcript.mjs";
|
|
10
11
|
import { notify } from "./notify.mjs";
|
|
11
12
|
import { readCodexRateLimits, recordUsageSnapshot, readUsageSnapshots, usageStatus } from "./usage-limits.mjs";
|
|
12
|
-
import {
|
|
13
|
+
import { projectDirProblem } from "./server-context.mjs";
|
|
14
|
+
import { recentModelRefusal, refusedModelIn, recordModelRefusal, clearModelRefusal } from "./health.mjs";
|
|
13
15
|
import { writeLease, removeLease, liveLeases, liveSlots, acquireSlot } from "./slots.mjs";
|
|
14
16
|
import { loadRun, runTotals, runAdmissionProblems, recordRunJob, detectUsageLimit } from "./runs.mjs";
|
|
15
17
|
import { agentProviderId, hostToolsImplementProblem } from "./agents.mjs";
|
|
@@ -22,7 +24,7 @@ import { policyAdmissionProblems } from "./outcome.mjs";
|
|
|
22
24
|
// Claude or Codex job took llama-server's only slot and blocked local work
|
|
23
25
|
// it never competed with -- reported from a real Senti run.
|
|
24
26
|
export function jobLane(job) {
|
|
25
|
-
return job.pool || job.subscription_worker ? "remote" : "local";
|
|
27
|
+
return job.mode === "verify" || job.pool || job.subscription_worker ? "remote" : "local";
|
|
26
28
|
}
|
|
27
29
|
|
|
28
30
|
// A static, operator-declared ceiling on how many remote jobs (api and
|
|
@@ -114,7 +116,7 @@ export function createJobRuntime(deps) {
|
|
|
114
116
|
* batch queue for a slot instead of failing.
|
|
115
117
|
*/
|
|
116
118
|
function withAgentSlot(args, jobId, fn, { waitMs = 0 } = {}) {
|
|
117
|
-
const max = args.agentName ? agentMaxConcurrent(args.agentName) : null;
|
|
119
|
+
const max = args.mode !== "verify" && args.agentName ? agentMaxConcurrent(args.agentName) : null;
|
|
118
120
|
if (!max) return fn();
|
|
119
121
|
return (async () => {
|
|
120
122
|
const slot = await acquireSlot(slotsRoot, args.agentName, max, { jobId, waitMs });
|
|
@@ -148,8 +150,14 @@ export function createJobRuntime(deps) {
|
|
|
148
150
|
} catch { /* Usage telemetry must never affect the job result. */ }
|
|
149
151
|
if (!entry.lane) return; // only tracked jobs, never internal helpers
|
|
150
152
|
const m = result?.manifest ?? {};
|
|
153
|
+
// Remember a model its vendor refused, until something on it works.
|
|
154
|
+
try {
|
|
155
|
+
const refused = refusedModelIn(m);
|
|
156
|
+
if (refused) recordModelRefusal(stateRoot, refused, String(m.workerError ?? m.worker?.error ?? "").split("\n")[0].slice(0, 300) || null);
|
|
157
|
+
else if (m.worker?.provider && m.worker?.model && !m.workerError && !error) clearModelRefusal(stateRoot, `${m.worker.provider}/${m.worker.model}`);
|
|
158
|
+
} catch { /* best-effort, like the usage reading */ }
|
|
151
159
|
const outcome = error ? "failed" : String(m.outcome ?? (result?.ok ? "done" : "finished")).toLowerCase().replace(/_/g, " ");
|
|
152
|
-
const who = entry.agent ? `${entry.agent}${entry.model ? `/${entry.model}` : ""}` : "local model";
|
|
160
|
+
const who = entry.mode === "verify" ? "verification runner" : entry.agent ? `${entry.agent}${entry.model ? `/${entry.model}` : ""}` : "local model";
|
|
153
161
|
const took = Math.round((Date.now() - Date.parse(entry.startedAt)) / 60000);
|
|
154
162
|
const ok = !error && (result?.ok || m.coordinatorStatus === "complete");
|
|
155
163
|
notify(`nomArmy: ${entry.role ?? entry.mode ?? "job"} ${ok ? "done" : outcome}`, `${entry.workerId ?? entry.jobId} on ${who}: ${outcome} after ${took}m. ${ok ? "Ready for the General's review." : "Needs a look."}`);
|
|
@@ -173,12 +181,14 @@ export function createJobRuntime(deps) {
|
|
|
173
181
|
};
|
|
174
182
|
}
|
|
175
183
|
async function admit(jobs) {
|
|
176
|
-
await deps.budgetState.refresh();
|
|
177
|
-
if (jobs.some((j) => jobLane(j) === "remote")) await modelCatalogReady();
|
|
184
|
+
if (jobs.some(j => j.mode !== "verify")) await deps.budgetState.refresh();
|
|
185
|
+
if (jobs.some((j) => j.mode !== "verify" && jobLane(j) === "remote")) await modelCatalogReady();
|
|
178
186
|
const problems = [];
|
|
187
|
+
const notARepo = (deps.projectDirProblem ?? projectDirProblem)(projectDir);
|
|
188
|
+
if (notARepo) problems.push(notARepo);
|
|
179
189
|
const snapshots = readUsageSnapshots(stateRoot);
|
|
180
190
|
jobs.forEach((j, i) => {
|
|
181
|
-
if (!j.agentName || j.confirm_over_limit === true) return;
|
|
191
|
+
if (j.mode === "verify" || !j.agentName || j.confirm_over_limit === true) return;
|
|
182
192
|
let provider;
|
|
183
193
|
try { provider = agentProviderId(agentsConfig().agents[j.agentName]); } catch { return; }
|
|
184
194
|
const snapshot = snapshots[provider];
|
|
@@ -195,6 +205,13 @@ export function createJobRuntime(deps) {
|
|
|
195
205
|
// (see budgetsForSubscriptionWorker) -- there's no "which entry" unknown
|
|
196
206
|
// the way a weighted pool has, since the name given IS the entry.
|
|
197
207
|
jobs.forEach((j, i) => {
|
|
208
|
+
if (j.mode === "verify") {
|
|
209
|
+
try {
|
|
210
|
+
const profiles = Object.keys(loadConfig(projectDir)?.config?.verification ?? {});
|
|
211
|
+
if (!j.verification || !profiles.includes(j.verification)) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}verify requires a verification profile from .nomarmy.yml; available profiles: ${profiles.join(", ") || "(none)"}`);
|
|
212
|
+
} catch (error) { problems.push(error.message); }
|
|
213
|
+
return;
|
|
214
|
+
}
|
|
198
215
|
const jobBudgets = budgetsForJob(j);
|
|
199
216
|
for (const p of checkBrief(j, jobBudgets)) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
|
|
200
217
|
});
|
|
@@ -213,6 +230,7 @@ export function createJobRuntime(deps) {
|
|
|
213
230
|
// clean, so a missing on_behalf_of is never reported twice in two
|
|
214
231
|
// different shapes.
|
|
215
232
|
jobs.forEach((j, i) => {
|
|
233
|
+
if (j.mode === "verify") return;
|
|
216
234
|
const fieldProblems = subscriptionJobFieldProblems(j);
|
|
217
235
|
for (const p of fieldProblems) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
|
|
218
236
|
if (fieldProblems.length === 0 && j.on_behalf_of) {
|
|
@@ -228,7 +246,7 @@ export function createJobRuntime(deps) {
|
|
|
228
246
|
// Claude CLI) isn't bounded by the sandbox, so it's refused unless that
|
|
229
247
|
// agent says allow_host_tools (lib/agents.mjs). Scouts and reviews still run.
|
|
230
248
|
jobs.forEach((j, i) => {
|
|
231
|
-
if (!j.agentName || (j.mode ?? "implement") !== "implement") return;
|
|
249
|
+
if (j.mode === "verify" || !j.agentName || (j.mode ?? "implement") !== "implement") return;
|
|
232
250
|
let problem = null;
|
|
233
251
|
try { problem = hostToolsImplementProblem(j.agentName, agentsConfig().agents[j.agentName]); } catch { return; }
|
|
234
252
|
if (problem) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}${problem}`);
|
|
@@ -236,7 +254,7 @@ export function createJobRuntime(deps) {
|
|
|
236
254
|
// A model its vendor refused on a job today, with nothing working on it
|
|
237
255
|
// since, isn't sent another job (lib/health.mjs recentModelRefusal).
|
|
238
256
|
jobs.forEach((j, i) => {
|
|
239
|
-
if (!j.agentName || !j.model) return;
|
|
257
|
+
if (j.mode === "verify" || !j.agentName || !j.model) return;
|
|
240
258
|
let provider = null;
|
|
241
259
|
try { provider = agentProviderId(agentsConfig().agents[j.agentName]); } catch { return; }
|
|
242
260
|
if (!provider) return;
|
|
@@ -247,7 +265,7 @@ export function createJobRuntime(deps) {
|
|
|
247
265
|
// queue for its slot at launch instead (withAgentSlot's waitMs).
|
|
248
266
|
if (jobs.length === 1) {
|
|
249
267
|
const [j] = jobs;
|
|
250
|
-
const max = j.agentName ? agentMaxConcurrent(j.agentName) : null;
|
|
268
|
+
const max = j.mode !== "verify" && j.agentName ? agentMaxConcurrent(j.agentName) : null;
|
|
251
269
|
const held = max ? liveSlots(slotsRoot, j.agentName) : 0;
|
|
252
270
|
if (max && held >= max) problems.push(`not admitted (capacity): agent "${j.agentName}" already has ${held} job(s) running across this machine's nomArmy sessions, at its max_concurrent of ${max}`);
|
|
253
271
|
}
|
|
@@ -258,7 +276,7 @@ export function createJobRuntime(deps) {
|
|
|
258
276
|
const run = loadRun(runsRoot, j.run_id);
|
|
259
277
|
if (run.repo !== projectDir) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}run "${run.id}" belongs to ${run.repo}, not this repository`);
|
|
260
278
|
const running = liveLeases(leasesRoot, { runId: run.id }).length + jobs.slice(0, i).filter((o) => o.run_id === run.id).length;
|
|
261
|
-
for (const p of runAdmissionProblems(run, { agentName: j.agentName ?? "local", running })) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
|
|
279
|
+
for (const p of runAdmissionProblems(run, { agentName: j.mode === "verify" ? null : j.agentName ?? "local", running })) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
|
|
262
280
|
} catch (error) { problems.push(jobs.length > 1 ? `job ${i + 1}: ${error.message}` : error.message); }
|
|
263
281
|
});
|
|
264
282
|
// Slot capacity only concerns local jobs: a remote job's inference runs
|
|
@@ -271,7 +289,7 @@ export function createJobRuntime(deps) {
|
|
|
271
289
|
const anyLocal = jobs.some((j) => jobLane(j) === "local");
|
|
272
290
|
const admission = anyLocal
|
|
273
291
|
? assessAdmission({ hardware: admissionHardware(), runningJobs: runningCount("local"), slots: deps.budgetState.contextInfo.slots, maxWorkers: currentMaxWorkers() })
|
|
274
|
-
: assessAdmission({ hardware: admissionHardware(), runningJobs: 0, slots: null, maxWorkers: Infinity });
|
|
292
|
+
: assessAdmission({ hardware: jobs.every(j => j.mode === "verify") ? null : admissionHardware(), runningJobs: 0, slots: null, maxWorkers: Infinity });
|
|
275
293
|
if (!admission.admit) problems.push(...admission.reasons.map(r => `not admitted (${admission.level}): ${r}`));
|
|
276
294
|
if (jobs.some((j) => jobLane(j) === "remote")) {
|
|
277
295
|
const remoteCeiling = currentMaxPoolWorkers(), runningRemote = runningCount("remote");
|
|
@@ -302,17 +320,17 @@ export function createJobRuntime(deps) {
|
|
|
302
320
|
*/
|
|
303
321
|
function recordJobInRun(args, jobId, result, error = null) {
|
|
304
322
|
if (!args.run_id) return;
|
|
305
|
-
const kind = args.pool ? "api" : args.subscription_worker ? "subscription" : "local";
|
|
323
|
+
const kind = args.mode === "verify" ? "verify" : args.pool ? "api" : args.subscription_worker ? "subscription" : "local";
|
|
306
324
|
const m = result?.manifest ?? {};
|
|
307
325
|
const errorLines = [m.worker?.error, error?.message,
|
|
308
326
|
...String(m.workerError ?? "").split(/\r?\n/).filter((l) => /error|limit|429/i.test(l))].filter(Boolean).join("\n");
|
|
309
|
-
const usageLimit = kind === "local" ? null : detectUsageLimit(errorLines);
|
|
327
|
+
const usageLimit = (kind === "local" || kind === "verify") ? null : detectUsageLimit(errorLines);
|
|
310
328
|
try {
|
|
311
329
|
const before = runTotals(loadRun(runsRoot, args.run_id)).warnings;
|
|
312
330
|
const updated = recordRunJob(runsRoot, args.run_id, {
|
|
313
|
-
jobId, agent: args.agentName ?? "local", kind, model: args.model ?? null, role: args.armyRole ?? null, mode: args.mode,
|
|
314
|
-
outcome: m.outcome ?? (error ? "ERROR" : null), costUsd: m.metrics?.worker_cost_usd ?? null,
|
|
315
|
-
tokens: m.metrics?.worker_tokens_total ?? null, usageLimit,
|
|
331
|
+
jobId, agent: kind === "verify" ? null : args.agentName ?? "local", kind, model: kind === "verify" ? null : args.model ?? null, role: args.armyRole ?? null, mode: args.mode,
|
|
332
|
+
outcome: m.outcome ?? (error ? "ERROR" : null), costUsd: kind === "verify" ? 0 : m.metrics?.worker_cost_usd ?? null,
|
|
333
|
+
tokens: kind === "verify" ? 0 : m.metrics?.worker_tokens_total ?? null, usageLimit,
|
|
316
334
|
});
|
|
317
335
|
// A limit crossed or an agent paused by this job is worth interrupting for.
|
|
318
336
|
const fresh = runTotals(updated).warnings.filter((w) => !before.includes(w) && /OVER|paused/.test(w));
|
|
@@ -380,6 +398,7 @@ export function createJobRuntime(deps) {
|
|
|
380
398
|
timeoutSeconds: status?.timeoutSeconds ?? null, coordinatorStatus: meta?.coordinatorStatus ?? null, outcome: meta?.outcome ?? null,
|
|
381
399
|
reviewRequired: meta?.reviewRequired ?? null, issues: (meta?.issues ?? []).slice(0, 6), worktree: meta?.worktree ?? null, branch: meta?.branch ?? null,
|
|
382
400
|
commit: meta?.commit?.sha ?? null, scout: meta?.scout ? { supported: meta.scout.supported, unsupported: meta.scout.unsupported } : null };
|
|
401
|
+
if (meta?.mode === "verify") Object.assign(out, { verification: meta.verification, baseRef: meta.baseRef, baseSha: meta.baseSha });
|
|
383
402
|
if (entry && !entry.settled) out.state = "running";
|
|
384
403
|
else if (entry?.error) { out.state = "failed"; out.error = String(entry.error.message ?? entry.error).split("\n")[0]; }
|
|
385
404
|
else if (meta) out.state = "finished";
|
package/lib/army.mjs
CHANGED
|
@@ -404,7 +404,7 @@ export function describeArmy(loaded, { agents = {}, describeAgent = null, usageS
|
|
|
404
404
|
roles,
|
|
405
405
|
layers: loaded.layers,
|
|
406
406
|
howToDispatch: Object.keys(roles).length
|
|
407
|
-
? "Pass army_role: \"<role>\" on a job (plus on_behalf_of when its agent is a subscription). The role's description goes at the top of the brief; set `mode` on the job yourself, the role's is only a suggestion. A role with model \"auto\" needs model on the job, picked from that agent's models below; any job may pass model to override the role's. Use agent: \"<name>\" instead to pick an agent directly."
|
|
407
|
+
? "Pass army_role: \"<role>\" on a job (plus on_behalf_of when its agent is a subscription). The role's description goes at the top of the brief; set `mode` on the job yourself, the role's is only a suggestion. A role with model \"auto\" needs model on the job, picked from that agent's models below (never its refusedModels: its vendor turned those down on a job); any job may pass model to override the role's. Use agent: \"<name>\" instead to pick an agent directly."
|
|
408
408
|
: "No army is configured. Run `nomarmy army init` for the default roster, or dispatch with agent: \"<name>\".",
|
|
409
409
|
};
|
|
410
410
|
}
|
package/lib/connect.mjs
CHANGED
|
@@ -45,8 +45,10 @@ export function defaultInstallDir() {
|
|
|
45
45
|
// `<!-- nomarmy:... -->` marker; a same-named file WITHOUT it is the
|
|
46
46
|
// operator's own and is never overwritten, only reported as skipped.
|
|
47
47
|
// Claude Code ~/.claude/commands/feature.md -> /feature <request>
|
|
48
|
-
// Codex ~/.
|
|
49
|
-
//
|
|
48
|
+
// Codex ~/.agents/skills/nomarmy-feature/SKILL.md (where the Codex
|
|
49
|
+
// CLI, desktop app and IDE extension all look for user
|
|
50
|
+
// skills; older nomArmy put it in $CODEX_HOME/skills, which
|
|
51
|
+
// connectCodex cleans up)
|
|
50
52
|
// Cursor ~/.cursor/commands/feature.md (Cursor's documented
|
|
51
53
|
// user-level commands directory; not verified against a
|
|
52
54
|
// real install here)
|
|
@@ -63,7 +65,7 @@ export function defaultCommandDirs(env = process.env) {
|
|
|
63
65
|
const home = os.homedir();
|
|
64
66
|
return {
|
|
65
67
|
claude: env.NOMARMY_CLAUDE_COMMANDS_DIR ? path.resolve(env.NOMARMY_CLAUDE_COMMANDS_DIR) : path.join(home, ".claude", "commands"),
|
|
66
|
-
codex: env.NOMARMY_CODEX_SKILLS_DIR ? path.resolve(env.NOMARMY_CODEX_SKILLS_DIR) : path.join(
|
|
68
|
+
codex: env.NOMARMY_CODEX_SKILLS_DIR ? path.resolve(env.NOMARMY_CODEX_SKILLS_DIR) : path.join(home, ".agents", "skills"),
|
|
67
69
|
cursor: env.NOMARMY_CURSOR_COMMANDS_DIR ? path.resolve(env.NOMARMY_CURSOR_COMMANDS_DIR) : path.join(home, ".cursor", "commands"),
|
|
68
70
|
};
|
|
69
71
|
}
|
|
@@ -334,15 +336,32 @@ export function connectClaude({ nomarmyRoot, installDir = defaultInstallDir(), r
|
|
|
334
336
|
/**
|
|
335
337
|
* @param {{ nomarmyRoot: string, installDir?: string, run?: Function }} input
|
|
336
338
|
*/
|
|
337
|
-
export function connectCodex({ nomarmyRoot, installDir = defaultInstallDir(), run = defaultRun, skillsDir = defaultCommandDirs().codex }) {
|
|
339
|
+
export function connectCodex({ nomarmyRoot, installDir = defaultInstallDir(), run = defaultRun, configDir = globalConfigDir(), skillsDir = defaultCommandDirs().codex, legacySkillsDir = path.join(process.env.CODEX_HOME || path.join(os.homedir(), ".codex"), "skills") }) {
|
|
338
340
|
installMcpCopy({ nomarmyRoot, installDir, run });
|
|
339
341
|
const commands = installPlaybooks({ nomarmyRoot, target: "codex", dir: skillsDir });
|
|
342
|
+
removeLegacyCodexSkill(legacySkillsDir, skillsDir);
|
|
340
343
|
const notifier = buildNotifierApp({ nomarmyRoot });
|
|
344
|
+
// The same environment Claude's registration gets (connectClaude): the
|
|
345
|
+
// install's execution mode and worker model always win, anything else an
|
|
346
|
+
// operator set on the old registration is kept.
|
|
347
|
+
let preservedEnv = {};
|
|
348
|
+
try { preservedEnv = JSON.parse(run("codex", ["mcp", "get", SERVER_NAME, "--json"], { stdio: ["ignore", "pipe", "ignore"], encoding: "utf8" }))?.transport?.env ?? {}; } catch { /* not registered yet */ }
|
|
349
|
+
const finalEnv = { ...derivePoolAuthEnvPlaceholders(configDir), ...preservedEnv, ...deriveWorkerModelEnv(nomarmyRoot) };
|
|
341
350
|
quietRun(run, "codex", ["mcp", "remove", SERVER_NAME]);
|
|
342
351
|
const serverPath = path.join(installDir, "mcp", "server.mjs");
|
|
343
|
-
|
|
352
|
+
const envArgs = Object.entries(finalEnv).flatMap(([k, v]) => ["--env", `${k}=${v}`]);
|
|
353
|
+
run("codex", ["mcp", "add", SERVER_NAME, ...envArgs, "--", "node", serverPath]);
|
|
344
354
|
run("codex", ["mcp", "list"]);
|
|
345
|
-
return { installDir, serverPath, commands, notifier };
|
|
355
|
+
return { installDir, serverPath, preservedEnv: finalEnv, commands, notifier };
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/** Remove the skill older nomArmy installed under $CODEX_HOME/skills (only nomArmy's own copy, by its marker). */
|
|
359
|
+
function removeLegacyCodexSkill(legacyDir, currentDir) {
|
|
360
|
+
const file = path.join(legacyDir, "nomarmy-feature", "SKILL.md");
|
|
361
|
+
if (path.resolve(legacyDir) === path.resolve(currentDir)) return;
|
|
362
|
+
try {
|
|
363
|
+
if (fs.readFileSync(file, "utf8").includes(COMMAND_MARKER)) fs.rmSync(path.dirname(file), { recursive: true, force: true });
|
|
364
|
+
} catch { /* nothing there */ }
|
|
346
365
|
}
|
|
347
366
|
|
|
348
367
|
export function defaultCursorConfigPath() {
|
|
@@ -381,7 +400,7 @@ export function readCursorConfig(configPath) {
|
|
|
381
400
|
/**
|
|
382
401
|
* @param {{ nomarmyRoot: string, installDir?: string, configPath?: string }} input
|
|
383
402
|
*/
|
|
384
|
-
export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), configPath = defaultCursorConfigPath(), run = defaultRun, commandsDir = defaultCommandDirs().cursor }) {
|
|
403
|
+
export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), configPath = defaultCursorConfigPath(), run = defaultRun, configDir = globalConfigDir(), commandsDir = defaultCommandDirs().cursor }) {
|
|
385
404
|
installMcpCopy({ nomarmyRoot, installDir, run });
|
|
386
405
|
const commands = installPlaybooks({ nomarmyRoot, target: "cursor", dir: commandsDir });
|
|
387
406
|
const notifier = buildNotifierApp({ nomarmyRoot });
|
|
@@ -392,8 +411,12 @@ export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), c
|
|
|
392
411
|
// connectClaude's captureExistingEnv guards against: silently dropping an
|
|
393
412
|
// operator-set NOMARMY_WORKER_MODEL on every reinstall) -- every OTHER
|
|
394
413
|
// configured server in the file is left completely untouched.
|
|
414
|
+
// As for Claude and Codex, the install's own settings (execution mode,
|
|
415
|
+
// worker model, api-key placeholders) win over whatever the entry had.
|
|
395
416
|
const existing = config.mcpServers[SERVER_NAME];
|
|
396
|
-
|
|
417
|
+
// Cursor starts a global MCP server in the home folder, not the open
|
|
418
|
+
// project; ${workspaceFolder} is Cursor's own placeholder for the latter.
|
|
419
|
+
const preservedEnv = { ...derivePoolAuthEnvPlaceholders(configDir), ...((existing && typeof existing.env === "object" && existing.env) || {}), ...deriveWorkerModelEnv(nomarmyRoot), NOMARMY_PROJECT_DIR: "${workspaceFolder}" };
|
|
397
420
|
config.mcpServers[SERVER_NAME] = { command: "node", args: [serverPath], env: preservedEnv };
|
|
398
421
|
fs.mkdirSync(path.dirname(configPath), { recursive: true });
|
|
399
422
|
fs.writeFileSync(configPath, JSON.stringify(config, null, 2) + "\n");
|
|
@@ -13,6 +13,7 @@ Before dispatching:
|
|
|
13
13
|
- Call army to see this repo's roles and which agent each runs on; dispatch by army_role when a role fits. A job on a subscription agent needs on_behalf_of set to that agent's owner.
|
|
14
14
|
- Agents' usage limits show in army and local_worker_capacity; a job on an agent at its limit is held. Ask the operator before resubmitting with confirm_over_limit: true, or move the job to another agent. Never set it on your own.
|
|
15
15
|
- A Claude subscription agent (claude-cli) runs its tools on this machine, outside the sandbox: use it for scout and review work. nomArmy refuses implement jobs on it unless the operator set allow_host_tools; send build work to a sandboxed agent.
|
|
16
|
+
- To run tests without changing anything, use mode: verify; it costs no model usage.
|
|
16
17
|
- Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
|
|
17
18
|
- Prefer local_worker_start + local_worker_status for anything longer than a few minutes. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
18
19
|
|
package/lib/execute.mjs
CHANGED
|
@@ -40,7 +40,7 @@ export function createExecutor(deps) {
|
|
|
40
40
|
ensureJobsRoot();
|
|
41
41
|
// Fire-and-forget: sweeps whatever this or any other nomArmy install left
|
|
42
42
|
// behind, without adding container-CLI round-trip latency to this job's own start.
|
|
43
|
-
sweepStaleSandboxContainers().catch(() => {});
|
|
43
|
+
if (mode !== "verify") sweepStaleSandboxContainers().catch(() => {});
|
|
44
44
|
const jobStartedMs = Date.now();
|
|
45
45
|
const base = await resolveBase(baseRef), jobId = presetJobId || slug(workerId || (mode === "scout" ? "scout" : mode === "decompose" ? "decompose" : "worker")), jobDir = path.join(jobsRoot, jobId), runtimeDir = path.join(jobDir, "runtime");
|
|
46
46
|
fs.mkdirSync(runtimeDir, { recursive: true });
|
|
@@ -48,13 +48,44 @@ export function createExecutor(deps) {
|
|
|
48
48
|
jobId, workerId: workerId || jobId, mode, phase, state: phase === "finished" ? "finished" : "running",
|
|
49
49
|
serverPid: process.pid, baseSha: base.sha, timeoutSeconds, ...extra
|
|
50
50
|
});
|
|
51
|
-
progress("starting", { startedAt: new Date().toISOString(), agent: pool ?? subscriptionWorker ?? "local", model: model ?? null });
|
|
51
|
+
progress("starting", { startedAt: new Date().toISOString(), agent: mode === "verify" ? null : pool ?? subscriptionWorker ?? "local", model: model ?? null });
|
|
52
52
|
const common = { task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, workerId, progress, jobStartedMs };
|
|
53
|
+
if (mode === "verify") return executeVerify({ ...common, verification });
|
|
53
54
|
if (mode === "scout") return executeScout(common);
|
|
54
55
|
if (mode === "decompose") return executeDecompose(common);
|
|
55
56
|
return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor });
|
|
56
57
|
}
|
|
57
58
|
|
|
59
|
+
async function executeVerify({ task, verification: profile, base, jobId, jobDir, workerId, progress, jobStartedMs }) {
|
|
60
|
+
const worktree = path.join(jobDir, "worktree");
|
|
61
|
+
let verification, record = null, error = null;
|
|
62
|
+
try {
|
|
63
|
+
progress("worktree");
|
|
64
|
+
await run("git", ["worktree", "add", "--detach", worktree, base.sha], { cwd: projectDir });
|
|
65
|
+
record = await collectGitRecord({ cwd: worktree, baseSha: base.sha, baseRef: base.ref, branch: null, jobId });
|
|
66
|
+
progress("verification");
|
|
67
|
+
verification = await runIndependentVerification({ profile, cwd: worktree, jobId, baseSha: base.sha, branch: null, mode: "verify", record, logFile: path.join(jobDir, "verification.log") });
|
|
68
|
+
} catch (err) {
|
|
69
|
+
error = err.message;
|
|
70
|
+
verification = normalizeVerification({ status: "not_run", reason: error }, profile);
|
|
71
|
+
} finally {
|
|
72
|
+
try { await run("git", ["worktree", "remove", "--force", worktree], { cwd: projectDir }); }
|
|
73
|
+
catch (err) { error = [error, err.message].filter(Boolean).join("\n"); }
|
|
74
|
+
}
|
|
75
|
+
const retained = fs.existsSync(worktree);
|
|
76
|
+
const outcome = retained ? OUTCOMES.VERIFICATION_NOT_RUN : {
|
|
77
|
+
pass: OUTCOMES.VERIFIED, fail: OUTCOMES.VERIFICATION_FAILED, not_run: OUTCOMES.VERIFICATION_NOT_RUN
|
|
78
|
+
}[verification.status];
|
|
79
|
+
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode: "verify", task,
|
|
80
|
+
baseRef: base.ref, baseSha: base.sha, branch: null, worktree: retained ? worktree : null, worktreeRetained: retained,
|
|
81
|
+
startedAt: new Date(jobStartedMs).toISOString(), finishedAt: new Date().toISOString(),
|
|
82
|
+
outcome, coordinatorStatus: COORDINATOR_STATUS_BY_OUTCOME[outcome], verification, git: record, error,
|
|
83
|
+
metrics: { total_elapsed: Date.now() - jobStartedMs, worker_cost_usd: 0, worker_tokens_total: 0, model_calls: 0 } };
|
|
84
|
+
fs.writeFileSync(path.join(jobDir, "metadata.json"), JSON.stringify(manifest, null, 2));
|
|
85
|
+
progress("finished", { outcome, coordinatorStatus: manifest.coordinatorStatus });
|
|
86
|
+
return { ok: outcome === OUTCOMES.VERIFIED, manifest, jobDir, report: "" };
|
|
87
|
+
}
|
|
88
|
+
|
|
58
89
|
async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, progress, jobStartedMs }) {
|
|
59
90
|
const mode = "implement";
|
|
60
91
|
let branch = `agent/${jobId}`, worktree = path.join(jobDir, "worktree");
|
package/lib/health.mjs
CHANGED
|
@@ -94,16 +94,21 @@ export function armyIssues(summary, mode = "local") {
|
|
|
94
94
|
* from job records. A listed model isn't proof it runs: Muse was listed
|
|
95
95
|
* while every job on it failed this way (a real Senti review).
|
|
96
96
|
*/
|
|
97
|
+
/** The "<provider>/<model>" a job record says its vendor refused, or null. */
|
|
98
|
+
export function refusedModelIn(record) {
|
|
99
|
+
const text = `${record?.workerError ?? ""} ${record?.worker?.error ?? ""}`;
|
|
100
|
+
// A job's own model_not_found line first (mcp/server.mjs names the model
|
|
101
|
+
// it attempted), then any raw vendor wording in older records.
|
|
102
|
+
const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
|
|
103
|
+
return (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, "") ?? null; // the sentence's own trailing period isn't part of the name
|
|
104
|
+
}
|
|
105
|
+
|
|
97
106
|
export function unknownModelIssues(jobRecords, { now = Date.now(), inUse = null, probedOk = {} } = {}) {
|
|
98
107
|
const failed = new Map(), lastOk = new Map(Object.entries(probedOk));
|
|
99
108
|
for (const m of jobRecords) {
|
|
100
109
|
const at = Date.parse(m.finishedAt ?? "");
|
|
101
110
|
if (now - at > DAY) continue;
|
|
102
|
-
const
|
|
103
|
-
// A job's own model_not_found line first (mcp/server.mjs names the model
|
|
104
|
-
// it attempted), then any raw vendor wording in older records.
|
|
105
|
-
const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
|
|
106
|
-
const model = (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, ""); // the sentence's own trailing period isn't part of the name
|
|
111
|
+
const model = refusedModelIn(m);
|
|
107
112
|
if (model) { const f = failed.get(model) ?? { n: 0, last: 0 }; f.n++; f.last = Math.max(f.last, at); failed.set(model, f); }
|
|
108
113
|
else if (m.worker?.provider && m.worker?.model && !m.workerError) {
|
|
109
114
|
const ran = `${m.worker.provider}/${m.worker.model}`;
|
|
@@ -148,6 +153,28 @@ export function recordProbeSuccess(stateRoot, model, { now = Date.now() } = {})
|
|
|
148
153
|
try { seen = JSON.parse(fs.readFileSync(file, "utf8")); } catch { /* first */ }
|
|
149
154
|
seen[model] = new Date(now).toISOString();
|
|
150
155
|
try { fs.writeFileSync(file, JSON.stringify(seen, null, 2)); } catch { /* best-effort */ }
|
|
156
|
+
clearModelRefusal(stateRoot, model);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// Models a vendor refused on a job, kept until a job or test call on them
|
|
160
|
+
// works. A plan's model limit doesn't expire overnight (gpt-6-luna was
|
|
161
|
+
// refused again days after the first time), so this isn't a 24-hour memory.
|
|
162
|
+
// <stateRoot>/model-refusals.json: { "<provider>/<model>": { at, reason } }.
|
|
163
|
+
const REFUSALS_FILE = "model-refusals.json";
|
|
164
|
+
export function modelRefusals(stateRoot) {
|
|
165
|
+
try { const data = JSON.parse(fs.readFileSync(path.join(stateRoot, REFUSALS_FILE), "utf8")); return data && typeof data === "object" && !Array.isArray(data) ? data : {}; } catch { return {}; }
|
|
166
|
+
}
|
|
167
|
+
function writeRefusals(stateRoot, data) {
|
|
168
|
+
try { fs.mkdirSync(stateRoot, { recursive: true }); fs.writeFileSync(path.join(stateRoot, REFUSALS_FILE), JSON.stringify(data, null, 2)); } catch { /* best-effort */ }
|
|
169
|
+
}
|
|
170
|
+
export function recordModelRefusal(stateRoot, model, reason = null, { now = Date.now() } = {}) {
|
|
171
|
+
writeRefusals(stateRoot, { ...modelRefusals(stateRoot), [model]: { at: new Date(now).toISOString(), reason } });
|
|
172
|
+
}
|
|
173
|
+
export function clearModelRefusal(stateRoot, model) {
|
|
174
|
+
const data = modelRefusals(stateRoot);
|
|
175
|
+
if (!(model in data)) return;
|
|
176
|
+
delete data[model];
|
|
177
|
+
writeRefusals(stateRoot, data);
|
|
151
178
|
}
|
|
152
179
|
function readProbeSuccesses(stateRoot) {
|
|
153
180
|
try { return Object.fromEntries(Object.entries(JSON.parse(fs.readFileSync(path.join(stateRoot, "probe-ok.json"), "utf8"))).map(([k, v]) => [k, Date.parse(v)])); } catch { return {}; }
|
|
@@ -160,6 +187,8 @@ function readProbeSuccesses(stateRoot) {
|
|
|
160
187
|
* ChatGPT plan failed job after job before health noticed). The issue, or null.
|
|
161
188
|
*/
|
|
162
189
|
export function recentModelRefusal(stateRoot, model, { now = Date.now() } = {}) {
|
|
190
|
+
const kept = modelRefusals(stateRoot)[model];
|
|
191
|
+
if (kept) return { id: `unknown-model:${model}`, severity: "warn", title: `${model} was refused on a job (${kept.at.slice(0, 10)}) and hasn't worked since`, detail: kept.reason ?? "The provider won't run it, though it may be listed in the catalog.", fix: "nomarmy army assign <role> <agent> <another model>", short: `${model.split("/").pop()} not running` };
|
|
163
192
|
const jobsRoot = path.join(stateRoot, "jobs");
|
|
164
193
|
const records = [];
|
|
165
194
|
let names = [];
|
package/lib/job-format.mjs
CHANGED
|
@@ -77,7 +77,14 @@ function compactImplementRecord(m) {
|
|
|
77
77
|
startedAt: m.startedAt ?? null, finishedAt: m.finishedAt ?? null,
|
|
78
78
|
issues: m.issues ?? [], error: m.error ?? null };
|
|
79
79
|
}
|
|
80
|
+
function compactVerifyRecord(m) {
|
|
81
|
+
return { jobId: m.jobId, workerId: m.workerId, mode: m.mode, outcome: m.outcome,
|
|
82
|
+
coordinatorStatus: m.coordinatorStatus, baseRef: m.baseRef, baseSha: m.baseSha,
|
|
83
|
+
verification: m.verification, elapsedSeconds: Math.round(m.metrics.total_elapsed / 1000),
|
|
84
|
+
worktreeRetained: m.worktreeRetained, error: m.error };
|
|
85
|
+
}
|
|
80
86
|
export function compactJobRecord(meta) {
|
|
87
|
+
if (meta.mode === "verify") return compactVerifyRecord(meta);
|
|
81
88
|
return meta.mode === "scout" ? compactScoutRecord(meta) : meta.mode === "decompose" ? compactDecomposeRecord(meta) : compactImplementRecord(meta);
|
|
82
89
|
}
|
|
83
90
|
|
|
@@ -87,6 +94,7 @@ export function compactJobRecord(meta) {
|
|
|
87
94
|
// worker said, including a truncated or garbled reply -- exactly backwards
|
|
88
95
|
// for a tool whose whole premise is not trusting that reply.
|
|
89
96
|
export function formatResult(r) {
|
|
97
|
+
if (r.manifest?.mode === "verify") return `OUTCOME: ${r.manifest.outcome}\n\n--- VERIFICATION RECORD ---\n${JSON.stringify(compactVerifyRecord(r.manifest), null, 2)}\n\nJob artifacts: ${r.jobDir}`;
|
|
90
98
|
const banner = orchestratorTrust === "degraded" ? DEGRADED_BANNER : "";
|
|
91
99
|
const outcomeLine = r.manifest?.outcome ? `OUTCOME: ${r.manifest.outcome}\n\n` : "";
|
|
92
100
|
const workerReport = `--- WORKER REPORT (a claim, not evidence) ---\n${r.report}`;
|
package/lib/outcomes.mjs
CHANGED
|
@@ -2,6 +2,9 @@ import { SCOUT_OUTCOMES, SCOUT_STATUS_BY_OUTCOME } from "./scout.mjs";
|
|
|
2
2
|
import { DECOMPOSE_STATUS_BY_OUTCOME } from "./decompose.mjs";
|
|
3
3
|
|
|
4
4
|
export const OUTCOMES = Object.freeze({
|
|
5
|
+
VERIFIED: "VERIFIED",
|
|
6
|
+
VERIFICATION_FAILED: "VERIFICATION_FAILED",
|
|
7
|
+
VERIFICATION_NOT_RUN: "VERIFICATION_NOT_RUN",
|
|
5
8
|
WORKER_DONE: "WORKER_DONE",
|
|
6
9
|
WORKER_PARTIAL: "WORKER_PARTIAL",
|
|
7
10
|
WORKER_BLOCKED: "WORKER_BLOCKED",
|
|
@@ -14,6 +17,9 @@ export const OUTCOMES = Object.freeze({
|
|
|
14
17
|
});
|
|
15
18
|
|
|
16
19
|
export const COORDINATOR_STATUS_BY_OUTCOME = Object.freeze({
|
|
20
|
+
[OUTCOMES.VERIFIED]: "complete",
|
|
21
|
+
[OUTCOMES.VERIFICATION_FAILED]: "failed",
|
|
22
|
+
[OUTCOMES.VERIFICATION_NOT_RUN]: "incomplete",
|
|
17
23
|
[OUTCOMES.WORKER_DONE]: "complete",
|
|
18
24
|
[OUTCOMES.RECOVERED_SUCCESS]: "complete",
|
|
19
25
|
[OUTCOMES.WORKER_BLOCKED]: "blocked",
|
package/lib/runs.mjs
CHANGED
|
@@ -80,10 +80,16 @@ function writeRun(runsDir, run) {
|
|
|
80
80
|
export function createRun(runsDir, { name, repo, limits, now = Date.now() }) {
|
|
81
81
|
const slug = String(name ?? "feature").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "").slice(0, 40) || "feature";
|
|
82
82
|
const id = `run-${slug}-${crypto.randomBytes(3).toString("hex")}`;
|
|
83
|
-
|
|
83
|
+
const run = writeRun(runsDir, {
|
|
84
84
|
id, name: name ?? slug, repo, status: "running", createdAt: new Date(now).toISOString(),
|
|
85
85
|
limits, jobs: [], pausedAgents: {}, logPath: path.join(runsDir, `${id}.md`), finishedAt: null, summary: null,
|
|
86
86
|
});
|
|
87
|
+
// The General keeps this log (playbooks/feature.md); start it with the
|
|
88
|
+
// sections it fills in, so the path it's handed already exists.
|
|
89
|
+
if (!fs.existsSync(run.logPath)) {
|
|
90
|
+
fs.writeFileSync(run.logPath, `# ${run.name} (${id})\n\nRepository: ${repo}\nStarted: ${run.createdAt}\n\n## Plan\n\n## Decisions\n\n## Jobs\n\n## Review\n\n## What's left\n`);
|
|
91
|
+
}
|
|
92
|
+
return run;
|
|
87
93
|
}
|
|
88
94
|
|
|
89
95
|
export function loadRun(runsDir, id) {
|
package/lib/server-context.mjs
CHANGED
|
@@ -1,8 +1,17 @@
|
|
|
1
1
|
import os from "node:os";
|
|
2
2
|
import path from "node:path";
|
|
3
|
+
import { spawnSync } from "node:child_process";
|
|
3
4
|
|
|
5
|
+
// The repository jobs run against: NOMARMY_PROJECT_DIR (set by `nomarmy
|
|
6
|
+
// connect cursor` to Cursor's ${workspaceFolder}), else Claude Code's
|
|
7
|
+
// CLAUDE_PROJECT_DIR, else the folder the server was started in (Claude Code
|
|
8
|
+
// and Codex start it in the open project; Cursor starts a global server in
|
|
9
|
+
// the home folder). A value still holding an unexpanded ${...} counts as
|
|
10
|
+
// unset, and a leading ~ is the home folder (Cursor fills ${workspaceFolder}
|
|
11
|
+
// in as "~/...", and nothing on the way expands it).
|
|
4
12
|
export function createServerContext({ env = process.env, homedir = os.homedir(), cwd = process.cwd() } = {}) {
|
|
5
|
-
const
|
|
13
|
+
const fromEnv = [env.NOMARMY_PROJECT_DIR, env.CLAUDE_PROJECT_DIR].find((v) => v && !v.includes("${"));
|
|
14
|
+
const projectDir = path.resolve(fromEnv ? fromEnv.replace(/^~(?=$|[\\/])/, homedir) : cwd);
|
|
6
15
|
const stateRoot = env.NOMARMY_AGENT_STATE || path.join(homedir, ".local", "share", "nomarmy-local-agents");
|
|
7
16
|
const jobsRoot = path.join(stateRoot, "jobs");
|
|
8
17
|
const runsRoot = path.join(stateRoot, "runs");
|
|
@@ -11,3 +20,14 @@ export function createServerContext({ env = process.env, homedir = os.homedir(),
|
|
|
11
20
|
const slotsRoot = path.join(stateRoot, "slots");
|
|
12
21
|
return Object.freeze({ projectDir, stateRoot, jobsRoot, runsRoot, leasesRoot, slotsRoot });
|
|
13
22
|
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Why jobs can't run against projectDir, or null. Jobs need a git
|
|
26
|
+
* repository (worktrees, nomArmy's commits); anything else, like the home
|
|
27
|
+
* folder a coordinator started the server in, is refused before dispatch.
|
|
28
|
+
*/
|
|
29
|
+
export function projectDirProblem(projectDir, { run = (args) => spawnSync("git", args, { encoding: "utf8" }) } = {}) {
|
|
30
|
+
const result = run(["-C", projectDir, "rev-parse", "--is-inside-work-tree"]);
|
|
31
|
+
if (result.status === 0 && String(result.stdout).trim() === "true") return null;
|
|
32
|
+
return `nomArmy's project folder is ${projectDir}, which isn't a git repository, so no job was sent. The coordinator started nomArmy's server outside the project: set NOMARMY_PROJECT_DIR to the repository in its MCP settings (\`nomarmy connect cursor\` does this for Cursor), or start the coordinator from inside the repository.`;
|
|
33
|
+
}
|
|
@@ -230,7 +230,15 @@ export function createVerificationFlow(deps) {
|
|
|
230
230
|
return normalizeVerification({ status: "not_run", basis: "none",
|
|
231
231
|
reason: "no verification runner registered; profile execution is owned by the verification component" }, context.profile);
|
|
232
232
|
}
|
|
233
|
-
try {
|
|
233
|
+
try {
|
|
234
|
+
const value = await verificationRunner(context);
|
|
235
|
+
const normalized = normalizeVerification(value, context.profile);
|
|
236
|
+
// A caller that asks for a log (mode: verify) gets the full output kept.
|
|
237
|
+
if (context.logFile && typeof value?.output === "string") {
|
|
238
|
+
try { fs.writeFileSync(context.logFile, value.output); normalized.log = context.logFile; } catch { /* best-effort */ }
|
|
239
|
+
}
|
|
240
|
+
return normalized;
|
|
241
|
+
}
|
|
234
242
|
catch (error) {
|
|
235
243
|
// A crashed runner produced no evidence. `not_run` is the truthful state:
|
|
236
244
|
// it can never promote a recovery to success, and it never fabricates a
|
package/lib/verify.mjs
CHANGED
|
@@ -693,6 +693,8 @@ export function createVerificationRunner(options = {}) {
|
|
|
693
693
|
basis: `${basis}; ${executed} executed in ${sandboxImage}`,
|
|
694
694
|
reason: null,
|
|
695
695
|
detail: `${verdict.detail}${detailSuffix}`,
|
|
696
|
+
// Each command's full (capped) output, for a caller that keeps a log.
|
|
697
|
+
output: results.map((r) => `$ ${r.command} (exit ${r.exitCode ?? "none"}${r.timedOut ? ", timed out" : ""})\n${r.stdout}${r.stderr ? `\n[stderr]\n${r.stderr}` : ""}`).join("\n\n"),
|
|
696
698
|
};
|
|
697
699
|
};
|
|
698
700
|
}
|
package/mcp/server.mjs
CHANGED
|
@@ -33,6 +33,7 @@ import { createRun, loadRun, runTotals, finishRun, resolveRunLimits, describeLow
|
|
|
33
33
|
import { agentDispatchFields, resolveAgentModel, agentProviderId, describeAgent } from "../lib/agents.mjs";
|
|
34
34
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "../lib/outcomes.mjs";
|
|
35
35
|
import { readUsageSnapshots, usageStatus } from "../lib/usage-limits.mjs";
|
|
36
|
+
import { modelRefusals } from "../lib/health.mjs";
|
|
36
37
|
import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
|
|
37
38
|
import { compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
|
|
38
39
|
|
|
@@ -216,6 +217,10 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
|
|
|
216
217
|
const expanded = jobs.map((job, i) => {
|
|
217
218
|
try {
|
|
218
219
|
let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
|
|
220
|
+
if (j.mode === "verify") {
|
|
221
|
+
const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
|
|
222
|
+
return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
|
|
223
|
+
}
|
|
219
224
|
if (j.army_role) { army ??= getArmy().army; j = expandArmyRole(j, army); }
|
|
220
225
|
const { agent, roleModel = null, ...rest } = j;
|
|
221
226
|
if (!agent) {
|
|
@@ -273,7 +278,7 @@ export const jobSchema = z.object({
|
|
|
273
278
|
verify_regression: z.boolean().optional().describe(
|
|
274
279
|
"implement only: after the diff passes `verification` and touches production files, temporarily revert just those production files, re-run the SAME verification profile (expected to fail without the fix), then restore them. A re-run that still PASSES proves no test would catch this regression, and the outcome is downgraded to NEEDS_REVIEW regardless of the worker's report -- never silently committed as done. This is the ONLY mechanism that catches a verification profile that passes for the wrong reason (a test-selection flag that accidentally excludes the changed file's own tests reports a real, honest, green run that never touched the diff -- exit-code checking alone cannot see the difference). Defaults to true whenever `verification` is set, since that gap is exactly what nomArmy's trust boundary claims to close; pass `false` explicitly to skip the doubled wall-clock cost (can matter on repos with thousands of tests) and accept the risk instead. No effect with no `verification` profile -- there is nothing to re-run. Ignored by scouts."
|
|
275
280
|
),
|
|
276
|
-
mode: z.enum(["scout", "implement", "decompose"]).default("implement").describe("implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
|
|
281
|
+
mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
|
|
277
282
|
base_ref: z.string().optional(),
|
|
278
283
|
timeout_seconds: z.number().int().min(30).max(1800).default(600),
|
|
279
284
|
reasoning: z.enum(["low", "medium", "high"]).default("medium").describe("Thinking level passed to the worker model. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
|
|
@@ -356,9 +361,9 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
|
|
|
356
361
|
poll: { tool: "local_worker_status", job_id: entry.jobId, wait_seconds: MAX_STATUS_WAIT_SECONDS },
|
|
357
362
|
// This job's own lane and budget: a subscription job used to be
|
|
358
363
|
// reported with the local model's figures.
|
|
359
|
-
lane: jobLane(args), agent: args.agentName ?? "local", model: args.model ?? null,
|
|
364
|
+
lane: jobLane(args), agent: args.mode === "verify" ? null : args.agentName ?? "local", model: args.model ?? null,
|
|
360
365
|
...(args.run_id ? { run: runBrief(args.run_id) } : {}),
|
|
361
|
-
admission: { level: admission.level, notes: admission.reasons }, budgets: describeBudgets(budgetsForJob(args)) }, null, 2));
|
|
366
|
+
admission: { level: admission.level, notes: admission.reasons }, budgets: args.mode === "verify" ? null : describeBudgets(budgetsForJob(args)) }, null, 2));
|
|
362
367
|
});
|
|
363
368
|
// A long poll must return inside the MCP client's own idle-timeout: it aborts
|
|
364
369
|
// a tool call after N seconds with no response or progress notification,
|
|
@@ -481,12 +486,17 @@ server.tool("army", "Who you, the General, are and who you call for what in this
|
|
|
481
486
|
// Each agent's models, from OpenClaw's catalog, so the General can pick
|
|
482
487
|
// one for a role set to "auto". The catalog can lag a brand-new model.
|
|
483
488
|
const catalog = await modelCatalogReady();
|
|
489
|
+
// A model its vendor refused on a job is listed apart, so a role on
|
|
490
|
+
// "auto" isn't sent to it (the catalog lists what a plan may refuse).
|
|
491
|
+
const refusals = modelRefusals(stateRoot);
|
|
484
492
|
summary.agents = Object.fromEntries(Object.entries(agents).map(([name, agent]) => {
|
|
485
493
|
const provider = agentProviderId(agent);
|
|
486
|
-
const
|
|
494
|
+
const listed = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
|
|
495
|
+
const models = listed.filter((m) => !refusals[`${provider}/${m}`]);
|
|
496
|
+
const refusedModels = listed.filter((m) => refusals[`${provider}/${m}`]);
|
|
487
497
|
const snapshot = usageSnapshots[provider];
|
|
488
498
|
const usage = snapshot ? (() => { const { level, text } = usageStatus(snapshot); return { level, text }; })() : null;
|
|
489
|
-
return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, usage }];
|
|
499
|
+
return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, ...(refusedModels.length ? { refusedModels } : {}), usage }];
|
|
490
500
|
}));
|
|
491
501
|
// A pinned model missing from the catalog isn't necessarily wrong:
|
|
492
502
|
// `army assign` proves an unlisted model with a real test call, and the
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "A harness for AI coding workers whose claims are never trusted: your coding assistant stays in charge while workers implement and test in sandboxes, on local models, API keys or your own subscriptions.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.5",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|
package/playbooks/feature.md
CHANGED
|
@@ -15,7 +15,7 @@ Follow the army's workflow, calling only the roles the work needs:
|
|
|
15
15
|
1. **Plan.** Scout the repo as needed (`repo_evidence` first; a scout only for research that would pull many files into your context). Write the plan into the run log: the outcome, acceptance criteria, the pieces, and which role gets each.
|
|
16
16
|
2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work.
|
|
17
17
|
3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Send what they find back to the builders as new, bounded jobs.
|
|
18
|
-
4. **Acceptance.** PO and stakeholder test end to end. Fix what they find the same way.
|
|
18
|
+
4. **Acceptance.** PO and stakeholder test end to end. Checks that only run existing tests use mode: verify; writing new e2e checks is still an implement job. Fix what they find the same way.
|
|
19
19
|
5. **Integrate.** Review every diff against nomArmy's verified record -- a worker's report is a claim, not evidence -- and bring the accepted work together on one branch. **Never merge into the developer's branch, and never push.** The finished state is a branch ready for the operator to review and merge.
|
|
20
20
|
|
|
21
21
|
## Decisions along the way
|