nomarmy 0.1.0-alpha.4 → 0.1.0-alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/admission.mjs CHANGED
@@ -1,6 +1,7 @@
1
1
  import fs from "node:fs";
2
2
  import path from "node:path";
3
3
  import { executionMode } from "./execution.mjs";
4
+ import { loadConfig } from "./config.mjs";
4
5
  import { clampInt } from "./budget-state.mjs";
5
6
  import { checkBrief, assessAdmission, describeBudgets } from "./budget.mjs";
6
7
  import { parseStatusPorcelainZ, isRuntimeJunk } from "./git-record.mjs";
@@ -9,7 +10,8 @@ import { readOpenClawTranscriptTail } from "./transcript.mjs";
9
10
  import { readClaudeSessionTranscript } from "./claude-transcript.mjs";
10
11
  import { notify } from "./notify.mjs";
11
12
  import { readCodexRateLimits, recordUsageSnapshot, readUsageSnapshots, usageStatus } from "./usage-limits.mjs";
12
- import { recentModelRefusal } from "./health.mjs";
13
+ import { projectDirProblem } from "./server-context.mjs";
14
+ import { recentModelRefusal, refusedModelIn, recordModelRefusal, clearModelRefusal } from "./health.mjs";
13
15
  import { writeLease, removeLease, liveLeases, liveSlots, acquireSlot } from "./slots.mjs";
14
16
  import { loadRun, runTotals, runAdmissionProblems, recordRunJob, detectUsageLimit } from "./runs.mjs";
15
17
  import { agentProviderId, hostToolsImplementProblem } from "./agents.mjs";
@@ -22,7 +24,7 @@ import { policyAdmissionProblems } from "./outcome.mjs";
22
24
  // Claude or Codex job took llama-server's only slot and blocked local work
23
25
  // it never competed with -- reported from a real Senti run.
24
26
  export function jobLane(job) {
25
- return job.pool || job.subscription_worker ? "remote" : "local";
27
+ return job.mode === "verify" || job.pool || job.subscription_worker ? "remote" : "local";
26
28
  }
27
29
 
28
30
  // A static, operator-declared ceiling on how many remote jobs (api and
@@ -114,7 +116,7 @@ export function createJobRuntime(deps) {
114
116
  * batch queue for a slot instead of failing.
115
117
  */
116
118
  function withAgentSlot(args, jobId, fn, { waitMs = 0 } = {}) {
117
- const max = args.agentName ? agentMaxConcurrent(args.agentName) : null;
119
+ const max = args.mode !== "verify" && args.agentName ? agentMaxConcurrent(args.agentName) : null;
118
120
  if (!max) return fn();
119
121
  return (async () => {
120
122
  const slot = await acquireSlot(slotsRoot, args.agentName, max, { jobId, waitMs });
@@ -148,8 +150,14 @@ export function createJobRuntime(deps) {
148
150
  } catch { /* Usage telemetry must never affect the job result. */ }
149
151
  if (!entry.lane) return; // only tracked jobs, never internal helpers
150
152
  const m = result?.manifest ?? {};
153
+ // Remember a model its vendor refused, until something on it works.
154
+ try {
155
+ const refused = refusedModelIn(m);
156
+ if (refused) recordModelRefusal(stateRoot, refused, String(m.workerError ?? m.worker?.error ?? "").split("\n")[0].slice(0, 300) || null);
157
+ else if (m.worker?.provider && m.worker?.model && !m.workerError && !error) clearModelRefusal(stateRoot, `${m.worker.provider}/${m.worker.model}`);
158
+ } catch { /* best-effort, like the usage reading */ }
151
159
  const outcome = error ? "failed" : String(m.outcome ?? (result?.ok ? "done" : "finished")).toLowerCase().replace(/_/g, " ");
152
- const who = entry.agent ? `${entry.agent}${entry.model ? `/${entry.model}` : ""}` : "local model";
160
+ const who = entry.mode === "verify" ? "verification runner" : entry.agent ? `${entry.agent}${entry.model ? `/${entry.model}` : ""}` : "local model";
153
161
  const took = Math.round((Date.now() - Date.parse(entry.startedAt)) / 60000);
154
162
  const ok = !error && (result?.ok || m.coordinatorStatus === "complete");
155
163
  notify(`nomArmy: ${entry.role ?? entry.mode ?? "job"} ${ok ? "done" : outcome}`, `${entry.workerId ?? entry.jobId} on ${who}: ${outcome} after ${took}m. ${ok ? "Ready for the General's review." : "Needs a look."}`);
@@ -173,12 +181,14 @@ export function createJobRuntime(deps) {
173
181
  };
174
182
  }
175
183
  async function admit(jobs) {
176
- await deps.budgetState.refresh();
177
- if (jobs.some((j) => jobLane(j) === "remote")) await modelCatalogReady();
184
+ if (jobs.some(j => j.mode !== "verify")) await deps.budgetState.refresh();
185
+ if (jobs.some((j) => j.mode !== "verify" && jobLane(j) === "remote")) await modelCatalogReady();
178
186
  const problems = [];
187
+ const notARepo = (deps.projectDirProblem ?? projectDirProblem)(projectDir);
188
+ if (notARepo) problems.push(notARepo);
179
189
  const snapshots = readUsageSnapshots(stateRoot);
180
190
  jobs.forEach((j, i) => {
181
- if (!j.agentName || j.confirm_over_limit === true) return;
191
+ if (j.mode === "verify" || !j.agentName || j.confirm_over_limit === true) return;
182
192
  let provider;
183
193
  try { provider = agentProviderId(agentsConfig().agents[j.agentName]); } catch { return; }
184
194
  const snapshot = snapshots[provider];
@@ -195,6 +205,13 @@ export function createJobRuntime(deps) {
195
205
  // (see budgetsForSubscriptionWorker) -- there's no "which entry" unknown
196
206
  // the way a weighted pool has, since the name given IS the entry.
197
207
  jobs.forEach((j, i) => {
208
+ if (j.mode === "verify") {
209
+ try {
210
+ const profiles = Object.keys(loadConfig(projectDir)?.config?.verification ?? {});
211
+ if (!j.verification || !profiles.includes(j.verification)) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}verify requires a verification profile from .nomarmy.yml; available profiles: ${profiles.join(", ") || "(none)"}`);
212
+ } catch (error) { problems.push(error.message); }
213
+ return;
214
+ }
198
215
  const jobBudgets = budgetsForJob(j);
199
216
  for (const p of checkBrief(j, jobBudgets)) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
200
217
  });
@@ -213,6 +230,7 @@ export function createJobRuntime(deps) {
213
230
  // clean, so a missing on_behalf_of is never reported twice in two
214
231
  // different shapes.
215
232
  jobs.forEach((j, i) => {
233
+ if (j.mode === "verify") return;
216
234
  const fieldProblems = subscriptionJobFieldProblems(j);
217
235
  for (const p of fieldProblems) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
218
236
  if (fieldProblems.length === 0 && j.on_behalf_of) {
@@ -228,7 +246,7 @@ export function createJobRuntime(deps) {
228
246
  // Claude CLI) isn't bounded by the sandbox, so it's refused unless that
229
247
  // agent says allow_host_tools (lib/agents.mjs). Scouts and reviews still run.
230
248
  jobs.forEach((j, i) => {
231
- if (!j.agentName || (j.mode ?? "implement") !== "implement") return;
249
+ if (j.mode === "verify" || !j.agentName || (j.mode ?? "implement") !== "implement") return;
232
250
  let problem = null;
233
251
  try { problem = hostToolsImplementProblem(j.agentName, agentsConfig().agents[j.agentName]); } catch { return; }
234
252
  if (problem) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}${problem}`);
@@ -236,7 +254,7 @@ export function createJobRuntime(deps) {
236
254
  // A model its vendor refused on a job today, with nothing working on it
237
255
  // since, isn't sent another job (lib/health.mjs recentModelRefusal).
238
256
  jobs.forEach((j, i) => {
239
- if (!j.agentName || !j.model) return;
257
+ if (j.mode === "verify" || !j.agentName || !j.model) return;
240
258
  let provider = null;
241
259
  try { provider = agentProviderId(agentsConfig().agents[j.agentName]); } catch { return; }
242
260
  if (!provider) return;
@@ -247,7 +265,7 @@ export function createJobRuntime(deps) {
247
265
  // queue for its slot at launch instead (withAgentSlot's waitMs).
248
266
  if (jobs.length === 1) {
249
267
  const [j] = jobs;
250
- const max = j.agentName ? agentMaxConcurrent(j.agentName) : null;
268
+ const max = j.mode !== "verify" && j.agentName ? agentMaxConcurrent(j.agentName) : null;
251
269
  const held = max ? liveSlots(slotsRoot, j.agentName) : 0;
252
270
  if (max && held >= max) problems.push(`not admitted (capacity): agent "${j.agentName}" already has ${held} job(s) running across this machine's nomArmy sessions, at its max_concurrent of ${max}`);
253
271
  }
@@ -258,7 +276,7 @@ export function createJobRuntime(deps) {
258
276
  const run = loadRun(runsRoot, j.run_id);
259
277
  if (run.repo !== projectDir) problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}run "${run.id}" belongs to ${run.repo}, not this repository`);
260
278
  const running = liveLeases(leasesRoot, { runId: run.id }).length + jobs.slice(0, i).filter((o) => o.run_id === run.id).length;
261
- for (const p of runAdmissionProblems(run, { agentName: j.agentName ?? "local", running })) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
279
+ for (const p of runAdmissionProblems(run, { agentName: j.mode === "verify" ? null : j.agentName ?? "local", running })) problems.push(jobs.length > 1 ? `job ${i + 1}: ${p}` : p);
262
280
  } catch (error) { problems.push(jobs.length > 1 ? `job ${i + 1}: ${error.message}` : error.message); }
263
281
  });
264
282
  // Slot capacity only concerns local jobs: a remote job's inference runs
@@ -271,7 +289,7 @@ export function createJobRuntime(deps) {
271
289
  const anyLocal = jobs.some((j) => jobLane(j) === "local");
272
290
  const admission = anyLocal
273
291
  ? assessAdmission({ hardware: admissionHardware(), runningJobs: runningCount("local"), slots: deps.budgetState.contextInfo.slots, maxWorkers: currentMaxWorkers() })
274
- : assessAdmission({ hardware: admissionHardware(), runningJobs: 0, slots: null, maxWorkers: Infinity });
292
+ : assessAdmission({ hardware: jobs.every(j => j.mode === "verify") ? null : admissionHardware(), runningJobs: 0, slots: null, maxWorkers: Infinity });
275
293
  if (!admission.admit) problems.push(...admission.reasons.map(r => `not admitted (${admission.level}): ${r}`));
276
294
  if (jobs.some((j) => jobLane(j) === "remote")) {
277
295
  const remoteCeiling = currentMaxPoolWorkers(), runningRemote = runningCount("remote");
@@ -302,17 +320,17 @@ export function createJobRuntime(deps) {
302
320
  */
303
321
  function recordJobInRun(args, jobId, result, error = null) {
304
322
  if (!args.run_id) return;
305
- const kind = args.pool ? "api" : args.subscription_worker ? "subscription" : "local";
323
+ const kind = args.mode === "verify" ? "verify" : args.pool ? "api" : args.subscription_worker ? "subscription" : "local";
306
324
  const m = result?.manifest ?? {};
307
325
  const errorLines = [m.worker?.error, error?.message,
308
326
  ...String(m.workerError ?? "").split(/\r?\n/).filter((l) => /error|limit|429/i.test(l))].filter(Boolean).join("\n");
309
- const usageLimit = kind === "local" ? null : detectUsageLimit(errorLines);
327
+ const usageLimit = (kind === "local" || kind === "verify") ? null : detectUsageLimit(errorLines);
310
328
  try {
311
329
  const before = runTotals(loadRun(runsRoot, args.run_id)).warnings;
312
330
  const updated = recordRunJob(runsRoot, args.run_id, {
313
- jobId, agent: args.agentName ?? "local", kind, model: args.model ?? null, role: args.armyRole ?? null, mode: args.mode,
314
- outcome: m.outcome ?? (error ? "ERROR" : null), costUsd: m.metrics?.worker_cost_usd ?? null,
315
- tokens: m.metrics?.worker_tokens_total ?? null, usageLimit,
331
+ jobId, agent: kind === "verify" ? null : args.agentName ?? "local", kind, model: kind === "verify" ? null : args.model ?? null, role: args.armyRole ?? null, mode: args.mode,
332
+ outcome: m.outcome ?? (error ? "ERROR" : null), costUsd: kind === "verify" ? 0 : m.metrics?.worker_cost_usd ?? null,
333
+ tokens: kind === "verify" ? 0 : m.metrics?.worker_tokens_total ?? null, usageLimit,
316
334
  });
317
335
  // A limit crossed or an agent paused by this job is worth interrupting for.
318
336
  const fresh = runTotals(updated).warnings.filter((w) => !before.includes(w) && /OVER|paused/.test(w));
@@ -380,6 +398,7 @@ export function createJobRuntime(deps) {
380
398
  timeoutSeconds: status?.timeoutSeconds ?? null, coordinatorStatus: meta?.coordinatorStatus ?? null, outcome: meta?.outcome ?? null,
381
399
  reviewRequired: meta?.reviewRequired ?? null, issues: (meta?.issues ?? []).slice(0, 6), worktree: meta?.worktree ?? null, branch: meta?.branch ?? null,
382
400
  commit: meta?.commit?.sha ?? null, scout: meta?.scout ? { supported: meta.scout.supported, unsupported: meta.scout.unsupported } : null };
401
+ if (meta?.mode === "verify") Object.assign(out, { verification: meta.verification, baseRef: meta.baseRef, baseSha: meta.baseSha });
383
402
  if (entry && !entry.settled) out.state = "running";
384
403
  else if (entry?.error) { out.state = "failed"; out.error = String(entry.error.message ?? entry.error).split("\n")[0]; }
385
404
  else if (meta) out.state = "finished";
package/lib/army.mjs CHANGED
@@ -404,7 +404,7 @@ export function describeArmy(loaded, { agents = {}, describeAgent = null, usageS
404
404
  roles,
405
405
  layers: loaded.layers,
406
406
  howToDispatch: Object.keys(roles).length
407
- ? "Pass army_role: \"<role>\" on a job (plus on_behalf_of when its agent is a subscription). The role's description goes at the top of the brief; set `mode` on the job yourself, the role's is only a suggestion. A role with model \"auto\" needs model on the job, picked from that agent's models below; any job may pass model to override the role's. Use agent: \"<name>\" instead to pick an agent directly."
407
+ ? "Pass army_role: \"<role>\" on a job (plus on_behalf_of when its agent is a subscription). The role's description goes at the top of the brief; set `mode` on the job yourself, the role's is only a suggestion. A role with model \"auto\" needs model on the job, picked from that agent's models below (never its refusedModels: its vendor turned those down on a job); any job may pass model to override the role's. Use agent: \"<name>\" instead to pick an agent directly."
408
408
  : "No army is configured. Run `nomarmy army init` for the default roster, or dispatch with agent: \"<name>\".",
409
409
  };
410
410
  }
package/lib/connect.mjs CHANGED
@@ -45,8 +45,10 @@ export function defaultInstallDir() {
45
45
  // `<!-- nomarmy:... -->` marker; a same-named file WITHOUT it is the
46
46
  // operator's own and is never overwritten, only reported as skipped.
47
47
  // Claude Code ~/.claude/commands/feature.md -> /feature <request>
48
- // Codex ~/.codex/skills/nomarmy-feature/SKILL.md (confirmed from
49
- // codex 0.156's own $CODEX_HOME/skills loader)
48
+ // Codex ~/.agents/skills/nomarmy-feature/SKILL.md (where the Codex
49
+ // CLI, desktop app and IDE extension all look for user
50
+ // skills; older nomArmy put it in $CODEX_HOME/skills, which
51
+ // connectCodex cleans up)
50
52
  // Cursor ~/.cursor/commands/feature.md (Cursor's documented
51
53
  // user-level commands directory; not verified against a
52
54
  // real install here)
@@ -63,7 +65,7 @@ export function defaultCommandDirs(env = process.env) {
63
65
  const home = os.homedir();
64
66
  return {
65
67
  claude: env.NOMARMY_CLAUDE_COMMANDS_DIR ? path.resolve(env.NOMARMY_CLAUDE_COMMANDS_DIR) : path.join(home, ".claude", "commands"),
66
- codex: env.NOMARMY_CODEX_SKILLS_DIR ? path.resolve(env.NOMARMY_CODEX_SKILLS_DIR) : path.join(env.CODEX_HOME || path.join(home, ".codex"), "skills"),
68
+ codex: env.NOMARMY_CODEX_SKILLS_DIR ? path.resolve(env.NOMARMY_CODEX_SKILLS_DIR) : path.join(home, ".agents", "skills"),
67
69
  cursor: env.NOMARMY_CURSOR_COMMANDS_DIR ? path.resolve(env.NOMARMY_CURSOR_COMMANDS_DIR) : path.join(home, ".cursor", "commands"),
68
70
  };
69
71
  }
@@ -334,15 +336,32 @@ export function connectClaude({ nomarmyRoot, installDir = defaultInstallDir(), r
334
336
  /**
335
337
  * @param {{ nomarmyRoot: string, installDir?: string, run?: Function }} input
336
338
  */
337
- export function connectCodex({ nomarmyRoot, installDir = defaultInstallDir(), run = defaultRun, skillsDir = defaultCommandDirs().codex }) {
339
+ export function connectCodex({ nomarmyRoot, installDir = defaultInstallDir(), run = defaultRun, configDir = globalConfigDir(), skillsDir = defaultCommandDirs().codex, legacySkillsDir = path.join(process.env.CODEX_HOME || path.join(os.homedir(), ".codex"), "skills") }) {
338
340
  installMcpCopy({ nomarmyRoot, installDir, run });
339
341
  const commands = installPlaybooks({ nomarmyRoot, target: "codex", dir: skillsDir });
342
+ removeLegacyCodexSkill(legacySkillsDir, skillsDir);
340
343
  const notifier = buildNotifierApp({ nomarmyRoot });
344
+ // The same environment Claude's registration gets (connectClaude): the
345
+ // install's execution mode and worker model always win, anything else an
346
+ // operator set on the old registration is kept.
347
+ let preservedEnv = {};
348
+ try { preservedEnv = JSON.parse(run("codex", ["mcp", "get", SERVER_NAME, "--json"], { stdio: ["ignore", "pipe", "ignore"], encoding: "utf8" }))?.transport?.env ?? {}; } catch { /* not registered yet */ }
349
+ const finalEnv = { ...derivePoolAuthEnvPlaceholders(configDir), ...preservedEnv, ...deriveWorkerModelEnv(nomarmyRoot) };
341
350
  quietRun(run, "codex", ["mcp", "remove", SERVER_NAME]);
342
351
  const serverPath = path.join(installDir, "mcp", "server.mjs");
343
- run("codex", ["mcp", "add", SERVER_NAME, "--", "node", serverPath]);
352
+ const envArgs = Object.entries(finalEnv).flatMap(([k, v]) => ["--env", `${k}=${v}`]);
353
+ run("codex", ["mcp", "add", SERVER_NAME, ...envArgs, "--", "node", serverPath]);
344
354
  run("codex", ["mcp", "list"]);
345
- return { installDir, serverPath, commands, notifier };
355
+ return { installDir, serverPath, preservedEnv: finalEnv, commands, notifier };
356
+ }
357
+
358
+ /** Remove the skill older nomArmy installed under $CODEX_HOME/skills (only nomArmy's own copy, by its marker). */
359
+ function removeLegacyCodexSkill(legacyDir, currentDir) {
360
+ const file = path.join(legacyDir, "nomarmy-feature", "SKILL.md");
361
+ if (path.resolve(legacyDir) === path.resolve(currentDir)) return;
362
+ try {
363
+ if (fs.readFileSync(file, "utf8").includes(COMMAND_MARKER)) fs.rmSync(path.dirname(file), { recursive: true, force: true });
364
+ } catch { /* nothing there */ }
346
365
  }
347
366
 
348
367
  export function defaultCursorConfigPath() {
@@ -381,7 +400,7 @@ export function readCursorConfig(configPath) {
381
400
  /**
382
401
  * @param {{ nomarmyRoot: string, installDir?: string, configPath?: string }} input
383
402
  */
384
- export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), configPath = defaultCursorConfigPath(), run = defaultRun, commandsDir = defaultCommandDirs().cursor }) {
403
+ export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), configPath = defaultCursorConfigPath(), run = defaultRun, configDir = globalConfigDir(), commandsDir = defaultCommandDirs().cursor }) {
385
404
  installMcpCopy({ nomarmyRoot, installDir, run });
386
405
  const commands = installPlaybooks({ nomarmyRoot, target: "cursor", dir: commandsDir });
387
406
  const notifier = buildNotifierApp({ nomarmyRoot });
@@ -392,8 +411,12 @@ export function connectCursor({ nomarmyRoot, installDir = defaultInstallDir(), c
392
411
  // connectClaude's captureExistingEnv guards against: silently dropping an
393
412
  // operator-set NOMARMY_WORKER_MODEL on every reinstall) -- every OTHER
394
413
  // configured server in the file is left completely untouched.
414
+ // As for Claude and Codex, the install's own settings (execution mode,
415
+ // worker model, api-key placeholders) win over whatever the entry had.
395
416
  const existing = config.mcpServers[SERVER_NAME];
396
- const preservedEnv = (existing && typeof existing.env === "object" && existing.env) || {};
417
+ // Cursor starts a global MCP server in the home folder, not the open
418
+ // project; ${workspaceFolder} is Cursor's own placeholder for the latter.
419
+ const preservedEnv = { ...derivePoolAuthEnvPlaceholders(configDir), ...((existing && typeof existing.env === "object" && existing.env) || {}), ...deriveWorkerModelEnv(nomarmyRoot), NOMARMY_PROJECT_DIR: "${workspaceFolder}" };
397
420
  config.mcpServers[SERVER_NAME] = { command: "node", args: [serverPath], env: preservedEnv };
398
421
  fs.mkdirSync(path.dirname(configPath), { recursive: true });
399
422
  fs.writeFileSync(configPath, JSON.stringify(config, null, 2) + "\n");
@@ -13,6 +13,7 @@ Before dispatching:
13
13
  - Call army to see this repo's roles and which agent each runs on; dispatch by army_role when a role fits. A job on a subscription agent needs on_behalf_of set to that agent's owner.
14
14
  - Agents' usage limits show in army and local_worker_capacity; a job on an agent at its limit is held. Ask the operator before resubmitting with confirm_over_limit: true, or move the job to another agent. Never set it on your own.
15
15
  - A Claude subscription agent (claude-cli) runs its tools on this machine, outside the sandbox: use it for scout and review work. nomArmy refuses implement jobs on it unless the operator set allow_host_tools; send build work to a sandboxed agent.
16
+ - To run tests without changing anything, use mode: verify; it costs no model usage.
16
17
  - Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
17
18
  - Prefer local_worker_start + local_worker_status for anything longer than a few minutes. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
18
19
 
package/lib/execute.mjs CHANGED
@@ -40,7 +40,7 @@ export function createExecutor(deps) {
40
40
  ensureJobsRoot();
41
41
  // Fire-and-forget: sweeps whatever this or any other nomArmy install left
42
42
  // behind, without adding container-CLI round-trip latency to this job's own start.
43
- sweepStaleSandboxContainers().catch(() => {});
43
+ if (mode !== "verify") sweepStaleSandboxContainers().catch(() => {});
44
44
  const jobStartedMs = Date.now();
45
45
  const base = await resolveBase(baseRef), jobId = presetJobId || slug(workerId || (mode === "scout" ? "scout" : mode === "decompose" ? "decompose" : "worker")), jobDir = path.join(jobsRoot, jobId), runtimeDir = path.join(jobDir, "runtime");
46
46
  fs.mkdirSync(runtimeDir, { recursive: true });
@@ -48,13 +48,44 @@ export function createExecutor(deps) {
48
48
  jobId, workerId: workerId || jobId, mode, phase, state: phase === "finished" ? "finished" : "running",
49
49
  serverPid: process.pid, baseSha: base.sha, timeoutSeconds, ...extra
50
50
  });
51
- progress("starting", { startedAt: new Date().toISOString(), agent: pool ?? subscriptionWorker ?? "local", model: model ?? null });
51
+ progress("starting", { startedAt: new Date().toISOString(), agent: mode === "verify" ? null : pool ?? subscriptionWorker ?? "local", model: model ?? null });
52
52
  const common = { task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, workerId, progress, jobStartedMs };
53
+ if (mode === "verify") return executeVerify({ ...common, verification });
53
54
  if (mode === "scout") return executeScout(common);
54
55
  if (mode === "decompose") return executeDecompose(common);
55
56
  return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor });
56
57
  }
57
58
 
59
+ async function executeVerify({ task, verification: profile, base, jobId, jobDir, workerId, progress, jobStartedMs }) {
60
+ const worktree = path.join(jobDir, "worktree");
61
+ let verification, record = null, error = null;
62
+ try {
63
+ progress("worktree");
64
+ await run("git", ["worktree", "add", "--detach", worktree, base.sha], { cwd: projectDir });
65
+ record = await collectGitRecord({ cwd: worktree, baseSha: base.sha, baseRef: base.ref, branch: null, jobId });
66
+ progress("verification");
67
+ verification = await runIndependentVerification({ profile, cwd: worktree, jobId, baseSha: base.sha, branch: null, mode: "verify", record, logFile: path.join(jobDir, "verification.log") });
68
+ } catch (err) {
69
+ error = err.message;
70
+ verification = normalizeVerification({ status: "not_run", reason: error }, profile);
71
+ } finally {
72
+ try { await run("git", ["worktree", "remove", "--force", worktree], { cwd: projectDir }); }
73
+ catch (err) { error = [error, err.message].filter(Boolean).join("\n"); }
74
+ }
75
+ const retained = fs.existsSync(worktree);
76
+ const outcome = retained ? OUTCOMES.VERIFICATION_NOT_RUN : {
77
+ pass: OUTCOMES.VERIFIED, fail: OUTCOMES.VERIFICATION_FAILED, not_run: OUTCOMES.VERIFICATION_NOT_RUN
78
+ }[verification.status];
79
+ const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode: "verify", task,
80
+ baseRef: base.ref, baseSha: base.sha, branch: null, worktree: retained ? worktree : null, worktreeRetained: retained,
81
+ startedAt: new Date(jobStartedMs).toISOString(), finishedAt: new Date().toISOString(),
82
+ outcome, coordinatorStatus: COORDINATOR_STATUS_BY_OUTCOME[outcome], verification, git: record, error,
83
+ metrics: { total_elapsed: Date.now() - jobStartedMs, worker_cost_usd: 0, worker_tokens_total: 0, model_calls: 0 } };
84
+ fs.writeFileSync(path.join(jobDir, "metadata.json"), JSON.stringify(manifest, null, 2));
85
+ progress("finished", { outcome, coordinatorStatus: manifest.coordinatorStatus });
86
+ return { ok: outcome === OUTCOMES.VERIFIED, manifest, jobDir, report: "" };
87
+ }
88
+
58
89
  async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, progress, jobStartedMs }) {
59
90
  const mode = "implement";
60
91
  let branch = `agent/${jobId}`, worktree = path.join(jobDir, "worktree");
package/lib/health.mjs CHANGED
@@ -94,16 +94,21 @@ export function armyIssues(summary, mode = "local") {
94
94
  * from job records. A listed model isn't proof it runs: Muse was listed
95
95
  * while every job on it failed this way (a real Senti review).
96
96
  */
97
+ /** The "<provider>/<model>" a job record says its vendor refused, or null. */
98
+ export function refusedModelIn(record) {
99
+ const text = `${record?.workerError ?? ""} ${record?.worker?.error ?? ""}`;
100
+ // A job's own model_not_found line first (mcp/server.mjs names the model
101
+ // it attempted), then any raw vendor wording in older records.
102
+ const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
103
+ return (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, "") ?? null; // the sentence's own trailing period isn't part of the name
104
+ }
105
+
97
106
  export function unknownModelIssues(jobRecords, { now = Date.now(), inUse = null, probedOk = {} } = {}) {
98
107
  const failed = new Map(), lastOk = new Map(Object.entries(probedOk));
99
108
  for (const m of jobRecords) {
100
109
  const at = Date.parse(m.finishedAt ?? "");
101
110
  if (now - at > DAY) continue;
102
- const text = `${m.workerError ?? ""} ${m.worker?.error ?? ""}`;
103
- // A job's own model_not_found line first (mcp/server.mjs names the model
104
- // it attempted), then any raw vendor wording in older records.
105
- const tagged = /model_not_found: ([\w.-]+\/[\w.:-]+)/.exec(text)?.[1];
106
- const model = (tagged ?? modelRejection(text)?.model)?.replace(/[.:]+$/, ""); // the sentence's own trailing period isn't part of the name
111
+ const model = refusedModelIn(m);
107
112
  if (model) { const f = failed.get(model) ?? { n: 0, last: 0 }; f.n++; f.last = Math.max(f.last, at); failed.set(model, f); }
108
113
  else if (m.worker?.provider && m.worker?.model && !m.workerError) {
109
114
  const ran = `${m.worker.provider}/${m.worker.model}`;
@@ -148,6 +153,28 @@ export function recordProbeSuccess(stateRoot, model, { now = Date.now() } = {})
148
153
  try { seen = JSON.parse(fs.readFileSync(file, "utf8")); } catch { /* first */ }
149
154
  seen[model] = new Date(now).toISOString();
150
155
  try { fs.writeFileSync(file, JSON.stringify(seen, null, 2)); } catch { /* best-effort */ }
156
+ clearModelRefusal(stateRoot, model);
157
+ }
158
+
159
+ // Models a vendor refused on a job, kept until a job or test call on them
160
+ // works. A plan's model limit doesn't expire overnight (gpt-6-luna was
161
+ // refused again days after the first time), so this isn't a 24-hour memory.
162
+ // <stateRoot>/model-refusals.json: { "<provider>/<model>": { at, reason } }.
163
+ const REFUSALS_FILE = "model-refusals.json";
164
+ export function modelRefusals(stateRoot) {
165
+ try { const data = JSON.parse(fs.readFileSync(path.join(stateRoot, REFUSALS_FILE), "utf8")); return data && typeof data === "object" && !Array.isArray(data) ? data : {}; } catch { return {}; }
166
+ }
167
+ function writeRefusals(stateRoot, data) {
168
+ try { fs.mkdirSync(stateRoot, { recursive: true }); fs.writeFileSync(path.join(stateRoot, REFUSALS_FILE), JSON.stringify(data, null, 2)); } catch { /* best-effort */ }
169
+ }
170
+ export function recordModelRefusal(stateRoot, model, reason = null, { now = Date.now() } = {}) {
171
+ writeRefusals(stateRoot, { ...modelRefusals(stateRoot), [model]: { at: new Date(now).toISOString(), reason } });
172
+ }
173
+ export function clearModelRefusal(stateRoot, model) {
174
+ const data = modelRefusals(stateRoot);
175
+ if (!(model in data)) return;
176
+ delete data[model];
177
+ writeRefusals(stateRoot, data);
151
178
  }
152
179
  function readProbeSuccesses(stateRoot) {
153
180
  try { return Object.fromEntries(Object.entries(JSON.parse(fs.readFileSync(path.join(stateRoot, "probe-ok.json"), "utf8"))).map(([k, v]) => [k, Date.parse(v)])); } catch { return {}; }
@@ -160,6 +187,8 @@ function readProbeSuccesses(stateRoot) {
160
187
  * ChatGPT plan failed job after job before health noticed). The issue, or null.
161
188
  */
162
189
  export function recentModelRefusal(stateRoot, model, { now = Date.now() } = {}) {
190
+ const kept = modelRefusals(stateRoot)[model];
191
+ if (kept) return { id: `unknown-model:${model}`, severity: "warn", title: `${model} was refused on a job (${kept.at.slice(0, 10)}) and hasn't worked since`, detail: kept.reason ?? "The provider won't run it, though it may be listed in the catalog.", fix: "nomarmy army assign <role> <agent> <another model>", short: `${model.split("/").pop()} not running` };
163
192
  const jobsRoot = path.join(stateRoot, "jobs");
164
193
  const records = [];
165
194
  let names = [];
@@ -77,7 +77,14 @@ function compactImplementRecord(m) {
77
77
  startedAt: m.startedAt ?? null, finishedAt: m.finishedAt ?? null,
78
78
  issues: m.issues ?? [], error: m.error ?? null };
79
79
  }
80
+ function compactVerifyRecord(m) {
81
+ return { jobId: m.jobId, workerId: m.workerId, mode: m.mode, outcome: m.outcome,
82
+ coordinatorStatus: m.coordinatorStatus, baseRef: m.baseRef, baseSha: m.baseSha,
83
+ verification: m.verification, elapsedSeconds: Math.round(m.metrics.total_elapsed / 1000),
84
+ worktreeRetained: m.worktreeRetained, error: m.error };
85
+ }
80
86
  export function compactJobRecord(meta) {
87
+ if (meta.mode === "verify") return compactVerifyRecord(meta);
81
88
  return meta.mode === "scout" ? compactScoutRecord(meta) : meta.mode === "decompose" ? compactDecomposeRecord(meta) : compactImplementRecord(meta);
82
89
  }
83
90
 
@@ -87,6 +94,7 @@ export function compactJobRecord(meta) {
87
94
  // worker said, including a truncated or garbled reply -- exactly backwards
88
95
  // for a tool whose whole premise is not trusting that reply.
89
96
  export function formatResult(r) {
97
+ if (r.manifest?.mode === "verify") return `OUTCOME: ${r.manifest.outcome}\n\n--- VERIFICATION RECORD ---\n${JSON.stringify(compactVerifyRecord(r.manifest), null, 2)}\n\nJob artifacts: ${r.jobDir}`;
90
98
  const banner = orchestratorTrust === "degraded" ? DEGRADED_BANNER : "";
91
99
  const outcomeLine = r.manifest?.outcome ? `OUTCOME: ${r.manifest.outcome}\n\n` : "";
92
100
  const workerReport = `--- WORKER REPORT (a claim, not evidence) ---\n${r.report}`;
package/lib/outcomes.mjs CHANGED
@@ -2,6 +2,9 @@ import { SCOUT_OUTCOMES, SCOUT_STATUS_BY_OUTCOME } from "./scout.mjs";
2
2
  import { DECOMPOSE_STATUS_BY_OUTCOME } from "./decompose.mjs";
3
3
 
4
4
  export const OUTCOMES = Object.freeze({
5
+ VERIFIED: "VERIFIED",
6
+ VERIFICATION_FAILED: "VERIFICATION_FAILED",
7
+ VERIFICATION_NOT_RUN: "VERIFICATION_NOT_RUN",
5
8
  WORKER_DONE: "WORKER_DONE",
6
9
  WORKER_PARTIAL: "WORKER_PARTIAL",
7
10
  WORKER_BLOCKED: "WORKER_BLOCKED",
@@ -14,6 +17,9 @@ export const OUTCOMES = Object.freeze({
14
17
  });
15
18
 
16
19
  export const COORDINATOR_STATUS_BY_OUTCOME = Object.freeze({
20
+ [OUTCOMES.VERIFIED]: "complete",
21
+ [OUTCOMES.VERIFICATION_FAILED]: "failed",
22
+ [OUTCOMES.VERIFICATION_NOT_RUN]: "incomplete",
17
23
  [OUTCOMES.WORKER_DONE]: "complete",
18
24
  [OUTCOMES.RECOVERED_SUCCESS]: "complete",
19
25
  [OUTCOMES.WORKER_BLOCKED]: "blocked",
package/lib/runs.mjs CHANGED
@@ -80,10 +80,16 @@ function writeRun(runsDir, run) {
80
80
  export function createRun(runsDir, { name, repo, limits, now = Date.now() }) {
81
81
  const slug = String(name ?? "feature").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "").slice(0, 40) || "feature";
82
82
  const id = `run-${slug}-${crypto.randomBytes(3).toString("hex")}`;
83
- return writeRun(runsDir, {
83
+ const run = writeRun(runsDir, {
84
84
  id, name: name ?? slug, repo, status: "running", createdAt: new Date(now).toISOString(),
85
85
  limits, jobs: [], pausedAgents: {}, logPath: path.join(runsDir, `${id}.md`), finishedAt: null, summary: null,
86
86
  });
87
+ // The General keeps this log (playbooks/feature.md); start it with the
88
+ // sections it fills in, so the path it's handed already exists.
89
+ if (!fs.existsSync(run.logPath)) {
90
+ fs.writeFileSync(run.logPath, `# ${run.name} (${id})\n\nRepository: ${repo}\nStarted: ${run.createdAt}\n\n## Plan\n\n## Decisions\n\n## Jobs\n\n## Review\n\n## What's left\n`);
91
+ }
92
+ return run;
87
93
  }
88
94
 
89
95
  export function loadRun(runsDir, id) {
@@ -1,8 +1,17 @@
1
1
  import os from "node:os";
2
2
  import path from "node:path";
3
+ import { spawnSync } from "node:child_process";
3
4
 
5
+ // The repository jobs run against: NOMARMY_PROJECT_DIR (set by `nomarmy
6
+ // connect cursor` to Cursor's ${workspaceFolder}), else Claude Code's
7
+ // CLAUDE_PROJECT_DIR, else the folder the server was started in (Claude Code
8
+ // and Codex start it in the open project; Cursor starts a global server in
9
+ // the home folder). A value still holding an unexpanded ${...} counts as
10
+ // unset, and a leading ~ is the home folder (Cursor fills ${workspaceFolder}
11
+ // in as "~/...", and nothing on the way expands it).
4
12
  export function createServerContext({ env = process.env, homedir = os.homedir(), cwd = process.cwd() } = {}) {
5
- const projectDir = path.resolve(env.CLAUDE_PROJECT_DIR || cwd);
13
+ const fromEnv = [env.NOMARMY_PROJECT_DIR, env.CLAUDE_PROJECT_DIR].find((v) => v && !v.includes("${"));
14
+ const projectDir = path.resolve(fromEnv ? fromEnv.replace(/^~(?=$|[\\/])/, homedir) : cwd);
6
15
  const stateRoot = env.NOMARMY_AGENT_STATE || path.join(homedir, ".local", "share", "nomarmy-local-agents");
7
16
  const jobsRoot = path.join(stateRoot, "jobs");
8
17
  const runsRoot = path.join(stateRoot, "runs");
@@ -11,3 +20,14 @@ export function createServerContext({ env = process.env, homedir = os.homedir(),
11
20
  const slotsRoot = path.join(stateRoot, "slots");
12
21
  return Object.freeze({ projectDir, stateRoot, jobsRoot, runsRoot, leasesRoot, slotsRoot });
13
22
  }
23
+
24
+ /**
25
+ * Why jobs can't run against projectDir, or null. Jobs need a git
26
+ * repository (worktrees, nomArmy's commits); anything else, like the home
27
+ * folder a coordinator started the server in, is refused before dispatch.
28
+ */
29
+ export function projectDirProblem(projectDir, { run = (args) => spawnSync("git", args, { encoding: "utf8" }) } = {}) {
30
+ const result = run(["-C", projectDir, "rev-parse", "--is-inside-work-tree"]);
31
+ if (result.status === 0 && String(result.stdout).trim() === "true") return null;
32
+ return `nomArmy's project folder is ${projectDir}, which isn't a git repository, so no job was sent. The coordinator started nomArmy's server outside the project: set NOMARMY_PROJECT_DIR to the repository in its MCP settings (\`nomarmy connect cursor\` does this for Cursor), or start the coordinator from inside the repository.`;
33
+ }
@@ -230,7 +230,15 @@ export function createVerificationFlow(deps) {
230
230
  return normalizeVerification({ status: "not_run", basis: "none",
231
231
  reason: "no verification runner registered; profile execution is owned by the verification component" }, context.profile);
232
232
  }
233
- try { return normalizeVerification(await verificationRunner(context), context.profile); }
233
+ try {
234
+ const value = await verificationRunner(context);
235
+ const normalized = normalizeVerification(value, context.profile);
236
+ // A caller that asks for a log (mode: verify) gets the full output kept.
237
+ if (context.logFile && typeof value?.output === "string") {
238
+ try { fs.writeFileSync(context.logFile, value.output); normalized.log = context.logFile; } catch { /* best-effort */ }
239
+ }
240
+ return normalized;
241
+ }
234
242
  catch (error) {
235
243
  // A crashed runner produced no evidence. `not_run` is the truthful state:
236
244
  // it can never promote a recovery to success, and it never fabricates a
package/lib/verify.mjs CHANGED
@@ -693,6 +693,8 @@ export function createVerificationRunner(options = {}) {
693
693
  basis: `${basis}; ${executed} executed in ${sandboxImage}`,
694
694
  reason: null,
695
695
  detail: `${verdict.detail}${detailSuffix}`,
696
+ // Each command's full (capped) output, for a caller that keeps a log.
697
+ output: results.map((r) => `$ ${r.command} (exit ${r.exitCode ?? "none"}${r.timedOut ? ", timed out" : ""})\n${r.stdout}${r.stderr ? `\n[stderr]\n${r.stderr}` : ""}`).join("\n\n"),
696
698
  };
697
699
  };
698
700
  }
package/mcp/server.mjs CHANGED
@@ -33,6 +33,7 @@ import { createRun, loadRun, runTotals, finishRun, resolveRunLimits, describeLow
33
33
  import { agentDispatchFields, resolveAgentModel, agentProviderId, describeAgent } from "../lib/agents.mjs";
34
34
  import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "../lib/outcomes.mjs";
35
35
  import { readUsageSnapshots, usageStatus } from "../lib/usage-limits.mjs";
36
+ import { modelRefusals } from "../lib/health.mjs";
36
37
  import { createBuildMetrics, resolveOutcome, finalText, workerMetadata, usageMetrics, policyAdmissionProblems, applyRefactorContract, applyVerificationPolicy, resolveVerifyRegression } from "../lib/outcome.mjs";
37
38
  import { compactJobRecord, formatResult, formatUnion, testChangeBanner, regressionCheckBanner, decomposeOverlapBanner } from "../lib/job-format.mjs";
38
39
 
@@ -216,6 +217,10 @@ export function expandJobs(jobs, { getArmy = currentArmy, getAgents = () => agen
216
217
  const expanded = jobs.map((job, i) => {
217
218
  try {
218
219
  let j = runId && !job.run_id ? { ...job, run_id: runId } : job;
220
+ if (j.mode === "verify") {
221
+ const { agent, model, army_role, on_behalf_of, agentName, pool, subscription_worker, roleModel, ...rest } = j;
222
+ return { ...rest, ...(army_role ? { armyRole: army_role } : {}) };
223
+ }
219
224
  if (j.army_role) { army ??= getArmy().army; j = expandArmyRole(j, army); }
220
225
  const { agent, roleModel = null, ...rest } = j;
221
226
  if (!agent) {
@@ -273,7 +278,7 @@ export const jobSchema = z.object({
273
278
  verify_regression: z.boolean().optional().describe(
274
279
  "implement only: after the diff passes `verification` and touches production files, temporarily revert just those production files, re-run the SAME verification profile (expected to fail without the fix), then restore them. A re-run that still PASSES proves no test would catch this regression, and the outcome is downgraded to NEEDS_REVIEW regardless of the worker's report -- never silently committed as done. This is the ONLY mechanism that catches a verification profile that passes for the wrong reason (a test-selection flag that accidentally excludes the changed file's own tests reports a real, honest, green run that never touched the diff -- exit-code checking alone cannot see the difference). Defaults to true whenever `verification` is set, since that gap is exactly what nomArmy's trust boundary claims to close; pass `false` explicitly to skip the doubled wall-clock cost (can matter on repos with thousands of tests) and accept the risk instead. No effect with no `verification` profile -- there is nothing to re-run. Ignored by scouts."
275
280
  ),
276
- mode: z.enum(["scout", "implement", "decompose"]).default("implement").describe("implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
281
+ mode: z.enum(["scout", "implement", "decompose", "verify"]).default("implement").describe("verify: run a required verification profile with no worker and no model tokens; base_ref selects the branch or commit (default current HEAD), task is a short record label, agent/model are unused and army_role is only a label. implement: edit in an isolated worktree, coordinator commits on a valid report. scout: read-only research; every finding must cite [path:start-end] and nomArmy attaches the cited lines after verifying them against the base commit. decompose: read-only; proposes 2+ independent, evidence-grounded subtasks for a broad objective instead of doing everything in one worker turn. Never auto-dispatched -- the proposal is reviewed like a scout's findings, and the coordinator makes its own separate dispatch call with whatever subtasks it chooses to use."),
277
282
  base_ref: z.string().optional(),
278
283
  timeout_seconds: z.number().int().min(30).max(1800).default(600),
279
284
  reasoning: z.enum(["low", "medium", "high"]).default("medium").describe("Thinking level passed to the worker model. On the local model it takes effect when that model supports thinking (NOMARMY_MODEL_THINKING); on an api or subscription agent it applies per that agent's own `thinking` setting (false = off, a fixed level = always that level). Default is medium, not high, on real measured evidence: on an identical ticket, gpt-oss-20b at high took 318s with 21 tool calls and 4 failures, and at medium took 62s with 9 calls and 0 failures -- high did not produce a better answer, it thrashed. A separate open-ended task made Qwen3.6-27B time out completely at high (630s, zero output) and succeed at medium. Do not raise this to high by default reasoning that more thinking should help -- it has only ever hurt or timed out in testing so far. Reach for high only after a task has already failed once at medium and the failure looks like an under-thinking problem specifically (wrong root cause, not a formatting or scope issue)."),
@@ -356,9 +361,9 @@ server.tool("local_worker_start", "Start one worker or scout in the background a
356
361
  poll: { tool: "local_worker_status", job_id: entry.jobId, wait_seconds: MAX_STATUS_WAIT_SECONDS },
357
362
  // This job's own lane and budget: a subscription job used to be
358
363
  // reported with the local model's figures.
359
- lane: jobLane(args), agent: args.agentName ?? "local", model: args.model ?? null,
364
+ lane: jobLane(args), agent: args.mode === "verify" ? null : args.agentName ?? "local", model: args.model ?? null,
360
365
  ...(args.run_id ? { run: runBrief(args.run_id) } : {}),
361
- admission: { level: admission.level, notes: admission.reasons }, budgets: describeBudgets(budgetsForJob(args)) }, null, 2));
366
+ admission: { level: admission.level, notes: admission.reasons }, budgets: args.mode === "verify" ? null : describeBudgets(budgetsForJob(args)) }, null, 2));
362
367
  });
363
368
  // A long poll must return inside the MCP client's own idle-timeout: it aborts
364
369
  // a tool call after N seconds with no response or progress notification,
@@ -481,12 +486,17 @@ server.tool("army", "Who you, the General, are and who you call for what in this
481
486
  // Each agent's models, from OpenClaw's catalog, so the General can pick
482
487
  // one for a role set to "auto". The catalog can lag a brand-new model.
483
488
  const catalog = await modelCatalogReady();
489
+ // A model its vendor refused on a job is listed apart, so a role on
490
+ // "auto" isn't sent to it (the catalog lists what a plan may refuse).
491
+ const refusals = modelRefusals(stateRoot);
484
492
  summary.agents = Object.fromEntries(Object.entries(agents).map(([name, agent]) => {
485
493
  const provider = agentProviderId(agent);
486
- const models = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
494
+ const listed = provider && catalog ? [...catalog.keys()].filter((k) => k.startsWith(`${provider}/`)).map((k) => k.slice(provider.length + 1)) : [];
495
+ const models = listed.filter((m) => !refusals[`${provider}/${m}`]);
496
+ const refusedModels = listed.filter((m) => refusals[`${provider}/${m}`]);
487
497
  const snapshot = usageSnapshots[provider];
488
498
  const usage = snapshot ? (() => { const { level, text } = usageStatus(snapshot); return { level, text }; })() : null;
489
- return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, usage }];
499
+ return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, ...(refusedModels.length ? { refusedModels } : {}), usage }];
490
500
  }));
491
501
  // A pinned model missing from the catalog isn't necessarily wrong:
492
502
  // `army assign` proves an unlisted model with a real test call, and the
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "description": "A harness for AI coding workers whose claims are never trusted: your coding assistant stays in charge while workers implement and test in sandboxes, on local models, API keys or your own subscriptions.",
4
4
  "author": "Rayson Technologies",
5
5
  "license": "Apache-2.0",
6
- "version": "0.1.0-alpha.4",
6
+ "version": "0.1.0-alpha.5",
7
7
  "private": false,
8
8
  "type": "module",
9
9
  "engines": {
@@ -15,7 +15,7 @@ Follow the army's workflow, calling only the roles the work needs:
15
15
  1. **Plan.** Scout the repo as needed (`repo_evidence` first; a scout only for research that would pull many files into your context). Write the plan into the run log: the outcome, acceptance criteria, the pieces, and which role gets each.
16
16
  2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work.
17
17
  3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Send what they find back to the builders as new, bounded jobs.
18
- 4. **Acceptance.** PO and stakeholder test end to end. Fix what they find the same way.
18
+ 4. **Acceptance.** PO and stakeholder test end to end. Checks that only run existing tests use mode: verify; writing new e2e checks is still an implement job. Fix what they find the same way.
19
19
  5. **Integrate.** Review every diff against nomArmy's verified record -- a worker's report is a claim, not evidence -- and bring the accepted work together on one branch. **Never merge into the developer's branch, and never push.** The finished state is a branch ready for the operator to review and merge.
20
20
 
21
21
  ## Decisions along the way