nomarmy 0.1.0-alpha.13 → 0.1.0-alpha.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nomarmy.mjs +4 -2
- package/lib/admission.mjs +9 -0
- package/lib/coordinator-instructions.mjs +1 -0
- package/lib/execute.mjs +11 -10
- package/lib/health.mjs +8 -0
- package/lib/outcome.mjs +9 -0
- package/lib/stats.mjs +24 -1
- package/lib/suggestions.mjs +118 -0
- package/mcp/server.mjs +14 -3
- package/package.json +1 -1
- package/playbooks/feature.md +3 -2
package/bin/nomarmy.mjs
CHANGED
|
@@ -20,7 +20,7 @@ import { readGGUFMetadata, resolveModelPath, totalSplitBytes } from "../lib/gguf
|
|
|
20
20
|
import { recommend, customRecommendation, evaluateConfig, bytesPerKvElementForCacheTypes, MIN_CONTEXT_PER_NOM } from "../lib/sizing.mjs";
|
|
21
21
|
import { connectClaude, connectCodex, connectCursor, cursorAlreadyConnected, deriveWorkerModelEnv, defaultInstallDir, installMcpCopy, SCOPES, claudeUserScoped, portableServerLaunch } from "../lib/connect.mjs";
|
|
22
22
|
import { compareVersions, readPackageVersion, readInstallVersions, copyIsStale } from "../lib/install-freshness.mjs";
|
|
23
|
-
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo } from "../lib/stats.mjs";
|
|
23
|
+
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
24
24
|
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
25
25
|
import { loadValidators, saveJevKey, removeJev, jevSettings, askJev, validatorsPath, JEV_CHECKS, saveJudge, removeJudge, judgeSettings } from "../lib/validators.mjs";
|
|
26
26
|
import { probeModel } from "../lib/model-probe.mjs";
|
|
@@ -2712,7 +2712,9 @@ function cmdStats() {
|
|
|
2712
2712
|
try { repo = execFileSync("git", ["rev-parse", "--show-toplevel"], { cwd: repoDir, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }).trim(); }
|
|
2713
2713
|
catch { throw new Error(`${repoDir} isn't inside a git repository; run nomarmy stats from one, or pass --repo <name> or --all-repos`); }
|
|
2714
2714
|
}
|
|
2715
|
-
|
|
2715
|
+
let agentFor = () => null;
|
|
2716
|
+
try { agentFor = agentLookup(loadAgents(globalConfigDir()).agents, agentProviderId); } catch { /* no agents.yml: commands name <agent> */ }
|
|
2717
|
+
const stats = computeStats(records, { repo, sinceMs: parseSince(value("since")), untilMs: parseSince(value("until")), role: value("role"), model: value("model"), agentFor });
|
|
2716
2718
|
if (json) return out(stats);
|
|
2717
2719
|
console.log(formatStats(stats));
|
|
2718
2720
|
}
|
package/lib/admission.mjs
CHANGED
|
@@ -236,6 +236,15 @@ export function createJobRuntime(deps) {
|
|
|
236
236
|
// verify_regression re-runs `verification`; with no profile set there is
|
|
237
237
|
// nothing to re-run. Refuse before starting anything, matching every other
|
|
238
238
|
// admission check here, rather than silently no-op at runtime.
|
|
239
|
+
// stakes applies to what gets committed; reviews names the job a scout reviews.
|
|
240
|
+
jobs.forEach((j, i) => {
|
|
241
|
+
const at = jobs.length > 1 ? `job ${i + 1}: ` : "";
|
|
242
|
+
if (j.stakes === "high" && (j.mode ?? "implement") !== "implement") problems.push(`${at}stakes: high applies to implement jobs; for a review, send a scout with reviews: <job id>`);
|
|
243
|
+
if (j.reviews) {
|
|
244
|
+
if (j.mode !== "scout") problems.push(`${at}reviews: <job id> marks a scout as a review of that job; this job is ${j.mode ?? "implement"}`);
|
|
245
|
+
else if (!fs.existsSync(path.join(jobsRoot, j.reviews, "metadata.json"))) problems.push(`${at}reviews: no finished job ${j.reviews} to review (see local_worker_jobs)`);
|
|
246
|
+
}
|
|
247
|
+
});
|
|
239
248
|
// continue_from: the retained job must exist, be unfinished and uncommitted,
|
|
240
249
|
// and belong to this repo; see lib/continue-from.mjs.
|
|
241
250
|
jobs.forEach((j, i) => {
|
|
@@ -14,6 +14,7 @@ Before dispatching:
|
|
|
14
14
|
- Agents' usage limits show in army and local_worker_capacity; a job on an agent at its limit is held. Ask the operator before resubmitting with confirm_over_limit: true, or move the job to another agent. Never set it on your own.
|
|
15
15
|
- A Claude subscription agent (claude-cli) runs its tools on this machine, outside the sandbox: use it for scout and review work. nomArmy refuses implement jobs on it unless the operator set allow_host_tools; send build work to a sandboxed agent.
|
|
16
16
|
- To run tests without changing anything, use mode: verify; it costs no model usage.
|
|
17
|
+
- Mark an implement job stakes: high when a mistake would be costly (security, access control, personal or tenant data, data loss, money, irreversible changes), however small it is. It then needs a verification profile, keeps the revert check, and needs an independent review before you accept it: a scout on another vendor with reviews: <job id>.
|
|
17
18
|
- Brief outcomes, not edits: a task, explicit acceptance criteria, and the tests that prove it. Put facts you've already resolved in evidence.
|
|
18
19
|
- Prefer local_worker_start for anything longer than a few minutes. Right after, if you can run a background command, run \`nomarmy jobs --wait <job_id>\` in the background so you're told the moment it finishes and can tell the operator (for several jobs, \`nomarmy jobs --events --until-done\`, which exits when they've all finished); otherwise poll local_worker_status with wait_seconds. Never run the plain \`nomarmy jobs --events\` stream as a background command: it only reports when it exits, so you'd never hear; it's for a monitor that reads each line. For a whole feature, use /feature (run_start keeps a run's jobs, spend and hours bounded).
|
|
19
20
|
|
package/lib/execute.mjs
CHANGED
|
@@ -20,7 +20,7 @@ import { detectTestSabotage, addedLinesOf, loadDependencyNames } from "./sabotag
|
|
|
20
20
|
import { describeRecoveryChanges, reportRecoveryPrompt } from "./worker-prompt.mjs";
|
|
21
21
|
import { parseWorkerReport } from "./report.mjs";
|
|
22
22
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "./outcomes.mjs";
|
|
23
|
-
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy } from "./outcome.mjs";
|
|
23
|
+
import { resolveOutcome, finalText, workerMetadata, applyRefactorContract, applyVerificationPolicy, HIGH_STAKES_NOTE } from "./outcome.mjs";
|
|
24
24
|
import { parseAddedLineNumbers, isTestPath, isDocumentationPath, detectScopedTestSelectionRisk, detectUnwiredNewDefinitions, detectMislabeledTestNames, detectPossibleSecrets, detectVerificationInputChanges } from "./diff-checks.mjs";
|
|
25
25
|
|
|
26
26
|
// ---------------------------------------------------------------------------
|
|
@@ -42,7 +42,7 @@ export function createExecutor(deps) {
|
|
|
42
42
|
// Optional Jev checks (lib/validators.mjs): the server wires the settings; without them (tests), none run.
|
|
43
43
|
const jevSettingsFor = (check) => { try { const s = deps.jevSettings?.(); return s?.checks?.includes(check) ? s : null; } catch { return null; } };
|
|
44
44
|
|
|
45
|
-
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, jobId: presetJobId = null }) {
|
|
45
|
+
async function executeJob({ task, acceptance, verification, mode = "implement", baseRef, timeoutSeconds = 600, profile = "coder", reasoning = "high", pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence = null, verifyRegression = false, commitSubject = null, refactor = false, continueFrom = null, stakes = null, reviews = null, jobId: presetJobId = null }) {
|
|
46
46
|
await assertRepo();
|
|
47
47
|
// A continuation starts from the retained job's own base commit.
|
|
48
48
|
let continuation = null;
|
|
@@ -66,9 +66,9 @@ export function createExecutor(deps) {
|
|
|
66
66
|
progress("starting", { startedAt: new Date().toISOString(), agent: mode === "verify" ? null : pool ?? subscriptionWorker ?? "local", model: model ?? null });
|
|
67
67
|
const common = { task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool, subscriptionWorker, onBehalfOf, model, reportSize, workerId, progress, jobStartedMs };
|
|
68
68
|
if (mode === "verify") return executeVerify({ ...common, verification });
|
|
69
|
-
if (mode === "scout") return executeScout(common);
|
|
69
|
+
if (mode === "scout") return executeScout({ ...common, reviews });
|
|
70
70
|
if (mode === "decompose") return executeDecompose(common);
|
|
71
|
-
return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor, continuation });
|
|
71
|
+
return executeImplement({ ...common, verification, evidence, verifyRegression, commitSubject, refactor, continuation, stakes });
|
|
72
72
|
}
|
|
73
73
|
|
|
74
74
|
async function executeVerify({ task, verification: profile, base, jobId, jobDir, workerId, progress, jobStartedMs }) {
|
|
@@ -103,7 +103,7 @@ export function createExecutor(deps) {
|
|
|
103
103
|
return { ok: outcome === OUTCOMES.VERIFIED, manifest, jobDir, report: "" };
|
|
104
104
|
}
|
|
105
105
|
|
|
106
|
-
async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, continuation = null, progress, jobStartedMs }) {
|
|
106
|
+
async function executeImplement({ task, acceptance, verification, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, evidence, verifyRegression = false, commitSubject = null, refactor = false, continuation = null, stakes = null, progress, jobStartedMs }) {
|
|
107
107
|
const mode = "implement";
|
|
108
108
|
let branch = `agent/${jobId}`, worktree = path.join(jobDir, "worktree");
|
|
109
109
|
try {
|
|
@@ -525,7 +525,8 @@ export function createExecutor(deps) {
|
|
|
525
525
|
: mutation?.status === "survivors"
|
|
526
526
|
? { ...afterConfig, reviewRequired: true, reasons: [...afterConfig.reasons, describeSurvivors(mutation, verification)] }
|
|
527
527
|
: afterConfig;
|
|
528
|
-
const
|
|
528
|
+
const afterStakes = stakes === "high" ? { ...afterMutation, reviewRequired: true, reasons: [...afterMutation.reasons, HIGH_STAKES_NOTE] } : afterMutation;
|
|
529
|
+
const finalOutcome = applyRefactorContract(applyVerificationPolicy(afterStakes, independentVerification.status, repoPolicy()),
|
|
529
530
|
{ refactor, verificationStatus: independentVerification.status, testChanges: preCommit.testChanges });
|
|
530
531
|
|
|
531
532
|
progress("commit");
|
|
@@ -571,9 +572,9 @@ export function createExecutor(deps) {
|
|
|
571
572
|
|
|
572
573
|
const metrics = buildMetrics({ result: result ?? attempted, record, reportValidation, outcome: finalOutcome, workerElapsedMs, totalElapsedMs: Date.now() - jobStartedMs, regressionCheckElapsedMs, transientAbortRetried });
|
|
573
574
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree, branch, startedAt, finishedAt,
|
|
574
|
-
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}),
|
|
575
|
+
objective: task, acceptance: acceptance ?? [], verificationProfile: verification ?? null, ...(continuedFrom ? { continuedFrom } : {}), ...(stakes ? { stakes } : {}),
|
|
575
576
|
...(mutation ? { mutation: { ...mutation, elapsedSeconds: Math.round((mutationElapsedMs ?? 0) / 1000) } } : {}),
|
|
576
|
-
...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
|
|
577
|
+
...(jevClaims || judged ? { validators: { ...(jevClaims ? { jev: { check: "report-claims", verdict: jevClaims.verdict, flagged: Boolean(jevClaims.flag), error: jevClaims.error, truncated: Boolean(jevClaims.truncated), inputTokens: jevClaims.usage } } : {}), ...(judged ? { judge: { agent: judge?.agent ?? null, provider: judge?.provider ?? null, model: judge?.model ?? null, answer: judged.answer, flags: judged.flags, error: judged.error } } : {}) } } : {}),
|
|
577
578
|
outcome: finalOutcome.outcome, recovered: finalOutcome.recovered, recoveryAttempted: finalOutcome.recoveryAttempted,
|
|
578
579
|
reportRecoveryAttempted, reportRecovered,
|
|
579
580
|
reviewRequired: finalOutcome.reviewRequired || record.testChanges.reviewRequired,
|
|
@@ -616,7 +617,7 @@ export function createExecutor(deps) {
|
|
|
616
617
|
// the worktree, so a scout that wrote to its snapshot cannot forge evidence.
|
|
617
618
|
// A clean scout worktree holds no work and is removed; a dirty one is retained
|
|
618
619
|
// because a scout that wrote is a scout that misbehaved, and that is worth a look.
|
|
619
|
-
async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs }) {
|
|
620
|
+
async function executeScout({ task, acceptance, base, jobId, jobDir, runtimeDir, timeoutSeconds, profile, reasoning, pool = null, subscriptionWorker = null, onBehalfOf = null, model = null, reportSize = null, workerId, progress, jobStartedMs, reviews = null }) {
|
|
620
621
|
const mode = "scout", worktree = path.join(jobDir, "worktree");
|
|
621
622
|
let worktreeRetained = false;
|
|
622
623
|
try {
|
|
@@ -765,7 +766,7 @@ export function createExecutor(deps) {
|
|
|
765
766
|
displacement_verdict: displacement.verdict
|
|
766
767
|
};
|
|
767
768
|
const manifest = { version: VERSION, jobId, workerId: workerId || jobId, mode, projectDir, worktree: worktreeRetained ? worktree : null, branch: null, baseSha: base.sha, startedAt, finishedAt,
|
|
768
|
-
objective: task, mustCover: acceptance ?? [],
|
|
769
|
+
objective: task, mustCover: acceptance ?? [], ...(reviews ? { reviews } : {}),
|
|
769
770
|
...(jevCitations ? { validators: { jev: { check: "scout-citations", checked: jevCitations.checked, flags: jevCitations.flags, errors: jevCitations.errors, inputTokens: jevCitations.usage } } } : {}),
|
|
770
771
|
outcome: outcome.outcome, coordinatorStatus: outcome.coordinatorStatus, reviewRequired: outcome.reviewRequired, issues,
|
|
771
772
|
scout: { question: report.question, confidence: report.confidence, notFound: report.notFound,
|
package/lib/health.mjs
CHANGED
|
@@ -333,6 +333,14 @@ export async function checkAndRecordHealth({ projectDir, stateRoot, configDir, n
|
|
|
333
333
|
const install = { ...readInstallVersions(installDir), copyHarnesses: Object.keys(loadHarnesses(path.join(installDir, "harnesses")).harnesses).length };
|
|
334
334
|
const result = await runHealthChecks({ now, mode, armySummary, agentsError, jobsRoot: path.join(stateRoot, "jobs"), pidAlive,
|
|
335
335
|
agents, openclawConfig: readOpenclawConfig(), vendors: SUBSCRIPTION_VENDORS, modelsInUse, autoPruned, usageSnapshots: readUsageSnapshots(stateRoot), install });
|
|
336
|
+
try {
|
|
337
|
+
const { loadJobRecords, agentLookup } = await import("./stats.mjs");
|
|
338
|
+
const { recentSuggestions } = await import("./suggestions.mjs");
|
|
339
|
+
for (const s of recentSuggestions(loadJobRecords(path.join(stateRoot, "jobs")), { projectDir, agentFor: agents ? agentLookup(agents, agentProviderId) : () => null, now })) {
|
|
340
|
+
if (s.level !== "warn") continue;
|
|
341
|
+
result.issues.push({ id: `suggestion:${s.key}`, severity: "warn", title: s.title, detail: s.evidence, fix: s.command ?? "nomarmy stats (routing suggestions)", short: "routing tip" });
|
|
342
|
+
}
|
|
343
|
+
} catch { /* suggestions never break a health check */ }
|
|
336
344
|
const toNotify = recordHealth(path.join(stateRoot, "health.json"), result, { now });
|
|
337
345
|
return { result, toNotify };
|
|
338
346
|
}
|
package/lib/outcome.mjs
CHANGED
|
@@ -207,8 +207,17 @@ export function policyAdmissionProblems(job, policy) {
|
|
|
207
207
|
// A refactor meets the regression requirement through its own contract
|
|
208
208
|
// (applyRefactorContract), not the revert check.
|
|
209
209
|
if (policy.require_regression_check && job.verify_regression === false && !job.refactor) problems.push("this repo requires the revert check (policy.require_regression_check in .nomarmy.yml): verify_regression can't be false (a behavior-preserving change can declare refactor: true instead)");
|
|
210
|
+
// stakes: high (security, data loss, irreversible): the checks a General
|
|
211
|
+
// could skip on routine work are mandatory, whatever the repo's policy.
|
|
212
|
+
if (job.stakes === "high") {
|
|
213
|
+
if (!job.verification && !job.refactor) problems.push("a stakes: high job needs a `verification` profile: its result is only as good as the tests that prove it");
|
|
214
|
+
if (job.verify_regression === false && !job.refactor) problems.push("a stakes: high job can't turn the revert check off (verify_regression: false): it's the proof a test catches the change");
|
|
215
|
+
}
|
|
210
216
|
return problems;
|
|
211
217
|
}
|
|
218
|
+
|
|
219
|
+
/** The review line every high-stakes job carries, whatever its outcome. */
|
|
220
|
+
export const HIGH_STAKES_NOTE = "HIGH STAKES: accept this only after an independent review: a scout on another vendor (army_role security-analyst or similar) with reviews: <this job id>, or a judge on another vendor. Passing its tests isn't enough on its own; most defects that pass every check are in security, data or deploy-only paths.";
|
|
212
221
|
/**
|
|
213
222
|
* A declared refactor commits only when verification passed and no test
|
|
214
223
|
* file was added, changed or deleted. Mechanical, not the General's call:
|
package/lib/stats.mjs
CHANGED
|
@@ -5,6 +5,9 @@
|
|
|
5
5
|
|
|
6
6
|
import fs from "node:fs";
|
|
7
7
|
import path from "node:path";
|
|
8
|
+
import { computeSuggestions, formatSuggestions, reviewOf } from "./suggestions.mjs";
|
|
9
|
+
|
|
10
|
+
const SUGGESTION_WINDOW_MS = 14 * 86400000;
|
|
8
11
|
|
|
9
12
|
/** Every readable job record under jobsRoot. */
|
|
10
13
|
export function loadJobRecords(jobsRoot) {
|
|
@@ -17,6 +20,15 @@ export function loadJobRecords(jobsRoot) {
|
|
|
17
20
|
return records;
|
|
18
21
|
}
|
|
19
22
|
|
|
23
|
+
/** An agent name from agents.yml for a provider id, when exactly one agent uses it. */
|
|
24
|
+
export function agentLookup(agents = {}, providerOf) {
|
|
25
|
+
return (provider) => {
|
|
26
|
+
if (!provider) return null;
|
|
27
|
+
const names = Object.entries(agents).filter(([, a]) => { try { return providerOf(a) === provider; } catch { return false; } }).map(([name]) => name);
|
|
28
|
+
return names.length === 1 ? names[0] : null;
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
|
|
20
32
|
/** "7d", "24h", or a date; returns epoch ms or null. */
|
|
21
33
|
export function parseSince(value, now = Date.now()) {
|
|
22
34
|
if (!value) return null;
|
|
@@ -87,7 +99,7 @@ export function resolveRepo(records, value) {
|
|
|
87
99
|
* @param {object[]} records
|
|
88
100
|
* @param {{ repo?: string|null, sinceMs?: number|null, untilMs?: number|null, role?: string|null, model?: string|null }} filter
|
|
89
101
|
*/
|
|
90
|
-
export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null } = {}) {
|
|
102
|
+
export function computeStats(records, { repo = null, sinceMs = null, untilMs = null, role = null, model = null, agentFor = () => null, now = Date.now() } = {}) {
|
|
91
103
|
const inRange = records.filter((r) => {
|
|
92
104
|
const at = Date.parse(r.startedAt ?? r.finishedAt ?? "");
|
|
93
105
|
if (sinceMs != null && !(at >= sinceMs)) return false;
|
|
@@ -169,6 +181,13 @@ export function computeStats(records, { repo = null, sinceMs = null, untilMs = n
|
|
|
169
181
|
notCompleted: sortDesc(count(jobs.filter((r) => !/^(WORKER_DONE|RECOVERED_SUCCESS|VERIFIED|SCOUT_DONE|DECOMPOSE_DONE|SCOUT_NOT_FOUND)$/.test(r.outcome ?? "")), (r) => r.outcome)),
|
|
170
182
|
reviewers,
|
|
171
183
|
signals: sortDesc(signals),
|
|
184
|
+
highStakes: (() => {
|
|
185
|
+
const high = implement.filter((r) => r.stakes === "high");
|
|
186
|
+
return { jobs: high.length, reviewed: high.filter((r) => reviewOf(r, records)).length };
|
|
187
|
+
})(),
|
|
188
|
+
// How you're set up now: the last 14 days unless a period was asked for.
|
|
189
|
+
suggestions: computeSuggestions(sinceMs == null ? jobs.filter((r) => Date.parse(r.startedAt ?? "") >= now - SUGGESTION_WINDOW_MS) : jobs, { agentFor }),
|
|
190
|
+
suggestionWindow: sinceMs == null ? "the last 14 days" : "this period",
|
|
172
191
|
};
|
|
173
192
|
}
|
|
174
193
|
|
|
@@ -183,6 +202,9 @@ export function formatStats(s) {
|
|
|
183
202
|
const lines = [
|
|
184
203
|
`nomArmy stats${s.repo ? ` for ${s.repo}` : " (all repositories)"}${s.role ? `, role ${s.role}` : ""}${s.model ? `, model ${s.model}` : ""}, ${s.period.from ? `${s.period.from.slice(0, 10)} to ${s.period.to.slice(0, 10)}` : "no jobs"}`,
|
|
185
204
|
"",
|
|
205
|
+
`SUGGESTIONS (from ${s.suggestionWindow ?? "this period"}; never applied for you)`,
|
|
206
|
+
...formatSuggestions(s.suggestions ?? []),
|
|
207
|
+
"",
|
|
186
208
|
"VOLUME",
|
|
187
209
|
` Jobs ${s.volume.jobs}: ${list(s.volume.byMode)}${s.unplacedVerifyRuns ? ` (plus ${s.unplacedVerifyRuns} older verify run(s) that don't record their repository)` : ""}`,
|
|
188
210
|
` By role ${list(s.volume.byRole)}`,
|
|
@@ -198,6 +220,7 @@ export function formatStats(s) {
|
|
|
198
220
|
` Passed, but reverting still passed ${c.revertStillPassed}${pct(c.revertStillPassed, c.claimedDone)}`,
|
|
199
221
|
` Passed both ${c.passedBoth}${pct(c.passedBoth, c.claimedDone)}`,
|
|
200
222
|
` of those, flagged by another check ${c.flaggedAfterPassing} (mutants, Jev, judge, rewritten checks)`,
|
|
223
|
+
` High-stakes jobs ${s.highStakes?.jobs ?? 0}, ${s.highStakes?.reviewed ?? 0} with an independent review`,
|
|
201
224
|
" Defects the General found at integration aren't in the records; count them in your own review.",
|
|
202
225
|
"",
|
|
203
226
|
"DIDN'T COMPLETE",
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// Routing suggestions from the job records: which role and model pairings
|
|
2
|
+
// are working, which aren't, and what a change would be. Evidence first,
|
|
3
|
+
// with minimum sample sizes; roles do different work, so a comparison across
|
|
4
|
+
// roles is worded as something to try, not a verdict. Never applied: the
|
|
5
|
+
// operator or the General decides, and each suggestion carries the command.
|
|
6
|
+
|
|
7
|
+
import { jobRole } from "./stats.mjs";
|
|
8
|
+
|
|
9
|
+
export const MIN_JOBS = 5;
|
|
10
|
+
const OK = /^(WORKER_DONE|RECOVERED_SUCCESS|SCOUT_DONE|SCOUT_NOT_FOUND|DECOMPOSE_DONE|VERIFIED)$/;
|
|
11
|
+
|
|
12
|
+
const provider = (r) => r.metrics?.worker_provider ?? r.worker?.provider ?? null;
|
|
13
|
+
const model = (r) => r.metrics?.worker_model ?? r.worker?.model ?? null;
|
|
14
|
+
const agent = (r) => r.labels?.agent ?? null;
|
|
15
|
+
const pct = (n, of) => Math.round((100 * n) / of);
|
|
16
|
+
|
|
17
|
+
/** Whether a high-stakes job has had an independent review: a scout, or a judge, on another vendor. */
|
|
18
|
+
export function reviewOf(job, records) {
|
|
19
|
+
const workerProvider = provider(job);
|
|
20
|
+
const scout = records.find((r) => r.mode === "scout" && r.reviews === job.jobId && provider(r) && provider(r) !== workerProvider);
|
|
21
|
+
if (scout) return { by: "scout", jobId: scout.jobId, provider: provider(scout) };
|
|
22
|
+
const judge = job.validators?.judge;
|
|
23
|
+
if (judge?.answer && judge.provider && judge.provider !== workerProvider) return { by: "judge", provider: judge.provider };
|
|
24
|
+
return null;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* @param {object[]} records this repo's records, already filtered to a period
|
|
29
|
+
* @returns {{ level: "warn"|"info", key: string, title: string, evidence: string, command: string|null }[]}
|
|
30
|
+
*/
|
|
31
|
+
export function computeSuggestions(records, { minJobs = MIN_JOBS, agentFor = () => null } = {}) {
|
|
32
|
+
const out = [];
|
|
33
|
+
const work = records.filter((r) => r.mode === "implement" || r.mode === "scout");
|
|
34
|
+
|
|
35
|
+
// Per role and model.
|
|
36
|
+
const groups = new Map();
|
|
37
|
+
for (const r of work) {
|
|
38
|
+
const key = `${jobRole(r) ?? ""}|${model(r) ?? ""}|${r.mode}`;
|
|
39
|
+
const g = groups.get(key) ?? { role: jobRole(r), model: model(r), mode: r.mode, agent: agent(r), provider: provider(r), jobs: 0, ok: 0, timeout: 0, unsupported: 0, tokens: 0 };
|
|
40
|
+
g.jobs++; if (OK.test(r.outcome ?? "")) g.ok++;
|
|
41
|
+
if (r.outcome === "WORKER_TIMEOUT") g.timeout++;
|
|
42
|
+
if (r.outcome === "SCOUT_UNSUPPORTED") g.unsupported++;
|
|
43
|
+
g.tokens += r.metrics?.worker_tokens_total ?? 0;
|
|
44
|
+
g.agent = g.agent ?? agent(r) ?? agentFor(provider(r));
|
|
45
|
+
groups.set(key, g);
|
|
46
|
+
}
|
|
47
|
+
const assign = (g, to = "<another model>") => (g.role ? `nomarmy army assign ${g.role} ${g.agent ?? "<agent>"} ${to}` : null);
|
|
48
|
+
|
|
49
|
+
for (const g of groups.values()) {
|
|
50
|
+
if (!g.model) continue;
|
|
51
|
+
const kind = g.mode === "scout" ? "scouts" : "implement jobs";
|
|
52
|
+
const who = g.role ? `${g.role} on ${g.model}` : `${kind} with no role on ${g.model}`;
|
|
53
|
+
// Scouts that come back empty.
|
|
54
|
+
if (g.mode === "scout" && g.jobs >= 3 && g.unsupported / g.jobs >= 0.4) {
|
|
55
|
+
out.push({ level: "warn", key: `scout-unsupported:${g.role}:${g.model}`, title: `${who}: ${g.unsupported} of ${g.jobs} scouts came back unsupported`,
|
|
56
|
+
evidence: "Their findings couldn't be tied to cited lines. A different agent, or report: full, usually fixes it.", command: assign(g) });
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
if (g.jobs < minJobs) continue;
|
|
60
|
+
// A pairing that rarely finishes.
|
|
61
|
+
if (g.ok / g.jobs < 0.5) {
|
|
62
|
+
const better = [...groups.values()].filter((o) => o !== g && o.mode === g.mode && o.jobs >= minJobs && o.model && o.ok / o.jobs >= g.ok / g.jobs + 0.2)
|
|
63
|
+
.sort((a, b) => b.ok / b.jobs - a.ok / a.jobs)[0];
|
|
64
|
+
out.push({ level: "warn", key: `low-success:${g.role}:${g.model}:${g.mode}`, title: `${who} finished ${g.ok} of ${g.jobs} ${g.role ? kind : ""}`.trim() + ` (${pct(g.ok, g.jobs)}%)`,
|
|
65
|
+
evidence: (better ? `${better.model} finished ${pct(better.ok, better.jobs)}% of its ${better.jobs} ${kind} here${better.role ? ` (as ${better.role})` : ""}.` : "No other model has enough jobs here to compare.") + (g.role ? "" : " These ran with no army role: send this kind of work to a role on a stronger agent instead."),
|
|
66
|
+
command: g.role ? assign(g, better?.model ?? "<another model>") : null });
|
|
67
|
+
}
|
|
68
|
+
// Timeouts.
|
|
69
|
+
if (g.timeout >= 3 && g.timeout / g.jobs >= 0.25) {
|
|
70
|
+
out.push({ level: "info", key: `timeouts:${g.role}:${g.model}`, title: `${who} timed out on ${g.timeout} of ${g.jobs} jobs`,
|
|
71
|
+
evidence: "Smaller briefs (one outcome each), a longer timeout_seconds, or a faster model would help.", command: null });
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// A lighter model doing as well on implement work, for much less.
|
|
76
|
+
const impl = [...groups.values()].filter((g) => g.mode === "implement" && g.model && g.jobs >= minJobs && g.tokens > 0);
|
|
77
|
+
for (const heavy of impl) {
|
|
78
|
+
for (const light of impl) {
|
|
79
|
+
if (light === heavy || light.model === heavy.model) continue;
|
|
80
|
+
const lightRate = light.ok / light.jobs, heavyRate = heavy.ok / heavy.jobs;
|
|
81
|
+
if (lightRate >= heavyRate - 0.05 && light.tokens / light.jobs <= 0.5 * (heavy.tokens / heavy.jobs)) {
|
|
82
|
+
out.push({ level: "info", key: `lighter:${heavy.role}:${heavy.model}:${light.model}`,
|
|
83
|
+
title: `${light.model} finished ${pct(light.ok, light.jobs)}% of its jobs${light.role ? ` (as ${light.role})` : ""} on ${Math.round((light.tokens / light.jobs) / 1000)}k tokens a job; ${heavy.model}${heavy.role ? ` (as ${heavy.role})` : ""} finished ${pct(heavy.ok, heavy.jobs)}% on ${Math.round((heavy.tokens / heavy.jobs) / 1000)}k`,
|
|
84
|
+
evidence: "They did different work, so try it rather than switch outright: put the role on auto so the General picks per job, or send its routine pieces to the lighter model.",
|
|
85
|
+
command: heavy.role ? `nomarmy army assign ${heavy.role} ${heavy.agent ?? "<agent>"} auto` : null });
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// Where the money goes.
|
|
91
|
+
const spend = new Map();
|
|
92
|
+
for (const r of work) if (Number.isFinite(r.metrics?.worker_cost_usd) && r.metrics.worker_cost_usd > 0) spend.set(model(r), (spend.get(model(r)) ?? 0) + r.metrics.worker_cost_usd);
|
|
93
|
+
const total = [...spend.values()].reduce((a, b) => a + b, 0);
|
|
94
|
+
const [top, topUsd] = [...spend.entries()].sort((a, b) => b[1] - a[1])[0] ?? [];
|
|
95
|
+
if (top && topUsd >= 5 && topUsd / total >= 0.5) {
|
|
96
|
+
out.push({ level: "info", key: `spend:${top}`, title: `${top} is ${pct(topUsd, total)}% of API spend ($${topUsd.toFixed(2)} of $${total.toFixed(2)})`,
|
|
97
|
+
evidence: "Worth knowing rather than changing if it's catching real problems; check its reviews' findings before moving it.", command: null });
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// High-stakes work without an independent review.
|
|
101
|
+
const unreviewed = records.filter((r) => r.mode === "implement" && r.stakes === "high" && !reviewOf(r, records));
|
|
102
|
+
if (unreviewed.length) {
|
|
103
|
+
out.push({ level: "warn", key: `unreviewed:${unreviewed.map((r) => r.jobId).sort().join(",")}`, title: `${unreviewed.length} high-stakes job(s) without an independent review: ${unreviewed.slice(0, 5).map((r) => r.jobId).join(", ")}`,
|
|
104
|
+
evidence: "Send a scout on another vendor with reviews: <job id> before accepting them (or configure a judge on another vendor).", command: null });
|
|
105
|
+
}
|
|
106
|
+
return out;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** This repository's suggestions from its last 14 days of jobs. */
|
|
110
|
+
export function recentSuggestions(records, { projectDir, agentFor = () => null, now = Date.now(), days = 14 } = {}) {
|
|
111
|
+
const since = now - days * 86400000;
|
|
112
|
+
return computeSuggestions(records.filter((r) => r.projectDir === projectDir && Date.parse(r.startedAt ?? "") >= since), { agentFor });
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export function formatSuggestions(list) {
|
|
116
|
+
if (!list.length) return [" none: nothing in the records suggests a routing change"];
|
|
117
|
+
return list.flatMap((s) => [` ${s.level === "warn" ? "!" : "-"} ${s.title}`, ` ${s.evidence}`, ...(s.command ? [` ${s.command}`] : [])]);
|
|
118
|
+
}
|
package/mcp/server.mjs
CHANGED
|
@@ -37,7 +37,8 @@ import { modelRefusals } from "../lib/health.mjs";
|
|
|
37
37
|
import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
|
|
38
38
|
import { restartNotice } from "../lib/install-freshness.mjs";
|
|
39
39
|
import { requestJobStop } from "../lib/openclaw-run.mjs";
|
|
40
|
-
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo } from "../lib/stats.mjs";
|
|
40
|
+
import { loadJobRecords, computeStats, formatStats, parseSince, resolveRepo, agentLookup } from "../lib/stats.mjs";
|
|
41
|
+
import { recentSuggestions } from "../lib/suggestions.mjs";
|
|
41
42
|
import { probeModel } from "../lib/model-probe.mjs";
|
|
42
43
|
import { jevSettings, judgeSettings } from "../lib/validators.mjs";
|
|
43
44
|
import { agentRunsToolsOnHost } from "../lib/dispatch-schema.mjs";
|
|
@@ -302,6 +303,8 @@ export const jobSchema = z.object({
|
|
|
302
303
|
model: z.string().regex(/^\S{1,200}$/).optional().describe("The model to run on the job's agent (an api or subscription agent), e.g. \"gpt-6-sol\". Overrides the role's model and the agent's default. Required when the role's model is \"auto\" or the agent has no default. The `army` tool lists each agent's models. Refused on the local agent, whose model `nomarmy model` sets."),
|
|
303
304
|
run_id: z.string().regex(/^run-[a-z0-9-]{1,80}$/).optional().describe("The /feature run this job belongs to (from run_start). Admission then enforces the run's limits (jobs, api spend, hours) and refuses an agent the run has paused after a vendor usage-limit error; the finished job is recorded into the run."),
|
|
304
305
|
report: z.enum(["brief", "standard", "full"]).optional().describe("How much the worker may report back, capped by its agent's tier: brief (today's local-sized report), standard (the default), full (the frontier ceiling: about 2k tokens for implement, 4k for a scout). An api or subscription scout defaults to full because its findings are the point; other jobs default to standard. The report lands in your own context and is re-read every later turn. No effect on the local model, whose caps are calibrated."),
|
|
306
|
+
stakes: z.enum(["normal", "high"]).optional().describe("implement: how much a mistake would cost, separate from how hard the work is. high for anything touching security or access control, personal or tenant data, data loss, money, or changes that can't be undone: a verification profile is then required, the revert check can't be turned off, and the job always comes back needing review until an independent review (a scout on another vendor with reviews: <job id>, or a judge on another vendor) has looked at it. A one-line auth change is simple and high-stakes."),
|
|
307
|
+
reviews: z.string().regex(/^[A-Za-z0-9._-]{1,120}$/).optional().describe("scout: the job id this scout independently reviews, so the review is recorded against that job (nomarmy stats shows high-stakes jobs with and without one). Use a different vendor than the job's worker."),
|
|
305
308
|
commit_subject: z.string().max(200).optional().describe("implement: the subject line of the commit nomArmy makes on the worker branch, e.g. \"Keep held-back tables in the list_tables cache\". Defaults to the task's first sentence; the body is the worker's NOTE, and the job id is a trailer."),
|
|
306
309
|
army_role: z.string().regex(/^[a-z][a-z0-9-]{0,63}$/).optional().describe("Dispatch by army role (e.g. \"sr-dev\", \"security-analyst\"): nomArmy runs it on the agent this repo assigns to that role and puts the role's description at the top of the brief. Call the `army` tool first to see this repo's roles. Mutually exclusive with agent. Add on_behalf_of in case the role's agent is a subscription; it's ignored otherwise."),
|
|
307
310
|
confirm_over_limit: z.boolean().optional().describe("Override a reached usage limit: the General must ask the operator before resubmitting with confirm_over_limit: true, or send the job to another agent. nomArmy never sets it itself."),
|
|
@@ -353,7 +356,7 @@ function jobArgs(args, workerId) {
|
|
|
353
356
|
return { task: args.task, acceptance: args.acceptance, verification: args.verification, mode: args.mode, baseRef: args.base_ref,
|
|
354
357
|
timeoutSeconds: args.timeout_seconds, profile: args.profile, reasoning: args.reasoning, pool: args.pool,
|
|
355
358
|
subscriptionWorker, onBehalfOf: args.on_behalf_of, model: args.model ?? null, reportSize: args.report ?? null, evidence: args.evidence,
|
|
356
|
-
verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), continueFrom: args.continue_from ?? null, workerId };
|
|
359
|
+
verifyRegression: resolveVerifyRegression(args), commitSubject: args.commit_subject ?? null, refactor: Boolean(args.refactor), continueFrom: args.continue_from ?? null, stakes: args.stakes ?? null, reviews: args.reviews ?? null, workerId };
|
|
357
360
|
}
|
|
358
361
|
server.tool("local_worker", "Run one isolated local worker and wait for it. mode=implement edits in its own worktree and the coordinator commits only on a valid done report (or a recovered job that passed independent verification); failed or incomplete worktrees are retained. mode=scout answers a question from a read-only snapshot with mandatory [path:line] citations that nomArmy verifies and expands. mode=decompose (also read-only) proposes 2+ independent subtasks for a broad objective instead of one worker turn trying to do too much; the proposal is never auto-dispatched, review it and make a separate call with the subtasks you choose. Refuses under memory pressure or over capacity; use local_worker_start + local_worker_status to avoid blocking.", jobSchema.shape,
|
|
359
362
|
async rawArgs => {
|
|
@@ -439,7 +442,9 @@ server.tool("stats", "What nomArmy's own job records show for this repository (o
|
|
|
439
442
|
}, async ({ since, until, all_repos, repo, role, model, format }) => {
|
|
440
443
|
try {
|
|
441
444
|
const records = loadJobRecords(jobsRoot);
|
|
442
|
-
|
|
445
|
+
let agentFor = () => null;
|
|
446
|
+
try { agentFor = agentLookup(agentsConfig().agents, agentProviderId); } catch { /* commands name <agent> */ }
|
|
447
|
+
const stats = computeStats(records, { repo: repo ? resolveRepo(records, repo) : all_repos ? null : projectDir, sinceMs: parseSince(since), untilMs: parseSince(until), role: role ?? null, model: model ?? null, agentFor });
|
|
443
448
|
return toolText(format === "json" ? JSON.stringify(stats, null, 2) : formatStats(stats));
|
|
444
449
|
} catch (error) { return toolText(error.message, true); }
|
|
445
450
|
});
|
|
@@ -561,6 +566,12 @@ server.tool("army", "Who you, the General, are and who you call for what in this
|
|
|
561
566
|
role.modelNote = `${role.model} isn't in OpenClaw's catalog for ${role.agent}; \`army assign\` checked it with a real test call when it was set, and the catalog can lag new models. Use it as assigned; if a job reports "Unknown model", reassign.`;
|
|
562
567
|
}
|
|
563
568
|
}
|
|
569
|
+
// Routing suggestions from this repo's recent jobs (lib/suggestions.mjs):
|
|
570
|
+
// tell the operator about them; never apply one without their say-so.
|
|
571
|
+
try {
|
|
572
|
+
const list = recentSuggestions(loadJobRecords(jobsRoot), { projectDir, agentFor: agentLookup(agents, agentProviderId) });
|
|
573
|
+
if (list.length) summary.suggestions = { note: "From this repo's last 14 days of jobs. Tell the operator; change routing only with their say-so (nomarmy army assign).", items: list.map(({ level, title, evidence, command }) => ({ level, title, evidence, command })) };
|
|
574
|
+
} catch { /* suggestions are a bonus; the army summary stands without them */ }
|
|
564
575
|
return toolText(JSON.stringify(withRestartNotice(summary), null, 2));
|
|
565
576
|
} catch (error) {
|
|
566
577
|
return toolText(error.message, true);
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.14",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|
package/playbooks/feature.md
CHANGED
|
@@ -12,9 +12,10 @@ You are the General. Build this feature end to end with nomArmy's army and come
|
|
|
12
12
|
|
|
13
13
|
Follow the army's workflow, calling only the roles the work needs:
|
|
14
14
|
|
|
15
|
+
0. **Routing.** The `army` tool's `suggestions` come from this repo's recent jobs (a role that keeps failing, a cheaper model doing as well, unreviewed high-stakes work). Tell the operator about any at the start and in your final summary; change routing only if they say so.
|
|
15
16
|
1. **Plan.** Scout the repo as needed (`repo_evidence` first; a scout only for research that would pull many files into your context). Write the plan into the run log: the outcome, acceptance criteria, the pieces, and which role gets each.
|
|
16
|
-
2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work.
|
|
17
|
-
3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Send what they find back to the builders as new, bounded jobs. A build job that comes back partial, blocked or failing verification is finished with `continue_from: <its job id>` and a brief of just the correction, never by fixing its files yourself: only verified work lands.
|
|
17
|
+
2. **Build.** Dispatch with `army_role` (and `on_behalf_of` when the role's agent is a subscription). The Sr Dev takes the core and harder work; the Jr Dev takes simple, fully specified pieces; UI/UX takes UI. For a role on `auto`, pick the model from the agent's list in the `army` tool: the lighter model for routine work, the frontier one for subtle work. Mark a job `stakes: high` when a mistake would be costly (security or access control, personal or tenant data, data loss, money, anything irreversible), however small the change: that's separate from how hard it is.
|
|
18
|
+
3. **Review.** When the build is in, call the specialists that apply (data architect for data work, security analyst for anything touching auth, input, secrets or data exposure), then the PM against the plan. Every `stakes: high` build job gets an independent review before you accept it: a scout on a different vendor than its worker, with `reviews: <that job id>`. Send what they find back to the builders as new, bounded jobs. A build job that comes back partial, blocked or failing verification is finished with `continue_from: <its job id>` and a brief of just the correction, never by fixing its files yourself: only verified work lands.
|
|
18
19
|
4. **Acceptance.** PO and stakeholder test end to end. Checks that only run existing tests use mode: verify; writing new e2e checks is still an implement job. Fix what they find the same way.
|
|
19
20
|
5. **Integrate.** Review every diff against nomArmy's verified record -- a worker's report is a claim, not evidence -- and bring the accepted work together on one branch. **Never merge into the developer's branch, and never push.** The finished state is a branch ready for the operator to review and merge.
|
|
20
21
|
|