nomarmy 0.1.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +25 -0
- package/README.md +484 -0
- package/bin/nomarmy.mjs +2248 -0
- package/config/agents.yml.example +63 -0
- package/config/common.env +31 -0
- package/config/profiles/bedrock-cheap.env +26 -0
- package/config/profiles/bedrock.env +28 -0
- package/config/profiles/cpu-linux.env +8 -0
- package/config/profiles/dgx-spark.env +12 -0
- package/config/profiles/macbook-pro.env +9 -0
- package/config/profiles/nvidia-linux.env +9 -0
- package/docker/Dockerfile +15 -0
- package/docker/Dockerfile.go +29 -0
- package/docker/Dockerfile.rust +19 -0
- package/e2e.sh +153 -0
- package/install.sh +125 -0
- package/lib/agents.mjs +285 -0
- package/lib/army.mjs +400 -0
- package/lib/budget.mjs +368 -0
- package/lib/claude-transcript.mjs +150 -0
- package/lib/config.mjs +193 -0
- package/lib/connect.mjs +409 -0
- package/lib/coordinator-instructions.mjs +23 -0
- package/lib/decompose.mjs +389 -0
- package/lib/dispatch-config.mjs +164 -0
- package/lib/dispatch-schema.mjs +280 -0
- package/lib/doctor.mjs +443 -0
- package/lib/evidence.mjs +679 -0
- package/lib/gguf.mjs +589 -0
- package/lib/hardware.mjs +476 -0
- package/lib/health.mjs +278 -0
- package/lib/model-catalog.mjs +71 -0
- package/lib/notifier-app.mjs +95 -0
- package/lib/notify.mjs +66 -0
- package/lib/openclaw-config.mjs +65 -0
- package/lib/openclaw-errors.mjs +40 -0
- package/lib/propose.mjs +110 -0
- package/lib/prune.mjs +77 -0
- package/lib/repo-query.mjs +267 -0
- package/lib/runs.mjs +150 -0
- package/lib/sabotage.mjs +128 -0
- package/lib/sandbox-images.mjs +434 -0
- package/lib/scan.mjs +1538 -0
- package/lib/schema.mjs +288 -0
- package/lib/scout.mjs +544 -0
- package/lib/sizing.mjs +1322 -0
- package/lib/slots.mjs +112 -0
- package/lib/statusline.mjs +126 -0
- package/lib/subscription-config.mjs +68 -0
- package/lib/subscription-setup.mjs +217 -0
- package/lib/transcript.mjs +195 -0
- package/lib/verify.mjs +700 -0
- package/mcp/server.mjs +4206 -0
- package/notifier/icon.swift +34 -0
- package/notifier/main.swift +52 -0
- package/notifier/nomarmy-icon.png +0 -0
- package/package.json +67 -0
- package/playbooks/feature.md +43 -0
- package/policies/coder.md +49 -0
- package/policies/orchestrator.md +35 -0
- package/policies/reviewer.md +35 -0
- package/policies/scout.md +65 -0
- package/scripts/configure-openclaw.sh +96 -0
- package/scripts/configure-orchestrator.sh +84 -0
- package/scripts/install-llama-cpp.sh +16 -0
- package/scripts/lib.sh +198 -0
- package/scripts/select-model.mjs +96 -0
- package/scripts/select-model.sh +4 -0
- package/scripts/setup-sandbox.sh +38 -0
- package/scripts/start-inference.sh +46 -0
- package/scripts/stop-inference.sh +5 -0
- package/scripts/uninstall.sh +6 -0
- package/scripts/verify-install.sh +68 -0
package/lib/verify.mjs
ADDED
|
@@ -0,0 +1,700 @@
|
|
|
1
|
+
// nomArmy independent verification runner (v1.3).
|
|
2
|
+
//
|
|
3
|
+
// The coordinator's verdict must not depend on the worker's claim, so this
|
|
4
|
+
// module re-runs the repository's own verification profile and reports what it
|
|
5
|
+
// observed. `mcp/server.mjs` consumes the verdict through
|
|
6
|
+
// `registerVerificationRunner`.
|
|
7
|
+
//
|
|
8
|
+
// THE SECURITY RULE THIS FILE EXISTS TO HOLD
|
|
9
|
+
// -----------------------------------------
|
|
10
|
+
// Verification commands come from `.nomarmy.yml` inside the repository being
|
|
11
|
+
// worked on, and repository content is untrusted input in this system's threat
|
|
12
|
+
// model. Running those commands on the host would hand coordinator privileges
|
|
13
|
+
// to repo-controlled code, the exact privilege the worker itself is denied.
|
|
14
|
+
//
|
|
15
|
+
// Therefore commands only ever execute inside the Podman sandbox image, as a
|
|
16
|
+
// non-root user, with `--network none`. If Podman is missing, the image is
|
|
17
|
+
// absent, or the container fails to start, the verdict is `not_run`. There is
|
|
18
|
+
// no host fallback path in this file, deliberately: a missing sandbox is
|
|
19
|
+
// absence of evidence, not permission to take a shortcut.
|
|
20
|
+
//
|
|
21
|
+
// Nothing from configuration ever reaches a host-side shell. Every host process
|
|
22
|
+
// is spawned with an argv array and `shell: false`. A command string is handed
|
|
23
|
+
// to `/bin/sh -c` *inside* the container, which is inside the sandbox boundary.
|
|
24
|
+
|
|
25
|
+
import { spawn } from "node:child_process";
|
|
26
|
+
import fs from "node:fs";
|
|
27
|
+
import path from "node:path";
|
|
28
|
+
|
|
29
|
+
import { loadConfig as defaultLoadConfig } from "./config.mjs";
|
|
30
|
+
import { resolveSandboxImage, nodeDependencyFiles, linkNodePackages, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
|
|
31
|
+
|
|
32
|
+
/** Sandbox image used when `NOMARMY_AGENT_IMAGE` is unset. */
|
|
33
|
+
export const DEFAULT_AGENT_IMAGE = "openclaw-nomarmy-coder:bookworm";
|
|
34
|
+
|
|
35
|
+
/** Matches the base image's non-root `node` user (uid 1000) baked into `docker/Dockerfile`. */
|
|
36
|
+
export const DEFAULT_CONTAINER_USER = "1000:1000";
|
|
37
|
+
|
|
38
|
+
/** Mount point for the job worktree inside the container. */
|
|
39
|
+
export const DEFAULT_WORKDIR = "/workspace";
|
|
40
|
+
|
|
41
|
+
/** Only this environment level is executable without provisioned services. */
|
|
42
|
+
export const EXECUTABLE_ENVIRONMENT = "none";
|
|
43
|
+
|
|
44
|
+
export const DEFAULT_COMMAND_TIMEOUT_MS = 10 * 60 * 1000;
|
|
45
|
+
export const DEFAULT_OVERALL_TIMEOUT_MS = 30 * 60 * 1000;
|
|
46
|
+
export const DEFAULT_MAX_OUTPUT_BYTES = 64 * 1024;
|
|
47
|
+
|
|
48
|
+
/** Marker written into captured output when a cap drops bytes. */
|
|
49
|
+
export const TRUNCATION_MARKER = "[nomarmy] output truncated";
|
|
50
|
+
|
|
51
|
+
/** Length of the stderr excerpt attached to a failure detail. */
|
|
52
|
+
const DETAIL_TAIL_CHARS = 400;
|
|
53
|
+
|
|
54
|
+
// ---------------------------------------------------------------------------
|
|
55
|
+
// pure: profile resolution
|
|
56
|
+
// ---------------------------------------------------------------------------
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Pick the verification profile to execute. Performs no I/O.
|
|
60
|
+
*
|
|
61
|
+
* Absence is never an error here. Most repositories have no `.nomarmy.yml`
|
|
62
|
+
* yet, and "there is nothing to run" is a legitimate, reportable state.
|
|
63
|
+
*
|
|
64
|
+
* @param {{ config?: object|null, profile?: string|null }} input
|
|
65
|
+
* @returns {{ found: boolean, name: string|null, commands: string[], environment: string|null, reason: string|null }}
|
|
66
|
+
*/
|
|
67
|
+
export function resolveProfile({ config = null, profile = null } = {}) {
|
|
68
|
+
const name = typeof profile === "string" && profile.trim() ? profile.trim() : null;
|
|
69
|
+
const miss = (reason) => ({ found: false, name, commands: [], environment: null, reason });
|
|
70
|
+
|
|
71
|
+
if (!name) {
|
|
72
|
+
return miss("no verification profile was requested for this job");
|
|
73
|
+
}
|
|
74
|
+
if (!config || typeof config !== "object") {
|
|
75
|
+
return miss(
|
|
76
|
+
`no .nomarmy.yml in the repository, so verification profile '${name}' is not defined`,
|
|
77
|
+
);
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
const block = config.verification;
|
|
81
|
+
if (!block || typeof block !== "object" || Array.isArray(block)) {
|
|
82
|
+
return miss(
|
|
83
|
+
`.nomarmy.yml has no 'verification' block, so profile '${name}' is not defined`,
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const available = Object.keys(block);
|
|
88
|
+
const entry = Object.prototype.hasOwnProperty.call(block, name) ? block[name] : null;
|
|
89
|
+
if (!entry || typeof entry !== "object" || Array.isArray(entry)) {
|
|
90
|
+
const known = available.length
|
|
91
|
+
? `known profiles: ${available.join(", ")}`
|
|
92
|
+
: "the 'verification' block is empty";
|
|
93
|
+
return miss(`.nomarmy.yml defines no verification profile '${name}'; ${known}`);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const environment =
|
|
97
|
+
typeof entry.environment === "string" && entry.environment.trim()
|
|
98
|
+
? entry.environment.trim()
|
|
99
|
+
: EXECUTABLE_ENVIRONMENT;
|
|
100
|
+
|
|
101
|
+
const commands = Array.isArray(entry.commands)
|
|
102
|
+
? entry.commands.filter((c) => typeof c === "string" && c.trim().length > 0)
|
|
103
|
+
: [];
|
|
104
|
+
|
|
105
|
+
return { found: true, name, commands, environment, reason: null };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// ---------------------------------------------------------------------------
|
|
109
|
+
// pure: result classification
|
|
110
|
+
// ---------------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
function tailOf(result) {
|
|
113
|
+
const raw = typeof result?.stderr === "string" && result.stderr.trim()
|
|
114
|
+
? result.stderr
|
|
115
|
+
: typeof result?.stdout === "string"
|
|
116
|
+
? result.stdout
|
|
117
|
+
: "";
|
|
118
|
+
const trimmed = raw.trim();
|
|
119
|
+
if (!trimmed) return "";
|
|
120
|
+
const tail = trimmed.length > DETAIL_TAIL_CHARS
|
|
121
|
+
? trimmed.slice(trimmed.length - DETAIL_TAIL_CHARS)
|
|
122
|
+
: trimmed;
|
|
123
|
+
return ` (last output: ${tail.replace(/\s+/g, " ")})`;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Turn per-command results into a verdict. Performs no I/O.
|
|
128
|
+
*
|
|
129
|
+
* A command that ran and failed (or timed out) is evidence of failure. A
|
|
130
|
+
* command that never started is absence of evidence, so if nothing ever ran the
|
|
131
|
+
* verdict is `not_run`, never `fail`.
|
|
132
|
+
*
|
|
133
|
+
* @param {Array<object>} results
|
|
134
|
+
* @returns {{ status: "pass"|"fail"|"not_run", detail: string }}
|
|
135
|
+
*/
|
|
136
|
+
export function classifyResults(results) {
|
|
137
|
+
const list = Array.isArray(results) ? results.filter(Boolean) : [];
|
|
138
|
+
if (list.length === 0) {
|
|
139
|
+
return { status: "not_run", detail: "no commands were executed" };
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
if (list.every((r) => r.started === false)) {
|
|
143
|
+
const first = list[0];
|
|
144
|
+
const why = first.reason || first.error || "the sandbox never started a command";
|
|
145
|
+
return { status: "not_run", detail: `no command was executed: ${why}` };
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const total = list.length;
|
|
149
|
+
for (let index = 0; index < total; index += 1) {
|
|
150
|
+
const result = list[index];
|
|
151
|
+
const label = `command ${index + 1} of ${total} (\`${result.command ?? "?"}\`)`;
|
|
152
|
+
|
|
153
|
+
if (result.timedOut) {
|
|
154
|
+
const budget = result.timeoutMs ?? result.durationMs;
|
|
155
|
+
const after = Number.isFinite(budget) ? ` after ${budget}ms` : "";
|
|
156
|
+
return { status: "fail", detail: `${label} timed out${after}${tailOf(result)}` };
|
|
157
|
+
}
|
|
158
|
+
if (result.started === false) {
|
|
159
|
+
const why = result.reason || result.error || "unknown sandbox failure";
|
|
160
|
+
return { status: "fail", detail: `${label} never started: ${why}` };
|
|
161
|
+
}
|
|
162
|
+
const code = Number.isInteger(result.exitCode) ? result.exitCode : null;
|
|
163
|
+
if (code === null) {
|
|
164
|
+
return { status: "fail", detail: `${label} produced no exit code${tailOf(result)}` };
|
|
165
|
+
}
|
|
166
|
+
if (code !== 0) {
|
|
167
|
+
return {
|
|
168
|
+
status: "fail",
|
|
169
|
+
detail: `${label} failed with exit code ${code}${tailOf(result)}`,
|
|
170
|
+
};
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
return { status: "pass", detail: `${total} of ${total} commands passed` };
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// ---------------------------------------------------------------------------
|
|
178
|
+
// pure: output capping
|
|
179
|
+
// ---------------------------------------------------------------------------
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Cap captured output at `limit` bytes and say how much was dropped. A runaway
|
|
183
|
+
* test suite must not be able to exhaust the coordinator's memory or context.
|
|
184
|
+
*
|
|
185
|
+
* @param {unknown} value
|
|
186
|
+
* @param {number} limit
|
|
187
|
+
* @returns {{ text: string, dropped: number, truncated: boolean }}
|
|
188
|
+
*/
|
|
189
|
+
export function capOutput(value, limit = DEFAULT_MAX_OUTPUT_BYTES) {
|
|
190
|
+
const text = value === null || value === undefined ? "" : String(value);
|
|
191
|
+
const max = Number.isFinite(limit) && limit >= 0 ? Math.floor(limit) : DEFAULT_MAX_OUTPUT_BYTES;
|
|
192
|
+
const buffer = Buffer.from(text, "utf8");
|
|
193
|
+
if (buffer.length <= max) {
|
|
194
|
+
return { text, dropped: 0, truncated: false };
|
|
195
|
+
}
|
|
196
|
+
const dropped = buffer.length - max;
|
|
197
|
+
const kept = buffer.subarray(0, max).toString("utf8");
|
|
198
|
+
return {
|
|
199
|
+
text: `${kept}\n${TRUNCATION_MARKER}: ${dropped} of ${buffer.length} bytes dropped`,
|
|
200
|
+
dropped,
|
|
201
|
+
truncated: true,
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// ---------------------------------------------------------------------------
|
|
206
|
+
// pure: podman invocation
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Build the argv for one containerised command. Exported so the invocation is
|
|
211
|
+
* assertable in tests without Podman.
|
|
212
|
+
*
|
|
213
|
+
* The repository-controlled `command` is the FINAL argv element, handed to
|
|
214
|
+
* `/bin/sh -c` inside the container. It is never concatenated into a host
|
|
215
|
+
* string and the host process is always spawned with `shell: false`.
|
|
216
|
+
*
|
|
217
|
+
* @param {{ image?: string, cwd: string, command: string, jobId?: string|null,
|
|
218
|
+
* network?: string, user?: string, workdir?: string,
|
|
219
|
+
* nodeModulesSource?: string|null }} input
|
|
220
|
+
* @returns {string[]} argv for `podman`
|
|
221
|
+
*/
|
|
222
|
+
export function buildPodmanArgs({
|
|
223
|
+
image = DEFAULT_AGENT_IMAGE,
|
|
224
|
+
cwd,
|
|
225
|
+
command,
|
|
226
|
+
jobId = null,
|
|
227
|
+
network = "none",
|
|
228
|
+
user = DEFAULT_CONTAINER_USER,
|
|
229
|
+
workdir = DEFAULT_WORKDIR,
|
|
230
|
+
nodeModulesSource = null,
|
|
231
|
+
env = {},
|
|
232
|
+
} = {}) {
|
|
233
|
+
const args = [
|
|
234
|
+
"run",
|
|
235
|
+
"--rm",
|
|
236
|
+
`--network=${network}`,
|
|
237
|
+
`--user=${user}`,
|
|
238
|
+
`--workdir=${workdir}`,
|
|
239
|
+
"--cap-drop=ALL",
|
|
240
|
+
"--security-opt=no-new-privileges",
|
|
241
|
+
"--pids-limit=512",
|
|
242
|
+
"--env",
|
|
243
|
+
"HOME=/tmp",
|
|
244
|
+
"--env",
|
|
245
|
+
"NOMARMY_VERIFICATION=1",
|
|
246
|
+
...Object.entries(SANDBOX_NPM_ENV).flatMap(([k, v]) => ["--env", `${k}=${v}`]),
|
|
247
|
+
"--mount",
|
|
248
|
+
`type=bind,source=${cwd},target=${workdir}`,
|
|
249
|
+
];
|
|
250
|
+
// Extra env vars a caller wants inside the sandbox -- e.g. the diff's own
|
|
251
|
+
// changed-file lists (see createVerificationRunner's NOMARMY_CHANGED_*).
|
|
252
|
+
// Passed as real podman --env args, never spliced into `command` itself:
|
|
253
|
+
// this process is spawned with shell:false, so a value can never break out
|
|
254
|
+
// of its own argv slot on the HOST side the way string-templating the
|
|
255
|
+
// command itself would risk. A verification command that then references
|
|
256
|
+
// the var unquoted inside its OWN `/bin/sh -c` still word-splits on spaces
|
|
257
|
+
// as usual -- that is the operator's shell, not this one, and is worth
|
|
258
|
+
// knowing when a repo's paths might contain spaces.
|
|
259
|
+
for (const [name, value] of Object.entries(env)) args.push("--env", `${name}=${value}`);
|
|
260
|
+
// `git worktree add` never copies node_modules, and this container has
|
|
261
|
+
// --network none, so a Node repo's own verification commands can never
|
|
262
|
+
// install what they need. The coordinator's own already-installed tree is
|
|
263
|
+
// trusted (it was npm-installed on the host by a human, not produced by
|
|
264
|
+
// repo-controlled code), so it is safe to hand the sandbox READ ONLY --
|
|
265
|
+
// resolveNodeModulesMount (below) only supplies this when the worktree's
|
|
266
|
+
// package-lock.json matches the coordinator's, so a job that legitimately
|
|
267
|
+
// changed dependencies never gets silently tested against a stale tree.
|
|
268
|
+
if (nodeModulesSource) {
|
|
269
|
+
args.push("--mount", `type=bind,source=${nodeModulesSource},target=${workdir}/node_modules,readonly`);
|
|
270
|
+
}
|
|
271
|
+
if (jobId) args.push("--label", `nomarmy.job=${jobId}`);
|
|
272
|
+
args.push("--entrypoint", "/bin/sh", image, "-c", command);
|
|
273
|
+
return args;
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
// ---------------------------------------------------------------------------
|
|
277
|
+
// the real Podman executor (the only place a host process is spawned)
|
|
278
|
+
// ---------------------------------------------------------------------------
|
|
279
|
+
|
|
280
|
+
function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd } = {}) {
|
|
281
|
+
return new Promise((resolve) => {
|
|
282
|
+
let child;
|
|
283
|
+
try {
|
|
284
|
+
// shell:false is the whole point: nothing here is ever parsed by a host
|
|
285
|
+
// shell, so a config value cannot break out of its argv slot.
|
|
286
|
+
child = spawn(file, args, { cwd, shell: false, windowsHide: true });
|
|
287
|
+
} catch (error) {
|
|
288
|
+
resolve({ spawned: false, code: null, stdout: "", stderr: String(error?.message || error), timedOut: false });
|
|
289
|
+
return;
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
const cap = Number.isFinite(maxOutputBytes) ? maxOutputBytes : DEFAULT_MAX_OUTPUT_BYTES;
|
|
293
|
+
const out = [];
|
|
294
|
+
const err = [];
|
|
295
|
+
let outBytes = 0;
|
|
296
|
+
let errBytes = 0;
|
|
297
|
+
let settled = false;
|
|
298
|
+
let timedOut = false;
|
|
299
|
+
|
|
300
|
+
const collect = (chunks, chunk, bytes) => {
|
|
301
|
+
const room = cap - bytes;
|
|
302
|
+
if (room > 0) chunks.push(chunk.subarray(0, room));
|
|
303
|
+
return bytes + chunk.length;
|
|
304
|
+
};
|
|
305
|
+
|
|
306
|
+
child.stdout?.on("data", (chunk) => { outBytes = collect(out, chunk, outBytes); });
|
|
307
|
+
child.stderr?.on("data", (chunk) => { errBytes = collect(err, chunk, errBytes); });
|
|
308
|
+
|
|
309
|
+
const timer = Number.isFinite(timeoutMs) && timeoutMs > 0
|
|
310
|
+
? setTimeout(() => {
|
|
311
|
+
timedOut = true;
|
|
312
|
+
try { child.kill("SIGKILL"); } catch { /* already gone */ }
|
|
313
|
+
}, timeoutMs)
|
|
314
|
+
: null;
|
|
315
|
+
|
|
316
|
+
const finish = (code, spawnError) => {
|
|
317
|
+
if (settled) return;
|
|
318
|
+
settled = true;
|
|
319
|
+
if (timer) clearTimeout(timer);
|
|
320
|
+
resolve({
|
|
321
|
+
spawned: !spawnError,
|
|
322
|
+
code: typeof code === "number" ? code : null,
|
|
323
|
+
stdout: Buffer.concat(out).toString("utf8"),
|
|
324
|
+
stderr: spawnError
|
|
325
|
+
? `${Buffer.concat(err).toString("utf8")}${String(spawnError.message || spawnError)}`
|
|
326
|
+
: Buffer.concat(err).toString("utf8"),
|
|
327
|
+
timedOut,
|
|
328
|
+
});
|
|
329
|
+
};
|
|
330
|
+
|
|
331
|
+
child.on("error", (error) => finish(null, error));
|
|
332
|
+
child.on("close", (code) => finish(code, null));
|
|
333
|
+
});
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
/**
|
|
337
|
+
* Executor backed by the real Podman CLI. Tests inject a fake in its place.
|
|
338
|
+
* @param {{ podman?: string }} [options]
|
|
339
|
+
*/
|
|
340
|
+
export function createPodmanExecutor({ podman = "podman" } = {}) {
|
|
341
|
+
return {
|
|
342
|
+
/** Is a usable sandbox present? Never throws. */
|
|
343
|
+
async probe({ image }) {
|
|
344
|
+
const version = await spawnCollect(podman, ["version", "--format", "{{.Server.Version}}"], {
|
|
345
|
+
timeoutMs: 20_000,
|
|
346
|
+
maxOutputBytes: 4096,
|
|
347
|
+
});
|
|
348
|
+
if (!version.spawned) {
|
|
349
|
+
return { available: false, reason: `podman CLI is not available: ${version.stderr.trim() || "spawn failed"}` };
|
|
350
|
+
}
|
|
351
|
+
if (version.timedOut) {
|
|
352
|
+
return { available: false, reason: "podman did not respond within 20s" };
|
|
353
|
+
}
|
|
354
|
+
if (version.code !== 0) {
|
|
355
|
+
return {
|
|
356
|
+
available: false,
|
|
357
|
+
reason: `podman is not usable: ${version.stderr.trim() || `exit ${version.code}`}`,
|
|
358
|
+
};
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
const inspect = await spawnCollect(podman, ["image", "inspect", image], {
|
|
362
|
+
timeoutMs: 20_000,
|
|
363
|
+
maxOutputBytes: 4096,
|
|
364
|
+
});
|
|
365
|
+
if (!inspect.spawned || inspect.timedOut || inspect.code !== 0) {
|
|
366
|
+
return {
|
|
367
|
+
available: false,
|
|
368
|
+
reason: `sandbox image '${image}' is not present locally; build it with scripts/setup-sandbox.sh`,
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
return { available: true, reason: null };
|
|
372
|
+
},
|
|
373
|
+
|
|
374
|
+
/** Run one command inside the sandbox. Never throws. */
|
|
375
|
+
async run({ command, cwd, image, jobId, network, user, workdir, timeoutMs, maxOutputBytes, nodeModulesSource, env }) {
|
|
376
|
+
const args = buildPodmanArgs({ image, cwd, command, jobId, network, user, workdir, nodeModulesSource, env });
|
|
377
|
+
const started = Date.now();
|
|
378
|
+
const outcome = await spawnCollect(podman, args, { timeoutMs, maxOutputBytes });
|
|
379
|
+
const durationMs = Date.now() - started;
|
|
380
|
+
|
|
381
|
+
if (!outcome.spawned) {
|
|
382
|
+
return {
|
|
383
|
+
started: false,
|
|
384
|
+
reason: `podman could not be executed: ${outcome.stderr.trim() || "spawn failed"}`,
|
|
385
|
+
stdout: "",
|
|
386
|
+
stderr: outcome.stderr,
|
|
387
|
+
durationMs,
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
if (outcome.timedOut) {
|
|
391
|
+
return {
|
|
392
|
+
started: true,
|
|
393
|
+
timedOut: true,
|
|
394
|
+
exitCode: null,
|
|
395
|
+
stdout: outcome.stdout,
|
|
396
|
+
stderr: outcome.stderr,
|
|
397
|
+
durationMs,
|
|
398
|
+
};
|
|
399
|
+
}
|
|
400
|
+
// 125 is Podman's own "the run itself failed" code, mirroring Docker's
|
|
401
|
+
// convention: the container never started, so this is absence of
|
|
402
|
+
// evidence rather than a failing test.
|
|
403
|
+
if (outcome.code === 125) {
|
|
404
|
+
return {
|
|
405
|
+
started: false,
|
|
406
|
+
reason: `container failed to start: ${outcome.stderr.trim() || "podman exit 125"}`,
|
|
407
|
+
stdout: outcome.stdout,
|
|
408
|
+
stderr: outcome.stderr,
|
|
409
|
+
durationMs,
|
|
410
|
+
};
|
|
411
|
+
}
|
|
412
|
+
return {
|
|
413
|
+
started: true,
|
|
414
|
+
timedOut: false,
|
|
415
|
+
exitCode: outcome.code,
|
|
416
|
+
stdout: outcome.stdout,
|
|
417
|
+
stderr: outcome.stderr,
|
|
418
|
+
durationMs,
|
|
419
|
+
};
|
|
420
|
+
},
|
|
421
|
+
};
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
// ---------------------------------------------------------------------------
|
|
425
|
+
// the runner
|
|
426
|
+
// ---------------------------------------------------------------------------
|
|
427
|
+
|
|
428
|
+
function notRun(reason, basis, extra = {}) {
|
|
429
|
+
return { status: "not_run", basis, reason, detail: null, ...extra };
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
/**
|
|
433
|
+
* Decide whether the coordinator's own already-installed node_modules is safe
|
|
434
|
+
* to hand a verification run, read only. Never throws.
|
|
435
|
+
*
|
|
436
|
+
* `hostProjectDir` has no node_modules at all: nothing to offer, no-op (most
|
|
437
|
+
* repos are not Node projects, or the coordinator was never `npm install`ed).
|
|
438
|
+
* Both sides have a package-lock.json and they differ: the worktree changed
|
|
439
|
+
* dependencies, so the host's tree would not reflect what the worktree
|
|
440
|
+
* actually needs -- block rather than risk a false pass or a misleading
|
|
441
|
+
* "module not found" fail that is really an environment gap.
|
|
442
|
+
* Otherwise: match (or the worktree never touched its lockfile) -- mount it.
|
|
443
|
+
*
|
|
444
|
+
* @param {{ hostProjectDir: string, worktreeCwd: string }} input
|
|
445
|
+
* @returns {{ source: string|null, blockedReason: string|null }}
|
|
446
|
+
*/
|
|
447
|
+
export function resolveNodeModulesMount({ hostProjectDir, worktreeCwd }) {
|
|
448
|
+
const hostNodeModules = path.join(hostProjectDir, "node_modules");
|
|
449
|
+
if (!fs.existsSync(hostNodeModules)) return { source: null, blockedReason: null };
|
|
450
|
+
|
|
451
|
+
let hostLock = null, worktreeLock = null;
|
|
452
|
+
try { hostLock = fs.readFileSync(path.join(hostProjectDir, "package-lock.json")); } catch { hostLock = null; }
|
|
453
|
+
try { worktreeLock = fs.readFileSync(path.join(worktreeCwd, "package-lock.json")); } catch { worktreeLock = null; }
|
|
454
|
+
|
|
455
|
+
if (hostLock && worktreeLock && !hostLock.equals(worktreeLock)) {
|
|
456
|
+
return {
|
|
457
|
+
source: null,
|
|
458
|
+
blockedReason: "this worktree's package-lock.json differs from the coordinator's own; the coordinator's node_modules would not reflect the worktree's actual dependencies",
|
|
459
|
+
};
|
|
460
|
+
}
|
|
461
|
+
return { source: hostNodeModules, blockedReason: null };
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
function pluralCommands(count, name) {
|
|
465
|
+
return `${count} command${count === 1 ? "" : "s"} in profile '${name}'`;
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
/**
|
|
469
|
+
* Build the verification runner to hand to `registerVerificationRunner`.
|
|
470
|
+
*
|
|
471
|
+
* @param {{
|
|
472
|
+
* loadConfig?: (repoDir: string) => object,
|
|
473
|
+
* executor?: { probe: Function, run: Function },
|
|
474
|
+
* image?: string,
|
|
475
|
+
* network?: string,
|
|
476
|
+
* user?: string,
|
|
477
|
+
* workdir?: string,
|
|
478
|
+
* commandTimeoutMs?: number,
|
|
479
|
+
* overallTimeoutMs?: number,
|
|
480
|
+
* maxOutputBytes?: number,
|
|
481
|
+
* now?: () => number,
|
|
482
|
+
* hostProjectDir?: string|null,
|
|
483
|
+
* }} [options]
|
|
484
|
+
* @returns {(context: object) => Promise<{status: string, basis?: string, reason?: string|null, detail?: string|null}>}
|
|
485
|
+
*/
|
|
486
|
+
export function createVerificationRunner(options = {}) {
|
|
487
|
+
const {
|
|
488
|
+
loadConfig = defaultLoadConfig,
|
|
489
|
+
executor = createPodmanExecutor(),
|
|
490
|
+
image = null,
|
|
491
|
+
network = "none",
|
|
492
|
+
user = DEFAULT_CONTAINER_USER,
|
|
493
|
+
workdir = DEFAULT_WORKDIR,
|
|
494
|
+
commandTimeoutMs = DEFAULT_COMMAND_TIMEOUT_MS,
|
|
495
|
+
overallTimeoutMs = DEFAULT_OVERALL_TIMEOUT_MS,
|
|
496
|
+
maxOutputBytes = DEFAULT_MAX_OUTPUT_BYTES,
|
|
497
|
+
now = () => Date.now(),
|
|
498
|
+
// The coordinator's own checkout, source of a trusted, already-installed
|
|
499
|
+
// node_modules a job's worktree never has (git worktree add does not copy
|
|
500
|
+
// it). Unset by default: existing callers/tests see no behavior change.
|
|
501
|
+
hostProjectDir = null,
|
|
502
|
+
// Shells out for the lazy Go/Rust image build/existence check -- distinct
|
|
503
|
+
// from `executor`, which runs commands INSIDE an already-built sandbox.
|
|
504
|
+
// Injectable so tests never invoke real Podman just by pointing `cwd` at
|
|
505
|
+
// a directory that happens to contain a go.mod/Cargo.toml.
|
|
506
|
+
sandboxImageRun = undefined,
|
|
507
|
+
} = options;
|
|
508
|
+
|
|
509
|
+
return async function runVerification(context = {}) {
|
|
510
|
+
const { profile = null, cwd = null, jobId = null, record = null } = context;
|
|
511
|
+
|
|
512
|
+
if (!cwd || typeof cwd !== "string") {
|
|
513
|
+
return notRun("no worktree path was supplied, so nothing could be verified", "no-worktree");
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
// Language-agnostic by design: nomArmy already knows exactly which files
|
|
517
|
+
// this diff touched (classifyTestChanges already handles Python/Go/JS/
|
|
518
|
+
// TS/Ruby/JVM naming conventions), and exposes that as plain env vars
|
|
519
|
+
// rather than trying to know every test runner's own CLI convention for
|
|
520
|
+
// "run just these files" (pytest and jest take file paths directly; Go
|
|
521
|
+
// wants package directories; cargo wants test binary names -- nomArmy
|
|
522
|
+
// has no business encoding all of that). A verification command that
|
|
523
|
+
// never references these behaves exactly as before this existed.
|
|
524
|
+
// Deleted test files are never included -- nothing to run.
|
|
525
|
+
const testChanges = record?.testChanges;
|
|
526
|
+
const verificationEnv = {
|
|
527
|
+
NOMARMY_CHANGED_TEST_FILES: [...(testChanges?.new_tests_added ?? []), ...(testChanges?.existing_tests_modified ?? [])].join(" "),
|
|
528
|
+
NOMARMY_CHANGED_PRODUCTION_FILES: (testChanges?.production_files_changed ?? []).join(" "),
|
|
529
|
+
};
|
|
530
|
+
|
|
531
|
+
// Loaded before sandbox image resolution, not after: a Python repo's
|
|
532
|
+
// dependency image needs environment.python.requirements from this same
|
|
533
|
+
// config to know what to install (see pythonRequirementsFor).
|
|
534
|
+
let loaded;
|
|
535
|
+
try {
|
|
536
|
+
loaded = loadConfig(cwd);
|
|
537
|
+
} catch (error) {
|
|
538
|
+
// A broken contract is not a failing test suite. Say so and stop.
|
|
539
|
+
return notRun(
|
|
540
|
+
`.nomarmy.yml could not be loaded: ${String(error?.message || error).split("\n")[0]}`,
|
|
541
|
+
"config-error",
|
|
542
|
+
);
|
|
543
|
+
}
|
|
544
|
+
const config = loaded && loaded.found ? loaded.config : null;
|
|
545
|
+
|
|
546
|
+
const explicitImage = image || process.env.NOMARMY_AGENT_IMAGE || null;
|
|
547
|
+
let sandboxImage;
|
|
548
|
+
try {
|
|
549
|
+
sandboxImage = resolveSandboxImage({
|
|
550
|
+
cwd, explicitImage, defaultImage: DEFAULT_AGENT_IMAGE, config,
|
|
551
|
+
...(sandboxImageRun ? { run: sandboxImageRun } : {}),
|
|
552
|
+
});
|
|
553
|
+
} catch (error) {
|
|
554
|
+
// A lazy Go/Rust/Python image build failed (offline, disk full,
|
|
555
|
+
// apt/rustup/pip error). Running the job's real commands against the
|
|
556
|
+
// wrong image would look like a genuine test failure rather than the
|
|
557
|
+
// infrastructure gap it actually is -- not_run is the honest state.
|
|
558
|
+
return notRun(error.message, "sandbox-image-build-failed");
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
// --- resolve what to run -------------------------------------------
|
|
562
|
+
const resolved = resolveProfile({ config, profile: profile ?? null });
|
|
563
|
+
if (!resolved.found) {
|
|
564
|
+
const basis = !profile ? "no-profile-requested" : !config ? "no-config" : "unknown-profile";
|
|
565
|
+
return notRun(resolved.reason, basis);
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
if (resolved.environment !== EXECUTABLE_ENVIRONMENT) {
|
|
569
|
+
// Running these commands without their services would produce a
|
|
570
|
+
// misleading `fail`. An unmet requirement is a `not_run`.
|
|
571
|
+
return notRun(
|
|
572
|
+
`profile '${resolved.name}' requires environment '${resolved.environment}', which needs provisioned services this component does not create; only 'environment: none' is executable today`,
|
|
573
|
+
"environment-not-provisioned",
|
|
574
|
+
);
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
const commands = resolved.commands;
|
|
578
|
+
if (commands.length === 0) {
|
|
579
|
+
return notRun(
|
|
580
|
+
`verification profile '${resolved.name}' lists no commands, so there is nothing to verify`,
|
|
581
|
+
"no-commands",
|
|
582
|
+
);
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
const basis = pluralCommands(commands.length, resolved.name);
|
|
586
|
+
|
|
587
|
+
// --- require the sandbox -------------------------------------------
|
|
588
|
+
// No host fallback exists below this line, by design.
|
|
589
|
+
let probe;
|
|
590
|
+
try {
|
|
591
|
+
probe = await executor.probe({ image: sandboxImage });
|
|
592
|
+
} catch (error) {
|
|
593
|
+
probe = { available: false, reason: `sandbox probe failed: ${String(error?.message || error)}` };
|
|
594
|
+
}
|
|
595
|
+
if (!probe || probe.available !== true) {
|
|
596
|
+
return notRun(
|
|
597
|
+
`${probe?.reason || "the Podman sandbox is unavailable"}; verification commands come from the repository and are never executed on the host`,
|
|
598
|
+
"sandbox-unavailable",
|
|
599
|
+
);
|
|
600
|
+
}
|
|
601
|
+
|
|
602
|
+
// --- offer the coordinator's own node_modules, if it is safe to -----
|
|
603
|
+
// Not when the sandbox image already holds this job's own dependencies
|
|
604
|
+
// (an npm lockfile: lib/sandbox-images.mjs). Those are built on Linux
|
|
605
|
+
// from the worktree's lockfile; the host's tree is built for the host
|
|
606
|
+
// (a Mac's esbuild and rollup binaries don't run in the Linux sandbox)
|
|
607
|
+
// and would shadow them. The mount stays for yarn, pnpm and workspaces.
|
|
608
|
+
let nodeModulesSource = null;
|
|
609
|
+
// A union or recovery worktree nomArmy made itself may not have the
|
|
610
|
+
// package links yet; idempotent.
|
|
611
|
+
try { linkNodePackages(cwd, config); } catch { /* best-effort */ }
|
|
612
|
+
if (hostProjectDir && !nodeDependencyFiles(cwd, config).files.length) {
|
|
613
|
+
const mount = resolveNodeModulesMount({ hostProjectDir, worktreeCwd: cwd });
|
|
614
|
+
if (mount.blockedReason) return notRun(mount.blockedReason, "dependency-drift");
|
|
615
|
+
nodeModulesSource = mount.source;
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
// --- execute, sequentially, inside the sandbox ----------------------
|
|
619
|
+
const deadline = now() + overallTimeoutMs;
|
|
620
|
+
const results = [];
|
|
621
|
+
|
|
622
|
+
for (const command of commands) {
|
|
623
|
+
const remaining = deadline - now();
|
|
624
|
+
if (remaining <= 0) {
|
|
625
|
+
results.push({
|
|
626
|
+
command,
|
|
627
|
+
started: false,
|
|
628
|
+
timedOut: true,
|
|
629
|
+
reason: `overall verification budget of ${overallTimeoutMs}ms was exhausted before this command started`,
|
|
630
|
+
timeoutMs: overallTimeoutMs,
|
|
631
|
+
});
|
|
632
|
+
break;
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
const timeoutMs = Math.max(1, Math.min(commandTimeoutMs, remaining));
|
|
636
|
+
let raw;
|
|
637
|
+
try {
|
|
638
|
+
raw = await executor.run({
|
|
639
|
+
command,
|
|
640
|
+
cwd,
|
|
641
|
+
image: sandboxImage,
|
|
642
|
+
jobId,
|
|
643
|
+
network,
|
|
644
|
+
user,
|
|
645
|
+
workdir,
|
|
646
|
+
timeoutMs,
|
|
647
|
+
maxOutputBytes,
|
|
648
|
+
nodeModulesSource,
|
|
649
|
+
env: verificationEnv,
|
|
650
|
+
});
|
|
651
|
+
} catch (error) {
|
|
652
|
+
raw = { started: false, reason: `sandbox execution threw: ${String(error?.message || error)}` };
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
const stdout = capOutput(raw?.stdout, maxOutputBytes);
|
|
656
|
+
const stderr = capOutput(raw?.stderr, maxOutputBytes);
|
|
657
|
+
const entry = {
|
|
658
|
+
command,
|
|
659
|
+
started: raw?.started !== false,
|
|
660
|
+
timedOut: Boolean(raw?.timedOut),
|
|
661
|
+
exitCode: Number.isInteger(raw?.exitCode) ? raw.exitCode : null,
|
|
662
|
+
reason: raw?.reason ?? null,
|
|
663
|
+
durationMs: Number.isFinite(raw?.durationMs) ? raw.durationMs : null,
|
|
664
|
+
timeoutMs,
|
|
665
|
+
stdout: stdout.text,
|
|
666
|
+
stderr: stderr.text,
|
|
667
|
+
truncated: { stdout: stdout.dropped, stderr: stderr.dropped },
|
|
668
|
+
};
|
|
669
|
+
results.push(entry);
|
|
670
|
+
|
|
671
|
+
// Stop at the first command that did not cleanly pass: later commands
|
|
672
|
+
// would run against an already-broken tree and add nothing.
|
|
673
|
+
if (!entry.started || entry.timedOut || entry.exitCode !== 0) break;
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
const verdict = classifyResults(results);
|
|
677
|
+
const executed = results.filter((r) => r.started).length;
|
|
678
|
+
const dropped = results.reduce(
|
|
679
|
+
(total, r) => total + (r.truncated?.stdout ?? 0) + (r.truncated?.stderr ?? 0),
|
|
680
|
+
0,
|
|
681
|
+
);
|
|
682
|
+
|
|
683
|
+
const detailSuffix = dropped > 0
|
|
684
|
+
? ` [${dropped} bytes of output dropped by the ${maxOutputBytes}-byte cap]`
|
|
685
|
+
: "";
|
|
686
|
+
|
|
687
|
+
if (verdict.status === "not_run") {
|
|
688
|
+
return notRun(verdict.detail, "sandbox-unavailable");
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
return {
|
|
692
|
+
status: verdict.status,
|
|
693
|
+
basis: `${basis}; ${executed} executed in ${sandboxImage}`,
|
|
694
|
+
reason: null,
|
|
695
|
+
detail: `${verdict.detail}${detailSuffix}`,
|
|
696
|
+
};
|
|
697
|
+
};
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
export default createVerificationRunner;
|