nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -480
- package/bin/nomarmy.mjs +1081 -185
- package/docker/Dockerfile +2 -2
- package/docker/Dockerfile.go +6 -4
- package/docker/Dockerfile.rust +17 -2
- package/harnesses/_template/README.md +27 -0
- package/harnesses/_template/harness.yml +26 -0
- package/harnesses/browser-playwright/README.md +35 -0
- package/harnesses/browser-playwright/fixture/package.json +1 -0
- package/harnesses/browser-playwright/fixture/page.html +1 -0
- package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
- package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
- package/harnesses/browser-playwright/harness.yml +18 -0
- package/harnesses/go/README.md +45 -0
- package/harnesses/go/harness.yml +14 -0
- package/harnesses/mock-oidc/README.md +31 -0
- package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
- package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
- package/harnesses/mock-oidc/harness.yml +19 -0
- package/harnesses/node/README.md +53 -0
- package/harnesses/node/harness.yml +18 -0
- package/harnesses/python/README.md +46 -0
- package/harnesses/python/harness.yml +16 -0
- package/harnesses/rust/README.md +45 -0
- package/harnesses/rust/harness.yml +13 -0
- package/install.sh +29 -9
- package/lib/admission.mjs +178 -30
- package/lib/agents.mjs +8 -6
- package/lib/army.mjs +25 -10
- package/lib/codex-link.mjs +37 -0
- package/lib/config.mjs +15 -0
- package/lib/connect.mjs +232 -19
- package/lib/continue-from.mjs +103 -0
- package/lib/coordinator-instructions.mjs +5 -1
- package/lib/diff-checks.mjs +114 -0
- package/lib/dispatch-schema.mjs +14 -12
- package/lib/doctor.mjs +98 -9
- package/lib/egress-proxy.mjs +116 -0
- package/lib/execute.mjs +241 -33
- package/lib/git-record.mjs +27 -3
- package/lib/harness-schema.mjs +61 -0
- package/lib/harnesses.mjs +99 -0
- package/lib/health.mjs +162 -18
- package/lib/install-freshness.mjs +114 -0
- package/lib/jev-checks.mjs +110 -0
- package/lib/job-format.mjs +54 -0
- package/lib/judge.mjs +130 -0
- package/lib/limits.mjs +77 -0
- package/lib/model-probe.mjs +61 -0
- package/lib/mutation.mjs +159 -0
- package/lib/notify.mjs +30 -3
- package/lib/openclaw-install.mjs +122 -0
- package/lib/openclaw-path.mjs +28 -0
- package/lib/openclaw-run.mjs +74 -12
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/outcome.mjs +21 -2
- package/lib/outcomes.mjs +6 -0
- package/lib/path-utils.mjs +4 -0
- package/lib/podman-health.mjs +41 -0
- package/lib/process.mjs +4 -1
- package/lib/propose.mjs +10 -11
- package/lib/refusal-retry.mjs +16 -0
- package/lib/registry-python.mjs +98 -0
- package/lib/registry-secrets.mjs +140 -0
- package/lib/repo-query.mjs +13 -7
- package/lib/runs.mjs +7 -1
- package/lib/same-path.mjs +14 -0
- package/lib/sandbox-images.mjs +499 -83
- package/lib/sandbox-vm.mjs +32 -0
- package/lib/scan.mjs +5 -1
- package/lib/schema.mjs +20 -11
- package/lib/scout.mjs +21 -3
- package/lib/server-context.mjs +21 -1
- package/lib/setup-steps.mjs +55 -0
- package/lib/share.mjs +82 -0
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +315 -0
- package/lib/statusline.mjs +32 -6
- package/lib/subscription-setup.mjs +13 -0
- package/lib/suggestions.mjs +153 -0
- package/lib/thinking.mjs +23 -0
- package/lib/transcript.mjs +30 -5
- package/lib/usage-limits.mjs +329 -0
- package/lib/user-config.mjs +106 -0
- package/lib/validators.mjs +220 -0
- package/lib/verification-artifacts.mjs +46 -0
- package/lib/verification-flow.mjs +52 -7
- package/lib/verification-network.mjs +66 -0
- package/lib/verify.mjs +338 -85
- package/lib/worker-prompt.mjs +5 -2
- package/lib/wsl-cli.mjs +152 -0
- package/lib/wsl.mjs +230 -0
- package/lib/zod-issues.mjs +15 -0
- package/mcp/server.mjs +165 -34
- package/package.json +7 -5
- package/playbooks/feature.md +8 -5
- package/scripts/configure-openclaw.sh +4 -2
- package/scripts/generate-harness-docs.mjs +42 -0
- package/scripts/install-openclaw.mjs +23 -0
- package/scripts/lib.sh +9 -2
- package/scripts/select-model.mjs +12 -5
- package/scripts/start-inference.sh +2 -2
package/lib/verify.mjs
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
// to repo-controlled code, the exact privilege the worker itself is denied.
|
|
14
14
|
//
|
|
15
15
|
// Therefore commands only ever execute inside the Podman sandbox image, as a
|
|
16
|
-
// non-root user, with `--network none
|
|
16
|
+
// non-root user, with `--network none` or a verification-only internal service network. If Podman is missing, the image is
|
|
17
17
|
// absent, or the container fails to start, the verdict is `not_run`. There is
|
|
18
18
|
// no host fallback path in this file, deliberately: a missing sandbox is
|
|
19
19
|
// absence of evidence, not permission to take a shortcut.
|
|
@@ -22,12 +22,17 @@
|
|
|
22
22
|
// is spawned with an argv array and `shell: false`. A command string is handed
|
|
23
23
|
// to `/bin/sh -c` *inside* the container, which is inside the sandbox boundary.
|
|
24
24
|
|
|
25
|
+
import { randomUUID, randomBytes } from "node:crypto";
|
|
25
26
|
import { spawn } from "node:child_process";
|
|
26
27
|
import fs from "node:fs";
|
|
28
|
+
import os from "node:os";
|
|
29
|
+
import { collectVerificationArtifacts } from "./verification-artifacts.mjs";
|
|
27
30
|
import path from "node:path";
|
|
31
|
+
import { fileURLToPath } from "node:url";
|
|
32
|
+
import { loadVerificationNetwork, redactCredentials } from "./verification-network.mjs";
|
|
28
33
|
|
|
29
34
|
import { loadConfig as defaultLoadConfig } from "./config.mjs";
|
|
30
|
-
import { resolveSandboxImage, nodeDependencyFiles, linkNodePackages, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
|
|
35
|
+
import { resolveSandboxImage, sandboxHarnesses, nodeDependencyFiles, linkNodePackages, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
|
|
31
36
|
|
|
32
37
|
/** Sandbox image used when `NOMARMY_AGENT_IMAGE` is unset. */
|
|
33
38
|
export const DEFAULT_AGENT_IMAGE = "openclaw-nomarmy-coder:bookworm";
|
|
@@ -130,10 +135,19 @@ function tailOf(result) {
|
|
|
130
135
|
* command that never started is absence of evidence, so if nothing ever ran the
|
|
131
136
|
* verdict is `not_run`, never `fail`.
|
|
132
137
|
*
|
|
138
|
+
* A command guarded on nomArmy's changed-file variables (`if [ -n
|
|
139
|
+
* "$NOMARMY_CHANGED_TEST_FILES" ]; then ...; fi`) exits 0 without running
|
|
140
|
+
* anything when those are empty, as on an unchanged checkout in mode: verify.
|
|
141
|
+
* That's skipped, not passed: exit 0, no output at all, and every
|
|
142
|
+
* NOMARMY_CHANGED_* variable the command names empty. A command with a
|
|
143
|
+
* fallback (`${NOMARMY_CHANGED_TEST_FILES:-tests/}`) prints test output, so
|
|
144
|
+
* it counts as run. If every command skipped, nothing was verified: not_run.
|
|
145
|
+
*
|
|
133
146
|
* @param {Array<object>} results
|
|
134
|
-
* @
|
|
147
|
+
* @param {{ env?: Record<string, string> }} [context] the NOMARMY_CHANGED_* values the commands saw
|
|
148
|
+
* @returns {{ status: "pass"|"fail"|"not_run", detail: string, basis?: string }}
|
|
135
149
|
*/
|
|
136
|
-
export function classifyResults(results) {
|
|
150
|
+
export function classifyResults(results, { env = {} } = {}) {
|
|
137
151
|
const list = Array.isArray(results) ? results.filter(Boolean) : [];
|
|
138
152
|
if (list.length === 0) {
|
|
139
153
|
return { status: "not_run", detail: "no commands were executed" };
|
|
@@ -171,9 +185,28 @@ export function classifyResults(results) {
|
|
|
171
185
|
}
|
|
172
186
|
}
|
|
173
187
|
|
|
188
|
+
const skipped = list.map((result, index) => ({ index, why: guardedNoOp(result, env) })).filter((s) => s.why);
|
|
189
|
+
if (skipped.length === total) {
|
|
190
|
+
return { status: "not_run", basis: "nothing-changed", detail: `every command is scoped to changed files and nothing changed, so none ran (${[...new Set(skipped.map((s) => s.why))].join("; ")}); add an unscoped profile to run the full suite` };
|
|
191
|
+
}
|
|
192
|
+
if (skipped.length) {
|
|
193
|
+
const which = skipped.map((s) => `command ${s.index + 1} (\`${list[s.index].command ?? "?"}\`) did nothing: ${s.why}`).join("; ");
|
|
194
|
+
return { status: "pass", detail: `${total - skipped.length} of ${total} commands passed; ${skipped.length} skipped: ${which}` };
|
|
195
|
+
}
|
|
174
196
|
return { status: "pass", detail: `${total} of ${total} commands passed` };
|
|
175
197
|
}
|
|
176
198
|
|
|
199
|
+
const CHANGED_VAR_RE = /\$\{?(NOMARMY_CHANGED_[A-Z_]+)/g;
|
|
200
|
+
|
|
201
|
+
/** Why a zero-exit, silent command was a guarded no-op, or null if it ran. */
|
|
202
|
+
function guardedNoOp(result, env) {
|
|
203
|
+
if (result.exitCode !== 0 || result.timedOut || result.started === false) return null;
|
|
204
|
+
if (result.silent === false || (result.silent === undefined && `${result.stdout ?? ""}${result.stderr ?? ""}`.trim())) return null;
|
|
205
|
+
const names = [...new Set([...String(result.command ?? "").matchAll(CHANGED_VAR_RE)].map((m) => m[1]))];
|
|
206
|
+
if (!names.length || names.some((name) => String(env[name] ?? "").trim())) return null;
|
|
207
|
+
return `${names.join(" and ")} ${names.length > 1 ? "are" : "is"} empty`;
|
|
208
|
+
}
|
|
209
|
+
|
|
177
210
|
// ---------------------------------------------------------------------------
|
|
178
211
|
// pure: output capping
|
|
179
212
|
// ---------------------------------------------------------------------------
|
|
@@ -229,6 +262,8 @@ export function buildPodmanArgs({
|
|
|
229
262
|
workdir = DEFAULT_WORKDIR,
|
|
230
263
|
nodeModulesSource = null,
|
|
231
264
|
env = {},
|
|
265
|
+
secretEnv = [],
|
|
266
|
+
shmMb = null,
|
|
232
267
|
} = {}) {
|
|
233
268
|
const args = [
|
|
234
269
|
"run",
|
|
@@ -256,7 +291,7 @@ export function buildPodmanArgs({
|
|
|
256
291
|
// the var unquoted inside its OWN `/bin/sh -c` still word-splits on spaces
|
|
257
292
|
// as usual -- that is the operator's shell, not this one, and is worth
|
|
258
293
|
// knowing when a repo's paths might contain spaces.
|
|
259
|
-
for (const [name, value] of Object.entries(env)) args.push("--env", `${name}=${value}`);
|
|
294
|
+
for (const [name, value] of Object.entries(env)) args.push("--env", secretEnv.includes(name) ? name : `${name}=${value}`);
|
|
260
295
|
// `git worktree add` never copies node_modules, and this container has
|
|
261
296
|
// --network none, so a Node repo's own verification commands can never
|
|
262
297
|
// install what they need. The coordinator's own already-installed tree is
|
|
@@ -268,6 +303,7 @@ export function buildPodmanArgs({
|
|
|
268
303
|
if (nodeModulesSource) {
|
|
269
304
|
args.push("--mount", `type=bind,source=${nodeModulesSource},target=${workdir}/node_modules,readonly`);
|
|
270
305
|
}
|
|
306
|
+
if (shmMb > 0) args.push(`--shm-size=${shmMb}m`);
|
|
271
307
|
if (jobId) args.push("--label", `nomarmy.job=${jobId}`);
|
|
272
308
|
args.push("--entrypoint", "/bin/sh", image, "-c", command);
|
|
273
309
|
return args;
|
|
@@ -277,13 +313,13 @@ export function buildPodmanArgs({
|
|
|
277
313
|
// the real Podman executor (the only place a host process is spawned)
|
|
278
314
|
// ---------------------------------------------------------------------------
|
|
279
315
|
|
|
280
|
-
function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd } = {}) {
|
|
316
|
+
function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd, env } = {}) {
|
|
281
317
|
return new Promise((resolve) => {
|
|
282
318
|
let child;
|
|
283
319
|
try {
|
|
284
320
|
// shell:false is the whole point: nothing here is ever parsed by a host
|
|
285
321
|
// shell, so a config value cannot break out of its argv slot.
|
|
286
|
-
child = spawn(file, args, { cwd, shell: false, windowsHide: true });
|
|
322
|
+
child = spawn(file, args, { cwd, env, shell: false, windowsHide: true });
|
|
287
323
|
} catch (error) {
|
|
288
324
|
resolve({ spawned: false, code: null, stdout: "", stderr: String(error?.message || error), timedOut: false });
|
|
289
325
|
return;
|
|
@@ -337,11 +373,116 @@ function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd } = {}) {
|
|
|
337
373
|
* Executor backed by the real Podman CLI. Tests inject a fake in its place.
|
|
338
374
|
* @param {{ podman?: string }} [options]
|
|
339
375
|
*/
|
|
340
|
-
export function createPodmanExecutor({ podman = "podman" } = {}) {
|
|
376
|
+
export function createPodmanExecutor({ podman = "podman", collect = spawnCollect, healthTimeoutMs = 30_000, healthIntervalMs = 250 } = {}) {
|
|
341
377
|
return {
|
|
378
|
+
/** Own only resources created by this run; cleanup is also used on setup failure. */
|
|
379
|
+
async startServices({ services, image, jobId, allow = [], token }) {
|
|
380
|
+
const network = `nomarmy-${String(jobId || randomUUID()).replace(/[^a-zA-Z0-9_-]/g, "-")}`;
|
|
381
|
+
const containers = [];
|
|
382
|
+
let created = false;
|
|
383
|
+
let externalCreated = false;
|
|
384
|
+
const external = `${network}-external`;
|
|
385
|
+
const proxyName = `${network}-egress`;
|
|
386
|
+
const call = (args, timeoutMs = 20_000) => collect(podman, args, { timeoutMs, maxOutputBytes: 4096, env: { ...process.env, NOMARMY_EGRESS_TOKEN: token } });
|
|
387
|
+
const ok = (result) => result.spawned && !result.timedOut && result.code === 0;
|
|
388
|
+
const requireSuccess = async (args, label, timeoutMs) => {
|
|
389
|
+
if (!ok(await call(args, timeoutMs))) throw new Error(label);
|
|
390
|
+
};
|
|
391
|
+
const cleanup = async () => {
|
|
392
|
+
const failures = [];
|
|
393
|
+
try { await this.cleanupVerification({ jobId }); }
|
|
394
|
+
catch { failures.push("could not remove verification containers"); }
|
|
395
|
+
for (const name of containers.reverse()) {
|
|
396
|
+
try { await requireSuccess(["rm", "--force", "--ignore", name], `could not remove service container ${name}`); }
|
|
397
|
+
catch (error) { failures.push(error.message); }
|
|
398
|
+
}
|
|
399
|
+
if (externalCreated) {
|
|
400
|
+
try { await requireSuccess(["network", "rm", external], "could not remove egress network"); }
|
|
401
|
+
catch (error) { failures.push(error.message); }
|
|
402
|
+
}
|
|
403
|
+
if (created) {
|
|
404
|
+
try { await requireSuccess(["network", "rm", network], `could not remove service network ${network}`); }
|
|
405
|
+
catch (error) { failures.push(error.message); }
|
|
406
|
+
}
|
|
407
|
+
if (failures.length) throw new Error(failures.join("; "));
|
|
408
|
+
};
|
|
409
|
+
try {
|
|
410
|
+
await requireSuccess(["network", "create", "--internal", network], `could not create internal service network ${network}`);
|
|
411
|
+
created = true;
|
|
412
|
+
if (allow.length) {
|
|
413
|
+
await requireSuccess(["network", "create", external], "could not create egress network");
|
|
414
|
+
externalCreated = true;
|
|
415
|
+
containers.push(proxyName);
|
|
416
|
+
await requireSuccess(["run", "--detach", "--name", proxyName,
|
|
417
|
+
"--network", `${network}:alias=egress`, "--network", external,
|
|
418
|
+
`--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
|
|
419
|
+
"--read-only", "--pids-limit=128", "--memory=128m", "--cpus=1",
|
|
420
|
+
"--mount", `type=bind,source=${fileURLToPath(new URL("./egress-proxy.mjs", import.meta.url))},target=/egress-proxy.mjs,readonly`,
|
|
421
|
+
"--env", "NOMARMY_EGRESS_TOKEN",
|
|
422
|
+
"--env", `NOMARMY_EGRESS_ALLOW=${JSON.stringify(allow)}`,
|
|
423
|
+
"--entrypoint", "node", DEFAULT_AGENT_IMAGE, "/egress-proxy.mjs"], "could not start egress proxy");
|
|
424
|
+
const probeName = `${network}-egress-health`;
|
|
425
|
+
containers.push(probeName);
|
|
426
|
+
const script = `const net=await import('node:net');const end=Date.now()+${healthTimeoutMs};while(Date.now()<end){const ok=await new Promise(r=>{const s=net.connect(3128,'egress');s.setTimeout(1000);s.on('connect',()=>{s.destroy();r(true)});s.on('error',()=>r(false));s.on('timeout',()=>{s.destroy();r(false)});});if(ok)process.exit(0);await new Promise(r=>setTimeout(r,${healthIntervalMs}));}process.exit(1);`;
|
|
427
|
+
await requireSuccess(["run", "--rm", "--name", probeName, "--network", network,
|
|
428
|
+
`--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
|
|
429
|
+
"--entrypoint", "node", DEFAULT_AGENT_IMAGE, "--input-type=module", "-e", script], "egress proxy did not become ready", healthTimeoutMs + 5000);
|
|
430
|
+
}
|
|
431
|
+
for (const service of services) {
|
|
432
|
+
if (!ok(await call(["image", "exists", service.image]))) {
|
|
433
|
+
await requireSuccess(["pull", service.image], `could not pull service image ${service.image}`, 120_000);
|
|
434
|
+
}
|
|
435
|
+
const name = `${network}-service-${service.name}`;
|
|
436
|
+
containers.push(name);
|
|
437
|
+
await requireSuccess(["run", "--detach", "--name", name, "--network", network,
|
|
438
|
+
"--network-alias", service.name, "--cap-drop=ALL", "--security-opt=no-new-privileges",
|
|
439
|
+
"--pids-limit=512", ...Object.entries(service.env ?? {}).flatMap(([k, v]) => ["--env", `${k}=${v}`]),
|
|
440
|
+
service.image], `could not start service ${service.name}`);
|
|
441
|
+
}
|
|
442
|
+
for (const service of services.filter((spec) => spec.health)) {
|
|
443
|
+
const probeName = `${network}-health-${service.name}`;
|
|
444
|
+
containers.push(probeName);
|
|
445
|
+
const url = `http://${service.name}:${service.port}${service.health}`;
|
|
446
|
+
// Poll inside the same internal network, never on the host. The base
|
|
447
|
+
// verification image supplies Node, so services need no curl binary.
|
|
448
|
+
const script = `const deadline=Date.now()+${healthTimeoutMs}; while(Date.now()<deadline){try{const r=await fetch(${JSON.stringify(url)},{redirect:'error',signal:AbortSignal.timeout(Math.min(2000,Math.max(1,deadline-Date.now())))});if(r.status>=200&&r.status<300)process.exit(0);}catch{}await new Promise(r=>setTimeout(r,${healthIntervalMs}));}process.exit(1);`;
|
|
449
|
+
await requireSuccess(["run", "--rm", "--name", probeName, "--network", network,
|
|
450
|
+
`--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
|
|
451
|
+
"--entrypoint", "node", image, "--input-type=module", "-e", script],
|
|
452
|
+
`service ${service.name} did not become healthy within ${healthTimeoutMs}ms (${url})`, healthTimeoutMs + 5000);
|
|
453
|
+
}
|
|
454
|
+
return { network, cleanup, ...(allow.length ? { logs: async () => {
|
|
455
|
+
// Freeze the audit trail before reading it; no late DNS completion or
|
|
456
|
+
// service connection may append a verdict after this snapshot.
|
|
457
|
+
await requireSuccess(["stop", "--time", "2", proxyName], "could not stop egress proxy");
|
|
458
|
+
const logLimit = 4 * 1024 * 1024;
|
|
459
|
+
const result = await collect(podman, ["logs", proxyName], { timeoutMs: 20_000, maxOutputBytes: logLimit });
|
|
460
|
+
// Never accept verification with a silently incomplete audit trail.
|
|
461
|
+
if (!ok(result) || Buffer.byteLength(result.stdout) >= logLimit) throw new Error("could not capture complete egress verdict log");
|
|
462
|
+
// Accept only the proxy's fixed verdict format; never relay diagnostics.
|
|
463
|
+
return result.stdout.split("\n").filter(line => /^[a-z0-9.-]+:[0-9]+ (connected|denied)$/.test(line)).join("\n");
|
|
464
|
+
} } : {}) };
|
|
465
|
+
} catch (error) {
|
|
466
|
+
try { await cleanup(); } catch (cleanupError) { throw new Error(`${error.message}; ${cleanupError.message}`); }
|
|
467
|
+
throw error;
|
|
468
|
+
}
|
|
469
|
+
},
|
|
470
|
+
|
|
471
|
+
async cleanupVerification({ jobId }) {
|
|
472
|
+
if (!jobId) return;
|
|
473
|
+
const listed = await collect(podman, ["ps", "-aq", "--filter", `label=nomarmy.job=${jobId}`], { timeoutMs: 20_000, maxOutputBytes: 1024 * 1024 });
|
|
474
|
+
if (!listed.spawned || listed.timedOut || listed.code !== 0 || Buffer.byteLength(listed.stdout) >= 1024 * 1024) throw new Error("could not list verification containers");
|
|
475
|
+
const ids = listed.stdout.trim().split(/\s+/).filter(Boolean);
|
|
476
|
+
if (ids.some(id => !/^[a-f0-9]+$/.test(id))) throw new Error("invalid verification container id");
|
|
477
|
+
if (ids.length) {
|
|
478
|
+
const removed = await collect(podman, ["rm", "--force", ...ids], { timeoutMs: 20_000, maxOutputBytes: 4096 });
|
|
479
|
+
if (!removed.spawned || removed.timedOut || removed.code !== 0) throw new Error("could not remove verification containers");
|
|
480
|
+
}
|
|
481
|
+
},
|
|
482
|
+
|
|
342
483
|
/** Is a usable sandbox present? Never throws. */
|
|
343
484
|
async probe({ image }) {
|
|
344
|
-
const version = await
|
|
485
|
+
const version = await collect(podman, ["version", "--format", "{{.Server.Version}}"], {
|
|
345
486
|
timeoutMs: 20_000,
|
|
346
487
|
maxOutputBytes: 4096,
|
|
347
488
|
});
|
|
@@ -358,7 +499,7 @@ export function createPodmanExecutor({ podman = "podman" } = {}) {
|
|
|
358
499
|
};
|
|
359
500
|
}
|
|
360
501
|
|
|
361
|
-
const inspect = await
|
|
502
|
+
const inspect = await collect(podman, ["image", "inspect", image], {
|
|
362
503
|
timeoutMs: 20_000,
|
|
363
504
|
maxOutputBytes: 4096,
|
|
364
505
|
});
|
|
@@ -372,10 +513,10 @@ export function createPodmanExecutor({ podman = "podman" } = {}) {
|
|
|
372
513
|
},
|
|
373
514
|
|
|
374
515
|
/** Run one command inside the sandbox. Never throws. */
|
|
375
|
-
async run({ command, cwd, image, jobId, network, user, workdir, timeoutMs, maxOutputBytes, nodeModulesSource, env }) {
|
|
376
|
-
const args = buildPodmanArgs({ image, cwd, command, jobId, network, user, workdir, nodeModulesSource, env });
|
|
516
|
+
async run({ command, cwd, image, jobId, network, user, workdir, timeoutMs, maxOutputBytes, nodeModulesSource, env, secretEnv, shmMb }) {
|
|
517
|
+
const args = buildPodmanArgs({ image, cwd, command, jobId, network, user, workdir, nodeModulesSource, env, secretEnv, shmMb });
|
|
377
518
|
const started = Date.now();
|
|
378
|
-
const outcome = await
|
|
519
|
+
const outcome = await collect(podman, args, { timeoutMs, maxOutputBytes, env: { ...process.env, ...Object.fromEntries((secretEnv ?? []).filter(name => Object.hasOwn(env ?? {}, name)).map(name => [name, env[name]])) } });
|
|
379
520
|
const durationMs = Date.now() - started;
|
|
380
521
|
|
|
381
522
|
if (!outcome.spawned) {
|
|
@@ -465,6 +606,11 @@ function pluralCommands(count, name) {
|
|
|
465
606
|
return `${count} command${count === 1 ? "" : "s"} in profile '${name}'`;
|
|
466
607
|
}
|
|
467
608
|
|
|
609
|
+
/** Largest matched shared-memory requirement in MiB; zero leaves Podman defaults. */
|
|
610
|
+
export function verificationShmMb(cwd, config, selection = sandboxHarnesses(cwd, config)) {
|
|
611
|
+
return Math.max(0, ...selection.matched.map((name) => selection.harnesses[name].requires.shmMb ?? 0));
|
|
612
|
+
}
|
|
613
|
+
|
|
468
614
|
/**
|
|
469
615
|
* Build the verification runner to hand to `registerVerificationRunner`.
|
|
470
616
|
*
|
|
@@ -506,8 +652,24 @@ export function createVerificationRunner(options = {}) {
|
|
|
506
652
|
sandboxImageRun = undefined,
|
|
507
653
|
} = options;
|
|
508
654
|
|
|
509
|
-
|
|
510
|
-
|
|
655
|
+
// Snapshot operator authority before any worker runs. Never load local policy
|
|
656
|
+
// from context.cwd, config objects, or harness definitions.
|
|
657
|
+
let networkPolicy = null, networkPolicyError = null;
|
|
658
|
+
try { networkPolicy = (options.loadVerificationNetwork ?? loadVerificationNetwork)(hostProjectDir); }
|
|
659
|
+
catch (error) { networkPolicyError = error.message; }
|
|
660
|
+
|
|
661
|
+
async function runVerification(context = {}) {
|
|
662
|
+
let { profile = null, cwd = null, jobId = randomUUID(), record = null } = context;
|
|
663
|
+
jobId ||= randomUUID();
|
|
664
|
+
let registryNote = null;
|
|
665
|
+
const notRun = (reason, basis, extra = {}) => ({ status: "not_run", basis, reason, detail: registryNote, ...extra });
|
|
666
|
+
const onRegistryNote = (note) => {
|
|
667
|
+
registryNote = note;
|
|
668
|
+
if (record) {
|
|
669
|
+
record.issues ??= [];
|
|
670
|
+
if (!record.issues.includes(note)) record.issues.push(note);
|
|
671
|
+
}
|
|
672
|
+
};
|
|
511
673
|
|
|
512
674
|
if (!cwd || typeof cwd !== "string") {
|
|
513
675
|
return notRun("no worktree path was supplied, so nothing could be verified", "no-worktree");
|
|
@@ -537,17 +699,46 @@ export function createVerificationRunner(options = {}) {
|
|
|
537
699
|
} catch (error) {
|
|
538
700
|
// A broken contract is not a failing test suite. Say so and stop.
|
|
539
701
|
return notRun(
|
|
540
|
-
`.nomarmy.yml could not be loaded: ${String(error?.message || error).split("\n")[0]}`,
|
|
702
|
+
`.nomarmy.yml could not be loaded: ${error.errors?.find(line => line.startsWith("verification_network ")) ?? String(error?.message || error).split("\n")[0]}`,
|
|
541
703
|
"config-error",
|
|
542
704
|
);
|
|
543
705
|
}
|
|
544
706
|
const config = loaded && loaded.found ? loaded.config : null;
|
|
707
|
+
if (config && Object.hasOwn(config, "verification_network")) return notRun("verification_network is operator-local only: use .nomarmy.local.yml", "config-error");
|
|
708
|
+
if (networkPolicyError) return notRun(networkPolicyError, "config-error");
|
|
709
|
+
const credentialEnv = {}, secrets = [];
|
|
710
|
+
const networkRecord = networkPolicy ? { allowlist: networkPolicy.allow, reached: [], credentials: Object.keys(networkPolicy.env) } : null;
|
|
711
|
+
for (const [name, source] of Object.entries(networkPolicy?.env ?? {})) {
|
|
712
|
+
const value = (options.hostEnv ?? process.env)[source];
|
|
713
|
+
if (typeof value !== "string" || !value.length) return notRun(`missing host environment variable ${source}`, "missing-credential", { network: networkRecord });
|
|
714
|
+
credentialEnv[name] = value;
|
|
715
|
+
secrets.push(value);
|
|
716
|
+
}
|
|
717
|
+
const credentialsInUse = secrets.length > 0;
|
|
718
|
+
const withheld = "output withheld: credentials were in use (set verification_network.keep_output: true in .nomarmy.local.yml to keep it, redacted)";
|
|
719
|
+
const hideOutput = credentialsInUse && networkPolicy.keep_output !== true;
|
|
720
|
+
const token = networkPolicy ? randomBytes(32).toString("hex") : null;
|
|
721
|
+
if (token) secrets.push(token);
|
|
722
|
+
const redact = value => redactCredentials(value, secrets);
|
|
723
|
+
const networkInfo = networkRecord ? { network: networkRecord, issues: [`verification had network access to ${networkRecord.allowlist.join(", ")}`] } : {};
|
|
724
|
+
|
|
725
|
+
let selection;
|
|
726
|
+
try { selection = sandboxHarnesses(cwd, config); }
|
|
727
|
+
catch (error) { return notRun(error.message, "config-error"); }
|
|
728
|
+
const specs = selection.matched.map((name) => selection.harnesses[name]);
|
|
729
|
+
if (!networkPolicy && specs.some((spec) => spec.network === "allowlist")) return notRun("allowlist harnesses require operator-local provisioning", "environment-not-provisioned");
|
|
730
|
+
const services = specs.flatMap((spec) => spec.services ?? []);
|
|
731
|
+
if (new Set(services.map((service) => service.name)).size !== services.length) return notRun("duplicate harness service names", "config-error");
|
|
732
|
+
if (networkPolicy && services.some(s => s.name === "egress")) return notRun("service alias egress is reserved for the verification proxy", "config-error");
|
|
733
|
+
const harnessEnv = Object.assign({}, ...specs.map((spec) => spec.env ?? {}));
|
|
734
|
+
const shmMb = verificationShmMb(cwd, config, selection);
|
|
545
735
|
|
|
546
736
|
const explicitImage = image || process.env.NOMARMY_AGENT_IMAGE || null;
|
|
547
737
|
let sandboxImage;
|
|
548
738
|
try {
|
|
549
739
|
sandboxImage = resolveSandboxImage({
|
|
550
740
|
cwd, explicitImage, defaultImage: DEFAULT_AGENT_IMAGE, config,
|
|
741
|
+
trustedDir: hostProjectDir, onNote: onRegistryNote,
|
|
551
742
|
...(sandboxImageRun ? { run: sandboxImageRun } : {}),
|
|
552
743
|
});
|
|
553
744
|
} catch (error) {
|
|
@@ -615,85 +806,147 @@ export function createVerificationRunner(options = {}) {
|
|
|
615
806
|
nodeModulesSource = mount.source;
|
|
616
807
|
}
|
|
617
808
|
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
timedOut: true,
|
|
629
|
-
reason: `overall verification budget of ${overallTimeoutMs}ms was exhausted before this command started`,
|
|
630
|
-
timeoutMs: overallTimeoutMs,
|
|
631
|
-
});
|
|
632
|
-
break;
|
|
809
|
+
let disposable;
|
|
810
|
+
try {
|
|
811
|
+
// Whenever a proxy token or credentials are issued, run on a throwaway
|
|
812
|
+
// copy: anything a command writes (a token in a new file) can't reach
|
|
813
|
+
// the worktree nomArmy commits from.
|
|
814
|
+
if (credentialsInUse || token) {
|
|
815
|
+
disposable = fs.mkdtempSync(path.join(os.tmpdir(), "nomarmy-verification-"));
|
|
816
|
+
const copy = path.join(disposable, "worktree");
|
|
817
|
+
fs.cpSync(cwd, copy, { recursive: true, dereference: false, verbatimSymlinks: true, filter: source => path.basename(source) !== ".git" });
|
|
818
|
+
cwd = copy;
|
|
633
819
|
}
|
|
820
|
+
// --- execute, sequentially, inside the sandbox ----------------------
|
|
821
|
+
const deadline = now() + overallTimeoutMs;
|
|
822
|
+
const results = [];
|
|
634
823
|
|
|
635
|
-
|
|
636
|
-
let
|
|
824
|
+
let serviceRun;
|
|
825
|
+
let proxyLog = "";
|
|
826
|
+
const renderOutput = () => (proxyLog ? `${proxyLog}\n\n` : "") + results.map((r) => `$ ${r.command} (exit ${r.exitCode ?? "none"}${r.timedOut ? ", timed out" : ""}${networkPolicy ? `, ${r.durationMs ?? 0}ms` : ""})\n${r.stdout ?? ""}${r.stderr ? `\n[stderr]\n${r.stderr}` : ""}`).join("\n\n");
|
|
637
827
|
try {
|
|
638
|
-
|
|
639
|
-
command,
|
|
640
|
-
cwd,
|
|
641
|
-
image: sandboxImage,
|
|
642
|
-
jobId,
|
|
643
|
-
network,
|
|
644
|
-
user,
|
|
645
|
-
workdir,
|
|
646
|
-
timeoutMs,
|
|
647
|
-
maxOutputBytes,
|
|
648
|
-
nodeModulesSource,
|
|
649
|
-
env: verificationEnv,
|
|
650
|
-
});
|
|
828
|
+
if (services.length || networkPolicy) serviceRun = await executor.startServices({ services, image: sandboxImage, jobId, ...(networkPolicy ? { allow: networkPolicy.allow, token } : {}) });
|
|
651
829
|
} catch (error) {
|
|
652
|
-
|
|
830
|
+
return notRun(redact(`service setup failed: ${error.message}`), "services-unavailable");
|
|
831
|
+
}
|
|
832
|
+
const proxyEnv = networkPolicy ? Object.fromEntries([
|
|
833
|
+
...["HTTP_PROXY", "HTTPS_PROXY", "http_proxy", "https_proxy"].map(name => [name, `http://nomarmy:${token}@egress:3128`]),
|
|
834
|
+
...["NO_PROXY", "no_proxy"].map(name => [name, services.map(s => s.name).join(",")]),
|
|
835
|
+
]) : {};
|
|
836
|
+
try {
|
|
837
|
+
for (const command of commands) {
|
|
838
|
+
const remaining = deadline - now();
|
|
839
|
+
if (remaining <= 0) {
|
|
840
|
+
results.push({
|
|
841
|
+
command,
|
|
842
|
+
started: false,
|
|
843
|
+
timedOut: true,
|
|
844
|
+
reason: `overall verification budget of ${overallTimeoutMs}ms was exhausted before this command started`,
|
|
845
|
+
timeoutMs: overallTimeoutMs,
|
|
846
|
+
});
|
|
847
|
+
break;
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
const timeoutMs = Math.max(1, Math.min(commandTimeoutMs, remaining));
|
|
851
|
+
let raw;
|
|
852
|
+
try {
|
|
853
|
+
raw = await executor.run({
|
|
854
|
+
command,
|
|
855
|
+
cwd,
|
|
856
|
+
image: sandboxImage,
|
|
857
|
+
jobId,
|
|
858
|
+
network: serviceRun?.network ?? network,
|
|
859
|
+
user,
|
|
860
|
+
workdir,
|
|
861
|
+
timeoutMs,
|
|
862
|
+
maxOutputBytes,
|
|
863
|
+
nodeModulesSource,
|
|
864
|
+
env: { ...harnessEnv, ...verificationEnv, ...credentialEnv, ...proxyEnv },
|
|
865
|
+
secretEnv: [...Object.keys(credentialEnv), ...["HTTP_PROXY", "HTTPS_PROXY", "http_proxy", "https_proxy"]],
|
|
866
|
+
shmMb,
|
|
867
|
+
});
|
|
868
|
+
} catch (error) {
|
|
869
|
+
raw = { started: false, reason: `sandbox execution threw: ${String(error?.message || error)}` };
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
const stdout = hideOutput ? { text: withheld, dropped: 0 } : capOutput(redact(raw?.stdout), maxOutputBytes);
|
|
873
|
+
const stderr = capOutput(hideOutput ? "" : redact(raw?.stderr), maxOutputBytes);
|
|
874
|
+
const entry = {
|
|
875
|
+
command: redact(command),
|
|
876
|
+
started: raw?.started !== false,
|
|
877
|
+
timedOut: Boolean(raw?.timedOut),
|
|
878
|
+
exitCode: Number.isInteger(raw?.exitCode) ? raw.exitCode : null,
|
|
879
|
+
reason: raw?.reason == null ? null : hideOutput ? withheld : redact(raw.reason),
|
|
880
|
+
durationMs: Number.isFinite(raw?.durationMs) ? raw.durationMs : null,
|
|
881
|
+
timeoutMs,
|
|
882
|
+
stdout: stdout.text,
|
|
883
|
+
stderr: stderr.text,
|
|
884
|
+
// Whether the command printed nothing at all, judged before any
|
|
885
|
+
// output is withheld, so a guarded no-op is still recognized.
|
|
886
|
+
silent: !String(raw?.stdout ?? "").trim() && !String(raw?.stderr ?? "").trim(),
|
|
887
|
+
truncated: { stdout: stdout.dropped, stderr: stderr.dropped },
|
|
888
|
+
};
|
|
889
|
+
results.push(entry);
|
|
890
|
+
|
|
891
|
+
// Stop at the first command that did not cleanly pass: later commands
|
|
892
|
+
// would run against an already-broken tree and add nothing.
|
|
893
|
+
if (!entry.started || entry.timedOut || entry.exitCode !== 0) break;
|
|
894
|
+
}
|
|
895
|
+
|
|
896
|
+
} finally {
|
|
897
|
+
if (serviceRun) {
|
|
898
|
+
let logError;
|
|
899
|
+
try { if (serviceRun.logs) proxyLog = redactCredentials(await serviceRun.logs(), secrets, { truncated: false }); }
|
|
900
|
+
catch { logError = true; }
|
|
901
|
+
if (networkRecord) networkRecord.reached = [...new Set(proxyLog.split("\n").filter(line => line.endsWith(" connected")).map(line => line.split(" ")[0]))];
|
|
902
|
+
try { await serviceRun.cleanup(); }
|
|
903
|
+
catch (error) { return notRun(redact(`service cleanup failed: ${error.message}`), "services-cleanup-failed", { ...networkInfo, output: renderOutput() }); }
|
|
904
|
+
if (logError) return notRun("could not capture egress verdict log", "egress-log-failed", { ...networkInfo, output: renderOutput() });
|
|
905
|
+
} else if (executor.cleanupVerification) {
|
|
906
|
+
await executor.cleanupVerification({ jobId });
|
|
907
|
+
}
|
|
653
908
|
}
|
|
654
909
|
|
|
655
|
-
const
|
|
656
|
-
const
|
|
657
|
-
const
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
exitCode: Number.isInteger(raw?.exitCode) ? raw.exitCode : null,
|
|
662
|
-
reason: raw?.reason ?? null,
|
|
663
|
-
durationMs: Number.isFinite(raw?.durationMs) ? raw.durationMs : null,
|
|
664
|
-
timeoutMs,
|
|
665
|
-
stdout: stdout.text,
|
|
666
|
-
stderr: stderr.text,
|
|
667
|
-
truncated: { stdout: stdout.dropped, stderr: stderr.dropped },
|
|
668
|
-
};
|
|
669
|
-
results.push(entry);
|
|
670
|
-
|
|
671
|
-
// Stop at the first command that did not cleanly pass: later commands
|
|
672
|
-
// would run against an already-broken tree and add nothing.
|
|
673
|
-
if (!entry.started || entry.timedOut || entry.exitCode !== 0) break;
|
|
674
|
-
}
|
|
910
|
+
const verdict = classifyResults(results, { env: verificationEnv });
|
|
911
|
+
const executed = results.filter((r) => r.started).length;
|
|
912
|
+
const dropped = results.reduce(
|
|
913
|
+
(total, r) => total + (r.truncated?.stdout ?? 0) + (r.truncated?.stderr ?? 0),
|
|
914
|
+
0,
|
|
915
|
+
);
|
|
675
916
|
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
(total, r) => total + (r.truncated?.stdout ?? 0) + (r.truncated?.stderr ?? 0),
|
|
680
|
-
0,
|
|
681
|
-
);
|
|
917
|
+
const detailSuffix = dropped > 0
|
|
918
|
+
? ` [${dropped} bytes of output dropped by the ${maxOutputBytes}-byte cap]`
|
|
919
|
+
: "";
|
|
682
920
|
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
921
|
+
if (verdict.status === "not_run") {
|
|
922
|
+
return notRun(verdict.detail, verdict.basis ?? "sandbox-unavailable", { ...networkInfo, output: renderOutput() });
|
|
923
|
+
}
|
|
686
924
|
|
|
687
|
-
|
|
688
|
-
|
|
925
|
+
return {
|
|
926
|
+
...networkInfo,
|
|
927
|
+
// No evidence files are kept while credentials or a proxy token are in
|
|
928
|
+
// use: a scan can't catch every encoding (UTF-16, wrapped, compressed).
|
|
929
|
+
...(credentialsInUse || token ? { artifacts: [], artifactsCapped: false, artifactsNote: "artifacts not kept: credentials or a proxy token were in use" } : {}),
|
|
930
|
+
status: verdict.status,
|
|
931
|
+
basis: `${basis}; ${executed} executed in ${sandboxImage}`,
|
|
932
|
+
reason: null,
|
|
933
|
+
detail: `${verdict.detail}${detailSuffix}${registryNote ? "; " + registryNote : ""}`,
|
|
934
|
+
// Each command's full (capped) output, for a caller that keeps a log.
|
|
935
|
+
output: renderOutput(),
|
|
936
|
+
};
|
|
937
|
+
} finally {
|
|
938
|
+
if (disposable) fs.rmSync(disposable, { recursive: true, force: true });
|
|
689
939
|
}
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
940
|
+
}
|
|
941
|
+
return async (context = {}) => {
|
|
942
|
+
let result;
|
|
943
|
+
try { result = await runVerification(context); }
|
|
944
|
+
catch (error) {
|
|
945
|
+
if (!networkPolicy) throw error;
|
|
946
|
+
result = notRun("verification could not complete safely", "runner-error");
|
|
947
|
+
}
|
|
948
|
+
if (networkPolicy && !result.network) result.network = { allowlist: networkPolicy.allow, reached: [], credentials: Object.keys(networkPolicy.env) };
|
|
949
|
+
return result;
|
|
697
950
|
};
|
|
698
951
|
}
|
|
699
952
|
|
package/lib/worker-prompt.mjs
CHANGED
|
@@ -69,10 +69,13 @@ export function describeRecoveryChanges(record) {
|
|
|
69
69
|
return `${record.repoStatusFiles.length} file(s) differ from a clean checkout: ${record.repoStatusFiles.join(", ")}`;
|
|
70
70
|
}
|
|
71
71
|
|
|
72
|
-
export function reportRecoveryPrompt({ report = { targetTokens: 256, hardCapTokens: 512 }, changes = null } = {}) {
|
|
72
|
+
export function reportRecoveryPrompt({ report = { targetTokens: 256, hardCapTokens: 512 }, changes = null, task = null } = {}) {
|
|
73
|
+
// The objective itself, so STATUS is judged against it even when the
|
|
74
|
+
// resumed session lost it with the cut-off reply.
|
|
75
|
+
const objective = task ? `\nThe objective you were working on:\n${String(task).trim()}\n` : "";
|
|
73
76
|
const changesLine = changes
|
|
74
77
|
? `\nThe repository (checked independently just now, not from your memory of this session) already shows: ${changes}. Trust this over any uncertainty about what you did or did not do.\n`
|
|
75
78
|
: `\nThe repository (checked independently just now, not from your memory of this session) shows no changes at all.\n`;
|
|
76
|
-
return `Your previous reply ended without the required final report, or was cut off before completing it.\n${changesLine}\nDo not repeat, redo, retry, or describe any action you already took. Do not call any tool. Reply with ONLY the four lines below, nothing before them, nothing after them:\n\nSTATUS: done | partial | blocked\nTESTS: pass | fail | not_run\nNOT_DONE: none | <brief>\nNOTE: <brief implementation or risk note>\n\nUse the exact field names above, including the underscore in NOT_DONE. Target ${report.targetTokens} tokens; ${report.hardCapTokens} is the hard cap. Base STATUS on the repository state above, not on what you recall attempting: if it shows the edit landed, you may report done; if it shows nothing relevant, report blocked or partial rather than guessing done.`;
|
|
79
|
+
return `Your previous reply ended without the required final report, or was cut off before completing it.\n${objective}${changesLine}\nDo not repeat, redo, retry, or describe any action you already took. Do not call any tool. Reply with ONLY the four lines below, nothing before them, nothing after them:\n\nSTATUS: done | partial | blocked\nTESTS: pass | fail | not_run\nNOT_DONE: none | <brief>\nNOTE: <brief implementation or risk note>\n\nUse the exact field names above, including the underscore in NOT_DONE. Target ${report.targetTokens} tokens; ${report.hardCapTokens} is the hard cap. Base STATUS on the repository state above, not on what you recall attempting: if it shows the edit landed, you may report done; if it shows nothing relevant, report blocked or partial rather than guessing done.`;
|
|
77
80
|
}
|
|
78
81
|
|