nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +86 -480
  2. package/bin/nomarmy.mjs +1081 -185
  3. package/docker/Dockerfile +2 -2
  4. package/docker/Dockerfile.go +6 -4
  5. package/docker/Dockerfile.rust +17 -2
  6. package/harnesses/_template/README.md +27 -0
  7. package/harnesses/_template/harness.yml +26 -0
  8. package/harnesses/browser-playwright/README.md +35 -0
  9. package/harnesses/browser-playwright/fixture/package.json +1 -0
  10. package/harnesses/browser-playwright/fixture/page.html +1 -0
  11. package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
  12. package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
  13. package/harnesses/browser-playwright/harness.yml +18 -0
  14. package/harnesses/go/README.md +45 -0
  15. package/harnesses/go/harness.yml +14 -0
  16. package/harnesses/mock-oidc/README.md +31 -0
  17. package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
  18. package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
  19. package/harnesses/mock-oidc/harness.yml +19 -0
  20. package/harnesses/node/README.md +53 -0
  21. package/harnesses/node/harness.yml +18 -0
  22. package/harnesses/python/README.md +46 -0
  23. package/harnesses/python/harness.yml +16 -0
  24. package/harnesses/rust/README.md +45 -0
  25. package/harnesses/rust/harness.yml +13 -0
  26. package/install.sh +29 -9
  27. package/lib/admission.mjs +178 -30
  28. package/lib/agents.mjs +8 -6
  29. package/lib/army.mjs +25 -10
  30. package/lib/codex-link.mjs +37 -0
  31. package/lib/config.mjs +15 -0
  32. package/lib/connect.mjs +232 -19
  33. package/lib/continue-from.mjs +103 -0
  34. package/lib/coordinator-instructions.mjs +5 -1
  35. package/lib/diff-checks.mjs +114 -0
  36. package/lib/dispatch-schema.mjs +14 -12
  37. package/lib/doctor.mjs +98 -9
  38. package/lib/egress-proxy.mjs +116 -0
  39. package/lib/execute.mjs +241 -33
  40. package/lib/git-record.mjs +27 -3
  41. package/lib/harness-schema.mjs +61 -0
  42. package/lib/harnesses.mjs +99 -0
  43. package/lib/health.mjs +162 -18
  44. package/lib/install-freshness.mjs +114 -0
  45. package/lib/jev-checks.mjs +110 -0
  46. package/lib/job-format.mjs +54 -0
  47. package/lib/judge.mjs +130 -0
  48. package/lib/limits.mjs +77 -0
  49. package/lib/model-probe.mjs +61 -0
  50. package/lib/mutation.mjs +159 -0
  51. package/lib/notify.mjs +30 -3
  52. package/lib/openclaw-install.mjs +122 -0
  53. package/lib/openclaw-path.mjs +28 -0
  54. package/lib/openclaw-run.mjs +74 -12
  55. package/lib/openclaw-runtime-health.mjs +56 -0
  56. package/lib/outcome.mjs +21 -2
  57. package/lib/outcomes.mjs +6 -0
  58. package/lib/path-utils.mjs +4 -0
  59. package/lib/podman-health.mjs +41 -0
  60. package/lib/process.mjs +4 -1
  61. package/lib/propose.mjs +10 -11
  62. package/lib/refusal-retry.mjs +16 -0
  63. package/lib/registry-python.mjs +98 -0
  64. package/lib/registry-secrets.mjs +140 -0
  65. package/lib/repo-query.mjs +13 -7
  66. package/lib/runs.mjs +7 -1
  67. package/lib/same-path.mjs +14 -0
  68. package/lib/sandbox-images.mjs +499 -83
  69. package/lib/sandbox-vm.mjs +32 -0
  70. package/lib/scan.mjs +5 -1
  71. package/lib/schema.mjs +20 -11
  72. package/lib/scout.mjs +21 -3
  73. package/lib/server-context.mjs +21 -1
  74. package/lib/setup-steps.mjs +55 -0
  75. package/lib/share.mjs +82 -0
  76. package/lib/stale-sessions.mjs +60 -0
  77. package/lib/stats.mjs +315 -0
  78. package/lib/statusline.mjs +32 -6
  79. package/lib/subscription-setup.mjs +13 -0
  80. package/lib/suggestions.mjs +153 -0
  81. package/lib/thinking.mjs +23 -0
  82. package/lib/transcript.mjs +30 -5
  83. package/lib/usage-limits.mjs +329 -0
  84. package/lib/user-config.mjs +106 -0
  85. package/lib/validators.mjs +220 -0
  86. package/lib/verification-artifacts.mjs +46 -0
  87. package/lib/verification-flow.mjs +52 -7
  88. package/lib/verification-network.mjs +66 -0
  89. package/lib/verify.mjs +338 -85
  90. package/lib/worker-prompt.mjs +5 -2
  91. package/lib/wsl-cli.mjs +152 -0
  92. package/lib/wsl.mjs +230 -0
  93. package/lib/zod-issues.mjs +15 -0
  94. package/mcp/server.mjs +165 -34
  95. package/package.json +7 -5
  96. package/playbooks/feature.md +8 -5
  97. package/scripts/configure-openclaw.sh +4 -2
  98. package/scripts/generate-harness-docs.mjs +42 -0
  99. package/scripts/install-openclaw.mjs +23 -0
  100. package/scripts/lib.sh +9 -2
  101. package/scripts/select-model.mjs +12 -5
  102. package/scripts/start-inference.sh +2 -2
package/lib/verify.mjs CHANGED
@@ -13,7 +13,7 @@
13
13
  // to repo-controlled code, the exact privilege the worker itself is denied.
14
14
  //
15
15
  // Therefore commands only ever execute inside the Podman sandbox image, as a
16
- // non-root user, with `--network none`. If Podman is missing, the image is
16
+ // non-root user, with `--network none` or a verification-only internal service network. If Podman is missing, the image is
17
17
  // absent, or the container fails to start, the verdict is `not_run`. There is
18
18
  // no host fallback path in this file, deliberately: a missing sandbox is
19
19
  // absence of evidence, not permission to take a shortcut.
@@ -22,12 +22,17 @@
22
22
  // is spawned with an argv array and `shell: false`. A command string is handed
23
23
  // to `/bin/sh -c` *inside* the container, which is inside the sandbox boundary.
24
24
 
25
+ import { randomUUID, randomBytes } from "node:crypto";
25
26
  import { spawn } from "node:child_process";
26
27
  import fs from "node:fs";
28
+ import os from "node:os";
29
+ import { collectVerificationArtifacts } from "./verification-artifacts.mjs";
27
30
  import path from "node:path";
31
+ import { fileURLToPath } from "node:url";
32
+ import { loadVerificationNetwork, redactCredentials } from "./verification-network.mjs";
28
33
 
29
34
  import { loadConfig as defaultLoadConfig } from "./config.mjs";
30
- import { resolveSandboxImage, nodeDependencyFiles, linkNodePackages, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
35
+ import { resolveSandboxImage, sandboxHarnesses, nodeDependencyFiles, linkNodePackages, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
31
36
 
32
37
  /** Sandbox image used when `NOMARMY_AGENT_IMAGE` is unset. */
33
38
  export const DEFAULT_AGENT_IMAGE = "openclaw-nomarmy-coder:bookworm";
@@ -130,10 +135,19 @@ function tailOf(result) {
130
135
  * command that never started is absence of evidence, so if nothing ever ran the
131
136
  * verdict is `not_run`, never `fail`.
132
137
  *
138
+ * A command guarded on nomArmy's changed-file variables (`if [ -n
139
+ * "$NOMARMY_CHANGED_TEST_FILES" ]; then ...; fi`) exits 0 without running
140
+ * anything when those are empty, as on an unchanged checkout in mode: verify.
141
+ * That's skipped, not passed: exit 0, no output at all, and every
142
+ * NOMARMY_CHANGED_* variable the command names empty. A command with a
143
+ * fallback (`${NOMARMY_CHANGED_TEST_FILES:-tests/}`) prints test output, so
144
+ * it counts as run. If every command skipped, nothing was verified: not_run.
145
+ *
133
146
  * @param {Array<object>} results
134
- * @returns {{ status: "pass"|"fail"|"not_run", detail: string }}
147
+ * @param {{ env?: Record<string, string> }} [context] the NOMARMY_CHANGED_* values the commands saw
148
+ * @returns {{ status: "pass"|"fail"|"not_run", detail: string, basis?: string }}
135
149
  */
136
- export function classifyResults(results) {
150
+ export function classifyResults(results, { env = {} } = {}) {
137
151
  const list = Array.isArray(results) ? results.filter(Boolean) : [];
138
152
  if (list.length === 0) {
139
153
  return { status: "not_run", detail: "no commands were executed" };
@@ -171,9 +185,28 @@ export function classifyResults(results) {
171
185
  }
172
186
  }
173
187
 
188
+ const skipped = list.map((result, index) => ({ index, why: guardedNoOp(result, env) })).filter((s) => s.why);
189
+ if (skipped.length === total) {
190
+ return { status: "not_run", basis: "nothing-changed", detail: `every command is scoped to changed files and nothing changed, so none ran (${[...new Set(skipped.map((s) => s.why))].join("; ")}); add an unscoped profile to run the full suite` };
191
+ }
192
+ if (skipped.length) {
193
+ const which = skipped.map((s) => `command ${s.index + 1} (\`${list[s.index].command ?? "?"}\`) did nothing: ${s.why}`).join("; ");
194
+ return { status: "pass", detail: `${total - skipped.length} of ${total} commands passed; ${skipped.length} skipped: ${which}` };
195
+ }
174
196
  return { status: "pass", detail: `${total} of ${total} commands passed` };
175
197
  }
176
198
 
199
+ const CHANGED_VAR_RE = /\$\{?(NOMARMY_CHANGED_[A-Z_]+)/g;
200
+
201
+ /** Why a zero-exit, silent command was a guarded no-op, or null if it ran. */
202
+ function guardedNoOp(result, env) {
203
+ if (result.exitCode !== 0 || result.timedOut || result.started === false) return null;
204
+ if (result.silent === false || (result.silent === undefined && `${result.stdout ?? ""}${result.stderr ?? ""}`.trim())) return null;
205
+ const names = [...new Set([...String(result.command ?? "").matchAll(CHANGED_VAR_RE)].map((m) => m[1]))];
206
+ if (!names.length || names.some((name) => String(env[name] ?? "").trim())) return null;
207
+ return `${names.join(" and ")} ${names.length > 1 ? "are" : "is"} empty`;
208
+ }
209
+
177
210
  // ---------------------------------------------------------------------------
178
211
  // pure: output capping
179
212
  // ---------------------------------------------------------------------------
@@ -229,6 +262,8 @@ export function buildPodmanArgs({
229
262
  workdir = DEFAULT_WORKDIR,
230
263
  nodeModulesSource = null,
231
264
  env = {},
265
+ secretEnv = [],
266
+ shmMb = null,
232
267
  } = {}) {
233
268
  const args = [
234
269
  "run",
@@ -256,7 +291,7 @@ export function buildPodmanArgs({
256
291
  // the var unquoted inside its OWN `/bin/sh -c` still word-splits on spaces
257
292
  // as usual -- that is the operator's shell, not this one, and is worth
258
293
  // knowing when a repo's paths might contain spaces.
259
- for (const [name, value] of Object.entries(env)) args.push("--env", `${name}=${value}`);
294
+ for (const [name, value] of Object.entries(env)) args.push("--env", secretEnv.includes(name) ? name : `${name}=${value}`);
260
295
  // `git worktree add` never copies node_modules, and this container has
261
296
  // --network none, so a Node repo's own verification commands can never
262
297
  // install what they need. The coordinator's own already-installed tree is
@@ -268,6 +303,7 @@ export function buildPodmanArgs({
268
303
  if (nodeModulesSource) {
269
304
  args.push("--mount", `type=bind,source=${nodeModulesSource},target=${workdir}/node_modules,readonly`);
270
305
  }
306
+ if (shmMb > 0) args.push(`--shm-size=${shmMb}m`);
271
307
  if (jobId) args.push("--label", `nomarmy.job=${jobId}`);
272
308
  args.push("--entrypoint", "/bin/sh", image, "-c", command);
273
309
  return args;
@@ -277,13 +313,13 @@ export function buildPodmanArgs({
277
313
  // the real Podman executor (the only place a host process is spawned)
278
314
  // ---------------------------------------------------------------------------
279
315
 
280
- function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd } = {}) {
316
+ function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd, env } = {}) {
281
317
  return new Promise((resolve) => {
282
318
  let child;
283
319
  try {
284
320
  // shell:false is the whole point: nothing here is ever parsed by a host
285
321
  // shell, so a config value cannot break out of its argv slot.
286
- child = spawn(file, args, { cwd, shell: false, windowsHide: true });
322
+ child = spawn(file, args, { cwd, env, shell: false, windowsHide: true });
287
323
  } catch (error) {
288
324
  resolve({ spawned: false, code: null, stdout: "", stderr: String(error?.message || error), timedOut: false });
289
325
  return;
@@ -337,11 +373,116 @@ function spawnCollect(file, args, { timeoutMs, maxOutputBytes, cwd } = {}) {
337
373
  * Executor backed by the real Podman CLI. Tests inject a fake in its place.
338
374
  * @param {{ podman?: string }} [options]
339
375
  */
340
- export function createPodmanExecutor({ podman = "podman" } = {}) {
376
+ export function createPodmanExecutor({ podman = "podman", collect = spawnCollect, healthTimeoutMs = 30_000, healthIntervalMs = 250 } = {}) {
341
377
  return {
378
+ /** Own only resources created by this run; cleanup is also used on setup failure. */
379
+ async startServices({ services, image, jobId, allow = [], token }) {
380
+ const network = `nomarmy-${String(jobId || randomUUID()).replace(/[^a-zA-Z0-9_-]/g, "-")}`;
381
+ const containers = [];
382
+ let created = false;
383
+ let externalCreated = false;
384
+ const external = `${network}-external`;
385
+ const proxyName = `${network}-egress`;
386
+ const call = (args, timeoutMs = 20_000) => collect(podman, args, { timeoutMs, maxOutputBytes: 4096, env: { ...process.env, NOMARMY_EGRESS_TOKEN: token } });
387
+ const ok = (result) => result.spawned && !result.timedOut && result.code === 0;
388
+ const requireSuccess = async (args, label, timeoutMs) => {
389
+ if (!ok(await call(args, timeoutMs))) throw new Error(label);
390
+ };
391
+ const cleanup = async () => {
392
+ const failures = [];
393
+ try { await this.cleanupVerification({ jobId }); }
394
+ catch { failures.push("could not remove verification containers"); }
395
+ for (const name of containers.reverse()) {
396
+ try { await requireSuccess(["rm", "--force", "--ignore", name], `could not remove service container ${name}`); }
397
+ catch (error) { failures.push(error.message); }
398
+ }
399
+ if (externalCreated) {
400
+ try { await requireSuccess(["network", "rm", external], "could not remove egress network"); }
401
+ catch (error) { failures.push(error.message); }
402
+ }
403
+ if (created) {
404
+ try { await requireSuccess(["network", "rm", network], `could not remove service network ${network}`); }
405
+ catch (error) { failures.push(error.message); }
406
+ }
407
+ if (failures.length) throw new Error(failures.join("; "));
408
+ };
409
+ try {
410
+ await requireSuccess(["network", "create", "--internal", network], `could not create internal service network ${network}`);
411
+ created = true;
412
+ if (allow.length) {
413
+ await requireSuccess(["network", "create", external], "could not create egress network");
414
+ externalCreated = true;
415
+ containers.push(proxyName);
416
+ await requireSuccess(["run", "--detach", "--name", proxyName,
417
+ "--network", `${network}:alias=egress`, "--network", external,
418
+ `--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
419
+ "--read-only", "--pids-limit=128", "--memory=128m", "--cpus=1",
420
+ "--mount", `type=bind,source=${fileURLToPath(new URL("./egress-proxy.mjs", import.meta.url))},target=/egress-proxy.mjs,readonly`,
421
+ "--env", "NOMARMY_EGRESS_TOKEN",
422
+ "--env", `NOMARMY_EGRESS_ALLOW=${JSON.stringify(allow)}`,
423
+ "--entrypoint", "node", DEFAULT_AGENT_IMAGE, "/egress-proxy.mjs"], "could not start egress proxy");
424
+ const probeName = `${network}-egress-health`;
425
+ containers.push(probeName);
426
+ const script = `const net=await import('node:net');const end=Date.now()+${healthTimeoutMs};while(Date.now()<end){const ok=await new Promise(r=>{const s=net.connect(3128,'egress');s.setTimeout(1000);s.on('connect',()=>{s.destroy();r(true)});s.on('error',()=>r(false));s.on('timeout',()=>{s.destroy();r(false)});});if(ok)process.exit(0);await new Promise(r=>setTimeout(r,${healthIntervalMs}));}process.exit(1);`;
427
+ await requireSuccess(["run", "--rm", "--name", probeName, "--network", network,
428
+ `--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
429
+ "--entrypoint", "node", DEFAULT_AGENT_IMAGE, "--input-type=module", "-e", script], "egress proxy did not become ready", healthTimeoutMs + 5000);
430
+ }
431
+ for (const service of services) {
432
+ if (!ok(await call(["image", "exists", service.image]))) {
433
+ await requireSuccess(["pull", service.image], `could not pull service image ${service.image}`, 120_000);
434
+ }
435
+ const name = `${network}-service-${service.name}`;
436
+ containers.push(name);
437
+ await requireSuccess(["run", "--detach", "--name", name, "--network", network,
438
+ "--network-alias", service.name, "--cap-drop=ALL", "--security-opt=no-new-privileges",
439
+ "--pids-limit=512", ...Object.entries(service.env ?? {}).flatMap(([k, v]) => ["--env", `${k}=${v}`]),
440
+ service.image], `could not start service ${service.name}`);
441
+ }
442
+ for (const service of services.filter((spec) => spec.health)) {
443
+ const probeName = `${network}-health-${service.name}`;
444
+ containers.push(probeName);
445
+ const url = `http://${service.name}:${service.port}${service.health}`;
446
+ // Poll inside the same internal network, never on the host. The base
447
+ // verification image supplies Node, so services need no curl binary.
448
+ const script = `const deadline=Date.now()+${healthTimeoutMs}; while(Date.now()<deadline){try{const r=await fetch(${JSON.stringify(url)},{redirect:'error',signal:AbortSignal.timeout(Math.min(2000,Math.max(1,deadline-Date.now())))});if(r.status>=200&&r.status<300)process.exit(0);}catch{}await new Promise(r=>setTimeout(r,${healthIntervalMs}));}process.exit(1);`;
449
+ await requireSuccess(["run", "--rm", "--name", probeName, "--network", network,
450
+ `--user=${DEFAULT_CONTAINER_USER}`, "--cap-drop=ALL", "--security-opt=no-new-privileges",
451
+ "--entrypoint", "node", image, "--input-type=module", "-e", script],
452
+ `service ${service.name} did not become healthy within ${healthTimeoutMs}ms (${url})`, healthTimeoutMs + 5000);
453
+ }
454
+ return { network, cleanup, ...(allow.length ? { logs: async () => {
455
+ // Freeze the audit trail before reading it; no late DNS completion or
456
+ // service connection may append a verdict after this snapshot.
457
+ await requireSuccess(["stop", "--time", "2", proxyName], "could not stop egress proxy");
458
+ const logLimit = 4 * 1024 * 1024;
459
+ const result = await collect(podman, ["logs", proxyName], { timeoutMs: 20_000, maxOutputBytes: logLimit });
460
+ // Never accept verification with a silently incomplete audit trail.
461
+ if (!ok(result) || Buffer.byteLength(result.stdout) >= logLimit) throw new Error("could not capture complete egress verdict log");
462
+ // Accept only the proxy's fixed verdict format; never relay diagnostics.
463
+ return result.stdout.split("\n").filter(line => /^[a-z0-9.-]+:[0-9]+ (connected|denied)$/.test(line)).join("\n");
464
+ } } : {}) };
465
+ } catch (error) {
466
+ try { await cleanup(); } catch (cleanupError) { throw new Error(`${error.message}; ${cleanupError.message}`); }
467
+ throw error;
468
+ }
469
+ },
470
+
471
+ async cleanupVerification({ jobId }) {
472
+ if (!jobId) return;
473
+ const listed = await collect(podman, ["ps", "-aq", "--filter", `label=nomarmy.job=${jobId}`], { timeoutMs: 20_000, maxOutputBytes: 1024 * 1024 });
474
+ if (!listed.spawned || listed.timedOut || listed.code !== 0 || Buffer.byteLength(listed.stdout) >= 1024 * 1024) throw new Error("could not list verification containers");
475
+ const ids = listed.stdout.trim().split(/\s+/).filter(Boolean);
476
+ if (ids.some(id => !/^[a-f0-9]+$/.test(id))) throw new Error("invalid verification container id");
477
+ if (ids.length) {
478
+ const removed = await collect(podman, ["rm", "--force", ...ids], { timeoutMs: 20_000, maxOutputBytes: 4096 });
479
+ if (!removed.spawned || removed.timedOut || removed.code !== 0) throw new Error("could not remove verification containers");
480
+ }
481
+ },
482
+
342
483
  /** Is a usable sandbox present? Never throws. */
343
484
  async probe({ image }) {
344
- const version = await spawnCollect(podman, ["version", "--format", "{{.Server.Version}}"], {
485
+ const version = await collect(podman, ["version", "--format", "{{.Server.Version}}"], {
345
486
  timeoutMs: 20_000,
346
487
  maxOutputBytes: 4096,
347
488
  });
@@ -358,7 +499,7 @@ export function createPodmanExecutor({ podman = "podman" } = {}) {
358
499
  };
359
500
  }
360
501
 
361
- const inspect = await spawnCollect(podman, ["image", "inspect", image], {
502
+ const inspect = await collect(podman, ["image", "inspect", image], {
362
503
  timeoutMs: 20_000,
363
504
  maxOutputBytes: 4096,
364
505
  });
@@ -372,10 +513,10 @@ export function createPodmanExecutor({ podman = "podman" } = {}) {
372
513
  },
373
514
 
374
515
  /** Run one command inside the sandbox. Never throws. */
375
- async run({ command, cwd, image, jobId, network, user, workdir, timeoutMs, maxOutputBytes, nodeModulesSource, env }) {
376
- const args = buildPodmanArgs({ image, cwd, command, jobId, network, user, workdir, nodeModulesSource, env });
516
+ async run({ command, cwd, image, jobId, network, user, workdir, timeoutMs, maxOutputBytes, nodeModulesSource, env, secretEnv, shmMb }) {
517
+ const args = buildPodmanArgs({ image, cwd, command, jobId, network, user, workdir, nodeModulesSource, env, secretEnv, shmMb });
377
518
  const started = Date.now();
378
- const outcome = await spawnCollect(podman, args, { timeoutMs, maxOutputBytes });
519
+ const outcome = await collect(podman, args, { timeoutMs, maxOutputBytes, env: { ...process.env, ...Object.fromEntries((secretEnv ?? []).filter(name => Object.hasOwn(env ?? {}, name)).map(name => [name, env[name]])) } });
379
520
  const durationMs = Date.now() - started;
380
521
 
381
522
  if (!outcome.spawned) {
@@ -465,6 +606,11 @@ function pluralCommands(count, name) {
465
606
  return `${count} command${count === 1 ? "" : "s"} in profile '${name}'`;
466
607
  }
467
608
 
609
+ /** Largest matched shared-memory requirement in MiB; zero leaves Podman defaults. */
610
+ export function verificationShmMb(cwd, config, selection = sandboxHarnesses(cwd, config)) {
611
+ return Math.max(0, ...selection.matched.map((name) => selection.harnesses[name].requires.shmMb ?? 0));
612
+ }
613
+
468
614
  /**
469
615
  * Build the verification runner to hand to `registerVerificationRunner`.
470
616
  *
@@ -506,8 +652,24 @@ export function createVerificationRunner(options = {}) {
506
652
  sandboxImageRun = undefined,
507
653
  } = options;
508
654
 
509
- return async function runVerification(context = {}) {
510
- const { profile = null, cwd = null, jobId = null, record = null } = context;
655
+ // Snapshot operator authority before any worker runs. Never load local policy
656
+ // from context.cwd, config objects, or harness definitions.
657
+ let networkPolicy = null, networkPolicyError = null;
658
+ try { networkPolicy = (options.loadVerificationNetwork ?? loadVerificationNetwork)(hostProjectDir); }
659
+ catch (error) { networkPolicyError = error.message; }
660
+
661
+ async function runVerification(context = {}) {
662
+ let { profile = null, cwd = null, jobId = randomUUID(), record = null } = context;
663
+ jobId ||= randomUUID();
664
+ let registryNote = null;
665
+ const notRun = (reason, basis, extra = {}) => ({ status: "not_run", basis, reason, detail: registryNote, ...extra });
666
+ const onRegistryNote = (note) => {
667
+ registryNote = note;
668
+ if (record) {
669
+ record.issues ??= [];
670
+ if (!record.issues.includes(note)) record.issues.push(note);
671
+ }
672
+ };
511
673
 
512
674
  if (!cwd || typeof cwd !== "string") {
513
675
  return notRun("no worktree path was supplied, so nothing could be verified", "no-worktree");
@@ -537,17 +699,46 @@ export function createVerificationRunner(options = {}) {
537
699
  } catch (error) {
538
700
  // A broken contract is not a failing test suite. Say so and stop.
539
701
  return notRun(
540
- `.nomarmy.yml could not be loaded: ${String(error?.message || error).split("\n")[0]}`,
702
+ `.nomarmy.yml could not be loaded: ${error.errors?.find(line => line.startsWith("verification_network ")) ?? String(error?.message || error).split("\n")[0]}`,
541
703
  "config-error",
542
704
  );
543
705
  }
544
706
  const config = loaded && loaded.found ? loaded.config : null;
707
+ if (config && Object.hasOwn(config, "verification_network")) return notRun("verification_network is operator-local only: use .nomarmy.local.yml", "config-error");
708
+ if (networkPolicyError) return notRun(networkPolicyError, "config-error");
709
+ const credentialEnv = {}, secrets = [];
710
+ const networkRecord = networkPolicy ? { allowlist: networkPolicy.allow, reached: [], credentials: Object.keys(networkPolicy.env) } : null;
711
+ for (const [name, source] of Object.entries(networkPolicy?.env ?? {})) {
712
+ const value = (options.hostEnv ?? process.env)[source];
713
+ if (typeof value !== "string" || !value.length) return notRun(`missing host environment variable ${source}`, "missing-credential", { network: networkRecord });
714
+ credentialEnv[name] = value;
715
+ secrets.push(value);
716
+ }
717
+ const credentialsInUse = secrets.length > 0;
718
+ const withheld = "output withheld: credentials were in use (set verification_network.keep_output: true in .nomarmy.local.yml to keep it, redacted)";
719
+ const hideOutput = credentialsInUse && networkPolicy.keep_output !== true;
720
+ const token = networkPolicy ? randomBytes(32).toString("hex") : null;
721
+ if (token) secrets.push(token);
722
+ const redact = value => redactCredentials(value, secrets);
723
+ const networkInfo = networkRecord ? { network: networkRecord, issues: [`verification had network access to ${networkRecord.allowlist.join(", ")}`] } : {};
724
+
725
+ let selection;
726
+ try { selection = sandboxHarnesses(cwd, config); }
727
+ catch (error) { return notRun(error.message, "config-error"); }
728
+ const specs = selection.matched.map((name) => selection.harnesses[name]);
729
+ if (!networkPolicy && specs.some((spec) => spec.network === "allowlist")) return notRun("allowlist harnesses require operator-local provisioning", "environment-not-provisioned");
730
+ const services = specs.flatMap((spec) => spec.services ?? []);
731
+ if (new Set(services.map((service) => service.name)).size !== services.length) return notRun("duplicate harness service names", "config-error");
732
+ if (networkPolicy && services.some(s => s.name === "egress")) return notRun("service alias egress is reserved for the verification proxy", "config-error");
733
+ const harnessEnv = Object.assign({}, ...specs.map((spec) => spec.env ?? {}));
734
+ const shmMb = verificationShmMb(cwd, config, selection);
545
735
 
546
736
  const explicitImage = image || process.env.NOMARMY_AGENT_IMAGE || null;
547
737
  let sandboxImage;
548
738
  try {
549
739
  sandboxImage = resolveSandboxImage({
550
740
  cwd, explicitImage, defaultImage: DEFAULT_AGENT_IMAGE, config,
741
+ trustedDir: hostProjectDir, onNote: onRegistryNote,
551
742
  ...(sandboxImageRun ? { run: sandboxImageRun } : {}),
552
743
  });
553
744
  } catch (error) {
@@ -615,85 +806,147 @@ export function createVerificationRunner(options = {}) {
615
806
  nodeModulesSource = mount.source;
616
807
  }
617
808
 
618
- // --- execute, sequentially, inside the sandbox ----------------------
619
- const deadline = now() + overallTimeoutMs;
620
- const results = [];
621
-
622
- for (const command of commands) {
623
- const remaining = deadline - now();
624
- if (remaining <= 0) {
625
- results.push({
626
- command,
627
- started: false,
628
- timedOut: true,
629
- reason: `overall verification budget of ${overallTimeoutMs}ms was exhausted before this command started`,
630
- timeoutMs: overallTimeoutMs,
631
- });
632
- break;
809
+ let disposable;
810
+ try {
811
+ // Whenever a proxy token or credentials are issued, run on a throwaway
812
+ // copy: anything a command writes (a token in a new file) can't reach
813
+ // the worktree nomArmy commits from.
814
+ if (credentialsInUse || token) {
815
+ disposable = fs.mkdtempSync(path.join(os.tmpdir(), "nomarmy-verification-"));
816
+ const copy = path.join(disposable, "worktree");
817
+ fs.cpSync(cwd, copy, { recursive: true, dereference: false, verbatimSymlinks: true, filter: source => path.basename(source) !== ".git" });
818
+ cwd = copy;
633
819
  }
820
+ // --- execute, sequentially, inside the sandbox ----------------------
821
+ const deadline = now() + overallTimeoutMs;
822
+ const results = [];
634
823
 
635
- const timeoutMs = Math.max(1, Math.min(commandTimeoutMs, remaining));
636
- let raw;
824
+ let serviceRun;
825
+ let proxyLog = "";
826
+ const renderOutput = () => (proxyLog ? `${proxyLog}\n\n` : "") + results.map((r) => `$ ${r.command} (exit ${r.exitCode ?? "none"}${r.timedOut ? ", timed out" : ""}${networkPolicy ? `, ${r.durationMs ?? 0}ms` : ""})\n${r.stdout ?? ""}${r.stderr ? `\n[stderr]\n${r.stderr}` : ""}`).join("\n\n");
637
827
  try {
638
- raw = await executor.run({
639
- command,
640
- cwd,
641
- image: sandboxImage,
642
- jobId,
643
- network,
644
- user,
645
- workdir,
646
- timeoutMs,
647
- maxOutputBytes,
648
- nodeModulesSource,
649
- env: verificationEnv,
650
- });
828
+ if (services.length || networkPolicy) serviceRun = await executor.startServices({ services, image: sandboxImage, jobId, ...(networkPolicy ? { allow: networkPolicy.allow, token } : {}) });
651
829
  } catch (error) {
652
- raw = { started: false, reason: `sandbox execution threw: ${String(error?.message || error)}` };
830
+ return notRun(redact(`service setup failed: ${error.message}`), "services-unavailable");
831
+ }
832
+ const proxyEnv = networkPolicy ? Object.fromEntries([
833
+ ...["HTTP_PROXY", "HTTPS_PROXY", "http_proxy", "https_proxy"].map(name => [name, `http://nomarmy:${token}@egress:3128`]),
834
+ ...["NO_PROXY", "no_proxy"].map(name => [name, services.map(s => s.name).join(",")]),
835
+ ]) : {};
836
+ try {
837
+ for (const command of commands) {
838
+ const remaining = deadline - now();
839
+ if (remaining <= 0) {
840
+ results.push({
841
+ command,
842
+ started: false,
843
+ timedOut: true,
844
+ reason: `overall verification budget of ${overallTimeoutMs}ms was exhausted before this command started`,
845
+ timeoutMs: overallTimeoutMs,
846
+ });
847
+ break;
848
+ }
849
+
850
+ const timeoutMs = Math.max(1, Math.min(commandTimeoutMs, remaining));
851
+ let raw;
852
+ try {
853
+ raw = await executor.run({
854
+ command,
855
+ cwd,
856
+ image: sandboxImage,
857
+ jobId,
858
+ network: serviceRun?.network ?? network,
859
+ user,
860
+ workdir,
861
+ timeoutMs,
862
+ maxOutputBytes,
863
+ nodeModulesSource,
864
+ env: { ...harnessEnv, ...verificationEnv, ...credentialEnv, ...proxyEnv },
865
+ secretEnv: [...Object.keys(credentialEnv), ...["HTTP_PROXY", "HTTPS_PROXY", "http_proxy", "https_proxy"]],
866
+ shmMb,
867
+ });
868
+ } catch (error) {
869
+ raw = { started: false, reason: `sandbox execution threw: ${String(error?.message || error)}` };
870
+ }
871
+
872
+ const stdout = hideOutput ? { text: withheld, dropped: 0 } : capOutput(redact(raw?.stdout), maxOutputBytes);
873
+ const stderr = capOutput(hideOutput ? "" : redact(raw?.stderr), maxOutputBytes);
874
+ const entry = {
875
+ command: redact(command),
876
+ started: raw?.started !== false,
877
+ timedOut: Boolean(raw?.timedOut),
878
+ exitCode: Number.isInteger(raw?.exitCode) ? raw.exitCode : null,
879
+ reason: raw?.reason == null ? null : hideOutput ? withheld : redact(raw.reason),
880
+ durationMs: Number.isFinite(raw?.durationMs) ? raw.durationMs : null,
881
+ timeoutMs,
882
+ stdout: stdout.text,
883
+ stderr: stderr.text,
884
+ // Whether the command printed nothing at all, judged before any
885
+ // output is withheld, so a guarded no-op is still recognized.
886
+ silent: !String(raw?.stdout ?? "").trim() && !String(raw?.stderr ?? "").trim(),
887
+ truncated: { stdout: stdout.dropped, stderr: stderr.dropped },
888
+ };
889
+ results.push(entry);
890
+
891
+ // Stop at the first command that did not cleanly pass: later commands
892
+ // would run against an already-broken tree and add nothing.
893
+ if (!entry.started || entry.timedOut || entry.exitCode !== 0) break;
894
+ }
895
+
896
+ } finally {
897
+ if (serviceRun) {
898
+ let logError;
899
+ try { if (serviceRun.logs) proxyLog = redactCredentials(await serviceRun.logs(), secrets, { truncated: false }); }
900
+ catch { logError = true; }
901
+ if (networkRecord) networkRecord.reached = [...new Set(proxyLog.split("\n").filter(line => line.endsWith(" connected")).map(line => line.split(" ")[0]))];
902
+ try { await serviceRun.cleanup(); }
903
+ catch (error) { return notRun(redact(`service cleanup failed: ${error.message}`), "services-cleanup-failed", { ...networkInfo, output: renderOutput() }); }
904
+ if (logError) return notRun("could not capture egress verdict log", "egress-log-failed", { ...networkInfo, output: renderOutput() });
905
+ } else if (executor.cleanupVerification) {
906
+ await executor.cleanupVerification({ jobId });
907
+ }
653
908
  }
654
909
 
655
- const stdout = capOutput(raw?.stdout, maxOutputBytes);
656
- const stderr = capOutput(raw?.stderr, maxOutputBytes);
657
- const entry = {
658
- command,
659
- started: raw?.started !== false,
660
- timedOut: Boolean(raw?.timedOut),
661
- exitCode: Number.isInteger(raw?.exitCode) ? raw.exitCode : null,
662
- reason: raw?.reason ?? null,
663
- durationMs: Number.isFinite(raw?.durationMs) ? raw.durationMs : null,
664
- timeoutMs,
665
- stdout: stdout.text,
666
- stderr: stderr.text,
667
- truncated: { stdout: stdout.dropped, stderr: stderr.dropped },
668
- };
669
- results.push(entry);
670
-
671
- // Stop at the first command that did not cleanly pass: later commands
672
- // would run against an already-broken tree and add nothing.
673
- if (!entry.started || entry.timedOut || entry.exitCode !== 0) break;
674
- }
910
+ const verdict = classifyResults(results, { env: verificationEnv });
911
+ const executed = results.filter((r) => r.started).length;
912
+ const dropped = results.reduce(
913
+ (total, r) => total + (r.truncated?.stdout ?? 0) + (r.truncated?.stderr ?? 0),
914
+ 0,
915
+ );
675
916
 
676
- const verdict = classifyResults(results);
677
- const executed = results.filter((r) => r.started).length;
678
- const dropped = results.reduce(
679
- (total, r) => total + (r.truncated?.stdout ?? 0) + (r.truncated?.stderr ?? 0),
680
- 0,
681
- );
917
+ const detailSuffix = dropped > 0
918
+ ? ` [${dropped} bytes of output dropped by the ${maxOutputBytes}-byte cap]`
919
+ : "";
682
920
 
683
- const detailSuffix = dropped > 0
684
- ? ` [${dropped} bytes of output dropped by the ${maxOutputBytes}-byte cap]`
685
- : "";
921
+ if (verdict.status === "not_run") {
922
+ return notRun(verdict.detail, verdict.basis ?? "sandbox-unavailable", { ...networkInfo, output: renderOutput() });
923
+ }
686
924
 
687
- if (verdict.status === "not_run") {
688
- return notRun(verdict.detail, "sandbox-unavailable");
925
+ return {
926
+ ...networkInfo,
927
+ // No evidence files are kept while credentials or a proxy token are in
928
+ // use: a scan can't catch every encoding (UTF-16, wrapped, compressed).
929
+ ...(credentialsInUse || token ? { artifacts: [], artifactsCapped: false, artifactsNote: "artifacts not kept: credentials or a proxy token were in use" } : {}),
930
+ status: verdict.status,
931
+ basis: `${basis}; ${executed} executed in ${sandboxImage}`,
932
+ reason: null,
933
+ detail: `${verdict.detail}${detailSuffix}${registryNote ? "; " + registryNote : ""}`,
934
+ // Each command's full (capped) output, for a caller that keeps a log.
935
+ output: renderOutput(),
936
+ };
937
+ } finally {
938
+ if (disposable) fs.rmSync(disposable, { recursive: true, force: true });
689
939
  }
690
-
691
- return {
692
- status: verdict.status,
693
- basis: `${basis}; ${executed} executed in ${sandboxImage}`,
694
- reason: null,
695
- detail: `${verdict.detail}${detailSuffix}`,
696
- };
940
+ }
941
+ return async (context = {}) => {
942
+ let result;
943
+ try { result = await runVerification(context); }
944
+ catch (error) {
945
+ if (!networkPolicy) throw error;
946
+ result = notRun("verification could not complete safely", "runner-error");
947
+ }
948
+ if (networkPolicy && !result.network) result.network = { allowlist: networkPolicy.allow, reached: [], credentials: Object.keys(networkPolicy.env) };
949
+ return result;
697
950
  };
698
951
  }
699
952
 
@@ -69,10 +69,13 @@ export function describeRecoveryChanges(record) {
69
69
  return `${record.repoStatusFiles.length} file(s) differ from a clean checkout: ${record.repoStatusFiles.join(", ")}`;
70
70
  }
71
71
 
72
- export function reportRecoveryPrompt({ report = { targetTokens: 256, hardCapTokens: 512 }, changes = null } = {}) {
72
+ export function reportRecoveryPrompt({ report = { targetTokens: 256, hardCapTokens: 512 }, changes = null, task = null } = {}) {
73
+ // The objective itself, so STATUS is judged against it even when the
74
+ // resumed session lost it with the cut-off reply.
75
+ const objective = task ? `\nThe objective you were working on:\n${String(task).trim()}\n` : "";
73
76
  const changesLine = changes
74
77
  ? `\nThe repository (checked independently just now, not from your memory of this session) already shows: ${changes}. Trust this over any uncertainty about what you did or did not do.\n`
75
78
  : `\nThe repository (checked independently just now, not from your memory of this session) shows no changes at all.\n`;
76
- return `Your previous reply ended without the required final report, or was cut off before completing it.\n${changesLine}\nDo not repeat, redo, retry, or describe any action you already took. Do not call any tool. Reply with ONLY the four lines below, nothing before them, nothing after them:\n\nSTATUS: done | partial | blocked\nTESTS: pass | fail | not_run\nNOT_DONE: none | <brief>\nNOTE: <brief implementation or risk note>\n\nUse the exact field names above, including the underscore in NOT_DONE. Target ${report.targetTokens} tokens; ${report.hardCapTokens} is the hard cap. Base STATUS on the repository state above, not on what you recall attempting: if it shows the edit landed, you may report done; if it shows nothing relevant, report blocked or partial rather than guessing done.`;
79
+ return `Your previous reply ended without the required final report, or was cut off before completing it.\n${objective}${changesLine}\nDo not repeat, redo, retry, or describe any action you already took. Do not call any tool. Reply with ONLY the four lines below, nothing before them, nothing after them:\n\nSTATUS: done | partial | blocked\nTESTS: pass | fail | not_run\nNOT_DONE: none | <brief>\nNOTE: <brief implementation or risk note>\n\nUse the exact field names above, including the underscore in NOT_DONE. Target ${report.targetTokens} tokens; ${report.hardCapTokens} is the hard cap. Base STATUS on the repository state above, not on what you recall attempting: if it shows the edit landed, you may report done; if it shows nothing relevant, report blocked or partial rather than guessing done.`;
77
80
  }
78
81