nomarmy 0.1.0-alpha.2 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -480
- package/bin/nomarmy.mjs +1081 -185
- package/docker/Dockerfile +2 -2
- package/docker/Dockerfile.go +6 -4
- package/docker/Dockerfile.rust +17 -2
- package/harnesses/_template/README.md +27 -0
- package/harnesses/_template/harness.yml +26 -0
- package/harnesses/browser-playwright/README.md +35 -0
- package/harnesses/browser-playwright/fixture/package.json +1 -0
- package/harnesses/browser-playwright/fixture/page.html +1 -0
- package/harnesses/browser-playwright/fixture/page.spec.js +5 -0
- package/harnesses/browser-playwright/fixture/playwright.config.js +8 -0
- package/harnesses/browser-playwright/harness.yml +18 -0
- package/harnesses/go/README.md +45 -0
- package/harnesses/go/harness.yml +14 -0
- package/harnesses/mock-oidc/README.md +31 -0
- package/harnesses/mock-oidc/fixture/.nomarmy.yml +4 -0
- package/harnesses/mock-oidc/fixture/discovery.test.mjs +16 -0
- package/harnesses/mock-oidc/harness.yml +19 -0
- package/harnesses/node/README.md +53 -0
- package/harnesses/node/harness.yml +18 -0
- package/harnesses/python/README.md +46 -0
- package/harnesses/python/harness.yml +16 -0
- package/harnesses/rust/README.md +45 -0
- package/harnesses/rust/harness.yml +13 -0
- package/install.sh +29 -9
- package/lib/admission.mjs +178 -30
- package/lib/agents.mjs +8 -6
- package/lib/army.mjs +25 -10
- package/lib/codex-link.mjs +37 -0
- package/lib/config.mjs +15 -0
- package/lib/connect.mjs +232 -19
- package/lib/continue-from.mjs +103 -0
- package/lib/coordinator-instructions.mjs +5 -1
- package/lib/diff-checks.mjs +114 -0
- package/lib/dispatch-schema.mjs +14 -12
- package/lib/doctor.mjs +98 -9
- package/lib/egress-proxy.mjs +116 -0
- package/lib/execute.mjs +241 -33
- package/lib/git-record.mjs +27 -3
- package/lib/harness-schema.mjs +61 -0
- package/lib/harnesses.mjs +99 -0
- package/lib/health.mjs +162 -18
- package/lib/install-freshness.mjs +114 -0
- package/lib/jev-checks.mjs +110 -0
- package/lib/job-format.mjs +54 -0
- package/lib/judge.mjs +130 -0
- package/lib/limits.mjs +77 -0
- package/lib/model-probe.mjs +61 -0
- package/lib/mutation.mjs +159 -0
- package/lib/notify.mjs +30 -3
- package/lib/openclaw-install.mjs +122 -0
- package/lib/openclaw-path.mjs +28 -0
- package/lib/openclaw-run.mjs +74 -12
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/outcome.mjs +21 -2
- package/lib/outcomes.mjs +6 -0
- package/lib/path-utils.mjs +4 -0
- package/lib/podman-health.mjs +41 -0
- package/lib/process.mjs +4 -1
- package/lib/propose.mjs +10 -11
- package/lib/refusal-retry.mjs +16 -0
- package/lib/registry-python.mjs +98 -0
- package/lib/registry-secrets.mjs +140 -0
- package/lib/repo-query.mjs +13 -7
- package/lib/runs.mjs +7 -1
- package/lib/same-path.mjs +14 -0
- package/lib/sandbox-images.mjs +499 -83
- package/lib/sandbox-vm.mjs +32 -0
- package/lib/scan.mjs +5 -1
- package/lib/schema.mjs +20 -11
- package/lib/scout.mjs +21 -3
- package/lib/server-context.mjs +21 -1
- package/lib/setup-steps.mjs +55 -0
- package/lib/share.mjs +82 -0
- package/lib/stale-sessions.mjs +60 -0
- package/lib/stats.mjs +315 -0
- package/lib/statusline.mjs +32 -6
- package/lib/subscription-setup.mjs +13 -0
- package/lib/suggestions.mjs +153 -0
- package/lib/thinking.mjs +23 -0
- package/lib/transcript.mjs +30 -5
- package/lib/usage-limits.mjs +329 -0
- package/lib/user-config.mjs +106 -0
- package/lib/validators.mjs +220 -0
- package/lib/verification-artifacts.mjs +46 -0
- package/lib/verification-flow.mjs +52 -7
- package/lib/verification-network.mjs +66 -0
- package/lib/verify.mjs +338 -85
- package/lib/worker-prompt.mjs +5 -2
- package/lib/wsl-cli.mjs +152 -0
- package/lib/wsl.mjs +230 -0
- package/lib/zod-issues.mjs +15 -0
- package/mcp/server.mjs +165 -34
- package/package.json +7 -5
- package/playbooks/feature.md +8 -5
- package/scripts/configure-openclaw.sh +4 -2
- package/scripts/generate-harness-docs.mjs +42 -0
- package/scripts/install-openclaw.mjs +23 -0
- package/scripts/lib.sh +9 -2
- package/scripts/select-model.mjs +12 -5
- package/scripts/start-inference.sh +2 -2
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
// One tested release for every nomArmy-managed OpenClaw installation.
|
|
2
|
+
// Never follow npm's dist-tag or downgrade a newer installation.
|
|
3
|
+
import { spawnSync } from "node:child_process";
|
|
4
|
+
import { SUBSCRIPTION_VENDORS, parseOpenclawVersion, versionAtLeast } from "./subscription-setup.mjs";
|
|
5
|
+
|
|
6
|
+
export const PINNED_OPENCLAW_VERSION = "2026.9.6";
|
|
7
|
+
export const OPENCLAW_INSTALL_ARGS = Object.freeze(["install", "-g", `openclaw@${PINNED_OPENCLAW_VERSION}`]);
|
|
8
|
+
|
|
9
|
+
export function runOpenclawCommand(command, args, { timeoutMs = command === "npm" ? 600000 : 60000 } = {}) {
|
|
10
|
+
const r = spawnSync(command, args, { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], timeout: timeoutMs });
|
|
11
|
+
const timedOut = r.error?.code === "ETIMEDOUT";
|
|
12
|
+
return { ok: !timedOut && r.status === 0, stdout: r.stdout ?? "",
|
|
13
|
+
stderr: timedOut ? `Command timed out after ${timeoutMs} ms.` : r.stderr ?? "", timedOut };
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export function openclawInstallPlan(installed, { prefix = null } = {}) {
|
|
17
|
+
const have = parseOpenclawVersion(installed);
|
|
18
|
+
if (versionAtLeast(have, PINNED_OPENCLAW_VERSION)) return [];
|
|
19
|
+
return [{
|
|
20
|
+
description: `${have ? "Upgrade" : "Install"} OpenClaw to ${PINNED_OPENCLAW_VERSION}`,
|
|
21
|
+
command: "npm",
|
|
22
|
+
args: [...OPENCLAW_INSTALL_ARGS, ...(prefix ? ["--prefix", prefix] : [])],
|
|
23
|
+
}];
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function configuredSubscriptionVendors(agents = {}, extra = []) {
|
|
27
|
+
return [...new Set([...extra, ...Object.values(agents)
|
|
28
|
+
.filter((a) => a?.kind === "subscription")
|
|
29
|
+
.map((a) => Object.keys(SUBSCRIPTION_VENDORS).find((key) => SUBSCRIPTION_VENDORS[key].provider === a.provider))])].filter(Boolean);
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** Read-only postflight using structured status and plugin compatibility data. */
|
|
33
|
+
function parseCommandJson(stdout) {
|
|
34
|
+
const start = (stdout ?? "").indexOf("{");
|
|
35
|
+
if (start < 0) throw new Error("No JSON object in command output");
|
|
36
|
+
return JSON.parse(stdout.slice(start));
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function verifyOpenclaw({ run = runOpenclawCommand, command = "openclaw", vendors = [] } = {}) {
|
|
40
|
+
const checks = [];
|
|
41
|
+
const version = run(command, ["--version"]);
|
|
42
|
+
const have = version.ok ? parseOpenclawVersion(version.stdout) : null;
|
|
43
|
+
checks.push({
|
|
44
|
+
id: "openclaw", ok: versionAtLeast(have, PINNED_OPENCLAW_VERSION),
|
|
45
|
+
message: have ? `OpenClaw ${have.join(".")}${have.join(".") !== PINNED_OPENCLAW_VERSION && versionAtLeast(have, PINNED_OPENCLAW_VERSION) ? " is newer than the tested release; left unchanged" : ""}.` : version.timedOut ? "OpenClaw version check timed out." : "OpenClaw version could not be verified.",
|
|
46
|
+
fix: `npm ${OPENCLAW_INSTALL_ARGS.join(" ")}`,
|
|
47
|
+
});
|
|
48
|
+
if (!have) return checks;
|
|
49
|
+
for (const key of [...new Set(vendors)]) {
|
|
50
|
+
const vendor = SUBSCRIPTION_VENDORS[key];
|
|
51
|
+
if (!vendor) throw new Error(`Unknown subscription vendor: ${key}`);
|
|
52
|
+
if (!vendor.plugin) continue;
|
|
53
|
+
const { id, spec } = vendor.plugin;
|
|
54
|
+
const inspected = run(command, ["plugins", "inspect", id, "--json"]);
|
|
55
|
+
let plugin;
|
|
56
|
+
try { plugin = parseCommandJson(inspected.stdout).plugin; } catch { /* Unverified output fails below. */ }
|
|
57
|
+
const pluginVersion = plugin?.builtWithOpenClawVersion === undefined ? plugin?.version : plugin.builtWithOpenClawVersion;
|
|
58
|
+
const installedVersion = have.join(".");
|
|
59
|
+
const ok = inspected.ok && plugin?.enabled === true && pluginVersion !== undefined && versionAtLeast(have, pluginVersion);
|
|
60
|
+
checks.push({
|
|
61
|
+
id: `openclaw-plugin:${id}`, ok,
|
|
62
|
+
message: ok ? `OpenClaw plugin ${id} ${pluginVersion} is ready (built for ${pluginVersion}; OpenClaw is ${installedVersion}).` : `OpenClaw plugin ${id} ${inspected.timedOut ? "check timed out." : plugin?.enabled === true && pluginVersion === undefined ? "version could not be verified." : `is missing, disabled, unreadable, or built for a newer OpenClaw than ${installedVersion}.`}`,
|
|
63
|
+
fix: `openclaw plugins install ${spec}`,
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
const status = run(command, ["update", "status", "--json"]);
|
|
67
|
+
let warnings;
|
|
68
|
+
try {
|
|
69
|
+
if (!status.ok) throw new Error("Status command failed");
|
|
70
|
+
const data = parseCommandJson(status.stdout);
|
|
71
|
+
warnings = data.migrationWarnings === undefined ? [] : data.migrationWarnings;
|
|
72
|
+
if (!Array.isArray(warnings) || warnings.some((warning) => typeof warning !== "string")) throw new Error("Invalid migration warnings");
|
|
73
|
+
} catch {
|
|
74
|
+
checks.push({ id: "openclaw-migrations", ok: false, message: status.timedOut ? "OpenClaw migration check timed out." : "OpenClaw migrations could not be verified. Run openclaw update status --json by hand.", fix: "openclaw update status --json" });
|
|
75
|
+
return checks;
|
|
76
|
+
}
|
|
77
|
+
checks.push({
|
|
78
|
+
id: "openclaw-migrations", ok: warnings.length === 0,
|
|
79
|
+
message: warnings.length ? warnings.join("\n") : "No pending OpenClaw migrations.",
|
|
80
|
+
fix: "openclaw update repair",
|
|
81
|
+
});
|
|
82
|
+
return checks;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** Print the complete mutation plan before a single default-no consent gate. */
|
|
86
|
+
export async function repairOpenclaw({
|
|
87
|
+
run = runOpenclawCommand, command = "openclaw", vendors = [], prefix = null,
|
|
88
|
+
yes = false, isTTY = false, ask = async () => "", print = console.log,
|
|
89
|
+
} = {}) {
|
|
90
|
+
const version = run(command, ["--version"]);
|
|
91
|
+
const actions = openclawInstallPlan(version.ok ? version.stdout : null, { prefix });
|
|
92
|
+
print("Planned changes:");
|
|
93
|
+
for (const action of actions) print(` ${action.description}: ${action.command} ${action.args.join(" ")}`);
|
|
94
|
+
if (!actions.length) print(" None. The installed OpenClaw is not older than the tested release.");
|
|
95
|
+
print("Postflight checks (read-only):");
|
|
96
|
+
print(` ${command} --version`);
|
|
97
|
+
for (const key of [...new Set(vendors)]) {
|
|
98
|
+
const vendor = SUBSCRIPTION_VENDORS[key];
|
|
99
|
+
if (!vendor) throw new Error(`Unknown subscription vendor: ${key}`);
|
|
100
|
+
if (vendor.plugin) print(` ${command} plugins inspect ${vendor.plugin.id} --json`);
|
|
101
|
+
}
|
|
102
|
+
print(` ${command} update status --json`);
|
|
103
|
+
print(" Migration repairs are not run automatically. If needed, run openclaw update repair --yes separately.");
|
|
104
|
+
const changed = [];
|
|
105
|
+
if (actions.length && !yes && (!isTTY || !/^(y|yes)$/i.test((await ask("Apply these changes? [y/N] ")).trim()))) {
|
|
106
|
+
print(isTTY ? "No changes made." : "No changes made. Non-interactive repair requires --yes.");
|
|
107
|
+
return { ok: false, actions, changed, checks: [] };
|
|
108
|
+
}
|
|
109
|
+
for (const action of actions) {
|
|
110
|
+
const result = run(action.command, action.args);
|
|
111
|
+
if (!result.ok) {
|
|
112
|
+
print(`Failed: ${action.description}. Fix: ${action.command} ${action.args.join(" ")}`);
|
|
113
|
+
print(`Completed changes: ${changed.length ? changed.join("; ") : "none"}. The failed install may have partially modified OpenClaw.`);
|
|
114
|
+
return { ok: false, actions, changed, checks: [] };
|
|
115
|
+
}
|
|
116
|
+
changed.push(action.description);
|
|
117
|
+
}
|
|
118
|
+
print(`Changed: ${changed.length ? changed.join("; ") : "nothing"}.`);
|
|
119
|
+
const checks = verifyOpenclaw({ run, command, vendors });
|
|
120
|
+
for (const check of checks) print(`${check.ok ? "OK" : "FAIL"}: ${check.message}${check.ok ? "" : ` Fix: ${check.fix}`}`);
|
|
121
|
+
return { ok: checks.every((check) => check.ok), actions, changed, checks };
|
|
122
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
// Find OpenClaw where install.sh may have put it. With no writable npm global
|
|
2
|
+
// folder (common on Linux), it installs into ~/.npm-global/bin, which isn't on
|
|
3
|
+
// most PATHs: a fresh-install practice run ended "OpenClaw not found", and
|
|
4
|
+
// every job would then fail. The CLI and the MCP server call this first, so
|
|
5
|
+
// they and everything they spawn find it.
|
|
6
|
+
|
|
7
|
+
import fs from "node:fs";
|
|
8
|
+
import os from "node:os";
|
|
9
|
+
import path from "node:path";
|
|
10
|
+
|
|
11
|
+
const isExecutable = (file) => { try { fs.accessSync(file, fs.constants.X_OK); return fs.statSync(file).isFile(); } catch { return false; } };
|
|
12
|
+
|
|
13
|
+
export function openclawFallbackDirs(home = os.homedir()) {
|
|
14
|
+
return [path.join(home, ".npm-global", "bin"), path.join(home, ".local", "bin")];
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** Adds the folder holding openclaw to env.PATH when it isn't already reachable. Returns the folder added, or null. */
|
|
18
|
+
export function ensureOpenClawOnPath(env = process.env, { home = os.homedir(), platform = process.platform } = {}) {
|
|
19
|
+
if (env.NOMARMY_OPENCLAW_CMD) return null;
|
|
20
|
+
const sep = platform === "win32" ? ";" : ":";
|
|
21
|
+
const names = platform === "win32" ? ["openclaw.cmd", "openclaw.exe", "openclaw"] : ["openclaw"];
|
|
22
|
+
const dirs = String(env.PATH ?? "").split(sep).filter(Boolean);
|
|
23
|
+
if (dirs.some((d) => names.some((n) => isExecutable(path.join(d, n))))) return null;
|
|
24
|
+
const found = openclawFallbackDirs(home).find((d) => names.some((n) => isExecutable(path.join(d, n))));
|
|
25
|
+
if (!found) return null;
|
|
26
|
+
env.PATH = [found, ...dirs].join(sep);
|
|
27
|
+
return found;
|
|
28
|
+
}
|
package/lib/openclaw-run.mjs
CHANGED
|
@@ -4,7 +4,7 @@ import path from "node:path";
|
|
|
4
4
|
import { resolveExecutable } from "./process.mjs";
|
|
5
5
|
import { readOpenClawTranscriptTail } from "./transcript.mjs";
|
|
6
6
|
import { loadConfig } from "./config.mjs";
|
|
7
|
-
import { resolveSandboxImage,
|
|
7
|
+
import { resolveSandboxImage, sandboxPathEntries, SANDBOX_NPM_ENV } from "./sandbox-images.mjs";
|
|
8
8
|
import { DEFAULT_AGENT_IMAGE } from "./verify.mjs";
|
|
9
9
|
import { scoutPrompt, isScoutReportUnusable } from "./scout.mjs";
|
|
10
10
|
import { decomposePrompt } from "./decompose.mjs";
|
|
@@ -13,6 +13,7 @@ import { deriveBudgets } from "./budget.mjs";
|
|
|
13
13
|
import { entryContextPerNom } from "./dispatch-config.mjs";
|
|
14
14
|
import { readClaudeSessionUsage } from "./claude-transcript.mjs";
|
|
15
15
|
import { modelRejection, modelRejectionLine } from "./openclaw-errors.mjs";
|
|
16
|
+
import { nearestThinkingLevel } from "./thinking.mjs";
|
|
16
17
|
|
|
17
18
|
const sleep = ms => new Promise(resolve => setTimeout(resolve, ms));
|
|
18
19
|
|
|
@@ -61,6 +62,35 @@ export function makeAbandonedBackgroundProcessTick(stateDir, { idleMs, minElapse
|
|
|
61
62
|
* tool call and files changed so far (liveProgress), and when. Never asks
|
|
62
63
|
* to stop; a failed read just skips that beat.
|
|
63
64
|
*/
|
|
65
|
+
// local_worker_stop (or `nomarmy jobs --stop`) writes this into the job's
|
|
66
|
+
// folder, so any session can stop any job; the next tick ends the worker.
|
|
67
|
+
export const STOP_REQUEST_FILE = "stop-request.json";
|
|
68
|
+
export function makeStopRequestTick(jobDir) {
|
|
69
|
+
return async () => (fs.existsSync(path.join(jobDir, STOP_REQUEST_FILE)) ? { stop: true, reason: "stopped" } : { stop: false });
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Ask a running job to stop. Refuses what isn't running; never kills
|
|
73
|
+
* anything itself (the job's own tick does, within one tick).
|
|
74
|
+
* @returns {{ ok: boolean, message: string }}
|
|
75
|
+
*/
|
|
76
|
+
export function requestJobStop({ jobsRoot, jobId, reason = null, now = new Date() }) {
|
|
77
|
+
if (!/^[A-Za-z0-9._-]{1,120}$/.test(String(jobId ?? ""))) return { ok: false, message: "not a job id" };
|
|
78
|
+
const jobDir = path.join(jobsRoot, jobId);
|
|
79
|
+
let status = null;
|
|
80
|
+
try { status = JSON.parse(fs.readFileSync(path.join(jobDir, "status.json"), "utf8")); } catch { return { ok: false, message: `no job ${jobId} (see local_worker_jobs)` }; }
|
|
81
|
+
if (status.state !== "running") return { ok: false, message: `job ${jobId} isn't running (it's ${status.state ?? "unknown"})` };
|
|
82
|
+
if (status.phase && status.phase !== "worker" && status.phase !== "starting" && status.phase !== "worktree") {
|
|
83
|
+
return { ok: false, message: `job ${jobId} is past its worker (phase ${status.phase}); it's finishing on its own and spends no more model usage` };
|
|
84
|
+
}
|
|
85
|
+
if (fs.existsSync(path.join(jobDir, STOP_REQUEST_FILE))) return { ok: true, message: `a stop was already requested for ${jobId}` };
|
|
86
|
+
fs.writeFileSync(path.join(jobDir, STOP_REQUEST_FILE), JSON.stringify({ at: now.toISOString(), reason: reason ? String(reason).slice(0, 300) : null }));
|
|
87
|
+
return { ok: true, message: `stop requested for ${jobId}: its worker ends within about 15 seconds, verification is skipped, and the worktree is kept uncommitted, so continue_from can pick the work up (on another model too)` };
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function readStopRequest(jobDir) {
|
|
91
|
+
try { return JSON.parse(fs.readFileSync(path.join(jobDir, STOP_REQUEST_FILE), "utf8")); } catch { return null; }
|
|
92
|
+
}
|
|
93
|
+
|
|
64
94
|
export function makeHeartbeatTick(jobDir, liveProgress) {
|
|
65
95
|
// Never two beats at once: a slow beat used to overlap the next.
|
|
66
96
|
let busy = false;
|
|
@@ -128,12 +158,44 @@ export function parseUnsupportedThinkingError(errorMessage) {
|
|
|
128
158
|
// worker that ran out of room, but has valid session state worth resuming)
|
|
129
159
|
// it exists for. Checked against the real captured envelope from that
|
|
130
160
|
// incident, not a synthesized shape.
|
|
131
|
-
export function parseOpenClawInternalTimeout(stdout) {
|
|
161
|
+
export function parseOpenClawInternalTimeout(stdout, stderr = "") {
|
|
162
|
+
// OpenClaw's own timer can end a run in the middle of a tool call. The
|
|
163
|
+
// envelope then reports that call's failure ("Read failed", status
|
|
164
|
+
// "error"), not a timeout; its run log still says so. Seen live on a Grok
|
|
165
|
+
// security review cut off mid-read at 600s: recorded as a crash, so its
|
|
166
|
+
// findings were never recovered.
|
|
167
|
+
if (/embedded run timeout: /.test(String(stderr))) return true;
|
|
132
168
|
let parsed;
|
|
133
169
|
try { parsed = JSON.parse(stdout); } catch { return false; }
|
|
134
170
|
return parsed?.ok === false && (parsed?.status === "timeout" || parsed?.error?.kind === "timeout");
|
|
135
171
|
}
|
|
136
172
|
|
|
173
|
+
/** What a failed run's envelope still says it used, so a failed job's spend is recorded too. */
|
|
174
|
+
export function failedRunUsage(stdout) {
|
|
175
|
+
let parsed;
|
|
176
|
+
try { parsed = JSON.parse(stdout); } catch { return {}; }
|
|
177
|
+
if (!parsed || typeof parsed !== "object") return {};
|
|
178
|
+
const out = {};
|
|
179
|
+
if (parsed.usage && typeof parsed.usage === "object") out.usage = parsed.usage;
|
|
180
|
+
if (Number.isFinite(parsed.costUsd)) out.costUsd = parsed.costUsd;
|
|
181
|
+
if (parsed.toolSummary && typeof parsed.toolSummary === "object") out.toolSummary = parsed.toolSummary;
|
|
182
|
+
if (typeof parsed.sessionId === "string") out.sessionId = parsed.sessionId;
|
|
183
|
+
return out;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* The time a run has, told to the worker. Without it a Grok scout read files
|
|
188
|
+
* for its whole 10 minutes and was cut off before writing a word of its report.
|
|
189
|
+
*/
|
|
190
|
+
export function timeBudgetNote(timeoutSeconds, mode) {
|
|
191
|
+
const total = Number(timeoutSeconds);
|
|
192
|
+
if (!Number.isFinite(total) || total <= 0) return "";
|
|
193
|
+
const minutes = Math.max(1, Math.round(total / 60));
|
|
194
|
+
const wrapAt = Math.max(1, Math.floor((total * 0.8) / 60));
|
|
195
|
+
const what = mode === "scout" || mode === "decompose" ? "stop exploring and write your report from what you have" : "stop starting new work, finish verification and write your report";
|
|
196
|
+
return `\n\nTIME\nThis run has about ${minutes} minute(s), then it is cut off. By minute ${wrapAt}, ${what}. A report on part of the question, with what's left under NOT DONE, is far more useful than none.`;
|
|
197
|
+
}
|
|
198
|
+
|
|
137
199
|
/**
|
|
138
200
|
* A run whose work finished but whose exit failed: OpenClaw logged the run
|
|
139
201
|
* ending normally (stopReason=stop) and then errored, e.g. "Codex one-shot
|
|
@@ -337,7 +399,7 @@ export function createOpenClawRunner(deps) {
|
|
|
337
399
|
// which is the one per-call lever that does reach the sandbox OpenClaw
|
|
338
400
|
// starts for that run. This clones the ambient config, points
|
|
339
401
|
// agents.defaults.sandbox.docker.image at the resolved image, and adds
|
|
340
|
-
//
|
|
402
|
+
// the matched harnesses' extra PATH entries -- verified live: OpenClaw's exec
|
|
341
403
|
// tool does not inherit a sandbox image's own baked ENV PATH on its own (a
|
|
342
404
|
// freshly built Go image's `go` resolved fine under a direct `podman exec`
|
|
343
405
|
// but came back "not found" through `openclaw agent exec` until
|
|
@@ -360,7 +422,7 @@ export function createOpenClawRunner(deps) {
|
|
|
360
422
|
// contract verification does.
|
|
361
423
|
loadConfigFn = () => loadConfig(projectDir),
|
|
362
424
|
resolveSandboxImageFn = resolveSandboxImage,
|
|
363
|
-
|
|
425
|
+
sandboxPathEntriesFn = sandboxPathEntries,
|
|
364
426
|
ambientConfigPathFn = ambientOpenClawConfigPath,
|
|
365
427
|
readAmbientConfig = (p) => JSON.parse(fs.readFileSync(p, "utf8")),
|
|
366
428
|
} = {}) {
|
|
@@ -372,7 +434,7 @@ export function createOpenClawRunner(deps) {
|
|
|
372
434
|
|
|
373
435
|
let image;
|
|
374
436
|
try {
|
|
375
|
-
image = resolveSandboxImageFn({ cwd, explicitImage: process.env.NOMARMY_AGENT_IMAGE || null, defaultImage: DEFAULT_AGENT_IMAGE, config });
|
|
437
|
+
image = resolveSandboxImageFn({ cwd, trustedDir: projectDir, explicitImage: process.env.NOMARMY_AGENT_IMAGE || null, defaultImage: DEFAULT_AGENT_IMAGE, config });
|
|
376
438
|
} catch {
|
|
377
439
|
// A lazy Go/Rust/Python image build failure here should not fail the
|
|
378
440
|
// worker's turn -- it runs in the default image instead, same as before
|
|
@@ -399,8 +461,7 @@ export function createOpenClawRunner(deps) {
|
|
|
399
461
|
// own verification runs: outside the worktree, and off.
|
|
400
462
|
overridden.agents.defaults.sandbox.docker.env = { ...(overridden.agents.defaults.sandbox.docker.env ?? {}), ...SANDBOX_NPM_ENV };
|
|
401
463
|
|
|
402
|
-
const
|
|
403
|
-
const pathPrepend = EXEC_PATH_PREPEND[lang] || [];
|
|
464
|
+
const pathPrepend = sandboxPathEntriesFn(cwd, config);
|
|
404
465
|
if (pathPrepend.length) {
|
|
405
466
|
overridden.tools ??= {};
|
|
406
467
|
overridden.tools.exec ??= {};
|
|
@@ -443,7 +504,7 @@ export function createOpenClawRunner(deps) {
|
|
|
443
504
|
? scoutPrompt({ question: task, mustCover: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.scout, report: jobBudgets.report.scout, evidenceTool })
|
|
444
505
|
: mode === "decompose"
|
|
445
506
|
? decomposePrompt({ objective: task, constraints: acceptance, baseRef, baseSha, workerId, limits: jobBudgets.decompose, report: jobBudgets.report.decompose, evidenceTool })
|
|
446
|
-
: workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement }));
|
|
507
|
+
: workerPrompt({ task, acceptance, verification, mode, baseRef, baseSha, workerId, evidence, report: jobBudgets.report.implement })) + (overridePrompt ? "" : timeBudgetNote(timeoutSeconds, mode));
|
|
447
508
|
fs.writeFileSync(path.join(jobDir, `brief${logSuffix}.txt`), prompt + "\n");
|
|
448
509
|
// --state-dir keeps OpenClaw's session state (its transcript database among
|
|
449
510
|
// it) inside the job directory instead of a temp dir it deletes on exit.
|
|
@@ -476,8 +537,9 @@ export function createOpenClawRunner(deps) {
|
|
|
476
537
|
// The heartbeat runs for every job, so a running job is always watchable
|
|
477
538
|
// (status.json used to keep its launch-time updatedAt until the end).
|
|
478
539
|
const onTick = combineTicks([
|
|
540
|
+
makeStopRequestTick(jobDir),
|
|
479
541
|
makeHeartbeatTick(jobDir),
|
|
480
|
-
...(idleDiff ? [makeIdleDiffTick(cwd, idleDiff), makeAbandonedBackgroundProcessTick(stateDir, idleDiff)] : []),
|
|
542
|
+
...(idleDiff ? [makeIdleDiffTick(cwd, { ...idleDiff, activity: async () => { const t = await readOpenClawTranscriptTail(stateDir, { limit: 5 }); return t.available ? { events: t.events, toolInFlight: t.toolInFlight } : null; } }), makeAbandonedBackgroundProcessTick(stateDir, idleDiff)] : []),
|
|
481
543
|
]);
|
|
482
544
|
const liveLogs = { stdout: path.join(jobDir, `openclaw${logSuffix}.stdout.log`), stderr: path.join(jobDir, `openclaw${logSuffix}.stderr.log`) };
|
|
483
545
|
// Where this call's own transcript events will start (see salvageFinishedRun).
|
|
@@ -519,7 +581,7 @@ export function createOpenClawRunner(deps) {
|
|
|
519
581
|
// -- never loop on the same level, and never mask a real, unrelated
|
|
520
582
|
// failure as a thinking-level problem it isn't.
|
|
521
583
|
if (unsupported && !unsupported.supported.includes(selected.thinking)) {
|
|
522
|
-
const fallback = unsupported.supported
|
|
584
|
+
const fallback = nearestThinkingLevel(selected.thinking, unsupported.supported);
|
|
523
585
|
fs.appendFileSync(path.join(jobDir, "coordinator.log"),
|
|
524
586
|
`${new Date().toISOString()} "${selected.model}" rejected thinking level "${selected.thinking}" (OpenClaw supports: ${unsupported.supported.join(", ")}) -- retrying once with "${fallback}"\n`);
|
|
525
587
|
selected.thinking = fallback; // the manifest's requestedReasoning field should reflect what was ACTUALLY used, not the level that failed
|
|
@@ -557,7 +619,7 @@ export function createOpenClawRunner(deps) {
|
|
|
557
619
|
// a graceful internal timeout, not an opaque crash -- relabel it so
|
|
558
620
|
// executeImplement/executeScout's workerTimedOut check (and therefore
|
|
559
621
|
// report recovery) sees it correctly.
|
|
560
|
-
if (!error.timedOut && parseOpenClawInternalTimeout(error.stdout)) {
|
|
622
|
+
if (!error.timedOut && error.stopReason !== "stopped" && parseOpenClawInternalTimeout(error.stdout, error.stderr)) {
|
|
561
623
|
error.timedOut = true;
|
|
562
624
|
error.stopReason = error.stopReason ?? "openclaw_internal_timeout";
|
|
563
625
|
}
|
|
@@ -582,7 +644,7 @@ export function createOpenClawRunner(deps) {
|
|
|
582
644
|
}
|
|
583
645
|
// What was attempted, for the job record: a failed job used to be
|
|
584
646
|
// labeled with the local default model, whatever it really ran on.
|
|
585
|
-
error.partialResult = { model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
|
|
647
|
+
error.partialResult = { ...failedRunUsage(error.stdout), model: bareModel, provider: selected.entry?.provider ?? workerProvider, budgetsUsed: jobBudgets };
|
|
586
648
|
throw error;
|
|
587
649
|
} finally {
|
|
588
650
|
// A cloned copy of the ambient OpenClaw config (which may carry a real
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
// Metadata-only checks for the runtime used by nomArmy's real test calls.
|
|
2
|
+
import { agentProviderId } from "./agents.mjs";
|
|
3
|
+
export const CODEX_IMPORT_RECOVERY = "openclaw migrate apply codex --from ~/.codex --agent main --include-secrets --item auth:openai --yes";
|
|
4
|
+
export function codexImportRecovery({ profiles = [] } = {}) {
|
|
5
|
+
return [...emailOpenaiProfiles(profiles).map((p) => `openclaw models auth logout ${shellId(p.id)}`), CODEX_IMPORT_RECOVERY].join(" && ");
|
|
6
|
+
}
|
|
7
|
+
|
|
8
|
+
export const CODEX_IMPORT_ARGS = ["migrate", "apply", "codex", "--from", "~/.codex", "--agent", "main", "--include-secrets", "--item", "auth:openai", "--yes"];
|
|
9
|
+
export function emailOpenaiProfiles(profiles) {
|
|
10
|
+
return profiles.filter((p) => /^openai:.*@/.test(p.id ?? ""));
|
|
11
|
+
}
|
|
12
|
+
export function shellId(id) {
|
|
13
|
+
return /^[\w@.+:-]+$/.test(id) ? id : "'" + id.replaceAll("'", "'\\''") + "'";
|
|
14
|
+
}
|
|
15
|
+
export function usableCodexImport(profiles, now = Date.now()) {
|
|
16
|
+
return profiles.some((p) => p.provider === "openai" && /^openai:account-/.test(p.id ?? "") &&
|
|
17
|
+
[p.label, p.name, p.displayName].some((s) => typeof s === "string" && s.includes("(Codex import)")) &&
|
|
18
|
+
(p.expiresAt == null || Date.parse(p.expiresAt) > now));
|
|
19
|
+
}
|
|
20
|
+
export function codexImportIssues(profiles, { now = Date.now() } = {}) {
|
|
21
|
+
const issues = [];
|
|
22
|
+
const emails = emailOpenaiProfiles(profiles);
|
|
23
|
+
const fix = codexImportRecovery({ profiles });
|
|
24
|
+
if (!usableCodexImport(profiles, now)) issues.push({
|
|
25
|
+
id: "codex-import:missing", severity: "error", title: "Codex has no unexpired account-ID (Codex import) profile",
|
|
26
|
+
detail: "The Codex app-server cannot use an email-keyed OpenAI login. Confirm Codex CLI login, then re-import it.",
|
|
27
|
+
fix, short: "Codex import missing",
|
|
28
|
+
});
|
|
29
|
+
if (emails.length) issues.push({
|
|
30
|
+
id: "codex-import:shadowed", severity: "error", title: "Email-keyed OpenAI profiles capture the Codex import",
|
|
31
|
+
detail: `Remove before importing: ${emails.map((p) => p.id).join(", ")}. A models status probe is not an app-server login test.`,
|
|
32
|
+
fix, short: "Codex import shadowed",
|
|
33
|
+
});
|
|
34
|
+
return issues;
|
|
35
|
+
}
|
|
36
|
+
export function modelPolicyIssues({ policies, paths, agents = {}, armySummary = null, modelsInUse = null }) {
|
|
37
|
+
const models = new Set(modelsInUse ?? []);
|
|
38
|
+
for (const a of Object.values(agents ?? {})) {
|
|
39
|
+
const provider = agentProviderId(a);
|
|
40
|
+
if (provider && a.model && a.model !== "auto") models.add(`${provider}/${a.model}`);
|
|
41
|
+
}
|
|
42
|
+
for (const role of Object.values(armySummary?.roles ?? {})) {
|
|
43
|
+
const provider = agentProviderId(agents?.[role.agent]);
|
|
44
|
+
if (provider && role.model && role.model !== "auto" && !role.modelIsAuto) models.add(`${provider}/${role.model}`);
|
|
45
|
+
}
|
|
46
|
+
return policies.flatMap((result, i) => {
|
|
47
|
+
let allow;
|
|
48
|
+
try { allow = result.ok ? JSON.parse(result.stdout) : null; } catch { return []; }
|
|
49
|
+
if (!Array.isArray(allow)) return [];
|
|
50
|
+
const missing = [...models].filter((m) => !allow.includes(m)).sort();
|
|
51
|
+
if (!missing.length) return [];
|
|
52
|
+
return [{ id: `model-policy:${paths[i]}`, severity: "error", title: `OpenClaw model policy excludes: ${missing.join(", ")}`,
|
|
53
|
+
detail: `${paths[i]} refuses models used by nomArmy agents or roles.`,
|
|
54
|
+
fix: `openclaw config unset ${paths[i].replace(/\.allow$/, "")} (or add the missing models to ${paths[i]})`, short: "models blocked" }];
|
|
55
|
+
});
|
|
56
|
+
}
|
package/lib/outcome.mjs
CHANGED
|
@@ -7,6 +7,16 @@ import { parseWorkerReport } from "./report.mjs";
|
|
|
7
7
|
// failed. Recovery exists so that a mangled REPORT cannot destroy correct WORK.
|
|
8
8
|
// It does not exist to launder a failure into a success.
|
|
9
9
|
// ---------------------------------------------------------------------------
|
|
10
|
+
// A failed verification's own detail (the command, its exit code, the end of
|
|
11
|
+
// its output), capped so an issue line stays readable; the full output is in
|
|
12
|
+
// the job's verification.log.
|
|
13
|
+
const FAILURE_DETAIL_CHARS = 600;
|
|
14
|
+
function failureDetail(independentVerification) {
|
|
15
|
+
const detail = String(independentVerification?.detail ?? independentVerification?.reason ?? "").trim();
|
|
16
|
+
if (!detail) return "";
|
|
17
|
+
return `: ${detail.length > FAILURE_DETAIL_CHARS ? `${detail.slice(0, FAILURE_DETAIL_CHARS)}...` : detail}`;
|
|
18
|
+
}
|
|
19
|
+
|
|
10
20
|
export function resolveOutcome({ report, repositoryChanged = false, independentVerification = null, regressionCheck = null, workerFailed = false, workerTimedOut = false, mode = "implement" }) {
|
|
11
21
|
const verification = independentVerification?.status ?? "not_run";
|
|
12
22
|
const parsed = report ?? parseWorkerReport("");
|
|
@@ -34,7 +44,7 @@ export function resolveOutcome({ report, repositoryChanged = false, independentV
|
|
|
34
44
|
if (verification === "fail") {
|
|
35
45
|
return { ...base, outcome: OUTCOMES.NEEDS_REVIEW, reviewRequired: true,
|
|
36
46
|
commitBlockedReason: "independent verification failed despite a clean done/pass report",
|
|
37
|
-
reasons: [
|
|
47
|
+
reasons: [`worker claimed done/pass but independent verification failed${failureDetail(independentVerification)}`] };
|
|
38
48
|
}
|
|
39
49
|
// verify_regression: reverting just the production files and re-running
|
|
40
50
|
// the SAME verification profile still passed (or came back genuinely
|
|
@@ -92,7 +102,7 @@ export function resolveOutcome({ report, repositoryChanged = false, independentV
|
|
|
92
102
|
if (verification === "fail") {
|
|
93
103
|
return { ...recovery, outcome: OUTCOMES.WORKER_REPORT_INVALID,
|
|
94
104
|
commitBlockedReason: "independent verification failed; recovery cannot promote a failure",
|
|
95
|
-
reasons: [...recovery.reasons,
|
|
105
|
+
reasons: [...recovery.reasons, `independent verification FAILED${failureDetail(independentVerification)}`] };
|
|
96
106
|
}
|
|
97
107
|
if (verification === "pass") {
|
|
98
108
|
// A leniently recovered `done` plus a passing independent check is the
|
|
@@ -197,8 +207,17 @@ export function policyAdmissionProblems(job, policy) {
|
|
|
197
207
|
// A refactor meets the regression requirement through its own contract
|
|
198
208
|
// (applyRefactorContract), not the revert check.
|
|
199
209
|
if (policy.require_regression_check && job.verify_regression === false && !job.refactor) problems.push("this repo requires the revert check (policy.require_regression_check in .nomarmy.yml): verify_regression can't be false (a behavior-preserving change can declare refactor: true instead)");
|
|
210
|
+
// stakes: high (security, data loss, irreversible): the checks a General
|
|
211
|
+
// could skip on routine work are mandatory, whatever the repo's policy.
|
|
212
|
+
if (job.stakes === "high") {
|
|
213
|
+
if (!job.verification && !job.refactor) problems.push("a stakes: high job needs a `verification` profile: its result is only as good as the tests that prove it");
|
|
214
|
+
if (job.verify_regression === false && !job.refactor) problems.push("a stakes: high job can't turn the revert check off (verify_regression: false): it's the proof a test catches the change");
|
|
215
|
+
}
|
|
200
216
|
return problems;
|
|
201
217
|
}
|
|
218
|
+
|
|
219
|
+
/** The review line every high-stakes job carries, whatever its outcome. */
|
|
220
|
+
export const HIGH_STAKES_NOTE = "HIGH STAKES: accept this only after an independent review: a scout on another vendor (army_role security-analyst or similar) with reviews: <this job id>, or a judge on another vendor. Passing its tests isn't enough on its own; most defects that pass every check are in security, data or deploy-only paths.";
|
|
202
221
|
/**
|
|
203
222
|
* A declared refactor commits only when verification passed and no test
|
|
204
223
|
* file was added, changed or deleted. Mechanical, not the General's call:
|
package/lib/outcomes.mjs
CHANGED
|
@@ -2,6 +2,9 @@ import { SCOUT_OUTCOMES, SCOUT_STATUS_BY_OUTCOME } from "./scout.mjs";
|
|
|
2
2
|
import { DECOMPOSE_STATUS_BY_OUTCOME } from "./decompose.mjs";
|
|
3
3
|
|
|
4
4
|
export const OUTCOMES = Object.freeze({
|
|
5
|
+
VERIFIED: "VERIFIED",
|
|
6
|
+
VERIFICATION_FAILED: "VERIFICATION_FAILED",
|
|
7
|
+
VERIFICATION_NOT_RUN: "VERIFICATION_NOT_RUN",
|
|
5
8
|
WORKER_DONE: "WORKER_DONE",
|
|
6
9
|
WORKER_PARTIAL: "WORKER_PARTIAL",
|
|
7
10
|
WORKER_BLOCKED: "WORKER_BLOCKED",
|
|
@@ -14,6 +17,9 @@ export const OUTCOMES = Object.freeze({
|
|
|
14
17
|
});
|
|
15
18
|
|
|
16
19
|
export const COORDINATOR_STATUS_BY_OUTCOME = Object.freeze({
|
|
20
|
+
[OUTCOMES.VERIFIED]: "complete",
|
|
21
|
+
[OUTCOMES.VERIFICATION_FAILED]: "failed",
|
|
22
|
+
[OUTCOMES.VERIFICATION_NOT_RUN]: "incomplete",
|
|
17
23
|
[OUTCOMES.WORKER_DONE]: "complete",
|
|
18
24
|
[OUTCOMES.RECOVERED_SUCCESS]: "complete",
|
|
19
25
|
[OUTCOMES.WORKER_BLOCKED]: "blocked",
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
// Is Podman answering, and when did its VM last start? A job dispatched into
|
|
2
|
+
// a stopped Podman, or running while its VM restarts (resizing it with
|
|
3
|
+
// `nomarmy sandbox --memory`, say), fails with nothing but "openclaw exited
|
|
4
|
+
// 2" unless nomArmy checks. Seen live: two jobs lost to a VM resize.
|
|
5
|
+
|
|
6
|
+
import { spawnSync } from "node:child_process";
|
|
7
|
+
|
|
8
|
+
// A loaded machine can take several seconds to answer; 15 seconds keeps a
|
|
9
|
+
// slow Podman from being reported as a stopped one (5 wasn't enough under a
|
|
10
|
+
// load average in the 40s to 80s).
|
|
11
|
+
export const PODMAN_ANSWER_TIMEOUT_MS = 15000;
|
|
12
|
+
const defaultRun = (args) => spawnSync("podman", args, { encoding: "utf8", timeout: PODMAN_ANSWER_TIMEOUT_MS });
|
|
13
|
+
|
|
14
|
+
/** Why jobs can't start their sandbox right now, or null. */
|
|
15
|
+
export function podmanProblem({ run = defaultRun } = {}) {
|
|
16
|
+
const r = run(["info", "--format", "{{.Host.Arch}}"]);
|
|
17
|
+
if (r.status === 0) return null;
|
|
18
|
+
const why = String(r.stderr || r.error?.message || "").split("\n").find((l) => l.trim()) ?? "no answer";
|
|
19
|
+
if (r.error?.code === "ETIMEDOUT") {
|
|
20
|
+
return `Podman didn't answer within ${PODMAN_ANSWER_TIMEOUT_MS / 1000} seconds, so no job was sent. It may be running but starved: check the machine's load (Docker, a local model, other jobs) and retry, or run \`nomarmy sandbox\`. Don't restart Podman while jobs are running; that kills their sandboxes.`;
|
|
21
|
+
}
|
|
22
|
+
return `Podman isn't answering (${why.trim().slice(0, 160)}), so no job was sent: its sandbox couldn't start. Start it with \`podman machine start\` (macOS, Windows), then check with \`nomarmy sandbox\`.`;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** When the Podman VM last started (macOS, Windows), or null where there's no VM. */
|
|
26
|
+
export function podmanVmStartedAt({ run = defaultRun, platform = process.platform } = {}) {
|
|
27
|
+
if (!["darwin", "win32"].includes(platform)) return null;
|
|
28
|
+
try {
|
|
29
|
+
const machines = JSON.parse(run(["machine", "inspect"]).stdout || "[]");
|
|
30
|
+
const m = machines.find((x) => x.State === "running") ?? machines[0];
|
|
31
|
+
return m?.LastUp ?? null;
|
|
32
|
+
} catch { return null; }
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** A job issue when the VM restarted, or stopped, between two readings. */
|
|
36
|
+
export function vmRestartIssue(before, after) {
|
|
37
|
+
if (!before || before === after) return null;
|
|
38
|
+
return after
|
|
39
|
+
? "the Podman VM restarted during this job, which removed its sandbox: the failure is most likely that, not the work. Re-dispatch once Podman is up (nomarmy sandbox)."
|
|
40
|
+
: "the Podman VM stopped during this job, which removed its sandbox: the failure is most likely that, not the work. Start it (podman machine start) and re-dispatch.";
|
|
41
|
+
}
|
package/lib/process.mjs
CHANGED
|
@@ -73,7 +73,10 @@ export function createProcess(ctx) {
|
|
|
73
73
|
if (ticker) clearInterval(ticker);
|
|
74
74
|
child.kill("SIGTERM");
|
|
75
75
|
const error = new Error(message);
|
|
76
|
-
|
|
76
|
+
// A stop someone asked for (local_worker_stop) isn't a timeout: a
|
|
77
|
+
// timeout invites a report-recovery call, which would spend exactly
|
|
78
|
+
// the usage the stop was meant to save.
|
|
79
|
+
error.timedOut = stopReason !== "stopped";
|
|
77
80
|
if (stopReason) error.stopReason = stopReason;
|
|
78
81
|
reject(error);
|
|
79
82
|
};
|
package/lib/propose.mjs
CHANGED
|
@@ -50,21 +50,17 @@ export function buildConfigProposal(evidence) {
|
|
|
50
50
|
}
|
|
51
51
|
|
|
52
52
|
// --- environment.python.requirements -------------------------------------
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
// the sandbox builder can't actually act on would be worse than proposing
|
|
59
|
-
// nothing. A repo whose real Python dependency source is poetry/uv with no
|
|
60
|
-
// requirements.txt anywhere gets no proposal here -- a distinct, real gap
|
|
61
|
-
// (the sandbox has no poetry/uv install path at all yet), not this one.
|
|
53
|
+
// Root project manifests are installed automatically by the Python harness.
|
|
54
|
+
// Do not override that manager selection with incidental requirements files.
|
|
55
|
+
const projectSources = new Set(["pyproject.toml", "uv.lock", "poetry.lock"]);
|
|
56
|
+
const pythonProject = (evidence?.files?.items ?? []).some((f) => projectSources.has(f.path) && !fixturePaths.has(f.path)) ||
|
|
57
|
+
(evidence?.tooling?.items ?? []).some((t) => projectSources.has(t.source) && !fixturePaths.has(t.source));
|
|
62
58
|
const pipTooling = (evidence?.tooling?.items ?? []).filter((t) => t.name === "pip" && t.category === "package-manager" && t.source);
|
|
63
59
|
for (const t of pipTooling) if (fixturePaths.has(t.source)) excludedFixturePaths.add(t.source);
|
|
64
60
|
const requirementsFiles = [...new Set(pipTooling.filter((t) => !fixturePaths.has(t.source)).map((t) => t.source))].sort();
|
|
65
|
-
if (requirementsFiles.length === 1) {
|
|
61
|
+
if (!pythonProject && requirementsFiles.length === 1) {
|
|
66
62
|
proposal.environment = { ...proposal.environment, python: { requirements: requirementsFiles } };
|
|
67
|
-
} else if (requirementsFiles.length > 1) {
|
|
63
|
+
} else if (!pythonProject && requirementsFiles.length > 1) {
|
|
68
64
|
// The exact ambiguity this file's own header already treats as a human
|
|
69
65
|
// judgment call for compose files: several requirements.txt-shaped files
|
|
70
66
|
// (app, dev, a sub-package's own) with no single conventional
|
|
@@ -88,6 +84,9 @@ export function buildConfigProposal(evidence) {
|
|
|
88
84
|
if (firstCommand) {
|
|
89
85
|
notes.push(`verification.quick.commands was seeded from a command found in evidence (${firstCommand}) -- confirm this is actually the right check before trusting it.`);
|
|
90
86
|
proposal.verification = { quick: { environment: "none", commands: [firstCommand] } };
|
|
87
|
+
} else if (pythonProject) {
|
|
88
|
+
notes.push("Python project detected: pytest is a proposed check using the sandbox venv; confirm the project declares pytest and this is the right test command.");
|
|
89
|
+
proposal.verification = { quick: { environment: "none", commands: ["pytest"] } };
|
|
91
90
|
} else {
|
|
92
91
|
notes.push("no command was found anywhere in evidence; verification.quick.commands is a fail-loud placeholder -- replace it before this profile is usable.");
|
|
93
92
|
proposal.verification = { quick: { environment: "none", commands: [PLACEHOLDER_COMMAND] } };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { claimModelRefusalRetry, modelRefusals, recordModelRefusal, recordProbeSuccess } from "./health.mjs";
|
|
2
|
+
|
|
3
|
+
/** Start eligible one-token probes without waiting for them in the caller. */
|
|
4
|
+
export function retryRefusedModelsInBackground(stateRoot, { probeModel, now = Date.now() } = {}) {
|
|
5
|
+
if (!probeModel) return;
|
|
6
|
+
for (const key of Object.keys(modelRefusals(stateRoot))) {
|
|
7
|
+
if (!claimModelRefusalRetry(stateRoot, key, { now })) continue;
|
|
8
|
+
const slash = key.indexOf("/");
|
|
9
|
+
if (slash < 1 || slash === key.length - 1) continue;
|
|
10
|
+
const provider = key.slice(0, slash), model = key.slice(slash + 1);
|
|
11
|
+
Promise.resolve().then(() => probeModel({ provider, model, stateRoot })).then((result) => {
|
|
12
|
+
if (result?.ok) recordProbeSuccess(stateRoot, key, { now });
|
|
13
|
+
else if (result?.refused) recordModelRefusal(stateRoot, key, result.reason, { now });
|
|
14
|
+
}).catch(() => { /* An inconclusive probe keeps the refusal and retry timestamp. */ });
|
|
15
|
+
}
|
|
16
|
+
}
|