nomarmy 0.1.0-alpha.20 → 0.1.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nomarmy.mjs +70 -23
- package/lib/admission.mjs +51 -12
- package/lib/army.mjs +6 -4
- package/lib/codex-link.mjs +37 -0
- package/lib/connect.mjs +3 -3
- package/lib/doctor.mjs +4 -1
- package/lib/health.mjs +73 -22
- package/lib/openclaw-runtime-health.mjs +56 -0
- package/lib/statusline.mjs +2 -2
- package/lib/subscription-setup.mjs +13 -0
- package/lib/usage-limits.mjs +224 -21
- package/mcp/server.mjs +17 -10
- package/package.json +1 -1
package/bin/nomarmy.mjs
CHANGED
|
@@ -31,16 +31,17 @@ import { loadAgents, readAgentsFile, writeAgentsFile, agentsConfigPath, apiAgent
|
|
|
31
31
|
import { loadArmy, mergeArmy, describeArmy, readArmyFile, updateArmyInFile, assignRoleInFile, parseTargetSpec, armyLayerPath, globalConfigDir, DEFAULT_ARMY, ARMY_PHASES, LOCAL_CONFIG_FILENAME } from "../lib/army.mjs";
|
|
32
32
|
import { parseLlamaUrl } from "../lib/execution.mjs";
|
|
33
33
|
import { setupSteps, formatSetupSteps, runSetupPlaybook } from "../lib/setup-steps.mjs";
|
|
34
|
-
import { readUsageSnapshots } from "../lib/usage-limits.mjs";
|
|
34
|
+
import { readUsageSnapshots, refreshStaleOverLimitReadings } from "../lib/usage-limits.mjs";
|
|
35
35
|
import { pickMachine, planResize } from "../lib/sandbox-vm.mjs";
|
|
36
36
|
import { listProcesses, staleSessions, formatStaleSessions } from "../lib/stale-sessions.mjs";
|
|
37
37
|
import { readSetting, writeSetting, writeEnvLine, userCommonPath, userProfilePath, profilePathFor, tildePath } from "../lib/user-config.mjs";
|
|
38
38
|
import { MIN_PODMAN_VM_MB } from "../lib/doctor.mjs";
|
|
39
39
|
import { liveLeases } from "../lib/slots.mjs";
|
|
40
40
|
import { ensureProviderConfig } from "../lib/openclaw-config.mjs";
|
|
41
|
-
import {
|
|
41
|
+
import { linkCodex } from "../lib/codex-link.mjs";
|
|
42
|
+
import { recordProbeSuccess, parseOpenclawAuthProfiles, codexImportRecovery } from "../lib/health.mjs";
|
|
42
43
|
import { pruneJobRuntime } from "../lib/prune.mjs";
|
|
43
|
-
import { SUBSCRIPTION_VENDORS, parseOpenclawVersion, versionAtLeast, parseCatalogModels, parseCliLoginStatus, probeOutcome, parseMuseAuthDescriptor, extractMintedKey } from "../lib/subscription-setup.mjs";
|
|
44
|
+
import { SUBSCRIPTION_VENDORS, parseOpenclawVersion, versionAtLeast, parseCatalogModels, parseCliLoginStatus, probeOutcome, openclawSignInFailure, parseMuseAuthDescriptor, extractMintedKey } from "../lib/subscription-setup.mjs";
|
|
44
45
|
import { PINNED_OPENCLAW_VERSION, repairOpenclaw, verifyOpenclaw, configuredSubscriptionVendors } from "../lib/openclaw-install.mjs";
|
|
45
46
|
import { ensureOpenClawOnPath } from "../lib/openclaw-path.mjs";
|
|
46
47
|
import { THINKING_LEVELS } from "../lib/thinking.mjs";
|
|
@@ -174,7 +175,12 @@ Usage: nomarmy <command> [options]
|
|
|
174
175
|
--model --auth-env [--base-url]
|
|
175
176
|
[--openclaw-provider --plugin] [--register]
|
|
176
177
|
[--update-mcp] (api); --provider --model
|
|
177
|
-
--owner (subscription)
|
|
178
|
+
--owner (subscription). Codex JSON setup can
|
|
179
|
+
--link-openclaw after CLI login; removing a
|
|
180
|
+
capturing email profile non-interactively needs
|
|
181
|
+
--remove-email-profiles. Interactive Codex setup
|
|
182
|
+
offers removal (default yes), then imports the
|
|
183
|
+
CLI login. For any kind, optionally
|
|
178
184
|
--max-concurrent --context-window
|
|
179
185
|
--thinking [minimal|low|medium|high|xhigh|adaptive|max|ultra] --no-thinking
|
|
180
186
|
update <name>
|
|
@@ -186,7 +192,7 @@ Usage: nomarmy <command> [options]
|
|
|
186
192
|
--no-model clears the default model, so every
|
|
187
193
|
role or job names its own.
|
|
188
194
|
(--json with the matching flags; --probe to
|
|
189
|
-
test-call a
|
|
195
|
+
test-call the agent; without a name, all agents)
|
|
190
196
|
remove <name>
|
|
191
197
|
remove one agent
|
|
192
198
|
army <show|init|assign|general>
|
|
@@ -1256,6 +1262,7 @@ function catalogModelsFor(provider) {
|
|
|
1256
1262
|
/** One real, one-token completion through OpenClaw to test a model. */
|
|
1257
1263
|
// Why the last probeWorker() call failed, in the vendor's words when it said.
|
|
1258
1264
|
let lastProbeFailure = null;
|
|
1265
|
+
let lastProbeText = "";
|
|
1259
1266
|
function probeWorker(provider, model) {
|
|
1260
1267
|
// The route a job takes: the ambient OpenClaw config and a state dir of
|
|
1261
1268
|
// its own, never --isolated. --isolated skips that config, and with it the
|
|
@@ -1271,7 +1278,10 @@ function probeWorker(provider, model) {
|
|
|
1271
1278
|
"--json", "--cwd", cwd, "--state-dir", stateDir, "--timeout", "90"], { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], cwd });
|
|
1272
1279
|
// Streams kept apart: OpenClaw logs a "run ... ended" line to stderr
|
|
1273
1280
|
// AFTER the JSON envelope, and the merged text doesn't parse.
|
|
1274
|
-
const
|
|
1281
|
+
const stdout = result.stdout ?? "";
|
|
1282
|
+
const stderr = result.stderr ?? "";
|
|
1283
|
+
const outcome = probeOutcome({ stdout, stderr });
|
|
1284
|
+
lastProbeText = `${stderr}\n${stdout}`;
|
|
1275
1285
|
lastProbeFailure = outcome.ok ? null : outcome.reason;
|
|
1276
1286
|
if (outcome.ok) recordProbeSuccess(agentStateRoot(), `${provider}/${model}`);
|
|
1277
1287
|
return outcome.ok;
|
|
@@ -1303,10 +1313,15 @@ function reapProbeSandbox(stateDir) {
|
|
|
1303
1313
|
/**
|
|
1304
1314
|
* OpenClaw's own provider login, for the vendors whose plugin wants one on
|
|
1305
1315
|
* top of the vendor CLI's login (Claude doesn't: OpenClaw reuses the CLI
|
|
1306
|
-
* session directly).
|
|
1307
|
-
*
|
|
1316
|
+
* session directly). Codex always refreshes its imported copy after CLI
|
|
1317
|
+
* login is confirmed; other providers link when catalog or auth checks fail.
|
|
1308
1318
|
*/
|
|
1309
|
-
function openclawProviderLogin(vendor) {
|
|
1319
|
+
async function openclawProviderLogin(vendor, rl) {
|
|
1320
|
+
if (vendor === SUBSCRIPTION_VENDORS.codex) return linkCodex({
|
|
1321
|
+
run: runQuiet, command: openclawCmd(), isTTY: Boolean(input.isTTY),
|
|
1322
|
+
removeEmailProfiles: flag("remove-email-profiles"),
|
|
1323
|
+
confirm: (prompt, opts) => confirm(rl, prompt, opts), print: (message) => console.log(c.dim(message)),
|
|
1324
|
+
});
|
|
1310
1325
|
const provider = vendor.credential.loginProvider ?? vendor.provider;
|
|
1311
1326
|
console.log(c.dim(`Linking OpenClaw to it (\`openclaw models auth login --provider ${provider}\`) -- follow its prompts:`));
|
|
1312
1327
|
const ok = runInteractive(openclawCmd(), ["models", "auth", "login", "--provider", provider]);
|
|
@@ -1575,16 +1590,23 @@ async function addSubscriptionAgent(rl, agents) {
|
|
|
1575
1590
|
console.log(`\n${c.bold("→")} Models`);
|
|
1576
1591
|
const needsOpenclawLogin = vendor.credential.kind === "openclaw-login";
|
|
1577
1592
|
let linkedOpenclaw = false;
|
|
1593
|
+
if (vendorKey === "codex") {
|
|
1594
|
+
linkedOpenclaw = await openclawProviderLogin(vendor, rl);
|
|
1595
|
+
if (!linkedOpenclaw) {
|
|
1596
|
+
console.log(c.red("✗ Sign-in has no usable auth profile. Retry with `nomarmy agents add subscription codex`."));
|
|
1597
|
+
process.exitCode = 1; return;
|
|
1598
|
+
}
|
|
1599
|
+
}
|
|
1578
1600
|
let models = catalogModelsFor(vendor.provider);
|
|
1579
|
-
if (!models.length && needsOpenclawLogin) {
|
|
1580
|
-
linkedOpenclaw = openclawProviderLogin(vendor);
|
|
1601
|
+
if (!models.length && needsOpenclawLogin && !linkedOpenclaw) {
|
|
1602
|
+
linkedOpenclaw = await openclawProviderLogin(vendor, rl);
|
|
1581
1603
|
if (!linkedOpenclaw) { console.log(c.red(`✗ Sign-in failed or was canceled. Retry with \`nomarmy agents add subscription ${vendorKey}\`.`)); process.exitCode = 1; return; }
|
|
1582
1604
|
models = catalogModelsFor(vendor.provider);
|
|
1583
1605
|
}
|
|
1584
1606
|
// Catalog entries can be cached, and a particular model can refuse a
|
|
1585
1607
|
// working login. Check OpenClaw's auth profile before asking for details.
|
|
1586
1608
|
if (needsOpenclawLogin && !hasUsableOpenclawAuthProfile(vendor.provider)) {
|
|
1587
|
-
if (linkedOpenclaw || !openclawProviderLogin(vendor) || !hasUsableOpenclawAuthProfile(vendor.provider)) {
|
|
1609
|
+
if (linkedOpenclaw || !(await openclawProviderLogin(vendor, rl)) || !hasUsableOpenclawAuthProfile(vendor.provider)) {
|
|
1588
1610
|
console.log(c.red(`✗ Sign-in has no usable auth profile. Retry with \`nomarmy agents add subscription ${vendorKey}\`.`));
|
|
1589
1611
|
process.exitCode = 1;
|
|
1590
1612
|
return;
|
|
@@ -1598,7 +1620,8 @@ async function addSubscriptionAgent(rl, agents) {
|
|
|
1598
1620
|
const pick = (await rl.question(c.bold(`Default model, optional${models.length ? " (a number or an id)" : ""}; blank = pick per role: `))).trim();
|
|
1599
1621
|
const model = /^\d+$/.test(pick) && models.length ? models[Number(pick) - 1] : pick || null;
|
|
1600
1622
|
if (pick && !model) throw new Error(`Not a valid choice: "${pick}".`);
|
|
1601
|
-
//
|
|
1623
|
+
// A usable profile is only a fast precondition. The test call is what
|
|
1624
|
+
// proves sign-in: an unexpired Codex import can still answer 401.
|
|
1602
1625
|
const probeModel = model ?? models[0] ?? vendor.defaultModel;
|
|
1603
1626
|
|
|
1604
1627
|
const ownerDefault = validOwnerEmail(auth.email) ? auth.email : gitUserEmail();
|
|
@@ -1611,9 +1634,17 @@ async function addSubscriptionAgent(rl, agents) {
|
|
|
1611
1634
|
if (!name) { console.log(c.dim("Stopped; nothing was written.")); return; }
|
|
1612
1635
|
|
|
1613
1636
|
console.log(`\n${c.bold("→")} Test call`);
|
|
1614
|
-
const
|
|
1637
|
+
const probed = Boolean(probeModel);
|
|
1638
|
+
const works = probed ? probeWorker(vendor.provider, probeModel) : false;
|
|
1615
1639
|
if (works) console.log(c.green(`✓ ${vendor.provider}/${probeModel} answered a real test prompt.`));
|
|
1616
|
-
else {
|
|
1640
|
+
else if (needsOpenclawLogin && probed && openclawSignInFailure(`${lastProbeFailure ?? ""}\n${lastProbeText}`)) {
|
|
1641
|
+
const why = lastProbeFailure ? `: ${lastProbeFailure}` : "";
|
|
1642
|
+
console.log(c.red(`✗ Sign-in failed${why}.`));
|
|
1643
|
+
console.log(` fix: ${codexImportRecovery()}`);
|
|
1644
|
+
console.log(c.dim("Stopped; nothing was written."));
|
|
1645
|
+
process.exitCode = 1;
|
|
1646
|
+
return;
|
|
1647
|
+
} else {
|
|
1617
1648
|
console.log(c.yellow(probeModel ? `⚠ The login works, but ${vendor.provider}/${probeModel} didn't answer a real test prompt.` : "⚠ The login works, but no model is available for a test call."));
|
|
1618
1649
|
console.log(c.dim(`You can retry later with \`nomarmy agents update ${name} --probe\`.`));
|
|
1619
1650
|
if (!(await confirm(rl, "Save the agent anyway?", { defaultYes: false }))) {
|
|
@@ -1648,6 +1679,15 @@ async function cmdAgentsAddJson() {
|
|
|
1648
1679
|
const thinking = resolveThinkingFlag();
|
|
1649
1680
|
if (thinking !== undefined) agent.thinking = thinking;
|
|
1650
1681
|
}
|
|
1682
|
+
if (kind === "subscription" && agent.provider === "openai" && (flag("link-openclaw") || flag("remove-email-profiles"))) {
|
|
1683
|
+
if (!readLoginStatus("codex").loggedIn) throw new Error("Confirm Codex CLI login first: codex login");
|
|
1684
|
+
const linked = await linkCodex({ run: runQuiet, command: openclawCmd(),
|
|
1685
|
+
removeEmailProfiles: flag("remove-email-profiles"), print: (message) => console.error(message) });
|
|
1686
|
+
if (!linked) throw new Error("Codex import failed; nothing was written.");
|
|
1687
|
+
if (!probeWorker("openai", agent.model ?? SUBSCRIPTION_VENDORS.codex.defaultModel)) {
|
|
1688
|
+
throw new Error(`Codex test call failed; nothing was written. Fix: ${codexImportRecovery()}`);
|
|
1689
|
+
}
|
|
1690
|
+
}
|
|
1651
1691
|
const written = saveAgents({ ...agents, [name]: agent });
|
|
1652
1692
|
const saved = written[name];
|
|
1653
1693
|
let registered = null, mcpUpdated = false;
|
|
@@ -1730,7 +1770,7 @@ async function cmdAgentsUpdate() {
|
|
|
1730
1770
|
if (flag("owner") || value("owner") !== null || value("provider") !== null || value("kind") !== null) {
|
|
1731
1771
|
throw new Error("Kind, provider and owner can't be changed -- that's a different agent. Use `nomarmy agents add`.");
|
|
1732
1772
|
}
|
|
1733
|
-
probe = flag("probe")
|
|
1773
|
+
probe = flag("probe");
|
|
1734
1774
|
if (!Object.keys(changes).length && !probe) throw new Error("Nothing to update -- pass at least one field flag (see `nomarmy agents update` usage).");
|
|
1735
1775
|
} else {
|
|
1736
1776
|
if (!process.stdin.isTTY) throw new Error("nomarmy agents update needs an interactive terminal, or the flags to change (e.g. --max-concurrent 3).");
|
|
@@ -1779,7 +1819,7 @@ async function cmdAgentsUpdate() {
|
|
|
1779
1819
|
}
|
|
1780
1820
|
|
|
1781
1821
|
if (probe && !Object.keys(changes).length) { probeConfiguredAgent(name, current); return; }
|
|
1782
|
-
if (probe && changes.model && !probeWorker(current
|
|
1822
|
+
if (probe && changes.model && !probeWorker(agentProviderId(current), changes.model)) {
|
|
1783
1823
|
out({ error: `a real test prompt to ${current.provider}/${changes.model} didn't come back -- nothing was written` });
|
|
1784
1824
|
process.exit(1);
|
|
1785
1825
|
}
|
|
@@ -2405,7 +2445,7 @@ function armyLayerFlag(fallback = "global") {
|
|
|
2405
2445
|
return chosen[0] ?? fallback;
|
|
2406
2446
|
}
|
|
2407
2447
|
|
|
2408
|
-
function loadArmyForCli({ globalOnly = false } = {}) {
|
|
2448
|
+
function loadArmyForCli({ globalOnly = false, usageRefresh = null } = {}) {
|
|
2409
2449
|
const agents = loadAgentsOrExit().agents;
|
|
2410
2450
|
const loaded = globalOnly
|
|
2411
2451
|
? (() => {
|
|
@@ -2414,8 +2454,8 @@ function loadArmyForCli({ globalOnly = false } = {}) {
|
|
|
2414
2454
|
return { ...mergeArmy([{ layer: "global", army }]), layers: [{ layer: "global", path: filePath, exists: fs.existsSync(filePath), hasArmy: Boolean(army) }] };
|
|
2415
2455
|
})()
|
|
2416
2456
|
: loadArmy({ projectDir: repoDir });
|
|
2417
|
-
const usageSnapshots = readUsageSnapshots(process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents"));
|
|
2418
|
-
return { loaded, agents, summary: describeArmy(loaded, { agents, describeAgent: describeAgentLabel, usageSnapshots, agentProviderId }) };
|
|
2457
|
+
const usageSnapshots = usageRefresh?.snapshots ?? readUsageSnapshots(process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents"));
|
|
2458
|
+
return { loaded, agents, summary: describeArmy(loaded, { agents, describeAgent: describeAgentLabel, usageSnapshots, agentProviderId, usageRefreshError: usageRefresh?.error ?? null, usageRefreshFailed: usageRefresh?.failedProviders ?? null }) };
|
|
2419
2459
|
}
|
|
2420
2460
|
|
|
2421
2461
|
// Claude Code adds settings.local.json to .gitignore for the same reason:
|
|
@@ -2453,7 +2493,9 @@ function repositoryHere(dir) {
|
|
|
2453
2493
|
|
|
2454
2494
|
async function cmdArmyShow() {
|
|
2455
2495
|
const inRepository = repositoryHere(repoDir);
|
|
2456
|
-
const
|
|
2496
|
+
const stateRoot = process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents");
|
|
2497
|
+
const usageRefresh = await refreshStaleOverLimitReadings(stateRoot);
|
|
2498
|
+
const { summary } = loadArmyForCli({ globalOnly: !inRepository, usageRefresh });
|
|
2457
2499
|
if (json) return out(summary);
|
|
2458
2500
|
const g = summary.general;
|
|
2459
2501
|
console.log(c.bold("🪖 nomArmy") + (inRepository ? c.dim(` (${repoDir})`) : ""));
|
|
@@ -2920,6 +2962,7 @@ async function cmdHealth() {
|
|
|
2920
2962
|
const { checkAndRecordHealth } = await import("../lib/health.mjs");
|
|
2921
2963
|
const stateRoot = process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents");
|
|
2922
2964
|
const { result } = await checkAndRecordHealth({ projectDir: repoDir, stateRoot, configDir: globalConfigDir(), env: installEnv() });
|
|
2965
|
+
if (result.issues.some((i) => i.severity === "error")) process.exitCode = 1;
|
|
2923
2966
|
if (json) return out(result);
|
|
2924
2967
|
console.log(c.bold("🍪 nomArmy health") + c.dim(` ${new Date(result.checkedAt).toLocaleString()}`));
|
|
2925
2968
|
if (!result.issues.length) { console.log(c.green("\n✓ Nothing to fix.")); return; }
|
|
@@ -3150,7 +3193,10 @@ async function cmdDoctor() {
|
|
|
3150
3193
|
}
|
|
3151
3194
|
// Import lazily to avoid circular dependencies
|
|
3152
3195
|
const { runDoctor } = await import("../lib/doctor.mjs");
|
|
3153
|
-
const
|
|
3196
|
+
const agents = loadAgentsOrExit().agents;
|
|
3197
|
+
const vendors = configuredSubscriptionVendors(agents);
|
|
3198
|
+
let armySummary = null;
|
|
3199
|
+
try { armySummary = describeArmy(loadArmy({ projectDir: repoDir }), { agents }); } catch { /* other doctor checks still run */ }
|
|
3154
3200
|
let checks;
|
|
3155
3201
|
if (flag("fix")) {
|
|
3156
3202
|
const repaired = await repairOpenclaw({
|
|
@@ -3166,7 +3212,8 @@ async function cmdDoctor() {
|
|
|
3166
3212
|
} else {
|
|
3167
3213
|
checks = verifyOpenclaw({ command: openclawCmd(), vendors });
|
|
3168
3214
|
}
|
|
3169
|
-
await runDoctor({ json, exit: true, env: installEnv(), additionalChecks: checks
|
|
3215
|
+
await runDoctor({ json, exit: true, env: installEnv(), additionalChecks: checks,
|
|
3216
|
+
runtime: { agents, armySummary, openclawCmd: openclawCmd() } });
|
|
3170
3217
|
}
|
|
3171
3218
|
commands.doctor = cmdDoctor;
|
|
3172
3219
|
if (windowsFrontEnd() && windowsPlan(argv) === "FORWARD") {
|
package/lib/admission.mjs
CHANGED
|
@@ -11,7 +11,7 @@ import { readJson } from "./openclaw-run.mjs";
|
|
|
11
11
|
import { readOpenClawTranscriptTail } from "./transcript.mjs";
|
|
12
12
|
import { readClaudeSessionTranscript } from "./claude-transcript.mjs";
|
|
13
13
|
import { notify } from "./notify.mjs";
|
|
14
|
-
import { readCodexRateLimits, recordUsageSnapshot, readUsageSnapshots, usageStatus } from "./usage-limits.mjs";
|
|
14
|
+
import { readCodexRateLimits, recordUsageSnapshot, readUsageSnapshots, usageStatus, usageDisplayText, refreshStaleOverLimitReadings, staleUsageRefreshedNote, USAGE_STALE_MINUTES } from "./usage-limits.mjs";
|
|
15
15
|
import { projectDirProblem } from "./server-context.mjs";
|
|
16
16
|
import { recentModelRefusal, refusedModelIn, recordModelRefusal, recordProbeSuccess, modelProven, modelRefusals, claimModelRefusalRetry } from "./health.mjs";
|
|
17
17
|
import { writeLease, removeLease, liveLeases, liveSlots, acquireSlot } from "./slots.mjs";
|
|
@@ -181,7 +181,7 @@ export function createJobRuntime(deps) {
|
|
|
181
181
|
function admissionHardware() {
|
|
182
182
|
return executionMode(deps.env).managesModelServer ? deps.budgetState.hardwareSnapshot : null;
|
|
183
183
|
}
|
|
184
|
-
function capacitySnapshot() {
|
|
184
|
+
function capacitySnapshot(usageOverride = null) {
|
|
185
185
|
const admission = assessAdmission({ hardware: admissionHardware(), runningJobs: runningCount("local"), slots: deps.budgetState.contextInfo.slots, maxWorkers: currentMaxWorkers() });
|
|
186
186
|
return {
|
|
187
187
|
// The local model's budget. An api or subscription job's scales with
|
|
@@ -191,7 +191,10 @@ export function createJobRuntime(deps) {
|
|
|
191
191
|
admission,
|
|
192
192
|
memory: deps.budgetState.hardwareSnapshot?.memory ?? null,
|
|
193
193
|
running: [...activeJobs.values()].filter(j => !j.settled).map(j => ({ jobId: j.jobId, workerId: j.workerId, mode: j.mode, lane: j.lane, startedAt: j.startedAt, phase: readJson(path.join(jobsRoot, j.jobId, "status.json"))?.phase ?? "starting" })),
|
|
194
|
-
usageLimits: Object.fromEntries(Object.entries(readUsageSnapshots(stateRoot)).map(([provider, snapshot]) =>
|
|
194
|
+
usageLimits: Object.fromEntries(Object.entries(usageOverride ?? readUsageSnapshots(stateRoot)).map(([provider, snapshot]) => {
|
|
195
|
+
const status = usageStatus(snapshot);
|
|
196
|
+
return [provider, { ...status, text: usageDisplayText(status) }];
|
|
197
|
+
})),
|
|
195
198
|
maxWorkers: currentMaxWorkers(),
|
|
196
199
|
remote: (() => { const limit = maxJobs(); return { running: runningCount("remote"), maxWorkers: limit.value, setBy: limit.source === "file" ? limit.path : limit.source === "env" ? "NOMARMY_MAX_POOL_WORKERS" : "default", note: "api and subscription agents; each agent's own max_concurrent also applies. Change with `nomarmy config max-jobs <n>`." }; })()
|
|
197
200
|
};
|
|
@@ -205,16 +208,38 @@ export function createJobRuntime(deps) {
|
|
|
205
208
|
// Every job needs its sandbox; the server wires the Podman check (tests don't).
|
|
206
209
|
const noSandbox = deps.sandboxProblem ? deps.sandboxProblem() : null;
|
|
207
210
|
if (noSandbox) problems.push(noSandbox);
|
|
208
|
-
const
|
|
211
|
+
const usageNow = deps.now?.() ?? Date.now();
|
|
212
|
+
let snapshots = readUsageSnapshots(stateRoot);
|
|
213
|
+
const providerFor = (job) => {
|
|
214
|
+
if (job.mode === "verify" || !job.agentName || job.confirm_over_limit === true) return null;
|
|
215
|
+
try { return agentProviderId(agentsConfig().agents[job.agentName]) ?? null; } catch { return null; }
|
|
216
|
+
};
|
|
217
|
+
const held = [];
|
|
209
218
|
jobs.forEach((j, i) => {
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
if (!snapshot) return;
|
|
215
|
-
const status = usageStatus(snapshot);
|
|
216
|
-
if (status.level === "over") problems.push(`${jobs.length > 1 ? `job ${i + 1}: ` : ""}agent "${j.agentName}" is held at its usage limit: ${status.text} (reading ${status.ageMinutes} minutes old). Ask the operator before resubmitting with confirm_over_limit: true, or send the job to another agent.`);
|
|
219
|
+
const provider = providerFor(j);
|
|
220
|
+
if (!provider || !snapshots[provider]) return;
|
|
221
|
+
const status = usageStatus(snapshots[provider], usageNow);
|
|
222
|
+
if (status.level === "over") held.push({ i, j, provider, status });
|
|
217
223
|
});
|
|
224
|
+
const usageNotes = [];
|
|
225
|
+
let failedProviders = new Set();
|
|
226
|
+
let refreshError = null;
|
|
227
|
+
if (held.some((item) => item.status.ageMinutes >= USAGE_STALE_MINUTES)) {
|
|
228
|
+
const refreshed = await refreshStaleOverLimitReadings(stateRoot, { run: deps.usageRun, now: usageNow, snapshots, timeoutMs: deps.usageRefreshTimeoutMs });
|
|
229
|
+
snapshots = refreshed.snapshots;
|
|
230
|
+
refreshError = refreshed.error;
|
|
231
|
+
failedProviders = new Set(refreshed.failedProviders ?? []);
|
|
232
|
+
}
|
|
233
|
+
const cleared = new Set();
|
|
234
|
+
for (const item of held) {
|
|
235
|
+
const status = snapshots[item.provider] ? usageStatus(snapshots[item.provider], usageNow) : item.status;
|
|
236
|
+
if (status.level !== "over") {
|
|
237
|
+
if (!cleared.has(item.provider)) { cleared.add(item.provider); usageNotes.push(staleUsageRefreshedNote(item.provider)); }
|
|
238
|
+
continue;
|
|
239
|
+
}
|
|
240
|
+
const extra = failedProviders.has(item.provider) ? ` ${refreshError ?? "OpenClaw usage refresh failed"}.` : "";
|
|
241
|
+
problems.push(`${jobs.length > 1 ? `job ${item.i + 1}: ` : ""}agent "${item.j.agentName}" is held at its usage limit: ${status.text} (reading ${status.ageMinutes} minutes old).${extra} Ask the operator before resubmitting with confirm_over_limit: true, or send the job to another agent.`);
|
|
242
|
+
}
|
|
218
243
|
// A pool-routed job is checked against that pool's OWN (model-dependent)
|
|
219
244
|
// budget, not the local-derived global one -- see budgetsForPool. Which
|
|
220
245
|
// specific entry pickProvider will land on isn't known yet at admission
|
|
@@ -370,9 +395,23 @@ export function createJobRuntime(deps) {
|
|
|
370
395
|
problems.push(`not admitted (capacity): ${runningRemote} remote job(s) (api or subscription agents) already running, the limit of ${remoteCeiling} at once (raise it with \`nomarmy config max-jobs <n>\`)`);
|
|
371
396
|
}
|
|
372
397
|
}
|
|
398
|
+
for (const note of usageNotes) admission.reasons.push(note);
|
|
373
399
|
return { problems, admission };
|
|
374
400
|
}
|
|
375
401
|
|
|
402
|
+
async function displayedCapacity() {
|
|
403
|
+
const now = deps.now?.() ?? Date.now();
|
|
404
|
+
const refreshed = await refreshStaleOverLimitReadings(stateRoot, { run: deps.usageRun, now, timeoutMs: deps.usageRefreshTimeoutMs });
|
|
405
|
+
const view = capacitySnapshot(refreshed.snapshots);
|
|
406
|
+
if (refreshed.error) {
|
|
407
|
+
for (const provider of refreshed.failedProviders ?? []) {
|
|
408
|
+
if (!view.usageLimits[provider]) continue;
|
|
409
|
+
view.usageLimits[provider] = { ...view.usageLimits[provider], text: `${view.usageLimits[provider].text}. ${refreshed.error}` };
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
return view;
|
|
413
|
+
}
|
|
414
|
+
|
|
376
415
|
function refusal(problems) {
|
|
377
416
|
return toolText(refusalText(problems, capacitySnapshot), true);
|
|
378
417
|
}
|
|
@@ -481,5 +520,5 @@ export function createJobRuntime(deps) {
|
|
|
481
520
|
return out;
|
|
482
521
|
}
|
|
483
522
|
|
|
484
|
-
return { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize };
|
|
523
|
+
return { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, displayedCapacity, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize };
|
|
485
524
|
}
|
package/lib/army.mjs
CHANGED
|
@@ -31,7 +31,7 @@ import YAML from "yaml";
|
|
|
31
31
|
import { z } from "zod";
|
|
32
32
|
import { typeError } from "./zod-issues.mjs";
|
|
33
33
|
import { agentRunsToolsOnHost } from "./dispatch-schema.mjs";
|
|
34
|
-
import { usageStatus } from "./usage-limits.mjs";
|
|
34
|
+
import { usageStatus, usageDisplayText } from "./usage-limits.mjs";
|
|
35
35
|
|
|
36
36
|
export const GLOBAL_CONFIG_FILENAME = "config.yml";
|
|
37
37
|
export const PROJECT_CONFIG_FILENAMES = Object.freeze([".nomarmy.yml", ".nomarmy.yaml"]);
|
|
@@ -365,7 +365,7 @@ export function generalOverlap(army, agents = {}) {
|
|
|
365
365
|
* description, phase, suggested mode, agent, any problem or overlap with
|
|
366
366
|
* the General, and which layer set each field.
|
|
367
367
|
*/
|
|
368
|
-
export function describeArmy(loaded, { agents = {}, describeAgent = null, usageSnapshots = null, agentProviderId = null, now = Date.now() } = {}) {
|
|
368
|
+
export function describeArmy(loaded, { agents = {}, describeAgent = null, usageSnapshots = null, agentProviderId = null, now = Date.now(), usageRefreshError = null, usageRefreshFailed = null } = {}) {
|
|
369
369
|
const army = loaded.army;
|
|
370
370
|
const problems = armyTargetProblems(army, agents);
|
|
371
371
|
const overlap = generalOverlap(army, agents);
|
|
@@ -376,8 +376,10 @@ export function describeArmy(loaded, { agents = {}, describeAgent = null, usageS
|
|
|
376
376
|
try { provider = agentProviderId(agents[name]); } catch { return null; }
|
|
377
377
|
const snapshot = usageSnapshots[provider];
|
|
378
378
|
if (!snapshot) return null;
|
|
379
|
-
const
|
|
380
|
-
|
|
379
|
+
const status = usageStatus(snapshot, now);
|
|
380
|
+
const failed = new Set(usageRefreshFailed ?? []);
|
|
381
|
+
const text = failed.has(provider) && usageRefreshError ? `${usageDisplayText(status)}. ${usageRefreshError}` : usageDisplayText(status);
|
|
382
|
+
return { level: status.level, text };
|
|
381
383
|
};
|
|
382
384
|
// A subscription whose owner isn't the General's own is worth a word,
|
|
383
385
|
// not a refusal: it may be the same person's other account (a real Senti
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import os from "node:os";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { parseOpenclawAuthProfiles } from "./health.mjs";
|
|
4
|
+
import { CODEX_IMPORT_ARGS, CODEX_IMPORT_RECOVERY, emailOpenaiProfiles, usableCodexImport } from "./openclaw-runtime-health.mjs";
|
|
5
|
+
|
|
6
|
+
// Never relay migration output: --include-secrets may expose credential data.
|
|
7
|
+
export async function linkCodex({ run, command = "openclaw", isTTY = false, removeEmailProfiles = false, confirm = async () => false, print = () => {}, codexDir = path.join(os.homedir(), ".codex"), now = Date.now() }) {
|
|
8
|
+
const read = async () => {
|
|
9
|
+
const result = await run(command, ["models", "auth", "list", "--json"]);
|
|
10
|
+
return result.ok ? parseOpenclawAuthProfiles(result.stdout) : null;
|
|
11
|
+
};
|
|
12
|
+
const profiles = await read();
|
|
13
|
+
if (!profiles) { print("Cannot inspect OpenClaw auth profiles; nothing imported."); return false; }
|
|
14
|
+
const emails = emailOpenaiProfiles(profiles);
|
|
15
|
+
if (emails.length) {
|
|
16
|
+
print(`Email-keyed profiles will capture the Codex import: ${emails.map((p) => p.id).join(", ")}.`);
|
|
17
|
+
if (!removeEmailProfiles && !(isTTY && await confirm("Remove these email-keyed profiles before importing?", { defaultYes: true }))) {
|
|
18
|
+
print("Stopped; use --remove-email-profiles to allow removal non-interactively.");
|
|
19
|
+
return false;
|
|
20
|
+
}
|
|
21
|
+
for (const p of emails) {
|
|
22
|
+
if (!(await run(command, ["models", "auth", "logout", p.id])).ok) { print("OpenClaw profile logout failed; nothing imported."); return false; }
|
|
23
|
+
}
|
|
24
|
+
const remaining = await read();
|
|
25
|
+
if (!remaining || emailOpenaiProfiles(remaining).length) { print("Email-keyed profiles still exist; nothing imported."); return false; }
|
|
26
|
+
}
|
|
27
|
+
print(`Linking OpenClaw: ${CODEX_IMPORT_RECOVERY}`);
|
|
28
|
+
const args = CODEX_IMPORT_ARGS.map((arg) => arg === "~/.codex" ? codexDir : arg);
|
|
29
|
+
if (!(await run(command, args)).ok) { print("Codex import failed."); return false; }
|
|
30
|
+
const imported = await read();
|
|
31
|
+
if (!imported || emailOpenaiProfiles(imported).length || !usableCodexImport(imported, now)) {
|
|
32
|
+
print("Sign-in has no usable auth profile: expected an unexpired openai:account- profile labeled (Codex import).");
|
|
33
|
+
return false;
|
|
34
|
+
}
|
|
35
|
+
print("Confirmed an unexpired openai:account- (Codex import) profile.");
|
|
36
|
+
return true;
|
|
37
|
+
}
|
package/lib/connect.mjs
CHANGED
|
@@ -434,7 +434,7 @@ export function portableServerLaunch({ nomarmyRoot, installDir = defaultInstallD
|
|
|
434
434
|
}
|
|
435
435
|
|
|
436
436
|
/** The on-disk server scripts registered by connect for this repository. */
|
|
437
|
-
export async function registeredMcpCopies({ projectDir = null, installDir = defaultInstallDir(), nomarmyRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."), run }) {
|
|
437
|
+
export async function registeredMcpCopies({ projectDir = null, installDir = defaultInstallDir(), nomarmyRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."), cursorConfigPath = defaultCursorConfigPath(), homeDir = os.homedir(), run }) {
|
|
438
438
|
const copies = [];
|
|
439
439
|
const add = (target, scope, command, args = []) => {
|
|
440
440
|
let serverPath = null;
|
|
@@ -458,7 +458,7 @@ export async function registeredMcpCopies({ projectDir = null, installDir = defa
|
|
|
458
458
|
if (scope && command) add("claude", scope, command, args ? [args] : []);
|
|
459
459
|
};
|
|
460
460
|
claude(await output("claude", ["mcp", "get", SERVER_NAME], projectDir || undefined));
|
|
461
|
-
if (projectDir) claude(await output("claude", ["mcp", "get", SERVER_NAME],
|
|
461
|
+
if (projectDir) claude(await output("claude", ["mcp", "get", SERVER_NAME], homeDir));
|
|
462
462
|
const codex = await output("codex", ["mcp", "get", SERVER_NAME, "--json"]);
|
|
463
463
|
if (codex) {
|
|
464
464
|
try {
|
|
@@ -466,7 +466,7 @@ export async function registeredMcpCopies({ projectDir = null, installDir = defa
|
|
|
466
466
|
if (transport) add("codex", "user", transport.command, transport.args);
|
|
467
467
|
} catch { /* no registration */ }
|
|
468
468
|
}
|
|
469
|
-
for (const [scope, configPath] of [["user",
|
|
469
|
+
for (const [scope, configPath] of [["user", cursorConfigPath], ...(projectDir ? [["local", path.join(projectDir, ".cursor", "mcp.json")]] : [])]) {
|
|
470
470
|
try {
|
|
471
471
|
const entry = readCursorConfig(configPath).mcpServers?.[SERVER_NAME];
|
|
472
472
|
if (entry) add("cursor", scope === "local" && entry.command === PORTABLE_SERVER.command ? "project" : scope, entry.command, entry.args);
|
package/lib/doctor.mjs
CHANGED
|
@@ -553,7 +553,10 @@ export function evaluateChecks(facts) {
|
|
|
553
553
|
export async function runDoctor(opts = {}) {
|
|
554
554
|
const { json = false, exit = false, env = process.env } = opts;
|
|
555
555
|
const facts = opts.facts ?? await collectFacts(env);
|
|
556
|
-
const
|
|
556
|
+
const { runtimeIssues } = await import("./health.mjs");
|
|
557
|
+
const runtime = opts.runtime ? await runtimeIssues(opts.runtime) : [];
|
|
558
|
+
const checks = [...evaluateChecks(facts), ...(opts.additionalChecks ?? []),
|
|
559
|
+
...runtime.map((issue) => ({ id: issue.id, ok: false, message: `${issue.title}. ${issue.detail}`, fix: issue.fix }))];
|
|
557
560
|
const allOk = checks.every((c) => c.ok);
|
|
558
561
|
const result = { ok: allOk, checks };
|
|
559
562
|
|
package/lib/health.mjs
CHANGED
|
@@ -4,7 +4,8 @@
|
|
|
4
4
|
// stall the server that runs it.
|
|
5
5
|
//
|
|
6
6
|
// login-expiry an OAuth login (the ChatGPT plan's Codex import) about
|
|
7
|
-
// to expire or already expired
|
|
7
|
+
// to expire or already expired. The import is a copy of
|
|
8
|
+
// Codex's own login; re-import it, don't run agents add.
|
|
8
9
|
// openclaw-update OpenClaw older than npm's latest (a stale 2026.9.5
|
|
9
10
|
// catalog made gpt-6-sol look unavailable)
|
|
10
11
|
// plugin-skew an OpenClaw plugin older than OpenClaw itself
|
|
@@ -21,10 +22,11 @@
|
|
|
21
22
|
import { execFile } from "node:child_process";
|
|
22
23
|
import fs from "node:fs";
|
|
23
24
|
import path from "node:path";
|
|
25
|
+
import { codexImportRecovery, codexImportIssues, modelPolicyIssues } from "./openclaw-runtime-health.mjs";
|
|
24
26
|
import { executionMode } from "./execution.mjs";
|
|
25
27
|
import { modelRejection } from "./openclaw-errors.mjs";
|
|
26
28
|
import { providerConfigured, readOpenclawConfig } from "./openclaw-config.mjs";
|
|
27
|
-
import { readUsageSnapshots, usageStatus } from "./usage-limits.mjs";
|
|
29
|
+
import { readUsageSnapshots, usageStatus, usageDisplayText, usageReadingIsStale, refreshStaleOverLimitReadings } from "./usage-limits.mjs";
|
|
28
30
|
import { freshnessIssues, readInstallVersions } from "./install-freshness.mjs";
|
|
29
31
|
|
|
30
32
|
import { PINNED_OPENCLAW_VERSION } from "./openclaw-install.mjs";
|
|
@@ -49,18 +51,59 @@ export function parseOpenclawAuthProfiles(stdout) {
|
|
|
49
51
|
} catch { return null; }
|
|
50
52
|
}
|
|
51
53
|
|
|
54
|
+
export { CODEX_IMPORT_RECOVERY, codexImportRecovery } from "./openclaw-runtime-health.mjs";
|
|
55
|
+
|
|
56
|
+
function profileUsable(profile, now) {
|
|
57
|
+
if (profile?.expiresAt == null || profile.expiresAt === "") return true;
|
|
58
|
+
const expires = Date.parse(profile.expiresAt);
|
|
59
|
+
return Number.isFinite(expires) && expires > now;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function profileName(profile) {
|
|
63
|
+
const id = String(profile.id ?? profile.provider ?? "unknown");
|
|
64
|
+
const raw = [profile.label, profile.name, profile.displayName].find((value) => typeof value === "string" && value.trim());
|
|
65
|
+
if (!raw) return id;
|
|
66
|
+
const label = raw.trim();
|
|
67
|
+
if (label === id || label.startsWith(`${id} `)) return label;
|
|
68
|
+
return label.startsWith("(") ? `${id} ${label}` : `${id} (${label})`;
|
|
69
|
+
}
|
|
70
|
+
|
|
52
71
|
/** OAuth profiles near or past expiry, from `openclaw models auth list --json`. */
|
|
53
72
|
export function loginExpiryIssues(authJson, { now = Date.now(), warnDays = 7 } = {}) {
|
|
73
|
+
const profiles = authJson?.profiles ?? [];
|
|
54
74
|
const issues = [];
|
|
55
|
-
for (const p of
|
|
75
|
+
for (const p of profiles) {
|
|
56
76
|
const at = Date.parse(p.expiresAt ?? "");
|
|
57
77
|
if (!Number.isFinite(at)) continue;
|
|
58
78
|
const days = Math.floor((at - now) / DAY);
|
|
59
79
|
const who = p.provider === "openai" ? "Codex (ChatGPT plan)" : p.provider;
|
|
60
|
-
const
|
|
61
|
-
|
|
62
|
-
|
|
80
|
+
const name = profileName(p);
|
|
81
|
+
const expired = at <= now;
|
|
82
|
+
const siblingValid = profiles.some((other) => other !== p && other.provider === p.provider && other.id !== p.id && profileUsable(other, now));
|
|
83
|
+
const fix = p.provider === "openai" ? codexImportRecovery({ profiles }) : `openclaw models auth login --provider ${p.provider}`;
|
|
84
|
+
if (expired) {
|
|
85
|
+
const detail = siblingValid
|
|
86
|
+
? `Auth profile ${name} expired ${new Date(at).toISOString().slice(0, 10)}. OpenClaw may still pick it.`
|
|
87
|
+
: `Auth profile ${name} expired ${new Date(at).toISOString().slice(0, 10)}; every job on it will fail.`;
|
|
88
|
+
issues.push({ id: `login-expired:${p.id}`, severity: "error", title: `${who} login has expired: ${name}`, detail, fix, short: `${p.provider} login expired` });
|
|
89
|
+
} else if (at - now <= warnDays * DAY) issues.push({ id: `login-expiring:${p.id}`, severity: "warn", title: `${who} login expires in ${days} day${days === 1 ? "" : "s"}: ${name}`, detail: `Auth profile ${name} expires ${new Date(at).toISOString().slice(0, 10)}.`, fix, short: `${p.provider} login ${days}d` });
|
|
90
|
+
}
|
|
91
|
+
return issues;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Shared runtime preconditions for doctor and health. */
|
|
95
|
+
export async function runtimeIssues({ run = runBounded, openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", openclawAgent = "main", agents = {}, armySummary = null, modelsInUse = null, now = Date.now() } = {}) {
|
|
96
|
+
const paths = [`agents.entries.${openclawAgent}.modelPolicy.allow`, "agents.defaults.modelPolicy.allow"];
|
|
97
|
+
const [auth, ...policies] = await Promise.all([
|
|
98
|
+
run(openclawCmd, ["models", "auth", "list", "--json"]),
|
|
99
|
+
...paths.map((p) => run(openclawCmd, ["config", "get", p, "--json"])),
|
|
100
|
+
]);
|
|
101
|
+
const profiles = auth.ok ? parseOpenclawAuthProfiles(auth.stdout) : null;
|
|
102
|
+
const issues = profiles ? loginExpiryIssues({ profiles }, { now }) : [];
|
|
103
|
+
if (Object.values(agents ?? {}).some((a) => a.kind === "subscription" && a.provider === "openai")) {
|
|
104
|
+
issues.push(...codexImportIssues(profiles ?? [], { now }));
|
|
63
105
|
}
|
|
106
|
+
issues.push(...modelPolicyIssues({ policies, paths, agents, armySummary, modelsInUse }));
|
|
64
107
|
return issues;
|
|
65
108
|
}
|
|
66
109
|
|
|
@@ -241,33 +284,41 @@ export function leftoverIssues({ retainedWorktrees = 0, jobsBytes = 0, staleRunn
|
|
|
241
284
|
* Run every check. `env` supplies what each needs, with real defaults;
|
|
242
285
|
* tests pass their own.
|
|
243
286
|
*/
|
|
244
|
-
export async function runHealthChecks({ now = Date.now(), mode = "local", openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", run = runBounded, armySummary = null, agentsError = null, jobsRoot = null, pidAlive = () => true, agents = null, openclawConfig = null, vendors = {}, modelsInUse = null, autoPruned = null, usageSnapshots = null, install = null } = {}) {
|
|
287
|
+
export async function runHealthChecks({ now = Date.now(), mode = "local", openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw", run = runBounded, armySummary = null, agentsError = null, jobsRoot = null, pidAlive = () => true, agents = null, openclawConfig = null, vendors = {}, modelsInUse = null, autoPruned = null, usageSnapshots = null, stateRoot = null, install = null } = {}) {
|
|
245
288
|
const issues = [];
|
|
246
289
|
if (autoPruned?.freedBytes) issues.push({ id: `auto-prune:${new Date(now).toISOString()}`, severity: "info",
|
|
247
290
|
title: `Freed ${(autoPruned.freedBytes / 1024 ** 3).toFixed(2)} GB: ${[autoPruned.pruned ? `runtime data of ${autoPruned.pruned} finished job${autoPruned.pruned === 1 ? "" : "s"} older than ${autoPruned.olderThanHours}h` : null, autoPruned.scratchCleared ? `OpenClaw scratch files of ${autoPruned.scratchCleared} more` : null].filter(Boolean).join(", ")}`,
|
|
248
291
|
detail: "Automatic; each job's record and report are kept. NOMARMY_AUTO_PRUNE_HOURS sets the age (0 turns it off).", fix: null, short: null });
|
|
249
292
|
if (agents) issues.push(...providerConfigIssues({ agents, openclawConfig, vendors }));
|
|
250
|
-
|
|
251
|
-
|
|
293
|
+
usageSnapshots ??= stateRoot ? readUsageSnapshots(stateRoot) : null;
|
|
294
|
+
const staleOver = usageSnapshots && Object.values(usageSnapshots).some((snapshot) => usageReadingIsStale(usageStatus(snapshot, now)));
|
|
295
|
+
const [runtime, version, latest, plugins, nomarmyLatest, refreshed] = await Promise.all([
|
|
296
|
+
runtimeIssues({ run, openclawCmd, agents, armySummary, modelsInUse, now }),
|
|
297
|
+
run(openclawCmd, ["--version"]),
|
|
298
|
+
run("npm", ["view", "openclaw", "version"], { timeoutMs: 15000 }),
|
|
299
|
+
run(openclawCmd, ["plugins", "inspect", "codex"]),
|
|
300
|
+
install ? run("npm", ["view", "nomarmy", "dist-tags.alpha"], { timeoutMs: 15000 }) : null,
|
|
301
|
+
staleOver ? refreshStaleOverLimitReadings(stateRoot, { run, now, snapshots: usageSnapshots, openclawCmd }) : null,
|
|
302
|
+
]);
|
|
303
|
+
const snapshots = refreshed?.snapshots ?? usageSnapshots;
|
|
304
|
+
const failedProviders = new Set(refreshed?.failedProviders ?? []);
|
|
305
|
+
if (snapshots) {
|
|
306
|
+
for (const [provider, snapshot] of Object.entries(snapshots)) {
|
|
252
307
|
const status = usageStatus(snapshot, now);
|
|
253
308
|
if (status.level === "ok") continue;
|
|
254
309
|
const short = `${provider} ${status.short}`;
|
|
310
|
+
const stale = usageReadingIsStale(status);
|
|
311
|
+
const detail = stale
|
|
312
|
+
? `${usageDisplayText(status)}.${failedProviders.has(provider) ? ` ${refreshed.error}.` : ""}`
|
|
313
|
+
: `${status.text} (reading ${status.ageMinutes} minutes old).`;
|
|
255
314
|
issues.push({ id: `usage:${provider}:${status.level}`, severity: "warn", title: status.level === "over" ? `${provider} is at its usage limit` : `${provider} usage is ${status.short}`,
|
|
256
|
-
detail
|
|
315
|
+
detail,
|
|
257
316
|
fix: status.level === "over" ? "wait for the reset, or move its roles with nomarmy army assign" : "plan remaining work or move its roles with nomarmy army assign",
|
|
258
317
|
short });
|
|
259
318
|
}
|
|
260
319
|
}
|
|
261
|
-
const [auth, version, latest, plugins, nomarmyLatest] = await Promise.all([
|
|
262
|
-
run(openclawCmd, ["models", "auth", "list", "--json"]),
|
|
263
|
-
run(openclawCmd, ["--version"]),
|
|
264
|
-
run("npm", ["view", "openclaw", "version"], { timeoutMs: 15000 }),
|
|
265
|
-
run(openclawCmd, ["plugins", "inspect", "codex"]),
|
|
266
|
-
install ? run("npm", ["view", "nomarmy", "dist-tags.alpha"], { timeoutMs: 15000 }) : null,
|
|
267
|
-
]);
|
|
268
320
|
if (install) issues.push(...freshnessIssues({ ...install, latestVersion: nomarmyLatest?.ok ? nomarmyLatest.stdout : null }));
|
|
269
|
-
|
|
270
|
-
if (authProfiles) issues.push(...loginExpiryIssues({ profiles: authProfiles }, { now }));
|
|
321
|
+
issues.push(...runtime);
|
|
271
322
|
const pluginVersion = /Version:\s*(\S+)/.exec(plugins.stdout ?? "")?.[1];
|
|
272
323
|
if (version.ok) issues.push(...versionIssues({ installed: version.stdout, latest: latest.ok ? latest.stdout : null, plugins: pluginVersion ? [{ id: "codex", version: pluginVersion }] : [] }));
|
|
273
324
|
if (agentsError) issues.push({ id: `config:agents:${agentsError}`, severity: "error", title: "agents.yml can't be loaded", detail: agentsError, fix: "nomarmy agents list (shows the problem)", short: "agents.yml broken" });
|
|
@@ -318,10 +369,10 @@ export function recordHealth(file, result, { now = Date.now() } = {}) {
|
|
|
318
369
|
}
|
|
319
370
|
|
|
320
371
|
/** Read versions and harnesses only from MCP registrations that launch nomArmy. */
|
|
321
|
-
export async function readRegisteredInstall({ projectDir, run = runBounded, installDir } = {}) {
|
|
372
|
+
export async function readRegisteredInstall({ projectDir, run = runBounded, installDir, cursorConfigPath, homeDir } = {}) {
|
|
322
373
|
const { registeredMcpCopies } = await import("./connect.mjs");
|
|
323
374
|
const { loadHarnesses } = await import("./harnesses.mjs");
|
|
324
|
-
const registered = await registeredMcpCopies({ projectDir, run, installDir });
|
|
375
|
+
const registered = await registeredMcpCopies({ projectDir, run, installDir, cursorConfigPath, homeDir });
|
|
325
376
|
if (!registered.length) return null;
|
|
326
377
|
return { registrations: registered.map(({ target, scope, serverPath }) => {
|
|
327
378
|
const copyDir = path.dirname(path.dirname(serverPath));
|
|
@@ -363,7 +414,7 @@ export async function checkAndRecordHealth({ projectDir, stateRoot, configDir, n
|
|
|
363
414
|
const mode = executionMode(env).mode;
|
|
364
415
|
const install = await readRegisteredInstall({ projectDir });
|
|
365
416
|
const result = await runHealthChecks({ now, mode, armySummary, agentsError, jobsRoot: path.join(stateRoot, "jobs"), pidAlive,
|
|
366
|
-
agents, openclawConfig: readOpenclawConfig(), vendors: SUBSCRIPTION_VENDORS, modelsInUse, autoPruned, usageSnapshots: readUsageSnapshots(stateRoot), install });
|
|
417
|
+
agents, openclawConfig: readOpenclawConfig(), vendors: SUBSCRIPTION_VENDORS, modelsInUse, autoPruned, usageSnapshots: readUsageSnapshots(stateRoot), stateRoot, install });
|
|
367
418
|
try {
|
|
368
419
|
const { loadJobRecords, agentLookup } = await import("./stats.mjs");
|
|
369
420
|
const { recentSuggestions } = await import("./suggestions.mjs");
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
// Metadata-only checks for the runtime used by nomArmy's real test calls.
|
|
2
|
+
import { agentProviderId } from "./agents.mjs";
|
|
3
|
+
export const CODEX_IMPORT_RECOVERY = "openclaw migrate apply codex --from ~/.codex --agent main --include-secrets --item auth:openai --yes";
|
|
4
|
+
export function codexImportRecovery({ profiles = [] } = {}) {
|
|
5
|
+
return [...emailOpenaiProfiles(profiles).map((p) => `openclaw models auth logout ${shellId(p.id)}`), CODEX_IMPORT_RECOVERY].join(" && ");
|
|
6
|
+
}
|
|
7
|
+
|
|
8
|
+
export const CODEX_IMPORT_ARGS = ["migrate", "apply", "codex", "--from", "~/.codex", "--agent", "main", "--include-secrets", "--item", "auth:openai", "--yes"];
|
|
9
|
+
export function emailOpenaiProfiles(profiles) {
|
|
10
|
+
return profiles.filter((p) => /^openai:.*@/.test(p.id ?? ""));
|
|
11
|
+
}
|
|
12
|
+
export function shellId(id) {
|
|
13
|
+
return /^[\w@.+:-]+$/.test(id) ? id : "'" + id.replaceAll("'", "'\\''") + "'";
|
|
14
|
+
}
|
|
15
|
+
export function usableCodexImport(profiles, now = Date.now()) {
|
|
16
|
+
return profiles.some((p) => p.provider === "openai" && /^openai:account-/.test(p.id ?? "") &&
|
|
17
|
+
[p.label, p.name, p.displayName].some((s) => typeof s === "string" && s.includes("(Codex import)")) &&
|
|
18
|
+
(p.expiresAt == null || Date.parse(p.expiresAt) > now));
|
|
19
|
+
}
|
|
20
|
+
export function codexImportIssues(profiles, { now = Date.now() } = {}) {
|
|
21
|
+
const issues = [];
|
|
22
|
+
const emails = emailOpenaiProfiles(profiles);
|
|
23
|
+
const fix = codexImportRecovery({ profiles });
|
|
24
|
+
if (!usableCodexImport(profiles, now)) issues.push({
|
|
25
|
+
id: "codex-import:missing", severity: "error", title: "Codex has no unexpired account-ID (Codex import) profile",
|
|
26
|
+
detail: "The Codex app-server cannot use an email-keyed OpenAI login. Confirm Codex CLI login, then re-import it.",
|
|
27
|
+
fix, short: "Codex import missing",
|
|
28
|
+
});
|
|
29
|
+
if (emails.length) issues.push({
|
|
30
|
+
id: "codex-import:shadowed", severity: "error", title: "Email-keyed OpenAI profiles capture the Codex import",
|
|
31
|
+
detail: `Remove before importing: ${emails.map((p) => p.id).join(", ")}. A models status probe is not an app-server login test.`,
|
|
32
|
+
fix, short: "Codex import shadowed",
|
|
33
|
+
});
|
|
34
|
+
return issues;
|
|
35
|
+
}
|
|
36
|
+
export function modelPolicyIssues({ policies, paths, agents = {}, armySummary = null, modelsInUse = null }) {
|
|
37
|
+
const models = new Set(modelsInUse ?? []);
|
|
38
|
+
for (const a of Object.values(agents ?? {})) {
|
|
39
|
+
const provider = agentProviderId(a);
|
|
40
|
+
if (provider && a.model && a.model !== "auto") models.add(`${provider}/${a.model}`);
|
|
41
|
+
}
|
|
42
|
+
for (const role of Object.values(armySummary?.roles ?? {})) {
|
|
43
|
+
const provider = agentProviderId(agents?.[role.agent]);
|
|
44
|
+
if (provider && role.model && role.model !== "auto" && !role.modelIsAuto) models.add(`${provider}/${role.model}`);
|
|
45
|
+
}
|
|
46
|
+
return policies.flatMap((result, i) => {
|
|
47
|
+
let allow;
|
|
48
|
+
try { allow = result.ok ? JSON.parse(result.stdout) : null; } catch { return []; }
|
|
49
|
+
if (!Array.isArray(allow)) return [];
|
|
50
|
+
const missing = [...models].filter((m) => !allow.includes(m)).sort();
|
|
51
|
+
if (!missing.length) return [];
|
|
52
|
+
return [{ id: `model-policy:${paths[i]}`, severity: "error", title: `OpenClaw model policy excludes: ${missing.join(", ")}`,
|
|
53
|
+
detail: `${paths[i]} refuses models used by nomArmy agents or roles.`,
|
|
54
|
+
fix: `openclaw config unset ${paths[i].replace(/\.allow$/, "")} (or add the missing models to ${paths[i]})`, short: "models blocked" }];
|
|
55
|
+
});
|
|
56
|
+
}
|
package/lib/statusline.mjs
CHANGED
|
@@ -50,12 +50,12 @@ export function statusLineText({ session = {}, stateRoot, now = Date.now(), maxL
|
|
|
50
50
|
const root = stateRoot ?? (process.env.NOMARMY_AGENT_STATE || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents"));
|
|
51
51
|
// Every Claude Code window redraws this line with its own last-seen
|
|
52
52
|
// limits, idle ones included, so a reading is merged rather than trusted
|
|
53
|
-
// (mergeUsageSnapshot)
|
|
53
|
+
// (mergeUsageSnapshot). Unchanged observations refresh on a bounded interval.
|
|
54
54
|
try {
|
|
55
55
|
const snapshot = normalizeClaudeRateLimits(session.rate_limits);
|
|
56
56
|
if (snapshot) {
|
|
57
57
|
const previous = readUsageSnapshots(root)["claude-cli"];
|
|
58
|
-
const merged = mergeUsageSnapshot(previous, { ...snapshot, observedAt: now }, now);
|
|
58
|
+
const merged = mergeUsageSnapshot(previous, { ...snapshot, sourceId: session.session_id, observedAt: now }, now);
|
|
59
59
|
if (merged !== previous) recordUsageSnapshot(root, "claude-cli", merged);
|
|
60
60
|
}
|
|
61
61
|
} catch { /* usage capture must never break the status line */ }
|
|
@@ -187,6 +187,19 @@ export function probeOutcome({ stdout = "", stderr = "" } = {}) {
|
|
|
187
187
|
return { ok: false, reason: message ? message.replace(/\\"/g, '"').slice(0, 300) : null };
|
|
188
188
|
}
|
|
189
189
|
|
|
190
|
+
/**
|
|
191
|
+
* A probe answer that means the login itself failed, not the model.
|
|
192
|
+
* 401, a missing bearer, no usable profiles, or an unavailable selected
|
|
193
|
+
* profile are sign-in failures. A refusal, rate limit, or timeout is not.
|
|
194
|
+
*/
|
|
195
|
+
export function openclawSignInFailure(text) {
|
|
196
|
+
const s = String(text ?? "");
|
|
197
|
+
return /\b401\b/.test(s)
|
|
198
|
+
|| /missing bearer/i.test(s)
|
|
199
|
+
|| /no usable profiles/i.test(s)
|
|
200
|
+
|| /selected auth profile\b[^]{0,240}?\bis unavailable/i.test(s);
|
|
201
|
+
}
|
|
202
|
+
|
|
190
203
|
/**
|
|
191
204
|
* Muse Code's non-secret login descriptor (~/.config/muse/auth.json) ->
|
|
192
205
|
* { loggedIn, email }. Reads only descriptor fields; the credential itself
|
package/lib/usage-limits.mjs
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { execFile } from "node:child_process";
|
|
2
|
+
import { isDeepStrictEqual, stripVTControlCharacters } from "node:util";
|
|
1
3
|
import fs from "node:fs";
|
|
2
4
|
import path from "node:path";
|
|
3
5
|
import { randomUUID } from "node:crypto";
|
|
@@ -7,10 +9,25 @@ import { randomUUID } from "node:crypto";
|
|
|
7
9
|
// passes them to the status line. Kept per OpenClaw provider id (one host
|
|
8
10
|
// login per provider) in <stateRoot>/usage-limits.json. Admission holds a job
|
|
9
11
|
// on an agent that's over its limit until the General confirms.
|
|
12
|
+
//
|
|
13
|
+
// Those readings only update when a job finishes. A reading taken before a
|
|
14
|
+
// reset, or before the operator bought more usage, would then hold every
|
|
15
|
+
// job, and the reading would never refresh. A reading older than a few
|
|
16
|
+
// minutes is refreshed from `openclaw models status` before that hold, and
|
|
17
|
+
// again when health or the army/capacity views show it. The refresh is
|
|
18
|
+
// async: spawnSync would freeze the server's event loop.
|
|
10
19
|
|
|
11
20
|
const object = value => value !== null && typeof value === "object" && !Array.isArray(value);
|
|
12
21
|
const finite = value => typeof value === "number" && Number.isFinite(value);
|
|
13
22
|
const windowName = minutes => ({ 300: "5h", 10080: "week", 1440: "day" })[minutes] ?? `${minutes}m`;
|
|
23
|
+
const USAGE_SOURCES = ["codex", "claude", "openclaw"];
|
|
24
|
+
// Older than this, an over-limit reading is refreshed before it holds a job.
|
|
25
|
+
export const USAGE_STALE_MINUTES = 10;
|
|
26
|
+
// Source observations age out independently, even when the reset stays the same.
|
|
27
|
+
export const CLAUDE_USAGE_SOURCE_TTL_MS = 30 * 60 * 1000;
|
|
28
|
+
const CLAUDE_USAGE_SOURCE_REFRESH_MS = 5 * 60 * 1000;
|
|
29
|
+
export const USAGE_REFRESH_TIMEOUT_MS = 20000;
|
|
30
|
+
const RESET_UNIT_MS = { d: 86400000, h: 3600000, m: 60000, s: 1000 };
|
|
14
31
|
function window(value, name, minutes, percentKey) {
|
|
15
32
|
if (!object(value) || !finite(value[percentKey]) || value[percentKey] < 0) return null;
|
|
16
33
|
return { name, usedPercent: value[percentKey], windowMinutes: minutes,
|
|
@@ -65,38 +82,59 @@ export function normalizeClaudeRateLimits(rateLimits) {
|
|
|
65
82
|
return windows.length ? { source: "claude", plan: null, limitReached: false, observedAt: Date.now(), windows } : null;
|
|
66
83
|
}
|
|
67
84
|
|
|
85
|
+
/** Maximum of the recent sources for one reset window. */
|
|
86
|
+
function recentUsageWindow(w, observedAt, now, sourceTtlMs) {
|
|
87
|
+
if (w.resetsAt !== null && w.resetsAt <= now) return null;
|
|
88
|
+
const sources = (w.sources ?? [{ usedPercent: w.usedPercent, observedAt }])
|
|
89
|
+
.filter(s => now - s.observedAt <= sourceTtlMs);
|
|
90
|
+
if (!sources.length) return null;
|
|
91
|
+
return { ...w, usedPercent: Math.max(...sources.map(s => s.usedPercent)), sources };
|
|
92
|
+
}
|
|
93
|
+
|
|
68
94
|
/**
|
|
69
|
-
*
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
* kept while it's live. Returns the stored snapshot itself when nothing
|
|
75
|
-
* changed, so callers can skip the write.
|
|
95
|
+
* Keep the latest observation per source and window. At the same reset,
|
|
96
|
+
* recent sources compete by percentage; a later reset replaces all sources
|
|
97
|
+
* for the earlier reset. Legacy snapshots count as one source at observedAt.
|
|
98
|
+
* Anonymous readings each get their own source. Unchanged redraws refresh
|
|
99
|
+
* source timestamps at most once every five minutes, before sources expire.
|
|
76
100
|
*/
|
|
77
|
-
export function mergeUsageSnapshot(previous, incoming, now = Date.now()) {
|
|
78
|
-
|
|
79
|
-
const
|
|
80
|
-
const
|
|
81
|
-
|
|
82
|
-
|
|
101
|
+
export function mergeUsageSnapshot(previous, incoming, now = Date.now(), sourceTtlMs = CLAUDE_USAGE_SOURCE_TTL_MS) {
|
|
102
|
+
const sourceId = typeof incoming.sourceId === "string" && incoming.sourceId ? incoming.sourceId : randomUUID();
|
|
103
|
+
const legacyId = previous?.sourceId ?? randomUUID();
|
|
104
|
+
const byName = new Map((previous?.windows ?? [])
|
|
105
|
+
.filter(w => w.resetsAt === null || w.resetsAt > now)
|
|
106
|
+
.map(w => [w.name, { ...w, sources: w.sources ?? [{ sourceId: legacyId, observedAt: previous.observedAt, usedPercent: w.usedPercent }] }]));
|
|
83
107
|
for (const w of incoming.windows) {
|
|
108
|
+
if (w.resetsAt !== null && w.resetsAt <= now) continue;
|
|
84
109
|
const stored = byName.get(w.name);
|
|
85
|
-
if (
|
|
110
|
+
if (stored && (stored.resetsAt ?? 0) > (w.resetsAt ?? 0)) continue;
|
|
111
|
+
const sources = stored && stored.resetsAt === w.resetsAt ? stored.sources : [];
|
|
112
|
+
const last = sources.find(s => s.sourceId === sourceId);
|
|
113
|
+
if (last && last.observedAt > incoming.observedAt) continue;
|
|
114
|
+
if (last && last.usedPercent === w.usedPercent && stored.windowMinutes === w.windowMinutes
|
|
115
|
+
&& incoming.observedAt - last.observedAt < Math.min(CLAUDE_USAGE_SOURCE_REFRESH_MS, sourceTtlMs / 2)) continue;
|
|
116
|
+
byName.set(w.name, { ...w, sources: [
|
|
117
|
+
...sources.filter(s => s.sourceId !== sourceId),
|
|
118
|
+
{ sourceId, observedAt: incoming.observedAt, usedPercent: w.usedPercent },
|
|
119
|
+
] });
|
|
86
120
|
}
|
|
87
|
-
const
|
|
88
|
-
|
|
89
|
-
|
|
121
|
+
const windows = [...byName.values()].map(w => recentUsageWindow(w, incoming.observedAt, now, sourceTtlMs)).filter(Boolean);
|
|
122
|
+
const { sourceId: ignored, ...snapshot } = incoming;
|
|
123
|
+
const merged = { ...snapshot, limitReached: incoming.limitReached || Boolean(previous?.limitReached), observedAt: now, windows };
|
|
124
|
+
// Keep identity for callers that avoid writes when only the redraw time changed.
|
|
125
|
+
return previous && isDeepStrictEqual({ ...merged, observedAt: previous.observedAt }, previous) ? previous : merged;
|
|
90
126
|
}
|
|
91
127
|
|
|
92
128
|
export function readUsageSnapshots(stateRoot) {
|
|
93
129
|
try {
|
|
94
130
|
const data = JSON.parse(fs.readFileSync(path.join(stateRoot, "usage-limits.json"), "utf8"));
|
|
95
131
|
if (!object(data)) return {};
|
|
96
|
-
return Object.fromEntries(Object.entries(data).filter(([, s]) => object(s) &&
|
|
132
|
+
return Object.fromEntries(Object.entries(data).filter(([, s]) => object(s) && USAGE_SOURCES.includes(s.source)
|
|
97
133
|
&& finite(s.observedAt) && typeof s.limitReached === "boolean" && (s.plan === null || typeof s.plan === "string")
|
|
98
134
|
&& Array.isArray(s.windows) && s.windows.every(w => object(w) && typeof w.name === "string" && finite(w.usedPercent)
|
|
99
|
-
&& (w.windowMinutes === null || finite(w.windowMinutes)) && (w.resetsAt === null || finite(w.resetsAt))
|
|
135
|
+
&& (w.windowMinutes === null || finite(w.windowMinutes)) && (w.resetsAt === null || finite(w.resetsAt))
|
|
136
|
+
&& (w.sources === undefined || (Array.isArray(w.sources) && w.sources.every(s => object(s)
|
|
137
|
+
&& typeof s.sourceId === "string" && finite(s.observedAt) && finite(s.usedPercent) && s.usedPercent >= 0))))));
|
|
100
138
|
} catch { return {}; }
|
|
101
139
|
}
|
|
102
140
|
|
|
@@ -113,8 +151,11 @@ export function recordUsageSnapshot(stateRoot, provider, snapshot) {
|
|
|
113
151
|
} finally { fs.rmSync(tmp, { force: true }); }
|
|
114
152
|
}
|
|
115
153
|
|
|
116
|
-
export function usageStatus(snapshot, now = Date.now()) {
|
|
117
|
-
const
|
|
154
|
+
export function usageStatus(snapshot, now = Date.now(), sourceTtlMs = CLAUDE_USAGE_SOURCE_TTL_MS) {
|
|
155
|
+
const windows = snapshot.source === "claude"
|
|
156
|
+
? snapshot.windows.map(w => recentUsageWindow(w, snapshot.observedAt, now, sourceTtlMs)).filter(Boolean)
|
|
157
|
+
: snapshot.windows;
|
|
158
|
+
const live = windows.filter(w => w.resetsAt === null || w.resetsAt > now).sort((a, b) => b.usedPercent - a.usedPercent);
|
|
118
159
|
const reached = snapshot.limitReached && (snapshot.windows.length === 0 || live.length > 0);
|
|
119
160
|
const highest = live[0];
|
|
120
161
|
const level = reached || highest?.usedPercent >= 100 ? "over" : highest?.usedPercent >= 80 ? "high" : "ok";
|
|
@@ -124,3 +165,165 @@ export function usageStatus(snapshot, now = Date.now()) {
|
|
|
124
165
|
return { level, text, short, resetsAt: level === "over" ? highest?.resetsAt ?? null : null,
|
|
125
166
|
ageMinutes: Math.max(0, Math.floor((now - snapshot.observedAt) / 60000)) };
|
|
126
167
|
}
|
|
168
|
+
|
|
169
|
+
/** An over-limit reading old enough that the window may already have reset. */
|
|
170
|
+
export function usageReadingIsStale(status) {
|
|
171
|
+
return status?.level === "over" && status.ageMinutes >= USAGE_STALE_MINUTES;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** View text. A stale over-limit reading is marked; a fresh one is unchanged. */
|
|
175
|
+
export function usageDisplayText(status) {
|
|
176
|
+
if (!usageReadingIsStale(status)) return status.text;
|
|
177
|
+
return `${status.text}; possibly stale (${status.ageMinutes} minutes old)`;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
export function staleUsageRefreshedNote(provider) {
|
|
181
|
+
return `stale usage reading for ${provider} was refreshed from OpenClaw`;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
function refreshFailureMessage(error) {
|
|
185
|
+
const message = String(error?.message ?? error ?? "");
|
|
186
|
+
return /timed out|timeout/i.test(message) ? "OpenClaw usage refresh failed (timed out)" : "OpenClaw usage refresh failed";
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
function withTimeout(promise, timeoutMs) {
|
|
190
|
+
return new Promise((resolve, reject) => {
|
|
191
|
+
const timer = setTimeout(() => reject(new Error("OpenClaw usage refresh timed out")), timeoutMs);
|
|
192
|
+
Promise.resolve(promise).then(
|
|
193
|
+
(value) => { clearTimeout(timer); resolve(value); },
|
|
194
|
+
(error) => { clearTimeout(timer); reject(error); },
|
|
195
|
+
);
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** Async `openclaw` runner. Never spawnSync: that blocks every other request. */
|
|
200
|
+
export function runOpenClawStatus(cmd, args, { timeoutMs = USAGE_REFRESH_TIMEOUT_MS } = {}) {
|
|
201
|
+
return new Promise((resolve) => {
|
|
202
|
+
execFile(cmd, args, { encoding: "utf8", timeout: timeoutMs, maxBuffer: 4 * 1024 * 1024, windowsHide: true }, (error, stdout) => {
|
|
203
|
+
resolve({ ok: !error, stdout: stdout ?? "", error: error ? String(error.message ?? error) : null });
|
|
204
|
+
});
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
function windowFromLabel(label) {
|
|
209
|
+
const text = String(label ?? "").trim();
|
|
210
|
+
const hours = /^(\d+(?:\.\d+)?)h$/i.exec(text);
|
|
211
|
+
if (hours) {
|
|
212
|
+
const minutes = Math.round(Number(hours[1]) * 60);
|
|
213
|
+
return { name: windowName(minutes), windowMinutes: minutes };
|
|
214
|
+
}
|
|
215
|
+
const days = /^(\d+(?:\.\d+)?)d$/i.exec(text);
|
|
216
|
+
if (days) {
|
|
217
|
+
const minutes = Math.round(Number(days[1]) * 1440);
|
|
218
|
+
return { name: windowName(minutes), windowMinutes: minutes };
|
|
219
|
+
}
|
|
220
|
+
const mins = /^(\d+(?:\.\d+)?)m$/i.exec(text);
|
|
221
|
+
if (mins) {
|
|
222
|
+
const minutes = Math.round(Number(mins[1]));
|
|
223
|
+
return { name: windowName(minutes), windowMinutes: minutes };
|
|
224
|
+
}
|
|
225
|
+
const named = { week: 10080, day: 1440, "5h": 300, spend: null };
|
|
226
|
+
if (Object.hasOwn(named, text)) return { name: text === "week" || text === "day" || text === "5h" ? windowName(named[text]) : text, windowMinutes: named[text] };
|
|
227
|
+
return { name: text || "unknown", windowMinutes: null };
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function resetAtFromDelay(delay, now) {
|
|
231
|
+
if (delay == null || delay === "") return null;
|
|
232
|
+
if (typeof delay === "number" && Number.isFinite(delay)) {
|
|
233
|
+
if (delay > 1e12) return delay;
|
|
234
|
+
if (delay > 1e9) return delay * 1000;
|
|
235
|
+
return null;
|
|
236
|
+
}
|
|
237
|
+
const text = String(delay).trim();
|
|
238
|
+
if (!text) return null;
|
|
239
|
+
if (/^\d{4}-\d{2}-\d{2}/.test(text)) {
|
|
240
|
+
const iso = Date.parse(text);
|
|
241
|
+
return Number.isFinite(iso) ? iso : null;
|
|
242
|
+
}
|
|
243
|
+
let ms = 0, matched = false;
|
|
244
|
+
for (const part of text.matchAll(/(\d+(?:\.\d+)?)\s*([dhms])/gi)) {
|
|
245
|
+
matched = true;
|
|
246
|
+
ms += Number(part[1]) * RESET_UNIT_MS[part[2].toLowerCase()];
|
|
247
|
+
}
|
|
248
|
+
return matched ? now + ms : null;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
function snapshotFromWindows(windows, now) {
|
|
252
|
+
if (!windows.length) return null;
|
|
253
|
+
return { source: "openclaw", plan: null, limitReached: false, observedAt: now, windows };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
function parseUsageRemainder(remainder, now) {
|
|
257
|
+
const windows = [];
|
|
258
|
+
const re = /(\S+)\s+(\d+(?:\.\d+)?)%\s+left(?:\s*⏱\uFE0F?\s*((?:\d+(?:\.\d+)?\s*[dhms]\s*)+))?/gi;
|
|
259
|
+
for (const match of String(remainder).matchAll(re)) {
|
|
260
|
+
const built = windowFromLabel(match[1]);
|
|
261
|
+
const usedPercent = 100 - Number(match[2]);
|
|
262
|
+
if (!Number.isFinite(usedPercent) || usedPercent < 0) continue;
|
|
263
|
+
windows.push({ name: built.name, usedPercent, windowMinutes: built.windowMinutes, resetsAt: resetAtFromDelay(match[3] ?? null, now) });
|
|
264
|
+
}
|
|
265
|
+
return windows;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
function parseTextUsage(raw, now) {
|
|
269
|
+
const out = {};
|
|
270
|
+
const lineRe = /(?:^|[\n"\s\u2500-\u257f])([A-Za-z0-9._-]+)\s+usage:\s+([^"\n]+)/g;
|
|
271
|
+
for (const line of String(raw).matchAll(lineRe)) {
|
|
272
|
+
const snapshot = snapshotFromWindows(parseUsageRemainder(line[2], now), now);
|
|
273
|
+
if (snapshot) out[line[1]] = snapshot;
|
|
274
|
+
}
|
|
275
|
+
return out;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* OpenClaw usage from plain-text `models status`:
|
|
280
|
+
* `- <provider> usage: <window> <N>% left ⏱<duration>`.
|
|
281
|
+
* "N% left" becomes used percent (100 - N). Returns provider -> snapshot.
|
|
282
|
+
* The JSON status output does not include usage readings.
|
|
283
|
+
*/
|
|
284
|
+
export function parseOpenClawUsageOutput(raw, now = Date.now()) {
|
|
285
|
+
return parseTextUsage(stripVTControlCharacters(String(raw ?? "")), now);
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** Ask OpenClaw for current usage without blocking the event loop. */
|
|
289
|
+
export async function fetchOpenClawUsage({ run = runOpenClawStatus, now = Date.now(), timeoutMs = USAGE_REFRESH_TIMEOUT_MS, openclawCmd = process.env.NOMARMY_OPENCLAW_CMD || "openclaw" } = {}) {
|
|
290
|
+
const failure = (error) => ({ ok: false, snapshots: {}, error: refreshFailureMessage(error) });
|
|
291
|
+
try {
|
|
292
|
+
return await withTimeout((async () => {
|
|
293
|
+
const result = await run(openclawCmd, ["models", "status"], { timeoutMs });
|
|
294
|
+
// Partial stdout from a failed command is not a fresh usage reading.
|
|
295
|
+
if (!result?.ok) return failure(result?.error);
|
|
296
|
+
const snapshots = parseOpenClawUsageOutput(result.stdout, now);
|
|
297
|
+
return Object.keys(snapshots).length ? { ok: true, snapshots, error: null } : failure(null);
|
|
298
|
+
})(), timeoutMs);
|
|
299
|
+
} catch (error) { return failure(error); }
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* Refresh over-limit readings older than USAGE_STALE_MINUTES. No OpenClaw
|
|
304
|
+
* call when nothing is stale. A failed refresh leaves the stored readings
|
|
305
|
+
* and says so. `snapshots` lets a caller pass what it already read.
|
|
306
|
+
*/
|
|
307
|
+
export async function refreshStaleOverLimitReadings(stateRoot, { run, now = Date.now(), snapshots = null, timeoutMs = USAGE_REFRESH_TIMEOUT_MS, openclawCmd } = {}) {
|
|
308
|
+
const current = snapshots ?? (stateRoot ? readUsageSnapshots(stateRoot) : {});
|
|
309
|
+
const stale = Object.entries(current).filter(([, snapshot]) => usageReadingIsStale(usageStatus(snapshot, now))).map(([provider]) => provider);
|
|
310
|
+
if (!stale.length) return { called: false, ok: true, snapshots: current, error: null, failedProviders: [] };
|
|
311
|
+
let fetched;
|
|
312
|
+
try { fetched = await fetchOpenClawUsage({ run, now, timeoutMs, openclawCmd }); }
|
|
313
|
+
catch (error) { return { called: true, ok: false, snapshots: current, error: refreshFailureMessage(error), failedProviders: stale }; }
|
|
314
|
+
if (!fetched.ok) return { called: true, ok: false, snapshots: current, error: fetched.error ?? "OpenClaw usage refresh failed", failedProviders: stale };
|
|
315
|
+
const updated = { ...current };
|
|
316
|
+
const seen = new Set();
|
|
317
|
+
for (const provider of stale) {
|
|
318
|
+
const snapshot = fetched.snapshots[provider];
|
|
319
|
+
if (!snapshot) continue;
|
|
320
|
+
try {
|
|
321
|
+
if (stateRoot) recordUsageSnapshot(stateRoot, provider, snapshot);
|
|
322
|
+
// A job may have saved a newer observation while the refresh was pending.
|
|
323
|
+
updated[provider] = stateRoot ? readUsageSnapshots(stateRoot)[provider] ?? snapshot : snapshot;
|
|
324
|
+
seen.add(provider);
|
|
325
|
+
} catch { /* Keep the hold if the fresh snapshot could not be recorded. */ }
|
|
326
|
+
}
|
|
327
|
+
const failedProviders = stale.filter((provider) => !seen.has(provider));
|
|
328
|
+
return { called: true, ok: failedProviders.length === 0, snapshots: updated, error: failedProviders.length ? "OpenClaw usage refresh failed" : null, failedProviders };
|
|
329
|
+
}
|
package/mcp/server.mjs
CHANGED
|
@@ -32,7 +32,7 @@ import { liveLeases } from "../lib/slots.mjs";
|
|
|
32
32
|
import { createRun, loadRun, runTotals, finishRun, resolveRunLimits, describeLoweredLimits } from "../lib/runs.mjs";
|
|
33
33
|
import { agentDispatchFields, resolveAgentModel, agentProviderId, describeAgent } from "../lib/agents.mjs";
|
|
34
34
|
import { OUTCOMES, COORDINATOR_STATUS_BY_OUTCOME } from "../lib/outcomes.mjs";
|
|
35
|
-
import { readUsageSnapshots, usageStatus } from "../lib/usage-limits.mjs";
|
|
35
|
+
import { readUsageSnapshots, usageStatus, usageDisplayText, refreshStaleOverLimitReadings } from "../lib/usage-limits.mjs";
|
|
36
36
|
import { modelRefusals } from "../lib/health.mjs";
|
|
37
37
|
import { retryRefusedModelsInBackground } from "../lib/refusal-retry.mjs";
|
|
38
38
|
import { podmanProblem, podmanVmStartedAt } from "../lib/podman-health.mjs";
|
|
@@ -307,7 +307,7 @@ const { executeJob, executeImplement, executeScout, executeDecompose } = createE
|
|
|
307
307
|
});
|
|
308
308
|
export { executeJob };
|
|
309
309
|
|
|
310
|
-
const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
|
|
310
|
+
const { WORKER_START_STAGGER_MS, activeJobs, runningCount, agentMaxConcurrent, withAgentSlot, track, notifyJobFinished, capacitySnapshot, displayedCapacity, admit, refusal, runBrief, recordJobInRun, trackInRun, launch, liveProgress, summarize } = createJobRuntime({
|
|
311
311
|
projectDir, stateRoot, jobsRoot, runsRoot, leasesRoot, slotsRoot, run, currentMaxWorkers, slug, agentsConfig, modelCatalogReady, budgetsForJob, resolveSubscriptionSelection, executeJob, subscriptionJobFieldProblems, repoPolicy, jobArgs,
|
|
312
312
|
env: process.env, budgetState, getActiveRunId: () => activeRunId,
|
|
313
313
|
sandboxProblem: () => podmanChecks?.problem() ?? null,
|
|
@@ -395,10 +395,11 @@ server.tool("local_worker", "Run one isolated local worker and wait for it. mode
|
|
|
395
395
|
const expanded = expandJobs([rawArgs]);
|
|
396
396
|
if (expanded.problems.length) return refusal(expanded.problems);
|
|
397
397
|
const [args] = expanded.jobs;
|
|
398
|
-
const { problems } = await admit([args]);
|
|
398
|
+
const { problems, admission } = await admit([args]);
|
|
399
399
|
if (problems.length) return refusal(problems);
|
|
400
400
|
const r = await launch(args).promise;
|
|
401
|
-
|
|
401
|
+
const refreshed = (admission.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
|
|
402
|
+
return toolText(refreshed.length ? `${coordinatorResult(r)}\n\n${refreshed.join("\n")}` : coordinatorResult(r), !r.ok);
|
|
402
403
|
});
|
|
403
404
|
server.tool("local_worker_start", "Start one worker or scout in the background and return immediately with a job_id. Poll it with local_worker_status (optionally long-polling with wait_seconds). Same admission rules as local_worker: refuses under memory pressure or when NOMARMY_MAX_WORKERS jobs are already running.", jobSchema.shape,
|
|
404
405
|
async rawArgs => {
|
|
@@ -494,7 +495,7 @@ server.tool("local_worker_stop", "Stop a running job's worker, for example one b
|
|
|
494
495
|
|
|
495
496
|
server.tool("local_worker_capacity", "What this host can take right now: context per nom and the brief/report budgets derived from it, memory pressure and whether another job would be admitted, and the jobs currently running. Read-only.", {}, async () => {
|
|
496
497
|
await budgetState.refresh();
|
|
497
|
-
return toolText(JSON.stringify(withRestartNotice(
|
|
498
|
+
return toolText(JSON.stringify(withRestartNotice(await displayedCapacity()), null, 2));
|
|
498
499
|
});
|
|
499
500
|
// The only way to know what `verification`/`union_verification`/
|
|
500
501
|
// `verify_regression` profile names are actually valid for this repo used to
|
|
@@ -577,8 +578,9 @@ server.tool("run_finish", "Close a /feature run as complete or stopped, with a o
|
|
|
577
578
|
server.tool("army", "Who you, the General, are and who you call for what in this repository: your fixed charter and the agent you're defined as, the army's workflow, then each role's description, phase (build, review, acceptance), suggested mode, and the agent it runs on, with which config layer set each value (global, project .nomarmy.yml, local .nomarmy.local.yml). Flags roles with no usable agent, and roles that share your model or subscription (not an independent review). Dispatch a role with `army_role`, or an agent directly with `agent`. Read-only, re-read on every call.", {}, async () => {
|
|
578
579
|
try {
|
|
579
580
|
const agents = agentsConfig().agents;
|
|
580
|
-
const
|
|
581
|
-
const
|
|
581
|
+
const usageRefresh = await refreshStaleOverLimitReadings(stateRoot);
|
|
582
|
+
const usageSnapshots = usageRefresh.snapshots;
|
|
583
|
+
const summary = describeArmy(currentArmy(), { agents, describeAgent, usageSnapshots, agentProviderId, usageRefreshError: usageRefresh.error, usageRefreshFailed: usageRefresh.failedProviders });
|
|
582
584
|
// Each agent's models, from OpenClaw's catalog, so the General can pick
|
|
583
585
|
// one for a role set to "auto". The catalog can lag a brand-new model.
|
|
584
586
|
const catalog = await modelCatalogReady();
|
|
@@ -592,7 +594,11 @@ server.tool("army", "Who you, the General, are and who you call for what in this
|
|
|
592
594
|
const models = listed.filter((m) => !refusals[`${provider}/${m}`]);
|
|
593
595
|
const refusedModels = listed.filter((m) => refusals[`${provider}/${m}`]);
|
|
594
596
|
const snapshot = usageSnapshots[provider];
|
|
595
|
-
const usage = snapshot ? (() => {
|
|
597
|
+
const usage = snapshot ? (() => {
|
|
598
|
+
const status = usageStatus(snapshot);
|
|
599
|
+
const failed = usageRefresh.failedProviders.includes(provider);
|
|
600
|
+
return { level: status.level, text: `${usageDisplayText(status)}${failed ? `. ${usageRefresh.error}` : ""}` };
|
|
601
|
+
})() : null;
|
|
596
602
|
return [name, { runsOn: describeAgent(agent), defaultModel: agent.model ?? null, models, ...(refusedModels.length ? { refusedModels } : {}), usage }];
|
|
597
603
|
}));
|
|
598
604
|
// A pinned model missing from the catalog isn't necessarily wrong:
|
|
@@ -632,7 +638,7 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
|
|
|
632
638
|
const expanded = expandJobs(rawJobs);
|
|
633
639
|
if (expanded.problems.length) return refusal(expanded.problems);
|
|
634
640
|
const { jobs } = expanded;
|
|
635
|
-
const { problems } = await admit(jobs);
|
|
641
|
+
const { problems, admission } = await admit(jobs);
|
|
636
642
|
let forcedBase = null;
|
|
637
643
|
if (auto_union) {
|
|
638
644
|
const refs = [...new Set(jobs.map(j => j.base_ref).filter(Boolean))];
|
|
@@ -697,7 +703,8 @@ server.tool("local_workers", "Run independent jobs (implement or scout) with bou
|
|
|
697
703
|
jobs: results.map(r => ({ jobId: r.manifest.jobId, workerId: r.manifest.workerId, mode: r.manifest.mode, outcome: r.manifest.outcome || OUTCOMES.WORKER_FAILED, recovered: Boolean(r.manifest.recovered), status: r.manifest.coordinatorStatus || "failed", branch: r.manifest.branch, commit: r.manifest.commit?.sha || null, worktree: r.manifest.worktree, jobDir: r.jobDir })),
|
|
698
704
|
...(union ? { union } : {}) };
|
|
699
705
|
const unionSection = union ? `UNION\n\n${formatUnion(withWindowsPaths(union))}\n\n` : "";
|
|
700
|
-
const
|
|
706
|
+
const refreshed = (admission?.reasons ?? []).filter((line) => line.startsWith("stale usage reading "));
|
|
707
|
+
const text = `BATCH EXECUTION RECORD\n${coordinatorJson(summary)}\n\n${unionSection}WORKER RESULTS\n\n${results.map((r, i) => `===== WORKER ${i + 1} =====\n${coordinatorResult(r)}`).join("\n\n")}${refreshed.length ? `\n\n${refreshed.join("\n")}` : ""}`;
|
|
701
708
|
return toolText(text, results.some(r => !r.ok) || union?.status === "union_verification_failed" || union?.status === "union_error");
|
|
702
709
|
});
|
|
703
710
|
// No model, no sandbox, no tokens spent on a worker: the coordinator asks the
|
package/package.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"description": "Every byte verified: a harness for AI coding workers whose claims are never trusted. Your coding assistant stays in charge while workers implement and test in sandboxes, and nomArmy checks every change before it is committed.",
|
|
4
4
|
"author": "Rayson Technologies",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"version": "0.1.0-alpha.
|
|
6
|
+
"version": "0.1.0-alpha.21",
|
|
7
7
|
"private": false,
|
|
8
8
|
"type": "module",
|
|
9
9
|
"engines": {
|