nomarmy 0.1.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +25 -0
- package/README.md +484 -0
- package/bin/nomarmy.mjs +2248 -0
- package/config/agents.yml.example +63 -0
- package/config/common.env +31 -0
- package/config/profiles/bedrock-cheap.env +26 -0
- package/config/profiles/bedrock.env +28 -0
- package/config/profiles/cpu-linux.env +8 -0
- package/config/profiles/dgx-spark.env +12 -0
- package/config/profiles/macbook-pro.env +9 -0
- package/config/profiles/nvidia-linux.env +9 -0
- package/docker/Dockerfile +15 -0
- package/docker/Dockerfile.go +29 -0
- package/docker/Dockerfile.rust +19 -0
- package/e2e.sh +153 -0
- package/install.sh +125 -0
- package/lib/agents.mjs +285 -0
- package/lib/army.mjs +400 -0
- package/lib/budget.mjs +368 -0
- package/lib/claude-transcript.mjs +150 -0
- package/lib/config.mjs +193 -0
- package/lib/connect.mjs +409 -0
- package/lib/coordinator-instructions.mjs +23 -0
- package/lib/decompose.mjs +389 -0
- package/lib/dispatch-config.mjs +164 -0
- package/lib/dispatch-schema.mjs +280 -0
- package/lib/doctor.mjs +443 -0
- package/lib/evidence.mjs +679 -0
- package/lib/gguf.mjs +589 -0
- package/lib/hardware.mjs +476 -0
- package/lib/health.mjs +278 -0
- package/lib/model-catalog.mjs +71 -0
- package/lib/notifier-app.mjs +95 -0
- package/lib/notify.mjs +66 -0
- package/lib/openclaw-config.mjs +65 -0
- package/lib/openclaw-errors.mjs +40 -0
- package/lib/propose.mjs +110 -0
- package/lib/prune.mjs +77 -0
- package/lib/repo-query.mjs +267 -0
- package/lib/runs.mjs +150 -0
- package/lib/sabotage.mjs +128 -0
- package/lib/sandbox-images.mjs +434 -0
- package/lib/scan.mjs +1538 -0
- package/lib/schema.mjs +288 -0
- package/lib/scout.mjs +544 -0
- package/lib/sizing.mjs +1322 -0
- package/lib/slots.mjs +112 -0
- package/lib/statusline.mjs +126 -0
- package/lib/subscription-config.mjs +68 -0
- package/lib/subscription-setup.mjs +217 -0
- package/lib/transcript.mjs +195 -0
- package/lib/verify.mjs +700 -0
- package/mcp/server.mjs +4206 -0
- package/notifier/icon.swift +34 -0
- package/notifier/main.swift +52 -0
- package/notifier/nomarmy-icon.png +0 -0
- package/package.json +67 -0
- package/playbooks/feature.md +43 -0
- package/policies/coder.md +49 -0
- package/policies/orchestrator.md +35 -0
- package/policies/reviewer.md +35 -0
- package/policies/scout.md +65 -0
- package/scripts/configure-openclaw.sh +96 -0
- package/scripts/configure-orchestrator.sh +84 -0
- package/scripts/install-llama-cpp.sh +16 -0
- package/scripts/lib.sh +198 -0
- package/scripts/select-model.mjs +96 -0
- package/scripts/select-model.sh +4 -0
- package/scripts/setup-sandbox.sh +38 -0
- package/scripts/start-inference.sh +46 -0
- package/scripts/stop-inference.sh +5 -0
- package/scripts/uninstall.sh +6 -0
- package/scripts/verify-install.sh +68 -0
package/lib/budget.mjs
ADDED
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
// Worker budgets derived from the context a nom actually has, and an
|
|
2
|
+
// admission check derived from the memory this host actually has free.
|
|
3
|
+
//
|
|
4
|
+
// Two different resources, two different levers, deliberately not conflated:
|
|
5
|
+
//
|
|
6
|
+
// * Context per nom bounds TEXT: how long a brief may be, how long a report
|
|
7
|
+
// may be, how many scout findings and excerpt lines are worth carrying.
|
|
8
|
+
// llama-server allocates the whole KV cache for every slot at startup, so
|
|
9
|
+
// a shorter brief does not save memory -- it saves the nom's attention.
|
|
10
|
+
//
|
|
11
|
+
// * Free memory bounds ADMISSION: whether starting one more job (which adds
|
|
12
|
+
// a Podman sandbox and an OpenClaw process, never a second copy of the
|
|
13
|
+
// model) is safe right now. Under pressure nomArmy refuses to start; it
|
|
14
|
+
// does not shrink the brief and hope.
|
|
15
|
+
//
|
|
16
|
+
// Everything here is pure except `resolveContextPerNom`, which may ask a
|
|
17
|
+
// running llama-server what its slots actually are; the probe is injectable.
|
|
18
|
+
import { DEFAULT_TARGET_CONTEXT_PER_NOM, MIN_CONTEXT_PER_NOM, RESERVES, GIB, formatBytes, isCloudExecution } from "./sizing.mjs";
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Hard ceilings the MCP tool schema enforces regardless of hardware. The
|
|
22
|
+
* derived budget may be lower than these, never higher, unless an explicit
|
|
23
|
+
* environment override says so. 3000 characters is the calibrated value: a
|
|
24
|
+
* ten-file, ~3.5k-character brief produced zero edits before a small model ran
|
|
25
|
+
* out of output budget.
|
|
26
|
+
*/
|
|
27
|
+
export const CALIBRATED = Object.freeze({
|
|
28
|
+
taskChars: 3000,
|
|
29
|
+
acceptanceItemChars: 300,
|
|
30
|
+
charsPerToken: 4,
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Ceilings for a frontier worker (any api or subscription agent). The
|
|
35
|
+
* CALIBRATED numbers above were measured on a ~20B local model, where a long
|
|
36
|
+
* brief made it thrash; applied to a 272k-context frontier model they only
|
|
37
|
+
* starve it of the spec the coordinator already has. These are estimates,
|
|
38
|
+
* not measurements yet -- revisit them against real frontier jobs.
|
|
39
|
+
*/
|
|
40
|
+
export const FRONTIER = Object.freeze({
|
|
41
|
+
taskChars: 16000,
|
|
42
|
+
acceptanceItemChars: 600,
|
|
43
|
+
evidenceChars: 24000,
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
/** Evidence ceiling for the local tier (NOMARMY_MAX_EVIDENCE_CHARS overrides it). */
|
|
47
|
+
export const LOCAL_EVIDENCE_CHARS = 6000;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Report ceilings in tokens by tier and requested size. A report lands in
|
|
51
|
+
* the coordinator's own context and is re-read on every later turn, so the
|
|
52
|
+
* coordinator picks the size per job (`report`), capped by the worker's
|
|
53
|
+
* tier. The local tier never grows past its calibrated caps.
|
|
54
|
+
*/
|
|
55
|
+
export const REPORT_CAPS = Object.freeze({
|
|
56
|
+
local: Object.freeze({
|
|
57
|
+
brief: { implement: 512, scout: 1536, decompose: 2048 },
|
|
58
|
+
standard: { implement: 512, scout: 1536, decompose: 2048 },
|
|
59
|
+
full: { implement: 512, scout: 1536, decompose: 2048 },
|
|
60
|
+
}),
|
|
61
|
+
frontier: Object.freeze({
|
|
62
|
+
brief: { implement: 512, scout: 1536, decompose: 2048 },
|
|
63
|
+
standard: { implement: 1024, scout: 2048, decompose: 2048 },
|
|
64
|
+
full: { implement: 2048, scout: 4096, decompose: 4096 },
|
|
65
|
+
}),
|
|
66
|
+
});
|
|
67
|
+
export const REPORT_SIZES = Object.freeze(["brief", "standard", "full"]);
|
|
68
|
+
|
|
69
|
+
/** The rules of thumb, in one place so a reviewer can argue with them. */
|
|
70
|
+
export const BUDGET_RULES = Object.freeze({
|
|
71
|
+
/** Share of the nom's context a brief may occupy, then clamped. */
|
|
72
|
+
briefFraction: 0.05,
|
|
73
|
+
briefTokensMin: 300,
|
|
74
|
+
briefTokensMax: CALIBRATED.taskChars / CALIBRATED.charsPerToken, // 750 -> 3000 chars
|
|
75
|
+
/** Implement report: four lines. The cap only shrinks on tiny contexts. */
|
|
76
|
+
implementReportFraction: 0.02,
|
|
77
|
+
implementReportTargetTokens: 256,
|
|
78
|
+
implementReportCapMin: 320,
|
|
79
|
+
implementReportCapMax: 512,
|
|
80
|
+
/** Scout report: one line per finding plus citations; needs more room. */
|
|
81
|
+
scoutReportFraction: 0.04,
|
|
82
|
+
scoutReportCapMin: 384,
|
|
83
|
+
scoutReportCapMax: 1536,
|
|
84
|
+
/** Scout structure. Excerpt lines are frontier context, so they stay modest. */
|
|
85
|
+
scoutFindingsMin: 4,
|
|
86
|
+
scoutFindingsMax: 12,
|
|
87
|
+
scoutCitationsPerFinding: 4,
|
|
88
|
+
scoutExcerptLinesPerCitation: 12,
|
|
89
|
+
scoutExcerptLinesTotalMin: 60,
|
|
90
|
+
scoutExcerptLinesTotalMax: 160,
|
|
91
|
+
scoutFindingChars: 300,
|
|
92
|
+
/** Decompose report: one subtask block per finding; heavier than scout. */
|
|
93
|
+
decomposeReportFraction: 0.06,
|
|
94
|
+
decomposeReportCapMin: 512,
|
|
95
|
+
decomposeReportCapMax: 2048,
|
|
96
|
+
decomposeSubtasksMin: 2,
|
|
97
|
+
decomposeSubtasksMax: 6,
|
|
98
|
+
decomposeAcceptancePerSubtask: 3,
|
|
99
|
+
decomposeFilesPerSubtask: 4,
|
|
100
|
+
decomposeSubtaskChars: 200,
|
|
101
|
+
/** Admission: what one more job needs free, and the floor below it. */
|
|
102
|
+
admissionFloorBytes: 1 * GIB,
|
|
103
|
+
tightFraction: 0.15,
|
|
104
|
+
/**
|
|
105
|
+
* Wall-clock split of a job's requested timeout: a work phase and a
|
|
106
|
+
* reserved report phase. A worker that spends its entire budget "working"
|
|
107
|
+
* has nothing left to emit even the four-line report if OpenClaw's own
|
|
108
|
+
* per-turn output budget runs out mid-reply -- observed directly: a run
|
|
109
|
+
* returned ok:true with a reply cut off mid-sentence, rescued only by a
|
|
110
|
+
* follow-up call squeezed into whatever time happened to be left.
|
|
111
|
+
* Reserving a slice up front guarantees that follow-up has real room
|
|
112
|
+
* instead of fighting the same deadline the work phase already used up.
|
|
113
|
+
*/
|
|
114
|
+
reportReserveFraction: 0.12,
|
|
115
|
+
reportReserveMin: 30,
|
|
116
|
+
reportReserveMax: 180,
|
|
117
|
+
workTimeoutMin: 20,
|
|
118
|
+
/**
|
|
119
|
+
* Idle-diff circuit breaker: end the work phase early once the worktree
|
|
120
|
+
* stops changing, instead of running to the deadline after the work is
|
|
121
|
+
* already done. Never fires before idleMinElapsedFraction of the work
|
|
122
|
+
* phase has passed, and never before any change has been observed at all
|
|
123
|
+
* (a job that has made no edits yet is not "idle", it just hasn't started).
|
|
124
|
+
*/
|
|
125
|
+
idleBreakFraction: 0.35,
|
|
126
|
+
idleBreakMin: 60,
|
|
127
|
+
idleBreakMax: 300,
|
|
128
|
+
idleMinElapsedFraction: 0.15,
|
|
129
|
+
idleMinElapsedMin: 30,
|
|
130
|
+
idleMinElapsedMax: 120,
|
|
131
|
+
idlePollSeconds: 15,
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
function clamp(n, lo, hi) { return Math.max(lo, Math.min(hi, n)); }
|
|
135
|
+
function envInt(env, name) {
|
|
136
|
+
const n = Number.parseInt(env?.[name] ?? "", 10);
|
|
137
|
+
return Number.isFinite(n) && n > 0 ? n : null;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Derive text budgets from the context one nom has.
|
|
142
|
+
*
|
|
143
|
+
* @param {{ contextPerNom?: number|null, source?: string, env?: object }} input
|
|
144
|
+
* @returns {object} budgets, with every applied override named
|
|
145
|
+
*/
|
|
146
|
+
/**
|
|
147
|
+
* `tier`: "local" (the local model; every calibrated number, and the
|
|
148
|
+
* NOMARMY_MAX_* env overrides, apply) or "frontier" (an api or
|
|
149
|
+
* subscription agent; FRONTIER and REPORT_CAPS.frontier ceilings).
|
|
150
|
+
* `reportSize`: the coordinator's per-job choice of how much comes back,
|
|
151
|
+
* capped by the tier ("standard" when omitted).
|
|
152
|
+
*/
|
|
153
|
+
export function deriveBudgets({ contextPerNom = null, source = "default", env = process.env, tier = "local", reportSize = "standard" } = {}) {
|
|
154
|
+
const ctx = Number.isFinite(contextPerNom) && contextPerNom > 0 ? Math.floor(contextPerNom) : DEFAULT_TARGET_CONTEXT_PER_NOM;
|
|
155
|
+
const effectiveSource = Number.isFinite(contextPerNom) && contextPerNom > 0 ? source : "default";
|
|
156
|
+
const R = BUDGET_RULES, cpt = CALIBRATED.charsPerToken;
|
|
157
|
+
const frontier = tier === "frontier";
|
|
158
|
+
const caps = REPORT_CAPS[frontier ? "frontier" : "local"][REPORT_SIZES.includes(reportSize) ? reportSize : "standard"];
|
|
159
|
+
const overrides = [];
|
|
160
|
+
|
|
161
|
+
const briefTokens = clamp(Math.floor(ctx * R.briefFraction), R.briefTokensMin, (frontier ? FRONTIER.taskChars : CALIBRATED.taskChars) / cpt);
|
|
162
|
+
let maxTaskChars = briefTokens * cpt;
|
|
163
|
+
let maxAcceptanceItemChars = Math.min(frontier ? FRONTIER.acceptanceItemChars : CALIBRATED.acceptanceItemChars, Math.max(120, Math.floor(maxTaskChars / 10)));
|
|
164
|
+
let maxEvidenceChars = frontier ? FRONTIER.evidenceChars : LOCAL_EVIDENCE_CHARS;
|
|
165
|
+
// The env overrides tune the local model, where the calibration came
|
|
166
|
+
// from; a frontier worker's ceilings don't follow them.
|
|
167
|
+
if (!frontier) {
|
|
168
|
+
const taskOverride = envInt(env, "NOMARMY_MAX_TASK_CHARS");
|
|
169
|
+
if (taskOverride) { maxTaskChars = taskOverride; overrides.push("NOMARMY_MAX_TASK_CHARS"); }
|
|
170
|
+
const itemOverride = envInt(env, "NOMARMY_MAX_ACCEPTANCE_ITEM_CHARS");
|
|
171
|
+
if (itemOverride) { maxAcceptanceItemChars = itemOverride; overrides.push("NOMARMY_MAX_ACCEPTANCE_ITEM_CHARS"); }
|
|
172
|
+
const evidenceOverride = envInt(env, "NOMARMY_MAX_EVIDENCE_CHARS");
|
|
173
|
+
if (evidenceOverride) { maxEvidenceChars = evidenceOverride; overrides.push("NOMARMY_MAX_EVIDENCE_CHARS"); }
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const implementCap = clamp(Math.floor(ctx * R.implementReportFraction), R.implementReportCapMin, caps.implement);
|
|
177
|
+
const implementTarget = Math.min(frontier ? Math.floor(caps.implement / 2) : R.implementReportTargetTokens, Math.floor(implementCap / 2) + 64);
|
|
178
|
+
const scoutCap = clamp(Math.floor(ctx * R.scoutReportFraction), R.scoutReportCapMin, caps.scout);
|
|
179
|
+
const scoutTarget = Math.floor(scoutCap * 0.6);
|
|
180
|
+
|
|
181
|
+
// Findings scale with the report cap: one finding with citations is roughly
|
|
182
|
+
// 60 tokens, and the header/footer lines take about 40.
|
|
183
|
+
const maxFindings = clamp(Math.floor((scoutCap - 40) / 60), R.scoutFindingsMin, frontier ? 24 : R.scoutFindingsMax);
|
|
184
|
+
const decomposeCap = clamp(Math.floor(ctx * R.decomposeReportFraction), R.decomposeReportCapMin, caps.decompose);
|
|
185
|
+
const decomposeTarget = Math.floor(decomposeCap * 0.6);
|
|
186
|
+
const maxSubtasks = clamp(Math.floor((decomposeCap - 40) / 150), R.decomposeSubtasksMin, R.decomposeSubtasksMax);
|
|
187
|
+
let maxExcerptLinesTotal = clamp(Math.floor(ctx / 512), R.scoutExcerptLinesTotalMin, frontier ? 320 : R.scoutExcerptLinesTotalMax);
|
|
188
|
+
const excerptOverride = envInt(env, "NOMARMY_SCOUT_MAX_EXCERPT_LINES");
|
|
189
|
+
if (excerptOverride && !frontier) { maxExcerptLinesTotal = excerptOverride; overrides.push("NOMARMY_SCOUT_MAX_EXCERPT_LINES"); }
|
|
190
|
+
|
|
191
|
+
return {
|
|
192
|
+
contextPerNom: ctx,
|
|
193
|
+
source: effectiveSource,
|
|
194
|
+
tier: frontier ? "frontier" : "local",
|
|
195
|
+
reportSize: REPORT_SIZES.includes(reportSize) ? reportSize : "standard",
|
|
196
|
+
brief: { maxTaskChars, maxAcceptanceItemChars, maxAcceptanceItems: 20, maxEvidenceChars },
|
|
197
|
+
report: {
|
|
198
|
+
implement: { targetTokens: implementTarget, hardCapTokens: implementCap },
|
|
199
|
+
scout: { targetTokens: scoutTarget, hardCapTokens: scoutCap },
|
|
200
|
+
decompose: { targetTokens: decomposeTarget, hardCapTokens: decomposeCap },
|
|
201
|
+
},
|
|
202
|
+
scout: {
|
|
203
|
+
maxFindings,
|
|
204
|
+
maxCitationsPerFinding: R.scoutCitationsPerFinding,
|
|
205
|
+
maxExcerptLinesPerCitation: R.scoutExcerptLinesPerCitation,
|
|
206
|
+
maxExcerptLinesTotal,
|
|
207
|
+
maxFindingChars: R.scoutFindingChars,
|
|
208
|
+
},
|
|
209
|
+
decompose: {
|
|
210
|
+
maxSubtasks,
|
|
211
|
+
maxAcceptancePerSubtask: R.decomposeAcceptancePerSubtask,
|
|
212
|
+
maxFilesPerSubtask: R.decomposeFilesPerSubtask,
|
|
213
|
+
maxSubtaskChars: R.decomposeSubtaskChars,
|
|
214
|
+
},
|
|
215
|
+
overrides,
|
|
216
|
+
tooSmall: ctx < MIN_CONTEXT_PER_NOM,
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Split a job's requested wall-clock timeout into a work phase and a
|
|
222
|
+
* reserved report phase, plus the idle-diff circuit-breaker thresholds
|
|
223
|
+
* derived from the work phase. Pure and deterministic; env overrides use the
|
|
224
|
+
* same envInt() convention as everything else here.
|
|
225
|
+
*
|
|
226
|
+
* @param {{ timeoutSeconds: number, env?: object }} input
|
|
227
|
+
*/
|
|
228
|
+
export function deriveTimeBudget({ timeoutSeconds, env = process.env } = {}) {
|
|
229
|
+
const R = BUDGET_RULES;
|
|
230
|
+
const total = Math.max(30, Math.floor(Number(timeoutSeconds) || 600));
|
|
231
|
+
|
|
232
|
+
let reportReserveSeconds = clamp(Math.round(total * R.reportReserveFraction), R.reportReserveMin, R.reportReserveMax);
|
|
233
|
+
const reserveOverride = envInt(env, "NOMARMY_REPORT_RESERVE_SECONDS");
|
|
234
|
+
if (reserveOverride) reportReserveSeconds = reserveOverride;
|
|
235
|
+
const workTimeoutSeconds = Math.max(R.workTimeoutMin, total - reportReserveSeconds);
|
|
236
|
+
reportReserveSeconds = total - workTimeoutSeconds; // recomputed so work + reserve always equals what the caller asked for
|
|
237
|
+
|
|
238
|
+
let idleBreakSeconds = Math.min(clamp(Math.round(workTimeoutSeconds * R.idleBreakFraction), R.idleBreakMin, R.idleBreakMax), workTimeoutSeconds);
|
|
239
|
+
const idleOverride = envInt(env, "NOMARMY_IDLE_BREAK_SECONDS");
|
|
240
|
+
if (idleOverride) idleBreakSeconds = Math.min(idleOverride, workTimeoutSeconds);
|
|
241
|
+
const idleMinElapsedSeconds = Math.min(clamp(Math.round(workTimeoutSeconds * R.idleMinElapsedFraction), R.idleMinElapsedMin, R.idleMinElapsedMax), workTimeoutSeconds);
|
|
242
|
+
|
|
243
|
+
return { timeoutSeconds: total, workTimeoutSeconds, reportReserveSeconds, idleBreakSeconds, idleMinElapsedSeconds, idlePollSeconds: R.idlePollSeconds };
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* Check a brief against a derived budget. Returns a list of problems; empty
|
|
248
|
+
* means admissible. This is the soft, hardware-aware check; the tool schema's
|
|
249
|
+
* ceiling is the hard one.
|
|
250
|
+
*/
|
|
251
|
+
export function checkBrief({ task, acceptance = [], evidence = null }, budgets) {
|
|
252
|
+
const problems = [];
|
|
253
|
+
const b = budgets.brief;
|
|
254
|
+
const who = budgets.tier === "frontier" ? "this agent's model" : "this nom";
|
|
255
|
+
const taskLen = String(task ?? "").length;
|
|
256
|
+
if (taskLen > b.maxTaskChars) {
|
|
257
|
+
problems.push(`objective is ${taskLen} characters; ${who}'s ${budgets.contextPerNom}-token context (${budgets.source}) allows ${b.maxTaskChars}. Split the job, or reference paths and line ranges instead of pasting content.`);
|
|
258
|
+
}
|
|
259
|
+
const evidenceLen = String(evidence ?? "").length;
|
|
260
|
+
if (b.maxEvidenceChars && evidenceLen > b.maxEvidenceChars) {
|
|
261
|
+
problems.push(`evidence is ${evidenceLen} characters; ${who} allows ${b.maxEvidenceChars}. Hand over less, or a path and line range instead of the material.`);
|
|
262
|
+
}
|
|
263
|
+
(acceptance ?? []).forEach((item, i) => {
|
|
264
|
+
const len = String(item ?? "").length;
|
|
265
|
+
if (len > b.maxAcceptanceItemChars) problems.push(`acceptance item ${i + 1} is ${len} characters; the budget allows ${b.maxAcceptanceItemChars}.`);
|
|
266
|
+
});
|
|
267
|
+
if ((acceptance ?? []).length > b.maxAcceptanceItems) problems.push(`${acceptance.length} acceptance items; the budget allows ${b.maxAcceptanceItems}.`);
|
|
268
|
+
return problems;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/**
|
|
272
|
+
* Parse a llama-server `/props` body. `default_generation_settings.n_ctx` is
|
|
273
|
+
* the PER-SLOT context (llama-server divides `-c` across `-np`), and
|
|
274
|
+
* `total_slots` is the slot count. Anything else is ignored.
|
|
275
|
+
*/
|
|
276
|
+
export function parseLlamaProps(body) {
|
|
277
|
+
let obj = body;
|
|
278
|
+
if (typeof body === "string") { try { obj = JSON.parse(body); } catch { return null; } }
|
|
279
|
+
if (!obj || typeof obj !== "object") return null;
|
|
280
|
+
const perSlot = Number(obj.default_generation_settings?.n_ctx);
|
|
281
|
+
const slots = Number(obj.total_slots);
|
|
282
|
+
if (!Number.isFinite(perSlot) || perSlot <= 0) return null;
|
|
283
|
+
return {
|
|
284
|
+
contextPerNom: Math.floor(perSlot),
|
|
285
|
+
slots: Number.isFinite(slots) && slots > 0 ? Math.floor(slots) : null,
|
|
286
|
+
modelPath: typeof obj.model_path === "string" ? obj.model_path : null,
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
async function defaultProbe(env) {
|
|
291
|
+
const host = env.NOMARMY_LLAMA_HOST || "127.0.0.1", port = env.NOMARMY_LLAMA_PORT || "8080";
|
|
292
|
+
const controller = new AbortController();
|
|
293
|
+
const timer = setTimeout(() => controller.abort(), 2000);
|
|
294
|
+
try {
|
|
295
|
+
const res = await fetch(`http://${host}:${port}/props`, { signal: controller.signal });
|
|
296
|
+
if (!res.ok) return null;
|
|
297
|
+
return parseLlamaProps(await res.text());
|
|
298
|
+
} catch { return null; } finally { clearTimeout(timer); }
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/**
|
|
302
|
+
* Work out how much context one nom has, most authoritative source first:
|
|
303
|
+
* 1. NOMARMY_CONTEXT_PER_NOM (lib.sh exports it when a profile is loaded)
|
|
304
|
+
* 2. NOMARMY_LLAMA_CONTEXT / NOMARMY_LLAMA_PARALLEL
|
|
305
|
+
* 3. the running llama-server's own /props (local execution only)
|
|
306
|
+
* 4. NOMARMY_CONTEXT_LIMIT, the older per-worker hint
|
|
307
|
+
* 5. the v1.3 design target, labeled as an assumption
|
|
308
|
+
*/
|
|
309
|
+
export async function resolveContextPerNom({ env = process.env, probe = defaultProbe } = {}) {
|
|
310
|
+
const direct = envInt(env, "NOMARMY_CONTEXT_PER_NOM");
|
|
311
|
+
if (direct) return { contextPerNom: direct, slots: envInt(env, "NOMARMY_LLAMA_PARALLEL"), source: "profile (NOMARMY_CONTEXT_PER_NOM)" };
|
|
312
|
+
const total = envInt(env, "NOMARMY_LLAMA_CONTEXT"), parallel = envInt(env, "NOMARMY_LLAMA_PARALLEL");
|
|
313
|
+
if (total && parallel) return { contextPerNom: Math.floor(total / parallel), slots: parallel, source: "profile (NOMARMY_LLAMA_CONTEXT / NOMARMY_LLAMA_PARALLEL)" };
|
|
314
|
+
if (!isCloudExecution(env.NOMARMY_EXECUTION || "local")) {
|
|
315
|
+
const probed = await probe(env);
|
|
316
|
+
if (probed) return { contextPerNom: probed.contextPerNom, slots: probed.slots, source: "llama-server /props", modelPath: probed.modelPath };
|
|
317
|
+
}
|
|
318
|
+
const hint = envInt(env, "NOMARMY_CONTEXT_LIMIT") ?? envInt(env, "NOMARMY_WORKER_CONTEXT_LIMIT");
|
|
319
|
+
if (hint) return { contextPerNom: hint, slots: null, source: "NOMARMY_CONTEXT_LIMIT" };
|
|
320
|
+
return { contextPerNom: DEFAULT_TARGET_CONTEXT_PER_NOM, slots: null, source: "assumed v1.3 target; no profile loaded and no llama-server answered" };
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* Decide whether one more job may start, from the memory free right now.
|
|
325
|
+
* A new job costs a sandbox container plus an OpenClaw process, not a second
|
|
326
|
+
* model, so the need is the per-nom sandbox reserve above a floor.
|
|
327
|
+
*
|
|
328
|
+
* @param {{ hardware: object|null, runningJobs?: number, slots?: number|null, maxWorkers?: number }} input
|
|
329
|
+
*/
|
|
330
|
+
export function assessAdmission({ hardware, runningJobs = 0, slots = null, maxWorkers = 1 } = {}) {
|
|
331
|
+
const total = Number(hardware?.memory?.totalBytes) || null;
|
|
332
|
+
const available = Number(hardware?.memory?.availableBytes) || null;
|
|
333
|
+
const need = RESERVES.sandboxPerNomBytes;
|
|
334
|
+
const out = { admit: true, level: "ok", reasons: [], totalBytes: total, availableBytes: available, needBytes: need, runningJobs, maxWorkers, slots };
|
|
335
|
+
|
|
336
|
+
if (runningJobs >= maxWorkers) {
|
|
337
|
+
out.admit = false; out.level = "capacity";
|
|
338
|
+
out.reasons.push(`${runningJobs} job(s) already running; NOMARMY_MAX_WORKERS is ${maxWorkers}. Wait for one to finish or poll with local_worker_status.`);
|
|
339
|
+
}
|
|
340
|
+
if (slots && runningJobs >= slots) {
|
|
341
|
+
out.admit = false; out.level = "capacity";
|
|
342
|
+
out.reasons.push(`${runningJobs} job(s) already running against ${slots} llama-server slot(s); another would queue inside the server and time out.`);
|
|
343
|
+
}
|
|
344
|
+
if (available === null) {
|
|
345
|
+
out.reasons.push("free memory could not be read; admitting on capacity alone");
|
|
346
|
+
return out;
|
|
347
|
+
}
|
|
348
|
+
if (available < need + BUDGET_RULES.admissionFloorBytes) {
|
|
349
|
+
out.admit = false; out.level = "critical";
|
|
350
|
+
out.reasons.push(`only ${formatBytes(available)} free; one more sandbox needs about ${formatBytes(need)} plus a ${formatBytes(BUDGET_RULES.admissionFloorBytes)} floor. Free memory before starting another nom.`);
|
|
351
|
+
} else if (total && available < total * BUDGET_RULES.tightFraction) {
|
|
352
|
+
out.level = out.level === "ok" ? "tight" : out.level;
|
|
353
|
+
out.reasons.push(`${formatBytes(available)} of ${formatBytes(total)} free; this job fits but the next may not.`);
|
|
354
|
+
}
|
|
355
|
+
return out;
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/** Human-readable budget summary for the CLI and the capacity tool. */
|
|
359
|
+
export function describeBudgets(b) {
|
|
360
|
+
return [
|
|
361
|
+
`${b.tier === "frontier" ? "frontier agent" : "local model"}: ${b.contextPerNom}-token context (${b.source})${b.tier === "frontier" ? `, ${b.reportSize ?? "standard"} report` : ""}`,
|
|
362
|
+
`brief: objective <= ${b.brief.maxTaskChars} chars, acceptance items <= ${b.brief.maxAcceptanceItemChars} chars`,
|
|
363
|
+
`implement report: ~${b.report.implement.targetTokens} tokens, cap ${b.report.implement.hardCapTokens}`,
|
|
364
|
+
`scout report: ~${b.report.scout.targetTokens} tokens, cap ${b.report.scout.hardCapTokens}; <= ${b.scout.maxFindings} findings, <= ${b.scout.maxExcerptLinesTotal} excerpt lines returned`,
|
|
365
|
+
...(b.overrides.length ? [`overridden by: ${b.overrides.join(", ")}`] : []),
|
|
366
|
+
...(b.tooSmall ? [`WARNING: ${b.contextPerNom} tokens is below the ${MIN_CONTEXT_PER_NOM}-token floor; expect truncated reports`] : []),
|
|
367
|
+
];
|
|
368
|
+
}
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
// What a Claude CLI worker actually did, from Claude Code's own transcript.
|
|
2
|
+
//
|
|
3
|
+
// OpenClaw's claude-cli provider runs `claude -p` in the job's worktree, and
|
|
4
|
+
// Claude Code runs its own tools there, so OpenClaw's transcript never sees
|
|
5
|
+
// them (a real Senti scout: 44 tool calls, none in OpenClaw's record, so
|
|
6
|
+
// nomArmy could only check final citations, not what was read). Claude Code
|
|
7
|
+
// does record them itself, as one JSONL file per session under
|
|
8
|
+
// ~/.claude/projects/<the working directory, every non-alphanumeric
|
|
9
|
+
// character turned into "-">/ (confirmed from real job transcripts). This
|
|
10
|
+
// reads that file into the same summary shape lib/transcript.mjs produces
|
|
11
|
+
// from OpenClaw's, so read tracking and the heartbeat work unchanged.
|
|
12
|
+
|
|
13
|
+
import fs from "node:fs";
|
|
14
|
+
import os from "node:os";
|
|
15
|
+
import path from "node:path";
|
|
16
|
+
|
|
17
|
+
// Claude Code's tools that bring repository content into the worker's context.
|
|
18
|
+
const CLAUDE_REPO_READ_TOOLS = ["read", "grep", "glob", "bash", "ls", "notebookread"];
|
|
19
|
+
|
|
20
|
+
/** Claude Code's per-project transcript directory for a working directory. */
|
|
21
|
+
export function claudeProjectDir(cwd, { home = os.homedir() } = {}) {
|
|
22
|
+
return path.join(home, ".claude", "projects", String(cwd).replace(/[^A-Za-z0-9]/g, "-"));
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function textOf(content) {
|
|
26
|
+
if (typeof content === "string") return content;
|
|
27
|
+
if (Array.isArray(content)) return content.map((c) => (typeof c === "string" ? c : typeof c?.text === "string" ? c.text : "")).join("\n");
|
|
28
|
+
return "";
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** Summarize Claude Code transcript events (parsed JSONL lines). */
|
|
32
|
+
export function summarizeClaudeEvents(events) {
|
|
33
|
+
const out = { modelCalls: 0, toolCalls: [], toolResultChars: 0, repoReadChars: 0, harnessChars: 0, assistantChars: 0,
|
|
34
|
+
filesRead: [], commands: [], lastToolResultText: null, lastAssistantText: null };
|
|
35
|
+
const byId = new Map();
|
|
36
|
+
for (const e of events ?? []) {
|
|
37
|
+
const content = Array.isArray(e?.message?.content) ? e.message.content : [];
|
|
38
|
+
if (e?.type === "assistant") {
|
|
39
|
+
out.modelCalls++;
|
|
40
|
+
const texts = [];
|
|
41
|
+
for (const c of content) {
|
|
42
|
+
if (c?.type === "tool_use") {
|
|
43
|
+
const input = c.input ?? {};
|
|
44
|
+
const tool = c.name ?? null;
|
|
45
|
+
const call = {
|
|
46
|
+
tool, path: typeof input.file_path === "string" ? input.file_path : typeof input.path === "string" ? input.path : null,
|
|
47
|
+
command: typeof input.command === "string" ? input.command : typeof input.pattern === "string" ? input.pattern : null,
|
|
48
|
+
resultChars: 0, repoRead: CLAUDE_REPO_READ_TOOLS.includes(String(tool ?? "").toLowerCase()),
|
|
49
|
+
};
|
|
50
|
+
out.toolCalls.push(call);
|
|
51
|
+
if (c.id) byId.set(c.id, call);
|
|
52
|
+
if (call.path && /^read$/i.test(tool ?? "")) out.filesRead.push(call.path);
|
|
53
|
+
if (typeof input.command === "string") out.commands.push(input.command);
|
|
54
|
+
} else if (c?.type === "text" && typeof c.text === "string") {
|
|
55
|
+
out.assistantChars += c.text.length;
|
|
56
|
+
texts.push(c.text);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
if (texts.length) out.lastAssistantText = texts.join("\n");
|
|
60
|
+
} else if (e?.type === "user") {
|
|
61
|
+
for (const c of content) {
|
|
62
|
+
if (c?.type !== "tool_result") continue;
|
|
63
|
+
const text = textOf(c.content);
|
|
64
|
+
out.toolResultChars += text.length;
|
|
65
|
+
out.lastToolResultText = text;
|
|
66
|
+
const call = byId.get(c.tool_use_id);
|
|
67
|
+
if (call) {
|
|
68
|
+
call.resultChars += text.length;
|
|
69
|
+
if (call.repoRead) out.repoReadChars += text.length; else out.harnessChars += text.length;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
return out;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* The transcript of the Claude Code session that ran in `cwd` since
|
|
79
|
+
* `sinceMs` (the newest such file), summarized; or { available: false }.
|
|
80
|
+
*/
|
|
81
|
+
// `tailBytes` reads only the file's last N bytes -- enough for a watcher
|
|
82
|
+
// that wants the latest tool call, without parsing a whole long session.
|
|
83
|
+
export function readClaudeSessionTranscript(cwd, { sinceMs = 0, home = os.homedir(), tailBytes = null } = {}) {
|
|
84
|
+
const candidates = [cwd];
|
|
85
|
+
try { candidates.push(fs.realpathSync(cwd)); } catch { /* the worktree may be gone already */ }
|
|
86
|
+
for (const dir of [...new Set(candidates.map((c) => claudeProjectDir(c, { home })))]) {
|
|
87
|
+
let files;
|
|
88
|
+
try { files = fs.readdirSync(dir).filter((f) => f.endsWith(".jsonl")).map((f) => path.join(dir, f)); } catch { continue; }
|
|
89
|
+
const recent = files.map((f) => { const st = fs.statSync(f); return { f, mtime: st.mtimeMs, size: st.size }; }).filter((x) => x.mtime >= sinceMs).sort((a, b) => b.mtime - a.mtime);
|
|
90
|
+
if (!recent.length) continue;
|
|
91
|
+
const events = [];
|
|
92
|
+
let text;
|
|
93
|
+
if (tailBytes && recent[0].size > tailBytes) {
|
|
94
|
+
const fd = fs.openSync(recent[0].f, "r");
|
|
95
|
+
try {
|
|
96
|
+
const buf = Buffer.alloc(tailBytes);
|
|
97
|
+
fs.readSync(fd, buf, 0, tailBytes, recent[0].size - tailBytes);
|
|
98
|
+
text = buf.toString("utf8").split("\n").slice(1).join("\n"); // drop the partial first line
|
|
99
|
+
} finally { fs.closeSync(fd); }
|
|
100
|
+
} else text = fs.readFileSync(recent[0].f, "utf8");
|
|
101
|
+
for (const line of text.split("\n")) {
|
|
102
|
+
if (!line.trim()) continue;
|
|
103
|
+
try { events.push(JSON.parse(line)); } catch { /* a torn last line while the session is still writing */ }
|
|
104
|
+
}
|
|
105
|
+
return { available: true, reason: null, source: "claude-code-session", dbPath: recent[0].f, events: events.length, ...summarizeClaudeEvents(events) };
|
|
106
|
+
}
|
|
107
|
+
return { available: false, reason: "no Claude Code session transcript found for this job's working directory" };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Token usage of every Claude Code session that ran in `cwd` since
|
|
112
|
+
* `sinceMs`, summed: { input, output, cacheRead, cacheWrite, calls } in
|
|
113
|
+
* OpenClaw's usage shape, or null when there's none. A claude-cli job's
|
|
114
|
+
* OpenClaw envelope carries only the final reply's usage (input 2, output
|
|
115
|
+
* 8 for a 29-call job), since the CLI runs its whole loop itself; its own
|
|
116
|
+
* session log has every call. Each call is logged once per content block
|
|
117
|
+
* with the same message id and usage, so ids count once.
|
|
118
|
+
*/
|
|
119
|
+
export function readClaudeSessionUsage(cwd, { sinceMs = 0, home = os.homedir() } = {}) {
|
|
120
|
+
const candidates = [cwd];
|
|
121
|
+
try { candidates.push(fs.realpathSync(cwd)); } catch { /* the worktree may be gone already */ }
|
|
122
|
+
const seen = new Set(), files = new Set();
|
|
123
|
+
const total = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, calls: 0 };
|
|
124
|
+
for (const dir of new Set(candidates.map((c) => claudeProjectDir(c, { home })))) {
|
|
125
|
+
let names = [];
|
|
126
|
+
try { names = fs.readdirSync(dir).filter((f) => f.endsWith(".jsonl")); } catch { continue; }
|
|
127
|
+
for (const name of names) {
|
|
128
|
+
const f = path.join(dir, name);
|
|
129
|
+
try { if (fs.statSync(f).mtimeMs < sinceMs || files.has(fs.realpathSync(f))) continue; files.add(fs.realpathSync(f)); } catch { continue; }
|
|
130
|
+
let text = "";
|
|
131
|
+
try { text = fs.readFileSync(f, "utf8"); } catch { continue; }
|
|
132
|
+
for (const line of text.split("\n")) {
|
|
133
|
+
if (!line.includes("\"usage\"")) continue;
|
|
134
|
+
let e; try { e = JSON.parse(line); } catch { continue; }
|
|
135
|
+
const m = e?.message, u = m?.usage;
|
|
136
|
+
if (e?.type !== "assistant" || !u || typeof u !== "object") continue;
|
|
137
|
+
const id = m.id ?? e.uuid;
|
|
138
|
+
if (id && seen.has(id)) continue;
|
|
139
|
+
if (id) seen.add(id);
|
|
140
|
+
total.calls++;
|
|
141
|
+
total.input += Number(u.input_tokens) || 0;
|
|
142
|
+
total.output += Number(u.output_tokens) || 0;
|
|
143
|
+
total.cacheRead += Number(u.cache_read_input_tokens) || 0;
|
|
144
|
+
total.cacheWrite += Number(u.cache_creation_input_tokens) || 0;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return total.calls ? total : null;
|
|
149
|
+
}
|
|
150
|
+
|