nomarmy 0.1.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +25 -0
- package/README.md +484 -0
- package/bin/nomarmy.mjs +2248 -0
- package/config/agents.yml.example +63 -0
- package/config/common.env +31 -0
- package/config/profiles/bedrock-cheap.env +26 -0
- package/config/profiles/bedrock.env +28 -0
- package/config/profiles/cpu-linux.env +8 -0
- package/config/profiles/dgx-spark.env +12 -0
- package/config/profiles/macbook-pro.env +9 -0
- package/config/profiles/nvidia-linux.env +9 -0
- package/docker/Dockerfile +15 -0
- package/docker/Dockerfile.go +29 -0
- package/docker/Dockerfile.rust +19 -0
- package/e2e.sh +153 -0
- package/install.sh +125 -0
- package/lib/agents.mjs +285 -0
- package/lib/army.mjs +400 -0
- package/lib/budget.mjs +368 -0
- package/lib/claude-transcript.mjs +150 -0
- package/lib/config.mjs +193 -0
- package/lib/connect.mjs +409 -0
- package/lib/coordinator-instructions.mjs +23 -0
- package/lib/decompose.mjs +389 -0
- package/lib/dispatch-config.mjs +164 -0
- package/lib/dispatch-schema.mjs +280 -0
- package/lib/doctor.mjs +443 -0
- package/lib/evidence.mjs +679 -0
- package/lib/gguf.mjs +589 -0
- package/lib/hardware.mjs +476 -0
- package/lib/health.mjs +278 -0
- package/lib/model-catalog.mjs +71 -0
- package/lib/notifier-app.mjs +95 -0
- package/lib/notify.mjs +66 -0
- package/lib/openclaw-config.mjs +65 -0
- package/lib/openclaw-errors.mjs +40 -0
- package/lib/propose.mjs +110 -0
- package/lib/prune.mjs +77 -0
- package/lib/repo-query.mjs +267 -0
- package/lib/runs.mjs +150 -0
- package/lib/sabotage.mjs +128 -0
- package/lib/sandbox-images.mjs +434 -0
- package/lib/scan.mjs +1538 -0
- package/lib/schema.mjs +288 -0
- package/lib/scout.mjs +544 -0
- package/lib/sizing.mjs +1322 -0
- package/lib/slots.mjs +112 -0
- package/lib/statusline.mjs +126 -0
- package/lib/subscription-config.mjs +68 -0
- package/lib/subscription-setup.mjs +217 -0
- package/lib/transcript.mjs +195 -0
- package/lib/verify.mjs +700 -0
- package/mcp/server.mjs +4206 -0
- package/notifier/icon.swift +34 -0
- package/notifier/main.swift +52 -0
- package/notifier/nomarmy-icon.png +0 -0
- package/package.json +67 -0
- package/playbooks/feature.md +43 -0
- package/policies/coder.md +49 -0
- package/policies/orchestrator.md +35 -0
- package/policies/reviewer.md +35 -0
- package/policies/scout.md +65 -0
- package/scripts/configure-openclaw.sh +96 -0
- package/scripts/configure-orchestrator.sh +84 -0
- package/scripts/install-llama-cpp.sh +16 -0
- package/scripts/lib.sh +198 -0
- package/scripts/select-model.mjs +96 -0
- package/scripts/select-model.sh +4 -0
- package/scripts/setup-sandbox.sh +38 -0
- package/scripts/start-inference.sh +46 -0
- package/scripts/stop-inference.sh +5 -0
- package/scripts/uninstall.sh +6 -0
- package/scripts/verify-install.sh +68 -0
package/lib/sizing.mjs
ADDED
|
@@ -0,0 +1,1322 @@
|
|
|
1
|
+
// nomArmy capacity sizing (v1.3).
|
|
2
|
+
//
|
|
3
|
+
// Turns hardware facts (`lib/hardware.mjs`) and model facts (`lib/gguf.mjs`)
|
|
4
|
+
// into a *recommendation* for the three coupled knobs. It never writes config:
|
|
5
|
+
// a human reads the recommendation and applies it.
|
|
6
|
+
//
|
|
7
|
+
// THE THING EVERYBODY GETS WRONG
|
|
8
|
+
// ------------------------------
|
|
9
|
+
// In llama.cpp's server, `-c` is the TOTAL context, shared out across the `-np`
|
|
10
|
+
// inference slots. Each slot gets `ctx_total / n_parallel`. So:
|
|
11
|
+
//
|
|
12
|
+
// -c 65536 -np 2 -> 32K per nom, not 64K per nom
|
|
13
|
+
//
|
|
14
|
+
// The v1.3 target is 64K *per nom*, so two noms need `-c 131072`. Every input
|
|
15
|
+
// and output here is therefore named `contextPerNom` or `contextTotal`. There
|
|
16
|
+
// is no bare "context" in this module, on purpose.
|
|
17
|
+
//
|
|
18
|
+
// THE SECOND THING
|
|
19
|
+
// ----------------
|
|
20
|
+
// `NOMARMY_MAX_WORKERS` above `NOMARMY_LLAMA_PARALLEL` does not buy throughput:
|
|
21
|
+
// the extra jobs queue on inference slots. That concurrency is fictional and
|
|
22
|
+
// only adds latency, so it is flagged as oversubscription and the default
|
|
23
|
+
// recommendation keeps the two equal for local execution.
|
|
24
|
+
//
|
|
25
|
+
// Both `recommend()` and `evaluateConfig()` are pure: facts in, structure out,
|
|
26
|
+
// no I/O at all.
|
|
27
|
+
|
|
28
|
+
import { deriveHeadDim } from "./gguf.mjs";
|
|
29
|
+
|
|
30
|
+
export const MIB = 1024 * 1024;
|
|
31
|
+
export const GIB = 1024 * 1024 * 1024;
|
|
32
|
+
|
|
33
|
+
/** v1.3 target: the autonomous explore/implement/test/repair loop needs room. */
|
|
34
|
+
export const DEFAULT_TARGET_CONTEXT_PER_NOM = 65536;
|
|
35
|
+
/** Architectural ceiling on bounded workers (see CLAUDE.md). */
|
|
36
|
+
export const MAX_NOMS = 8;
|
|
37
|
+
/** Never recommend a slot context below this; the loop cannot work in less. */
|
|
38
|
+
export const MIN_CONTEXT_PER_NOM = 8192;
|
|
39
|
+
/** Cloud execution is bounded by quota and budget, not by this machine. */
|
|
40
|
+
export const CLOUD_DEFAULT_MAX_WORKERS = 4;
|
|
41
|
+
|
|
42
|
+
/** Execution values that mean "inference does not happen on this machine". */
|
|
43
|
+
export const CLOUD_EXECUTIONS = Object.freeze(["bedrock", "bedrock-cheap", "cloud", "hosted"]);
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Explicit, generous headroom. Everything here is memory that is NOT available
|
|
47
|
+
* for weights or KV cache. Swapping mid-inference is far worse than running one
|
|
48
|
+
* fewer nom, so these are deliberately not tight.
|
|
49
|
+
*/
|
|
50
|
+
export const RESERVES = Object.freeze({
|
|
51
|
+
/** OS, page cache, desktop, editor. */
|
|
52
|
+
osBytes: 3 * GIB,
|
|
53
|
+
/** The frontier coordinator process, its MCP server and Git work. */
|
|
54
|
+
coordinatorBytes: 2 * GIB,
|
|
55
|
+
/** The Podman sandbox container nomArmy starts for every job. */
|
|
56
|
+
sandboxPerNomBytes: Math.round(1.5 * GIB),
|
|
57
|
+
/** llama-server runtime: CUDA/Metal context, graph and scratch allocations. */
|
|
58
|
+
runtimeOverheadBytes: 1 * GIB,
|
|
59
|
+
/** Per-slot activation/compute buffers, which are not KV cache. */
|
|
60
|
+
computeBufferPerSlotBytes: Math.round(0.5 * GIB),
|
|
61
|
+
/** Never plan to fill more than this fraction of the pool. */
|
|
62
|
+
safetyFraction: 0.9,
|
|
63
|
+
/**
|
|
64
|
+
* Share of Apple unified memory the GPU may hold wired. 0.75 was an
|
|
65
|
+
* unverified rule of thumb carried since the initial commit. Measured
|
|
66
|
+
* directly on an M3 Max 64 GiB running Qwen3-Coder-Next Q4_K_M (45.1 GiB
|
|
67
|
+
* weights): 3 noms at 64K each (~52.1 GiB estimated GPU-resident, ~81% of
|
|
68
|
+
* the pool) completed a real job correctly; 4 noms at 64K each (~54.1 GiB,
|
|
69
|
+
* ~85%) failed to load with a genuine Metal allocation error
|
|
70
|
+
* (ggml_metal_synchronize: command buffer failed, kIOGPUCommandBufferCallbackErrorOutOfMemory)
|
|
71
|
+
* that /health did not catch (see the completion-based health check in
|
|
72
|
+
* lib/doctor.mjs / scripts/start-inference.sh). 0.80 sits with real margin
|
|
73
|
+
* below the observed 81% success point and well below the 85% failure
|
|
74
|
+
* point -- still a deliberate margin, now one anchored to a measurement
|
|
75
|
+
* instead of a guess. This is one machine and one (hybrid-architecture,
|
|
76
|
+
* small-KV-footprint) model; revisit if a dense model or different
|
|
77
|
+
* hardware class shows a different real ceiling.
|
|
78
|
+
*/
|
|
79
|
+
unifiedGpuFraction: 0.80,
|
|
80
|
+
/**
|
|
81
|
+
* Coherent-memory detection. On a GB10 / DGX Spark class machine the
|
|
82
|
+
* "VRAM" nvidia-smi reports and the system RAM are the same silicon, so they
|
|
83
|
+
* must not be budgeted twice. The signal is that the two figures are nearly
|
|
84
|
+
* equal -- a band, not a floor: a large discrete card in a RAM-poor box has
|
|
85
|
+
* VRAM well *above* system RAM and is not coherent.
|
|
86
|
+
*/
|
|
87
|
+
coherentMemoryRatioMin: 0.85,
|
|
88
|
+
/** Grace/Jetson class hosts are ARM64, where the same closeness is stronger evidence. */
|
|
89
|
+
coherentMemoryRatioMinArm: 0.6,
|
|
90
|
+
coherentMemoryRatioMax: 1.2,
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* fp16 KV cache. llama.cpp defaults to f16 for both K and V unless started with
|
|
95
|
+
* `--cache-type-k` / `--cache-type-v`. Override via `bytesPerKvElement` when you
|
|
96
|
+
* quantize the cache; the assumption is always stated in the result.
|
|
97
|
+
*/
|
|
98
|
+
export const DEFAULT_BYTES_PER_KV_ELEMENT = 2;
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Bytes per element for llama-server's `--cache-type-k` / `--cache-type-v`
|
|
102
|
+
* values, so `NOMARMY_LLAMA_CACHE_TYPE_K/V` (see scripts/start-inference.sh)
|
|
103
|
+
* can be reflected in the sizing math instead of silently assuming fp16 once
|
|
104
|
+
* someone quantizes the KV cache to fit more context. K and V are modeled as
|
|
105
|
+
* one shared element size (see bytesPerKvElementForCacheTypes below); llama.cpp
|
|
106
|
+
* allows setting them independently, but mixed-precision KV is not separately
|
|
107
|
+
* modeled here.
|
|
108
|
+
*/
|
|
109
|
+
export const KV_CACHE_TYPE_BYTES = Object.freeze({
|
|
110
|
+
f32: 4,
|
|
111
|
+
f16: 2,
|
|
112
|
+
bf16: 2,
|
|
113
|
+
q8_0: 1,
|
|
114
|
+
q5_1: 0.6875,
|
|
115
|
+
q5_0: 0.625,
|
|
116
|
+
q4_1: 0.5625,
|
|
117
|
+
q4_0: 0.5,
|
|
118
|
+
iq4_nl: 0.5,
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Resolve the KV element size to use for sizing math from the K/V cache-type
|
|
123
|
+
* strings a caller may have set (matching llama-server's own flag values,
|
|
124
|
+
* case-insensitively). Unset or unrecognized values fall back to the fp16
|
|
125
|
+
* default. When K and V differ, the smaller (cheaper) of the two is used --
|
|
126
|
+
* an optimistic estimate is flagged in the assumptions text by the caller,
|
|
127
|
+
* rather than silently averaging two different real allocations.
|
|
128
|
+
*/
|
|
129
|
+
export function bytesPerKvElementForCacheTypes(cacheTypeK, cacheTypeV) {
|
|
130
|
+
const resolve = (t) => (t ? KV_CACHE_TYPE_BYTES[String(t).toLowerCase()] : undefined);
|
|
131
|
+
const kExplicit = resolve(cacheTypeK);
|
|
132
|
+
const vExplicit = resolve(cacheTypeV);
|
|
133
|
+
if (kExplicit === undefined && vExplicit === undefined) return DEFAULT_BYTES_PER_KV_ELEMENT;
|
|
134
|
+
// An unset side is not "ignore it" -- llama.cpp still allocates it at the
|
|
135
|
+
// real fp16 default. Setting only one of K/V (e.g. quantizing V while
|
|
136
|
+
// leaving K untouched) must compare against that real default, not vanish
|
|
137
|
+
// from the comparison entirely, or the still-fp16 side is silently dropped
|
|
138
|
+
// from consideration. This still returns ONE scalar applied to both K and V
|
|
139
|
+
// in kvBytesPerSlot's formula, so a genuinely asymmetric K/V is approximated
|
|
140
|
+
// by its smaller side either way -- that approximation is unchanged and
|
|
141
|
+
// intentional (see doc comment above); this only makes the unset-side
|
|
142
|
+
// comparison honest instead of skipping it.
|
|
143
|
+
return Math.min(kExplicit ?? DEFAULT_BYTES_PER_KV_ELEMENT, vExplicit ?? DEFAULT_BYTES_PER_KV_ELEMENT);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Used only when the GGUF is absent or unreadable, which is the normal state of
|
|
148
|
+
* a fresh install. These are deliberately pessimistic stand-ins for a ~30B-class
|
|
149
|
+
* GQA coder model at Q4_K_M. They are ASSUMPTIONS, not measurements, and any
|
|
150
|
+
* result built on them is marked `confidence: "low"`.
|
|
151
|
+
*/
|
|
152
|
+
export const FALLBACK_MODEL = Object.freeze({
|
|
153
|
+
weightsBytes: 18 * GIB,
|
|
154
|
+
blockCount: 48,
|
|
155
|
+
headCountKv: 8,
|
|
156
|
+
headDim: 128,
|
|
157
|
+
contextLength: 262144,
|
|
158
|
+
label: "~30B-class GQA coder model at Q4_K_M",
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* KV cache only grows with context on layers that do full (softmax)
|
|
163
|
+
* attention. A hybrid architecture mixes those with recurrent/state-space
|
|
164
|
+
* layers (Mamba-style; GGUF exposes this as `.ssm.*` header keys) whose state
|
|
165
|
+
* is a small, roughly context-INDEPENDENT size -- `block_count` alone cannot
|
|
166
|
+
* tell full attention layers from recurrent ones apart, and there is no
|
|
167
|
+
* per-layer array in the GGUF header either (checked directly against a real
|
|
168
|
+
* Qwen3-Coder-Next-GGUF file: `attention.head_count_kv` and `block_count` are
|
|
169
|
+
* both plain scalars, not arrays). The ratio is a fixed fact of each specific
|
|
170
|
+
* hybrid architecture's design, not something derivable from the file, so
|
|
171
|
+
* this is a small, explicit, named table rather than a formula -- add an
|
|
172
|
+
* entry only once the ratio is verified against that architecture's own
|
|
173
|
+
* documentation, the way qwen3next's was.
|
|
174
|
+
*
|
|
175
|
+
* qwen3next (Qwen3-Next, including Qwen3-Coder-Next): a fixed 3:1 layout of
|
|
176
|
+
* three Gated DeltaNet (linear-attention) blocks then one full-attention
|
|
177
|
+
* block, repeating -- 1 in 4 layers is full attention. Sources: Qwen's own
|
|
178
|
+
* announcement (https://qwen.ai/blog?id=4074cca80393150c248e508aa62983f9cb7d27cd)
|
|
179
|
+
* and NVIDIA's technical writeup of the architecture.
|
|
180
|
+
*/
|
|
181
|
+
export const HYBRID_ATTENTION_LAYER_FRACTION = Object.freeze({
|
|
182
|
+
qwen3next: 1 / 4,
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
// ---------------------------------------------------------------------------
|
|
186
|
+
// Small helpers
|
|
187
|
+
// ---------------------------------------------------------------------------
|
|
188
|
+
|
|
189
|
+
function num(value) {
|
|
190
|
+
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : null;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** Human-readable byte size, for warning text and for display. */
|
|
194
|
+
export function formatBytes(bytes) {
|
|
195
|
+
if (bytes === null || bytes === undefined || !Number.isFinite(bytes)) return "unknown";
|
|
196
|
+
const sign = bytes < 0 ? "-" : "";
|
|
197
|
+
const abs = Math.abs(bytes);
|
|
198
|
+
if (abs >= GIB) return `${sign}${(abs / GIB).toFixed(1)} GiB`;
|
|
199
|
+
if (abs >= MIB) return `${sign}${Math.round(abs / MIB)} MiB`;
|
|
200
|
+
return `${sign}${abs} B`;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/** Human-readable context size, e.g. 65536 -> "64K". */
|
|
204
|
+
export function formatContext(tokens) {
|
|
205
|
+
if (!Number.isFinite(tokens)) return "unknown";
|
|
206
|
+
return tokens % 1024 === 0 ? `${tokens / 1024}K` : `${tokens}`;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
function warning(code, severity, message) {
|
|
210
|
+
return { code, severity, message };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
export function isCloudExecution(execution) {
|
|
214
|
+
return CLOUD_EXECUTIONS.includes(String(execution || "").toLowerCase());
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// ---------------------------------------------------------------------------
|
|
218
|
+
// Model facts
|
|
219
|
+
// ---------------------------------------------------------------------------
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Resolve the architecture numbers needed for the KV formula, recording which
|
|
223
|
+
* of them came from the file and which were assumed.
|
|
224
|
+
*
|
|
225
|
+
* @param {object|null} gguf result of `readGGUFMetadata`
|
|
226
|
+
* @returns {object}
|
|
227
|
+
*/
|
|
228
|
+
export function resolveModel(gguf) {
|
|
229
|
+
const found = Boolean(gguf && gguf.found);
|
|
230
|
+
const params = (gguf && gguf.params) || {};
|
|
231
|
+
const assumed = [];
|
|
232
|
+
|
|
233
|
+
let layers = found ? num(params.blockCount) : null;
|
|
234
|
+
if (!layers) {
|
|
235
|
+
layers = FALLBACK_MODEL.blockCount;
|
|
236
|
+
assumed.push(`block_count assumed to be ${FALLBACK_MODEL.blockCount}`);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
let kvHeads = found ? num(params.headCountKv) || num(params.headCount) : null;
|
|
240
|
+
if (!kvHeads) {
|
|
241
|
+
kvHeads = FALLBACK_MODEL.headCountKv;
|
|
242
|
+
assumed.push(`attention.head_count_kv assumed to be ${FALLBACK_MODEL.headCountKv} (GQA)`);
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
const derived = found ? deriveHeadDim(params) : { headDim: null, source: null };
|
|
246
|
+
let keyLength = found ? num(params.keyLength) || derived.headDim : null;
|
|
247
|
+
if (!keyLength) {
|
|
248
|
+
keyLength = FALLBACK_MODEL.headDim;
|
|
249
|
+
assumed.push(`attention.key_length assumed to be ${FALLBACK_MODEL.headDim}`);
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
let valueLength = found ? num(params.valueLength) || (num(params.keyLength) ? params.keyLength : derived.headDim) : null;
|
|
253
|
+
if (!valueLength) {
|
|
254
|
+
valueLength = keyLength;
|
|
255
|
+
if (!found) assumed.push(`attention.value_length assumed equal to key_length (${keyLength})`);
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
let weightsBytes = found ? num(gguf && gguf.fileSizeBytes) : null;
|
|
259
|
+
if (!weightsBytes) {
|
|
260
|
+
weightsBytes = FALLBACK_MODEL.weightsBytes;
|
|
261
|
+
assumed.push(
|
|
262
|
+
`model weights assumed to be ${formatBytes(FALLBACK_MODEL.weightsBytes)} (${FALLBACK_MODEL.label})`,
|
|
263
|
+
);
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const modelContextLength = found ? num(params.contextLength) : null;
|
|
267
|
+
|
|
268
|
+
// `layers` above is the TOTAL block count, used for weights and display.
|
|
269
|
+
// `kvLayers` is the count that actually belongs in the KV-cache-per-context
|
|
270
|
+
// formula: on a hybrid architecture only a fraction of layers do full
|
|
271
|
+
// attention (see HYBRID_ATTENTION_LAYER_FRACTION), the rest carry a
|
|
272
|
+
// recurrent state whose size does not scale with context. This is a known,
|
|
273
|
+
// cited correction, not a guess filling in missing data -- it is kept out
|
|
274
|
+
// of `assumed`/`complete`/confidence, which mean "the file didn't say and
|
|
275
|
+
// we substituted something"; applying it makes the estimate MORE accurate,
|
|
276
|
+
// not less certain.
|
|
277
|
+
const arch = (gguf && gguf.arch) || null;
|
|
278
|
+
const hasSsmLayers = found && [params.ssmStateSize, params.ssmInnerSize, params.ssmConvKernel, params.ssmGroupCount]
|
|
279
|
+
.some((v) => num(v));
|
|
280
|
+
const hybridFraction = arch && Object.prototype.hasOwnProperty.call(HYBRID_ATTENTION_LAYER_FRACTION, arch)
|
|
281
|
+
? HYBRID_ATTENTION_LAYER_FRACTION[arch]
|
|
282
|
+
: null;
|
|
283
|
+
let kvLayers = layers;
|
|
284
|
+
let hybridNote = null;
|
|
285
|
+
if (hasSsmLayers && hybridFraction !== null) {
|
|
286
|
+
kvLayers = Math.max(1, Math.round(layers * hybridFraction));
|
|
287
|
+
hybridNote = `${arch} is a known hybrid architecture (${Math.round(hybridFraction * 100)}% full attention); ` +
|
|
288
|
+
`KV cache is sized for ${kvLayers} of ${layers} layers, not all of them, since the rest carry a ` +
|
|
289
|
+
"recurrent state that does not grow with context.";
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
return {
|
|
293
|
+
source: found ? "gguf" : "fallback",
|
|
294
|
+
found,
|
|
295
|
+
truncated: Boolean(gguf && gguf.truncated),
|
|
296
|
+
arch,
|
|
297
|
+
layers,
|
|
298
|
+
kvLayers,
|
|
299
|
+
isHybrid: hasSsmLayers,
|
|
300
|
+
hybridRecognized: hybridNote !== null,
|
|
301
|
+
hybridNote,
|
|
302
|
+
kvHeads,
|
|
303
|
+
keyLength,
|
|
304
|
+
valueLength,
|
|
305
|
+
weightsBytes,
|
|
306
|
+
modelContextLength,
|
|
307
|
+
assumed,
|
|
308
|
+
complete: found && assumed.length === 0 && !(gguf && gguf.truncated),
|
|
309
|
+
};
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/**
|
|
313
|
+
* KV cache bytes for ONE slot at `contextPerNom` tokens.
|
|
314
|
+
*
|
|
315
|
+
* kv_per_slot = n_layers * n_kv_heads * (key_length + value_length)
|
|
316
|
+
* * context_per_nom * bytes_per_element
|
|
317
|
+
*
|
|
318
|
+
* which is the familiar `2 * layers * kv_heads * head_dim * ctx * bytes` when
|
|
319
|
+
* key and value dimensions are equal, and stays correct when they are not.
|
|
320
|
+
*
|
|
321
|
+
* Uses `model.kvLayers` when present -- the count of layers that actually do
|
|
322
|
+
* full attention, which is `layers` itself except on a recognized hybrid
|
|
323
|
+
* architecture (see resolveModel) -- falling back to `model.layers` for a
|
|
324
|
+
* hand-built model object that predates this field.
|
|
325
|
+
*
|
|
326
|
+
* @param {object} model result of `resolveModel`
|
|
327
|
+
* @param {number} contextPerNom
|
|
328
|
+
* @param {number} bytesPerElement
|
|
329
|
+
* @returns {number}
|
|
330
|
+
*/
|
|
331
|
+
export function kvBytesPerSlot(model, contextPerNom, bytesPerElement = DEFAULT_BYTES_PER_KV_ELEMENT) {
|
|
332
|
+
const layers = model.kvLayers ?? model.layers;
|
|
333
|
+
return (
|
|
334
|
+
layers * model.kvHeads * (model.keyLength + model.valueLength) * contextPerNom * bytesPerElement
|
|
335
|
+
);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
// ---------------------------------------------------------------------------
|
|
339
|
+
// Memory pool
|
|
340
|
+
// ---------------------------------------------------------------------------
|
|
341
|
+
|
|
342
|
+
/**
|
|
343
|
+
* Work out which pool the weights and KV cache actually land in.
|
|
344
|
+
*
|
|
345
|
+
* kinds:
|
|
346
|
+
* "vram" discrete NVIDIA VRAM; Podman/OS/coordinator live in system RAM
|
|
347
|
+
* "unified" one pool shared by CPU and GPU (Apple Silicon, GB10 coherent)
|
|
348
|
+
* "system" no GPU at all; everything is system RAM
|
|
349
|
+
* "unknown" nothing could be measured
|
|
350
|
+
*
|
|
351
|
+
* @param {object|null} hardware
|
|
352
|
+
* @returns {object}
|
|
353
|
+
*/
|
|
354
|
+
/**
|
|
355
|
+
* Memory Podman has carved out of host RAM and that a local model can never use.
|
|
356
|
+
*
|
|
357
|
+
* On a macOS/Windows Podman machine, containers run inside a VM with a fixed
|
|
358
|
+
* memory allocation, reserved whether or not anything is running. That
|
|
359
|
+
* allocation is unavailable to llama-server, so it must come off the pool.
|
|
360
|
+
* On native Linux, containers share the host kernel and draw from the same
|
|
361
|
+
* pool as everything else, so only the per-nom sandbox charge applies.
|
|
362
|
+
*
|
|
363
|
+
* The absolute numbers here can be dramatic: on one real machine, switching
|
|
364
|
+
* this sandbox from Docker Desktop to Podman took the configured VM ceiling
|
|
365
|
+
* from ~31.2 GiB down to Podman machine's 2 GiB default, over 15x less
|
|
366
|
+
* memory permanently carved out of the pool a local model can use.
|
|
367
|
+
*/
|
|
368
|
+
export function podmanVmReservation(hardware) {
|
|
369
|
+
const hw = hardware || {};
|
|
370
|
+
const total = Number(hw.podman && hw.podman.totalMemoryBytes) || 0;
|
|
371
|
+
if (!total) return 0;
|
|
372
|
+
const desktopHost = hw.platform === "win32" || hw.platform === "darwin";
|
|
373
|
+
if (!desktopHost) return 0;
|
|
374
|
+
// MemTotal is the VM CEILING, not a standing reservation: the WSL2 and
|
|
375
|
+
// virtiofs backends grow and release within it. Charging all of it made an
|
|
376
|
+
// 11.3 GiB model that demonstrably runs here report "does not fit", so half
|
|
377
|
+
// the ceiling is used as a working estimate of steady-state pressure. This
|
|
378
|
+
// is a heuristic, not a measurement, and it is stated as one in the output.
|
|
379
|
+
return Math.floor(total / 2);
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
export function resolveMemoryPool(hardware, options = {}) {
|
|
383
|
+
// A Podman machine holds its VM allocation out of host RAM permanently, so
|
|
384
|
+
// the pool a local model can actually use is smaller than total RAM suggests.
|
|
385
|
+
const vmReservationBytes = podmanVmReservation(hardware);
|
|
386
|
+
const hw = hardware || {};
|
|
387
|
+
const mem = hw.memory || {};
|
|
388
|
+
const gpu = hw.gpu || {};
|
|
389
|
+
const systemTotal = num(mem.totalBytes);
|
|
390
|
+
const systemAvailable = num(mem.availableBytes);
|
|
391
|
+
const gpuCount = Number.isFinite(gpu.count) ? gpu.count : 0;
|
|
392
|
+
const vramTotal = num(gpu.totalVramBytes);
|
|
393
|
+
const vramFree = num(gpu.freeVramBytes) || vramTotal;
|
|
394
|
+
|
|
395
|
+
if (gpuCount > 0 && vramTotal) {
|
|
396
|
+
const ratio = systemTotal ? vramTotal / systemTotal : null;
|
|
397
|
+
const minRatio =
|
|
398
|
+
hw.arch === "arm64" ? RESERVES.coherentMemoryRatioMinArm : RESERVES.coherentMemoryRatioMin;
|
|
399
|
+
const coherent = Boolean(
|
|
400
|
+
ratio !== null && ratio >= minRatio && ratio <= RESERVES.coherentMemoryRatioMax,
|
|
401
|
+
);
|
|
402
|
+
if (coherent) {
|
|
403
|
+
// Grace Blackwell / DGX Spark class: the reported "VRAM" is the same
|
|
404
|
+
// silicon as system RAM. Budget it once.
|
|
405
|
+
return {
|
|
406
|
+
kind: "unified", vmReservationBytes,
|
|
407
|
+
coherent: true,
|
|
408
|
+
poolBytes: Math.min(systemTotal, vramTotal),
|
|
409
|
+
gpuResidentCapBytes: vramFree,
|
|
410
|
+
systemTotalBytes: systemTotal,
|
|
411
|
+
systemAvailableBytes: systemAvailable,
|
|
412
|
+
gpuCount,
|
|
413
|
+
source: "nvidia-smi on a coherent-memory system",
|
|
414
|
+
};
|
|
415
|
+
}
|
|
416
|
+
return {
|
|
417
|
+
kind: "vram", vmReservationBytes,
|
|
418
|
+
coherent: false,
|
|
419
|
+
poolBytes: vramFree,
|
|
420
|
+
gpuResidentCapBytes: vramFree,
|
|
421
|
+
systemTotalBytes: systemTotal,
|
|
422
|
+
systemAvailableBytes: systemAvailable,
|
|
423
|
+
gpuCount,
|
|
424
|
+
source: "nvidia-smi free VRAM",
|
|
425
|
+
};
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
if (mem.unified) {
|
|
429
|
+
const unified = num(mem.unifiedBytes) || systemTotal;
|
|
430
|
+
if (unified) {
|
|
431
|
+
return {
|
|
432
|
+
kind: "unified", vmReservationBytes,
|
|
433
|
+
coherent: true,
|
|
434
|
+
poolBytes: unified,
|
|
435
|
+
gpuResidentCapBytes: Math.floor(unified * RESERVES.unifiedGpuFraction),
|
|
436
|
+
systemTotalBytes: systemTotal || unified,
|
|
437
|
+
systemAvailableBytes: systemAvailable,
|
|
438
|
+
gpuCount: 0,
|
|
439
|
+
// Apple Silicon is not the only coherent-memory platform nomArmy targets:
|
|
440
|
+
// DGX Spark (Grace Blackwell, Linux arm64) shares one pool too. Naming the
|
|
441
|
+
// wrong platform in a sizing warning misleads exactly the users who most
|
|
442
|
+
// need it, so the label follows the detected platform.
|
|
443
|
+
appleUnified: hardware?.platform === "darwin" || hardware?.appleSilicon === true,
|
|
444
|
+
source: hardware?.platform === "darwin" || hardware?.appleSilicon === true
|
|
445
|
+
? "Apple Silicon unified memory"
|
|
446
|
+
: "coherent unified memory",
|
|
447
|
+
};
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
if (systemAvailable || systemTotal) {
|
|
452
|
+
return {
|
|
453
|
+
kind: "system",
|
|
454
|
+
coherent: false,
|
|
455
|
+
// Capacity planning sizes against what the machine HAS, not what happens to
|
|
456
|
+
// be free while a browser is open -- otherwise the same machine yields a
|
|
457
|
+
// different recommendation hour to hour. useAvailable opts into the
|
|
458
|
+
// right-now view; a large gap between the two is reported as a warning.
|
|
459
|
+
poolBytes: (options.useAvailable ? systemAvailable : systemTotal) || systemAvailable || systemTotal,
|
|
460
|
+
gpuResidentCapBytes: null,
|
|
461
|
+
systemTotalBytes: systemTotal,
|
|
462
|
+
systemAvailableBytes: systemAvailable,
|
|
463
|
+
vmReservationBytes,
|
|
464
|
+
|
|
465
|
+
gpuCount: 0,
|
|
466
|
+
source: "system RAM (no GPU detected)",
|
|
467
|
+
};
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
return {
|
|
471
|
+
kind: "unknown",
|
|
472
|
+
coherent: false,
|
|
473
|
+
poolBytes: null,
|
|
474
|
+
gpuResidentCapBytes: null,
|
|
475
|
+
systemTotalBytes: null,
|
|
476
|
+
systemAvailableBytes: null,
|
|
477
|
+
gpuCount: 0,
|
|
478
|
+
source: "nothing measurable",
|
|
479
|
+
};
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* How many noms fit at a given per-nom context, and the breakdown that produced
|
|
484
|
+
* the answer. `rawCapacity` may be 0 or negative: that is the signal that not
|
|
485
|
+
* even one nom fits inside the reserved headroom.
|
|
486
|
+
*/
|
|
487
|
+
function capacityFor(pool, model, contextPerNom, bytesPerElement) {
|
|
488
|
+
const kvPerSlot = kvBytesPerSlot(model, contextPerNom, bytesPerElement);
|
|
489
|
+
const R = RESERVES;
|
|
490
|
+
|
|
491
|
+
if (pool.kind === "unknown") {
|
|
492
|
+
return { rawCapacity: 1, kvPerSlot, budgetBytes: null, perSlotBytes: null, limitedBy: "unknown-hardware" };
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
if (pool.kind === "vram") {
|
|
496
|
+
// Weights and KV live in VRAM; sandboxes and the coordinator do not.
|
|
497
|
+
const gpuBudget = pool.poolBytes * R.safetyFraction - R.runtimeOverheadBytes - model.weightsBytes;
|
|
498
|
+
const perSlotGpu = kvPerSlot + R.computeBufferPerSlotBytes;
|
|
499
|
+
const byGpu = Math.floor(gpuBudget / perSlotGpu);
|
|
500
|
+
|
|
501
|
+
const sysPool = pool.systemAvailableBytes || pool.systemTotalBytes;
|
|
502
|
+
let bySystem = Number.POSITIVE_INFINITY;
|
|
503
|
+
let systemBudget = null;
|
|
504
|
+
if (sysPool) {
|
|
505
|
+
// Must mirror the unified/system branch below: a Podman machine VM
|
|
506
|
+
// allocation comes off system RAM here too, not just out of the GPU pool.
|
|
507
|
+
systemBudget = sysPool * R.safetyFraction - R.osBytes - R.coordinatorBytes - (pool.vmReservationBytes ?? 0);
|
|
508
|
+
bySystem = Math.floor(systemBudget / R.sandboxPerNomBytes);
|
|
509
|
+
}
|
|
510
|
+
return {
|
|
511
|
+
rawCapacity: Math.min(byGpu, bySystem),
|
|
512
|
+
kvPerSlot,
|
|
513
|
+
budgetBytes: gpuBudget,
|
|
514
|
+
systemBudgetBytes: systemBudget,
|
|
515
|
+
perSlotBytes: perSlotGpu,
|
|
516
|
+
limitedBy: byGpu <= bySystem ? "vram" : "system-ram",
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
// One pool: everything comes out of the same bytes.
|
|
521
|
+
// Must mirror memoryBreakdown exactly: a Podman machine VM allocation comes
|
|
522
|
+
// off the pool, and its sandboxes are then inside the VM rather than extra.
|
|
523
|
+
const vmReservation = pool.vmReservationBytes ?? 0;
|
|
524
|
+
const budget =
|
|
525
|
+
pool.poolBytes * R.safetyFraction -
|
|
526
|
+
R.osBytes -
|
|
527
|
+
R.coordinatorBytes -
|
|
528
|
+
R.runtimeOverheadBytes -
|
|
529
|
+
vmReservation -
|
|
530
|
+
model.weightsBytes;
|
|
531
|
+
const perSlot =
|
|
532
|
+
kvPerSlot + R.computeBufferPerSlotBytes + (vmReservation > 0 ? 0 : R.sandboxPerNomBytes);
|
|
533
|
+
let raw = Math.floor(budget / perSlot);
|
|
534
|
+
let limitedBy = pool.kind === "system" ? "system-ram" : "unified-memory";
|
|
535
|
+
|
|
536
|
+
if (pool.gpuResidentCapBytes) {
|
|
537
|
+
// The GPU may not be allowed to hold the whole pool wired.
|
|
538
|
+
const gpuBudget = pool.gpuResidentCapBytes - R.runtimeOverheadBytes - model.weightsBytes;
|
|
539
|
+
const byGpu = Math.floor(gpuBudget / (kvPerSlot + R.computeBufferPerSlotBytes));
|
|
540
|
+
if (byGpu < raw) {
|
|
541
|
+
raw = byGpu;
|
|
542
|
+
limitedBy = "gpu-resident-cap";
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
|
|
546
|
+
return { rawCapacity: raw, kvPerSlot, budgetBytes: budget, perSlotBytes: perSlot, limitedBy };
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
/** Full memory breakdown for a concrete (noms, contextPerNom) choice. */
|
|
550
|
+
function memoryBreakdown(pool, model, noms, contextPerNom, bytesPerElement) {
|
|
551
|
+
const R = RESERVES;
|
|
552
|
+
const kvPerSlot = kvBytesPerSlot(model, contextPerNom, bytesPerElement);
|
|
553
|
+
const kvTotal = kvPerSlot * noms;
|
|
554
|
+
const computeBuffers = R.computeBufferPerSlotBytes * noms;
|
|
555
|
+
// On a Podman machine host the sandboxes live inside the VM, so charging them
|
|
556
|
+
// per nom on top of the whole VM allocation would double-count.
|
|
557
|
+
const vmReservationBytes = pool.vmReservationBytes ?? 0;
|
|
558
|
+
const sandboxes = vmReservationBytes > 0 ? 0 : R.sandboxPerNomBytes * noms;
|
|
559
|
+
|
|
560
|
+
const gpuResidentBytes = model.weightsBytes + kvTotal + computeBuffers + R.runtimeOverheadBytes;
|
|
561
|
+
const systemResidentBytes = sandboxes + vmReservationBytes + R.osBytes + R.coordinatorBytes;
|
|
562
|
+
const estimatedTotalBytes =
|
|
563
|
+
pool.kind === "vram" ? gpuResidentBytes : gpuResidentBytes + systemResidentBytes;
|
|
564
|
+
|
|
565
|
+
const budgetBytes = pool.poolBytes === null ? null : pool.poolBytes * R.safetyFraction;
|
|
566
|
+
const headroomBytes = budgetBytes === null ? null : budgetBytes - estimatedTotalBytes;
|
|
567
|
+
|
|
568
|
+
return {
|
|
569
|
+
pool: {
|
|
570
|
+
kind: pool.kind,
|
|
571
|
+
coherent: pool.coherent,
|
|
572
|
+
bytes: pool.poolBytes,
|
|
573
|
+
source: pool.source,
|
|
574
|
+
gpuResidentCapBytes: pool.gpuResidentCapBytes,
|
|
575
|
+
systemTotalBytes: pool.systemTotalBytes,
|
|
576
|
+
systemAvailableBytes: pool.systemAvailableBytes,
|
|
577
|
+
},
|
|
578
|
+
bytesPerKvElement: bytesPerElement,
|
|
579
|
+
modelWeightsBytes: model.weightsBytes,
|
|
580
|
+
kvBytesPerSlot: kvPerSlot,
|
|
581
|
+
kvBytesTotal: kvTotal,
|
|
582
|
+
reserved: {
|
|
583
|
+
osBytes: R.osBytes,
|
|
584
|
+
coordinatorBytes: R.coordinatorBytes,
|
|
585
|
+
vmReservationBytes,
|
|
586
|
+
sandboxBytes: sandboxes,
|
|
587
|
+
sandboxPerNomBytes: R.sandboxPerNomBytes,
|
|
588
|
+
runtimeOverheadBytes: R.runtimeOverheadBytes,
|
|
589
|
+
computeBufferBytes: computeBuffers,
|
|
590
|
+
safetyFraction: R.safetyFraction,
|
|
591
|
+
safetyMarginBytes: pool.poolBytes === null ? null : pool.poolBytes * (1 - R.safetyFraction),
|
|
592
|
+
totalBytes:
|
|
593
|
+
(pool.kind === "vram" ? 0 : R.osBytes + R.coordinatorBytes + sandboxes) +
|
|
594
|
+
R.runtimeOverheadBytes +
|
|
595
|
+
computeBuffers +
|
|
596
|
+
(pool.poolBytes === null ? 0 : pool.poolBytes * (1 - R.safetyFraction)),
|
|
597
|
+
},
|
|
598
|
+
gpuResidentBytes,
|
|
599
|
+
systemResidentBytes,
|
|
600
|
+
estimatedTotalBytes,
|
|
601
|
+
budgetBytes,
|
|
602
|
+
headroomBytes,
|
|
603
|
+
fits: headroomBytes === null ? null : headroomBytes >= 0,
|
|
604
|
+
};
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
// ---------------------------------------------------------------------------
|
|
608
|
+
// Shared warning builders, so recommend() and evaluateConfig() agree on shape
|
|
609
|
+
// ---------------------------------------------------------------------------
|
|
610
|
+
|
|
611
|
+
function oversubscriptionWarning(maxWorkers, llamaParallel) {
|
|
612
|
+
return warning(
|
|
613
|
+
"oversubscription",
|
|
614
|
+
"warning",
|
|
615
|
+
`NOMARMY_MAX_WORKERS=${maxWorkers} exceeds NOMARMY_LLAMA_PARALLEL=${llamaParallel}: ` +
|
|
616
|
+
`${maxWorkers - llamaParallel} nom(s) will queue waiting for an inference slot. ` +
|
|
617
|
+
"That concurrency is fictional; it adds latency without adding throughput. " +
|
|
618
|
+
`Set max workers to ${llamaParallel}, or raise parallel slots (and the total context with it).`,
|
|
619
|
+
);
|
|
620
|
+
}
|
|
621
|
+
|
|
622
|
+
function contextBelowTargetWarning(contextPerNom, target, contextTotal, llamaParallel) {
|
|
623
|
+
return warning(
|
|
624
|
+
"context_below_target",
|
|
625
|
+
"warning",
|
|
626
|
+
`-c ${contextTotal} shared across -np ${llamaParallel} gives each nom ` +
|
|
627
|
+
`${formatContext(contextPerNom)} (${contextPerNom} tokens), below the ` +
|
|
628
|
+
`${formatContext(target)} v1.3 target. llama.cpp divides total context by slots: ` +
|
|
629
|
+
`for ${llamaParallel} nom(s) at ${formatContext(target)} each, set ` +
|
|
630
|
+
`NOMARMY_LLAMA_CONTEXT=${target * llamaParallel}.`,
|
|
631
|
+
);
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
function environmentWarnings(hardware, pool, model) {
|
|
635
|
+
const out = [];
|
|
636
|
+
const hw = hardware || {};
|
|
637
|
+
|
|
638
|
+
if (pool.kind === "unknown") {
|
|
639
|
+
out.push(
|
|
640
|
+
warning(
|
|
641
|
+
"hardware_unknown",
|
|
642
|
+
"error",
|
|
643
|
+
"No memory or GPU facts could be measured on this machine, so this is a default, not a measurement.",
|
|
644
|
+
),
|
|
645
|
+
);
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
if (pool.kind === "system") {
|
|
649
|
+
out.push(
|
|
650
|
+
warning(
|
|
651
|
+
"no_gpu",
|
|
652
|
+
"warning",
|
|
653
|
+
"No NVIDIA GPU and no unified-memory GPU detected: inference will run on CPU. " +
|
|
654
|
+
"Expect a large latency penalty; consider the cpu-linux profile's smaller context, or a cloud profile.",
|
|
655
|
+
),
|
|
656
|
+
);
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
if (pool.kind === "unified") {
|
|
660
|
+
out.push(
|
|
661
|
+
warning(
|
|
662
|
+
"unified_memory",
|
|
663
|
+
"info",
|
|
664
|
+
pool.coherent && pool.gpuCount > 0
|
|
665
|
+
? "GPU memory and system RAM are the same physical pool on this machine (coherent memory), " +
|
|
666
|
+
"so VRAM and RAM must not be added together. Weights, KV cache, the OS, the coordinator and " +
|
|
667
|
+
"every Podman sandbox all draw on the one pool."
|
|
668
|
+
: pool.appleUnified
|
|
669
|
+
? "Apple Silicon unified memory: CPU and GPU share one pool, and macOS caps how much of it the GPU " +
|
|
670
|
+
`may hold wired (budgeted here at ${Math.round(RESERVES.unifiedGpuFraction * 100)}%). ` +
|
|
671
|
+
"There is no separate VRAM to spend."
|
|
672
|
+
: "This machine has coherent unified memory: CPU and GPU share one physical pool, so there is no " +
|
|
673
|
+
`separate VRAM to spend and only about ${Math.round(RESERVES.unifiedGpuFraction * 100)}% of the ` +
|
|
674
|
+
"pool is budgeted for model residency. Weights, KV cache, the OS, the coordinator and every " +
|
|
675
|
+
"Podman sandbox all draw on the one pool.",
|
|
676
|
+
),
|
|
677
|
+
);
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
if (hw.isWSL) {
|
|
681
|
+
out.push(
|
|
682
|
+
warning(
|
|
683
|
+
"wsl_memory",
|
|
684
|
+
"info",
|
|
685
|
+
"Running under WSL2: the memory seen here is what WSL was granted, not the host's RAM. " +
|
|
686
|
+
"Raise it in .wslconfig before treating this as the machine's capacity.",
|
|
687
|
+
),
|
|
688
|
+
);
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
if (pool.gpuCount > 1) {
|
|
692
|
+
out.push(
|
|
693
|
+
warning(
|
|
694
|
+
"multi_gpu",
|
|
695
|
+
"info",
|
|
696
|
+
`${pool.gpuCount} GPUs detected. VRAM is summed here, which assumes llama.cpp splits the model across ` +
|
|
697
|
+
"them; a single-GPU run is bounded by the smallest card instead.",
|
|
698
|
+
),
|
|
699
|
+
);
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
if (!model.found) {
|
|
703
|
+
out.push(
|
|
704
|
+
warning(
|
|
705
|
+
"unknown_model",
|
|
706
|
+
"warning",
|
|
707
|
+
"No readable GGUF was found, so the KV-cache maths uses assumed architecture numbers " +
|
|
708
|
+
`(${FALLBACK_MODEL.label}): ${model.assumed.join("; ")}. This is a guess, not a measurement. ` +
|
|
709
|
+
"Re-run after the model has been downloaded for a real number.",
|
|
710
|
+
),
|
|
711
|
+
);
|
|
712
|
+
} else if (model.assumed.length > 0 || model.truncated) {
|
|
713
|
+
out.push(
|
|
714
|
+
warning(
|
|
715
|
+
"partial_model_metadata",
|
|
716
|
+
"warning",
|
|
717
|
+
`The GGUF header was read but incomplete${model.truncated ? " (truncated)" : ""}; ` +
|
|
718
|
+
`substituted: ${model.assumed.join("; ") || "none"}.`,
|
|
719
|
+
),
|
|
720
|
+
);
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
if (model.found && model.hybridRecognized) {
|
|
724
|
+
out.push(warning("hybrid_architecture_corrected", "info", model.hybridNote));
|
|
725
|
+
} else if (model.found && model.isHybrid) {
|
|
726
|
+
out.push(
|
|
727
|
+
warning(
|
|
728
|
+
"unrecognized_hybrid_architecture",
|
|
729
|
+
"warning",
|
|
730
|
+
`${model.arch ?? "this architecture"} has recurrent/state-space layers (per its GGUF .ssm.* header ` +
|
|
731
|
+
"keys), but nomArmy has no verified full-attention ratio for it, so the KV-cache estimate below " +
|
|
732
|
+
"assumes every layer is full attention. That almost certainly OVERSTATES real memory need -- treat " +
|
|
733
|
+
"the recommendation as conservative, not tight.",
|
|
734
|
+
),
|
|
735
|
+
);
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
if (hw.podman && hw.podman.available === false) {
|
|
739
|
+
out.push(
|
|
740
|
+
warning(
|
|
741
|
+
"podman_unavailable",
|
|
742
|
+
"warning",
|
|
743
|
+
"Podman did not respond. nomArmy starts a sandbox container per job, so worker jobs will not run " +
|
|
744
|
+
"until Podman is available (on macOS, 'podman machine start'). Its memory ceiling could not be " +
|
|
745
|
+
"measured either way, so this plan assumes ZERO reservation for it -- if Podman is actually running " +
|
|
746
|
+
"and just slow to answer (e.g. a machine still starting up), the real number is available and this " +
|
|
747
|
+
"plan is optimistic until you re-run sizing once Podman responds.",
|
|
748
|
+
),
|
|
749
|
+
);
|
|
750
|
+
}
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
// The pool is sized against total RAM so the answer does not change hour to
|
|
754
|
+
// hour. That is only honest if we say when the machine cannot deliver it now.
|
|
755
|
+
const totalRam = pool?.systemTotalBytes ?? null;
|
|
756
|
+
const freeRam = pool?.systemAvailableBytes ?? null;
|
|
757
|
+
if (totalRam && freeRam && freeRam < totalRam * 0.6) {
|
|
758
|
+
out.push(
|
|
759
|
+
warning(
|
|
760
|
+
"memory_pressure",
|
|
761
|
+
"info",
|
|
762
|
+
`Sized against this machine's full ${formatBytes(totalRam)}, but only ` +
|
|
763
|
+
`${formatBytes(freeRam)} is free right now. Free memory before starting noms, ` +
|
|
764
|
+
`or size against current load instead if you need a right-now answer.`,
|
|
765
|
+
),
|
|
766
|
+
);
|
|
767
|
+
}
|
|
768
|
+
return out;
|
|
769
|
+
}
|
|
770
|
+
|
|
771
|
+
function confidenceFor(model, pool) {
|
|
772
|
+
if (!model.found || pool.kind === "unknown") return "low";
|
|
773
|
+
if (model.assumed.length > 0 || model.truncated) return "medium";
|
|
774
|
+
if (!pool.poolBytes) return "low";
|
|
775
|
+
return "high";
|
|
776
|
+
}
|
|
777
|
+
|
|
778
|
+
function assumptionsFor(model, pool, bytesPerElement) {
|
|
779
|
+
const list = [
|
|
780
|
+
`KV cache is ${bytesPerElement} bytes per element (${bytesPerElement === 2 ? "fp16, llama.cpp's default for both K and V" : "caller-supplied"}).`,
|
|
781
|
+
model.isHybrid
|
|
782
|
+
? `kv_per_slot = kv_layers * kv_heads * (key_length + value_length) * context_per_nom * bytes_per_element, ` +
|
|
783
|
+
`where kv_layers (${model.kvLayers} of ${model.layers}) is the layer count that actually does full ` +
|
|
784
|
+
"attention -- see the hybrid-architecture note above for how that count was determined."
|
|
785
|
+
: "kv_per_slot = layers * kv_heads * (key_length + value_length) * context_per_nom * bytes_per_element.",
|
|
786
|
+
`Headroom reserved: ${formatBytes(RESERVES.osBytes)} OS, ${formatBytes(RESERVES.coordinatorBytes)} coordinator, ` +
|
|
787
|
+
(pool.vmReservationBytes ? `${formatBytes(pool.vmReservationBytes)} for the Podman machine VM (half its reported ceiling, a heuristic ` +
|
|
788
|
+
`for steady-state pressure -- sandboxes run inside it and are not charged again), ` : `${formatBytes(RESERVES.sandboxPerNomBytes)} Podman sandbox per nom, `) +
|
|
789
|
+
`${formatBytes(RESERVES.runtimeOverheadBytes)} llama-server runtime, ` +
|
|
790
|
+
`${formatBytes(RESERVES.computeBufferPerSlotBytes)} compute buffers per slot, ` +
|
|
791
|
+
`and ${Math.round((1 - RESERVES.safetyFraction) * 100)}% of the pool left unallocated.`,
|
|
792
|
+
`Model weights taken as ${model.source === "gguf" ? "the GGUF file size on disk" : "an assumed figure"} ` +
|
|
793
|
+
`(${formatBytes(model.weightsBytes)}).`,
|
|
794
|
+
`Memory pool: ${pool.source}${pool.poolBytes ? ` (${formatBytes(pool.poolBytes)})` : ""}.`,
|
|
795
|
+
];
|
|
796
|
+
for (const item of model.assumed) list.push(`ASSUMED: ${item}`);
|
|
797
|
+
return list;
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
// ---------------------------------------------------------------------------
|
|
801
|
+
// recommend
|
|
802
|
+
// ---------------------------------------------------------------------------
|
|
803
|
+
|
|
804
|
+
/**
|
|
805
|
+
* Recommend context/parallel/worker settings.
|
|
806
|
+
*
|
|
807
|
+
* @param {{
|
|
808
|
+
* hardware?: object|null,
|
|
809
|
+
* gguf?: object|null,
|
|
810
|
+
* execution?: string,
|
|
811
|
+
* targetContextPerNom?: number,
|
|
812
|
+
* maxNoms?: number,
|
|
813
|
+
* bytesPerKvElement?: number
|
|
814
|
+
* }} input
|
|
815
|
+
* @returns {object}
|
|
816
|
+
*/
|
|
817
|
+
export function recommend(input = {}) {
|
|
818
|
+
const {
|
|
819
|
+
hardware = null,
|
|
820
|
+
gguf = null,
|
|
821
|
+
execution = "local",
|
|
822
|
+
targetContextPerNom = DEFAULT_TARGET_CONTEXT_PER_NOM,
|
|
823
|
+
maxNoms = MAX_NOMS,
|
|
824
|
+
bytesPerKvElement = DEFAULT_BYTES_PER_KV_ELEMENT,
|
|
825
|
+
useAvailable = false,
|
|
826
|
+
} = input;
|
|
827
|
+
|
|
828
|
+
const requestedTarget = num(targetContextPerNom) || DEFAULT_TARGET_CONTEXT_PER_NOM;
|
|
829
|
+
const nomCeiling = Math.max(1, Math.min(num(maxNoms) || MAX_NOMS, MAX_NOMS));
|
|
830
|
+
|
|
831
|
+
if (isCloudExecution(execution)) {
|
|
832
|
+
return cloudRecommendation(execution, requestedTarget);
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
const model = resolveModel(gguf);
|
|
836
|
+
const pool = resolveMemoryPool(hardware, { useAvailable: input.useAvailable === true });
|
|
837
|
+
const warnings = environmentWarnings(hardware, pool, model);
|
|
838
|
+
|
|
839
|
+
// Step down to the largest context that actually fits before giving up.
|
|
840
|
+
// Recommending an unusable configuration and burying the working ones in an
|
|
841
|
+
// "alternatives" table below is the wrong way round: the recommendation
|
|
842
|
+
// should be the best option that works, with the shortfall stated plainly.
|
|
843
|
+
let target = requestedTarget;
|
|
844
|
+
let capacity = capacityFor(pool, model, target, bytesPerKvElement);
|
|
845
|
+
let steppedDownFrom = null;
|
|
846
|
+
if (capacity.rawCapacity < 1) {
|
|
847
|
+
for (let ctx = Math.floor(target / 2); ctx >= MIN_CONTEXT_PER_NOM; ctx = Math.floor(ctx / 2)) {
|
|
848
|
+
const c = capacityFor(pool, model, ctx, bytesPerKvElement);
|
|
849
|
+
if (c.rawCapacity >= 1) {
|
|
850
|
+
steppedDownFrom = target;
|
|
851
|
+
target = ctx;
|
|
852
|
+
capacity = c;
|
|
853
|
+
break;
|
|
854
|
+
}
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
if (steppedDownFrom) {
|
|
858
|
+
warnings.push(
|
|
859
|
+
warning(
|
|
860
|
+
"context_stepped_down",
|
|
861
|
+
"warning",
|
|
862
|
+
`${formatContext(steppedDownFrom)} per nom does not fit, so the recommendation steps down to ` +
|
|
863
|
+
`${formatContext(target)}. Free memory, use a smaller quantization, or quantize the KV cache ` +
|
|
864
|
+
`to reach the ${formatContext(steppedDownFrom)} target.`,
|
|
865
|
+
),
|
|
866
|
+
);
|
|
867
|
+
}
|
|
868
|
+
const feasible = Math.min(Math.max(capacity.rawCapacity, 0), nomCeiling);
|
|
869
|
+
const noms = Math.max(1, feasible);
|
|
870
|
+
|
|
871
|
+
if (capacity.rawCapacity < 1) {
|
|
872
|
+
warnings.push(
|
|
873
|
+
warning(
|
|
874
|
+
"low_memory",
|
|
875
|
+
"error",
|
|
876
|
+
`Not even one nom at ${formatContext(target)} fits inside the reserved headroom: ` +
|
|
877
|
+
`${formatBytes(pool.poolBytes)} pool, ${formatBytes(model.weightsBytes)} weights and ` +
|
|
878
|
+
`${formatBytes(capacity.kvPerSlot)} KV per slot. Recommending 1 nom anyway, but expect swapping or an ` +
|
|
879
|
+
"out-of-memory failure. Reduce the per-nom context, use a smaller quantization, or quantize the KV cache.",
|
|
880
|
+
),
|
|
881
|
+
);
|
|
882
|
+
} else if (capacity.rawCapacity < 2 && pool.kind !== "unknown") {
|
|
883
|
+
warnings.push(
|
|
884
|
+
warning(
|
|
885
|
+
"single_nom_only",
|
|
886
|
+
"info",
|
|
887
|
+
`This machine has room for one nom at ${formatContext(target)} (limited by ${capacity.limitedBy}). ` +
|
|
888
|
+
"A second nom needs roughly " +
|
|
889
|
+
`${formatBytes(capacity.perSlotBytes)} more.`,
|
|
890
|
+
),
|
|
891
|
+
);
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
if (model.modelContextLength && target > model.modelContextLength) {
|
|
895
|
+
warnings.push(
|
|
896
|
+
warning(
|
|
897
|
+
"context_exceeds_model",
|
|
898
|
+
"warning",
|
|
899
|
+
`Target ${formatContext(target)} per nom exceeds the model's trained context of ` +
|
|
900
|
+
`${formatContext(model.modelContextLength)}.`,
|
|
901
|
+
),
|
|
902
|
+
);
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
const llamaParallel = noms;
|
|
906
|
+
const maxWorkers = noms; // Deliberately equal: see the oversubscription note.
|
|
907
|
+
const contextTotal = target * llamaParallel;
|
|
908
|
+
const memory = memoryBreakdown(pool, model, noms, target, bytesPerKvElement);
|
|
909
|
+
|
|
910
|
+
// Capacity planning sizes against total RAM, because configuration persists
|
|
911
|
+
// and should not change with whatever happens to be open. But a plan you
|
|
912
|
+
// cannot act on right now is a trap: this check compares what must actually
|
|
913
|
+
// be resident against what is free, and says how much to free before starting.
|
|
914
|
+
const residentNeed =
|
|
915
|
+
memory.modelWeightsBytes + memory.kvBytesTotal +
|
|
916
|
+
(RESERVES.computeBufferPerSlotBytes * noms) + RESERVES.runtimeOverheadBytes;
|
|
917
|
+
const freeNow = pool.systemAvailableBytes;
|
|
918
|
+
if (freeNow && residentNeed > freeNow) {
|
|
919
|
+
warnings.push(
|
|
920
|
+
warning(
|
|
921
|
+
"cannot_start_now",
|
|
922
|
+
"error",
|
|
923
|
+
`This configuration needs about ${formatBytes(residentNeed)} resident, but only ` +
|
|
924
|
+
`${formatBytes(freeNow)} is free right now. Free at least ` +
|
|
925
|
+
`${formatBytes(residentNeed - freeNow)} before starting, or it will not load ` +
|
|
926
|
+
`and may take other processes down with it.`,
|
|
927
|
+
),
|
|
928
|
+
);
|
|
929
|
+
}
|
|
930
|
+
|
|
931
|
+
// "More noms" (the recommendation above) answers "what fits in memory" --
|
|
932
|
+
// this project's own README is explicit that this is a different question
|
|
933
|
+
// from "is this fast enough to be useful", and that raising worker count
|
|
934
|
+
// is the one documented contention/speed cost sizing has no model for.
|
|
935
|
+
// Every profile actually shipped in config/profiles/*.env uses 1-2 workers
|
|
936
|
+
// regardless of how much more memory-headroom exists, so "nominal" is that
|
|
937
|
+
// convention made explicit: 1 worker at the SAME target context, not a
|
|
938
|
+
// separately computed number.
|
|
939
|
+
//
|
|
940
|
+
// NOT always guaranteed to fit, caught by review: the step-down loop above
|
|
941
|
+
// only updates `target`/`capacity` when it actually FINDS a context that
|
|
942
|
+
// fits at least 1 nom. On a machine where nothing fits even at
|
|
943
|
+
// MIN_CONTEXT_PER_NOM (the genuine "NOTHING FITS" case cmdSizing already
|
|
944
|
+
// has to handle for the primary recommendation), the loop exhausts without
|
|
945
|
+
// ever breaking, `target` stays at its original infeasible value, and
|
|
946
|
+
// `capacity.rawCapacity` stays below 1 -- nominal would silently inherit
|
|
947
|
+
// that same infeasibility with no way for a caller to tell. `fits` makes
|
|
948
|
+
// that visible, mirroring the same field `alternatives` entries already
|
|
949
|
+
// carry for exactly this reason.
|
|
950
|
+
const nominal = {
|
|
951
|
+
label: `1 nom @ ${formatContext(target)} (nominal -- matches this project's own shipped profiles, regardless of how many more would fit)`,
|
|
952
|
+
contextPerNom: target,
|
|
953
|
+
contextTotal: target,
|
|
954
|
+
llamaParallel: 1,
|
|
955
|
+
maxWorkers: 1,
|
|
956
|
+
env: { NOMARMY_LLAMA_CONTEXT: target, NOMARMY_LLAMA_PARALLEL: 1, NOMARMY_MAX_WORKERS: 1 },
|
|
957
|
+
summary: `1 nom at ${formatContext(target)} -> NOMARMY_LLAMA_CONTEXT=${target}, NOMARMY_LLAMA_PARALLEL=1, NOMARMY_MAX_WORKERS=1.`,
|
|
958
|
+
fits: capacity.rawCapacity >= 1,
|
|
959
|
+
// True when "more noms" already recommended exactly 1 -- nothing to
|
|
960
|
+
// choose between in that case, both options are the same configuration.
|
|
961
|
+
sameAsRecommended: noms === 1,
|
|
962
|
+
};
|
|
963
|
+
|
|
964
|
+
return {
|
|
965
|
+
kind: "local",
|
|
966
|
+
execution: "local",
|
|
967
|
+
hardwareDerived: true,
|
|
968
|
+
contextPerNom: target,
|
|
969
|
+
contextTotal,
|
|
970
|
+
llamaParallel,
|
|
971
|
+
maxWorkers,
|
|
972
|
+
targetContextPerNom: target,
|
|
973
|
+
env: {
|
|
974
|
+
NOMARMY_LLAMA_CONTEXT: contextTotal,
|
|
975
|
+
NOMARMY_LLAMA_PARALLEL: llamaParallel,
|
|
976
|
+
NOMARMY_MAX_WORKERS: maxWorkers,
|
|
977
|
+
},
|
|
978
|
+
model: {
|
|
979
|
+
source: model.source,
|
|
980
|
+
arch: model.arch,
|
|
981
|
+
layers: model.layers,
|
|
982
|
+
kvHeads: model.kvHeads,
|
|
983
|
+
keyLength: model.keyLength,
|
|
984
|
+
valueLength: model.valueLength,
|
|
985
|
+
weightsBytes: model.weightsBytes,
|
|
986
|
+
modelContextLength: model.modelContextLength,
|
|
987
|
+
},
|
|
988
|
+
memory,
|
|
989
|
+
limitedBy: capacity.limitedBy,
|
|
990
|
+
confidence: confidenceFor(model, pool),
|
|
991
|
+
assumptions: assumptionsFor(model, pool, bytesPerKvElement),
|
|
992
|
+
warnings,
|
|
993
|
+
alternatives: buildAlternatives(pool, model, target, bytesPerKvElement, nomCeiling, noms),
|
|
994
|
+
nominal,
|
|
995
|
+
summary:
|
|
996
|
+
`${noms} nom(s) at ${formatContext(target)} each -> NOMARMY_LLAMA_CONTEXT=${contextTotal}, ` +
|
|
997
|
+
`NOMARMY_LLAMA_PARALLEL=${llamaParallel}, NOMARMY_MAX_WORKERS=${maxWorkers}.`,
|
|
998
|
+
};
|
|
999
|
+
}
|
|
1000
|
+
|
|
1001
|
+
/**
|
|
1002
|
+
* Size for an EXACT worker count the caller picked, rather than solving for
|
|
1003
|
+
* the max that fits ("more noms") or the fixed shipped-profile convention
|
|
1004
|
+
* ("nominal"). Same step-down behavior as recommend(): the requested target
|
|
1005
|
+
* context is used if `noms` fits at it, otherwise context steps down until
|
|
1006
|
+
* `noms` fits or the machine genuinely cannot run that many at all (fits:
|
|
1007
|
+
* false, never silently substituted with a smaller count the caller didn't
|
|
1008
|
+
* ask for -- that decision belongs to the human who typed the number).
|
|
1009
|
+
*
|
|
1010
|
+
* @param {{
|
|
1011
|
+
* hardware?: object|null, gguf?: object|null, execution?: string,
|
|
1012
|
+
* noms: number, targetContextPerNom?: number, bytesPerKvElement?: number
|
|
1013
|
+
* }} input
|
|
1014
|
+
* @returns {object}
|
|
1015
|
+
*/
|
|
1016
|
+
export function customRecommendation(input = {}) {
|
|
1017
|
+
const {
|
|
1018
|
+
hardware = null, gguf = null, execution = "local",
|
|
1019
|
+
targetContextPerNom = DEFAULT_TARGET_CONTEXT_PER_NOM,
|
|
1020
|
+
bytesPerKvElement = DEFAULT_BYTES_PER_KV_ELEMENT,
|
|
1021
|
+
} = input;
|
|
1022
|
+
const noms = Math.max(1, Math.floor(num(input.noms) || 1));
|
|
1023
|
+
const requestedTarget = num(targetContextPerNom) || DEFAULT_TARGET_CONTEXT_PER_NOM;
|
|
1024
|
+
|
|
1025
|
+
if (isCloudExecution(execution)) {
|
|
1026
|
+
// No local slots to divide -- concurrency is quota/budget-bound, so the
|
|
1027
|
+
// requested count is honored as-is, same as cloudRecommendation's own
|
|
1028
|
+
// NOMARMY_MAX_WORKERS.
|
|
1029
|
+
return {
|
|
1030
|
+
kind: "cloud", execution: String(execution).toLowerCase(), requestedNoms: noms,
|
|
1031
|
+
contextPerNom: requestedTarget, contextTotal: null, llamaParallel: null, maxWorkers: noms,
|
|
1032
|
+
env: { NOMARMY_MAX_WORKERS: noms }, fits: true, steppedDownFrom: null, memory: null,
|
|
1033
|
+
confidence: "not-applicable", limitedBy: "api-quota-and-budget",
|
|
1034
|
+
summary: `Hosted execution: NOMARMY_MAX_WORKERS=${noms} as requested, bounded by quota/budget, not local memory.`,
|
|
1035
|
+
};
|
|
1036
|
+
}
|
|
1037
|
+
|
|
1038
|
+
const model = resolveModel(gguf);
|
|
1039
|
+
const pool = resolveMemoryPool(hardware, {});
|
|
1040
|
+
let target = requestedTarget;
|
|
1041
|
+
let capacity = capacityFor(pool, model, target, bytesPerKvElement);
|
|
1042
|
+
let fits = capacity.rawCapacity >= noms;
|
|
1043
|
+
let steppedDownFrom = null;
|
|
1044
|
+
if (!fits) {
|
|
1045
|
+
for (let ctx = Math.floor(target / 2); ctx >= MIN_CONTEXT_PER_NOM; ctx = Math.floor(ctx / 2)) {
|
|
1046
|
+
const c = capacityFor(pool, model, ctx, bytesPerKvElement);
|
|
1047
|
+
if (c.rawCapacity >= noms) { steppedDownFrom = target; target = ctx; capacity = c; fits = true; break; }
|
|
1048
|
+
}
|
|
1049
|
+
}
|
|
1050
|
+
const contextTotal = target * noms;
|
|
1051
|
+
const memory = memoryBreakdown(pool, model, noms, target, bytesPerKvElement);
|
|
1052
|
+
return {
|
|
1053
|
+
kind: "local", execution, requestedNoms: noms,
|
|
1054
|
+
contextPerNom: target, contextTotal, llamaParallel: noms, maxWorkers: noms,
|
|
1055
|
+
env: { NOMARMY_LLAMA_CONTEXT: contextTotal, NOMARMY_LLAMA_PARALLEL: noms, NOMARMY_MAX_WORKERS: noms },
|
|
1056
|
+
fits, steppedDownFrom, memory, limitedBy: capacity.limitedBy,
|
|
1057
|
+
confidence: confidenceFor(model, pool),
|
|
1058
|
+
assumptions: assumptionsFor(model, pool, bytesPerKvElement),
|
|
1059
|
+
summary: fits
|
|
1060
|
+
? `${noms} nom(s) at ${formatContext(target)} each -> NOMARMY_LLAMA_CONTEXT=${contextTotal}, ` +
|
|
1061
|
+
`NOMARMY_LLAMA_PARALLEL=${noms}, NOMARMY_MAX_WORKERS=${noms}.`
|
|
1062
|
+
: `${noms} nom(s) does not fit on this machine even at the minimum context (${formatContext(MIN_CONTEXT_PER_NOM)}).`,
|
|
1063
|
+
};
|
|
1064
|
+
}
|
|
1065
|
+
|
|
1066
|
+
/**
|
|
1067
|
+
* Cloud execution: local hardware is irrelevant. There are no inference slots
|
|
1068
|
+
* to divide, and the bound is API quota and budget.
|
|
1069
|
+
*/
|
|
1070
|
+
function cloudRecommendation(execution, target) {
|
|
1071
|
+
return {
|
|
1072
|
+
kind: "cloud",
|
|
1073
|
+
execution: String(execution).toLowerCase(),
|
|
1074
|
+
hardwareDerived: false,
|
|
1075
|
+
contextPerNom: target,
|
|
1076
|
+
contextTotal: null,
|
|
1077
|
+
llamaParallel: null,
|
|
1078
|
+
maxWorkers: CLOUD_DEFAULT_MAX_WORKERS,
|
|
1079
|
+
targetContextPerNom: target,
|
|
1080
|
+
env: {
|
|
1081
|
+
NOMARMY_MAX_WORKERS: CLOUD_DEFAULT_MAX_WORKERS,
|
|
1082
|
+
},
|
|
1083
|
+
model: null,
|
|
1084
|
+
memory: null,
|
|
1085
|
+
limitedBy: "api-quota-and-budget",
|
|
1086
|
+
confidence: "not-applicable",
|
|
1087
|
+
assumptions: [
|
|
1088
|
+
"Execution is hosted, so no local weights, KV cache or inference slots exist on this machine.",
|
|
1089
|
+
"NOMARMY_LLAMA_CONTEXT and NOMARMY_LLAMA_PARALLEL do not apply and are left unset.",
|
|
1090
|
+
`NOMARMY_MAX_WORKERS=${CLOUD_DEFAULT_MAX_WORKERS} is the profile default, not a hardware-derived number.`,
|
|
1091
|
+
],
|
|
1092
|
+
warnings: [
|
|
1093
|
+
warning(
|
|
1094
|
+
"cloud_execution",
|
|
1095
|
+
"info",
|
|
1096
|
+
`NOMARMY_EXECUTION=${execution}: worker concurrency is bounded by your Bedrock TPM quota and your budget, ` +
|
|
1097
|
+
"not by this machine. Measure first-pass accept rate before raising it; coordinator review tokens " +
|
|
1098
|
+
"dominate worker tokens.",
|
|
1099
|
+
),
|
|
1100
|
+
warning(
|
|
1101
|
+
"cloud_data_flow",
|
|
1102
|
+
"info",
|
|
1103
|
+
"Cloud profiles send repository content off the machine. That is a data-residency decision, not a sizing one.",
|
|
1104
|
+
),
|
|
1105
|
+
],
|
|
1106
|
+
alternatives: [],
|
|
1107
|
+
// "More noms" vs "nominal" is a local memory/contention tradeoff; a
|
|
1108
|
+
// cloud profile has no local slots to trade off in the first place.
|
|
1109
|
+
nominal: null,
|
|
1110
|
+
summary:
|
|
1111
|
+
`Hosted execution: no local slots. NOMARMY_MAX_WORKERS=${CLOUD_DEFAULT_MAX_WORKERS} bounded by quota/budget.`,
|
|
1112
|
+
};
|
|
1113
|
+
}
|
|
1114
|
+
|
|
1115
|
+
/**
|
|
1116
|
+
* Show the tradeoff rather than handing back one number: fewer noms with more
|
|
1117
|
+
* context each, or more noms with less.
|
|
1118
|
+
*/
|
|
1119
|
+
function buildAlternatives(pool, model, target, bytesPerElement, nomCeiling, recommendedNoms) {
|
|
1120
|
+
const contexts = [];
|
|
1121
|
+
for (let ctx = target; ctx >= MIN_CONTEXT_PER_NOM; ctx = Math.floor(ctx / 2)) {
|
|
1122
|
+
contexts.push(ctx);
|
|
1123
|
+
if (contexts.length >= 4) break;
|
|
1124
|
+
}
|
|
1125
|
+
|
|
1126
|
+
const out = [];
|
|
1127
|
+
const seen = new Set();
|
|
1128
|
+
for (const ctx of contexts) {
|
|
1129
|
+
const capacity = capacityFor(pool, model, ctx, bytesPerElement);
|
|
1130
|
+
const noms = Math.max(1, Math.min(Math.max(capacity.rawCapacity, 0), nomCeiling));
|
|
1131
|
+
const key = `${noms}@${ctx}`;
|
|
1132
|
+
if (seen.has(key)) continue;
|
|
1133
|
+
seen.add(key);
|
|
1134
|
+
const memory = memoryBreakdown(pool, model, noms, ctx, bytesPerElement);
|
|
1135
|
+
out.push({
|
|
1136
|
+
label: `${noms} nom${noms === 1 ? "" : "s"} @ ${formatContext(ctx)}`,
|
|
1137
|
+
noms,
|
|
1138
|
+
contextPerNom: ctx,
|
|
1139
|
+
contextTotal: ctx * noms,
|
|
1140
|
+
llamaParallel: noms,
|
|
1141
|
+
maxWorkers: noms,
|
|
1142
|
+
estimatedTotalBytes: memory.estimatedTotalBytes,
|
|
1143
|
+
headroomBytes: memory.headroomBytes,
|
|
1144
|
+
fits: capacity.rawCapacity >= 1,
|
|
1145
|
+
recommended: ctx === target && noms === recommendedNoms,
|
|
1146
|
+
});
|
|
1147
|
+
}
|
|
1148
|
+
return out;
|
|
1149
|
+
}
|
|
1150
|
+
|
|
1151
|
+
// ---------------------------------------------------------------------------
|
|
1152
|
+
// evaluateConfig
|
|
1153
|
+
// ---------------------------------------------------------------------------
|
|
1154
|
+
|
|
1155
|
+
/**
|
|
1156
|
+
* Check an EXISTING profile's numbers and return the same warning shapes as
|
|
1157
|
+
* `recommend()`. This is what tells a user their current profile is
|
|
1158
|
+
* oversubscribed, over-committed on memory, or quietly giving each nom half the
|
|
1159
|
+
* context they think it has.
|
|
1160
|
+
*
|
|
1161
|
+
* @param {{
|
|
1162
|
+
* hardware?: object|null,
|
|
1163
|
+
* gguf?: object|null,
|
|
1164
|
+
* contextTotal: number,
|
|
1165
|
+
* llamaParallel: number,
|
|
1166
|
+
* maxWorkers: number,
|
|
1167
|
+
* targetContextPerNom?: number,
|
|
1168
|
+
* execution?: string,
|
|
1169
|
+
* bytesPerKvElement?: number
|
|
1170
|
+
* }} input
|
|
1171
|
+
* @returns {object}
|
|
1172
|
+
*/
|
|
1173
|
+
export function evaluateConfig(input = {}) {
|
|
1174
|
+
const {
|
|
1175
|
+
hardware = null,
|
|
1176
|
+
gguf = null,
|
|
1177
|
+
contextTotal,
|
|
1178
|
+
llamaParallel,
|
|
1179
|
+
maxWorkers,
|
|
1180
|
+
targetContextPerNom = DEFAULT_TARGET_CONTEXT_PER_NOM,
|
|
1181
|
+
execution = "local",
|
|
1182
|
+
bytesPerKvElement = DEFAULT_BYTES_PER_KV_ELEMENT,
|
|
1183
|
+
useAvailable = false,
|
|
1184
|
+
} = input;
|
|
1185
|
+
|
|
1186
|
+
const target = num(targetContextPerNom) || DEFAULT_TARGET_CONTEXT_PER_NOM;
|
|
1187
|
+
const warnings = [];
|
|
1188
|
+
|
|
1189
|
+
if (isCloudExecution(execution)) {
|
|
1190
|
+
if (num(contextTotal) || num(llamaParallel)) {
|
|
1191
|
+
warnings.push(
|
|
1192
|
+
warning(
|
|
1193
|
+
"cloud_execution",
|
|
1194
|
+
"info",
|
|
1195
|
+
`NOMARMY_EXECUTION=${execution}: NOMARMY_LLAMA_CONTEXT and NOMARMY_LLAMA_PARALLEL are ignored, ` +
|
|
1196
|
+
"since no llama-server runs on this machine. Worker concurrency is bounded by quota and budget.",
|
|
1197
|
+
),
|
|
1198
|
+
);
|
|
1199
|
+
}
|
|
1200
|
+
return {
|
|
1201
|
+
kind: "cloud",
|
|
1202
|
+
execution: String(execution).toLowerCase(),
|
|
1203
|
+
hardwareDerived: false,
|
|
1204
|
+
ok: true,
|
|
1205
|
+
contextPerNom: null,
|
|
1206
|
+
contextTotal: num(contextTotal) || null,
|
|
1207
|
+
llamaParallel: num(llamaParallel) || null,
|
|
1208
|
+
maxWorkers: num(maxWorkers) || null,
|
|
1209
|
+
memory: null,
|
|
1210
|
+
confidence: "not-applicable",
|
|
1211
|
+
assumptions: ["Hosted execution: local memory and inference slots do not apply."],
|
|
1212
|
+
warnings,
|
|
1213
|
+
};
|
|
1214
|
+
}
|
|
1215
|
+
|
|
1216
|
+
const parallel = num(llamaParallel);
|
|
1217
|
+
const total = num(contextTotal);
|
|
1218
|
+
const workers = num(maxWorkers);
|
|
1219
|
+
|
|
1220
|
+
if (!parallel || !total || !workers) {
|
|
1221
|
+
warnings.push(
|
|
1222
|
+
warning(
|
|
1223
|
+
"invalid_config",
|
|
1224
|
+
"error",
|
|
1225
|
+
"contextTotal, llamaParallel and maxWorkers must each be a positive number to be evaluated " +
|
|
1226
|
+
`(got contextTotal=${contextTotal}, llamaParallel=${llamaParallel}, maxWorkers=${maxWorkers}).`,
|
|
1227
|
+
),
|
|
1228
|
+
);
|
|
1229
|
+
return {
|
|
1230
|
+
kind: "local",
|
|
1231
|
+
execution: "local",
|
|
1232
|
+
hardwareDerived: true,
|
|
1233
|
+
ok: false,
|
|
1234
|
+
contextPerNom: null,
|
|
1235
|
+
contextTotal: total || null,
|
|
1236
|
+
llamaParallel: parallel || null,
|
|
1237
|
+
maxWorkers: workers || null,
|
|
1238
|
+
memory: null,
|
|
1239
|
+
confidence: "low",
|
|
1240
|
+
assumptions: [],
|
|
1241
|
+
warnings,
|
|
1242
|
+
};
|
|
1243
|
+
}
|
|
1244
|
+
|
|
1245
|
+
const contextPerNom = Math.floor(total / parallel);
|
|
1246
|
+
const model = resolveModel(gguf);
|
|
1247
|
+
const pool = resolveMemoryPool(hardware, { useAvailable: input.useAvailable === true });
|
|
1248
|
+
warnings.push(...environmentWarnings(hardware, pool, model));
|
|
1249
|
+
|
|
1250
|
+
if (contextPerNom < target) {
|
|
1251
|
+
warnings.push(contextBelowTargetWarning(contextPerNom, target, total, parallel));
|
|
1252
|
+
}
|
|
1253
|
+
if (total % parallel !== 0) {
|
|
1254
|
+
warnings.push(
|
|
1255
|
+
warning(
|
|
1256
|
+
"context_not_divisible",
|
|
1257
|
+
"info",
|
|
1258
|
+
`NOMARMY_LLAMA_CONTEXT=${total} does not divide evenly by ${parallel} slots; ` +
|
|
1259
|
+
`each nom gets ${contextPerNom} tokens and ${total % parallel} are wasted.`,
|
|
1260
|
+
),
|
|
1261
|
+
);
|
|
1262
|
+
}
|
|
1263
|
+
if (workers > parallel) {
|
|
1264
|
+
warnings.push(oversubscriptionWarning(workers, parallel));
|
|
1265
|
+
} else if (workers < parallel) {
|
|
1266
|
+
warnings.push(
|
|
1267
|
+
warning(
|
|
1268
|
+
"idle_slots",
|
|
1269
|
+
"info",
|
|
1270
|
+
`NOMARMY_MAX_WORKERS=${workers} is below NOMARMY_LLAMA_PARALLEL=${parallel}: ` +
|
|
1271
|
+
`${parallel - workers} inference slot(s) hold reserved KV cache that no nom will ever use. ` +
|
|
1272
|
+
"Lower the parallel count to give the remaining noms more context each.",
|
|
1273
|
+
),
|
|
1274
|
+
);
|
|
1275
|
+
}
|
|
1276
|
+
|
|
1277
|
+
const memory = memoryBreakdown(pool, model, parallel, contextPerNom, bytesPerKvElement);
|
|
1278
|
+
if (memory.fits === false) {
|
|
1279
|
+
warnings.push(
|
|
1280
|
+
warning(
|
|
1281
|
+
"memory_over_commit",
|
|
1282
|
+
"error",
|
|
1283
|
+
`This configuration is estimated at ${formatBytes(memory.estimatedTotalBytes)} against a ` +
|
|
1284
|
+
`${formatBytes(memory.budgetBytes)} budget (${formatBytes(pool.poolBytes)} pool at ` +
|
|
1285
|
+
`${Math.round(RESERVES.safetyFraction * 100)}%): over by ${formatBytes(-memory.headroomBytes)}. ` +
|
|
1286
|
+
"Reduce NOMARMY_LLAMA_CONTEXT or NOMARMY_LLAMA_PARALLEL.",
|
|
1287
|
+
),
|
|
1288
|
+
);
|
|
1289
|
+
}
|
|
1290
|
+
|
|
1291
|
+
const ok = !warnings.some((w) => w.severity === "error");
|
|
1292
|
+
return {
|
|
1293
|
+
kind: "local",
|
|
1294
|
+
execution: "local",
|
|
1295
|
+
hardwareDerived: true,
|
|
1296
|
+
ok,
|
|
1297
|
+
contextPerNom,
|
|
1298
|
+
contextTotal: total,
|
|
1299
|
+
llamaParallel: parallel,
|
|
1300
|
+
maxWorkers: workers,
|
|
1301
|
+
targetContextPerNom: target,
|
|
1302
|
+
model: {
|
|
1303
|
+
source: model.source,
|
|
1304
|
+
arch: model.arch,
|
|
1305
|
+
layers: model.layers,
|
|
1306
|
+
kvHeads: model.kvHeads,
|
|
1307
|
+
keyLength: model.keyLength,
|
|
1308
|
+
valueLength: model.valueLength,
|
|
1309
|
+
weightsBytes: model.weightsBytes,
|
|
1310
|
+
},
|
|
1311
|
+
memory,
|
|
1312
|
+
confidence: confidenceFor(model, pool),
|
|
1313
|
+
assumptions: assumptionsFor(model, pool, bytesPerKvElement),
|
|
1314
|
+
warnings,
|
|
1315
|
+
summary:
|
|
1316
|
+
`-c ${total} / -np ${parallel} = ${formatContext(contextPerNom)} per nom, ` +
|
|
1317
|
+
`${workers} worker(s), estimated ${formatBytes(memory.estimatedTotalBytes)} of ` +
|
|
1318
|
+
`${formatBytes(memory.budgetBytes)} budget.`,
|
|
1319
|
+
};
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
export default { recommend, customRecommendation, evaluateConfig, RESERVES, DEFAULT_TARGET_CONTEXT_PER_NOM, KV_CACHE_TYPE_BYTES, bytesPerKvElementForCacheTypes };
|