nomarmy 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +202 -0
  2. package/NOTICE +25 -0
  3. package/README.md +484 -0
  4. package/bin/nomarmy.mjs +2248 -0
  5. package/config/agents.yml.example +63 -0
  6. package/config/common.env +31 -0
  7. package/config/profiles/bedrock-cheap.env +26 -0
  8. package/config/profiles/bedrock.env +28 -0
  9. package/config/profiles/cpu-linux.env +8 -0
  10. package/config/profiles/dgx-spark.env +12 -0
  11. package/config/profiles/macbook-pro.env +9 -0
  12. package/config/profiles/nvidia-linux.env +9 -0
  13. package/docker/Dockerfile +15 -0
  14. package/docker/Dockerfile.go +29 -0
  15. package/docker/Dockerfile.rust +19 -0
  16. package/e2e.sh +153 -0
  17. package/install.sh +125 -0
  18. package/lib/agents.mjs +285 -0
  19. package/lib/army.mjs +400 -0
  20. package/lib/budget.mjs +368 -0
  21. package/lib/claude-transcript.mjs +150 -0
  22. package/lib/config.mjs +193 -0
  23. package/lib/connect.mjs +409 -0
  24. package/lib/coordinator-instructions.mjs +23 -0
  25. package/lib/decompose.mjs +389 -0
  26. package/lib/dispatch-config.mjs +164 -0
  27. package/lib/dispatch-schema.mjs +280 -0
  28. package/lib/doctor.mjs +443 -0
  29. package/lib/evidence.mjs +679 -0
  30. package/lib/gguf.mjs +589 -0
  31. package/lib/hardware.mjs +476 -0
  32. package/lib/health.mjs +278 -0
  33. package/lib/model-catalog.mjs +71 -0
  34. package/lib/notifier-app.mjs +95 -0
  35. package/lib/notify.mjs +66 -0
  36. package/lib/openclaw-config.mjs +65 -0
  37. package/lib/openclaw-errors.mjs +40 -0
  38. package/lib/propose.mjs +110 -0
  39. package/lib/prune.mjs +77 -0
  40. package/lib/repo-query.mjs +267 -0
  41. package/lib/runs.mjs +150 -0
  42. package/lib/sabotage.mjs +128 -0
  43. package/lib/sandbox-images.mjs +434 -0
  44. package/lib/scan.mjs +1538 -0
  45. package/lib/schema.mjs +288 -0
  46. package/lib/scout.mjs +544 -0
  47. package/lib/sizing.mjs +1322 -0
  48. package/lib/slots.mjs +112 -0
  49. package/lib/statusline.mjs +126 -0
  50. package/lib/subscription-config.mjs +68 -0
  51. package/lib/subscription-setup.mjs +217 -0
  52. package/lib/transcript.mjs +195 -0
  53. package/lib/verify.mjs +700 -0
  54. package/mcp/server.mjs +4206 -0
  55. package/notifier/icon.swift +34 -0
  56. package/notifier/main.swift +52 -0
  57. package/notifier/nomarmy-icon.png +0 -0
  58. package/package.json +67 -0
  59. package/playbooks/feature.md +43 -0
  60. package/policies/coder.md +49 -0
  61. package/policies/orchestrator.md +35 -0
  62. package/policies/reviewer.md +35 -0
  63. package/policies/scout.md +65 -0
  64. package/scripts/configure-openclaw.sh +96 -0
  65. package/scripts/configure-orchestrator.sh +84 -0
  66. package/scripts/install-llama-cpp.sh +16 -0
  67. package/scripts/lib.sh +198 -0
  68. package/scripts/select-model.mjs +96 -0
  69. package/scripts/select-model.sh +4 -0
  70. package/scripts/setup-sandbox.sh +38 -0
  71. package/scripts/start-inference.sh +46 -0
  72. package/scripts/stop-inference.sh +5 -0
  73. package/scripts/uninstall.sh +6 -0
  74. package/scripts/verify-install.sh +68 -0
package/lib/gguf.mjs ADDED
@@ -0,0 +1,589 @@
1
+ // Minimal GGUF header reader (v1.3).
2
+ //
3
+ // The accurate way to size a KV cache is the model's own architecture, and a
4
+ // GGUF file states it in its header. This reader parses *only* the header
5
+ // key/value block. It never reads tensor data, never memory-maps the file, and
6
+ // never loads weights.
7
+ //
8
+ // Layout (little-endian):
9
+ // magic 4 bytes "GGUF"
10
+ // version u32 1, 2 or 3
11
+ // tensor_count u64 (u32 in version 1)
12
+ // metadata_kv_cnt u64 (u32 in version 1)
13
+ // then metadata_kv_cnt records of:
14
+ // key gguf_string (u64 length + raw bytes, u32 length in v1)
15
+ // value_type u32 (see GGUF_TYPE)
16
+ // value typed
17
+ //
18
+ // Defensive posture: a fresh nomArmy install has no model downloaded yet. That
19
+ // is the common case, not an error. Anything unreadable, absent or malformed
20
+ // returns `{ found: false, ... }` with a reason; this function never throws.
21
+
22
+ import fs from "node:fs";
23
+ import os from "node:os";
24
+ import path from "node:path";
25
+
26
+ /** GGUF metadata value types. */
27
+ export const GGUF_TYPE = Object.freeze({
28
+ UINT8: 0,
29
+ INT8: 1,
30
+ UINT16: 2,
31
+ INT16: 3,
32
+ UINT32: 4,
33
+ INT32: 5,
34
+ FLOAT32: 6,
35
+ BOOL: 7,
36
+ STRING: 8,
37
+ ARRAY: 9,
38
+ UINT64: 10,
39
+ INT64: 11,
40
+ FLOAT64: 12,
41
+ });
42
+
43
+ const SCALAR_SIZE = Object.freeze({
44
+ 0: 1, // UINT8
45
+ 1: 1, // INT8
46
+ 2: 2, // UINT16
47
+ 3: 2, // INT16
48
+ 4: 4, // UINT32
49
+ 5: 4, // INT32
50
+ 6: 4, // FLOAT32
51
+ 7: 1, // BOOL
52
+ 10: 8, // UINT64
53
+ 11: 8, // INT64
54
+ 12: 8, // FLOAT64
55
+ });
56
+
57
+ /** How far into the file we are willing to walk before giving up. */
58
+ export const DEFAULT_HEADER_BYTE_LIMIT = 64 * 1024 * 1024;
59
+ /** Sliding window used for the small reads (keys, scalars, short strings). */
60
+ const WINDOW_BYTES = 256 * 1024;
61
+ /** A single string longer than this is treated as malformed rather than read. */
62
+ const MAX_STRING_BYTES = 4 * 1024 * 1024;
63
+ /** Guard rails against a corrupt header claiming absurd counts. */
64
+ const MAX_KV_COUNT = 1_000_000;
65
+ const MAX_TENSOR_COUNT = 10_000_000;
66
+ const MAX_ARRAY_DEPTH = 4;
67
+
68
+ /** Thrown internally when the header runs past the byte limit or file end. */
69
+ class TruncatedHeader extends Error {
70
+ constructor(message) {
71
+ super(message);
72
+ this.name = "TruncatedHeader";
73
+ }
74
+ }
75
+
76
+ /** Thrown internally when the bytes do not describe a valid GGUF header. */
77
+ class MalformedHeader extends Error {
78
+ constructor(message) {
79
+ super(message);
80
+ this.name = "MalformedHeader";
81
+ }
82
+ }
83
+
84
+ /**
85
+ * A bounded forward cursor over a file descriptor. It keeps a small window in
86
+ * memory and can `skip` arbitrarily large regions (vocabulary arrays run to
87
+ * megabytes) without ever holding them.
88
+ */
89
+ class Cursor {
90
+ /**
91
+ * @param {number} fd
92
+ * @param {number} fileSize
93
+ * @param {number} limit absolute offset we refuse to read past
94
+ */
95
+ constructor(fd, fileSize, limit) {
96
+ this.fd = fd;
97
+ this.fileSize = fileSize;
98
+ this.limit = Math.min(limit, fileSize);
99
+ this.pos = 0;
100
+ this.window = Buffer.alloc(0);
101
+ this.windowStart = 0;
102
+ }
103
+
104
+ /** Ensure `n` bytes from the current position are in the window. */
105
+ ensure(n) {
106
+ if (n > WINDOW_BYTES) {
107
+ throw new MalformedHeader(`refusing to buffer ${n} bytes for a single field`);
108
+ }
109
+ if (this.pos + n > this.limit) {
110
+ throw new TruncatedHeader(
111
+ `header field at offset ${this.pos} extends past the ${this.limit}-byte read limit`,
112
+ );
113
+ }
114
+ const end = this.windowStart + this.window.length;
115
+ if (this.pos >= this.windowStart && this.pos + n <= end) return;
116
+ const want = Math.min(WINDOW_BYTES, this.fileSize - this.pos);
117
+ const buf = Buffer.alloc(want);
118
+ const read = fs.readSync(this.fd, buf, 0, want, this.pos);
119
+ this.window = buf.subarray(0, read);
120
+ this.windowStart = this.pos;
121
+ if (read < n) {
122
+ throw new TruncatedHeader(`short read at offset ${this.pos}`);
123
+ }
124
+ }
125
+
126
+ /** Advance without reading. Used to step over tensor-free bulk arrays. */
127
+ skip(n) {
128
+ if (n < 0) throw new MalformedHeader(`negative skip (${n})`);
129
+ if (this.pos + n > this.limit) {
130
+ throw new TruncatedHeader(
131
+ `skipping ${n} bytes from ${this.pos} would pass the ${this.limit}-byte read limit`,
132
+ );
133
+ }
134
+ this.pos += n;
135
+ }
136
+
137
+ slice(n) {
138
+ this.ensure(n);
139
+ const offset = this.pos - this.windowStart;
140
+ const out = this.window.subarray(offset, offset + n);
141
+ this.pos += n;
142
+ return out;
143
+ }
144
+
145
+ u8() {
146
+ return this.slice(1).readUInt8(0);
147
+ }
148
+
149
+ u32() {
150
+ return this.slice(4).readUInt32LE(0);
151
+ }
152
+
153
+ /** u64 as a Number, rejecting values that would lose precision. */
154
+ u64() {
155
+ const value = this.slice(8).readBigUInt64LE(0);
156
+ if (value > BigInt(Number.MAX_SAFE_INTEGER)) {
157
+ throw new MalformedHeader(`64-bit length ${value} exceeds the safe integer range`);
158
+ }
159
+ return Number(value);
160
+ }
161
+ }
162
+
163
+ /**
164
+ * GGUF v1 used 32-bit lengths and counts; v2 and v3 use 64-bit.
165
+ *
166
+ * @param {Cursor} cur
167
+ * @param {number} version
168
+ */
169
+ function readLength(cur, version) {
170
+ return version === 1 ? cur.u32() : cur.u64();
171
+ }
172
+
173
+ function readString(cur, version) {
174
+ const len = readLength(cur, version);
175
+ if (len > MAX_STRING_BYTES) {
176
+ throw new MalformedHeader(`string length ${len} is implausible for a header field`);
177
+ }
178
+ if (len === 0) return "";
179
+ // Long strings still have to be read whole, so chunk them through the window.
180
+ if (len <= WINDOW_BYTES) {
181
+ return cur.slice(len).toString("utf8");
182
+ }
183
+ const parts = [];
184
+ let remaining = len;
185
+ while (remaining > 0) {
186
+ const take = Math.min(remaining, WINDOW_BYTES);
187
+ parts.push(Buffer.from(cur.slice(take)));
188
+ remaining -= take;
189
+ }
190
+ return Buffer.concat(parts).toString("utf8");
191
+ }
192
+
193
+ /**
194
+ * Read one typed value. Scalars and strings are returned; arrays are skipped
195
+ * (we never need their contents) but must be stepped over exactly, otherwise
196
+ * every subsequent key is garbage.
197
+ *
198
+ * @returns {{ value: unknown, kind: "scalar"|"string"|"array" }}
199
+ */
200
+ function readValue(cur, type, version, depth = 0) {
201
+ switch (type) {
202
+ case GGUF_TYPE.UINT8:
203
+ return { value: cur.u8(), kind: "scalar" };
204
+ case GGUF_TYPE.INT8:
205
+ return { value: cur.slice(1).readInt8(0), kind: "scalar" };
206
+ case GGUF_TYPE.UINT16:
207
+ return { value: cur.slice(2).readUInt16LE(0), kind: "scalar" };
208
+ case GGUF_TYPE.INT16:
209
+ return { value: cur.slice(2).readInt16LE(0), kind: "scalar" };
210
+ case GGUF_TYPE.UINT32:
211
+ return { value: cur.u32(), kind: "scalar" };
212
+ case GGUF_TYPE.INT32:
213
+ return { value: cur.slice(4).readInt32LE(0), kind: "scalar" };
214
+ case GGUF_TYPE.FLOAT32:
215
+ return { value: cur.slice(4).readFloatLE(0), kind: "scalar" };
216
+ case GGUF_TYPE.BOOL:
217
+ return { value: cur.u8() !== 0, kind: "scalar" };
218
+ case GGUF_TYPE.UINT64:
219
+ return { value: cur.u64(), kind: "scalar" };
220
+ case GGUF_TYPE.INT64: {
221
+ const raw = cur.slice(8).readBigInt64LE(0);
222
+ return { value: Number(raw), kind: "scalar" };
223
+ }
224
+ case GGUF_TYPE.FLOAT64:
225
+ return { value: cur.slice(8).readDoubleLE(0), kind: "scalar" };
226
+ case GGUF_TYPE.STRING:
227
+ return { value: readString(cur, version), kind: "string" };
228
+ case GGUF_TYPE.ARRAY: {
229
+ if (depth >= MAX_ARRAY_DEPTH) {
230
+ throw new MalformedHeader("array nesting is deeper than this reader allows");
231
+ }
232
+ const elemType = cur.u32();
233
+ const count = readLength(cur, version);
234
+ skipArrayElements(cur, elemType, count, version, depth + 1);
235
+ return { value: { elementType: elemType, length: count }, kind: "array" };
236
+ }
237
+ default:
238
+ // An unknown value type means we can no longer compute where the next key
239
+ // begins. Stopping is the honest outcome; guessing is not.
240
+ throw new MalformedHeader(`unknown GGUF value type ${type}`);
241
+ }
242
+ }
243
+
244
+ function skipArrayElements(cur, elemType, count, version, depth) {
245
+ const size = SCALAR_SIZE[elemType];
246
+ if (size !== undefined) {
247
+ cur.skip(size * count);
248
+ return;
249
+ }
250
+ if (elemType === GGUF_TYPE.STRING) {
251
+ // Variable-length: each element carries its own length prefix.
252
+ for (let i = 0; i < count; i += 1) {
253
+ const len = readLength(cur, version);
254
+ if (len > MAX_STRING_BYTES) {
255
+ throw new MalformedHeader(`array string length ${len} is implausible`);
256
+ }
257
+ cur.skip(len);
258
+ }
259
+ return;
260
+ }
261
+ if (elemType === GGUF_TYPE.ARRAY) {
262
+ for (let i = 0; i < count; i += 1) {
263
+ readValue(cur, GGUF_TYPE.ARRAY, version, depth);
264
+ }
265
+ return;
266
+ }
267
+ throw new MalformedHeader(`unknown GGUF array element type ${elemType}`);
268
+ }
269
+
270
+ /**
271
+ * Keys we care about, matched by suffix so they work for any architecture.
272
+ * The `.ssm.*` keys only exist on hybrid/recurrent architectures (Mamba-style
273
+ * state-space layers mixed with regular attention, e.g. Qwen3-Next's Gated
274
+ * DeltaNet blocks) -- their mere presence is architecture-agnostic evidence
275
+ * that not every layer is full attention, even before anything knows the
276
+ * exact ratio (see lib/sizing.mjs's HYBRID_ATTENTION_LAYER_FRACTION).
277
+ */
278
+ const WANTED_SUFFIXES = Object.freeze({
279
+ ".block_count": "blockCount",
280
+ ".attention.head_count_kv": "headCountKv",
281
+ ".attention.head_count": "headCount",
282
+ ".attention.key_length": "keyLength",
283
+ ".attention.value_length": "valueLength",
284
+ ".embedding_length": "embeddingLength",
285
+ ".context_length": "contextLength",
286
+ ".ssm.conv_kernel": "ssmConvKernel",
287
+ ".ssm.state_size": "ssmStateSize",
288
+ ".ssm.group_count": "ssmGroupCount",
289
+ ".ssm.time_step_rank": "ssmTimeStepRank",
290
+ ".ssm.inner_size": "ssmInnerSize",
291
+ });
292
+
293
+ function matchWanted(key) {
294
+ // Longest suffix first, so `.attention.head_count_kv` is not stolen by
295
+ // `.attention.head_count`.
296
+ const suffixes = Object.keys(WANTED_SUFFIXES).sort((a, b) => b.length - a.length);
297
+ for (const suffix of suffixes) {
298
+ if (key.endsWith(suffix)) return WANTED_SUFFIXES[suffix];
299
+ }
300
+ return null;
301
+ }
302
+
303
+ function numeric(value) {
304
+ return typeof value === "number" && Number.isFinite(value) ? value : null;
305
+ }
306
+
307
+ /**
308
+ * Read the architecture facts out of a GGUF file's header.
309
+ *
310
+ * @param {string} filePath
311
+ * @param {{ headerByteLimit?: number }} [options]
312
+ * @returns {{
313
+ * found: boolean,
314
+ * path: string|null,
315
+ * fileSizeBytes: number|null,
316
+ * arch: string|null,
317
+ * params: object,
318
+ * truncated: boolean,
319
+ * version: number|null,
320
+ * kvCount: number|null,
321
+ * tensorCount: number|null,
322
+ * general: object,
323
+ * reason: string|null
324
+ * }}
325
+ */
326
+ export function readGGUFMetadata(filePath, options = {}) {
327
+ const limit = Number.isFinite(options.headerByteLimit)
328
+ ? options.headerByteLimit
329
+ : DEFAULT_HEADER_BYTE_LIMIT;
330
+
331
+ const result = {
332
+ found: false,
333
+ path: typeof filePath === "string" ? filePath : null,
334
+ fileSizeBytes: null,
335
+ arch: null,
336
+ params: {
337
+ blockCount: null,
338
+ headCount: null,
339
+ headCountKv: null,
340
+ keyLength: null,
341
+ valueLength: null,
342
+ embeddingLength: null,
343
+ contextLength: null,
344
+ ssmConvKernel: null,
345
+ ssmStateSize: null,
346
+ ssmGroupCount: null,
347
+ ssmTimeStepRank: null,
348
+ ssmInnerSize: null,
349
+ },
350
+ truncated: false,
351
+ version: null,
352
+ kvCount: null,
353
+ tensorCount: null,
354
+ general: { name: null, fileType: null, sizeLabel: null },
355
+ reason: null,
356
+ };
357
+
358
+ if (typeof filePath !== "string" || filePath.length === 0) {
359
+ result.reason = "no model path supplied";
360
+ return result;
361
+ }
362
+
363
+ let fd = null;
364
+ try {
365
+ const stat = fs.statSync(filePath);
366
+ if (!stat.isFile()) {
367
+ result.reason = "path is not a regular file";
368
+ return result;
369
+ }
370
+ result.fileSizeBytes = stat.size;
371
+
372
+ fd = fs.openSync(filePath, "r");
373
+ const cur = new Cursor(fd, stat.size, limit);
374
+
375
+ const magic = cur.slice(4).toString("latin1");
376
+ if (magic !== "GGUF") {
377
+ result.reason = `not a GGUF file (magic ${JSON.stringify(magic)})`;
378
+ return result;
379
+ }
380
+ const version = cur.u32();
381
+ if (!Number.isFinite(version) || version < 1 || version > 3) {
382
+ result.reason = `unsupported GGUF version ${version}`;
383
+ return result;
384
+ }
385
+ result.version = version;
386
+
387
+ const tensorCount = readLength(cur, version);
388
+ const kvCount = readLength(cur, version);
389
+ if (tensorCount > MAX_TENSOR_COUNT || kvCount > MAX_KV_COUNT) {
390
+ result.reason = `implausible header counts (tensors=${tensorCount}, kv=${kvCount})`;
391
+ return result;
392
+ }
393
+ result.tensorCount = tensorCount;
394
+ result.kvCount = kvCount;
395
+
396
+ // From here the file is a GGUF: whatever we manage to read is real, and
397
+ // anything we cannot read is reported as `truncated`, not as "not found".
398
+ result.found = true;
399
+
400
+ const collected = Object.create(null);
401
+ for (let i = 0; i < kvCount; i += 1) {
402
+ const key = readString(cur, version);
403
+ const type = cur.u32();
404
+ const { value } = readValue(cur, type, version);
405
+
406
+ if (key === "general.architecture" && typeof value === "string") {
407
+ result.arch = value;
408
+ } else if (key === "general.name" && typeof value === "string") {
409
+ result.general.name = value;
410
+ } else if (key === "general.size_label" && typeof value === "string") {
411
+ result.general.sizeLabel = value;
412
+ } else if (key === "general.file_type") {
413
+ result.general.fileType = numeric(value);
414
+ } else {
415
+ const field = matchWanted(key);
416
+ if (field && numeric(value) !== null) {
417
+ // Prefer keys that belong to the declared architecture; a stray
418
+ // `<other>.block_count` must not overwrite the real one.
419
+ const archPrefixed = result.arch && key.startsWith(`${result.arch}.`);
420
+ if (collected[field] === undefined || archPrefixed) {
421
+ collected[field] = numeric(value);
422
+ }
423
+ }
424
+ }
425
+ }
426
+ Object.assign(result.params, collected);
427
+ return result;
428
+ } catch (err) {
429
+ if (err instanceof TruncatedHeader) {
430
+ result.truncated = true;
431
+ result.reason = err.message;
432
+ return result;
433
+ }
434
+ if (err instanceof MalformedHeader) {
435
+ result.reason = err.message;
436
+ // A malformed header after a valid magic means the numbers we did read
437
+ // cannot be trusted to be complete; say so rather than silently shipping
438
+ // half an architecture.
439
+ result.truncated = result.found;
440
+ return result;
441
+ }
442
+ result.found = false;
443
+ result.reason = String(err && err.message ? err.message : err);
444
+ return result;
445
+ } finally {
446
+ if (fd !== null) {
447
+ try {
448
+ fs.closeSync(fd);
449
+ } catch {
450
+ /* closing a descriptor we already hold cannot usefully fail here */
451
+ }
452
+ }
453
+ }
454
+ }
455
+
456
+ /**
457
+ * Derive the per-head dimension from whatever the header gave us.
458
+ * `key_length` is authoritative when present; otherwise fall back to
459
+ * embedding_length / head_count.
460
+ *
461
+ * @param {object} params
462
+ * @returns {{ headDim: number|null, source: string|null }}
463
+ */
464
+ export function deriveHeadDim(params) {
465
+ if (!params) return { headDim: null, source: null };
466
+ if (numeric(params.keyLength)) {
467
+ return { headDim: params.keyLength, source: "attention.key_length" };
468
+ }
469
+ if (numeric(params.embeddingLength) && numeric(params.headCount) && params.headCount > 0) {
470
+ return {
471
+ headDim: Math.floor(params.embeddingLength / params.headCount),
472
+ source: "embedding_length / attention.head_count",
473
+ };
474
+ }
475
+ return { headDim: null, source: null };
476
+ }
477
+
478
+ // A GGUF split into shards names each part `<name>-NNNNN-of-MMMMM.gguf`. Only
479
+ // the first shard carries the architecture header a reader needs; the rest
480
+ // are pure tensor data. Matches llama.cpp's own split convention.
481
+ export const GGUF_SPLIT_RE = /^(.*)-(\d+)-of-(\d+)\.gguf$/i;
482
+
483
+ /**
484
+ * Walk a directory tree for .gguf files, following symlinks (the Hugging
485
+ * Face cache stores every blob as a symlink from its snapshot directory).
486
+ * Never throws: an unreadable directory or a broken symlink is skipped, not
487
+ * an error -- a fresh install with no model yet is the common case.
488
+ *
489
+ * @param {string} root
490
+ * @returns {{ path: string, mtimeMs: number }[]}
491
+ */
492
+ export function findGgufFiles(root) {
493
+ const found = [];
494
+ const stack = [root];
495
+ while (stack.length) {
496
+ const dir = stack.pop();
497
+ let entries;
498
+ try { entries = fs.readdirSync(dir, { withFileTypes: true }); } catch { continue; }
499
+ for (const e of entries) {
500
+ const p = path.join(dir, e.name);
501
+ let stat;
502
+ try { stat = fs.statSync(p); } catch { continue; } // broken symlink, permission denied, etc.
503
+ if (stat.isDirectory()) stack.push(p);
504
+ else if (stat.isFile() && e.name.toLowerCase().endsWith(".gguf")) found.push({ path: p, mtimeMs: stat.mtimeMs });
505
+ }
506
+ }
507
+ return found;
508
+ }
509
+
510
+ /**
511
+ * Locate a GGUF file to size against. llama.cpp caches Hugging Face `-hf`
512
+ * pulls in the standard Hugging Face hub cache layout
513
+ * (~/.cache/huggingface/hub/models--<org>--<repo>/snapshots/<hash>/<file>),
514
+ * not a llama.cpp-specific directory, so that is checked alongside the older
515
+ * locations rather than making the caller pass a path. Absence is normal on
516
+ * a fresh install.
517
+ *
518
+ * @param {{ explicit?: string|null, env?: object, roots?: string[]|null }} [input]
519
+ * `roots` overrides the search list entirely (tests use this to avoid
520
+ * scanning the real machine's caches); omit it for the real locations.
521
+ * @returns {string|null}
522
+ */
523
+ export function resolveModelPath({ explicit = null, env = process.env, roots = null } = {}) {
524
+ if (explicit) return explicit;
525
+ const installRoot = env.NOMARMY_INSTALL_ROOT
526
+ || path.join(os.homedir(), ".local", "share", "nomarmy-local-agents");
527
+ const searchRoots = roots ?? [
528
+ env.NOMARMY_MODEL_PATH,
529
+ path.join(os.homedir(), ".cache", "huggingface", "hub"),
530
+ path.join(os.homedir(), ".cache", "llama.cpp"),
531
+ path.join(installRoot, "models"),
532
+ ].filter(Boolean);
533
+
534
+ const candidates = [];
535
+ for (const root of searchRoots) {
536
+ try {
537
+ if (!fs.existsSync(root)) continue;
538
+ const stat = fs.statSync(root);
539
+ if (stat.isFile()) {
540
+ if (root.toLowerCase().endsWith(".gguf")) candidates.push({ path: root, mtimeMs: stat.mtimeMs });
541
+ continue;
542
+ }
543
+ candidates.push(...findGgufFiles(root));
544
+ } catch {
545
+ // An unreadable cache directory is a normal outcome, not an error.
546
+ }
547
+ }
548
+ if (!candidates.length) return null;
549
+
550
+ // A split model contributes many files; only its first shard is a usable
551
+ // representative (it alone carries the header). Collapse each split down to
552
+ // that one file, keeping the most recently touched file otherwise -- a
553
+ // proxy for "the model most recently pulled or used" absent a loaded
554
+ // profile to name one explicitly.
555
+ const representative = candidates.filter((c) => {
556
+ const m = path.basename(c.path).match(GGUF_SPLIT_RE);
557
+ return !m || Number(m[2]) === 1;
558
+ });
559
+ representative.sort((a, b) => b.mtimeMs - a.mtimeMs);
560
+ return representative[0].path;
561
+ }
562
+
563
+ /**
564
+ * The representative file resolveModelPath returns may be shard 1 of many;
565
+ * `fileSizeBytes` from a header read of just that file would then understate
566
+ * total weight size by a factor of the shard count, which is worse for the
567
+ * sizing math than not finding the model at all. Sum every sibling shard's
568
+ * real size instead. A non-split file just returns its own size.
569
+ *
570
+ * @param {string} modelPath
571
+ * @returns {number}
572
+ */
573
+ export function totalSplitBytes(modelPath) {
574
+ const name = path.basename(modelPath);
575
+ const m = name.match(GGUF_SPLIT_RE);
576
+ if (!m) return fs.statSync(modelPath).size;
577
+ const [, prefix, shardDigits, totalStr] = m;
578
+ const total = Number(totalStr);
579
+ const dir = path.dirname(modelPath);
580
+ let sum = 0;
581
+ for (let i = 1; i <= total; i += 1) {
582
+ const shardName = `${prefix}-${String(i).padStart(shardDigits.length, "0")}-of-${totalStr}.gguf`;
583
+ try { sum += fs.statSync(path.join(dir, shardName)).size; }
584
+ catch { return fs.statSync(modelPath).size; } // a missing sibling means the set is incomplete; fall back to what we can measure
585
+ }
586
+ return sum;
587
+ }
588
+
589
+ export default { readGGUFMetadata, deriveHeadDim, GGUF_TYPE, findGgufFiles, resolveModelPath, totalSplitBytes };