dotmd-cli 0.87.0 → 0.87.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/model.mjs ADDED
@@ -0,0 +1,523 @@
1
+ import { spawn, spawnSync } from 'node:child_process';
2
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
3
+ import os from 'node:os';
4
+ import path from 'node:path';
5
+ import { fileURLToPath } from 'node:url';
6
+ import { die, warn } from './util.mjs';
7
+ import { bold, dim, green, yellow } from './color.mjs';
8
+
9
+ // One local model server, talked to over HTTP. The model loads once, serves
10
+ // one request at a time, and unloads after it sits idle. runlist never starts
11
+ // the server on its own: only `runlist model start` does, and a model is never
12
+ // pulled for you.
13
+ //
14
+ // Settings are the machine's, not the repo's, because memory is: they live in
15
+ // ~/.runlist/model.json (RUNLIST_MODEL_SETTINGS moves it), written by
16
+ // `runlist model use` and `runlist model cap`. RUNLIST_MODEL,
17
+ // RUNLIST_MODEL_CAP_GB and RUNLIST_MODEL_ENDPOINT override them for one shell, and `--model` for one run.
18
+ //
19
+ // Runtimes: `ollama` (the default; its own API, so load state, memory and
20
+ // unload are visible) and `openai` (any OpenAI-compatible server: mlx_lm.server,
21
+ // llama-server, LM Studio), which can only be asked to generate.
22
+ //
23
+ // Memory is guarded twice. The cap, unless set, is a quarter of the machine's
24
+ // memory up to 12 GB, so an 8 GB laptop gets 2 GB and no candidate: model
25
+ // features stay off there and say why. And before a model loads on this
26
+ // machine, the memory free right now must hold it plus headroom, with the
27
+ // system's memory pressure normal; otherwise the request is refused and the
28
+ // command carries on without it. A server on another machine is that
29
+ // machine's memory, so neither check applies to it.
30
+
31
+ const REQUEST = fileURLToPath(new URL('./model-request.mjs', import.meta.url));
32
+
33
+ // Best first. None is picked on a machine until `runlist model measure` has
34
+ // recorded its peak memory there; the first measured one under the cap is used
35
+ // when no model is named.
36
+ export const CANDIDATES = Object.freeze(['gemma4:12b', 'qwen3.5:9b', 'qwen3.5:4b']);
37
+
38
+ export const DEFAULTS = Object.freeze({
39
+ runtime: 'ollama',
40
+ endpoint: 'http://127.0.0.1:11434',
41
+ model: null,
42
+ capGb: null,
43
+ headroomGb: 1.5,
44
+ keepAlive: '5m',
45
+ contextTokens: 16384,
46
+ });
47
+
48
+ const GB = 1e9;
49
+ const MAX_CAP_GB = 12;
50
+
51
+ // A quarter of the machine's memory, in half-gigabyte steps, at most 12 GB.
52
+ export function autoCapGb(totalBytes = os.totalmem()) {
53
+ return Math.min(MAX_CAP_GB, Math.floor((totalBytes / GB / 4) * 2) / 2);
54
+ }
55
+
56
+ const PRESSURE = { 1: 'normal', 2: 'warn', 4: 'critical' };
57
+
58
+ // Total, free-now and pressure for this machine. RUNLIST_MODEL_AVAILABLE_GB
59
+ // and RUNLIST_MODEL_PRESSURE replace the reading where it is wrong, and in tests.
60
+ export function memoryReading() {
61
+ const totalGb = os.totalmem() / GB;
62
+ let availableGb = os.freemem() / GB;
63
+ let pressure = 'normal';
64
+ if (process.platform === 'darwin') {
65
+ const r = spawnSync('sysctl', ['-n', 'kern.memorystatus_level', 'kern.memorystatus_vm_pressure_level'], { encoding: 'utf8' });
66
+ const [level, press] = (r.stdout ?? '').trim().split('\n').map(Number);
67
+ if (level > 0) availableGb = totalGb * level / 100;
68
+ if (PRESSURE[press]) pressure = PRESSURE[press];
69
+ } else if (process.platform === 'linux') {
70
+ const m = /^MemAvailable:\s+(\d+) kB/m.exec(readFileSync('/proc/meminfo', 'utf8'));
71
+ if (m) availableGb = Number(m[1]) * 1024 / GB;
72
+ }
73
+ if (process.env.RUNLIST_MODEL_AVAILABLE_GB) availableGb = Number(process.env.RUNLIST_MODEL_AVAILABLE_GB);
74
+ if (process.env.RUNLIST_MODEL_PRESSURE) pressure = process.env.RUNLIST_MODEL_PRESSURE;
75
+ return { totalGb, availableGb, pressure };
76
+ }
77
+
78
+ export function isLocalEndpoint(endpoint) {
79
+ try {
80
+ return ['127.0.0.1', 'localhost', '[::1]', '::1'].includes(new URL(endpoint).hostname);
81
+ } catch { return true; }
82
+ }
83
+
84
+ export function settingsFile() {
85
+ return process.env.RUNLIST_MODEL_SETTINGS || path.join(os.homedir(), '.runlist', 'model.json');
86
+ }
87
+
88
+ export function measurementsFile() {
89
+ return path.join(path.dirname(settingsFile()), 'model-measurements.json');
90
+ }
91
+
92
+ export function readMeasurements() {
93
+ const file = measurementsFile();
94
+ if (!existsSync(file)) return {};
95
+ try { return JSON.parse(readFileSync(file, 'utf8')) ?? {}; } catch { return {}; }
96
+ }
97
+
98
+ const peakOf = (name, measured = readMeasurements()) => measured[name]?.peakGb ?? null;
99
+
100
+ function readSettingsFile() {
101
+ const file = settingsFile();
102
+ if (!existsSync(file)) return {};
103
+ try { return JSON.parse(readFileSync(file, 'utf8')) ?? {}; } catch { return {}; }
104
+ }
105
+
106
+ export function writeSettings(patch) {
107
+ const file = settingsFile();
108
+ const next = { ...readSettingsFile(), ...patch };
109
+ for (const [k, v] of Object.entries(next)) if (v === null) delete next[k];
110
+ mkdirSync(path.dirname(file), { recursive: true });
111
+ writeFileSync(file, `${JSON.stringify(next, null, 2)}\n`);
112
+ return next;
113
+ }
114
+
115
+ export function modelSettings(overrides = {}) {
116
+ const s = { ...DEFAULTS, ...readSettingsFile() };
117
+ if (process.env.RUNLIST_MODEL_ENDPOINT) s.endpoint = process.env.RUNLIST_MODEL_ENDPOINT;
118
+ if (process.env.RUNLIST_MODEL) s.model = process.env.RUNLIST_MODEL;
119
+ if (process.env.RUNLIST_MODEL_CAP_GB) s.capGb = Number(process.env.RUNLIST_MODEL_CAP_GB);
120
+ for (const [k, v] of Object.entries(overrides)) if (v !== undefined && v !== null) s[k] = v;
121
+ s.endpoint = String(s.endpoint).replace(/\/+$/, '');
122
+ s.capSource = Number(s.capGb) > 0 ? 'set' : 'machine';
123
+ s.capGb = Number(s.capGb) > 0 ? Number(s.capGb) : autoCapGb();
124
+ return s;
125
+ }
126
+
127
+ // The named model, or the first measured candidate that fits the cap. A
128
+ // candidate not yet measured is never picked on its own.
129
+ export function pickModel(settings) {
130
+ if (settings.model) return { name: settings.model, why: 'named' };
131
+ const measured = readMeasurements();
132
+ const fit = CANDIDATES.find(name => peakOf(name, measured) !== null && peakOf(name, measured) <= settings.capGb);
133
+ if (fit) return { name: fit, why: `measured here at ${peakOf(fit, measured)} GB, under the ${settings.capGb} GB cap` };
134
+ return { name: null, why: `no measured model fits the ${settings.capGb} GB cap${settings.capSource === 'machine' ? ' this machine allows' : ''}` };
135
+ }
136
+
137
+ const FAILED = Object.freeze({ ok: false, status: 0, json: null, error: 'request process failed' });
138
+
139
+ // Sends the requests in order from one child process and returns their results.
140
+ export function requests(list) {
141
+ const total = list.reduce((sum, r) => sum + (r.timeoutMs ?? 10000), 0);
142
+ const result = spawnSync(process.execPath, [REQUEST], {
143
+ input: JSON.stringify(list),
144
+ encoding: 'utf8',
145
+ timeout: total + 10000,
146
+ });
147
+ if (result.status !== 0 || !result.stdout) return list.map(() => FAILED);
148
+ try { return JSON.parse(result.stdout); } catch { return list.map(() => FAILED); }
149
+ }
150
+
151
+ export function request(url, { method = 'GET', body, timeoutMs = 10000 } = {}) {
152
+ return requests([{ url, method, body, timeoutMs }])[0];
153
+ }
154
+
155
+ export function serverVersion(settings) {
156
+ const res = settings.runtime === 'ollama'
157
+ ? request(`${settings.endpoint}/api/version`, { timeoutMs: 5000 })
158
+ : request(`${settings.endpoint}/v1/models`, { timeoutMs: 5000 });
159
+ if (!res.ok) return null;
160
+ return res.json?.version ?? 'up';
161
+ }
162
+
163
+ export function loadedModels(settings) {
164
+ if (settings.runtime !== 'ollama') return [];
165
+ const res = request(`${settings.endpoint}/api/ps`, { timeoutMs: 5000 });
166
+ return res.ok ? (res.json?.models ?? []).map(m => ({ name: m.name, bytes: m.size, expiresAt: m.expires_at })) : [];
167
+ }
168
+
169
+ export function pulledModels(settings) {
170
+ if (settings.runtime !== 'ollama') return null;
171
+ const res = request(`${settings.endpoint}/api/tags`, { timeoutMs: 5000 });
172
+ return res.ok ? (res.json?.models ?? []).map(m => ({ name: m.name, bytes: m.size })) : null;
173
+ }
174
+
175
+ // Why a model cannot load on this machine right now, or null when it can.
176
+ export function roomRefusal(name, needGb, settings, reading = memoryReading()) {
177
+ if (reading.pressure !== 'normal') return `Memory pressure is ${reading.pressure}; ${name} was not loaded.`;
178
+ const want = needGb + settings.headroomGb;
179
+ if (reading.availableGb < want) {
180
+ return `${name} needs about ${want.toFixed(1)} GB free with headroom and ${reading.availableGb.toFixed(1)} GB is; it was not loaded.`;
181
+ }
182
+ return null;
183
+ }
184
+
185
+ const sameModel = (a, b) => a === b || a === `${b}:latest` || b === `${a}:latest`;
186
+
187
+ const warned = new Set();
188
+ // A model that passed the server, pull and cap checks once is not re-checked
189
+ // for every document in the same run.
190
+ const ready = new Set();
191
+ function warnOnce(key, message) {
192
+ if (warned.has(key)) return;
193
+ warned.add(key);
194
+ warn(message);
195
+ }
196
+
197
+ // The last request's timing and footprint, for `runlist model` and measurement.
198
+ export let lastRun = null;
199
+
200
+ // Sends one chat request and returns the reply text, or null with one warning
201
+ // saying why. Callers treat null as "no model available" and carry on.
202
+ export function generate(messages, opts = {}) {
203
+ const settings = modelSettings({ model: opts.model });
204
+ const { name } = pickModel(settings);
205
+ if (!name) {
206
+ const hint = settings.capSource === 'machine'
207
+ ? 'Model features are off on this machine; a server on another machine can be named with RUNLIST_MODEL_ENDPOINT.'
208
+ : '`runlist model use <name>` names one.';
209
+ warnOnce('no-model', `No local model: ${pickModel(settings).why}. ${hint}`);
210
+ return null;
211
+ }
212
+
213
+ const readyKey = `${settings.endpoint} ${name} ${settings.capGb}`;
214
+ if (!ready.has(readyKey)) {
215
+ const [up, tags, ps] = settings.runtime === 'ollama'
216
+ ? requests(['version', 'tags', 'ps'].map(p => ({ url: `${settings.endpoint}/api/${p}`, timeoutMs: 5000 })))
217
+ : [request(`${settings.endpoint}/v1/models`, { timeoutMs: 5000 }), null, null];
218
+ if (!up.ok) {
219
+ warnOnce('no-server', `The model server is not running at ${settings.endpoint}. \`runlist model start\` starts it.`);
220
+ return null;
221
+ }
222
+ if (settings.runtime === 'ollama') {
223
+ const onDisk = (tags.json?.models ?? []).find(m => sameModel(m.name, name));
224
+ if (!onDisk) { warnOnce(`pull:${name}`, `${name} is not pulled. \`ollama pull ${name}\` downloads it.`); return null; }
225
+ const floorGb = peakOf(name) ?? onDisk.size / GB;
226
+ const local = isLocalEndpoint(settings.endpoint);
227
+ if (local && floorGb > settings.capGb) {
228
+ warnOnce(`cap:${name}`, `${name} needs about ${floorGb.toFixed(1)} GB, over this machine's ${settings.capGb} GB cap. \`runlist model cap <gb>\` raises it.`);
229
+ return null;
230
+ }
231
+ const loaded = (ps.json?.models ?? []).some(m => sameModel(m.name, name));
232
+ const refusal = local && !loaded ? roomRefusal(name, floorGb, settings) : null;
233
+ if (refusal) { warnOnce(`room:${name}`, refusal); return null; }
234
+ }
235
+ ready.add(readyKey);
236
+ }
237
+
238
+ const started = Date.now();
239
+ const timeoutMs = opts.timeoutMs ?? 300000;
240
+ let res;
241
+ let text;
242
+ let resident;
243
+ if (settings.runtime === 'ollama') {
244
+ // The chat and the memory reading after it go in one child process.
245
+ const [chat, ps] = requests([
246
+ {
247
+ url: `${settings.endpoint}/api/chat`,
248
+ method: 'POST',
249
+ timeoutMs,
250
+ body: {
251
+ model: name,
252
+ messages,
253
+ stream: false,
254
+ think: false,
255
+ keep_alive: settings.keepAlive,
256
+ options: { num_ctx: settings.contextTokens, num_predict: opts.maxTokens ?? 200, temperature: 0 },
257
+ },
258
+ },
259
+ { url: `${settings.endpoint}/api/ps`, timeoutMs: 5000 },
260
+ ]);
261
+ res = chat;
262
+ text = chat.json?.message?.content;
263
+ const m = (ps.json?.models ?? []).find(x => sameModel(x.name, name));
264
+ resident = m ? { bytes: m.size } : null;
265
+ } else {
266
+ res = request(`${settings.endpoint}/v1/chat/completions`, {
267
+ method: 'POST',
268
+ timeoutMs,
269
+ body: { model: name, messages, max_tokens: opts.maxTokens ?? 200, temperature: 0 },
270
+ });
271
+ text = res.json?.choices?.[0]?.message?.content;
272
+ }
273
+
274
+ if (!res.ok) { warnOnce(`fail:${name}`, `${name} failed: ${res.error}`); return null; }
275
+ lastRun = {
276
+ model: name,
277
+ ms: Date.now() - started,
278
+ residentBytes: resident?.bytes ?? null,
279
+ evalCount: res.json?.eval_count ?? null,
280
+ evalNs: res.json?.eval_duration ?? null,
281
+ };
282
+ if (resident && isLocalEndpoint(settings.endpoint) && resident.bytes / GB > settings.capGb) {
283
+ unload(settings, name);
284
+ warnOnce(`over:${name}`, `${name} took ${(resident.bytes / GB).toFixed(1)} GB, over the ${settings.capGb} GB cap, and was unloaded.`);
285
+ }
286
+ return text?.trim() || null;
287
+ }
288
+
289
+ export function unload(settings, name) {
290
+ if (settings.runtime !== 'ollama') return false;
291
+ return request(`${settings.endpoint}/api/generate`, { method: 'POST', body: { model: name, keep_alive: 0 } }).ok;
292
+ }
293
+
294
+ function sleepMs(ms) {
295
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
296
+ }
297
+
298
+ // Starts `ollama serve` detached, holding one model and one request at a time.
299
+ // Refuses, unless forced, on a machine where no model could run.
300
+ export function startServer(settings, { force = false } = {}) {
301
+ if (serverVersion(settings)) return { started: false, already: true };
302
+ if (settings.runtime !== 'ollama') die(`runlist starts only Ollama; start the ${settings.runtime} server at ${settings.endpoint} yourself.`);
303
+ if (!isLocalEndpoint(settings.endpoint)) die(`The model server is on another machine (${settings.endpoint}); start it there.`);
304
+ const pick = pickModel(settings);
305
+ if (!pick.name && !force) die(`Not started: ${pick.why}. Model features stay off on this machine; \`runlist model start --force\` starts the server anyway.`);
306
+ const installed = spawnSync('ollama', ['--version'], { encoding: 'utf8' });
307
+ if (installed.error) die('Ollama is not installed (https://ollama.com). Or point runlist at a server on another machine: RUNLIST_MODEL_ENDPOINT, or "endpoint" in the settings file.');
308
+ const port = new URL(settings.endpoint).port || '11434';
309
+ const child = spawn('ollama', ['serve'], {
310
+ detached: true,
311
+ stdio: 'ignore',
312
+ env: { ...process.env, OLLAMA_HOST: `127.0.0.1:${port}`, OLLAMA_MAX_LOADED_MODELS: '1', OLLAMA_NUM_PARALLEL: '1' },
313
+ });
314
+ child.unref();
315
+ for (let i = 0; i < 40; i++) {
316
+ sleepMs(250);
317
+ if (serverVersion(settings)) return { started: true, pid: child.pid };
318
+ }
319
+ die(`ollama serve did not answer at ${settings.endpoint} within 10 seconds.`);
320
+ }
321
+
322
+ const gb = bytes => `${(bytes / GB).toFixed(1)} GB`;
323
+
324
+ function statusData(settings) {
325
+ const version = serverVersion(settings);
326
+ const local = isLocalEndpoint(settings.endpoint);
327
+ const reading = local ? memoryReading() : null;
328
+ const pick = pickModel(settings);
329
+ const pulled = version ? pulledModels(settings) : null;
330
+ const loaded = version ? loadedModels(settings) : [];
331
+ const resident = loaded.find(m => pick.name && sameModel(m.name, pick.name)) ?? loaded[0] ?? null;
332
+ return {
333
+ // The one-line reading: is the server up, which model (the one loaded now,
334
+ // else the one runlist would load), and what it holds in memory (MiB, null
335
+ // when nothing is loaded).
336
+ running: !!version,
337
+ name: resident?.name ?? pick.name,
338
+ memoryMb: resident ? Math.round(resident.bytes / 2 ** 20) : null,
339
+ runtime: settings.runtime,
340
+ endpoint: settings.endpoint,
341
+ server: version ? { running: true, version } : { running: false },
342
+ model: pick.name,
343
+ why: pick.why,
344
+ pulled: pulled && pick.name ? pulled.some(m => sameModel(m.name, pick.name)) : null,
345
+ capGb: settings.capGb,
346
+ capSource: settings.capSource,
347
+ memory: reading && { totalGb: +reading.totalGb.toFixed(1), availableGb: +reading.availableGb.toFixed(1), pressure: reading.pressure },
348
+ local,
349
+ keepAlive: settings.keepAlive,
350
+ contextTokens: settings.contextTokens,
351
+ loaded,
352
+ candidates: CANDIDATES.map(name => {
353
+ const peakGb = peakOf(name);
354
+ return {
355
+ name,
356
+ peakGb,
357
+ pulled: pulled ? pulled.some(m => sameModel(m.name, name)) : null,
358
+ fits: peakGb !== null ? peakGb <= settings.capGb : null,
359
+ };
360
+ }),
361
+ settingsFile: settingsFile(),
362
+ };
363
+ }
364
+
365
+ function printStatus(d) {
366
+ const out = [];
367
+ out.push(d.server.running
368
+ ? `${bold('Server')} ${green('running')} (${d.runtime} ${d.server.version}) at ${d.endpoint}`
369
+ : `${bold('Server')} ${yellow('not running')} at ${d.endpoint}. \`runlist model start\` starts it.`);
370
+ const pulledNote = d.pulled === false ? yellow(` not pulled: \`ollama pull ${d.model}\``) : '';
371
+ out.push(`${bold('Model')} ${d.model ?? yellow('none')} ${dim(`(${d.why})`)}${pulledNote}`);
372
+ const capWhy = d.capSource === 'set' ? 'set' : 'a quarter of this machine\'s memory, at most 12 GB';
373
+ out.push(`${bold('Cap')} ${d.capGb} GB ${dim(`(${capWhy}); idle unload ${d.keepAlive}, context ${d.contextTokens} tokens`)}`);
374
+ out.push(d.memory
375
+ ? `${bold('Memory')} ${d.memory.totalGb} GB total, ${d.memory.availableGb} GB free now, pressure ${d.memory.pressure === 'normal' ? d.memory.pressure : yellow(d.memory.pressure)}`
376
+ : `${bold('Memory')} ${dim('the server is on another machine; its memory is its own')}`);
377
+ if (d.loaded.length) {
378
+ for (const m of d.loaded) out.push(`${bold('Loaded')} ${m.name} ${gb(m.bytes)}${m.expiresAt ? dim(`, unloads ${new Date(m.expiresAt).toLocaleTimeString()}`) : ''}`);
379
+ } else if (d.server.running) {
380
+ out.push(`${bold('Loaded')} nothing`);
381
+ }
382
+ out.push('', bold('Candidates'));
383
+ for (const c of d.candidates) {
384
+ const peak = c.peakGb !== null ? `${c.peakGb} GB peak here` : 'not measured here';
385
+ const fits = c.fits === null ? '' : c.fits ? '' : yellow(' over the cap');
386
+ const pulled = c.pulled === null ? '' : c.pulled ? '' : dim(' not pulled');
387
+ out.push(` ${c.name} ${dim(peak)}${fits}${pulled}`);
388
+ }
389
+ out.push('', dim(`Settings: ${d.settingsFile}`));
390
+ process.stdout.write(`${out.join('\n')}\n`);
391
+ }
392
+
393
+ const median = xs => [...xs].sort((a, b) => a - b)[Math.floor(xs.length / 2)];
394
+
395
+ // The documents a measurement summarises: the largest few in the corpus, so
396
+ // the reading is taken on real text of the size the features will see.
397
+ async function measureDocs(config, count = 3) {
398
+ const { buildIndex } = await import('./index.mjs');
399
+ const { extractFrontmatter } = await import('./frontmatter.mjs');
400
+ return buildIndex(config).docs
401
+ .filter(d => !d.path.includes('/archived/'))
402
+ .map(d => ({ doc: d, body: extractFrontmatter(readFileSync(path.resolve(config.repoRoot, d.path), 'utf8')).body ?? '' }))
403
+ .sort((a, b) => b.body.length - a.body.length)
404
+ .slice(0, count);
405
+ }
406
+
407
+ function reniceServer() {
408
+ const r = spawnSync('pgrep', ['-x', 'ollama'], { encoding: 'utf8' });
409
+ const pids = (r.stdout ?? '').trim().split('\n').filter(Boolean).map(Number);
410
+ for (const pid of pids) { try { os.setPriority(pid, 10); } catch { /* not ours to renice */ } }
411
+ return pids.length;
412
+ }
413
+
414
+ // Loads each named (or pulled) candidate in turn, summarises the largest
415
+ // documents, records the peak memory and speed on this machine, and unloads it.
416
+ // Skips any model the memory free right now cannot hold.
417
+ async function measure(names, settings, config) {
418
+ if (settings.runtime !== 'ollama') die('`runlist model measure` reads memory from Ollama; it cannot measure another runtime.');
419
+ if (!isLocalEndpoint(settings.endpoint)) die('Measure on the machine that runs the server; its memory is what is recorded.');
420
+ if (!serverVersion(settings)) die('The model server is not running. `runlist model start --force` starts it for a measurement.');
421
+ const pulled = pulledModels(settings) ?? [];
422
+ const targets = names.length ? names : CANDIDATES.filter(n => pulled.some(m => sameModel(m.name, n)));
423
+ if (!targets.length) die(`None of the candidates is pulled: ${CANDIDATES.map(n => `\`ollama pull ${n}\``).join(', ')}.`);
424
+ const docs = await measureDocs(config);
425
+ if (!docs.length) die('No documents to summarise in this corpus.');
426
+ const { summarizeDocBody } = await import('./ai.mjs');
427
+ const reniced = reniceServer();
428
+ process.stderr.write(dim(`Measuring on ${docs.length} documents; ${reniced ? 'the server runs at lower priority' : 'could not lower the server\'s priority'}.\n`));
429
+
430
+ const results = readMeasurements();
431
+ let recorded = 0;
432
+ for (const name of targets) {
433
+ const onDisk = pulled.find(m => sameModel(m.name, name));
434
+ if (!onDisk) { process.stdout.write(`${name}: not pulled, skipped.\n`); continue; }
435
+ for (const m of loadedModels(settings)) if (CANDIDATES.some(c => sameModel(m.name, c))) unload(settings, m.name);
436
+ const refusal = roomRefusal(name, onDisk.bytes / GB, settings);
437
+ if (refusal) { process.stdout.write(`${name}: skipped. ${refusal}\n`); continue; }
438
+
439
+ const runs = [];
440
+ let peak = 0;
441
+ for (const { doc, body } of docs) {
442
+ const summary = summarizeDocBody(body, { title: doc.title ?? doc.path, status: doc.status }, { model: name, maxTokens: 120 });
443
+ if (!summary || !lastRun) break;
444
+ runs.push({ ms: lastRun.ms, tokensPerSec: lastRun.evalNs ? lastRun.evalCount / (lastRun.evalNs / 1e9) : null });
445
+ peak = Math.max(peak, lastRun.residentBytes ?? 0);
446
+ }
447
+ unload(settings, name);
448
+ if (runs.length < docs.length) { process.stdout.write(`${name}: stopped after ${runs.length} of ${docs.length} documents; nothing recorded.\n`); continue; }
449
+
450
+ results[name] = {
451
+ peakGb: +(peak / GB).toFixed(2),
452
+ contextTokens: settings.contextTokens,
453
+ firstMs: runs[0].ms,
454
+ medianMs: median(runs.slice(1).map(r => r.ms)) ?? runs[0].ms,
455
+ tokensPerSec: runs.every(r => r.tokensPerSec) ? +median(runs.map(r => r.tokensPerSec)).toFixed(1) : null,
456
+ documents: docs.length,
457
+ server: serverVersion(settings),
458
+ measuredAt: new Date().toISOString(),
459
+ };
460
+ recorded += 1;
461
+ mkdirSync(path.dirname(measurementsFile()), { recursive: true });
462
+ writeFileSync(measurementsFile(), `${JSON.stringify(results, null, 2)}\n`);
463
+ const r = results[name];
464
+ process.stdout.write(`${name}: ${r.peakGb} GB peak, first ${(r.firstMs / 1000).toFixed(1)}s (load included), then ${(r.medianMs / 1000).toFixed(1)}s per summary${r.tokensPerSec ? `, ${r.tokensPerSec} tokens/s` : ''}.\n`);
465
+ }
466
+ if (recorded) process.stdout.write(dim(`Recorded in ${measurementsFile()}.\n`));
467
+ }
468
+
469
+ export async function runModel(argv, config) {
470
+ const json = argv.includes('--json');
471
+ const [sub = 'status', arg, ...more] = argv.filter(a => !a.startsWith('--'));
472
+ const settings = modelSettings();
473
+
474
+ if (sub === 'status') {
475
+ const d = statusData(settings);
476
+ if (json) process.stdout.write(`${JSON.stringify(d, null, 2)}\n`);
477
+ else printStatus(d);
478
+ return;
479
+ }
480
+ if (sub === 'start') {
481
+ const r = startServer(settings, { force: argv.includes('--force') });
482
+ process.stdout.write(r.already
483
+ ? `The model server is already running at ${settings.endpoint}.\n`
484
+ : `Started ollama serve (pid ${r.pid}) at ${settings.endpoint}. Nothing is loaded until a model command runs.\n`);
485
+ return;
486
+ }
487
+ if (sub === 'stop') {
488
+ if (settings.runtime !== 'ollama') die(`runlist can unload only from Ollama; stop the ${settings.runtime} server yourself.`);
489
+ if (!serverVersion(settings)) { process.stdout.write('The model server is not running; nothing is loaded.\n'); return; }
490
+ const loaded = loadedModels(settings);
491
+ const name = pickModel(settings).name;
492
+ const targets = argv.includes('--all') ? loaded : loaded.filter(m => sameModel(m.name, name));
493
+ if (!targets.length) { process.stdout.write(`Nothing of runlist's is loaded${loaded.length ? ` (${loaded.map(m => m.name).join(', ')} loaded by others; --all unloads them)` : ''}.\n`); return; }
494
+ for (const m of targets) {
495
+ unload(settings, m.name);
496
+ process.stdout.write(`Unloaded ${m.name}, ${gb(m.bytes)} freed.\n`);
497
+ }
498
+ return;
499
+ }
500
+ if (sub === 'measure') {
501
+ await measure([arg, ...more].filter(Boolean), settings, config);
502
+ return;
503
+ }
504
+ if (sub === 'use') {
505
+ if (!arg) die('Usage: runlist model use <name> (`runlist model use auto` goes back to the first candidate under the cap)');
506
+ const next = writeSettings({ model: arg === 'auto' ? null : arg });
507
+ process.stdout.write(`Model: ${next.model ?? 'auto'}. Written to ${settingsFile()}.\n`);
508
+ return;
509
+ }
510
+ if (sub === 'cap') {
511
+ if (arg === 'auto') {
512
+ writeSettings({ capGb: null });
513
+ process.stdout.write(`Cap: ${autoCapGb()} GB, from this machine's memory. Written to ${settingsFile()}.\n`);
514
+ return;
515
+ }
516
+ const n = Number(arg);
517
+ if (!(n > 0)) die('Usage: runlist model cap <gb|auto>');
518
+ writeSettings({ capGb: n });
519
+ process.stdout.write(`Cap: ${n} GB. Written to ${settingsFile()}.\n`);
520
+ return;
521
+ }
522
+ die(`Unknown: runlist model ${sub}. Use status, start, stop, measure, use <name> or cap <gb>.`);
523
+ }
package/src/show.mjs ADDED
@@ -0,0 +1,101 @@
1
+ import { readFileSync } from 'node:fs';
2
+ import path from 'node:path';
3
+ import { extractFrontmatter, parseSimpleFrontmatter } from './frontmatter.mjs';
4
+ import { parseDocFile, resolveDocArg } from './index.mjs';
5
+ import { resolveBodyLinkTarget } from './body-link.mjs';
6
+ import { die, normalizeStringList, resolveRefPath, toRepoPath } from './util.mjs';
7
+
8
+ // `runlist show <file...>` — one document's card as data: its title, status,
9
+ // next step, blockers and checklist, the plans and docs its frontmatter names,
10
+ // and the documents its body links to. Read-only; nothing is claimed.
11
+
12
+ // Frontmatter lists that name other documents. The configured reference fields
13
+ // are read too; `related_docs` is read whether or not a repo configures it.
14
+ const RELATED = ['related_plans', 'related_docs', 'supports_plans', 'parent_plan', 'runlist'];
15
+
16
+ function relatedFields(config) {
17
+ const configured = [...(config.referenceFields?.bidirectional ?? []), ...(config.referenceFields?.unidirectional ?? [])];
18
+ return [...new Set([...RELATED, ...configured])];
19
+ }
20
+
21
+ function brief(abs, config) {
22
+ try {
23
+ const d = parseDocFile(abs, config, { fast: true });
24
+ return { title: d.title, status: d.status, type: d.type };
25
+ } catch {
26
+ return { title: null, status: null, type: null };
27
+ }
28
+ }
29
+
30
+ /** The card for one document, by absolute path. */
31
+ export function showDoc(abs, config) {
32
+ const doc = parseDocFile(abs, config, { fast: true });
33
+ const raw = readFileSync(abs, 'utf8');
34
+ const fm = parseSimpleFrontmatter(extractFrontmatter(raw).frontmatter ?? '', []);
35
+ const dir = path.dirname(abs);
36
+ const seen = new Set([doc.path]);
37
+
38
+ const related = [];
39
+ for (const field of relatedFields(config)) {
40
+ for (const entry of normalizeStringList(fm[field])) {
41
+ const ref = String(entry).replace(/^>\s*/, '').replace(/#.*$/, '').trim();
42
+ if (!ref) continue;
43
+ const target = resolveRefPath(ref, dir, config.repoRoot);
44
+ const rel = target ? toRepoPath(target, config.repoRoot) : null;
45
+ if (rel && seen.has(`${field}:${rel}`)) continue;
46
+ if (rel) seen.add(`${field}:${rel}`);
47
+ related.push({ field, ref, path: rel, exists: Boolean(target), ...(target ? brief(target, config) : { title: null, status: null, type: null }) });
48
+ }
49
+ }
50
+
51
+ const named = new Set(related.map(r => r.path).filter(Boolean));
52
+ const links = [];
53
+ for (const link of doc.bodyLinks ?? []) {
54
+ if (link.targetKind !== 'document') continue;
55
+ const target = resolveBodyLinkTarget(link.href, dir, config.repoRoot);
56
+ if (!target.ok) continue;
57
+ const rel = toRepoPath(target.path, config.repoRoot);
58
+ if (seen.has(rel) || named.has(rel)) continue;
59
+ seen.add(rel);
60
+ links.push({ path: rel, ...brief(target.path, config) });
61
+ }
62
+
63
+ return {
64
+ path: doc.path,
65
+ type: doc.type,
66
+ title: doc.title,
67
+ status: doc.status,
68
+ summary: doc.summary,
69
+ currentState: doc.currentState === 'No current_state set' ? null : doc.currentState,
70
+ nextStep: doc.nextStep,
71
+ blockers: doc.blockers,
72
+ checklist: doc.checklist,
73
+ updated: doc.updated,
74
+ related,
75
+ links,
76
+ };
77
+ }
78
+
79
+ export function runShow(args, config) {
80
+ const json = args.includes('--json');
81
+ const targets = args.filter(a => !a.startsWith('-'));
82
+ if (!targets.length) die('Usage: runlist show <file...> [--json]');
83
+ const cards = [];
84
+ for (const t of targets) {
85
+ const abs = resolveDocArg(t, config, { dieOnMiss: !json });
86
+ cards.push(abs ? showDoc(abs, config) : { path: t, error: 'not found' });
87
+ }
88
+ if (json) {
89
+ process.stdout.write(`${JSON.stringify(cards, null, 2)}\n`);
90
+ return;
91
+ }
92
+ cards.forEach((c, n) => {
93
+ if (n) process.stdout.write('\n');
94
+ process.stdout.write(`${c.title} (${c.status ?? 'no status'})\n${c.path}\n`);
95
+ if (c.nextStep) process.stdout.write(`Next: ${c.nextStep}\n`);
96
+ for (const b of c.blockers) process.stdout.write(`Blocker: ${b}\n`);
97
+ if (c.checklist?.total) process.stdout.write(`Checklist: ${c.checklist.completed} of ${c.checklist.total} done\n`);
98
+ for (const r of c.related) process.stdout.write(`${r.field}: ${r.path ?? `${r.ref} (missing)`}${r.status ? ` (${r.status})` : ''}\n`);
99
+ if (c.links.length) process.stdout.write(`Linked from the body: ${c.links.map(l => l.path).join(', ')}\n`);
100
+ });
101
+ }
package/src/summary.mjs CHANGED
@@ -69,6 +69,6 @@ export function runSummary(argv, config) {
69
69
  } else if (summary) {
70
70
  process.stdout.write(`${summary}\n`);
71
71
  } else {
72
- process.stdout.write(dim('Summary unavailable (model call failed or uv not installed).') + '\n');
72
+ process.stdout.write(dim('Summary unavailable (no model available; `runlist model` says why).') + '\n');
73
73
  }
74
74
  }
package/src/validate.mjs CHANGED
@@ -272,7 +272,7 @@ export function validateDoc(doc, frontmatter, headingTitle, config) {
272
272
  }
273
273
 
274
274
  for (const link of (doc.bodyLinks || [])) {
275
- const resolution = resolveBodyLinkTarget(link.href, docDir, config.repoRoot);
275
+ const resolution = resolveBodyLinkTarget(link.href, docDir, config.repoRoot, config.externalBodyLinkRoots);
276
276
  if (!resolution.ok) {
277
277
  const shownHref = link.rawHref ?? link.href;
278
278
  doc.errors.push({