@otto-code/brain 0.7.5 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bench/context-corpus.js +3 -3
- package/dist/bench/corpus.js +2 -2
- package/dist/bench/curated-repos.js +3 -3
- package/dist/bench/health.d.ts +1 -1
- package/dist/bench/health.js +2 -2
- package/dist/bench/tasks.js +2 -2
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +1 -1
- package/dist/commands/bench.d.ts +1 -1
- package/dist/commands/bench.js +27 -4
- package/dist/commands/calibrate.d.ts +1 -1
- package/dist/commands/calibrate.js +5 -2
- package/dist/commands/catalog.d.ts +1 -1
- package/dist/commands/config.d.ts +1 -1
- package/dist/commands/config.js +9 -0
- package/dist/commands/lifecycle.js +1 -1
- package/dist/commands/pull.d.ts +1 -1
- package/dist/commands/pull.js +9 -4
- package/dist/commands/report.d.ts +1 -1
- package/dist/commands/rescore.d.ts +1 -1
- package/dist/commands/rescore.js +1 -1
- package/dist/commands/runtime.d.ts +1 -1
- package/dist/commands/scan.d.ts +1 -1
- package/dist/commands/scan.js +6 -1
- package/dist/commands/search.d.ts +1 -1
- package/dist/commands/share.js +2 -2
- package/dist/commands/sweep.d.ts +2 -2
- package/dist/commands/sweep.js +22 -8
- package/dist/commands/ui.d.ts +1 -1
- package/dist/commands/ui.js +37 -5
- package/dist/config/index.d.ts +2 -1
- package/dist/config/index.js +2 -1
- package/dist/config/otto-home.js +1 -1
- package/dist/config/paths.d.ts +1 -0
- package/dist/config/paths.js +4 -0
- package/dist/config/profile-edit.d.ts +94 -0
- package/dist/config/profile-edit.js +269 -0
- package/dist/config/profiles.d.ts +2 -2
- package/dist/config/profiles.js +1 -1
- package/dist/config/schema.d.ts +4 -4
- package/dist/config/schema.js +6 -6
- package/dist/config/store.d.ts +1 -1
- package/dist/config/store.js +1 -1
- package/dist/models/download.js +1 -1
- package/dist/models/enrich.d.ts +2 -2
- package/dist/models/index.d.ts +1 -1
- package/dist/models/index.js +1 -1
- package/dist/models/pick.d.ts +9 -0
- package/dist/models/pick.js +24 -1
- package/dist/ops/report.js +28 -28
- package/dist/ops/results.d.ts +128 -9
- package/dist/ops/results.js +77 -5
- package/dist/output/render.js +1 -1
- package/dist/output/types.d.ts +1 -1
- package/dist/runtime/args.d.ts +1 -1
- package/dist/runtime/args.js +1 -1
- package/dist/runtime/managed.js +4 -4
- package/dist/service/activity.d.ts +83 -0
- package/dist/service/activity.js +216 -0
- package/dist/service/host-api.d.ts +132 -0
- package/dist/service/host-api.js +397 -0
- package/dist/service/http-util.d.ts +27 -0
- package/dist/service/http-util.js +72 -0
- package/dist/service/model-selector.d.ts +2 -2
- package/dist/service/model-selector.js +6 -6
- package/dist/service/router.d.ts +28 -4
- package/dist/service/router.js +128 -94
- package/dist/service/scheduler.d.ts +2 -2
- package/dist/service/scheduler.js +1 -1
- package/dist/service/serve.d.ts +8 -1
- package/dist/service/serve.js +69 -16
- package/dist/service/supervisor.d.ts +6 -0
- package/dist/service/supervisor.js +2 -0
- package/dist/service/tailscale.js +1 -1
- package/dist/service/tls.d.ts +4 -4
- package/dist/service/tls.js +3 -3
- package/dist/sysmon.d.ts +20 -4
- package/dist/sysmon.js +42 -18
- package/dist/tui/app.d.ts +12 -2
- package/dist/tui/app.js +48 -23
- package/dist/vram.d.ts +8 -1
- package/dist/vram.js +6 -3
- package/package.json +1 -1
package/dist/service/tls.js
CHANGED
|
@@ -4,11 +4,11 @@
|
|
|
4
4
|
* generalized from otto-brain-relay's `cert.js`.
|
|
5
5
|
*
|
|
6
6
|
* Modes (resolved from `config.tls`):
|
|
7
|
-
* - `files`
|
|
8
|
-
* - `self-signed`
|
|
7
|
+
* - `files` - read the cert/key paths you provide; no renewal.
|
|
8
|
+
* - `self-signed` - generate a local keypair on first run, cached under
|
|
9
9
|
* `certDir`; regenerated when it nears expiry. Clients must
|
|
10
10
|
* trust it (or pass `-k`); there is no chain of trust.
|
|
11
|
-
* - `tailscale`
|
|
11
|
+
* - `tailscale` - issue/renew a real Let's Encrypt cert for this machine's
|
|
12
12
|
* MagicDNS name via `tailscaled`, so tailnet clients see no
|
|
13
13
|
* warnings. Renewal is hot: `renewed` fires with the new
|
|
14
14
|
* { cert, key } and the server swaps its secure context
|
package/dist/sysmon.d.ts
CHANGED
|
@@ -7,17 +7,29 @@ import { query } from "./gpu.js";
|
|
|
7
7
|
* inferred from request counting, so it reflects what the engine is really
|
|
8
8
|
* doing. That matters for agentic use: several concurrent requests are only
|
|
9
9
|
* genuinely parallel if there are free slots to take them, and slots share one
|
|
10
|
-
* KV pool
|
|
10
|
+
* KV pool - so more concurrency costs context per request.
|
|
11
11
|
*/
|
|
12
12
|
/** A CPU busy-fraction sampler: returns the fraction in [0,1], or null. */
|
|
13
13
|
export interface CpuSampler {
|
|
14
14
|
(): number | null;
|
|
15
15
|
}
|
|
16
|
-
/**
|
|
16
|
+
/**
|
|
17
|
+
* Slot occupancy from the running server.
|
|
18
|
+
*
|
|
19
|
+
* `busy` is the sum of `prefill` and `decode`. The split matters because the two
|
|
20
|
+
* phases feel completely different from outside: prefill is a single batched
|
|
21
|
+
* pass over the prompt that pins the GPU and returns nothing, decode is the
|
|
22
|
+
* token-at-a-time stream. A UI that can only say "busy" cannot tell a long
|
|
23
|
+
* prompt being ingested from a model that has started answering.
|
|
24
|
+
*/
|
|
17
25
|
export interface SlotInfo {
|
|
18
26
|
total: number;
|
|
19
27
|
busy: number;
|
|
20
28
|
idle: number;
|
|
29
|
+
/** Slots ingesting a prompt: processing, but not a single token emitted yet. */
|
|
30
|
+
prefill: number;
|
|
31
|
+
/** Slots emitting tokens. */
|
|
32
|
+
decode: number;
|
|
21
33
|
contexts: number[];
|
|
22
34
|
}
|
|
23
35
|
/** One combined reading for the status panel. */
|
|
@@ -37,9 +49,13 @@ interface Endpoint {
|
|
|
37
49
|
/** CPU busy fraction, sampled between successive calls. */
|
|
38
50
|
export declare function createCpuSampler(): CpuSampler;
|
|
39
51
|
/**
|
|
40
|
-
*
|
|
41
|
-
*
|
|
52
|
+
* Reduce llama-server's `/slots` array to the occupancy the UI shows.
|
|
53
|
+
*
|
|
54
|
+
* Exported separately from the fetch so the phase split can be tested against
|
|
55
|
+
* the field spellings real llama.cpp builds emit, without a live server.
|
|
42
56
|
*/
|
|
57
|
+
export declare function summariseSlots(rows: unknown[]): SlotInfo;
|
|
58
|
+
/** Slot occupancy from the running server. */
|
|
43
59
|
declare function slots({ host, port }: {
|
|
44
60
|
host: string;
|
|
45
61
|
port: number;
|
package/dist/sysmon.js
CHANGED
|
@@ -51,28 +51,45 @@ function fetchJson({ host, port, path: urlPath, timeout = 2500, }) {
|
|
|
51
51
|
req.on("error", () => resolve(null));
|
|
52
52
|
});
|
|
53
53
|
}
|
|
54
|
+
/** Whether a slot is doing anything. Field naming has varied across versions. */
|
|
55
|
+
function isProcessing(rec) {
|
|
56
|
+
if (typeof rec.is_processing === "boolean")
|
|
57
|
+
return rec.is_processing;
|
|
58
|
+
if (typeof rec.state === "number")
|
|
59
|
+
return rec.state !== 0;
|
|
60
|
+
return false;
|
|
61
|
+
}
|
|
62
|
+
/** How many tokens this slot has emitted for the request it is on. */
|
|
63
|
+
function decodedTokens(rec) {
|
|
64
|
+
for (const key of ["n_decoded", "n_decoded_tokens", "tokens_predicted"]) {
|
|
65
|
+
const value = rec[key];
|
|
66
|
+
if (typeof value === "number" && Number.isFinite(value))
|
|
67
|
+
return value;
|
|
68
|
+
}
|
|
69
|
+
// No counter at all: report a non-zero so the slot lands in decode rather than
|
|
70
|
+
// claiming a prefill that may never have been happening.
|
|
71
|
+
return 1;
|
|
72
|
+
}
|
|
54
73
|
/**
|
|
55
|
-
*
|
|
56
|
-
*
|
|
74
|
+
* Reduce llama-server's `/slots` array to the occupancy the UI shows.
|
|
75
|
+
*
|
|
76
|
+
* Exported separately from the fetch so the phase split can be tested against
|
|
77
|
+
* the field spellings real llama.cpp builds emit, without a live server.
|
|
57
78
|
*/
|
|
58
|
-
|
|
59
|
-
const
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
//
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
if (typeof rec.is_processing === "boolean")
|
|
67
|
-
return rec.is_processing;
|
|
68
|
-
if (typeof rec.state === "number")
|
|
69
|
-
return rec.state !== 0;
|
|
70
|
-
return false;
|
|
71
|
-
}).length;
|
|
79
|
+
export function summariseSlots(rows) {
|
|
80
|
+
const busyRows = rows.filter((s) => isProcessing(s));
|
|
81
|
+
// A busy slot that has not yet decoded a token is still ingesting its prompt.
|
|
82
|
+
// The decoded counter is the only field that separates the two phases, and it
|
|
83
|
+
// has been spelled three ways across llama.cpp versions; a slot that reports
|
|
84
|
+
// none of them counts as decode, because a busy slot that cannot prove it is
|
|
85
|
+
// still prefilling is far more likely to be mid-answer than mid-prompt.
|
|
86
|
+
const prefill = busyRows.filter((s) => decodedTokens(s) === 0).length;
|
|
72
87
|
return {
|
|
73
88
|
total: rows.length,
|
|
74
|
-
busy,
|
|
75
|
-
idle: rows.length -
|
|
89
|
+
busy: busyRows.length,
|
|
90
|
+
idle: rows.length - busyRows.length,
|
|
91
|
+
prefill,
|
|
92
|
+
decode: busyRows.length - prefill,
|
|
76
93
|
contexts: rows.map((s) => {
|
|
77
94
|
const rec = s;
|
|
78
95
|
const nCtx = typeof rec.n_ctx === "number" ? rec.n_ctx : undefined;
|
|
@@ -81,6 +98,13 @@ async function slots({ host, port }) {
|
|
|
81
98
|
}),
|
|
82
99
|
};
|
|
83
100
|
}
|
|
101
|
+
/** Slot occupancy from the running server. */
|
|
102
|
+
async function slots({ host, port }) {
|
|
103
|
+
const data = await fetchJson({ host, port, path: "/slots" });
|
|
104
|
+
if (!Array.isArray(data))
|
|
105
|
+
return null;
|
|
106
|
+
return summariseSlots(data);
|
|
107
|
+
}
|
|
84
108
|
export { slots };
|
|
85
109
|
/** One combined reading for the status panel. */
|
|
86
110
|
export async function sample(sampler, { host, port } = {}) {
|
package/dist/tui/app.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { Screen } from "./screen.js";
|
|
2
2
|
import type { DiskUsage, QuantOption, RepoQuants, ModelSearchResult } from "../models/index.js";
|
|
3
|
+
import * as vram from "../vram.js";
|
|
3
4
|
import { Supervisor } from "../service/supervisor.js";
|
|
4
5
|
import { Telemetry } from "../service/router.js";
|
|
5
6
|
import http from "node:http";
|
|
@@ -103,6 +104,15 @@ export declare class App {
|
|
|
103
104
|
telemetry: Telemetry;
|
|
104
105
|
supervisor: Supervisor;
|
|
105
106
|
routerServer: http.Server | null;
|
|
107
|
+
/**
|
|
108
|
+
* The VRAM fit behind the resident model, so a benchmark can record whether
|
|
109
|
+
* the profile it measured was the profile the user configured. Keyed by model
|
|
110
|
+
* id because a fit computed for one model says nothing about the next.
|
|
111
|
+
*/
|
|
112
|
+
lastFit: {
|
|
113
|
+
modelId: string;
|
|
114
|
+
fit: vram.FitResult;
|
|
115
|
+
} | null;
|
|
106
116
|
filterMode: boolean;
|
|
107
117
|
confirming: ConfirmState | null;
|
|
108
118
|
picker: PickerState | null;
|
|
@@ -199,8 +209,8 @@ export declare class App {
|
|
|
199
209
|
drawHelp(cols: number): void;
|
|
200
210
|
/**
|
|
201
211
|
* The key hints for the current mode, as an array of lines. Groups are
|
|
202
|
-
* deliberately broken onto separate lines
|
|
203
|
-
* actions
|
|
212
|
+
* deliberately broken onto separate lines - navigation first, then the
|
|
213
|
+
* actions - and each group wraps further only if the terminal is too narrow.
|
|
204
214
|
*/
|
|
205
215
|
keybindings(cols: number): string[];
|
|
206
216
|
}
|
package/dist/tui/app.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { Screen, box, meter, style, pad, truncate, width, onKeys } from "./screen.js";
|
|
2
2
|
import { scanModels, managedModelsDir, diskUsage, totalModelBytes, planDelete, deleteModelFiles, listRepoQuants, searchModels, downloadRepoFiles, resolveHfToken, } from "../models/index.js";
|
|
3
3
|
import * as profiles from "../config/profiles.js";
|
|
4
|
-
import { loadBrainConfig, loadProfilesStore, saveProfilesStore } from "../config/index.js";
|
|
4
|
+
import { formatReasoningBudget, loadBrainConfig, loadProfilesStore, saveProfilesStore, UNRESTRICTED_REASONING_BUDGET, } from "../config/index.js";
|
|
5
5
|
import * as vram from "../vram.js";
|
|
6
6
|
import * as gpu from "../gpu.js";
|
|
7
7
|
import { calibrate } from "../ops/calibrate.js";
|
|
@@ -11,6 +11,7 @@ import * as archive from "../ops/archive.js";
|
|
|
11
11
|
import { Supervisor } from "../service/supervisor.js";
|
|
12
12
|
import { createRouter, Telemetry } from "../service/router.js";
|
|
13
13
|
import * as sysmon from "../sysmon.js";
|
|
14
|
+
import { resolveVersion } from "../version.js";
|
|
14
15
|
import http from "node:http";
|
|
15
16
|
// The config panel holds short fields, so keep it compact and give the rest of
|
|
16
17
|
// the width to the model list (long model names need the room).
|
|
@@ -83,8 +84,8 @@ export const FIELDS = [
|
|
|
83
84
|
kind: "cycle",
|
|
84
85
|
values: REASONING_CYCLE,
|
|
85
86
|
format: (p) => {
|
|
86
|
-
if (p.reasoningBudget ===
|
|
87
|
-
return `${style.red}
|
|
87
|
+
if (p.reasoningBudget === UNRESTRICTED_REASONING_BUDGET)
|
|
88
|
+
return `${style.red}${formatReasoningBudget(p.reasoningBudget)}${style.reset}`;
|
|
88
89
|
if (p.reasoningBudget === 0)
|
|
89
90
|
return `${style.cyan}thinking off${style.reset}`;
|
|
90
91
|
return `${p.reasoningBudget} tokens`;
|
|
@@ -113,6 +114,12 @@ export const FIELDS = [
|
|
|
113
114
|
];
|
|
114
115
|
export class App {
|
|
115
116
|
constructor({ runtime, listenPort, listenHost }) {
|
|
117
|
+
/**
|
|
118
|
+
* The VRAM fit behind the resident model, so a benchmark can record whether
|
|
119
|
+
* the profile it measured was the profile the user configured. Keyed by model
|
|
120
|
+
* id because a fit computed for one model says nothing about the next.
|
|
121
|
+
*/
|
|
122
|
+
this.lastFit = null;
|
|
116
123
|
this.filterMode = false;
|
|
117
124
|
this.confirming = null;
|
|
118
125
|
this.picker = null;
|
|
@@ -198,6 +205,7 @@ export class App {
|
|
|
198
205
|
async loadModelFitted(target) {
|
|
199
206
|
const info = this.gpuInfo || (await gpu.query());
|
|
200
207
|
let profile = profiles.forModel(this.store, target);
|
|
208
|
+
this.lastFit = null;
|
|
201
209
|
if (info) {
|
|
202
210
|
const fit = vram.fitToBudget({
|
|
203
211
|
model: target,
|
|
@@ -207,6 +215,9 @@ export class App {
|
|
|
207
215
|
});
|
|
208
216
|
if (!fit.adjusted && !fit.budget.fits)
|
|
209
217
|
throw new Error(fit.reason ?? undefined);
|
|
218
|
+
// Held for the benchmark to record: once `fit.profile` is applied, the
|
|
219
|
+
// context the user actually asked for is gone from every other source.
|
|
220
|
+
this.lastFit = { modelId: target.id, fit };
|
|
210
221
|
profile = fit.profile;
|
|
211
222
|
}
|
|
212
223
|
this.setStatus(`switching to ${target.displayName}…`, "info");
|
|
@@ -614,7 +625,7 @@ export class App {
|
|
|
614
625
|
}
|
|
615
626
|
const plan = planDelete(model);
|
|
616
627
|
this.confirming = { kind: "delete", model };
|
|
617
|
-
this.setStatus(`delete ${model.displayName}
|
|
628
|
+
this.setStatus(`delete ${model.displayName} - frees ${vram.formatGiB(plan.bytes)}` +
|
|
618
629
|
`${plan.includesProjector ? " (incl. projector)" : ""}? y / n`, "warn");
|
|
619
630
|
this.draw();
|
|
620
631
|
}
|
|
@@ -628,7 +639,7 @@ export class App {
|
|
|
628
639
|
const plan = deleteModelFiles(confirming.model);
|
|
629
640
|
this.reload();
|
|
630
641
|
void this.refreshDisk();
|
|
631
|
-
this.setStatus(`deleted ${confirming.model.displayName}
|
|
642
|
+
this.setStatus(`deleted ${confirming.model.displayName} - freed ${vram.formatGiB(plan.bytes)}`, "good");
|
|
632
643
|
});
|
|
633
644
|
}
|
|
634
645
|
else if (key === "n" || key === "escape") {
|
|
@@ -892,7 +903,7 @@ export class App {
|
|
|
892
903
|
}
|
|
893
904
|
const lines = [this.header(cols), ""];
|
|
894
905
|
lines.push(...box({
|
|
895
|
-
title: `Download quant
|
|
906
|
+
title: `Download quant - ${truncate(picker.repo, Math.max(8, inner - 18))}`,
|
|
896
907
|
lines: rows,
|
|
897
908
|
innerWidth: inner,
|
|
898
909
|
footer: picker.loading
|
|
@@ -1100,7 +1111,7 @@ export class App {
|
|
|
1100
1111
|
const archiveId = archive.runId(model);
|
|
1101
1112
|
const healthSampler = health.start();
|
|
1102
1113
|
// stop() is idempotent (clearInterval); guarantee the 1s nvidia-smi
|
|
1103
|
-
// sampler never outlives the run even if runSuite throws
|
|
1114
|
+
// sampler never outlives the run even if runSuite throws - otherwise a
|
|
1104
1115
|
// failed bench leaks a recurring subprocess.
|
|
1105
1116
|
let sampled = false;
|
|
1106
1117
|
try {
|
|
@@ -1119,7 +1130,7 @@ export class App {
|
|
|
1119
1130
|
`${p.title}: ${(p.score * 100).toFixed(0)}% ${p.summary}`;
|
|
1120
1131
|
}
|
|
1121
1132
|
else if (p.phase === "failed")
|
|
1122
|
-
this.benchProgress.push(`${p.title}: failed
|
|
1133
|
+
this.benchProgress.push(`${p.title}: failed - ${p.summary}`);
|
|
1123
1134
|
this.draw();
|
|
1124
1135
|
},
|
|
1125
1136
|
});
|
|
@@ -1135,10 +1146,24 @@ export class App {
|
|
|
1135
1146
|
runtime: `${this.runtime.label} v${this.runtime.version}`,
|
|
1136
1147
|
system: report.system,
|
|
1137
1148
|
archiveId,
|
|
1149
|
+
args: this.supervisor.args,
|
|
1150
|
+
// Only when it belongs to the model actually being benchmarked - the
|
|
1151
|
+
// model may have been resident since before this fit was computed.
|
|
1152
|
+
fit: this.lastFit?.modelId === model.id ? this.lastFit.fit : null,
|
|
1153
|
+
calibration: profile ? profiles.getCalibration(this.store, model, profile) : null,
|
|
1154
|
+
suite: {
|
|
1155
|
+
// The TUI runs the full static suite; only concurrency varies, and
|
|
1156
|
+
// it tracks the profile's slot count the same way runSuite is called.
|
|
1157
|
+
execute: true,
|
|
1158
|
+
concurrency: Math.max(1, profile?.parallelSlots || 3),
|
|
1159
|
+
depths: null,
|
|
1160
|
+
only: null,
|
|
1161
|
+
mined: false,
|
|
1162
|
+
},
|
|
1138
1163
|
});
|
|
1139
1164
|
this.loadBenchResults();
|
|
1140
1165
|
this.loadRankings();
|
|
1141
|
-
this.benchProgress.push(`done
|
|
1166
|
+
this.benchProgress.push(`done - overall ${(report.overall * 100).toFixed(0)}% (${report.grade})`);
|
|
1142
1167
|
this.setStatus(`benchmark complete: ${model.displayName} ${(report.overall * 100).toFixed(0)}% (${report.grade})`, "good");
|
|
1143
1168
|
}
|
|
1144
1169
|
finally {
|
|
@@ -1346,7 +1371,7 @@ export class App {
|
|
|
1346
1371
|
return box({
|
|
1347
1372
|
title: "Leaderboard",
|
|
1348
1373
|
lines: [
|
|
1349
|
-
`${style.grey}no benchmarks yet
|
|
1374
|
+
`${style.grey}no benchmarks yet - select a model and press r to run one${style.reset}`,
|
|
1350
1375
|
],
|
|
1351
1376
|
innerWidth: innerWidth - 2,
|
|
1352
1377
|
});
|
|
@@ -1382,7 +1407,7 @@ export class App {
|
|
|
1382
1407
|
const bodyRows = Math.max(1, this.screen.rows - 1 - overhead);
|
|
1383
1408
|
const body = [];
|
|
1384
1409
|
if (logs.length === 0) {
|
|
1385
|
-
body.push(`${style.grey}no logs yet
|
|
1410
|
+
body.push(`${style.grey}no logs yet - start a model with s to see llama-server output${style.reset}`);
|
|
1386
1411
|
}
|
|
1387
1412
|
else {
|
|
1388
1413
|
for (const line of logs.slice(Math.max(0, logs.length - bodyRows)))
|
|
@@ -1394,7 +1419,7 @@ export class App {
|
|
|
1394
1419
|
this.renderFitted(lines);
|
|
1395
1420
|
}
|
|
1396
1421
|
header(cols) {
|
|
1397
|
-
const title = `${style.bold}${style.brightCyan}Otto Brain${style.reset}`;
|
|
1422
|
+
const title = `${style.bold}${style.brightCyan}Otto Brain${style.reset}${style.grey} v${resolveVersion()}${style.reset}`;
|
|
1398
1423
|
const rt = `${style.grey}llama.cpp ${this.runtime.label} v${this.runtime.version}${style.reset}`;
|
|
1399
1424
|
const g = this.gpuInfo
|
|
1400
1425
|
? `${style.grey}${this.gpuInfo.name} · ${vram.formatGiB(this.gpuInfo.usedBytes)}/${vram.formatGiB(this.gpuInfo.totalBytes)}${style.reset}`
|
|
@@ -1657,19 +1682,19 @@ export class App {
|
|
|
1657
1682
|
section("Change settings (Configuration panel)");
|
|
1658
1683
|
item("← →", "change the selected field (toggle / cycle / ± step)");
|
|
1659
1684
|
item("- +", "same as ← →");
|
|
1660
|
-
item("Enter", "edit a number field
|
|
1685
|
+
item("Enter", "edit a number field - type digits, Enter saves, Esc cancels");
|
|
1661
1686
|
item("m", "set context to the largest size that fits in VRAM");
|
|
1662
1687
|
section("Run the model");
|
|
1663
1688
|
item("s", "start / load the selected model");
|
|
1664
1689
|
item("x", "stop the running model");
|
|
1665
|
-
item("c", "calibrate
|
|
1666
|
-
item("w", "sweep
|
|
1690
|
+
item("c", "calibrate - measure real VRAM per token");
|
|
1691
|
+
item("w", "sweep - find the best reasoning budget");
|
|
1667
1692
|
section("Manage models");
|
|
1668
|
-
item("f", "find on Hugging Face
|
|
1669
|
-
item("g", "get a quant
|
|
1693
|
+
item("f", "find on Hugging Face - search and add a new model");
|
|
1694
|
+
item("g", "get a quant - pick Q4/Q5/Q6… to download for this repo");
|
|
1670
1695
|
item("D", "delete the selected model (frees disk, asks to confirm)");
|
|
1671
1696
|
section("Views");
|
|
1672
|
-
item("b", "benchmark mode
|
|
1697
|
+
item("b", "benchmark mode - rank models, run the coding suite");
|
|
1673
1698
|
item("l", "view the live llama-server log");
|
|
1674
1699
|
item("/", "filter the model list (Enter apply, Esc clear)");
|
|
1675
1700
|
item("r", "rescan the models folder");
|
|
@@ -1679,7 +1704,7 @@ export class App {
|
|
|
1679
1704
|
item("q Ctrl-C", "quit Otto Brain");
|
|
1680
1705
|
const lines = [this.header(cols), ""];
|
|
1681
1706
|
lines.push(...box({
|
|
1682
|
-
title: "Help
|
|
1707
|
+
title: "Help - every key and what it does",
|
|
1683
1708
|
lines: rows,
|
|
1684
1709
|
innerWidth: inner,
|
|
1685
1710
|
footer: `${style.grey}esc or ? to go back${style.reset}`,
|
|
@@ -1689,8 +1714,8 @@ export class App {
|
|
|
1689
1714
|
}
|
|
1690
1715
|
/**
|
|
1691
1716
|
* The key hints for the current mode, as an array of lines. Groups are
|
|
1692
|
-
* deliberately broken onto separate lines
|
|
1693
|
-
* actions
|
|
1717
|
+
* deliberately broken onto separate lines - navigation first, then the
|
|
1718
|
+
* actions - and each group wraps further only if the terminal is too narrow.
|
|
1694
1719
|
*/
|
|
1695
1720
|
keybindings(cols) {
|
|
1696
1721
|
let groups;
|
|
@@ -1751,14 +1776,14 @@ export class App {
|
|
|
1751
1776
|
}
|
|
1752
1777
|
else {
|
|
1753
1778
|
groups = [
|
|
1754
|
-
// Navigation
|
|
1779
|
+
// Navigation - line one.
|
|
1755
1780
|
[
|
|
1756
1781
|
["↑↓", "select"],
|
|
1757
1782
|
["tab", "panel"],
|
|
1758
1783
|
["←→", "change"],
|
|
1759
1784
|
["enter", "edit"],
|
|
1760
1785
|
],
|
|
1761
|
-
// Actions
|
|
1786
|
+
// Actions - line two onward.
|
|
1762
1787
|
[
|
|
1763
1788
|
["s", "start"],
|
|
1764
1789
|
["x", "stop"],
|
package/dist/vram.d.ts
CHANGED
|
@@ -53,12 +53,19 @@ export interface FitResult {
|
|
|
53
53
|
adjusted: boolean;
|
|
54
54
|
reason: string | null;
|
|
55
55
|
budget: Budget;
|
|
56
|
+
/**
|
|
57
|
+
* The context the caller asked for, before any adjustment. `profile` is the
|
|
58
|
+
* one that will actually run, so once a fit has been applied the original is
|
|
59
|
+
* unrecoverable from it - and a benchmark that does not record what it asked
|
|
60
|
+
* for cannot later explain why it scored the way it did.
|
|
61
|
+
*/
|
|
62
|
+
requestedContextSize: number;
|
|
56
63
|
}
|
|
57
64
|
/**
|
|
58
65
|
* Adapt a profile to the hardware it is about to run on.
|
|
59
66
|
*
|
|
60
67
|
* Refusing to load because a saved profile asks for more context than this
|
|
61
|
-
* machine has is unhelpful when a slightly smaller context would work
|
|
68
|
+
* machine has is unhelpful when a slightly smaller context would work - and it
|
|
62
69
|
* is the difference between a model being usable on a 32GB desktop and a 24GB
|
|
63
70
|
* laptop. Clamp the context instead, and report what changed.
|
|
64
71
|
*/
|
package/dist/vram.js
CHANGED
|
@@ -99,14 +99,15 @@ export function maxContextThatFits({ model, profile, calibration, totalVramBytes
|
|
|
99
99
|
* Adapt a profile to the hardware it is about to run on.
|
|
100
100
|
*
|
|
101
101
|
* Refusing to load because a saved profile asks for more context than this
|
|
102
|
-
* machine has is unhelpful when a slightly smaller context would work
|
|
102
|
+
* machine has is unhelpful when a slightly smaller context would work - and it
|
|
103
103
|
* is the difference between a model being usable on a 32GB desktop and a 24GB
|
|
104
104
|
* laptop. Clamp the context instead, and report what changed.
|
|
105
105
|
*/
|
|
106
106
|
export function fitToBudget({ model, profile, calibration, totalVramBytes, reserveBytes = 1.5 * GIB, }) {
|
|
107
|
+
const requestedContextSize = profile.contextSize;
|
|
107
108
|
const initial = budget({ model, profile, calibration, totalVramBytes, reserveBytes });
|
|
108
109
|
if (initial.fits) {
|
|
109
|
-
return { profile, adjusted: false, reason: null, budget: initial };
|
|
110
|
+
return { profile, adjusted: false, reason: null, budget: initial, requestedContextSize };
|
|
110
111
|
}
|
|
111
112
|
const max = maxContextThatFits({ model, profile, calibration, totalVramBytes, reserveBytes });
|
|
112
113
|
if (!max || max < 4096) {
|
|
@@ -116,14 +117,16 @@ export function fitToBudget({ model, profile, calibration, totalVramBytes, reser
|
|
|
116
117
|
reason: `does not fit at any usable context (needs ${formatGiB(initial.totalBytes)}, ` +
|
|
117
118
|
`${formatGiB(initial.usableBytes)} usable)`,
|
|
118
119
|
budget: initial,
|
|
120
|
+
requestedContextSize,
|
|
119
121
|
};
|
|
120
122
|
}
|
|
121
123
|
const fitted = { ...profile, contextSize: max };
|
|
122
124
|
return {
|
|
123
125
|
profile: fitted,
|
|
124
126
|
adjusted: true,
|
|
125
|
-
reason: `context reduced ${
|
|
127
|
+
reason: `context reduced ${requestedContextSize.toLocaleString()} -> ${max.toLocaleString()} to fit ${formatGiB(totalVramBytes)} of VRAM`,
|
|
126
128
|
budget: budget({ model, profile: fitted, calibration, totalVramBytes, reserveBytes }),
|
|
129
|
+
requestedContextSize,
|
|
127
130
|
};
|
|
128
131
|
}
|
|
129
132
|
export function formatGiB(bytes, digits = 1) {
|
package/package.json
CHANGED