@1agh/maude 0.54.0 → 0.56.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/apps/studio/annotations-layer.tsx +11 -1
- package/apps/studio/bin/_smart-frames.mjs +187 -20
- package/apps/studio/bin/_smart-frames.test.mjs +59 -4
- package/apps/studio/bin/_transcribe.mjs +40 -3
- package/apps/studio/bin/smoke.sh +7 -1
- package/apps/studio/canvas-build-sandbox.ts +101 -9
- package/apps/studio/canvas-build-worker.ts +7 -1
- package/apps/studio/canvas-build.ts +9 -1
- package/apps/studio/client/app.jsx +49 -3
- package/apps/studio/client/github.js +7 -0
- package/apps/studio/client/panels/CloudBar.jsx +505 -45
- package/apps/studio/client/panels/GitPanel.jsx +38 -25
- package/apps/studio/client/panels/SettingsPanel.jsx +239 -27
- package/apps/studio/client/styles/3-shell-maude.css +30 -0
- package/apps/studio/client/styles/4-components.css +72 -0
- package/apps/studio/cloud/endpoints.ts +104 -9
- package/apps/studio/collab/persistence.ts +29 -2
- package/apps/studio/config.schema.json +3 -3
- package/apps/studio/context.ts +41 -0
- package/apps/studio/dist/client.bundle.js +1545 -1545
- package/apps/studio/dist/comment-mount.js +2 -2
- package/apps/studio/dist/styles.css +1 -1
- package/apps/studio/generation/gemma-models.ts +312 -14
- package/apps/studio/generation/prefs.ts +7 -2
- package/apps/studio/generation/runtime-probe.ts +50 -0
- package/apps/studio/generation/whisper-models.ts +124 -0
- package/apps/studio/hmr-broadcast.ts +67 -0
- package/apps/studio/http.ts +210 -110
- package/apps/studio/input-router.tsx +55 -2
- package/apps/studio/server.ts +11 -9
- package/apps/studio/sync/autocommit.ts +61 -2
- package/apps/studio/sync/cell-pairing.ts +174 -0
- package/apps/studio/sync/codec.ts +11 -5
- package/apps/studio/sync/index.ts +239 -26
- package/apps/studio/sync/limits.ts +49 -0
- package/apps/studio/sync/loopback.ts +21 -0
- package/apps/studio/sync/projection.ts +47 -12
- package/apps/studio/sync/supervisor.ts +178 -0
- package/apps/studio/test/cloud-endpoints.test.ts +326 -4
- package/apps/studio/test/csrf-write-guard.test.ts +19 -2
- package/apps/studio/test/gemma-models.test.ts +245 -0
- package/apps/studio/test/hmr-broadcast.test.ts +57 -1
- package/apps/studio/test/input-router.test.ts +95 -0
- package/apps/studio/test/shared-doc-cell-pairing.test.ts +639 -0
- package/apps/studio/test/sync-autocommit.test.ts +47 -0
- package/apps/studio/test/sync-supervisor.test.ts +212 -0
- package/apps/studio/test/trusted-request-host.test.ts +66 -0
- package/apps/studio/test/whisper-setup.test.ts +97 -0
- package/apps/studio/whats-new.json +45 -0
- package/apps/studio/ws.ts +9 -1
- package/cli/commands/kg.mjs +9 -2
- package/package.json +8 -8
- package/plugins/design/dependencies.json +21 -3
|
@@ -21,6 +21,39 @@ import { spawn, spawnSync } from 'node:child_process';
|
|
|
21
21
|
import { existsSync, readdirSync, statSync } from 'node:fs';
|
|
22
22
|
import { homedir } from 'node:os';
|
|
23
23
|
import { join } from 'node:path';
|
|
24
|
+
import { cachedProbe, hasCommandCached, type SetupOption } from './runtime-probe.ts';
|
|
25
|
+
|
|
26
|
+
/** Maude-managed venv for the mlx-vlm runtime. Lives next to the identity cache
|
|
27
|
+
* (`$XDG_CACHE_HOME/maude` else `~/.maude`) — re-creatable with one command, so
|
|
28
|
+
* cache-tier is fine. The install stays USER-RUN in a terminal (DDR-183: the
|
|
29
|
+
* runtime is a manual step), but pointing the command at a dedicated venv makes
|
|
30
|
+
* it PEP 668-proof (a bare `pip install` is refused by Homebrew/system Pythons)
|
|
31
|
+
* and gives the probe a well-known place to look, so the app picks the install
|
|
32
|
+
* up automatically. */
|
|
33
|
+
export function mlxVenvDir(): string {
|
|
34
|
+
const xdg = process.env.XDG_CACHE_HOME;
|
|
35
|
+
const base = xdg && xdg.length > 0 ? join(xdg, 'maude') : join(homedir(), '.maude');
|
|
36
|
+
return join(base, 'mlx-venv');
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** POSIX single-quote a path for a command the USER will paste into a shell.
|
|
40
|
+
* Double quotes would let `$(…)`, backticks and `"` inside $XDG_CACHE_HOME /
|
|
41
|
+
* $HOME survive into a command run unread; single quotes disarm everything but
|
|
42
|
+
* `'` itself, which is closed-escaped-reopened. */
|
|
43
|
+
function shellQuote(s: string): string {
|
|
44
|
+
return `'${s.replaceAll("'", `'\\''`)}'`;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** The copy/paste one-liner the Settings card shows. Computed from the SAME path
|
|
48
|
+
* the probe checks, so the two can never drift. Returns null when the resolved
|
|
49
|
+
* path holds a newline or control char — no quoting makes a multi-line paste
|
|
50
|
+
* safe, so the card shows no command rather than a dangerous one. */
|
|
51
|
+
export function mlxInstallCommand(): string | null {
|
|
52
|
+
const dir = mlxVenvDir();
|
|
53
|
+
// biome-ignore lint/suspicious/noControlCharactersInRegex: refusing control chars is the point.
|
|
54
|
+
if (/[\x00-\x1f\x7f]/.test(dir)) return null;
|
|
55
|
+
return `python3 -m venv ${shellQuote(dir)} && ${shellQuote(join(dir, 'bin', 'pip'))} install -U mlx-vlm`;
|
|
56
|
+
}
|
|
24
57
|
|
|
25
58
|
export interface GemmaModelDescriptor {
|
|
26
59
|
/** Stable id used by the pref + route. */
|
|
@@ -103,11 +136,16 @@ export function listGemmaModels(): GemmaModelStatus[] {
|
|
|
103
136
|
return GEMMA_MODELS.map((m) => ({ ...m, downloaded: gemmaModelDownloaded(m) }));
|
|
104
137
|
}
|
|
105
138
|
|
|
106
|
-
/** Resolve a Python that can `import mlx_vlm
|
|
139
|
+
/** Resolve a Python that can `import mlx_vlm`, or null. Order: explicit
|
|
140
|
+
* $MAUDE_MLX_PYTHON → the Maude-managed venv → PATH pythons. */
|
|
107
141
|
export function resolveMlxPython(): string | null {
|
|
108
|
-
const
|
|
109
|
-
|
|
110
|
-
|
|
142
|
+
const venvPy = join(mlxVenvDir(), 'bin', 'python3');
|
|
143
|
+
const candidates = [
|
|
144
|
+
process.env.MAUDE_MLX_PYTHON,
|
|
145
|
+
existsSync(venvPy) ? venvPy : null,
|
|
146
|
+
'python3',
|
|
147
|
+
'python',
|
|
148
|
+
].filter(Boolean) as string[];
|
|
111
149
|
for (const py of candidates) {
|
|
112
150
|
const r = spawnSync(py, ['-c', 'import mlx_vlm'], { stdio: 'ignore' });
|
|
113
151
|
if (r.status === 0) return py;
|
|
@@ -122,22 +160,282 @@ export function resolveMlxPython(): string | null {
|
|
|
122
160
|
// Memoize with a short TTL: DoS-bounded to at most one probe per PROBE_TTL_MS
|
|
123
161
|
// regardless of request rate, while still picking up a mid-session `pip install` /
|
|
124
162
|
// PATH change within the TTL.
|
|
163
|
+
// The TTL cache + PATH probe live in runtime-probe.ts — shared with the whisper
|
|
164
|
+
// stack, which has the same "never print an impossible instruction" problem.
|
|
125
165
|
const PROBE_TTL_MS = 30_000;
|
|
126
|
-
const probeCache = new Map<string, { at: number; value: boolean }>();
|
|
127
|
-
|
|
128
|
-
function cachedProbe(key: string, compute: () => boolean): boolean {
|
|
129
|
-
const now = Date.now();
|
|
130
|
-
const hit = probeCache.get(key);
|
|
131
|
-
if (hit && now - hit.at < PROBE_TTL_MS) return hit.value;
|
|
132
|
-
const value = compute();
|
|
133
|
-
probeCache.set(key, { at: now, value });
|
|
134
|
-
return value;
|
|
135
|
-
}
|
|
136
166
|
|
|
137
167
|
export function mlxVlmAvailable(): boolean {
|
|
138
168
|
return cachedProbe('mlx', () => resolveMlxPython() !== null);
|
|
139
169
|
}
|
|
140
170
|
|
|
171
|
+
// ─── Ollama alternative runtime ──────────────────────────────────────────────
|
|
172
|
+
// The scout can also run through a local Ollama server (gemma3 vision) — a far
|
|
173
|
+
// simpler install story than mlx-vlm for most people: one app, `ollama pull`,
|
|
174
|
+
// no Python. mlx stays preferred when both are present (it's the benchmarked
|
|
175
|
+
// path, DDR-183); Ollama is the accessible alternative.
|
|
176
|
+
|
|
177
|
+
/** The model the install/pull commands recommend (vision-capable, ~3.3 GB). */
|
|
178
|
+
export const OLLAMA_RECOMMENDED_MODEL = 'gemma3:4b';
|
|
179
|
+
|
|
180
|
+
/** Loopback-only literal hosts (mirrors the DDR-185 curl-local posture). No DNS
|
|
181
|
+
* names besides `localhost` — a resolvable name could point anywhere. */
|
|
182
|
+
export function isLoopbackHostname(hostname: string): boolean {
|
|
183
|
+
const h = (hostname || '').toLowerCase().replace(/^\[|\]$/g, '');
|
|
184
|
+
return h === 'localhost' || h === '::1' || /^127(\.\d{1,3}){3}$/.test(h);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/** Local Ollama endpoint. Honors $OLLAMA_HOST (with or without scheme) but pins
|
|
188
|
+
* it to loopback — the scout is egress-free by design (DDR-183): a remote/LAN
|
|
189
|
+
* Ollama would silently upload the user's video frames, and on the server side
|
|
190
|
+
* an unvalidated host would turn the probe route into an SSRF emitter. Returns
|
|
191
|
+
* null when the value isn't loopback. */
|
|
192
|
+
export function ollamaHost(): string | null {
|
|
193
|
+
const raw = (process.env.OLLAMA_HOST || '').trim();
|
|
194
|
+
if (!raw) return 'http://127.0.0.1:11434';
|
|
195
|
+
const url = /^https?:\/\//.test(raw) ? raw.replace(/\/$/, '') : `http://${raw}`;
|
|
196
|
+
try {
|
|
197
|
+
return isLoopbackHostname(new URL(url).hostname) ? url : null;
|
|
198
|
+
} catch {
|
|
199
|
+
return null;
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
export interface OllamaStatus {
|
|
204
|
+
/** The server answered /api/tags. */
|
|
205
|
+
available: boolean;
|
|
206
|
+
/** A usable vision-capable gemma tag ($MAUDE_OLLAMA_MODEL wins), or null. */
|
|
207
|
+
model: string | null;
|
|
208
|
+
/** Every tag the local server has (drives the per-model "downloaded" state). */
|
|
209
|
+
tags: string[];
|
|
210
|
+
/** The `ollama` binary is on PATH. Installed-but-not-running is a DIFFERENT
|
|
211
|
+
* state from not-installed, and it needs the opposite advice ("start it",
|
|
212
|
+
* not "install it") — the card would otherwise tell you to reinstall
|
|
213
|
+
* software you already have. */
|
|
214
|
+
installed: boolean;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Curated allowlist of Ollama scout models, mirroring GEMMA_MODELS for the mlx
|
|
218
|
+
// runtime. The tag is the frozen pull target — NEVER user input, so the pull
|
|
219
|
+
// can't be steered at an arbitrary repo. Ids are namespaced so one route can
|
|
220
|
+
// serve both runtimes without ambiguity.
|
|
221
|
+
export const OLLAMA_MODELS = [
|
|
222
|
+
{
|
|
223
|
+
id: 'ollama:gemma3:4b',
|
|
224
|
+
tag: 'gemma3:4b',
|
|
225
|
+
label: 'Gemma 3 4B (Ollama)',
|
|
226
|
+
sizeMB: 3300,
|
|
227
|
+
note: 'The recommended Ollama scout. ~3.3 GB. Pulled and managed by Ollama.',
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
id: 'ollama:gemma3:12b',
|
|
231
|
+
tag: 'gemma3:12b',
|
|
232
|
+
label: 'Gemma 3 12B (Ollama)',
|
|
233
|
+
sizeMB: 8100,
|
|
234
|
+
note: 'Sharper beats, needs more RAM. ~8.1 GB.',
|
|
235
|
+
},
|
|
236
|
+
] as const;
|
|
237
|
+
|
|
238
|
+
export function getOllamaModel(id: unknown): (typeof OLLAMA_MODELS)[number] | null {
|
|
239
|
+
return OLLAMA_MODELS.find((m) => m.id === id) ?? null;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** The scout models offered for BOTH runtimes, each tagged with the runtime it
|
|
243
|
+
* belongs to and whether it's already on disk — so one Settings list can show
|
|
244
|
+
* "download" against whichever runtime the machine actually has. */
|
|
245
|
+
export function listScoutModels(status: OllamaStatus) {
|
|
246
|
+
const mlx = listGemmaModels().map((m) => ({ ...m, runtime: 'mlx' as const }));
|
|
247
|
+
const ollama = OLLAMA_MODELS.map((m) => ({
|
|
248
|
+
...m,
|
|
249
|
+
runtime: 'ollama' as const,
|
|
250
|
+
// An exact tag match; `gemma3:4b` and `gemma3:4b-it-q4_K_M` are different pulls.
|
|
251
|
+
downloaded: status.tags.includes(m.tag),
|
|
252
|
+
}));
|
|
253
|
+
// Ollama first when it's the runtime that can actually act on a click.
|
|
254
|
+
return status.available ? [...ollama, ...mlx] : [...mlx, ...ollama];
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Pull a model through the local Ollama server (`POST /api/pull`, NDJSON
|
|
259
|
+
* progress stream). Same egress discipline as everything else here: the host is
|
|
260
|
+
* loopback-pinned, redirects are refused, and the tag comes from the frozen
|
|
261
|
+
* allowlist — never from the request body.
|
|
262
|
+
*/
|
|
263
|
+
export async function pullOllamaModel(
|
|
264
|
+
id: string,
|
|
265
|
+
onProgress: (received: number, total: number) => void,
|
|
266
|
+
signal?: AbortSignal
|
|
267
|
+
): Promise<void> {
|
|
268
|
+
const m = getOllamaModel(id);
|
|
269
|
+
if (!m) throw new Error(`unknown ollama model: ${id}`);
|
|
270
|
+
const host = ollamaHost();
|
|
271
|
+
if (!host) throw new Error('OLLAMA_HOST is not a loopback address — refusing to pull');
|
|
272
|
+
|
|
273
|
+
const res = await fetch(`${host}/api/pull`, {
|
|
274
|
+
method: 'POST',
|
|
275
|
+
headers: { 'content-type': 'application/json' },
|
|
276
|
+
body: JSON.stringify({ model: m.tag, stream: true }),
|
|
277
|
+
redirect: 'manual',
|
|
278
|
+
signal,
|
|
279
|
+
});
|
|
280
|
+
if (!res.ok || !res.body) throw new Error(`ollama pull failed (HTTP ${res.status})`);
|
|
281
|
+
|
|
282
|
+
// NDJSON: {"status":"pulling …","total":N,"completed":M} … {"status":"success"}
|
|
283
|
+
const reader = res.body.getReader();
|
|
284
|
+
const decoder = new TextDecoder();
|
|
285
|
+
let buf = '';
|
|
286
|
+
let ok = false;
|
|
287
|
+
let lastErr = '';
|
|
288
|
+
while (true) {
|
|
289
|
+
const { done, value } = await reader.read();
|
|
290
|
+
if (done) break;
|
|
291
|
+
buf += decoder.decode(value, { stream: true });
|
|
292
|
+
const lines = buf.split('\n');
|
|
293
|
+
buf = lines.pop() ?? '';
|
|
294
|
+
for (const line of lines) {
|
|
295
|
+
if (!line.trim()) continue;
|
|
296
|
+
try {
|
|
297
|
+
const ev = JSON.parse(line) as {
|
|
298
|
+
status?: string;
|
|
299
|
+
total?: number;
|
|
300
|
+
completed?: number;
|
|
301
|
+
error?: string;
|
|
302
|
+
};
|
|
303
|
+
if (ev.error) lastErr = ev.error;
|
|
304
|
+
if (typeof ev.total === 'number' && ev.total > 0)
|
|
305
|
+
onProgress(Math.min(ev.total, ev.completed ?? 0), ev.total);
|
|
306
|
+
if (ev.status === 'success') ok = true;
|
|
307
|
+
} catch {
|
|
308
|
+
/* a partial/garbage line — the next read completes it */
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
if (!ok) throw new Error(lastErr || 'ollama pull did not report success');
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
const OLLAMA_DOWNLOAD_URL = 'https://ollama.com/download';
|
|
316
|
+
|
|
317
|
+
/**
|
|
318
|
+
* The install routes that actually work on THIS machine, best first.
|
|
319
|
+
*
|
|
320
|
+
* The point is to never show an impossible instruction: `brew install` is noise
|
|
321
|
+
* without Homebrew, and the venv one-liner is noise without python3. The
|
|
322
|
+
* universal fallback is Ollama's official install script, which handles macOS
|
|
323
|
+
* AND Linux (on macOS it fetches Ollama-darwin.zip into /Applications) and needs
|
|
324
|
+
* only curl — present on both by default, so it works where brew doesn't.
|
|
325
|
+
*/
|
|
326
|
+
export function ollamaSetupOptions(status: OllamaStatus): SetupOption[] {
|
|
327
|
+
const opts: SetupOption[] = [];
|
|
328
|
+
if (status.installed && !status.available) {
|
|
329
|
+
opts.push({
|
|
330
|
+
id: 'start',
|
|
331
|
+
kind: 'command',
|
|
332
|
+
label: 'Ollama is installed but not running — start it',
|
|
333
|
+
command: 'ollama serve',
|
|
334
|
+
note: 'Or just open the Ollama app; it runs in the menu bar.',
|
|
335
|
+
});
|
|
336
|
+
return opts;
|
|
337
|
+
}
|
|
338
|
+
if (status.available) {
|
|
339
|
+
opts.push({
|
|
340
|
+
id: 'pull',
|
|
341
|
+
kind: 'command',
|
|
342
|
+
label: 'Ollama is running — it just needs a vision model',
|
|
343
|
+
command: `ollama pull ${OLLAMA_RECOMMENDED_MODEL}`,
|
|
344
|
+
});
|
|
345
|
+
return opts;
|
|
346
|
+
}
|
|
347
|
+
const plat = process.platform;
|
|
348
|
+
if ((plat === 'darwin' || plat === 'linux') && hasCommandCached('curl'))
|
|
349
|
+
opts.push({
|
|
350
|
+
id: 'script',
|
|
351
|
+
kind: 'command',
|
|
352
|
+
label: 'Install Ollama (official script — no Homebrew needed)',
|
|
353
|
+
command: `curl -fsSL https://ollama.com/install.sh | sh && ollama pull ${OLLAMA_RECOMMENDED_MODEL}`,
|
|
354
|
+
note: 'Works on macOS and Linux; needs only curl.',
|
|
355
|
+
});
|
|
356
|
+
if (plat === 'darwin' && hasCommandCached('brew'))
|
|
357
|
+
opts.push({
|
|
358
|
+
id: 'brew',
|
|
359
|
+
kind: 'command',
|
|
360
|
+
label: 'Install with Homebrew',
|
|
361
|
+
command: `brew install ollama && brew services start ollama && ollama pull ${OLLAMA_RECOMMENDED_MODEL}`,
|
|
362
|
+
});
|
|
363
|
+
opts.push({
|
|
364
|
+
id: 'download',
|
|
365
|
+
kind: 'link',
|
|
366
|
+
label: 'Download the Ollama app',
|
|
367
|
+
url: OLLAMA_DOWNLOAD_URL,
|
|
368
|
+
note: `Then run \`ollama pull ${OLLAMA_RECOMMENDED_MODEL}\` in a terminal.`,
|
|
369
|
+
});
|
|
370
|
+
return opts;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
/** The mlx-vlm half, which only exists on Apple Silicon and only when a python3
|
|
374
|
+
* is around to build the venv with. Returns the reason when it can't run, so
|
|
375
|
+
* the card can say why instead of showing a command that would fail. */
|
|
376
|
+
export function mlxSetup(): { supported: boolean; reason?: string; command?: string } {
|
|
377
|
+
if (process.platform !== 'darwin')
|
|
378
|
+
return { supported: false, reason: 'mlx-vlm is Apple-Silicon only — use Ollama instead.' };
|
|
379
|
+
if (process.arch !== 'arm64')
|
|
380
|
+
return {
|
|
381
|
+
supported: false,
|
|
382
|
+
reason: 'mlx needs Apple Silicon (this Mac is Intel) — use Ollama instead.',
|
|
383
|
+
};
|
|
384
|
+
if (!hasCommandCached('python3'))
|
|
385
|
+
return {
|
|
386
|
+
supported: false,
|
|
387
|
+
reason: 'No python3 on PATH — install Python 3, or just use Ollama (no Python needed).',
|
|
388
|
+
};
|
|
389
|
+
const command = mlxInstallCommand();
|
|
390
|
+
if (!command)
|
|
391
|
+
return {
|
|
392
|
+
supported: false,
|
|
393
|
+
reason: 'Can’t build a safe install command for this machine’s cache path — use Ollama.',
|
|
394
|
+
};
|
|
395
|
+
return { supported: true, command };
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/** Pick a vision-capable gemma tag from an /api/tags listing. gemma3:1b is
|
|
399
|
+
* text-only and gemma3n is not multimodal in Ollama — exclude both. */
|
|
400
|
+
export function pickOllamaGemmaTag(tags: string[]): string | null {
|
|
401
|
+
const explicit = process.env.MAUDE_OLLAMA_MODEL;
|
|
402
|
+
if (explicit && explicit.length > 0) return explicit;
|
|
403
|
+
return tags.find((t) => /^gemma3:(?!1b)/.test(t) || t === 'gemma3') ?? null;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
const ollamaCache = { at: 0, value: null as OllamaStatus | null };
|
|
407
|
+
|
|
408
|
+
export async function ollamaStatus(): Promise<OllamaStatus> {
|
|
409
|
+
const now = Date.now();
|
|
410
|
+
if (ollamaCache.value && now - ollamaCache.at < PROBE_TTL_MS) return ollamaCache.value;
|
|
411
|
+
let value: OllamaStatus = {
|
|
412
|
+
available: false,
|
|
413
|
+
model: null,
|
|
414
|
+
tags: [],
|
|
415
|
+
installed: hasCommandCached('ollama'),
|
|
416
|
+
};
|
|
417
|
+
const host = ollamaHost(); // null = $OLLAMA_HOST steered off loopback — refused
|
|
418
|
+
if (host) {
|
|
419
|
+
try {
|
|
420
|
+
// redirect: 'manual' — never follow a redirect off the pinned loopback host.
|
|
421
|
+
const res = await fetch(`${host}/api/tags`, {
|
|
422
|
+
signal: AbortSignal.timeout(1500),
|
|
423
|
+
redirect: 'manual',
|
|
424
|
+
});
|
|
425
|
+
if (res.ok && Number(res.headers.get('content-length') || 0) <= 1024 * 1024) {
|
|
426
|
+
const body = (await res.json()) as { models?: Array<{ name?: string }> };
|
|
427
|
+
const tags = (body.models ?? []).map((m) => m.name).filter(Boolean) as string[];
|
|
428
|
+
value = { ...value, available: true, tags, model: pickOllamaGemmaTag(tags) };
|
|
429
|
+
}
|
|
430
|
+
} catch {
|
|
431
|
+
/* not running / not installed */
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
ollamaCache.at = now;
|
|
435
|
+
ollamaCache.value = value;
|
|
436
|
+
return value;
|
|
437
|
+
}
|
|
438
|
+
|
|
141
439
|
export function ffmpegAvailable(): boolean {
|
|
142
440
|
return cachedProbe('ffmpeg', () => {
|
|
143
441
|
const finder = process.platform === 'win32' ? 'where' : 'which';
|
|
@@ -12,8 +12,13 @@
|
|
|
12
12
|
import { existsSync, readFileSync } from 'node:fs';
|
|
13
13
|
import { join } from 'node:path';
|
|
14
14
|
|
|
15
|
-
/** The transcription engines the selector offers (mirrors the config schema enum).
|
|
16
|
-
|
|
15
|
+
/** The transcription engines the selector offers (mirrors the config schema enum).
|
|
16
|
+
* `auto` picks the best engine the machine is equipped for — a cloud engine
|
|
17
|
+
* when its key is set, else local whisper — and the UI states what it resolved
|
|
18
|
+
* to. It is a SELECTED mode, not a silent fallback: DDR-164's rule (Maude never
|
|
19
|
+
* switches to a paid/off-machine engine behind your back) survives because
|
|
20
|
+
* choosing `auto` is itself the explicit act, and the resolution is shown. */
|
|
21
|
+
export const TRANSCRIPTION_PROVIDERS = ['auto', 'whisper', 'elevenlabs', 'groq'] as const;
|
|
17
22
|
export type TranscriptionProvider = (typeof TRANSCRIPTION_PROVIDERS)[number];
|
|
18
23
|
|
|
19
24
|
export function isTranscriptionProvider(v: unknown): v is TranscriptionProvider {
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
// generation/runtime-probe.ts — shared "what can this machine actually do"
|
|
2
|
+
// probing for the optional local runtimes (the Gemma scout's mlx-vlm/Ollama, the
|
|
3
|
+
// subtitle stack's whisper.cpp).
|
|
4
|
+
//
|
|
5
|
+
// Two jobs, both learned the hard way:
|
|
6
|
+
//
|
|
7
|
+
// 1. Probes SPAWN. They sit behind un-authenticated (same-origin + loopback)
|
|
8
|
+
// GETs, so an unthrottled probe is a way to stall the single-threaded Bun
|
|
9
|
+
// event loop. Everything here is TTL-cached: at most one spawn per key per
|
|
10
|
+
// PROBE_TTL_MS regardless of request rate, while still noticing a
|
|
11
|
+
// mid-session install within the TTL.
|
|
12
|
+
//
|
|
13
|
+
// 2. A card must never print an instruction this machine can't follow.
|
|
14
|
+
// `brew install …` is noise without Homebrew; a venv one-liner is noise
|
|
15
|
+
// without python3. Callers build their setup routes from these probes and
|
|
16
|
+
// return only the viable ones, best first.
|
|
17
|
+
|
|
18
|
+
import { spawnSync } from 'node:child_process';
|
|
19
|
+
|
|
20
|
+
const PROBE_TTL_MS = 30_000;
|
|
21
|
+
const probeCache = new Map<string, { at: number; value: boolean }>();
|
|
22
|
+
|
|
23
|
+
export function cachedProbe(key: string, compute: () => boolean): boolean {
|
|
24
|
+
const now = Date.now();
|
|
25
|
+
const hit = probeCache.get(key);
|
|
26
|
+
if (hit && now - hit.at < PROBE_TTL_MS) return hit.value;
|
|
27
|
+
const value = compute();
|
|
28
|
+
probeCache.set(key, { at: now, value });
|
|
29
|
+
return value;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** Is a command on PATH? */
|
|
33
|
+
export function hasCommandCached(cmd: string): boolean {
|
|
34
|
+
return cachedProbe(`has:${cmd}`, () => {
|
|
35
|
+
const finder = process.platform === 'win32' ? 'where' : 'which';
|
|
36
|
+
return spawnSync(finder, [cmd], { stdio: 'ignore' }).status === 0;
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** One way to get a runtime onto THIS machine. `command` is copy/paste; `link`
|
|
41
|
+
* is a URL — the desktop shell has no general URL opener by design (DDR-054),
|
|
42
|
+
* so the client renders a copyable link rather than a button that can't work. */
|
|
43
|
+
export interface SetupOption {
|
|
44
|
+
id: string;
|
|
45
|
+
kind: 'command' | 'link';
|
|
46
|
+
label: string;
|
|
47
|
+
command?: string;
|
|
48
|
+
url?: string;
|
|
49
|
+
note?: string;
|
|
50
|
+
}
|
|
@@ -13,10 +13,12 @@
|
|
|
13
13
|
// The actual download (streamed, SSRF-hardened, progress-tracked) lives in the
|
|
14
14
|
// http route so the egress discipline sits next to the other provider egress.
|
|
15
15
|
|
|
16
|
+
import { spawnSync } from 'node:child_process';
|
|
16
17
|
import { existsSync, readdirSync, statSync } from 'node:fs';
|
|
17
18
|
import { mkdir, rename, rm } from 'node:fs/promises';
|
|
18
19
|
import { homedir } from 'node:os';
|
|
19
20
|
import { join } from 'node:path';
|
|
21
|
+
import { cachedProbe, hasCommandCached, type SetupOption } from './runtime-probe.ts';
|
|
20
22
|
|
|
21
23
|
export interface WhisperModelDescriptor {
|
|
22
24
|
/** Stable id used by the config + route (`base`, `base.en`, …). */
|
|
@@ -270,3 +272,125 @@ export async function removeWhisperModel(id: string): Promise<boolean> {
|
|
|
270
272
|
await rm(p, { force: true });
|
|
271
273
|
return true;
|
|
272
274
|
}
|
|
275
|
+
|
|
276
|
+
// ─── Runtime detection + setup routes ────────────────────────────────────────
|
|
277
|
+
// The card used to print `brew install whisper-cpp` unconditionally and never
|
|
278
|
+
// checked whether the binary was actually there — so a machine without Homebrew
|
|
279
|
+
// got an instruction it couldn't follow, and nobody learned the engine was
|
|
280
|
+
// missing until a transcription failed.
|
|
281
|
+
//
|
|
282
|
+
// Unlike Ollama there is NO universal one-liner here: whisper.cpp's releases
|
|
283
|
+
// ship prebuilt binaries for Ubuntu and Windows but NOT macOS (the xcframework
|
|
284
|
+
// is an Xcode library, not a CLI), so on a Mac without brew the honest routes
|
|
285
|
+
// are a cloud engine (already supported, needs only a key) or a source build.
|
|
286
|
+
|
|
287
|
+
/** Resolve the whisper.cpp CLI, or null.
|
|
288
|
+
*
|
|
289
|
+
* MUST stay byte-identical to `bin/_transcribe.mjs` `resolveWhisper()` — the
|
|
290
|
+
* card would otherwise claim an engine the transcriber can't find. That
|
|
291
|
+
* includes the security rule it documents: the bare name `main` is
|
|
292
|
+
* DELIBERATELY not probed (it's a common executable name and the probe
|
|
293
|
+
* EXECUTES each candidate, so `.` on $PATH inside an untrusted repo could
|
|
294
|
+
* auto-run a seeded `main`). */
|
|
295
|
+
export function resolveWhisperCli(): string | null {
|
|
296
|
+
const candidates = [process.env.MAUDE_WHISPER_CLI, 'whisper-cli', 'whisper'].filter(
|
|
297
|
+
Boolean
|
|
298
|
+
) as string[];
|
|
299
|
+
for (const c of candidates) {
|
|
300
|
+
const probe = spawnSync(c, ['--help'], { stdio: 'ignore' });
|
|
301
|
+
if (!probe.error) return c;
|
|
302
|
+
}
|
|
303
|
+
return null;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
export function whisperCliAvailable(): boolean {
|
|
307
|
+
return cachedProbe('whisper-cli', () => resolveWhisperCli() !== null);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
export interface WhisperSetup {
|
|
311
|
+
/** The binary resolves — local transcription can actually run. */
|
|
312
|
+
installed: boolean;
|
|
313
|
+
/** Routes to get it, best first. Empty when it's already installed. */
|
|
314
|
+
options: SetupOption[];
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
export function whisperSetup(): WhisperSetup {
|
|
318
|
+
if (whisperCliAvailable()) return { installed: true, options: [] };
|
|
319
|
+
const options: SetupOption[] = [];
|
|
320
|
+
if (hasCommandCached('brew'))
|
|
321
|
+
options.push({
|
|
322
|
+
id: 'brew',
|
|
323
|
+
kind: 'command',
|
|
324
|
+
label: 'Install whisper.cpp with Homebrew',
|
|
325
|
+
command: 'brew install whisper-cpp',
|
|
326
|
+
});
|
|
327
|
+
else
|
|
328
|
+
options.push({
|
|
329
|
+
id: 'cloud',
|
|
330
|
+
kind: 'link',
|
|
331
|
+
label: 'No Homebrew — a cloud engine needs no install at all',
|
|
332
|
+
url: 'https://elevenlabs.io/app/settings/api-keys',
|
|
333
|
+
note: 'Pick ElevenLabs Scribe or Groq above and paste a key. Audio is uploaded to that provider and billed to your account.',
|
|
334
|
+
});
|
|
335
|
+
options.push({
|
|
336
|
+
id: 'source',
|
|
337
|
+
kind: 'link',
|
|
338
|
+
label: 'Build from source',
|
|
339
|
+
url: 'https://github.com/ggml-org/whisper.cpp',
|
|
340
|
+
note: 'whisper.cpp ships prebuilt binaries for Linux and Windows, but not macOS — a Mac build needs cmake + Xcode command-line tools.',
|
|
341
|
+
});
|
|
342
|
+
return { installed: false, options };
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// ─── Automatic engine choice ─────────────────────────────────────────────────
|
|
346
|
+
// Task 2.6 / DDR-164 made the engine an EXPLICIT choice because a silent
|
|
347
|
+
// fallback to a cloud engine has two consequences the user never asked for:
|
|
348
|
+
// their audio leaves the machine, and their account is billed. That rule is
|
|
349
|
+
// kept — what changes is that "auto" becomes a choice the user can MAKE, and
|
|
350
|
+
// one that always SAYS what it currently resolves to (in this card and in the
|
|
351
|
+
// transcriber's own output). Nothing switches behind your back; `auto` is a
|
|
352
|
+
// selected mode, not a hidden default override.
|
|
353
|
+
//
|
|
354
|
+
// The bite worth knowing: one ElevenLabs key covers audio generation AND
|
|
355
|
+
// Scribe, so a key added for music/TTS is enough to make `auto` route
|
|
356
|
+
// transcription to the cloud. That's why the card states the resolution
|
|
357
|
+
// out loud and pinning `whisper` stays one click away.
|
|
358
|
+
|
|
359
|
+
export type TranscriptionEngine = 'auto' | 'whisper' | 'elevenlabs' | 'groq';
|
|
360
|
+
|
|
361
|
+
export interface AutoEngineResolution {
|
|
362
|
+
/** What `auto` picks right now. */
|
|
363
|
+
engine: Exclude<TranscriptionEngine, 'auto'>;
|
|
364
|
+
/** Why — rendered verbatim so the choice is never a mystery. */
|
|
365
|
+
reason: string;
|
|
366
|
+
/** True when the resolved engine uploads audio and bills a provider. */
|
|
367
|
+
cloud: boolean;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/**
|
|
371
|
+
* Resolve `auto`: prefer the cloud engine whose key is present (ElevenLabs
|
|
372
|
+
* Scribe first — better accuracy than a local base model), else local
|
|
373
|
+
* whisper.cpp. `configured` is the set of providers holding a key, passed in so
|
|
374
|
+
* this stays a pure function (the keychain read is the caller's).
|
|
375
|
+
*/
|
|
376
|
+
export function resolveAutoEngine(
|
|
377
|
+
configured: Iterable<string>,
|
|
378
|
+
whisperInstalled: boolean
|
|
379
|
+
): AutoEngineResolution {
|
|
380
|
+
const keys = new Set(configured);
|
|
381
|
+
if (keys.has('elevenlabs'))
|
|
382
|
+
return {
|
|
383
|
+
engine: 'elevenlabs',
|
|
384
|
+
reason: 'ElevenLabs key is set — Scribe is more accurate than a local base model.',
|
|
385
|
+
cloud: true,
|
|
386
|
+
};
|
|
387
|
+
if (keys.has('groq'))
|
|
388
|
+
return { engine: 'groq', reason: 'Groq key is set — fast cloud transcription.', cloud: true };
|
|
389
|
+
return {
|
|
390
|
+
engine: 'whisper',
|
|
391
|
+
reason: whisperInstalled
|
|
392
|
+
? 'No cloud key set — using local whisper.cpp (free, offline, nothing leaves this machine).'
|
|
393
|
+
: 'No cloud key set — will use local whisper.cpp once its binary is installed.',
|
|
394
|
+
cloud: false,
|
|
395
|
+
};
|
|
396
|
+
}
|
|
@@ -171,3 +171,70 @@ export function classifyChange(
|
|
|
171
171
|
}
|
|
172
172
|
return null;
|
|
173
173
|
}
|
|
174
|
+
|
|
175
|
+
// ---------------------------------------------------------------------------
|
|
176
|
+
// Container write bridge — inspector-edits-live-render RCA.
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Delay before synthesising the `fs:any` a container's `fs.watch` failed to
|
|
180
|
+
* emit. The write is `await`ed inside the op before it returns, so a few hundred
|
|
181
|
+
* ms is ample; kept well under activity's 2.5 s `SUPPRESS_TTL_MS` so the
|
|
182
|
+
* user-write rim stays muted when the synthetic event lands.
|
|
183
|
+
*/
|
|
184
|
+
export const SYNTHETIC_FS_DELAY_MS = 250;
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* In a CELL the recursive `fs.watch` (Linux inotify) does NOT fire for the
|
|
188
|
+
* atomic tmp+rename writes `canvas-edit.ts` makes into `designRoot` subdirs —
|
|
189
|
+
* verified live: after a `200` `edit-css` on a cell, a connected `canvas-hmr`
|
|
190
|
+
* socket received nothing. So `fs:any` never fires, the HMR broadcaster never
|
|
191
|
+
* runs, and a PEER's canvas iframe sits on the pre-edit module until a manual
|
|
192
|
+
* reload. (Annotations are spared: they cross via the collab room, not `fs:any`.)
|
|
193
|
+
*
|
|
194
|
+
* Every write path arms `activity:suppress(rel)` immediately before writing — a
|
|
195
|
+
* reliable, complete signal that a write to `rel` is imminent. Treat it as the
|
|
196
|
+
* `fs:any` the watcher owes us and synthesise one once the write has settled. A
|
|
197
|
+
* no-op or failed edit disarms via `activity:unsuppress`, which cancels the
|
|
198
|
+
* pending emit, so an equal-length or rejected edit never reloads peers.
|
|
199
|
+
*
|
|
200
|
+
* WORKSPACE-MODE ONLY (the caller gates it): locally `fs.watch` fires, and a
|
|
201
|
+
* second source would double-reload every canvas on every edit — the HMR
|
|
202
|
+
* broadcaster's 50 ms per-path debounce cannot coalesce two events ~250 ms apart.
|
|
203
|
+
*/
|
|
204
|
+
export function createContainerWriteBridge(ctx: Context): { stop(): void } {
|
|
205
|
+
const pending = new Map<string, ReturnType<typeof setTimeout>>();
|
|
206
|
+
const norm = (rel: string) => rel.replace(/\\/g, '/');
|
|
207
|
+
|
|
208
|
+
const offSuppress = ctx.bus.on('activity:suppress', (rel: string) => {
|
|
209
|
+
if (typeof rel !== 'string' || !rel) return;
|
|
210
|
+
const key = norm(rel);
|
|
211
|
+
const prev = pending.get(key);
|
|
212
|
+
if (prev) clearTimeout(prev);
|
|
213
|
+
pending.set(
|
|
214
|
+
key,
|
|
215
|
+
setTimeout(() => {
|
|
216
|
+
pending.delete(key);
|
|
217
|
+
ctx.bus.emit('fs:any', key);
|
|
218
|
+
}, SYNTHETIC_FS_DELAY_MS)
|
|
219
|
+
);
|
|
220
|
+
});
|
|
221
|
+
|
|
222
|
+
const cancel = (rel: string) => {
|
|
223
|
+
if (typeof rel !== 'string' || !rel) return;
|
|
224
|
+
const t = pending.get(norm(rel));
|
|
225
|
+
if (t) {
|
|
226
|
+
clearTimeout(t);
|
|
227
|
+
pending.delete(norm(rel));
|
|
228
|
+
}
|
|
229
|
+
};
|
|
230
|
+
const offUnsuppress = ctx.bus.on('activity:unsuppress', cancel);
|
|
231
|
+
|
|
232
|
+
return {
|
|
233
|
+
stop() {
|
|
234
|
+
offSuppress();
|
|
235
|
+
offUnsuppress();
|
|
236
|
+
for (const t of pending.values()) clearTimeout(t);
|
|
237
|
+
pending.clear();
|
|
238
|
+
},
|
|
239
|
+
};
|
|
240
|
+
}
|