@1agh/maude 0.54.0 → 0.56.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/apps/studio/annotations-layer.tsx +11 -1
- package/apps/studio/bin/_smart-frames.mjs +187 -20
- package/apps/studio/bin/_smart-frames.test.mjs +59 -4
- package/apps/studio/bin/_transcribe.mjs +40 -3
- package/apps/studio/bin/smoke.sh +7 -1
- package/apps/studio/canvas-build-sandbox.ts +101 -9
- package/apps/studio/canvas-build-worker.ts +7 -1
- package/apps/studio/canvas-build.ts +9 -1
- package/apps/studio/client/app.jsx +49 -3
- package/apps/studio/client/github.js +7 -0
- package/apps/studio/client/panels/CloudBar.jsx +505 -45
- package/apps/studio/client/panels/GitPanel.jsx +38 -25
- package/apps/studio/client/panels/SettingsPanel.jsx +239 -27
- package/apps/studio/client/styles/3-shell-maude.css +30 -0
- package/apps/studio/client/styles/4-components.css +72 -0
- package/apps/studio/cloud/endpoints.ts +104 -9
- package/apps/studio/collab/persistence.ts +29 -2
- package/apps/studio/config.schema.json +3 -3
- package/apps/studio/context.ts +41 -0
- package/apps/studio/dist/client.bundle.js +1545 -1545
- package/apps/studio/dist/comment-mount.js +2 -2
- package/apps/studio/dist/styles.css +1 -1
- package/apps/studio/generation/gemma-models.ts +312 -14
- package/apps/studio/generation/prefs.ts +7 -2
- package/apps/studio/generation/runtime-probe.ts +50 -0
- package/apps/studio/generation/whisper-models.ts +124 -0
- package/apps/studio/hmr-broadcast.ts +67 -0
- package/apps/studio/http.ts +210 -110
- package/apps/studio/input-router.tsx +55 -2
- package/apps/studio/server.ts +11 -9
- package/apps/studio/sync/autocommit.ts +61 -2
- package/apps/studio/sync/cell-pairing.ts +174 -0
- package/apps/studio/sync/codec.ts +11 -5
- package/apps/studio/sync/index.ts +239 -26
- package/apps/studio/sync/limits.ts +49 -0
- package/apps/studio/sync/loopback.ts +21 -0
- package/apps/studio/sync/projection.ts +47 -12
- package/apps/studio/sync/supervisor.ts +178 -0
- package/apps/studio/test/cloud-endpoints.test.ts +326 -4
- package/apps/studio/test/csrf-write-guard.test.ts +19 -2
- package/apps/studio/test/gemma-models.test.ts +245 -0
- package/apps/studio/test/hmr-broadcast.test.ts +57 -1
- package/apps/studio/test/input-router.test.ts +95 -0
- package/apps/studio/test/shared-doc-cell-pairing.test.ts +639 -0
- package/apps/studio/test/sync-autocommit.test.ts +47 -0
- package/apps/studio/test/sync-supervisor.test.ts +212 -0
- package/apps/studio/test/trusted-request-host.test.ts +66 -0
- package/apps/studio/test/whisper-setup.test.ts +97 -0
- package/apps/studio/whats-new.json +45 -0
- package/apps/studio/ws.ts +9 -1
- package/cli/commands/kg.mjs +9 -2
- package/package.json +8 -8
- package/plugins/design/dependencies.json +21 -3
|
@@ -1252,7 +1252,17 @@ export function AnnotationsLayer() {
|
|
|
1252
1252
|
headers: { 'Content-Type': 'application/json' },
|
|
1253
1253
|
body: JSON.stringify({ file, svg }),
|
|
1254
1254
|
})
|
|
1255
|
-
.then(() =>
|
|
1255
|
+
.then((r) => {
|
|
1256
|
+
// A refused save (403 read-only, 405 at a proxy door) previously
|
|
1257
|
+
// dissolved here without a trace — the user kept drawing on state
|
|
1258
|
+
// that never reached disk, a peer, or a reload (the cloud
|
|
1259
|
+
// canvas-writes RCA). Optimistic local state is still the right
|
|
1260
|
+
// UX; a persistence failure being INVISIBLE is not.
|
|
1261
|
+
if (!r.ok) {
|
|
1262
|
+
console.warn(`[annotations] save refused (${r.status}) — strokes are local-only`);
|
|
1263
|
+
}
|
|
1264
|
+
return undefined;
|
|
1265
|
+
})
|
|
1256
1266
|
.catch(() => {
|
|
1257
1267
|
/* swallow — user sees uncommitted state until the next stroke */
|
|
1258
1268
|
});
|
|
@@ -22,7 +22,8 @@
|
|
|
22
22
|
// are unit-tested in _smart-frames.test.mjs.
|
|
23
23
|
|
|
24
24
|
import { spawnSync } from 'node:child_process';
|
|
25
|
-
import { existsSync, mkdirSync,
|
|
25
|
+
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync } from 'node:fs';
|
|
26
|
+
import { homedir, tmpdir } from 'node:os';
|
|
26
27
|
import { dirname, join } from 'node:path';
|
|
27
28
|
import { fileURLToPath } from 'node:url';
|
|
28
29
|
|
|
@@ -37,9 +38,24 @@ export function hasCommand(cmd) {
|
|
|
37
38
|
return r.status === 0;
|
|
38
39
|
}
|
|
39
40
|
|
|
40
|
-
/**
|
|
41
|
+
/** The Maude-managed mlx-vlm venv python (mirrors generation/gemma-models.ts —
|
|
42
|
+
* the Settings install command creates this venv, so the CLI must look there). */
|
|
43
|
+
export function mlxVenvPython(env = process.env) {
|
|
44
|
+
const xdg = env.XDG_CACHE_HOME;
|
|
45
|
+
const base = xdg && xdg.length > 0 ? join(xdg, 'maude') : join(homedir(), '.maude');
|
|
46
|
+
return join(base, 'mlx-venv', 'bin', 'python3');
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Resolve a python that can `import mlx_vlm`, or null. Order: $MAUDE_MLX_PYTHON
|
|
50
|
+
* → the Maude-managed venv → PATH pythons. */
|
|
41
51
|
export function resolveMlxPython(env = process.env) {
|
|
42
|
-
const
|
|
52
|
+
const venvPy = mlxVenvPython(env);
|
|
53
|
+
const candidates = [
|
|
54
|
+
env.MAUDE_MLX_PYTHON,
|
|
55
|
+
existsSync(venvPy) ? venvPy : null,
|
|
56
|
+
'python3',
|
|
57
|
+
'python',
|
|
58
|
+
].filter(Boolean);
|
|
43
59
|
for (const py of candidates) {
|
|
44
60
|
const r = spawnSync(py, ['-c', 'import mlx_vlm'], { stdio: 'ignore' });
|
|
45
61
|
if (r.status === 0) return py;
|
|
@@ -47,22 +63,88 @@ export function resolveMlxPython(env = process.env) {
|
|
|
47
63
|
return null;
|
|
48
64
|
}
|
|
49
65
|
|
|
50
|
-
|
|
66
|
+
/** Loopback-only literal hosts (mirrors the DDR-185 curl-local posture). No DNS
|
|
67
|
+
* names besides `localhost` — a resolvable name could point anywhere (and
|
|
68
|
+
* rebind between probe and use). */
|
|
69
|
+
export function isLoopbackHostname(hostname) {
|
|
70
|
+
const h = (hostname || '').toLowerCase().replace(/^\[|\]$/g, '');
|
|
71
|
+
return h === 'localhost' || h === '::1' || /^127(\.\d{1,3}){3}$/.test(h);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Local Ollama endpoint. Honors $OLLAMA_HOST (with or without scheme) but pins
|
|
75
|
+
* it to loopback: the footage pipeline is egress-free by design (DDR-183) —
|
|
76
|
+
* frames must never leave this machine, so a remote/LAN Ollama is refused
|
|
77
|
+
* rather than silently uploaded to. Returns null when the value isn't loopback. */
|
|
78
|
+
export function ollamaHost(env = process.env) {
|
|
79
|
+
const raw = (env.OLLAMA_HOST || '').trim();
|
|
80
|
+
if (!raw) return 'http://127.0.0.1:11434';
|
|
81
|
+
const url = /^https?:\/\//.test(raw) ? raw.replace(/\/$/, '') : `http://${raw}`;
|
|
82
|
+
try {
|
|
83
|
+
return isLoopbackHostname(new URL(url).hostname) ? url : null;
|
|
84
|
+
} catch {
|
|
85
|
+
return null;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Pick a vision-capable gemma tag from an /api/tags listing. gemma3:1b is
|
|
90
|
+
* text-only and gemma3n is not multimodal in Ollama — exclude both.
|
|
91
|
+
* $MAUDE_OLLAMA_MODEL wins verbatim. */
|
|
92
|
+
export function pickOllamaGemmaTag(tags, env = process.env) {
|
|
93
|
+
const explicit = env.MAUDE_OLLAMA_MODEL;
|
|
94
|
+
if (explicit && explicit.length > 0) return explicit;
|
|
95
|
+
// Keep this predicate byte-identical to generation/gemma-models.ts's copy.
|
|
96
|
+
return tags.find((t) => /^gemma3:(?!1b)/.test(t) || t === 'gemma3') ?? null;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Probe a running Ollama server for a usable gemma vision model.
|
|
100
|
+
* Returns { host, model } or null. */
|
|
101
|
+
export async function detectOllama(env = process.env) {
|
|
102
|
+
const host = ollamaHost(env);
|
|
103
|
+
if (!host) return null; // $OLLAMA_HOST steered off loopback — refused
|
|
104
|
+
try {
|
|
105
|
+
// redirect: 'manual' — a redirecting "Ollama" is not Ollama; following one
|
|
106
|
+
// could carry the request off loopback.
|
|
107
|
+
const res = await fetch(`${host}/api/tags`, {
|
|
108
|
+
signal: AbortSignal.timeout(1500),
|
|
109
|
+
redirect: 'manual',
|
|
110
|
+
});
|
|
111
|
+
if (!res.ok) return null;
|
|
112
|
+
if (Number(res.headers.get('content-length') || 0) > 1024 * 1024) return null;
|
|
113
|
+
const body = await res.json();
|
|
114
|
+
const tags = (body.models || []).map((m) => m?.name).filter(Boolean);
|
|
115
|
+
const model = pickOllamaGemmaTag(tags, env);
|
|
116
|
+
return model ? { host, model } : null;
|
|
117
|
+
} catch {
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Probe what's installed. `engine` lets us SKIP the Ollama network probe when it
|
|
123
|
+
* can't change the outcome — an explicit `blind`/`ffmpeg` run, no ffmpeg at all,
|
|
124
|
+
* or mlx-vlm already resolved (mlx wins anyway). A `--engine blind` run should
|
|
125
|
+
* touch no sockets. */
|
|
126
|
+
export async function detectAvailability(env = process.env, engine = 'auto') {
|
|
127
|
+
const ffmpeg = hasCommand('ffmpeg') && hasCommand('ffprobe');
|
|
128
|
+
const mlxPython = resolveMlxPython(env);
|
|
129
|
+
const ollamaCouldMatter =
|
|
130
|
+
ffmpeg && !mlxPython && (engine === 'auto' || engine === 'gemma' || !engine);
|
|
51
131
|
return {
|
|
52
|
-
ffmpeg
|
|
53
|
-
mlxPython
|
|
132
|
+
ffmpeg,
|
|
133
|
+
mlxPython,
|
|
134
|
+
ollama: ollamaCouldMatter ? await detectOllama(env) : null,
|
|
54
135
|
};
|
|
55
136
|
}
|
|
56
137
|
|
|
57
138
|
/** Pick the tier to actually run. Explicit engine wins (and errors if its deps are
|
|
58
139
|
* missing — no silent downgrade when the user asked for one); `auto` degrades
|
|
59
|
-
* gemma → ffmpeg → blind based on what's installed.
|
|
140
|
+
* gemma → ffmpeg → blind based on what's installed. The gemma tier runs on
|
|
141
|
+
* either scout runtime: mlx-vlm (preferred — the benchmarked path) or Ollama. */
|
|
60
142
|
export function selectTier(engine, avail) {
|
|
61
|
-
const gemmaOk = Boolean(avail.ffmpeg && avail.mlxPython);
|
|
143
|
+
const gemmaOk = Boolean(avail.ffmpeg && (avail.mlxPython || avail.ollama));
|
|
62
144
|
if (engine && engine !== 'auto') {
|
|
63
145
|
if (engine === 'gemma' && !gemmaOk)
|
|
64
146
|
throw new Error(
|
|
65
|
-
'engine "gemma" needs ffmpeg + mlx-vlm (Apple Silicon). Install
|
|
147
|
+
'engine "gemma" needs ffmpeg + a scout runtime — mlx-vlm (Apple Silicon) or Ollama with a gemma3 vision model. Install one or use --engine ffmpeg|blind|auto.'
|
|
66
148
|
);
|
|
67
149
|
if (engine === 'ffmpeg' && !avail.ffmpeg)
|
|
68
150
|
throw new Error(
|
|
@@ -87,6 +169,16 @@ export function parseSceneCuts(stderr) {
|
|
|
87
169
|
return out.filter((t) => Number.isFinite(t));
|
|
88
170
|
}
|
|
89
171
|
|
|
172
|
+
/** Model-authored free text that lands in the manifest (and later in agent
|
|
173
|
+
* prompts): strip control chars, cap the length. Injection-hardening — the
|
|
174
|
+
* scout text is untrusted model output (DDR-183). */
|
|
175
|
+
// biome-ignore lint/suspicious/noControlCharactersInRegex: matching control chars is the point.
|
|
176
|
+
const CONTROL_CHARS = /[\x00-\x1f\x7f]/g;
|
|
177
|
+
|
|
178
|
+
function sanitizeWhat(s) {
|
|
179
|
+
return (s || '').replace(CONTROL_CHARS, ' ').trim().slice(0, 120);
|
|
180
|
+
}
|
|
181
|
+
|
|
90
182
|
/** Parse a Gemma scout's text → beat timestamps. Accepts both `TIME=<sec>` and the
|
|
91
183
|
* `M:SS | what` shape the small model tends to emit. */
|
|
92
184
|
export function parseBeats(text, durationSec) {
|
|
@@ -98,7 +190,7 @@ export function parseBeats(text, durationSec) {
|
|
|
98
190
|
const reMS = /(?:^|\n)\s*(\d+):(\d{2}(?:\.\d+)?)\s*\|\s*([^\n]*)/g;
|
|
99
191
|
// biome-ignore lint/suspicious/noAssignInExpressions: standard regex-exec loop.
|
|
100
192
|
while ((m = reMS.exec(text)))
|
|
101
|
-
beats.push({ t: Number(m[1]) * 60 + Number(m[2]), what: (m[3]
|
|
193
|
+
beats.push({ t: Number(m[1]) * 60 + Number(m[2]), what: sanitizeWhat(m[3]) });
|
|
102
194
|
return beats.filter((b) => Number.isFinite(b.t) && b.t >= 0 && b.t <= durationSec);
|
|
103
195
|
}
|
|
104
196
|
|
|
@@ -194,6 +286,65 @@ function extractFrame(clip, t, outPath) {
|
|
|
194
286
|
const SCOUT_PROMPT = (dur) =>
|
|
195
287
|
`You are a shot-list scout watching a ${dur}-second video clip. List the KEY moments an editor must see: the start of each distinct shot AND any peak action beat (a snap, a catch, a big movement, a reveal) — even inside a continuous shot. Do NOT space them evenly; pick only moments where something meaningful happens or changes. Output ONE line per moment, formatted EXACTLY as: TIME=<seconds> | <a few words>. Seconds must be real numbers between 0 and ${dur}.`;
|
|
196
288
|
|
|
289
|
+
/** Ollama scout — same job as the mlx scout, different transport. Ollama has no
|
|
290
|
+
* --video input, so we sample ≤16 evenly-spaced frames ourselves (ffmpeg is
|
|
291
|
+
* guaranteed present in this tier), tell the model each frame's timestamp, and
|
|
292
|
+
* send them as images to /api/chat. Coarser than mlx's native video path, but
|
|
293
|
+
* scout beats only ADD candidate frames — ffmpeg cuts stay the precise backbone
|
|
294
|
+
* (DDR-183), so coarse is acceptable. */
|
|
295
|
+
export function ollamaScoutPrompt(durationSec, times) {
|
|
296
|
+
const map = times.map((t, i) => `frame ${i + 1} = t=${t.toFixed(2)}s`).join(', ');
|
|
297
|
+
return (
|
|
298
|
+
`${SCOUT_PROMPT(durationSec)}\n` +
|
|
299
|
+
`You are given ${times.length} frames sampled from the clip: ${map}. ` +
|
|
300
|
+
`Use these timestamps (or values between adjacent ones) as your TIME= values.`
|
|
301
|
+
);
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
async function ollamaScout(clip, durationSec, { host, model, fps }) {
|
|
305
|
+
const n = Math.max(4, Math.min(16, Math.round(durationSec * fps)));
|
|
306
|
+
const times = [];
|
|
307
|
+
for (let k = 0; k < n; k++)
|
|
308
|
+
times.push(
|
|
309
|
+
+Math.min(Math.max(0.05, (k * durationSec) / Math.max(1, n - 1)), durationSec - 0.05).toFixed(
|
|
310
|
+
3
|
|
311
|
+
)
|
|
312
|
+
);
|
|
313
|
+
const dir = mkdtempSync(join(tmpdir(), 'smart-frames-scout-'));
|
|
314
|
+
try {
|
|
315
|
+
const images = [];
|
|
316
|
+
for (const [i, t] of times.entries()) {
|
|
317
|
+
const png = join(dir, `s_${String(i + 1).padStart(2, '0')}.png`);
|
|
318
|
+
if (extractFrame(clip, t, png)) images.push(readFileSync(png).toString('base64'));
|
|
319
|
+
}
|
|
320
|
+
if (!images.length) return { beats: [], ok: false, raw: 'no scout frames extracted' };
|
|
321
|
+
const res = await fetch(`${host}/api/chat`, {
|
|
322
|
+
method: 'POST',
|
|
323
|
+
headers: { 'content-type': 'application/json' },
|
|
324
|
+
body: JSON.stringify({
|
|
325
|
+
model,
|
|
326
|
+
stream: false,
|
|
327
|
+
messages: [{ role: 'user', content: ollamaScoutPrompt(durationSec, times), images }],
|
|
328
|
+
options: { num_predict: 300 },
|
|
329
|
+
}),
|
|
330
|
+
// big vision models on CPU are slow — generous, but bounded
|
|
331
|
+
signal: AbortSignal.timeout(300_000),
|
|
332
|
+
redirect: 'manual', // never follow a redirect off the pinned loopback host
|
|
333
|
+
});
|
|
334
|
+
if (!res.ok) return { beats: [], ok: false, raw: `ollama HTTP ${res.status}` };
|
|
335
|
+
if (Number(res.headers.get('content-length') || 0) > 4 * 1024 * 1024)
|
|
336
|
+
return { beats: [], ok: false, raw: 'ollama response too large' };
|
|
337
|
+
const body = await res.json();
|
|
338
|
+
// A 300-token completion is a few KB — cap what enters the parse/manifest.
|
|
339
|
+
const text = (body?.message?.content || '').slice(0, 64 * 1024);
|
|
340
|
+
return { beats: parseBeats(text, durationSec), ok: true, raw: text };
|
|
341
|
+
} catch (e) {
|
|
342
|
+
return { beats: [], ok: false, raw: e?.message ? e.message : 'ollama scout failed' };
|
|
343
|
+
} finally {
|
|
344
|
+
rmSync(dir, { recursive: true, force: true });
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
197
348
|
function gemmaScout(clip, durationSec, { python, model, fps }) {
|
|
198
349
|
const r = spawnSync(
|
|
199
350
|
python,
|
|
@@ -301,7 +452,7 @@ function parseArgs(argv) {
|
|
|
301
452
|
return o;
|
|
302
453
|
}
|
|
303
454
|
|
|
304
|
-
function main() {
|
|
455
|
+
async function main() {
|
|
305
456
|
const o = parseArgs(process.argv.slice(2));
|
|
306
457
|
if (o.help || !o.asset) {
|
|
307
458
|
process.stdout.write(
|
|
@@ -318,7 +469,7 @@ function main() {
|
|
|
318
469
|
if (pref) o.engine = pref;
|
|
319
470
|
}
|
|
320
471
|
|
|
321
|
-
const avail = detectAvailability();
|
|
472
|
+
const avail = await detectAvailability(process.env, o.engine);
|
|
322
473
|
let tier;
|
|
323
474
|
try {
|
|
324
475
|
tier = selectTier(o.engine, avail);
|
|
@@ -362,17 +513,32 @@ function main() {
|
|
|
362
513
|
const cuts = ffmpegSceneCuts(clip, o.sceneThresh);
|
|
363
514
|
let beats = [];
|
|
364
515
|
let method = 'ffmpeg';
|
|
516
|
+
let scoutKind = null;
|
|
365
517
|
if (tier === 'gemma') {
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
518
|
+
// mlx-vlm preferred (benchmarked, native video input); Ollama is the
|
|
519
|
+
// accessible alternative runtime when mlx isn't installed.
|
|
520
|
+
let scout;
|
|
521
|
+
if (avail.mlxPython) {
|
|
522
|
+
scoutKind = 'mlx';
|
|
523
|
+
const model = process.env.MAUDE_GEMMA_MODEL || 'mlx-community/gemma-4-e4b-it-4bit';
|
|
524
|
+
scout = gemmaScout(clip, meta.durationSec, {
|
|
525
|
+
python: avail.mlxPython,
|
|
526
|
+
model,
|
|
527
|
+
fps: o.scoutFps,
|
|
528
|
+
});
|
|
529
|
+
} else {
|
|
530
|
+
scoutKind = 'ollama';
|
|
531
|
+
scout = await ollamaScout(clip, meta.durationSec, {
|
|
532
|
+
host: avail.ollama.host,
|
|
533
|
+
model: avail.ollama.model,
|
|
534
|
+
fps: o.scoutFps,
|
|
535
|
+
});
|
|
536
|
+
}
|
|
372
537
|
if (scout.ok && scout.beats.length) {
|
|
373
538
|
beats = scout.beats;
|
|
374
539
|
method = 'gemma';
|
|
375
540
|
} else {
|
|
541
|
+
scoutKind = null;
|
|
376
542
|
process.stderr.write(
|
|
377
543
|
'smart-frames: gemma scout produced no beats — falling back to ffmpeg tier\n'
|
|
378
544
|
);
|
|
@@ -404,6 +570,7 @@ function main() {
|
|
|
404
570
|
width: meta.width,
|
|
405
571
|
height: meta.height,
|
|
406
572
|
method,
|
|
573
|
+
scout: scoutKind,
|
|
407
574
|
sceneCuts: cuts,
|
|
408
575
|
scoutBeats: beats,
|
|
409
576
|
outDir,
|
|
@@ -411,9 +578,9 @@ function main() {
|
|
|
411
578
|
};
|
|
412
579
|
process.stdout.write(`${JSON.stringify(manifest)}\n`);
|
|
413
580
|
process.stderr.write(
|
|
414
|
-
`smart-frames: engine=${method} · ${frames.length} frames · ${cuts.length} scene cuts · ${beats.length} scout beats\n`
|
|
581
|
+
`smart-frames: engine=${method}${scoutKind ? ` (${scoutKind})` : ''} · ${frames.length} frames · ${cuts.length} scene cuts · ${beats.length} scout beats\n`
|
|
415
582
|
);
|
|
416
583
|
}
|
|
417
584
|
|
|
418
585
|
// run only as a CLI, not when imported by the test
|
|
419
|
-
if (process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1]) main();
|
|
586
|
+
if (process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1]) await main();
|
|
@@ -10,8 +10,11 @@ import { join } from 'node:path';
|
|
|
10
10
|
import {
|
|
11
11
|
clipCandidates,
|
|
12
12
|
mergeTimestamps,
|
|
13
|
+
ollamaHost,
|
|
14
|
+
ollamaScoutPrompt,
|
|
13
15
|
parseBeats,
|
|
14
16
|
parseSceneCuts,
|
|
17
|
+
pickOllamaGemmaTag,
|
|
15
18
|
readEnginePref,
|
|
16
19
|
selectTier,
|
|
17
20
|
} from './_smart-frames.mjs';
|
|
@@ -85,17 +88,26 @@ describe('mergeTimestamps', () => {
|
|
|
85
88
|
});
|
|
86
89
|
|
|
87
90
|
describe('selectTier', () => {
|
|
88
|
-
const gemmaReady = { ffmpeg: true, mlxPython: 'python3' };
|
|
89
|
-
const
|
|
90
|
-
|
|
91
|
+
const gemmaReady = { ffmpeg: true, mlxPython: 'python3', ollama: null };
|
|
92
|
+
const ollamaReady = {
|
|
93
|
+
ffmpeg: true,
|
|
94
|
+
mlxPython: null,
|
|
95
|
+
ollama: { host: 'http://127.0.0.1:11434', model: 'gemma3:4b' },
|
|
96
|
+
};
|
|
97
|
+
const ffmpegOnly = { ffmpeg: true, mlxPython: null, ollama: null };
|
|
98
|
+
const bare = { ffmpeg: false, mlxPython: null, ollama: null };
|
|
91
99
|
|
|
92
100
|
test('auto degrades gemma → ffmpeg → blind', () => {
|
|
93
101
|
expect(selectTier('auto', gemmaReady)).toBe('gemma');
|
|
94
102
|
expect(selectTier('auto', ffmpegOnly)).toBe('ffmpeg');
|
|
95
103
|
expect(selectTier('auto', bare)).toBe('blind');
|
|
96
104
|
});
|
|
105
|
+
test('ollama alone unlocks the gemma tier (no mlx-vlm needed)', () => {
|
|
106
|
+
expect(selectTier('auto', ollamaReady)).toBe('gemma');
|
|
107
|
+
expect(selectTier('gemma', ollamaReady)).toBe('gemma');
|
|
108
|
+
});
|
|
97
109
|
test('explicit engine errors when its deps are missing (no silent downgrade)', () => {
|
|
98
|
-
expect(() => selectTier('gemma', ffmpegOnly)).toThrow(/mlx-vlm/);
|
|
110
|
+
expect(() => selectTier('gemma', ffmpegOnly)).toThrow(/mlx-vlm|Ollama/);
|
|
99
111
|
expect(() => selectTier('ffmpeg', bare)).toThrow(/ffmpeg/);
|
|
100
112
|
});
|
|
101
113
|
test('blind is always allowed', () => {
|
|
@@ -103,6 +115,49 @@ describe('selectTier', () => {
|
|
|
103
115
|
});
|
|
104
116
|
});
|
|
105
117
|
|
|
118
|
+
describe('ollama runtime helpers', () => {
|
|
119
|
+
test('pickOllamaGemmaTag picks a vision-capable gemma3 tag', () => {
|
|
120
|
+
expect(pickOllamaGemmaTag(['llama3:8b', 'gemma3:4b'], {})).toBe('gemma3:4b');
|
|
121
|
+
expect(pickOllamaGemmaTag(['gemma3:latest'], {})).toBe('gemma3:latest');
|
|
122
|
+
expect(pickOllamaGemmaTag(['gemma3:12b'], {})).toBe('gemma3:12b');
|
|
123
|
+
});
|
|
124
|
+
test('pickOllamaGemmaTag excludes text-only tags', () => {
|
|
125
|
+
expect(pickOllamaGemmaTag(['gemma3:1b'], {})).toBe(null);
|
|
126
|
+
expect(pickOllamaGemmaTag(['gemma3n:e4b'], {})).toBe(null);
|
|
127
|
+
expect(pickOllamaGemmaTag(['llama3:8b'], {})).toBe(null);
|
|
128
|
+
});
|
|
129
|
+
test('$MAUDE_OLLAMA_MODEL wins verbatim', () => {
|
|
130
|
+
expect(pickOllamaGemmaTag(['gemma3:4b'], { MAUDE_OLLAMA_MODEL: 'gemma3:27b' })).toBe(
|
|
131
|
+
'gemma3:27b'
|
|
132
|
+
);
|
|
133
|
+
});
|
|
134
|
+
test('ollamaHost normalizes $OLLAMA_HOST and pins it to loopback', () => {
|
|
135
|
+
expect(ollamaHost({})).toBe('http://127.0.0.1:11434');
|
|
136
|
+
expect(ollamaHost({ OLLAMA_HOST: 'http://localhost:11434/' })).toBe('http://localhost:11434');
|
|
137
|
+
expect(ollamaHost({ OLLAMA_HOST: '127.0.0.1:11434' })).toBe('http://127.0.0.1:11434');
|
|
138
|
+
expect(ollamaHost({ OLLAMA_HOST: '[::1]:11434' })).toBe('http://[::1]:11434');
|
|
139
|
+
// non-loopback = refused (frames must never leave this machine — DDR-183)
|
|
140
|
+
expect(ollamaHost({ OLLAMA_HOST: '0.0.0.0:11434' })).toBe(null);
|
|
141
|
+
expect(ollamaHost({ OLLAMA_HOST: '192.168.1.20:11434' })).toBe(null);
|
|
142
|
+
expect(ollamaHost({ OLLAMA_HOST: 'http://ollama.internal:11434' })).toBe(null);
|
|
143
|
+
expect(ollamaHost({ OLLAMA_HOST: 'http://169.254.169.254' })).toBe(null);
|
|
144
|
+
});
|
|
145
|
+
test('parseBeats sanitizes model-authored labels (control chars stripped, capped)', () => {
|
|
146
|
+
const evil = `0:05 | a\x00b\x1bc ${'x'.repeat(300)}`;
|
|
147
|
+
const beats = parseBeats(evil, 10);
|
|
148
|
+
expect(beats).toHaveLength(1);
|
|
149
|
+
expect(beats[0].what.length).toBeLessThanOrEqual(120);
|
|
150
|
+
// biome-ignore lint/suspicious/noControlCharactersInRegex: asserting that control chars were stripped is the point.
|
|
151
|
+
expect(beats[0].what).not.toMatch(/[\x00-\x1f]/);
|
|
152
|
+
});
|
|
153
|
+
test('ollamaScoutPrompt maps frames to timestamps', () => {
|
|
154
|
+
const p = ollamaScoutPrompt(10, [0.05, 5, 9.95]);
|
|
155
|
+
expect(p).toContain('frame 1 = t=0.05s');
|
|
156
|
+
expect(p).toContain('frame 3 = t=9.95s');
|
|
157
|
+
expect(p).toContain('TIME=');
|
|
158
|
+
});
|
|
159
|
+
});
|
|
160
|
+
|
|
106
161
|
describe('clipCandidates', () => {
|
|
107
162
|
test('tries raw, root-relative, and designRoot/assets locations', () => {
|
|
108
163
|
const c = clipCandidates('assets/abc12345.mp4', '/repo', '.design');
|
|
@@ -123,7 +123,7 @@ function parseArgs(argv) {
|
|
|
123
123
|
}
|
|
124
124
|
|
|
125
125
|
const CLOUD_PROVIDERS = new Set(['elevenlabs', 'groq']);
|
|
126
|
-
const KNOWN_PROVIDERS = new Set(['whisper', 'elevenlabs', 'groq']);
|
|
126
|
+
const KNOWN_PROVIDERS = new Set(['auto', 'whisper', 'elevenlabs', 'groq']);
|
|
127
127
|
|
|
128
128
|
/**
|
|
129
129
|
* Resolve the transcription engine as an EXPLICIT choice (Task 2.6): the
|
|
@@ -136,7 +136,7 @@ export function resolveProvider(flag, repo) {
|
|
|
136
136
|
if (flag) {
|
|
137
137
|
if (!KNOWN_PROVIDERS.has(flag)) {
|
|
138
138
|
process.stderr.write(
|
|
139
|
-
`transcribe: unknown --provider '${flag}' (use whisper | elevenlabs | groq)\n`
|
|
139
|
+
`transcribe: unknown --provider '${flag}' (use auto | whisper | elevenlabs | groq)\n`
|
|
140
140
|
);
|
|
141
141
|
process.exit(2);
|
|
142
142
|
}
|
|
@@ -148,6 +148,39 @@ export function resolveProvider(flag, repo) {
|
|
|
148
148
|
return { provider: 'whisper', defaulted: true };
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
+
/**
|
|
152
|
+
* Settle an `auto` choice into a concrete engine. `auto` prefers a cloud engine
|
|
153
|
+
* whose key is set (ElevenLabs Scribe first), else local whisper — mirroring
|
|
154
|
+
* generation/whisper-models.ts `resolveAutoEngine`. Key presence lives
|
|
155
|
+
* server-side (the keychain), so this asks the local dev server; with no server
|
|
156
|
+
* reachable there is no key to use anyway, so it settles on local whisper.
|
|
157
|
+
*
|
|
158
|
+
* Still no SILENT cloud switch (DDR-164): `auto` is a chosen mode, and the
|
|
159
|
+
* caller prints which engine it resolved to before transcribing.
|
|
160
|
+
*/
|
|
161
|
+
export async function settleAutoProvider(provider, port) {
|
|
162
|
+
if (provider !== 'auto') return { provider, autoReason: null };
|
|
163
|
+
const base = `http://127.0.0.1:${port || process.env.MAUDE_PORT || 4399}`;
|
|
164
|
+
try {
|
|
165
|
+
const res = await fetch(`${base}/_api/generate/providers`, {
|
|
166
|
+
signal: AbortSignal.timeout(1500),
|
|
167
|
+
});
|
|
168
|
+
if (res.ok) {
|
|
169
|
+
const body = await res.json();
|
|
170
|
+
const keyed = new Set(
|
|
171
|
+
(body.providers || []).filter((p) => p && p.configured).map((p) => p.id)
|
|
172
|
+
);
|
|
173
|
+
if (keyed.has('elevenlabs'))
|
|
174
|
+
return { provider: 'elevenlabs', autoReason: 'auto → ElevenLabs Scribe (key is set)' };
|
|
175
|
+
if (keyed.has('groq'))
|
|
176
|
+
return { provider: 'groq', autoReason: 'auto → Groq Whisper (key is set)' };
|
|
177
|
+
}
|
|
178
|
+
} catch {
|
|
179
|
+
/* no server / no keys — local whisper is the honest answer */
|
|
180
|
+
}
|
|
181
|
+
return { provider: 'whisper', autoReason: 'auto → local whisper.cpp (no cloud key set)' };
|
|
182
|
+
}
|
|
183
|
+
|
|
151
184
|
/** Read `.design/config.json` → generation.transcription.provider, or null. */
|
|
152
185
|
function readConfigProvider(repo) {
|
|
153
186
|
try {
|
|
@@ -303,7 +336,11 @@ async function main() {
|
|
|
303
336
|
// Resolve the engine as an explicit choice (Task 2.6). A chosen cloud engine
|
|
304
337
|
// routes through the dev server (key server-side); the local default stays
|
|
305
338
|
// whisper. We NEVER auto-switch between them.
|
|
306
|
-
const { provider, defaulted } = resolveProvider(args.provider, repo);
|
|
339
|
+
const { provider: chosen, defaulted } = resolveProvider(args.provider, repo);
|
|
340
|
+
// `auto` settles into a concrete engine here, and we SAY which — the engine is
|
|
341
|
+
// never a mystery even when the user delegated the pick (DDR-164).
|
|
342
|
+
const { provider, autoReason } = await settleAutoProvider(chosen, args.port);
|
|
343
|
+
if (autoReason) process.stderr.write(`transcribe: ${autoReason}\n`);
|
|
307
344
|
if (defaulted)
|
|
308
345
|
process.stderr.write(
|
|
309
346
|
'transcribe: no engine configured — using local whisper ' +
|
package/apps/studio/bin/smoke.sh
CHANGED
|
@@ -157,7 +157,13 @@ if [ "$CHANGED_ONLY" = "1" ]; then
|
|
|
157
157
|
git -C "$REPO" ls-files --others --exclude-standard -- .design/ui .design/system 2>/dev/null
|
|
158
158
|
} | sort -u
|
|
159
159
|
)
|
|
160
|
-
|
|
160
|
+
# `apps/studio/` is the dev server. The pattern used to say `dev-server/`,
|
|
161
|
+
# which is where it lived before the move — so from the rename until
|
|
162
|
+
# 2026-08-06 the escalation silently never fired, and a dev-server change
|
|
163
|
+
# got the narrow "only canvases that changed" sweep instead of the full one.
|
|
164
|
+
# Kept alongside the old path rather than replaced: a --changed-only run in
|
|
165
|
+
# an older checkout should still escalate.
|
|
166
|
+
if printf '%s\n' "$CHANGED" | grep -qE 'apps/studio/|dev-server/|canvas-lib\.tsx|canvas[^/]*\.tsx\.template'; then
|
|
161
167
|
echo "→ --changed-only: dev-server / canvas-lib / template changed — escalating to FULL set" >&2
|
|
162
168
|
else
|
|
163
169
|
# Keep only canvases whose repo-relative path is in the changed set.
|
|
@@ -43,9 +43,21 @@ export interface SandboxBuildFail {
|
|
|
43
43
|
ok: false;
|
|
44
44
|
error: string;
|
|
45
45
|
kind: 'build' | 'timeout' | 'memory' | 'runtime';
|
|
46
|
+
/** How many imports the allowlist refused. A COUNT — never the specifiers. */
|
|
47
|
+
rejectedImports?: number;
|
|
46
48
|
}
|
|
47
49
|
export type SandboxBuildResult = SandboxBuildOk | SandboxBuildFail;
|
|
48
50
|
|
|
51
|
+
/**
|
|
52
|
+
* How much history the counters hold — Cloud Phase 26 Stage 4.
|
|
53
|
+
*
|
|
54
|
+
* A ROLLING WINDOW, not a lifetime total: the sweep reads this hourly, and a
|
|
55
|
+
* monotonic counter read at intervals turns every "how busy was the last hour"
|
|
56
|
+
* question into a subtraction the reader has to get right. An hour of history
|
|
57
|
+
* answers it directly.
|
|
58
|
+
*/
|
|
59
|
+
export const STATS_WINDOW_MS = Number(process.env.MAUDE_CANVAS_STATS_WINDOW_MS ?? 3_600_000);
|
|
60
|
+
|
|
49
61
|
const counters = {
|
|
50
62
|
builds: 0,
|
|
51
63
|
cacheHits: 0,
|
|
@@ -53,12 +65,58 @@ const counters = {
|
|
|
53
65
|
failures: 0,
|
|
54
66
|
timeouts: 0,
|
|
55
67
|
memoryKills: 0,
|
|
68
|
+
/** Imports the allowlist refused. The COUNT only — see buildStats(). */
|
|
69
|
+
rejectedImports: 0,
|
|
70
|
+
/** Summed wall-clock of completed builds, the closest thing to Active-CPU
|
|
71
|
+
* this side of the boundary can honestly report. */
|
|
72
|
+
cpuMsTotal: 0,
|
|
73
|
+
/** The biggest import graph any build had to read, in bytes. */
|
|
74
|
+
largestGraphBytes: 0,
|
|
56
75
|
/** Every completed build's wall-clock, newest last, capped. */
|
|
57
76
|
durationsMs: [] as number[],
|
|
77
|
+
/** When this window started. THE FIELD THAT MAKES A ZERO READABLE. */
|
|
78
|
+
windowStartedAt: Date.now(),
|
|
58
79
|
};
|
|
59
80
|
|
|
60
|
-
/**
|
|
81
|
+
/** Start a fresh window, keeping the cache (which is not a counter). */
|
|
82
|
+
function rollWindow(now: number): void {
|
|
83
|
+
counters.builds = 0;
|
|
84
|
+
counters.cacheHits = 0;
|
|
85
|
+
counters.cacheMisses = 0;
|
|
86
|
+
counters.failures = 0;
|
|
87
|
+
counters.timeouts = 0;
|
|
88
|
+
counters.memoryKills = 0;
|
|
89
|
+
counters.rejectedImports = 0;
|
|
90
|
+
counters.cpuMsTotal = 0;
|
|
91
|
+
counters.largestGraphBytes = 0;
|
|
92
|
+
counters.durationsMs = [];
|
|
93
|
+
counters.windowStartedAt = now;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function maybeRoll(now = Date.now()): void {
|
|
97
|
+
if (now - counters.windowStartedAt >= STATS_WINDOW_MS) rollWindow(now);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* A snapshot for the operator surface + the cost lane.
|
|
102
|
+
*
|
|
103
|
+
* COUNTS AND DURATIONS ONLY. Never a canvas name, never a path, and never the
|
|
104
|
+
* text of a rejected import specifier — that specifier is written by the
|
|
105
|
+
* tenant, so it is their content; the operational fact is that a rejection
|
|
106
|
+
* happened, not what it said.
|
|
107
|
+
*
|
|
108
|
+
* `windowStartedAt` travels with the counters and is load-bearing. The window
|
|
109
|
+
* lives in memory, so a cell restart resets it — and without the start time a
|
|
110
|
+
* row of zeroes from a cell that rebooted a minute ago is indistinguishable
|
|
111
|
+
* from a genuinely quiet hour. One of those is nothing to do; the other is a
|
|
112
|
+
* crash loop.
|
|
113
|
+
*
|
|
114
|
+
* `cacheHitRatio` is offered for a human reading this payload directly, but
|
|
115
|
+
* the two COUNTS are what travel to the operator board, which divides at read
|
|
116
|
+
* time. A ratio computed over a window that reset is a confident lie.
|
|
117
|
+
*/
|
|
61
118
|
export function buildStats() {
|
|
119
|
+
maybeRoll();
|
|
62
120
|
const d = [...counters.durationsMs].sort((a, b) => a - b);
|
|
63
121
|
const p = (q: number) =>
|
|
64
122
|
d.length === 0 ? null : d[Math.min(d.length - 1, Math.floor(d.length * q))];
|
|
@@ -71,20 +129,26 @@ export function buildStats() {
|
|
|
71
129
|
failures: counters.failures,
|
|
72
130
|
timeouts: counters.timeouts,
|
|
73
131
|
memoryKills: counters.memoryKills,
|
|
132
|
+
rejectedImports: counters.rejectedImports,
|
|
133
|
+
cpuMsTotal: counters.cpuMsTotal,
|
|
134
|
+
largestGraphBytes: counters.largestGraphBytes,
|
|
74
135
|
p50Ms: p(0.5),
|
|
75
136
|
p95Ms: p(0.95),
|
|
137
|
+
maxMs: d.length === 0 ? null : d[d.length - 1],
|
|
138
|
+
windowMs: STATS_WINDOW_MS,
|
|
139
|
+
windowStartedAt: counters.windowStartedAt,
|
|
76
140
|
};
|
|
77
141
|
}
|
|
78
142
|
|
|
143
|
+
/** Record one import the allowlist refused. The specifier is NOT passed in. */
|
|
144
|
+
export function noteRejectedImport(): void {
|
|
145
|
+
maybeRoll();
|
|
146
|
+
counters.rejectedImports++;
|
|
147
|
+
}
|
|
148
|
+
|
|
79
149
|
/** Test seam. */
|
|
80
150
|
export function _resetBuildStats(): void {
|
|
81
|
-
|
|
82
|
-
counters.cacheHits = 0;
|
|
83
|
-
counters.cacheMisses = 0;
|
|
84
|
-
counters.failures = 0;
|
|
85
|
-
counters.timeouts = 0;
|
|
86
|
-
counters.memoryKills = 0;
|
|
87
|
-
counters.durationsMs = [];
|
|
151
|
+
rollWindow(Date.now());
|
|
88
152
|
cache.clear();
|
|
89
153
|
}
|
|
90
154
|
|
|
@@ -154,6 +218,19 @@ function resolveCandidates(fromDir: string, spec: string): string[] {
|
|
|
154
218
|
];
|
|
155
219
|
}
|
|
156
220
|
|
|
221
|
+
/** Total bytes of every source a build could have to read. Sizes only. */
|
|
222
|
+
function graphBytes(designRoot: string, canvasAbs: string): number {
|
|
223
|
+
let total = 0;
|
|
224
|
+
for (const file of relevantSources(designRoot, canvasAbs)) {
|
|
225
|
+
try {
|
|
226
|
+
total += statSync(file).size;
|
|
227
|
+
} catch {
|
|
228
|
+
/* a file that vanished contributes nothing rather than failing a build */
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
return total;
|
|
232
|
+
}
|
|
233
|
+
|
|
157
234
|
function remember(key: string, value: { js: string; locator: unknown; etag: string }): void {
|
|
158
235
|
cache.set(key, value);
|
|
159
236
|
while (cache.size > CACHE_MAX_ENTRIES) {
|
|
@@ -258,6 +335,7 @@ export async function buildCanvasSandboxed({
|
|
|
258
335
|
canvasAbs: string;
|
|
259
336
|
env?: NodeJS.ProcessEnv;
|
|
260
337
|
}): Promise<SandboxBuildResult> {
|
|
338
|
+
maybeRoll();
|
|
261
339
|
const key = cacheKeyFor(designRoot, canvasAbs);
|
|
262
340
|
const hit = cache.get(key);
|
|
263
341
|
if (hit) {
|
|
@@ -265,10 +343,18 @@ export async function buildCanvasSandboxed({
|
|
|
265
343
|
return { ...hit, ok: true, cached: true };
|
|
266
344
|
}
|
|
267
345
|
counters.cacheMisses++;
|
|
346
|
+
// The pathological-canvas signal, measured rather than argued: a huge import
|
|
347
|
+
// graph burns our Active-CPU while the tenant pays a flat rate. Bytes, never
|
|
348
|
+
// filenames.
|
|
349
|
+
counters.largestGraphBytes = Math.max(
|
|
350
|
+
counters.largestGraphBytes,
|
|
351
|
+
graphBytes(designRoot, canvasAbs)
|
|
352
|
+
);
|
|
268
353
|
|
|
269
354
|
const started = Date.now();
|
|
270
355
|
const result = await runWorker({ designRoot, canvasAbs, env });
|
|
271
356
|
const elapsed = Date.now() - started;
|
|
357
|
+
counters.cpuMsTotal += elapsed;
|
|
272
358
|
counters.durationsMs.push(elapsed);
|
|
273
359
|
if (counters.durationsMs.length > 200) counters.durationsMs.shift();
|
|
274
360
|
|
|
@@ -281,6 +367,7 @@ export async function buildCanvasSandboxed({
|
|
|
281
367
|
counters.failures++;
|
|
282
368
|
if (result.kind === 'timeout') counters.timeouts++;
|
|
283
369
|
if (result.kind === 'memory') counters.memoryKills++;
|
|
370
|
+
counters.rejectedImports += result.rejectedImports ?? 0;
|
|
284
371
|
return result;
|
|
285
372
|
}
|
|
286
373
|
|
|
@@ -362,7 +449,12 @@ async function runWorker({
|
|
|
362
449
|
if (parsed.ok) {
|
|
363
450
|
return { ok: true, js: parsed.js, locator: parsed.locator, etag: parsed.etag, cached: false };
|
|
364
451
|
}
|
|
365
|
-
return {
|
|
452
|
+
return {
|
|
453
|
+
ok: false,
|
|
454
|
+
kind: 'build',
|
|
455
|
+
error: String(parsed.error),
|
|
456
|
+
rejectedImports: Number(parsed.rejectedImports ?? 0) || 0,
|
|
457
|
+
};
|
|
366
458
|
} catch {
|
|
367
459
|
return {
|
|
368
460
|
ok: false,
|