@turbojev/runtime-llamacpp-wasm 0.29.0 → 0.29.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +121 -39
- package/package.json +1 -1
- package/src/index.js +262 -123
- package/src/media-worker-pool.js +18 -0
package/README.md
CHANGED
|
@@ -1,39 +1,121 @@
|
|
|
1
|
-
# TurboJev llama.cpp WASM runtime
|
|
2
|
-
|
|
3
|
-
Browser decision adapter for
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
The adapter sends the model's chat-formatted request to llama.cpp, constrains
|
|
9
|
-
the next answer to `A`–`T` with grammar, and maps returned log-probabilities
|
|
10
|
-
back to the original candidate order. It does not contain model-specific token
|
|
11
|
-
IDs. One inference call is made per decision task
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
await engine.
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
1
|
+
# TurboJev llama.cpp WASM runtime
|
|
2
|
+
|
|
3
|
+
Browser decision adapter for compatible llama.cpp GGUF models, using Wllama's
|
|
4
|
+
WebAssembly build. It implements TurboJev's `BrowserDecisionRuntime` contract
|
|
5
|
+
and returns one score per candidate. The Rust/WASM engine validates the scores
|
|
6
|
+
and creates the shared `EvaluationResponse`.
|
|
7
|
+
|
|
8
|
+
The adapter sends the model's chat-formatted request to llama.cpp, constrains
|
|
9
|
+
the next answer to `A`–`T` with grammar, and maps returned log-probabilities
|
|
10
|
+
back to the original candidate order. It does not contain model-specific token
|
|
11
|
+
IDs. One inference call is made per decision task unless cyclic order correction
|
|
12
|
+
is requested. Wllama 3.8.1 supplies
|
|
13
|
+
llama.cpp's WebAssembly multimodal path: set `mmprojUrl` to the projector paired
|
|
14
|
+
with the GGUF to enable supported image input. Video is passed as ordered
|
|
15
|
+
timestamped image frames; it does not guarantee temporal reasoning. When Wllama
|
|
16
|
+
reports an audio encoder, the adapter forwards audio bytes before the prompt
|
|
17
|
+
text. Audio needs cross-origin isolation and a prestarted worker pool for mtmd
|
|
18
|
+
preprocessing. The adapter reserves those workers independently of `nThreads`;
|
|
19
|
+
the requested inference thread count is preserved.
|
|
20
|
+
|
|
21
|
+
```js
|
|
22
|
+
import { TurboJevWeb } from '@turbojev/web';
|
|
23
|
+
|
|
24
|
+
const engine = await TurboJevWeb.load({
|
|
25
|
+
model: 'Qwen3 0.6B Q4_K_M',
|
|
26
|
+
runtimeModuleUrl: '/llamacpp/index.js',
|
|
27
|
+
wasmModuleUrl: '/turbojev_wasm.js',
|
|
28
|
+
runtimeOptions: {
|
|
29
|
+
modelUrl: 'https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/1208e45d782fe18602c5eaf10e5758d5b0f24c03/Qwen3-0.6B-Q4_K_M.gguf',
|
|
30
|
+
device: 'wasm', // use 'webgpu' to request GPU execution
|
|
31
|
+
},
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
const result = await engine.evaluate({
|
|
35
|
+
state: { message: 'My package arrived broken' },
|
|
36
|
+
questions: { department: {
|
|
37
|
+
type: 'choice',
|
|
38
|
+
instructions: 'Which department should handle this message?',
|
|
39
|
+
criteria: { billing: 'Billing', shipping: 'Shipping', technical: 'Technical', other: 'Other' },
|
|
40
|
+
} },
|
|
41
|
+
});
|
|
42
|
+
await engine.close();
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
For image input, provide the matching projector and declare the modality when
|
|
46
|
+
loading:
|
|
47
|
+
|
|
48
|
+
```js
|
|
49
|
+
const engine = await TurboJevWeb.load({
|
|
50
|
+
model: 'LFM2-VL 450M Q4_0',
|
|
51
|
+
runtimeModuleUrl: '/llamacpp/index.js',
|
|
52
|
+
wasmModuleUrl: '/turbojev_wasm.js',
|
|
53
|
+
runtimeOptions: {
|
|
54
|
+
modelUrl: 'https://huggingface.co/runanywhere/LFM2-VL-450M-GGUF/resolve/main/LFM2-VL-450M-Q4_0.gguf',
|
|
55
|
+
mmprojUrl: 'https://huggingface.co/runanywhere/LFM2-VL-450M-GGUF/resolve/main/mmproj-LFM2-VL-450M-Q8_0.gguf',
|
|
56
|
+
modality: 'image',
|
|
57
|
+
device: 'wasm',
|
|
58
|
+
},
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
import { mediaEvidence } from '@turbojev/web';
|
|
62
|
+
|
|
63
|
+
const evidence = await mediaEvidence(imageFile, 'image', 'Read the traffic sign.');
|
|
64
|
+
const result = await engine.evaluate({
|
|
65
|
+
state: evidence,
|
|
66
|
+
questions: { sign: {
|
|
67
|
+
type: 'choice',
|
|
68
|
+
instructions: 'Which sign is visible?',
|
|
69
|
+
criteria: { stop: 'A stop sign', yield: 'A yield sign' },
|
|
70
|
+
} },
|
|
71
|
+
});
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
For spoken-command decisions, load the matching LFM2 Audio GGUF and projector:
|
|
75
|
+
|
|
76
|
+
```js
|
|
77
|
+
const engine = await TurboJevWeb.load({
|
|
78
|
+
model: 'LFM2-Audio-1.5B-Q8_0',
|
|
79
|
+
runtimeModuleUrl: '/llamacpp/index.js',
|
|
80
|
+
wasmModuleUrl: '/turbojev_wasm.js',
|
|
81
|
+
cyclicOrderRobustness: true, // Optional: one evaluation per choice; false is one pass.
|
|
82
|
+
runtimeOptions: {
|
|
83
|
+
modelUrl: 'https://huggingface.co/ggml-org/LFM2-Audio-1.5B-GGUF/resolve/main/LFM2-Audio-1.5B-Q8_0.gguf',
|
|
84
|
+
mmprojUrl: 'https://huggingface.co/ggml-org/LFM2-Audio-1.5B-GGUF/resolve/main/mmproj-LFM2-Audio-1.5B-Q8_0.gguf',
|
|
85
|
+
modality: 'audio', device: 'wasm', nThreads: 4, jinja: true, warmup: false,
|
|
86
|
+
},
|
|
87
|
+
});
|
|
88
|
+
const state = await mediaEvidence(audioFile, 'audio');
|
|
89
|
+
const result = await engine.evaluate({ state, questions: { command: {
|
|
90
|
+
type: 'choice', instructions: 'Which command does the speaker say?',
|
|
91
|
+
criteria: { on: 'Turn on the light', off: 'Turn off the light' },
|
|
92
|
+
} } });
|
|
93
|
+
await engine.close();
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Serve with `Cross-Origin-Opener-Policy: same-origin` and
|
|
97
|
+
`Cross-Origin-Embedder-Policy: require-corp`. Audio fails before loading if
|
|
98
|
+
isolation is absent, rather than hanging during preprocessing. The language
|
|
99
|
+
model and projector download about 1.58 GB together; CPU/WASM inference is
|
|
100
|
+
expensive. This example does not imply that arbitrary audio models accept the
|
|
101
|
+
same template. The standalone real-WASM check is
|
|
102
|
+
`scripts/validate_multimodal_wasm_audio.mjs`; browser qualification is separate.
|
|
103
|
+
|
|
104
|
+
`modality` is checked against Wllama after loading. Image input is discovered
|
|
105
|
+
from the loaded model/projector; video requires image support. The reported
|
|
106
|
+
audio flag is only an encoder capability check; it does not prove that inference
|
|
107
|
+
completes. These flags do not guarantee that a model's chat template accepts
|
|
108
|
+
multimodal content blocks. Evaluation returns
|
|
109
|
+
`MODEL_PROMPT_FORMAT_UNSUPPORTED` when Wllama cannot format that model's media
|
|
110
|
+
prompt. Set `jinja` or pass `chatTemplate` when an image model requires a
|
|
111
|
+
different template. The repository's `py website/serve.py` development server
|
|
112
|
+
sets COOP/COEP headers for browser WebAssembly threading; they do not by
|
|
113
|
+
themselves qualify the audio path. The
|
|
114
|
+
`@turbojev/web` media helpers encode input bytes but do not decode compressed
|
|
115
|
+
video containers. Use `videoFrameEvidence` with ordered image frames for video.
|
|
116
|
+
|
|
117
|
+
Copy the entire `src/` directory to `/llamacpp/`, including
|
|
118
|
+
`media-worker-pool.js`, or bundle `src/index.js` as one ES module. The host must
|
|
119
|
+
serve this adapter, TurboJev's generated WASM files, and the page
|
|
120
|
+
over localhost or HTTPS. Model files are fetched from their URL and remain in
|
|
121
|
+
the browser. Wllama 3.8.1 and its WASM binary are loaded from jsDelivr.
|
package/package.json
CHANGED
package/src/index.js
CHANGED
|
@@ -1,123 +1,262 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
const
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
});
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
}
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
1
|
+
import { reserveMediaWorkers } from './media-worker-pool.js';
|
|
2
|
+
|
|
3
|
+
const WLLAMA_VERSION = '3.8.1';
|
|
4
|
+
const WLLAMA_ENTRY = `https://cdn.jsdelivr.net/npm/@wllama/wllama@${WLLAMA_VERSION}/esm/index.js`;
|
|
5
|
+
const WLLAMA_WASM = `https://cdn.jsdelivr.net/npm/@wllama/wllama@${WLLAMA_VERSION}/esm/wasm/wllama.wasm`;
|
|
6
|
+
const WLLAMA_WORKER = `https://cdn.jsdelivr.net/npm/@wllama/wllama@${WLLAMA_VERSION}/src/wasm/wllama.js`;
|
|
7
|
+
|
|
8
|
+
export async function createRuntime(dependencies = {}) {
|
|
9
|
+
let engine;
|
|
10
|
+
let model = '';
|
|
11
|
+
let modalities = ['text'];
|
|
12
|
+
let debug = false;
|
|
13
|
+
const loadWllama = dependencies.loadWllama || (() => import(WLLAMA_ENTRY));
|
|
14
|
+
const loadWorkerCode = dependencies.loadWorkerCode || (async (url) => {
|
|
15
|
+
const response = await fetch(url);
|
|
16
|
+
if (!response.ok) throw new Error(`Could not load Wllama worker: HTTP ${response.status}`);
|
|
17
|
+
return response.text();
|
|
18
|
+
});
|
|
19
|
+
|
|
20
|
+
return {
|
|
21
|
+
async initialize(options) {
|
|
22
|
+
if (options.thinking === true) {
|
|
23
|
+
throw new Error('THINKING_UNSUPPORTED: this adapter supports direct decision scores only; use the native GGUF text runtime for thinking.');
|
|
24
|
+
}
|
|
25
|
+
debug = options.debug === true;
|
|
26
|
+
if (!options.modelUrl) {
|
|
27
|
+
throw new Error('Choose a GGUF model with a direct download URL.');
|
|
28
|
+
}
|
|
29
|
+
if (options.modality === 'audio' && globalThis.crossOriginIsolated === false) {
|
|
30
|
+
throw new Error('GGUF audio requires cross-origin isolation. Serve the page with COOP: same-origin and COEP: require-corp to enable WebAssembly workers.');
|
|
31
|
+
}
|
|
32
|
+
const inferenceThreads = Number.isInteger(options.nThreads) && options.nThreads > 0
|
|
33
|
+
? options.nThreads : Math.max(1, Math.floor((globalThis.navigator?.hardwareConcurrency || 2) / 2));
|
|
34
|
+
|
|
35
|
+
const { Wllama, LoggerWithoutDebug } = await loadWllama();
|
|
36
|
+
engine = new Wllama({ default: WLLAMA_WASM }, {
|
|
37
|
+
logger: debug ? console : LoggerWithoutDebug,
|
|
38
|
+
suppressNativeLog: !debug,
|
|
39
|
+
parallelDownloads: 4,
|
|
40
|
+
});
|
|
41
|
+
if (options.mmprojUrl && globalThis.crossOriginIsolated !== false) {
|
|
42
|
+
const getResources = engine.getWorkerResources.bind(engine);
|
|
43
|
+
const resources = getResources();
|
|
44
|
+
const source = resources.jsPath?.code || await loadWorkerCode(
|
|
45
|
+
typeof resources.jsPath === 'string' ? resources.jsPath : WLLAMA_WORKER,
|
|
46
|
+
);
|
|
47
|
+
const worker = { code: reserveMediaWorkers(source, inferenceThreads) };
|
|
48
|
+
engine.getWorkerResources = () => ({ ...getResources(), jsPath: worker });
|
|
49
|
+
}
|
|
50
|
+
model = options.modelLabel || options.modelUrl;
|
|
51
|
+
|
|
52
|
+
const modelSource = options.mmprojUrl
|
|
53
|
+
? { url: options.modelUrl, mmprojUrl: options.mmprojUrl }
|
|
54
|
+
: options.modelUrl;
|
|
55
|
+
const loadOptions = {
|
|
56
|
+
n_ctx: 2048,
|
|
57
|
+
n_batch: 512,
|
|
58
|
+
n_gpu_layers: options.device === 'webgpu' ? 999 : 0,
|
|
59
|
+
n_threads: inferenceThreads,
|
|
60
|
+
cache_prompt: false,
|
|
61
|
+
jinja: options.jinja ?? !options.mmprojUrl,
|
|
62
|
+
prefill_assistant: true,
|
|
63
|
+
warmup: options.warmup ?? options.modality !== 'audio',
|
|
64
|
+
progressCallback: ({ loaded = 0, total = 0 }) => {
|
|
65
|
+
options.progressCallback?.({
|
|
66
|
+
status: 'progress',
|
|
67
|
+
file: 'GGUF model',
|
|
68
|
+
loaded,
|
|
69
|
+
total,
|
|
70
|
+
progress: total > 0 ? Math.min(100, loaded / total * 100) : 0,
|
|
71
|
+
});
|
|
72
|
+
},
|
|
73
|
+
};
|
|
74
|
+
if (typeof options.chatTemplate === 'string' && options.chatTemplate.length > 0) {
|
|
75
|
+
loadOptions.chat_template = options.chatTemplate;
|
|
76
|
+
}
|
|
77
|
+
await engine.loadModelFromUrl(modelSource, loadOptions);
|
|
78
|
+
|
|
79
|
+
const supportsImage = engine.supportInputModality('image');
|
|
80
|
+
const supportsAudio = engine.supportInputModality('audio');
|
|
81
|
+
modalities = ['text'];
|
|
82
|
+
if (supportsImage) modalities.push('image', 'video');
|
|
83
|
+
if (supportsAudio) modalities.push('audio');
|
|
84
|
+
|
|
85
|
+
const requested = options.modality;
|
|
86
|
+
const requestedSupported = requested === undefined || requested === 'text'
|
|
87
|
+
|| (requested === 'image' && supportsImage)
|
|
88
|
+
|| (requested === 'audio' && supportsAudio)
|
|
89
|
+
|| (requested === 'video' && supportsImage);
|
|
90
|
+
if (!requestedSupported) {
|
|
91
|
+
await engine.exit();
|
|
92
|
+
engine = undefined;
|
|
93
|
+
throw new Error(`UNSUPPORTED_MODALITY: the loaded GGUF/projector does not support ${requested} input.`);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
if (requested !== 'audio') {
|
|
97
|
+
const warmup = await engine.createChatCompletion({
|
|
98
|
+
messages: [{ role: 'user', content: 'Reply with one word: ready.' }],
|
|
99
|
+
max_tokens: 1,
|
|
100
|
+
temperature: 0,
|
|
101
|
+
cache_prompt: false,
|
|
102
|
+
chat_template_kwargs: { enable_thinking: false },
|
|
103
|
+
});
|
|
104
|
+
if (!warmup?.choices?.length) throw new Error('llama.cpp WASM loaded the model but warmup returned no completion.');
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
return {
|
|
108
|
+
name: 'llama.cpp-wasm',
|
|
109
|
+
model,
|
|
110
|
+
codecProfile: 'llamacpp-wasm-json-prefill-v2',
|
|
111
|
+
scoringMode: 'decisions',
|
|
112
|
+
modalities,
|
|
113
|
+
capabilities: {
|
|
114
|
+
nextTokenLogits: false,
|
|
115
|
+
sequenceScores: false,
|
|
116
|
+
sharedPrefix: false,
|
|
117
|
+
parallelSuffixes: false,
|
|
118
|
+
selectiveOutputHead: false,
|
|
119
|
+
localInference: true,
|
|
120
|
+
},
|
|
121
|
+
};
|
|
122
|
+
},
|
|
123
|
+
|
|
124
|
+
async scoreDecisionBatch(batchJson) {
|
|
125
|
+
if (!engine) throw new Error('Load a GGUF model first.');
|
|
126
|
+
const batch = JSON.parse(batchJson);
|
|
127
|
+
debugLog(debug, `decision batch: ${batch.tasks?.length || 0} task(s), modality=${batch.state?.modality || 'text'}`);
|
|
128
|
+
const scores = [];
|
|
129
|
+
let inputTokens = 0;
|
|
130
|
+
|
|
131
|
+
for (const task of batch.tasks) {
|
|
132
|
+
const candidates = task.candidates || [];
|
|
133
|
+
if (candidates.length < 2 || candidates.length > 20) {
|
|
134
|
+
throw new Error(`llama.cpp WASM supports 2–20 choices per decision (received ${candidates.length}).`);
|
|
135
|
+
}
|
|
136
|
+
const labels = Array.from({ length: candidates.length }, (_, i) => String.fromCharCode(65 + i));
|
|
137
|
+
const isBestMatch = String(task.intent).toLowerCase() === 'bestmatch';
|
|
138
|
+
const criterion = task.instruction || (isBestMatch ? 'Choose the option that best matches the state.' : 'Choose the correct option.');
|
|
139
|
+
const optionsText = candidates.map((candidate, i) => {
|
|
140
|
+
const meaning = candidate.description && candidate.description !== candidate.key
|
|
141
|
+
? `${candidate.key}: ${candidate.description}`
|
|
142
|
+
: candidate.description || candidate.key;
|
|
143
|
+
return `${labels[i]}. ${meaning}`;
|
|
144
|
+
}).join('\n');
|
|
145
|
+
const media = batch.state?.contract === 'turbojev.media.v1' ? batch.state : null;
|
|
146
|
+
const systemPrompt = 'Make the requested decision from the supplied state. Follow the requested output format exactly.';
|
|
147
|
+
const userContent = media
|
|
148
|
+
? multimodalContent(media, criterion, optionsText, labels)
|
|
149
|
+
: `State:\n${typeof batch.state === 'string' ? batch.state : JSON.stringify(batch.state)}\n\nQuestion:\n${criterion}\n\nAllowed options:\n${optionsText}\n\nAnswer as {"answer": "<option letter>"}.`;
|
|
150
|
+
const messages = [
|
|
151
|
+
{ role: 'system', content: systemPrompt },
|
|
152
|
+
{ role: 'user', content: userContent },
|
|
153
|
+
{ role: 'assistant', content: '{"answer": "' },
|
|
154
|
+
];
|
|
155
|
+
const grammar = `root ::= ${labels.map(label => `"${label}"`).join(' | ')}`;
|
|
156
|
+
if (media) {
|
|
157
|
+
const needsImage = media.modality === 'image' || media.modality === 'video';
|
|
158
|
+
if (needsImage && !engine.supportInputModality('image')) {
|
|
159
|
+
throw new Error(`UNSUPPORTED_MODALITY: loaded llama.cpp WASM model does not support ${media.modality} input.`);
|
|
160
|
+
}
|
|
161
|
+
if (media.modality === 'audio' && !engine.supportInputModality('audio')) {
|
|
162
|
+
throw new Error('UNSUPPORTED_MODALITY: loaded llama.cpp WASM model does not support audio input.');
|
|
163
|
+
}
|
|
164
|
+
if (media.modality === 'audio' && globalThis.crossOriginIsolated === false) {
|
|
165
|
+
throw new Error('GGUF audio requires cross-origin isolation and WebAssembly workers.');
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
let response;
|
|
169
|
+
try {
|
|
170
|
+
debugLog(debug, `requesting model scores for task ${task.id}`);
|
|
171
|
+
response = await engine.createChatCompletion({
|
|
172
|
+
messages,
|
|
173
|
+
prefill_assistant: true,
|
|
174
|
+
max_tokens: 1,
|
|
175
|
+
temperature: 1,
|
|
176
|
+
top_k: 0,
|
|
177
|
+
top_p: 1,
|
|
178
|
+
logprobs: true,
|
|
179
|
+
top_logprobs: 20,
|
|
180
|
+
grammar,
|
|
181
|
+
cache_prompt: false,
|
|
182
|
+
chat_template_kwargs: { enable_thinking: false },
|
|
183
|
+
});
|
|
184
|
+
debugLog(debug, `model returned scores for task ${task.id}`);
|
|
185
|
+
} catch (error) {
|
|
186
|
+
const message = error?.message || String(error);
|
|
187
|
+
if (media && /Failed to format input|Failed to tokenize prompt/i.test(message)) {
|
|
188
|
+
throw new Error(`MODEL_PROMPT_FORMAT_UNSUPPORTED: Wllama could not tokenize this model's ${media.modality} input. Use a GGUF and projector verified with the selected Wllama version.`);
|
|
189
|
+
}
|
|
190
|
+
throw error;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
inputTokens += response.usage?.prompt_tokens || 0;
|
|
194
|
+
const alternatives = response.choices?.[0]?.logprobs?.content?.[0]?.top_logprobs || [];
|
|
195
|
+
const logits = labels.map(label => {
|
|
196
|
+
const byte = label.charCodeAt(0);
|
|
197
|
+
const match = alternatives.find(item => item.token === label || (item.bytes?.length === 1 && item.bytes[0] === byte));
|
|
198
|
+
return Number(match?.logprob);
|
|
199
|
+
});
|
|
200
|
+
if (logits.some(value => !Number.isFinite(value))) {
|
|
201
|
+
throw new Error(`llama.cpp did not return a score for every allowed answer (${labels.join(', ')}). Try another GGUF model or runtime version.`);
|
|
202
|
+
}
|
|
203
|
+
scores.push({ id: task.id, logits });
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
return JSON.stringify({ model, scores, metrics: { input_tokens: inputTokens } });
|
|
207
|
+
},
|
|
208
|
+
|
|
209
|
+
async close() {
|
|
210
|
+
const loaded = engine;
|
|
211
|
+
engine = undefined;
|
|
212
|
+
if (loaded) await loaded.exit();
|
|
213
|
+
},
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function multimodalContent(media, criterion, optionsText, labels) {
|
|
218
|
+
const context = `State:\n${media.text || ''}\n\nEvidence:\n`;
|
|
219
|
+
const content = [];
|
|
220
|
+
if (media.modality === 'audio') {
|
|
221
|
+
// Wllama's own chat client places the media block before the user text.
|
|
222
|
+
// Some audio projectors only recognize the marker in that position.
|
|
223
|
+
content.push({ type: 'audio', data: decodeBase64(media.base64) });
|
|
224
|
+
content.push({ type: 'text', text: context });
|
|
225
|
+
} else {
|
|
226
|
+
content.push({ type: 'text', text: context });
|
|
227
|
+
}
|
|
228
|
+
if (media.modality === 'image') {
|
|
229
|
+
content.push({ type: 'image', data: decodeBase64(media.base64) });
|
|
230
|
+
} else if (media.modality === 'video') {
|
|
231
|
+
for (const frame of media.frames || []) {
|
|
232
|
+
content.push({ type: 'text', text: `[frame at ${frame.timestamp_ms} ms]\n` });
|
|
233
|
+
content.push({ type: 'image', data: decodeBase64(frame.base64) });
|
|
234
|
+
}
|
|
235
|
+
} else if (media.modality !== 'audio') {
|
|
236
|
+
throw new Error(`UNSUPPORTED_MODALITY: unknown media modality ${media.modality}.`);
|
|
237
|
+
}
|
|
238
|
+
content.push({
|
|
239
|
+
type: 'text',
|
|
240
|
+
text: `\nQuestion:\n${criterion}\n\nAllowed options:\n${optionsText}\n\nAnswer as {"answer": "<option letter>"}.`,
|
|
241
|
+
});
|
|
242
|
+
return content;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
function decodeBase64(value) {
|
|
246
|
+
if (typeof value !== 'string' || value.length === 0) {
|
|
247
|
+
throw new Error('INVALID_MEDIA: media payload is empty.');
|
|
248
|
+
}
|
|
249
|
+
let binary;
|
|
250
|
+
try {
|
|
251
|
+
binary = atob(value);
|
|
252
|
+
} catch {
|
|
253
|
+
throw new Error('INVALID_MEDIA: media payload is not valid base64.');
|
|
254
|
+
}
|
|
255
|
+
const bytes = new Uint8Array(binary.length);
|
|
256
|
+
for (let i = 0; i < binary.length; i++) bytes[i] = binary.charCodeAt(i);
|
|
257
|
+
return bytes.buffer;
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
function debugLog(enabled, message) {
|
|
261
|
+
if (enabled) console.debug(`[turbojev llama.cpp wasm] ${message}`);
|
|
262
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
// mtmd's mel preprocessors spawn three pthreads independently of n_threads.
|
|
2
|
+
// llama and the media encoder also retain CPU thread pools. Emscripten must
|
|
3
|
+
// prestart these workers: spawning one while a synchronous join is waiting can
|
|
4
|
+
// deadlock before the new worker receives its initialization message.
|
|
5
|
+
export function reserveMediaWorkers(source, inferenceThreads) {
|
|
6
|
+
const header = 'var Module=typeof Module!="undefined"?Module:{};';
|
|
7
|
+
if (!source.startsWith(header)) {
|
|
8
|
+
throw new Error('Unsupported Wllama worker script: cannot reserve multimodal workers safely.');
|
|
9
|
+
}
|
|
10
|
+
const poolSize = Math.max(4, 2 * inferenceThreads + 4);
|
|
11
|
+
const setup = `\nModule.pthreadPoolSize = Math.max(Module.pthreadPoolSize || 0, ${poolSize});
|
|
12
|
+
if (Module.mainScriptUrlOrBlob === 'throw new Error("Multithreading is not enabled")') {
|
|
13
|
+
Module.mainScriptUrlOrBlob = new Blob([${JSON.stringify(source)}], { type: 'text/javascript' });
|
|
14
|
+
}\n`;
|
|
15
|
+
// Keep the original declaration first: Wllama replaces that declaration when
|
|
16
|
+
// embedding the glue in its main worker. Pthreads load the unmodified glue.
|
|
17
|
+
return header + setup + source.slice(header.length);
|
|
18
|
+
}
|