@turbojev/runtime-llamacpp-wasm 0.28.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +34 -0
  2. package/package.json +15 -0
  3. package/src/index.js +123 -0
package/README.md ADDED
@@ -0,0 +1,34 @@
1
+ # TurboJev llama.cpp WASM runtime
2
+
3
+ Browser decision adapter for GGUF text models supported by llama.cpp, using the
4
+ Wllama WebAssembly build. It implements TurboJev's `BrowserDecisionRuntime`
5
+ contract and returns one score per candidate. The Rust/WASM engine validates
6
+ the scores and creates the shared `EvaluationResponse`.
7
+
8
+ The adapter sends the model's chat-formatted request to llama.cpp, constrains
9
+ the next answer to `A`–`T` with grammar, and maps returned log-probabilities
10
+ back to the original candidate order. It does not contain model-specific token
11
+ IDs. One inference call is made per decision task.
12
+
13
+ ```js
14
+ import { TurboJevWeb } from '@turbojev/web';
15
+
16
+ const engine = await TurboJevWeb.load({
17
+ model: 'Qwen3 0.6B Q4_K_M',
18
+ runtimeModuleUrl: '/llamacpp-runtime.js',
19
+ wasmModuleUrl: '/turbojev_wasm.js',
20
+ runtimeOptions: {
21
+ modelUrl: 'https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/1208e45d782fe18602c5eaf10e5758d5b0f24c03/Qwen3-0.6B-Q4_K_M.gguf',
22
+ device: 'wasm', // use 'webgpu' to request GPU execution
23
+ },
24
+ });
25
+
26
+ const result = await engine.classify('My package arrived broken', [
27
+ 'billing', 'shipping', 'technical', 'other',
28
+ ]);
29
+ await engine.close();
30
+ ```
31
+
32
+ The host must serve this adapter, TurboJev's generated WASM files, and the page
33
+ over localhost or HTTPS. Model files are fetched from their URL and remain in
34
+ the browser. Wllama 3.6.1 and its WASM binary are loaded from jsDelivr.
package/package.json ADDED
@@ -0,0 +1,15 @@
1
+ {
2
+ "name": "@turbojev/runtime-llamacpp-wasm",
3
+ "version": "0.28.6",
4
+ "description": "Browser-side llama.cpp WASM decision runtime for TurboJev",
5
+ "license": "Apache-2.0",
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "https://github.com/DestroyerDarkNess/TurboJev.git",
9
+ "directory": "bindings/web-runtimes/llamacpp-wasm"
10
+ },
11
+ "publishConfig": { "access": "public" },
12
+ "type": "module",
13
+ "exports": "./src/index.js",
14
+ "files": ["src", "README.md"]
15
+ }
package/src/index.js ADDED
@@ -0,0 +1,123 @@
1
+ const WLLAMA_VERSION = '3.6.1';
2
+ const WLLAMA_ENTRY = `https://cdn.jsdelivr.net/npm/@wllama/wllama@${WLLAMA_VERSION}/esm/index.js`;
3
+ const WLLAMA_WASM = `https://cdn.jsdelivr.net/npm/@wllama/wllama@${WLLAMA_VERSION}/esm/wasm/wllama.wasm`;
4
+
5
+ export async function createRuntime() {
6
+ let engine;
7
+ let model = '';
8
+
9
+ return {
10
+ async initialize(options) {
11
+ if (options.modality !== 'text') {
12
+ throw new Error('llama.cpp WASM currently supports text GGUF models only.');
13
+ }
14
+ if (!options.modelUrl) {
15
+ throw new Error('Choose a GGUF model with a direct download URL.');
16
+ }
17
+
18
+ const { Wllama, LoggerWithoutDebug } = await import(WLLAMA_ENTRY);
19
+ engine = new Wllama({ default: WLLAMA_WASM }, {
20
+ logger: LoggerWithoutDebug,
21
+ suppressNativeLog: true,
22
+ parallelDownloads: 4,
23
+ });
24
+ model = options.modelLabel || options.modelUrl;
25
+
26
+ await engine.loadModelFromUrl(options.modelUrl, {
27
+ n_ctx: 2048,
28
+ n_batch: 512,
29
+ n_gpu_layers: options.device === 'webgpu' ? 999 : 0,
30
+ cache_prompt: false,
31
+ progressCallback: ({ loaded = 0, total = 0 }) => {
32
+ options.progressCallback?.({
33
+ status: 'progress',
34
+ file: 'GGUF model',
35
+ loaded,
36
+ total,
37
+ progress: total > 0 ? Math.min(100, loaded / total * 100) : 0,
38
+ });
39
+ },
40
+ });
41
+
42
+ const warmup = await engine.createChatCompletion({
43
+ messages: [{ role: 'user', content: 'Reply with one word: ready.' }],
44
+ max_tokens: 1,
45
+ temperature: 0,
46
+ cache_prompt: false,
47
+ chat_template_kwargs: { enable_thinking: false },
48
+ });
49
+ if (!warmup?.choices?.length) throw new Error('llama.cpp WASM loaded the model but warmup returned no completion.');
50
+
51
+ return {
52
+ name: 'llama.cpp-wasm',
53
+ model,
54
+ codecProfile: 'llamacpp-wasm-constrained-choice-v1',
55
+ scoringMode: 'decisions',
56
+ modalities: ['text'],
57
+ capabilities: {
58
+ nextTokenLogits: false,
59
+ sequenceScores: false,
60
+ sharedPrefix: false,
61
+ parallelSuffixes: false,
62
+ selectiveOutputHead: false,
63
+ localInference: true,
64
+ },
65
+ };
66
+ },
67
+
68
+ async scoreDecisionBatch(batchJson) {
69
+ if (!engine) throw new Error('Load a GGUF model first.');
70
+ const batch = JSON.parse(batchJson);
71
+ const scores = [];
72
+ let inputTokens = 0;
73
+
74
+ for (const task of batch.tasks) {
75
+ const candidates = task.candidates || [];
76
+ if (candidates.length < 2 || candidates.length > 20) {
77
+ throw new Error(`llama.cpp WASM supports 2–20 choices per decision (received ${candidates.length}).`);
78
+ }
79
+ const labels = Array.from({ length: candidates.length }, (_, i) => String.fromCharCode(65 + i));
80
+ const isBestMatch = String(task.intent).toLowerCase() === 'bestmatch';
81
+ const criterion = task.instruction || (isBestMatch ? 'Choose the option that best matches the state.' : 'Choose the correct option.');
82
+ const optionsText = candidates.map((candidate, i) => `${labels[i]}. ${candidate.description || candidate.key}`).join('\n');
83
+ const messages = [
84
+ { role: 'system', content: 'Make the requested decision from the supplied state. Follow the output format exactly.' },
85
+ { role: 'user', content: `State:\n${typeof batch.state === 'string' ? batch.state : JSON.stringify(batch.state)}\n\nQuestion:\n${criterion}\n\nAllowed options:\n${optionsText}\n\nReply with exactly one option letter: ${labels.join(', ')}.` },
86
+ ];
87
+ const grammar = `root ::= ${labels.map(label => `"${label}"`).join(' | ')}`;
88
+ const response = await engine.createChatCompletion({
89
+ messages,
90
+ max_tokens: 1,
91
+ temperature: 1,
92
+ top_k: 0,
93
+ top_p: 1,
94
+ logprobs: true,
95
+ top_logprobs: 20,
96
+ grammar,
97
+ cache_prompt: false,
98
+ chat_template_kwargs: { enable_thinking: false },
99
+ });
100
+
101
+ inputTokens += response.usage?.prompt_tokens || 0;
102
+ const alternatives = response.choices?.[0]?.logprobs?.content?.[0]?.top_logprobs || [];
103
+ const logits = labels.map(label => {
104
+ const byte = label.charCodeAt(0);
105
+ const match = alternatives.find(item => item.token === label || (item.bytes?.length === 1 && item.bytes[0] === byte));
106
+ return Number(match?.logprob);
107
+ });
108
+ if (logits.some(value => !Number.isFinite(value))) {
109
+ throw new Error(`llama.cpp did not return a score for every allowed answer (${labels.join(', ')}). Try another GGUF model or runtime version.`);
110
+ }
111
+ scores.push({ id: task.id, logits });
112
+ }
113
+
114
+ return JSON.stringify({ model, scores, metrics: { input_tokens: inputTokens } });
115
+ },
116
+
117
+ async close() {
118
+ const loaded = engine;
119
+ engine = undefined;
120
+ if (loaded) await loaded.exit();
121
+ },
122
+ };
123
+ }