@turbojev/runtime-llamacpp-wasm 0.29.2 → 0.30.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/index.js +58 -12
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@turbojev/runtime-llamacpp-wasm",
3
- "version": "0.29.2",
3
+ "version": "0.30.1",
4
4
  "description": "Browser-side llama.cpp WASM decision runtime for TurboJev",
5
5
  "license": "Apache-2.0",
6
6
  "repository": {
package/src/index.js CHANGED
@@ -107,7 +107,7 @@ export async function createRuntime(dependencies = {}) {
107
107
  return {
108
108
  name: 'llama.cpp-wasm',
109
109
  model,
110
- codecProfile: 'llamacpp-wasm-json-prefill-v2',
110
+ codecProfile: 'llamacpp-wasm-json-prefill-v3',
111
111
  scoringMode: 'decisions',
112
112
  modalities,
113
113
  capabilities: {
@@ -130,9 +130,14 @@ export async function createRuntime(dependencies = {}) {
130
130
 
131
131
  for (const task of batch.tasks) {
132
132
  const candidates = task.candidates || [];
133
- if (candidates.length < 2 || candidates.length > 20) {
134
- throw new Error(`llama.cpp WASM supports 2–20 choices per decision (received ${candidates.length}).`);
133
+ if (candidates.length < 2 || candidates.length > 24) {
134
+ throw new Error(`llama.cpp WASM supports 2–24 choices per decision (received ${candidates.length}).`);
135
135
  }
136
+ // Wllama's Jinja autoparser ignores custom GBNF for this path.
137
+ // Resolve textual labels through its actual model tokenizer and give
138
+ // every allowed token the SAME additive offset. Relative log-odds among
139
+ // candidates are unchanged; post-sampling top-N then includes them all.
140
+ // Never supply guessed token IDs or boost one answer over another.
136
141
  const labels = Array.from({ length: candidates.length }, (_, i) => String.fromCharCode(65 + i));
137
142
  const isBestMatch = String(task.intent).toLowerCase() === 'bestmatch';
138
143
  const criterion = task.instruction || (isBestMatch ? 'Choose the option that best matches the state.' : 'Choose the correct option.');
@@ -152,7 +157,7 @@ export async function createRuntime(dependencies = {}) {
152
157
  { role: 'user', content: userContent },
153
158
  { role: 'assistant', content: '{"answer": "' },
154
159
  ];
155
- const grammar = `root ::= ${labels.map(label => `"${label}"`).join(' | ')}`;
160
+
156
161
  if (media) {
157
162
  const needsImage = media.modality === 'image' || media.modality === 'video';
158
163
  if (needsImage && !engine.supportInputModality('image')) {
@@ -175,9 +180,12 @@ export async function createRuntime(dependencies = {}) {
175
180
  temperature: 1,
176
181
  top_k: 0,
177
182
  top_p: 1,
183
+ min_p: 0,
184
+ post_sampling_probs: true,
178
185
  logprobs: true,
179
- top_logprobs: 20,
180
- grammar,
186
+ top_logprobs: labels.length,
187
+ logit_bias: Object.fromEntries(labels.map(label => [label, 100])),
188
+
181
189
  cache_prompt: false,
182
190
  chat_template_kwargs: { enable_thinking: false },
183
191
  });
@@ -191,12 +199,17 @@ export async function createRuntime(dependencies = {}) {
191
199
  }
192
200
 
193
201
  inputTokens += response.usage?.prompt_tokens || 0;
194
- const alternatives = response.choices?.[0]?.logprobs?.content?.[0]?.top_logprobs || [];
195
- const logits = labels.map(label => {
196
- const byte = label.charCodeAt(0);
197
- const match = alternatives.find(item => item.token === label || (item.bytes?.length === 1 && item.bytes[0] === byte));
198
- return Number(match?.logprob);
199
- });
202
+ const project = response => {
203
+ const tokenScores = response.choices?.[0]?.logprobs?.content?.[0];
204
+ const alternatives = tokenScores?.top_probs || tokenScores?.top_logprobs || [];
205
+ return labels.map(label => {
206
+ const byte = label.charCodeAt(0);
207
+ const match = alternatives.find(item => item.token === label || (item.bytes?.length === 1 && item.bytes[0] === byte));
208
+ if (typeof match?.prob === 'number' && match.prob > 0) return Math.log(match.prob);
209
+ return typeof match?.logprob === 'number' ? match.logprob : NaN;
210
+ });
211
+ };
212
+ const logits = project(response);
200
213
  if (logits.some(value => !Number.isFinite(value))) {
201
214
  throw new Error(`llama.cpp did not return a score for every allowed answer (${labels.join(', ')}). Try another GGUF model or runtime version.`);
202
215
  }
@@ -206,6 +219,39 @@ export async function createRuntime(dependencies = {}) {
206
219
  return JSON.stringify({ model, scores, metrics: { input_tokens: inputTokens } });
207
220
  },
208
221
 
222
+ async inferField(taskJson) {
223
+ if (!engine) throw new Error('Load a GGUF model first.');
224
+ const task = JSON.parse(taskJson);
225
+ if (task.thinking) throw new Error('THINKING_UNSUPPORTED: per-field reasoning requires the native GGUF runtime.');
226
+ if (task.values) {
227
+ if (task.values.length === 1) return JSON.stringify({ logits: [0], thinking: '', input_tokens: 0, generated_tokens: 0 });
228
+ const output = JSON.parse(await this.scoreDecisionBatch(JSON.stringify({
229
+ state: task.state,
230
+ tasks: [{ id: task.name, intent: 'criterion', instruction: task.instructions,
231
+ candidates: task.values.map(value => ({ key: JSON.stringify(value), description: JSON.stringify(value) })) }],
232
+ })));
233
+ return JSON.stringify({ logits: output.scores[0].logits, thinking: '', input_tokens: output.metrics.input_tokens, generated_tokens: 0 });
234
+ }
235
+ const schema = { type: 'object', properties: { value: task.output_schema }, required: ['value'], additionalProperties: false };
236
+ const question = `${task.instructions}\nReturn exactly one JSON object with key value. Schema: ${JSON.stringify(task.output_schema)}`;
237
+ const media = task.state?.contract === 'turbojev.media.v1' ? task.state : null;
238
+ const user = media ? multimodalContent(media, question, '', [])
239
+ : `State:\n${typeof task.state === 'string' ? task.state : JSON.stringify(task.state)}\n\nQuestion:\n${question}`;
240
+ // For media, replace the categorical suffix while retaining the same bytes.
241
+ if (media) user.at(-1).text = `\nQuestion:\n${question}`;
242
+ const response = await engine.createChatCompletion({
243
+ messages: [{ role: 'system', content: 'Answer the requested field from the supplied evidence. Return only the requested JSON object.' }, { role: 'user', content: user }],
244
+ max_tokens: task.max_tokens, temperature: 0, cache_prompt: false,
245
+ chat_template_kwargs: { enable_thinking: false },
246
+ response_format: { type: 'json_schema', json_schema: { name: 'field', schema, strict: true } },
247
+ });
248
+ const choice = response.choices?.[0];
249
+ if (choice?.finish_reason === 'length') throw new Error('GENERATION_BUDGET_EXHAUSTED: unfinished field value.');
250
+ const value = JSON.parse(choice?.message?.content ?? '');
251
+ if (Object.keys(value).length !== 1 || !Object.hasOwn(value, 'value')) throw new Error('Invalid field generation response.');
252
+ return JSON.stringify({ value: value.value, logits: [], thinking: '', input_tokens: response.usage?.prompt_tokens, generated_tokens: response.usage?.completion_tokens ?? 0 });
253
+ },
254
+
209
255
  async close() {
210
256
  const loaded = engine;
211
257
  engine = undefined;