@turbojev/runtime-llamacpp-wasm 0.29.2 → 0.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/index.js +58 -12
package/package.json
CHANGED
package/src/index.js
CHANGED
|
@@ -107,7 +107,7 @@ export async function createRuntime(dependencies = {}) {
|
|
|
107
107
|
return {
|
|
108
108
|
name: 'llama.cpp-wasm',
|
|
109
109
|
model,
|
|
110
|
-
codecProfile: 'llamacpp-wasm-json-prefill-
|
|
110
|
+
codecProfile: 'llamacpp-wasm-json-prefill-v3',
|
|
111
111
|
scoringMode: 'decisions',
|
|
112
112
|
modalities,
|
|
113
113
|
capabilities: {
|
|
@@ -130,9 +130,14 @@ export async function createRuntime(dependencies = {}) {
|
|
|
130
130
|
|
|
131
131
|
for (const task of batch.tasks) {
|
|
132
132
|
const candidates = task.candidates || [];
|
|
133
|
-
if (candidates.length < 2 || candidates.length >
|
|
134
|
-
throw new Error(`llama.cpp WASM supports 2–
|
|
133
|
+
if (candidates.length < 2 || candidates.length > 24) {
|
|
134
|
+
throw new Error(`llama.cpp WASM supports 2–24 choices per decision (received ${candidates.length}).`);
|
|
135
135
|
}
|
|
136
|
+
// Wllama's Jinja autoparser ignores custom GBNF for this path.
|
|
137
|
+
// Resolve textual labels through its actual model tokenizer and give
|
|
138
|
+
// every allowed token the SAME additive offset. Relative log-odds among
|
|
139
|
+
// candidates are unchanged; post-sampling top-N then includes them all.
|
|
140
|
+
// Never supply guessed token IDs or boost one answer over another.
|
|
136
141
|
const labels = Array.from({ length: candidates.length }, (_, i) => String.fromCharCode(65 + i));
|
|
137
142
|
const isBestMatch = String(task.intent).toLowerCase() === 'bestmatch';
|
|
138
143
|
const criterion = task.instruction || (isBestMatch ? 'Choose the option that best matches the state.' : 'Choose the correct option.');
|
|
@@ -152,7 +157,7 @@ export async function createRuntime(dependencies = {}) {
|
|
|
152
157
|
{ role: 'user', content: userContent },
|
|
153
158
|
{ role: 'assistant', content: '{"answer": "' },
|
|
154
159
|
];
|
|
155
|
-
|
|
160
|
+
|
|
156
161
|
if (media) {
|
|
157
162
|
const needsImage = media.modality === 'image' || media.modality === 'video';
|
|
158
163
|
if (needsImage && !engine.supportInputModality('image')) {
|
|
@@ -175,9 +180,12 @@ export async function createRuntime(dependencies = {}) {
|
|
|
175
180
|
temperature: 1,
|
|
176
181
|
top_k: 0,
|
|
177
182
|
top_p: 1,
|
|
183
|
+
min_p: 0,
|
|
184
|
+
post_sampling_probs: true,
|
|
178
185
|
logprobs: true,
|
|
179
|
-
top_logprobs:
|
|
180
|
-
|
|
186
|
+
top_logprobs: labels.length,
|
|
187
|
+
logit_bias: Object.fromEntries(labels.map(label => [label, 100])),
|
|
188
|
+
|
|
181
189
|
cache_prompt: false,
|
|
182
190
|
chat_template_kwargs: { enable_thinking: false },
|
|
183
191
|
});
|
|
@@ -191,12 +199,17 @@ export async function createRuntime(dependencies = {}) {
|
|
|
191
199
|
}
|
|
192
200
|
|
|
193
201
|
inputTokens += response.usage?.prompt_tokens || 0;
|
|
194
|
-
const
|
|
195
|
-
|
|
196
|
-
const
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
202
|
+
const project = response => {
|
|
203
|
+
const tokenScores = response.choices?.[0]?.logprobs?.content?.[0];
|
|
204
|
+
const alternatives = tokenScores?.top_probs || tokenScores?.top_logprobs || [];
|
|
205
|
+
return labels.map(label => {
|
|
206
|
+
const byte = label.charCodeAt(0);
|
|
207
|
+
const match = alternatives.find(item => item.token === label || (item.bytes?.length === 1 && item.bytes[0] === byte));
|
|
208
|
+
if (typeof match?.prob === 'number' && match.prob > 0) return Math.log(match.prob);
|
|
209
|
+
return typeof match?.logprob === 'number' ? match.logprob : NaN;
|
|
210
|
+
});
|
|
211
|
+
};
|
|
212
|
+
const logits = project(response);
|
|
200
213
|
if (logits.some(value => !Number.isFinite(value))) {
|
|
201
214
|
throw new Error(`llama.cpp did not return a score for every allowed answer (${labels.join(', ')}). Try another GGUF model or runtime version.`);
|
|
202
215
|
}
|
|
@@ -206,6 +219,39 @@ export async function createRuntime(dependencies = {}) {
|
|
|
206
219
|
return JSON.stringify({ model, scores, metrics: { input_tokens: inputTokens } });
|
|
207
220
|
},
|
|
208
221
|
|
|
222
|
+
async inferField(taskJson) {
|
|
223
|
+
if (!engine) throw new Error('Load a GGUF model first.');
|
|
224
|
+
const task = JSON.parse(taskJson);
|
|
225
|
+
if (task.thinking) throw new Error('THINKING_UNSUPPORTED: per-field reasoning requires the native GGUF runtime.');
|
|
226
|
+
if (task.values) {
|
|
227
|
+
if (task.values.length === 1) return JSON.stringify({ logits: [0], thinking: '', input_tokens: 0, generated_tokens: 0 });
|
|
228
|
+
const output = JSON.parse(await this.scoreDecisionBatch(JSON.stringify({
|
|
229
|
+
state: task.state,
|
|
230
|
+
tasks: [{ id: task.name, intent: 'criterion', instruction: task.instructions,
|
|
231
|
+
candidates: task.values.map(value => ({ key: JSON.stringify(value), description: JSON.stringify(value) })) }],
|
|
232
|
+
})));
|
|
233
|
+
return JSON.stringify({ logits: output.scores[0].logits, thinking: '', input_tokens: output.metrics.input_tokens, generated_tokens: 0 });
|
|
234
|
+
}
|
|
235
|
+
const schema = { type: 'object', properties: { value: task.output_schema }, required: ['value'], additionalProperties: false };
|
|
236
|
+
const question = `${task.instructions}\nReturn exactly one JSON object with key value. Schema: ${JSON.stringify(task.output_schema)}`;
|
|
237
|
+
const media = task.state?.contract === 'turbojev.media.v1' ? task.state : null;
|
|
238
|
+
const user = media ? multimodalContent(media, question, '', [])
|
|
239
|
+
: `State:\n${typeof task.state === 'string' ? task.state : JSON.stringify(task.state)}\n\nQuestion:\n${question}`;
|
|
240
|
+
// For media, replace the categorical suffix while retaining the same bytes.
|
|
241
|
+
if (media) user.at(-1).text = `\nQuestion:\n${question}`;
|
|
242
|
+
const response = await engine.createChatCompletion({
|
|
243
|
+
messages: [{ role: 'system', content: 'Answer the requested field from the supplied evidence. Return only the requested JSON object.' }, { role: 'user', content: user }],
|
|
244
|
+
max_tokens: task.max_tokens, temperature: 0, cache_prompt: false,
|
|
245
|
+
chat_template_kwargs: { enable_thinking: false },
|
|
246
|
+
response_format: { type: 'json_schema', json_schema: { name: 'field', schema, strict: true } },
|
|
247
|
+
});
|
|
248
|
+
const choice = response.choices?.[0];
|
|
249
|
+
if (choice?.finish_reason === 'length') throw new Error('GENERATION_BUDGET_EXHAUSTED: unfinished field value.');
|
|
250
|
+
const value = JSON.parse(choice?.message?.content ?? '');
|
|
251
|
+
if (Object.keys(value).length !== 1 || !Object.hasOwn(value, 'value')) throw new Error('Invalid field generation response.');
|
|
252
|
+
return JSON.stringify({ value: value.value, logits: [], thinking: '', input_tokens: response.usage?.prompt_tokens, generated_tokens: response.usage?.completion_tokens ?? 0 });
|
|
253
|
+
},
|
|
254
|
+
|
|
209
255
|
async close() {
|
|
210
256
|
const loaded = engine;
|
|
211
257
|
engine = undefined;
|