@kaleidorg/mind 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +96 -8
- package/dist/context/compress.d.ts.map +1 -1
- package/dist/context/compress.js +1 -0
- package/dist/context/compress.js.map +1 -1
- package/dist/context/rgb-units.d.ts +18 -0
- package/dist/context/rgb-units.d.ts.map +1 -0
- package/dist/context/rgb-units.js +86 -0
- package/dist/context/rgb-units.js.map +1 -0
- package/dist/engine.d.ts +53 -1
- package/dist/engine.d.ts.map +1 -1
- package/dist/engine.js +179 -17
- package/dist/engine.js.map +1 -1
- package/dist/funnel.d.ts.map +1 -1
- package/dist/funnel.js +23 -4
- package/dist/funnel.js.map +1 -1
- package/dist/guards.d.ts +63 -0
- package/dist/guards.d.ts.map +1 -0
- package/dist/guards.js +284 -0
- package/dist/guards.js.map +1 -0
- package/dist/index.d.ts +9 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -0
- package/dist/index.js.map +1 -1
- package/dist/kaleidoswap/contract.d.ts +3 -4
- package/dist/kaleidoswap/contract.d.ts.map +1 -1
- package/dist/kaleidoswap/contract.js +3 -17
- package/dist/kaleidoswap/contract.js.map +1 -1
- package/dist/providers/openai.d.ts +64 -0
- package/dist/providers/openai.d.ts.map +1 -0
- package/dist/providers/openai.js +233 -0
- package/dist/providers/openai.js.map +1 -0
- package/dist/providers/types.d.ts +20 -0
- package/dist/providers/types.d.ts.map +1 -1
- package/dist/qvac/config.d.ts +1 -1
- package/dist/qvac/config.js +1 -1
- package/dist/qvac/index.d.ts +1 -0
- package/dist/qvac/index.d.ts.map +1 -1
- package/dist/qvac/index.js +1 -0
- package/dist/qvac/index.js.map +1 -1
- package/dist/qvac/models.d.ts +31 -0
- package/dist/qvac/models.d.ts.map +1 -0
- package/dist/qvac/models.js +80 -0
- package/dist/qvac/models.js.map +1 -0
- package/dist/qvac/parse.d.ts +11 -0
- package/dist/qvac/parse.d.ts.map +1 -1
- package/dist/qvac/parse.js +29 -7
- package/dist/qvac/parse.js.map +1 -1
- package/dist/qvac/provider.d.ts +5 -4
- package/dist/qvac/provider.d.ts.map +1 -1
- package/dist/qvac/provider.js +22 -8
- package/dist/qvac/provider.js.map +1 -1
- package/dist/qvac/stream.d.ts +3 -5
- package/dist/qvac/stream.d.ts.map +1 -1
- package/dist/qvac/stream.js.map +1 -1
- package/dist/qvac/tools.d.ts +5 -0
- package/dist/qvac/tools.d.ts.map +1 -1
- package/dist/qvac/tools.js +10 -0
- package/dist/qvac/tools.js.map +1 -1
- package/dist/recipe/issue-asset.d.ts.map +1 -1
- package/dist/recipe/issue-asset.js +42 -16
- package/dist/recipe/issue-asset.js.map +1 -1
- package/dist/recipe/runner.d.ts.map +1 -1
- package/dist/recipe/runner.js +1 -0
- package/dist/recipe/runner.js.map +1 -1
- package/dist/recipe/submarine-pay.d.ts +17 -0
- package/dist/recipe/submarine-pay.d.ts.map +1 -0
- package/dist/recipe/submarine-pay.js +79 -0
- package/dist/recipe/submarine-pay.js.map +1 -0
- package/dist/skills/registry.js +1 -1
- package/dist/skills/select.d.ts +20 -0
- package/dist/skills/select.d.ts.map +1 -0
- package/dist/skills/select.js +32 -0
- package/dist/skills/select.js.map +1 -0
- package/dist/submarine/contract.d.ts +42 -0
- package/dist/submarine/contract.d.ts.map +1 -0
- package/dist/submarine/contract.js +73 -0
- package/dist/submarine/contract.js.map +1 -0
- package/dist/testing/mock-wallet.d.ts +24 -0
- package/dist/testing/mock-wallet.d.ts.map +1 -1
- package/dist/testing/mock-wallet.js +75 -6
- package/dist/testing/mock-wallet.js.map +1 -1
- package/dist/wallet/confirm.d.ts.map +1 -1
- package/dist/wallet/confirm.js +27 -9
- package/dist/wallet/confirm.js.map +1 -1
- package/package.json +6 -1
- package/skills/README.md +1 -0
- package/skills/kaleido-trading/SKILL.md +22 -28
- package/skills/kaleido-trading/references/api.md +3 -3
- package/skills/rgb-lightning-node/SKILL.md +32 -12
- package/skills/spark-wallet/SKILL.md +1 -0
- package/skills/submarine-swaps/SKILL.md +48 -0
- package/src/context/compress.ts +1 -0
- package/src/context/rgb-units.test.ts +42 -0
- package/src/context/rgb-units.ts +89 -0
- package/src/engine.test.ts +94 -3
- package/src/engine.ts +234 -19
- package/src/funnel.mind.test.ts +4 -2
- package/src/funnel.ts +23 -4
- package/src/guards.test.ts +399 -0
- package/src/guards.ts +299 -0
- package/src/index.ts +37 -1
- package/src/kaleidoswap/contract.test.ts +8 -16
- package/src/kaleidoswap/contract.ts +4 -32
- package/src/providers/openai.test.ts +127 -0
- package/src/providers/openai.ts +282 -0
- package/src/providers/types.ts +22 -0
- package/src/qvac/config.ts +1 -1
- package/src/qvac/index.ts +10 -0
- package/src/qvac/models.ts +98 -0
- package/src/qvac/parse.test.ts +7 -0
- package/src/qvac/parse.ts +33 -1
- package/src/qvac/provider.test.ts +34 -1
- package/src/qvac/provider.ts +30 -13
- package/src/qvac/stream.ts +3 -5
- package/src/qvac/tools.ts +10 -0
- package/src/recipe/issue-asset.test.ts +39 -0
- package/src/recipe/issue-asset.ts +41 -13
- package/src/recipe/runner.ts +1 -0
- package/src/recipe/submarine-pay.test.ts +87 -0
- package/src/recipe/submarine-pay.ts +78 -0
- package/src/skills/registry.ts +1 -1
- package/src/skills/select.ts +41 -0
- package/src/submarine/contract.ts +112 -0
- package/src/testing/mock-wallet.ts +73 -6
- package/src/wallet/confirm.ts +27 -9
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* createOpenAICompatibleProvider — an LLMProvider over any server that speaks
|
|
3
|
+
* the OpenAI Chat Completions API with tool calling: Ollama, llama.cpp
|
|
4
|
+
* `llama-server`, LM Studio, vLLM, `qvac serve`, or a hosted API.
|
|
5
|
+
*
|
|
6
|
+
* No dependencies: it uses `fetch` (pass your own for older runtimes or tests).
|
|
7
|
+
*
|
|
8
|
+
* The Engine keeps history as plain `{ role, content }` messages. Tool calls
|
|
9
|
+
* are written into the assistant's `rawContent` as `<tool_call>{json}</tool_call>`
|
|
10
|
+
* blocks and turned back into `tool_calls` / `tool_call_id` pairs on the next
|
|
11
|
+
* request, so the server always receives a valid tool-calling conversation.
|
|
12
|
+
*/
|
|
13
|
+
import type { Message, ToolCall } from '../types.js';
|
|
14
|
+
import type { InferenceMetrics, LLMProvider, ToolCallError, ToolChoice, TurnInput, TurnOutput } from './types.js';
|
|
15
|
+
import { extractTextToolCalls } from '../qvac/parse.js';
|
|
16
|
+
import { toolParametersSchema } from '../qvac/tools.js';
|
|
17
|
+
|
|
18
|
+
export interface OpenAICompatibleOptions {
|
|
19
|
+
/** API root including the version, e.g. `http://localhost:11434/v1`. */
|
|
20
|
+
baseUrl: string;
|
|
21
|
+
/** Model name as the server knows it, e.g. `qwen3.5:4b`. */
|
|
22
|
+
model: string;
|
|
23
|
+
/** Sent as `Authorization: Bearer …` when set. */
|
|
24
|
+
apiKey?: string;
|
|
25
|
+
defaultTemperature?: number;
|
|
26
|
+
defaultMaxTokens?: number;
|
|
27
|
+
/** Stream tokens (default true). Set false for servers without SSE. */
|
|
28
|
+
stream?: boolean;
|
|
29
|
+
/** Extra JSON merged into every request body (e.g. `{ reasoning_effort: 'low' }`). */
|
|
30
|
+
extraBody?: Record<string, unknown>;
|
|
31
|
+
/** Extra request headers. */
|
|
32
|
+
headers?: Record<string, string>;
|
|
33
|
+
/** Reasoning deltas (`reasoning_content` / `reasoning`), when the server sends them. */
|
|
34
|
+
onThinking?: (token: string) => void;
|
|
35
|
+
fetch?: typeof fetch;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface OpenAITurnInput extends TurnInput {
|
|
39
|
+
temperature?: number;
|
|
40
|
+
maxTokens?: number;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
type WireMessage =
|
|
44
|
+
| { role: 'system' | 'user'; content: string }
|
|
45
|
+
| { role: 'assistant'; content: string | null; tool_calls?: WireToolCall[] }
|
|
46
|
+
| { role: 'tool'; content: string; tool_call_id: string };
|
|
47
|
+
|
|
48
|
+
interface WireToolCall {
|
|
49
|
+
id: string;
|
|
50
|
+
type: 'function';
|
|
51
|
+
function: { name: string; arguments: string };
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const TOOL_CALL_BLOCK = /<tool_call\b[^>]*>[\s\S]*?<\/tool_call>/gi;
|
|
55
|
+
|
|
56
|
+
/** The assistant frame the Engine stores: visible text plus one block per call. */
|
|
57
|
+
export function encodeToolCalls(text: string, calls: ToolCall[]): string {
|
|
58
|
+
const blocks = calls.map((c) => `<tool_call>${JSON.stringify({ name: c.name, arguments: c.arguments })}</tool_call>`);
|
|
59
|
+
return [text.trim(), ...blocks].filter(Boolean).join('\n');
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Engine history → Chat Completions messages, with ids pairing calls and results. */
|
|
63
|
+
export function toWireMessages(messages: Message[], system?: string): WireMessage[] {
|
|
64
|
+
const out: WireMessage[] = system ? [{ role: 'system', content: system }] : [];
|
|
65
|
+
let pending: string[] = [];
|
|
66
|
+
messages.forEach((m, i) => {
|
|
67
|
+
if (m.role === 'assistant') {
|
|
68
|
+
const calls = extractTextToolCalls(m.content);
|
|
69
|
+
const text = m.content.replace(TOOL_CALL_BLOCK, '').trim();
|
|
70
|
+
pending = calls.map((_, j) => `call_${i}_${j}`);
|
|
71
|
+
out.push(
|
|
72
|
+
calls.length
|
|
73
|
+
? {
|
|
74
|
+
role: 'assistant',
|
|
75
|
+
content: text || null,
|
|
76
|
+
tool_calls: calls.map((c, j) => ({
|
|
77
|
+
id: pending[j]!,
|
|
78
|
+
type: 'function',
|
|
79
|
+
function: { name: c.name, arguments: JSON.stringify(c.arguments) },
|
|
80
|
+
})),
|
|
81
|
+
}
|
|
82
|
+
: { role: 'assistant', content: m.content },
|
|
83
|
+
);
|
|
84
|
+
} else if (m.role === 'tool') {
|
|
85
|
+
const id = pending.shift();
|
|
86
|
+
// A tool message with no call to answer (e.g. a parse-error note) would
|
|
87
|
+
// be rejected by the server; send it as user context instead.
|
|
88
|
+
out.push(id ? { role: 'tool', content: m.content, tool_call_id: id } : { role: 'user', content: `Tool result: ${m.content}` });
|
|
89
|
+
} else {
|
|
90
|
+
pending = [];
|
|
91
|
+
out.push({ role: m.role as 'system' | 'user', content: m.content });
|
|
92
|
+
}
|
|
93
|
+
});
|
|
94
|
+
return out;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function wireToolChoice(choice: ToolChoice | undefined): unknown {
|
|
98
|
+
if (!choice) return undefined;
|
|
99
|
+
if (choice === 'auto' || choice === 'none' || choice === 'required') return choice;
|
|
100
|
+
return { type: 'function', function: { name: choice } };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
interface Accumulated {
|
|
104
|
+
content: string;
|
|
105
|
+
calls: Map<number, { id?: string; name: string; args: string }>;
|
|
106
|
+
finishReason?: string;
|
|
107
|
+
usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function absorbDelta(acc: Accumulated, delta: any, onToken?: (t: string) => void, onThinking?: (t: string) => void) {
|
|
111
|
+
if (typeof delta?.content === 'string' && delta.content) {
|
|
112
|
+
acc.content += delta.content;
|
|
113
|
+
onToken?.(delta.content);
|
|
114
|
+
}
|
|
115
|
+
const reasoning = delta?.reasoning_content ?? delta?.reasoning;
|
|
116
|
+
if (typeof reasoning === 'string' && reasoning) onThinking?.(reasoning);
|
|
117
|
+
for (const tc of delta?.tool_calls ?? []) {
|
|
118
|
+
const index = typeof tc.index === 'number' ? tc.index : acc.calls.size;
|
|
119
|
+
const cur = acc.calls.get(index) ?? { name: '', args: '' };
|
|
120
|
+
if (tc.id) cur.id = tc.id;
|
|
121
|
+
if (tc.function?.name) cur.name += tc.function.name;
|
|
122
|
+
if (tc.function?.arguments) {
|
|
123
|
+
cur.args += typeof tc.function.arguments === 'string' ? tc.function.arguments : JSON.stringify(tc.function.arguments);
|
|
124
|
+
}
|
|
125
|
+
acc.calls.set(index, cur);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
async function readSse(body: ReadableStream<Uint8Array>, onEvent: (data: string) => void): Promise<void> {
|
|
130
|
+
const reader = body.getReader();
|
|
131
|
+
const decoder = new TextDecoder();
|
|
132
|
+
let buffer = '';
|
|
133
|
+
for (;;) {
|
|
134
|
+
const { value, done } = await reader.read();
|
|
135
|
+
if (done) break;
|
|
136
|
+
buffer += decoder.decode(value, { stream: true });
|
|
137
|
+
let nl: number;
|
|
138
|
+
while ((nl = buffer.indexOf('\n')) >= 0) {
|
|
139
|
+
const line = buffer.slice(0, nl).trim();
|
|
140
|
+
buffer = buffer.slice(nl + 1);
|
|
141
|
+
if (line.startsWith('data:')) onEvent(line.slice(5).trim());
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
const last = buffer.trim();
|
|
145
|
+
if (last.startsWith('data:')) onEvent(last.slice(5).trim());
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function createOpenAICompatibleProvider(options: OpenAICompatibleOptions): LLMProvider {
|
|
149
|
+
const doFetch = options.fetch ?? globalThis.fetch;
|
|
150
|
+
if (!doFetch) throw new Error('No fetch available; pass options.fetch');
|
|
151
|
+
const url = `${options.baseUrl.replace(/\/+$/, '')}/chat/completions`;
|
|
152
|
+
const inflight = new Map<string, AbortController>();
|
|
153
|
+
let seq = 0;
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
name: 'openai-compatible',
|
|
157
|
+
|
|
158
|
+
async runTurn(input: OpenAITurnInput): Promise<TurnOutput> {
|
|
159
|
+
const requestId = `oai-${Date.now().toString(36)}-${++seq}`;
|
|
160
|
+
const controller = new AbortController();
|
|
161
|
+
inflight.set(requestId, controller);
|
|
162
|
+
const onAbort = () => controller.abort();
|
|
163
|
+
input.signal?.addEventListener('abort', onAbort, { once: true });
|
|
164
|
+
if (input.signal?.aborted) controller.abort();
|
|
165
|
+
|
|
166
|
+
const stream = options.stream ?? true;
|
|
167
|
+
const temperature = input.temperature ?? options.defaultTemperature;
|
|
168
|
+
const maxTokens = input.maxTokens ?? options.defaultMaxTokens;
|
|
169
|
+
const tools = input.tools.map((t) => ({
|
|
170
|
+
type: 'function' as const,
|
|
171
|
+
function: { name: t.name, description: t.description ?? '', parameters: toolParametersSchema(t) },
|
|
172
|
+
}));
|
|
173
|
+
const toolChoice = tools.length ? wireToolChoice(input.toolChoice) : undefined;
|
|
174
|
+
const body = {
|
|
175
|
+
model: options.model,
|
|
176
|
+
messages: toWireMessages(input.messages, input.system),
|
|
177
|
+
stream,
|
|
178
|
+
...(stream ? { stream_options: { include_usage: true } } : {}),
|
|
179
|
+
...(temperature !== undefined ? { temperature } : {}),
|
|
180
|
+
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
181
|
+
...(tools.length ? { tools } : {}),
|
|
182
|
+
...(toolChoice ? { tool_choice: toolChoice } : {}),
|
|
183
|
+
...options.extraBody,
|
|
184
|
+
};
|
|
185
|
+
|
|
186
|
+
const startedAt = Date.now();
|
|
187
|
+
let firstTokenAt: number | undefined;
|
|
188
|
+
const mark = (f?: (t: string) => void) => (t: string) => {
|
|
189
|
+
firstTokenAt ??= Date.now();
|
|
190
|
+
f?.(t);
|
|
191
|
+
};
|
|
192
|
+
const acc: Accumulated = { content: '', calls: new Map() };
|
|
193
|
+
let cancelled = false;
|
|
194
|
+
try {
|
|
195
|
+
const res = await doFetch(url, {
|
|
196
|
+
method: 'POST',
|
|
197
|
+
headers: {
|
|
198
|
+
'content-type': 'application/json',
|
|
199
|
+
...(options.apiKey ? { authorization: `Bearer ${options.apiKey}` } : {}),
|
|
200
|
+
...options.headers,
|
|
201
|
+
},
|
|
202
|
+
body: JSON.stringify(body),
|
|
203
|
+
signal: controller.signal,
|
|
204
|
+
});
|
|
205
|
+
if (!res.ok) {
|
|
206
|
+
const detail = await res.text().catch(() => '');
|
|
207
|
+
throw new Error(`${options.baseUrl} returned ${res.status}: ${detail.slice(0, 300)}`);
|
|
208
|
+
}
|
|
209
|
+
if (stream && res.body) {
|
|
210
|
+
await readSse(res.body, (data) => {
|
|
211
|
+
if (data === '[DONE]') return;
|
|
212
|
+
let chunk: any;
|
|
213
|
+
try {
|
|
214
|
+
chunk = JSON.parse(data);
|
|
215
|
+
} catch {
|
|
216
|
+
return;
|
|
217
|
+
}
|
|
218
|
+
if (chunk.usage) acc.usage = chunk.usage;
|
|
219
|
+
const choice = chunk.choices?.[0];
|
|
220
|
+
if (!choice) return;
|
|
221
|
+
absorbDelta(acc, choice.delta, mark(input.onToken), mark(options.onThinking));
|
|
222
|
+
if (choice.finish_reason) acc.finishReason = choice.finish_reason;
|
|
223
|
+
});
|
|
224
|
+
} else {
|
|
225
|
+
const json: any = await res.json();
|
|
226
|
+
const choice = json.choices?.[0];
|
|
227
|
+
absorbDelta(acc, { ...choice?.message, tool_calls: choice?.message?.tool_calls?.map((c: any, index: number) => ({ index, ...c })) }, input.onToken, options.onThinking);
|
|
228
|
+
acc.finishReason = choice?.finish_reason;
|
|
229
|
+
acc.usage = json.usage;
|
|
230
|
+
}
|
|
231
|
+
} catch (err) {
|
|
232
|
+
if (!controller.signal.aborted) throw err;
|
|
233
|
+
cancelled = true;
|
|
234
|
+
} finally {
|
|
235
|
+
inflight.delete(requestId);
|
|
236
|
+
input.signal?.removeEventListener('abort', onAbort);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
const toolCalls: ToolCall[] = [];
|
|
240
|
+
const toolErrors: ToolCallError[] = [];
|
|
241
|
+
for (const c of [...acc.calls.entries()].sort(([a], [b]) => a - b).map(([, v]) => v)) {
|
|
242
|
+
if (!c.name) continue;
|
|
243
|
+
try {
|
|
244
|
+
const args = c.args.trim() ? JSON.parse(c.args) : {};
|
|
245
|
+
toolCalls.push({ ...(c.id ? { id: c.id } : {}), name: c.name, arguments: args && typeof args === 'object' ? args : {} });
|
|
246
|
+
} catch {
|
|
247
|
+
toolErrors.push({ code: 'PARSE_ERROR', message: `arguments for ${c.name} are not valid JSON`, raw: c.args });
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
// Models that ignore the tools API sometimes write the call as text.
|
|
251
|
+
if (!toolCalls.length && !toolErrors.length) {
|
|
252
|
+
for (const c of extractTextToolCalls(acc.content)) toolCalls.push(c);
|
|
253
|
+
}
|
|
254
|
+
const text = acc.content.replace(TOOL_CALL_BLOCK, '').trim();
|
|
255
|
+
|
|
256
|
+
const truncated = acc.finishReason === 'length';
|
|
257
|
+
const inference: InferenceMetrics = {
|
|
258
|
+
requestId,
|
|
259
|
+
durationMs: Date.now() - startedAt,
|
|
260
|
+
status: cancelled ? 'cancelled' : truncated ? 'truncated' : 'completed',
|
|
261
|
+
...(firstTokenAt !== undefined ? { ttftMs: firstTokenAt - startedAt } : {}),
|
|
262
|
+
...(acc.usage?.prompt_tokens !== undefined ? { promptTokens: acc.usage.prompt_tokens } : {}),
|
|
263
|
+
...(acc.usage?.completion_tokens !== undefined ? { completionTokens: acc.usage.completion_tokens } : {}),
|
|
264
|
+
...(acc.usage?.total_tokens !== undefined ? { totalTokens: acc.usage.total_tokens } : {}),
|
|
265
|
+
...(acc.finishReason ? { stopReason: acc.finishReason } : {}),
|
|
266
|
+
};
|
|
267
|
+
|
|
268
|
+
return {
|
|
269
|
+
text,
|
|
270
|
+
rawContent: encodeToolCalls(text, toolCalls),
|
|
271
|
+
toolCalls,
|
|
272
|
+
...(toolErrors.length && !toolCalls.length ? { toolErrors } : {}),
|
|
273
|
+
requestId,
|
|
274
|
+
inference,
|
|
275
|
+
};
|
|
276
|
+
},
|
|
277
|
+
|
|
278
|
+
async cancel(requestId: string): Promise<void> {
|
|
279
|
+
inflight.get(requestId)?.abort();
|
|
280
|
+
},
|
|
281
|
+
};
|
|
282
|
+
}
|
package/src/providers/types.ts
CHANGED
|
@@ -17,11 +17,26 @@ export interface TurnInput {
|
|
|
17
17
|
tools: ToolDef[];
|
|
18
18
|
/** System prompt, when not already present as a message. */
|
|
19
19
|
system?: string;
|
|
20
|
+
/**
|
|
21
|
+
* `'required'` forces a tool call, a tool name forces that tool, `'none'`
|
|
22
|
+
* forbids tools. Omit for the model's own choice. Ignored when `tools` is
|
|
23
|
+
* empty or the provider has no such control.
|
|
24
|
+
*/
|
|
25
|
+
toolChoice?: ToolChoice;
|
|
20
26
|
/** Visible content tokens as they stream. */
|
|
21
27
|
onToken?: (token: string) => void;
|
|
22
28
|
signal?: AbortSignal;
|
|
23
29
|
}
|
|
24
30
|
|
|
31
|
+
export type ToolChoice = 'auto' | 'none' | 'required' | (string & {});
|
|
32
|
+
|
|
33
|
+
/** A tool-call region the model emitted that the provider could not turn into a call. */
|
|
34
|
+
export interface ToolCallError {
|
|
35
|
+
code: 'PARSE_ERROR' | 'VALIDATION_ERROR' | 'UNKNOWN_TOOL' | (string & {});
|
|
36
|
+
message: string;
|
|
37
|
+
raw?: string;
|
|
38
|
+
}
|
|
39
|
+
|
|
25
40
|
/** Judge-auditable metrics for one provider inference request. */
|
|
26
41
|
export interface InferenceMetrics {
|
|
27
42
|
requestId?: string;
|
|
@@ -50,10 +65,17 @@ export interface TurnOutput {
|
|
|
50
65
|
rawContent: string;
|
|
51
66
|
/** Tool calls the model requested this turn (empty ⇒ final answer). */
|
|
52
67
|
toolCalls: ToolCall[];
|
|
68
|
+
/** Tool-call attempts that failed to parse or validate this turn. */
|
|
69
|
+
toolErrors?: ToolCallError[];
|
|
53
70
|
/** Provider request id, for cancellation. */
|
|
54
71
|
requestId?: string;
|
|
55
72
|
/** Optional local-inference receipt. Hosts may persist this as JSONL evidence. */
|
|
56
73
|
inference?: InferenceMetrics;
|
|
74
|
+
/**
|
|
75
|
+
* True when the turn produced no visible answer because it ran out of budget
|
|
76
|
+
* (e.g. reasoning used the whole output cap). `text` may hold a placeholder.
|
|
77
|
+
*/
|
|
78
|
+
incomplete?: boolean;
|
|
57
79
|
}
|
|
58
80
|
|
|
59
81
|
export interface LLMProvider {
|
package/src/qvac/config.ts
CHANGED
|
@@ -30,7 +30,7 @@ export const LOCAL_LLM_CONFIG_GPU = {
|
|
|
30
30
|
|
|
31
31
|
/**
|
|
32
32
|
* Delegated to a desktop provider — it has the RAM to run a big context, so give
|
|
33
|
-
* the agentic prompt plenty of room (Qwen3
|
|
33
|
+
* the agentic prompt plenty of room (Qwen3.5 supports far more than this). 2048
|
|
34
34
|
* overflowed with the system prompt + tool/skill definitions alone.
|
|
35
35
|
*/
|
|
36
36
|
export const DELEGATE_LLM_CONFIG = {
|
package/src/qvac/index.ts
CHANGED
|
@@ -23,6 +23,16 @@ export {
|
|
|
23
23
|
normalizeWhisperLang,
|
|
24
24
|
} from './config.js';
|
|
25
25
|
|
|
26
|
+
export {
|
|
27
|
+
QWEN35_MODELS,
|
|
28
|
+
DEFAULT_MODEL_ID,
|
|
29
|
+
DEFAULT_SMALL_DEVICE_MODEL_ID,
|
|
30
|
+
DEFAULT_QVAC_MODEL,
|
|
31
|
+
DEFAULT_SMALL_DEVICE_QVAC_MODEL,
|
|
32
|
+
getRecommendedModel,
|
|
33
|
+
type RecommendedModel,
|
|
34
|
+
} from './models.js';
|
|
35
|
+
|
|
26
36
|
export {
|
|
27
37
|
finalToTurn,
|
|
28
38
|
type QvacFinalLike,
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recommended local models. Plain data (no SDK import): `qvacConstant` is the
|
|
3
|
+
* name of the matching @qvac/sdk registry export (present since @qvac/sdk
|
|
4
|
+
* 0.13.1), `hfRepo`/`hfFile` the same GGUF on Hugging Face for hosts that
|
|
5
|
+
* download directly. Sizes are the exact file sizes.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
export interface RecommendedModel {
|
|
9
|
+
id: string;
|
|
10
|
+
family: string;
|
|
11
|
+
displayName: string;
|
|
12
|
+
/** Name of the @qvac/sdk model constant, e.g. `QWEN3_5_4B_MULTIMODAL_Q4_K_M`. */
|
|
13
|
+
qvacConstant: string;
|
|
14
|
+
quant: string;
|
|
15
|
+
sizeBytes: number;
|
|
16
|
+
hfRepo: string;
|
|
17
|
+
hfFile: string;
|
|
18
|
+
/** Rough RAM needed to run it with the agent's context window. */
|
|
19
|
+
ramHintGb: number;
|
|
20
|
+
notes: string;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export const QWEN35_MODELS: readonly RecommendedModel[] = [
|
|
24
|
+
{
|
|
25
|
+
id: 'qwen3.5-0.8b-q4_k_m',
|
|
26
|
+
family: 'qwen3.5',
|
|
27
|
+
displayName: 'Qwen 3.5 · 0.8B',
|
|
28
|
+
qvacConstant: 'QWEN3_5_0_8B_MULTIMODAL_Q4_K_M',
|
|
29
|
+
quant: 'Q4_K_M',
|
|
30
|
+
sizeBytes: 532_517_120,
|
|
31
|
+
hfRepo: 'unsloth/Qwen3.5-0.8B-GGUF',
|
|
32
|
+
hfFile: 'Qwen3.5-0.8B-Q4_K_M.gguf',
|
|
33
|
+
ramHintGb: 1.5,
|
|
34
|
+
notes: 'Smoke tests only. Loops on wallet actions in our signet bench; fine for chat and single read-only calls.',
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
id: 'qwen3.5-2b-q4_k_m',
|
|
38
|
+
family: 'qwen3.5',
|
|
39
|
+
displayName: 'Qwen 3.5 · 2B',
|
|
40
|
+
qvacConstant: 'QWEN3_5_2B_MULTIMODAL_Q4_K_M',
|
|
41
|
+
quant: 'Q4_K_M',
|
|
42
|
+
sizeBytes: 1_280_835_840,
|
|
43
|
+
hfRepo: 'unsloth/Qwen3.5-2B-GGUF',
|
|
44
|
+
hfFile: 'Qwen3.5-2B-Q4_K_M.gguf',
|
|
45
|
+
ramHintGb: 3,
|
|
46
|
+
notes: 'Recommended default. Passed all 7 wallet tasks of our signet RGB bench at 45–150 s per question on an M4 laptop; also fits phones.',
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
id: 'qwen3.5-4b-q4_k_m',
|
|
50
|
+
family: 'qwen3.5',
|
|
51
|
+
displayName: 'Qwen 3.5 · 4B',
|
|
52
|
+
qvacConstant: 'QWEN3_5_4B_MULTIMODAL_Q4_K_M',
|
|
53
|
+
quant: 'Q4_K_M',
|
|
54
|
+
sizeBytes: 2_740_937_888,
|
|
55
|
+
hfRepo: 'unsloth/Qwen3.5-4B-GGUF',
|
|
56
|
+
hfFile: 'Qwen3.5-4B-Q4_K_M.gguf',
|
|
57
|
+
ramHintGb: 5,
|
|
58
|
+
notes: 'Same correctness as 2B in our bench, about twice as slow (105–330 s per question on an M4). Pick it for longer free-form answers.',
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
id: 'qwen3.5-9b-q4_k_m',
|
|
62
|
+
family: 'qwen3.5',
|
|
63
|
+
displayName: 'Qwen 3.5 · 9B',
|
|
64
|
+
qvacConstant: 'QWEN3_5_9B_MULTIMODAL_Q4_K_M',
|
|
65
|
+
quant: 'Q4_K_M',
|
|
66
|
+
sizeBytes: 5_680_522_464,
|
|
67
|
+
hfRepo: 'unsloth/Qwen3.5-9B-GGUF',
|
|
68
|
+
hfFile: 'Qwen3.5-9B-Q4_K_M.gguf',
|
|
69
|
+
ramHintGb: 9,
|
|
70
|
+
notes: 'Needs 16 GB of RAM. Slow for interactive use (180–690 s per question on an M4); for unattended multi-step tasks.',
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
id: 'qwen3.6-35b-a3b-q4_k_m',
|
|
74
|
+
family: 'qwen3.6',
|
|
75
|
+
displayName: 'Qwen 3.6 · 35B-A3B (MoE)',
|
|
76
|
+
qvacConstant: 'QWEN3_6_35B_A3B_MULTIMODAL_Q4_K_M',
|
|
77
|
+
quant: 'UD-Q4_K_M',
|
|
78
|
+
sizeBytes: 22_134_528_992,
|
|
79
|
+
hfRepo: 'unsloth/Qwen3.6-35B-A3B-GGUF',
|
|
80
|
+
hfFile: 'Qwen3.6-35B-A3B-UD-Q4_K_M.gguf',
|
|
81
|
+
ramHintGb: 26,
|
|
82
|
+
notes: 'Big machines only (32 GB+). MoE with ~3B active parameters: best quality, 22 GB download.',
|
|
83
|
+
},
|
|
84
|
+
];
|
|
85
|
+
|
|
86
|
+
/** Default model for every host (provider sidecar, CLI, examples). */
|
|
87
|
+
export const DEFAULT_MODEL_ID = 'qwen3.5-2b-q4_k_m';
|
|
88
|
+
/** Default model for phones and other small devices. */
|
|
89
|
+
export const DEFAULT_SMALL_DEVICE_MODEL_ID = 'qwen3.5-2b-q4_k_m';
|
|
90
|
+
|
|
91
|
+
export function getRecommendedModel(id: string): RecommendedModel | undefined {
|
|
92
|
+
return QWEN35_MODELS.find((m) => m.id === id);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** @qvac/sdk constant name of the default, e.g. for `sdk[DEFAULT_QVAC_MODEL]`. */
|
|
96
|
+
export const DEFAULT_QVAC_MODEL = getRecommendedModel(DEFAULT_MODEL_ID)!.qvacConstant;
|
|
97
|
+
/** @qvac/sdk constant name of the small-device default. */
|
|
98
|
+
export const DEFAULT_SMALL_DEVICE_QVAC_MODEL = getRecommendedModel(DEFAULT_SMALL_DEVICE_MODEL_ID)!.qvacConstant;
|
package/src/qvac/parse.test.ts
CHANGED
|
@@ -114,6 +114,13 @@ describe('finalToTurn', () => {
|
|
|
114
114
|
]);
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
+
it('recovers a Qwen3.5 XML-style call', () => {
|
|
118
|
+
const calls = extractTextToolCalls(
|
|
119
|
+
'<tool_call>\n<function=rln_issue_asset>\n<parameter=name>\nHack\n</parameter>\n<parameter=amount>\n1000\n</parameter>\n</function>\n</tool_call>',
|
|
120
|
+
);
|
|
121
|
+
expect(calls).toEqual([{ name: 'rln_issue_asset', arguments: { name: 'Hack', amount: 1000 } }]);
|
|
122
|
+
});
|
|
123
|
+
|
|
117
124
|
it('returns [] for plain prose', () => {
|
|
118
125
|
expect(extractTextToolCalls('just a normal answer')).toEqual([]);
|
|
119
126
|
});
|
package/src/qvac/parse.ts
CHANGED
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
* is testable without loading a model, and so the same mapping runs on mobile,
|
|
5
5
|
* desktop, and the eval harness.
|
|
6
6
|
*/
|
|
7
|
+
import type { ToolCallError } from '../providers/types.js';
|
|
7
8
|
import { cleanAssistantVisibleText } from './text.js';
|
|
8
9
|
|
|
9
10
|
/**
|
|
@@ -32,6 +33,8 @@ export interface QvacFinalLike {
|
|
|
32
33
|
raw?: { fullText?: string };
|
|
33
34
|
/** Tool calls the model requested this turn (empty ⇒ final answer). */
|
|
34
35
|
toolCalls?: Array<{ id?: string; name: string; arguments?: Record<string, unknown> }>;
|
|
36
|
+
/** Tool-call regions that failed to parse or validate (QVAC 0.20+; omitted when none). */
|
|
37
|
+
toolErrors?: ToolCallError[];
|
|
35
38
|
/**
|
|
36
39
|
* Why generation stopped: `"length"` when the token budget is exhausted,
|
|
37
40
|
* `"cancelled"` on abort, `"eos"`/`"stopSequence"`/`undefined` on a natural stop. We surface
|
|
@@ -49,6 +52,8 @@ export interface ParsedTurn {
|
|
|
49
52
|
rawContent: string;
|
|
50
53
|
/** Tool calls the model requested (arguments defaulted to `{}`). */
|
|
51
54
|
toolCalls: Array<{ id?: string; name: string; arguments: Record<string, unknown> }>;
|
|
55
|
+
/** Tool-call attempts the SDK could not parse, when no call was recovered from text. */
|
|
56
|
+
toolErrors?: ToolCallError[];
|
|
52
57
|
/** True when generation was cut off by the token budget (incomplete output). */
|
|
53
58
|
truncated: boolean;
|
|
54
59
|
/** Raw stop reason from the SDK, when provided. */
|
|
@@ -89,6 +94,31 @@ function parseCallObject(
|
|
|
89
94
|
return null;
|
|
90
95
|
}
|
|
91
96
|
|
|
97
|
+
/**
|
|
98
|
+
* Parse Qwen3.5's XML call body:
|
|
99
|
+
* `<function=name><parameter=key>value</parameter>…</function>`. Values that
|
|
100
|
+
* read as JSON (numbers, booleans, arrays, objects) are decoded; the rest stay
|
|
101
|
+
* strings.
|
|
102
|
+
*/
|
|
103
|
+
function parseXmlCall(s: string): { name: string; arguments: Record<string, unknown> } | null {
|
|
104
|
+
const fn = s.match(/<function=([^>\s]+)\s*>([\s\S]*?)(?:<\/function>|$)/i);
|
|
105
|
+
if (!fn?.[1]) return null;
|
|
106
|
+
const args: Record<string, unknown> = {};
|
|
107
|
+
for (const m of (fn[2] ?? '').matchAll(/<parameter=([^>\s]+)\s*>([\s\S]*?)<\/parameter>/gi)) {
|
|
108
|
+
const raw = (m[2] ?? '').trim();
|
|
109
|
+
let value: unknown = raw;
|
|
110
|
+
if (/^(-?\d+(\.\d+)?|true|false|null|\[[\s\S]*\]|\{[\s\S]*\})$/.test(raw)) {
|
|
111
|
+
try {
|
|
112
|
+
value = JSON.parse(raw);
|
|
113
|
+
} catch {
|
|
114
|
+
value = raw;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
args[m[1]!] = value;
|
|
118
|
+
}
|
|
119
|
+
return { name: fn[1], arguments: args };
|
|
120
|
+
}
|
|
121
|
+
|
|
92
122
|
/**
|
|
93
123
|
* Recover tool calls a model emitted as PLAIN TEXT instead of structured frames
|
|
94
124
|
* — `<tool_call>{"name":…,"arguments":…}</tool_call>` (Qwen/Hermes) or a bare
|
|
@@ -101,7 +131,8 @@ export function extractTextToolCalls(
|
|
|
101
131
|
): Array<{ name: string; arguments: Record<string, unknown> }> {
|
|
102
132
|
const calls: Array<{ name: string; arguments: Record<string, unknown> }> = [];
|
|
103
133
|
for (const m of text.matchAll(/<tool_call\b[^>]*>([\s\S]*?)<\/tool_call>/gi)) {
|
|
104
|
-
const
|
|
134
|
+
const body = m[1] ?? '';
|
|
135
|
+
const c = parseCallObject(body) ?? parseXmlCall(body);
|
|
105
136
|
if (c) calls.push(c);
|
|
106
137
|
}
|
|
107
138
|
if (calls.length) return calls;
|
|
@@ -139,6 +170,7 @@ export function finalToTurn(final: QvacFinalLike, streamed = ''): ParsedTurn {
|
|
|
139
170
|
text,
|
|
140
171
|
rawContent: final.raw?.fullText ?? rawText,
|
|
141
172
|
toolCalls,
|
|
173
|
+
...(toolCalls.length === 0 && final.toolErrors?.length ? { toolErrors: final.toolErrors } : {}),
|
|
142
174
|
truncated: final.stopReason === 'length',
|
|
143
175
|
stopReason: final.stopReason,
|
|
144
176
|
stats: final.stats,
|
|
@@ -100,7 +100,7 @@ describe('createQvacProvider.runTurn', () => {
|
|
|
100
100
|
const cancel = vi.fn(async () => {});
|
|
101
101
|
const { fn } = fakeCompletion(
|
|
102
102
|
{ contentText: '', toolCalls: [], raw: { fullText: '' }, stopReason: 'cancelled' },
|
|
103
|
-
[{ type: 'thinkingDelta', text: 'z'.repeat(
|
|
103
|
+
[{ type: 'thinkingDelta', text: 'z'.repeat(400) }], // ~100 tokens, budget 4 (+ backstop headroom)
|
|
104
104
|
);
|
|
105
105
|
const p = createQvacProvider({
|
|
106
106
|
completion: fn as any,
|
|
@@ -113,6 +113,39 @@ describe('createQvacProvider.runTurn', () => {
|
|
|
113
113
|
expect(out.text).toMatch(/thinking budget/i);
|
|
114
114
|
});
|
|
115
115
|
|
|
116
|
+
it('sends the thinking cap as the SDK reasoning_budget', async () => {
|
|
117
|
+
const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
|
|
118
|
+
const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', maxThinkingTokens: 128 });
|
|
119
|
+
await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
|
|
120
|
+
expect(calls[0].generationParams).toEqual({ reasoning_budget: 128 });
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
it('keeps the reasoning budget below the output cap', async () => {
|
|
124
|
+
const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
|
|
125
|
+
const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', defaultMaxTokens: 512, maxThinkingTokens: 512 });
|
|
126
|
+
await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
|
|
127
|
+
expect(calls[0].generationParams).toEqual({ predict: 512, reasoning_budget: 256 });
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
it('forwards toolChoice only when tools are present', async () => {
|
|
131
|
+
const tool = { name: 'get_balance', description: 'b', parameters: { type: 'object', properties: {} } };
|
|
132
|
+
const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
|
|
133
|
+
const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
|
|
134
|
+
await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [tool as any], toolChoice: 'required' });
|
|
135
|
+
await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], toolChoice: 'required' });
|
|
136
|
+
expect(calls[0].generationParams).toEqual({ tool_choice: 'required' });
|
|
137
|
+
expect(calls[1].generationParams).toBeUndefined();
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
it('returns toolErrors when the model emitted a tool call that did not parse', async () => {
|
|
141
|
+
const toolErrors = [{ code: 'PARSE_ERROR', message: 'unterminated string', raw: '{"ticker":"HCK' }];
|
|
142
|
+
const { fn } = fakeCompletion({ contentText: '', toolCalls: [], toolErrors, raw: { fullText: '<tool_call>{"ticker":"HCK' } });
|
|
143
|
+
const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
|
|
144
|
+
const out = await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
|
|
145
|
+
expect(out.toolCalls).toEqual([]);
|
|
146
|
+
expect(out.toolErrors).toEqual(toolErrors);
|
|
147
|
+
});
|
|
148
|
+
|
|
116
149
|
it('returns a cancelled turn when the SDK rejects final on abort', async () => {
|
|
117
150
|
const cancel = vi.fn(async () => {});
|
|
118
151
|
const fn = () => ({
|
package/src/qvac/provider.ts
CHANGED
|
@@ -41,10 +41,11 @@ export interface QvacProviderOptions {
|
|
|
41
41
|
/** Default max output tokens — caps a turn so it can't ramble. Omit for uncapped. */
|
|
42
42
|
defaultMaxTokens?: number;
|
|
43
43
|
/**
|
|
44
|
-
* Cap `<think>` reasoning at this many TOKENS (not seconds — tok/s varies
|
|
45
|
-
* the SDK
|
|
46
|
-
* the
|
|
47
|
-
* "Thinking…". Omit for
|
|
44
|
+
* Cap `<think>` reasoning at this many TOKENS (not seconds — tok/s varies).
|
|
45
|
+
* Sent as the SDK's `reasoning_budget`, so the model closes its reasoning and
|
|
46
|
+
* answers. If the stream still runs well past it, the run is cancelled and a
|
|
47
|
+
* short fallback is returned instead of hanging on "Thinking…". Omit for
|
|
48
|
+
* unlimited reasoning.
|
|
48
49
|
*/
|
|
49
50
|
maxThinkingTokens?: number;
|
|
50
51
|
/** Stream the model's `<think>` reasoning, when a host wants to surface it. */
|
|
@@ -93,13 +94,23 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
|
|
|
93
94
|
// when a value is set so a host that passes neither keeps SDK defaults.
|
|
94
95
|
const temp = input.temperature ?? options.defaultTemperature;
|
|
95
96
|
const predict = input.maxTokens ?? options.defaultMaxTokens;
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
:
|
|
97
|
+
// A thinking budget at or above the output cap never binds: the model can
|
|
98
|
+
// spend the whole turn reasoning and return no answer. Keep half for it.
|
|
99
|
+
const thinkingCap = input.maxThinkingTokens ?? options.maxThinkingTokens;
|
|
100
|
+
const maxThinkingTokens =
|
|
101
|
+
thinkingCap !== undefined && predict !== undefined && thinkingCap >= predict
|
|
102
|
+
? Math.floor(predict / 2)
|
|
103
|
+
: thinkingCap;
|
|
104
|
+
// `tool_choice` is only meaningful with tools; the SDK rejects a named
|
|
105
|
+
// choice that isn't among them.
|
|
106
|
+
const toolChoice = tools && input.toolChoice ? input.toolChoice : undefined;
|
|
107
|
+
const generationParamsRaw = {
|
|
108
|
+
...(temp !== undefined ? { temp } : {}),
|
|
109
|
+
...(predict !== undefined ? { predict } : {}),
|
|
110
|
+
...(maxThinkingTokens !== undefined ? { reasoning_budget: maxThinkingTokens } : {}),
|
|
111
|
+
...(toolChoice ? { tool_choice: toolChoice } : {}),
|
|
112
|
+
};
|
|
113
|
+
const generationParams = Object.keys(generationParamsRaw).length ? generationParamsRaw : undefined;
|
|
103
114
|
|
|
104
115
|
const run = options.completion({
|
|
105
116
|
modelId,
|
|
@@ -129,11 +140,13 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
|
|
|
129
140
|
}
|
|
130
141
|
}
|
|
131
142
|
|
|
132
|
-
const maxThinkingTokens = input.maxThinkingTokens ?? options.maxThinkingTokens;
|
|
133
143
|
const result = await consumeRun(run, {
|
|
134
144
|
onToken: input.onToken,
|
|
135
145
|
onThinking: input.onThinking ?? options.onThinking,
|
|
136
|
-
|
|
146
|
+
// Backstop only: the SDK enforces the budget itself, and our count is a
|
|
147
|
+
// char-based estimate, so leave headroom before cancelling.
|
|
148
|
+
maxThinkingTokens:
|
|
149
|
+
maxThinkingTokens === undefined ? undefined : Math.ceil(maxThinkingTokens * 1.25) + 32,
|
|
137
150
|
// Cancel the in-flight run the moment the thinking budget is blown — the
|
|
138
151
|
// SDK keeps generating otherwise. Fire-and-forget; `final` then resolves.
|
|
139
152
|
onThinkingBudgetExceeded: () => {
|
|
@@ -175,12 +188,16 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
|
|
|
175
188
|
...(result.stopReason ? { stopReason: result.stopReason } : {}),
|
|
176
189
|
};
|
|
177
190
|
|
|
191
|
+
const incomplete =
|
|
192
|
+
!result.text && result.toolCalls.length === 0 && (result.thinkingBudgetExceeded || !!result.truncated);
|
|
178
193
|
return {
|
|
179
194
|
text,
|
|
180
195
|
rawContent: result.rawContent,
|
|
181
196
|
toolCalls: result.toolCalls,
|
|
197
|
+
...(result.toolErrors ? { toolErrors: result.toolErrors } : {}),
|
|
182
198
|
requestId: result.requestId,
|
|
183
199
|
inference,
|
|
200
|
+
...(incomplete ? { incomplete: true } : {}),
|
|
184
201
|
};
|
|
185
202
|
},
|
|
186
203
|
|