@kaleidorg/mind 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/README.md +96 -8
  2. package/dist/context/compress.d.ts.map +1 -1
  3. package/dist/context/compress.js +1 -0
  4. package/dist/context/compress.js.map +1 -1
  5. package/dist/context/rgb-units.d.ts +18 -0
  6. package/dist/context/rgb-units.d.ts.map +1 -0
  7. package/dist/context/rgb-units.js +86 -0
  8. package/dist/context/rgb-units.js.map +1 -0
  9. package/dist/engine.d.ts +53 -1
  10. package/dist/engine.d.ts.map +1 -1
  11. package/dist/engine.js +179 -17
  12. package/dist/engine.js.map +1 -1
  13. package/dist/funnel.d.ts.map +1 -1
  14. package/dist/funnel.js +23 -4
  15. package/dist/funnel.js.map +1 -1
  16. package/dist/guards.d.ts +63 -0
  17. package/dist/guards.d.ts.map +1 -0
  18. package/dist/guards.js +284 -0
  19. package/dist/guards.js.map +1 -0
  20. package/dist/index.d.ts +9 -2
  21. package/dist/index.d.ts.map +1 -1
  22. package/dist/index.js +7 -0
  23. package/dist/index.js.map +1 -1
  24. package/dist/kaleidoswap/contract.d.ts +3 -4
  25. package/dist/kaleidoswap/contract.d.ts.map +1 -1
  26. package/dist/kaleidoswap/contract.js +3 -17
  27. package/dist/kaleidoswap/contract.js.map +1 -1
  28. package/dist/providers/openai.d.ts +64 -0
  29. package/dist/providers/openai.d.ts.map +1 -0
  30. package/dist/providers/openai.js +233 -0
  31. package/dist/providers/openai.js.map +1 -0
  32. package/dist/providers/types.d.ts +20 -0
  33. package/dist/providers/types.d.ts.map +1 -1
  34. package/dist/qvac/config.d.ts +1 -1
  35. package/dist/qvac/config.js +1 -1
  36. package/dist/qvac/index.d.ts +1 -0
  37. package/dist/qvac/index.d.ts.map +1 -1
  38. package/dist/qvac/index.js +1 -0
  39. package/dist/qvac/index.js.map +1 -1
  40. package/dist/qvac/models.d.ts +31 -0
  41. package/dist/qvac/models.d.ts.map +1 -0
  42. package/dist/qvac/models.js +80 -0
  43. package/dist/qvac/models.js.map +1 -0
  44. package/dist/qvac/parse.d.ts +11 -0
  45. package/dist/qvac/parse.d.ts.map +1 -1
  46. package/dist/qvac/parse.js +29 -7
  47. package/dist/qvac/parse.js.map +1 -1
  48. package/dist/qvac/provider.d.ts +5 -4
  49. package/dist/qvac/provider.d.ts.map +1 -1
  50. package/dist/qvac/provider.js +22 -8
  51. package/dist/qvac/provider.js.map +1 -1
  52. package/dist/qvac/stream.d.ts +3 -5
  53. package/dist/qvac/stream.d.ts.map +1 -1
  54. package/dist/qvac/stream.js.map +1 -1
  55. package/dist/qvac/tools.d.ts +5 -0
  56. package/dist/qvac/tools.d.ts.map +1 -1
  57. package/dist/qvac/tools.js +10 -0
  58. package/dist/qvac/tools.js.map +1 -1
  59. package/dist/recipe/issue-asset.d.ts.map +1 -1
  60. package/dist/recipe/issue-asset.js +42 -16
  61. package/dist/recipe/issue-asset.js.map +1 -1
  62. package/dist/recipe/runner.d.ts.map +1 -1
  63. package/dist/recipe/runner.js +1 -0
  64. package/dist/recipe/runner.js.map +1 -1
  65. package/dist/recipe/submarine-pay.d.ts +17 -0
  66. package/dist/recipe/submarine-pay.d.ts.map +1 -0
  67. package/dist/recipe/submarine-pay.js +79 -0
  68. package/dist/recipe/submarine-pay.js.map +1 -0
  69. package/dist/skills/registry.js +1 -1
  70. package/dist/skills/select.d.ts +20 -0
  71. package/dist/skills/select.d.ts.map +1 -0
  72. package/dist/skills/select.js +32 -0
  73. package/dist/skills/select.js.map +1 -0
  74. package/dist/submarine/contract.d.ts +42 -0
  75. package/dist/submarine/contract.d.ts.map +1 -0
  76. package/dist/submarine/contract.js +73 -0
  77. package/dist/submarine/contract.js.map +1 -0
  78. package/dist/testing/mock-wallet.d.ts +24 -0
  79. package/dist/testing/mock-wallet.d.ts.map +1 -1
  80. package/dist/testing/mock-wallet.js +75 -6
  81. package/dist/testing/mock-wallet.js.map +1 -1
  82. package/dist/wallet/confirm.d.ts.map +1 -1
  83. package/dist/wallet/confirm.js +27 -9
  84. package/dist/wallet/confirm.js.map +1 -1
  85. package/package.json +6 -1
  86. package/skills/README.md +1 -0
  87. package/skills/kaleido-trading/SKILL.md +22 -28
  88. package/skills/kaleido-trading/references/api.md +3 -3
  89. package/skills/rgb-lightning-node/SKILL.md +32 -12
  90. package/skills/spark-wallet/SKILL.md +1 -0
  91. package/skills/submarine-swaps/SKILL.md +48 -0
  92. package/src/context/compress.ts +1 -0
  93. package/src/context/rgb-units.test.ts +42 -0
  94. package/src/context/rgb-units.ts +89 -0
  95. package/src/engine.test.ts +94 -3
  96. package/src/engine.ts +234 -19
  97. package/src/funnel.mind.test.ts +4 -2
  98. package/src/funnel.ts +23 -4
  99. package/src/guards.test.ts +399 -0
  100. package/src/guards.ts +299 -0
  101. package/src/index.ts +37 -1
  102. package/src/kaleidoswap/contract.test.ts +8 -16
  103. package/src/kaleidoswap/contract.ts +4 -32
  104. package/src/providers/openai.test.ts +127 -0
  105. package/src/providers/openai.ts +282 -0
  106. package/src/providers/types.ts +22 -0
  107. package/src/qvac/config.ts +1 -1
  108. package/src/qvac/index.ts +10 -0
  109. package/src/qvac/models.ts +98 -0
  110. package/src/qvac/parse.test.ts +7 -0
  111. package/src/qvac/parse.ts +33 -1
  112. package/src/qvac/provider.test.ts +34 -1
  113. package/src/qvac/provider.ts +30 -13
  114. package/src/qvac/stream.ts +3 -5
  115. package/src/qvac/tools.ts +10 -0
  116. package/src/recipe/issue-asset.test.ts +39 -0
  117. package/src/recipe/issue-asset.ts +41 -13
  118. package/src/recipe/runner.ts +1 -0
  119. package/src/recipe/submarine-pay.test.ts +87 -0
  120. package/src/recipe/submarine-pay.ts +78 -0
  121. package/src/skills/registry.ts +1 -1
  122. package/src/skills/select.ts +41 -0
  123. package/src/submarine/contract.ts +112 -0
  124. package/src/testing/mock-wallet.ts +73 -6
  125. package/src/wallet/confirm.ts +27 -9
@@ -0,0 +1,282 @@
1
+ /**
2
+ * createOpenAICompatibleProvider — an LLMProvider over any server that speaks
3
+ * the OpenAI Chat Completions API with tool calling: Ollama, llama.cpp
4
+ * `llama-server`, LM Studio, vLLM, `qvac serve`, or a hosted API.
5
+ *
6
+ * No dependencies: it uses `fetch` (pass your own for older runtimes or tests).
7
+ *
8
+ * The Engine keeps history as plain `{ role, content }` messages. Tool calls
9
+ * are written into the assistant's `rawContent` as `<tool_call>{json}</tool_call>`
10
+ * blocks and turned back into `tool_calls` / `tool_call_id` pairs on the next
11
+ * request, so the server always receives a valid tool-calling conversation.
12
+ */
13
+ import type { Message, ToolCall } from '../types.js';
14
+ import type { InferenceMetrics, LLMProvider, ToolCallError, ToolChoice, TurnInput, TurnOutput } from './types.js';
15
+ import { extractTextToolCalls } from '../qvac/parse.js';
16
+ import { toolParametersSchema } from '../qvac/tools.js';
17
+
18
+ export interface OpenAICompatibleOptions {
19
+ /** API root including the version, e.g. `http://localhost:11434/v1`. */
20
+ baseUrl: string;
21
+ /** Model name as the server knows it, e.g. `qwen3.5:4b`. */
22
+ model: string;
23
+ /** Sent as `Authorization: Bearer …` when set. */
24
+ apiKey?: string;
25
+ defaultTemperature?: number;
26
+ defaultMaxTokens?: number;
27
+ /** Stream tokens (default true). Set false for servers without SSE. */
28
+ stream?: boolean;
29
+ /** Extra JSON merged into every request body (e.g. `{ reasoning_effort: 'low' }`). */
30
+ extraBody?: Record<string, unknown>;
31
+ /** Extra request headers. */
32
+ headers?: Record<string, string>;
33
+ /** Reasoning deltas (`reasoning_content` / `reasoning`), when the server sends them. */
34
+ onThinking?: (token: string) => void;
35
+ fetch?: typeof fetch;
36
+ }
37
+
38
+ export interface OpenAITurnInput extends TurnInput {
39
+ temperature?: number;
40
+ maxTokens?: number;
41
+ }
42
+
43
+ type WireMessage =
44
+ | { role: 'system' | 'user'; content: string }
45
+ | { role: 'assistant'; content: string | null; tool_calls?: WireToolCall[] }
46
+ | { role: 'tool'; content: string; tool_call_id: string };
47
+
48
+ interface WireToolCall {
49
+ id: string;
50
+ type: 'function';
51
+ function: { name: string; arguments: string };
52
+ }
53
+
54
+ const TOOL_CALL_BLOCK = /<tool_call\b[^>]*>[\s\S]*?<\/tool_call>/gi;
55
+
56
+ /** The assistant frame the Engine stores: visible text plus one block per call. */
57
+ export function encodeToolCalls(text: string, calls: ToolCall[]): string {
58
+ const blocks = calls.map((c) => `<tool_call>${JSON.stringify({ name: c.name, arguments: c.arguments })}</tool_call>`);
59
+ return [text.trim(), ...blocks].filter(Boolean).join('\n');
60
+ }
61
+
62
+ /** Engine history → Chat Completions messages, with ids pairing calls and results. */
63
+ export function toWireMessages(messages: Message[], system?: string): WireMessage[] {
64
+ const out: WireMessage[] = system ? [{ role: 'system', content: system }] : [];
65
+ let pending: string[] = [];
66
+ messages.forEach((m, i) => {
67
+ if (m.role === 'assistant') {
68
+ const calls = extractTextToolCalls(m.content);
69
+ const text = m.content.replace(TOOL_CALL_BLOCK, '').trim();
70
+ pending = calls.map((_, j) => `call_${i}_${j}`);
71
+ out.push(
72
+ calls.length
73
+ ? {
74
+ role: 'assistant',
75
+ content: text || null,
76
+ tool_calls: calls.map((c, j) => ({
77
+ id: pending[j]!,
78
+ type: 'function',
79
+ function: { name: c.name, arguments: JSON.stringify(c.arguments) },
80
+ })),
81
+ }
82
+ : { role: 'assistant', content: m.content },
83
+ );
84
+ } else if (m.role === 'tool') {
85
+ const id = pending.shift();
86
+ // A tool message with no call to answer (e.g. a parse-error note) would
87
+ // be rejected by the server; send it as user context instead.
88
+ out.push(id ? { role: 'tool', content: m.content, tool_call_id: id } : { role: 'user', content: `Tool result: ${m.content}` });
89
+ } else {
90
+ pending = [];
91
+ out.push({ role: m.role as 'system' | 'user', content: m.content });
92
+ }
93
+ });
94
+ return out;
95
+ }
96
+
97
+ function wireToolChoice(choice: ToolChoice | undefined): unknown {
98
+ if (!choice) return undefined;
99
+ if (choice === 'auto' || choice === 'none' || choice === 'required') return choice;
100
+ return { type: 'function', function: { name: choice } };
101
+ }
102
+
103
+ interface Accumulated {
104
+ content: string;
105
+ calls: Map<number, { id?: string; name: string; args: string }>;
106
+ finishReason?: string;
107
+ usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number };
108
+ }
109
+
110
+ function absorbDelta(acc: Accumulated, delta: any, onToken?: (t: string) => void, onThinking?: (t: string) => void) {
111
+ if (typeof delta?.content === 'string' && delta.content) {
112
+ acc.content += delta.content;
113
+ onToken?.(delta.content);
114
+ }
115
+ const reasoning = delta?.reasoning_content ?? delta?.reasoning;
116
+ if (typeof reasoning === 'string' && reasoning) onThinking?.(reasoning);
117
+ for (const tc of delta?.tool_calls ?? []) {
118
+ const index = typeof tc.index === 'number' ? tc.index : acc.calls.size;
119
+ const cur = acc.calls.get(index) ?? { name: '', args: '' };
120
+ if (tc.id) cur.id = tc.id;
121
+ if (tc.function?.name) cur.name += tc.function.name;
122
+ if (tc.function?.arguments) {
123
+ cur.args += typeof tc.function.arguments === 'string' ? tc.function.arguments : JSON.stringify(tc.function.arguments);
124
+ }
125
+ acc.calls.set(index, cur);
126
+ }
127
+ }
128
+
129
+ async function readSse(body: ReadableStream<Uint8Array>, onEvent: (data: string) => void): Promise<void> {
130
+ const reader = body.getReader();
131
+ const decoder = new TextDecoder();
132
+ let buffer = '';
133
+ for (;;) {
134
+ const { value, done } = await reader.read();
135
+ if (done) break;
136
+ buffer += decoder.decode(value, { stream: true });
137
+ let nl: number;
138
+ while ((nl = buffer.indexOf('\n')) >= 0) {
139
+ const line = buffer.slice(0, nl).trim();
140
+ buffer = buffer.slice(nl + 1);
141
+ if (line.startsWith('data:')) onEvent(line.slice(5).trim());
142
+ }
143
+ }
144
+ const last = buffer.trim();
145
+ if (last.startsWith('data:')) onEvent(last.slice(5).trim());
146
+ }
147
+
148
+ export function createOpenAICompatibleProvider(options: OpenAICompatibleOptions): LLMProvider {
149
+ const doFetch = options.fetch ?? globalThis.fetch;
150
+ if (!doFetch) throw new Error('No fetch available; pass options.fetch');
151
+ const url = `${options.baseUrl.replace(/\/+$/, '')}/chat/completions`;
152
+ const inflight = new Map<string, AbortController>();
153
+ let seq = 0;
154
+
155
+ return {
156
+ name: 'openai-compatible',
157
+
158
+ async runTurn(input: OpenAITurnInput): Promise<TurnOutput> {
159
+ const requestId = `oai-${Date.now().toString(36)}-${++seq}`;
160
+ const controller = new AbortController();
161
+ inflight.set(requestId, controller);
162
+ const onAbort = () => controller.abort();
163
+ input.signal?.addEventListener('abort', onAbort, { once: true });
164
+ if (input.signal?.aborted) controller.abort();
165
+
166
+ const stream = options.stream ?? true;
167
+ const temperature = input.temperature ?? options.defaultTemperature;
168
+ const maxTokens = input.maxTokens ?? options.defaultMaxTokens;
169
+ const tools = input.tools.map((t) => ({
170
+ type: 'function' as const,
171
+ function: { name: t.name, description: t.description ?? '', parameters: toolParametersSchema(t) },
172
+ }));
173
+ const toolChoice = tools.length ? wireToolChoice(input.toolChoice) : undefined;
174
+ const body = {
175
+ model: options.model,
176
+ messages: toWireMessages(input.messages, input.system),
177
+ stream,
178
+ ...(stream ? { stream_options: { include_usage: true } } : {}),
179
+ ...(temperature !== undefined ? { temperature } : {}),
180
+ ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
181
+ ...(tools.length ? { tools } : {}),
182
+ ...(toolChoice ? { tool_choice: toolChoice } : {}),
183
+ ...options.extraBody,
184
+ };
185
+
186
+ const startedAt = Date.now();
187
+ let firstTokenAt: number | undefined;
188
+ const mark = (f?: (t: string) => void) => (t: string) => {
189
+ firstTokenAt ??= Date.now();
190
+ f?.(t);
191
+ };
192
+ const acc: Accumulated = { content: '', calls: new Map() };
193
+ let cancelled = false;
194
+ try {
195
+ const res = await doFetch(url, {
196
+ method: 'POST',
197
+ headers: {
198
+ 'content-type': 'application/json',
199
+ ...(options.apiKey ? { authorization: `Bearer ${options.apiKey}` } : {}),
200
+ ...options.headers,
201
+ },
202
+ body: JSON.stringify(body),
203
+ signal: controller.signal,
204
+ });
205
+ if (!res.ok) {
206
+ const detail = await res.text().catch(() => '');
207
+ throw new Error(`${options.baseUrl} returned ${res.status}: ${detail.slice(0, 300)}`);
208
+ }
209
+ if (stream && res.body) {
210
+ await readSse(res.body, (data) => {
211
+ if (data === '[DONE]') return;
212
+ let chunk: any;
213
+ try {
214
+ chunk = JSON.parse(data);
215
+ } catch {
216
+ return;
217
+ }
218
+ if (chunk.usage) acc.usage = chunk.usage;
219
+ const choice = chunk.choices?.[0];
220
+ if (!choice) return;
221
+ absorbDelta(acc, choice.delta, mark(input.onToken), mark(options.onThinking));
222
+ if (choice.finish_reason) acc.finishReason = choice.finish_reason;
223
+ });
224
+ } else {
225
+ const json: any = await res.json();
226
+ const choice = json.choices?.[0];
227
+ absorbDelta(acc, { ...choice?.message, tool_calls: choice?.message?.tool_calls?.map((c: any, index: number) => ({ index, ...c })) }, input.onToken, options.onThinking);
228
+ acc.finishReason = choice?.finish_reason;
229
+ acc.usage = json.usage;
230
+ }
231
+ } catch (err) {
232
+ if (!controller.signal.aborted) throw err;
233
+ cancelled = true;
234
+ } finally {
235
+ inflight.delete(requestId);
236
+ input.signal?.removeEventListener('abort', onAbort);
237
+ }
238
+
239
+ const toolCalls: ToolCall[] = [];
240
+ const toolErrors: ToolCallError[] = [];
241
+ for (const c of [...acc.calls.entries()].sort(([a], [b]) => a - b).map(([, v]) => v)) {
242
+ if (!c.name) continue;
243
+ try {
244
+ const args = c.args.trim() ? JSON.parse(c.args) : {};
245
+ toolCalls.push({ ...(c.id ? { id: c.id } : {}), name: c.name, arguments: args && typeof args === 'object' ? args : {} });
246
+ } catch {
247
+ toolErrors.push({ code: 'PARSE_ERROR', message: `arguments for ${c.name} are not valid JSON`, raw: c.args });
248
+ }
249
+ }
250
+ // Models that ignore the tools API sometimes write the call as text.
251
+ if (!toolCalls.length && !toolErrors.length) {
252
+ for (const c of extractTextToolCalls(acc.content)) toolCalls.push(c);
253
+ }
254
+ const text = acc.content.replace(TOOL_CALL_BLOCK, '').trim();
255
+
256
+ const truncated = acc.finishReason === 'length';
257
+ const inference: InferenceMetrics = {
258
+ requestId,
259
+ durationMs: Date.now() - startedAt,
260
+ status: cancelled ? 'cancelled' : truncated ? 'truncated' : 'completed',
261
+ ...(firstTokenAt !== undefined ? { ttftMs: firstTokenAt - startedAt } : {}),
262
+ ...(acc.usage?.prompt_tokens !== undefined ? { promptTokens: acc.usage.prompt_tokens } : {}),
263
+ ...(acc.usage?.completion_tokens !== undefined ? { completionTokens: acc.usage.completion_tokens } : {}),
264
+ ...(acc.usage?.total_tokens !== undefined ? { totalTokens: acc.usage.total_tokens } : {}),
265
+ ...(acc.finishReason ? { stopReason: acc.finishReason } : {}),
266
+ };
267
+
268
+ return {
269
+ text,
270
+ rawContent: encodeToolCalls(text, toolCalls),
271
+ toolCalls,
272
+ ...(toolErrors.length && !toolCalls.length ? { toolErrors } : {}),
273
+ requestId,
274
+ inference,
275
+ };
276
+ },
277
+
278
+ async cancel(requestId: string): Promise<void> {
279
+ inflight.get(requestId)?.abort();
280
+ },
281
+ };
282
+ }
@@ -17,11 +17,26 @@ export interface TurnInput {
17
17
  tools: ToolDef[];
18
18
  /** System prompt, when not already present as a message. */
19
19
  system?: string;
20
+ /**
21
+ * `'required'` forces a tool call, a tool name forces that tool, `'none'`
22
+ * forbids tools. Omit for the model's own choice. Ignored when `tools` is
23
+ * empty or the provider has no such control.
24
+ */
25
+ toolChoice?: ToolChoice;
20
26
  /** Visible content tokens as they stream. */
21
27
  onToken?: (token: string) => void;
22
28
  signal?: AbortSignal;
23
29
  }
24
30
 
31
+ export type ToolChoice = 'auto' | 'none' | 'required' | (string & {});
32
+
33
+ /** A tool-call region the model emitted that the provider could not turn into a call. */
34
+ export interface ToolCallError {
35
+ code: 'PARSE_ERROR' | 'VALIDATION_ERROR' | 'UNKNOWN_TOOL' | (string & {});
36
+ message: string;
37
+ raw?: string;
38
+ }
39
+
25
40
  /** Judge-auditable metrics for one provider inference request. */
26
41
  export interface InferenceMetrics {
27
42
  requestId?: string;
@@ -50,10 +65,17 @@ export interface TurnOutput {
50
65
  rawContent: string;
51
66
  /** Tool calls the model requested this turn (empty ⇒ final answer). */
52
67
  toolCalls: ToolCall[];
68
+ /** Tool-call attempts that failed to parse or validate this turn. */
69
+ toolErrors?: ToolCallError[];
53
70
  /** Provider request id, for cancellation. */
54
71
  requestId?: string;
55
72
  /** Optional local-inference receipt. Hosts may persist this as JSONL evidence. */
56
73
  inference?: InferenceMetrics;
74
+ /**
75
+ * True when the turn produced no visible answer because it ran out of budget
76
+ * (e.g. reasoning used the whole output cap). `text` may hold a placeholder.
77
+ */
78
+ incomplete?: boolean;
57
79
  }
58
80
 
59
81
  export interface LLMProvider {
@@ -30,7 +30,7 @@ export const LOCAL_LLM_CONFIG_GPU = {
30
30
 
31
31
  /**
32
32
  * Delegated to a desktop provider — it has the RAM to run a big context, so give
33
- * the agentic prompt plenty of room (Qwen3-600M supports up to 32k). 2048
33
+ * the agentic prompt plenty of room (Qwen3.5 supports far more than this). 2048
34
34
  * overflowed with the system prompt + tool/skill definitions alone.
35
35
  */
36
36
  export const DELEGATE_LLM_CONFIG = {
package/src/qvac/index.ts CHANGED
@@ -23,6 +23,16 @@ export {
23
23
  normalizeWhisperLang,
24
24
  } from './config.js';
25
25
 
26
+ export {
27
+ QWEN35_MODELS,
28
+ DEFAULT_MODEL_ID,
29
+ DEFAULT_SMALL_DEVICE_MODEL_ID,
30
+ DEFAULT_QVAC_MODEL,
31
+ DEFAULT_SMALL_DEVICE_QVAC_MODEL,
32
+ getRecommendedModel,
33
+ type RecommendedModel,
34
+ } from './models.js';
35
+
26
36
  export {
27
37
  finalToTurn,
28
38
  type QvacFinalLike,
@@ -0,0 +1,98 @@
1
+ /**
2
+ * Recommended local models. Plain data (no SDK import): `qvacConstant` is the
3
+ * name of the matching @qvac/sdk registry export (present since @qvac/sdk
4
+ * 0.13.1), `hfRepo`/`hfFile` the same GGUF on Hugging Face for hosts that
5
+ * download directly. Sizes are the exact file sizes.
6
+ */
7
+
8
+ export interface RecommendedModel {
9
+ id: string;
10
+ family: string;
11
+ displayName: string;
12
+ /** Name of the @qvac/sdk model constant, e.g. `QWEN3_5_4B_MULTIMODAL_Q4_K_M`. */
13
+ qvacConstant: string;
14
+ quant: string;
15
+ sizeBytes: number;
16
+ hfRepo: string;
17
+ hfFile: string;
18
+ /** Rough RAM needed to run it with the agent's context window. */
19
+ ramHintGb: number;
20
+ notes: string;
21
+ }
22
+
23
+ export const QWEN35_MODELS: readonly RecommendedModel[] = [
24
+ {
25
+ id: 'qwen3.5-0.8b-q4_k_m',
26
+ family: 'qwen3.5',
27
+ displayName: 'Qwen 3.5 · 0.8B',
28
+ qvacConstant: 'QWEN3_5_0_8B_MULTIMODAL_Q4_K_M',
29
+ quant: 'Q4_K_M',
30
+ sizeBytes: 532_517_120,
31
+ hfRepo: 'unsloth/Qwen3.5-0.8B-GGUF',
32
+ hfFile: 'Qwen3.5-0.8B-Q4_K_M.gguf',
33
+ ramHintGb: 1.5,
34
+ notes: 'Smoke tests only. Loops on wallet actions in our signet bench; fine for chat and single read-only calls.',
35
+ },
36
+ {
37
+ id: 'qwen3.5-2b-q4_k_m',
38
+ family: 'qwen3.5',
39
+ displayName: 'Qwen 3.5 · 2B',
40
+ qvacConstant: 'QWEN3_5_2B_MULTIMODAL_Q4_K_M',
41
+ quant: 'Q4_K_M',
42
+ sizeBytes: 1_280_835_840,
43
+ hfRepo: 'unsloth/Qwen3.5-2B-GGUF',
44
+ hfFile: 'Qwen3.5-2B-Q4_K_M.gguf',
45
+ ramHintGb: 3,
46
+ notes: 'Recommended default. Passed all 7 wallet tasks of our signet RGB bench at 45–150 s per question on an M4 laptop; also fits phones.',
47
+ },
48
+ {
49
+ id: 'qwen3.5-4b-q4_k_m',
50
+ family: 'qwen3.5',
51
+ displayName: 'Qwen 3.5 · 4B',
52
+ qvacConstant: 'QWEN3_5_4B_MULTIMODAL_Q4_K_M',
53
+ quant: 'Q4_K_M',
54
+ sizeBytes: 2_740_937_888,
55
+ hfRepo: 'unsloth/Qwen3.5-4B-GGUF',
56
+ hfFile: 'Qwen3.5-4B-Q4_K_M.gguf',
57
+ ramHintGb: 5,
58
+ notes: 'Same correctness as 2B in our bench, about twice as slow (105–330 s per question on an M4). Pick it for longer free-form answers.',
59
+ },
60
+ {
61
+ id: 'qwen3.5-9b-q4_k_m',
62
+ family: 'qwen3.5',
63
+ displayName: 'Qwen 3.5 · 9B',
64
+ qvacConstant: 'QWEN3_5_9B_MULTIMODAL_Q4_K_M',
65
+ quant: 'Q4_K_M',
66
+ sizeBytes: 5_680_522_464,
67
+ hfRepo: 'unsloth/Qwen3.5-9B-GGUF',
68
+ hfFile: 'Qwen3.5-9B-Q4_K_M.gguf',
69
+ ramHintGb: 9,
70
+ notes: 'Needs 16 GB of RAM. Slow for interactive use (180–690 s per question on an M4); for unattended multi-step tasks.',
71
+ },
72
+ {
73
+ id: 'qwen3.6-35b-a3b-q4_k_m',
74
+ family: 'qwen3.6',
75
+ displayName: 'Qwen 3.6 · 35B-A3B (MoE)',
76
+ qvacConstant: 'QWEN3_6_35B_A3B_MULTIMODAL_Q4_K_M',
77
+ quant: 'UD-Q4_K_M',
78
+ sizeBytes: 22_134_528_992,
79
+ hfRepo: 'unsloth/Qwen3.6-35B-A3B-GGUF',
80
+ hfFile: 'Qwen3.6-35B-A3B-UD-Q4_K_M.gguf',
81
+ ramHintGb: 26,
82
+ notes: 'Big machines only (32 GB+). MoE with ~3B active parameters: best quality, 22 GB download.',
83
+ },
84
+ ];
85
+
86
+ /** Default model for every host (provider sidecar, CLI, examples). */
87
+ export const DEFAULT_MODEL_ID = 'qwen3.5-2b-q4_k_m';
88
+ /** Default model for phones and other small devices. */
89
+ export const DEFAULT_SMALL_DEVICE_MODEL_ID = 'qwen3.5-2b-q4_k_m';
90
+
91
+ export function getRecommendedModel(id: string): RecommendedModel | undefined {
92
+ return QWEN35_MODELS.find((m) => m.id === id);
93
+ }
94
+
95
+ /** @qvac/sdk constant name of the default, e.g. for `sdk[DEFAULT_QVAC_MODEL]`. */
96
+ export const DEFAULT_QVAC_MODEL = getRecommendedModel(DEFAULT_MODEL_ID)!.qvacConstant;
97
+ /** @qvac/sdk constant name of the small-device default. */
98
+ export const DEFAULT_SMALL_DEVICE_QVAC_MODEL = getRecommendedModel(DEFAULT_SMALL_DEVICE_MODEL_ID)!.qvacConstant;
@@ -114,6 +114,13 @@ describe('finalToTurn', () => {
114
114
  ]);
115
115
  });
116
116
 
117
+ it('recovers a Qwen3.5 XML-style call', () => {
118
+ const calls = extractTextToolCalls(
119
+ '<tool_call>\n<function=rln_issue_asset>\n<parameter=name>\nHack\n</parameter>\n<parameter=amount>\n1000\n</parameter>\n</function>\n</tool_call>',
120
+ );
121
+ expect(calls).toEqual([{ name: 'rln_issue_asset', arguments: { name: 'Hack', amount: 1000 } }]);
122
+ });
123
+
117
124
  it('returns [] for plain prose', () => {
118
125
  expect(extractTextToolCalls('just a normal answer')).toEqual([]);
119
126
  });
package/src/qvac/parse.ts CHANGED
@@ -4,6 +4,7 @@
4
4
  * is testable without loading a model, and so the same mapping runs on mobile,
5
5
  * desktop, and the eval harness.
6
6
  */
7
+ import type { ToolCallError } from '../providers/types.js';
7
8
  import { cleanAssistantVisibleText } from './text.js';
8
9
 
9
10
  /**
@@ -32,6 +33,8 @@ export interface QvacFinalLike {
32
33
  raw?: { fullText?: string };
33
34
  /** Tool calls the model requested this turn (empty ⇒ final answer). */
34
35
  toolCalls?: Array<{ id?: string; name: string; arguments?: Record<string, unknown> }>;
36
+ /** Tool-call regions that failed to parse or validate (QVAC 0.20+; omitted when none). */
37
+ toolErrors?: ToolCallError[];
35
38
  /**
36
39
  * Why generation stopped: `"length"` when the token budget is exhausted,
37
40
  * `"cancelled"` on abort, `"eos"`/`"stopSequence"`/`undefined` on a natural stop. We surface
@@ -49,6 +52,8 @@ export interface ParsedTurn {
49
52
  rawContent: string;
50
53
  /** Tool calls the model requested (arguments defaulted to `{}`). */
51
54
  toolCalls: Array<{ id?: string; name: string; arguments: Record<string, unknown> }>;
55
+ /** Tool-call attempts the SDK could not parse, when no call was recovered from text. */
56
+ toolErrors?: ToolCallError[];
52
57
  /** True when generation was cut off by the token budget (incomplete output). */
53
58
  truncated: boolean;
54
59
  /** Raw stop reason from the SDK, when provided. */
@@ -89,6 +94,31 @@ function parseCallObject(
89
94
  return null;
90
95
  }
91
96
 
97
+ /**
98
+ * Parse Qwen3.5's XML call body:
99
+ * `<function=name><parameter=key>value</parameter>…</function>`. Values that
100
+ * read as JSON (numbers, booleans, arrays, objects) are decoded; the rest stay
101
+ * strings.
102
+ */
103
+ function parseXmlCall(s: string): { name: string; arguments: Record<string, unknown> } | null {
104
+ const fn = s.match(/<function=([^>\s]+)\s*>([\s\S]*?)(?:<\/function>|$)/i);
105
+ if (!fn?.[1]) return null;
106
+ const args: Record<string, unknown> = {};
107
+ for (const m of (fn[2] ?? '').matchAll(/<parameter=([^>\s]+)\s*>([\s\S]*?)<\/parameter>/gi)) {
108
+ const raw = (m[2] ?? '').trim();
109
+ let value: unknown = raw;
110
+ if (/^(-?\d+(\.\d+)?|true|false|null|\[[\s\S]*\]|\{[\s\S]*\})$/.test(raw)) {
111
+ try {
112
+ value = JSON.parse(raw);
113
+ } catch {
114
+ value = raw;
115
+ }
116
+ }
117
+ args[m[1]!] = value;
118
+ }
119
+ return { name: fn[1], arguments: args };
120
+ }
121
+
92
122
  /**
93
123
  * Recover tool calls a model emitted as PLAIN TEXT instead of structured frames
94
124
  * — `<tool_call>{"name":…,"arguments":…}</tool_call>` (Qwen/Hermes) or a bare
@@ -101,7 +131,8 @@ export function extractTextToolCalls(
101
131
  ): Array<{ name: string; arguments: Record<string, unknown> }> {
102
132
  const calls: Array<{ name: string; arguments: Record<string, unknown> }> = [];
103
133
  for (const m of text.matchAll(/<tool_call\b[^>]*>([\s\S]*?)<\/tool_call>/gi)) {
104
- const c = parseCallObject(m[1] ?? '');
134
+ const body = m[1] ?? '';
135
+ const c = parseCallObject(body) ?? parseXmlCall(body);
105
136
  if (c) calls.push(c);
106
137
  }
107
138
  if (calls.length) return calls;
@@ -139,6 +170,7 @@ export function finalToTurn(final: QvacFinalLike, streamed = ''): ParsedTurn {
139
170
  text,
140
171
  rawContent: final.raw?.fullText ?? rawText,
141
172
  toolCalls,
173
+ ...(toolCalls.length === 0 && final.toolErrors?.length ? { toolErrors: final.toolErrors } : {}),
142
174
  truncated: final.stopReason === 'length',
143
175
  stopReason: final.stopReason,
144
176
  stats: final.stats,
@@ -100,7 +100,7 @@ describe('createQvacProvider.runTurn', () => {
100
100
  const cancel = vi.fn(async () => {});
101
101
  const { fn } = fakeCompletion(
102
102
  { contentText: '', toolCalls: [], raw: { fullText: '' }, stopReason: 'cancelled' },
103
- [{ type: 'thinkingDelta', text: 'z'.repeat(40) }], // ~10 tokens, budget 4
103
+ [{ type: 'thinkingDelta', text: 'z'.repeat(400) }], // ~100 tokens, budget 4 (+ backstop headroom)
104
104
  );
105
105
  const p = createQvacProvider({
106
106
  completion: fn as any,
@@ -113,6 +113,39 @@ describe('createQvacProvider.runTurn', () => {
113
113
  expect(out.text).toMatch(/thinking budget/i);
114
114
  });
115
115
 
116
+ it('sends the thinking cap as the SDK reasoning_budget', async () => {
117
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
118
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', maxThinkingTokens: 128 });
119
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
120
+ expect(calls[0].generationParams).toEqual({ reasoning_budget: 128 });
121
+ });
122
+
123
+ it('keeps the reasoning budget below the output cap', async () => {
124
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
125
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', defaultMaxTokens: 512, maxThinkingTokens: 512 });
126
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
127
+ expect(calls[0].generationParams).toEqual({ predict: 512, reasoning_budget: 256 });
128
+ });
129
+
130
+ it('forwards toolChoice only when tools are present', async () => {
131
+ const tool = { name: 'get_balance', description: 'b', parameters: { type: 'object', properties: {} } };
132
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
133
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
134
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [tool as any], toolChoice: 'required' });
135
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], toolChoice: 'required' });
136
+ expect(calls[0].generationParams).toEqual({ tool_choice: 'required' });
137
+ expect(calls[1].generationParams).toBeUndefined();
138
+ });
139
+
140
+ it('returns toolErrors when the model emitted a tool call that did not parse', async () => {
141
+ const toolErrors = [{ code: 'PARSE_ERROR', message: 'unterminated string', raw: '{"ticker":"HCK' }];
142
+ const { fn } = fakeCompletion({ contentText: '', toolCalls: [], toolErrors, raw: { fullText: '<tool_call>{"ticker":"HCK' } });
143
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
144
+ const out = await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
145
+ expect(out.toolCalls).toEqual([]);
146
+ expect(out.toolErrors).toEqual(toolErrors);
147
+ });
148
+
116
149
  it('returns a cancelled turn when the SDK rejects final on abort', async () => {
117
150
  const cancel = vi.fn(async () => {});
118
151
  const fn = () => ({
@@ -41,10 +41,11 @@ export interface QvacProviderOptions {
41
41
  /** Default max output tokens — caps a turn so it can't ramble. Omit for uncapped. */
42
42
  defaultMaxTokens?: number;
43
43
  /**
44
- * Cap `<think>` reasoning at this many TOKENS (not seconds — tok/s varies, and
45
- * the SDK has no numeric reasoning budget). When a turn's thinking exceeds it,
46
- * the run is cancelled and a short fallback is returned instead of hanging on
47
- * "Thinking…". Omit for unlimited reasoning.
44
+ * Cap `<think>` reasoning at this many TOKENS (not seconds — tok/s varies).
45
+ * Sent as the SDK's `reasoning_budget`, so the model closes its reasoning and
46
+ * answers. If the stream still runs well past it, the run is cancelled and a
47
+ * short fallback is returned instead of hanging on "Thinking…". Omit for
48
+ * unlimited reasoning.
48
49
  */
49
50
  maxThinkingTokens?: number;
50
51
  /** Stream the model's `<think>` reasoning, when a host wants to surface it. */
@@ -93,13 +94,23 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
93
94
  // when a value is set so a host that passes neither keeps SDK defaults.
94
95
  const temp = input.temperature ?? options.defaultTemperature;
95
96
  const predict = input.maxTokens ?? options.defaultMaxTokens;
96
- const generationParams =
97
- temp !== undefined || predict !== undefined
98
- ? {
99
- ...(temp !== undefined ? { temp } : {}),
100
- ...(predict !== undefined ? { predict } : {}),
101
- }
102
- : undefined;
97
+ // A thinking budget at or above the output cap never binds: the model can
98
+ // spend the whole turn reasoning and return no answer. Keep half for it.
99
+ const thinkingCap = input.maxThinkingTokens ?? options.maxThinkingTokens;
100
+ const maxThinkingTokens =
101
+ thinkingCap !== undefined && predict !== undefined && thinkingCap >= predict
102
+ ? Math.floor(predict / 2)
103
+ : thinkingCap;
104
+ // `tool_choice` is only meaningful with tools; the SDK rejects a named
105
+ // choice that isn't among them.
106
+ const toolChoice = tools && input.toolChoice ? input.toolChoice : undefined;
107
+ const generationParamsRaw = {
108
+ ...(temp !== undefined ? { temp } : {}),
109
+ ...(predict !== undefined ? { predict } : {}),
110
+ ...(maxThinkingTokens !== undefined ? { reasoning_budget: maxThinkingTokens } : {}),
111
+ ...(toolChoice ? { tool_choice: toolChoice } : {}),
112
+ };
113
+ const generationParams = Object.keys(generationParamsRaw).length ? generationParamsRaw : undefined;
103
114
 
104
115
  const run = options.completion({
105
116
  modelId,
@@ -129,11 +140,13 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
129
140
  }
130
141
  }
131
142
 
132
- const maxThinkingTokens = input.maxThinkingTokens ?? options.maxThinkingTokens;
133
143
  const result = await consumeRun(run, {
134
144
  onToken: input.onToken,
135
145
  onThinking: input.onThinking ?? options.onThinking,
136
- maxThinkingTokens,
146
+ // Backstop only: the SDK enforces the budget itself, and our count is a
147
+ // char-based estimate, so leave headroom before cancelling.
148
+ maxThinkingTokens:
149
+ maxThinkingTokens === undefined ? undefined : Math.ceil(maxThinkingTokens * 1.25) + 32,
137
150
  // Cancel the in-flight run the moment the thinking budget is blown — the
138
151
  // SDK keeps generating otherwise. Fire-and-forget; `final` then resolves.
139
152
  onThinkingBudgetExceeded: () => {
@@ -175,12 +188,16 @@ export function createQvacProvider(options: QvacProviderOptions): LLMProvider {
175
188
  ...(result.stopReason ? { stopReason: result.stopReason } : {}),
176
189
  };
177
190
 
191
+ const incomplete =
192
+ !result.text && result.toolCalls.length === 0 && (result.thinkingBudgetExceeded || !!result.truncated);
178
193
  return {
179
194
  text,
180
195
  rawContent: result.rawContent,
181
196
  toolCalls: result.toolCalls,
197
+ ...(result.toolErrors ? { toolErrors: result.toolErrors } : {}),
182
198
  requestId: result.requestId,
183
199
  inference,
200
+ ...(incomplete ? { incomplete: true } : {}),
184
201
  };
185
202
  },
186
203