@kaleidorg/mind 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. package/README.md +96 -8
  2. package/dist/context/compress.d.ts.map +1 -1
  3. package/dist/context/compress.js +1 -0
  4. package/dist/context/compress.js.map +1 -1
  5. package/dist/context/rgb-units.d.ts +18 -0
  6. package/dist/context/rgb-units.d.ts.map +1 -0
  7. package/dist/context/rgb-units.js +86 -0
  8. package/dist/context/rgb-units.js.map +1 -0
  9. package/dist/engine.d.ts +59 -1
  10. package/dist/engine.d.ts.map +1 -1
  11. package/dist/engine.js +196 -17
  12. package/dist/engine.js.map +1 -1
  13. package/dist/fastpath/fastpath.d.ts +6 -1
  14. package/dist/fastpath/fastpath.d.ts.map +1 -1
  15. package/dist/fastpath/fastpath.js +16 -2
  16. package/dist/fastpath/fastpath.js.map +1 -1
  17. package/dist/fastpath/render.d.ts +2 -0
  18. package/dist/fastpath/render.d.ts.map +1 -0
  19. package/dist/fastpath/render.js +58 -0
  20. package/dist/fastpath/render.js.map +1 -0
  21. package/dist/funnel.d.ts.map +1 -1
  22. package/dist/funnel.js +34 -18
  23. package/dist/funnel.js.map +1 -1
  24. package/dist/guards.d.ts +63 -0
  25. package/dist/guards.d.ts.map +1 -0
  26. package/dist/guards.js +284 -0
  27. package/dist/guards.js.map +1 -0
  28. package/dist/index.d.ts +9 -2
  29. package/dist/index.d.ts.map +1 -1
  30. package/dist/index.js +7 -0
  31. package/dist/index.js.map +1 -1
  32. package/dist/kaleidoswap/contract.d.ts +3 -4
  33. package/dist/kaleidoswap/contract.d.ts.map +1 -1
  34. package/dist/kaleidoswap/contract.js +3 -17
  35. package/dist/kaleidoswap/contract.js.map +1 -1
  36. package/dist/providers/openai.d.ts +64 -0
  37. package/dist/providers/openai.d.ts.map +1 -0
  38. package/dist/providers/openai.js +233 -0
  39. package/dist/providers/openai.js.map +1 -0
  40. package/dist/providers/types.d.ts +33 -0
  41. package/dist/providers/types.d.ts.map +1 -1
  42. package/dist/qvac/config.d.ts +1 -1
  43. package/dist/qvac/config.js +1 -1
  44. package/dist/qvac/index.d.ts +1 -0
  45. package/dist/qvac/index.d.ts.map +1 -1
  46. package/dist/qvac/index.js +1 -0
  47. package/dist/qvac/index.js.map +1 -1
  48. package/dist/qvac/models.d.ts +31 -0
  49. package/dist/qvac/models.d.ts.map +1 -0
  50. package/dist/qvac/models.js +80 -0
  51. package/dist/qvac/models.js.map +1 -0
  52. package/dist/qvac/parse.d.ts +11 -0
  53. package/dist/qvac/parse.d.ts.map +1 -1
  54. package/dist/qvac/parse.js +29 -7
  55. package/dist/qvac/parse.js.map +1 -1
  56. package/dist/qvac/provider.d.ts +14 -4
  57. package/dist/qvac/provider.d.ts.map +1 -1
  58. package/dist/qvac/provider.js +33 -8
  59. package/dist/qvac/provider.js.map +1 -1
  60. package/dist/qvac/stream.d.ts +3 -5
  61. package/dist/qvac/stream.d.ts.map +1 -1
  62. package/dist/qvac/stream.js.map +1 -1
  63. package/dist/qvac/tools.d.ts +5 -0
  64. package/dist/qvac/tools.d.ts.map +1 -1
  65. package/dist/qvac/tools.js +10 -0
  66. package/dist/qvac/tools.js.map +1 -1
  67. package/dist/recipe/issue-asset.d.ts.map +1 -1
  68. package/dist/recipe/issue-asset.js +42 -16
  69. package/dist/recipe/issue-asset.js.map +1 -1
  70. package/dist/recipe/runner.d.ts.map +1 -1
  71. package/dist/recipe/runner.js +1 -0
  72. package/dist/recipe/runner.js.map +1 -1
  73. package/dist/recipe/submarine-pay.d.ts +17 -0
  74. package/dist/recipe/submarine-pay.d.ts.map +1 -0
  75. package/dist/recipe/submarine-pay.js +79 -0
  76. package/dist/recipe/submarine-pay.js.map +1 -0
  77. package/dist/skills/registry.js +1 -1
  78. package/dist/skills/select.d.ts +20 -0
  79. package/dist/skills/select.d.ts.map +1 -0
  80. package/dist/skills/select.js +32 -0
  81. package/dist/skills/select.js.map +1 -0
  82. package/dist/submarine/contract.d.ts +42 -0
  83. package/dist/submarine/contract.d.ts.map +1 -0
  84. package/dist/submarine/contract.js +73 -0
  85. package/dist/submarine/contract.js.map +1 -0
  86. package/dist/testing/mock-wallet.d.ts +24 -0
  87. package/dist/testing/mock-wallet.d.ts.map +1 -1
  88. package/dist/testing/mock-wallet.js +75 -6
  89. package/dist/testing/mock-wallet.js.map +1 -1
  90. package/dist/wallet/confirm.d.ts.map +1 -1
  91. package/dist/wallet/confirm.js +27 -9
  92. package/dist/wallet/confirm.js.map +1 -1
  93. package/package.json +6 -1
  94. package/skills/README.md +1 -0
  95. package/skills/kaleido-trading/SKILL.md +22 -28
  96. package/skills/kaleido-trading/references/api.md +3 -3
  97. package/skills/rgb-lightning-node/SKILL.md +32 -12
  98. package/skills/spark-wallet/SKILL.md +1 -0
  99. package/skills/submarine-swaps/SKILL.md +48 -0
  100. package/src/context/compress.ts +1 -0
  101. package/src/context/rgb-units.test.ts +42 -0
  102. package/src/context/rgb-units.ts +89 -0
  103. package/src/engine.test.ts +133 -3
  104. package/src/engine.ts +254 -19
  105. package/src/fastpath/fastpath.ts +22 -2
  106. package/src/fastpath/render.test.ts +39 -0
  107. package/src/fastpath/render.ts +56 -0
  108. package/src/funnel.mind.test.ts +15 -17
  109. package/src/funnel.ts +34 -19
  110. package/src/guards.test.ts +399 -0
  111. package/src/guards.ts +299 -0
  112. package/src/index.ts +37 -1
  113. package/src/kaleidoswap/contract.test.ts +8 -16
  114. package/src/kaleidoswap/contract.ts +4 -32
  115. package/src/providers/openai.test.ts +127 -0
  116. package/src/providers/openai.ts +282 -0
  117. package/src/providers/types.ts +35 -0
  118. package/src/qvac/config.ts +1 -1
  119. package/src/qvac/index.ts +10 -0
  120. package/src/qvac/models.ts +98 -0
  121. package/src/qvac/parse.test.ts +7 -0
  122. package/src/qvac/parse.ts +33 -1
  123. package/src/qvac/provider.test.ts +54 -1
  124. package/src/qvac/provider.ts +49 -13
  125. package/src/qvac/stream.ts +3 -5
  126. package/src/qvac/tools.ts +10 -0
  127. package/src/recipe/issue-asset.test.ts +39 -0
  128. package/src/recipe/issue-asset.ts +41 -13
  129. package/src/recipe/runner.ts +1 -0
  130. package/src/recipe/submarine-pay.test.ts +87 -0
  131. package/src/recipe/submarine-pay.ts +78 -0
  132. package/src/skills/registry.ts +1 -1
  133. package/src/skills/select.ts +41 -0
  134. package/src/submarine/contract.ts +112 -0
  135. package/src/testing/mock-wallet.ts +73 -6
  136. package/src/wallet/confirm.ts +27 -9
@@ -0,0 +1,282 @@
1
+ /**
2
+ * createOpenAICompatibleProvider — an LLMProvider over any server that speaks
3
+ * the OpenAI Chat Completions API with tool calling: Ollama, llama.cpp
4
+ * `llama-server`, LM Studio, vLLM, `qvac serve`, or a hosted API.
5
+ *
6
+ * No dependencies: it uses `fetch` (pass your own for older runtimes or tests).
7
+ *
8
+ * The Engine keeps history as plain `{ role, content }` messages. Tool calls
9
+ * are written into the assistant's `rawContent` as `<tool_call>{json}</tool_call>`
10
+ * blocks and turned back into `tool_calls` / `tool_call_id` pairs on the next
11
+ * request, so the server always receives a valid tool-calling conversation.
12
+ */
13
+ import type { Message, ToolCall } from '../types.js';
14
+ import type { InferenceMetrics, LLMProvider, ToolCallError, ToolChoice, TurnInput, TurnOutput } from './types.js';
15
+ import { extractTextToolCalls } from '../qvac/parse.js';
16
+ import { toolParametersSchema } from '../qvac/tools.js';
17
+
18
+ export interface OpenAICompatibleOptions {
19
+ /** API root including the version, e.g. `http://localhost:11434/v1`. */
20
+ baseUrl: string;
21
+ /** Model name as the server knows it, e.g. `qwen3.5:4b`. */
22
+ model: string;
23
+ /** Sent as `Authorization: Bearer …` when set. */
24
+ apiKey?: string;
25
+ defaultTemperature?: number;
26
+ defaultMaxTokens?: number;
27
+ /** Stream tokens (default true). Set false for servers without SSE. */
28
+ stream?: boolean;
29
+ /** Extra JSON merged into every request body (e.g. `{ reasoning_effort: 'low' }`). */
30
+ extraBody?: Record<string, unknown>;
31
+ /** Extra request headers. */
32
+ headers?: Record<string, string>;
33
+ /** Reasoning deltas (`reasoning_content` / `reasoning`), when the server sends them. */
34
+ onThinking?: (token: string) => void;
35
+ fetch?: typeof fetch;
36
+ }
37
+
38
+ export interface OpenAITurnInput extends TurnInput {
39
+ temperature?: number;
40
+ maxTokens?: number;
41
+ }
42
+
43
+ type WireMessage =
44
+ | { role: 'system' | 'user'; content: string }
45
+ | { role: 'assistant'; content: string | null; tool_calls?: WireToolCall[] }
46
+ | { role: 'tool'; content: string; tool_call_id: string };
47
+
48
+ interface WireToolCall {
49
+ id: string;
50
+ type: 'function';
51
+ function: { name: string; arguments: string };
52
+ }
53
+
54
+ const TOOL_CALL_BLOCK = /<tool_call\b[^>]*>[\s\S]*?<\/tool_call>/gi;
55
+
56
+ /** The assistant frame the Engine stores: visible text plus one block per call. */
57
+ export function encodeToolCalls(text: string, calls: ToolCall[]): string {
58
+ const blocks = calls.map((c) => `<tool_call>${JSON.stringify({ name: c.name, arguments: c.arguments })}</tool_call>`);
59
+ return [text.trim(), ...blocks].filter(Boolean).join('\n');
60
+ }
61
+
62
+ /** Engine history → Chat Completions messages, with ids pairing calls and results. */
63
+ export function toWireMessages(messages: Message[], system?: string): WireMessage[] {
64
+ const out: WireMessage[] = system ? [{ role: 'system', content: system }] : [];
65
+ let pending: string[] = [];
66
+ messages.forEach((m, i) => {
67
+ if (m.role === 'assistant') {
68
+ const calls = extractTextToolCalls(m.content);
69
+ const text = m.content.replace(TOOL_CALL_BLOCK, '').trim();
70
+ pending = calls.map((_, j) => `call_${i}_${j}`);
71
+ out.push(
72
+ calls.length
73
+ ? {
74
+ role: 'assistant',
75
+ content: text || null,
76
+ tool_calls: calls.map((c, j) => ({
77
+ id: pending[j]!,
78
+ type: 'function',
79
+ function: { name: c.name, arguments: JSON.stringify(c.arguments) },
80
+ })),
81
+ }
82
+ : { role: 'assistant', content: m.content },
83
+ );
84
+ } else if (m.role === 'tool') {
85
+ const id = pending.shift();
86
+ // A tool message with no call to answer (e.g. a parse-error note) would
87
+ // be rejected by the server; send it as user context instead.
88
+ out.push(id ? { role: 'tool', content: m.content, tool_call_id: id } : { role: 'user', content: `Tool result: ${m.content}` });
89
+ } else {
90
+ pending = [];
91
+ out.push({ role: m.role as 'system' | 'user', content: m.content });
92
+ }
93
+ });
94
+ return out;
95
+ }
96
+
97
+ function wireToolChoice(choice: ToolChoice | undefined): unknown {
98
+ if (!choice) return undefined;
99
+ if (choice === 'auto' || choice === 'none' || choice === 'required') return choice;
100
+ return { type: 'function', function: { name: choice } };
101
+ }
102
+
103
+ interface Accumulated {
104
+ content: string;
105
+ calls: Map<number, { id?: string; name: string; args: string }>;
106
+ finishReason?: string;
107
+ usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number };
108
+ }
109
+
110
+ function absorbDelta(acc: Accumulated, delta: any, onToken?: (t: string) => void, onThinking?: (t: string) => void) {
111
+ if (typeof delta?.content === 'string' && delta.content) {
112
+ acc.content += delta.content;
113
+ onToken?.(delta.content);
114
+ }
115
+ const reasoning = delta?.reasoning_content ?? delta?.reasoning;
116
+ if (typeof reasoning === 'string' && reasoning) onThinking?.(reasoning);
117
+ for (const tc of delta?.tool_calls ?? []) {
118
+ const index = typeof tc.index === 'number' ? tc.index : acc.calls.size;
119
+ const cur = acc.calls.get(index) ?? { name: '', args: '' };
120
+ if (tc.id) cur.id = tc.id;
121
+ if (tc.function?.name) cur.name += tc.function.name;
122
+ if (tc.function?.arguments) {
123
+ cur.args += typeof tc.function.arguments === 'string' ? tc.function.arguments : JSON.stringify(tc.function.arguments);
124
+ }
125
+ acc.calls.set(index, cur);
126
+ }
127
+ }
128
+
129
+ async function readSse(body: ReadableStream<Uint8Array>, onEvent: (data: string) => void): Promise<void> {
130
+ const reader = body.getReader();
131
+ const decoder = new TextDecoder();
132
+ let buffer = '';
133
+ for (;;) {
134
+ const { value, done } = await reader.read();
135
+ if (done) break;
136
+ buffer += decoder.decode(value, { stream: true });
137
+ let nl: number;
138
+ while ((nl = buffer.indexOf('\n')) >= 0) {
139
+ const line = buffer.slice(0, nl).trim();
140
+ buffer = buffer.slice(nl + 1);
141
+ if (line.startsWith('data:')) onEvent(line.slice(5).trim());
142
+ }
143
+ }
144
+ const last = buffer.trim();
145
+ if (last.startsWith('data:')) onEvent(last.slice(5).trim());
146
+ }
147
+
148
+ export function createOpenAICompatibleProvider(options: OpenAICompatibleOptions): LLMProvider {
149
+ const doFetch = options.fetch ?? globalThis.fetch;
150
+ if (!doFetch) throw new Error('No fetch available; pass options.fetch');
151
+ const url = `${options.baseUrl.replace(/\/+$/, '')}/chat/completions`;
152
+ const inflight = new Map<string, AbortController>();
153
+ let seq = 0;
154
+
155
+ return {
156
+ name: 'openai-compatible',
157
+
158
+ async runTurn(input: OpenAITurnInput): Promise<TurnOutput> {
159
+ const requestId = `oai-${Date.now().toString(36)}-${++seq}`;
160
+ const controller = new AbortController();
161
+ inflight.set(requestId, controller);
162
+ const onAbort = () => controller.abort();
163
+ input.signal?.addEventListener('abort', onAbort, { once: true });
164
+ if (input.signal?.aborted) controller.abort();
165
+
166
+ const stream = options.stream ?? true;
167
+ const temperature = input.temperature ?? options.defaultTemperature;
168
+ const maxTokens = input.maxTokens ?? options.defaultMaxTokens;
169
+ const tools = input.tools.map((t) => ({
170
+ type: 'function' as const,
171
+ function: { name: t.name, description: t.description ?? '', parameters: toolParametersSchema(t) },
172
+ }));
173
+ const toolChoice = tools.length ? wireToolChoice(input.toolChoice) : undefined;
174
+ const body = {
175
+ model: options.model,
176
+ messages: toWireMessages(input.messages, input.system),
177
+ stream,
178
+ ...(stream ? { stream_options: { include_usage: true } } : {}),
179
+ ...(temperature !== undefined ? { temperature } : {}),
180
+ ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
181
+ ...(tools.length ? { tools } : {}),
182
+ ...(toolChoice ? { tool_choice: toolChoice } : {}),
183
+ ...options.extraBody,
184
+ };
185
+
186
+ const startedAt = Date.now();
187
+ let firstTokenAt: number | undefined;
188
+ const mark = (f?: (t: string) => void) => (t: string) => {
189
+ firstTokenAt ??= Date.now();
190
+ f?.(t);
191
+ };
192
+ const acc: Accumulated = { content: '', calls: new Map() };
193
+ let cancelled = false;
194
+ try {
195
+ const res = await doFetch(url, {
196
+ method: 'POST',
197
+ headers: {
198
+ 'content-type': 'application/json',
199
+ ...(options.apiKey ? { authorization: `Bearer ${options.apiKey}` } : {}),
200
+ ...options.headers,
201
+ },
202
+ body: JSON.stringify(body),
203
+ signal: controller.signal,
204
+ });
205
+ if (!res.ok) {
206
+ const detail = await res.text().catch(() => '');
207
+ throw new Error(`${options.baseUrl} returned ${res.status}: ${detail.slice(0, 300)}`);
208
+ }
209
+ if (stream && res.body) {
210
+ await readSse(res.body, (data) => {
211
+ if (data === '[DONE]') return;
212
+ let chunk: any;
213
+ try {
214
+ chunk = JSON.parse(data);
215
+ } catch {
216
+ return;
217
+ }
218
+ if (chunk.usage) acc.usage = chunk.usage;
219
+ const choice = chunk.choices?.[0];
220
+ if (!choice) return;
221
+ absorbDelta(acc, choice.delta, mark(input.onToken), mark(options.onThinking));
222
+ if (choice.finish_reason) acc.finishReason = choice.finish_reason;
223
+ });
224
+ } else {
225
+ const json: any = await res.json();
226
+ const choice = json.choices?.[0];
227
+ absorbDelta(acc, { ...choice?.message, tool_calls: choice?.message?.tool_calls?.map((c: any, index: number) => ({ index, ...c })) }, input.onToken, options.onThinking);
228
+ acc.finishReason = choice?.finish_reason;
229
+ acc.usage = json.usage;
230
+ }
231
+ } catch (err) {
232
+ if (!controller.signal.aborted) throw err;
233
+ cancelled = true;
234
+ } finally {
235
+ inflight.delete(requestId);
236
+ input.signal?.removeEventListener('abort', onAbort);
237
+ }
238
+
239
+ const toolCalls: ToolCall[] = [];
240
+ const toolErrors: ToolCallError[] = [];
241
+ for (const c of [...acc.calls.entries()].sort(([a], [b]) => a - b).map(([, v]) => v)) {
242
+ if (!c.name) continue;
243
+ try {
244
+ const args = c.args.trim() ? JSON.parse(c.args) : {};
245
+ toolCalls.push({ ...(c.id ? { id: c.id } : {}), name: c.name, arguments: args && typeof args === 'object' ? args : {} });
246
+ } catch {
247
+ toolErrors.push({ code: 'PARSE_ERROR', message: `arguments for ${c.name} are not valid JSON`, raw: c.args });
248
+ }
249
+ }
250
+ // Models that ignore the tools API sometimes write the call as text.
251
+ if (!toolCalls.length && !toolErrors.length) {
252
+ for (const c of extractTextToolCalls(acc.content)) toolCalls.push(c);
253
+ }
254
+ const text = acc.content.replace(TOOL_CALL_BLOCK, '').trim();
255
+
256
+ const truncated = acc.finishReason === 'length';
257
+ const inference: InferenceMetrics = {
258
+ requestId,
259
+ durationMs: Date.now() - startedAt,
260
+ status: cancelled ? 'cancelled' : truncated ? 'truncated' : 'completed',
261
+ ...(firstTokenAt !== undefined ? { ttftMs: firstTokenAt - startedAt } : {}),
262
+ ...(acc.usage?.prompt_tokens !== undefined ? { promptTokens: acc.usage.prompt_tokens } : {}),
263
+ ...(acc.usage?.completion_tokens !== undefined ? { completionTokens: acc.usage.completion_tokens } : {}),
264
+ ...(acc.usage?.total_tokens !== undefined ? { totalTokens: acc.usage.total_tokens } : {}),
265
+ ...(acc.finishReason ? { stopReason: acc.finishReason } : {}),
266
+ };
267
+
268
+ return {
269
+ text,
270
+ rawContent: encodeToolCalls(text, toolCalls),
271
+ toolCalls,
272
+ ...(toolErrors.length && !toolCalls.length ? { toolErrors } : {}),
273
+ requestId,
274
+ inference,
275
+ };
276
+ },
277
+
278
+ async cancel(requestId: string): Promise<void> {
279
+ inflight.get(requestId)?.abort();
280
+ },
281
+ };
282
+ }
@@ -17,11 +17,37 @@ export interface TurnInput {
17
17
  tools: ToolDef[];
18
18
  /** System prompt, when not already present as a message. */
19
19
  system?: string;
20
+ /**
21
+ * `'required'` forces a tool call, a tool name forces that tool, `'none'`
22
+ * forbids tools. Omit for the model's own choice. Ignored when `tools` is
23
+ * empty or the provider has no such control.
24
+ */
25
+ toolChoice?: ToolChoice;
26
+ /**
27
+ * `'off'` asks the provider to skip reasoning for this turn (e.g. a forced
28
+ * tool call). Ignored by providers without reasoning control.
29
+ */
30
+ thinking?: 'off';
31
+ /**
32
+ * Same value on every call of one agentic run, whose history only grows.
33
+ * Providers with a session cache (QVAC `kvCache`) can then send only the new
34
+ * message instead of the whole prompt.
35
+ */
36
+ sessionKey?: string;
20
37
  /** Visible content tokens as they stream. */
21
38
  onToken?: (token: string) => void;
22
39
  signal?: AbortSignal;
23
40
  }
24
41
 
42
+ export type ToolChoice = 'auto' | 'none' | 'required' | (string & {});
43
+
44
+ /** A tool-call region the model emitted that the provider could not turn into a call. */
45
+ export interface ToolCallError {
46
+ code: 'PARSE_ERROR' | 'VALIDATION_ERROR' | 'UNKNOWN_TOOL' | (string & {});
47
+ message: string;
48
+ raw?: string;
49
+ }
50
+
25
51
  /** Judge-auditable metrics for one provider inference request. */
26
52
  export interface InferenceMetrics {
27
53
  requestId?: string;
@@ -50,16 +76,25 @@ export interface TurnOutput {
50
76
  rawContent: string;
51
77
  /** Tool calls the model requested this turn (empty ⇒ final answer). */
52
78
  toolCalls: ToolCall[];
79
+ /** Tool-call attempts that failed to parse or validate this turn. */
80
+ toolErrors?: ToolCallError[];
53
81
  /** Provider request id, for cancellation. */
54
82
  requestId?: string;
55
83
  /** Optional local-inference receipt. Hosts may persist this as JSONL evidence. */
56
84
  inference?: InferenceMetrics;
85
+ /**
86
+ * True when the turn produced no visible answer because it ran out of budget
87
+ * (e.g. reasoning used the whole output cap). `text` may hold a placeholder.
88
+ */
89
+ incomplete?: boolean;
57
90
  }
58
91
 
59
92
  export interface LLMProvider {
60
93
  readonly name: string;
61
94
  /** Run one completion turn. */
62
95
  runTurn(input: TurnInput): Promise<TurnOutput>;
96
+ /** Drop whatever the provider cached for `sessionKey` (end of an agentic run). */
97
+ endSession?(sessionKey: string): Promise<void>;
63
98
  /** Cancel an in-flight turn by request id, if the provider supports it. */
64
99
  cancel?(requestId: string): Promise<void>;
65
100
  }
@@ -30,7 +30,7 @@ export const LOCAL_LLM_CONFIG_GPU = {
30
30
 
31
31
  /**
32
32
  * Delegated to a desktop provider — it has the RAM to run a big context, so give
33
- * the agentic prompt plenty of room (Qwen3-600M supports up to 32k). 2048
33
+ * the agentic prompt plenty of room (Qwen3.5 supports far more than this). 2048
34
34
  * overflowed with the system prompt + tool/skill definitions alone.
35
35
  */
36
36
  export const DELEGATE_LLM_CONFIG = {
package/src/qvac/index.ts CHANGED
@@ -23,6 +23,16 @@ export {
23
23
  normalizeWhisperLang,
24
24
  } from './config.js';
25
25
 
26
+ export {
27
+ QWEN35_MODELS,
28
+ DEFAULT_MODEL_ID,
29
+ DEFAULT_SMALL_DEVICE_MODEL_ID,
30
+ DEFAULT_QVAC_MODEL,
31
+ DEFAULT_SMALL_DEVICE_QVAC_MODEL,
32
+ getRecommendedModel,
33
+ type RecommendedModel,
34
+ } from './models.js';
35
+
26
36
  export {
27
37
  finalToTurn,
28
38
  type QvacFinalLike,
@@ -0,0 +1,98 @@
1
+ /**
2
+ * Recommended local models. Plain data (no SDK import): `qvacConstant` is the
3
+ * name of the matching @qvac/sdk registry export (present since @qvac/sdk
4
+ * 0.13.1), `hfRepo`/`hfFile` the same GGUF on Hugging Face for hosts that
5
+ * download directly. Sizes are the exact file sizes.
6
+ */
7
+
8
+ export interface RecommendedModel {
9
+ id: string;
10
+ family: string;
11
+ displayName: string;
12
+ /** Name of the @qvac/sdk model constant, e.g. `QWEN3_5_4B_MULTIMODAL_Q4_K_M`. */
13
+ qvacConstant: string;
14
+ quant: string;
15
+ sizeBytes: number;
16
+ hfRepo: string;
17
+ hfFile: string;
18
+ /** Rough RAM needed to run it with the agent's context window. */
19
+ ramHintGb: number;
20
+ notes: string;
21
+ }
22
+
23
+ export const QWEN35_MODELS: readonly RecommendedModel[] = [
24
+ {
25
+ id: 'qwen3.5-0.8b-q4_k_m',
26
+ family: 'qwen3.5',
27
+ displayName: 'Qwen 3.5 · 0.8B',
28
+ qvacConstant: 'QWEN3_5_0_8B_MULTIMODAL_Q4_K_M',
29
+ quant: 'Q4_K_M',
30
+ sizeBytes: 532_517_120,
31
+ hfRepo: 'unsloth/Qwen3.5-0.8B-GGUF',
32
+ hfFile: 'Qwen3.5-0.8B-Q4_K_M.gguf',
33
+ ramHintGb: 1.5,
34
+ notes: 'Smoke tests only. Loops on wallet actions in our signet bench; fine for chat and single read-only calls.',
35
+ },
36
+ {
37
+ id: 'qwen3.5-2b-q4_k_m',
38
+ family: 'qwen3.5',
39
+ displayName: 'Qwen 3.5 · 2B',
40
+ qvacConstant: 'QWEN3_5_2B_MULTIMODAL_Q4_K_M',
41
+ quant: 'Q4_K_M',
42
+ sizeBytes: 1_280_835_840,
43
+ hfRepo: 'unsloth/Qwen3.5-2B-GGUF',
44
+ hfFile: 'Qwen3.5-2B-Q4_K_M.gguf',
45
+ ramHintGb: 3,
46
+ notes: 'Recommended default. Passed all 7 wallet tasks of our signet RGB bench at 45–150 s per question on an M4 laptop; also fits phones.',
47
+ },
48
+ {
49
+ id: 'qwen3.5-4b-q4_k_m',
50
+ family: 'qwen3.5',
51
+ displayName: 'Qwen 3.5 · 4B',
52
+ qvacConstant: 'QWEN3_5_4B_MULTIMODAL_Q4_K_M',
53
+ quant: 'Q4_K_M',
54
+ sizeBytes: 2_740_937_888,
55
+ hfRepo: 'unsloth/Qwen3.5-4B-GGUF',
56
+ hfFile: 'Qwen3.5-4B-Q4_K_M.gguf',
57
+ ramHintGb: 5,
58
+ notes: 'Same correctness as 2B in our bench, about twice as slow (105–330 s per question on an M4). Pick it for longer free-form answers.',
59
+ },
60
+ {
61
+ id: 'qwen3.5-9b-q4_k_m',
62
+ family: 'qwen3.5',
63
+ displayName: 'Qwen 3.5 · 9B',
64
+ qvacConstant: 'QWEN3_5_9B_MULTIMODAL_Q4_K_M',
65
+ quant: 'Q4_K_M',
66
+ sizeBytes: 5_680_522_464,
67
+ hfRepo: 'unsloth/Qwen3.5-9B-GGUF',
68
+ hfFile: 'Qwen3.5-9B-Q4_K_M.gguf',
69
+ ramHintGb: 9,
70
+ notes: 'Needs 16 GB of RAM. Slow for interactive use (180–690 s per question on an M4); for unattended multi-step tasks.',
71
+ },
72
+ {
73
+ id: 'qwen3.6-35b-a3b-q4_k_m',
74
+ family: 'qwen3.6',
75
+ displayName: 'Qwen 3.6 · 35B-A3B (MoE)',
76
+ qvacConstant: 'QWEN3_6_35B_A3B_MULTIMODAL_Q4_K_M',
77
+ quant: 'UD-Q4_K_M',
78
+ sizeBytes: 22_134_528_992,
79
+ hfRepo: 'unsloth/Qwen3.6-35B-A3B-GGUF',
80
+ hfFile: 'Qwen3.6-35B-A3B-UD-Q4_K_M.gguf',
81
+ ramHintGb: 26,
82
+ notes: 'Big machines only (32 GB+). MoE with ~3B active parameters: best quality, 22 GB download.',
83
+ },
84
+ ];
85
+
86
+ /** Default model for every host (provider sidecar, CLI, examples). */
87
+ export const DEFAULT_MODEL_ID = 'qwen3.5-2b-q4_k_m';
88
+ /** Default model for phones and other small devices. */
89
+ export const DEFAULT_SMALL_DEVICE_MODEL_ID = 'qwen3.5-2b-q4_k_m';
90
+
91
+ export function getRecommendedModel(id: string): RecommendedModel | undefined {
92
+ return QWEN35_MODELS.find((m) => m.id === id);
93
+ }
94
+
95
+ /** @qvac/sdk constant name of the default, e.g. for `sdk[DEFAULT_QVAC_MODEL]`. */
96
+ export const DEFAULT_QVAC_MODEL = getRecommendedModel(DEFAULT_MODEL_ID)!.qvacConstant;
97
+ /** @qvac/sdk constant name of the small-device default. */
98
+ export const DEFAULT_SMALL_DEVICE_QVAC_MODEL = getRecommendedModel(DEFAULT_SMALL_DEVICE_MODEL_ID)!.qvacConstant;
@@ -114,6 +114,13 @@ describe('finalToTurn', () => {
114
114
  ]);
115
115
  });
116
116
 
117
+ it('recovers a Qwen3.5 XML-style call', () => {
118
+ const calls = extractTextToolCalls(
119
+ '<tool_call>\n<function=rln_issue_asset>\n<parameter=name>\nHack\n</parameter>\n<parameter=amount>\n1000\n</parameter>\n</function>\n</tool_call>',
120
+ );
121
+ expect(calls).toEqual([{ name: 'rln_issue_asset', arguments: { name: 'Hack', amount: 1000 } }]);
122
+ });
123
+
117
124
  it('returns [] for plain prose', () => {
118
125
  expect(extractTextToolCalls('just a normal answer')).toEqual([]);
119
126
  });
package/src/qvac/parse.ts CHANGED
@@ -4,6 +4,7 @@
4
4
  * is testable without loading a model, and so the same mapping runs on mobile,
5
5
  * desktop, and the eval harness.
6
6
  */
7
+ import type { ToolCallError } from '../providers/types.js';
7
8
  import { cleanAssistantVisibleText } from './text.js';
8
9
 
9
10
  /**
@@ -32,6 +33,8 @@ export interface QvacFinalLike {
32
33
  raw?: { fullText?: string };
33
34
  /** Tool calls the model requested this turn (empty ⇒ final answer). */
34
35
  toolCalls?: Array<{ id?: string; name: string; arguments?: Record<string, unknown> }>;
36
+ /** Tool-call regions that failed to parse or validate (QVAC 0.20+; omitted when none). */
37
+ toolErrors?: ToolCallError[];
35
38
  /**
36
39
  * Why generation stopped: `"length"` when the token budget is exhausted,
37
40
  * `"cancelled"` on abort, `"eos"`/`"stopSequence"`/`undefined` on a natural stop. We surface
@@ -49,6 +52,8 @@ export interface ParsedTurn {
49
52
  rawContent: string;
50
53
  /** Tool calls the model requested (arguments defaulted to `{}`). */
51
54
  toolCalls: Array<{ id?: string; name: string; arguments: Record<string, unknown> }>;
55
+ /** Tool-call attempts the SDK could not parse, when no call was recovered from text. */
56
+ toolErrors?: ToolCallError[];
52
57
  /** True when generation was cut off by the token budget (incomplete output). */
53
58
  truncated: boolean;
54
59
  /** Raw stop reason from the SDK, when provided. */
@@ -89,6 +94,31 @@ function parseCallObject(
89
94
  return null;
90
95
  }
91
96
 
97
+ /**
98
+ * Parse Qwen3.5's XML call body:
99
+ * `<function=name><parameter=key>value</parameter>…</function>`. Values that
100
+ * read as JSON (numbers, booleans, arrays, objects) are decoded; the rest stay
101
+ * strings.
102
+ */
103
+ function parseXmlCall(s: string): { name: string; arguments: Record<string, unknown> } | null {
104
+ const fn = s.match(/<function=([^>\s]+)\s*>([\s\S]*?)(?:<\/function>|$)/i);
105
+ if (!fn?.[1]) return null;
106
+ const args: Record<string, unknown> = {};
107
+ for (const m of (fn[2] ?? '').matchAll(/<parameter=([^>\s]+)\s*>([\s\S]*?)<\/parameter>/gi)) {
108
+ const raw = (m[2] ?? '').trim();
109
+ let value: unknown = raw;
110
+ if (/^(-?\d+(\.\d+)?|true|false|null|\[[\s\S]*\]|\{[\s\S]*\})$/.test(raw)) {
111
+ try {
112
+ value = JSON.parse(raw);
113
+ } catch {
114
+ value = raw;
115
+ }
116
+ }
117
+ args[m[1]!] = value;
118
+ }
119
+ return { name: fn[1], arguments: args };
120
+ }
121
+
92
122
  /**
93
123
  * Recover tool calls a model emitted as PLAIN TEXT instead of structured frames
94
124
  * — `<tool_call>{"name":…,"arguments":…}</tool_call>` (Qwen/Hermes) or a bare
@@ -101,7 +131,8 @@ export function extractTextToolCalls(
101
131
  ): Array<{ name: string; arguments: Record<string, unknown> }> {
102
132
  const calls: Array<{ name: string; arguments: Record<string, unknown> }> = [];
103
133
  for (const m of text.matchAll(/<tool_call\b[^>]*>([\s\S]*?)<\/tool_call>/gi)) {
104
- const c = parseCallObject(m[1] ?? '');
134
+ const body = m[1] ?? '';
135
+ const c = parseCallObject(body) ?? parseXmlCall(body);
105
136
  if (c) calls.push(c);
106
137
  }
107
138
  if (calls.length) return calls;
@@ -139,6 +170,7 @@ export function finalToTurn(final: QvacFinalLike, streamed = ''): ParsedTurn {
139
170
  text,
140
171
  rawContent: final.raw?.fullText ?? rawText,
141
172
  toolCalls,
173
+ ...(toolCalls.length === 0 && final.toolErrors?.length ? { toolErrors: final.toolErrors } : {}),
142
174
  truncated: final.stopReason === 'length',
143
175
  stopReason: final.stopReason,
144
176
  stats: final.stats,
@@ -100,7 +100,7 @@ describe('createQvacProvider.runTurn', () => {
100
100
  const cancel = vi.fn(async () => {});
101
101
  const { fn } = fakeCompletion(
102
102
  { contentText: '', toolCalls: [], raw: { fullText: '' }, stopReason: 'cancelled' },
103
- [{ type: 'thinkingDelta', text: 'z'.repeat(40) }], // ~10 tokens, budget 4
103
+ [{ type: 'thinkingDelta', text: 'z'.repeat(400) }], // ~100 tokens, budget 4 (+ backstop headroom)
104
104
  );
105
105
  const p = createQvacProvider({
106
106
  completion: fn as any,
@@ -113,6 +113,59 @@ describe('createQvacProvider.runTurn', () => {
113
113
  expect(out.text).toMatch(/thinking budget/i);
114
114
  });
115
115
 
116
+ it('sends the thinking cap as the SDK reasoning_budget', async () => {
117
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
118
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', maxThinkingTokens: 128 });
119
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
120
+ expect(calls[0].generationParams).toEqual({ reasoning_budget: 128 });
121
+ });
122
+
123
+ it("sends reasoning_budget 0 when a turn asks for thinking 'off'", async () => {
124
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
125
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', maxThinkingTokens: 128 });
126
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], thinking: 'off' });
127
+ expect(calls[0].generationParams).toEqual({ reasoning_budget: 0 });
128
+ });
129
+
130
+ it('uses the session key as kvCache when sessionCache is on, and deletes it at the end', async () => {
131
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
132
+ const deleted: unknown[] = [];
133
+ const on = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', sessionCache: true, deleteCache: (async (p: unknown) => void deleted.push(p)) as any });
134
+ const off = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
135
+ await on.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], sessionKey: 'run-1' });
136
+ await off.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], sessionKey: 'run-1' });
137
+ await on.endSession!('run-1');
138
+ expect(calls[0].kvCache).toBe('run-1');
139
+ expect(calls[1].kvCache).toBeUndefined();
140
+ expect(deleted).toEqual([{ kvCacheKey: 'run-1' }]);
141
+ });
142
+
143
+ it('keeps the reasoning budget below the output cap', async () => {
144
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
145
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1', defaultMaxTokens: 512, maxThinkingTokens: 512 });
146
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
147
+ expect(calls[0].generationParams).toEqual({ predict: 512, reasoning_budget: 256 });
148
+ });
149
+
150
+ it('forwards toolChoice only when tools are present', async () => {
151
+ const tool = { name: 'get_balance', description: 'b', parameters: { type: 'object', properties: {} } };
152
+ const { fn, calls } = fakeCompletion({ contentText: 'ok', toolCalls: [], raw: { fullText: 'ok' } });
153
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
154
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [tool as any], toolChoice: 'required' });
155
+ await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [], toolChoice: 'required' });
156
+ expect(calls[0].generationParams).toEqual({ tool_choice: 'required' });
157
+ expect(calls[1].generationParams).toBeUndefined();
158
+ });
159
+
160
+ it('returns toolErrors when the model emitted a tool call that did not parse', async () => {
161
+ const toolErrors = [{ code: 'PARSE_ERROR', message: 'unterminated string', raw: '{"ticker":"HCK' }];
162
+ const { fn } = fakeCompletion({ contentText: '', toolCalls: [], toolErrors, raw: { fullText: '<tool_call>{"ticker":"HCK' } });
163
+ const p = createQvacProvider({ completion: fn as any, cancel: noopCancel, getModelId: () => 'm1' });
164
+ const out = await p.runTurn({ messages: [{ role: 'user', content: 'x' }], tools: [] });
165
+ expect(out.toolCalls).toEqual([]);
166
+ expect(out.toolErrors).toEqual(toolErrors);
167
+ });
168
+
116
169
  it('returns a cancelled turn when the SDK rejects final on abort', async () => {
117
170
  const cancel = vi.fn(async () => {});
118
171
  const fn = () => ({