@vierratale/ai 0.1.0-beta.14 → 0.1.0-beta.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,25 +2,6 @@ import { BaseProvider } from './base.js';
2
2
  import { Catalog } from '../catalog.js';
3
3
  import { Config } from '../config.js';
4
4
 
5
- // Keep the prompt a local engine must re-process per call bounded: the
6
- // system prompt plus the most recent messages. Sending the whole growing
7
- // history makes prompt-processing time unbounded on slow hardware.
8
- const MAX_CONTEXT_MESSAGES = 12;
9
-
10
- function buildEngineMessages(systemPrompt, messages) {
11
- const engineMessages = [];
12
- if (systemPrompt) {
13
- engineMessages.push({ role: 'system', content: systemPrompt });
14
- }
15
- for (const msg of messages) {
16
- engineMessages.push({ role: msg.role, content: msg.content });
17
- }
18
- if (engineMessages.length > MAX_CONTEXT_MESSAGES) {
19
- return [engineMessages[0], ...engineMessages.slice(-(MAX_CONTEXT_MESSAGES - 1))];
20
- }
21
- return engineMessages;
22
- }
23
-
24
5
  export class CortexProvider extends BaseProvider {
25
6
  constructor() {
26
7
  super('cortex');
@@ -61,238 +42,74 @@ export class CortexProvider extends BaseProvider {
61
42
 
62
43
  async *stream(messages, options = {}) {
63
44
  const model = Catalog.getRealModel(options.model || Config.get('model'));
64
- const engineMessages = buildEngineMessages(options.systemPrompt || '', messages);
45
+ const systemPrompt = options.systemPrompt || '';
65
46
 
66
- const controller = new AbortController();
67
- const connectTimer = setTimeout(() => controller.abort(), 600000);
68
-
69
- let resp;
70
- try {
71
- resp = await fetch(`${this.host}/api/chat`, {
72
- method: 'POST',
73
- headers: { 'Content-Type': 'application/json' },
74
- body: JSON.stringify({
75
- model,
76
- messages: engineMessages,
77
- stream: true,
78
- keep_alive: Config.get('keepAlive'),
79
- options: {
80
- num_ctx: Config.get('numCtx'),
81
- num_predict: options.maxTokens ?? Config.get('maxTokens'),
82
- temperature: options.temperature || Config.get('temperature'),
83
- num_thread: Config.get('numThreads'),
84
- },
85
- }),
86
- signal: controller.signal,
87
- });
88
- } catch (err) {
89
- if (err.name === 'AbortError') {
90
- throw new Error('[ERR-0001] Could not reach the engine (no response within 600s). Make sure the local engine is running.');
91
- }
92
- throw new Error('[ERR-0001] Could not reach the engine (no response). Make sure the local engine is running.');
93
- } finally {
94
- clearTimeout(connectTimer);
47
+ const engineMessages = [];
48
+ if (systemPrompt) {
49
+ engineMessages.push({ role: 'system', content: systemPrompt });
50
+ }
51
+ for (const msg of messages) {
52
+ engineMessages.push({ role: msg.role, content: msg.content });
95
53
  }
96
54
 
55
+ const resp = await fetch(`${this.host}/api/chat`, {
56
+ method: 'POST',
57
+ headers: { 'Content-Type': 'application/json' },
58
+ body: JSON.stringify({
59
+ model,
60
+ messages: engineMessages,
61
+ stream: true,
62
+ keep_alive: Config.get('keepAlive'),
63
+ options: {
64
+ num_ctx: Config.get('numCtx'),
65
+ temperature: options.temperature || Config.get('temperature'),
66
+ },
67
+ }),
68
+ });
69
+
97
70
  if (!resp.ok) {
98
- throw new Error(await describeEngineError(resp, model));
71
+ throw new Error('Failed to connect to backend engine');
99
72
  }
100
73
 
101
74
  const reader = resp.body.getReader();
102
75
  const decoder = new TextDecoder();
103
76
  let buffer = '';
104
- let idle = null;
105
- const settle = () => {
106
- if (idle) clearTimeout(idle);
107
- idle = setTimeout(() => controller.abort(), 420000);
108
- };
109
- settle();
110
- try {
111
- while (true) {
112
- let value, done;
113
- try {
114
- ({ done, value } = await reader.read());
115
- } catch (err) {
116
- if (err.name === 'AbortError') {
117
- throw new Error('[ERR-0001] Lost the connection to the engine (no data for 420s).');
118
- }
119
- throw err;
120
- }
121
- if (done) break;
122
- settle();
123
- buffer += decoder.decode(value, { stream: true });
124
- const lines = buffer.split('\n');
125
- buffer = lines.pop() || '';
126
-
127
- for (const line of lines) {
128
- if (!line.trim()) continue;
129
- try {
130
- const json = JSON.parse(line);
131
- if (json.message?.content) {
132
- yield json.message.content;
133
- }
134
- if (json.done) return;
135
- } catch {}
136
- }
137
- }
138
- } finally {
139
- if (idle) clearTimeout(idle);
140
- }
141
- }
142
-
143
- // Non-streaming completion (used by the tool-planning step). Reads the
144
- // streamed reply incrementally: on a slow engine a full answer can take
145
- // minutes, so only a connection with no response for 180s (or an idle
146
- // stream for 120s) counts as a failure - not a slowly progressing one.
147
- async complete(messages, options = {}) {
148
- const model = Catalog.getRealModel(options.model || Config.get('model'));
149
- const engineMessages = buildEngineMessages(options.systemPrompt || '', messages);
150
-
151
- const controller = new AbortController();
152
- const connectTimer = setTimeout(() => controller.abort(), 600000);
153
-
154
- if (options.signal) {
155
- if (options.signal.aborted) controller.abort();
156
- else options.signal.addEventListener('abort', () => controller.abort(), { once: true });
157
- }
158
77
 
159
- let resp;
160
- try {
161
- resp = await fetch(`${this.host}/api/chat`, {
162
- method: 'POST',
163
- headers: { 'Content-Type': 'application/json' },
164
- body: JSON.stringify({
165
- model,
166
- messages: engineMessages,
167
- stream: true,
168
- keep_alive: Config.get('keepAlive'),
169
- options: {
170
- num_ctx: Config.get('numCtx'),
171
- num_predict: options.maxTokens ?? Config.get('maxTokens'),
172
- temperature: options.temperature || Config.get('temperature'),
173
- num_thread: Config.get('numThreads'),
174
- },
175
- }),
176
- signal: controller.signal,
177
- });
178
- } catch (err) {
179
- if (err.name === 'AbortError') {
180
- throw new Error('[ERR-0001] Could not complete the request (no response within 600s).');
181
- }
182
- throw new Error('[ERR-0001] Could not reach the engine (no response). Make sure the local engine is running.');
183
- } finally {
184
- clearTimeout(connectTimer);
185
- }
78
+ while (true) {
79
+ const { done, value } = await reader.read();
80
+ if (done) break;
186
81
 
187
- if (!resp.ok) {
188
- throw new Error(await describeEngineError(resp, model));
189
- }
82
+ buffer += decoder.decode(value, { stream: true });
83
+ const lines = buffer.split('\n');
84
+ buffer = lines.pop() || '';
190
85
 
191
- const reader = resp.body.getReader();
192
- const decoder = new TextDecoder();
193
- let buffer = '';
194
- let idle = null;
195
- const settle = () => {
196
- if (idle) clearTimeout(idle);
197
- idle = setTimeout(() => controller.abort(), 120000);
198
- };
199
- settle();
200
- let output = '';
201
- try {
202
- while (true) {
203
- let value, done;
86
+ for (const line of lines) {
87
+ if (!line.trim()) continue;
204
88
  try {
205
- ({ done, value } = await reader.read());
206
- } catch (err) {
207
- if (err.name === 'AbortError') {
208
- throw new Error('[ERR-0001] Lost the connection to the engine (no data for 420s).');
89
+ const json = JSON.parse(line);
90
+ if (json.message?.content) {
91
+ yield json.message.content;
209
92
  }
210
- throw err;
211
- }
212
- if (done) break;
213
- settle();
214
- buffer += decoder.decode(value, { stream: true });
215
- const lines = buffer.split('\n');
216
- buffer = lines.pop() || '';
217
-
218
- for (const line of lines) {
219
- if (!line.trim()) continue;
220
- try {
221
- const json = JSON.parse(line);
222
- if (json.message?.content) {
223
- output += json.message.content;
224
- }
225
- if (json.done) {
226
- buffer = '';
227
- break;
228
- }
229
- } catch {}
230
- }
93
+ if (json.done) return;
94
+ } catch {}
231
95
  }
232
- } finally {
233
- if (idle) clearTimeout(idle);
234
- controller.abort();
235
96
  }
236
- return output;
237
97
  }
238
98
 
239
99
  async warmup() {
240
- // Fire-and-forget: ask the engine to load the model eagerly, then abort as
241
- // soon as generation starts. The engine runs with a single slot, so an
242
- // in-flight non-streaming 'hi' would block the first real request.
100
+ // Fire-and-forget: send a tiny request so ollama loads the model into
101
+ // memory eagerly, so the first real query doesn't pay the cold-load cost.
243
102
  const model = Catalog.getRealModel(Config.get('model'));
244
- const controller = new AbortController();
245
103
  fetch(`${this.host}/api/chat`, {
246
104
  method: 'POST',
247
105
  headers: { 'Content-Type': 'application/json' },
248
106
  body: JSON.stringify({
249
107
  model,
250
108
  messages: [{ role: 'user', content: 'hi' }],
251
- stream: true,
109
+ stream: false,
252
110
  keep_alive: Config.get('keepAlive'),
253
111
  options: { num_ctx: Config.get('numCtx') },
254
112
  }),
255
- signal: controller.signal,
256
- })
257
- .then(async (resp) => {
258
- if (!resp.ok || !resp.body) return;
259
- const reader = resp.body.getReader();
260
- const decoder = new TextDecoder();
261
- let buffer = '';
262
- for (let i = 0; i < 100; i++) {
263
- const { done, value } = await reader.read();
264
- if (done) break;
265
- buffer += decoder.decode(value, { stream: true });
266
- const lines = buffer.split('\n');
267
- buffer = lines.pop() || '';
268
- for (const line of lines) {
269
- if (!line.trim()) continue;
270
- try {
271
- if (JSON.parse(line).message?.content) {
272
- controller.abort();
273
- return;
274
- }
275
- } catch {}
276
- }
277
- }
278
- })
279
- .catch(() => {});
280
- }
281
- }
282
-
283
- // Build a user-actionable message when the engine rejects a request (e.g. a
284
- // cloud model name posted to a local engine -> 404 "model not found"). The
285
- // engine host is deliberately kept out of the message; errors carry codes.
286
- async function describeEngineError(resp, model) {
287
- if (/not found|does not exist|model.*missing/i.test(String(resp.statusText))) {
288
- return `[ERR-0002] Model "${model}" is not installed on the engine. If you picked a cloud model, use /model VRTL-6.pro or /provider openai.`;
289
- }
290
- let body = '';
291
- try {
292
- body = (await resp.text()).slice(0, 300);
293
- } catch {}
294
- if (/not found|does not exist|model.*missing/i.test(body)) {
295
- return `[ERR-0002] Model "${model}" is not installed on the engine. If you picked a cloud model, use /model VRTL-6.pro or /provider openai.`;
113
+ }).catch(() => {});
296
114
  }
297
- return `[ERR-0003] The engine returned HTTP ${resp.status}. Check the engine status or run /model VRTL-2.fast.`;
298
115
  }
@@ -39,7 +39,7 @@ export class GeminiProvider extends BaseProvider {
39
39
 
40
40
  async *stream(messages, options = {}) {
41
41
  const key = Config.get('geminiApiKey');
42
- if (!key) throw new Error('[ERR-0004] Cloud API key not configured. Add it to the config and try again.');
42
+ if (!key) throw new Error('Gemini API key not configured');
43
43
 
44
44
  const model = this._resolveModel(options);
45
45
  const systemPrompt = options.systemPrompt || '';
@@ -52,77 +52,46 @@ export class GeminiProvider extends BaseProvider {
52
52
  contents.unshift({ role: 'user', parts: [{ text: `System: ${systemPrompt}` }] });
53
53
  }
54
54
 
55
- const controller = new AbortController();
56
- const connectTimer = setTimeout(() => controller.abort(), 240000);
57
-
58
- let resp;
59
- try {
60
- resp = await fetch(
61
- `${GEMINI_URL}/${model}:streamGenerateContent?key=${key}&alt=sse`,
62
- {
63
- method: 'POST',
64
- headers: { 'Content-Type': 'application/json' },
65
- body: JSON.stringify({
66
- contents,
67
- generationConfig: {
68
- temperature: options.temperature || Config.get('temperature'),
69
- maxOutputTokens: options.maxTokens || Config.get('maxTokens'),
70
- },
71
- }),
72
- signal: controller.signal,
73
- }
74
- );
75
- } catch (err) {
76
- if (err.name === 'AbortError') {
77
- throw new Error('[ERR-0001] Could not reach the API (no response within 60s). Check your network connection.');
55
+ const resp = await fetch(
56
+ `${GEMINI_URL}/${model}:streamGenerateContent?key=${key}&alt=sse`,
57
+ {
58
+ method: 'POST',
59
+ headers: { 'Content-Type': 'application/json' },
60
+ body: JSON.stringify({
61
+ contents,
62
+ generationConfig: {
63
+ temperature: options.temperature || Config.get('temperature'),
64
+ maxOutputTokens: options.maxTokens || Config.get('maxTokens'),
65
+ },
66
+ }),
78
67
  }
79
- throw new Error('[ERR-0001] Could not reach the API (no response). Check your network connection.');
80
- } finally {
81
- clearTimeout(connectTimer);
82
- }
68
+ );
83
69
 
84
70
  if (!resp.ok) {
85
- throw new Error('[ERR-0003] The cloud provider returned HTTP ' + resp.status + '. Check your API key in config.');
71
+ throw new Error('Failed to connect to cloud engine');
86
72
  }
87
73
 
88
74
  const reader = resp.body.getReader();
89
75
  const decoder = new TextDecoder();
90
76
  let buffer = '';
91
- let idle = null;
92
- const settle = () => {
93
- if (idle) clearTimeout(idle);
94
- idle = setTimeout(() => controller.abort(), 240000);
95
- };
96
- settle();
97
- try {
98
- while (true) {
99
- let value, done;
100
- try {
101
- ({ done, value } = await reader.read());
102
- } catch (err) {
103
- if (err.name === 'AbortError') {
104
- throw new Error('[ERR-0001] Lost the connection to the cloud API (no data for 60s).');
105
- }
106
- throw err;
107
- }
108
- if (done) break;
109
- settle();
110
- buffer += decoder.decode(value, { stream: true });
111
- const lines = buffer.split('\n');
112
- buffer = lines.pop() || '';
113
77
 
114
- for (const line of lines) {
115
- if (!line.startsWith('data: ')) continue;
116
- const data = line.slice(6);
117
- try {
118
- const json = JSON.parse(data);
119
- const text = json.candidates?.[0]?.content?.parts?.[0]?.text;
120
- if (text) yield text;
121
- } catch {}
122
- }
78
+ while (true) {
79
+ const { done, value } = await reader.read();
80
+ if (done) break;
81
+
82
+ buffer += decoder.decode(value, { stream: true });
83
+ const lines = buffer.split('\n');
84
+ buffer = lines.pop() || '';
85
+
86
+ for (const line of lines) {
87
+ if (!line.startsWith('data: ')) continue;
88
+ const data = line.slice(6);
89
+ try {
90
+ const json = JSON.parse(data);
91
+ const text = json.candidates?.[0]?.content?.parts?.[0]?.text;
92
+ if (text) yield text;
93
+ } catch {}
123
94
  }
124
- } finally {
125
- if (idle) clearTimeout(idle);
126
95
  }
127
96
  }
128
97
  }
@@ -57,7 +57,7 @@ export class OpenAIProvider extends BaseProvider {
57
57
 
58
58
  async *stream(messages, options = {}) {
59
59
  const key = Config.get('openaiApiKey');
60
- if (!key) throw new Error('[ERR-0004] Cloud API key not configured. Add it to the config and try again.');
60
+ if (!key) throw new Error('Cloud API key not configured');
61
61
 
62
62
  const model = this._resolveModel(options);
63
63
  const systemPrompt = options.systemPrompt || '';
@@ -70,78 +70,47 @@ export class OpenAIProvider extends BaseProvider {
70
70
  apiMessages.push({ role: msg.role, content: msg.content });
71
71
  }
72
72
 
73
- const controller = new AbortController();
74
- const connectTimer = setTimeout(() => controller.abort(), 240000);
75
-
76
- let resp;
77
- try {
78
- resp = await fetch('https://api.openai.com/v1/chat/completions', {
79
- method: 'POST',
80
- headers: {
81
- 'Content-Type': 'application/json',
82
- Authorization: `Bearer ${key}`,
83
- },
84
- body: JSON.stringify({
85
- model,
86
- messages: apiMessages,
87
- stream: true,
88
- temperature: options.temperature || Config.get('temperature'),
89
- max_tokens: options.maxTokens || Config.get('maxTokens'),
90
- }),
91
- signal: controller.signal,
92
- });
93
- } catch (err) {
94
- if (err.name === 'AbortError') {
95
- throw new Error('[ERR-0001] Could not reach the API (no response within 60s). Check your network connection.');
96
- }
97
- throw new Error('[ERR-0001] Could not reach the API (no response). Check your network connection.');
98
- } finally {
99
- clearTimeout(connectTimer);
100
- }
73
+ const resp = await fetch('https://api.openai.com/v1/chat/completions', {
74
+ method: 'POST',
75
+ headers: {
76
+ 'Content-Type': 'application/json',
77
+ Authorization: `Bearer ${key}`,
78
+ },
79
+ body: JSON.stringify({
80
+ model,
81
+ messages: apiMessages,
82
+ stream: true,
83
+ temperature: options.temperature || Config.get('temperature'),
84
+ max_tokens: options.maxTokens || Config.get('maxTokens'),
85
+ }),
86
+ });
101
87
 
102
88
  if (!resp.ok) {
103
- throw new Error('[ERR-0003] The cloud provider returned HTTP ' + resp.status + '. Check your API key in config.');
89
+ throw new Error('Failed to connect to cloud engine');
104
90
  }
105
91
 
106
92
  const reader = resp.body.getReader();
107
93
  const decoder = new TextDecoder();
108
94
  let buffer = '';
109
- let idle = null;
110
- const settle = () => {
111
- if (idle) clearTimeout(idle);
112
- idle = setTimeout(() => controller.abort(), 240000);
113
- };
114
- settle();
115
- try {
116
- while (true) {
117
- let value, done;
118
- try {
119
- ({ done, value } = await reader.read());
120
- } catch (err) {
121
- if (err.name === 'AbortError') {
122
- throw new Error('[ERR-0001] Lost the connection to the cloud API (no data for 60s).');
123
- }
124
- throw err;
125
- }
126
- if (done) break;
127
- settle();
128
- buffer += decoder.decode(value, { stream: true });
129
- const lines = buffer.split('\n');
130
- buffer = lines.pop() || '';
131
95
 
132
- for (const line of lines) {
133
- if (!line.startsWith('data: ')) continue;
134
- const data = line.slice(6);
135
- if (data === '[DONE]') return;
136
- try {
137
- const json = JSON.parse(data);
138
- const content = json.choices?.[0]?.delta?.content;
139
- if (content) yield content;
140
- } catch {}
141
- }
96
+ while (true) {
97
+ const { done, value } = await reader.read();
98
+ if (done) break;
99
+
100
+ buffer += decoder.decode(value, { stream: true });
101
+ const lines = buffer.split('\n');
102
+ buffer = lines.pop() || '';
103
+
104
+ for (const line of lines) {
105
+ if (!line.startsWith('data: ')) continue;
106
+ const data = line.slice(6);
107
+ if (data === '[DONE]') return;
108
+ try {
109
+ const json = JSON.parse(data);
110
+ const content = json.choices?.[0]?.delta?.content;
111
+ if (content) yield content;
112
+ } catch {}
142
113
  }
143
- } finally {
144
- if (idle) clearTimeout(idle);
145
114
  }
146
115
  }
147
116
  }