@vierratale/ai 0.1.0-beta.8 → 0.1.0-beta.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,7 +39,7 @@ export class AnthropicProvider extends BaseProvider {
39
39
 
40
40
  async *stream(messages, options = {}) {
41
41
  const key = Config.get('anthropicApiKey');
42
- if (!key) throw new Error('Anthropic API key not configured');
42
+ if (!key) throw new Error('[ERR-0004] Cloud API key not configured. Add it to the config and try again.');
43
43
 
44
44
  const model = this._resolveModel(options);
45
45
  const systemPrompt = options.systemPrompt || '';
@@ -51,49 +51,80 @@ export class AnthropicProvider extends BaseProvider {
51
51
  content: m.content,
52
52
  }));
53
53
 
54
- const resp = await fetch(ANTHROPIC_URL, {
55
- method: 'POST',
56
- headers: {
57
- 'Content-Type': 'application/json',
58
- 'x-api-key': key,
59
- 'anthropic-version': '2023-06-01',
60
- },
61
- body: JSON.stringify({
62
- model,
63
- system: system.length ? system : undefined,
64
- messages: apiMessages,
65
- max_tokens: options.maxTokens || Config.get('maxTokens'),
66
- stream: true,
67
- temperature: options.temperature || Config.get('temperature'),
68
- }),
69
- });
54
+ const controller = new AbortController();
55
+ const connectTimer = setTimeout(() => controller.abort(), 240000);
56
+
57
+ let resp;
58
+ try {
59
+ resp = await fetch(ANTHROPIC_URL, {
60
+ method: 'POST',
61
+ headers: {
62
+ 'Content-Type': 'application/json',
63
+ 'x-api-key': key,
64
+ 'anthropic-version': '2023-06-01',
65
+ },
66
+ body: JSON.stringify({
67
+ model,
68
+ system: system.length ? system : undefined,
69
+ messages: apiMessages,
70
+ max_tokens: options.maxTokens || Config.get('maxTokens'),
71
+ stream: true,
72
+ temperature: options.temperature || Config.get('temperature'),
73
+ }),
74
+ signal: controller.signal,
75
+ });
76
+ } catch (err) {
77
+ if (err.name === 'AbortError') {
78
+ throw new Error('[ERR-0001] Could not reach the API (no response within 60s). Check your network connection.');
79
+ }
80
+ throw new Error('[ERR-0001] Could not reach the API (no response). Check your network connection.');
81
+ } finally {
82
+ clearTimeout(connectTimer);
83
+ }
70
84
 
71
85
  if (!resp.ok) {
72
- throw new Error('Failed to connect to cloud engine');
86
+ throw new Error('[ERR-0003] The cloud provider returned HTTP ' + resp.status + '. Check your API key in config.');
73
87
  }
74
88
 
75
89
  const reader = resp.body.getReader();
76
90
  const decoder = new TextDecoder();
77
91
  let buffer = '';
78
-
79
- while (true) {
80
- const { done, value } = await reader.read();
81
- if (done) break;
82
-
83
- buffer += decoder.decode(value, { stream: true });
84
- const lines = buffer.split('\n');
85
- buffer = lines.pop() || '';
86
-
87
- for (const line of lines) {
88
- if (!line.startsWith('data: ')) continue;
89
- const data = line.slice(6);
92
+ let idle = null;
93
+ const settle = () => {
94
+ if (idle) clearTimeout(idle);
95
+ idle = setTimeout(() => controller.abort(), 240000);
96
+ };
97
+ settle();
98
+ try {
99
+ while (true) {
100
+ let value, done;
90
101
  try {
91
- const json = JSON.parse(data);
92
- if (json.type === 'content_block_delta' && json.delta?.text) {
93
- yield json.delta.text;
102
+ ({ done, value } = await reader.read());
103
+ } catch (err) {
104
+ if (err.name === 'AbortError') {
105
+ throw new Error('[ERR-0001] Lost the connection to the cloud API (no data for 60s).');
94
106
  }
95
- } catch {}
107
+ throw err;
108
+ }
109
+ if (done) break;
110
+ settle();
111
+ buffer += decoder.decode(value, { stream: true });
112
+ const lines = buffer.split('\n');
113
+ buffer = lines.pop() || '';
114
+
115
+ for (const line of lines) {
116
+ if (!line.startsWith('data: ')) continue;
117
+ const data = line.slice(6);
118
+ try {
119
+ const json = JSON.parse(data);
120
+ if (json.type === 'content_block_delta' && json.delta?.text) {
121
+ yield json.delta.text;
122
+ }
123
+ } catch {}
124
+ }
96
125
  }
126
+ } finally {
127
+ if (idle) clearTimeout(idle);
97
128
  }
98
129
  }
99
130
  }
@@ -15,6 +15,16 @@ export class BaseProvider {
15
15
  throw new Error('stream() must be implemented');
16
16
  }
17
17
 
18
+ // Non-streaming convenience: collects the stream. Providers may override
19
+ // with a proper single-shot request (see CortexProvider).
20
+ async complete(messages, options = {}) {
21
+ let out = '';
22
+ for await (const chunk of this.stream(messages, options)) {
23
+ out += chunk;
24
+ }
25
+ return out;
26
+ }
27
+
18
28
  async listModels() {
19
29
  throw new Error('listModels() must be implemented');
20
30
  }
@@ -2,6 +2,25 @@ import { BaseProvider } from './base.js';
2
2
  import { Catalog } from '../catalog.js';
3
3
  import { Config } from '../config.js';
4
4
 
5
+ // Keep the prompt a local engine must re-process per call bounded: the
6
+ // system prompt plus the most recent messages. Sending the whole growing
7
+ // history makes prompt-processing time unbounded on slow hardware.
8
+ const MAX_CONTEXT_MESSAGES = 12;
9
+
10
+ function buildEngineMessages(systemPrompt, messages) {
11
+ const engineMessages = [];
12
+ if (systemPrompt) {
13
+ engineMessages.push({ role: 'system', content: systemPrompt });
14
+ }
15
+ for (const msg of messages) {
16
+ engineMessages.push({ role: msg.role, content: msg.content });
17
+ }
18
+ if (engineMessages.length > MAX_CONTEXT_MESSAGES) {
19
+ return [engineMessages[0], ...engineMessages.slice(-(MAX_CONTEXT_MESSAGES - 1))];
20
+ }
21
+ return engineMessages;
22
+ }
23
+
5
24
  export class CortexProvider extends BaseProvider {
6
25
  constructor() {
7
26
  super('cortex');
@@ -42,74 +61,238 @@ export class CortexProvider extends BaseProvider {
42
61
 
43
62
  async *stream(messages, options = {}) {
44
63
  const model = Catalog.getRealModel(options.model || Config.get('model'));
45
- const systemPrompt = options.systemPrompt || '';
64
+ const engineMessages = buildEngineMessages(options.systemPrompt || '', messages);
46
65
 
47
- const engineMessages = [];
48
- if (systemPrompt) {
49
- engineMessages.push({ role: 'system', content: systemPrompt });
50
- }
51
- for (const msg of messages) {
52
- engineMessages.push({ role: msg.role, content: msg.content });
53
- }
66
+ const controller = new AbortController();
67
+ const connectTimer = setTimeout(() => controller.abort(), 600000);
54
68
 
55
- const resp = await fetch(`${this.host}/api/chat`, {
56
- method: 'POST',
57
- headers: { 'Content-Type': 'application/json' },
58
- body: JSON.stringify({
59
- model,
60
- messages: engineMessages,
61
- stream: true,
62
- keep_alive: Config.get('keepAlive'),
63
- options: {
64
- num_ctx: Config.get('numCtx'),
65
- temperature: options.temperature || Config.get('temperature'),
66
- },
67
- }),
68
- });
69
+ let resp;
70
+ try {
71
+ resp = await fetch(`${this.host}/api/chat`, {
72
+ method: 'POST',
73
+ headers: { 'Content-Type': 'application/json' },
74
+ body: JSON.stringify({
75
+ model,
76
+ messages: engineMessages,
77
+ stream: true,
78
+ keep_alive: Config.get('keepAlive'),
79
+ options: {
80
+ num_ctx: Config.get('numCtx'),
81
+ num_predict: options.maxTokens ?? Config.get('maxTokens'),
82
+ temperature: options.temperature || Config.get('temperature'),
83
+ num_thread: Config.get('numThreads'),
84
+ },
85
+ }),
86
+ signal: controller.signal,
87
+ });
88
+ } catch (err) {
89
+ if (err.name === 'AbortError') {
90
+ throw new Error('[ERR-0001] Could not reach the engine (no response within 600s). Make sure the local engine is running.');
91
+ }
92
+ throw new Error('[ERR-0001] Could not reach the engine (no response). Make sure the local engine is running.');
93
+ } finally {
94
+ clearTimeout(connectTimer);
95
+ }
69
96
 
70
97
  if (!resp.ok) {
71
- throw new Error('Failed to connect to backend engine');
98
+ throw new Error(await describeEngineError(resp, model));
72
99
  }
73
100
 
74
101
  const reader = resp.body.getReader();
75
102
  const decoder = new TextDecoder();
76
103
  let buffer = '';
104
+ let idle = null;
105
+ const settle = () => {
106
+ if (idle) clearTimeout(idle);
107
+ idle = setTimeout(() => controller.abort(), 420000);
108
+ };
109
+ settle();
110
+ try {
111
+ while (true) {
112
+ let value, done;
113
+ try {
114
+ ({ done, value } = await reader.read());
115
+ } catch (err) {
116
+ if (err.name === 'AbortError') {
117
+ throw new Error('[ERR-0001] Lost the connection to the engine (no data for 420s).');
118
+ }
119
+ throw err;
120
+ }
121
+ if (done) break;
122
+ settle();
123
+ buffer += decoder.decode(value, { stream: true });
124
+ const lines = buffer.split('\n');
125
+ buffer = lines.pop() || '';
126
+
127
+ for (const line of lines) {
128
+ if (!line.trim()) continue;
129
+ try {
130
+ const json = JSON.parse(line);
131
+ if (json.message?.content) {
132
+ yield json.message.content;
133
+ }
134
+ if (json.done) return;
135
+ } catch {}
136
+ }
137
+ }
138
+ } finally {
139
+ if (idle) clearTimeout(idle);
140
+ }
141
+ }
142
+
143
+ // Non-streaming completion (used by the tool-planning step). Reads the
144
+ // streamed reply incrementally: on a slow engine a full answer can take
145
+ // minutes, so only a connection with no response for 180s (or an idle
146
+ // stream for 120s) counts as a failure - not a slowly progressing one.
147
+ async complete(messages, options = {}) {
148
+ const model = Catalog.getRealModel(options.model || Config.get('model'));
149
+ const engineMessages = buildEngineMessages(options.systemPrompt || '', messages);
150
+
151
+ const controller = new AbortController();
152
+ const connectTimer = setTimeout(() => controller.abort(), 600000);
153
+
154
+ if (options.signal) {
155
+ if (options.signal.aborted) controller.abort();
156
+ else options.signal.addEventListener('abort', () => controller.abort(), { once: true });
157
+ }
77
158
 
78
- while (true) {
79
- const { done, value } = await reader.read();
80
- if (done) break;
159
+ let resp;
160
+ try {
161
+ resp = await fetch(`${this.host}/api/chat`, {
162
+ method: 'POST',
163
+ headers: { 'Content-Type': 'application/json' },
164
+ body: JSON.stringify({
165
+ model,
166
+ messages: engineMessages,
167
+ stream: true,
168
+ keep_alive: Config.get('keepAlive'),
169
+ options: {
170
+ num_ctx: Config.get('numCtx'),
171
+ num_predict: options.maxTokens ?? Config.get('maxTokens'),
172
+ temperature: options.temperature || Config.get('temperature'),
173
+ num_thread: Config.get('numThreads'),
174
+ },
175
+ }),
176
+ signal: controller.signal,
177
+ });
178
+ } catch (err) {
179
+ if (err.name === 'AbortError') {
180
+ throw new Error('[ERR-0001] Could not complete the request (no response within 600s).');
181
+ }
182
+ throw new Error('[ERR-0001] Could not reach the engine (no response). Make sure the local engine is running.');
183
+ } finally {
184
+ clearTimeout(connectTimer);
185
+ }
81
186
 
82
- buffer += decoder.decode(value, { stream: true });
83
- const lines = buffer.split('\n');
84
- buffer = lines.pop() || '';
187
+ if (!resp.ok) {
188
+ throw new Error(await describeEngineError(resp, model));
189
+ }
85
190
 
86
- for (const line of lines) {
87
- if (!line.trim()) continue;
191
+ const reader = resp.body.getReader();
192
+ const decoder = new TextDecoder();
193
+ let buffer = '';
194
+ let idle = null;
195
+ const settle = () => {
196
+ if (idle) clearTimeout(idle);
197
+ idle = setTimeout(() => controller.abort(), 120000);
198
+ };
199
+ settle();
200
+ let output = '';
201
+ try {
202
+ while (true) {
203
+ let value, done;
88
204
  try {
89
- const json = JSON.parse(line);
90
- if (json.message?.content) {
91
- yield json.message.content;
205
+ ({ done, value } = await reader.read());
206
+ } catch (err) {
207
+ if (err.name === 'AbortError') {
208
+ throw new Error('[ERR-0001] Lost the connection to the engine (no data for 420s).');
92
209
  }
93
- if (json.done) return;
94
- } catch {}
210
+ throw err;
211
+ }
212
+ if (done) break;
213
+ settle();
214
+ buffer += decoder.decode(value, { stream: true });
215
+ const lines = buffer.split('\n');
216
+ buffer = lines.pop() || '';
217
+
218
+ for (const line of lines) {
219
+ if (!line.trim()) continue;
220
+ try {
221
+ const json = JSON.parse(line);
222
+ if (json.message?.content) {
223
+ output += json.message.content;
224
+ }
225
+ if (json.done) {
226
+ buffer = '';
227
+ break;
228
+ }
229
+ } catch {}
230
+ }
95
231
  }
232
+ } finally {
233
+ if (idle) clearTimeout(idle);
234
+ controller.abort();
96
235
  }
236
+ return output;
97
237
  }
98
238
 
99
239
  async warmup() {
100
- // Fire-and-forget: send a tiny request so ollama loads the model into
101
- // memory eagerly, so the first real query doesn't pay the cold-load cost.
240
+ // Fire-and-forget: ask the engine to load the model eagerly, then abort as
241
+ // soon as generation starts. The engine runs with a single slot, so an
242
+ // in-flight non-streaming 'hi' would block the first real request.
102
243
  const model = Catalog.getRealModel(Config.get('model'));
244
+ const controller = new AbortController();
103
245
  fetch(`${this.host}/api/chat`, {
104
246
  method: 'POST',
105
247
  headers: { 'Content-Type': 'application/json' },
106
248
  body: JSON.stringify({
107
249
  model,
108
250
  messages: [{ role: 'user', content: 'hi' }],
109
- stream: false,
251
+ stream: true,
110
252
  keep_alive: Config.get('keepAlive'),
111
253
  options: { num_ctx: Config.get('numCtx') },
112
254
  }),
113
- }).catch(() => {});
255
+ signal: controller.signal,
256
+ })
257
+ .then(async (resp) => {
258
+ if (!resp.ok || !resp.body) return;
259
+ const reader = resp.body.getReader();
260
+ const decoder = new TextDecoder();
261
+ let buffer = '';
262
+ for (let i = 0; i < 100; i++) {
263
+ const { done, value } = await reader.read();
264
+ if (done) break;
265
+ buffer += decoder.decode(value, { stream: true });
266
+ const lines = buffer.split('\n');
267
+ buffer = lines.pop() || '';
268
+ for (const line of lines) {
269
+ if (!line.trim()) continue;
270
+ try {
271
+ if (JSON.parse(line).message?.content) {
272
+ controller.abort();
273
+ return;
274
+ }
275
+ } catch {}
276
+ }
277
+ }
278
+ })
279
+ .catch(() => {});
280
+ }
281
+ }
282
+
283
+ // Build a user-actionable message when the engine rejects a request (e.g. a
284
+ // cloud model name posted to a local engine -> 404 "model not found"). The
285
+ // engine host is deliberately kept out of the message; errors carry codes.
286
+ async function describeEngineError(resp, model) {
287
+ if (/not found|does not exist|model.*missing/i.test(String(resp.statusText))) {
288
+ return `[ERR-0002] Model "${model}" is not installed on the engine. If you picked a cloud model, use /model VRTL-6.pro or /provider openai.`;
289
+ }
290
+ let body = '';
291
+ try {
292
+ body = (await resp.text()).slice(0, 300);
293
+ } catch {}
294
+ if (/not found|does not exist|model.*missing/i.test(body)) {
295
+ return `[ERR-0002] Model "${model}" is not installed on the engine. If you picked a cloud model, use /model VRTL-6.pro or /provider openai.`;
114
296
  }
297
+ return `[ERR-0003] The engine returned HTTP ${resp.status}. Check the engine status or run /model VRTL-2.fast.`;
115
298
  }
@@ -39,7 +39,7 @@ export class GeminiProvider extends BaseProvider {
39
39
 
40
40
  async *stream(messages, options = {}) {
41
41
  const key = Config.get('geminiApiKey');
42
- if (!key) throw new Error('Gemini API key not configured');
42
+ if (!key) throw new Error('[ERR-0004] Cloud API key not configured. Add it to the config and try again.');
43
43
 
44
44
  const model = this._resolveModel(options);
45
45
  const systemPrompt = options.systemPrompt || '';
@@ -52,46 +52,77 @@ export class GeminiProvider extends BaseProvider {
52
52
  contents.unshift({ role: 'user', parts: [{ text: `System: ${systemPrompt}` }] });
53
53
  }
54
54
 
55
- const resp = await fetch(
56
- `${GEMINI_URL}/${model}:streamGenerateContent?key=${key}&alt=sse`,
57
- {
58
- method: 'POST',
59
- headers: { 'Content-Type': 'application/json' },
60
- body: JSON.stringify({
61
- contents,
62
- generationConfig: {
63
- temperature: options.temperature || Config.get('temperature'),
64
- maxOutputTokens: options.maxTokens || Config.get('maxTokens'),
65
- },
66
- }),
55
+ const controller = new AbortController();
56
+ const connectTimer = setTimeout(() => controller.abort(), 240000);
57
+
58
+ let resp;
59
+ try {
60
+ resp = await fetch(
61
+ `${GEMINI_URL}/${model}:streamGenerateContent?key=${key}&alt=sse`,
62
+ {
63
+ method: 'POST',
64
+ headers: { 'Content-Type': 'application/json' },
65
+ body: JSON.stringify({
66
+ contents,
67
+ generationConfig: {
68
+ temperature: options.temperature || Config.get('temperature'),
69
+ maxOutputTokens: options.maxTokens || Config.get('maxTokens'),
70
+ },
71
+ }),
72
+ signal: controller.signal,
73
+ }
74
+ );
75
+ } catch (err) {
76
+ if (err.name === 'AbortError') {
77
+ throw new Error('[ERR-0001] Could not reach the API (no response within 60s). Check your network connection.');
67
78
  }
68
- );
79
+ throw new Error('[ERR-0001] Could not reach the API (no response). Check your network connection.');
80
+ } finally {
81
+ clearTimeout(connectTimer);
82
+ }
69
83
 
70
84
  if (!resp.ok) {
71
- throw new Error('Failed to connect to cloud engine');
85
+ throw new Error('[ERR-0003] The cloud provider returned HTTP ' + resp.status + '. Check your API key in config.');
72
86
  }
73
87
 
74
88
  const reader = resp.body.getReader();
75
89
  const decoder = new TextDecoder();
76
90
  let buffer = '';
77
-
78
- while (true) {
79
- const { done, value } = await reader.read();
80
- if (done) break;
81
-
82
- buffer += decoder.decode(value, { stream: true });
83
- const lines = buffer.split('\n');
84
- buffer = lines.pop() || '';
85
-
86
- for (const line of lines) {
87
- if (!line.startsWith('data: ')) continue;
88
- const data = line.slice(6);
91
+ let idle = null;
92
+ const settle = () => {
93
+ if (idle) clearTimeout(idle);
94
+ idle = setTimeout(() => controller.abort(), 240000);
95
+ };
96
+ settle();
97
+ try {
98
+ while (true) {
99
+ let value, done;
89
100
  try {
90
- const json = JSON.parse(data);
91
- const text = json.candidates?.[0]?.content?.parts?.[0]?.text;
92
- if (text) yield text;
93
- } catch {}
101
+ ({ done, value } = await reader.read());
102
+ } catch (err) {
103
+ if (err.name === 'AbortError') {
104
+ throw new Error('[ERR-0001] Lost the connection to the cloud API (no data for 60s).');
105
+ }
106
+ throw err;
107
+ }
108
+ if (done) break;
109
+ settle();
110
+ buffer += decoder.decode(value, { stream: true });
111
+ const lines = buffer.split('\n');
112
+ buffer = lines.pop() || '';
113
+
114
+ for (const line of lines) {
115
+ if (!line.startsWith('data: ')) continue;
116
+ const data = line.slice(6);
117
+ try {
118
+ const json = JSON.parse(data);
119
+ const text = json.candidates?.[0]?.content?.parts?.[0]?.text;
120
+ if (text) yield text;
121
+ } catch {}
122
+ }
94
123
  }
124
+ } finally {
125
+ if (idle) clearTimeout(idle);
95
126
  }
96
127
  }
97
128
  }