ucode-agent 1.26.1 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -7
- package/package.json +1 -1
- package/skills/build-app/DIGEST.md +94 -0
- package/skills/build-app/SKILL.md +222 -181
- package/skills/ui-ux/DIGEST.md +135 -0
- package/src/core/context.js +164 -151
- package/src/core/loop.js +209 -39
- package/src/core/provider.js +874 -868
- package/src/core/skills.js +189 -165
- package/src/core/window.js +27 -2
- package/src/tools/blocks.js +117 -27
- package/src/tools/index.js +71 -23
- package/src/tools/scaffold.js +104 -14
- package/src/ui/plain.js +4 -2
- package/src/ui/screen.js +14 -5
- package/src/ui/theme.js +100 -24
- package/templates/blocks/plain/filter-bar.js +133 -0
- package/templates/blocks/plain/item-list.js +249 -0
- package/templates/blocks/plain/modal.js +141 -0
- package/templates/blocks/plain/store.js +93 -0
- package/templates/blocks/plain/theme-toggle.js +116 -0
- package/templates/blocks/plain/toast.js +107 -0
- package/templates/plain-html/styles.css +4 -0
package/src/core/provider.js
CHANGED
|
@@ -1,868 +1,874 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* provider.js — the only module that knows a model provider exists.
|
|
3
|
-
*
|
|
4
|
-
* Everything above this file speaks one small neutral message format and calls
|
|
5
|
-
* ask(). Moving ucode to a different host means rewriting this file and
|
|
6
|
-
* nothing else.
|
|
7
|
-
*
|
|
8
|
-
* The neutral formats:
|
|
9
|
-
* { role: 'system', content }
|
|
10
|
-
* { role: 'user', content, images?: [dataUrl] }
|
|
11
|
-
* { role: 'assistant', content?, toolCalls?: [{ id, name, args }] }
|
|
12
|
-
* { role: 'tool', toolCallId, name, content }
|
|
13
|
-
*
|
|
14
|
-
* tool: { name, description, parameters: <JSON Schema> }
|
|
15
|
-
*
|
|
16
|
-
* ask() resolves to:
|
|
17
|
-
* { text, reasoning, toolCalls, usage, finishReason, model }
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
import { join, dirname } from 'node:path';
|
|
21
|
-
import { homedir } from 'node:os';
|
|
22
|
-
import { fileURLToPath } from 'node:url';
|
|
23
|
-
import dotenv from 'dotenv';
|
|
24
|
-
import OpenAI from 'openai';
|
|
25
|
-
import { jsonrepair } from 'jsonrepair';
|
|
26
|
-
import { Failure } from './failure.js';
|
|
27
|
-
|
|
28
|
-
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
29
|
-
const PACKAGE_ROOT = join(HERE, '..', '..');
|
|
30
|
-
|
|
31
|
-
/** Personal config lives here, outside the package, so upgrades never touch it. */
|
|
32
|
-
export const UCODE_HOME = join(homedir(), '.ucode');
|
|
33
|
-
export const ENV_FILE = join(UCODE_HOME, '.env');
|
|
34
|
-
|
|
35
|
-
export const BASE_URL = 'https://openrouter.ai/api/v1';
|
|
36
|
-
export const PROVIDER = 'OpenRouter';
|
|
37
|
-
|
|
38
|
-
// First definition wins — dotenv never overwrites a variable that already
|
|
39
|
-
// exists — so the order here is the precedence order:
|
|
40
|
-
// real environment > ./.env (this project) > ~/.ucode/.env (this machine)
|
|
41
|
-
// > the checkout's own .env (only when developing on a clone)
|
|
42
|
-
dotenv.config({ path: join(process.cwd(), '.env'), quiet: true });
|
|
43
|
-
dotenv.config({ path: ENV_FILE, quiet: true });
|
|
44
|
-
dotenv.config({ path: join(PACKAGE_ROOT, '.env'), quiet: true });
|
|
45
|
-
|
|
46
|
-
/**
|
|
47
|
-
* The whole model list. Not a starting point — the list.
|
|
48
|
-
*
|
|
49
|
-
* ucode runs on NVIDIA and Cohere only. Both vendors serve genuinely capable
|
|
50
|
-
* models free through OpenRouter, both handle tool calling properly, and
|
|
51
|
-
* keeping the set to five means every one of them has been used in anger
|
|
52
|
-
* rather than listed on the strength of a benchmark. A picker offering sixty
|
|
53
|
-
* models is a picker nobody reads.
|
|
54
|
-
*
|
|
55
|
-
* `name` is what the status bar shows. `note` is what the picker shows.
|
|
56
|
-
*/
|
|
57
|
-
export const MODELS = {
|
|
58
|
-
'nvidia/nemotron-3-ultra-550b-a55b:free': {
|
|
59
|
-
name: 'Nemotron 3 Ultra',
|
|
60
|
-
context: 1_000_000,
|
|
61
|
-
star: true,
|
|
62
|
-
note: 'deepest reasoning, 1M context — slowest to answer',
|
|
63
|
-
},
|
|
64
|
-
'nvidia/nemotron-3.5-lightning:free': {
|
|
65
|
-
name: 'Nemotron 3.5 Lightning',
|
|
66
|
-
context: 1_000_000,
|
|
67
|
-
note: 'same huge window, answers much sooner',
|
|
68
|
-
},
|
|
69
|
-
'nvidia/nemotron-3-super-120b-a12b:free': {
|
|
70
|
-
name: 'Nemotron 3 Super',
|
|
71
|
-
context: 262_144,
|
|
72
|
-
note: 'strong all-rounder, quick to first token',
|
|
73
|
-
},
|
|
74
|
-
'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free': {
|
|
75
|
-
name: 'Nemotron 3 Nano Omni',
|
|
76
|
-
context: 256_000,
|
|
77
|
-
note: 'small and fast, reasoning tuned',
|
|
78
|
-
},
|
|
79
|
-
'cohere/north-mini-code:free': {
|
|
80
|
-
name: 'North Mini Code',
|
|
81
|
-
context: 256_000,
|
|
82
|
-
star: true,
|
|
83
|
-
note: 'the default — built for code and interface work, quick to answer',
|
|
84
|
-
},
|
|
85
|
-
};
|
|
86
|
-
|
|
87
|
-
/**
|
|
88
|
-
* North Mini Code is the default: it is built for code and interface work,
|
|
89
|
-
* which is what ucode is mostly asked to do, and it answers far sooner than
|
|
90
|
-
* the big reasoning models. /model moves to Ultra when a problem needs the
|
|
91
|
-
* million-token window and the long think more than it needs the speed.
|
|
92
|
-
*/
|
|
93
|
-
export const DEFAULT_MODEL = 'cohere/north-mini-code:free';
|
|
94
|
-
|
|
95
|
-
/**
|
|
96
|
-
* Where to go when a model is busy, in order of preference. Each is served by
|
|
97
|
-
* a different upstream, so a rate limit on one rarely means a limit on the
|
|
98
|
-
* next — which is what lets a long build keep going instead of stopping at
|
|
99
|
-
* the first "too many requests".
|
|
100
|
-
*/
|
|
101
|
-
export const FALLBACKS = [
|
|
102
|
-
'cohere/north-mini-code:free',
|
|
103
|
-
'nvidia/nemotron-3.5-lightning:free',
|
|
104
|
-
'nvidia/nemotron-3-super-120b-a12b:free',
|
|
105
|
-
'nvidia/nemotron-3-ultra-550b-a55b:free',
|
|
106
|
-
];
|
|
107
|
-
|
|
108
|
-
/** The next model to try after `id`, skipping any already tried this round. */
|
|
109
|
-
export function fallbackFor(id, tried = new Set()) {
|
|
110
|
-
const start = Math.max(0, FALLBACKS.indexOf(id));
|
|
111
|
-
for (let i = 1; i <= FALLBACKS.length; i++) {
|
|
112
|
-
const next = FALLBACKS[(start + i) % FALLBACKS.length];
|
|
113
|
-
if (next !== id && !tried.has(next)) return next;
|
|
114
|
-
}
|
|
115
|
-
return null;
|
|
116
|
-
}
|
|
117
|
-
|
|
118
|
-
/** Seconds to wait on successive rate limits that come with no retry-after. */
|
|
119
|
-
const RATE_LIMIT_BACKOFF = [5, 10, 20];
|
|
120
|
-
|
|
121
|
-
let current = process.env.UCODE_MODEL || DEFAULT_MODEL;
|
|
122
|
-
let client = null;
|
|
123
|
-
|
|
124
|
-
export function model() {
|
|
125
|
-
return current;
|
|
126
|
-
}
|
|
127
|
-
|
|
128
|
-
export function setModel(id) {
|
|
129
|
-
const wanted = String(id ?? '').trim();
|
|
130
|
-
if (!wanted) {
|
|
131
|
-
throw new Failure({
|
|
132
|
-
kind: 'bad_model',
|
|
133
|
-
attempted: 'switching model',
|
|
134
|
-
failed: 'No model name was given.',
|
|
135
|
-
fix: `Pick one of: ${Object.keys(MODELS).join(', ')}`,
|
|
136
|
-
});
|
|
137
|
-
}
|
|
138
|
-
if (!MODELS[wanted]) {
|
|
139
|
-
throw new Failure({
|
|
140
|
-
kind: 'bad_model',
|
|
141
|
-
attempted: `switching to "${wanted}"`,
|
|
142
|
-
failed: 'ucode only runs NVIDIA and Cohere models, and that is not one of them.',
|
|
143
|
-
fix: `Run /model to choose from: ${Object.keys(MODELS).join(', ')}`,
|
|
144
|
-
});
|
|
145
|
-
}
|
|
146
|
-
current = wanted;
|
|
147
|
-
return current;
|
|
148
|
-
}
|
|
149
|
-
|
|
150
|
-
/** The short name for a model id: "Nemotron 3 Ultra". */
|
|
151
|
-
export function modelName(id = current) {
|
|
152
|
-
return MODELS[id]?.name ?? id;
|
|
153
|
-
}
|
|
154
|
-
|
|
155
|
-
/** Every model, in list order, annotated with whether it is the active one. */
|
|
156
|
-
export function modelList() {
|
|
157
|
-
return Object.entries(MODELS).map(([id, info]) => ({
|
|
158
|
-
id,
|
|
159
|
-
...info,
|
|
160
|
-
active: id === current,
|
|
161
|
-
}));
|
|
162
|
-
}
|
|
163
|
-
|
|
164
|
-
/**
|
|
165
|
-
* How many tokens one request may occupy.
|
|
166
|
-
*
|
|
167
|
-
* OpenRouter meters requests rather than tokens, so nothing here is rationing
|
|
168
|
-
* a quota — the binding limit is simply the window the model has. On the two
|
|
169
|
-
* million-token models compaction essentially never fires.
|
|
170
|
-
*/
|
|
171
|
-
export function contextLimit(id = current) {
|
|
172
|
-
const override = Number(process.env.UCODE_MAX_CONTEXT_TOKENS);
|
|
173
|
-
if (Number.isFinite(override) && override > 0) return override;
|
|
174
|
-
return MODELS[id]?.context ?? 128_000;
|
|
175
|
-
}
|
|
176
|
-
|
|
177
|
-
/**
|
|
178
|
-
* A rough local token count, used to decide when to compact *before* a
|
|
179
|
-
* request goes out. Real numbers come back in the response usage; this only
|
|
180
|
-
* has to be close enough to trigger at the right time.
|
|
181
|
-
*/
|
|
182
|
-
export function estimateTokens(text) {
|
|
183
|
-
return text ? Math.ceil(String(text).length / 4) : 0;
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
export function estimateConversation(messages) {
|
|
187
|
-
let total = 0;
|
|
188
|
-
for (const m of messages) {
|
|
189
|
-
total += estimateTokens(m.content || '');
|
|
190
|
-
for (const call of m.toolCalls || []) {
|
|
191
|
-
total += estimateTokens(call.name) + estimateTokens(JSON.stringify(call.args || {}));
|
|
192
|
-
}
|
|
193
|
-
total += 4; // role and framing overhead per message
|
|
194
|
-
}
|
|
195
|
-
return total;
|
|
196
|
-
}
|
|
197
|
-
|
|
198
|
-
// ---------------------------------------------------------------------------
|
|
199
|
-
// Client
|
|
200
|
-
// ---------------------------------------------------------------------------
|
|
201
|
-
|
|
202
|
-
/**
|
|
203
|
-
* No key is bundled here, and none ever should be. A key committed to a public
|
|
204
|
-
* package is readable by anyone who runs `npm pack ucode-agent`, and no amount
|
|
205
|
-
* of first-run convenience is worth handing out a live credential.
|
|
206
|
-
*/
|
|
207
|
-
function apiKey() {
|
|
208
|
-
// UCODE_API_KEY is the documented name. The provider's own variable name is
|
|
209
|
-
// still read, so a key set up for another tool keeps working here.
|
|
210
|
-
const key = (process.env.UCODE_API_KEY || process.env.OPENROUTER_API_KEY || '').trim();
|
|
211
|
-
if (!key) {
|
|
212
|
-
throw new Failure({
|
|
213
|
-
kind: 'no_api_key',
|
|
214
|
-
attempted: 'connecting to the model',
|
|
215
|
-
failed: 'No API key is set - UCODE_API_KEY is missing from the environment and from every .env file.',
|
|
216
|
-
fix:
|
|
217
|
-
`Put UCODE_API_KEY=your-key in ${ENV_FILE} — that applies to every ` +
|
|
218
|
-
'project on this machine — or in a .env file beside your code. ' +
|
|
219
|
-
'Free keys: https://openrouter.ai/keys',
|
|
220
|
-
});
|
|
221
|
-
}
|
|
222
|
-
return key;
|
|
223
|
-
}
|
|
224
|
-
|
|
225
|
-
function connection() {
|
|
226
|
-
if (client) return client;
|
|
227
|
-
client = new OpenAI({
|
|
228
|
-
apiKey: apiKey(),
|
|
229
|
-
baseURL: process.env.UCODE_BASE_URL || BASE_URL,
|
|
230
|
-
// Nemotron Ultra can think for a long time before its first token, so the
|
|
231
|
-
// ceiling is deliberately generous. maxRetries is 0 because ask() owns
|
|
232
|
-
// retrying: its attempts are narrated on screen instead of happening
|
|
233
|
-
// silently somewhere inside the SDK.
|
|
234
|
-
timeout: Number(process.env.UCODE_REQUEST_TIMEOUT_MS) || 300_000,
|
|
235
|
-
maxRetries: 0,
|
|
236
|
-
defaultHeaders: {
|
|
237
|
-
'HTTP-Referer': 'https://github.com/sppideey/ucode-agent',
|
|
238
|
-
'X-Title': 'ucode',
|
|
239
|
-
},
|
|
240
|
-
});
|
|
241
|
-
return client;
|
|
242
|
-
}
|
|
243
|
-
|
|
244
|
-
/** Drop the cached client so the next request picks up a changed key. */
|
|
245
|
-
export function resetConnection() {
|
|
246
|
-
client = null;
|
|
247
|
-
}
|
|
248
|
-
|
|
249
|
-
// ---------------------------------------------------------------------------
|
|
250
|
-
// Live quota, taken from whatever rate-limit headers come back
|
|
251
|
-
// ---------------------------------------------------------------------------
|
|
252
|
-
|
|
253
|
-
let limits = null;
|
|
254
|
-
|
|
255
|
-
export function rateLimits() {
|
|
256
|
-
return limits;
|
|
257
|
-
}
|
|
258
|
-
|
|
259
|
-
/** "1.5s", "2m59.56s", "1h2m" -> seconds */
|
|
260
|
-
function seconds(value) {
|
|
261
|
-
if (!value) return null;
|
|
262
|
-
const m = /^(?:(\d+(?:\.\d+)?)h)?(?:(\d+(?:\.\d+)?)m(?!s))?(?:(\d+(?:\.\d+)?)m?s)?$/
|
|
263
|
-
.exec(String(value).trim());
|
|
264
|
-
if (!m) return null;
|
|
265
|
-
const total = (parseFloat(m[1]) || 0) * 3600 + (parseFloat(m[2]) || 0) * 60 + (parseFloat(m[3]) || 0);
|
|
266
|
-
return total > 0 ? total : null;
|
|
267
|
-
}
|
|
268
|
-
|
|
269
|
-
function noteLimits(headers) {
|
|
270
|
-
if (!headers?.get) return;
|
|
271
|
-
const num = (name) => {
|
|
272
|
-
const n = Number(headers.get(name));
|
|
273
|
-
return Number.isFinite(n) ? n : null;
|
|
274
|
-
};
|
|
275
|
-
limits = {
|
|
276
|
-
requestsLimit: num('x-ratelimit-limit-requests'),
|
|
277
|
-
requestsRemaining: num('x-ratelimit-remaining-requests'),
|
|
278
|
-
requestsReset: seconds(headers.get('x-ratelimit-reset-requests')),
|
|
279
|
-
tokensLimit: num('x-ratelimit-limit-tokens'),
|
|
280
|
-
tokensRemaining: num('x-ratelimit-remaining-tokens'),
|
|
281
|
-
tokensReset: seconds(headers.get('x-ratelimit-reset-tokens')),
|
|
282
|
-
at: Date.now(),
|
|
283
|
-
};
|
|
284
|
-
}
|
|
285
|
-
|
|
286
|
-
// ---------------------------------------------------------------------------
|
|
287
|
-
// Neutral format -> wire format
|
|
288
|
-
// ---------------------------------------------------------------------------
|
|
289
|
-
|
|
290
|
-
function wireTools(tools) {
|
|
291
|
-
if (!tools?.length) return undefined;
|
|
292
|
-
return tools.map((t) => ({
|
|
293
|
-
type: 'function',
|
|
294
|
-
function: {
|
|
295
|
-
name: t.name,
|
|
296
|
-
description: t.description,
|
|
297
|
-
parameters: t.parameters ?? { type: 'object', properties: {} },
|
|
298
|
-
},
|
|
299
|
-
}));
|
|
300
|
-
}
|
|
301
|
-
|
|
302
|
-
function wireMessages(messages) {
|
|
303
|
-
const out = [];
|
|
304
|
-
for (const m of messages) {
|
|
305
|
-
if (m.role === 'user' && m.images?.length) {
|
|
306
|
-
out.push({
|
|
307
|
-
role: 'user',
|
|
308
|
-
content: [
|
|
309
|
-
{ type: 'text', text: m.content ?? '' },
|
|
310
|
-
...m.images.map((url) => ({ type: 'image_url', image_url: { url } })),
|
|
311
|
-
],
|
|
312
|
-
});
|
|
313
|
-
} else if (m.role === 'system' || m.role === 'user') {
|
|
314
|
-
out.push({ role: m.role, content: m.content ?? '' });
|
|
315
|
-
} else if (m.role === 'assistant') {
|
|
316
|
-
const wire = { role: 'assistant', content: m.content || '' };
|
|
317
|
-
if (m.toolCalls?.length) {
|
|
318
|
-
wire.tool_calls = m.toolCalls.map((c) => ({
|
|
319
|
-
id: c.id,
|
|
320
|
-
type: 'function',
|
|
321
|
-
function: { name: c.name, arguments: JSON.stringify(c.args ?? {}) },
|
|
322
|
-
}));
|
|
323
|
-
}
|
|
324
|
-
out.push(wire);
|
|
325
|
-
} else if (m.role === 'tool') {
|
|
326
|
-
out.push({ role: 'tool', tool_call_id: m.toolCallId, content: m.content ?? '' });
|
|
327
|
-
}
|
|
328
|
-
}
|
|
329
|
-
return out;
|
|
330
|
-
}
|
|
331
|
-
|
|
332
|
-
// ---------------------------------------------------------------------------
|
|
333
|
-
// Turning provider errors into something a person can act on
|
|
334
|
-
// ---------------------------------------------------------------------------
|
|
335
|
-
|
|
336
|
-
function bodyOf(err) {
|
|
337
|
-
if (err?.error) return { error: err.error };
|
|
338
|
-
const raw = String(err?.message ?? '');
|
|
339
|
-
const start = raw.indexOf('{');
|
|
340
|
-
if (start === -1) return null;
|
|
341
|
-
try {
|
|
342
|
-
return JSON.parse(raw.slice(start));
|
|
343
|
-
} catch {
|
|
344
|
-
return null;
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
|
|
348
|
-
export function explain(err, id) {
|
|
349
|
-
if (err instanceof Failure) return err;
|
|
350
|
-
|
|
351
|
-
const status = err?.status ?? err?.statusCode ?? null;
|
|
352
|
-
const body = bodyOf(err);
|
|
353
|
-
const detail = body?.error?.message ?? String(err?.message ?? err);
|
|
354
|
-
const attempted = `asking ${modelName(id)} for a reply`;
|
|
355
|
-
|
|
356
|
-
// The useful text is often not on the error itself. undici reports a socket
|
|
357
|
-
// that closed mid-response as a bare `TypeError: terminated` and puts the
|
|
358
|
-
// real reason on `cause`, so the whole chain is matched rather than the top
|
|
359
|
-
// message alone — otherwise an ordinary dropped connection, which is worth
|
|
360
|
-
// retrying, gets reported as an unknown fault, which is not.
|
|
361
|
-
const chain = [err?.message, err?.code, err?.cause?.message, err?.cause?.code]
|
|
362
|
-
.filter(Boolean)
|
|
363
|
-
.join(' | ');
|
|
364
|
-
const raw = chain || String(err);
|
|
365
|
-
|
|
366
|
-
if (err?.name === 'AbortError' || /aborted|The operation was aborted/i.test(raw)) {
|
|
367
|
-
return new Failure({
|
|
368
|
-
kind: 'aborted',
|
|
369
|
-
attempted,
|
|
370
|
-
failed: 'The request was cancelled.',
|
|
371
|
-
fix: 'Send the message again when you are ready.',
|
|
372
|
-
cause: err,
|
|
373
|
-
});
|
|
374
|
-
}
|
|
375
|
-
|
|
376
|
-
if (status === 401 || status === 403 || /invalid[_ ]api[_ ]key/i.test(raw)) {
|
|
377
|
-
return new Failure({
|
|
378
|
-
kind: 'invalid_api_key',
|
|
379
|
-
attempted,
|
|
380
|
-
failed: `The API key was rejected (HTTP ${status ?? 401}).`,
|
|
381
|
-
fix:
|
|
382
|
-
'Check UCODE_API_KEY in ~/.ucode/.env for a typo or trailing space, and ' +
|
|
383
|
-
'confirm the key is still active in your account.',
|
|
384
|
-
cause: err,
|
|
385
|
-
});
|
|
386
|
-
}
|
|
387
|
-
|
|
388
|
-
if (status === 429 || /rate[_ ]limit/i.test(raw)) {
|
|
389
|
-
// `??` cannot be used to chain through Number(): Number(undefined) is NaN,
|
|
390
|
-
// which is neither null nor undefined, so it would swallow every fallback
|
|
391
|
-
// after it and the wait would silently never be found.
|
|
392
|
-
const header = err?.headers?.get?.('retry-after');
|
|
393
|
-
const asNumber = Number(header);
|
|
394
|
-
const retryAfter =
|
|
395
|
-
seconds(header) ??
|
|
396
|
-
(Number.isFinite(asNumber) && asNumber > 0 ? asNumber : null) ??
|
|
397
|
-
seconds(/try again in ([\dhms.]+)/i.exec(detail)?.[1]) ??
|
|
398
|
-
null;
|
|
399
|
-
// The daily cap reads "free-models-per-day-high-balance", with hyphens, and
|
|
400
|
-
// names its source in the metadata. It is one cap across every free model,
|
|
401
|
-
// so it is reported at once rather than waited on model after model.
|
|
402
|
-
const meta = body?.error?.metadata ?? {};
|
|
403
|
-
const daily = /per[- ]day|RPD|TPD|daily/i.test(`${detail} ${meta.limit_source ?? ''}`);
|
|
404
|
-
const resetMs = Number(meta.headers?.['X-RateLimit-Reset'] ?? err?.headers?.get?.('x-ratelimit-reset'));
|
|
405
|
-
const resetAt = daily && Number.isFinite(resetMs) && resetMs > Date.now() ? new Date(resetMs) : null;
|
|
406
|
-
const cap = Number(meta.headers?.['X-RateLimit-Limit']) || null;
|
|
407
|
-
const wait = Number.isFinite(retryAfter) && retryAfter
|
|
408
|
-
? (retryAfter >= 60 ? `${Math.ceil(retryAfter / 60)} min` : `${Math.ceil(retryAfter)}s`)
|
|
409
|
-
: null;
|
|
410
|
-
|
|
411
|
-
return new Failure({
|
|
412
|
-
kind: 'rate_limit',
|
|
413
|
-
attempted,
|
|
414
|
-
failed: daily
|
|
415
|
-
? `This key's free daily limit${cap ? ` of ${cap} requests` : ''} is used up. It covers every free model, so switching will not help.`
|
|
416
|
-
: `Too many requests for ${modelName(id)} just now${wait ? ` — clear in ${wait}` : ''}.`,
|
|
417
|
-
fix: daily
|
|
418
|
-
? `It resets ${resetAt ? `at ${resetAt.toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' })}` : 'once a day'}. ` +
|
|
419
|
-
'Adding credit to your account raises the limit.'
|
|
420
|
-
: 'ucode waits these out on its own. Free endpoints are shared, so it usually ' +
|
|
421
|
-
'clears in seconds; /model moves to a quieter one.',
|
|
422
|
-
detail: { retryAfter, daily, resetAt: resetAt?.getTime() ?? null },
|
|
423
|
-
cause: err,
|
|
424
|
-
});
|
|
425
|
-
}
|
|
426
|
-
|
|
427
|
-
// A 404 on a model in this list is almost never a bad name. It is OpenRouter
|
|
428
|
-
// having no upstream free to serve it at that instant, and it clears by
|
|
429
|
-
// itself — so it is retried rather than reported as a missing model.
|
|
430
|
-
if (status === 404 || /does not exist|not found|decommissioned/i.test(raw)) {
|
|
431
|
-
if (MODELS[id]) {
|
|
432
|
-
return new Failure({
|
|
433
|
-
kind: 'server',
|
|
434
|
-
attempted,
|
|
435
|
-
failed: `No provider was free to serve ${modelName(id)} at that moment.`,
|
|
436
|
-
fix: 'ucode retries this by itself. If it keeps up, /model switches.',
|
|
437
|
-
detail: { status },
|
|
438
|
-
cause: err,
|
|
439
|
-
});
|
|
440
|
-
}
|
|
441
|
-
return new Failure({
|
|
442
|
-
kind: 'bad_model',
|
|
443
|
-
attempted,
|
|
444
|
-
failed: `No model "${id}" is available to this key.`,
|
|
445
|
-
fix: `Run /model. ucode ships with: ${Object.keys(MODELS).join(', ')}`,
|
|
446
|
-
cause: err,
|
|
447
|
-
});
|
|
448
|
-
}
|
|
449
|
-
|
|
450
|
-
// Some upstreams validate tool calls themselves and reject an invented one,
|
|
451
|
-
// which fails the whole request. Recoverable: the loop feeds this back.
|
|
452
|
-
if (body?.error?.code === 'tool_use_failed' || /tool call validation failed/i.test(detail)) {
|
|
453
|
-
const attemptedName = /call tool '([^']+)'/.exec(detail)?.[1];
|
|
454
|
-
return new Failure({
|
|
455
|
-
kind: 'bad_tool_call',
|
|
456
|
-
attempted,
|
|
457
|
-
failed: attemptedName
|
|
458
|
-
? `The model tried to call "${attemptedName}", which is not one of its tools.`
|
|
459
|
-
: `The model produced a tool call the provider rejected: ${detail}`,
|
|
460
|
-
fix: 'Use only the tools supplied with the request.',
|
|
461
|
-
detail: { attemptedName },
|
|
462
|
-
cause: err,
|
|
463
|
-
});
|
|
464
|
-
}
|
|
465
|
-
|
|
466
|
-
if (status === 400) {
|
|
467
|
-
const noTools = /tool calling.*not supported/i.test(detail);
|
|
468
|
-
return new Failure({
|
|
469
|
-
kind: noTools ? 'no_tool_support' : 'bad_request',
|
|
470
|
-
attempted,
|
|
471
|
-
failed: noTools
|
|
472
|
-
? `${modelName(id)} cannot call tools, which ucode needs for every task.`
|
|
473
|
-
: `The request was rejected as malformed (HTTP 400): ${detail}`,
|
|
474
|
-
fix: noTools
|
|
475
|
-
? 'Run /model and pick another one.'
|
|
476
|
-
: '
|
|
477
|
-
cause: err,
|
|
478
|
-
});
|
|
479
|
-
}
|
|
480
|
-
|
|
481
|
-
if (status === 413 || /too large|context.*length/i.test(detail)) {
|
|
482
|
-
return new Failure({
|
|
483
|
-
kind: 'too_large',
|
|
484
|
-
attempted,
|
|
485
|
-
failed: `The conversation no longer fits in ${modelName(id)}: ${detail}`,
|
|
486
|
-
fix: 'Run /new for a fresh session, or lower UCODE_MAX_CONTEXT_TOKENS so ucode folds older turns away sooner.',
|
|
487
|
-
cause: err,
|
|
488
|
-
});
|
|
489
|
-
}
|
|
490
|
-
|
|
491
|
-
// OpenRouter drops a request when the upstream goes quiet on it. That is the
|
|
492
|
-
// ordinary failure mode of a busy free endpoint, and of a big reasoning model
|
|
493
|
-
// that spends a long time thinking before its first token. Nothing was
|
|
494
|
-
// generated, so retrying is safe — and ask() does it before anyone notices.
|
|
495
|
-
if (
|
|
496
|
-
status === 408 || status === 504 || status === 524 || status === 522 ||
|
|
497
|
-
err?.name === 'APIConnectionTimeoutError' ||
|
|
498
|
-
/idle timeout|timed out|timeout/i.test(detail) || /idle timeout|timed out/i.test(raw)
|
|
499
|
-
) {
|
|
500
|
-
return new Failure({
|
|
501
|
-
kind: 'timeout',
|
|
502
|
-
attempted,
|
|
503
|
-
failed: `${modelName(id)} sent nothing back in time — the provider dropped the request.`,
|
|
504
|
-
fix:
|
|
505
|
-
'Free endpoints stall under load, and the largest reasoning models are the ' +
|
|
506
|
-
'first to. ucode already retried. If it keeps happening, /model to Nemotron ' +
|
|
507
|
-
'3.5 Lightning or North Mini Code, which answer sooner.',
|
|
508
|
-
detail: { status },
|
|
509
|
-
cause: err,
|
|
510
|
-
});
|
|
511
|
-
}
|
|
512
|
-
|
|
513
|
-
if (status >= 500 || /internal|unavailable|overloaded/i.test(raw)) {
|
|
514
|
-
return new Failure({
|
|
515
|
-
kind: 'server',
|
|
516
|
-
attempted,
|
|
517
|
-
failed: `The provider returned HTTP ${status}. That is their side, not yours.`,
|
|
518
|
-
fix: 'Wait a few seconds and send again. If it persists, /model to another one.',
|
|
519
|
-
cause: err,
|
|
520
|
-
});
|
|
521
|
-
}
|
|
522
|
-
|
|
523
|
-
// `terminated` and `premature close` are what a connection dropped part way
|
|
524
|
-
// through a reply looks like. Nothing usable arrived, so it is safe to send
|
|
525
|
-
// again — and on a free endpoint under load it happens often enough that
|
|
526
|
-
// treating it as fatal would be the single most visible flaw in the agent.
|
|
527
|
-
if (
|
|
528
|
-
/ENOTFOUND|ECONNREFUSED|ECONNRESET|EAI_AGAIN|ETIMEDOUT|EPIPE|network|fetch failed|socket hang up|terminated|premature close|other side closed|UND_ERR/i.test(raw) ||
|
|
529
|
-
err?.name === 'APIConnectionError' ||
|
|
530
|
-
(err instanceof TypeError && /terminated/i.test(raw))
|
|
531
|
-
) {
|
|
532
|
-
return new Failure({
|
|
533
|
-
kind: 'network',
|
|
534
|
-
attempted,
|
|
535
|
-
failed: `The connection to the model dropped: ${raw}`,
|
|
536
|
-
fix:
|
|
537
|
-
'ucode retries this by itself. If it keeps happening, check your connection, ' +
|
|
538
|
-
'VPN and any corporate proxy (HTTPS_PROXY) — or /model to a lighter one, since ' +
|
|
539
|
-
'a long think on a busy free endpoint is the usual cause.',
|
|
540
|
-
cause: err,
|
|
541
|
-
});
|
|
542
|
-
}
|
|
543
|
-
|
|
544
|
-
return new Failure({
|
|
545
|
-
kind: 'unknown',
|
|
546
|
-
attempted,
|
|
547
|
-
failed: detail,
|
|
548
|
-
fix: 'Retry once. If it repeats, run ucode --debug for the full trace.',
|
|
549
|
-
cause: err,
|
|
550
|
-
});
|
|
551
|
-
}
|
|
552
|
-
|
|
553
|
-
const pause = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
554
|
-
|
|
555
|
-
// ---------------------------------------------------------------------------
|
|
556
|
-
// The one call
|
|
557
|
-
// ---------------------------------------------------------------------------
|
|
558
|
-
|
|
559
|
-
/**
|
|
560
|
-
* Send a conversation and get a normalized reply.
|
|
561
|
-
*
|
|
562
|
-
* @param {Array} messages neutral messages
|
|
563
|
-
* @param {Array} [tools] neutral tool definitions
|
|
564
|
-
* @param {object} [opts] { model, temperature, signal, maxOutputTokens,
|
|
565
|
-
* reasoning, attempts, onText, onThinking, onWait }
|
|
566
|
-
*/
|
|
567
|
-
export async function ask(messages, tools = [], opts = {}) {
|
|
568
|
-
const id = opts.model || current;
|
|
569
|
-
|
|
570
|
-
const request = {
|
|
571
|
-
model: id,
|
|
572
|
-
messages: wireMessages(messages),
|
|
573
|
-
// Ask for the working. Without it OpenRouter withholds reasoning entirely,
|
|
574
|
-
// and on a reasoning model that *is* the whole reply until the very end:
|
|
575
|
-
// the socket sits silent for the length of the think, the screen shows
|
|
576
|
-
// nothing, and the provider eventually drops the request as idle. Asking
|
|
577
|
-
// for it fixes the blank screen and the dropped request together. Models
|
|
578
|
-
// that do not reason ignore the flag.
|
|
579
|
-
include_reasoning: true,
|
|
580
|
-
};
|
|
581
|
-
|
|
582
|
-
const wired = wireTools(tools);
|
|
583
|
-
if (wired) {
|
|
584
|
-
request.tools = wired;
|
|
585
|
-
request.tool_choice = 'auto';
|
|
586
|
-
}
|
|
587
|
-
if (opts.temperature !== undefined) request.temperature = opts.temperature;
|
|
588
|
-
if (opts.maxOutputTokens) request.max_tokens = opts.maxOutputTokens;
|
|
589
|
-
if (opts.reasoning) request.reasoning = opts.reasoning;
|
|
590
|
-
|
|
591
|
-
// A side call (the design review) passes fewer: it is better skipped than
|
|
592
|
-
// waited on through a string of rate-limit pauses.
|
|
593
|
-
const attempts = opts.attempts ?? 4;
|
|
594
|
-
let problem;
|
|
595
|
-
|
|
596
|
-
// Text already on screen cannot be unprinted, so a stream is only safe to
|
|
597
|
-
// retry while it is still silent. Every timeout worth retrying happens
|
|
598
|
-
// before the first token, so this costs nothing in practice.
|
|
599
|
-
let printed = 0;
|
|
600
|
-
const callOpts = opts.onText
|
|
601
|
-
? { ...opts, onText: (d) => { printed += d.length; opts.onText(d); } }
|
|
602
|
-
: opts;
|
|
603
|
-
|
|
604
|
-
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
605
|
-
try {
|
|
606
|
-
if (opts.onText) return await streamed(request, callOpts, id);
|
|
607
|
-
const { data, response } = await connection().chat.completions
|
|
608
|
-
.create(request, { signal: opts.signal })
|
|
609
|
-
.withResponse();
|
|
610
|
-
noteLimits(response?.headers);
|
|
611
|
-
return normalize(data, id);
|
|
612
|
-
} catch (err) {
|
|
613
|
-
noteLimits(err?.headers);
|
|
614
|
-
problem = explain(err, id);
|
|
615
|
-
|
|
616
|
-
// A per-minute limit is a wait, not a failure. Sit it out rather than
|
|
617
|
-
// making the user retype their message. Free endpoints often refuse
|
|
618
|
-
// without saying how long to wait, so when there is no retry-after the
|
|
619
|
-
// pauses grow on their own — 5s, 10s, 20s — and only then does the
|
|
620
|
-
// error go up to the loop, which moves to another model.
|
|
621
|
-
const told = problem.detail?.retryAfter;
|
|
622
|
-
const wait = Number.isFinite(told) && told > 0 && told <= 90 ? told : RATE_LIMIT_BACKOFF[attempt - 1];
|
|
623
|
-
if (
|
|
624
|
-
problem.kind === 'rate_limit' && !problem.detail?.daily &&
|
|
625
|
-
wait && attempt < attempts && printed === 0 && !opts.signal?.aborted
|
|
626
|
-
) {
|
|
627
|
-
const until = Date.now() + wait * 1000;
|
|
628
|
-
while (Date.now() < until && !opts.signal?.aborted) {
|
|
629
|
-
opts.onWait?.(`rate limited — resuming in ${Math.ceil((until - Date.now()) / 1000)}s`);
|
|
630
|
-
await pause(Math.min(1000, until - Date.now()));
|
|
631
|
-
}
|
|
632
|
-
if (opts.signal?.aborted) break;
|
|
633
|
-
continue;
|
|
634
|
-
}
|
|
635
|
-
|
|
636
|
-
const worthRetrying =
|
|
637
|
-
problem.kind === 'server' || problem.kind === 'network' || problem.kind === 'timeout';
|
|
638
|
-
if (!worthRetrying || attempt === attempts || opts.signal?.aborted) break;
|
|
639
|
-
if (printed > 0) break; // half an answer is on screen; do not print it twice
|
|
640
|
-
|
|
641
|
-
// A stalled provider needs longer to come back than a dropped socket
|
|
642
|
-
// does, and the wait is narrated so a slow turn never looks like a hang.
|
|
643
|
-
const backoff = problem.kind === 'timeout'
|
|
644
|
-
? 1500 * 2 ** (attempt - 1)
|
|
645
|
-
: 400 * 2 ** (attempt - 1);
|
|
646
|
-
opts.onWait?.(
|
|
647
|
-
`${problem.kind === 'timeout' ? 'provider stalled' : 'connection failed'} — ` +
|
|
648
|
-
`retrying (${attempt + 1}/${attempts})`
|
|
649
|
-
);
|
|
650
|
-
await pause(backoff);
|
|
651
|
-
}
|
|
652
|
-
}
|
|
653
|
-
|
|
654
|
-
throw problem;
|
|
655
|
-
}
|
|
656
|
-
|
|
657
|
-
/** Collect a streamed reply, handing deltas out as they land. */
|
|
658
|
-
async function streamed(request, opts, id) {
|
|
659
|
-
const { data: stream, response } = await connection().chat.completions
|
|
660
|
-
.create(
|
|
661
|
-
{ ...request, stream: true, stream_options: { include_usage: true } },
|
|
662
|
-
{ signal: opts.signal }
|
|
663
|
-
)
|
|
664
|
-
.withResponse();
|
|
665
|
-
noteLimits(response?.headers);
|
|
666
|
-
|
|
667
|
-
let text = '';
|
|
668
|
-
let reasoning = '';
|
|
669
|
-
let finishReason = 'stop';
|
|
670
|
-
let usage = null;
|
|
671
|
-
const partial = new Map();
|
|
672
|
-
const handed = new Set();
|
|
673
|
-
let highest = -1;
|
|
674
|
-
|
|
675
|
-
for await (const chunk of stream) {
|
|
676
|
-
if (opts.signal?.aborted) break;
|
|
677
|
-
if (chunk.usage) usage = chunk.usage;
|
|
678
|
-
|
|
679
|
-
const choice = chunk.choices?.[0];
|
|
680
|
-
if (!choice) continue;
|
|
681
|
-
if (choice.finish_reason) finishReason = choice.finish_reason;
|
|
682
|
-
const delta = choice.delta ?? {};
|
|
683
|
-
|
|
684
|
-
// Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
|
|
685
|
-
// `reasoning_content` on some upstreams.
|
|
686
|
-
const thinking = delta.reasoning ?? delta.reasoning_content;
|
|
687
|
-
if (thinking) {
|
|
688
|
-
reasoning += thinking;
|
|
689
|
-
opts.onThinking?.(thinking);
|
|
690
|
-
}
|
|
691
|
-
|
|
692
|
-
if (delta.content) {
|
|
693
|
-
text += delta.content;
|
|
694
|
-
opts.onText(delta.content);
|
|
695
|
-
}
|
|
696
|
-
|
|
697
|
-
// A tool call's name and arguments arrive across several chunks, keyed by
|
|
698
|
-
// index, so they are stitched back together here.
|
|
699
|
-
for (const call of delta.tool_calls ?? []) {
|
|
700
|
-
// Calls arrive one after another, so the first chunk of call N means
|
|
701
|
-
// every call before it is complete. Those are handed over at once, and
|
|
702
|
-
// the caller can start running them while the rest are still being
|
|
703
|
-
// written — the reply streaming and the tools working overlap.
|
|
704
|
-
if (opts.onToolCall && call.index > highest) {
|
|
705
|
-
for (const [index, slot] of partial) {
|
|
706
|
-
if (index < call.index && !handed.has(index)) {
|
|
707
|
-
handed.add(index);
|
|
708
|
-
opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args }));
|
|
709
|
-
}
|
|
710
|
-
}
|
|
711
|
-
highest = call.index;
|
|
712
|
-
}
|
|
713
|
-
|
|
714
|
-
const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
|
|
715
|
-
if (call.id) slot.id = call.id;
|
|
716
|
-
if (call.function?.name) slot.name += call.function.name;
|
|
717
|
-
if (call.function?.arguments) slot.args += call.function.arguments;
|
|
718
|
-
partial.set(call.index, slot);
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
toolCalls,
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
*
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
const
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
const
|
|
843
|
-
|
|
844
|
-
const
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
1
|
+
/**
|
|
2
|
+
* provider.js — the only module that knows a model provider exists.
|
|
3
|
+
*
|
|
4
|
+
* Everything above this file speaks one small neutral message format and calls
|
|
5
|
+
* ask(). Moving ucode to a different host means rewriting this file and
|
|
6
|
+
* nothing else.
|
|
7
|
+
*
|
|
8
|
+
* The neutral formats:
|
|
9
|
+
* { role: 'system', content }
|
|
10
|
+
* { role: 'user', content, images?: [dataUrl] }
|
|
11
|
+
* { role: 'assistant', content?, toolCalls?: [{ id, name, args }] }
|
|
12
|
+
* { role: 'tool', toolCallId, name, content }
|
|
13
|
+
*
|
|
14
|
+
* tool: { name, description, parameters: <JSON Schema> }
|
|
15
|
+
*
|
|
16
|
+
* ask() resolves to:
|
|
17
|
+
* { text, reasoning, toolCalls, usage, finishReason, model }
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import { join, dirname } from 'node:path';
|
|
21
|
+
import { homedir } from 'node:os';
|
|
22
|
+
import { fileURLToPath } from 'node:url';
|
|
23
|
+
import dotenv from 'dotenv';
|
|
24
|
+
import OpenAI from 'openai';
|
|
25
|
+
import { jsonrepair } from 'jsonrepair';
|
|
26
|
+
import { Failure } from './failure.js';
|
|
27
|
+
|
|
28
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
29
|
+
const PACKAGE_ROOT = join(HERE, '..', '..');
|
|
30
|
+
|
|
31
|
+
/** Personal config lives here, outside the package, so upgrades never touch it. */
|
|
32
|
+
export const UCODE_HOME = join(homedir(), '.ucode');
|
|
33
|
+
export const ENV_FILE = join(UCODE_HOME, '.env');
|
|
34
|
+
|
|
35
|
+
export const BASE_URL = 'https://openrouter.ai/api/v1';
|
|
36
|
+
export const PROVIDER = 'OpenRouter';
|
|
37
|
+
|
|
38
|
+
// First definition wins — dotenv never overwrites a variable that already
|
|
39
|
+
// exists — so the order here is the precedence order:
|
|
40
|
+
// real environment > ./.env (this project) > ~/.ucode/.env (this machine)
|
|
41
|
+
// > the checkout's own .env (only when developing on a clone)
|
|
42
|
+
dotenv.config({ path: join(process.cwd(), '.env'), quiet: true });
|
|
43
|
+
dotenv.config({ path: ENV_FILE, quiet: true });
|
|
44
|
+
dotenv.config({ path: join(PACKAGE_ROOT, '.env'), quiet: true });
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* The whole model list. Not a starting point — the list.
|
|
48
|
+
*
|
|
49
|
+
* ucode runs on NVIDIA and Cohere only. Both vendors serve genuinely capable
|
|
50
|
+
* models free through OpenRouter, both handle tool calling properly, and
|
|
51
|
+
* keeping the set to five means every one of them has been used in anger
|
|
52
|
+
* rather than listed on the strength of a benchmark. A picker offering sixty
|
|
53
|
+
* models is a picker nobody reads.
|
|
54
|
+
*
|
|
55
|
+
* `name` is what the status bar shows. `note` is what the picker shows.
|
|
56
|
+
*/
|
|
57
|
+
export const MODELS = {
|
|
58
|
+
'nvidia/nemotron-3-ultra-550b-a55b:free': {
|
|
59
|
+
name: 'Nemotron 3 Ultra',
|
|
60
|
+
context: 1_000_000,
|
|
61
|
+
star: true,
|
|
62
|
+
note: 'deepest reasoning, 1M context — slowest to answer',
|
|
63
|
+
},
|
|
64
|
+
'nvidia/nemotron-3.5-lightning:free': {
|
|
65
|
+
name: 'Nemotron 3.5 Lightning',
|
|
66
|
+
context: 1_000_000,
|
|
67
|
+
note: 'same huge window, answers much sooner',
|
|
68
|
+
},
|
|
69
|
+
'nvidia/nemotron-3-super-120b-a12b:free': {
|
|
70
|
+
name: 'Nemotron 3 Super',
|
|
71
|
+
context: 262_144,
|
|
72
|
+
note: 'strong all-rounder, quick to first token',
|
|
73
|
+
},
|
|
74
|
+
'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free': {
|
|
75
|
+
name: 'Nemotron 3 Nano Omni',
|
|
76
|
+
context: 256_000,
|
|
77
|
+
note: 'small and fast, reasoning tuned',
|
|
78
|
+
},
|
|
79
|
+
'cohere/north-mini-code:free': {
|
|
80
|
+
name: 'North Mini Code',
|
|
81
|
+
context: 256_000,
|
|
82
|
+
star: true,
|
|
83
|
+
note: 'the default — built for code and interface work, quick to answer',
|
|
84
|
+
},
|
|
85
|
+
};
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* North Mini Code is the default: it is built for code and interface work,
|
|
89
|
+
* which is what ucode is mostly asked to do, and it answers far sooner than
|
|
90
|
+
* the big reasoning models. /model moves to Ultra when a problem needs the
|
|
91
|
+
* million-token window and the long think more than it needs the speed.
|
|
92
|
+
*/
|
|
93
|
+
export const DEFAULT_MODEL = 'cohere/north-mini-code:free';
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Where to go when a model is busy, in order of preference. Each is served by
|
|
97
|
+
* a different upstream, so a rate limit on one rarely means a limit on the
|
|
98
|
+
* next — which is what lets a long build keep going instead of stopping at
|
|
99
|
+
* the first "too many requests".
|
|
100
|
+
*/
|
|
101
|
+
export const FALLBACKS = [
|
|
102
|
+
'cohere/north-mini-code:free',
|
|
103
|
+
'nvidia/nemotron-3.5-lightning:free',
|
|
104
|
+
'nvidia/nemotron-3-super-120b-a12b:free',
|
|
105
|
+
'nvidia/nemotron-3-ultra-550b-a55b:free',
|
|
106
|
+
];
|
|
107
|
+
|
|
108
|
+
/** The next model to try after `id`, skipping any already tried this round. */
|
|
109
|
+
export function fallbackFor(id, tried = new Set()) {
|
|
110
|
+
const start = Math.max(0, FALLBACKS.indexOf(id));
|
|
111
|
+
for (let i = 1; i <= FALLBACKS.length; i++) {
|
|
112
|
+
const next = FALLBACKS[(start + i) % FALLBACKS.length];
|
|
113
|
+
if (next !== id && !tried.has(next)) return next;
|
|
114
|
+
}
|
|
115
|
+
return null;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** Seconds to wait on successive rate limits that come with no retry-after. */
|
|
119
|
+
const RATE_LIMIT_BACKOFF = [5, 10, 20];
|
|
120
|
+
|
|
121
|
+
let current = process.env.UCODE_MODEL || DEFAULT_MODEL;
|
|
122
|
+
let client = null;
|
|
123
|
+
|
|
124
|
+
export function model() {
|
|
125
|
+
return current;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
export function setModel(id) {
|
|
129
|
+
const wanted = String(id ?? '').trim();
|
|
130
|
+
if (!wanted) {
|
|
131
|
+
throw new Failure({
|
|
132
|
+
kind: 'bad_model',
|
|
133
|
+
attempted: 'switching model',
|
|
134
|
+
failed: 'No model name was given.',
|
|
135
|
+
fix: `Pick one of: ${Object.keys(MODELS).join(', ')}`,
|
|
136
|
+
});
|
|
137
|
+
}
|
|
138
|
+
if (!MODELS[wanted]) {
|
|
139
|
+
throw new Failure({
|
|
140
|
+
kind: 'bad_model',
|
|
141
|
+
attempted: `switching to "${wanted}"`,
|
|
142
|
+
failed: 'ucode only runs NVIDIA and Cohere models, and that is not one of them.',
|
|
143
|
+
fix: `Run /model to choose from: ${Object.keys(MODELS).join(', ')}`,
|
|
144
|
+
});
|
|
145
|
+
}
|
|
146
|
+
current = wanted;
|
|
147
|
+
return current;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/** The short name for a model id: "Nemotron 3 Ultra". */
|
|
151
|
+
export function modelName(id = current) {
|
|
152
|
+
return MODELS[id]?.name ?? id;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** Every model, in list order, annotated with whether it is the active one. */
|
|
156
|
+
export function modelList() {
|
|
157
|
+
return Object.entries(MODELS).map(([id, info]) => ({
|
|
158
|
+
id,
|
|
159
|
+
...info,
|
|
160
|
+
active: id === current,
|
|
161
|
+
}));
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* How many tokens one request may occupy.
|
|
166
|
+
*
|
|
167
|
+
* OpenRouter meters requests rather than tokens, so nothing here is rationing
|
|
168
|
+
* a quota — the binding limit is simply the window the model has. On the two
|
|
169
|
+
* million-token models compaction essentially never fires.
|
|
170
|
+
*/
|
|
171
|
+
export function contextLimit(id = current) {
|
|
172
|
+
const override = Number(process.env.UCODE_MAX_CONTEXT_TOKENS);
|
|
173
|
+
if (Number.isFinite(override) && override > 0) return override;
|
|
174
|
+
return MODELS[id]?.context ?? 128_000;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* A rough local token count, used to decide when to compact *before* a
|
|
179
|
+
* request goes out. Real numbers come back in the response usage; this only
|
|
180
|
+
* has to be close enough to trigger at the right time.
|
|
181
|
+
*/
|
|
182
|
+
export function estimateTokens(text) {
|
|
183
|
+
return text ? Math.ceil(String(text).length / 4) : 0;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
export function estimateConversation(messages) {
|
|
187
|
+
let total = 0;
|
|
188
|
+
for (const m of messages) {
|
|
189
|
+
total += estimateTokens(m.content || '');
|
|
190
|
+
for (const call of m.toolCalls || []) {
|
|
191
|
+
total += estimateTokens(call.name) + estimateTokens(JSON.stringify(call.args || {}));
|
|
192
|
+
}
|
|
193
|
+
total += 4; // role and framing overhead per message
|
|
194
|
+
}
|
|
195
|
+
return total;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ---------------------------------------------------------------------------
|
|
199
|
+
// Client
|
|
200
|
+
// ---------------------------------------------------------------------------
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* No key is bundled here, and none ever should be. A key committed to a public
|
|
204
|
+
* package is readable by anyone who runs `npm pack ucode-agent`, and no amount
|
|
205
|
+
* of first-run convenience is worth handing out a live credential.
|
|
206
|
+
*/
|
|
207
|
+
function apiKey() {
|
|
208
|
+
// UCODE_API_KEY is the documented name. The provider's own variable name is
|
|
209
|
+
// still read, so a key set up for another tool keeps working here.
|
|
210
|
+
const key = (process.env.UCODE_API_KEY || process.env.OPENROUTER_API_KEY || '').trim();
|
|
211
|
+
if (!key) {
|
|
212
|
+
throw new Failure({
|
|
213
|
+
kind: 'no_api_key',
|
|
214
|
+
attempted: 'connecting to the model',
|
|
215
|
+
failed: 'No API key is set - UCODE_API_KEY is missing from the environment and from every .env file.',
|
|
216
|
+
fix:
|
|
217
|
+
`Put UCODE_API_KEY=your-key in ${ENV_FILE} — that applies to every ` +
|
|
218
|
+
'project on this machine — or in a .env file beside your code. ' +
|
|
219
|
+
'Free keys: https://openrouter.ai/keys',
|
|
220
|
+
});
|
|
221
|
+
}
|
|
222
|
+
return key;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
function connection() {
|
|
226
|
+
if (client) return client;
|
|
227
|
+
client = new OpenAI({
|
|
228
|
+
apiKey: apiKey(),
|
|
229
|
+
baseURL: process.env.UCODE_BASE_URL || BASE_URL,
|
|
230
|
+
// Nemotron Ultra can think for a long time before its first token, so the
|
|
231
|
+
// ceiling is deliberately generous. maxRetries is 0 because ask() owns
|
|
232
|
+
// retrying: its attempts are narrated on screen instead of happening
|
|
233
|
+
// silently somewhere inside the SDK.
|
|
234
|
+
timeout: Number(process.env.UCODE_REQUEST_TIMEOUT_MS) || 300_000,
|
|
235
|
+
maxRetries: 0,
|
|
236
|
+
defaultHeaders: {
|
|
237
|
+
'HTTP-Referer': 'https://github.com/sppideey/ucode-agent',
|
|
238
|
+
'X-Title': 'ucode',
|
|
239
|
+
},
|
|
240
|
+
});
|
|
241
|
+
return client;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/** Drop the cached client so the next request picks up a changed key. */
|
|
245
|
+
export function resetConnection() {
|
|
246
|
+
client = null;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// ---------------------------------------------------------------------------
|
|
250
|
+
// Live quota, taken from whatever rate-limit headers come back
|
|
251
|
+
// ---------------------------------------------------------------------------
|
|
252
|
+
|
|
253
|
+
let limits = null;
|
|
254
|
+
|
|
255
|
+
export function rateLimits() {
|
|
256
|
+
return limits;
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/** "1.5s", "2m59.56s", "1h2m" -> seconds */
|
|
260
|
+
function seconds(value) {
|
|
261
|
+
if (!value) return null;
|
|
262
|
+
const m = /^(?:(\d+(?:\.\d+)?)h)?(?:(\d+(?:\.\d+)?)m(?!s))?(?:(\d+(?:\.\d+)?)m?s)?$/
|
|
263
|
+
.exec(String(value).trim());
|
|
264
|
+
if (!m) return null;
|
|
265
|
+
const total = (parseFloat(m[1]) || 0) * 3600 + (parseFloat(m[2]) || 0) * 60 + (parseFloat(m[3]) || 0);
|
|
266
|
+
return total > 0 ? total : null;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
function noteLimits(headers) {
|
|
270
|
+
if (!headers?.get) return;
|
|
271
|
+
const num = (name) => {
|
|
272
|
+
const n = Number(headers.get(name));
|
|
273
|
+
return Number.isFinite(n) ? n : null;
|
|
274
|
+
};
|
|
275
|
+
limits = {
|
|
276
|
+
requestsLimit: num('x-ratelimit-limit-requests'),
|
|
277
|
+
requestsRemaining: num('x-ratelimit-remaining-requests'),
|
|
278
|
+
requestsReset: seconds(headers.get('x-ratelimit-reset-requests')),
|
|
279
|
+
tokensLimit: num('x-ratelimit-limit-tokens'),
|
|
280
|
+
tokensRemaining: num('x-ratelimit-remaining-tokens'),
|
|
281
|
+
tokensReset: seconds(headers.get('x-ratelimit-reset-tokens')),
|
|
282
|
+
at: Date.now(),
|
|
283
|
+
};
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
// ---------------------------------------------------------------------------
|
|
287
|
+
// Neutral format -> wire format
|
|
288
|
+
// ---------------------------------------------------------------------------
|
|
289
|
+
|
|
290
|
+
function wireTools(tools) {
|
|
291
|
+
if (!tools?.length) return undefined;
|
|
292
|
+
return tools.map((t) => ({
|
|
293
|
+
type: 'function',
|
|
294
|
+
function: {
|
|
295
|
+
name: t.name,
|
|
296
|
+
description: t.description,
|
|
297
|
+
parameters: t.parameters ?? { type: 'object', properties: {} },
|
|
298
|
+
},
|
|
299
|
+
}));
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
function wireMessages(messages) {
|
|
303
|
+
const out = [];
|
|
304
|
+
for (const m of messages) {
|
|
305
|
+
if (m.role === 'user' && m.images?.length) {
|
|
306
|
+
out.push({
|
|
307
|
+
role: 'user',
|
|
308
|
+
content: [
|
|
309
|
+
{ type: 'text', text: m.content ?? '' },
|
|
310
|
+
...m.images.map((url) => ({ type: 'image_url', image_url: { url } })),
|
|
311
|
+
],
|
|
312
|
+
});
|
|
313
|
+
} else if (m.role === 'system' || m.role === 'user') {
|
|
314
|
+
out.push({ role: m.role, content: m.content ?? '' });
|
|
315
|
+
} else if (m.role === 'assistant') {
|
|
316
|
+
const wire = { role: 'assistant', content: m.content || '' };
|
|
317
|
+
if (m.toolCalls?.length) {
|
|
318
|
+
wire.tool_calls = m.toolCalls.map((c) => ({
|
|
319
|
+
id: c.id,
|
|
320
|
+
type: 'function',
|
|
321
|
+
function: { name: c.name, arguments: JSON.stringify(c.args ?? {}) },
|
|
322
|
+
}));
|
|
323
|
+
}
|
|
324
|
+
out.push(wire);
|
|
325
|
+
} else if (m.role === 'tool') {
|
|
326
|
+
out.push({ role: 'tool', tool_call_id: m.toolCallId, content: m.content ?? '' });
|
|
327
|
+
}
|
|
328
|
+
}
|
|
329
|
+
return out;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
// ---------------------------------------------------------------------------
|
|
333
|
+
// Turning provider errors into something a person can act on
|
|
334
|
+
// ---------------------------------------------------------------------------
|
|
335
|
+
|
|
336
|
+
function bodyOf(err) {
|
|
337
|
+
if (err?.error) return { error: err.error };
|
|
338
|
+
const raw = String(err?.message ?? '');
|
|
339
|
+
const start = raw.indexOf('{');
|
|
340
|
+
if (start === -1) return null;
|
|
341
|
+
try {
|
|
342
|
+
return JSON.parse(raw.slice(start));
|
|
343
|
+
} catch {
|
|
344
|
+
return null;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
export function explain(err, id) {
|
|
349
|
+
if (err instanceof Failure) return err;
|
|
350
|
+
|
|
351
|
+
const status = err?.status ?? err?.statusCode ?? null;
|
|
352
|
+
const body = bodyOf(err);
|
|
353
|
+
const detail = body?.error?.message ?? String(err?.message ?? err);
|
|
354
|
+
const attempted = `asking ${modelName(id)} for a reply`;
|
|
355
|
+
|
|
356
|
+
// The useful text is often not on the error itself. undici reports a socket
|
|
357
|
+
// that closed mid-response as a bare `TypeError: terminated` and puts the
|
|
358
|
+
// real reason on `cause`, so the whole chain is matched rather than the top
|
|
359
|
+
// message alone — otherwise an ordinary dropped connection, which is worth
|
|
360
|
+
// retrying, gets reported as an unknown fault, which is not.
|
|
361
|
+
const chain = [err?.message, err?.code, err?.cause?.message, err?.cause?.code]
|
|
362
|
+
.filter(Boolean)
|
|
363
|
+
.join(' | ');
|
|
364
|
+
const raw = chain || String(err);
|
|
365
|
+
|
|
366
|
+
if (err?.name === 'AbortError' || /aborted|The operation was aborted/i.test(raw)) {
|
|
367
|
+
return new Failure({
|
|
368
|
+
kind: 'aborted',
|
|
369
|
+
attempted,
|
|
370
|
+
failed: 'The request was cancelled.',
|
|
371
|
+
fix: 'Send the message again when you are ready.',
|
|
372
|
+
cause: err,
|
|
373
|
+
});
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
if (status === 401 || status === 403 || /invalid[_ ]api[_ ]key/i.test(raw)) {
|
|
377
|
+
return new Failure({
|
|
378
|
+
kind: 'invalid_api_key',
|
|
379
|
+
attempted,
|
|
380
|
+
failed: `The API key was rejected (HTTP ${status ?? 401}).`,
|
|
381
|
+
fix:
|
|
382
|
+
'Check UCODE_API_KEY in ~/.ucode/.env for a typo or trailing space, and ' +
|
|
383
|
+
'confirm the key is still active in your account.',
|
|
384
|
+
cause: err,
|
|
385
|
+
});
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
if (status === 429 || /rate[_ ]limit/i.test(raw)) {
|
|
389
|
+
// `??` cannot be used to chain through Number(): Number(undefined) is NaN,
|
|
390
|
+
// which is neither null nor undefined, so it would swallow every fallback
|
|
391
|
+
// after it and the wait would silently never be found.
|
|
392
|
+
const header = err?.headers?.get?.('retry-after');
|
|
393
|
+
const asNumber = Number(header);
|
|
394
|
+
const retryAfter =
|
|
395
|
+
seconds(header) ??
|
|
396
|
+
(Number.isFinite(asNumber) && asNumber > 0 ? asNumber : null) ??
|
|
397
|
+
seconds(/try again in ([\dhms.]+)/i.exec(detail)?.[1]) ??
|
|
398
|
+
null;
|
|
399
|
+
// The daily cap reads "free-models-per-day-high-balance", with hyphens, and
|
|
400
|
+
// names its source in the metadata. It is one cap across every free model,
|
|
401
|
+
// so it is reported at once rather than waited on model after model.
|
|
402
|
+
const meta = body?.error?.metadata ?? {};
|
|
403
|
+
const daily = /per[- ]day|RPD|TPD|daily/i.test(`${detail} ${meta.limit_source ?? ''}`);
|
|
404
|
+
const resetMs = Number(meta.headers?.['X-RateLimit-Reset'] ?? err?.headers?.get?.('x-ratelimit-reset'));
|
|
405
|
+
const resetAt = daily && Number.isFinite(resetMs) && resetMs > Date.now() ? new Date(resetMs) : null;
|
|
406
|
+
const cap = Number(meta.headers?.['X-RateLimit-Limit']) || null;
|
|
407
|
+
const wait = Number.isFinite(retryAfter) && retryAfter
|
|
408
|
+
? (retryAfter >= 60 ? `${Math.ceil(retryAfter / 60)} min` : `${Math.ceil(retryAfter)}s`)
|
|
409
|
+
: null;
|
|
410
|
+
|
|
411
|
+
return new Failure({
|
|
412
|
+
kind: 'rate_limit',
|
|
413
|
+
attempted,
|
|
414
|
+
failed: daily
|
|
415
|
+
? `This key's free daily limit${cap ? ` of ${cap} requests` : ''} is used up. It covers every free model, so switching will not help.`
|
|
416
|
+
: `Too many requests for ${modelName(id)} just now${wait ? ` — clear in ${wait}` : ''}.`,
|
|
417
|
+
fix: daily
|
|
418
|
+
? `It resets ${resetAt ? `at ${resetAt.toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' })}` : 'once a day'}. ` +
|
|
419
|
+
'Adding credit to your account raises the limit.'
|
|
420
|
+
: 'ucode waits these out on its own. Free endpoints are shared, so it usually ' +
|
|
421
|
+
'clears in seconds; /model moves to a quieter one.',
|
|
422
|
+
detail: { retryAfter, daily, resetAt: resetAt?.getTime() ?? null },
|
|
423
|
+
cause: err,
|
|
424
|
+
});
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
// A 404 on a model in this list is almost never a bad name. It is OpenRouter
|
|
428
|
+
// having no upstream free to serve it at that instant, and it clears by
|
|
429
|
+
// itself — so it is retried rather than reported as a missing model.
|
|
430
|
+
if (status === 404 || /does not exist|not found|decommissioned/i.test(raw)) {
|
|
431
|
+
if (MODELS[id]) {
|
|
432
|
+
return new Failure({
|
|
433
|
+
kind: 'server',
|
|
434
|
+
attempted,
|
|
435
|
+
failed: `No provider was free to serve ${modelName(id)} at that moment.`,
|
|
436
|
+
fix: 'ucode retries this by itself. If it keeps up, /model switches.',
|
|
437
|
+
detail: { status },
|
|
438
|
+
cause: err,
|
|
439
|
+
});
|
|
440
|
+
}
|
|
441
|
+
return new Failure({
|
|
442
|
+
kind: 'bad_model',
|
|
443
|
+
attempted,
|
|
444
|
+
failed: `No model "${id}" is available to this key.`,
|
|
445
|
+
fix: `Run /model. ucode ships with: ${Object.keys(MODELS).join(', ')}`,
|
|
446
|
+
cause: err,
|
|
447
|
+
});
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
// Some upstreams validate tool calls themselves and reject an invented one,
|
|
451
|
+
// which fails the whole request. Recoverable: the loop feeds this back.
|
|
452
|
+
if (body?.error?.code === 'tool_use_failed' || /tool call validation failed/i.test(detail)) {
|
|
453
|
+
const attemptedName = /call tool '([^']+)'/.exec(detail)?.[1];
|
|
454
|
+
return new Failure({
|
|
455
|
+
kind: 'bad_tool_call',
|
|
456
|
+
attempted,
|
|
457
|
+
failed: attemptedName
|
|
458
|
+
? `The model tried to call "${attemptedName}", which is not one of its tools.`
|
|
459
|
+
: `The model produced a tool call the provider rejected: ${detail}`,
|
|
460
|
+
fix: 'Use only the tools supplied with the request.',
|
|
461
|
+
detail: { attemptedName },
|
|
462
|
+
cause: err,
|
|
463
|
+
});
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
if (status === 400) {
|
|
467
|
+
const noTools = /tool calling.*not supported/i.test(detail);
|
|
468
|
+
return new Failure({
|
|
469
|
+
kind: noTools ? 'no_tool_support' : 'bad_request',
|
|
470
|
+
attempted,
|
|
471
|
+
failed: noTools
|
|
472
|
+
? `${modelName(id)} cannot call tools, which ucode needs for every task.`
|
|
473
|
+
: `The request was rejected as malformed (HTTP 400): ${detail}`,
|
|
474
|
+
fix: noTools
|
|
475
|
+
? 'Run /model and pick another one.'
|
|
476
|
+
: 'The conversation has something in it the provider will not accept. /new starts a fresh one.',
|
|
477
|
+
cause: err,
|
|
478
|
+
});
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
if (status === 413 || /too large|context.*length/i.test(detail)) {
|
|
482
|
+
return new Failure({
|
|
483
|
+
kind: 'too_large',
|
|
484
|
+
attempted,
|
|
485
|
+
failed: `The conversation no longer fits in ${modelName(id)}: ${detail}`,
|
|
486
|
+
fix: 'Run /new for a fresh session, or lower UCODE_MAX_CONTEXT_TOKENS so ucode folds older turns away sooner.',
|
|
487
|
+
cause: err,
|
|
488
|
+
});
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
// OpenRouter drops a request when the upstream goes quiet on it. That is the
|
|
492
|
+
// ordinary failure mode of a busy free endpoint, and of a big reasoning model
|
|
493
|
+
// that spends a long time thinking before its first token. Nothing was
|
|
494
|
+
// generated, so retrying is safe — and ask() does it before anyone notices.
|
|
495
|
+
if (
|
|
496
|
+
status === 408 || status === 504 || status === 524 || status === 522 ||
|
|
497
|
+
err?.name === 'APIConnectionTimeoutError' ||
|
|
498
|
+
/idle timeout|timed out|timeout/i.test(detail) || /idle timeout|timed out/i.test(raw)
|
|
499
|
+
) {
|
|
500
|
+
return new Failure({
|
|
501
|
+
kind: 'timeout',
|
|
502
|
+
attempted,
|
|
503
|
+
failed: `${modelName(id)} sent nothing back in time — the provider dropped the request.`,
|
|
504
|
+
fix:
|
|
505
|
+
'Free endpoints stall under load, and the largest reasoning models are the ' +
|
|
506
|
+
'first to. ucode already retried. If it keeps happening, /model to Nemotron ' +
|
|
507
|
+
'3.5 Lightning or North Mini Code, which answer sooner.',
|
|
508
|
+
detail: { status },
|
|
509
|
+
cause: err,
|
|
510
|
+
});
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
if (status >= 500 || /internal|unavailable|overloaded/i.test(raw)) {
|
|
514
|
+
return new Failure({
|
|
515
|
+
kind: 'server',
|
|
516
|
+
attempted,
|
|
517
|
+
failed: `The provider returned HTTP ${status}. That is their side, not yours.`,
|
|
518
|
+
fix: 'Wait a few seconds and send again. If it persists, /model to another one.',
|
|
519
|
+
cause: err,
|
|
520
|
+
});
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
// `terminated` and `premature close` are what a connection dropped part way
|
|
524
|
+
// through a reply looks like. Nothing usable arrived, so it is safe to send
|
|
525
|
+
// again — and on a free endpoint under load it happens often enough that
|
|
526
|
+
// treating it as fatal would be the single most visible flaw in the agent.
|
|
527
|
+
if (
|
|
528
|
+
/ENOTFOUND|ECONNREFUSED|ECONNRESET|EAI_AGAIN|ETIMEDOUT|EPIPE|network|fetch failed|socket hang up|terminated|premature close|other side closed|UND_ERR/i.test(raw) ||
|
|
529
|
+
err?.name === 'APIConnectionError' ||
|
|
530
|
+
(err instanceof TypeError && /terminated/i.test(raw))
|
|
531
|
+
) {
|
|
532
|
+
return new Failure({
|
|
533
|
+
kind: 'network',
|
|
534
|
+
attempted,
|
|
535
|
+
failed: `The connection to the model dropped: ${raw}`,
|
|
536
|
+
fix:
|
|
537
|
+
'ucode retries this by itself. If it keeps happening, check your connection, ' +
|
|
538
|
+
'VPN and any corporate proxy (HTTPS_PROXY) — or /model to a lighter one, since ' +
|
|
539
|
+
'a long think on a busy free endpoint is the usual cause.',
|
|
540
|
+
cause: err,
|
|
541
|
+
});
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
return new Failure({
|
|
545
|
+
kind: 'unknown',
|
|
546
|
+
attempted,
|
|
547
|
+
failed: detail,
|
|
548
|
+
fix: 'Retry once. If it repeats, run ucode --debug for the full trace.',
|
|
549
|
+
cause: err,
|
|
550
|
+
});
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
const pause = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
554
|
+
|
|
555
|
+
// ---------------------------------------------------------------------------
|
|
556
|
+
// The one call
|
|
557
|
+
// ---------------------------------------------------------------------------
|
|
558
|
+
|
|
559
|
+
/**
|
|
560
|
+
* Send a conversation and get a normalized reply.
|
|
561
|
+
*
|
|
562
|
+
* @param {Array} messages neutral messages
|
|
563
|
+
* @param {Array} [tools] neutral tool definitions
|
|
564
|
+
* @param {object} [opts] { model, temperature, signal, maxOutputTokens,
|
|
565
|
+
* reasoning, attempts, onText, onThinking, onWait }
|
|
566
|
+
*/
|
|
567
|
+
export async function ask(messages, tools = [], opts = {}) {
|
|
568
|
+
const id = opts.model || current;
|
|
569
|
+
|
|
570
|
+
const request = {
|
|
571
|
+
model: id,
|
|
572
|
+
messages: wireMessages(messages),
|
|
573
|
+
// Ask for the working. Without it OpenRouter withholds reasoning entirely,
|
|
574
|
+
// and on a reasoning model that *is* the whole reply until the very end:
|
|
575
|
+
// the socket sits silent for the length of the think, the screen shows
|
|
576
|
+
// nothing, and the provider eventually drops the request as idle. Asking
|
|
577
|
+
// for it fixes the blank screen and the dropped request together. Models
|
|
578
|
+
// that do not reason ignore the flag.
|
|
579
|
+
include_reasoning: true,
|
|
580
|
+
};
|
|
581
|
+
|
|
582
|
+
const wired = wireTools(tools);
|
|
583
|
+
if (wired) {
|
|
584
|
+
request.tools = wired;
|
|
585
|
+
request.tool_choice = 'auto';
|
|
586
|
+
}
|
|
587
|
+
if (opts.temperature !== undefined) request.temperature = opts.temperature;
|
|
588
|
+
if (opts.maxOutputTokens) request.max_tokens = opts.maxOutputTokens;
|
|
589
|
+
if (opts.reasoning) request.reasoning = opts.reasoning;
|
|
590
|
+
|
|
591
|
+
// A side call (the design review) passes fewer: it is better skipped than
|
|
592
|
+
// waited on through a string of rate-limit pauses.
|
|
593
|
+
const attempts = opts.attempts ?? 4;
|
|
594
|
+
let problem;
|
|
595
|
+
|
|
596
|
+
// Text already on screen cannot be unprinted, so a stream is only safe to
|
|
597
|
+
// retry while it is still silent. Every timeout worth retrying happens
|
|
598
|
+
// before the first token, so this costs nothing in practice.
|
|
599
|
+
let printed = 0;
|
|
600
|
+
const callOpts = opts.onText
|
|
601
|
+
? { ...opts, onText: (d) => { printed += d.length; opts.onText(d); } }
|
|
602
|
+
: opts;
|
|
603
|
+
|
|
604
|
+
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
605
|
+
try {
|
|
606
|
+
if (opts.onText) return await streamed(request, callOpts, id);
|
|
607
|
+
const { data, response } = await connection().chat.completions
|
|
608
|
+
.create(request, { signal: opts.signal })
|
|
609
|
+
.withResponse();
|
|
610
|
+
noteLimits(response?.headers);
|
|
611
|
+
return normalize(data, id);
|
|
612
|
+
} catch (err) {
|
|
613
|
+
noteLimits(err?.headers);
|
|
614
|
+
problem = explain(err, id);
|
|
615
|
+
|
|
616
|
+
// A per-minute limit is a wait, not a failure. Sit it out rather than
|
|
617
|
+
// making the user retype their message. Free endpoints often refuse
|
|
618
|
+
// without saying how long to wait, so when there is no retry-after the
|
|
619
|
+
// pauses grow on their own — 5s, 10s, 20s — and only then does the
|
|
620
|
+
// error go up to the loop, which moves to another model.
|
|
621
|
+
const told = problem.detail?.retryAfter;
|
|
622
|
+
const wait = Number.isFinite(told) && told > 0 && told <= 90 ? told : RATE_LIMIT_BACKOFF[attempt - 1];
|
|
623
|
+
if (
|
|
624
|
+
problem.kind === 'rate_limit' && !problem.detail?.daily &&
|
|
625
|
+
wait && attempt < attempts && printed === 0 && !opts.signal?.aborted
|
|
626
|
+
) {
|
|
627
|
+
const until = Date.now() + wait * 1000;
|
|
628
|
+
while (Date.now() < until && !opts.signal?.aborted) {
|
|
629
|
+
opts.onWait?.(`rate limited — resuming in ${Math.ceil((until - Date.now()) / 1000)}s`);
|
|
630
|
+
await pause(Math.min(1000, until - Date.now()));
|
|
631
|
+
}
|
|
632
|
+
if (opts.signal?.aborted) break;
|
|
633
|
+
continue;
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
const worthRetrying =
|
|
637
|
+
problem.kind === 'server' || problem.kind === 'network' || problem.kind === 'timeout';
|
|
638
|
+
if (!worthRetrying || attempt === attempts || opts.signal?.aborted) break;
|
|
639
|
+
if (printed > 0) break; // half an answer is on screen; do not print it twice
|
|
640
|
+
|
|
641
|
+
// A stalled provider needs longer to come back than a dropped socket
|
|
642
|
+
// does, and the wait is narrated so a slow turn never looks like a hang.
|
|
643
|
+
const backoff = problem.kind === 'timeout'
|
|
644
|
+
? 1500 * 2 ** (attempt - 1)
|
|
645
|
+
: 400 * 2 ** (attempt - 1);
|
|
646
|
+
opts.onWait?.(
|
|
647
|
+
`${problem.kind === 'timeout' ? 'provider stalled' : 'connection failed'} — ` +
|
|
648
|
+
`retrying (${attempt + 1}/${attempts})`
|
|
649
|
+
);
|
|
650
|
+
await pause(backoff);
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
throw problem;
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
/** Collect a streamed reply, handing deltas out as they land. */
|
|
658
|
+
async function streamed(request, opts, id) {
|
|
659
|
+
const { data: stream, response } = await connection().chat.completions
|
|
660
|
+
.create(
|
|
661
|
+
{ ...request, stream: true, stream_options: { include_usage: true } },
|
|
662
|
+
{ signal: opts.signal }
|
|
663
|
+
)
|
|
664
|
+
.withResponse();
|
|
665
|
+
noteLimits(response?.headers);
|
|
666
|
+
|
|
667
|
+
let text = '';
|
|
668
|
+
let reasoning = '';
|
|
669
|
+
let finishReason = 'stop';
|
|
670
|
+
let usage = null;
|
|
671
|
+
const partial = new Map();
|
|
672
|
+
const handed = new Set();
|
|
673
|
+
let highest = -1;
|
|
674
|
+
|
|
675
|
+
for await (const chunk of stream) {
|
|
676
|
+
if (opts.signal?.aborted) break;
|
|
677
|
+
if (chunk.usage) usage = chunk.usage;
|
|
678
|
+
|
|
679
|
+
const choice = chunk.choices?.[0];
|
|
680
|
+
if (!choice) continue;
|
|
681
|
+
if (choice.finish_reason) finishReason = choice.finish_reason;
|
|
682
|
+
const delta = choice.delta ?? {};
|
|
683
|
+
|
|
684
|
+
// Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
|
|
685
|
+
// `reasoning_content` on some upstreams.
|
|
686
|
+
const thinking = delta.reasoning ?? delta.reasoning_content;
|
|
687
|
+
if (thinking) {
|
|
688
|
+
reasoning += thinking;
|
|
689
|
+
opts.onThinking?.(thinking);
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
if (delta.content) {
|
|
693
|
+
text += delta.content;
|
|
694
|
+
opts.onText(delta.content);
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
// A tool call's name and arguments arrive across several chunks, keyed by
|
|
698
|
+
// index, so they are stitched back together here.
|
|
699
|
+
for (const call of delta.tool_calls ?? []) {
|
|
700
|
+
// Calls arrive one after another, so the first chunk of call N means
|
|
701
|
+
// every call before it is complete. Those are handed over at once, and
|
|
702
|
+
// the caller can start running them while the rest are still being
|
|
703
|
+
// written — the reply streaming and the tools working overlap.
|
|
704
|
+
if (opts.onToolCall && call.index > highest) {
|
|
705
|
+
for (const [index, slot] of partial) {
|
|
706
|
+
if (index < call.index && !handed.has(index)) {
|
|
707
|
+
handed.add(index);
|
|
708
|
+
opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args }));
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
highest = call.index;
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
|
|
715
|
+
if (call.id) slot.id = call.id;
|
|
716
|
+
if (call.function?.name) slot.name += call.function.name;
|
|
717
|
+
if (call.function?.arguments) slot.args += call.function.arguments;
|
|
718
|
+
partial.set(call.index, slot);
|
|
719
|
+
|
|
720
|
+
// A whole app arrives as one enormous arguments string that takes a
|
|
721
|
+
// minute or two to write. Handing it over as it grows is what lets the
|
|
722
|
+
// caller say which file is being written right now, instead of showing
|
|
723
|
+
// a spinner that has meant nothing for ninety seconds.
|
|
724
|
+
if (call.function?.arguments) opts.onToolArgs?.({ index: call.index, name: slot.name, args: slot.args });
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
const toolCalls = [];
|
|
729
|
+
for (const [index, slot] of partial) {
|
|
730
|
+
toolCalls.push(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args, cutOff: finishReason === 'length' }));
|
|
731
|
+
}
|
|
732
|
+
|
|
733
|
+
return {
|
|
734
|
+
text: text.trim(),
|
|
735
|
+
reasoning: reasoning.trim(),
|
|
736
|
+
toolCalls,
|
|
737
|
+
usage: {
|
|
738
|
+
promptTokens: usage?.prompt_tokens ?? 0,
|
|
739
|
+
outputTokens: usage?.completion_tokens ?? 0,
|
|
740
|
+
totalTokens: usage?.total_tokens ?? 0,
|
|
741
|
+
},
|
|
742
|
+
finishReason,
|
|
743
|
+
model: id,
|
|
744
|
+
};
|
|
745
|
+
}
|
|
746
|
+
|
|
747
|
+
/**
|
|
748
|
+
* Recover the files from a file-write call whose JSON will not parse.
|
|
749
|
+
*
|
|
750
|
+
* The usual cause is a double quote inside the code that the model forgot to
|
|
751
|
+
* escape — `className="flex"` — which ends the JSON string early. No general
|
|
752
|
+
* repair can know which quote was meant, but a file write has a fixed shape:
|
|
753
|
+
* "path", then "content", then either the next file or the end. Splitting on
|
|
754
|
+
* that shape and escaping the stray quotes gets every file back.
|
|
755
|
+
*/
|
|
756
|
+
export function salvageWrites(text) {
|
|
757
|
+
const heads = [...text.matchAll(/"path"\s*:\s*"((?:[^"\\]|\\.)*)"\s*,\s*"content"\s*:\s*"/g)];
|
|
758
|
+
if (!heads.length) return null;
|
|
759
|
+
|
|
760
|
+
const files = [];
|
|
761
|
+
for (let i = 0; i < heads.length; i++) {
|
|
762
|
+
const from = heads[i].index + heads[i][0].length;
|
|
763
|
+
const to = i + 1 < heads.length ? heads[i + 1].index : text.length;
|
|
764
|
+
// The string ends at the last quote that is followed by nothing but JSON
|
|
765
|
+
// punctuation — `"}, {`, or `"}}, {` when the model added a brace, or `"}]}`.
|
|
766
|
+
const body = text.slice(from, to).replace(/"[\s,{}[\]]*$/, '');
|
|
767
|
+
const content = unescapeLoose(body);
|
|
768
|
+
const pathValue = unescapeLoose(heads[i][1]);
|
|
769
|
+
if (!pathValue || !content) return null; // not the shape we thought — leave it an honest error
|
|
770
|
+
files.push({ path: pathValue, content });
|
|
771
|
+
}
|
|
772
|
+
return files.length ? files : null;
|
|
773
|
+
}
|
|
774
|
+
|
|
775
|
+
/**
|
|
776
|
+
* Decode a JSON string body the forgiving way: the standard escapes are
|
|
777
|
+
* honoured, and everything JSON would reject — a raw line break, a tab, a stray
|
|
778
|
+
* quote, an escape JSON does not know — is kept as the character it plainly is.
|
|
779
|
+
*/
|
|
780
|
+
function unescapeLoose(s) {
|
|
781
|
+
const simple = { n: '\n', t: '\t', r: '\r', b: '\b', f: '\f', '"': '"', '\\': '\\', '/': '/' };
|
|
782
|
+
let out = '';
|
|
783
|
+
for (let i = 0; i < s.length; i++) {
|
|
784
|
+
const c = s[i];
|
|
785
|
+
if (c !== '\\' || i === s.length - 1) { out += c; continue; }
|
|
786
|
+
const next = s[++i];
|
|
787
|
+
if (next === 'u' && /^[0-9a-fA-F]{4}$/.test(s.slice(i + 1, i + 5))) {
|
|
788
|
+
out += String.fromCharCode(parseInt(s.slice(i + 1, i + 5), 16));
|
|
789
|
+
i += 4;
|
|
790
|
+
} else {
|
|
791
|
+
out += simple[next] ?? next;
|
|
792
|
+
}
|
|
793
|
+
}
|
|
794
|
+
return out;
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
/**
|
|
798
|
+
* Parse one tool call's arguments.
|
|
799
|
+
*
|
|
800
|
+
* A parse error is recorded on the call rather than thrown. The loop hands it
|
|
801
|
+
* back to the model, which usually fixes its own JSON on the next step —
|
|
802
|
+
* cheaper than failing the whole turn over a stray comma.
|
|
803
|
+
*/
|
|
804
|
+
export function readCall({ id, name, raw, cutOff = false }) {
|
|
805
|
+
const call = { id, name, args: {} };
|
|
806
|
+
const text = String(raw ?? '').trim();
|
|
807
|
+
if (!text) return call;
|
|
808
|
+
try {
|
|
809
|
+
const parsed = JSON.parse(text);
|
|
810
|
+
if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) call.args = parsed;
|
|
811
|
+
else call.parseError = `arguments must be a JSON object, got ${Array.isArray(parsed) ? 'array' : typeof parsed}`;
|
|
812
|
+
} catch (err) {
|
|
813
|
+
// A missing comma or a stray control character in a 3,000-token
|
|
814
|
+
// batch_write used to throw the whole step away — half a minute of output
|
|
815
|
+
// discarded over one character. Repair the usual slips instead. Never when
|
|
816
|
+
// the reply was cut off at the output limit, though: repairing that would
|
|
817
|
+
// close the string and quietly write half a file.
|
|
818
|
+
if (!cutOff) {
|
|
819
|
+
try {
|
|
820
|
+
const fixed = JSON.parse(jsonrepair(text));
|
|
821
|
+
if (fixed && typeof fixed === 'object' && !Array.isArray(fixed)) {
|
|
822
|
+
call.args = fixed;
|
|
823
|
+
call.repaired = true;
|
|
824
|
+
return call;
|
|
825
|
+
}
|
|
826
|
+
} catch { /* beyond general repair — try the file-write shape next */ }
|
|
827
|
+
|
|
828
|
+
const salvaged = (name === 'write_file' || name === 'batch_write') ? salvageWrites(text) : null;
|
|
829
|
+
if (salvaged) {
|
|
830
|
+
call.args = name === 'write_file' ? salvaged[0] : { files: salvaged };
|
|
831
|
+
call.repaired = true;
|
|
832
|
+
return call;
|
|
833
|
+
}
|
|
834
|
+
}
|
|
835
|
+
call.parseError = `${err.message} — the raw arguments were: ${text.slice(0, 300)}`;
|
|
836
|
+
}
|
|
837
|
+
return call;
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
function normalize(data, id) {
|
|
841
|
+
const choice = data?.choices?.[0];
|
|
842
|
+
const message = choice?.message ?? {};
|
|
843
|
+
|
|
844
|
+
const toolCalls = (message.tool_calls ?? []).map((c) =>
|
|
845
|
+
readCall({ id: c.id, name: c.function?.name, raw: c.function?.arguments, cutOff: choice?.finish_reason === 'length' })
|
|
846
|
+
);
|
|
847
|
+
|
|
848
|
+
const u = data?.usage ?? {};
|
|
849
|
+
const finishReason = choice?.finish_reason ?? 'stop';
|
|
850
|
+
const text = (message.content ?? '').trim();
|
|
851
|
+
|
|
852
|
+
if (!text && toolCalls.length === 0 && finishReason === 'length') {
|
|
853
|
+
throw new Failure({
|
|
854
|
+
kind: 'no_content',
|
|
855
|
+
attempted: `asking ${modelName(id)} for a reply`,
|
|
856
|
+
failed: 'The reply hit the output limit before producing anything at all.',
|
|
857
|
+
fix: 'Ask for something shorter, or split the task into steps.',
|
|
858
|
+
detail: { finishReason },
|
|
859
|
+
});
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
return {
|
|
863
|
+
text,
|
|
864
|
+
reasoning: (message.reasoning ?? '').trim(),
|
|
865
|
+
toolCalls,
|
|
866
|
+
usage: {
|
|
867
|
+
promptTokens: u.prompt_tokens ?? 0,
|
|
868
|
+
outputTokens: u.completion_tokens ?? 0,
|
|
869
|
+
totalTokens: u.total_tokens ?? 0,
|
|
870
|
+
},
|
|
871
|
+
finishReason,
|
|
872
|
+
model: id,
|
|
873
|
+
};
|
|
874
|
+
}
|