ucode-agent 1.54.0 → 1.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,1099 +1,1149 @@
1
- /**
2
- * provider.js — the only module that knows a model provider exists.
3
- *
4
- * Everything above this file speaks one small neutral message format and calls
5
- * ask(). Moving ucode to a different host means rewriting this file and
6
- * nothing else.
7
- *
8
- * The neutral formats:
9
- * { role: 'system', content }
10
- * { role: 'user', content, images?: [dataUrl] }
11
- * { role: 'assistant', content?, toolCalls?: [{ id, name, args }] }
12
- * { role: 'tool', toolCallId, name, content }
13
- *
14
- * tool: { name, description, parameters: <JSON Schema> }
15
- *
16
- * ask() resolves to:
17
- * { text, reasoning, toolCalls, usage, finishReason, model }
18
- */
19
-
20
- import { join, dirname } from 'node:path';
21
- import { homedir } from 'node:os';
22
- import { fileURLToPath } from 'node:url';
23
- import dotenv from 'dotenv';
24
- import OpenAI from 'openai';
25
- import { jsonrepair } from 'jsonrepair';
26
- import { Failure } from './failure.js';
27
-
28
- const HERE = dirname(fileURLToPath(import.meta.url));
29
- const PACKAGE_ROOT = join(HERE, '..', '..');
30
-
31
- /** Personal config lives here, outside the package, so upgrades never touch it. */
32
- export const UCODE_HOME = join(homedir(), '.ucode');
33
- export const ENV_FILE = join(UCODE_HOME, '.env');
34
-
35
- export const BASE_URL = 'https://openrouter.ai/api/v1';
36
- export const PROVIDER = 'OpenRouter';
37
-
38
- // First definition wins — dotenv never overwrites a variable that already
39
- // exists — so the order here is the precedence order:
40
- // real environment > ./.env (this project) > ~/.ucode/.env (this machine)
41
- // > the checkout's own .env (only when developing on a clone)
42
- dotenv.config({ path: join(process.cwd(), '.env'), quiet: true });
43
- dotenv.config({ path: ENV_FILE, quiet: true });
44
- dotenv.config({ path: join(PACKAGE_ROOT, '.env'), quiet: true });
45
-
46
- /**
47
- * The whole model list. Not a starting point — the list.
48
- *
49
- * ucode runs on NVIDIA, Cohere, Nex AGI, DeepSeek and Qwen only. All of them
50
- * serve genuinely capable models free through OpenRouter, all handle tool
51
- * calling properly, and keeping the set to eight means every one of them has
52
- * been used in anger rather than listed on the strength of a benchmark. A
53
- * picker offering sixty models is a picker nobody reads.
54
- *
55
- * `name` is what the status bar shows. `note` is what the picker shows.
56
- */
57
- export const MODELS = {
58
- 'nvidia/nemotron-3-ultra-550b-a55b:free': {
59
- name: 'Nemotron 3 Ultra',
60
- context: 1_000_000,
61
- star: true,
62
- note: 'deepest reasoning, 1M context — slowest to answer',
63
- },
64
- 'nvidia/nemotron-3.5-lightning:free': {
65
- name: 'Nemotron 3.5 Lightning',
66
- context: 1_000_000,
67
- note: 'same huge window, answers much sooner',
68
- },
69
- 'nvidia/nemotron-3-super-120b-a12b:free': {
70
- name: 'Nemotron 3 Super',
71
- context: 262_144,
72
- note: 'strong all-rounder, quick to first token',
73
- },
74
- 'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free': {
75
- name: 'Nemotron 3 Nano Omni',
76
- context: 256_000,
77
- note: 'small and fast, reasoning tuned',
78
- },
79
- 'cohere/north-mini-code:free': {
80
- name: 'North Mini Code',
81
- context: 256_000,
82
- star: true,
83
- note: 'the default — built for code and interface work, quick to answer',
84
- },
85
- // On trial, so not the default and not in FALLBACKS yet. Nex AGI is its only
86
- // upstream, so a stall there has nowhere else to go.
87
- 'nex-agi/nex-n2.5-pro:free': {
88
- name: 'Nex N2.5 Pro',
89
- context: 262_144,
90
- note: 'new agentic coder, on trial — can stall on big builds',
91
- },
92
- // Both free on OpenRouter with tool calling, added untested: on trial, not
93
- // the default and not in FALLBACKS until a real ucode build has run on them.
94
- 'deepseek/deepseek-v4-flash-0731:free': {
95
- name: 'DeepSeek V4 Flash',
96
- context: 1_048_576,
97
- // Left to itself it thinks for minutes before every step — over two
98
- // minutes of nothing on screen before its first word. Low effort answers
99
- // in seconds and still plans the build.
100
- reasoning: { effort: 'low' },
101
- note: 'fast DeepSeek coder, 1M context — on trial',
102
- },
103
- 'qwen/qwen3.8-27b:free': {
104
- name: 'Qwen 3.8',
105
- context: 262_144,
106
- note: 'compact Qwen coder, 262k context — on trial',
107
- },
108
- };
109
-
110
- /**
111
- * North Mini Code is the default: it is built for code and interface work,
112
- * which is what ucode is mostly asked to do, and it answers far sooner than
113
- * the big reasoning models. /model moves to Ultra when a problem needs the
114
- * million-token window and the long think more than it needs the speed.
115
- */
116
- export const DEFAULT_MODEL = 'cohere/north-mini-code:free';
117
-
118
- /**
119
- * Where to go when a model is busy, in order of preference. Each is served by
120
- * a different upstream, so a rate limit on one rarely means a limit on the
121
- * next — which is what lets a long build keep going instead of stopping at
122
- * the first "too many requests".
123
- */
124
- export const FALLBACKS = [
125
- 'cohere/north-mini-code:free',
126
- 'nvidia/nemotron-3.5-lightning:free',
127
- 'nvidia/nemotron-3-super-120b-a12b:free',
128
- 'nvidia/nemotron-3-ultra-550b-a55b:free',
129
- ];
130
-
131
- /** The next model to try after `id`, skipping any already tried this round. */
132
- export function fallbackFor(id, tried = new Set()) {
133
- // Off unless UCODE_FALLBACK=1. A build that starts on one model and finishes
134
- // on another finishes to a different standard, and the swap lands exactly
135
- // when the user is least placed to work out why the output changed —
136
- // mid-build, behind a note that scrolls past. Which model to run is the one
137
- // decision they made before starting; it is not one to take back for them.
138
- //
139
- // Every caller already reads "no fallback" as "wait, then try this one
140
- // again": failover waits a minute and returns to the same model, the
141
- // stuck-detector simply does not switch, and a worker retries its own. That
142
- // is why this can be a single gate rather than four.
143
- if (process.env.UCODE_FALLBACK !== '1') return null;
144
-
145
- const start = Math.max(0, FALLBACKS.indexOf(id));
146
- for (let i = 1; i <= FALLBACKS.length; i++) {
147
- const next = FALLBACKS[(start + i) % FALLBACKS.length];
148
- if (next !== id && !tried.has(next)) return next;
149
- }
150
- return null;
151
- }
152
-
153
- /** Seconds to wait on successive rate limits that come with no retry-after. */
154
- const RATE_LIMIT_BACKOFF = [5, 10, 20];
155
-
156
- /**
157
- * How long a stream may go without a single chunk before it counts as frozen.
158
- *
159
- * A free endpoint can accept a request and then send nothing at all, and the
160
- * only thing that used to end that was the five-minute request timeout — five
161
- * minutes of a spinner, then the same again on the retry. Reasoning streams
162
- * as it is produced, so even the slowest thinker sends something well inside
163
- * a minute; silence for that long means nobody is working on the reply.
164
- */
165
- export const stallLimit = () => Number(process.env.UCODE_STALL_MS) || 60_000;
166
-
167
- /** Freezes in a row before ucode stops asking and says to switch models. */
168
- export const MAX_STALLS = 3;
169
-
170
- /** How long a reply may go quiet once it has started writing a tool call. */
171
- const writingLimit = () => Number(process.env.UCODE_STALL_MS) || 600_000;
172
- let stalls = 0;
173
-
174
- let current = process.env.UCODE_MODEL || DEFAULT_MODEL;
175
- let client = null;
176
-
177
- export function model() {
178
- return current;
179
- }
180
-
181
- export function setModel(id) {
182
- const wanted = String(id ?? '').trim();
183
- if (!wanted) {
184
- throw new Failure({
185
- kind: 'bad_model',
186
- attempted: 'switching model',
187
- failed: 'No model name was given.',
188
- fix: `Pick one of: ${Object.keys(MODELS).join(', ')}`,
189
- });
190
- }
191
- if (!MODELS[wanted]) {
192
- throw new Failure({
193
- kind: 'bad_model',
194
- attempted: `switching to "${wanted}"`,
195
- failed: 'ucode only runs NVIDIA, Cohere, Nex AGI, DeepSeek and Qwen models, and that is not one of them.',
196
- fix: `Run /model to choose from: ${Object.keys(MODELS).join(', ')}`,
197
- });
198
- }
199
- current = wanted;
200
- return current;
201
- }
202
-
203
- /** The short name for a model id: "Nemotron 3 Ultra". */
204
- export function modelName(id = current) {
205
- return MODELS[id]?.name ?? id;
206
- }
207
-
208
- /** Every model, in list order, annotated with whether it is the active one. */
209
- export function modelList() {
210
- return Object.entries(MODELS).map(([id, info]) => ({
211
- id,
212
- ...info,
213
- active: id === current,
214
- }));
215
- }
216
-
217
- /**
218
- * How many tokens one request may occupy.
219
- *
220
- * OpenRouter meters requests rather than tokens, so nothing here is rationing
221
- * a quota — the binding limit is simply the window the model has. On the two
222
- * million-token models compaction essentially never fires.
223
- */
224
- export function contextLimit(id = current) {
225
- const override = Number(process.env.UCODE_MAX_CONTEXT_TOKENS);
226
- if (Number.isFinite(override) && override > 0) return override;
227
- return MODELS[id]?.context ?? 128_000;
228
- }
229
-
230
- /**
231
- * A rough local token count, used to decide when to compact *before* a
232
- * request goes out. Real numbers come back in the response usage; this only
233
- * has to be close enough to trigger at the right time.
234
- */
235
- export function estimateTokens(text) {
236
- return text ? Math.ceil(String(text).length / 4) : 0;
237
- }
238
-
239
- export function estimateConversation(messages) {
240
- let total = 0;
241
- for (const m of messages) {
242
- total += estimateTokens(m.content || '');
243
- for (const call of m.toolCalls || []) {
244
- total += estimateTokens(call.name) + estimateTokens(JSON.stringify(call.args || {}));
245
- }
246
- total += 4; // role and framing overhead per message
247
- }
248
- return total;
249
- }
250
-
251
- // ---------------------------------------------------------------------------
252
- // Client
253
- // ---------------------------------------------------------------------------
254
-
255
- /**
256
- * No key is bundled here, and none ever should be. A key committed to a public
257
- * package is readable by anyone who runs `npm pack ucode-agent`, and no amount
258
- * of first-run convenience is worth handing out a live credential.
259
- */
260
- function apiKey() {
261
- // UCODE_API_KEY is the documented name. The provider's own variable name is
262
- // still read, so a key set up for another tool keeps working here.
263
- const key = (process.env.UCODE_API_KEY || process.env.OPENROUTER_API_KEY || '').trim();
264
- if (!key) {
265
- throw new Failure({
266
- kind: 'no_api_key',
267
- attempted: 'connecting to the model',
268
- failed: 'No API key is set - UCODE_API_KEY is missing from the environment and from every .env file.',
269
- fix:
270
- `Put UCODE_API_KEY=your-key in ${ENV_FILE} — that applies to every ` +
271
- 'project on this machine — or in a .env file beside your code. ' +
272
- 'Free keys: https://openrouter.ai/keys',
273
- });
274
- }
275
- return key;
276
- }
277
-
278
- function connection() {
279
- if (client) return client;
280
- client = new OpenAI({
281
- apiKey: apiKey(),
282
- baseURL: process.env.UCODE_BASE_URL || BASE_URL,
283
- // Nemotron Ultra can think for a long time before its first token, so the
284
- // ceiling is deliberately generous. maxRetries is 0 because ask() owns
285
- // retrying: its attempts are narrated on screen instead of happening
286
- // silently somewhere inside the SDK.
287
- timeout: Number(process.env.UCODE_REQUEST_TIMEOUT_MS) || 300_000,
288
- maxRetries: 0,
289
- defaultHeaders: {
290
- 'HTTP-Referer': 'https://github.com/sppideey/ucode-agent',
291
- 'X-Title': 'ucode',
292
- },
293
- });
294
- return client;
295
- }
296
-
297
- /** Drop the cached client so the next request picks up a changed key. */
298
- export function resetConnection() {
299
- client = null;
300
- }
301
-
302
- // ---------------------------------------------------------------------------
303
- // Live quota, taken from whatever rate-limit headers come back
304
- // ---------------------------------------------------------------------------
305
-
306
- let limits = null;
307
-
308
- export function rateLimits() {
309
- return limits;
310
- }
311
-
312
- /** "1.5s", "2m59.56s", "1h2m" -> seconds */
313
- function seconds(value) {
314
- if (!value) return null;
315
- const m = /^(?:(\d+(?:\.\d+)?)h)?(?:(\d+(?:\.\d+)?)m(?!s))?(?:(\d+(?:\.\d+)?)m?s)?$/
316
- .exec(String(value).trim());
317
- if (!m) return null;
318
- const total = (parseFloat(m[1]) || 0) * 3600 + (parseFloat(m[2]) || 0) * 60 + (parseFloat(m[3]) || 0);
319
- return total > 0 ? total : null;
320
- }
321
-
322
- function noteLimits(headers) {
323
- if (!headers?.get) return;
324
- const num = (name) => {
325
- const n = Number(headers.get(name));
326
- return Number.isFinite(n) ? n : null;
327
- };
328
- limits = {
329
- requestsLimit: num('x-ratelimit-limit-requests'),
330
- requestsRemaining: num('x-ratelimit-remaining-requests'),
331
- requestsReset: seconds(headers.get('x-ratelimit-reset-requests')),
332
- tokensLimit: num('x-ratelimit-limit-tokens'),
333
- tokensRemaining: num('x-ratelimit-remaining-tokens'),
334
- tokensReset: seconds(headers.get('x-ratelimit-reset-tokens')),
335
- at: Date.now(),
336
- };
337
- }
338
-
339
- // ---------------------------------------------------------------------------
340
- // Neutral format -> wire format
341
- // ---------------------------------------------------------------------------
342
-
343
- function wireTools(tools) {
344
- if (!tools?.length) return undefined;
345
- return tools.map((t) => ({
346
- type: 'function',
347
- function: {
348
- name: t.name,
349
- description: t.description,
350
- parameters: t.parameters ?? { type: 'object', properties: {} },
351
- },
352
- }));
353
- }
354
-
355
- function wireMessages(messages) {
356
- const out = [];
357
- for (const m of messages) {
358
- if (m.role === 'user' && m.images?.length) {
359
- out.push({
360
- role: 'user',
361
- content: [
362
- { type: 'text', text: m.content ?? '' },
363
- ...m.images.map((url) => ({ type: 'image_url', image_url: { url } })),
364
- ],
365
- });
366
- } else if (m.role === 'system' || m.role === 'user') {
367
- out.push({ role: m.role, content: m.content ?? '' });
368
- } else if (m.role === 'assistant') {
369
- const wire = { role: 'assistant', content: m.content || '' };
370
- if (m.toolCalls?.length) {
371
- wire.tool_calls = m.toolCalls.map((c) => ({
372
- id: c.id,
373
- type: 'function',
374
- function: { name: c.name, arguments: JSON.stringify(c.args ?? {}) },
375
- }));
376
- }
377
- out.push(wire);
378
- } else if (m.role === 'tool') {
379
- out.push({ role: 'tool', tool_call_id: m.toolCallId, content: m.content ?? '' });
380
- }
381
- }
382
- return out;
383
- }
384
-
385
- // ---------------------------------------------------------------------------
386
- // Turning provider errors into something a person can act on
387
- // ---------------------------------------------------------------------------
388
-
389
- function bodyOf(err) {
390
- if (err?.error) return { error: err.error };
391
- const raw = String(err?.message ?? '');
392
- const start = raw.indexOf('{');
393
- if (start === -1) return null;
394
- try {
395
- return JSON.parse(raw.slice(start));
396
- } catch {
397
- return null;
398
- }
399
- }
400
-
401
- export function explain(err, id) {
402
- if (err instanceof Failure) return err;
403
-
404
- const status = err?.status ?? err?.statusCode ?? null;
405
- const body = bodyOf(err);
406
- const detail = body?.error?.message ?? String(err?.message ?? err);
407
- const attempted = `asking ${modelName(id)} for a reply`;
408
-
409
- // The useful text is often not on the error itself. undici reports a socket
410
- // that closed mid-response as a bare `TypeError: terminated` and puts the
411
- // real reason on `cause`, so the whole chain is matched rather than the top
412
- // message alone — otherwise an ordinary dropped connection, which is worth
413
- // retrying, gets reported as an unknown fault, which is not.
414
- const chain = [err?.message, err?.code, err?.cause?.message, err?.cause?.code]
415
- .filter(Boolean)
416
- .join(' | ');
417
- const raw = chain || String(err);
418
-
419
- if (err?.name === 'AbortError' || /aborted|The operation was aborted/i.test(raw)) {
420
- return new Failure({
421
- kind: 'aborted',
422
- attempted,
423
- failed: 'The request was cancelled.',
424
- fix: 'Send the message again when you are ready.',
425
- cause: err,
426
- });
427
- }
428
-
429
- if (status === 401 || status === 403 || /invalid[_ ]api[_ ]key/i.test(raw)) {
430
- return new Failure({
431
- kind: 'invalid_api_key',
432
- attempted,
433
- failed: `The API key was rejected (HTTP ${status ?? 401}).`,
434
- fix:
435
- 'Check UCODE_API_KEY in ~/.ucode/.env for a typo or trailing space, and ' +
436
- 'confirm the key is still active in your account.',
437
- cause: err,
438
- });
439
- }
440
-
441
- if (status === 429 || /rate[_ ]limit/i.test(raw)) {
442
- // `??` cannot be used to chain through Number(): Number(undefined) is NaN,
443
- // which is neither null nor undefined, so it would swallow every fallback
444
- // after it and the wait would silently never be found.
445
- const header = err?.headers?.get?.('retry-after');
446
- const asNumber = Number(header);
447
- const retryAfter =
448
- seconds(header) ??
449
- (Number.isFinite(asNumber) && asNumber > 0 ? asNumber : null) ??
450
- seconds(/try again in ([\dhms.]+)/i.exec(detail)?.[1]) ??
451
- null;
452
- // The daily cap reads "free-models-per-day-high-balance", with hyphens, and
453
- // names its source in the metadata. It is one cap across every free model,
454
- // so it is reported at once rather than waited on model after model.
455
- const meta = body?.error?.metadata ?? {};
456
- const daily = /per[- ]day|RPD|TPD|daily/i.test(`${detail} ${meta.limit_source ?? ''}`);
457
- const resetMs = Number(meta.headers?.['X-RateLimit-Reset'] ?? err?.headers?.get?.('x-ratelimit-reset'));
458
- const resetAt = daily && Number.isFinite(resetMs) && resetMs > Date.now() ? new Date(resetMs) : null;
459
- const cap = Number(meta.headers?.['X-RateLimit-Limit']) || null;
460
- const wait = Number.isFinite(retryAfter) && retryAfter
461
- ? (retryAfter >= 60 ? `${Math.ceil(retryAfter / 60)} min` : `${Math.ceil(retryAfter)}s`)
462
- : null;
463
-
464
- return new Failure({
465
- kind: 'rate_limit',
466
- attempted,
467
- failed: daily
468
- ? `This key's free daily limit${cap ? ` of ${cap} requests` : ''} is used up. It covers every free model, so switching will not help.`
469
- : `Too many requests for ${modelName(id)} just now${wait ? ` — clear in ${wait}` : ''}.`,
470
- fix: daily
471
- ? `It resets ${resetAt ? `at ${resetAt.toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' })}` : 'once a day'}. ` +
472
- 'Adding credit to your account raises the limit.'
473
- : 'ucode waits these out on its own. Free endpoints are shared, so it usually ' +
474
- 'clears in seconds; /model moves to a quieter one.',
475
- detail: { retryAfter, daily, resetAt: resetAt?.getTime() ?? null },
476
- cause: err,
477
- });
478
- }
479
-
480
- // A 404 on a model in this list is almost never a bad name. It is OpenRouter
481
- // having no upstream free to serve it at that instant, and it clears by
482
- // itself — so it is retried rather than reported as a missing model.
483
- if (status === 404 || /does not exist|not found|decommissioned/i.test(raw)) {
484
- if (MODELS[id]) {
485
- return new Failure({
486
- kind: 'server',
487
- attempted,
488
- failed: `No provider was free to serve ${modelName(id)} at that moment.`,
489
- fix: 'ucode retries this by itself. If it keeps up, /model switches.',
490
- detail: { status },
491
- cause: err,
492
- });
493
- }
494
- return new Failure({
495
- kind: 'bad_model',
496
- attempted,
497
- failed: `No model "${id}" is available to this key.`,
498
- fix: `Run /model. ucode ships with: ${Object.keys(MODELS).join(', ')}`,
499
- cause: err,
500
- });
501
- }
502
-
503
- // Some upstreams validate tool calls themselves and reject an invented one,
504
- // which fails the whole request. Recoverable: the loop feeds this back.
505
- if (body?.error?.code === 'tool_use_failed' || /tool call validation failed/i.test(detail)) {
506
- const attemptedName = /call tool '([^']+)'/.exec(detail)?.[1];
507
- return new Failure({
508
- kind: 'bad_tool_call',
509
- attempted,
510
- failed: attemptedName
511
- ? `The model tried to call "${attemptedName}", which is not one of its tools.`
512
- : `The model produced a tool call the provider rejected: ${detail}`,
513
- fix: 'Use only the tools supplied with the request.',
514
- detail: { attemptedName },
515
- cause: err,
516
- });
517
- }
518
-
519
- if (status === 400) {
520
- const noTools = /tool calling.*not supported/i.test(detail);
521
- return new Failure({
522
- kind: noTools ? 'no_tool_support' : 'bad_request',
523
- attempted,
524
- failed: noTools
525
- ? `${modelName(id)} cannot call tools, which ucode needs for every task.`
526
- : `The request was rejected as malformed (HTTP 400): ${detail}`,
527
- fix: noTools
528
- ? 'Run /model and pick another one.'
529
- : 'The conversation has something in it the provider will not accept. /new starts a fresh one.',
530
- cause: err,
531
- });
532
- }
533
-
534
- if (status === 413 || /too large|context.*length/i.test(detail)) {
535
- return new Failure({
536
- kind: 'too_large',
537
- attempted,
538
- failed: `The conversation no longer fits in ${modelName(id)}: ${detail}`,
539
- fix: 'Run /new for a fresh session, or lower UCODE_MAX_CONTEXT_TOKENS so ucode folds older turns away sooner.',
540
- cause: err,
541
- });
542
- }
543
-
544
- // OpenRouter drops a request when the upstream goes quiet on it. That is the
545
- // ordinary failure mode of a busy free endpoint, and of a big reasoning model
546
- // that spends a long time thinking before its first token. Nothing was
547
- // generated, so retrying is safe — and ask() does it before anyone notices.
548
- if (
549
- status === 408 || status === 504 || status === 524 || status === 522 ||
550
- err?.name === 'APIConnectionTimeoutError' ||
551
- /idle timeout|timed out|timeout/i.test(detail) || /idle timeout|timed out/i.test(raw)
552
- ) {
553
- return new Failure({
554
- kind: 'timeout',
555
- attempted,
556
- failed: `${modelName(id)} sent nothing back in time — the provider dropped the request.`,
557
- fix:
558
- 'Free endpoints stall under load, and the largest reasoning models are the ' +
559
- 'first to. ucode already retried. If it keeps happening, /model to Nemotron ' +
560
- '3.5 Lightning or North Mini Code, which answer sooner.',
561
- detail: { status },
562
- cause: err,
563
- });
564
- }
565
-
566
- if (status >= 500 || /internal|unavailable|overloaded/i.test(raw)) {
567
- return new Failure({
568
- kind: 'server',
569
- attempted,
570
- failed: `The provider returned HTTP ${status}. That is their side, not yours.`,
571
- fix: 'Wait a few seconds and send again. If it persists, /model to another one.',
572
- cause: err,
573
- });
574
- }
575
-
576
- // `terminated` and `premature close` are what a connection dropped part way
577
- // through a reply looks like. Nothing usable arrived, so it is safe to send
578
- // again — and on a free endpoint under load it happens often enough that
579
- // treating it as fatal would be the single most visible flaw in the agent.
580
- if (
581
- /ENOTFOUND|ECONNREFUSED|ECONNRESET|EAI_AGAIN|ETIMEDOUT|EPIPE|network|fetch failed|socket hang up|terminated|premature close|other side closed|UND_ERR/i.test(raw) ||
582
- err?.name === 'APIConnectionError' ||
583
- (err instanceof TypeError && /terminated/i.test(raw))
584
- ) {
585
- return new Failure({
586
- kind: 'network',
587
- attempted,
588
- failed: `The connection to the model dropped: ${raw}`,
589
- fix:
590
- 'ucode retries this by itself. If it keeps happening, check your connection, ' +
591
- 'VPN and any corporate proxy (HTTPS_PROXY) — or /model to a lighter one, since ' +
592
- 'a long think on a busy free endpoint is the usual cause.',
593
- cause: err,
594
- });
595
- }
596
-
597
- return new Failure({
598
- kind: 'unknown',
599
- attempted,
600
- failed: detail,
601
- fix: 'Retry once. If it repeats, run ucode --debug for the full trace.',
602
- cause: err,
603
- });
604
- }
605
-
606
- const pause = (ms) => new Promise((r) => setTimeout(r, ms));
607
-
608
- // ---------------------------------------------------------------------------
609
- // The one call
610
- // ---------------------------------------------------------------------------
611
-
612
- /**
613
- * Send a conversation and get a normalized reply.
614
- *
615
- * @param {Array} messages neutral messages
616
- * @param {Array} [tools] neutral tool definitions
617
- * @param {object} [opts] { model, temperature, signal, maxOutputTokens,
618
- * reasoning, attempts, onText, onThinking, onWait }
619
- */
620
- export async function ask(messages, tools = [], opts = {}) {
621
- const id = opts.model || current;
622
-
623
- const request = {
624
- model: id,
625
- messages: wireMessages(messages),
626
- // Ask for the working. Without it OpenRouter withholds reasoning entirely,
627
- // and on a reasoning model that *is* the whole reply until the very end:
628
- // the socket sits silent for the length of the think, the screen shows
629
- // nothing, and the provider eventually drops the request as idle. Asking
630
- // for it fixes the blank screen and the dropped request together. Models
631
- // that do not reason ignore the flag.
632
- include_reasoning: true,
633
- };
634
-
635
- const wired = wireTools(tools);
636
- if (wired) {
637
- request.tools = wired;
638
- request.tool_choice = 'auto';
639
- }
640
- if (opts.temperature !== undefined) request.temperature = opts.temperature;
641
- if (opts.maxOutputTokens) request.max_tokens = opts.maxOutputTokens;
642
- const reasoning = opts.reasoning ?? MODELS[id]?.reasoning;
643
- if (reasoning) request.reasoning = reasoning;
644
-
645
- // A side call (the design review) passes fewer: it is better skipped than
646
- // waited on through a string of rate-limit pauses.
647
- const attempts = opts.attempts ?? 4;
648
- let problem;
649
-
650
- // Text already on screen cannot be unprinted, so a stream is only safe to
651
- // retry while it is still silent. Every timeout worth retrying happens
652
- // before the first token, so this costs nothing in practice.
653
- let printed = 0;
654
- const callOpts = opts.onText
655
- ? { ...opts, onText: (d) => { printed += d.length; opts.onText(d); } }
656
- : opts;
657
-
658
- for (let attempt = 1; attempt <= attempts; attempt++) {
659
- try {
660
- if (opts.onText) {
661
- const reply = await streamed(request, callOpts, id);
662
- stalls = 0;
663
- return reply;
664
- }
665
- const { data, response } = await connection().chat.completions
666
- .create(request, { signal: opts.signal })
667
- .withResponse();
668
- noteLimits(response?.headers);
669
- stalls = 0;
670
- return normalize(data, id);
671
- } catch (err) {
672
- noteLimits(err?.headers);
673
- problem = explain(err, id);
674
-
675
- // Retrying a model that keeps freezing only repeats the wait. After a
676
- // few in a row, say so and hand the choice back.
677
- if (problem.detail?.stalled && ++stalls >= MAX_STALLS) {
678
- stalls = 0;
679
- throw new Failure({
680
- kind: 'stalled',
681
- attempted: `asking ${modelName(id)} for a reply`,
682
- failed: `${modelName(id)} froze ${MAX_STALLS} times in a row — it took the request and then sent nothing.`,
683
- fix: 'Its free endpoint is struggling right now. Run /model and pick another one; North Mini Code answers soonest.',
684
- cause: problem,
685
- });
686
- }
687
-
688
- // A per-minute limit is a wait, not a failure. Sit it out rather than
689
- // making the user retype their message. Free endpoints often refuse
690
- // without saying how long to wait, so when there is no retry-after the
691
- // pauses grow on their own — 5s, 10s, 20s — and only then does the
692
- // error go up to the loop, which moves to another model.
693
- const told = problem.detail?.retryAfter;
694
- const wait = Number.isFinite(told) && told > 0 && told <= 90 ? told : RATE_LIMIT_BACKOFF[attempt - 1];
695
- if (
696
- problem.kind === 'rate_limit' && !problem.detail?.daily &&
697
- wait && attempt < attempts && printed === 0 && !opts.signal?.aborted
698
- ) {
699
- const until = Date.now() + wait * 1000;
700
- while (Date.now() < until && !opts.signal?.aborted) {
701
- opts.onWait?.(`rate limited — resuming in ${Math.ceil((until - Date.now()) / 1000)}s`);
702
- await pause(Math.min(1000, until - Date.now()));
703
- }
704
- if (opts.signal?.aborted) break;
705
- continue;
706
- }
707
-
708
- const worthRetrying =
709
- problem.kind === 'server' || problem.kind === 'network' || problem.kind === 'timeout';
710
- if (!worthRetrying || attempt === attempts || opts.signal?.aborted) break;
711
- if (printed > 0) break; // half an answer is on screen; do not print it twice
712
- if (problem.detail?.handed) break; // its first tool calls are already running
713
-
714
- // A stalled provider needs longer to come back than a dropped socket
715
- // does, and the wait is narrated so a slow turn never looks like a hang.
716
- const backoff = problem.kind === 'timeout'
717
- ? 1500 * 2 ** (attempt - 1)
718
- : 400 * 2 ** (attempt - 1);
719
- opts.onWait?.(
720
- `${problem.kind === 'timeout' ? 'provider stalled' : 'connection failed'} — ` +
721
- `retrying (${attempt + 1}/${attempts})`
722
- );
723
- await pause(backoff);
724
- }
725
- }
726
-
727
- throw problem;
728
- }
729
-
730
- /** Collect a streamed reply, handing deltas out as they land. */
731
- async function streamed(request, opts, id) {
732
- // A watchdog of its own, so a frozen stream can be ended without it looking
733
- // like the user pressed stop.
734
- const quiet = new AbortController();
735
- const stop = () => quiet.abort();
736
- opts.signal?.addEventListener('abort', stop, { once: true });
737
- let stalled = false;
738
- let timer;
739
- const frozen = (cause) => new Failure({
740
- kind: 'timeout',
741
- attempted: `asking ${modelName(id)} for a reply`,
742
- failed: `${modelName(id)} went silent for too long, so ucode stopped waiting.`,
743
- fix: 'ucode asks again by itself. If it keeps freezing, /model to North Mini Code.',
744
- detail: { stalled: true, handed: handed.size },
745
- cause,
746
- });
747
- // Once a tool call has begun, the silence may be the call being written.
748
- // DeepSeek's free upstream holds a create_app back until the whole app is
749
- // done — two or three quiet minutes — and a one-minute watchdog killed every
750
- // build it started. So a call in progress gets ten minutes instead.
751
- const alive = () => {
752
- clearTimeout(timer);
753
- const limit = partial.size ? writingLimit() : stallLimit();
754
- timer = setTimeout(() => { stalled = true; quiet.abort(); }, limit);
755
- };
756
-
757
- let text = '';
758
- let reasoning = '';
759
- let finishReason = 'stop';
760
- let usage = null;
761
- const partial = new Map();
762
- const handed = new Set();
763
- let highest = -1;
764
-
765
- try {
766
- alive();
767
- const { data: stream, response } = await connection().chat.completions
768
- .create(
769
- { ...request, stream: true, stream_options: { include_usage: true } },
770
- { signal: quiet.signal }
771
- )
772
- .withResponse();
773
- noteLimits(response?.headers);
774
-
775
- for await (const chunk of stream) {
776
- alive();
777
- if (opts.signal?.aborted) break;
778
- if (chunk.usage) usage = chunk.usage;
779
-
780
- const choice = chunk.choices?.[0];
781
- if (!choice) continue;
782
- if (choice.finish_reason) finishReason = choice.finish_reason;
783
- const delta = choice.delta ?? {};
784
-
785
- // Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
786
- // `reasoning_content` on some upstreams.
787
- const thinking = delta.reasoning ?? delta.reasoning_content;
788
- if (thinking) {
789
- reasoning += thinking;
790
- opts.onThinking?.(thinking);
791
- }
792
-
793
- if (delta.content) {
794
- text += delta.content;
795
- opts.onText(delta.content);
796
- }
797
-
798
- // A tool call's name and arguments arrive across several chunks, keyed by
799
- // index, so they are stitched back together here.
800
- for (const call of delta.tool_calls ?? []) {
801
- // Calls arrive one after another, so the first chunk of call N means
802
- // every call before it is complete. Those are handed over at once, and
803
- // the caller can start running them while the rest are still being
804
- // written — the reply streaming and the tools working overlap.
805
- if (opts.onToolCall && call.index > highest) {
806
- for (const [index, slot] of partial) {
807
- if (index < call.index && !handed.has(index)) {
808
- handed.add(index);
809
- opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args }));
810
- }
811
- }
812
- highest = call.index;
813
- }
814
-
815
- const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
816
- if (call.id) slot.id = call.id;
817
- if (call.function?.name) slot.name += call.function.name;
818
- if (call.function?.arguments) slot.args += call.function.arguments;
819
- partial.set(call.index, slot);
820
-
821
- // A whole app arrives as one enormous arguments string that takes a
822
- // minute or two to write. Handing it over as it grows is what lets the
823
- // caller say which file is being written right now, instead of showing
824
- // a spinner that has meant nothing for ninety seconds.
825
- if (call.function?.arguments) opts.onToolArgs?.({ index: call.index, name: slot.name, args: slot.args });
826
- }
827
- }
828
- } catch (err) {
829
- if (!stalled || opts.signal?.aborted) throw err;
830
- throw frozen(err);
831
- } finally {
832
- clearTimeout(timer);
833
- opts.signal?.removeEventListener('abort', stop);
834
- }
835
- // The watchdog's abort can also end the stream quietly instead of throwing.
836
- // Carrying on from there ran a half-written create_app: JSON repair closed
837
- // it with no files, and five minutes of app were saved as an empty starter.
838
- if (stalled && !opts.signal?.aborted) throw frozen();
839
-
840
- const toolCalls = [];
841
- for (const [index, slot] of partial) {
842
- toolCalls.push(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args, cutOff: finishReason === 'length' }));
843
- }
844
-
845
- return {
846
- text: text.trim(),
847
- reasoning: reasoning.trim(),
848
- toolCalls,
849
- usage: {
850
- promptTokens: usage?.prompt_tokens ?? 0,
851
- outputTokens: usage?.completion_tokens ?? 0,
852
- totalTokens: usage?.total_tokens ?? 0,
853
- },
854
- finishReason,
855
- model: id,
856
- };
857
- }
858
-
859
- /**
860
- * Recover the files from a file-write call whose JSON will not parse.
861
- *
862
- * The usual cause is a double quote inside the code that the model forgot to
863
- * escape — `className="flex"` — which ends the JSON string early. No general
864
- * repair can know which quote was meant, but a file write has a fixed shape:
865
- * "path", then "content", then either the next file or the end. Splitting on
866
- * that shape and escaping the stray quotes gets every file back.
867
- */
868
- /**
869
- * Which calls can be rebuilt out of broken JSON.
870
- *
871
- * All three carry file contents — hundreds of lines of HTML, CSS and
872
- * JavaScript escaped into a JSON string — which is the one argument shape a
873
- * model gets wrong often enough to matter. Everything else is short enough
874
- * that a parse failure is a real mistake, worth reporting rather than guessing
875
- * around.
876
- */
877
- const SALVAGEABLE = new Set(['write_file', 'batch_write', 'create_app']);
878
-
879
- /**
880
- * The plain string arguments sitting beside a files array, read off the raw
881
- * text when the object as a whole will not parse.
882
- *
883
- * Only the keys create_app needs, and only from before the files begin, so a
884
- * "name" belonging to something nested inside a file cannot be mistaken for
885
- * the app's own.
886
- */
887
- function scalarArgs(text) {
888
- const at = text.search(/"files"\s*:/);
889
- const head = at > 0 ? text.slice(0, at) : text;
890
- const out = {};
891
- for (const key of ['folder', 'name', 'description', 'template', 'design']) {
892
- const hit = new RegExp(`"${key}"\\s*:\\s*"((?:[^"\\\\]|\\\\.)*)"`).exec(head);
893
- if (hit) out[key] = hit[1].replace(/\\(["\\/])/g, '$1');
894
- }
895
- return out;
896
- }
897
-
898
- /**
899
- * Rebuild the edits out of a broken edit_file or edit_files call.
900
- *
901
- * The same problem as a broken write, one level down: an edit carries two
902
- * slabs of somebody's source as escaped JSON strings, and one stray character
903
- * loses the call. edit_files failed this way three times in a row in a traced
904
- * build, at the end of a turn, which is a turn that ends having undone nothing
905
- * and fixed nothing.
906
- *
907
- * Safer than it sounds. Every edit is applied by replaceOnce, which requires
908
- * old_string to appear exactly once and refuses otherwise — so an edit
909
- * recovered wrongly does not corrupt a file, it fails to match and says so.
910
- * The risk of guessing is a clear error; the cost of not guessing is the whole
911
- * call. Truncated replies are still refused upstream, where half a string
912
- * would mean half a file.
913
- *
914
- * Each edit is attributed to the nearest "path" before it, which is how
915
- * edit_files nests them; edit_file has exactly one of each.
916
- */
917
- export function salvageEdits(text) {
918
- const heads = [...text.matchAll(/"old_string"\s*:\s*"/g)];
919
- if (!heads.length) return null;
920
-
921
- const paths = [...text.matchAll(/"path"\s*:\s*"((?:[^"\\]|\\.)*)"/g)];
922
- const NEW = '"new_string"';
923
- const out = [];
924
-
925
- for (let i = 0; i < heads.length; i++) {
926
- const oldFrom = heads[i].index + heads[i][0].length;
927
- const newKey = text.indexOf(NEW, oldFrom);
928
- if (newKey < 0) return null;
929
-
930
- const opens = /^\s*:\s*"/.exec(text.slice(newKey + NEW.length));
931
- if (!opens) return null;
932
- const newFrom = newKey + NEW.length + opens[0].length;
933
-
934
- // The value runs until whatever structure comes next — the following edit,
935
- // or the following file — and the trailing JSON punctuation comes off.
936
- const nextEdit = i + 1 < heads.length ? heads[i + 1].index : text.length;
937
- const nextPath = paths.find((m) => m.index > newFrom)?.index ?? text.length;
938
- const strip = (v) => v.replace(/"[\s,{}[\]]*$/, '');
939
-
940
- const old_string = unescapeLoose(strip(text.slice(oldFrom, newKey)));
941
- const new_string = unescapeLoose(strip(text.slice(newFrom, Math.min(nextEdit, nextPath))));
942
- const owner = paths.filter((m) => m.index < heads[i].index).pop();
943
-
944
- if (!owner || !old_string) return null;
945
- out.push({ path: unescapeLoose(owner[1]), old_string, new_string });
946
- }
947
-
948
- return out.length ? out : null;
949
- }
950
-
951
- /** The same edits, grouped under their file, which is edit_files' own shape. */
952
- export function groupEdits(edits) {
953
- const byPath = new Map();
954
- for (const { path, old_string, new_string } of edits) {
955
- if (!byPath.has(path)) byPath.set(path, []);
956
- byPath.get(path).push({ old_string, new_string });
957
- }
958
- return [...byPath].map(([path, list]) => ({ path, edits: list }));
959
- }
960
-
961
- export function salvageWrites(text) {
962
- const heads = [...text.matchAll(/"path"\s*:\s*"((?:[^"\\]|\\.)*)"\s*,\s*"content"\s*:\s*"/g)];
963
- if (!heads.length) return null;
964
-
965
- const files = [];
966
- for (let i = 0; i < heads.length; i++) {
967
- const from = heads[i].index + heads[i][0].length;
968
- const to = i + 1 < heads.length ? heads[i + 1].index : text.length;
969
- // The string ends at the last quote that is followed by nothing but JSON
970
- // punctuation — `"}, {`, or `"}}, {` when the model added a brace, or `"}]}`.
971
- const body = text.slice(from, to).replace(/"[\s,{}[\]]*$/, '');
972
- const content = unescapeLoose(body);
973
- const pathValue = unescapeLoose(heads[i][1]);
974
- if (!pathValue || !content) return null; // not the shape we thought — leave it an honest error
975
- files.push({ path: pathValue, content });
976
- }
977
- return files.length ? files : null;
978
- }
979
-
980
- /**
981
- * Decode a JSON string body the forgiving way: the standard escapes are
982
- * honoured, and everything JSON would reject — a raw line break, a tab, a stray
983
- * quote, an escape JSON does not know — is kept as the character it plainly is.
984
- */
985
- function unescapeLoose(s) {
986
- const simple = { n: '\n', t: '\t', r: '\r', b: '\b', f: '\f', '"': '"', '\\': '\\', '/': '/' };
987
- let out = '';
988
- for (let i = 0; i < s.length; i++) {
989
- const c = s[i];
990
- if (c !== '\\' || i === s.length - 1) { out += c; continue; }
991
- const next = s[++i];
992
- if (next === 'u' && /^[0-9a-fA-F]{4}$/.test(s.slice(i + 1, i + 5))) {
993
- out += String.fromCharCode(parseInt(s.slice(i + 1, i + 5), 16));
994
- i += 4;
995
- } else {
996
- out += simple[next] ?? next;
997
- }
998
- }
999
- return out;
1000
- }
1001
-
1002
- /**
1003
- * Parse one tool call's arguments.
1004
- *
1005
- * A parse error is recorded on the call rather than thrown. The loop hands it
1006
- * back to the model, which usually fixes its own JSON on the next step —
1007
- * cheaper than failing the whole turn over a stray comma.
1008
- */
1009
- export function readCall({ id, name, raw, cutOff = false }) {
1010
- const call = { id, name, args: {} };
1011
- const text = String(raw ?? '').trim();
1012
- if (!text) return call;
1013
- try {
1014
- const parsed = JSON.parse(text);
1015
- if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) call.args = parsed;
1016
- else call.parseError = `arguments must be a JSON object, got ${Array.isArray(parsed) ? 'array' : typeof parsed}`;
1017
- } catch (err) {
1018
- // A missing comma or a stray control character in a 3,000-token
1019
- // batch_write used to throw the whole step away — half a minute of output
1020
- // discarded over one character. Repair the usual slips instead. Never when
1021
- // the reply was cut off at the output limit, though: repairing that would
1022
- // close the string and quietly write half a file.
1023
- if (!cutOff) {
1024
- try {
1025
- const fixed = JSON.parse(jsonrepair(text));
1026
- if (fixed && typeof fixed === 'object' && !Array.isArray(fixed)) {
1027
- call.args = fixed;
1028
- call.repaired = true;
1029
- return call;
1030
- }
1031
- } catch { /* beyond general repair — try the file-write shape next */ }
1032
-
1033
- if (name === 'edit_file' || name === 'edit_files' || name === 'multi_edit') {
1034
- const edits = salvageEdits(text);
1035
- if (edits) {
1036
- if (name === 'edit_file') call.args = edits[0];
1037
- else if (name === 'multi_edit') {
1038
- call.args = { path: edits[0].path, edits: edits.map(({ old_string, new_string }) => ({ old_string, new_string })) };
1039
- } else call.args = { files: groupEdits(edits) };
1040
- call.repaired = true;
1041
- return call;
1042
- }
1043
- }
1044
-
1045
- const salvaged = SALVAGEABLE.has(name) ? salvageWrites(text) : null;
1046
- if (salvaged) {
1047
- if (name === 'write_file') call.args = salvaged[0];
1048
- // create_app carries the app in the same { path, content } shape, with
1049
- // a few plain strings beside it. Losing the entire call because one of
1050
- // several hundred lines of HTML held a raw newline is how a build ends
1051
- // with no files at all — which is what it did: refused three times,
1052
- // the folder never created, the turn over in sixty-six seconds having
1053
- // produced nothing. The scalars are read back off the same text.
1054
- else if (name === 'create_app') call.args = { ...scalarArgs(text), files: salvaged };
1055
- else call.args = { files: salvaged };
1056
- call.repaired = true;
1057
- return call;
1058
- }
1059
- }
1060
- call.parseError = `${err.message} — the raw arguments were: ${text.slice(0, 300)}`;
1061
- }
1062
- return call;
1063
- }
1064
-
1065
- function normalize(data, id) {
1066
- const choice = data?.choices?.[0];
1067
- const message = choice?.message ?? {};
1068
-
1069
- const toolCalls = (message.tool_calls ?? []).map((c) =>
1070
- readCall({ id: c.id, name: c.function?.name, raw: c.function?.arguments, cutOff: choice?.finish_reason === 'length' })
1071
- );
1072
-
1073
- const u = data?.usage ?? {};
1074
- const finishReason = choice?.finish_reason ?? 'stop';
1075
- const text = (message.content ?? '').trim();
1076
-
1077
- if (!text && toolCalls.length === 0 && finishReason === 'length') {
1078
- throw new Failure({
1079
- kind: 'no_content',
1080
- attempted: `asking ${modelName(id)} for a reply`,
1081
- failed: 'The reply hit the output limit before producing anything at all.',
1082
- fix: 'Ask for something shorter, or split the task into steps.',
1083
- detail: { finishReason },
1084
- });
1085
- }
1086
-
1087
- return {
1088
- text,
1089
- reasoning: (message.reasoning ?? '').trim(),
1090
- toolCalls,
1091
- usage: {
1092
- promptTokens: u.prompt_tokens ?? 0,
1093
- outputTokens: u.completion_tokens ?? 0,
1094
- totalTokens: u.total_tokens ?? 0,
1095
- },
1096
- finishReason,
1097
- model: id,
1098
- };
1099
- }
1
+ /**
2
+ * provider.js — the only module that knows a model provider exists.
3
+ *
4
+ * Everything above this file speaks one small neutral message format and calls
5
+ * ask(). Moving ucode to a different host means rewriting this file and
6
+ * nothing else.
7
+ *
8
+ * The neutral formats:
9
+ * { role: 'system', content }
10
+ * { role: 'user', content, images?: [dataUrl] }
11
+ * { role: 'assistant', content?, toolCalls?: [{ id, name, args }] }
12
+ * { role: 'tool', toolCallId, name, content }
13
+ *
14
+ * tool: { name, description, parameters: <JSON Schema> }
15
+ *
16
+ * ask() resolves to:
17
+ * { text, reasoning, toolCalls, usage, finishReason, model }
18
+ */
19
+
20
+ import { join, dirname } from 'node:path';
21
+ import { homedir } from 'node:os';
22
+ import { fileURLToPath } from 'node:url';
23
+ import dotenv from 'dotenv';
24
+ import OpenAI from 'openai';
25
+ import { jsonrepair } from 'jsonrepair';
26
+ import { Failure } from './failure.js';
27
+
28
+ const HERE = dirname(fileURLToPath(import.meta.url));
29
+ const PACKAGE_ROOT = join(HERE, '..', '..');
30
+
31
+ /** Personal config lives here, outside the package, so upgrades never touch it. */
32
+ export const UCODE_HOME = join(homedir(), '.ucode');
33
+ export const ENV_FILE = join(UCODE_HOME, '.env');
34
+
35
+ export const BASE_URL = 'https://openrouter.ai/api/v1';
36
+ export const PROVIDER = 'OpenRouter';
37
+
38
+ // First definition wins — dotenv never overwrites a variable that already
39
+ // exists — so the order here is the precedence order:
40
+ // real environment > ./.env (this project) > ~/.ucode/.env (this machine)
41
+ // > the checkout's own .env (only when developing on a clone)
42
+ dotenv.config({ path: join(process.cwd(), '.env'), quiet: true });
43
+ dotenv.config({ path: ENV_FILE, quiet: true });
44
+ dotenv.config({ path: join(PACKAGE_ROOT, '.env'), quiet: true });
45
+
46
+ /**
47
+ * The whole model list. Not a starting point — the list.
48
+ *
49
+ * ucode runs on NVIDIA, Cohere and Nex AGI only. All three serve genuinely
50
+ * capable models free through OpenRouter, all handle tool calling properly,
51
+ * and keeping the set to six means every one of them has
52
+ * been used in anger rather than listed on the strength of a benchmark. A
53
+ * picker offering sixty models is a picker nobody reads.
54
+ *
55
+ * `name` is what the status bar shows. `note` is what the picker shows.
56
+ */
57
+ export const MODELS = {
58
+ 'nvidia/nemotron-3-ultra-550b-a55b:free': {
59
+ name: 'Nemotron 3 Ultra',
60
+ context: 1_000_000,
61
+ star: true,
62
+ note: 'deepest reasoning, 1M context — slowest to answer',
63
+ },
64
+ 'nvidia/nemotron-3.5-lightning:free': {
65
+ name: 'Nemotron 3.5 Lightning',
66
+ context: 1_000_000,
67
+ note: 'same huge window, answers much sooner',
68
+ },
69
+ 'nvidia/nemotron-3-super-120b-a12b:free': {
70
+ name: 'Nemotron 3 Super',
71
+ context: 262_144,
72
+ note: 'strong all-rounder, quick to first token',
73
+ },
74
+ 'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free': {
75
+ name: 'Nemotron 3 Nano Omni',
76
+ context: 256_000,
77
+ note: 'small and fast, reasoning tuned',
78
+ },
79
+ 'cohere/north-mini-code:free': {
80
+ name: 'North Mini Code',
81
+ context: 256_000,
82
+ star: true,
83
+ note: 'the default — built for code and interface work, quick to answer',
84
+ },
85
+ // On trial, so not the default and not in FALLBACKS yet. Nex AGI is its only
86
+ // upstream, so a stall there has nowhere else to go.
87
+ 'nex-agi/nex-n2.5-pro:free': {
88
+ name: 'Nex N2.5 Pro',
89
+ context: 262_144,
90
+ note: 'new agentic coder, on trial — can stall on big builds',
91
+ },
92
+ };
93
+
94
+ /**
95
+ * North Mini Code is the default: it is built for code and interface work,
96
+ * which is what ucode is mostly asked to do, and it answers far sooner than
97
+ * the big reasoning models. /model moves to Ultra when a problem needs the
98
+ * million-token window and the long think more than it needs the speed.
99
+ */
100
+ export const DEFAULT_MODEL = 'cohere/north-mini-code:free';
101
+
102
+ /**
103
+ * Where to go when a model is busy, in order of preference. Each is served by
104
+ * a different upstream, so a rate limit on one rarely means a limit on the
105
+ * next — which is what lets a long build keep going instead of stopping at
106
+ * the first "too many requests".
107
+ */
108
+ export const FALLBACKS = [
109
+ 'cohere/north-mini-code:free',
110
+ 'nvidia/nemotron-3.5-lightning:free',
111
+ 'nvidia/nemotron-3-super-120b-a12b:free',
112
+ 'nvidia/nemotron-3-ultra-550b-a55b:free',
113
+ ];
114
+
115
+ /** The next model to try after `id`, skipping any already tried this round. */
116
+ export function fallbackFor(id, tried = new Set()) {
117
+ // Off unless UCODE_FALLBACK=1. A build that starts on one model and finishes
118
+ // on another finishes to a different standard, and the swap lands exactly
119
+ // when the user is least placed to work out why the output changed —
120
+ // mid-build, behind a note that scrolls past. Which model to run is the one
121
+ // decision they made before starting; it is not one to take back for them.
122
+ //
123
+ // Every caller already reads "no fallback" as "wait, then try this one
124
+ // again": failover waits a minute and returns to the same model, the
125
+ // stuck-detector simply does not switch, and a worker retries its own. That
126
+ // is why this can be a single gate rather than four.
127
+ if (process.env.UCODE_FALLBACK !== '1') return null;
128
+
129
+ const start = Math.max(0, FALLBACKS.indexOf(id));
130
+ for (let i = 1; i <= FALLBACKS.length; i++) {
131
+ const next = FALLBACKS[(start + i) % FALLBACKS.length];
132
+ if (next !== id && !tried.has(next)) return next;
133
+ }
134
+ return null;
135
+ }
136
+
137
+ /** Seconds to wait on successive rate limits that come with no retry-after. */
138
+ const RATE_LIMIT_BACKOFF = [5, 10, 20];
139
+
140
+ /**
141
+ * How long a stream may go without a single chunk before it counts as frozen.
142
+ *
143
+ * A free endpoint can accept a request and then send nothing at all, and the
144
+ * only thing that used to end that was the five-minute request timeout — five
145
+ * minutes of a spinner, then the same again on the retry. Reasoning streams
146
+ * as it is produced, so even the slowest thinker sends something well inside
147
+ * a minute; silence for that long means nobody is working on the reply.
148
+ */
149
+ export const stallLimit = () => Number(process.env.UCODE_STALL_MS) || 60_000;
150
+
151
+ /** Freezes in a row before ucode stops asking and says to switch models. */
152
+ export const MAX_STALLS = 3;
153
+ let stalls = 0;
154
+
155
+ let current = process.env.UCODE_MODEL || DEFAULT_MODEL;
156
+ let client = null;
157
+
158
+ export function model() {
159
+ return current;
160
+ }
161
+
162
+ export function setModel(id) {
163
+ const wanted = String(id ?? '').trim();
164
+ if (!wanted) {
165
+ throw new Failure({
166
+ kind: 'bad_model',
167
+ attempted: 'switching model',
168
+ failed: 'No model name was given.',
169
+ fix: `Pick one of: ${Object.keys(MODELS).join(', ')}`,
170
+ });
171
+ }
172
+ if (!MODELS[wanted]) {
173
+ throw new Failure({
174
+ kind: 'bad_model',
175
+ attempted: `switching to "${wanted}"`,
176
+ failed: 'ucode only runs NVIDIA, Cohere and Nex AGI models, and that is not one of them.',
177
+ fix: `Run /model to choose from: ${Object.keys(MODELS).join(', ')}`,
178
+ });
179
+ }
180
+ current = wanted;
181
+ return current;
182
+ }
183
+
184
+ /** The short name for a model id: "Nemotron 3 Ultra". */
185
+ export function modelName(id = current) {
186
+ return MODELS[id]?.name ?? id;
187
+ }
188
+
189
+ /** Every model, in list order, annotated with whether it is the active one. */
190
+ export function modelList() {
191
+ return Object.entries(MODELS).map(([id, info]) => ({
192
+ id,
193
+ ...info,
194
+ active: id === current,
195
+ }));
196
+ }
197
+
198
+ /**
199
+ * How many tokens one request may occupy.
200
+ *
201
+ * OpenRouter meters requests rather than tokens, so nothing here is rationing
202
+ * a quota — the binding limit is simply the window the model has. On the two
203
+ * million-token models compaction essentially never fires.
204
+ */
205
+ export function contextLimit(id = current) {
206
+ const override = Number(process.env.UCODE_MAX_CONTEXT_TOKENS);
207
+ if (Number.isFinite(override) && override > 0) return override;
208
+ return MODELS[id]?.context ?? 128_000;
209
+ }
210
+
211
+ /**
212
+ * A rough local token count, used to decide when to compact *before* a
213
+ * request goes out. Real numbers come back in the response usage; this only
214
+ * has to be close enough to trigger at the right time.
215
+ */
216
+ export function estimateTokens(text) {
217
+ return text ? Math.ceil(String(text).length / 4) : 0;
218
+ }
219
+
220
+ export function estimateConversation(messages) {
221
+ let total = 0;
222
+ for (const m of messages) {
223
+ total += estimateTokens(m.content || '');
224
+ for (const call of m.toolCalls || []) {
225
+ total += estimateTokens(call.name) + estimateTokens(JSON.stringify(call.args || {}));
226
+ }
227
+ total += 4; // role and framing overhead per message
228
+ }
229
+ return total;
230
+ }
231
+
232
+ // ---------------------------------------------------------------------------
233
+ // Client
234
+ // ---------------------------------------------------------------------------
235
+
236
+ /**
237
+ * No key is bundled here, and none ever should be. A key committed to a public
238
+ * package is readable by anyone who runs `npm pack ucode-agent`, and no amount
239
+ * of first-run convenience is worth handing out a live credential.
240
+ */
241
+ function apiKey() {
242
+ // UCODE_API_KEY is the documented name. The provider's own variable name is
243
+ // still read, so a key set up for another tool keeps working here.
244
+ const key = (process.env.UCODE_API_KEY || process.env.OPENROUTER_API_KEY || '').trim();
245
+ if (!key) {
246
+ throw new Failure({
247
+ kind: 'no_api_key',
248
+ attempted: 'connecting to the model',
249
+ failed: 'No API key is set - UCODE_API_KEY is missing from the environment and from every .env file.',
250
+ fix:
251
+ `Put UCODE_API_KEY=your-key in ${ENV_FILE} — that applies to every ` +
252
+ 'project on this machine — or in a .env file beside your code. ' +
253
+ 'Free keys: https://openrouter.ai/keys',
254
+ });
255
+ }
256
+ return key;
257
+ }
258
+
259
+ function connection() {
260
+ if (client) return client;
261
+ client = new OpenAI({
262
+ apiKey: apiKey(),
263
+ baseURL: process.env.UCODE_BASE_URL || BASE_URL,
264
+ // Nemotron Ultra can think for a long time before its first token, so the
265
+ // ceiling is deliberately generous. maxRetries is 0 because ask() owns
266
+ // retrying: its attempts are narrated on screen instead of happening
267
+ // silently somewhere inside the SDK.
268
+ timeout: Number(process.env.UCODE_REQUEST_TIMEOUT_MS) || 300_000,
269
+ maxRetries: 0,
270
+ defaultHeaders: {
271
+ 'HTTP-Referer': 'https://github.com/sppideey/ucode-agent',
272
+ 'X-Title': 'ucode',
273
+ },
274
+ });
275
+ return client;
276
+ }
277
+
278
+ /** Drop the cached client so the next request picks up a changed key. */
279
+ export function resetConnection() {
280
+ client = null;
281
+ }
282
+
283
+ // ---------------------------------------------------------------------------
284
+ // Live quota, taken from whatever rate-limit headers come back
285
+ // ---------------------------------------------------------------------------
286
+
287
+ let limits = null;
288
+
289
+ export function rateLimits() {
290
+ return limits;
291
+ }
292
+
293
+ /** "1.5s", "2m59.56s", "1h2m" -> seconds */
294
+ function seconds(value) {
295
+ if (!value) return null;
296
+ const m = /^(?:(\d+(?:\.\d+)?)h)?(?:(\d+(?:\.\d+)?)m(?!s))?(?:(\d+(?:\.\d+)?)m?s)?$/
297
+ .exec(String(value).trim());
298
+ if (!m) return null;
299
+ const total = (parseFloat(m[1]) || 0) * 3600 + (parseFloat(m[2]) || 0) * 60 + (parseFloat(m[3]) || 0);
300
+ return total > 0 ? total : null;
301
+ }
302
+
303
+ function noteLimits(headers) {
304
+ if (!headers?.get) return;
305
+ const num = (name) => {
306
+ const n = Number(headers.get(name));
307
+ return Number.isFinite(n) ? n : null;
308
+ };
309
+ limits = {
310
+ requestsLimit: num('x-ratelimit-limit-requests'),
311
+ requestsRemaining: num('x-ratelimit-remaining-requests'),
312
+ requestsReset: seconds(headers.get('x-ratelimit-reset-requests')),
313
+ tokensLimit: num('x-ratelimit-limit-tokens'),
314
+ tokensRemaining: num('x-ratelimit-remaining-tokens'),
315
+ tokensReset: seconds(headers.get('x-ratelimit-reset-tokens')),
316
+ at: Date.now(),
317
+ };
318
+ }
319
+
320
+ // ---------------------------------------------------------------------------
321
+ // Neutral format -> wire format
322
+ // ---------------------------------------------------------------------------
323
+
324
+ function wireTools(tools) {
325
+ if (!tools?.length) return undefined;
326
+ return tools.map((t) => ({
327
+ type: 'function',
328
+ function: {
329
+ name: t.name,
330
+ description: t.description,
331
+ parameters: t.parameters ?? { type: 'object', properties: {} },
332
+ },
333
+ }));
334
+ }
335
+
336
+ function wireMessages(messages) {
337
+ const out = [];
338
+ for (const m of messages) {
339
+ if (m.role === 'user' && m.images?.length) {
340
+ out.push({
341
+ role: 'user',
342
+ content: [
343
+ { type: 'text', text: m.content ?? '' },
344
+ ...m.images.map((url) => ({ type: 'image_url', image_url: { url } })),
345
+ ],
346
+ });
347
+ } else if (m.role === 'system' || m.role === 'user') {
348
+ out.push({ role: m.role, content: m.content ?? '' });
349
+ } else if (m.role === 'assistant') {
350
+ const wire = { role: 'assistant', content: m.content || '' };
351
+ if (m.toolCalls?.length) {
352
+ wire.tool_calls = m.toolCalls.map((c) => ({
353
+ id: c.id,
354
+ type: 'function',
355
+ function: { name: c.name, arguments: JSON.stringify(c.args ?? {}) },
356
+ }));
357
+ }
358
+ out.push(wire);
359
+ } else if (m.role === 'tool') {
360
+ out.push({ role: 'tool', tool_call_id: m.toolCallId, content: m.content ?? '' });
361
+ }
362
+ }
363
+ return out;
364
+ }
365
+
366
+ // ---------------------------------------------------------------------------
367
+ // Turning provider errors into something a person can act on
368
+ // ---------------------------------------------------------------------------
369
+
370
+ function bodyOf(err) {
371
+ if (err?.error) return { error: err.error };
372
+ const raw = String(err?.message ?? '');
373
+ const start = raw.indexOf('{');
374
+ if (start === -1) return null;
375
+ try {
376
+ return JSON.parse(raw.slice(start));
377
+ } catch {
378
+ return null;
379
+ }
380
+ }
381
+
382
+ /**
383
+ * How providers say the conversation no longer fits, from opencode's list
384
+ * (MIT, see THIRD_PARTY_NOTICES.md). Most arrive as a plain HTTP 400, which on
385
+ * its own reads as a malformed request and would end the turn; recognised,
386
+ * the conversation is folded and the request sent again.
387
+ */
388
+ const OVERFLOW = [
389
+ /prompt is too long/i, /request_too_large/i, /input is too long for requested model/i,
390
+ /exceeds the context window/i,
391
+ /exceeds (?:the )?(?:model'?s )?maximum context length(?: of [\d,]+ tokens?|\s*\([\d,]+\))/i,
392
+ /input token count.*exceeds the maximum/i, /tokens in request more than max tokens allowed/i,
393
+ /maximum prompt length is \d+/i, /reduce the length of the messages/i,
394
+ /maximum context length is \d+ tokens/i,
395
+ /exceeds (?:the )?maximum allowed input length of [\d,]+ tokens?/i,
396
+ /input \(\d+ tokens\) is longer than the model'?s context length \(\d+ tokens\)/i,
397
+ /exceeds the limit of \d+/i, /exceeds the available context size/i, /greater than the context length/i,
398
+ /context window exceeds limit/i, /exceeded model token limit/i, /context[_ ]length[_ ]exceeded/i,
399
+ /request entity too large/i, /context length is only \d+ tokens/i, /input length.*exceeds.*context length/i,
400
+ /prompt too long; exceeded (?:max )?context length/i, /too large for model with \d+ maximum context length/i,
401
+ /prompt has [\d,]+ tokens?, but the configured context size is [\d,]+ tokens?/i,
402
+ /model_context_window_exceeded/i, /too many tokens/i, /token limit exceeded/i,
403
+ ];
404
+ const THROTTLED = [/^(?:throttling error|service unavailable):/i, /rate limit/i, /too many requests/i];
405
+
406
+ /** A limit on the conversation's size, as opposed to a limit on how fast it is sent. */
407
+ export const overflowed = (text) =>
408
+ !THROTTLED.some((p) => p.test(text)) && OVERFLOW.some((p) => p.test(text));
409
+
410
+ /**
411
+ * The provider's own words for "busy, try again", also from opencode. OpenRouter
412
+ * passes an upstream's hiccup through as "Provider returned error" on a 400,
413
+ * and treating that as a bad request stopped builds that one retry would save.
414
+ */
415
+ const TRANSIENT = /overloaded|service[ _-]unavailable|internal[ _-]error|internal server error|server[ _-]error|provider[ _-]returned[ _-]error|resource[ _-]exhausted|try your request again|retry your request|\btry again (?:later|in\b)|\b(?:currently|temporarily) at capacity\b/i;
416
+
417
+ export function explain(err, id) {
418
+ if (err instanceof Failure) return err;
419
+
420
+ const status = err?.status ?? err?.statusCode ?? null;
421
+ const body = bodyOf(err);
422
+ const detail = body?.error?.message ?? String(err?.message ?? err);
423
+ const attempted = `asking ${modelName(id)} for a reply`;
424
+
425
+ // The useful text is often not on the error itself. undici reports a socket
426
+ // that closed mid-response as a bare `TypeError: terminated` and puts the
427
+ // real reason on `cause`, so the whole chain is matched rather than the top
428
+ // message alone — otherwise an ordinary dropped connection, which is worth
429
+ // retrying, gets reported as an unknown fault, which is not.
430
+ const chain = [err?.message, err?.code, err?.cause?.message, err?.cause?.code]
431
+ .filter(Boolean)
432
+ .join(' | ');
433
+ const raw = chain || String(err);
434
+
435
+ if (err?.name === 'AbortError' || /aborted|The operation was aborted/i.test(raw)) {
436
+ return new Failure({
437
+ kind: 'aborted',
438
+ attempted,
439
+ failed: 'The request was cancelled.',
440
+ fix: 'Send the message again when you are ready.',
441
+ cause: err,
442
+ });
443
+ }
444
+
445
+ if (status === 401 || status === 403 || /invalid[_ ]api[_ ]key/i.test(raw)) {
446
+ return new Failure({
447
+ kind: 'invalid_api_key',
448
+ attempted,
449
+ failed: `The API key was rejected (HTTP ${status ?? 401}).`,
450
+ fix:
451
+ 'Check UCODE_API_KEY in ~/.ucode/.env for a typo or trailing space, and ' +
452
+ 'confirm the key is still active in your account.',
453
+ cause: err,
454
+ });
455
+ }
456
+
457
+ if (status !== 429 && (status === 413 || overflowed(detail) || overflowed(raw))) {
458
+ return new Failure({
459
+ kind: 'too_large',
460
+ attempted,
461
+ failed: `The conversation no longer fits in ${modelName(id)}: ${detail}`,
462
+ fix: 'Run /new for a fresh session, or lower UCODE_MAX_CONTEXT_TOKENS so ucode folds older turns away sooner.',
463
+ cause: err,
464
+ });
465
+ }
466
+
467
+ if (status === 429 || /rate[_ ]limit/i.test(raw)) {
468
+ // `??` cannot be used to chain through Number(): Number(undefined) is NaN,
469
+ // which is neither null nor undefined, so it would swallow every fallback
470
+ // after it and the wait would silently never be found.
471
+ const header = err?.headers?.get?.('retry-after');
472
+ const asNumber = Number(header);
473
+ // retry-after-ms is exact and often well under a second; waiting the
474
+ // five-second fallback instead is time thrown away on every rate limit.
475
+ const exactMs = Number(err?.headers?.get?.('retry-after-ms'));
476
+ const asDate = header && !Number.isFinite(asNumber) ? (Date.parse(header) - Date.now()) / 1000 : NaN;
477
+ const retryAfter =
478
+ (Number.isFinite(exactMs) && exactMs > 0 ? exactMs / 1000 : null) ??
479
+ seconds(header) ??
480
+ (Number.isFinite(asNumber) && asNumber > 0 ? asNumber : null) ??
481
+ (asDate > 0 ? asDate : null) ??
482
+ seconds(/try again in ([\dhms.]+)/i.exec(detail)?.[1]) ??
483
+ null;
484
+ // The daily cap reads "free-models-per-day-high-balance", with hyphens, and
485
+ // names its source in the metadata. It is one cap across every free model,
486
+ // so it is reported at once rather than waited on model after model.
487
+ const meta = body?.error?.metadata ?? {};
488
+ const daily = /per[- ]day|RPD|TPD|daily/i.test(`${detail} ${meta.limit_source ?? ''}`);
489
+ const resetMs = Number(meta.headers?.['X-RateLimit-Reset'] ?? err?.headers?.get?.('x-ratelimit-reset'));
490
+ const resetAt = daily && Number.isFinite(resetMs) && resetMs > Date.now() ? new Date(resetMs) : null;
491
+ const cap = Number(meta.headers?.['X-RateLimit-Limit']) || null;
492
+ const wait = Number.isFinite(retryAfter) && retryAfter
493
+ ? (retryAfter >= 60 ? `${Math.ceil(retryAfter / 60)} min` : `${Math.ceil(retryAfter)}s`)
494
+ : null;
495
+
496
+ return new Failure({
497
+ kind: 'rate_limit',
498
+ attempted,
499
+ failed: daily
500
+ ? `This key's free daily limit${cap ? ` of ${cap} requests` : ''} is used up. It covers every free model, so switching will not help.`
501
+ : `Too many requests for ${modelName(id)} just now${wait ? ` — clear in ${wait}` : ''}.`,
502
+ fix: daily
503
+ ? `It resets ${resetAt ? `at ${resetAt.toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' })}` : 'once a day'}. ` +
504
+ 'Adding credit to your account raises the limit.'
505
+ : 'ucode waits these out on its own. Free endpoints are shared, so it usually ' +
506
+ 'clears in seconds; /model moves to a quieter one.',
507
+ detail: { retryAfter, daily, resetAt: resetAt?.getTime() ?? null },
508
+ cause: err,
509
+ });
510
+ }
511
+
512
+ // A 404 on a model in this list is almost never a bad name. It is OpenRouter
513
+ // having no upstream free to serve it at that instant, and it clears by
514
+ // itself — so it is retried rather than reported as a missing model.
515
+ if (status === 404 || /does not exist|not found|decommissioned/i.test(raw)) {
516
+ if (MODELS[id]) {
517
+ return new Failure({
518
+ kind: 'server',
519
+ attempted,
520
+ failed: `No provider was free to serve ${modelName(id)} at that moment.`,
521
+ fix: 'ucode retries this by itself. If it keeps up, /model switches.',
522
+ detail: { status },
523
+ cause: err,
524
+ });
525
+ }
526
+ return new Failure({
527
+ kind: 'bad_model',
528
+ attempted,
529
+ failed: `No model "${id}" is available to this key.`,
530
+ fix: `Run /model. ucode ships with: ${Object.keys(MODELS).join(', ')}`,
531
+ cause: err,
532
+ });
533
+ }
534
+
535
+ // Some upstreams validate tool calls themselves and reject an invented one,
536
+ // which fails the whole request. Recoverable: the loop feeds this back.
537
+ if (body?.error?.code === 'tool_use_failed' || /tool call validation failed/i.test(detail)) {
538
+ const attemptedName = /call tool '([^']+)'/.exec(detail)?.[1];
539
+ return new Failure({
540
+ kind: 'bad_tool_call',
541
+ attempted,
542
+ failed: attemptedName
543
+ ? `The model tried to call "${attemptedName}", which is not one of its tools.`
544
+ : `The model produced a tool call the provider rejected: ${detail}`,
545
+ fix: 'Use only the tools supplied with the request.',
546
+ detail: { attemptedName },
547
+ cause: err,
548
+ });
549
+ }
550
+
551
+ if (status === 400 && TRANSIENT.test(detail)) {
552
+ return new Failure({
553
+ kind: 'server',
554
+ attempted,
555
+ failed: `${modelName(id)}'s provider had a passing fault: ${detail}`,
556
+ fix: 'ucode retries this by itself. If it keeps up, /model switches.',
557
+ detail: { status },
558
+ cause: err,
559
+ });
560
+ }
561
+
562
+ if (status === 400) {
563
+ const noTools = /tool calling.*not supported/i.test(detail);
564
+ return new Failure({
565
+ kind: noTools ? 'no_tool_support' : 'bad_request',
566
+ attempted,
567
+ failed: noTools
568
+ ? `${modelName(id)} cannot call tools, which ucode needs for every task.`
569
+ : `The request was rejected as malformed (HTTP 400): ${detail}`,
570
+ fix: noTools
571
+ ? 'Run /model and pick another one.'
572
+ : 'The conversation has something in it the provider will not accept. /new starts a fresh one.',
573
+ cause: err,
574
+ });
575
+ }
576
+
577
+ if (status === 413 || /too large|context.*length/i.test(detail)) {
578
+ return new Failure({
579
+ kind: 'too_large',
580
+ attempted,
581
+ failed: `The conversation no longer fits in ${modelName(id)}: ${detail}`,
582
+ fix: 'Run /new for a fresh session, or lower UCODE_MAX_CONTEXT_TOKENS so ucode folds older turns away sooner.',
583
+ cause: err,
584
+ });
585
+ }
586
+
587
+ // OpenRouter drops a request when the upstream goes quiet on it. That is the
588
+ // ordinary failure mode of a busy free endpoint, and of a big reasoning model
589
+ // that spends a long time thinking before its first token. Nothing was
590
+ // generated, so retrying is safe — and ask() does it before anyone notices.
591
+ if (
592
+ status === 408 || status === 504 || status === 524 || status === 522 ||
593
+ err?.name === 'APIConnectionTimeoutError' ||
594
+ /idle timeout|timed out|timeout/i.test(detail) || /idle timeout|timed out/i.test(raw)
595
+ ) {
596
+ return new Failure({
597
+ kind: 'timeout',
598
+ attempted,
599
+ failed: `${modelName(id)} sent nothing back in time — the provider dropped the request.`,
600
+ fix:
601
+ 'Free endpoints stall under load, and the largest reasoning models are the ' +
602
+ 'first to. ucode already retried. If it keeps happening, /model to Nemotron ' +
603
+ '3.5 Lightning or North Mini Code, which answer sooner.',
604
+ detail: { status },
605
+ cause: err,
606
+ });
607
+ }
608
+
609
+ if (status >= 500 || /internal|unavailable|overloaded/i.test(raw)) {
610
+ return new Failure({
611
+ kind: 'server',
612
+ attempted,
613
+ failed: `The provider returned HTTP ${status}. That is their side, not yours.`,
614
+ fix: 'Wait a few seconds and send again. If it persists, /model to another one.',
615
+ cause: err,
616
+ });
617
+ }
618
+
619
+ // `terminated` and `premature close` are what a connection dropped part way
620
+ // through a reply looks like. Nothing usable arrived, so it is safe to send
621
+ // again — and on a free endpoint under load it happens often enough that
622
+ // treating it as fatal would be the single most visible flaw in the agent.
623
+ if (
624
+ /ENOTFOUND|ECONNREFUSED|ECONNRESET|EAI_AGAIN|ETIMEDOUT|EPIPE|network|fetch failed|failed to fetch|socket hang up|terminated|premature close|other side closed|UND_ERR|upstream connect|connection (?:error|refused|lost)|socket connection was closed|reset before headers|getaddrinfo/i.test(raw) ||
625
+ err?.name === 'APIConnectionError' ||
626
+ (err instanceof TypeError && /terminated/i.test(raw))
627
+ ) {
628
+ return new Failure({
629
+ kind: 'network',
630
+ attempted,
631
+ failed: `The connection to the model dropped: ${raw}`,
632
+ fix:
633
+ 'ucode retries this by itself. If it keeps happening, check your connection, ' +
634
+ 'VPN and any corporate proxy (HTTPS_PROXY) — or /model to a lighter one, since ' +
635
+ 'a long think on a busy free endpoint is the usual cause.',
636
+ cause: err,
637
+ });
638
+ }
639
+
640
+ return new Failure({
641
+ kind: 'unknown',
642
+ attempted,
643
+ failed: detail,
644
+ fix: 'Retry once. If it repeats, run ucode --debug for the full trace.',
645
+ cause: err,
646
+ });
647
+ }
648
+
649
+ const pause = (ms) => new Promise((r) => setTimeout(r, ms));
650
+
651
+ // ---------------------------------------------------------------------------
652
+ // The one call
653
+ // ---------------------------------------------------------------------------
654
+
655
+ /**
656
+ * Send a conversation and get a normalized reply.
657
+ *
658
+ * @param {Array} messages neutral messages
659
+ * @param {Array} [tools] neutral tool definitions
660
+ * @param {object} [opts] { model, temperature, signal, maxOutputTokens,
661
+ * reasoning, attempts, onText, onThinking, onWait }
662
+ */
663
+ export async function ask(messages, tools = [], opts = {}) {
664
+ const id = opts.model || current;
665
+
666
+ const request = {
667
+ model: id,
668
+ messages: wireMessages(messages),
669
+ // Ask for the working. Without it OpenRouter withholds reasoning entirely,
670
+ // and on a reasoning model that *is* the whole reply until the very end:
671
+ // the socket sits silent for the length of the think, the screen shows
672
+ // nothing, and the provider eventually drops the request as idle. Asking
673
+ // for it fixes the blank screen and the dropped request together. Models
674
+ // that do not reason ignore the flag.
675
+ include_reasoning: true,
676
+ };
677
+
678
+ const wired = wireTools(tools);
679
+ if (wired) {
680
+ request.tools = wired;
681
+ request.tool_choice = 'auto';
682
+ }
683
+ if (opts.temperature !== undefined) request.temperature = opts.temperature;
684
+ if (opts.maxOutputTokens) request.max_tokens = opts.maxOutputTokens;
685
+ if (opts.reasoning) request.reasoning = opts.reasoning;
686
+
687
+ // A side call (the design review) passes fewer: it is better skipped than
688
+ // waited on through a string of rate-limit pauses.
689
+ const attempts = opts.attempts ?? 4;
690
+ let problem;
691
+
692
+ // Text already on screen cannot be unprinted, so a stream is only safe to
693
+ // retry while it is still silent. Every timeout worth retrying happens
694
+ // before the first token, so this costs nothing in practice.
695
+ let printed = 0;
696
+ const callOpts = opts.onText
697
+ ? { ...opts, onText: (d) => { printed += d.length; opts.onText(d); } }
698
+ : opts;
699
+
700
+ for (let attempt = 1; attempt <= attempts; attempt++) {
701
+ try {
702
+ if (opts.onText) {
703
+ const reply = await streamed(request, callOpts, id);
704
+ stalls = 0;
705
+ return reply;
706
+ }
707
+ const { data, response } = await connection().chat.completions
708
+ .create(request, { signal: opts.signal })
709
+ .withResponse();
710
+ noteLimits(response?.headers);
711
+ stalls = 0;
712
+ return normalize(data, id, request.tools?.map((t) => t.function.name));
713
+ } catch (err) {
714
+ noteLimits(err?.headers);
715
+ problem = explain(err, id);
716
+
717
+ // Retrying a model that keeps freezing only repeats the wait. After a
718
+ // few in a row, say so and hand the choice back.
719
+ if (problem.detail?.stalled && ++stalls >= MAX_STALLS) {
720
+ stalls = 0;
721
+ throw new Failure({
722
+ kind: 'stalled',
723
+ attempted: `asking ${modelName(id)} for a reply`,
724
+ failed: `${modelName(id)} froze ${MAX_STALLS} times in a row — it took the request and then sent nothing.`,
725
+ fix: 'Its free endpoint is struggling right now. Run /model and pick another one; North Mini Code answers soonest.',
726
+ cause: problem,
727
+ });
728
+ }
729
+
730
+ // A per-minute limit is a wait, not a failure. Sit it out rather than
731
+ // making the user retype their message. Free endpoints often refuse
732
+ // without saying how long to wait, so when there is no retry-after the
733
+ // pauses grow on their own — 5s, 10s, 20s — and only then does the
734
+ // error go up to the loop, which moves to another model.
735
+ const told = problem.detail?.retryAfter;
736
+ const wait = Number.isFinite(told) && told > 0 && told <= 90 ? told : RATE_LIMIT_BACKOFF[attempt - 1];
737
+ if (
738
+ problem.kind === 'rate_limit' && !problem.detail?.daily &&
739
+ wait && attempt < attempts && printed === 0 && !opts.signal?.aborted
740
+ ) {
741
+ const until = Date.now() + wait * 1000;
742
+ while (Date.now() < until && !opts.signal?.aborted) {
743
+ opts.onWait?.(`rate limited — resuming in ${Math.ceil((until - Date.now()) / 1000)}s`);
744
+ await pause(Math.min(1000, until - Date.now()));
745
+ }
746
+ if (opts.signal?.aborted) break;
747
+ continue;
748
+ }
749
+
750
+ const worthRetrying =
751
+ problem.kind === 'server' || problem.kind === 'network' || problem.kind === 'timeout';
752
+ if (!worthRetrying || attempt === attempts || opts.signal?.aborted) break;
753
+ if (printed > 0) break; // half an answer is on screen; do not print it twice
754
+ if (problem.detail?.handed) break; // its first tool calls are already running
755
+
756
+ // A stalled provider needs longer to come back than a dropped socket
757
+ // does, and the wait is narrated so a slow turn never looks like a hang.
758
+ const backoff = problem.kind === 'timeout'
759
+ ? 1500 * 2 ** (attempt - 1)
760
+ : 400 * 2 ** (attempt - 1);
761
+ opts.onWait?.(
762
+ `${problem.kind === 'timeout' ? 'provider stalled' : 'connection failed'} — ` +
763
+ `retrying (${attempt + 1}/${attempts})`
764
+ );
765
+ await pause(backoff);
766
+ }
767
+ }
768
+
769
+ throw problem;
770
+ }
771
+
772
+ /** Collect a streamed reply, handing deltas out as they land. */
773
+ async function streamed(request, opts, id) {
774
+ // A watchdog of its own, so a frozen stream can be ended without it looking
775
+ // like the user pressed stop.
776
+ const quiet = new AbortController();
777
+ const stop = () => quiet.abort();
778
+ opts.signal?.addEventListener('abort', stop, { once: true });
779
+ let stalled = false;
780
+ let timer;
781
+ const frozen = (cause) => new Failure({
782
+ kind: 'timeout',
783
+ attempted: `asking ${modelName(id)} for a reply`,
784
+ failed: `${modelName(id)} went silent for ${Math.round(stallLimit() / 1000)}s, so ucode stopped waiting.`,
785
+ fix: 'ucode asks again by itself. If it keeps freezing, /model to North Mini Code.',
786
+ detail: { stalled: true, handed: handed.size },
787
+ cause,
788
+ });
789
+ const alive = () => {
790
+ clearTimeout(timer);
791
+ timer = setTimeout(() => { stalled = true; quiet.abort(); }, stallLimit());
792
+ };
793
+
794
+ let text = '';
795
+ let reasoning = '';
796
+ let finishReason = 'stop';
797
+ let usage = null;
798
+ const partial = new Map();
799
+ const handed = new Set();
800
+ let highest = -1;
801
+ const names = request.tools?.map((t) => t.function.name);
802
+
803
+ try {
804
+ alive();
805
+ const { data: stream, response } = await connection().chat.completions
806
+ .create(
807
+ { ...request, stream: true, stream_options: { include_usage: true } },
808
+ { signal: quiet.signal }
809
+ )
810
+ .withResponse();
811
+ noteLimits(response?.headers);
812
+
813
+ for await (const chunk of stream) {
814
+ alive();
815
+ if (opts.signal?.aborted) break;
816
+ if (chunk.usage) usage = chunk.usage;
817
+
818
+ const choice = chunk.choices?.[0];
819
+ if (!choice) continue;
820
+ if (choice.finish_reason) finishReason = choice.finish_reason;
821
+ const delta = choice.delta ?? {};
822
+
823
+ // Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
824
+ // `reasoning_content` on some upstreams.
825
+ const thinking = delta.reasoning ?? delta.reasoning_content;
826
+ if (thinking) {
827
+ reasoning += thinking;
828
+ opts.onThinking?.(thinking);
829
+ }
830
+
831
+ if (delta.content) {
832
+ text += delta.content;
833
+ opts.onText(delta.content);
834
+ }
835
+
836
+ // A tool call's name and arguments arrive across several chunks, keyed by
837
+ // index, so they are stitched back together here.
838
+ for (const call of delta.tool_calls ?? []) {
839
+ // Calls arrive one after another, so the first chunk of call N means
840
+ // every call before it is complete. Those are handed over at once, and
841
+ // the caller can start running them while the rest are still being
842
+ // written — the reply streaming and the tools working overlap.
843
+ if (opts.onToolCall && call.index > highest) {
844
+ for (const [index, slot] of partial) {
845
+ if (index < call.index && !handed.has(index)) {
846
+ handed.add(index);
847
+ opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args, names }));
848
+ }
849
+ }
850
+ highest = call.index;
851
+ }
852
+
853
+ const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
854
+ if (call.id) slot.id = call.id;
855
+ if (call.function?.name) slot.name += call.function.name;
856
+ if (call.function?.arguments) slot.args += call.function.arguments;
857
+ partial.set(call.index, slot);
858
+
859
+ // A whole app arrives as one enormous arguments string that takes a
860
+ // minute or two to write. Handing it over as it grows is what lets the
861
+ // caller say which file is being written right now, instead of showing
862
+ // a spinner that has meant nothing for ninety seconds.
863
+ if (call.function?.arguments) opts.onToolArgs?.({ index: call.index, name: slot.name, args: slot.args });
864
+ }
865
+ }
866
+ } catch (err) {
867
+ if (!stalled || opts.signal?.aborted) throw err;
868
+ throw frozen(err);
869
+ } finally {
870
+ clearTimeout(timer);
871
+ opts.signal?.removeEventListener('abort', stop);
872
+ }
873
+ // The watchdog's abort can also end the stream quietly instead of throwing.
874
+ // Carrying on from there would run a half-written tool call: JSON repair
875
+ // closes a cut-off file write, and the truncated file lands on disk.
876
+ if (stalled && !opts.signal?.aborted) throw frozen();
877
+
878
+ const toolCalls = [];
879
+ for (const [index, slot] of partial) {
880
+ toolCalls.push(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args, cutOff: finishReason === 'length', names }));
881
+ }
882
+
883
+ return {
884
+ text: text.trim(),
885
+ reasoning: reasoning.trim(),
886
+ toolCalls,
887
+ usage: {
888
+ promptTokens: usage?.prompt_tokens ?? 0,
889
+ outputTokens: usage?.completion_tokens ?? 0,
890
+ totalTokens: usage?.total_tokens ?? 0,
891
+ },
892
+ finishReason,
893
+ model: id,
894
+ };
895
+ }
896
+
897
+ /**
898
+ * Recover the files from a file-write call whose JSON will not parse.
899
+ *
900
+ * The usual cause is a double quote inside the code that the model forgot to
901
+ * escape — `className="flex"` — which ends the JSON string early. No general
902
+ * repair can know which quote was meant, but a file write has a fixed shape:
903
+ * "path", then "content", then either the next file or the end. Splitting on
904
+ * that shape and escaping the stray quotes gets every file back.
905
+ */
906
+ /**
907
+ * Which calls can be rebuilt out of broken JSON.
908
+ *
909
+ * All three carry file contents — hundreds of lines of HTML, CSS and
910
+ * JavaScript escaped into a JSON string — which is the one argument shape a
911
+ * model gets wrong often enough to matter. Everything else is short enough
912
+ * that a parse failure is a real mistake, worth reporting rather than guessing
913
+ * around.
914
+ */
915
+ const SALVAGEABLE = new Set(['write_file', 'batch_write', 'create_app']);
916
+
917
+ /**
918
+ * The plain string arguments sitting beside a files array, read off the raw
919
+ * text when the object as a whole will not parse.
920
+ *
921
+ * Only the keys create_app needs, and only from before the files begin, so a
922
+ * "name" belonging to something nested inside a file cannot be mistaken for
923
+ * the app's own.
924
+ */
925
+ function scalarArgs(text) {
926
+ const at = text.search(/"files"\s*:/);
927
+ const head = at > 0 ? text.slice(0, at) : text;
928
+ const out = {};
929
+ for (const key of ['folder', 'name', 'description', 'template', 'design']) {
930
+ const hit = new RegExp(`"${key}"\\s*:\\s*"((?:[^"\\\\]|\\\\.)*)"`).exec(head);
931
+ if (hit) out[key] = hit[1].replace(/\\(["\\/])/g, '$1');
932
+ }
933
+ return out;
934
+ }
935
+
936
+ /**
937
+ * Rebuild the edits out of a broken edit_file or edit_files call.
938
+ *
939
+ * The same problem as a broken write, one level down: an edit carries two
940
+ * slabs of somebody's source as escaped JSON strings, and one stray character
941
+ * loses the call. edit_files failed this way three times in a row in a traced
942
+ * build, at the end of a turn, which is a turn that ends having undone nothing
943
+ * and fixed nothing.
944
+ *
945
+ * Safer than it sounds. Every edit is applied by replaceOnce, which requires
946
+ * old_string to appear exactly once and refuses otherwise — so an edit
947
+ * recovered wrongly does not corrupt a file, it fails to match and says so.
948
+ * The risk of guessing is a clear error; the cost of not guessing is the whole
949
+ * call. Truncated replies are still refused upstream, where half a string
950
+ * would mean half a file.
951
+ *
952
+ * Each edit is attributed to the nearest "path" before it, which is how
953
+ * edit_files nests them; edit_file has exactly one of each.
954
+ */
955
+ export function salvageEdits(text) {
956
+ const heads = [...text.matchAll(/"old_string"\s*:\s*"/g)];
957
+ if (!heads.length) return null;
958
+
959
+ const paths = [...text.matchAll(/"path"\s*:\s*"((?:[^"\\]|\\.)*)"/g)];
960
+ const NEW = '"new_string"';
961
+ const out = [];
962
+
963
+ for (let i = 0; i < heads.length; i++) {
964
+ const oldFrom = heads[i].index + heads[i][0].length;
965
+ const newKey = text.indexOf(NEW, oldFrom);
966
+ if (newKey < 0) return null;
967
+
968
+ const opens = /^\s*:\s*"/.exec(text.slice(newKey + NEW.length));
969
+ if (!opens) return null;
970
+ const newFrom = newKey + NEW.length + opens[0].length;
971
+
972
+ // The value runs until whatever structure comes next — the following edit,
973
+ // or the following file — and the trailing JSON punctuation comes off.
974
+ const nextEdit = i + 1 < heads.length ? heads[i + 1].index : text.length;
975
+ const nextPath = paths.find((m) => m.index > newFrom)?.index ?? text.length;
976
+ const strip = (v) => v.replace(/"[\s,{}[\]]*$/, '');
977
+
978
+ const old_string = unescapeLoose(strip(text.slice(oldFrom, newKey)));
979
+ const new_string = unescapeLoose(strip(text.slice(newFrom, Math.min(nextEdit, nextPath))));
980
+ const owner = paths.filter((m) => m.index < heads[i].index).pop();
981
+
982
+ if (!owner || !old_string) return null;
983
+ out.push({ path: unescapeLoose(owner[1]), old_string, new_string });
984
+ }
985
+
986
+ return out.length ? out : null;
987
+ }
988
+
989
+ /** The same edits, grouped under their file, which is edit_files' own shape. */
990
+ export function groupEdits(edits) {
991
+ const byPath = new Map();
992
+ for (const { path, old_string, new_string } of edits) {
993
+ if (!byPath.has(path)) byPath.set(path, []);
994
+ byPath.get(path).push({ old_string, new_string });
995
+ }
996
+ return [...byPath].map(([path, list]) => ({ path, edits: list }));
997
+ }
998
+
999
+ export function salvageWrites(text) {
1000
+ const heads = [...text.matchAll(/"path"\s*:\s*"((?:[^"\\]|\\.)*)"\s*,\s*"content"\s*:\s*"/g)];
1001
+ if (!heads.length) return null;
1002
+
1003
+ const files = [];
1004
+ for (let i = 0; i < heads.length; i++) {
1005
+ const from = heads[i].index + heads[i][0].length;
1006
+ const to = i + 1 < heads.length ? heads[i + 1].index : text.length;
1007
+ // The string ends at the last quote that is followed by nothing but JSON
1008
+ // punctuation — `"}, {`, or `"}}, {` when the model added a brace, or `"}]}`.
1009
+ const body = text.slice(from, to).replace(/"[\s,{}[\]]*$/, '');
1010
+ const content = unescapeLoose(body);
1011
+ const pathValue = unescapeLoose(heads[i][1]);
1012
+ if (!pathValue || !content) return null; // not the shape we thought — leave it an honest error
1013
+ files.push({ path: pathValue, content });
1014
+ }
1015
+ return files.length ? files : null;
1016
+ }
1017
+
1018
+ /**
1019
+ * Decode a JSON string body the forgiving way: the standard escapes are
1020
+ * honoured, and everything JSON would reject — a raw line break, a tab, a stray
1021
+ * quote, an escape JSON does not know — is kept as the character it plainly is.
1022
+ */
1023
+ function unescapeLoose(s) {
1024
+ const simple = { n: '\n', t: '\t', r: '\r', b: '\b', f: '\f', '"': '"', '\\': '\\', '/': '/' };
1025
+ let out = '';
1026
+ for (let i = 0; i < s.length; i++) {
1027
+ const c = s[i];
1028
+ if (c !== '\\' || i === s.length - 1) { out += c; continue; }
1029
+ const next = s[++i];
1030
+ if (next === 'u' && /^[0-9a-fA-F]{4}$/.test(s.slice(i + 1, i + 5))) {
1031
+ out += String.fromCharCode(parseInt(s.slice(i + 1, i + 5), 16));
1032
+ i += 4;
1033
+ } else {
1034
+ out += simple[next] ?? next;
1035
+ }
1036
+ }
1037
+ return out;
1038
+ }
1039
+
1040
+ /**
1041
+ * A tool named with the wrong case or separators — "Read_File", "readFile",
1042
+ * "functions.read_file" — is the tool it plainly means. opencode repairs the
1043
+ * case the same way; refusing costs a whole round trip to learn a spelling.
1044
+ */
1045
+ export function fixName(name, names) {
1046
+ if (!names?.length || typeof name !== 'string' || names.includes(name)) return name;
1047
+ const squash = (s) => s.replace(/^(?:functions|tools?)[.:]/i, '').toLowerCase().replace(/[^a-z0-9]/g, '');
1048
+ return names.find((n) => squash(n) === squash(name)) ?? name;
1049
+ }
1050
+
1051
+ /**
1052
+ * Parse one tool call's arguments.
1053
+ *
1054
+ * A parse error is recorded on the call rather than thrown. The loop hands it
1055
+ * back to the model, which usually fixes its own JSON on the next step —
1056
+ * cheaper than failing the whole turn over a stray comma.
1057
+ */
1058
+ export function readCall({ id, name, raw, cutOff = false, names }) {
1059
+ const call = { id, name: fixName(name, names), args: {} };
1060
+ name = call.name;
1061
+ const text = String(raw ?? '').trim();
1062
+ if (!text) return call;
1063
+ try {
1064
+ const parsed = JSON.parse(text);
1065
+ if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) call.args = parsed;
1066
+ else call.parseError = `arguments must be a JSON object, got ${Array.isArray(parsed) ? 'array' : typeof parsed}`;
1067
+ } catch (err) {
1068
+ // A missing comma or a stray control character in a 3,000-token
1069
+ // batch_write used to throw the whole step away — half a minute of output
1070
+ // discarded over one character. Repair the usual slips instead. Never when
1071
+ // the reply was cut off at the output limit, though: repairing that would
1072
+ // close the string and quietly write half a file.
1073
+ if (!cutOff) {
1074
+ try {
1075
+ const fixed = JSON.parse(jsonrepair(text));
1076
+ if (fixed && typeof fixed === 'object' && !Array.isArray(fixed)) {
1077
+ call.args = fixed;
1078
+ call.repaired = true;
1079
+ return call;
1080
+ }
1081
+ } catch { /* beyond general repair — try the file-write shape next */ }
1082
+
1083
+ if (name === 'edit_file' || name === 'edit_files' || name === 'multi_edit') {
1084
+ const edits = salvageEdits(text);
1085
+ if (edits) {
1086
+ if (name === 'edit_file') call.args = edits[0];
1087
+ else if (name === 'multi_edit') {
1088
+ call.args = { path: edits[0].path, edits: edits.map(({ old_string, new_string }) => ({ old_string, new_string })) };
1089
+ } else call.args = { files: groupEdits(edits) };
1090
+ call.repaired = true;
1091
+ return call;
1092
+ }
1093
+ }
1094
+
1095
+ const salvaged = SALVAGEABLE.has(name) ? salvageWrites(text) : null;
1096
+ if (salvaged) {
1097
+ if (name === 'write_file') call.args = salvaged[0];
1098
+ // create_app carries the app in the same { path, content } shape, with
1099
+ // a few plain strings beside it. Losing the entire call because one of
1100
+ // several hundred lines of HTML held a raw newline is how a build ends
1101
+ // with no files at all — which is what it did: refused three times,
1102
+ // the folder never created, the turn over in sixty-six seconds having
1103
+ // produced nothing. The scalars are read back off the same text.
1104
+ else if (name === 'create_app') call.args = { ...scalarArgs(text), files: salvaged };
1105
+ else call.args = { files: salvaged };
1106
+ call.repaired = true;
1107
+ return call;
1108
+ }
1109
+ }
1110
+ call.parseError = `${err.message} — the raw arguments were: ${text.slice(0, 300)}`;
1111
+ }
1112
+ return call;
1113
+ }
1114
+
1115
+ function normalize(data, id, names) {
1116
+ const choice = data?.choices?.[0];
1117
+ const message = choice?.message ?? {};
1118
+
1119
+ const toolCalls = (message.tool_calls ?? []).map((c) =>
1120
+ readCall({ id: c.id, name: c.function?.name, raw: c.function?.arguments, cutOff: choice?.finish_reason === 'length', names })
1121
+ );
1122
+
1123
+ const u = data?.usage ?? {};
1124
+ const finishReason = choice?.finish_reason ?? 'stop';
1125
+ const text = (message.content ?? '').trim();
1126
+
1127
+ if (!text && toolCalls.length === 0 && finishReason === 'length') {
1128
+ throw new Failure({
1129
+ kind: 'no_content',
1130
+ attempted: `asking ${modelName(id)} for a reply`,
1131
+ failed: 'The reply hit the output limit before producing anything at all.',
1132
+ fix: 'Ask for something shorter, or split the task into steps.',
1133
+ detail: { finishReason },
1134
+ });
1135
+ }
1136
+
1137
+ return {
1138
+ text,
1139
+ reasoning: (message.reasoning ?? '').trim(),
1140
+ toolCalls,
1141
+ usage: {
1142
+ promptTokens: u.prompt_tokens ?? 0,
1143
+ outputTokens: u.completion_tokens ?? 0,
1144
+ totalTokens: u.total_tokens ?? 0,
1145
+ },
1146
+ finishReason,
1147
+ model: id,
1148
+ };
1149
+ }