ucode-agent 1.47.0 → 1.48.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -416,6 +416,7 @@ ucode [options]
416
416
  Environment overrides: `UCODE_MODEL`, `UCODE_WORKER_MODEL` (a faster model for
417
417
  parallel workers), `UCODE_WORKER_STEPS`, `UCODE_MAX_CONTEXT_TOKENS`,
418
418
  `UCODE_MAX_STEPS`, `UCODE_MAX_TOOL_OUTPUT`, `UCODE_REQUEST_TIMEOUT_MS`,
419
+ `UCODE_STALL_MS` (how long a silent reply is waited on before asking again, 60s),
419
420
  `UCODE_BASE_URL`, `UCODE_NO_UPDATE`.
420
421
 
421
422
  Web search needs a Tavily key — free, 1000 searches a month, no card. Without
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ucode-agent",
3
- "version": "1.47.0",
3
+ "version": "1.48.0",
4
4
  "description": "ucode - a terminal coding agent that reads, edits and runs your code, on NVIDIA, Cohere and Nex AGI models.",
5
5
  "type": "module",
6
6
  "main": "ucode.js",
package/src/core/loop.js CHANGED
@@ -1952,11 +1952,19 @@ export class Agent {
1952
1952
  const from = model();
1953
1953
  let next = fallbackFor(from, this.tried);
1954
1954
 
1955
+ const why = err.kind === 'rate_limit' ? 'busy' : err.kind === 'timeout' ? 'too slow to answer' : 'not answering';
1956
+
1955
1957
  if (!next) {
1956
- const until = Date.now() + 60_000;
1957
- this.ui.startSpinner('every model is busy');
1958
+ // With fallback off this is the ordinary path, so the wait names the one
1959
+ // model being waited on. A rate limit needs the full minute to clear; a
1960
+ // timeout has already cost minutes of silence, and waiting longer before
1961
+ // asking again buys nothing.
1962
+ const alone = process.env.UCODE_FALLBACK !== '1';
1963
+ const who = alone ? modelName(from) : 'every model';
1964
+ const until = Date.now() + (err.kind === 'rate_limit' || !alone ? 60_000 : 5_000);
1965
+ this.ui.startSpinner(`${who} is ${why}`);
1958
1966
  while (Date.now() < until && !this.abort?.signal.aborted) {
1959
- this.ui.updateSpinner(`every model is busy — trying again in ${Math.ceil((until - Date.now()) / 1000)}s`);
1967
+ this.ui.updateSpinner(`${who} is ${why} — trying again in ${Math.ceil((until - Date.now()) / 1000)}s`);
1960
1968
  await wait(1000);
1961
1969
  }
1962
1970
  this.ui.stopSpinner();
@@ -1968,8 +1976,9 @@ export class Agent {
1968
1976
  this.tried.add(next);
1969
1977
  setModel(next);
1970
1978
  this.cooldownUntil = Date.now() + COOLDOWN;
1971
- const why = err.kind === 'rate_limit' ? 'busy' : err.kind === 'timeout' ? 'too slow to answer' : 'not answering';
1972
- this.ui.note(`${modelName(from)} is ${why} — carrying on with ${modelName(next)}`);
1979
+ this.ui.note(next === from
1980
+ ? `${modelName(from)} was ${why} — asking it again`
1981
+ : `${modelName(from)} is ${why} — carrying on with ${modelName(next)}`);
1973
1982
  if (this.full) this.showHeader({ clear: false });
1974
1983
  return true;
1975
1984
  }
@@ -137,6 +137,21 @@ export function fallbackFor(id, tried = new Set()) {
137
137
  /** Seconds to wait on successive rate limits that come with no retry-after. */
138
138
  const RATE_LIMIT_BACKOFF = [5, 10, 20];
139
139
 
140
+ /**
141
+ * How long a stream may go without a single chunk before it counts as frozen.
142
+ *
143
+ * A free endpoint can accept a request and then send nothing at all, and the
144
+ * only thing that used to end that was the five-minute request timeout — five
145
+ * minutes of a spinner, then the same again on the retry. Reasoning streams
146
+ * as it is produced, so even the slowest thinker sends something well inside
147
+ * a minute; silence for that long means nobody is working on the reply.
148
+ */
149
+ export const stallLimit = () => Number(process.env.UCODE_STALL_MS) || 60_000;
150
+
151
+ /** Freezes in a row before ucode stops asking and says to switch models. */
152
+ export const MAX_STALLS = 3;
153
+ let stalls = 0;
154
+
140
155
  let current = process.env.UCODE_MODEL || DEFAULT_MODEL;
141
156
  let client = null;
142
157
 
@@ -622,16 +637,34 @@ export async function ask(messages, tools = [], opts = {}) {
622
637
 
623
638
  for (let attempt = 1; attempt <= attempts; attempt++) {
624
639
  try {
625
- if (opts.onText) return await streamed(request, callOpts, id);
640
+ if (opts.onText) {
641
+ const reply = await streamed(request, callOpts, id);
642
+ stalls = 0;
643
+ return reply;
644
+ }
626
645
  const { data, response } = await connection().chat.completions
627
646
  .create(request, { signal: opts.signal })
628
647
  .withResponse();
629
648
  noteLimits(response?.headers);
649
+ stalls = 0;
630
650
  return normalize(data, id);
631
651
  } catch (err) {
632
652
  noteLimits(err?.headers);
633
653
  problem = explain(err, id);
634
654
 
655
+ // Retrying a model that keeps freezing only repeats the wait. After a
656
+ // few in a row, say so and hand the choice back.
657
+ if (problem.detail?.stalled && ++stalls >= MAX_STALLS) {
658
+ stalls = 0;
659
+ throw new Failure({
660
+ kind: 'stalled',
661
+ attempted: `asking ${modelName(id)} for a reply`,
662
+ failed: `${modelName(id)} froze ${MAX_STALLS} times in a row — it took the request and then sent nothing.`,
663
+ fix: 'Its free endpoint is struggling right now. Run /model and pick another one; North Mini Code answers soonest.',
664
+ cause: problem,
665
+ });
666
+ }
667
+
635
668
  // A per-minute limit is a wait, not a failure. Sit it out rather than
636
669
  // making the user retype their message. Free endpoints often refuse
637
670
  // without saying how long to wait, so when there is no retry-after the
@@ -656,6 +689,7 @@ export async function ask(messages, tools = [], opts = {}) {
656
689
  problem.kind === 'server' || problem.kind === 'network' || problem.kind === 'timeout';
657
690
  if (!worthRetrying || attempt === attempts || opts.signal?.aborted) break;
658
691
  if (printed > 0) break; // half an answer is on screen; do not print it twice
692
+ if (problem.detail?.handed) break; // its first tool calls are already running
659
693
 
660
694
  // A stalled provider needs longer to come back than a dropped socket
661
695
  // does, and the wait is narrated so a slow turn never looks like a hang.
@@ -675,13 +709,17 @@ export async function ask(messages, tools = [], opts = {}) {
675
709
 
676
710
  /** Collect a streamed reply, handing deltas out as they land. */
677
711
  async function streamed(request, opts, id) {
678
- const { data: stream, response } = await connection().chat.completions
679
- .create(
680
- { ...request, stream: true, stream_options: { include_usage: true } },
681
- { signal: opts.signal }
682
- )
683
- .withResponse();
684
- noteLimits(response?.headers);
712
+ // A watchdog of its own, so a frozen stream can be ended without it looking
713
+ // like the user pressed stop.
714
+ const quiet = new AbortController();
715
+ const stop = () => quiet.abort();
716
+ opts.signal?.addEventListener('abort', stop, { once: true });
717
+ let stalled = false;
718
+ let timer;
719
+ const alive = () => {
720
+ clearTimeout(timer);
721
+ timer = setTimeout(() => { stalled = true; quiet.abort(); }, stallLimit());
722
+ };
685
723
 
686
724
  let text = '';
687
725
  let reasoning = '';
@@ -691,57 +729,82 @@ async function streamed(request, opts, id) {
691
729
  const handed = new Set();
692
730
  let highest = -1;
693
731
 
694
- for await (const chunk of stream) {
695
- if (opts.signal?.aborted) break;
696
- if (chunk.usage) usage = chunk.usage;
697
-
698
- const choice = chunk.choices?.[0];
699
- if (!choice) continue;
700
- if (choice.finish_reason) finishReason = choice.finish_reason;
701
- const delta = choice.delta ?? {};
702
-
703
- // Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
704
- // `reasoning_content` on some upstreams.
705
- const thinking = delta.reasoning ?? delta.reasoning_content;
706
- if (thinking) {
707
- reasoning += thinking;
708
- opts.onThinking?.(thinking);
709
- }
732
+ try {
733
+ alive();
734
+ const { data: stream, response } = await connection().chat.completions
735
+ .create(
736
+ { ...request, stream: true, stream_options: { include_usage: true } },
737
+ { signal: quiet.signal }
738
+ )
739
+ .withResponse();
740
+ noteLimits(response?.headers);
741
+
742
+ for await (const chunk of stream) {
743
+ alive();
744
+ if (opts.signal?.aborted) break;
745
+ if (chunk.usage) usage = chunk.usage;
746
+
747
+ const choice = chunk.choices?.[0];
748
+ if (!choice) continue;
749
+ if (choice.finish_reason) finishReason = choice.finish_reason;
750
+ const delta = choice.delta ?? {};
751
+
752
+ // Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
753
+ // `reasoning_content` on some upstreams.
754
+ const thinking = delta.reasoning ?? delta.reasoning_content;
755
+ if (thinking) {
756
+ reasoning += thinking;
757
+ opts.onThinking?.(thinking);
758
+ }
710
759
 
711
- if (delta.content) {
712
- text += delta.content;
713
- opts.onText(delta.content);
714
- }
760
+ if (delta.content) {
761
+ text += delta.content;
762
+ opts.onText(delta.content);
763
+ }
715
764
 
716
- // A tool call's name and arguments arrive across several chunks, keyed by
717
- // index, so they are stitched back together here.
718
- for (const call of delta.tool_calls ?? []) {
719
- // Calls arrive one after another, so the first chunk of call N means
720
- // every call before it is complete. Those are handed over at once, and
721
- // the caller can start running them while the rest are still being
722
- // written — the reply streaming and the tools working overlap.
723
- if (opts.onToolCall && call.index > highest) {
724
- for (const [index, slot] of partial) {
725
- if (index < call.index && !handed.has(index)) {
726
- handed.add(index);
727
- opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args }));
765
+ // A tool call's name and arguments arrive across several chunks, keyed by
766
+ // index, so they are stitched back together here.
767
+ for (const call of delta.tool_calls ?? []) {
768
+ // Calls arrive one after another, so the first chunk of call N means
769
+ // every call before it is complete. Those are handed over at once, and
770
+ // the caller can start running them while the rest are still being
771
+ // written — the reply streaming and the tools working overlap.
772
+ if (opts.onToolCall && call.index > highest) {
773
+ for (const [index, slot] of partial) {
774
+ if (index < call.index && !handed.has(index)) {
775
+ handed.add(index);
776
+ opts.onToolCall(readCall({ id: slot.id || `call_${index}`, name: slot.name, raw: slot.args }));
777
+ }
728
778
  }
779
+ highest = call.index;
729
780
  }
730
- highest = call.index;
731
- }
732
781
 
733
- const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
734
- if (call.id) slot.id = call.id;
735
- if (call.function?.name) slot.name += call.function.name;
736
- if (call.function?.arguments) slot.args += call.function.arguments;
737
- partial.set(call.index, slot);
738
-
739
- // A whole app arrives as one enormous arguments string that takes a
740
- // minute or two to write. Handing it over as it grows is what lets the
741
- // caller say which file is being written right now, instead of showing
742
- // a spinner that has meant nothing for ninety seconds.
743
- if (call.function?.arguments) opts.onToolArgs?.({ index: call.index, name: slot.name, args: slot.args });
782
+ const slot = partial.get(call.index) ?? { id: '', name: '', args: '' };
783
+ if (call.id) slot.id = call.id;
784
+ if (call.function?.name) slot.name += call.function.name;
785
+ if (call.function?.arguments) slot.args += call.function.arguments;
786
+ partial.set(call.index, slot);
787
+
788
+ // A whole app arrives as one enormous arguments string that takes a
789
+ // minute or two to write. Handing it over as it grows is what lets the
790
+ // caller say which file is being written right now, instead of showing
791
+ // a spinner that has meant nothing for ninety seconds.
792
+ if (call.function?.arguments) opts.onToolArgs?.({ index: call.index, name: slot.name, args: slot.args });
793
+ }
744
794
  }
795
+ } catch (err) {
796
+ if (!stalled || opts.signal?.aborted) throw err;
797
+ throw new Failure({
798
+ kind: 'timeout',
799
+ attempted: `asking ${modelName(id)} for a reply`,
800
+ failed: `${modelName(id)} went silent for ${Math.round(stallLimit() / 1000)}s, so ucode stopped waiting.`,
801
+ fix: 'ucode asks again by itself. If it keeps freezing, /model to North Mini Code.',
802
+ detail: { stalled: true, handed: handed.size },
803
+ cause: err,
804
+ });
805
+ } finally {
806
+ clearTimeout(timer);
807
+ opts.signal?.removeEventListener('abort', stop);
745
808
  }
746
809
 
747
810
  const toolCalls = [];