ucode-agent 1.60.1 → 1.60.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core/provider.js +21 -4
package/package.json
CHANGED
package/src/core/provider.js
CHANGED
|
@@ -118,6 +118,18 @@ const RATE_LIMIT_BACKOFF = [5, 10, 20];
|
|
|
118
118
|
*/
|
|
119
119
|
export const stallLimit = () => Number(process.env.UCODE_STALL_MS) || 60_000;
|
|
120
120
|
|
|
121
|
+
/**
|
|
122
|
+
* How long to wait for the first piece of a reply.
|
|
123
|
+
*
|
|
124
|
+
* Gemini sends a tool call whole, at the end, not piece by piece. A whole app
|
|
125
|
+
* in one create_app is 50-70 silent seconds before anything arrives, and the
|
|
126
|
+
* one-minute watchdog was cancelling those replies as frozen and starting them
|
|
127
|
+
* again — "provider stalled" on requests that were working fine. Silence before
|
|
128
|
+
* the reply starts gets this long; once it is arriving, stallLimit applies.
|
|
129
|
+
*/
|
|
130
|
+
export const firstReplyLimit = () =>
|
|
131
|
+
Number(process.env.UCODE_FIRST_REPLY_MS) || Number(process.env.UCODE_STALL_MS) || 180_000;
|
|
132
|
+
|
|
121
133
|
/** Freezes in a row before ucode stops asking and says to switch models. */
|
|
122
134
|
export const MAX_STALLS = 3;
|
|
123
135
|
let stalls = 0;
|
|
@@ -755,17 +767,21 @@ async function streamed(request, opts, id) {
|
|
|
755
767
|
opts.signal?.addEventListener('abort', stop, { once: true });
|
|
756
768
|
let stalled = false;
|
|
757
769
|
let timer;
|
|
770
|
+
// Nothing of the reply itself yet: text, reasoning or a tool call. Chunks
|
|
771
|
+
// that carry none of those (a role, usage) do not count as it having started.
|
|
772
|
+
let started = false;
|
|
773
|
+
const limit = () => (started ? stallLimit() : firstReplyLimit());
|
|
758
774
|
const frozen = (cause) => new Failure({
|
|
759
775
|
kind: 'timeout',
|
|
760
776
|
attempted: `asking ${modelName(id)} for a reply`,
|
|
761
|
-
failed: `${modelName(id)} went silent for ${Math.round(
|
|
777
|
+
failed: `${modelName(id)} went silent for ${Math.round(limit() / 1000)}s, so ucode stopped waiting.`,
|
|
762
778
|
fix: 'ucode asks again by itself. If it keeps freezing, /model to North Mini Code.',
|
|
763
779
|
detail: { stalled: true, handed: handed.size },
|
|
764
780
|
cause,
|
|
765
781
|
});
|
|
766
782
|
const alive = () => {
|
|
767
783
|
clearTimeout(timer);
|
|
768
|
-
timer = setTimeout(() => { stalled = true; quiet.abort(); },
|
|
784
|
+
timer = setTimeout(() => { stalled = true; quiet.abort(); }, limit());
|
|
769
785
|
};
|
|
770
786
|
|
|
771
787
|
let text = '';
|
|
@@ -788,14 +804,15 @@ async function streamed(request, opts, id) {
|
|
|
788
804
|
noteLimits(response?.headers);
|
|
789
805
|
|
|
790
806
|
for await (const chunk of stream) {
|
|
791
|
-
alive();
|
|
792
807
|
if (opts.signal?.aborted) break;
|
|
793
808
|
if (chunk.usage) usage = chunk.usage;
|
|
794
809
|
|
|
795
810
|
const choice = chunk.choices?.[0];
|
|
811
|
+
const delta = choice?.delta ?? {};
|
|
812
|
+
if (delta.content || delta.reasoning || delta.reasoning_content || delta.tool_calls?.length) started = true;
|
|
813
|
+
alive();
|
|
796
814
|
if (!choice) continue;
|
|
797
815
|
if (choice.finish_reason) finishReason = choice.finish_reason;
|
|
798
|
-
const delta = choice.delta ?? {};
|
|
799
816
|
|
|
800
817
|
// Reasoning arrives on a separate channel: `reasoning` on OpenRouter,
|
|
801
818
|
// `reasoning_content` on some upstreams.
|