@yeaft/webchat-agent 0.1.1023 → 0.1.1025
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/cli.js +118 -1
- package/package.json +1 -1
- package/yeaft/config.js +56 -0
- package/yeaft/engine.js +144 -2
- package/yeaft/llm/adapter.js +106 -0
- package/yeaft/llm/anthropic.js +33 -17
- package/yeaft/llm/openai-responses.js +11 -7
- package/yeaft/models.js +21 -0
- package/yeaft/web-bridge.js +17 -0
package/cli.js
CHANGED
|
@@ -24,6 +24,11 @@ import {
|
|
|
24
24
|
useOpenAICompatible,
|
|
25
25
|
writeLocalLlmConfig,
|
|
26
26
|
} from './llm-config-cli.js';
|
|
27
|
+
import {
|
|
28
|
+
discoverGitHubCopilotModels,
|
|
29
|
+
discoverOpenAICompatibleModels,
|
|
30
|
+
GITHUB_COPILOT_PROVIDER,
|
|
31
|
+
} from './llm-model-discovery.js';
|
|
27
32
|
|
|
28
33
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
29
34
|
const pkg = JSON.parse(readFileSync(join(__dirname, 'package.json'), 'utf-8'));
|
|
@@ -97,6 +102,7 @@ function printLlmHelp() {
|
|
|
97
102
|
|
|
98
103
|
Usage:
|
|
99
104
|
yeaft-agent llm show [--reveal]
|
|
105
|
+
yeaft-agent llm list-models [<provider-name>]
|
|
100
106
|
yeaft-agent llm setup
|
|
101
107
|
yeaft-agent llm use github-copilot --model <modelId> [--fast <modelId>] [--allow-unknown-model]
|
|
102
108
|
yeaft-agent llm use openai-compatible --name <name> --base-url <url> --api-key-env <ENV> --model <modelId> [--fast <modelId>]
|
|
@@ -112,6 +118,8 @@ function printLlmHelp() {
|
|
|
112
118
|
add-provider updates/replaces an existing provider with the same --name.
|
|
113
119
|
--api-key-env reads the environment variable value and writes it as apiKey.
|
|
114
120
|
set-model requires full provider/model references.
|
|
121
|
+
list-models with no provider lists the local config offline; with 'github-copilot'
|
|
122
|
+
or a configured provider name, it queries the live '/models' catalog.
|
|
115
123
|
--config <path> can target a config file for tests or scripted setup.
|
|
116
124
|
|
|
117
125
|
Examples:
|
|
@@ -127,7 +135,14 @@ async function handleLlmCommand(args) {
|
|
|
127
135
|
const subcommand = args[0];
|
|
128
136
|
|
|
129
137
|
try {
|
|
130
|
-
|
|
138
|
+
// `use <preset>` and `list-models <provider-name>` both put a positional
|
|
139
|
+
// arg right after the subcommand; trim it off before flag parsing so
|
|
140
|
+
// parseLlmArgs sees only flags.
|
|
141
|
+
const positionalAfterSub =
|
|
142
|
+
subcommand === 'use' ? 2
|
|
143
|
+
: (subcommand === 'list-models' && args[1] && !args[1].startsWith('--')) ? 2
|
|
144
|
+
: 1;
|
|
145
|
+
const options = parseLlmArgs(args.slice(positionalAfterSub));
|
|
131
146
|
const configPath = options.config || getDefaultYeaftConfigPath();
|
|
132
147
|
if (!subcommand || subcommand === '--help' || subcommand === '-h' || subcommand === 'help') {
|
|
133
148
|
printLlmHelp();
|
|
@@ -140,6 +155,19 @@ async function handleLlmCommand(args) {
|
|
|
140
155
|
return;
|
|
141
156
|
}
|
|
142
157
|
|
|
158
|
+
if (subcommand === 'list-models') {
|
|
159
|
+
// `yeaft-agent llm list-models` (no provider) — list models declared in
|
|
160
|
+
// the local config (offline; no network call).
|
|
161
|
+
// `yeaft-agent llm list-models <provider-name>` — live-discover models:
|
|
162
|
+
// - "github-copilot" uses the local Copilot credential
|
|
163
|
+
// - any other name must already exist in config.json (uses its
|
|
164
|
+
// baseUrl + apiKey for OpenAI-compatible /models discovery)
|
|
165
|
+
const providerName = (args[1] && !args[1].startsWith('--')) ? args[1] : null;
|
|
166
|
+
const config = readLocalLlmConfig(configPath);
|
|
167
|
+
await handleListModels(config, { providerName });
|
|
168
|
+
return;
|
|
169
|
+
}
|
|
170
|
+
|
|
143
171
|
const current = readLocalLlmConfig(configPath);
|
|
144
172
|
let result;
|
|
145
173
|
if (subcommand === 'setup') {
|
|
@@ -204,6 +232,95 @@ async function handleLlmCommand(args) {
|
|
|
204
232
|
}
|
|
205
233
|
}
|
|
206
234
|
|
|
235
|
+
/**
|
|
236
|
+
* `yeaft-agent llm list-models [<provider-name>]` handler.
|
|
237
|
+
*
|
|
238
|
+
* Three modes:
|
|
239
|
+
* - No provider — list all models declared in the local config (offline,
|
|
240
|
+
* no network call). Annotates `← primary` / `← fast` for clarity.
|
|
241
|
+
* - "github-copilot" — live-discover Copilot's model catalog using the
|
|
242
|
+
* local device credential. Missing/invalid credential prints an
|
|
243
|
+
* actionable hint ("Run `gh auth login` ...") and exits non-zero so
|
|
244
|
+
* scripts can detect the failure.
|
|
245
|
+
* - Any other name — must already exist in config.json; uses its
|
|
246
|
+
* baseUrl + apiKey for OpenAI-compatible `/models` discovery.
|
|
247
|
+
*/
|
|
248
|
+
export async function handleListModels(
|
|
249
|
+
config,
|
|
250
|
+
{ providerName = null, deps = {} } = {}
|
|
251
|
+
) {
|
|
252
|
+
const discoverCopilot = deps.discoverCopilot || discoverGitHubCopilotModels;
|
|
253
|
+
const discoverOpenAI = deps.discoverOpenAI || discoverOpenAICompatibleModels;
|
|
254
|
+
|
|
255
|
+
if (providerName === GITHUB_COPILOT_PROVIDER.name) {
|
|
256
|
+
try {
|
|
257
|
+
const result = await discoverCopilot();
|
|
258
|
+
console.log(`Available models from GitHub Copilot (source: ${result.source}):`);
|
|
259
|
+
for (const id of result.models) console.log(` ${id}`);
|
|
260
|
+
if (result.warning) console.log(`\nNote: ${result.warning}`);
|
|
261
|
+
return;
|
|
262
|
+
} catch (err) {
|
|
263
|
+
console.error(`GitHub Copilot model discovery failed: ${err.message}`);
|
|
264
|
+
if (err.code === 'COPILOT_CREDENTIAL_MISSING' || err.code === 'COPILOT_AUTH_INVALID') {
|
|
265
|
+
console.error('Tip: run `gh auth login` (or complete the Copilot device login) and re-run this command.');
|
|
266
|
+
}
|
|
267
|
+
process.exitCode = 1;
|
|
268
|
+
return;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
if (providerName) {
|
|
273
|
+
const providers = Array.isArray(config.providers) ? config.providers : [];
|
|
274
|
+
const target = providers.find(p => p && p.name === providerName);
|
|
275
|
+
if (!target) {
|
|
276
|
+
console.error(`Provider "${providerName}" not found in config.json.`);
|
|
277
|
+
if (providers.length === 0) {
|
|
278
|
+
console.error('No providers are configured. Run `yeaft-agent llm setup` or `yeaft-agent llm use github-copilot ...`.');
|
|
279
|
+
} else {
|
|
280
|
+
console.error('Configured providers:');
|
|
281
|
+
for (const p of providers) console.error(` ${p.name}`);
|
|
282
|
+
}
|
|
283
|
+
process.exitCode = 1;
|
|
284
|
+
return;
|
|
285
|
+
}
|
|
286
|
+
try {
|
|
287
|
+
const result = await discoverOpenAI({
|
|
288
|
+
baseUrl: target.baseUrl,
|
|
289
|
+
apiKey: target.apiKey,
|
|
290
|
+
});
|
|
291
|
+
console.log(`Available models from "${providerName}" (${target.baseUrl}, source: ${result.source}):`);
|
|
292
|
+
for (const id of result.models) console.log(` ${providerName}/${id}`);
|
|
293
|
+
return;
|
|
294
|
+
} catch (err) {
|
|
295
|
+
console.error(`Model discovery for "${providerName}" failed: ${err.message}`);
|
|
296
|
+
process.exitCode = 1;
|
|
297
|
+
return;
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// Default: list configured providers' declared models (no network call).
|
|
302
|
+
const providers = Array.isArray(config.providers) ? config.providers : [];
|
|
303
|
+
if (providers.length === 0) {
|
|
304
|
+
console.log('No providers configured in config.json.');
|
|
305
|
+
console.log('Run `yeaft-agent llm setup`, or `yeaft-agent llm list-models github-copilot` to discover the Copilot catalog.');
|
|
306
|
+
return;
|
|
307
|
+
}
|
|
308
|
+
console.log('Configured models:');
|
|
309
|
+
for (const provider of providers) {
|
|
310
|
+
const tag = provider.managed || provider.credentialProvider ? ' (managed)' : '';
|
|
311
|
+
console.log(` [${provider.name}]${tag} ${provider.baseUrl || ''}`.trimEnd());
|
|
312
|
+
if (!Array.isArray(provider.models)) continue;
|
|
313
|
+
for (const m of provider.models) {
|
|
314
|
+
const id = typeof m === 'string' ? m : m?.id;
|
|
315
|
+
if (!id) continue;
|
|
316
|
+
const ref = `${provider.name}/${id}`;
|
|
317
|
+
const annot = ref === config.primaryModel ? ' ← primary'
|
|
318
|
+
: ref === config.fastModel ? ' ← fast'
|
|
319
|
+
: '';
|
|
320
|
+
console.log(` ${ref}${annot}`);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
}
|
|
207
324
|
|
|
208
325
|
async function runLlmSetup(current, configPath) {
|
|
209
326
|
if (!process.stdin.isTTY || !process.stdout.isTTY) {
|
package/package.json
CHANGED
package/yeaft/config.js
CHANGED
|
@@ -54,8 +54,62 @@ const DEFAULTS = {
|
|
|
54
54
|
// block is injected). Hand-edited values are NOT clamped — we let
|
|
55
55
|
// power users opt into larger docs at their own context-window risk.
|
|
56
56
|
projectDocMaxBytes: 32 * 1024,
|
|
57
|
+
// ─── LLM retry policy ──────────────────────────────────────
|
|
58
|
+
// How the engine reacts when adapter.stream()/call() throws a
|
|
59
|
+
// retryable error (429 / 529 / 5xx / transport failure). Each field
|
|
60
|
+
// is overridable from ~/.yeaft/config.json under "llmRetry": {…}.
|
|
61
|
+
// • maxRetries: cap on CONSECUTIVE retryable failures per turn.
|
|
62
|
+
// 0 disables retry entirely (caller falls straight through to
|
|
63
|
+
// fallbackModel / error). Default 3.
|
|
64
|
+
// • baseDelayMs / maxDelayMs: exponential backoff bounds used when
|
|
65
|
+
// the server didn't send a Retry-After header.
|
|
66
|
+
// • jitterRatio: ± random fraction applied to backoff; 0 disables.
|
|
67
|
+
llmRetry: {
|
|
68
|
+
maxRetries: 3,
|
|
69
|
+
baseDelayMs: 1_000,
|
|
70
|
+
maxDelayMs: 30_000,
|
|
71
|
+
jitterRatio: 0.25,
|
|
72
|
+
},
|
|
57
73
|
};
|
|
58
74
|
|
|
75
|
+
// ─── llmRetry normalizer ────────────────────────────────────────
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Merge user-supplied retry overrides with DEFAULTS.llmRetry, clamping
|
|
79
|
+
* each field to its safe range. The output is always a fully-populated
|
|
80
|
+
* object so the engine never has to deal with `undefined`. Unknown keys
|
|
81
|
+
* in the input are dropped.
|
|
82
|
+
*
|
|
83
|
+
* Precedence: overrides > fileConfig > DEFAULTS.
|
|
84
|
+
*
|
|
85
|
+
* @param {object | null | undefined} fileConfig
|
|
86
|
+
* @param {object | null | undefined} overrides
|
|
87
|
+
* @returns {{ maxRetries: number, baseDelayMs: number, maxDelayMs: number, jitterRatio: number }}
|
|
88
|
+
*/
|
|
89
|
+
export function normalizeLlmRetry(fileConfig, overrides) {
|
|
90
|
+
const base = DEFAULTS.llmRetry;
|
|
91
|
+
const out = { ...base };
|
|
92
|
+
const apply = (src) => {
|
|
93
|
+
if (!src || typeof src !== 'object') return;
|
|
94
|
+
if (Number.isFinite(src.maxRetries) && src.maxRetries >= 0) {
|
|
95
|
+
out.maxRetries = Math.min(20, Math.floor(src.maxRetries));
|
|
96
|
+
}
|
|
97
|
+
if (Number.isFinite(src.baseDelayMs) && src.baseDelayMs >= 0) {
|
|
98
|
+
out.baseDelayMs = Math.min(60_000, Math.floor(src.baseDelayMs));
|
|
99
|
+
}
|
|
100
|
+
if (Number.isFinite(src.maxDelayMs) && src.maxDelayMs >= 0) {
|
|
101
|
+
out.maxDelayMs = Math.min(600_000, Math.floor(src.maxDelayMs));
|
|
102
|
+
}
|
|
103
|
+
if (Number.isFinite(src.jitterRatio) && src.jitterRatio >= 0) {
|
|
104
|
+
out.jitterRatio = Math.min(1, src.jitterRatio);
|
|
105
|
+
}
|
|
106
|
+
};
|
|
107
|
+
apply(fileConfig);
|
|
108
|
+
apply(overrides);
|
|
109
|
+
if (out.maxDelayMs < out.baseDelayMs) out.maxDelayMs = out.baseDelayMs;
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
112
|
+
|
|
59
113
|
// ─── config.json reader ─────────────────────────────────────────
|
|
60
114
|
|
|
61
115
|
/**
|
|
@@ -220,6 +274,7 @@ function loadLegacyConfig(dir, overrides) {
|
|
|
220
274
|
const config = {
|
|
221
275
|
model: overrides.model || env.YEAFT_MODEL || fileConfig.model || 'claude-sonnet-4-20250514',
|
|
222
276
|
fallbackModel: overrides.fallbackModel || env.YEAFT_FALLBACK_MODEL || fileConfig.fallbackModel || null,
|
|
277
|
+
llmRetry: normalizeLlmRetry(fileConfig.llmRetry, overrides.llmRetry),
|
|
223
278
|
language: overrides.language || env.YEAFT_LANGUAGE || fileConfig.language || DEFAULTS.language,
|
|
224
279
|
apiKey: overrides.apiKey || env.YEAFT_API_KEY || fileConfig.apiKey || null,
|
|
225
280
|
openaiApiKey: overrides.openaiApiKey || env.YEAFT_OPENAI_API_KEY || fileConfig.openaiApiKey || null,
|
|
@@ -341,6 +396,7 @@ export function loadConfig(overrides = {}) {
|
|
|
341
396
|
fastModel: overrides.fastModel || fastModel,
|
|
342
397
|
fastModelId: fastModelId,
|
|
343
398
|
fallbackModel: jsonConfig.fallbackModel || null,
|
|
399
|
+
llmRetry: normalizeLlmRetry(jsonConfig.llmRetry, overrides.llmRetry),
|
|
344
400
|
modelInfo: modelInfo || null,
|
|
345
401
|
|
|
346
402
|
// Providers
|
package/yeaft/engine.js
CHANGED
|
@@ -22,7 +22,7 @@ import { promises as fsp } from 'fs';
|
|
|
22
22
|
import { join, resolve as resolvePath } from 'path';
|
|
23
23
|
import { buildSystemPrompt, buildWorkerPrompt } from './prompts.js';
|
|
24
24
|
import { getRuntimePlatformInfo } from './runtime-platform.js';
|
|
25
|
-
import { LLMContextError, LLMAbortError } from './llm/adapter.js';
|
|
25
|
+
import { LLMContextError, LLMAbortError, LLMRateLimitError, LLMServerError } from './llm/adapter.js';
|
|
26
26
|
import { runMemoryPreflow, buildRelevantScopes } from './sessions/pre-flow.js';
|
|
27
27
|
import { readProjectDoc, pickProjectDocFile, DEFAULT_PROJECT_DOC_MAX_BYTES } from './sessions/project-doc.js';
|
|
28
28
|
import { partitionMessages } from './compact/partition.js';
|
|
@@ -76,6 +76,80 @@ import {
|
|
|
76
76
|
/** Maximum auto-continue turns when stopReason is 'max_tokens'. */
|
|
77
77
|
const MAX_CONTINUE_TURNS = 3;
|
|
78
78
|
|
|
79
|
+
// ─── LLM retry policy defaults ──────────────────────────────────
|
|
80
|
+
// Hard-coded floor / ceiling for retry behaviour. The engine reads the
|
|
81
|
+
// effective policy from `config.llmRetry` so users can dial these via
|
|
82
|
+
// ~/.yeaft/config.json without touching code; the constants here are the
|
|
83
|
+
// fallback when the config is missing or partial.
|
|
84
|
+
// • maxRetries: how many times we re-issue the same turn on a
|
|
85
|
+
// retryable failure before giving up / falling back
|
|
86
|
+
// • baseDelayMs: starting backoff for transient (5xx / network) errors
|
|
87
|
+
// • maxDelayMs: ceiling we never exceed regardless of backoff growth
|
|
88
|
+
// • jitterRatio: ± random fraction of the delay (0.25 = ±25 %)
|
|
89
|
+
// Rate-limit waits prefer the server's Retry-After header; we only fall
|
|
90
|
+
// back to exponential backoff when the header is missing.
|
|
91
|
+
const RETRY_DEFAULTS = Object.freeze({
|
|
92
|
+
maxRetries: 3,
|
|
93
|
+
baseDelayMs: 1_000,
|
|
94
|
+
maxDelayMs: 30_000,
|
|
95
|
+
jitterRatio: 0.25,
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
function resolveRetryPolicy(config) {
|
|
99
|
+
const raw = config?.llmRetry || {};
|
|
100
|
+
const num = (v, d) => (Number.isFinite(v) && v >= 0 ? v : d);
|
|
101
|
+
return {
|
|
102
|
+
maxRetries: Math.max(0, Math.floor(num(raw.maxRetries, RETRY_DEFAULTS.maxRetries))),
|
|
103
|
+
baseDelayMs: Math.max(0, Math.floor(num(raw.baseDelayMs, RETRY_DEFAULTS.baseDelayMs))),
|
|
104
|
+
maxDelayMs: Math.max(0, Math.floor(num(raw.maxDelayMs, RETRY_DEFAULTS.maxDelayMs))),
|
|
105
|
+
jitterRatio: Math.min(1, Math.max(0, num(raw.jitterRatio, RETRY_DEFAULTS.jitterRatio))),
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Compute the next backoff delay (ms) for a transient error.
|
|
111
|
+
* Exponential growth (base * 2^attempt) capped at maxDelayMs, with optional
|
|
112
|
+
* +/- jitter to avoid synchronized retry stampedes.
|
|
113
|
+
*
|
|
114
|
+
* @param {{ baseDelayMs: number, maxDelayMs: number, jitterRatio: number }} policy
|
|
115
|
+
* @param {number} attempt - 0-indexed retry attempt (0 = first retry)
|
|
116
|
+
* @returns {number} milliseconds to sleep
|
|
117
|
+
*/
|
|
118
|
+
export function computeBackoffDelay(policy, attempt) {
|
|
119
|
+
const safeAttempt = Math.max(0, Math.floor(attempt));
|
|
120
|
+
const grown = policy.baseDelayMs * Math.pow(2, safeAttempt);
|
|
121
|
+
const capped = Math.min(policy.maxDelayMs, grown);
|
|
122
|
+
if (policy.jitterRatio <= 0) return capped;
|
|
123
|
+
const jitter = capped * policy.jitterRatio;
|
|
124
|
+
const offset = (Math.random() * 2 - 1) * jitter; // [-jitter, +jitter]
|
|
125
|
+
return Math.max(0, Math.round(capped + offset));
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Promise-based sleep that resolves early if the caller's AbortSignal fires.
|
|
130
|
+
* Returns true when the sleep completed normally, false when aborted.
|
|
131
|
+
*
|
|
132
|
+
* @param {number} ms
|
|
133
|
+
* @param {AbortSignal | undefined} signal
|
|
134
|
+
* @returns {Promise<boolean>}
|
|
135
|
+
*/
|
|
136
|
+
function sleepWithAbort(ms, signal) {
|
|
137
|
+
if (ms <= 0) return Promise.resolve(true);
|
|
138
|
+
if (signal?.aborted) return Promise.resolve(false);
|
|
139
|
+
return new Promise((resolve) => {
|
|
140
|
+
const timer = setTimeout(() => {
|
|
141
|
+
if (signal) signal.removeEventListener('abort', onAbort);
|
|
142
|
+
resolve(true);
|
|
143
|
+
}, ms);
|
|
144
|
+
const onAbort = () => {
|
|
145
|
+
clearTimeout(timer);
|
|
146
|
+
if (signal) signal.removeEventListener('abort', onAbort);
|
|
147
|
+
resolve(false);
|
|
148
|
+
};
|
|
149
|
+
if (signal) signal.addEventListener('abort', onAbort, { once: true });
|
|
150
|
+
});
|
|
151
|
+
}
|
|
152
|
+
|
|
79
153
|
/**
|
|
80
154
|
* task-331 — Map a conversationMessages entry into the snapshot shape used
|
|
81
155
|
* by `debug_turn.messages`. Preserves the function-calling metadata that
|
|
@@ -226,8 +300,9 @@ export function shouldAllowGroupReflection({
|
|
|
226
300
|
* @typedef {{ type: 'consolidate', archivedCount: number, extractedCount: number }} ConsolidateEvent
|
|
227
301
|
* @typedef {{ type: 'recall', entryCount: number, cached: boolean }} RecallEvent
|
|
228
302
|
* @typedef {{ type: 'fallback', from: string, to: string, reason: string }} FallbackEvent
|
|
303
|
+
* @typedef {{ type: 'llm_retry', attempt: number, maxRetries: number, delayMs: number, reason: 'rate_limit_retry_after'|'rate_limit_backoff'|'transient_backoff', errorName: string, statusCode: number|null, message: string }} LlmRetryEvent
|
|
229
304
|
*
|
|
230
|
-
* @typedef {import('./llm/adapter.js').StreamEvent | TurnStartEvent | TurnEndEvent | ToolStartEvent | ToolEndEvent | ConsolidateEvent | RecallEvent | FallbackEvent} EngineEvent
|
|
305
|
+
* @typedef {import('./llm/adapter.js').StreamEvent | TurnStartEvent | TurnEndEvent | ToolStartEvent | ToolEndEvent | ConsolidateEvent | RecallEvent | FallbackEvent | LlmRetryEvent} EngineEvent
|
|
231
306
|
*/
|
|
232
307
|
|
|
233
308
|
// ─── Engine ──────────────────────────────────────────────────────
|
|
@@ -1760,6 +1835,16 @@ export class Engine {
|
|
|
1760
1835
|
// outer-loop iteration so the flag never carries across turns.
|
|
1761
1836
|
let endTurnRequested = null;
|
|
1762
1837
|
|
|
1838
|
+
// LLM retry bookkeeping (rate-limit / 5xx / transient network errors).
|
|
1839
|
+
// Counts CONSECUTIVE retryable failures on the same turn — reset to 0
|
|
1840
|
+
// on any successful stream() iteration, and also on a fallback-model
|
|
1841
|
+
// switch (the new model gets a fresh budget). Reaching maxRetries
|
|
1842
|
+
// gives up: we either fall back to a backup model or surface the
|
|
1843
|
+
// error to the user. LLMContextError has its own compact-retry path
|
|
1844
|
+
// and does NOT count against this budget.
|
|
1845
|
+
const retryPolicy = resolveRetryPolicy(this.#config);
|
|
1846
|
+
let consecutiveRetryableErrors = 0;
|
|
1847
|
+
|
|
1763
1848
|
while (true) {
|
|
1764
1849
|
turnNumber++;
|
|
1765
1850
|
|
|
@@ -2011,6 +2096,11 @@ export class Engine {
|
|
|
2011
2096
|
break;
|
|
2012
2097
|
}
|
|
2013
2098
|
}
|
|
2099
|
+
// Stream completed without throwing — reset the retry counter so
|
|
2100
|
+
// the next turn starts with a clean budget. Note: this includes
|
|
2101
|
+
// stream() that emitted an in-band `error` event (server didn't
|
|
2102
|
+
// throw), since those are policy-specific and not transport-level.
|
|
2103
|
+
consecutiveRetryableErrors = 0;
|
|
2014
2104
|
} catch (err) {
|
|
2015
2105
|
const latencyMs = Date.now() - startTime;
|
|
2016
2106
|
this.#trace.endTurn(turnId, {
|
|
@@ -2087,12 +2177,64 @@ export class Engine {
|
|
|
2087
2177
|
}
|
|
2088
2178
|
}
|
|
2089
2179
|
|
|
2180
|
+
// ─── Rate-limit / transient retry ─────────────────
|
|
2181
|
+
// Honour server-supplied Retry-After for 429/529; fall back to
|
|
2182
|
+
// exponential backoff for 5xx and transport failures wrapped as
|
|
2183
|
+
// LLMServerError. Counts against retryPolicy.maxRetries; on
|
|
2184
|
+
// exhaustion we fall through to the fallback-model path (and
|
|
2185
|
+
// ultimately the error event) without further waiting.
|
|
2186
|
+
const isRateLimit = err instanceof LLMRateLimitError;
|
|
2187
|
+
const isTransient = err instanceof LLMServerError;
|
|
2188
|
+
if (isRateLimit || isTransient) {
|
|
2189
|
+
if (consecutiveRetryableErrors < retryPolicy.maxRetries) {
|
|
2190
|
+
consecutiveRetryableErrors += 1;
|
|
2191
|
+
let delayMs;
|
|
2192
|
+
let reason;
|
|
2193
|
+
if (isRateLimit && Number.isFinite(err.retryAfterMs) && err.retryAfterMs > 0) {
|
|
2194
|
+
// Server told us exactly when to come back. Respect it,
|
|
2195
|
+
// but still cap to maxDelayMs so a misconfigured upstream
|
|
2196
|
+
// can't make us hang forever on one turn.
|
|
2197
|
+
delayMs = Math.min(retryPolicy.maxDelayMs, err.retryAfterMs);
|
|
2198
|
+
reason = 'rate_limit_retry_after';
|
|
2199
|
+
} else if (isRateLimit) {
|
|
2200
|
+
// No header — use backoff but start one step higher so the
|
|
2201
|
+
// first retry isn't immediate (rate-limit windows are
|
|
2202
|
+
// usually >= 1s wide).
|
|
2203
|
+
delayMs = computeBackoffDelay(retryPolicy, consecutiveRetryableErrors);
|
|
2204
|
+
reason = 'rate_limit_backoff';
|
|
2205
|
+
} else {
|
|
2206
|
+
delayMs = computeBackoffDelay(retryPolicy, consecutiveRetryableErrors - 1);
|
|
2207
|
+
reason = 'transient_backoff';
|
|
2208
|
+
}
|
|
2209
|
+
yield {
|
|
2210
|
+
type: 'llm_retry',
|
|
2211
|
+
attempt: consecutiveRetryableErrors,
|
|
2212
|
+
maxRetries: retryPolicy.maxRetries,
|
|
2213
|
+
delayMs,
|
|
2214
|
+
reason,
|
|
2215
|
+
errorName: err.name,
|
|
2216
|
+
statusCode: err.statusCode ?? null,
|
|
2217
|
+
message: String(err.message || '').slice(0, 300),
|
|
2218
|
+
};
|
|
2219
|
+
const slept = await sleepWithAbort(delayMs, signal);
|
|
2220
|
+
if (!slept || signal?.aborted) {
|
|
2221
|
+
yield { type: 'aborted', reason: this.#abortReason || 'external', turnNumber, threadId };
|
|
2222
|
+
yield { type: 'turn_end', turnNumber, stopReason: 'aborted', threadId };
|
|
2223
|
+
break;
|
|
2224
|
+
}
|
|
2225
|
+
yield { type: 'turn_end', turnNumber, stopReason: 'llm_retry', threadId };
|
|
2226
|
+
continue; // retry the same turn with the same model
|
|
2227
|
+
}
|
|
2228
|
+
// Exhausted — fall through to fallback-model / error paths.
|
|
2229
|
+
}
|
|
2230
|
+
|
|
2090
2231
|
// ─── Fallback model ──────────────────────────────
|
|
2091
2232
|
const fallbackModel = this.#config.fallbackModel;
|
|
2092
2233
|
if (fallbackModel && fallbackModel !== currentModel &&
|
|
2093
2234
|
(err.name === 'LLMRateLimitError' || err.name === 'LLMServerError')) {
|
|
2094
2235
|
yield { type: 'fallback', from: currentModel, to: fallbackModel, reason: err.message };
|
|
2095
2236
|
currentModel = fallbackModel;
|
|
2237
|
+
consecutiveRetryableErrors = 0; // new model, fresh retry budget
|
|
2096
2238
|
yield { type: 'turn_end', turnNumber, stopReason: 'fallback_retry', threadId };
|
|
2097
2239
|
continue; // retry with fallback model
|
|
2098
2240
|
}
|
package/yeaft/llm/adapter.js
CHANGED
|
@@ -108,6 +108,112 @@ export class LLMAbortError extends Error {
|
|
|
108
108
|
}
|
|
109
109
|
}
|
|
110
110
|
|
|
111
|
+
// ─── Retry helpers ─────────────────────────────────────────────
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Parse an HTTP `Retry-After` header value into milliseconds.
|
|
115
|
+
*
|
|
116
|
+
* The header may be either:
|
|
117
|
+
* • An integer number of seconds (delta-seconds) — `"30"`
|
|
118
|
+
* • An HTTP-date — `"Fri, 31 Dec 1999 23:59:59 GMT"`
|
|
119
|
+
*
|
|
120
|
+
* Returns null when the header is missing, malformed, or yields a non-positive
|
|
121
|
+
* delay. Callers should treat null as "no server hint" and fall back to their
|
|
122
|
+
* own backoff schedule.
|
|
123
|
+
*
|
|
124
|
+
* @param {string | null | undefined} headerValue
|
|
125
|
+
* @returns {number | null} milliseconds to wait, or null
|
|
126
|
+
*/
|
|
127
|
+
export function parseRetryAfterMs(headerValue) {
|
|
128
|
+
if (!headerValue) return null;
|
|
129
|
+
const trimmed = String(headerValue).trim();
|
|
130
|
+
if (!trimmed) return null;
|
|
131
|
+
// Integer seconds path
|
|
132
|
+
if (/^\d+(\.\d+)?$/.test(trimmed)) {
|
|
133
|
+
const seconds = Number(trimmed);
|
|
134
|
+
if (!Number.isFinite(seconds) || seconds < 0) return null;
|
|
135
|
+
return Math.round(seconds * 1000);
|
|
136
|
+
}
|
|
137
|
+
// HTTP-date path
|
|
138
|
+
const ts = Date.parse(trimmed);
|
|
139
|
+
if (!Number.isFinite(ts)) return null;
|
|
140
|
+
const delta = ts - Date.now();
|
|
141
|
+
return delta > 0 ? delta : null;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Read the `retry-after` header from a fetch Response in a case-insensitive
|
|
146
|
+
* way that tolerates both real `Headers` instances and the plain object
|
|
147
|
+
* stand-ins our tests sometimes hand us.
|
|
148
|
+
*
|
|
149
|
+
* @param {Response | { headers?: Record<string, string> | Headers } | null | undefined} response
|
|
150
|
+
* @returns {number | null} milliseconds, or null when not present
|
|
151
|
+
*/
|
|
152
|
+
export function retryAfterFromResponse(response) {
|
|
153
|
+
const headers = response?.headers;
|
|
154
|
+
if (!headers) return null;
|
|
155
|
+
let raw = null;
|
|
156
|
+
if (typeof headers.get === 'function') {
|
|
157
|
+
raw = headers.get('retry-after');
|
|
158
|
+
} else {
|
|
159
|
+
for (const key of Object.keys(headers)) {
|
|
160
|
+
if (key.toLowerCase() === 'retry-after') {
|
|
161
|
+
raw = headers[key];
|
|
162
|
+
break;
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
return parseRetryAfterMs(raw);
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Classify a raw error thrown by `fetch()` (or while reading a streaming
|
|
171
|
+
* body) into one of the unified LLM error types when it represents a
|
|
172
|
+
* retryable transport-level failure (DNS, ECONN*, socket reset, fetch
|
|
173
|
+
* `TypeError`, undici `UND_ERR_*`, …).
|
|
174
|
+
*
|
|
175
|
+
* Returns:
|
|
176
|
+
* • LLMAbortError — caller-initiated abort; engine short-circuits.
|
|
177
|
+
* • LLMServerError — transient transport failure; engine should retry.
|
|
178
|
+
* • the original error — anything that doesn't match a known transient
|
|
179
|
+
* pattern. We never invent retryability we can't
|
|
180
|
+
* prove from the error shape.
|
|
181
|
+
*
|
|
182
|
+
* @param {unknown} err
|
|
183
|
+
* @param {{ providerLabel?: string }} [opts]
|
|
184
|
+
* @returns {Error}
|
|
185
|
+
*/
|
|
186
|
+
export function classifyFetchError(err, opts = {}) {
|
|
187
|
+
if (!(err instanceof Error)) return err instanceof Object ? err : new Error(String(err));
|
|
188
|
+
if (err.name === 'AbortError' || err.name === 'LLMAbortError') return new LLMAbortError();
|
|
189
|
+
// Anything already classified: keep as-is.
|
|
190
|
+
if (err instanceof LLMRateLimitError
|
|
191
|
+
|| err instanceof LLMAuthError
|
|
192
|
+
|| err instanceof LLMContextError
|
|
193
|
+
|| err instanceof LLMServerError
|
|
194
|
+
|| err instanceof LLMAbortError) {
|
|
195
|
+
return err;
|
|
196
|
+
}
|
|
197
|
+
const label = opts.providerLabel ? `${opts.providerLabel}: ` : '';
|
|
198
|
+
const code = err.cause?.code || err.code || null;
|
|
199
|
+
const transientCodes = new Set([
|
|
200
|
+
'ECONNRESET', 'ECONNREFUSED', 'ECONNABORTED', 'ETIMEDOUT',
|
|
201
|
+
'ENOTFOUND', 'EAI_AGAIN', 'EPIPE', 'EHOSTUNREACH', 'ENETUNREACH',
|
|
202
|
+
'UND_ERR_SOCKET', 'UND_ERR_CLOSED', 'UND_ERR_BODY_TIMEOUT',
|
|
203
|
+
'UND_ERR_HEADERS_TIMEOUT', 'UND_ERR_CONNECT_TIMEOUT',
|
|
204
|
+
]);
|
|
205
|
+
if (code && transientCodes.has(code)) {
|
|
206
|
+
return new LLMServerError(`${label}network error (${code}): ${err.message}`, 0);
|
|
207
|
+
}
|
|
208
|
+
// Node 20+ fetch surfaces network problems as a plain TypeError
|
|
209
|
+
// with `cause` set to the underlying undici error. We treat any
|
|
210
|
+
// such TypeError as transient — abort already returned above.
|
|
211
|
+
if (err.name === 'TypeError' && /fetch failed|terminated|network|socket/i.test(err.message || '')) {
|
|
212
|
+
return new LLMServerError(`${label}fetch failed: ${err.message}`, 0);
|
|
213
|
+
}
|
|
214
|
+
return err;
|
|
215
|
+
}
|
|
216
|
+
|
|
111
217
|
// ─── Raw payload redaction helper ──────────────────────────────
|
|
112
218
|
|
|
113
219
|
/**
|
package/yeaft/llm/anthropic.js
CHANGED
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
import {
|
|
10
10
|
LLMAdapter,
|
|
11
11
|
LLMRateLimitError,
|
|
12
|
+
classifyFetchError,
|
|
13
|
+
retryAfterFromResponse,
|
|
12
14
|
LLMAuthError,
|
|
13
15
|
LLMContextError,
|
|
14
16
|
LLMServerError,
|
|
@@ -175,18 +177,22 @@ export class AnthropicAdapter extends LLMAdapter {
|
|
|
175
177
|
* Classify HTTP errors into our typed errors.
|
|
176
178
|
* @param {number} status
|
|
177
179
|
* @param {string} body
|
|
180
|
+
* @param {Response | { headers?: Record<string, string> } | null} [response] -
|
|
181
|
+
* When provided, reads the `retry-after` header so the engine can honor
|
|
182
|
+
* the server hint instead of guessing a backoff delay.
|
|
178
183
|
*/
|
|
179
|
-
#classifyError(status, body) {
|
|
184
|
+
#classifyError(status, body, response = null) {
|
|
180
185
|
const authHint = `auth=${this.#authHeaderMode}`;
|
|
181
186
|
if (status === 401 || status === 403) {
|
|
182
187
|
return new LLMAuthError(`Anthropic auth error (${authHint}): ${body}`, status);
|
|
183
188
|
}
|
|
184
189
|
if (status === 429) {
|
|
185
|
-
const retryAfter =
|
|
190
|
+
const retryAfter = retryAfterFromResponse(response);
|
|
186
191
|
return new LLMRateLimitError(`Anthropic rate limit (${authHint}): ${body}`, status, retryAfter);
|
|
187
192
|
}
|
|
188
193
|
if (status === 529) {
|
|
189
|
-
|
|
194
|
+
const retryAfter = retryAfterFromResponse(response);
|
|
195
|
+
return new LLMRateLimitError(`Anthropic overloaded (${authHint}): ${body}`, status, retryAfter);
|
|
190
196
|
}
|
|
191
197
|
if (body.includes('prompt is too long') || body.includes('max_tokens')) {
|
|
192
198
|
return new LLMContextError(`Anthropic context error (${authHint}): ${body}`);
|
|
@@ -231,12 +237,17 @@ export class AnthropicAdapter extends LLMAdapter {
|
|
|
231
237
|
// exactly what we POST to the LLM.
|
|
232
238
|
const rawRequest = redactRawRequest({ url, method: 'POST', headers, body });
|
|
233
239
|
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
+
let response;
|
|
241
|
+
try {
|
|
242
|
+
response = await fetch(url, {
|
|
243
|
+
method: 'POST',
|
|
244
|
+
headers,
|
|
245
|
+
body: JSON.stringify(body),
|
|
246
|
+
signal,
|
|
247
|
+
});
|
|
248
|
+
} catch (err) {
|
|
249
|
+
throw classifyFetchError(err, { providerLabel: 'Anthropic' });
|
|
250
|
+
}
|
|
240
251
|
|
|
241
252
|
if (!response.ok) {
|
|
242
253
|
const errorBody = await response.text();
|
|
@@ -253,7 +264,7 @@ export class AnthropicAdapter extends LLMAdapter {
|
|
|
253
264
|
});
|
|
254
265
|
} catch { /* ignore */ }
|
|
255
266
|
}
|
|
256
|
-
throw this.#classifyError(response.status, errorBody);
|
|
267
|
+
throw this.#classifyError(response.status, errorBody, response);
|
|
257
268
|
}
|
|
258
269
|
|
|
259
270
|
// Parse SSE stream
|
|
@@ -469,16 +480,21 @@ export class AnthropicAdapter extends LLMAdapter {
|
|
|
469
480
|
applyAnthropicThinking(body, model, normEffort);
|
|
470
481
|
}
|
|
471
482
|
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
483
|
+
let response;
|
|
484
|
+
try {
|
|
485
|
+
response = await fetch(`${this.#baseUrl}/v1/messages`, {
|
|
486
|
+
method: 'POST',
|
|
487
|
+
headers: this.#headers(),
|
|
488
|
+
body: JSON.stringify(body),
|
|
489
|
+
signal,
|
|
490
|
+
});
|
|
491
|
+
} catch (err) {
|
|
492
|
+
throw classifyFetchError(err, { providerLabel: 'Anthropic' });
|
|
493
|
+
}
|
|
478
494
|
|
|
479
495
|
if (!response.ok) {
|
|
480
496
|
const errorBody = await response.text();
|
|
481
|
-
throw this.#classifyError(response.status, errorBody);
|
|
497
|
+
throw this.#classifyError(response.status, errorBody, response);
|
|
482
498
|
}
|
|
483
499
|
|
|
484
500
|
const result = await response.json();
|
|
@@ -32,6 +32,8 @@ import {
|
|
|
32
32
|
LLMContextError,
|
|
33
33
|
LLMServerError,
|
|
34
34
|
LLMAbortError,
|
|
35
|
+
classifyFetchError,
|
|
36
|
+
retryAfterFromResponse,
|
|
35
37
|
redactRawRequest,
|
|
36
38
|
safeHeaders,
|
|
37
39
|
} from './adapter.js';
|
|
@@ -187,15 +189,17 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
|
|
|
187
189
|
|
|
188
190
|
// ─── Error classification ───────────────────────────────
|
|
189
191
|
|
|
190
|
-
#classifyError(status, body) {
|
|
192
|
+
#classifyError(status, body, response = null) {
|
|
191
193
|
if (status === 401 || status === 403) {
|
|
192
194
|
return new LLMAuthError(`Auth error: ${body}`, status);
|
|
193
195
|
}
|
|
194
196
|
if (status === 429) {
|
|
195
|
-
|
|
197
|
+
const retryAfter = retryAfterFromResponse(response);
|
|
198
|
+
return new LLMRateLimitError(`Rate limit: ${body}`, status, retryAfter);
|
|
196
199
|
}
|
|
197
200
|
if (status === 529) {
|
|
198
|
-
|
|
201
|
+
const retryAfter = retryAfterFromResponse(response);
|
|
202
|
+
return new LLMRateLimitError(`Overloaded: ${body}`, status, retryAfter);
|
|
199
203
|
}
|
|
200
204
|
if (status === 413 || body.includes('context_length_exceeded') || body.includes('maximum context length')) {
|
|
201
205
|
return new LLMContextError(`Context too long: ${body}`);
|
|
@@ -286,7 +290,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
|
|
|
286
290
|
});
|
|
287
291
|
} catch (err) {
|
|
288
292
|
if (err.name === 'AbortError') throw new LLMAbortError();
|
|
289
|
-
throw err;
|
|
293
|
+
throw classifyFetchError(err, { providerLabel: 'OpenAI' });
|
|
290
294
|
}
|
|
291
295
|
|
|
292
296
|
if (!response.ok) {
|
|
@@ -304,7 +308,7 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
|
|
|
304
308
|
});
|
|
305
309
|
} catch { /* ignore */ }
|
|
306
310
|
}
|
|
307
|
-
throw this.#classifyError(response.status, errorBody);
|
|
311
|
+
throw this.#classifyError(response.status, errorBody, response);
|
|
308
312
|
}
|
|
309
313
|
|
|
310
314
|
const reader = response.body.getReader();
|
|
@@ -512,12 +516,12 @@ export class OpenAIResponsesAdapter extends LLMAdapter {
|
|
|
512
516
|
});
|
|
513
517
|
} catch (err) {
|
|
514
518
|
if (err.name === 'AbortError') throw new LLMAbortError();
|
|
515
|
-
throw err;
|
|
519
|
+
throw classifyFetchError(err, { providerLabel: 'OpenAI' });
|
|
516
520
|
}
|
|
517
521
|
|
|
518
522
|
if (!response.ok) {
|
|
519
523
|
const errorBody = await response.text();
|
|
520
|
-
throw this.#classifyError(response.status, errorBody);
|
|
524
|
+
throw this.#classifyError(response.status, errorBody, response);
|
|
521
525
|
}
|
|
522
526
|
|
|
523
527
|
const result = await response.json();
|
package/yeaft/models.js
CHANGED
|
@@ -428,6 +428,13 @@ export const OPENAI_REASONING_EFFORT_OPTIONS = ['minimal', 'low', 'medium', 'hig
|
|
|
428
428
|
export const ANTHROPIC_MANUAL_EFFORT_OPTIONS = ['low', 'medium', 'high'];
|
|
429
429
|
export const ANTHROPIC_ADAPTIVE_EFFORT_OPTIONS = ['low', 'medium', 'high', 'xhigh', 'max'];
|
|
430
430
|
export const ANTHROPIC_ADAPTIVE_MAX_EFFORT_OPTIONS = ['low', 'medium', 'high', 'max'];
|
|
431
|
+
// DeepSeek reasoning models (deepseek-reasoner / deepseek-r1) expose a simple
|
|
432
|
+
// low/medium/high effort scale. DeepSeek's OpenAI-compatible surface accepts a
|
|
433
|
+
// reasoning effort hint; we send it through the standard openai-reasoning
|
|
434
|
+
// `reasoning.effort` path (relay/proxy adapts to DeepSeek's wire format). No
|
|
435
|
+
// `minimal` tier — DeepSeek documents only a high/max effort distinction, so we
|
|
436
|
+
// keep the user-facing scale to the three levels the user expects.
|
|
437
|
+
export const DEEPSEEK_REASONING_EFFORT_OPTIONS = ['low', 'medium', 'high'];
|
|
431
438
|
|
|
432
439
|
function inferThinkingCapability(model) {
|
|
433
440
|
const id = parseModelRef(model).modelId.toLowerCase();
|
|
@@ -468,6 +475,20 @@ function inferThinkingCapability(model) {
|
|
|
468
475
|
return { supportsThinking: true, thinkingProtocol: 'anthropic', defaultEffort: null, maxBudgetTokens };
|
|
469
476
|
}
|
|
470
477
|
|
|
478
|
+
// DeepSeek reasoning models expose a thinking effort hint. Only the reasoner
|
|
479
|
+
// family (deepseek-reasoner / deepseek-r1*) is a reasoning model — plain
|
|
480
|
+
// deepseek-chat stays effort-less. Effort travels over the openai-reasoning
|
|
481
|
+
// `reasoning.effort` path; user-facing scale is low/medium/high.
|
|
482
|
+
if (/^deepseek-(reasoner|r1)/.test(id)) {
|
|
483
|
+
return {
|
|
484
|
+
supportsThinking: true,
|
|
485
|
+
thinkingProtocol: 'openai-reasoning',
|
|
486
|
+
defaultEffort: null,
|
|
487
|
+
maxBudgetTokens: null,
|
|
488
|
+
effortOptions: DEEPSEEK_REASONING_EFFORT_OPTIONS,
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
|
|
471
492
|
return null;
|
|
472
493
|
}
|
|
473
494
|
|
package/yeaft/web-bridge.js
CHANGED
|
@@ -2264,6 +2264,23 @@ function handleEngineEvent(event, hctx) {
|
|
|
2264
2264
|
}, envelope);
|
|
2265
2265
|
break;
|
|
2266
2266
|
|
|
2267
|
+
case 'llm_retry':
|
|
2268
|
+
// Engine paused before re-issuing the same turn because the LLM
|
|
2269
|
+
// returned a retryable error (rate limit / 5xx / transient network).
|
|
2270
|
+
// Surface to the client so the UI can show "retrying in Xs (1/3)"
|
|
2271
|
+
// instead of looking frozen mid-turn.
|
|
2272
|
+
sendSessionEvent({
|
|
2273
|
+
type: 'llm_retry',
|
|
2274
|
+
attempt: event.attempt,
|
|
2275
|
+
maxRetries: event.maxRetries,
|
|
2276
|
+
delayMs: event.delayMs,
|
|
2277
|
+
reason: event.reason,
|
|
2278
|
+
errorName: event.errorName,
|
|
2279
|
+
statusCode: event.statusCode,
|
|
2280
|
+
message: event.message,
|
|
2281
|
+
}, envelope);
|
|
2282
|
+
break;
|
|
2283
|
+
|
|
2267
2284
|
case 'reflection':
|
|
2268
2285
|
sendSessionEvent({
|
|
2269
2286
|
type: 'reflection',
|