@latimer-woods-tech/llm 0.4.2 → 0.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +35 -0
- package/dist/index.d.mts +379 -0
- package/dist/index.mjs +910 -0
- package/dist/index.mjs.map +1 -0
- package/package.json +1 -1
package/dist/index.mjs
ADDED
|
@@ -0,0 +1,910 @@
|
|
|
1
|
+
// src/index.ts
|
|
2
|
+
import {
|
|
3
|
+
InternalError,
|
|
4
|
+
RateLimitError,
|
|
5
|
+
ValidationError,
|
|
6
|
+
toErrorResponse
|
|
7
|
+
} from "@latimer-woods-tech/errors";
|
|
8
|
+
function contentToText(content) {
|
|
9
|
+
if (typeof content === "string") return content;
|
|
10
|
+
return content.map((b) => b.type === "text" ? b.text : b.type === "tool_result" ? b.content : "").join("");
|
|
11
|
+
}
|
|
12
|
+
function systemText(opts, messages) {
|
|
13
|
+
if (opts.system !== void 0) return opts.system;
|
|
14
|
+
const c = messages.find((m) => m.role === "system")?.content;
|
|
15
|
+
return c === void 0 ? void 0 : contentToText(c);
|
|
16
|
+
}
|
|
17
|
+
var MODELS = {
|
|
18
|
+
anthropic: {
|
|
19
|
+
fast: "claude-haiku-4-20250514",
|
|
20
|
+
balanced: "claude-sonnet-4-6",
|
|
21
|
+
smart: "claude-opus-4-7"
|
|
22
|
+
},
|
|
23
|
+
gemini: {
|
|
24
|
+
smart: "gemini-2.5-pro"
|
|
25
|
+
},
|
|
26
|
+
groq: {
|
|
27
|
+
verifier: "llama-4-maverick"
|
|
28
|
+
},
|
|
29
|
+
grok: {
|
|
30
|
+
fast: "grok-4.3"
|
|
31
|
+
},
|
|
32
|
+
deepseek: {
|
|
33
|
+
workbench: "deepseek-chat"
|
|
34
|
+
}
|
|
35
|
+
};
|
|
36
|
+
var DEFAULT_MAX_TOKENS = 1024;
|
|
37
|
+
var DEFAULT_TEMPERATURE = 0.7;
|
|
38
|
+
var DEFAULT_LONG_CONTEXT_THRESHOLD = 15e4;
|
|
39
|
+
var BACKOFF_BASE_MS = 500;
|
|
40
|
+
var BACKOFF_CAP_MS = 8e3;
|
|
41
|
+
var BACKOFF_JITTER_MAX_MS = 250;
|
|
42
|
+
var PER_PROVIDER_MAX_ATTEMPTS = 3;
|
|
43
|
+
var providerCooldownUntil = /* @__PURE__ */ new Map();
|
|
44
|
+
var PROVIDER_COOLDOWN_MS = 3e4;
|
|
45
|
+
function isProviderCoolingDown(provider, now = Date.now) {
|
|
46
|
+
const until = providerCooldownUntil.get(provider);
|
|
47
|
+
if (until === void 0) return false;
|
|
48
|
+
return now() < until;
|
|
49
|
+
}
|
|
50
|
+
function isoDate(nowMs) {
|
|
51
|
+
return new Date(nowMs).toISOString().slice(0, 10);
|
|
52
|
+
}
|
|
53
|
+
async function recordOrgCostUsage(kv, todayKey, monthKey, costUsd, opts) {
|
|
54
|
+
if (opts.dailyCapUsd !== void 0) {
|
|
55
|
+
const raw = await kv.get(todayKey).catch(() => null);
|
|
56
|
+
const spent = parseFloat(raw ?? "0");
|
|
57
|
+
await kv.put(todayKey, String(spent + costUsd), {
|
|
58
|
+
expirationTtl: 172800
|
|
59
|
+
/* 48 h */
|
|
60
|
+
}).catch(() => void 0);
|
|
61
|
+
}
|
|
62
|
+
if (opts.monthlyCapUsd !== void 0) {
|
|
63
|
+
const raw = await kv.get(monthKey).catch(() => null);
|
|
64
|
+
const spent = parseFloat(raw ?? "0");
|
|
65
|
+
await kv.put(monthKey, String(spent + costUsd), {
|
|
66
|
+
expirationTtl: 3456e3
|
|
67
|
+
/* 40 d */
|
|
68
|
+
}).catch(() => void 0);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
var MODEL_PRICE_PER_1M = {
|
|
72
|
+
// Anthropic Haiku 4
|
|
73
|
+
"claude-haiku-4-20250514": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
|
|
74
|
+
"claude-haiku-4-5-20251001": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
|
|
75
|
+
// Anthropic Sonnet 4
|
|
76
|
+
"claude-sonnet-4-20250514": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
|
77
|
+
"claude-sonnet-4-6": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
|
|
78
|
+
// Anthropic Opus 4
|
|
79
|
+
"claude-opus-4-20250514": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
|
|
80
|
+
"claude-opus-4-7": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
|
|
81
|
+
// Gemini 2.5 Pro
|
|
82
|
+
"gemini-2.5-pro": { input: 1.25, output: 10, cacheRead: 0.31, cacheWrite: 4.5 },
|
|
83
|
+
// Groq Llama 4 Maverick
|
|
84
|
+
"llama-4-maverick": { input: 0.5, output: 0.77, cacheRead: 0.05, cacheWrite: 0.5 },
|
|
85
|
+
// Grok 4.3
|
|
86
|
+
"grok-4.3": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
|
|
87
|
+
// DeepSeek API pricing as of 2026-05: cache-write conservatively uses cache-miss input pricing.
|
|
88
|
+
"deepseek-chat": { input: 0.27, output: 1.1, cacheRead: 0.07, cacheWrite: 0.27 },
|
|
89
|
+
"deepseek-reasoner": { input: 0.55, output: 2.19, cacheRead: 0.14, cacheWrite: 0.55 },
|
|
90
|
+
// Deprecated aliases retained for historical ledger rows.
|
|
91
|
+
"grok-4-fast": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
|
|
92
|
+
"grok-3-mini-latest": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 }
|
|
93
|
+
};
|
|
94
|
+
var PRICE_FALLBACK = MODEL_PRICE_PER_1M["claude-opus-4-7"];
|
|
95
|
+
function estimateCostUsd(tokens, model) {
|
|
96
|
+
const price = MODEL_PRICE_PER_1M[model] ?? PRICE_FALLBACK;
|
|
97
|
+
return (tokens.input * price.input + tokens.output * price.output + (tokens.cacheRead ?? 0) * price.cacheRead + (tokens.cacheWrite ?? 0) * price.cacheWrite) / 1e6;
|
|
98
|
+
}
|
|
99
|
+
function isoMonth(nowMs) {
|
|
100
|
+
return new Date(nowMs).toISOString().slice(0, 7);
|
|
101
|
+
}
|
|
102
|
+
function markProviderCoolingDown(provider, now = Date.now) {
|
|
103
|
+
providerCooldownUntil.set(provider, now() + PROVIDER_COOLDOWN_MS);
|
|
104
|
+
}
|
|
105
|
+
function clearProviderCooldown(provider) {
|
|
106
|
+
providerCooldownUntil.delete(provider);
|
|
107
|
+
}
|
|
108
|
+
var BASE_BACKOFF_MS = 250;
|
|
109
|
+
function isRetryableForBackoff(status) {
|
|
110
|
+
return status === 429 || status >= 500 && status < 600;
|
|
111
|
+
}
|
|
112
|
+
function estimateTokens(messages, system) {
|
|
113
|
+
let chars = system?.length ?? 0;
|
|
114
|
+
for (const m of messages) chars += contentToText(m.content).length;
|
|
115
|
+
return Math.ceil(chars / 4);
|
|
116
|
+
}
|
|
117
|
+
function sleep(ms, signal) {
|
|
118
|
+
return new Promise((resolve, reject) => {
|
|
119
|
+
const t = setTimeout(resolve, ms);
|
|
120
|
+
if (signal) {
|
|
121
|
+
const onAbort = () => {
|
|
122
|
+
clearTimeout(t);
|
|
123
|
+
reject(new DOMException("Aborted", "AbortError"));
|
|
124
|
+
};
|
|
125
|
+
if (signal.aborted) onAbort();
|
|
126
|
+
else signal.addEventListener("abort", onAbort, { once: true });
|
|
127
|
+
}
|
|
128
|
+
});
|
|
129
|
+
}
|
|
130
|
+
function computeBackoffMs(attempt) {
|
|
131
|
+
const jitter = Math.floor(Math.random() * BACKOFF_JITTER_MAX_MS);
|
|
132
|
+
return Math.min(BACKOFF_BASE_MS * Math.pow(2, attempt) + jitter, BACKOFF_CAP_MS);
|
|
133
|
+
}
|
|
134
|
+
function buildAnthropicRequest(model, messages, opts, env, streaming = false) {
|
|
135
|
+
const sys = systemText(opts, messages);
|
|
136
|
+
const filtered = messages.filter((m) => m.role !== "system");
|
|
137
|
+
const body = {
|
|
138
|
+
model,
|
|
139
|
+
max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
140
|
+
temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
|
|
141
|
+
messages: filtered.map((m) => ({ role: m.role, content: m.content }))
|
|
142
|
+
};
|
|
143
|
+
if (streaming) {
|
|
144
|
+
body.stream = true;
|
|
145
|
+
}
|
|
146
|
+
if (sys) {
|
|
147
|
+
const cache = opts.promptCache ?? sys.length >= 4096;
|
|
148
|
+
body.system = cache ? [{ type: "text", text: sys, cache_control: { type: "ephemeral" } }] : sys;
|
|
149
|
+
}
|
|
150
|
+
if (opts.tools && opts.tools.length > 0) {
|
|
151
|
+
body.tools = opts.tools.map((t) => ({
|
|
152
|
+
name: t.name,
|
|
153
|
+
description: t.description ?? "",
|
|
154
|
+
input_schema: t.parameters
|
|
155
|
+
}));
|
|
156
|
+
const tc = opts.toolChoice ?? "auto";
|
|
157
|
+
body.tool_choice = tc === "auto" ? { type: "auto" } : tc === "none" ? { type: "none" } : { type: "tool", name: tc.name };
|
|
158
|
+
}
|
|
159
|
+
return {
|
|
160
|
+
url: `${env.AI_GATEWAY_BASE_URL}/anthropic/v1/messages`,
|
|
161
|
+
headers: {
|
|
162
|
+
"content-type": "application/json",
|
|
163
|
+
"x-api-key": env.ANTHROPIC_API_KEY,
|
|
164
|
+
"anthropic-version": "2023-06-01",
|
|
165
|
+
"anthropic-beta": "prompt-caching-2024-07-31"
|
|
166
|
+
},
|
|
167
|
+
body: JSON.stringify(body)
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
function buildGeminiRequest(model, messages, opts, env) {
|
|
171
|
+
const sys = systemText(opts, messages);
|
|
172
|
+
const contents = messages.filter((m) => m.role !== "system").map((m) => ({
|
|
173
|
+
role: m.role === "assistant" ? "model" : "user",
|
|
174
|
+
parts: [{ text: contentToText(m.content) }]
|
|
175
|
+
}));
|
|
176
|
+
const body = {
|
|
177
|
+
contents,
|
|
178
|
+
generationConfig: {
|
|
179
|
+
maxOutputTokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
180
|
+
temperature: opts.temperature ?? DEFAULT_TEMPERATURE
|
|
181
|
+
}
|
|
182
|
+
};
|
|
183
|
+
if (sys) {
|
|
184
|
+
body.systemInstruction = { parts: [{ text: sys }] };
|
|
185
|
+
}
|
|
186
|
+
const path = `v1/projects/${env.VERTEX_PROJECT}/locations/${env.VERTEX_LOCATION}/publishers/google/models/${model}:generateContent`;
|
|
187
|
+
return {
|
|
188
|
+
url: `${env.AI_GATEWAY_BASE_URL}/google-vertex-ai/${path}`,
|
|
189
|
+
headers: {
|
|
190
|
+
"content-type": "application/json",
|
|
191
|
+
authorization: `Bearer ${env.VERTEX_ACCESS_TOKEN}`
|
|
192
|
+
},
|
|
193
|
+
body: JSON.stringify(body)
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
function toOpenAiMessages(messages, sys) {
|
|
197
|
+
const out = [];
|
|
198
|
+
if (sys) out.push({ role: "system", content: sys });
|
|
199
|
+
for (const m of messages) {
|
|
200
|
+
if (m.role === "system") continue;
|
|
201
|
+
if (typeof m.content === "string") {
|
|
202
|
+
out.push({ role: m.role, content: m.content });
|
|
203
|
+
continue;
|
|
204
|
+
}
|
|
205
|
+
let text = "";
|
|
206
|
+
const toolCalls = [];
|
|
207
|
+
const results = [];
|
|
208
|
+
for (const b of m.content) {
|
|
209
|
+
if (b.type === "text") text += b.text;
|
|
210
|
+
else if (b.type === "tool_use")
|
|
211
|
+
toolCalls.push({ id: b.id, type: "function", function: { name: b.name, arguments: JSON.stringify(b.input) } });
|
|
212
|
+
else if (b.type === "tool_result") results.push({ tool_use_id: b.tool_use_id, content: b.content });
|
|
213
|
+
}
|
|
214
|
+
if (results.length > 0) {
|
|
215
|
+
for (const r of results) out.push({ role: "tool", tool_call_id: r.tool_use_id, content: r.content });
|
|
216
|
+
if (text) out.push({ role: "user", content: text });
|
|
217
|
+
} else if (toolCalls.length > 0) {
|
|
218
|
+
out.push({ role: "assistant", content: text || null, tool_calls: toolCalls });
|
|
219
|
+
} else {
|
|
220
|
+
out.push({ role: m.role, content: text });
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
return out;
|
|
224
|
+
}
|
|
225
|
+
function openAiTools(opts) {
|
|
226
|
+
if (!opts.tools || opts.tools.length === 0) return void 0;
|
|
227
|
+
return opts.tools.map((t) => ({
|
|
228
|
+
type: "function",
|
|
229
|
+
function: { name: t.name, description: t.description ?? "", parameters: t.parameters }
|
|
230
|
+
}));
|
|
231
|
+
}
|
|
232
|
+
function openAiToolChoice(tc) {
|
|
233
|
+
if (tc === void 0) return void 0;
|
|
234
|
+
if (tc === "auto" || tc === "none") return tc;
|
|
235
|
+
return { type: "function", function: { name: tc.name } };
|
|
236
|
+
}
|
|
237
|
+
function buildGroqRequest(model, messages, opts, env) {
|
|
238
|
+
const sys = systemText(opts, messages);
|
|
239
|
+
const merged = [];
|
|
240
|
+
if (sys) merged.push({ role: "system", content: sys });
|
|
241
|
+
for (const m of messages) if (m.role !== "system") merged.push({ role: m.role, content: contentToText(m.content) });
|
|
242
|
+
return {
|
|
243
|
+
url: `${env.AI_GATEWAY_BASE_URL}/groq/openai/v1/chat/completions`,
|
|
244
|
+
headers: {
|
|
245
|
+
"content-type": "application/json",
|
|
246
|
+
authorization: `Bearer ${env.GROQ_API_KEY}`
|
|
247
|
+
},
|
|
248
|
+
body: JSON.stringify({
|
|
249
|
+
model,
|
|
250
|
+
max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
251
|
+
temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
|
|
252
|
+
messages: merged
|
|
253
|
+
})
|
|
254
|
+
};
|
|
255
|
+
}
|
|
256
|
+
function buildGrokRequest(model, messages, opts, env) {
|
|
257
|
+
if (!env.GROK_API_KEY) {
|
|
258
|
+
throw new ValidationError("GROK_API_KEY required for grok-* model override");
|
|
259
|
+
}
|
|
260
|
+
const sys = systemText(opts, messages);
|
|
261
|
+
const body = {
|
|
262
|
+
model,
|
|
263
|
+
max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
264
|
+
temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
|
|
265
|
+
messages: toOpenAiMessages(messages, sys)
|
|
266
|
+
};
|
|
267
|
+
const tools = openAiTools(opts);
|
|
268
|
+
if (tools) {
|
|
269
|
+
body.tools = tools;
|
|
270
|
+
const tc = openAiToolChoice(opts.toolChoice ?? "auto");
|
|
271
|
+
if (tc !== void 0) body.tool_choice = tc;
|
|
272
|
+
}
|
|
273
|
+
if (model === MODELS.grok.fast) {
|
|
274
|
+
body.reasoning_effort = opts.reasoningEffort ?? "none";
|
|
275
|
+
}
|
|
276
|
+
return {
|
|
277
|
+
url: `${env.AI_GATEWAY_BASE_URL}/grok/v1/chat/completions`,
|
|
278
|
+
headers: {
|
|
279
|
+
"content-type": "application/json",
|
|
280
|
+
authorization: `Bearer ${env.GROK_API_KEY}`
|
|
281
|
+
},
|
|
282
|
+
body: JSON.stringify(body)
|
|
283
|
+
};
|
|
284
|
+
}
|
|
285
|
+
function buildDeepSeekRequest(model, messages, opts, env) {
|
|
286
|
+
if (!env.DEEPSEEK_API_KEY) {
|
|
287
|
+
throw new ValidationError("DEEPSEEK_API_KEY required for workbench tier or deepseek-* model override");
|
|
288
|
+
}
|
|
289
|
+
const sys = systemText(opts, messages);
|
|
290
|
+
const body = {
|
|
291
|
+
model,
|
|
292
|
+
max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
293
|
+
temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
|
|
294
|
+
messages: toOpenAiMessages(messages, sys)
|
|
295
|
+
};
|
|
296
|
+
const tools = openAiTools(opts);
|
|
297
|
+
if (tools) {
|
|
298
|
+
body.tools = tools;
|
|
299
|
+
const tc = openAiToolChoice(opts.toolChoice ?? "auto");
|
|
300
|
+
if (tc !== void 0) body.tool_choice = tc;
|
|
301
|
+
}
|
|
302
|
+
return {
|
|
303
|
+
url: `${env.AI_GATEWAY_BASE_URL}/deepseek/chat/completions`,
|
|
304
|
+
headers: {
|
|
305
|
+
"content-type": "application/json",
|
|
306
|
+
authorization: `Bearer ${env.DEEPSEEK_API_KEY}`
|
|
307
|
+
},
|
|
308
|
+
body: JSON.stringify(body)
|
|
309
|
+
};
|
|
310
|
+
}
|
|
311
|
+
function normalizeAnthropicStop(reason) {
|
|
312
|
+
switch (reason) {
|
|
313
|
+
case "end_turn":
|
|
314
|
+
case "stop_sequence":
|
|
315
|
+
return "end";
|
|
316
|
+
case "tool_use":
|
|
317
|
+
return "tool_use";
|
|
318
|
+
case "max_tokens":
|
|
319
|
+
return "max_tokens";
|
|
320
|
+
default:
|
|
321
|
+
return reason ? "other" : void 0;
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
function parseAnthropic(json) {
|
|
325
|
+
const r = json;
|
|
326
|
+
const toolCalls = (r.content ?? []).filter((c) => c.type === "tool_use" && typeof c.id === "string" && typeof c.name === "string").map((c) => ({ id: c.id, name: c.name, arguments: c.input ?? {} }));
|
|
327
|
+
return {
|
|
328
|
+
content: r.content?.find((c) => c.type === "text")?.text ?? "",
|
|
329
|
+
input: r.usage?.input_tokens ?? 0,
|
|
330
|
+
output: r.usage?.output_tokens ?? 0,
|
|
331
|
+
cacheRead: r.usage?.cache_read_input_tokens ?? 0,
|
|
332
|
+
cacheWrite: r.usage?.cache_creation_input_tokens ?? 0,
|
|
333
|
+
model: r.model,
|
|
334
|
+
toolCalls: toolCalls.length > 0 ? toolCalls : void 0,
|
|
335
|
+
stopReason: normalizeAnthropicStop(r.stop_reason)
|
|
336
|
+
};
|
|
337
|
+
}
|
|
338
|
+
function parseGemini(json) {
|
|
339
|
+
const r = json;
|
|
340
|
+
const text = r.candidates?.[0]?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
|
|
341
|
+
return {
|
|
342
|
+
content: text,
|
|
343
|
+
input: r.usageMetadata?.promptTokenCount ?? 0,
|
|
344
|
+
output: r.usageMetadata?.candidatesTokenCount ?? 0
|
|
345
|
+
};
|
|
346
|
+
}
|
|
347
|
+
function parseGroq(json) {
|
|
348
|
+
const r = json;
|
|
349
|
+
return {
|
|
350
|
+
content: r.choices?.[0]?.message?.content ?? "",
|
|
351
|
+
input: r.usage?.prompt_tokens ?? 0,
|
|
352
|
+
output: r.usage?.completion_tokens ?? 0,
|
|
353
|
+
model: r.model
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
function normalizeOpenAiStop(reason) {
|
|
357
|
+
switch (reason) {
|
|
358
|
+
case "stop":
|
|
359
|
+
return "end";
|
|
360
|
+
case "tool_calls":
|
|
361
|
+
case "function_call":
|
|
362
|
+
return "tool_use";
|
|
363
|
+
case "length":
|
|
364
|
+
return "max_tokens";
|
|
365
|
+
default:
|
|
366
|
+
return reason ? "other" : void 0;
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
function parseToolArgs(raw) {
|
|
370
|
+
if (!raw) return {};
|
|
371
|
+
try {
|
|
372
|
+
const v = JSON.parse(raw);
|
|
373
|
+
return typeof v === "object" && v !== null ? v : {};
|
|
374
|
+
} catch {
|
|
375
|
+
return {};
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
function parseOpenAi(json) {
|
|
379
|
+
const r = json;
|
|
380
|
+
const choice = r.choices?.[0];
|
|
381
|
+
const toolCalls = (choice?.message?.tool_calls ?? []).filter((c) => typeof c.function?.name === "string").map((c, i) => ({
|
|
382
|
+
id: c.id ?? `call_${i}`,
|
|
383
|
+
name: c.function.name,
|
|
384
|
+
arguments: parseToolArgs(c.function?.arguments)
|
|
385
|
+
}));
|
|
386
|
+
return {
|
|
387
|
+
content: choice?.message?.content ?? "",
|
|
388
|
+
input: r.usage?.prompt_tokens ?? 0,
|
|
389
|
+
output: r.usage?.completion_tokens ?? 0,
|
|
390
|
+
model: r.model,
|
|
391
|
+
toolCalls: toolCalls.length > 0 ? toolCalls : void 0,
|
|
392
|
+
stopReason: normalizeOpenAiStop(choice?.finish_reason)
|
|
393
|
+
};
|
|
394
|
+
}
|
|
395
|
+
async function callWithBackoff(provider, request, fetchImpl, signal, logger, nowFn) {
|
|
396
|
+
function exhaustAndThrow(err) {
|
|
397
|
+
markProviderCoolingDown(provider, nowFn ?? Date.now);
|
|
398
|
+
throw err;
|
|
399
|
+
}
|
|
400
|
+
let lastErr;
|
|
401
|
+
for (let attempt = 1; attempt <= PER_PROVIDER_MAX_ATTEMPTS; attempt++) {
|
|
402
|
+
try {
|
|
403
|
+
const response = await fetchImpl(request.url, {
|
|
404
|
+
method: "POST",
|
|
405
|
+
headers: request.headers,
|
|
406
|
+
body: request.body,
|
|
407
|
+
signal
|
|
408
|
+
});
|
|
409
|
+
if (!response.ok) {
|
|
410
|
+
const text = await response.text().catch(() => "");
|
|
411
|
+
const retryable = isRetryableForBackoff(response.status);
|
|
412
|
+
const err = {
|
|
413
|
+
provider,
|
|
414
|
+
status: response.status,
|
|
415
|
+
retryable,
|
|
416
|
+
message: `${provider} ${String(response.status)}: ${text.slice(0, 300)}`
|
|
417
|
+
};
|
|
418
|
+
logger?.warn?.("llm.provider.error", { provider, status: response.status, attempt });
|
|
419
|
+
if (!err.retryable || attempt === PER_PROVIDER_MAX_ATTEMPTS) {
|
|
420
|
+
if (err.retryable) exhaustAndThrow(err);
|
|
421
|
+
throw err;
|
|
422
|
+
}
|
|
423
|
+
lastErr = err;
|
|
424
|
+
} else {
|
|
425
|
+
const gatewayRequestId = response.headers.get("cf-aig-request-id") ?? void 0;
|
|
426
|
+
clearProviderCooldown(provider);
|
|
427
|
+
return { json: await response.json(), gatewayRequestId, attempts: attempt };
|
|
428
|
+
}
|
|
429
|
+
} catch (e) {
|
|
430
|
+
if (e instanceof DOMException && e.name === "AbortError") throw e;
|
|
431
|
+
if (typeof e === "object" && e !== null && "retryable" in e) {
|
|
432
|
+
const err = e;
|
|
433
|
+
if (!err.retryable || attempt === PER_PROVIDER_MAX_ATTEMPTS) {
|
|
434
|
+
if (err.retryable) exhaustAndThrow(err);
|
|
435
|
+
throw err;
|
|
436
|
+
}
|
|
437
|
+
lastErr = err;
|
|
438
|
+
} else {
|
|
439
|
+
const err = {
|
|
440
|
+
provider,
|
|
441
|
+
status: 0,
|
|
442
|
+
retryable: true,
|
|
443
|
+
message: e instanceof Error ? e.message : String(e)
|
|
444
|
+
};
|
|
445
|
+
if (attempt === PER_PROVIDER_MAX_ATTEMPTS) exhaustAndThrow(err);
|
|
446
|
+
lastErr = err;
|
|
447
|
+
}
|
|
448
|
+
}
|
|
449
|
+
const backoffMs = computeBackoffMs(attempt - 1);
|
|
450
|
+
await sleep(backoffMs, signal);
|
|
451
|
+
}
|
|
452
|
+
markProviderCoolingDown(provider, nowFn ?? Date.now);
|
|
453
|
+
throw lastErr ?? { provider, status: 0, retryable: false, message: "exhausted" };
|
|
454
|
+
}
|
|
455
|
+
function isProviderError(err) {
|
|
456
|
+
return typeof err === "object" && err !== null && typeof err.status === "number" && typeof err.message === "string" && typeof err.provider === "string";
|
|
457
|
+
}
|
|
458
|
+
var TOOL_CAPABLE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic", "grok", "deepseek"]);
|
|
459
|
+
function plan(tier, opts, tokenEstimate) {
|
|
460
|
+
if (opts.model) {
|
|
461
|
+
const m = opts.model;
|
|
462
|
+
if (m.startsWith("claude")) return { primary: { provider: "anthropic", model: m } };
|
|
463
|
+
if (m.startsWith("gemini")) return { primary: { provider: "gemini", model: m } };
|
|
464
|
+
if (m.startsWith("grok")) return { primary: { provider: "grok", model: m } };
|
|
465
|
+
if (m.startsWith("deepseek")) return { primary: { provider: "deepseek", model: m } };
|
|
466
|
+
return { primary: { provider: "groq", model: m } };
|
|
467
|
+
}
|
|
468
|
+
const longContext = tokenEstimate >= (opts.longContextThreshold ?? DEFAULT_LONG_CONTEXT_THRESHOLD);
|
|
469
|
+
switch (tier) {
|
|
470
|
+
case "workbench":
|
|
471
|
+
return {
|
|
472
|
+
primary: { provider: "deepseek", model: MODELS.deepseek.workbench },
|
|
473
|
+
fallback: { provider: "groq", model: MODELS.groq.verifier }
|
|
474
|
+
};
|
|
475
|
+
case "verifier":
|
|
476
|
+
return { primary: { provider: "groq", model: MODELS.groq.verifier } };
|
|
477
|
+
case "smart":
|
|
478
|
+
return longContext ? {
|
|
479
|
+
primary: { provider: "gemini", model: MODELS.gemini.smart },
|
|
480
|
+
fallback: { provider: "anthropic", model: MODELS.anthropic.smart }
|
|
481
|
+
} : {
|
|
482
|
+
primary: { provider: "anthropic", model: MODELS.anthropic.smart },
|
|
483
|
+
fallback: { provider: "gemini", model: MODELS.gemini.smart }
|
|
484
|
+
};
|
|
485
|
+
case "fast":
|
|
486
|
+
return {
|
|
487
|
+
primary: { provider: "grok", model: MODELS.grok.fast },
|
|
488
|
+
fallback: { provider: "anthropic", model: MODELS.anthropic.fast }
|
|
489
|
+
};
|
|
490
|
+
case "balanced":
|
|
491
|
+
default:
|
|
492
|
+
return longContext ? {
|
|
493
|
+
primary: { provider: "gemini", model: MODELS.gemini.smart },
|
|
494
|
+
fallback: { provider: "anthropic", model: MODELS.anthropic.balanced }
|
|
495
|
+
} : {
|
|
496
|
+
primary: { provider: "anthropic", model: MODELS.anthropic.balanced },
|
|
497
|
+
fallback: { provider: "gemini", model: MODELS.gemini.smart }
|
|
498
|
+
};
|
|
499
|
+
}
|
|
500
|
+
}
|
|
501
|
+
function buildAigMetadata(opts) {
|
|
502
|
+
const meta = {};
|
|
503
|
+
if (opts.project) meta.project = opts.project;
|
|
504
|
+
if (opts.workload) meta.workload = opts.workload;
|
|
505
|
+
if (opts.actor) meta.actor = opts.actor;
|
|
506
|
+
if (opts.runId) meta.runId = opts.runId;
|
|
507
|
+
return Object.keys(meta).length > 0 ? JSON.stringify(meta) : void 0;
|
|
508
|
+
}
|
|
509
|
+
async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
|
|
510
|
+
let req;
|
|
511
|
+
switch (leg.provider) {
|
|
512
|
+
case "anthropic":
|
|
513
|
+
req = buildAnthropicRequest(leg.model, messages, opts, env);
|
|
514
|
+
break;
|
|
515
|
+
case "gemini":
|
|
516
|
+
req = buildGeminiRequest(leg.model, messages, opts, env);
|
|
517
|
+
break;
|
|
518
|
+
case "groq":
|
|
519
|
+
req = buildGroqRequest(leg.model, messages, opts, env);
|
|
520
|
+
break;
|
|
521
|
+
case "grok":
|
|
522
|
+
req = buildGrokRequest(leg.model, messages, opts, env);
|
|
523
|
+
break;
|
|
524
|
+
case "deepseek":
|
|
525
|
+
req = buildDeepSeekRequest(leg.model, messages, opts, env);
|
|
526
|
+
break;
|
|
527
|
+
}
|
|
528
|
+
const aigMetadata = buildAigMetadata(opts);
|
|
529
|
+
if (aigMetadata) req.headers["cf-aig-metadata"] = aigMetadata;
|
|
530
|
+
const { json, gatewayRequestId, attempts } = await callWithBackoff(
|
|
531
|
+
leg.provider,
|
|
532
|
+
req,
|
|
533
|
+
fetchImpl,
|
|
534
|
+
opts.signal,
|
|
535
|
+
logger,
|
|
536
|
+
nowFn
|
|
537
|
+
);
|
|
538
|
+
switch (leg.provider) {
|
|
539
|
+
case "anthropic":
|
|
540
|
+
return { parsed: parseAnthropic(json), gatewayRequestId, attempts };
|
|
541
|
+
case "gemini":
|
|
542
|
+
return { parsed: parseGemini(json), gatewayRequestId, attempts };
|
|
543
|
+
case "groq":
|
|
544
|
+
return { parsed: parseGroq(json), gatewayRequestId, attempts };
|
|
545
|
+
case "grok":
|
|
546
|
+
return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
|
|
547
|
+
case "deepseek":
|
|
548
|
+
return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
async function complete(messages, env, opts = {}, deps = {}) {
|
|
552
|
+
if (messages.length === 0) {
|
|
553
|
+
throw new ValidationError("messages must not be empty");
|
|
554
|
+
}
|
|
555
|
+
if (!env.AI_GATEWAY_BASE_URL) {
|
|
556
|
+
throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
|
|
557
|
+
}
|
|
558
|
+
const fetchImpl = deps.fetch ?? fetch;
|
|
559
|
+
const now = deps.now ?? (() => Date.now());
|
|
560
|
+
const logger = deps.logger;
|
|
561
|
+
const startedAt = now();
|
|
562
|
+
const tier = opts.tier ?? "balanced";
|
|
563
|
+
const system = systemText(opts, messages);
|
|
564
|
+
const tokenEstimate = estimateTokens(messages, system);
|
|
565
|
+
const route = plan(tier, opts, tokenEstimate);
|
|
566
|
+
const kv = env.LLM_COST_KV;
|
|
567
|
+
const todayKey = `llm:daily-cost:${isoDate(now())}`;
|
|
568
|
+
const monthKey = `llm:monthly-cost:${isoMonth(now())}`;
|
|
569
|
+
if (kv) {
|
|
570
|
+
if (opts.dailyCapUsd !== void 0) {
|
|
571
|
+
const raw = await kv.get(todayKey).catch(() => null);
|
|
572
|
+
const spent = parseFloat(raw ?? "0");
|
|
573
|
+
if (spent >= opts.dailyCapUsd) {
|
|
574
|
+
return toErrorResponse(
|
|
575
|
+
new RateLimitError("LLM_DAILY_CAP_EXCEEDED", {
|
|
576
|
+
spentUsd: spent,
|
|
577
|
+
dailyCapUsd: opts.dailyCapUsd
|
|
578
|
+
})
|
|
579
|
+
);
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
if (opts.monthlyCapUsd !== void 0) {
|
|
583
|
+
const raw = await kv.get(monthKey).catch(() => null);
|
|
584
|
+
const spent = parseFloat(raw ?? "0");
|
|
585
|
+
if (spent >= opts.monthlyCapUsd) {
|
|
586
|
+
return toErrorResponse(
|
|
587
|
+
new RateLimitError("LLM_MONTHLY_CAP_EXCEEDED", {
|
|
588
|
+
spentUsd: spent,
|
|
589
|
+
monthlyCapUsd: opts.monthlyCapUsd
|
|
590
|
+
})
|
|
591
|
+
);
|
|
592
|
+
}
|
|
593
|
+
}
|
|
594
|
+
}
|
|
595
|
+
const attemptLog = [];
|
|
596
|
+
let routeLegs = [route.primary, route.fallback].filter(Boolean);
|
|
597
|
+
if (opts.tools && opts.tools.length > 0) {
|
|
598
|
+
routeLegs = routeLegs.filter((l) => TOOL_CAPABLE_PROVIDERS.has(l.provider));
|
|
599
|
+
if (routeLegs.length === 0) {
|
|
600
|
+
throw new ValidationError(
|
|
601
|
+
`tool-calling requires a tool-capable provider (${[...TOOL_CAPABLE_PROVIDERS].join(", ")}); tier '${tier}' has none \u2014 use tier fast/balanced/smart or a claude-* model override`
|
|
602
|
+
);
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
for (const [legIndex, leg] of routeLegs.entries()) {
|
|
606
|
+
if (isProviderCoolingDown(leg.provider, now)) {
|
|
607
|
+
logger?.warn?.("llm.provider.coolingDown", { provider: leg.provider });
|
|
608
|
+
attemptLog.push({ provider: leg.provider, message: "skipped: cooling down" });
|
|
609
|
+
continue;
|
|
610
|
+
}
|
|
611
|
+
if (opts.signal?.aborted) {
|
|
612
|
+
return toErrorResponse(
|
|
613
|
+
new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
|
|
614
|
+
);
|
|
615
|
+
}
|
|
616
|
+
try {
|
|
617
|
+
const result = await callOne(leg, messages, opts, env, fetchImpl, logger, now);
|
|
618
|
+
if (!result.parsed.content && !(result.parsed.toolCalls && result.parsed.toolCalls.length > 0)) {
|
|
619
|
+
throw { provider: leg.provider, status: 200, retryable: false, message: "empty content" };
|
|
620
|
+
}
|
|
621
|
+
logger?.info?.("llm.complete", {
|
|
622
|
+
provider: leg.provider,
|
|
623
|
+
model: leg.model,
|
|
624
|
+
tier,
|
|
625
|
+
tokenEstimate,
|
|
626
|
+
attempts: result.attempts,
|
|
627
|
+
runId: opts.runId,
|
|
628
|
+
project: opts.project,
|
|
629
|
+
actor: opts.actor,
|
|
630
|
+
workload: opts.workload
|
|
631
|
+
});
|
|
632
|
+
const llmResult = {
|
|
633
|
+
content: result.parsed.content,
|
|
634
|
+
provider: leg.provider,
|
|
635
|
+
model: result.parsed.model ?? leg.model,
|
|
636
|
+
tier,
|
|
637
|
+
tokens: {
|
|
638
|
+
input: result.parsed.input,
|
|
639
|
+
output: result.parsed.output,
|
|
640
|
+
cacheRead: result.parsed.cacheRead,
|
|
641
|
+
cacheWrite: result.parsed.cacheWrite
|
|
642
|
+
},
|
|
643
|
+
latency: now() - startedAt,
|
|
644
|
+
attempts: result.attempts,
|
|
645
|
+
gatewayRequestId: result.gatewayRequestId,
|
|
646
|
+
stopReason: result.parsed.stopReason,
|
|
647
|
+
toolCalls: result.parsed.toolCalls
|
|
648
|
+
};
|
|
649
|
+
const costUsd = estimateCostUsd(llmResult.tokens, llmResult.model);
|
|
650
|
+
if (opts.maxCostUsd !== void 0 && costUsd > opts.maxCostUsd) {
|
|
651
|
+
if (kv && (opts.dailyCapUsd !== void 0 || opts.monthlyCapUsd !== void 0)) {
|
|
652
|
+
await recordOrgCostUsage(kv, todayKey, monthKey, costUsd, opts);
|
|
653
|
+
}
|
|
654
|
+
return toErrorResponse(
|
|
655
|
+
new RateLimitError("LLM_COST_CAP_EXCEEDED", {
|
|
656
|
+
costUsd,
|
|
657
|
+
maxCostUsd: opts.maxCostUsd,
|
|
658
|
+
model: llmResult.model,
|
|
659
|
+
tokens: llmResult.tokens
|
|
660
|
+
})
|
|
661
|
+
);
|
|
662
|
+
}
|
|
663
|
+
if (kv && (opts.dailyCapUsd !== void 0 || opts.monthlyCapUsd !== void 0)) {
|
|
664
|
+
await recordOrgCostUsage(kv, todayKey, monthKey, costUsd, opts);
|
|
665
|
+
}
|
|
666
|
+
if (deps.onRecord && opts.ledger) {
|
|
667
|
+
const row = {
|
|
668
|
+
...opts.ledger,
|
|
669
|
+
model: llmResult.model,
|
|
670
|
+
provider: llmResult.provider,
|
|
671
|
+
tier: llmResult.tier,
|
|
672
|
+
inputTokens: llmResult.tokens.input,
|
|
673
|
+
outputTokens: llmResult.tokens.output,
|
|
674
|
+
cacheReadTokens: llmResult.tokens.cacheRead ?? 0,
|
|
675
|
+
cacheWriteTokens: llmResult.tokens.cacheWrite ?? 0,
|
|
676
|
+
latencyMs: llmResult.latency,
|
|
677
|
+
costUsd,
|
|
678
|
+
yyyyMm: isoMonth(now())
|
|
679
|
+
};
|
|
680
|
+
deps.onRecord(row).catch((e) => {
|
|
681
|
+
logger?.warn?.("llm.onRecord.error", { message: e instanceof Error ? e.message : String(e) });
|
|
682
|
+
});
|
|
683
|
+
}
|
|
684
|
+
return { data: llmResult, error: null };
|
|
685
|
+
} catch (e) {
|
|
686
|
+
if (e instanceof DOMException && e.name === "AbortError") {
|
|
687
|
+
return toErrorResponse(
|
|
688
|
+
new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
|
|
689
|
+
);
|
|
690
|
+
}
|
|
691
|
+
if (isProviderError(e)) {
|
|
692
|
+
attemptLog.push({ provider: e.provider, status: e.status, message: e.message });
|
|
693
|
+
if (e.status === 429 && legIndex === routeLegs.length - 1) {
|
|
694
|
+
return toErrorResponse(
|
|
695
|
+
new RateLimitError(`llm rate limited on ${e.provider}`, { attempts: attemptLog })
|
|
696
|
+
);
|
|
697
|
+
}
|
|
698
|
+
logger?.warn?.("llm.leg.failed", { provider: leg.provider, status: e.status });
|
|
699
|
+
continue;
|
|
700
|
+
}
|
|
701
|
+
attemptLog.push({ provider: leg.provider, message: e instanceof Error ? e.message : String(e) });
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
return toErrorResponse(
|
|
705
|
+
new InternalError("LLM_ALL_PROVIDERS_FAILED", { attempts: attemptLog, tier, tokenEstimate })
|
|
706
|
+
);
|
|
707
|
+
}
|
|
708
|
+
async function* completionStream(messages, env, opts = {}) {
|
|
709
|
+
if (messages.length === 0) {
|
|
710
|
+
throw new ValidationError("messages must not be empty");
|
|
711
|
+
}
|
|
712
|
+
if (!env.AI_GATEWAY_BASE_URL) {
|
|
713
|
+
throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
|
|
714
|
+
}
|
|
715
|
+
const deps = opts.deps ?? {};
|
|
716
|
+
const fetchImpl = deps.fetch ?? fetch;
|
|
717
|
+
const now = deps.now ?? (() => Date.now());
|
|
718
|
+
const logger = deps.logger;
|
|
719
|
+
const startedAt = now();
|
|
720
|
+
const tier = opts.tier ?? "balanced";
|
|
721
|
+
const system = systemText(opts, messages);
|
|
722
|
+
const tokenEstimate = estimateTokens(messages, system);
|
|
723
|
+
const route = plan(tier, opts, tokenEstimate);
|
|
724
|
+
const streamLeg = route.primary.provider === "grok" && !env.GROK_API_KEY && route.fallback?.provider === "anthropic" ? route.fallback : route.primary;
|
|
725
|
+
if (streamLeg.provider !== "anthropic") {
|
|
726
|
+
const result = await complete(messages, env, opts, deps);
|
|
727
|
+
if (result.error !== null || result.data === null) {
|
|
728
|
+
throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
|
|
729
|
+
}
|
|
730
|
+
yield result.data.content;
|
|
731
|
+
return result.data;
|
|
732
|
+
}
|
|
733
|
+
if (isProviderCoolingDown(streamLeg.provider, now)) {
|
|
734
|
+
logger?.warn?.("llm.provider.coolingDown", { provider: streamLeg.provider });
|
|
735
|
+
const result = await complete(messages, env, opts, deps);
|
|
736
|
+
if (result.error !== null || result.data === null) {
|
|
737
|
+
throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
|
|
738
|
+
}
|
|
739
|
+
yield result.data.content;
|
|
740
|
+
return result.data;
|
|
741
|
+
}
|
|
742
|
+
const req = buildAnthropicRequest(streamLeg.model, messages, opts, env, true);
|
|
743
|
+
const streamAigMetadata = buildAigMetadata(opts);
|
|
744
|
+
if (streamAigMetadata) req.headers["cf-aig-metadata"] = streamAigMetadata;
|
|
745
|
+
let response;
|
|
746
|
+
try {
|
|
747
|
+
response = await fetchImpl(req.url, {
|
|
748
|
+
method: "POST",
|
|
749
|
+
headers: req.headers,
|
|
750
|
+
body: req.body,
|
|
751
|
+
// Fall back to a 60 s default when the caller provides no signal — prevents
|
|
752
|
+
// a hung provider connection from consuming the Worker's wall-clock budget.
|
|
753
|
+
signal: opts.signal ?? AbortSignal.timeout(6e4)
|
|
754
|
+
});
|
|
755
|
+
} catch (e) {
|
|
756
|
+
if (e instanceof DOMException && e.name === "AbortError") {
|
|
757
|
+
throw new InternalError("llm call aborted", {
|
|
758
|
+
provider: streamLeg.provider,
|
|
759
|
+
model: streamLeg.model
|
|
760
|
+
});
|
|
761
|
+
}
|
|
762
|
+
throw new InternalError("llm stream fetch failed", {
|
|
763
|
+
message: e instanceof Error ? e.message : String(e)
|
|
764
|
+
});
|
|
765
|
+
}
|
|
766
|
+
if (!response.ok) {
|
|
767
|
+
const text = await response.text().catch(() => "");
|
|
768
|
+
const retryable = isRetryableForBackoff(response.status);
|
|
769
|
+
if (retryable && response.status === 429) {
|
|
770
|
+
markProviderCoolingDown(streamLeg.provider, now);
|
|
771
|
+
}
|
|
772
|
+
const result = await complete(messages, env, opts, deps);
|
|
773
|
+
if (result.error !== null || result.data === null) {
|
|
774
|
+
throw new InternalError("LLM_ALL_PROVIDERS_FAILED", {
|
|
775
|
+
streamError: `${streamLeg.provider} ${String(response.status)}: ${text.slice(0, 300)}`,
|
|
776
|
+
error: result.error
|
|
777
|
+
});
|
|
778
|
+
}
|
|
779
|
+
yield result.data.content;
|
|
780
|
+
return result.data;
|
|
781
|
+
}
|
|
782
|
+
if (!response.body) {
|
|
783
|
+
throw new InternalError("llm stream response body is null", {
|
|
784
|
+
provider: streamLeg.provider
|
|
785
|
+
});
|
|
786
|
+
}
|
|
787
|
+
const decoder = new TextDecoder();
|
|
788
|
+
let accumulatedText = "";
|
|
789
|
+
let inputTokens = 0;
|
|
790
|
+
let outputTokens = 0;
|
|
791
|
+
let cacheRead = 0;
|
|
792
|
+
let cacheWrite = 0;
|
|
793
|
+
let modelName;
|
|
794
|
+
const toolBlocks = /* @__PURE__ */ new Map();
|
|
795
|
+
let streamStopReason;
|
|
796
|
+
const gatewayRequestId = response.headers.get("cf-aig-request-id") ?? void 0;
|
|
797
|
+
const reader = response.body.getReader();
|
|
798
|
+
let buffer = "";
|
|
799
|
+
try {
|
|
800
|
+
while (true) {
|
|
801
|
+
const { done, value } = await reader.read();
|
|
802
|
+
if (done) break;
|
|
803
|
+
buffer += decoder.decode(value, { stream: true });
|
|
804
|
+
const lines = buffer.split("\n");
|
|
805
|
+
buffer = lines.pop() ?? "";
|
|
806
|
+
for (const line of lines) {
|
|
807
|
+
if (!line.startsWith("data: ")) continue;
|
|
808
|
+
const data = line.slice(6).trim();
|
|
809
|
+
if (data === "[DONE]") break;
|
|
810
|
+
let event;
|
|
811
|
+
try {
|
|
812
|
+
event = JSON.parse(data);
|
|
813
|
+
} catch {
|
|
814
|
+
continue;
|
|
815
|
+
}
|
|
816
|
+
switch (event.type) {
|
|
817
|
+
case "message_start":
|
|
818
|
+
inputTokens = event.message?.usage?.input_tokens ?? 0;
|
|
819
|
+
cacheRead = event.message?.usage?.cache_read_input_tokens ?? 0;
|
|
820
|
+
cacheWrite = event.message?.usage?.cache_creation_input_tokens ?? 0;
|
|
821
|
+
modelName = event.message?.model;
|
|
822
|
+
break;
|
|
823
|
+
case "content_block_start":
|
|
824
|
+
if (event.content_block?.type === "tool_use" && typeof event.index === "number" && typeof event.content_block.id === "string" && typeof event.content_block.name === "string") {
|
|
825
|
+
toolBlocks.set(event.index, { id: event.content_block.id, name: event.content_block.name, json: "" });
|
|
826
|
+
}
|
|
827
|
+
break;
|
|
828
|
+
case "content_block_delta":
|
|
829
|
+
if (event.delta?.type === "text_delta" && typeof event.delta.text === "string") {
|
|
830
|
+
accumulatedText += event.delta.text;
|
|
831
|
+
yield event.delta.text;
|
|
832
|
+
} else if (event.delta?.type === "input_json_delta" && typeof event.delta.partial_json === "string" && typeof event.index === "number") {
|
|
833
|
+
const block = toolBlocks.get(event.index);
|
|
834
|
+
if (block) block.json += event.delta.partial_json;
|
|
835
|
+
}
|
|
836
|
+
break;
|
|
837
|
+
case "message_delta":
|
|
838
|
+
outputTokens = event.usage?.output_tokens ?? outputTokens;
|
|
839
|
+
if (event.delta?.stop_reason) streamStopReason = normalizeAnthropicStop(event.delta.stop_reason);
|
|
840
|
+
break;
|
|
841
|
+
default:
|
|
842
|
+
break;
|
|
843
|
+
}
|
|
844
|
+
}
|
|
845
|
+
}
|
|
846
|
+
} finally {
|
|
847
|
+
reader.releaseLock();
|
|
848
|
+
}
|
|
849
|
+
clearProviderCooldown(streamLeg.provider);
|
|
850
|
+
logger?.info?.("llm.completionStream", {
|
|
851
|
+
provider: streamLeg.provider,
|
|
852
|
+
model: streamLeg.model,
|
|
853
|
+
tier,
|
|
854
|
+
tokenEstimate,
|
|
855
|
+
runId: opts.runId,
|
|
856
|
+
project: opts.project,
|
|
857
|
+
actor: opts.actor,
|
|
858
|
+
workload: opts.workload
|
|
859
|
+
});
|
|
860
|
+
const toolCalls = [...toolBlocks.values()].map((b) => ({
|
|
861
|
+
id: b.id,
|
|
862
|
+
name: b.name,
|
|
863
|
+
arguments: parseToolArgs(b.json)
|
|
864
|
+
}));
|
|
865
|
+
return {
|
|
866
|
+
content: accumulatedText,
|
|
867
|
+
provider: streamLeg.provider,
|
|
868
|
+
model: modelName ?? streamLeg.model,
|
|
869
|
+
tier,
|
|
870
|
+
tokens: { input: inputTokens, output: outputTokens, cacheRead, cacheWrite },
|
|
871
|
+
latency: now() - startedAt,
|
|
872
|
+
attempts: 1,
|
|
873
|
+
gatewayRequestId,
|
|
874
|
+
stopReason: streamStopReason,
|
|
875
|
+
toolCalls: toolCalls.length > 0 ? toolCalls : void 0
|
|
876
|
+
};
|
|
877
|
+
}
|
|
878
|
+
function assertGrounding(response, sources) {
|
|
879
|
+
if (sources.length === 0) return true;
|
|
880
|
+
const WINDOW = 5;
|
|
881
|
+
const responseTokens = response.split(/\s+/).filter((t) => t.length > 0);
|
|
882
|
+
if (responseTokens.length < WINDOW) return false;
|
|
883
|
+
const sourceNgrams = /* @__PURE__ */ new Set();
|
|
884
|
+
for (const source of sources) {
|
|
885
|
+
const tokens = source.split(/\s+/).filter((t) => t.length > 0);
|
|
886
|
+
for (let i = 0; i <= tokens.length - WINDOW; i++) {
|
|
887
|
+
const ngram = tokens.slice(i, i + WINDOW).join(" ");
|
|
888
|
+
sourceNgrams.add(ngram);
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
if (sourceNgrams.size === 0) return false;
|
|
892
|
+
for (let i = 0; i <= responseTokens.length - WINDOW; i++) {
|
|
893
|
+
const ngram = responseTokens.slice(i, i + WINDOW).join(" ");
|
|
894
|
+
if (sourceNgrams.has(ngram)) return true;
|
|
895
|
+
}
|
|
896
|
+
return false;
|
|
897
|
+
}
|
|
898
|
+
export {
|
|
899
|
+
BASE_BACKOFF_MS,
|
|
900
|
+
MODELS,
|
|
901
|
+
MODEL_PRICE_PER_1M,
|
|
902
|
+
PROVIDER_COOLDOWN_MS,
|
|
903
|
+
assertGrounding,
|
|
904
|
+
clearProviderCooldown,
|
|
905
|
+
complete,
|
|
906
|
+
completionStream,
|
|
907
|
+
isProviderCoolingDown,
|
|
908
|
+
markProviderCoolingDown
|
|
909
|
+
};
|
|
910
|
+
//# sourceMappingURL=index.mjs.map
|