agent-accelerator 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +83 -112
- package/{SYSTEM_PROMPT_AGENT.md → examples/prompts/SYSTEM_PROMPT_AGENT.md} +1 -1
- package/package.json +14 -9
- package/src/agent/agent.ts +191 -71
- package/src/agent/context.ts +1 -0
- package/src/agent/delegation.ts +61 -12
- package/src/agent/loop.ts +71 -48
- package/src/data/README.md +6 -6
- package/src/index.ts +130 -45
- package/src/models/catalog-cache.ts +60 -9
- package/src/models/catalog.ts +53 -7
- package/src/providers/google.ts +926 -0
- package/src/providers/openai-compat.ts +1147 -0
- package/src/providers/openai.ts +959 -0
- package/src/providers/openrouter-responses.ts +949 -0
- package/src/providers/openrouter.ts +1037 -0
- package/src/{ai-sdk → providers}/registry.ts +43 -55
- package/src/providers.ts +490 -0
- package/src/streaming/sse-parser.ts +6 -4
- package/src/tools/executor.ts +19 -6
- package/src/tools/schema.ts +21 -11
- package/src/types/agent.ts +8 -1
- package/src/types/core.ts +1 -1
- package/src/types/message.ts +5 -0
- package/src/types/model.ts +3 -7
- package/src/types/provider-payloads.ts +2 -84
- package/src/types/tool.ts +6 -0
- package/src/update-models.ts +56 -0
- package/src/utils/cache.ts +1 -1
- package/src/utils/documents.ts +517 -0
- package/src/utils/env.ts +0 -7
- package/src/{ai-sdk → utils}/errors.ts +61 -2
- package/src/utils/headers.ts +10 -20
- package/src/utils/media.ts +5 -2
- package/src/utils/retry.ts +89 -0
- package/src/utils/serialization.ts +15 -0
- package/src/ai-sdk/converters.ts +0 -342
- package/src/ai-sdk/executor.ts +0 -454
- package/src/ai-sdk/index.ts +0 -55
- package/src/ai-sdk/model-provider.ts +0 -303
- package/src/ai-sdk/options.ts +0 -306
- package/src/ai-sdk/provider.ts +0 -415
- package/src/tokens/counter.ts +0 -136
- /package/{SYSTEM_PROMPT.md → examples/prompts/SYSTEM_PROMPT.md} +0 -0
- /package/{SYSTEM_PROMPT_TOOLS.md → examples/prompts/SYSTEM_PROMPT_TOOLS.md} +0 -0
|
@@ -0,0 +1,1147 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Generic OpenAI-compatible Chat Completions provider
|
|
3
|
+
* (`POST {baseUrl}/chat/completions`, streaming on the same endpoint).
|
|
4
|
+
*
|
|
5
|
+
* Serves EVERY custom prefix (`groq/…`, `ollama/…`, `cerebras/…`, …) with one
|
|
6
|
+
* strict-subset implementation of the OpenAI Chat Completions wire — no SDK
|
|
7
|
+
* transport anywhere in this file.
|
|
8
|
+
*
|
|
9
|
+
* All endpoint-specific HTTP, headers, auth, request construction,
|
|
10
|
+
* response/SSE parsing, and wire transformations live HERE — never in the
|
|
11
|
+
* canonical layer (`src/providers.ts`).
|
|
12
|
+
*
|
|
13
|
+
* Strictness contract (differs from the old lenient passthrough on purpose):
|
|
14
|
+
* - Requests carry ONLY standard Chat Completions fields (`model`, `messages`,
|
|
15
|
+
* `tools`, `tool_choice`, `stream`). No `reasoning`, `service_tier`,
|
|
16
|
+
* `session_id`, `prompt_cache_key`, or other extras: strict endpoints
|
|
17
|
+
* (groq, ollama, …) fail unknown body properties, so anything unsupported
|
|
18
|
+
* is warn+drop, never sent.
|
|
19
|
+
* - `text`, `image_url`, and `input_audio` parts are sent; `video`/`file`
|
|
20
|
+
* parts fail fast with a one-line error naming the remedy (files point at
|
|
21
|
+
* `convertDocumentToMarkdown`). The old transport let these through to
|
|
22
|
+
* confusing provider 400s.
|
|
23
|
+
* - Prior-turn reasoning is NOT resent (chat history is messages + tool calls
|
|
24
|
+
* only), except echoed Gemini thought signatures (see below).
|
|
25
|
+
* - Gemini models served through OpenAI-compatible endpoints keep working:
|
|
26
|
+
* `extra_content.google.thought_signature` on tool calls is captured into
|
|
27
|
+
* the canonical `thoughtSignature` and echoed back verbatim on the next
|
|
28
|
+
* turn (otherwise the endpoint 400s on missing signatures). Echoes happen
|
|
29
|
+
* ONLY when a previous turn produced a signature — other endpoints never
|
|
30
|
+
* see the field.
|
|
31
|
+
*
|
|
32
|
+
* Notes:
|
|
33
|
+
* - STATELESS per request: every turn sends the full canonical `messages`
|
|
34
|
+
* array explicitly. Session affinity is headers-only best effort
|
|
35
|
+
* (`x-session-id`); bodies carry no affinity key.
|
|
36
|
+
* - IDs are provider-generated and echoed verbatim (`tool_call_id`). The
|
|
37
|
+
* `call_${random}` fallback in parsers is a local canonical correlation id
|
|
38
|
+
* only; it is never sent.
|
|
39
|
+
* - Local endpoints (localhost / loopback / RFC-1918 / *.local) may omit the
|
|
40
|
+
* API key (no `Authorization` header is sent then). Any other endpoint
|
|
41
|
+
* without a key fails fast instead of sending a dummy bearer into a 401.
|
|
42
|
+
*/
|
|
43
|
+
import type {
|
|
44
|
+
Provider,
|
|
45
|
+
ProviderId,
|
|
46
|
+
ModelSpec,
|
|
47
|
+
ProviderRequestOptions,
|
|
48
|
+
ProviderGenerateResult,
|
|
49
|
+
ProviderRawData,
|
|
50
|
+
} from "../types/model.ts";
|
|
51
|
+
import type { ProviderContext, ContentPart } from "../types/message.ts";
|
|
52
|
+
import type { StandardToolDeclaration, ToolCallRecord } from "../types/tool.ts";
|
|
53
|
+
import type { TokenUsage, ThinkingConfig } from "../types/core.ts";
|
|
54
|
+
import { AssistantMessageEventStream } from "../streaming/event-stream.ts";
|
|
55
|
+
import { SSEParser } from "../streaming/sse-parser.ts";
|
|
56
|
+
import { AgentResponse } from "../types/response.ts";
|
|
57
|
+
import { getApiKey, getEnv } from "../utils/env.ts";
|
|
58
|
+
import { buildSessionHeaders } from "../utils/headers.ts";
|
|
59
|
+
import { normalizeMediaInput } from "../utils/media.ts";
|
|
60
|
+
import { safeStringify } from "../utils/serialization.ts";
|
|
61
|
+
import { toConciseProviderError, assertModalitiesSupported } from "../utils/errors.ts";
|
|
62
|
+
import { withRetries } from "../utils/retry.ts";
|
|
63
|
+
import { getModelFromCatalog, createGenericModelSpec } from "../models/catalog.ts";
|
|
64
|
+
import {
|
|
65
|
+
applyCacheForCustom,
|
|
66
|
+
mapToolChoiceToOpenAI,
|
|
67
|
+
emitProviderWarning,
|
|
68
|
+
noteProviderTurn,
|
|
69
|
+
parseStreamedToolArguments,
|
|
70
|
+
} from "../providers.ts";
|
|
71
|
+
|
|
72
|
+
// ---------------------------------------------------------------------------
|
|
73
|
+
// Chat Completions wire shapes (strict standard subset)
|
|
74
|
+
// ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
type ChatContentPart =
|
|
77
|
+
| { type: "text"; text: string }
|
|
78
|
+
| { type: "image_url"; image_url: { url: string } }
|
|
79
|
+
| { type: "input_audio"; input_audio: { data: string; format: string } }
|
|
80
|
+
| { type: string; [k: string]: unknown };
|
|
81
|
+
|
|
82
|
+
type ChatMessage =
|
|
83
|
+
| { role: "system" | "user"; content: string | ChatContentPart[]; name?: string }
|
|
84
|
+
| {
|
|
85
|
+
role: "assistant";
|
|
86
|
+
content: string | null;
|
|
87
|
+
tool_calls?: Array<{
|
|
88
|
+
id: string;
|
|
89
|
+
type: "function";
|
|
90
|
+
function: { name: string; arguments: string };
|
|
91
|
+
extra_content?: { google?: { thought_signature?: string } };
|
|
92
|
+
}>;
|
|
93
|
+
name?: string;
|
|
94
|
+
}
|
|
95
|
+
| { role: "tool"; content: string; tool_call_id: string; name?: string }
|
|
96
|
+
| { type: string; [k: string]: unknown };
|
|
97
|
+
|
|
98
|
+
interface ChatRequestBody {
|
|
99
|
+
model: string;
|
|
100
|
+
messages: ChatMessage[];
|
|
101
|
+
tools?: Array<Record<string, unknown>>;
|
|
102
|
+
tool_choice?: string | { type: "function"; function: { name: string } };
|
|
103
|
+
stream?: boolean;
|
|
104
|
+
[k: string]: unknown;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
interface ChatChoice {
|
|
108
|
+
index?: number;
|
|
109
|
+
message?: {
|
|
110
|
+
role?: string;
|
|
111
|
+
content?: string | null;
|
|
112
|
+
tool_calls?: Array<{
|
|
113
|
+
id?: string;
|
|
114
|
+
type?: string;
|
|
115
|
+
index?: number;
|
|
116
|
+
function?: { name?: string; arguments?: string };
|
|
117
|
+
extra_content?: { google?: { thought_signature?: string } };
|
|
118
|
+
}>;
|
|
119
|
+
reasoning?: string | null;
|
|
120
|
+
reasoning_details?: Array<{ type?: string; text?: string }>;
|
|
121
|
+
refusal?: string | null;
|
|
122
|
+
};
|
|
123
|
+
delta?: {
|
|
124
|
+
role?: string;
|
|
125
|
+
content?: string | null;
|
|
126
|
+
tool_calls?: Array<{
|
|
127
|
+
index?: number;
|
|
128
|
+
id?: string;
|
|
129
|
+
type?: string;
|
|
130
|
+
function?: { name?: string; arguments?: string };
|
|
131
|
+
}>;
|
|
132
|
+
reasoning_content?: string;
|
|
133
|
+
reasoning?: string;
|
|
134
|
+
reasoning_details?: Array<{ type?: string; text?: string }>;
|
|
135
|
+
refusal?: string | null;
|
|
136
|
+
};
|
|
137
|
+
finish_reason?: string | null;
|
|
138
|
+
error?: { message?: string; code?: string | number };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
interface ChatResponse {
|
|
142
|
+
id?: string;
|
|
143
|
+
object?: string;
|
|
144
|
+
created?: number;
|
|
145
|
+
model?: string;
|
|
146
|
+
choices?: ChatChoice[];
|
|
147
|
+
usage?: {
|
|
148
|
+
prompt_tokens?: number;
|
|
149
|
+
completion_tokens?: number;
|
|
150
|
+
total_tokens?: number;
|
|
151
|
+
prompt_tokens_details?: { cached_tokens?: number; cache_write_tokens?: number };
|
|
152
|
+
completion_tokens_details?: { reasoning_tokens?: number };
|
|
153
|
+
cost?: number;
|
|
154
|
+
[k: string]: unknown;
|
|
155
|
+
};
|
|
156
|
+
error?: { message?: string; code?: string | number };
|
|
157
|
+
[k: string]: unknown;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
const DEFAULT_BASE_URL = "https://api.openai.com/v1";
|
|
161
|
+
|
|
162
|
+
/** Headers that must never leak onto native REST requests. */
|
|
163
|
+
const INTERNAL_HEADERS = new Set([
|
|
164
|
+
"x-thought-signature-map",
|
|
165
|
+
"x-cached-content-id",
|
|
166
|
+
"x-multimodal-user-content",
|
|
167
|
+
]);
|
|
168
|
+
|
|
169
|
+
// ---------------------------------------------------------------------------
|
|
170
|
+
// Request building (canonical -> Chat Completions)
|
|
171
|
+
// ---------------------------------------------------------------------------
|
|
172
|
+
|
|
173
|
+
function envPrefixOf(prefix: string): string {
|
|
174
|
+
return prefix.toUpperCase().replace(/[^A-Z0-9]/g, "_");
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
function resolveBaseUrl(
|
|
178
|
+
prefix: string,
|
|
179
|
+
configuredBaseUrl: string | undefined,
|
|
180
|
+
options?: ProviderRequestOptions
|
|
181
|
+
): string {
|
|
182
|
+
const envPrefix = envPrefixOf(prefix);
|
|
183
|
+
return (
|
|
184
|
+
options?.baseUrl ||
|
|
185
|
+
configuredBaseUrl ||
|
|
186
|
+
options?.env?.[`${envPrefix}_BASE_URL`] ||
|
|
187
|
+
options?.env?.[`${envPrefix}_BASEURL`] ||
|
|
188
|
+
options?.env?.[`${envPrefix}_API_BASE`] ||
|
|
189
|
+
getEnv(`${envPrefix}_BASE_URL`) ||
|
|
190
|
+
getEnv(`${envPrefix}_BASEURL`) ||
|
|
191
|
+
getEnv(`${envPrefix}_API_BASE`) ||
|
|
192
|
+
getEnv("OPENAI_BASE_URL") ||
|
|
193
|
+
getEnv("OPENAI_API_BASE") ||
|
|
194
|
+
DEFAULT_BASE_URL
|
|
195
|
+
).replace(/\/+$/, "");
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function resolveApiKey(
|
|
199
|
+
prefix: string,
|
|
200
|
+
configuredApiKey: string | undefined,
|
|
201
|
+
options?: ProviderRequestOptions
|
|
202
|
+
): string | undefined {
|
|
203
|
+
const envPrefix = envPrefixOf(prefix);
|
|
204
|
+
return (
|
|
205
|
+
options?.apiKey ||
|
|
206
|
+
configuredApiKey ||
|
|
207
|
+
options?.env?.[`${envPrefix}_API_KEY`] ||
|
|
208
|
+
options?.env?.[`${envPrefix}_BASE_API_KEY`] ||
|
|
209
|
+
options?.env?.["OPENAI_BASE_API_KEY"] ||
|
|
210
|
+
options?.env?.["OPENAI_API_KEY"] ||
|
|
211
|
+
getEnv(`${envPrefix}_API_KEY`) ||
|
|
212
|
+
getEnv(`${envPrefix}_BASE_API_KEY`) ||
|
|
213
|
+
getEnv("OPENAI_BASE_API_KEY") ||
|
|
214
|
+
getEnv("OPENAI_API_KEY") ||
|
|
215
|
+
getApiKey(prefix, undefined, options?.env) ||
|
|
216
|
+
undefined
|
|
217
|
+
);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/** Local endpoints (loopback / LAN / *.local) may omit the API key. */
|
|
221
|
+
function isLocalEndpoint(baseUrl: string): boolean {
|
|
222
|
+
try {
|
|
223
|
+
const host = new URL(baseUrl).hostname.toLowerCase();
|
|
224
|
+
if (
|
|
225
|
+
host === "localhost" ||
|
|
226
|
+
host === "127.0.0.1" ||
|
|
227
|
+
host === "0.0.0.0" ||
|
|
228
|
+
host === "::1" ||
|
|
229
|
+
host.endsWith(".local")
|
|
230
|
+
) {
|
|
231
|
+
return true;
|
|
232
|
+
}
|
|
233
|
+
if (/^10\./.test(host) || /^192\.168\./.test(host) || /^172\.(1[6-9]|2\d|3[01])\./.test(host)) {
|
|
234
|
+
return true;
|
|
235
|
+
}
|
|
236
|
+
return false;
|
|
237
|
+
} catch {
|
|
238
|
+
return false;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** Strips ONLY the `{prefix}/` scope. Bare ids pass through untouched. */
|
|
243
|
+
function cleanModelId(prefix: string, model: string | ModelSpec): string {
|
|
244
|
+
const rawId = typeof model === "string" ? model : model.id;
|
|
245
|
+
const escaped = prefix.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
246
|
+
return rawId.replace(new RegExp(`^${escaped}/`, "i"), "");
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
function toCompatTools(tools?: StandardToolDeclaration[]): Array<Record<string, unknown>> | undefined {
|
|
250
|
+
if (!tools || tools.length === 0) return undefined;
|
|
251
|
+
return tools.map((t) => ({
|
|
252
|
+
type: "function",
|
|
253
|
+
function: {
|
|
254
|
+
name: t.name,
|
|
255
|
+
description: t.description,
|
|
256
|
+
parameters: (t.parameters || { type: "object", properties: {} }) as Record<string, unknown>,
|
|
257
|
+
...(t.strict !== undefined ? { strict: t.strict } : {}),
|
|
258
|
+
},
|
|
259
|
+
}));
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/** Thinking levels have no portable wire shape on generic endpoints: warn once, send nothing. */
|
|
263
|
+
function warnThinkingDropped(prefix: string, thinking: ThinkingConfig | undefined, modelRef: string): void {
|
|
264
|
+
if (!thinking) return;
|
|
265
|
+
const level = thinking.level;
|
|
266
|
+
if (!level || level === "dynamic") return; // server default either way
|
|
267
|
+
emitProviderWarning({
|
|
268
|
+
provider: prefix,
|
|
269
|
+
capability: "thinking level",
|
|
270
|
+
requested: `${level} (${modelRef})`,
|
|
271
|
+
reason: "custom OpenAI-compatible endpoints define no portable reasoning control; levels vary by vendor and strict endpoints reject unknown fields.",
|
|
272
|
+
fallback: "the server default (no reasoning payload is sent)",
|
|
273
|
+
});
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
function audioFormatFor(mimeType?: string): string {
|
|
277
|
+
const mime = (mimeType || "").toLowerCase();
|
|
278
|
+
if (mime.includes("wav")) return "wav";
|
|
279
|
+
return "mp3";
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/**
|
|
283
|
+
* Rejects `video`/`file` parts before any network call: the strict Chat
|
|
284
|
+
* Completions subset has no shape for them, and a provider 400 would only say
|
|
285
|
+
* so confusingly. Images and audio keep flowing natively.
|
|
286
|
+
*/
|
|
287
|
+
function assertNoVideoOrFileParts(
|
|
288
|
+
context: ProviderContext,
|
|
289
|
+
prefix: string,
|
|
290
|
+
modelId: string
|
|
291
|
+
): void {
|
|
292
|
+
let kind: "video" | "file" | undefined;
|
|
293
|
+
for (const msg of context.messages) {
|
|
294
|
+
if (!Array.isArray((msg as { content?: unknown }).content)) continue;
|
|
295
|
+
for (const part of (msg as { content: Array<{ type?: unknown }> }).content) {
|
|
296
|
+
if ((part as { type?: unknown }).type === "video") {
|
|
297
|
+
kind = "video";
|
|
298
|
+
break;
|
|
299
|
+
}
|
|
300
|
+
if ((part as { type?: unknown }).type === "file") {
|
|
301
|
+
kind = "file";
|
|
302
|
+
break;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
if (kind) break;
|
|
306
|
+
}
|
|
307
|
+
if (!kind) return;
|
|
308
|
+
const message =
|
|
309
|
+
kind === "video"
|
|
310
|
+
? `[${prefix}/${modelId}] unsupported video input (generic OpenAI-compatible endpoints carry text/image/audio only). Use a video-capable model (e.g. google/gemini-*) or drop the video part.`
|
|
311
|
+
: `[${prefix}/${modelId}] unsupported file input (generic OpenAI-compatible endpoints carry text/image/audio only). Convert the document to Markdown with the convert_document_to_markdown tool (or Agent bypassInputFileModality) and send it as text instead.`;
|
|
312
|
+
const err = new Error(message);
|
|
313
|
+
err.name = "AgentAccelProviderError";
|
|
314
|
+
Object.defineProperties(err, {
|
|
315
|
+
provider: { value: prefix, enumerable: false },
|
|
316
|
+
model: { value: modelId, enumerable: false },
|
|
317
|
+
});
|
|
318
|
+
throw err;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
async function contentPartsToBlocks(parts: ContentPart[]): Promise<ChatContentPart[]> {
|
|
322
|
+
const blocks: ChatContentPart[] = [];
|
|
323
|
+
for (const part of parts) {
|
|
324
|
+
if (part.type === "text" && part.text) {
|
|
325
|
+
blocks.push({ type: "text", text: part.text });
|
|
326
|
+
} else if (part.type === "image") {
|
|
327
|
+
const raw = (part as { image?: unknown }).image;
|
|
328
|
+
if (typeof raw === "string" && (raw.startsWith("http://") || raw.startsWith("https://"))) {
|
|
329
|
+
blocks.push({ type: "image_url", image_url: { url: raw } });
|
|
330
|
+
continue;
|
|
331
|
+
}
|
|
332
|
+
const norm = await normalizeMediaInput(
|
|
333
|
+
raw as string | Uint8Array | ArrayBuffer,
|
|
334
|
+
(part as { mimeType?: string }).mimeType
|
|
335
|
+
);
|
|
336
|
+
blocks.push({ type: "image_url", image_url: { url: norm.dataUrl } });
|
|
337
|
+
} else if (part.type === "audio") {
|
|
338
|
+
const raw = (part as { audio?: unknown }).audio;
|
|
339
|
+
const mimeType = (part as { mimeType?: string }).mimeType;
|
|
340
|
+
// The OpenAI chat audio shape carries data only — always inline.
|
|
341
|
+
const norm = await normalizeMediaInput(
|
|
342
|
+
raw as string | Uint8Array | ArrayBuffer,
|
|
343
|
+
mimeType
|
|
344
|
+
);
|
|
345
|
+
blocks.push({
|
|
346
|
+
type: "input_audio",
|
|
347
|
+
input_audio: { data: norm.base64Data, format: audioFormatFor(norm.mimeType) },
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
// video/file never reach here: assertNoVideoOrFileParts runs first.
|
|
351
|
+
}
|
|
352
|
+
return blocks;
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
function resultToOutput(result: unknown): string {
|
|
356
|
+
if (typeof result === "string") return result;
|
|
357
|
+
return safeStringify(result);
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
/**
|
|
361
|
+
* Builds the FULL explicit history as Chat Completions `messages` (stateless
|
|
362
|
+
* transport: every turn carries everything). Prior-turn reasoning is NOT
|
|
363
|
+
* resent — except echoed Gemini thought signatures, which strict
|
|
364
|
+
* OpenAI-compatible endpoints require on tool calls they generated.
|
|
365
|
+
*/
|
|
366
|
+
async function fullHistoryMessages(context: ProviderContext): Promise<ChatMessage[]> {
|
|
367
|
+
const pairing = new Map<string, string>();
|
|
368
|
+
for (const m of context.messages) {
|
|
369
|
+
if (m.role === "assistant" && Array.isArray(m.content)) {
|
|
370
|
+
for (const part of m.content) {
|
|
371
|
+
if (part.type === "tool_call") {
|
|
372
|
+
pairing.set(part.id, part.callId || part.id);
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
const messages: ChatMessage[] = [];
|
|
379
|
+
if (context.systemPrompt) {
|
|
380
|
+
messages.push({ role: "system", content: context.systemPrompt });
|
|
381
|
+
}
|
|
382
|
+
for (const m of context.messages) {
|
|
383
|
+
if (m.role === "system") continue;
|
|
384
|
+
if (m.role === "user") {
|
|
385
|
+
if (typeof m.content === "string") {
|
|
386
|
+
if (m.content) messages.push({ role: "user", content: m.content });
|
|
387
|
+
} else {
|
|
388
|
+
const blocks = await contentPartsToBlocks(m.content);
|
|
389
|
+
if (blocks.length > 0) messages.push({ role: "user", content: blocks });
|
|
390
|
+
}
|
|
391
|
+
} else if (m.role === "assistant") {
|
|
392
|
+
if (typeof m.content === "string") {
|
|
393
|
+
if (m.content) {
|
|
394
|
+
messages.push({ role: "assistant", content: m.content });
|
|
395
|
+
}
|
|
396
|
+
continue;
|
|
397
|
+
}
|
|
398
|
+
const texts: string[] = [];
|
|
399
|
+
const calls: Array<{
|
|
400
|
+
id: string;
|
|
401
|
+
type: "function";
|
|
402
|
+
function: { name: string; arguments: string };
|
|
403
|
+
extra_content?: { google: { thought_signature: string } };
|
|
404
|
+
}> = [];
|
|
405
|
+
for (const part of m.content) {
|
|
406
|
+
if (part.type === "tool_call") {
|
|
407
|
+
const call: {
|
|
408
|
+
id: string;
|
|
409
|
+
type: "function";
|
|
410
|
+
function: { name: string; arguments: string };
|
|
411
|
+
extra_content?: { google: { thought_signature: string } };
|
|
412
|
+
} = {
|
|
413
|
+
id: part.callId || part.id,
|
|
414
|
+
type: "function",
|
|
415
|
+
function: { name: part.name, arguments: JSON.stringify(part.arguments || {}) },
|
|
416
|
+
};
|
|
417
|
+
// Echo provider-issued thought signatures verbatim so endpoints
|
|
418
|
+
// that minted them (Gemini behind a compat proxy) keep working.
|
|
419
|
+
// Only present when a previous turn captured one — other endpoints
|
|
420
|
+
// never see this field.
|
|
421
|
+
if (part.thoughtSignature) {
|
|
422
|
+
call.extra_content = { google: { thought_signature: part.thoughtSignature } };
|
|
423
|
+
}
|
|
424
|
+
calls.push(call);
|
|
425
|
+
} else if (part.type === "text" && part.text) {
|
|
426
|
+
texts.push(part.text);
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
messages.push({
|
|
430
|
+
role: "assistant",
|
|
431
|
+
content: texts.length > 0 ? texts.join("\n") : null,
|
|
432
|
+
...(calls.length > 0 ? { tool_calls: calls } : {}),
|
|
433
|
+
});
|
|
434
|
+
} else if (m.role === "tool") {
|
|
435
|
+
if (!Array.isArray(m.content)) continue;
|
|
436
|
+
for (const part of m.content) {
|
|
437
|
+
if (part.type === "tool_result") {
|
|
438
|
+
messages.push({
|
|
439
|
+
role: "tool",
|
|
440
|
+
tool_call_id: pairing.get(part.id) || part.id,
|
|
441
|
+
content: resultToOutput(part.result),
|
|
442
|
+
name: part.name,
|
|
443
|
+
});
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
}
|
|
448
|
+
return messages;
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
// ---------------------------------------------------------------------------
|
|
452
|
+
// Response mapping (Chat Completions -> canonical)
|
|
453
|
+
// ---------------------------------------------------------------------------
|
|
454
|
+
|
|
455
|
+
function mapUsage(raw?: ChatResponse["usage"]): TokenUsage {
|
|
456
|
+
const input = raw?.prompt_tokens ?? 0;
|
|
457
|
+
const output = raw?.completion_tokens ?? 0;
|
|
458
|
+
// Canonical invariant: cache hits are a SUBSET of input (no turn may
|
|
459
|
+
// report a >100% hit rate).
|
|
460
|
+
const cached = Math.min(raw?.prompt_tokens_details?.cached_tokens ?? 0, input);
|
|
461
|
+
const usage: TokenUsage = {
|
|
462
|
+
inputTokens: input,
|
|
463
|
+
outputTokens: output,
|
|
464
|
+
totalTokens: raw?.total_tokens ?? input + output,
|
|
465
|
+
cachedTokens: cached,
|
|
466
|
+
cacheReadTokens: cached,
|
|
467
|
+
cacheWriteTokens: raw?.prompt_tokens_details?.cache_write_tokens ?? 0,
|
|
468
|
+
thinkingTokens: raw?.completion_tokens_details?.reasoning_tokens ?? 0,
|
|
469
|
+
};
|
|
470
|
+
if (typeof raw?.cost === "number" && raw.cost > 0) {
|
|
471
|
+
usage.cost = { totalCost: raw.cost };
|
|
472
|
+
}
|
|
473
|
+
return usage;
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
function parseArguments(raw: string | undefined): Record<string, unknown> {
|
|
477
|
+
if (!raw) return {};
|
|
478
|
+
try {
|
|
479
|
+
const parsed: unknown = JSON.parse(raw);
|
|
480
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
481
|
+
return parsed as Record<string, unknown>;
|
|
482
|
+
}
|
|
483
|
+
return { raw };
|
|
484
|
+
} catch {
|
|
485
|
+
return { raw };
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
/** Trims/collapses assembled thinking parts (no edge-tripling). */
|
|
490
|
+
function normalizeThinkingParts(parts: string[]): string | undefined {
|
|
491
|
+
const cleaned = parts
|
|
492
|
+
.map((p) => p.replace(/\n{3,}/g, "\n\n").trim())
|
|
493
|
+
.filter((p) => p.length > 0);
|
|
494
|
+
return cleaned.length > 0 ? cleaned.join("\n") : undefined;
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
function extractSignature(rawTc: {
|
|
498
|
+
extra_content?: { google?: { thought_signature?: string } };
|
|
499
|
+
}): string | undefined {
|
|
500
|
+
const sig = rawTc?.extra_content?.google?.thought_signature;
|
|
501
|
+
return typeof sig === "string" && sig ? sig : undefined;
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
function throwResponseError(
|
|
505
|
+
response: ChatResponse,
|
|
506
|
+
prefix: string,
|
|
507
|
+
modelId: string
|
|
508
|
+
): void {
|
|
509
|
+
const message =
|
|
510
|
+
response.error && typeof response.error.message === "string" && response.error.message
|
|
511
|
+
? response.error.message
|
|
512
|
+
: "OpenAI-compatible response failed";
|
|
513
|
+
const failure: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
|
|
514
|
+
if (response.error?.code !== undefined) failure["code"] = response.error.code;
|
|
515
|
+
throw toConciseProviderError(failure, prefix, modelId);
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
function parseResponse(
|
|
519
|
+
response: ChatResponse,
|
|
520
|
+
prefix: string,
|
|
521
|
+
modelId: string,
|
|
522
|
+
durationMs: number,
|
|
523
|
+
raw: ProviderRawData
|
|
524
|
+
): ProviderGenerateResult {
|
|
525
|
+
// Provider-interrupted generations arrive as HTTP 200 carrying ONLY `error`
|
|
526
|
+
// (no `choices`) — must check the body, not just the status.
|
|
527
|
+
if (response.error) throwResponseError(response, prefix, modelId);
|
|
528
|
+
|
|
529
|
+
const choice = response.choices?.[0];
|
|
530
|
+
const message = choice?.message;
|
|
531
|
+
const text = typeof message?.content === "string" ? message.content : "";
|
|
532
|
+
// Thinking tolerance: plain `reasoning` plus per-block texts (deepseek-style
|
|
533
|
+
// `reasoning_details`). Details win when both ride along; both are
|
|
534
|
+
// display-only here — never resent.
|
|
535
|
+
const thinkingParts: string[] = [];
|
|
536
|
+
const detailTexts: string[] = [];
|
|
537
|
+
for (const block of message?.reasoning_details ?? []) {
|
|
538
|
+
if (block && typeof block.text === "string" && block.text) detailTexts.push(block.text);
|
|
539
|
+
}
|
|
540
|
+
if (detailTexts.length > 0) {
|
|
541
|
+
thinkingParts.push(...detailTexts);
|
|
542
|
+
} else if (typeof message?.reasoning === "string" && message.reasoning) {
|
|
543
|
+
thinkingParts.push(message.reasoning);
|
|
544
|
+
}
|
|
545
|
+
const toolCalls: ToolCallRecord[] = [];
|
|
546
|
+
for (const tc of message?.tool_calls ?? []) {
|
|
547
|
+
const args = parseArguments(tc.function?.arguments);
|
|
548
|
+
const sig = extractSignature(tc);
|
|
549
|
+
toolCalls.push({
|
|
550
|
+
id: tc.id || `call_${Math.random().toString(36).slice(2, 9)}`,
|
|
551
|
+
name: tc.function?.name || "unknown",
|
|
552
|
+
arguments: args,
|
|
553
|
+
rawArguments:
|
|
554
|
+
typeof tc.function?.arguments === "string"
|
|
555
|
+
? tc.function.arguments
|
|
556
|
+
: JSON.stringify(tc.function?.arguments ?? {}),
|
|
557
|
+
...(sig ? { thoughtSignature: sig } : {}),
|
|
558
|
+
});
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
const wireReason = choice?.finish_reason ?? undefined;
|
|
562
|
+
return {
|
|
563
|
+
text,
|
|
564
|
+
thinking: normalizeThinkingParts(thinkingParts),
|
|
565
|
+
thoughtSignature: toolCalls.map((c) => c.thoughtSignature).find(Boolean),
|
|
566
|
+
toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
|
|
567
|
+
usage: mapUsage(response.usage),
|
|
568
|
+
finishReason: toolCalls.length > 0 ? "tool_calls" : wireReason || "stop",
|
|
569
|
+
responseId: response.id,
|
|
570
|
+
model: modelId,
|
|
571
|
+
provider: prefix as ProviderId,
|
|
572
|
+
raw,
|
|
573
|
+
durationMs,
|
|
574
|
+
};
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
function redactedHeaders(headers: Record<string, string>): Record<string, string> {
|
|
578
|
+
const out: Record<string, string> = {};
|
|
579
|
+
for (const [k, v] of Object.entries(headers)) {
|
|
580
|
+
out[k] = k.toLowerCase() === "authorization" ? "[REDACTED]" : v;
|
|
581
|
+
}
|
|
582
|
+
return out;
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
function readErrorPayload(bodyText: string): { message: string; code?: string | number } {
|
|
586
|
+
try {
|
|
587
|
+
const parsed: unknown = JSON.parse(bodyText);
|
|
588
|
+
const first = Array.isArray(parsed) ? parsed[0] : parsed;
|
|
589
|
+
const err = (first as { error?: { message?: string; code?: string | number } })?.error;
|
|
590
|
+
if (err && typeof err.message === "string") {
|
|
591
|
+
return { message: err.message, code: err.code };
|
|
592
|
+
}
|
|
593
|
+
return { message: bodyText.slice(0, 300) };
|
|
594
|
+
} catch {
|
|
595
|
+
return { message: bodyText.slice(0, 300) };
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
// ---------------------------------------------------------------------------
|
|
600
|
+
// Provider
|
|
601
|
+
// ---------------------------------------------------------------------------
|
|
602
|
+
|
|
603
|
+
export interface CustomProviderOptions {
|
|
604
|
+
/** Human label, defaults to `${prefix} (OpenAI-compatible)` */
|
|
605
|
+
name?: string;
|
|
606
|
+
/** Static baseUrl — overrides env. Env `{PREFIX}_BASE_URL` still wins at runtime if set. */
|
|
607
|
+
baseUrl?: string;
|
|
608
|
+
/** Static apiKey — overrides env. Explicit per-call `apiKey` still wins. */
|
|
609
|
+
apiKey?: string;
|
|
610
|
+
/** Default baseUrl when no env is set. Defaults to OpenAI cloud. */
|
|
611
|
+
defaultBaseUrl?: string;
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
/**
|
|
615
|
+
* Generic OpenAI-compatible provider implemented directly on the Chat
|
|
616
|
+
* Completions REST API (`POST {baseUrl}/chat/completions`, streaming on the
|
|
617
|
+
* same endpoint). One class serves every custom prefix.
|
|
618
|
+
*/
|
|
619
|
+
export class OpenAICompatibleChatProvider implements Provider {
|
|
620
|
+
readonly id: ProviderId;
|
|
621
|
+
readonly name: string;
|
|
622
|
+
readonly models: ModelSpec[] = [];
|
|
623
|
+
private readonly prefix: string;
|
|
624
|
+
private readonly configuredBaseUrl?: string;
|
|
625
|
+
private readonly configuredApiKey?: string;
|
|
626
|
+
|
|
627
|
+
/**
|
|
628
|
+
* Creates an OpenAI-compatible provider for any endpoint prefix.
|
|
629
|
+
*
|
|
630
|
+
* @param prefix Prefix used in model strings and environment variables,
|
|
631
|
+
* such as `groq` for `groq/llama-3.3-70b-versatile`.
|
|
632
|
+
* @param opts Optional endpoint, key, and display-name overrides.
|
|
633
|
+
*
|
|
634
|
+
* @example
|
|
635
|
+
* ```ts
|
|
636
|
+
* const groq = new OpenAICompatibleChatProvider("groq", {
|
|
637
|
+
* baseUrl: "https://api.groq.com/openai/v1",
|
|
638
|
+
* apiKey: process.env.GROQ_API_KEY,
|
|
639
|
+
* });
|
|
640
|
+
* ```
|
|
641
|
+
*/
|
|
642
|
+
constructor(prefix: string, opts?: CustomProviderOptions) {
|
|
643
|
+
const norm = prefix.trim().toLowerCase().replace(/\/.*$/, "");
|
|
644
|
+
this.prefix = norm;
|
|
645
|
+
this.id = norm as ProviderId;
|
|
646
|
+
this.name = opts?.name ?? `${norm} (OpenAI-compatible)`;
|
|
647
|
+
this.configuredBaseUrl = opts?.baseUrl ?? opts?.defaultBaseUrl;
|
|
648
|
+
this.configuredApiKey = opts?.apiKey;
|
|
649
|
+
}
|
|
650
|
+
|
|
651
|
+
getModel(modelId: string): ModelSpec | undefined {
|
|
652
|
+
const clean = cleanModelId(this.prefix, modelId);
|
|
653
|
+
return (
|
|
654
|
+
getModelFromCatalog(this.id, modelId) ||
|
|
655
|
+
getModelFromCatalog(this.id, clean) ||
|
|
656
|
+
this.models.find((m) => m.id === modelId || m.id === clean) ||
|
|
657
|
+
createGenericModelSpec(this.id, clean)
|
|
658
|
+
);
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
private baseUrl(options?: ProviderRequestOptions): string {
|
|
662
|
+
return resolveBaseUrl(this.prefix, this.configuredBaseUrl, options);
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
private apiKey(options?: ProviderRequestOptions): string | undefined {
|
|
666
|
+
return resolveApiKey(this.prefix, this.configuredApiKey, options);
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
private requireApiKey(baseUrl: string, options?: ProviderRequestOptions): string | undefined {
|
|
670
|
+
const key = this.apiKey(options);
|
|
671
|
+
if (!key && !isLocalEndpoint(baseUrl)) {
|
|
672
|
+
throw new Error(
|
|
673
|
+
`[Agent Accelerator] Missing API key for "${this.prefix}". Set ${envPrefixOf(this.prefix)}_API_KEY or pass apiKey. (Local endpoints may omit the key.)`
|
|
674
|
+
);
|
|
675
|
+
}
|
|
676
|
+
return key;
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
private async buildBody(
|
|
680
|
+
modelId: string,
|
|
681
|
+
context: ProviderContext,
|
|
682
|
+
options: ProviderRequestOptions | undefined,
|
|
683
|
+
tools: StandardToolDeclaration[] | undefined,
|
|
684
|
+
stream: boolean
|
|
685
|
+
): Promise<ChatRequestBody> {
|
|
686
|
+
warnThinkingDropped(this.prefix, options?.thinking, `${this.prefix}/${modelId}`);
|
|
687
|
+
if (options?.serviceTier) {
|
|
688
|
+
emitProviderWarning({
|
|
689
|
+
provider: this.prefix,
|
|
690
|
+
capability: "service tier",
|
|
691
|
+
requested: `${options.serviceTier} (${this.prefix}/${modelId})`,
|
|
692
|
+
reason: "custom OpenAI-compatible endpoints define no service-tier primitive, and strict endpoints reject unknown body properties.",
|
|
693
|
+
fallback: "standard routing (no service_tier payload is sent)",
|
|
694
|
+
});
|
|
695
|
+
}
|
|
696
|
+
applyCacheForCustom(this.prefix, options?.cache, `${this.prefix}/${modelId}`);
|
|
697
|
+
|
|
698
|
+
const messages = await fullHistoryMessages(context);
|
|
699
|
+
const body: ChatRequestBody = {
|
|
700
|
+
model: modelId,
|
|
701
|
+
messages,
|
|
702
|
+
...(stream ? { stream: true } : {}),
|
|
703
|
+
};
|
|
704
|
+
const compatTools = toCompatTools(tools);
|
|
705
|
+
if (compatTools) body.tools = compatTools;
|
|
706
|
+
const toolChoice = mapToolChoiceToOpenAI(
|
|
707
|
+
options?.toolChoice as "auto" | "none" | "required" | { type: "function"; function: { name: string } } | undefined
|
|
708
|
+
);
|
|
709
|
+
if (toolChoice !== undefined) {
|
|
710
|
+
// Chat Completions nests pinned names under `function`; the canonical
|
|
711
|
+
// mapper emits the flat Responses shape, so normalize here.
|
|
712
|
+
body.tool_choice =
|
|
713
|
+
typeof toolChoice === "object" && "name" in toolChoice && !("function" in toolChoice)
|
|
714
|
+
? { type: "function", function: { name: (toolChoice as { name: string }).name } }
|
|
715
|
+
: (toolChoice as ChatRequestBody["tool_choice"]);
|
|
716
|
+
}
|
|
717
|
+
// No reasoning / service_tier / session / cache body primitives: strict
|
|
718
|
+
// endpoints reject unknown properties. Affinity is headers-only.
|
|
719
|
+
return body;
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
private requestInit(
|
|
723
|
+
body: ChatRequestBody,
|
|
724
|
+
options: ProviderRequestOptions | undefined,
|
|
725
|
+
apiKey: string | undefined,
|
|
726
|
+
sessionId: string | undefined,
|
|
727
|
+
signal?: AbortSignal
|
|
728
|
+
): RequestInit {
|
|
729
|
+
const headers: Record<string, string> = buildSessionHeaders(
|
|
730
|
+
this.prefix,
|
|
731
|
+
options?.cache,
|
|
732
|
+
options?.headers,
|
|
733
|
+
sessionId
|
|
734
|
+
);
|
|
735
|
+
for (const [k, v] of Object.entries(headers)) {
|
|
736
|
+
if (INTERNAL_HEADERS.has(k.toLowerCase())) delete headers[k];
|
|
737
|
+
}
|
|
738
|
+
headers["Content-Type"] = "application/json";
|
|
739
|
+
if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
|
|
740
|
+
return { method: "POST", headers, body: JSON.stringify(body), signal };
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
private async doFetch(
|
|
744
|
+
url: string,
|
|
745
|
+
body: ChatRequestBody,
|
|
746
|
+
options: ProviderRequestOptions | undefined,
|
|
747
|
+
apiKey: string | undefined,
|
|
748
|
+
sessionId: string | undefined,
|
|
749
|
+
signal?: AbortSignal
|
|
750
|
+
): Promise<{ status: number; statusText: string; headers: Record<string, string>; text: string }> {
|
|
751
|
+
const res = await fetch(url, this.requestInit(body, options, apiKey, sessionId, signal));
|
|
752
|
+
const text = await res.text();
|
|
753
|
+
const headers: Record<string, string> = {};
|
|
754
|
+
res.headers.forEach((v, k) => {
|
|
755
|
+
headers[k] = v;
|
|
756
|
+
});
|
|
757
|
+
return { status: res.status, statusText: res.statusText, headers, text };
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
private throwIfError(
|
|
761
|
+
status: number,
|
|
762
|
+
url: string,
|
|
763
|
+
bodyText: string,
|
|
764
|
+
modelId: string
|
|
765
|
+
): void {
|
|
766
|
+
if (status >= 200 && status < 300) return;
|
|
767
|
+
const { message, code } = readErrorPayload(bodyText);
|
|
768
|
+
const err: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
|
|
769
|
+
(err as Record<string, unknown>)["statusCode"] = status;
|
|
770
|
+
(err as Record<string, unknown>)["status"] = status;
|
|
771
|
+
(err as Record<string, unknown>)["responseBody"] = bodyText.slice(0, 500);
|
|
772
|
+
(err as Record<string, unknown>)["url"] = url.split("?")[0];
|
|
773
|
+
if (code !== undefined) (err as Record<string, unknown>)["code"] = code;
|
|
774
|
+
throw toConciseProviderError(err, this.prefix, modelId);
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
async generate(
|
|
778
|
+
model: string | ModelSpec,
|
|
779
|
+
context: ProviderContext,
|
|
780
|
+
options?: ProviderRequestOptions
|
|
781
|
+
): Promise<ProviderGenerateResult> {
|
|
782
|
+
const startTime = Date.now();
|
|
783
|
+
const clean = cleanModelId(this.prefix, model);
|
|
784
|
+
const baseUrl = this.baseUrl(options);
|
|
785
|
+
const apiKey = this.requireApiKey(baseUrl, options);
|
|
786
|
+
// Fail fast before any network call when the catalog knows the model
|
|
787
|
+
// lacks the requested modality. Unknown models skip the guard; the
|
|
788
|
+
// endpoint verdict surfaces concisely.
|
|
789
|
+
assertModalitiesSupported(context, this.prefix, clean);
|
|
790
|
+
assertNoVideoOrFileParts(context, this.prefix, clean);
|
|
791
|
+
const url = `${baseUrl}/chat/completions`;
|
|
792
|
+
const sessionId = options?.sessionId || options?.cache?.sessionId;
|
|
793
|
+
const tools = options?.tools as StandardToolDeclaration[] | undefined;
|
|
794
|
+
const body = await this.buildBody(clean, context, options, tools, false);
|
|
795
|
+
|
|
796
|
+
const auditRaw: Record<string, string> = {
|
|
797
|
+
...buildSessionHeaders(this.prefix, options?.cache, options?.headers, sessionId),
|
|
798
|
+
"Content-Type": "application/json",
|
|
799
|
+
...(apiKey ? { Authorization: "[REDACTED]" } : {}),
|
|
800
|
+
};
|
|
801
|
+
for (const k of Object.keys(auditRaw)) {
|
|
802
|
+
if (INTERNAL_HEADERS.has(k.toLowerCase())) delete auditRaw[k];
|
|
803
|
+
}
|
|
804
|
+
const rawRequest = {
|
|
805
|
+
url,
|
|
806
|
+
method: "POST",
|
|
807
|
+
headers: redactedHeaders(auditRaw),
|
|
808
|
+
body,
|
|
809
|
+
};
|
|
810
|
+
|
|
811
|
+
const doCall = async (): Promise<ProviderGenerateResult> => {
|
|
812
|
+
const res = await this.doFetch(url, body, options, apiKey, sessionId, options?.signal);
|
|
813
|
+
this.throwIfError(res.status, url, res.text, clean);
|
|
814
|
+
let response: ChatResponse;
|
|
815
|
+
try {
|
|
816
|
+
response = JSON.parse(res.text) as ChatResponse;
|
|
817
|
+
} catch {
|
|
818
|
+
throw toConciseProviderError(
|
|
819
|
+
Object.assign(new Error("Invalid JSON response from OpenAI-compatible Chat Completions API"), {
|
|
820
|
+
statusCode: res.status,
|
|
821
|
+
responseBody: res.text.slice(0, 500),
|
|
822
|
+
url,
|
|
823
|
+
}),
|
|
824
|
+
this.prefix,
|
|
825
|
+
clean
|
|
826
|
+
);
|
|
827
|
+
}
|
|
828
|
+
// Stateless turns never establish chains, but the session store still
|
|
829
|
+
// records the turn so other providers' switch detection keeps working.
|
|
830
|
+
noteProviderTurn(sessionId, this.prefix);
|
|
831
|
+
return parseResponse(
|
|
832
|
+
response,
|
|
833
|
+
this.prefix,
|
|
834
|
+
clean,
|
|
835
|
+
Date.now() - startTime,
|
|
836
|
+
{ request: rawRequest, response: { status: res.status, statusText: res.statusText, headers: res.headers, body: response } }
|
|
837
|
+
);
|
|
838
|
+
};
|
|
839
|
+
|
|
840
|
+
try {
|
|
841
|
+
return await withRetries(doCall, {
|
|
842
|
+
maxRetries: options?.maxRetries,
|
|
843
|
+
maxRetryDelayMs: options?.maxRetryDelayMs,
|
|
844
|
+
signal: options?.signal,
|
|
845
|
+
label: { providerId: this.prefix, modelId: clean },
|
|
846
|
+
});
|
|
847
|
+
} catch (err) {
|
|
848
|
+
if (err instanceof Error && (err as { name?: string }).name === "AbortError") throw err;
|
|
849
|
+
throw err;
|
|
850
|
+
}
|
|
851
|
+
}
|
|
852
|
+
|
|
853
|
+
stream(
|
|
854
|
+
model: string | ModelSpec,
|
|
855
|
+
context: ProviderContext,
|
|
856
|
+
options?: ProviderRequestOptions
|
|
857
|
+
): AssistantMessageEventStream {
|
|
858
|
+
const eventStream = new AssistantMessageEventStream();
|
|
859
|
+
const startTime = Date.now();
|
|
860
|
+
const prefix = this.prefix;
|
|
861
|
+
const clean = cleanModelId(this.prefix, model);
|
|
862
|
+
|
|
863
|
+
const linked = new AbortController();
|
|
864
|
+
const forwardUserAbort = () => {
|
|
865
|
+
try {
|
|
866
|
+
linked.abort((options?.signal as { reason?: unknown })?.reason);
|
|
867
|
+
} catch {
|
|
868
|
+
try {
|
|
869
|
+
linked.abort();
|
|
870
|
+
} catch {}
|
|
871
|
+
}
|
|
872
|
+
};
|
|
873
|
+
if (options?.signal?.aborted) forwardUserAbort();
|
|
874
|
+
else options?.signal?.addEventListener("abort", forwardUserAbort, { once: true });
|
|
875
|
+
const removeStreamCancel = eventStream.onCancel(() => {
|
|
876
|
+
try {
|
|
877
|
+
linked.abort();
|
|
878
|
+
} catch {}
|
|
879
|
+
});
|
|
880
|
+
|
|
881
|
+
(async () => {
|
|
882
|
+
try {
|
|
883
|
+
const baseUrl = this.baseUrl(options);
|
|
884
|
+
const apiKey = this.requireApiKey(baseUrl, options);
|
|
885
|
+
assertModalitiesSupported(context, prefix, clean);
|
|
886
|
+
assertNoVideoOrFileParts(context, prefix, clean);
|
|
887
|
+
const url = `${baseUrl}/chat/completions`;
|
|
888
|
+
const sessionId = options?.sessionId || options?.cache?.sessionId;
|
|
889
|
+
const tools = options?.tools as StandardToolDeclaration[] | undefined;
|
|
890
|
+
const body = await this.buildBody(clean, context, options, tools, true);
|
|
891
|
+
|
|
892
|
+
const streamAuditRaw: Record<string, string> = {
|
|
893
|
+
...buildSessionHeaders(prefix, options?.cache, options?.headers, sessionId),
|
|
894
|
+
"Content-Type": "application/json",
|
|
895
|
+
...(apiKey ? { Authorization: "[REDACTED]" } : {}),
|
|
896
|
+
};
|
|
897
|
+
for (const k of Object.keys(streamAuditRaw)) {
|
|
898
|
+
if (INTERNAL_HEADERS.has(k.toLowerCase())) delete streamAuditRaw[k];
|
|
899
|
+
}
|
|
900
|
+
const rawRequest = {
|
|
901
|
+
url,
|
|
902
|
+
method: "POST",
|
|
903
|
+
headers: redactedHeaders(streamAuditRaw),
|
|
904
|
+
body,
|
|
905
|
+
};
|
|
906
|
+
|
|
907
|
+
const res = await fetch(url, this.requestInit(body, options, apiKey, sessionId, linked.signal));
|
|
908
|
+
if (!res.ok || !res.body) {
|
|
909
|
+
const text = !res.ok ? await res.text().catch(() => "") : "";
|
|
910
|
+
if (!res.ok) this.throwIfError(res.status, url, text, clean);
|
|
911
|
+
throw toConciseProviderError(new Error("OpenAI-compatible streaming response had no body"), prefix, clean);
|
|
912
|
+
}
|
|
913
|
+
const responseHeaders: Record<string, string> = {};
|
|
914
|
+
res.headers.forEach((v, k) => {
|
|
915
|
+
responseHeaders[k] = v;
|
|
916
|
+
});
|
|
917
|
+
const responseMeta = { status: res.status, statusText: res.statusText, headers: responseHeaders };
|
|
918
|
+
eventStream.push({ type: "start", raw: { request: rawRequest } } as never);
|
|
919
|
+
|
|
920
|
+
const parser = new SSEParser();
|
|
921
|
+
const reader = res.body.getReader();
|
|
922
|
+
const decoder = new TextDecoder();
|
|
923
|
+
let text = "";
|
|
924
|
+
let thinking = "";
|
|
925
|
+
// Tool calls keyed by choice index: { id, name, startArgs, deltaArgs }.
|
|
926
|
+
const calls = new Map<number, { id: string; name: string; startArgs: string; deltaArgs: string }>();
|
|
927
|
+
let lastToolIndex = 0;
|
|
928
|
+
const signatures = new Map<number, string>();
|
|
929
|
+
let usage: TokenUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0 };
|
|
930
|
+
let finishReason = "stop";
|
|
931
|
+
let responseId: string | undefined;
|
|
932
|
+
let completedBody: unknown = undefined;
|
|
933
|
+
let aborted = false;
|
|
934
|
+
linked.signal.addEventListener(
|
|
935
|
+
"abort",
|
|
936
|
+
() => {
|
|
937
|
+
aborted = true;
|
|
938
|
+
try {
|
|
939
|
+
void reader.cancel();
|
|
940
|
+
} catch {}
|
|
941
|
+
},
|
|
942
|
+
{ once: true }
|
|
943
|
+
);
|
|
944
|
+
|
|
945
|
+
const handleMessage = (data: string): void => {
|
|
946
|
+
if (!data || data === "[DONE]") return;
|
|
947
|
+
let msg: Record<string, unknown>;
|
|
948
|
+
try {
|
|
949
|
+
msg = JSON.parse(data) as Record<string, unknown>;
|
|
950
|
+
} catch {
|
|
951
|
+
return;
|
|
952
|
+
}
|
|
953
|
+
// Mid-stream provider error: top-level `error`, HTTP stays 200.
|
|
954
|
+
// Must surface — resolving empty success hides it.
|
|
955
|
+
const topError = msg["error"] as { message?: string; code?: string | number } | undefined;
|
|
956
|
+
if (topError && typeof topError === "object") {
|
|
957
|
+
const message =
|
|
958
|
+
typeof topError.message === "string" && topError.message
|
|
959
|
+
? topError.message
|
|
960
|
+
: "OpenAI-compatible streaming error";
|
|
961
|
+
const failure: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
|
|
962
|
+
if (topError.code !== undefined) failure["code"] = topError.code;
|
|
963
|
+
failure["url"] = url;
|
|
964
|
+
throw toConciseProviderError(failure, prefix, clean);
|
|
965
|
+
}
|
|
966
|
+
if (typeof msg["id"] === "string" && !responseId) {
|
|
967
|
+
responseId = msg["id"] as string;
|
|
968
|
+
}
|
|
969
|
+
const choice = (Array.isArray(msg["choices"]) ? (msg["choices"] as Record<string, unknown>[])[0] : undefined) ?? {};
|
|
970
|
+
const delta = (choice["delta"] as Record<string, unknown>) ?? {};
|
|
971
|
+
// Text delta (content-free accounting frames carry "" — ignored).
|
|
972
|
+
if (typeof delta["content"] === "string" && delta["content"]) {
|
|
973
|
+
text += delta["content"] as string;
|
|
974
|
+
eventStream.push({ type: "text_delta", delta: delta["content"] as string, partialText: text });
|
|
975
|
+
}
|
|
976
|
+
// Tolerated reasoning deltas (deepseek-style): display-only, never resent.
|
|
977
|
+
const details = delta["reasoning_details"];
|
|
978
|
+
if (Array.isArray(details)) {
|
|
979
|
+
for (const block of details) {
|
|
980
|
+
const t = (block as { text?: string })?.text;
|
|
981
|
+
if (typeof t === "string" && t) {
|
|
982
|
+
thinking += t;
|
|
983
|
+
eventStream.push({ type: "thinking_delta", thinkingDelta: t, partialThinking: thinking });
|
|
984
|
+
}
|
|
985
|
+
}
|
|
986
|
+
} else if (typeof delta["reasoning_content"] === "string" && delta["reasoning_content"]) {
|
|
987
|
+
thinking += delta["reasoning_content"] as string;
|
|
988
|
+
eventStream.push({
|
|
989
|
+
type: "thinking_delta",
|
|
990
|
+
thinkingDelta: delta["reasoning_content"] as string,
|
|
991
|
+
partialThinking: thinking,
|
|
992
|
+
});
|
|
993
|
+
} else if (typeof delta["reasoning"] === "string" && delta["reasoning"]) {
|
|
994
|
+
thinking += delta["reasoning"] as string;
|
|
995
|
+
eventStream.push({
|
|
996
|
+
type: "thinking_delta",
|
|
997
|
+
thinkingDelta: delta["reasoning"] as string,
|
|
998
|
+
partialThinking: thinking,
|
|
999
|
+
});
|
|
1000
|
+
}
|
|
1001
|
+
// Tool-call deltas accumulate per choice index (unindexed chunks
|
|
1002
|
+
// continue the last call; defaulting to calls.size would fork
|
|
1003
|
+
// phantom calls).
|
|
1004
|
+
const deltaCalls = delta["tool_calls"];
|
|
1005
|
+
if (Array.isArray(deltaCalls)) {
|
|
1006
|
+
for (const tc of deltaCalls) {
|
|
1007
|
+
const entry = tc as {
|
|
1008
|
+
index?: number;
|
|
1009
|
+
id?: string;
|
|
1010
|
+
type?: string;
|
|
1011
|
+
function?: { name?: string; arguments?: string };
|
|
1012
|
+
extra_content?: { google?: { thought_signature?: string } };
|
|
1013
|
+
};
|
|
1014
|
+
const index = typeof entry.index === "number" ? (lastToolIndex = entry.index) : lastToolIndex;
|
|
1015
|
+
const sig = entry.extra_content?.google?.thought_signature;
|
|
1016
|
+
if (typeof sig === "string" && sig) signatures.set(index, sig);
|
|
1017
|
+
const existing = calls.get(index);
|
|
1018
|
+
if (existing) {
|
|
1019
|
+
if (entry.id) existing.id = entry.id;
|
|
1020
|
+
if (entry.function?.name) existing.name = entry.function.name;
|
|
1021
|
+
if (typeof entry.function?.arguments === "string") {
|
|
1022
|
+
existing.deltaArgs += entry.function.arguments;
|
|
1023
|
+
}
|
|
1024
|
+
} else {
|
|
1025
|
+
calls.set(index, {
|
|
1026
|
+
id: entry.id || "",
|
|
1027
|
+
name: entry.function?.name || "unknown",
|
|
1028
|
+
startArgs: "",
|
|
1029
|
+
deltaArgs: typeof entry.function?.arguments === "string" ? entry.function.arguments : "",
|
|
1030
|
+
});
|
|
1031
|
+
}
|
|
1032
|
+
}
|
|
1033
|
+
}
|
|
1034
|
+
// Terminal accounting: finish_reason repeats on the usage chunk.
|
|
1035
|
+
if (typeof choice["finish_reason"] === "string" && choice["finish_reason"]) {
|
|
1036
|
+
const fr = choice["finish_reason"] as string;
|
|
1037
|
+
if (fr === "error") {
|
|
1038
|
+
const failure: Record<string, unknown> & Error = new Error(
|
|
1039
|
+
"OpenAI-compatible stream terminated with finish_reason error"
|
|
1040
|
+
) as Record<string, unknown> & Error;
|
|
1041
|
+
failure["url"] = url;
|
|
1042
|
+
throw toConciseProviderError(failure, prefix, clean);
|
|
1043
|
+
}
|
|
1044
|
+
finishReason = fr;
|
|
1045
|
+
}
|
|
1046
|
+
const chunkUsage = msg["usage"] as ChatResponse["usage"] | undefined;
|
|
1047
|
+
if (chunkUsage) {
|
|
1048
|
+
usage = mapUsage(chunkUsage);
|
|
1049
|
+
completedBody = completedBody ?? msg;
|
|
1050
|
+
eventStream.push({ type: "usage", usage });
|
|
1051
|
+
}
|
|
1052
|
+
};
|
|
1053
|
+
|
|
1054
|
+
while (true) {
|
|
1055
|
+
if (linked.signal.aborted || eventStream.isCancelled()) {
|
|
1056
|
+
try {
|
|
1057
|
+
await reader.cancel();
|
|
1058
|
+
} catch {}
|
|
1059
|
+
throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
|
|
1060
|
+
}
|
|
1061
|
+
const { done, value } = await reader.read();
|
|
1062
|
+
if (done) break;
|
|
1063
|
+
const chunk = decoder.decode(value, { stream: true });
|
|
1064
|
+
for (const m of parser.feed(chunk)) handleMessage(m.data);
|
|
1065
|
+
}
|
|
1066
|
+
for (const m of parser.flush()) handleMessage(m.data);
|
|
1067
|
+
try {
|
|
1068
|
+
reader.releaseLock();
|
|
1069
|
+
} catch {}
|
|
1070
|
+
|
|
1071
|
+
if (linked.signal.aborted || eventStream.isCancelled() || aborted) {
|
|
1072
|
+
throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
|
|
1073
|
+
}
|
|
1074
|
+
|
|
1075
|
+
const toolCalls: ToolCallRecord[] = [];
|
|
1076
|
+
for (const [index, c] of calls) {
|
|
1077
|
+
const args = parseStreamedToolArguments(c.startArgs, c.deltaArgs);
|
|
1078
|
+
const sig = signatures.get(index);
|
|
1079
|
+
const record: ToolCallRecord = {
|
|
1080
|
+
id: c.id || `call_${Math.random().toString(36).slice(2, 9)}`,
|
|
1081
|
+
name: c.name,
|
|
1082
|
+
arguments: args,
|
|
1083
|
+
rawArguments: c.startArgs + c.deltaArgs,
|
|
1084
|
+
...(sig ? { thoughtSignature: sig } : {}),
|
|
1085
|
+
};
|
|
1086
|
+
toolCalls.push(record);
|
|
1087
|
+
eventStream.push({ type: "tool_call_complete", toolCall: record });
|
|
1088
|
+
}
|
|
1089
|
+
|
|
1090
|
+
noteProviderTurn(sessionId, prefix);
|
|
1091
|
+
if (toolCalls.length > 0) finishReason = "tool_calls";
|
|
1092
|
+
|
|
1093
|
+
const cleanThinking = thinking.replace(/\n{3,}/g, "\n\n").trim();
|
|
1094
|
+
const finalThoughtSignature = toolCalls.map((c) => c.thoughtSignature).find(Boolean);
|
|
1095
|
+
const finalResponse = new AgentResponse({
|
|
1096
|
+
text,
|
|
1097
|
+
thinking: cleanThinking || undefined,
|
|
1098
|
+
thoughtSignature: finalThoughtSignature,
|
|
1099
|
+
toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
|
|
1100
|
+
usage,
|
|
1101
|
+
finishReason,
|
|
1102
|
+
responseId,
|
|
1103
|
+
model: clean,
|
|
1104
|
+
provider: prefix as ProviderId,
|
|
1105
|
+
raw: { request: rawRequest, response: { ...responseMeta, body: completedBody } },
|
|
1106
|
+
durationMs: Date.now() - startTime,
|
|
1107
|
+
});
|
|
1108
|
+
eventStream.push({ type: "done", delta: "", usage, finishReason, responseId });
|
|
1109
|
+
eventStream.end(finalResponse);
|
|
1110
|
+
} catch (err: unknown) {
|
|
1111
|
+
const raw = err instanceof Error ? err : new Error(String(err));
|
|
1112
|
+
const isAbort =
|
|
1113
|
+
linked.signal.aborted ||
|
|
1114
|
+
eventStream.isCancelled() ||
|
|
1115
|
+
(raw as { name?: string }).name === "AbortError" ||
|
|
1116
|
+
/abort|cancell?ed/i.test(String((raw as { message?: string }).message ?? raw));
|
|
1117
|
+
eventStream.fail(
|
|
1118
|
+
isAbort ? Object.assign(new Error("Stream aborted"), { name: "AbortError" }) : raw
|
|
1119
|
+
);
|
|
1120
|
+
} finally {
|
|
1121
|
+
try {
|
|
1122
|
+
options?.signal?.removeEventListener("abort", forwardUserAbort);
|
|
1123
|
+
} catch {}
|
|
1124
|
+
try {
|
|
1125
|
+
removeStreamCancel();
|
|
1126
|
+
} catch {}
|
|
1127
|
+
}
|
|
1128
|
+
})();
|
|
1129
|
+
|
|
1130
|
+
return eventStream;
|
|
1131
|
+
}
|
|
1132
|
+
}
|
|
1133
|
+
|
|
1134
|
+
/**
|
|
1135
|
+
* Creates a custom OpenAI-compatible provider backed by native REST.
|
|
1136
|
+
* @example `const provider = createOpenAICompatibleProvider("groq", { baseUrl: "https://api.groq.com/openai/v1" });`
|
|
1137
|
+
*/
|
|
1138
|
+
export function createOpenAICompatibleProvider(
|
|
1139
|
+
prefix: string,
|
|
1140
|
+
opts?: CustomProviderOptions
|
|
1141
|
+
): OpenAICompatibleChatProvider {
|
|
1142
|
+
return new OpenAICompatibleChatProvider(prefix, opts);
|
|
1143
|
+
}
|
|
1144
|
+
/** Backward-compatible alias for {@link createOpenAICompatibleProvider}. @example `createCustomProvider("ollama")` */
|
|
1145
|
+
export const createCustomProvider = createOpenAICompatibleProvider;
|
|
1146
|
+
/** Backward-compatible class alias for {@link OpenAICompatibleChatProvider}. @example `new CustomProvider("ollama")` */
|
|
1147
|
+
export const CustomProvider = OpenAICompatibleChatProvider;
|