agent-accelerator 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +83 -112
  2. package/{SYSTEM_PROMPT_AGENT.md → examples/prompts/SYSTEM_PROMPT_AGENT.md} +1 -1
  3. package/package.json +14 -9
  4. package/src/agent/agent.ts +191 -71
  5. package/src/agent/context.ts +1 -0
  6. package/src/agent/delegation.ts +61 -12
  7. package/src/agent/loop.ts +71 -48
  8. package/src/data/README.md +6 -6
  9. package/src/index.ts +130 -45
  10. package/src/models/catalog-cache.ts +60 -9
  11. package/src/models/catalog.ts +53 -7
  12. package/src/providers/google.ts +926 -0
  13. package/src/providers/openai-compat.ts +1147 -0
  14. package/src/providers/openai.ts +959 -0
  15. package/src/providers/openrouter-responses.ts +949 -0
  16. package/src/providers/openrouter.ts +1037 -0
  17. package/src/{ai-sdk → providers}/registry.ts +43 -55
  18. package/src/providers.ts +490 -0
  19. package/src/streaming/sse-parser.ts +6 -4
  20. package/src/tools/executor.ts +19 -6
  21. package/src/tools/schema.ts +21 -11
  22. package/src/types/agent.ts +8 -1
  23. package/src/types/core.ts +1 -1
  24. package/src/types/message.ts +5 -0
  25. package/src/types/model.ts +3 -7
  26. package/src/types/provider-payloads.ts +2 -84
  27. package/src/types/tool.ts +6 -0
  28. package/src/update-models.ts +56 -0
  29. package/src/utils/cache.ts +1 -1
  30. package/src/utils/documents.ts +517 -0
  31. package/src/utils/env.ts +0 -7
  32. package/src/{ai-sdk → utils}/errors.ts +61 -2
  33. package/src/utils/headers.ts +10 -20
  34. package/src/utils/media.ts +5 -2
  35. package/src/utils/retry.ts +89 -0
  36. package/src/utils/serialization.ts +15 -0
  37. package/src/ai-sdk/converters.ts +0 -342
  38. package/src/ai-sdk/executor.ts +0 -454
  39. package/src/ai-sdk/index.ts +0 -55
  40. package/src/ai-sdk/model-provider.ts +0 -303
  41. package/src/ai-sdk/options.ts +0 -306
  42. package/src/ai-sdk/provider.ts +0 -415
  43. package/src/tokens/counter.ts +0 -136
  44. /package/{SYSTEM_PROMPT.md → examples/prompts/SYSTEM_PROMPT.md} +0 -0
  45. /package/{SYSTEM_PROMPT_TOOLS.md → examples/prompts/SYSTEM_PROMPT_TOOLS.md} +0 -0
@@ -0,0 +1,1147 @@
1
+ /**
2
+ * Generic OpenAI-compatible Chat Completions provider
3
+ * (`POST {baseUrl}/chat/completions`, streaming on the same endpoint).
4
+ *
5
+ * Serves EVERY custom prefix (`groq/…`, `ollama/…`, `cerebras/…`, …) with one
6
+ * strict-subset implementation of the OpenAI Chat Completions wire — no SDK
7
+ * transport anywhere in this file.
8
+ *
9
+ * All endpoint-specific HTTP, headers, auth, request construction,
10
+ * response/SSE parsing, and wire transformations live HERE — never in the
11
+ * canonical layer (`src/providers.ts`).
12
+ *
13
+ * Strictness contract (differs from the old lenient passthrough on purpose):
14
+ * - Requests carry ONLY standard Chat Completions fields (`model`, `messages`,
15
+ * `tools`, `tool_choice`, `stream`). No `reasoning`, `service_tier`,
16
+ * `session_id`, `prompt_cache_key`, or other extras: strict endpoints
17
+ * (groq, ollama, …) fail unknown body properties, so anything unsupported
18
+ * is warn+drop, never sent.
19
+ * - `text`, `image_url`, and `input_audio` parts are sent; `video`/`file`
20
+ * parts fail fast with a one-line error naming the remedy (files point at
21
+ * `convertDocumentToMarkdown`). The old transport let these through to
22
+ * confusing provider 400s.
23
+ * - Prior-turn reasoning is NOT resent (chat history is messages + tool calls
24
+ * only), except echoed Gemini thought signatures (see below).
25
+ * - Gemini models served through OpenAI-compatible endpoints keep working:
26
+ * `extra_content.google.thought_signature` on tool calls is captured into
27
+ * the canonical `thoughtSignature` and echoed back verbatim on the next
28
+ * turn (otherwise the endpoint 400s on missing signatures). Echoes happen
29
+ * ONLY when a previous turn produced a signature — other endpoints never
30
+ * see the field.
31
+ *
32
+ * Notes:
33
+ * - STATELESS per request: every turn sends the full canonical `messages`
34
+ * array explicitly. Session affinity is headers-only best effort
35
+ * (`x-session-id`); bodies carry no affinity key.
36
+ * - IDs are provider-generated and echoed verbatim (`tool_call_id`). The
37
+ * `call_${random}` fallback in parsers is a local canonical correlation id
38
+ * only; it is never sent.
39
+ * - Local endpoints (localhost / loopback / RFC-1918 / *.local) may omit the
40
+ * API key (no `Authorization` header is sent then). Any other endpoint
41
+ * without a key fails fast instead of sending a dummy bearer into a 401.
42
+ */
43
+ import type {
44
+ Provider,
45
+ ProviderId,
46
+ ModelSpec,
47
+ ProviderRequestOptions,
48
+ ProviderGenerateResult,
49
+ ProviderRawData,
50
+ } from "../types/model.ts";
51
+ import type { ProviderContext, ContentPart } from "../types/message.ts";
52
+ import type { StandardToolDeclaration, ToolCallRecord } from "../types/tool.ts";
53
+ import type { TokenUsage, ThinkingConfig } from "../types/core.ts";
54
+ import { AssistantMessageEventStream } from "../streaming/event-stream.ts";
55
+ import { SSEParser } from "../streaming/sse-parser.ts";
56
+ import { AgentResponse } from "../types/response.ts";
57
+ import { getApiKey, getEnv } from "../utils/env.ts";
58
+ import { buildSessionHeaders } from "../utils/headers.ts";
59
+ import { normalizeMediaInput } from "../utils/media.ts";
60
+ import { safeStringify } from "../utils/serialization.ts";
61
+ import { toConciseProviderError, assertModalitiesSupported } from "../utils/errors.ts";
62
+ import { withRetries } from "../utils/retry.ts";
63
+ import { getModelFromCatalog, createGenericModelSpec } from "../models/catalog.ts";
64
+ import {
65
+ applyCacheForCustom,
66
+ mapToolChoiceToOpenAI,
67
+ emitProviderWarning,
68
+ noteProviderTurn,
69
+ parseStreamedToolArguments,
70
+ } from "../providers.ts";
71
+
72
+ // ---------------------------------------------------------------------------
73
+ // Chat Completions wire shapes (strict standard subset)
74
+ // ---------------------------------------------------------------------------
75
+
76
+ type ChatContentPart =
77
+ | { type: "text"; text: string }
78
+ | { type: "image_url"; image_url: { url: string } }
79
+ | { type: "input_audio"; input_audio: { data: string; format: string } }
80
+ | { type: string; [k: string]: unknown };
81
+
82
+ type ChatMessage =
83
+ | { role: "system" | "user"; content: string | ChatContentPart[]; name?: string }
84
+ | {
85
+ role: "assistant";
86
+ content: string | null;
87
+ tool_calls?: Array<{
88
+ id: string;
89
+ type: "function";
90
+ function: { name: string; arguments: string };
91
+ extra_content?: { google?: { thought_signature?: string } };
92
+ }>;
93
+ name?: string;
94
+ }
95
+ | { role: "tool"; content: string; tool_call_id: string; name?: string }
96
+ | { type: string; [k: string]: unknown };
97
+
98
+ interface ChatRequestBody {
99
+ model: string;
100
+ messages: ChatMessage[];
101
+ tools?: Array<Record<string, unknown>>;
102
+ tool_choice?: string | { type: "function"; function: { name: string } };
103
+ stream?: boolean;
104
+ [k: string]: unknown;
105
+ }
106
+
107
+ interface ChatChoice {
108
+ index?: number;
109
+ message?: {
110
+ role?: string;
111
+ content?: string | null;
112
+ tool_calls?: Array<{
113
+ id?: string;
114
+ type?: string;
115
+ index?: number;
116
+ function?: { name?: string; arguments?: string };
117
+ extra_content?: { google?: { thought_signature?: string } };
118
+ }>;
119
+ reasoning?: string | null;
120
+ reasoning_details?: Array<{ type?: string; text?: string }>;
121
+ refusal?: string | null;
122
+ };
123
+ delta?: {
124
+ role?: string;
125
+ content?: string | null;
126
+ tool_calls?: Array<{
127
+ index?: number;
128
+ id?: string;
129
+ type?: string;
130
+ function?: { name?: string; arguments?: string };
131
+ }>;
132
+ reasoning_content?: string;
133
+ reasoning?: string;
134
+ reasoning_details?: Array<{ type?: string; text?: string }>;
135
+ refusal?: string | null;
136
+ };
137
+ finish_reason?: string | null;
138
+ error?: { message?: string; code?: string | number };
139
+ }
140
+
141
+ interface ChatResponse {
142
+ id?: string;
143
+ object?: string;
144
+ created?: number;
145
+ model?: string;
146
+ choices?: ChatChoice[];
147
+ usage?: {
148
+ prompt_tokens?: number;
149
+ completion_tokens?: number;
150
+ total_tokens?: number;
151
+ prompt_tokens_details?: { cached_tokens?: number; cache_write_tokens?: number };
152
+ completion_tokens_details?: { reasoning_tokens?: number };
153
+ cost?: number;
154
+ [k: string]: unknown;
155
+ };
156
+ error?: { message?: string; code?: string | number };
157
+ [k: string]: unknown;
158
+ }
159
+
160
+ const DEFAULT_BASE_URL = "https://api.openai.com/v1";
161
+
162
+ /** Headers that must never leak onto native REST requests. */
163
+ const INTERNAL_HEADERS = new Set([
164
+ "x-thought-signature-map",
165
+ "x-cached-content-id",
166
+ "x-multimodal-user-content",
167
+ ]);
168
+
169
+ // ---------------------------------------------------------------------------
170
+ // Request building (canonical -> Chat Completions)
171
+ // ---------------------------------------------------------------------------
172
+
173
+ function envPrefixOf(prefix: string): string {
174
+ return prefix.toUpperCase().replace(/[^A-Z0-9]/g, "_");
175
+ }
176
+
177
+ function resolveBaseUrl(
178
+ prefix: string,
179
+ configuredBaseUrl: string | undefined,
180
+ options?: ProviderRequestOptions
181
+ ): string {
182
+ const envPrefix = envPrefixOf(prefix);
183
+ return (
184
+ options?.baseUrl ||
185
+ configuredBaseUrl ||
186
+ options?.env?.[`${envPrefix}_BASE_URL`] ||
187
+ options?.env?.[`${envPrefix}_BASEURL`] ||
188
+ options?.env?.[`${envPrefix}_API_BASE`] ||
189
+ getEnv(`${envPrefix}_BASE_URL`) ||
190
+ getEnv(`${envPrefix}_BASEURL`) ||
191
+ getEnv(`${envPrefix}_API_BASE`) ||
192
+ getEnv("OPENAI_BASE_URL") ||
193
+ getEnv("OPENAI_API_BASE") ||
194
+ DEFAULT_BASE_URL
195
+ ).replace(/\/+$/, "");
196
+ }
197
+
198
+ function resolveApiKey(
199
+ prefix: string,
200
+ configuredApiKey: string | undefined,
201
+ options?: ProviderRequestOptions
202
+ ): string | undefined {
203
+ const envPrefix = envPrefixOf(prefix);
204
+ return (
205
+ options?.apiKey ||
206
+ configuredApiKey ||
207
+ options?.env?.[`${envPrefix}_API_KEY`] ||
208
+ options?.env?.[`${envPrefix}_BASE_API_KEY`] ||
209
+ options?.env?.["OPENAI_BASE_API_KEY"] ||
210
+ options?.env?.["OPENAI_API_KEY"] ||
211
+ getEnv(`${envPrefix}_API_KEY`) ||
212
+ getEnv(`${envPrefix}_BASE_API_KEY`) ||
213
+ getEnv("OPENAI_BASE_API_KEY") ||
214
+ getEnv("OPENAI_API_KEY") ||
215
+ getApiKey(prefix, undefined, options?.env) ||
216
+ undefined
217
+ );
218
+ }
219
+
220
+ /** Local endpoints (loopback / LAN / *.local) may omit the API key. */
221
+ function isLocalEndpoint(baseUrl: string): boolean {
222
+ try {
223
+ const host = new URL(baseUrl).hostname.toLowerCase();
224
+ if (
225
+ host === "localhost" ||
226
+ host === "127.0.0.1" ||
227
+ host === "0.0.0.0" ||
228
+ host === "::1" ||
229
+ host.endsWith(".local")
230
+ ) {
231
+ return true;
232
+ }
233
+ if (/^10\./.test(host) || /^192\.168\./.test(host) || /^172\.(1[6-9]|2\d|3[01])\./.test(host)) {
234
+ return true;
235
+ }
236
+ return false;
237
+ } catch {
238
+ return false;
239
+ }
240
+ }
241
+
242
+ /** Strips ONLY the `{prefix}/` scope. Bare ids pass through untouched. */
243
+ function cleanModelId(prefix: string, model: string | ModelSpec): string {
244
+ const rawId = typeof model === "string" ? model : model.id;
245
+ const escaped = prefix.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
246
+ return rawId.replace(new RegExp(`^${escaped}/`, "i"), "");
247
+ }
248
+
249
+ function toCompatTools(tools?: StandardToolDeclaration[]): Array<Record<string, unknown>> | undefined {
250
+ if (!tools || tools.length === 0) return undefined;
251
+ return tools.map((t) => ({
252
+ type: "function",
253
+ function: {
254
+ name: t.name,
255
+ description: t.description,
256
+ parameters: (t.parameters || { type: "object", properties: {} }) as Record<string, unknown>,
257
+ ...(t.strict !== undefined ? { strict: t.strict } : {}),
258
+ },
259
+ }));
260
+ }
261
+
262
+ /** Thinking levels have no portable wire shape on generic endpoints: warn once, send nothing. */
263
+ function warnThinkingDropped(prefix: string, thinking: ThinkingConfig | undefined, modelRef: string): void {
264
+ if (!thinking) return;
265
+ const level = thinking.level;
266
+ if (!level || level === "dynamic") return; // server default either way
267
+ emitProviderWarning({
268
+ provider: prefix,
269
+ capability: "thinking level",
270
+ requested: `${level} (${modelRef})`,
271
+ reason: "custom OpenAI-compatible endpoints define no portable reasoning control; levels vary by vendor and strict endpoints reject unknown fields.",
272
+ fallback: "the server default (no reasoning payload is sent)",
273
+ });
274
+ }
275
+
276
+ function audioFormatFor(mimeType?: string): string {
277
+ const mime = (mimeType || "").toLowerCase();
278
+ if (mime.includes("wav")) return "wav";
279
+ return "mp3";
280
+ }
281
+
282
+ /**
283
+ * Rejects `video`/`file` parts before any network call: the strict Chat
284
+ * Completions subset has no shape for them, and a provider 400 would only say
285
+ * so confusingly. Images and audio keep flowing natively.
286
+ */
287
+ function assertNoVideoOrFileParts(
288
+ context: ProviderContext,
289
+ prefix: string,
290
+ modelId: string
291
+ ): void {
292
+ let kind: "video" | "file" | undefined;
293
+ for (const msg of context.messages) {
294
+ if (!Array.isArray((msg as { content?: unknown }).content)) continue;
295
+ for (const part of (msg as { content: Array<{ type?: unknown }> }).content) {
296
+ if ((part as { type?: unknown }).type === "video") {
297
+ kind = "video";
298
+ break;
299
+ }
300
+ if ((part as { type?: unknown }).type === "file") {
301
+ kind = "file";
302
+ break;
303
+ }
304
+ }
305
+ if (kind) break;
306
+ }
307
+ if (!kind) return;
308
+ const message =
309
+ kind === "video"
310
+ ? `[${prefix}/${modelId}] unsupported video input (generic OpenAI-compatible endpoints carry text/image/audio only). Use a video-capable model (e.g. google/gemini-*) or drop the video part.`
311
+ : `[${prefix}/${modelId}] unsupported file input (generic OpenAI-compatible endpoints carry text/image/audio only). Convert the document to Markdown with the convert_document_to_markdown tool (or Agent bypassInputFileModality) and send it as text instead.`;
312
+ const err = new Error(message);
313
+ err.name = "AgentAccelProviderError";
314
+ Object.defineProperties(err, {
315
+ provider: { value: prefix, enumerable: false },
316
+ model: { value: modelId, enumerable: false },
317
+ });
318
+ throw err;
319
+ }
320
+
321
+ async function contentPartsToBlocks(parts: ContentPart[]): Promise<ChatContentPart[]> {
322
+ const blocks: ChatContentPart[] = [];
323
+ for (const part of parts) {
324
+ if (part.type === "text" && part.text) {
325
+ blocks.push({ type: "text", text: part.text });
326
+ } else if (part.type === "image") {
327
+ const raw = (part as { image?: unknown }).image;
328
+ if (typeof raw === "string" && (raw.startsWith("http://") || raw.startsWith("https://"))) {
329
+ blocks.push({ type: "image_url", image_url: { url: raw } });
330
+ continue;
331
+ }
332
+ const norm = await normalizeMediaInput(
333
+ raw as string | Uint8Array | ArrayBuffer,
334
+ (part as { mimeType?: string }).mimeType
335
+ );
336
+ blocks.push({ type: "image_url", image_url: { url: norm.dataUrl } });
337
+ } else if (part.type === "audio") {
338
+ const raw = (part as { audio?: unknown }).audio;
339
+ const mimeType = (part as { mimeType?: string }).mimeType;
340
+ // The OpenAI chat audio shape carries data only — always inline.
341
+ const norm = await normalizeMediaInput(
342
+ raw as string | Uint8Array | ArrayBuffer,
343
+ mimeType
344
+ );
345
+ blocks.push({
346
+ type: "input_audio",
347
+ input_audio: { data: norm.base64Data, format: audioFormatFor(norm.mimeType) },
348
+ });
349
+ }
350
+ // video/file never reach here: assertNoVideoOrFileParts runs first.
351
+ }
352
+ return blocks;
353
+ }
354
+
355
+ function resultToOutput(result: unknown): string {
356
+ if (typeof result === "string") return result;
357
+ return safeStringify(result);
358
+ }
359
+
360
+ /**
361
+ * Builds the FULL explicit history as Chat Completions `messages` (stateless
362
+ * transport: every turn carries everything). Prior-turn reasoning is NOT
363
+ * resent — except echoed Gemini thought signatures, which strict
364
+ * OpenAI-compatible endpoints require on tool calls they generated.
365
+ */
366
+ async function fullHistoryMessages(context: ProviderContext): Promise<ChatMessage[]> {
367
+ const pairing = new Map<string, string>();
368
+ for (const m of context.messages) {
369
+ if (m.role === "assistant" && Array.isArray(m.content)) {
370
+ for (const part of m.content) {
371
+ if (part.type === "tool_call") {
372
+ pairing.set(part.id, part.callId || part.id);
373
+ }
374
+ }
375
+ }
376
+ }
377
+
378
+ const messages: ChatMessage[] = [];
379
+ if (context.systemPrompt) {
380
+ messages.push({ role: "system", content: context.systemPrompt });
381
+ }
382
+ for (const m of context.messages) {
383
+ if (m.role === "system") continue;
384
+ if (m.role === "user") {
385
+ if (typeof m.content === "string") {
386
+ if (m.content) messages.push({ role: "user", content: m.content });
387
+ } else {
388
+ const blocks = await contentPartsToBlocks(m.content);
389
+ if (blocks.length > 0) messages.push({ role: "user", content: blocks });
390
+ }
391
+ } else if (m.role === "assistant") {
392
+ if (typeof m.content === "string") {
393
+ if (m.content) {
394
+ messages.push({ role: "assistant", content: m.content });
395
+ }
396
+ continue;
397
+ }
398
+ const texts: string[] = [];
399
+ const calls: Array<{
400
+ id: string;
401
+ type: "function";
402
+ function: { name: string; arguments: string };
403
+ extra_content?: { google: { thought_signature: string } };
404
+ }> = [];
405
+ for (const part of m.content) {
406
+ if (part.type === "tool_call") {
407
+ const call: {
408
+ id: string;
409
+ type: "function";
410
+ function: { name: string; arguments: string };
411
+ extra_content?: { google: { thought_signature: string } };
412
+ } = {
413
+ id: part.callId || part.id,
414
+ type: "function",
415
+ function: { name: part.name, arguments: JSON.stringify(part.arguments || {}) },
416
+ };
417
+ // Echo provider-issued thought signatures verbatim so endpoints
418
+ // that minted them (Gemini behind a compat proxy) keep working.
419
+ // Only present when a previous turn captured one — other endpoints
420
+ // never see this field.
421
+ if (part.thoughtSignature) {
422
+ call.extra_content = { google: { thought_signature: part.thoughtSignature } };
423
+ }
424
+ calls.push(call);
425
+ } else if (part.type === "text" && part.text) {
426
+ texts.push(part.text);
427
+ }
428
+ }
429
+ messages.push({
430
+ role: "assistant",
431
+ content: texts.length > 0 ? texts.join("\n") : null,
432
+ ...(calls.length > 0 ? { tool_calls: calls } : {}),
433
+ });
434
+ } else if (m.role === "tool") {
435
+ if (!Array.isArray(m.content)) continue;
436
+ for (const part of m.content) {
437
+ if (part.type === "tool_result") {
438
+ messages.push({
439
+ role: "tool",
440
+ tool_call_id: pairing.get(part.id) || part.id,
441
+ content: resultToOutput(part.result),
442
+ name: part.name,
443
+ });
444
+ }
445
+ }
446
+ }
447
+ }
448
+ return messages;
449
+ }
450
+
451
+ // ---------------------------------------------------------------------------
452
+ // Response mapping (Chat Completions -> canonical)
453
+ // ---------------------------------------------------------------------------
454
+
455
+ function mapUsage(raw?: ChatResponse["usage"]): TokenUsage {
456
+ const input = raw?.prompt_tokens ?? 0;
457
+ const output = raw?.completion_tokens ?? 0;
458
+ // Canonical invariant: cache hits are a SUBSET of input (no turn may
459
+ // report a >100% hit rate).
460
+ const cached = Math.min(raw?.prompt_tokens_details?.cached_tokens ?? 0, input);
461
+ const usage: TokenUsage = {
462
+ inputTokens: input,
463
+ outputTokens: output,
464
+ totalTokens: raw?.total_tokens ?? input + output,
465
+ cachedTokens: cached,
466
+ cacheReadTokens: cached,
467
+ cacheWriteTokens: raw?.prompt_tokens_details?.cache_write_tokens ?? 0,
468
+ thinkingTokens: raw?.completion_tokens_details?.reasoning_tokens ?? 0,
469
+ };
470
+ if (typeof raw?.cost === "number" && raw.cost > 0) {
471
+ usage.cost = { totalCost: raw.cost };
472
+ }
473
+ return usage;
474
+ }
475
+
476
+ function parseArguments(raw: string | undefined): Record<string, unknown> {
477
+ if (!raw) return {};
478
+ try {
479
+ const parsed: unknown = JSON.parse(raw);
480
+ if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
481
+ return parsed as Record<string, unknown>;
482
+ }
483
+ return { raw };
484
+ } catch {
485
+ return { raw };
486
+ }
487
+ }
488
+
489
+ /** Trims/collapses assembled thinking parts (no edge-tripling). */
490
+ function normalizeThinkingParts(parts: string[]): string | undefined {
491
+ const cleaned = parts
492
+ .map((p) => p.replace(/\n{3,}/g, "\n\n").trim())
493
+ .filter((p) => p.length > 0);
494
+ return cleaned.length > 0 ? cleaned.join("\n") : undefined;
495
+ }
496
+
497
+ function extractSignature(rawTc: {
498
+ extra_content?: { google?: { thought_signature?: string } };
499
+ }): string | undefined {
500
+ const sig = rawTc?.extra_content?.google?.thought_signature;
501
+ return typeof sig === "string" && sig ? sig : undefined;
502
+ }
503
+
504
+ function throwResponseError(
505
+ response: ChatResponse,
506
+ prefix: string,
507
+ modelId: string
508
+ ): void {
509
+ const message =
510
+ response.error && typeof response.error.message === "string" && response.error.message
511
+ ? response.error.message
512
+ : "OpenAI-compatible response failed";
513
+ const failure: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
514
+ if (response.error?.code !== undefined) failure["code"] = response.error.code;
515
+ throw toConciseProviderError(failure, prefix, modelId);
516
+ }
517
+
518
+ function parseResponse(
519
+ response: ChatResponse,
520
+ prefix: string,
521
+ modelId: string,
522
+ durationMs: number,
523
+ raw: ProviderRawData
524
+ ): ProviderGenerateResult {
525
+ // Provider-interrupted generations arrive as HTTP 200 carrying ONLY `error`
526
+ // (no `choices`) — must check the body, not just the status.
527
+ if (response.error) throwResponseError(response, prefix, modelId);
528
+
529
+ const choice = response.choices?.[0];
530
+ const message = choice?.message;
531
+ const text = typeof message?.content === "string" ? message.content : "";
532
+ // Thinking tolerance: plain `reasoning` plus per-block texts (deepseek-style
533
+ // `reasoning_details`). Details win when both ride along; both are
534
+ // display-only here — never resent.
535
+ const thinkingParts: string[] = [];
536
+ const detailTexts: string[] = [];
537
+ for (const block of message?.reasoning_details ?? []) {
538
+ if (block && typeof block.text === "string" && block.text) detailTexts.push(block.text);
539
+ }
540
+ if (detailTexts.length > 0) {
541
+ thinkingParts.push(...detailTexts);
542
+ } else if (typeof message?.reasoning === "string" && message.reasoning) {
543
+ thinkingParts.push(message.reasoning);
544
+ }
545
+ const toolCalls: ToolCallRecord[] = [];
546
+ for (const tc of message?.tool_calls ?? []) {
547
+ const args = parseArguments(tc.function?.arguments);
548
+ const sig = extractSignature(tc);
549
+ toolCalls.push({
550
+ id: tc.id || `call_${Math.random().toString(36).slice(2, 9)}`,
551
+ name: tc.function?.name || "unknown",
552
+ arguments: args,
553
+ rawArguments:
554
+ typeof tc.function?.arguments === "string"
555
+ ? tc.function.arguments
556
+ : JSON.stringify(tc.function?.arguments ?? {}),
557
+ ...(sig ? { thoughtSignature: sig } : {}),
558
+ });
559
+ }
560
+
561
+ const wireReason = choice?.finish_reason ?? undefined;
562
+ return {
563
+ text,
564
+ thinking: normalizeThinkingParts(thinkingParts),
565
+ thoughtSignature: toolCalls.map((c) => c.thoughtSignature).find(Boolean),
566
+ toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
567
+ usage: mapUsage(response.usage),
568
+ finishReason: toolCalls.length > 0 ? "tool_calls" : wireReason || "stop",
569
+ responseId: response.id,
570
+ model: modelId,
571
+ provider: prefix as ProviderId,
572
+ raw,
573
+ durationMs,
574
+ };
575
+ }
576
+
577
+ function redactedHeaders(headers: Record<string, string>): Record<string, string> {
578
+ const out: Record<string, string> = {};
579
+ for (const [k, v] of Object.entries(headers)) {
580
+ out[k] = k.toLowerCase() === "authorization" ? "[REDACTED]" : v;
581
+ }
582
+ return out;
583
+ }
584
+
585
+ function readErrorPayload(bodyText: string): { message: string; code?: string | number } {
586
+ try {
587
+ const parsed: unknown = JSON.parse(bodyText);
588
+ const first = Array.isArray(parsed) ? parsed[0] : parsed;
589
+ const err = (first as { error?: { message?: string; code?: string | number } })?.error;
590
+ if (err && typeof err.message === "string") {
591
+ return { message: err.message, code: err.code };
592
+ }
593
+ return { message: bodyText.slice(0, 300) };
594
+ } catch {
595
+ return { message: bodyText.slice(0, 300) };
596
+ }
597
+ }
598
+
599
+ // ---------------------------------------------------------------------------
600
+ // Provider
601
+ // ---------------------------------------------------------------------------
602
+
603
+ export interface CustomProviderOptions {
604
+ /** Human label, defaults to `${prefix} (OpenAI-compatible)` */
605
+ name?: string;
606
+ /** Static baseUrl — overrides env. Env `{PREFIX}_BASE_URL` still wins at runtime if set. */
607
+ baseUrl?: string;
608
+ /** Static apiKey — overrides env. Explicit per-call `apiKey` still wins. */
609
+ apiKey?: string;
610
+ /** Default baseUrl when no env is set. Defaults to OpenAI cloud. */
611
+ defaultBaseUrl?: string;
612
+ }
613
+
614
+ /**
615
+ * Generic OpenAI-compatible provider implemented directly on the Chat
616
+ * Completions REST API (`POST {baseUrl}/chat/completions`, streaming on the
617
+ * same endpoint). One class serves every custom prefix.
618
+ */
619
+ export class OpenAICompatibleChatProvider implements Provider {
620
+ readonly id: ProviderId;
621
+ readonly name: string;
622
+ readonly models: ModelSpec[] = [];
623
+ private readonly prefix: string;
624
+ private readonly configuredBaseUrl?: string;
625
+ private readonly configuredApiKey?: string;
626
+
627
+ /**
628
+ * Creates an OpenAI-compatible provider for any endpoint prefix.
629
+ *
630
+ * @param prefix Prefix used in model strings and environment variables,
631
+ * such as `groq` for `groq/llama-3.3-70b-versatile`.
632
+ * @param opts Optional endpoint, key, and display-name overrides.
633
+ *
634
+ * @example
635
+ * ```ts
636
+ * const groq = new OpenAICompatibleChatProvider("groq", {
637
+ * baseUrl: "https://api.groq.com/openai/v1",
638
+ * apiKey: process.env.GROQ_API_KEY,
639
+ * });
640
+ * ```
641
+ */
642
+ constructor(prefix: string, opts?: CustomProviderOptions) {
643
+ const norm = prefix.trim().toLowerCase().replace(/\/.*$/, "");
644
+ this.prefix = norm;
645
+ this.id = norm as ProviderId;
646
+ this.name = opts?.name ?? `${norm} (OpenAI-compatible)`;
647
+ this.configuredBaseUrl = opts?.baseUrl ?? opts?.defaultBaseUrl;
648
+ this.configuredApiKey = opts?.apiKey;
649
+ }
650
+
651
+ getModel(modelId: string): ModelSpec | undefined {
652
+ const clean = cleanModelId(this.prefix, modelId);
653
+ return (
654
+ getModelFromCatalog(this.id, modelId) ||
655
+ getModelFromCatalog(this.id, clean) ||
656
+ this.models.find((m) => m.id === modelId || m.id === clean) ||
657
+ createGenericModelSpec(this.id, clean)
658
+ );
659
+ }
660
+
661
+ private baseUrl(options?: ProviderRequestOptions): string {
662
+ return resolveBaseUrl(this.prefix, this.configuredBaseUrl, options);
663
+ }
664
+
665
+ private apiKey(options?: ProviderRequestOptions): string | undefined {
666
+ return resolveApiKey(this.prefix, this.configuredApiKey, options);
667
+ }
668
+
669
+ private requireApiKey(baseUrl: string, options?: ProviderRequestOptions): string | undefined {
670
+ const key = this.apiKey(options);
671
+ if (!key && !isLocalEndpoint(baseUrl)) {
672
+ throw new Error(
673
+ `[Agent Accelerator] Missing API key for "${this.prefix}". Set ${envPrefixOf(this.prefix)}_API_KEY or pass apiKey. (Local endpoints may omit the key.)`
674
+ );
675
+ }
676
+ return key;
677
+ }
678
+
679
+ private async buildBody(
680
+ modelId: string,
681
+ context: ProviderContext,
682
+ options: ProviderRequestOptions | undefined,
683
+ tools: StandardToolDeclaration[] | undefined,
684
+ stream: boolean
685
+ ): Promise<ChatRequestBody> {
686
+ warnThinkingDropped(this.prefix, options?.thinking, `${this.prefix}/${modelId}`);
687
+ if (options?.serviceTier) {
688
+ emitProviderWarning({
689
+ provider: this.prefix,
690
+ capability: "service tier",
691
+ requested: `${options.serviceTier} (${this.prefix}/${modelId})`,
692
+ reason: "custom OpenAI-compatible endpoints define no service-tier primitive, and strict endpoints reject unknown body properties.",
693
+ fallback: "standard routing (no service_tier payload is sent)",
694
+ });
695
+ }
696
+ applyCacheForCustom(this.prefix, options?.cache, `${this.prefix}/${modelId}`);
697
+
698
+ const messages = await fullHistoryMessages(context);
699
+ const body: ChatRequestBody = {
700
+ model: modelId,
701
+ messages,
702
+ ...(stream ? { stream: true } : {}),
703
+ };
704
+ const compatTools = toCompatTools(tools);
705
+ if (compatTools) body.tools = compatTools;
706
+ const toolChoice = mapToolChoiceToOpenAI(
707
+ options?.toolChoice as "auto" | "none" | "required" | { type: "function"; function: { name: string } } | undefined
708
+ );
709
+ if (toolChoice !== undefined) {
710
+ // Chat Completions nests pinned names under `function`; the canonical
711
+ // mapper emits the flat Responses shape, so normalize here.
712
+ body.tool_choice =
713
+ typeof toolChoice === "object" && "name" in toolChoice && !("function" in toolChoice)
714
+ ? { type: "function", function: { name: (toolChoice as { name: string }).name } }
715
+ : (toolChoice as ChatRequestBody["tool_choice"]);
716
+ }
717
+ // No reasoning / service_tier / session / cache body primitives: strict
718
+ // endpoints reject unknown properties. Affinity is headers-only.
719
+ return body;
720
+ }
721
+
722
+ private requestInit(
723
+ body: ChatRequestBody,
724
+ options: ProviderRequestOptions | undefined,
725
+ apiKey: string | undefined,
726
+ sessionId: string | undefined,
727
+ signal?: AbortSignal
728
+ ): RequestInit {
729
+ const headers: Record<string, string> = buildSessionHeaders(
730
+ this.prefix,
731
+ options?.cache,
732
+ options?.headers,
733
+ sessionId
734
+ );
735
+ for (const [k, v] of Object.entries(headers)) {
736
+ if (INTERNAL_HEADERS.has(k.toLowerCase())) delete headers[k];
737
+ }
738
+ headers["Content-Type"] = "application/json";
739
+ if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
740
+ return { method: "POST", headers, body: JSON.stringify(body), signal };
741
+ }
742
+
743
+ private async doFetch(
744
+ url: string,
745
+ body: ChatRequestBody,
746
+ options: ProviderRequestOptions | undefined,
747
+ apiKey: string | undefined,
748
+ sessionId: string | undefined,
749
+ signal?: AbortSignal
750
+ ): Promise<{ status: number; statusText: string; headers: Record<string, string>; text: string }> {
751
+ const res = await fetch(url, this.requestInit(body, options, apiKey, sessionId, signal));
752
+ const text = await res.text();
753
+ const headers: Record<string, string> = {};
754
+ res.headers.forEach((v, k) => {
755
+ headers[k] = v;
756
+ });
757
+ return { status: res.status, statusText: res.statusText, headers, text };
758
+ }
759
+
760
+ private throwIfError(
761
+ status: number,
762
+ url: string,
763
+ bodyText: string,
764
+ modelId: string
765
+ ): void {
766
+ if (status >= 200 && status < 300) return;
767
+ const { message, code } = readErrorPayload(bodyText);
768
+ const err: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
769
+ (err as Record<string, unknown>)["statusCode"] = status;
770
+ (err as Record<string, unknown>)["status"] = status;
771
+ (err as Record<string, unknown>)["responseBody"] = bodyText.slice(0, 500);
772
+ (err as Record<string, unknown>)["url"] = url.split("?")[0];
773
+ if (code !== undefined) (err as Record<string, unknown>)["code"] = code;
774
+ throw toConciseProviderError(err, this.prefix, modelId);
775
+ }
776
+
777
+ async generate(
778
+ model: string | ModelSpec,
779
+ context: ProviderContext,
780
+ options?: ProviderRequestOptions
781
+ ): Promise<ProviderGenerateResult> {
782
+ const startTime = Date.now();
783
+ const clean = cleanModelId(this.prefix, model);
784
+ const baseUrl = this.baseUrl(options);
785
+ const apiKey = this.requireApiKey(baseUrl, options);
786
+ // Fail fast before any network call when the catalog knows the model
787
+ // lacks the requested modality. Unknown models skip the guard; the
788
+ // endpoint verdict surfaces concisely.
789
+ assertModalitiesSupported(context, this.prefix, clean);
790
+ assertNoVideoOrFileParts(context, this.prefix, clean);
791
+ const url = `${baseUrl}/chat/completions`;
792
+ const sessionId = options?.sessionId || options?.cache?.sessionId;
793
+ const tools = options?.tools as StandardToolDeclaration[] | undefined;
794
+ const body = await this.buildBody(clean, context, options, tools, false);
795
+
796
+ const auditRaw: Record<string, string> = {
797
+ ...buildSessionHeaders(this.prefix, options?.cache, options?.headers, sessionId),
798
+ "Content-Type": "application/json",
799
+ ...(apiKey ? { Authorization: "[REDACTED]" } : {}),
800
+ };
801
+ for (const k of Object.keys(auditRaw)) {
802
+ if (INTERNAL_HEADERS.has(k.toLowerCase())) delete auditRaw[k];
803
+ }
804
+ const rawRequest = {
805
+ url,
806
+ method: "POST",
807
+ headers: redactedHeaders(auditRaw),
808
+ body,
809
+ };
810
+
811
+ const doCall = async (): Promise<ProviderGenerateResult> => {
812
+ const res = await this.doFetch(url, body, options, apiKey, sessionId, options?.signal);
813
+ this.throwIfError(res.status, url, res.text, clean);
814
+ let response: ChatResponse;
815
+ try {
816
+ response = JSON.parse(res.text) as ChatResponse;
817
+ } catch {
818
+ throw toConciseProviderError(
819
+ Object.assign(new Error("Invalid JSON response from OpenAI-compatible Chat Completions API"), {
820
+ statusCode: res.status,
821
+ responseBody: res.text.slice(0, 500),
822
+ url,
823
+ }),
824
+ this.prefix,
825
+ clean
826
+ );
827
+ }
828
+ // Stateless turns never establish chains, but the session store still
829
+ // records the turn so other providers' switch detection keeps working.
830
+ noteProviderTurn(sessionId, this.prefix);
831
+ return parseResponse(
832
+ response,
833
+ this.prefix,
834
+ clean,
835
+ Date.now() - startTime,
836
+ { request: rawRequest, response: { status: res.status, statusText: res.statusText, headers: res.headers, body: response } }
837
+ );
838
+ };
839
+
840
+ try {
841
+ return await withRetries(doCall, {
842
+ maxRetries: options?.maxRetries,
843
+ maxRetryDelayMs: options?.maxRetryDelayMs,
844
+ signal: options?.signal,
845
+ label: { providerId: this.prefix, modelId: clean },
846
+ });
847
+ } catch (err) {
848
+ if (err instanceof Error && (err as { name?: string }).name === "AbortError") throw err;
849
+ throw err;
850
+ }
851
+ }
852
+
853
+ stream(
854
+ model: string | ModelSpec,
855
+ context: ProviderContext,
856
+ options?: ProviderRequestOptions
857
+ ): AssistantMessageEventStream {
858
+ const eventStream = new AssistantMessageEventStream();
859
+ const startTime = Date.now();
860
+ const prefix = this.prefix;
861
+ const clean = cleanModelId(this.prefix, model);
862
+
863
+ const linked = new AbortController();
864
+ const forwardUserAbort = () => {
865
+ try {
866
+ linked.abort((options?.signal as { reason?: unknown })?.reason);
867
+ } catch {
868
+ try {
869
+ linked.abort();
870
+ } catch {}
871
+ }
872
+ };
873
+ if (options?.signal?.aborted) forwardUserAbort();
874
+ else options?.signal?.addEventListener("abort", forwardUserAbort, { once: true });
875
+ const removeStreamCancel = eventStream.onCancel(() => {
876
+ try {
877
+ linked.abort();
878
+ } catch {}
879
+ });
880
+
881
+ (async () => {
882
+ try {
883
+ const baseUrl = this.baseUrl(options);
884
+ const apiKey = this.requireApiKey(baseUrl, options);
885
+ assertModalitiesSupported(context, prefix, clean);
886
+ assertNoVideoOrFileParts(context, prefix, clean);
887
+ const url = `${baseUrl}/chat/completions`;
888
+ const sessionId = options?.sessionId || options?.cache?.sessionId;
889
+ const tools = options?.tools as StandardToolDeclaration[] | undefined;
890
+ const body = await this.buildBody(clean, context, options, tools, true);
891
+
892
+ const streamAuditRaw: Record<string, string> = {
893
+ ...buildSessionHeaders(prefix, options?.cache, options?.headers, sessionId),
894
+ "Content-Type": "application/json",
895
+ ...(apiKey ? { Authorization: "[REDACTED]" } : {}),
896
+ };
897
+ for (const k of Object.keys(streamAuditRaw)) {
898
+ if (INTERNAL_HEADERS.has(k.toLowerCase())) delete streamAuditRaw[k];
899
+ }
900
+ const rawRequest = {
901
+ url,
902
+ method: "POST",
903
+ headers: redactedHeaders(streamAuditRaw),
904
+ body,
905
+ };
906
+
907
+ const res = await fetch(url, this.requestInit(body, options, apiKey, sessionId, linked.signal));
908
+ if (!res.ok || !res.body) {
909
+ const text = !res.ok ? await res.text().catch(() => "") : "";
910
+ if (!res.ok) this.throwIfError(res.status, url, text, clean);
911
+ throw toConciseProviderError(new Error("OpenAI-compatible streaming response had no body"), prefix, clean);
912
+ }
913
+ const responseHeaders: Record<string, string> = {};
914
+ res.headers.forEach((v, k) => {
915
+ responseHeaders[k] = v;
916
+ });
917
+ const responseMeta = { status: res.status, statusText: res.statusText, headers: responseHeaders };
918
+ eventStream.push({ type: "start", raw: { request: rawRequest } } as never);
919
+
920
+ const parser = new SSEParser();
921
+ const reader = res.body.getReader();
922
+ const decoder = new TextDecoder();
923
+ let text = "";
924
+ let thinking = "";
925
+ // Tool calls keyed by choice index: { id, name, startArgs, deltaArgs }.
926
+ const calls = new Map<number, { id: string; name: string; startArgs: string; deltaArgs: string }>();
927
+ let lastToolIndex = 0;
928
+ const signatures = new Map<number, string>();
929
+ let usage: TokenUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0 };
930
+ let finishReason = "stop";
931
+ let responseId: string | undefined;
932
+ let completedBody: unknown = undefined;
933
+ let aborted = false;
934
+ linked.signal.addEventListener(
935
+ "abort",
936
+ () => {
937
+ aborted = true;
938
+ try {
939
+ void reader.cancel();
940
+ } catch {}
941
+ },
942
+ { once: true }
943
+ );
944
+
945
+ const handleMessage = (data: string): void => {
946
+ if (!data || data === "[DONE]") return;
947
+ let msg: Record<string, unknown>;
948
+ try {
949
+ msg = JSON.parse(data) as Record<string, unknown>;
950
+ } catch {
951
+ return;
952
+ }
953
+ // Mid-stream provider error: top-level `error`, HTTP stays 200.
954
+ // Must surface — resolving empty success hides it.
955
+ const topError = msg["error"] as { message?: string; code?: string | number } | undefined;
956
+ if (topError && typeof topError === "object") {
957
+ const message =
958
+ typeof topError.message === "string" && topError.message
959
+ ? topError.message
960
+ : "OpenAI-compatible streaming error";
961
+ const failure: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
962
+ if (topError.code !== undefined) failure["code"] = topError.code;
963
+ failure["url"] = url;
964
+ throw toConciseProviderError(failure, prefix, clean);
965
+ }
966
+ if (typeof msg["id"] === "string" && !responseId) {
967
+ responseId = msg["id"] as string;
968
+ }
969
+ const choice = (Array.isArray(msg["choices"]) ? (msg["choices"] as Record<string, unknown>[])[0] : undefined) ?? {};
970
+ const delta = (choice["delta"] as Record<string, unknown>) ?? {};
971
+ // Text delta (content-free accounting frames carry "" — ignored).
972
+ if (typeof delta["content"] === "string" && delta["content"]) {
973
+ text += delta["content"] as string;
974
+ eventStream.push({ type: "text_delta", delta: delta["content"] as string, partialText: text });
975
+ }
976
+ // Tolerated reasoning deltas (deepseek-style): display-only, never resent.
977
+ const details = delta["reasoning_details"];
978
+ if (Array.isArray(details)) {
979
+ for (const block of details) {
980
+ const t = (block as { text?: string })?.text;
981
+ if (typeof t === "string" && t) {
982
+ thinking += t;
983
+ eventStream.push({ type: "thinking_delta", thinkingDelta: t, partialThinking: thinking });
984
+ }
985
+ }
986
+ } else if (typeof delta["reasoning_content"] === "string" && delta["reasoning_content"]) {
987
+ thinking += delta["reasoning_content"] as string;
988
+ eventStream.push({
989
+ type: "thinking_delta",
990
+ thinkingDelta: delta["reasoning_content"] as string,
991
+ partialThinking: thinking,
992
+ });
993
+ } else if (typeof delta["reasoning"] === "string" && delta["reasoning"]) {
994
+ thinking += delta["reasoning"] as string;
995
+ eventStream.push({
996
+ type: "thinking_delta",
997
+ thinkingDelta: delta["reasoning"] as string,
998
+ partialThinking: thinking,
999
+ });
1000
+ }
1001
+ // Tool-call deltas accumulate per choice index (unindexed chunks
1002
+ // continue the last call; defaulting to calls.size would fork
1003
+ // phantom calls).
1004
+ const deltaCalls = delta["tool_calls"];
1005
+ if (Array.isArray(deltaCalls)) {
1006
+ for (const tc of deltaCalls) {
1007
+ const entry = tc as {
1008
+ index?: number;
1009
+ id?: string;
1010
+ type?: string;
1011
+ function?: { name?: string; arguments?: string };
1012
+ extra_content?: { google?: { thought_signature?: string } };
1013
+ };
1014
+ const index = typeof entry.index === "number" ? (lastToolIndex = entry.index) : lastToolIndex;
1015
+ const sig = entry.extra_content?.google?.thought_signature;
1016
+ if (typeof sig === "string" && sig) signatures.set(index, sig);
1017
+ const existing = calls.get(index);
1018
+ if (existing) {
1019
+ if (entry.id) existing.id = entry.id;
1020
+ if (entry.function?.name) existing.name = entry.function.name;
1021
+ if (typeof entry.function?.arguments === "string") {
1022
+ existing.deltaArgs += entry.function.arguments;
1023
+ }
1024
+ } else {
1025
+ calls.set(index, {
1026
+ id: entry.id || "",
1027
+ name: entry.function?.name || "unknown",
1028
+ startArgs: "",
1029
+ deltaArgs: typeof entry.function?.arguments === "string" ? entry.function.arguments : "",
1030
+ });
1031
+ }
1032
+ }
1033
+ }
1034
+ // Terminal accounting: finish_reason repeats on the usage chunk.
1035
+ if (typeof choice["finish_reason"] === "string" && choice["finish_reason"]) {
1036
+ const fr = choice["finish_reason"] as string;
1037
+ if (fr === "error") {
1038
+ const failure: Record<string, unknown> & Error = new Error(
1039
+ "OpenAI-compatible stream terminated with finish_reason error"
1040
+ ) as Record<string, unknown> & Error;
1041
+ failure["url"] = url;
1042
+ throw toConciseProviderError(failure, prefix, clean);
1043
+ }
1044
+ finishReason = fr;
1045
+ }
1046
+ const chunkUsage = msg["usage"] as ChatResponse["usage"] | undefined;
1047
+ if (chunkUsage) {
1048
+ usage = mapUsage(chunkUsage);
1049
+ completedBody = completedBody ?? msg;
1050
+ eventStream.push({ type: "usage", usage });
1051
+ }
1052
+ };
1053
+
1054
+ while (true) {
1055
+ if (linked.signal.aborted || eventStream.isCancelled()) {
1056
+ try {
1057
+ await reader.cancel();
1058
+ } catch {}
1059
+ throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
1060
+ }
1061
+ const { done, value } = await reader.read();
1062
+ if (done) break;
1063
+ const chunk = decoder.decode(value, { stream: true });
1064
+ for (const m of parser.feed(chunk)) handleMessage(m.data);
1065
+ }
1066
+ for (const m of parser.flush()) handleMessage(m.data);
1067
+ try {
1068
+ reader.releaseLock();
1069
+ } catch {}
1070
+
1071
+ if (linked.signal.aborted || eventStream.isCancelled() || aborted) {
1072
+ throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
1073
+ }
1074
+
1075
+ const toolCalls: ToolCallRecord[] = [];
1076
+ for (const [index, c] of calls) {
1077
+ const args = parseStreamedToolArguments(c.startArgs, c.deltaArgs);
1078
+ const sig = signatures.get(index);
1079
+ const record: ToolCallRecord = {
1080
+ id: c.id || `call_${Math.random().toString(36).slice(2, 9)}`,
1081
+ name: c.name,
1082
+ arguments: args,
1083
+ rawArguments: c.startArgs + c.deltaArgs,
1084
+ ...(sig ? { thoughtSignature: sig } : {}),
1085
+ };
1086
+ toolCalls.push(record);
1087
+ eventStream.push({ type: "tool_call_complete", toolCall: record });
1088
+ }
1089
+
1090
+ noteProviderTurn(sessionId, prefix);
1091
+ if (toolCalls.length > 0) finishReason = "tool_calls";
1092
+
1093
+ const cleanThinking = thinking.replace(/\n{3,}/g, "\n\n").trim();
1094
+ const finalThoughtSignature = toolCalls.map((c) => c.thoughtSignature).find(Boolean);
1095
+ const finalResponse = new AgentResponse({
1096
+ text,
1097
+ thinking: cleanThinking || undefined,
1098
+ thoughtSignature: finalThoughtSignature,
1099
+ toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
1100
+ usage,
1101
+ finishReason,
1102
+ responseId,
1103
+ model: clean,
1104
+ provider: prefix as ProviderId,
1105
+ raw: { request: rawRequest, response: { ...responseMeta, body: completedBody } },
1106
+ durationMs: Date.now() - startTime,
1107
+ });
1108
+ eventStream.push({ type: "done", delta: "", usage, finishReason, responseId });
1109
+ eventStream.end(finalResponse);
1110
+ } catch (err: unknown) {
1111
+ const raw = err instanceof Error ? err : new Error(String(err));
1112
+ const isAbort =
1113
+ linked.signal.aborted ||
1114
+ eventStream.isCancelled() ||
1115
+ (raw as { name?: string }).name === "AbortError" ||
1116
+ /abort|cancell?ed/i.test(String((raw as { message?: string }).message ?? raw));
1117
+ eventStream.fail(
1118
+ isAbort ? Object.assign(new Error("Stream aborted"), { name: "AbortError" }) : raw
1119
+ );
1120
+ } finally {
1121
+ try {
1122
+ options?.signal?.removeEventListener("abort", forwardUserAbort);
1123
+ } catch {}
1124
+ try {
1125
+ removeStreamCancel();
1126
+ } catch {}
1127
+ }
1128
+ })();
1129
+
1130
+ return eventStream;
1131
+ }
1132
+ }
1133
+
1134
+ /**
1135
+ * Creates a custom OpenAI-compatible provider backed by native REST.
1136
+ * @example `const provider = createOpenAICompatibleProvider("groq", { baseUrl: "https://api.groq.com/openai/v1" });`
1137
+ */
1138
+ export function createOpenAICompatibleProvider(
1139
+ prefix: string,
1140
+ opts?: CustomProviderOptions
1141
+ ): OpenAICompatibleChatProvider {
1142
+ return new OpenAICompatibleChatProvider(prefix, opts);
1143
+ }
1144
+ /** Backward-compatible alias for {@link createOpenAICompatibleProvider}. @example `createCustomProvider("ollama")` */
1145
+ export const createCustomProvider = createOpenAICompatibleProvider;
1146
+ /** Backward-compatible class alias for {@link OpenAICompatibleChatProvider}. @example `new CustomProvider("ollama")` */
1147
+ export const CustomProvider = OpenAICompatibleChatProvider;