agent-accelerator 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +83 -112
  2. package/{SYSTEM_PROMPT_AGENT.md → examples/prompts/SYSTEM_PROMPT_AGENT.md} +1 -1
  3. package/package.json +14 -9
  4. package/src/agent/agent.ts +191 -71
  5. package/src/agent/context.ts +1 -0
  6. package/src/agent/delegation.ts +61 -12
  7. package/src/agent/loop.ts +71 -48
  8. package/src/data/README.md +6 -6
  9. package/src/index.ts +130 -45
  10. package/src/models/catalog-cache.ts +60 -9
  11. package/src/models/catalog.ts +53 -7
  12. package/src/providers/google.ts +926 -0
  13. package/src/providers/openai-compat.ts +1147 -0
  14. package/src/providers/openai.ts +959 -0
  15. package/src/providers/openrouter-responses.ts +949 -0
  16. package/src/providers/openrouter.ts +1037 -0
  17. package/src/{ai-sdk → providers}/registry.ts +43 -55
  18. package/src/providers.ts +490 -0
  19. package/src/streaming/sse-parser.ts +6 -4
  20. package/src/tools/executor.ts +19 -6
  21. package/src/tools/schema.ts +21 -11
  22. package/src/types/agent.ts +8 -1
  23. package/src/types/core.ts +1 -1
  24. package/src/types/message.ts +5 -0
  25. package/src/types/model.ts +3 -7
  26. package/src/types/provider-payloads.ts +2 -84
  27. package/src/types/tool.ts +6 -0
  28. package/src/update-models.ts +56 -0
  29. package/src/utils/cache.ts +1 -1
  30. package/src/utils/documents.ts +517 -0
  31. package/src/utils/env.ts +0 -7
  32. package/src/{ai-sdk → utils}/errors.ts +61 -2
  33. package/src/utils/headers.ts +10 -20
  34. package/src/utils/media.ts +5 -2
  35. package/src/utils/retry.ts +89 -0
  36. package/src/utils/serialization.ts +15 -0
  37. package/src/ai-sdk/converters.ts +0 -342
  38. package/src/ai-sdk/executor.ts +0 -454
  39. package/src/ai-sdk/index.ts +0 -55
  40. package/src/ai-sdk/model-provider.ts +0 -303
  41. package/src/ai-sdk/options.ts +0 -306
  42. package/src/ai-sdk/provider.ts +0 -415
  43. package/src/tokens/counter.ts +0 -136
  44. /package/{SYSTEM_PROMPT.md → examples/prompts/SYSTEM_PROMPT.md} +0 -0
  45. /package/{SYSTEM_PROMPT_TOOLS.md → examples/prompts/SYSTEM_PROMPT_TOOLS.md} +0 -0
@@ -0,0 +1,926 @@
1
+ /**
2
+ * Google Interactions API provider (`POST /v1beta/interactions`).
3
+ *
4
+ * All Google-specific HTTP, endpoints, headers, auth, request construction,
5
+ * response/SSE parsing, and wire transformations live HERE — never in the
6
+ * canonical layer (`src/providers.ts`).
7
+ *
8
+ * Wire contract: `references/testings/CONTRACT.md` (docs + raw captures).
9
+ * Capability matrix: `references/testings/CAPABILITY-MATRIX.md`.
10
+ *
11
+ * Notes:
12
+ * - Interaction/tool ids are provider-generated (`v1_...`/`call_...`) and
13
+ * echoed verbatim (`call_id` in `function_result` steps). The
14
+ * `call_${random}` fallback in parsers is a local canonical correlation id
15
+ * only (used when the wire omits an id); it is never sent to the provider.
16
+ * - Stateful chaining via `previous_interaction_id` keyed by session; falls
17
+ * back to stateless full-history (`store: false`) when the session mixed
18
+ * providers, keeping the canonical conversation provider-agnostic.
19
+ * - Unlike the legacy transport path, Gemini 2.x is NOT rejected here —
20
+ * the Interactions API documents 2.5 support; the catalog still governs
21
+ * thinking validation upstream.
22
+ */
23
+ import type {
24
+ Provider,
25
+ ProviderId,
26
+ ModelSpec,
27
+ ProviderRequestOptions,
28
+ ProviderGenerateResult,
29
+ ProviderRawData,
30
+ } from "../types/model.ts";
31
+ import type { ProviderContext, Message, ContentPart } from "../types/message.ts";
32
+ import type { StandardToolDeclaration, ToolCallRecord } from "../types/tool.ts";
33
+ import type { TokenUsage } from "../types/core.ts";
34
+ import { AssistantMessageEventStream } from "../streaming/event-stream.ts";
35
+ import { SSEParser } from "../streaming/sse-parser.ts";
36
+ import { AgentResponse } from "../types/response.ts";
37
+ import { getApiKey, getEnv } from "../utils/env.ts";
38
+ import { normalizeMediaInput } from "../utils/media.ts";
39
+ import { safeStringify } from "../utils/serialization.ts";
40
+ import { stripSchemaForGoogle } from "../tools/schema.ts";
41
+ import { toConciseProviderError, assertModalitiesSupported } from "../utils/errors.ts";
42
+ import { withRetries } from "../utils/retry.ts";
43
+ import { createGenericModelSpec } from "../models/catalog.ts";
44
+ import { getModelFromCatalog, getModelsForProvider } from "../models/catalog.ts";
45
+ import {
46
+ mapThinkingLevelToGoogle,
47
+ mapServiceTierToGoogle,
48
+ applyCacheForGoogle,
49
+ normalizeToolChoice,
50
+ lastProviderFor,
51
+ noteProviderTurn,
52
+ parseStreamedToolArguments,
53
+ } from "../providers.ts";
54
+
55
+ // ---------------------------------------------------------------------------
56
+ // Interactions wire shapes (subset used by this adapter)
57
+ // ---------------------------------------------------------------------------
58
+
59
+ type InteractionContentBlock =
60
+ | { type: "text"; text: string }
61
+ | { type: "image"; mime_type: string; data?: string; uri?: string }
62
+ | { type: "audio"; mime_type: string; data?: string; uri?: string }
63
+ | { type: "video"; mime_type: string; data?: string; uri?: string }
64
+ | { type: "document"; mime_type: string; data?: string; uri?: string };
65
+
66
+ type InteractionInputStep =
67
+ | { type: "user_input"; content: string | InteractionContentBlock[] }
68
+ | { type: "thought"; signature: string; summary?: Array<{ type: string; text?: string }> }
69
+ | { type: "function_call"; id: string; name: string; arguments: Record<string, unknown> }
70
+ | {
71
+ type: "function_result";
72
+ name: string;
73
+ call_id: string;
74
+ result: InteractionContentBlock[];
75
+ }
76
+ | { type: string; [k: string]: unknown };
77
+
78
+ interface InteractionRequestBody {
79
+ model: string;
80
+ input: string | InteractionContentBlock[] | InteractionInputStep[];
81
+ system_instruction?: string;
82
+ tools?: Array<Record<string, unknown>>;
83
+ generation_config?: Record<string, unknown>;
84
+ response_format?: Record<string, unknown>;
85
+ service_tier?: string;
86
+ store?: boolean;
87
+ previous_interaction_id?: string;
88
+ stream?: boolean;
89
+ }
90
+
91
+ interface InteractionStep {
92
+ type: string;
93
+ id?: string;
94
+ name?: string;
95
+ arguments?: Record<string, unknown>;
96
+ signature?: string;
97
+ summary?: Array<{ type?: string; text?: string }>;
98
+ content?: Array<{ type?: string; text?: string }>;
99
+ call_id?: string;
100
+ }
101
+
102
+ interface InteractionObject {
103
+ id?: string;
104
+ object?: string;
105
+ model?: string;
106
+ status?: string;
107
+ created?: string;
108
+ updated?: string;
109
+ service_tier?: string;
110
+ steps?: InteractionStep[];
111
+ usage?: {
112
+ total_tokens?: number;
113
+ total_input_tokens?: number;
114
+ total_output_tokens?: number;
115
+ total_cached_tokens?: number;
116
+ total_thought_tokens?: number;
117
+ total_tool_use_tokens?: number;
118
+ input_tokens_by_modality?: Array<{ modality?: string; tokens?: number }>;
119
+ [k: string]: unknown;
120
+ };
121
+ [k: string]: unknown;
122
+ }
123
+
124
+ const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
125
+
126
+ /** Internal headers that must never leak onto Google REST requests. */
127
+ const INTERNAL_HEADERS = new Set([
128
+ "x-thought-signature-map",
129
+ "x-cached-content-id",
130
+ "x-multimodal-user-content",
131
+ ]);
132
+
133
+ /** Per-session interaction chaining (provider-generated ids only). */
134
+ interface ChainEntry {
135
+ interactionId: string;
136
+ model: string;
137
+ }
138
+ const interactionChains = new Map<string, ChainEntry>();
139
+
140
+ function chainFor(sessionId: string | undefined): ChainEntry | undefined {
141
+ if (!sessionId) return undefined;
142
+ return interactionChains.get(sessionId);
143
+ }
144
+
145
+ function setChain(sessionId: string | undefined, entry: ChainEntry): void {
146
+ if (!sessionId) return;
147
+ interactionChains.set(sessionId, entry);
148
+ }
149
+
150
+ /** Clears chained interaction state (mainly for tests). */
151
+ export function clearInteractionChains(sessionId?: string): void {
152
+ if (sessionId) interactionChains.delete(sessionId);
153
+ else interactionChains.clear();
154
+ }
155
+
156
+ // ---------------------------------------------------------------------------
157
+ // Request building (canonical -> Interactions)
158
+ // ---------------------------------------------------------------------------
159
+
160
+ function resolveBaseUrl(options?: ProviderRequestOptions): string {
161
+ return (
162
+ options?.baseUrl ||
163
+ options?.env?.["GOOGLE_BASE_URL"] ||
164
+ getEnv("GOOGLE_BASE_URL") ||
165
+ DEFAULT_BASE_URL
166
+ ).replace(/\/+$/, "");
167
+ }
168
+
169
+ function resolveApiKey(options?: ProviderRequestOptions): string | undefined {
170
+ return options?.apiKey || getApiKey("google", undefined, options?.env);
171
+ }
172
+
173
+ function cleanModelId(model: string | ModelSpec): string {
174
+ const rawId = typeof model === "string" ? model : model.id;
175
+ return rawId.replace(/^(google\/|gemini\/|models\/)/i, "");
176
+ }
177
+
178
+ function toGoogleTools(tools?: StandardToolDeclaration[]): Array<Record<string, unknown>> | undefined {
179
+ if (!tools || tools.length === 0) return undefined;
180
+ return tools.map((t) => ({
181
+ type: "function",
182
+ name: t.name,
183
+ description: t.description,
184
+ parameters: stripSchemaForGoogle(
185
+ (t.parameters || { type: "object", properties: {} }) as Record<string, unknown>
186
+ ),
187
+ }));
188
+ }
189
+
190
+ function toGoogleToolChoice(
191
+ choice: ProviderRequestOptions["toolChoice"],
192
+ toolNames: string[]
193
+ ): Record<string, unknown> | undefined {
194
+ const norm = normalizeToolChoice(choice, toolNames);
195
+ if (!norm) return undefined;
196
+ if (norm.mode === "none") return { allowed_tools: { mode: "none" } };
197
+ if (norm.tools && norm.tools.length > 0) {
198
+ return { allowed_tools: { mode: norm.mode, tools: norm.tools } };
199
+ }
200
+ return { allowed_tools: { mode: norm.mode } };
201
+ }
202
+
203
+ function mediaBlockType(partType: string): "image" | "audio" | "video" | "document" {
204
+ if (partType === "image" || partType === "audio" || partType === "video") return partType;
205
+ // Canonical `file` parts are PDFs/documents. The guides reference document
206
+ // processing with the same content-block pattern; the wire type is
207
+ // `document`. If the server ever rejects it, the concise error surfaces it.
208
+ return "document";
209
+ }
210
+
211
+ async function contentPartsToBlocks(parts: ContentPart[]): Promise<InteractionContentBlock[]> {
212
+ const blocks: InteractionContentBlock[] = [];
213
+ for (const part of parts) {
214
+ if (part.type === "text" && part.text) {
215
+ blocks.push({ type: "text", text: part.text });
216
+ } else if (
217
+ part.type === "image" ||
218
+ part.type === "audio" ||
219
+ part.type === "video" ||
220
+ part.type === "file"
221
+ ) {
222
+ const raw = (part as { image?: unknown; audio?: unknown; video?: unknown; file?: unknown }).image ??
223
+ (part as { audio?: unknown }).audio ??
224
+ (part as { video?: unknown }).video ??
225
+ (part as { file?: unknown }).file;
226
+ const norm = await normalizeMediaInput(
227
+ raw as string | Uint8Array | ArrayBuffer,
228
+ (part as { mimeType?: string }).mimeType
229
+ );
230
+ blocks.push({
231
+ type: mediaBlockType(part.type),
232
+ mime_type: norm.mimeType,
233
+ data: norm.base64Data,
234
+ } as InteractionContentBlock);
235
+ }
236
+ }
237
+ return blocks;
238
+ }
239
+
240
+ function resultToText(result: unknown): string {
241
+ if (typeof result === "string") return result;
242
+ return safeStringify(result);
243
+ }
244
+
245
+ /** Latest-turn input for stateful chaining (server already holds history). */
246
+ async function latestTurnInput(context: ProviderContext): Promise<string | InteractionContentBlock[] | InteractionInputStep[]> {
247
+ const messages = context.messages;
248
+ // Trailing tool messages -> function_result steps.
249
+ const trailingTool: Message[] = [];
250
+ for (let i = messages.length - 1; i >= 0; i--) {
251
+ const m = messages[i]!;
252
+ if (m.role === "tool") trailingTool.unshift(m);
253
+ else break;
254
+ }
255
+ if (trailingTool.length > 0) {
256
+ const steps: InteractionInputStep[] = [];
257
+ for (const m of trailingTool) {
258
+ if (!Array.isArray(m.content)) continue;
259
+ for (const part of m.content) {
260
+ if (part.type === "tool_result") {
261
+ steps.push({
262
+ type: "function_result",
263
+ name: part.name,
264
+ call_id: part.id,
265
+ result: [{ type: "text", text: resultToText(part.result) }],
266
+ });
267
+ }
268
+ }
269
+ }
270
+ return steps;
271
+ }
272
+ const last = messages[messages.length - 1];
273
+ if (!last) return "";
274
+ if (typeof last.content === "string") return last.content;
275
+ const blocks = await contentPartsToBlocks(last.content);
276
+ if (blocks.length === 1 && blocks[0]!.type === "text") {
277
+ return (blocks[0] as { type: "text"; text: string }).text;
278
+ }
279
+ return blocks;
280
+ }
281
+
282
+ /** Full-history Steps for stateless (provider-switch) turns. */
283
+ async function fullHistorySteps(context: ProviderContext): Promise<InteractionInputStep[]> {
284
+ const steps: InteractionInputStep[] = [];
285
+ for (const m of context.messages) {
286
+ if (m.role === "system") continue;
287
+ if (m.role === "user") {
288
+ if (typeof m.content === "string") {
289
+ steps.push({ type: "user_input", content: [{ type: "text", text: m.content }] });
290
+ } else {
291
+ const blocks = await contentPartsToBlocks(m.content);
292
+ if (blocks.length > 0) steps.push({ type: "user_input", content: blocks });
293
+ }
294
+ } else if (m.role === "assistant") {
295
+ if (typeof m.content === "string") {
296
+ if (m.content) {
297
+ steps.push({
298
+ type: "model_output",
299
+ content: [{ type: "text", text: m.content }],
300
+ } as InteractionInputStep);
301
+ }
302
+ continue;
303
+ }
304
+ const texts: string[] = [];
305
+ for (const part of m.content) {
306
+ if (part.type === "thinking" && part.thinking) {
307
+ steps.push({
308
+ type: "thought",
309
+ signature: part.thoughtSignature || m.thoughtSignature || "",
310
+ summary: [{ type: "text", text: part.thinking }],
311
+ });
312
+ } else if (part.type === "tool_call") {
313
+ steps.push({
314
+ type: "function_call",
315
+ id: part.id,
316
+ name: part.name,
317
+ arguments: part.arguments || {},
318
+ });
319
+ } else if (part.type === "text" && part.text) {
320
+ texts.push(part.text);
321
+ }
322
+ }
323
+ if (texts.length > 0) {
324
+ steps.push({
325
+ type: "model_output",
326
+ content: texts.map((t) => ({ type: "text", text: t })),
327
+ } as InteractionInputStep);
328
+ } else if (m.thoughtSignature && !steps.some((s) => s.type === "thought")) {
329
+ // Preserve a signature-only thought so stateless chaining keeps
330
+ // reasoning continuity even when no summary text was retained.
331
+ steps.push({ type: "thought", signature: m.thoughtSignature });
332
+ }
333
+ } else if (m.role === "tool") {
334
+ if (!Array.isArray(m.content)) continue;
335
+ for (const part of m.content) {
336
+ if (part.type === "tool_result") {
337
+ steps.push({
338
+ type: "function_result",
339
+ name: part.name,
340
+ call_id: part.id,
341
+ result: [{ type: "text", text: resultToText(part.result) }],
342
+ });
343
+ }
344
+ }
345
+ }
346
+ }
347
+ return steps;
348
+ }
349
+
350
+ // ---------------------------------------------------------------------------
351
+ // Response mapping (Interactions -> canonical)
352
+ // ---------------------------------------------------------------------------
353
+
354
+ function mapUsage(raw?: InteractionObject["usage"]): TokenUsage {
355
+ const input = raw?.total_input_tokens ?? 0;
356
+ const output = raw?.total_output_tokens ?? 0;
357
+ // Canonical invariant: cached hits are a SUBSET of input. Google usually
358
+ // honors this, but long chained runs have been observed reporting
359
+ // total_cached_tokens > total_input_tokens (which yields >100% hit rates
360
+ // and negative non-cached costs downstream). Clamp at the boundary.
361
+ const cached = Math.min(raw?.total_cached_tokens ?? 0, input);
362
+ return {
363
+ inputTokens: input,
364
+ outputTokens: output,
365
+ totalTokens: raw?.total_tokens ?? input + output,
366
+ cachedTokens: cached,
367
+ cacheReadTokens: cached,
368
+ cacheWriteTokens: 0,
369
+ thinkingTokens: raw?.total_thought_tokens ?? 0,
370
+ };
371
+ }
372
+
373
+ function mapFinishReason(status?: string): string {
374
+ if (status === "requires_action") return "tool_calls";
375
+ if (status === "incomplete") return "length";
376
+ return "stop";
377
+ }
378
+
379
+ function parseInteraction(
380
+ interaction: InteractionObject,
381
+ modelId: string,
382
+ durationMs: number,
383
+ raw: ProviderRawData
384
+ ): ProviderGenerateResult {
385
+ let text = "";
386
+ const thinkingParts: string[] = [];
387
+ let thoughtSignature: string | undefined;
388
+ const toolCalls: ToolCallRecord[] = [];
389
+
390
+ for (const step of interaction.steps ?? []) {
391
+ if (step.type === "thought") {
392
+ if (step.signature && !thoughtSignature) thoughtSignature = step.signature;
393
+ const summary = (step.summary ?? [])
394
+ .filter((b) => b.type === "text" && b.text)
395
+ .map((b) => b.text as string)
396
+ .join("");
397
+ // Provider summaries routinely trail with blank lines (observed
398
+ // `"...\n\n\n"`). Trim per-step and drop empties so joined thinking has
399
+ // no edge-tripling. Stateful turns resend nothing (server holds
400
+ // history), so this is cache-safe; stateless resends stay consistent.
401
+ const cleanSummary = summary.replace(/\n{3,}/g, "\n\n").trim();
402
+ if (cleanSummary) thinkingParts.push(cleanSummary);
403
+ } else if (step.type === "model_output") {
404
+ for (const block of step.content ?? []) {
405
+ if (block.type === "text" && block.text) text += block.text;
406
+ }
407
+ } else if (step.type === "function_call") {
408
+ const args =
409
+ step.arguments && typeof step.arguments === "object" ? step.arguments : { raw: step.arguments };
410
+ toolCalls.push({
411
+ id: step.id || `call_${Math.random().toString(36).slice(2, 9)}`,
412
+ name: step.name || "unknown",
413
+ arguments: args as Record<string, unknown>,
414
+ rawArguments: JSON.stringify(args),
415
+ });
416
+ }
417
+ }
418
+
419
+ return {
420
+ text,
421
+ thinking: thinkingParts.length > 0 ? thinkingParts.join("\n") : undefined,
422
+ thoughtSignature,
423
+ toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
424
+ usage: mapUsage(interaction.usage),
425
+ finishReason: mapFinishReason(interaction.status),
426
+ responseId: interaction.id,
427
+ model: modelId,
428
+ provider: "google",
429
+ raw,
430
+ durationMs,
431
+ };
432
+ }
433
+
434
+ function redactedHeaders(headers: Record<string, string>): Record<string, string> {
435
+ const out: Record<string, string> = {};
436
+ for (const [k, v] of Object.entries(headers)) {
437
+ out[k] = k.toLowerCase() === "x-goog-api-key" ? "[REDACTED]" : v;
438
+ }
439
+ return out;
440
+ }
441
+
442
+ function readErrorPayload(bodyText: string): { message: string; code?: string | number } {
443
+ try {
444
+ const parsed: unknown = JSON.parse(bodyText);
445
+ const first = Array.isArray(parsed) ? parsed[0] : parsed;
446
+ const err = (first as { error?: { message?: string; code?: string | number; status?: string } })?.error;
447
+ if (err && typeof err.message === "string") {
448
+ return { message: err.message, code: err.code ?? err.status };
449
+ }
450
+ return { message: bodyText.slice(0, 300) };
451
+ } catch {
452
+ return { message: bodyText.slice(0, 300) };
453
+ }
454
+ }
455
+
456
+ // ---------------------------------------------------------------------------
457
+ // Provider
458
+ // ---------------------------------------------------------------------------
459
+
460
+ /**
461
+ * Google provider implemented directly on the Interactions REST API
462
+ * (`POST {baseUrl}/interactions`, streaming via `?alt=sse`).
463
+ */
464
+ export class GoogleInteractionsProvider implements Provider {
465
+ readonly id: ProviderId = "google";
466
+ readonly name = "Google Interactions";
467
+
468
+ /** Live catalog view: a constructor snapshot would go stale after refresh. */
469
+ get models(): ModelSpec[] {
470
+ return getModelsForProvider("google");
471
+ }
472
+
473
+ getModel(modelId: string): ModelSpec | undefined {
474
+ const clean = cleanModelId(modelId);
475
+ return (
476
+ getModelFromCatalog(this.id, modelId) ||
477
+ getModelFromCatalog(this.id, clean) ||
478
+ this.models.find((m) => m.id === modelId || m.id === clean) ||
479
+ createGenericModelSpec(this.id, clean)
480
+ );
481
+ }
482
+
483
+ private buildBody(
484
+ modelId: string,
485
+ context: ProviderContext,
486
+ options: ProviderRequestOptions | undefined,
487
+ sessionId: string | undefined,
488
+ tools: StandardToolDeclaration[] | undefined,
489
+ stream: boolean
490
+ ): Promise<InteractionRequestBody> {
491
+ return (async () => {
492
+ const { thinkingLevel } = mapThinkingLevelToGoogle(options?.thinking?.level);
493
+ const serviceTier = mapServiceTierToGoogle(options?.serviceTier);
494
+ applyCacheForGoogle(options?.cache, `google/${modelId}`);
495
+
496
+ const toolNames = (tools ?? []).map((t) => t.name);
497
+ const toolChoice = toGoogleToolChoice(options?.toolChoice, toolNames);
498
+
499
+ const generationConfig: Record<string, unknown> = {};
500
+ if (thinkingLevel) generationConfig["thinking_level"] = thinkingLevel;
501
+ // Summaries surface reasoning as `thought.summary` / deltas; without it
502
+ // only signatures return and canonical `thinking` would always be empty.
503
+ // Enabled whenever thinking is active — including the default/dynamic
504
+ // case where no explicit level is sent — and off only when disabled.
505
+ const thinkingActive =
506
+ options?.thinking?.enabled !== false && (options?.thinking?.level ?? "dynamic") !== "none";
507
+ if (thinkingActive) generationConfig["thinking_summaries"] = "auto";
508
+ if (toolChoice) generationConfig["tool_choice"] = toolChoice;
509
+
510
+ const body: InteractionRequestBody = {
511
+ model: modelId,
512
+ input: "",
513
+ ...(stream ? { stream: true } : {}),
514
+ };
515
+ if (context.systemPrompt) body.system_instruction = context.systemPrompt;
516
+ const googleTools = toGoogleTools(tools);
517
+ if (googleTools) body.tools = googleTools;
518
+ if (Object.keys(generationConfig).length > 0) body.generation_config = generationConfig;
519
+ if (serviceTier) body.service_tier = serviceTier;
520
+
521
+ const prevProvider = lastProviderFor(sessionId);
522
+ const switched = !!prevProvider && prevProvider !== "google";
523
+ const chain = chainFor(sessionId);
524
+ // Interaction ids are model-scoped server-side: chaining a 2.5 id into
525
+ // a 3.5 request 400s, so a same-provider model switch resumes stateless.
526
+ const continuing = !!chain && !switched && chain.model === modelId && context.messages.length > 1;
527
+
528
+ if (switched) {
529
+ // Cross-provider turn: the Google server never saw the other
530
+ // provider's turns, so send the full canonical history explicitly and
531
+ // statelessly, then resume chaining fresh on the next turn.
532
+ body.input = await fullHistorySteps(context);
533
+ body.store = false;
534
+ } else if (continuing) {
535
+ body.input = await latestTurnInput(context);
536
+ body.previous_interaction_id = chain!.interactionId;
537
+ } else if (context.messages.length > 1) {
538
+ // No chain but multi-message history (restored session after a
539
+ // restart, or the turn right after a provider switch): send the full
540
+ // history once with default store so the response establishes a fresh
541
+ // chain and the next turn resumes stateful chaining.
542
+ body.input = await fullHistorySteps(context);
543
+ } else {
544
+ body.input = await latestTurnInput(context);
545
+ }
546
+ return body;
547
+ })();
548
+ }
549
+
550
+ private requestInit(
551
+ body: InteractionRequestBody,
552
+ options: ProviderRequestOptions | undefined,
553
+ apiKey: string,
554
+ signal?: AbortSignal
555
+ ): RequestInit {
556
+ const headers: Record<string, string> = { "Content-Type": "application/json" };
557
+ for (const [k, v] of Object.entries(options?.headers ?? {})) {
558
+ if (!INTERNAL_HEADERS.has(k.toLowerCase())) headers[k] = v;
559
+ }
560
+ headers["x-goog-api-key"] = apiKey;
561
+ return { method: "POST", headers, body: JSON.stringify(body), signal };
562
+ }
563
+
564
+ private async doFetch(
565
+ url: string,
566
+ body: InteractionRequestBody,
567
+ options: ProviderRequestOptions | undefined,
568
+ apiKey: string,
569
+ signal?: AbortSignal
570
+ ): Promise<{ status: number; statusText: string; headers: Record<string, string>; text: string }> {
571
+ const res = await fetch(url, this.requestInit(body, options, apiKey, signal));
572
+ const text = await res.text();
573
+ const headers: Record<string, string> = {};
574
+ res.headers.forEach((v, k) => {
575
+ headers[k] = v;
576
+ });
577
+ return { status: res.status, statusText: res.statusText, headers, text };
578
+ }
579
+
580
+ private throwIfError(
581
+ status: number,
582
+ statusText: string,
583
+ url: string,
584
+ bodyText: string,
585
+ modelId: string
586
+ ): void {
587
+ if (status >= 200 && status < 300) return;
588
+ const { message, code } = readErrorPayload(bodyText);
589
+ const err: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
590
+ (err as Record<string, unknown>)["statusCode"] = status;
591
+ (err as Record<string, unknown>)["status"] = status;
592
+ (err as Record<string, unknown>)["responseBody"] = bodyText.slice(0, 500);
593
+ (err as Record<string, unknown>)["url"] = url.split("?")[0];
594
+ if (code !== undefined) (err as Record<string, unknown>)["code"] = code;
595
+ throw toConciseProviderError(err, "google", modelId);
596
+ }
597
+
598
+ async generate(
599
+ model: string | ModelSpec,
600
+ context: ProviderContext,
601
+ options?: ProviderRequestOptions
602
+ ): Promise<ProviderGenerateResult> {
603
+ const startTime = Date.now();
604
+ const clean = cleanModelId(model);
605
+ const apiKey = resolveApiKey(options);
606
+ if (!apiKey) {
607
+ throw new Error(
608
+ "[Agent Accelerator] Missing Google API key. Set GEMINI_API_KEY (or GOOGLE_API_KEY) or pass apiKey."
609
+ );
610
+ }
611
+ // Fail fast before any network call when the catalog knows the model
612
+ // lacks the requested modality (clear one-liner instead of a provider 400).
613
+ // Unknown models skip the guard; the provider verdict surfaces concisely.
614
+ assertModalitiesSupported(context, "google", clean);
615
+ const baseUrl = resolveBaseUrl(options);
616
+ const url = `${baseUrl}/interactions`;
617
+ const sessionId = options?.sessionId || options?.cache?.sessionId;
618
+ const tools = options?.tools as StandardToolDeclaration[] | undefined;
619
+ const body = await this.buildBody(clean, context, options, sessionId, tools, false);
620
+
621
+ const rawRequest = {
622
+ url,
623
+ method: "POST",
624
+ headers: redactedHeaders({
625
+ "Content-Type": "application/json",
626
+ ...Object.fromEntries(
627
+ Object.entries(options?.headers ?? {}).filter(([k]) => !INTERNAL_HEADERS.has(k.toLowerCase()))
628
+ ),
629
+ "x-goog-api-key": "[REDACTED]",
630
+ }),
631
+ body,
632
+ };
633
+
634
+ const doCall = async (): Promise<ProviderGenerateResult> => {
635
+ const res = await this.doFetch(url, body, options, apiKey, options?.signal);
636
+ this.throwIfError(res.status, res.statusText, url, res.text, clean);
637
+ let interaction: InteractionObject;
638
+ try {
639
+ interaction = JSON.parse(res.text) as InteractionObject;
640
+ } catch {
641
+ throw toConciseProviderError(
642
+ Object.assign(new Error("Invalid JSON response from Google Interactions API"), {
643
+ statusCode: res.status,
644
+ responseBody: res.text.slice(0, 500),
645
+ url,
646
+ }),
647
+ "google",
648
+ clean
649
+ );
650
+ }
651
+ if (interaction.id && sessionId && body.store !== false) {
652
+ setChain(sessionId, { interactionId: interaction.id, model: clean });
653
+ } else if (body.store === false && sessionId) {
654
+ interactionChains.delete(sessionId);
655
+ }
656
+ noteProviderTurn(sessionId, "google");
657
+ return parseInteraction(
658
+ interaction,
659
+ clean,
660
+ Date.now() - startTime,
661
+ { request: rawRequest, response: { status: res.status, statusText: res.statusText, headers: res.headers, body: interaction } }
662
+ );
663
+ };
664
+
665
+ try {
666
+ return await withRetries(doCall, {
667
+ maxRetries: options?.maxRetries,
668
+ maxRetryDelayMs: options?.maxRetryDelayMs,
669
+ signal: options?.signal,
670
+ label: { providerId: "google", modelId: clean },
671
+ });
672
+ } catch (err) {
673
+ if (err instanceof Error && (err as { name?: string }).name === "AbortError") throw err;
674
+ throw err;
675
+ }
676
+ }
677
+
678
+ stream(
679
+ model: string | ModelSpec,
680
+ context: ProviderContext,
681
+ options?: ProviderRequestOptions
682
+ ): AssistantMessageEventStream {
683
+ const eventStream = new AssistantMessageEventStream();
684
+ const startTime = Date.now();
685
+ const clean = cleanModelId(model);
686
+
687
+ const linked = new AbortController();
688
+ const forwardUserAbort = () => {
689
+ try {
690
+ linked.abort((options?.signal as { reason?: unknown })?.reason);
691
+ } catch {
692
+ try {
693
+ linked.abort();
694
+ } catch {}
695
+ }
696
+ };
697
+ if (options?.signal?.aborted) forwardUserAbort();
698
+ else options?.signal?.addEventListener("abort", forwardUserAbort, { once: true });
699
+ const removeStreamCancel = eventStream.onCancel(() => {
700
+ try {
701
+ linked.abort();
702
+ } catch {}
703
+ });
704
+
705
+ (async () => {
706
+ try {
707
+ const apiKey = resolveApiKey(options);
708
+ if (!apiKey) {
709
+ throw new Error(
710
+ "[Agent Accelerator] Missing Google API key. Set GEMINI_API_KEY (or GOOGLE_API_KEY) or pass apiKey."
711
+ );
712
+ }
713
+ assertModalitiesSupported(context, "google", clean);
714
+ const baseUrl = resolveBaseUrl(options);
715
+ const url = `${baseUrl}/interactions?alt=sse`;
716
+ const sessionId = options?.sessionId || options?.cache?.sessionId;
717
+ const tools = options?.tools as StandardToolDeclaration[] | undefined;
718
+ const body = await this.buildBody(clean, context, options, sessionId, tools, true);
719
+
720
+ // Audit trail mirrors generate (custom headers included); the
721
+ // Interactions API sends no session-affinity headers (implicit
722
+ // caching + chaining instead), so none are recorded here either.
723
+ const rawRequest = {
724
+ url: url.split("?")[0]!,
725
+ method: "POST",
726
+ headers: redactedHeaders({
727
+ "Content-Type": "application/json",
728
+ ...Object.fromEntries(
729
+ Object.entries(options?.headers ?? {}).filter(([k]) => !INTERNAL_HEADERS.has(k.toLowerCase()))
730
+ ),
731
+ "x-goog-api-key": "[REDACTED]",
732
+ }),
733
+ body,
734
+ };
735
+
736
+ const res = await fetch(url, this.requestInit(body, options, apiKey, linked.signal));
737
+ if (!res.ok || !res.body) {
738
+ const text = !res.ok ? await res.text().catch(() => "") : "";
739
+ if (!res.ok) this.throwIfError(res.status, res.statusText, url, text, clean);
740
+ throw toConciseProviderError(new Error("Google streaming response had no body"), "google", clean);
741
+ }
742
+ const responseHeaders: Record<string, string> = {};
743
+ res.headers.forEach((v, k) => {
744
+ responseHeaders[k] = v;
745
+ });
746
+ const responseMeta = { status: res.status, statusText: res.statusText, headers: responseHeaders };
747
+ eventStream.push({ type: "start", raw: { request: rawRequest } } as never);
748
+
749
+ const parser = new SSEParser();
750
+ const reader = res.body.getReader();
751
+ const decoder = new TextDecoder();
752
+ let text = "";
753
+ let thinking = "";
754
+ let signature: string | undefined;
755
+ const calls = new Map<number, { id: string; name: string; startArgs: string; deltaArgs: string }>();
756
+ let usage: TokenUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0 };
757
+ let finishReason = "stop";
758
+ let responseId: string | undefined;
759
+ let completedBody: unknown = undefined;
760
+ let aborted = false;
761
+ linked.signal.addEventListener(
762
+ "abort",
763
+ () => {
764
+ aborted = true;
765
+ try {
766
+ void reader.cancel();
767
+ } catch {}
768
+ },
769
+ { once: true }
770
+ );
771
+
772
+ const handleMessage = (event: string | undefined, data: string): void => {
773
+ if (!data || data === "[DONE]") return;
774
+ let msg: Record<string, unknown>;
775
+ try {
776
+ msg = JSON.parse(data) as Record<string, unknown>;
777
+ } catch {
778
+ return;
779
+ }
780
+ const type = (msg["event_type"] as string) || event;
781
+ if (type === "error") {
782
+ // SSE error event (e.g. rate_limit_exceeded mid-stream). Must
783
+ // surface — silently ending with empty text + zero usage hides it.
784
+ const errObj = (msg["error"] as { message?: string; code?: string | number }) ?? {};
785
+ const message =
786
+ typeof errObj.message === "string" && errObj.message ? errObj.message : "Google streaming error";
787
+ const failure: Record<string, unknown> & Error = new Error(message) as Record<string, unknown> & Error;
788
+ if (errObj.code !== undefined) failure["code"] = errObj.code;
789
+ failure["url"] = url.split("?")[0];
790
+ throw toConciseProviderError(failure, "google", clean);
791
+ }
792
+ if (type === "interaction.created") {
793
+ const interaction = msg["interaction"] as { id?: string } | undefined;
794
+ if (interaction?.id) responseId = interaction.id;
795
+ } else if (type === "step.delta") {
796
+ const index = (msg["index"] as number) ?? 0;
797
+ const delta = (msg["delta"] as Record<string, unknown>) ?? {};
798
+ const dType = delta["type"] as string;
799
+ if (dType === "text" && typeof delta["text"] === "string") {
800
+ text += delta["text"] as string;
801
+ eventStream.push({ type: "text_delta", delta: delta["text"] as string, partialText: text });
802
+ } else if (dType === "thought_summary") {
803
+ const content = delta["content"] as { type?: string; text?: string } | undefined;
804
+ const t = content?.text ?? (typeof delta["text"] === "string" ? (delta["text"] as string) : "");
805
+ if (t) {
806
+ thinking += t;
807
+ eventStream.push({ type: "thinking_delta", thinkingDelta: t, partialThinking: thinking });
808
+ }
809
+ } else if (dType === "thought_signature" && typeof delta["signature"] === "string") {
810
+ // Wire truth: signature arrives as its own delta (no text event).
811
+ if (!signature) signature = delta["signature"] as string;
812
+ } else if (dType === "arguments" || dType === "arguments_delta") {
813
+ // Wire truth is `arguments`; prose says `partial_arguments` — accept both.
814
+ // Deltas accumulate separately from step.start (see
815
+ // parseStreamedToolArguments): start may be `{}` with the full
816
+ // JSON in one delta, or a genuine first partial.
817
+ const chunk =
818
+ (delta["arguments"] as string) ?? (delta["partial_arguments"] as string) ?? "";
819
+ const entry = calls.get(index);
820
+ if (entry && typeof chunk === "string") entry.deltaArgs += chunk;
821
+ }
822
+ } else if (type === "step.start") {
823
+ const index = (msg["index"] as number) ?? 0;
824
+ const step = (msg["step"] as Record<string, unknown>) ?? {};
825
+ if (step["type"] === "function_call") {
826
+ const args = step["arguments"];
827
+ const startArgs = typeof args === "string" ? args : args ? JSON.stringify(args) : "";
828
+ calls.set(index, {
829
+ id: (step["id"] as string) || `call_${Math.random().toString(36).slice(2, 9)}`,
830
+ name: (step["name"] as string) || "unknown",
831
+ startArgs,
832
+ deltaArgs: "",
833
+ });
834
+ }
835
+ } else if (type === "interaction.completed") {
836
+ const interaction = (msg["interaction"] as InteractionObject) ?? {};
837
+ completedBody = msg["interaction"];
838
+ if (interaction.id) responseId = interaction.id;
839
+ if (interaction.usage) usage = mapUsage(interaction.usage);
840
+ finishReason = mapFinishReason(interaction.status);
841
+ eventStream.push({ type: "usage", usage });
842
+ }
843
+ };
844
+
845
+ while (true) {
846
+ if (linked.signal.aborted || eventStream.isCancelled()) {
847
+ try {
848
+ await reader.cancel();
849
+ } catch {}
850
+ throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
851
+ }
852
+ const { done, value } = await reader.read();
853
+ if (done) break;
854
+ const chunk = decoder.decode(value, { stream: true });
855
+ for (const m of parser.feed(chunk)) handleMessage(m.event, m.data);
856
+ }
857
+ for (const m of parser.flush()) handleMessage(m.event, m.data);
858
+ try {
859
+ reader.releaseLock();
860
+ } catch {}
861
+
862
+ if (linked.signal.aborted || eventStream.isCancelled() || aborted) {
863
+ throw Object.assign(new Error("Stream aborted"), { name: "AbortError" });
864
+ }
865
+
866
+ const toolCalls: ToolCallRecord[] = [];
867
+ for (const c of calls.values()) {
868
+ const args = parseStreamedToolArguments(c.startArgs, c.deltaArgs);
869
+ const record: ToolCallRecord = {
870
+ id: c.id,
871
+ name: c.name,
872
+ arguments: args,
873
+ rawArguments: c.startArgs + c.deltaArgs,
874
+ };
875
+ toolCalls.push(record);
876
+ eventStream.push({ type: "tool_call_complete", toolCall: record });
877
+ }
878
+
879
+ if (sessionId) {
880
+ if (responseId && body.store !== false) setChain(sessionId, { interactionId: responseId, model: clean });
881
+ else if (body.store === false) interactionChains.delete(sessionId);
882
+ }
883
+ noteProviderTurn(sessionId, "google");
884
+
885
+ // Trim provider trailing blank lines from finalized thinking (live
886
+ // deltas already emitted raw; display collapse happens in the
887
+ // wrapThinking presentation layer). Stateful bodies unaffected.
888
+ const cleanThinking = thinking.replace(/\n{3,}/g, "\n\n").trim();
889
+ const finalResponse = new AgentResponse({
890
+ text,
891
+ thinking: cleanThinking || undefined,
892
+ thoughtSignature: signature,
893
+ toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
894
+ usage,
895
+ finishReason,
896
+ responseId,
897
+ model: clean,
898
+ provider: "google",
899
+ raw: { request: rawRequest, response: { ...responseMeta, body: completedBody } },
900
+ durationMs: Date.now() - startTime,
901
+ });
902
+ eventStream.push({ type: "done", delta: "", usage, finishReason, responseId });
903
+ eventStream.end(finalResponse);
904
+ } catch (err: unknown) {
905
+ const raw = err instanceof Error ? err : new Error(String(err));
906
+ const isAbort =
907
+ linked.signal.aborted ||
908
+ eventStream.isCancelled() ||
909
+ (raw as { name?: string }).name === "AbortError" ||
910
+ /abort|cancell?ed/i.test(String((raw as { message?: string }).message ?? raw));
911
+ eventStream.fail(
912
+ isAbort ? Object.assign(new Error("Stream aborted"), { name: "AbortError" }) : raw
913
+ );
914
+ } finally {
915
+ try {
916
+ options?.signal?.removeEventListener("abort", forwardUserAbort);
917
+ } catch {}
918
+ try {
919
+ removeStreamCancel();
920
+ } catch {}
921
+ }
922
+ })();
923
+
924
+ return eventStream;
925
+ }
926
+ }