@opencode/ai 2.0.15 → 2.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. package/README.md +286 -2
  2. package/dist/generation.d.ts +36 -22
  3. package/dist/generation.js +53 -24
  4. package/dist/image-client.d.ts +14 -7
  5. package/dist/image-client.js +24 -8
  6. package/dist/image.d.ts +398 -47
  7. package/dist/image.js +47 -45
  8. package/dist/index.d.ts +13 -1
  9. package/dist/index.js +9 -0
  10. package/dist/media-model.d.ts +44 -0
  11. package/dist/media-model.js +49 -0
  12. package/dist/media.d.ts +10 -9
  13. package/dist/media.js +9 -10
  14. package/dist/promise.d.ts +428 -8
  15. package/dist/promise.js +40 -3
  16. package/dist/protocols/alibaba-chat.d.ts +12 -0
  17. package/dist/protocols/alibaba-responses.d.ts +2 -2
  18. package/dist/protocols/anthropic-messages.js +1 -2
  19. package/dist/protocols/assemblyai-transcription.d.ts +40 -0
  20. package/dist/protocols/assemblyai-transcription.js +138 -0
  21. package/dist/protocols/bedrock-converse.js +5 -11
  22. package/dist/protocols/bfl-images.d.ts +32 -0
  23. package/dist/protocols/bfl-images.js +153 -0
  24. package/dist/protocols/cartesia-speech.d.ts +127 -0
  25. package/dist/protocols/cartesia-speech.js +126 -0
  26. package/dist/protocols/deepgram-speech.d.ts +119 -0
  27. package/dist/protocols/deepgram-speech.js +92 -0
  28. package/dist/protocols/deepgram-transcription.d.ts +25 -0
  29. package/dist/protocols/deepgram-transcription.js +129 -0
  30. package/dist/protocols/elevenlabs-speech.d.ts +122 -0
  31. package/dist/protocols/elevenlabs-speech.js +115 -0
  32. package/dist/protocols/fal-images.d.ts +24 -0
  33. package/dist/protocols/fal-images.js +114 -0
  34. package/dist/protocols/fal-video.d.ts +29 -0
  35. package/dist/protocols/fal-video.js +88 -0
  36. package/dist/protocols/gemini.d.ts +9 -9
  37. package/dist/protocols/gemini.js +8 -34
  38. package/dist/protocols/google-images.js +2 -14
  39. package/dist/protocols/google-speech.d.ts +130 -0
  40. package/dist/protocols/google-speech.js +84 -0
  41. package/dist/protocols/google-transcription.d.ts +173 -0
  42. package/dist/protocols/google-transcription.js +138 -0
  43. package/dist/protocols/google-video.d.ts +26 -0
  44. package/dist/protocols/google-video.js +158 -0
  45. package/dist/protocols/meta-images.js +2 -9
  46. package/dist/protocols/meta-responses.d.ts +4 -4
  47. package/dist/protocols/meta-responses.js +1 -1
  48. package/dist/protocols/open-responses.d.ts +6 -6
  49. package/dist/protocols/open-responses.js +1 -2
  50. package/dist/protocols/openai-chat.d.ts +84 -0
  51. package/dist/protocols/openai-chat.js +26 -14
  52. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  53. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  54. package/dist/protocols/openai-images.d.ts +124 -3
  55. package/dist/protocols/openai-images.js +107 -54
  56. package/dist/protocols/openai-responses.d.ts +15 -15
  57. package/dist/protocols/openai-responses.js +5 -6
  58. package/dist/protocols/openai-speech.d.ts +116 -0
  59. package/dist/protocols/openai-speech.js +98 -0
  60. package/dist/protocols/openai-transcription.d.ts +207 -0
  61. package/dist/protocols/openai-transcription.js +190 -0
  62. package/dist/protocols/replicate-images.d.ts +28 -0
  63. package/dist/protocols/replicate-images.js +133 -0
  64. package/dist/protocols/runway-video.d.ts +38 -0
  65. package/dist/protocols/runway-video.js +146 -0
  66. package/dist/protocols/shared.d.ts +13 -3
  67. package/dist/protocols/shared.js +23 -3
  68. package/dist/protocols/stability-images.d.ts +38 -0
  69. package/dist/protocols/stability-images.js +148 -0
  70. package/dist/protocols/utils/fal-queue.d.ts +28 -0
  71. package/dist/protocols/utils/fal-queue.js +69 -0
  72. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  73. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  74. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  75. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  76. package/dist/protocols/utils/media-input.d.ts +8 -0
  77. package/dist/protocols/utils/media-input.js +18 -0
  78. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  79. package/dist/protocols/utils/speech-stream.js +67 -0
  80. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  81. package/dist/protocols/utils/tool-schema.js +40 -17
  82. package/dist/protocols/xai-images.js +1 -12
  83. package/dist/protocols/xai-responses.d.ts +2 -2
  84. package/dist/protocols/xai-video.d.ts +34 -0
  85. package/dist/protocols/xai-video.js +147 -0
  86. package/dist/protocols/zai-chat.d.ts +13 -1
  87. package/dist/provider-error.js +3 -0
  88. package/dist/providers/alibaba.d.ts +14 -2
  89. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  90. package/dist/providers/assemblyai.d.ts +25 -0
  91. package/dist/providers/assemblyai.js +29 -0
  92. package/dist/providers/azure.d.ts +18 -6
  93. package/dist/providers/baseten.d.ts +24 -0
  94. package/dist/providers/black-forest-labs.d.ts +25 -0
  95. package/dist/providers/black-forest-labs.js +28 -0
  96. package/dist/providers/cartesia.d.ts +24 -0
  97. package/dist/providers/cartesia.js +22 -0
  98. package/dist/providers/cerebras.d.ts +24 -0
  99. package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
  100. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  101. package/dist/providers/deepgram.d.ts +29 -0
  102. package/dist/providers/deepgram.js +31 -0
  103. package/dist/providers/deepinfra.d.ts +24 -0
  104. package/dist/providers/deepseek.d.ts +24 -0
  105. package/dist/providers/elevenlabs.d.ts +24 -0
  106. package/dist/providers/elevenlabs.js +28 -0
  107. package/dist/providers/fal.d.ts +29 -0
  108. package/dist/providers/fal.js +33 -0
  109. package/dist/providers/fireworks.d.ts +24 -0
  110. package/dist/providers/google-vertex-chat.d.ts +12 -0
  111. package/dist/providers/google-vertex-responses.d.ts +2 -2
  112. package/dist/providers/google-vertex.d.ts +3 -3
  113. package/dist/providers/google.d.ts +18 -3
  114. package/dist/providers/google.js +11 -2
  115. package/dist/providers/groq.d.ts +24 -0
  116. package/dist/providers/index.d.ts +9 -0
  117. package/dist/providers/index.js +9 -0
  118. package/dist/providers/meta.d.ts +14 -2
  119. package/dist/providers/minimax.d.ts +14 -2
  120. package/dist/providers/moonshot.d.ts +14 -2
  121. package/dist/providers/moonshot.js +3 -3
  122. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  123. package/dist/providers/openai-compatible.d.ts +12 -0
  124. package/dist/providers/openai.d.ts +25 -3
  125. package/dist/providers/openai.js +10 -1
  126. package/dist/providers/openrouter.d.ts +48 -0
  127. package/dist/providers/replicate.d.ts +25 -0
  128. package/dist/providers/replicate.js +22 -0
  129. package/dist/providers/runway.d.ts +24 -0
  130. package/dist/providers/runway.js +22 -0
  131. package/dist/providers/stability.d.ts +28 -0
  132. package/dist/providers/stability.js +23 -0
  133. package/dist/providers/togetherai.d.ts +24 -0
  134. package/dist/providers/xai.d.ts +17 -0
  135. package/dist/providers/xai.js +5 -2
  136. package/dist/providers/zai-coding-plan.d.ts +15 -3
  137. package/dist/providers/zai.d.ts +13 -1
  138. package/dist/route/auth.d.ts +4 -1
  139. package/dist/route/auth.js +6 -0
  140. package/dist/route/framing.d.ts +5 -1
  141. package/dist/route/framing.js +9 -0
  142. package/dist/route/media-protocol.d.ts +116 -3
  143. package/dist/route/media-protocol.js +60 -4
  144. package/dist/route/media.d.ts +58 -7
  145. package/dist/route/media.js +201 -29
  146. package/dist/schema/events.d.ts +0 -6
  147. package/dist/schema/messages.d.ts +0 -3
  148. package/dist/schema/options.d.ts +4 -3
  149. package/dist/schema/options.js +3 -2
  150. package/dist/speech-client.d.ts +21 -0
  151. package/dist/speech-client.js +25 -0
  152. package/dist/speech.d.ts +1150 -0
  153. package/dist/speech.js +119 -0
  154. package/dist/transcription-client.d.ts +28 -0
  155. package/dist/transcription-client.js +44 -0
  156. package/dist/transcription.d.ts +1504 -0
  157. package/dist/transcription.js +133 -0
  158. package/dist/utils/bytes.d.ts +1 -0
  159. package/dist/utils/bytes.js +10 -0
  160. package/dist/utils/media-type.d.ts +1 -0
  161. package/dist/utils/media-type.js +22 -1
  162. package/dist/video-client.d.ts +28 -0
  163. package/dist/video-client.js +40 -0
  164. package/dist/video.d.ts +1359 -0
  165. package/dist/video.js +119 -0
  166. package/package.json +3 -3
  167. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  168. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
@@ -0,0 +1,114 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { ImageModel, ImageResponse } from "../image.js";
3
+ import { Media } from "../media.js";
4
+ import { MediaProtocol } from "../route/media-protocol.js";
5
+ import { MediaRoute } from "../route/media.js";
6
+ import { ProviderID, mergeJsonRecords } from "../schema/index.js";
7
+ import { ProviderShared, optionalNull } from "./shared.js";
8
+ import { FalQueue } from "./utils/fal-queue.js";
9
+ import { MediaInput } from "./utils/media-input.js";
10
+ const ADAPTER = "fal-images";
11
+ const NAME = "fal Images";
12
+ const PROVIDER = ProviderID.make("fal");
13
+ // ---------------------------------------------------------------------------
14
+ // 2. Response schema
15
+ // ---------------------------------------------------------------------------
16
+ const QueueResult = Schema.StructWithRest(Schema.Struct({
17
+ images: Schema.Array(Schema.Struct({
18
+ url: Schema.String,
19
+ width: optionalNull(Schema.Number),
20
+ height: optionalNull(Schema.Number),
21
+ content_type: optionalNull(Schema.String),
22
+ })),
23
+ seed: optionalNull(Schema.Number),
24
+ has_nsfw_concepts: optionalNull(Schema.Array(Schema.Boolean)),
25
+ }), [Schema.Record(Schema.String, Schema.Unknown)]);
26
+ // ---------------------------------------------------------------------------
27
+ // 5. Request body construction
28
+ // ---------------------------------------------------------------------------
29
+ const sizing = (model) => {
30
+ if (/^fal-ai\/(nano-banana|flux-pro\/v1\.1-ultra)/.test(model))
31
+ return "aspect_ratio";
32
+ if (model.startsWith("fal-ai/flux"))
33
+ return "image_size";
34
+ return undefined;
35
+ };
36
+ const unsupported = (model, field, message) => ProviderShared.unsupportedOperation({
37
+ operation: `media.${field}`,
38
+ provider: PROVIDER,
39
+ route: ADAPTER,
40
+ message: `${model} ${message}`,
41
+ });
42
+ const validate = (request) => {
43
+ const id = request.model.id;
44
+ const field = sizing(id);
45
+ if (request.size !== undefined && request.aspectRatio !== undefined)
46
+ return Effect.fail(ProviderShared.invalidRequest(`${NAME} accepts either size or aspectRatio, not both`));
47
+ if (request.size !== undefined && field === "aspect_ratio")
48
+ return Effect.fail(unsupported(id, "size", "sizes by aspectRatio"));
49
+ if (request.aspectRatio !== undefined && field === "image_size")
50
+ return Effect.fail(unsupported(id, "aspectRatio", "sizes by size (image_size)"));
51
+ if ((request.images?.length ?? 0) > 1 && !isEdit(id))
52
+ return Effect.fail(unsupported(id, "images", "takes one image_url; use an /edit endpoint for several images"));
53
+ return Effect.void;
54
+ };
55
+ // `/edit` endpoints take an `image_urls` list; image-to-image, fill, and Ultra take one `image_url` (beside `mask_url`).
56
+ const isEdit = (model) => model.endsWith("/edit");
57
+ const fromRequest = Effect.fn("FalImages.fromRequest")(function* (request) {
58
+ yield* validate(request);
59
+ const images = yield* Effect.forEach(request.images ?? [], (image) => FalQueue.mediaUrl(image, NAME));
60
+ const edit = isEdit(request.model.id);
61
+ return MediaProtocol.json(mergeJsonRecords({
62
+ prompt: request.prompt,
63
+ num_images: request.n,
64
+ seed: request.seed,
65
+ image_size: request.size === undefined ? undefined : MediaInput.dimensions(request.size),
66
+ aspect_ratio: request.aspectRatio,
67
+ output_format: request.format,
68
+ image_urls: edit && images.length > 0 ? images : undefined,
69
+ image_url: edit ? undefined : images[0],
70
+ mask_url: request.mask === undefined ? undefined : yield* FalQueue.mediaUrl(request.mask, NAME),
71
+ }, request.providerOptions, request.http?.body) ?? {});
72
+ });
73
+ // ---------------------------------------------------------------------------
74
+ // 6. Response decoding
75
+ // ---------------------------------------------------------------------------
76
+ const decodeQueueResult = MediaProtocol.decodeJson(ADAPTER, NAME, QueueResult);
77
+ const decodeResult = Effect.fn("FalImages.decodeResult")(function* (response, context) {
78
+ const output = yield* decodeQueueResult(response);
79
+ const { images, seed, has_nsfw_concepts, ...rest } = output.value;
80
+ if (images.length === 0)
81
+ return yield* output.invalid(`${NAME} returned no images`);
82
+ // With the safety checker on, flagged images come back blacked out rather than omitted.
83
+ const flagged = (has_nsfw_concepts ?? []).flatMap((value, index) => (value ? [index] : []));
84
+ return new ImageResponse({
85
+ images: images.map((image) => Media.url(image.url, {
86
+ mediaType: image.content_type ?? undefined,
87
+ info: { width: image.width ?? undefined, height: image.height ?? undefined },
88
+ })),
89
+ notices: flagged.length === 0
90
+ ? undefined
91
+ : flagged.map((index) => ({ type: "moderated", message: `${NAME} flagged image ${index} as NSFW` })),
92
+ providerMetadata: { fal: { requestId: context.token.requestID, seed: seed ?? undefined, ...rest } },
93
+ });
94
+ });
95
+ // ---------------------------------------------------------------------------
96
+ // 7. Protocol and route
97
+ // ---------------------------------------------------------------------------
98
+ export const protocol = FalQueue.protocol({
99
+ id: ADAPTER,
100
+ name: NAME,
101
+ from: fromRequest,
102
+ decodeResult,
103
+ });
104
+ export const model = (input) => ImageModel.fromRoute({
105
+ id: ADAPTER,
106
+ provider: PROVIDER,
107
+ protocol,
108
+ baseURL: FalQueue.DEFAULT_BASE_URL,
109
+ path: ({ request }) => `/${request.model.id}`,
110
+ }, input);
111
+ export const FalImages = {
112
+ protocol,
113
+ model,
114
+ };
@@ -0,0 +1,29 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { VideoModel, VideoResponse, type VideoRequestFor } from "../video.js";
4
+ export type FalVideoString<Known extends string> = Known | (string & {});
5
+ /**
6
+ * Provider-native input. fal video endpoints are model-specific: `duration` is a string enum whose values differ per
7
+ * model (`"8s"` for Veo, `"5"` for Kling), and last-frame fields are named per model (`end_image_url`,
8
+ * `last_frame_url`, `tail_image_url`), so those pass through here instead of lowering from common fields.
9
+ */
10
+ export type FalVideoOptions = {
11
+ readonly duration?: FalVideoString<"4s" | "6s" | "8s" | "5" | "10">;
12
+ } & Record<string, unknown>;
13
+ export type Request = VideoRequestFor<FalVideoOptions>;
14
+ export declare const protocol: MediaProtocol.Queued<Request, VideoResponse, {
15
+ readonly requestID: string;
16
+ readonly statusURL: string;
17
+ readonly responseURL: string;
18
+ readonly cancelURL: string;
19
+ }>;
20
+ export declare const model: (input: MediaRoute.ModelInput) => VideoModel<FalVideoOptions>;
21
+ export declare const FalVideo: {
22
+ readonly protocol: MediaProtocol.Queued<Request, VideoResponse, {
23
+ readonly requestID: string;
24
+ readonly statusURL: string;
25
+ readonly responseURL: string;
26
+ readonly cancelURL: string;
27
+ }>;
28
+ readonly model: (input: MediaRoute.ModelInput) => VideoModel<FalVideoOptions>;
29
+ };
@@ -0,0 +1,88 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { Media } from "../media.js";
3
+ import { MediaProtocol } from "../route/media-protocol.js";
4
+ import { MediaRoute } from "../route/media.js";
5
+ import { ProviderID, mergeJsonRecords } from "../schema/index.js";
6
+ import { VideoModel, VideoResponse } from "../video.js";
7
+ import { ProviderShared, optionalNull } from "./shared.js";
8
+ import { FalQueue } from "./utils/fal-queue.js";
9
+ const ADAPTER = "fal-video";
10
+ const NAME = "fal Video";
11
+ const PROVIDER = ProviderID.make("fal");
12
+ // ---------------------------------------------------------------------------
13
+ // 2. Response schema
14
+ // ---------------------------------------------------------------------------
15
+ const QueueResult = Schema.StructWithRest(Schema.Struct({
16
+ video: Schema.Struct({
17
+ url: Schema.String,
18
+ content_type: optionalNull(Schema.String),
19
+ file_name: optionalNull(Schema.String),
20
+ file_size: optionalNull(Schema.Number),
21
+ }),
22
+ seed: optionalNull(Schema.Number),
23
+ }), [Schema.Record(Schema.String, Schema.Unknown)]);
24
+ // ---------------------------------------------------------------------------
25
+ // 5. Request body construction
26
+ // ---------------------------------------------------------------------------
27
+ const fromRequest = Effect.fn("FalVideo.fromRequest")(function* (request) {
28
+ if (request.frames?.last !== undefined)
29
+ return yield* ProviderShared.unsupportedOperation({
30
+ operation: "video.frames.last",
31
+ provider: PROVIDER,
32
+ route: ADAPTER,
33
+ message: `${NAME} names the last frame per model; pass it through providerOptions (e.g. end_image_url) instead of frames.last`,
34
+ });
35
+ const imageUrl = request.frames?.first === undefined ? undefined : yield* FalQueue.mediaUrl(request.frames.first, NAME);
36
+ const videoUrl = request.video === undefined ? undefined : yield* FalQueue.mediaUrl(request.video, NAME);
37
+ return MediaProtocol.json(mergeJsonRecords({
38
+ prompt: request.prompt,
39
+ negative_prompt: request.negativePrompt,
40
+ seed: request.seed,
41
+ aspect_ratio: request.aspectRatio,
42
+ resolution: request.resolution,
43
+ generate_audio: request.audio,
44
+ image_url: imageUrl,
45
+ video_url: videoUrl,
46
+ }, request.providerOptions, request.http?.body) ?? {});
47
+ });
48
+ // ---------------------------------------------------------------------------
49
+ // 6. Response decoding
50
+ // ---------------------------------------------------------------------------
51
+ const decodeQueueResult = MediaProtocol.decodeJson(ADAPTER, NAME, QueueResult);
52
+ const decodeResult = Effect.fn("FalVideo.decodeResult")(function* (response, context) {
53
+ const output = yield* decodeQueueResult(response);
54
+ const { video, seed, ...rest } = output.value;
55
+ return new VideoResponse({
56
+ videos: [Media.url(video.url, { mediaType: video.content_type ?? "video/mp4" })],
57
+ providerMetadata: {
58
+ fal: {
59
+ requestId: context.token.requestID,
60
+ seed: seed ?? undefined,
61
+ fileName: video.file_name ?? undefined,
62
+ fileSize: video.file_size ?? undefined,
63
+ ...rest,
64
+ },
65
+ },
66
+ });
67
+ });
68
+ // ---------------------------------------------------------------------------
69
+ // 7. Protocol and route
70
+ // ---------------------------------------------------------------------------
71
+ export const protocol = FalQueue.protocol({
72
+ id: ADAPTER,
73
+ name: NAME,
74
+ unsupported: ["n", "durationSeconds", "references"],
75
+ from: fromRequest,
76
+ decodeResult,
77
+ });
78
+ export const model = (input) => VideoModel.fromRoute({
79
+ id: ADAPTER,
80
+ provider: PROVIDER,
81
+ protocol,
82
+ baseURL: FalQueue.DEFAULT_BASE_URL,
83
+ path: ({ request }) => `/${request.model.id}`,
84
+ }, input);
85
+ export const FalVideo = {
86
+ protocol,
87
+ model,
88
+ };
@@ -76,7 +76,7 @@ declare const GeminiBody: Schema.Struct<{
76
76
  readonly functionDeclarations: Schema.$Array<Schema.Struct<{
77
77
  readonly name: Schema.String;
78
78
  readonly description: Schema.String;
79
- readonly parameters: Schema.optional<Schema.$Record<Schema.String, Schema.Unknown>>;
79
+ readonly parametersJsonSchema: Schema.$Record<Schema.String, Schema.Unknown>;
80
80
  }>>;
81
81
  }>>>;
82
82
  toolConfig: Schema.optional<Schema.Struct<{
@@ -148,11 +148,11 @@ export declare const protocol: Protocol<{
148
148
  }[];
149
149
  readonly tools?: readonly {
150
150
  readonly functionDeclarations: readonly {
151
- readonly description: string;
152
151
  readonly name: string;
153
- readonly parameters?: {
152
+ readonly description: string;
153
+ readonly parametersJsonSchema: {
154
154
  readonly [x: string]: unknown;
155
- } | undefined;
155
+ };
156
156
  }[];
157
157
  }[] | undefined;
158
158
  readonly serviceTier?: string | undefined;
@@ -206,11 +206,11 @@ export declare const protocol: Protocol<{
206
206
  readonly safetyRatings?: unknown;
207
207
  } | null | undefined;
208
208
  readonly usageMetadata?: {
209
- readonly cachedContentTokenCount?: number | null | undefined;
210
- readonly thoughtsTokenCount?: number | null | undefined;
211
209
  readonly promptTokenCount?: number | null | undefined;
212
210
  readonly candidatesTokenCount?: number | null | undefined;
213
211
  readonly totalTokenCount?: number | null | undefined;
212
+ readonly cachedContentTokenCount?: number | null | undefined;
213
+ readonly thoughtsTokenCount?: number | null | undefined;
214
214
  } | null | undefined;
215
215
  }, {
216
216
  route: string;
@@ -262,11 +262,11 @@ export declare const route: Route<{
262
262
  }[];
263
263
  readonly tools?: readonly {
264
264
  readonly functionDeclarations: readonly {
265
- readonly description: string;
266
265
  readonly name: string;
267
- readonly parameters?: {
266
+ readonly description: string;
267
+ readonly parametersJsonSchema: {
268
268
  readonly [x: string]: unknown;
269
- } | undefined;
269
+ };
270
270
  }[];
271
271
  }[] | undefined;
272
272
  readonly serviceTier?: string | undefined;
@@ -9,7 +9,7 @@ import { AIError, LLMEvent, Usage, } from "../schema/index.js";
9
9
  import { classifyProviderFailure } from "../provider-error.js";
10
10
  import { Media } from "../media.js";
11
11
  import { JsonObject, knownString, lenient, optionalArray, optionalNull, ProviderShared } from "./shared.js";
12
- import { GeminiToolSchema } from "./utils/gemini-tool-schema.js";
12
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
13
13
  import { Lifecycle } from "./utils/lifecycle.js";
14
14
  import { ToolSchemaProjection } from "./utils/tool-schema.js";
15
15
  const ADAPTER = "gemini";
@@ -102,7 +102,7 @@ const GeminiSystemInstruction = Schema.Struct({
102
102
  const GeminiFunctionDeclaration = Schema.Struct({
103
103
  name: Schema.String,
104
104
  description: Schema.String,
105
- parameters: Schema.optional(JsonObject),
105
+ parametersJsonSchema: JsonObject,
106
106
  });
107
107
  const GeminiTool = Schema.Struct({
108
108
  functionDeclarations: Schema.Array(GeminiFunctionDeclaration),
@@ -186,34 +186,14 @@ const GeminiEvent = Schema.Struct({
186
186
  usageMetadata: optionalNull(GeminiUsage),
187
187
  });
188
188
  // =============================================================================
189
- // Tool Schema Conversion
190
- // =============================================================================
191
- // Tool-schema conversion has two distinct concerns:
192
- //
193
- // 1. Sanitize — fix common authoring mistakes Gemini rejects: integer/number
194
- // enums (must be strings), `required` entries that don't match a property,
195
- // untyped arrays (`items` must be present), and `properties`/`required`
196
- // keys on non-object scalars. Mirrors OpenCode's historical Gemini rules.
197
- //
198
- // 2. Project — lossy mapping from JSON Schema to Gemini's schema dialect:
199
- // drop empty root parameter schemas while preserving nested empty objects,
200
- // expand type arrays into `anyOf`, derive `nullable: true` from null members,
201
- // coerce `const` to `[const]` enum, recurse properties/items, and propagate
202
- // only an allowlisted set of keys (description, required, format, type,
203
- // nullable, enum, properties, items, allOf, anyOf, oneOf, minLength).
204
- // Anything outside the allowlist (e.g. `additionalProperties`, `$ref`) is
205
- // silently dropped.
206
- //
207
- // Sanitize runs first, then project. The implementation lives in
208
- // `utils/gemini-tool-schema` so this protocol keeps the same shape as the other
209
- // provider protocols.
210
- // =============================================================================
211
189
  // Request Lowering
212
190
  // =============================================================================
213
- const lowerTool = (tool, inputSchema) => ({
191
+ // Tool schemas go in `parametersJsonSchema`, which accepts standard JSON Schema. Gemini's schema
192
+ // rules are this API's default, including for tuned endpoints whose IDs do not name Gemini.
193
+ const lowerTool = (tool, model) => ({
214
194
  name: tool.name,
215
195
  description: tool.description,
216
- parameters: GeminiToolSchema.convert(inputSchema),
196
+ parametersJsonSchema: ToolSchemaProjection.modelCompatibility(tool.inputSchema, model, "gemini"),
217
197
  });
218
198
  const lowerToolConfig = (toolChoice) => ProviderShared.matchToolChoice("Gemini", toolChoice, {
219
199
  auto: () => ({ functionCallingConfig: { mode: "AUTO" } }),
@@ -221,15 +201,10 @@ const lowerToolConfig = (toolChoice) => ProviderShared.matchToolChoice("Gemini",
221
201
  required: () => ({ functionCallingConfig: { mode: "ANY" } }),
222
202
  tool: (name) => ({ functionCallingConfig: { mode: "ANY", allowedFunctionNames: [name] } }),
223
203
  });
224
- // Gemini does not fetch public URLs; inline payloads and Gemini Files references are the accepted inputs.
225
204
  const lowerContentPart = Effect.fn("Gemini.lowerContentPart")(function* (part) {
226
205
  if (part.type === "text")
227
206
  return { text: part.text };
228
- const source = part.media.source;
229
- if (source.type === "ref" && source.provider === "google")
230
- return { fileData: { mimeType: part.media.mediaType, fileUri: source.id } };
231
- const media = yield* ProviderShared.requireInlineMedia("Gemini", part.media);
232
- return { inlineData: { mimeType: media.mime, data: media.base64 } };
207
+ return yield* GeminiGenerateContent.mediaPart("Gemini", part.media);
233
208
  });
234
209
  const providerMetadata = (key, metadata) => ({ [key]: metadata });
235
210
  const thoughtSignature = (metadata, key) => {
@@ -382,7 +357,6 @@ const fromRequest = Effect.fn("Gemini.fromRequest")(function* (request) {
382
357
  const hasTools = flattened.tools.length > 0;
383
358
  const generation = request.generation;
384
359
  const options = yield* decodeOptions(request.providerOptions ?? {});
385
- const toolSchemaCompatibility = request.model.compatibility?.toolSchema;
386
360
  const generationConfig = {
387
361
  maxOutputTokens: generation?.maxTokens,
388
362
  temperature: generation?.temperature,
@@ -405,7 +379,7 @@ const fromRequest = Effect.fn("Gemini.fromRequest")(function* (request) {
405
379
  tools: hasTools
406
380
  ? [
407
381
  {
408
- functionDeclarations: flattened.tools.map((tool) => lowerTool(tool, ToolSchemaProjection.modelCompatibility(tool.inputSchema, toolSchemaCompatibility))),
382
+ functionDeclarations: flattened.tools.map((tool) => lowerTool(tool, request.model)),
409
383
  },
410
384
  ]
411
385
  : undefined,
@@ -1,10 +1,10 @@
1
1
  import { Effect, Schema } from "effect";
2
2
  import { ImageModel, ImageResponse } from "../image.js";
3
- import { Media } from "../media.js";
4
3
  import { MediaProtocol } from "../route/media-protocol.js";
5
4
  import { MediaRoute } from "../route/media.js";
6
5
  import { ProviderID, mergeJsonRecords } from "../schema/index.js";
7
6
  import { ProviderShared } from "./shared.js";
7
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
8
8
  import { MediaInput } from "./utils/media-input.js";
9
9
  const ADAPTER = "google-images";
10
10
  const NAME = "Google Images";
@@ -61,18 +61,6 @@ const generationConfig = (request) => {
61
61
  thinkingConfig: Object.values(thinkingConfig).some((value) => value !== undefined) ? thinkingConfig : undefined,
62
62
  }, native) ?? { responseModalities: ["IMAGE"] });
63
63
  };
64
- // Gemini does not fetch public URLs; inline payloads or Gemini Files references are the only accepted inputs.
65
- const imagePart = (asset) => {
66
- const inline = asset.inline();
67
- if (inline)
68
- return Effect.succeed({ inlineData: { mimeType: inline.mime, data: inline.base64 } });
69
- const id = MediaInput.refID(asset, PROVIDER);
70
- if (id)
71
- return Effect.succeed({ fileData: { mimeType: asset.mediaType, fileUri: id } });
72
- if (asset.source.type === "ref")
73
- return Effect.fail(ProviderShared.invalidRequest("Google generateContent requires Gemini file references rather than other providers' file IDs"));
74
- return Effect.fail(ProviderShared.invalidRequest("Google generateContent does not fetch public image URLs; use bytes, a data URL, or a Gemini file reference"));
75
- };
76
64
  const fromRequest = Effect.fn("GoogleImages.fromRequest")(function* (request) {
77
65
  if (request.n !== undefined && request.n > 1)
78
66
  return yield* ProviderShared.unsupportedOperation({
@@ -81,7 +69,7 @@ const fromRequest = Effect.fn("GoogleImages.fromRequest")(function* (request) {
81
69
  route: ADAPTER,
82
70
  message: `${NAME} generates one image per request; call it once per image instead of n=${request.n}`,
83
71
  });
84
- const parts = yield* Effect.forEach(request.images ?? [], imagePart);
72
+ const parts = yield* Effect.forEach(request.images ?? [], (image) => GeminiGenerateContent.mediaPart(NAME, image));
85
73
  return MediaProtocol.json(mergeJsonRecords({
86
74
  contents: [{ role: "user", parts: [{ text: request.prompt }, ...parts] }],
87
75
  generationConfig: generationConfig(request),
@@ -0,0 +1,130 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { SpeechModel, type SpeechRequestFor } from "../speech.js";
4
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
5
+ import { SpeechStream } from "./utils/speech-stream.js";
6
+ export declare const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
7
+ /** Style is directed in the text itself, and `speechConfig.multiSpeakerVoiceConfig` excludes `voice`. */
8
+ export type GoogleSpeechOptions = {
9
+ readonly temperature?: number;
10
+ readonly seed?: number;
11
+ readonly speechConfig?: {
12
+ readonly multiSpeakerVoiceConfig?: {
13
+ readonly speakerVoiceConfigs: ReadonlyArray<{
14
+ readonly speaker: string;
15
+ readonly voiceConfig: {
16
+ readonly prebuiltVoiceConfig: {
17
+ readonly voiceName: string;
18
+ };
19
+ };
20
+ }>;
21
+ };
22
+ };
23
+ } & Record<string, unknown>;
24
+ export type Request = SpeechRequestFor<GoogleSpeechOptions>;
25
+ interface State extends SpeechStream.Audio, GeminiGenerateContent.Metadata {
26
+ readonly mimeType?: string;
27
+ }
28
+ export declare const protocol: MediaProtocol.Streamed<Request, {
29
+ readonly type: "audio-delta";
30
+ readonly chunk: Uint8Array<ArrayBufferLike>;
31
+ } | {
32
+ readonly type: "timestamps";
33
+ readonly items: readonly {
34
+ readonly text: string;
35
+ readonly startSeconds: number;
36
+ readonly endSeconds: number;
37
+ }[];
38
+ } | {
39
+ readonly type: "finish";
40
+ readonly audio: import("../media.js").Asset;
41
+ readonly providerMetadata?: {
42
+ readonly [x: string]: {
43
+ readonly [x: string]: unknown;
44
+ };
45
+ } | undefined;
46
+ readonly usage?: {
47
+ readonly type: "tokens";
48
+ readonly input?: number | undefined;
49
+ readonly output?: number | undefined;
50
+ readonly total?: number | undefined;
51
+ readonly details?: {
52
+ readonly [x: string]: unknown;
53
+ } | undefined;
54
+ } | {
55
+ readonly type: "seconds";
56
+ readonly seconds: number;
57
+ } | {
58
+ readonly type: "characters";
59
+ readonly characters: number;
60
+ } | {
61
+ readonly type: "credits";
62
+ readonly credits: number;
63
+ } | {
64
+ readonly type: "compute";
65
+ readonly seconds: number;
66
+ } | undefined;
67
+ readonly notices?: readonly {
68
+ readonly type: "other" | "moderated" | "filtered";
69
+ readonly message: string;
70
+ readonly providerMetadata?: {
71
+ readonly [x: string]: {
72
+ readonly [x: string]: unknown;
73
+ };
74
+ } | undefined;
75
+ }[] | undefined;
76
+ }, string, State>;
77
+ export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
78
+ export declare const GoogleSpeech: {
79
+ readonly protocol: MediaProtocol.Streamed<Request, {
80
+ readonly type: "audio-delta";
81
+ readonly chunk: Uint8Array<ArrayBufferLike>;
82
+ } | {
83
+ readonly type: "timestamps";
84
+ readonly items: readonly {
85
+ readonly text: string;
86
+ readonly startSeconds: number;
87
+ readonly endSeconds: number;
88
+ }[];
89
+ } | {
90
+ readonly type: "finish";
91
+ readonly audio: import("../media.js").Asset;
92
+ readonly providerMetadata?: {
93
+ readonly [x: string]: {
94
+ readonly [x: string]: unknown;
95
+ };
96
+ } | undefined;
97
+ readonly usage?: {
98
+ readonly type: "tokens";
99
+ readonly input?: number | undefined;
100
+ readonly output?: number | undefined;
101
+ readonly total?: number | undefined;
102
+ readonly details?: {
103
+ readonly [x: string]: unknown;
104
+ } | undefined;
105
+ } | {
106
+ readonly type: "seconds";
107
+ readonly seconds: number;
108
+ } | {
109
+ readonly type: "characters";
110
+ readonly characters: number;
111
+ } | {
112
+ readonly type: "credits";
113
+ readonly credits: number;
114
+ } | {
115
+ readonly type: "compute";
116
+ readonly seconds: number;
117
+ } | undefined;
118
+ readonly notices?: readonly {
119
+ readonly type: "other" | "moderated" | "filtered";
120
+ readonly message: string;
121
+ readonly providerMetadata?: {
122
+ readonly [x: string]: {
123
+ readonly [x: string]: unknown;
124
+ };
125
+ } | undefined;
126
+ }[] | undefined;
127
+ }, string, State>;
128
+ readonly model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
129
+ };
130
+ export {};
@@ -0,0 +1,84 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { ProviderID, mergeJsonRecords } from "../schema/index.js";
5
+ import { SpeechModel } from "../speech.js";
6
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
7
+ import { SpeechStream } from "./utils/speech-stream.js";
8
+ const ADAPTER = "google-speech";
9
+ const NAME = "Google Speech";
10
+ const PROVIDER = ProviderID.make("google");
11
+ export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
12
+ const DEFAULT_SAMPLE_RATE = 24000;
13
+ // ---------------------------------------------------------------------------
14
+ // 3. Streaming event schema
15
+ // ---------------------------------------------------------------------------
16
+ const GenerateContentChunk = GeminiGenerateContent.chunk(Schema.Struct({
17
+ text: Schema.optional(Schema.String),
18
+ inlineData: Schema.optional(Schema.Struct({ mimeType: Schema.String, data: Schema.Uint8ArrayFromBase64 })),
19
+ }));
20
+ const decodeChunk = MediaProtocol.decodeFrame(ADAPTER, NAME, GenerateContentChunk);
21
+ // ---------------------------------------------------------------------------
22
+ // 5. Request body construction
23
+ // ---------------------------------------------------------------------------
24
+ const fromRequest = Effect.fn("GoogleSpeech.fromRequest")(function* (request) {
25
+ if (request.format !== undefined && request.format !== "pcm")
26
+ return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} only returns raw PCM; request format "pcm" or omit it, then wrap the samples yourself`);
27
+ const voiceName = SpeechStream.voiceID(request.voice);
28
+ return MediaProtocol.json(mergeJsonRecords({
29
+ contents: [{ role: "user", parts: [{ text: request.text }] }],
30
+ generationConfig: mergeJsonRecords({
31
+ responseModalities: ["AUDIO"],
32
+ speechConfig: {
33
+ voiceConfig: voiceName === undefined ? undefined : { prebuiltVoiceConfig: { voiceName } },
34
+ languageCode: request.language,
35
+ },
36
+ }, request.providerOptions),
37
+ }, request.http?.body) ?? {});
38
+ });
39
+ // ---------------------------------------------------------------------------
40
+ // 6. Stream parsing
41
+ // ---------------------------------------------------------------------------
42
+ const step = Effect.fn("GoogleSpeech.step")(function* (state, frame) {
43
+ const chunk = yield* decodeChunk(frame);
44
+ const blocked = GeminiGenerateContent.blocked(NAME, chunk, frame);
45
+ if (blocked !== undefined)
46
+ return yield* blocked;
47
+ const audio = (chunk.candidates?.[0]?.content?.parts ?? []).flatMap((part) => part.inlineData === undefined ? [] : [part.inlineData]);
48
+ const next = { ...GeminiGenerateContent.track(state, chunk), mimeType: state.mimeType ?? audio[0]?.mimeType };
49
+ return [next, audio.flatMap((part) => SpeechStream.delta(next, part.data)[1])];
50
+ });
51
+ const finish = (state) => {
52
+ const sampleRate = SpeechStream.sampleRate(state.mimeType) ?? DEFAULT_SAMPLE_RATE;
53
+ return SpeechStream.finish(ADAPTER, state, {
54
+ ...SpeechStream.pcm("pcm_s16le", sampleRate, state.mimeType ?? `audio/L16;codec=pcm;rate=${sampleRate}`),
55
+ usage: GeminiGenerateContent.usage(state.usage),
56
+ providerMetadata: GeminiGenerateContent.providerMetadata(state),
57
+ detail: state.finishReason === undefined ? undefined : `finish reason: ${state.finishReason}`,
58
+ });
59
+ };
60
+ // ---------------------------------------------------------------------------
61
+ // 7. Protocol and route
62
+ // ---------------------------------------------------------------------------
63
+ export const protocol = MediaProtocol.stream({
64
+ id: ADAPTER,
65
+ name: NAME,
66
+ unsupported: ["instructions", "speed", "timestamps"],
67
+ body: { from: fromRequest },
68
+ frames: (bytes, context) => GeminiGenerateContent.frames(bytes, context.request.mode),
69
+ initial: () => ({ chunks: [] }),
70
+ step,
71
+ finish,
72
+ });
73
+ export const model = (input) => SpeechModel.fromRoute({
74
+ id: ADAPTER,
75
+ provider: PROVIDER,
76
+ protocol,
77
+ baseURL: DEFAULT_BASE_URL,
78
+ // Only `gemini-3.1-flash-tts-preview` and later stream; earlier TTS models reject `streamGenerateContent`.
79
+ path: ({ request }) => GeminiGenerateContent.path(request.model.id, request.mode),
80
+ }, input);
81
+ export const GoogleSpeech = {
82
+ protocol,
83
+ model,
84
+ };