@opencode/ai 2.0.14 → 2.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +399 -56
  2. package/dist/experimental/evaluation-client.d.ts +1 -1
  3. package/dist/experimental/evaluation-client.js +39 -3
  4. package/dist/experimental/evaluation.d.ts +4 -4
  5. package/dist/experimental/evaluation.js +2 -2
  6. package/dist/experimental/system-one.d.ts +3 -3
  7. package/dist/experimental/system-one.js +40 -51
  8. package/dist/generation.d.ts +83 -0
  9. package/dist/generation.js +113 -0
  10. package/dist/image-client.d.ts +16 -7
  11. package/dist/image-client.js +29 -13
  12. package/dist/image.d.ts +1410 -81
  13. package/dist/image.js +97 -67
  14. package/dist/index.d.ts +17 -2
  15. package/dist/index.js +12 -1
  16. package/dist/llm.d.ts +9 -1
  17. package/dist/media-model.d.ts +44 -0
  18. package/dist/media-model.js +49 -0
  19. package/dist/media.d.ts +213 -0
  20. package/dist/media.js +227 -0
  21. package/dist/promise.d.ts +974 -0
  22. package/dist/promise.js +81 -0
  23. package/dist/protocols/alibaba-chat.d.ts +12 -0
  24. package/dist/protocols/alibaba-responses.d.ts +2 -2
  25. package/dist/protocols/anthropic-messages.js +8 -19
  26. package/dist/protocols/assemblyai-transcription.d.ts +40 -0
  27. package/dist/protocols/assemblyai-transcription.js +138 -0
  28. package/dist/protocols/bedrock-converse.d.ts +4 -4
  29. package/dist/protocols/bedrock-converse.js +6 -17
  30. package/dist/protocols/bfl-images.d.ts +32 -0
  31. package/dist/protocols/bfl-images.js +153 -0
  32. package/dist/protocols/cartesia-speech.d.ts +127 -0
  33. package/dist/protocols/cartesia-speech.js +126 -0
  34. package/dist/protocols/deepgram-speech.d.ts +119 -0
  35. package/dist/protocols/deepgram-speech.js +92 -0
  36. package/dist/protocols/deepgram-transcription.d.ts +25 -0
  37. package/dist/protocols/deepgram-transcription.js +129 -0
  38. package/dist/protocols/elevenlabs-speech.d.ts +122 -0
  39. package/dist/protocols/elevenlabs-speech.js +115 -0
  40. package/dist/protocols/fal-images.d.ts +24 -0
  41. package/dist/protocols/fal-images.js +114 -0
  42. package/dist/protocols/fal-video.d.ts +29 -0
  43. package/dist/protocols/fal-video.js +88 -0
  44. package/dist/protocols/gemini.d.ts +30 -9
  45. package/dist/protocols/gemini.js +45 -35
  46. package/dist/protocols/google-images.d.ts +9 -21
  47. package/dist/protocols/google-images.js +158 -133
  48. package/dist/protocols/google-speech.d.ts +130 -0
  49. package/dist/protocols/google-speech.js +84 -0
  50. package/dist/protocols/google-transcription.d.ts +173 -0
  51. package/dist/protocols/google-transcription.js +138 -0
  52. package/dist/protocols/google-video.d.ts +26 -0
  53. package/dist/protocols/google-video.js +158 -0
  54. package/dist/protocols/meta-images.d.ts +7 -12
  55. package/dist/protocols/meta-images.js +85 -66
  56. package/dist/protocols/meta-responses.d.ts +4 -4
  57. package/dist/protocols/meta-responses.js +1 -1
  58. package/dist/protocols/mistral-chat.js +7 -6
  59. package/dist/protocols/open-responses.d.ts +17 -9
  60. package/dist/protocols/open-responses.js +24 -14
  61. package/dist/protocols/openai-chat.d.ts +118 -1
  62. package/dist/protocols/openai-chat.js +125 -44
  63. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  64. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  65. package/dist/protocols/openai-images.d.ts +128 -18
  66. package/dist/protocols/openai-images.js +177 -154
  67. package/dist/protocols/openai-responses.d.ts +15 -15
  68. package/dist/protocols/openai-responses.js +5 -6
  69. package/dist/protocols/openai-speech.d.ts +116 -0
  70. package/dist/protocols/openai-speech.js +98 -0
  71. package/dist/protocols/openai-transcription.d.ts +207 -0
  72. package/dist/protocols/openai-transcription.js +190 -0
  73. package/dist/protocols/replicate-images.d.ts +28 -0
  74. package/dist/protocols/replicate-images.js +133 -0
  75. package/dist/protocols/runway-video.d.ts +38 -0
  76. package/dist/protocols/runway-video.js +146 -0
  77. package/dist/protocols/shared.d.ts +27 -17
  78. package/dist/protocols/shared.js +52 -35
  79. package/dist/protocols/stability-images.d.ts +38 -0
  80. package/dist/protocols/stability-images.js +148 -0
  81. package/dist/protocols/utils/bedrock-media.d.ts +2 -3
  82. package/dist/protocols/utils/bedrock-media.js +4 -4
  83. package/dist/protocols/utils/fal-queue.d.ts +28 -0
  84. package/dist/protocols/utils/fal-queue.js +69 -0
  85. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  86. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  87. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  88. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  89. package/dist/protocols/utils/media-input.d.ts +18 -0
  90. package/dist/protocols/utils/media-input.js +35 -0
  91. package/dist/protocols/utils/responses-compaction.js +6 -5
  92. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  93. package/dist/protocols/utils/speech-stream.js +67 -0
  94. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  95. package/dist/protocols/utils/tool-schema.js +40 -17
  96. package/dist/protocols/utils/tool-stream.d.ts +27 -3
  97. package/dist/protocols/xai-images.d.ts +9 -15
  98. package/dist/protocols/xai-images.js +75 -84
  99. package/dist/protocols/xai-responses.d.ts +2 -2
  100. package/dist/protocols/xai-video.d.ts +34 -0
  101. package/dist/protocols/xai-video.js +147 -0
  102. package/dist/protocols/zai-chat.d.ts +13 -1
  103. package/dist/protocols/zai-images.d.ts +9 -13
  104. package/dist/protocols/zai-images.js +59 -57
  105. package/dist/provider-error.js +3 -0
  106. package/dist/providers/alibaba.d.ts +14 -2
  107. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  108. package/dist/providers/amazon-bedrock.d.ts +2 -2
  109. package/dist/providers/assemblyai.d.ts +25 -0
  110. package/dist/providers/assemblyai.js +29 -0
  111. package/dist/providers/azure.d.ts +18 -6
  112. package/dist/providers/baseten.d.ts +24 -0
  113. package/dist/providers/black-forest-labs.d.ts +25 -0
  114. package/dist/providers/black-forest-labs.js +28 -0
  115. package/dist/providers/cartesia.d.ts +24 -0
  116. package/dist/providers/cartesia.js +22 -0
  117. package/dist/providers/cerebras.d.ts +24 -0
  118. package/dist/providers/cerebras.js +6 -1
  119. package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
  120. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  121. package/dist/providers/deepgram.d.ts +29 -0
  122. package/dist/providers/deepgram.js +31 -0
  123. package/dist/providers/deepinfra.d.ts +24 -0
  124. package/dist/providers/deepinfra.js +6 -1
  125. package/dist/providers/deepseek.d.ts +24 -0
  126. package/dist/providers/elevenlabs.d.ts +24 -0
  127. package/dist/providers/elevenlabs.js +28 -0
  128. package/dist/providers/fal.d.ts +29 -0
  129. package/dist/providers/fal.js +33 -0
  130. package/dist/providers/fireworks.d.ts +24 -0
  131. package/dist/providers/google-vertex-chat.d.ts +12 -0
  132. package/dist/providers/google-vertex-responses.d.ts +2 -2
  133. package/dist/providers/google-vertex.d.ts +10 -3
  134. package/dist/providers/google.d.ts +25 -3
  135. package/dist/providers/google.js +11 -2
  136. package/dist/providers/groq.d.ts +24 -0
  137. package/dist/providers/index.d.ts +10 -0
  138. package/dist/providers/index.js +10 -0
  139. package/dist/providers/meta.d.ts +14 -2
  140. package/dist/providers/minimax.d.ts +14 -2
  141. package/dist/providers/moonshot.d.ts +14 -2
  142. package/dist/providers/moonshot.js +3 -3
  143. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  144. package/dist/providers/openai-compatible.d.ts +12 -0
  145. package/dist/providers/openai.d.ts +25 -3
  146. package/dist/providers/openai.js +10 -1
  147. package/dist/providers/openrouter.d.ts +67 -0
  148. package/dist/providers/openrouter.js +13 -1
  149. package/dist/providers/replicate.d.ts +25 -0
  150. package/dist/providers/replicate.js +22 -0
  151. package/dist/providers/runway.d.ts +24 -0
  152. package/dist/providers/runway.js +22 -0
  153. package/dist/providers/stability.d.ts +28 -0
  154. package/dist/providers/stability.js +23 -0
  155. package/dist/providers/togetherai.d.ts +24 -0
  156. package/dist/providers/vercel-ai-gateway.d.ts +41 -0
  157. package/dist/providers/vercel-ai-gateway.js +85 -0
  158. package/dist/providers/xai.d.ts +17 -0
  159. package/dist/providers/xai.js +5 -2
  160. package/dist/providers/zai-coding-plan.d.ts +15 -3
  161. package/dist/providers/zai.d.ts +13 -1
  162. package/dist/route/auth.d.ts +4 -1
  163. package/dist/route/auth.js +6 -0
  164. package/dist/route/client.d.ts +9 -1
  165. package/dist/route/endpoint.d.ts +10 -10
  166. package/dist/route/executor-service.d.ts +12 -0
  167. package/dist/route/executor-service.js +3 -0
  168. package/dist/route/executor.d.ts +4 -9
  169. package/dist/route/executor.js +3 -3
  170. package/dist/route/framing.d.ts +5 -1
  171. package/dist/route/framing.js +9 -0
  172. package/dist/route/index.d.ts +2 -0
  173. package/dist/route/index.js +2 -0
  174. package/dist/route/media-protocol.d.ts +158 -0
  175. package/dist/route/media-protocol.js +96 -0
  176. package/dist/route/media.d.ts +97 -0
  177. package/dist/route/media.js +236 -0
  178. package/dist/schema/errors.d.ts +13 -3
  179. package/dist/schema/errors.js +7 -0
  180. package/dist/schema/events.d.ts +557 -40
  181. package/dist/schema/events.js +35 -2
  182. package/dist/schema/messages.d.ts +95 -8
  183. package/dist/schema/messages.js +8 -6
  184. package/dist/schema/options.d.ts +6 -3
  185. package/dist/schema/options.js +6 -2
  186. package/dist/speech-client.d.ts +21 -0
  187. package/dist/speech-client.js +25 -0
  188. package/dist/speech.d.ts +1150 -0
  189. package/dist/speech.js +119 -0
  190. package/dist/testing.d.ts +72 -8
  191. package/dist/transcription-client.d.ts +28 -0
  192. package/dist/transcription-client.js +44 -0
  193. package/dist/transcription.d.ts +1504 -0
  194. package/dist/transcription.js +133 -0
  195. package/dist/utils/bytes.d.ts +1 -0
  196. package/dist/utils/bytes.js +10 -0
  197. package/dist/utils/media-type.d.ts +7 -0
  198. package/dist/utils/media-type.js +70 -0
  199. package/dist/utils/sanitize.js +3 -1
  200. package/dist/video-client.d.ts +28 -0
  201. package/dist/video-client.js +40 -0
  202. package/dist/video.d.ts +1359 -0
  203. package/dist/video.js +119 -0
  204. package/package.json +7 -3
  205. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  206. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
  207. package/dist/protocols/utils/image-input.d.ts +0 -21
  208. package/dist/protocols/utils/image-input.js +0 -20
@@ -1,12 +1,18 @@
1
- import { Effect, Encoding, Schema } from "effect";
2
- import { Headers, HttpClientRequest } from "effect/unstable/http";
3
- import { GeneratedImage, ImageModel, ImageResponse, } from "../image.js";
4
- import { Auth } from "../route/auth.js";
5
- import { AIError, Usage, mergeHttpOptions, mergeJsonRecords } from "../schema/index.js";
1
+ import { Effect, Schema } from "effect";
2
+ import { ImageModel, ImageResponse } from "../image.js";
3
+ import { MediaProtocol } from "../route/media-protocol.js";
4
+ import { MediaRoute } from "../route/media.js";
5
+ import { ProviderID, mergeJsonRecords } from "../schema/index.js";
6
6
  import { ProviderShared } from "./shared.js";
7
- import { ImageInputs } from "./utils/image-input.js";
7
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
8
+ import { MediaInput } from "./utils/media-input.js";
8
9
  const ADAPTER = "google-images";
10
+ const NAME = "Google Images";
11
+ const PROVIDER = ProviderID.make("google");
9
12
  export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
13
+ // ---------------------------------------------------------------------------
14
+ // 2. Response schema
15
+ // ---------------------------------------------------------------------------
10
16
  const GoogleUsage = Schema.StructWithRest(Schema.Struct({
11
17
  cachedContentTokenCount: Schema.optional(Schema.Number),
12
18
  thoughtsTokenCount: Schema.optional(Schema.Number),
@@ -41,141 +47,160 @@ const GoogleImageResponse = Schema.Struct({
41
47
  responseId: Schema.optional(Schema.String),
42
48
  promptFeedback: Schema.optional(Schema.Unknown),
43
49
  });
44
- const nativeOptions = (options) => {
45
- const { aspectRatio, imageSize, seed, thinkingLevel, includeThoughts, ...native } = options ?? {};
46
- const image = {
47
- aspectRatio,
48
- imageSize,
49
- };
50
- const thinkingConfig = {
51
- thinkingLevel,
52
- includeThoughts,
53
- };
50
+ // ---------------------------------------------------------------------------
51
+ // 5. Request body construction
52
+ // ---------------------------------------------------------------------------
53
+ const generationConfig = (request) => {
54
+ const { imageSize, thinkingLevel, includeThoughts, ...native } = request.providerOptions ?? {};
55
+ const imageConfig = { aspectRatio: request.aspectRatio, imageSize };
56
+ const thinkingConfig = { thinkingLevel, includeThoughts };
54
57
  return (mergeJsonRecords({
55
58
  responseModalities: ["IMAGE"],
56
- imageConfig: Object.values(image).some((value) => value !== undefined) ? image : undefined,
57
- seed,
59
+ imageConfig: Object.values(imageConfig).some((value) => value !== undefined) ? imageConfig : undefined,
60
+ seed: request.seed,
58
61
  thinkingConfig: Object.values(thinkingConfig).some((value) => value !== undefined) ? thinkingConfig : undefined,
59
62
  }, native) ?? { responseModalities: ["IMAGE"] });
60
63
  };
61
- const applyQuery = (url, query) => {
62
- if (!query)
63
- return url;
64
- const next = new URL(url);
65
- Object.entries(query).forEach(([key, value]) => next.searchParams.set(key, value));
66
- return next.toString();
67
- };
68
- export const model = (input) => {
69
- const route = {
70
- id: ADAPTER,
71
- generate: Effect.fn("GoogleImages.generate")(function* (request, execute) {
72
- const imageParts = yield* Effect.forEach(request.images ?? [], googleImagePart);
73
- const http = mergeHttpOptions(request.model.http, request.http);
74
- const requestBody = mergeJsonRecords({
75
- contents: [{ role: "user", parts: [{ text: request.prompt }, ...imageParts] }],
76
- generationConfig: nativeOptions(request.options),
77
- }, http?.body);
78
- const text = ProviderShared.encodeJson(requestBody);
79
- const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}/models/${request.model.id}:generateContent`, http?.query);
80
- const headers = yield* Auth.toEffect(input.auth)({
81
- request,
82
- method: "POST",
83
- url,
84
- body: text,
85
- headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
86
- });
87
- const response = yield* execute(HttpClientRequest.post(url).pipe(HttpClientRequest.setHeaders(headers), HttpClientRequest.bodyText(text, "application/json")));
88
- const output = yield* ProviderShared.imageResponse(ADAPTER, "Google Images", response);
89
- const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(GoogleImageResponse))(output.body).pipe(Effect.mapError((cause) => output.invalid("Google Images returned an invalid response", cause)));
90
- const candidates = decoded.candidates ?? [];
91
- const candidateMetadata = candidates.map((candidate, candidateIndex) => ({
92
- index: candidate.index ?? candidateIndex,
93
- finishReason: candidate.finishReason,
94
- finishMessage: candidate.finishMessage,
95
- safetyRatings: candidate.safetyRatings,
96
- citationMetadata: candidate.citationMetadata,
97
- groundingMetadata: candidate.groundingMetadata,
98
- parts: (candidate.content?.parts ?? []).map((part) => part.inlineData === undefined
99
- ? {
100
- type: "text",
101
- text: part.text,
102
- thought: part.thought,
103
- thoughtSignature: part.thoughtSignature,
104
- }
105
- : {
106
- type: "inlineData",
107
- mediaType: part.inlineData.mimeType,
108
- thought: part.thought,
109
- thoughtSignature: part.thoughtSignature,
110
- }),
111
- }));
112
- const encoded = candidates.flatMap((candidate, candidateIndex) => (candidate.content?.parts ?? []).flatMap((part, partIndex) => part.inlineData === undefined || part.thought === true
113
- ? []
114
- : [{ candidate, candidateIndex, partIndex, inlineData: part.inlineData }]));
115
- const images = yield* Effect.forEach(encoded, (item) => Effect.fromResult(Encoding.decodeBase64(item.inlineData.data)).pipe(Effect.mapError((cause) => output.invalid(`Google Images candidate ${item.candidateIndex} part ${item.partIndex} contains invalid base64 data`, cause)), Effect.map((data) => new GeneratedImage({
116
- mediaType: item.inlineData.mimeType,
117
- data,
118
- providerMetadata: {
119
- google: {
120
- candidateIndex: item.candidate.index ?? item.candidateIndex,
121
- partIndex: item.partIndex,
122
- finishReason: item.candidate.finishReason,
123
- safetyRatings: item.candidate.safetyRatings,
124
- citationMetadata: item.candidate.citationMetadata,
125
- groundingMetadata: item.candidate.groundingMetadata,
126
- thoughtSignature: item.candidate.content?.parts[item.partIndex]?.thoughtSignature,
127
- },
64
+ const fromRequest = Effect.fn("GoogleImages.fromRequest")(function* (request) {
65
+ if (request.n !== undefined && request.n > 1)
66
+ return yield* ProviderShared.unsupportedOperation({
67
+ operation: "image.n",
68
+ provider: PROVIDER,
69
+ route: ADAPTER,
70
+ message: `${NAME} generates one image per request; call it once per image instead of n=${request.n}`,
71
+ });
72
+ const parts = yield* Effect.forEach(request.images ?? [], (image) => GeminiGenerateContent.mediaPart(NAME, image));
73
+ return MediaProtocol.json(mergeJsonRecords({
74
+ contents: [{ role: "user", parts: [{ text: request.prompt }, ...parts] }],
75
+ generationConfig: generationConfig(request),
76
+ }, request.http?.body) ?? {});
77
+ });
78
+ // ---------------------------------------------------------------------------
79
+ // 6. Response decoding
80
+ // ---------------------------------------------------------------------------
81
+ const decodeResponse = Effect.fn("GoogleImages.decodeResponse")(function* (response) {
82
+ const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, GoogleImageResponse)(response);
83
+ const decoded = output.value;
84
+ const candidates = decoded.candidates ?? [];
85
+ const candidateMetadata = candidates.map((candidate, candidateIndex) => ({
86
+ index: candidate.index ?? candidateIndex,
87
+ finishReason: candidate.finishReason,
88
+ finishMessage: candidate.finishMessage,
89
+ safetyRatings: candidate.safetyRatings,
90
+ citationMetadata: candidate.citationMetadata,
91
+ groundingMetadata: candidate.groundingMetadata,
92
+ parts: (candidate.content?.parts ?? []).map((part) => part.inlineData === undefined
93
+ ? { type: "text", text: part.text, thought: part.thought, thoughtSignature: part.thoughtSignature }
94
+ : {
95
+ type: "inlineData",
96
+ mediaType: part.inlineData.mimeType,
97
+ thought: part.thought,
98
+ thoughtSignature: part.thoughtSignature,
99
+ }),
100
+ }));
101
+ // Thought parts are drafts; only non-thought inline data is a final image.
102
+ const encoded = candidates.flatMap((candidate, candidateIndex) => (candidate.content?.parts ?? []).flatMap((part, partIndex) => part.inlineData === undefined || part.thought === true
103
+ ? []
104
+ : [
105
+ {
106
+ candidate,
107
+ candidateIndex,
108
+ partIndex,
109
+ inlineData: part.inlineData,
110
+ thoughtSignature: part.thoughtSignature,
111
+ },
112
+ ]));
113
+ const images = yield* Effect.forEach(encoded, (item) => MediaInput.decodedAsset(output.invalid, `${NAME} candidate ${item.candidateIndex} part ${item.partIndex}`, item.inlineData.data, item.inlineData.mimeType, {
114
+ providerMetadata: {
115
+ google: {
116
+ candidateIndex: item.candidate.index ?? item.candidateIndex,
117
+ partIndex: item.partIndex,
118
+ finishReason: item.candidate.finishReason,
119
+ safetyRatings: item.candidate.safetyRatings,
120
+ citationMetadata: item.candidate.citationMetadata,
121
+ groundingMetadata: item.candidate.groundingMetadata,
122
+ thoughtSignature: item.thoughtSignature,
123
+ },
124
+ },
125
+ }));
126
+ if (images.length === 0) {
127
+ const finishReasons = candidates.flatMap((candidate) => candidate.finishReason === undefined ? [] : [candidate.finishReason]);
128
+ return yield* output.invalid(`${NAME} returned no final images${finishReasons.length === 0 ? "" : ` (finish reasons: ${finishReasons.join(", ")})`}; inspect body for prompt feedback and candidate details`);
129
+ }
130
+ // Candidates that stopped for a safety or policy reason are partial results, not a silent drop.
131
+ const notices = [
132
+ ...(decoded.promptFeedback === undefined
133
+ ? []
134
+ : [
135
+ {
136
+ type: "filtered",
137
+ message: `${NAME} reported prompt feedback`,
138
+ providerMetadata: { google: { promptFeedback: decoded.promptFeedback } },
128
139
  },
129
- }))));
130
- if (images.length === 0) {
131
- const finishReasons = candidates.flatMap((candidate) => candidate.finishReason === undefined ? [] : [candidate.finishReason]);
132
- return yield* output.invalid(`Google Images returned no final images${finishReasons.length === 0 ? "" : ` (finish reasons: ${finishReasons.join(", ")})`}; inspect body for prompt feedback and candidate details`);
133
- }
134
- const usage = decoded.usageMetadata;
135
- const outputTokens = usage?.candidatesTokenCount === undefined
136
- ? undefined
137
- : usage.candidatesTokenCount + (usage.thoughtsTokenCount ?? 0);
138
- return new ImageResponse({
139
- images,
140
- usage: usage === undefined
141
- ? undefined
142
- : new Usage({
143
- inputTokens: usage.promptTokenCount,
144
- outputTokens,
145
- nonCachedInputTokens: ProviderShared.subtractTokens(usage.promptTokenCount, usage.cachedContentTokenCount),
146
- cacheReadInputTokens: usage.cachedContentTokenCount,
147
- reasoningTokens: usage.thoughtsTokenCount,
148
- totalTokens: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
149
- providerMetadata: { google: usage },
150
- }),
151
- providerMetadata: {
152
- google: {
153
- modelVersion: decoded.modelVersion,
154
- responseId: decoded.responseId,
155
- promptFeedback: decoded.promptFeedback,
156
- candidates: candidateMetadata,
140
+ ]),
141
+ ...candidates.flatMap((candidate, index) => candidate.finishReason === undefined || candidate.finishReason === "STOP"
142
+ ? []
143
+ : [
144
+ {
145
+ type: "filtered",
146
+ message: `${NAME} candidate ${candidate.index ?? index} finished with ${candidate.finishReason}${candidate.finishMessage === undefined ? "" : `: ${candidate.finishMessage}`}`,
147
+ providerMetadata: {
148
+ google: {
149
+ candidateIndex: candidate.index ?? index,
150
+ finishReason: candidate.finishReason,
151
+ finishMessage: candidate.finishMessage,
152
+ safetyRatings: candidate.safetyRatings,
153
+ },
157
154
  },
158
155
  },
159
- });
160
- }),
161
- };
162
- return ImageModel.make({ id: input.id, provider: "google", route, http: input.http });
163
- };
164
- const googleImagePart = (image) => {
165
- if (image.type === "bytes")
166
- return Effect.succeed({ inlineData: { mimeType: image.mediaType, data: Encoding.encodeBase64(image.data) } });
167
- if (image.type === "file-uri")
168
- return Effect.succeed({ fileData: { mimeType: image.mediaType, fileUri: image.uri } });
169
- if (image.type === "url")
170
- return ImageInputs.decodeDataUrl(image.url).pipe(Effect.flatMap((decoded) => {
171
- if (decoded === undefined)
172
- return Effect.fail(ImageInputs.invalid("Google generateContent does not fetch public image URLs; use bytes, a data URL, or a Gemini file URI"));
173
- return Effect.succeed({
174
- inlineData: { mimeType: decoded.mediaType, data: Encoding.encodeBase64(decoded.data) },
175
- });
176
- }));
177
- return Effect.fail(ImageInputs.invalid("Google generateContent requires Gemini file URIs rather than provider file IDs"));
178
- };
156
+ ]),
157
+ ];
158
+ const usage = decoded.usageMetadata;
159
+ const outputTokens = usage?.candidatesTokenCount === undefined ? undefined : usage.candidatesTokenCount + (usage.thoughtsTokenCount ?? 0);
160
+ return new ImageResponse({
161
+ images,
162
+ notices: notices.length === 0 ? undefined : notices,
163
+ usage: usage === undefined
164
+ ? undefined
165
+ : {
166
+ type: "tokens",
167
+ input: usage.promptTokenCount,
168
+ output: outputTokens,
169
+ total: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
170
+ details: {
171
+ reasoningTokens: usage.thoughtsTokenCount,
172
+ cacheReadInputTokens: usage.cachedContentTokenCount,
173
+ google: usage,
174
+ },
175
+ },
176
+ providerMetadata: {
177
+ google: {
178
+ modelVersion: decoded.modelVersion,
179
+ responseId: decoded.responseId,
180
+ promptFeedback: decoded.promptFeedback,
181
+ candidates: candidateMetadata,
182
+ },
183
+ },
184
+ });
185
+ });
186
+ // ---------------------------------------------------------------------------
187
+ // 7. Protocol and route
188
+ // ---------------------------------------------------------------------------
189
+ export const protocol = MediaProtocol.inline({
190
+ id: ADAPTER,
191
+ name: NAME,
192
+ unsupported: ["mask", "size", "format"],
193
+ body: { from: fromRequest },
194
+ response: { decode: decodeResponse },
195
+ });
196
+ export const model = (input) => ImageModel.fromRoute({
197
+ id: ADAPTER,
198
+ provider: PROVIDER,
199
+ protocol,
200
+ baseURL: DEFAULT_BASE_URL,
201
+ path: ({ request }) => `/models/${request.model.id}:generateContent`,
202
+ }, input);
179
203
  export const GoogleImages = {
204
+ protocol,
180
205
  model,
181
206
  };
@@ -0,0 +1,130 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { SpeechModel, type SpeechRequestFor } from "../speech.js";
4
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
5
+ import { SpeechStream } from "./utils/speech-stream.js";
6
+ export declare const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
7
+ /** Style is directed in the text itself, and `speechConfig.multiSpeakerVoiceConfig` excludes `voice`. */
8
+ export type GoogleSpeechOptions = {
9
+ readonly temperature?: number;
10
+ readonly seed?: number;
11
+ readonly speechConfig?: {
12
+ readonly multiSpeakerVoiceConfig?: {
13
+ readonly speakerVoiceConfigs: ReadonlyArray<{
14
+ readonly speaker: string;
15
+ readonly voiceConfig: {
16
+ readonly prebuiltVoiceConfig: {
17
+ readonly voiceName: string;
18
+ };
19
+ };
20
+ }>;
21
+ };
22
+ };
23
+ } & Record<string, unknown>;
24
+ export type Request = SpeechRequestFor<GoogleSpeechOptions>;
25
+ interface State extends SpeechStream.Audio, GeminiGenerateContent.Metadata {
26
+ readonly mimeType?: string;
27
+ }
28
+ export declare const protocol: MediaProtocol.Streamed<Request, {
29
+ readonly type: "audio-delta";
30
+ readonly chunk: Uint8Array<ArrayBufferLike>;
31
+ } | {
32
+ readonly type: "timestamps";
33
+ readonly items: readonly {
34
+ readonly text: string;
35
+ readonly startSeconds: number;
36
+ readonly endSeconds: number;
37
+ }[];
38
+ } | {
39
+ readonly type: "finish";
40
+ readonly audio: import("../media.js").Asset;
41
+ readonly providerMetadata?: {
42
+ readonly [x: string]: {
43
+ readonly [x: string]: unknown;
44
+ };
45
+ } | undefined;
46
+ readonly usage?: {
47
+ readonly type: "tokens";
48
+ readonly input?: number | undefined;
49
+ readonly output?: number | undefined;
50
+ readonly total?: number | undefined;
51
+ readonly details?: {
52
+ readonly [x: string]: unknown;
53
+ } | undefined;
54
+ } | {
55
+ readonly type: "seconds";
56
+ readonly seconds: number;
57
+ } | {
58
+ readonly type: "characters";
59
+ readonly characters: number;
60
+ } | {
61
+ readonly type: "credits";
62
+ readonly credits: number;
63
+ } | {
64
+ readonly type: "compute";
65
+ readonly seconds: number;
66
+ } | undefined;
67
+ readonly notices?: readonly {
68
+ readonly type: "other" | "moderated" | "filtered";
69
+ readonly message: string;
70
+ readonly providerMetadata?: {
71
+ readonly [x: string]: {
72
+ readonly [x: string]: unknown;
73
+ };
74
+ } | undefined;
75
+ }[] | undefined;
76
+ }, string, State>;
77
+ export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
78
+ export declare const GoogleSpeech: {
79
+ readonly protocol: MediaProtocol.Streamed<Request, {
80
+ readonly type: "audio-delta";
81
+ readonly chunk: Uint8Array<ArrayBufferLike>;
82
+ } | {
83
+ readonly type: "timestamps";
84
+ readonly items: readonly {
85
+ readonly text: string;
86
+ readonly startSeconds: number;
87
+ readonly endSeconds: number;
88
+ }[];
89
+ } | {
90
+ readonly type: "finish";
91
+ readonly audio: import("../media.js").Asset;
92
+ readonly providerMetadata?: {
93
+ readonly [x: string]: {
94
+ readonly [x: string]: unknown;
95
+ };
96
+ } | undefined;
97
+ readonly usage?: {
98
+ readonly type: "tokens";
99
+ readonly input?: number | undefined;
100
+ readonly output?: number | undefined;
101
+ readonly total?: number | undefined;
102
+ readonly details?: {
103
+ readonly [x: string]: unknown;
104
+ } | undefined;
105
+ } | {
106
+ readonly type: "seconds";
107
+ readonly seconds: number;
108
+ } | {
109
+ readonly type: "characters";
110
+ readonly characters: number;
111
+ } | {
112
+ readonly type: "credits";
113
+ readonly credits: number;
114
+ } | {
115
+ readonly type: "compute";
116
+ readonly seconds: number;
117
+ } | undefined;
118
+ readonly notices?: readonly {
119
+ readonly type: "other" | "moderated" | "filtered";
120
+ readonly message: string;
121
+ readonly providerMetadata?: {
122
+ readonly [x: string]: {
123
+ readonly [x: string]: unknown;
124
+ };
125
+ } | undefined;
126
+ }[] | undefined;
127
+ }, string, State>;
128
+ readonly model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
129
+ };
130
+ export {};
@@ -0,0 +1,84 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { ProviderID, mergeJsonRecords } from "../schema/index.js";
5
+ import { SpeechModel } from "../speech.js";
6
+ import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
7
+ import { SpeechStream } from "./utils/speech-stream.js";
8
+ const ADAPTER = "google-speech";
9
+ const NAME = "Google Speech";
10
+ const PROVIDER = ProviderID.make("google");
11
+ export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
12
+ const DEFAULT_SAMPLE_RATE = 24000;
13
+ // ---------------------------------------------------------------------------
14
+ // 3. Streaming event schema
15
+ // ---------------------------------------------------------------------------
16
+ const GenerateContentChunk = GeminiGenerateContent.chunk(Schema.Struct({
17
+ text: Schema.optional(Schema.String),
18
+ inlineData: Schema.optional(Schema.Struct({ mimeType: Schema.String, data: Schema.Uint8ArrayFromBase64 })),
19
+ }));
20
+ const decodeChunk = MediaProtocol.decodeFrame(ADAPTER, NAME, GenerateContentChunk);
21
+ // ---------------------------------------------------------------------------
22
+ // 5. Request body construction
23
+ // ---------------------------------------------------------------------------
24
+ const fromRequest = Effect.fn("GoogleSpeech.fromRequest")(function* (request) {
25
+ if (request.format !== undefined && request.format !== "pcm")
26
+ return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} only returns raw PCM; request format "pcm" or omit it, then wrap the samples yourself`);
27
+ const voiceName = SpeechStream.voiceID(request.voice);
28
+ return MediaProtocol.json(mergeJsonRecords({
29
+ contents: [{ role: "user", parts: [{ text: request.text }] }],
30
+ generationConfig: mergeJsonRecords({
31
+ responseModalities: ["AUDIO"],
32
+ speechConfig: {
33
+ voiceConfig: voiceName === undefined ? undefined : { prebuiltVoiceConfig: { voiceName } },
34
+ languageCode: request.language,
35
+ },
36
+ }, request.providerOptions),
37
+ }, request.http?.body) ?? {});
38
+ });
39
+ // ---------------------------------------------------------------------------
40
+ // 6. Stream parsing
41
+ // ---------------------------------------------------------------------------
42
+ const step = Effect.fn("GoogleSpeech.step")(function* (state, frame) {
43
+ const chunk = yield* decodeChunk(frame);
44
+ const blocked = GeminiGenerateContent.blocked(NAME, chunk, frame);
45
+ if (blocked !== undefined)
46
+ return yield* blocked;
47
+ const audio = (chunk.candidates?.[0]?.content?.parts ?? []).flatMap((part) => part.inlineData === undefined ? [] : [part.inlineData]);
48
+ const next = { ...GeminiGenerateContent.track(state, chunk), mimeType: state.mimeType ?? audio[0]?.mimeType };
49
+ return [next, audio.flatMap((part) => SpeechStream.delta(next, part.data)[1])];
50
+ });
51
+ const finish = (state) => {
52
+ const sampleRate = SpeechStream.sampleRate(state.mimeType) ?? DEFAULT_SAMPLE_RATE;
53
+ return SpeechStream.finish(ADAPTER, state, {
54
+ ...SpeechStream.pcm("pcm_s16le", sampleRate, state.mimeType ?? `audio/L16;codec=pcm;rate=${sampleRate}`),
55
+ usage: GeminiGenerateContent.usage(state.usage),
56
+ providerMetadata: GeminiGenerateContent.providerMetadata(state),
57
+ detail: state.finishReason === undefined ? undefined : `finish reason: ${state.finishReason}`,
58
+ });
59
+ };
60
+ // ---------------------------------------------------------------------------
61
+ // 7. Protocol and route
62
+ // ---------------------------------------------------------------------------
63
+ export const protocol = MediaProtocol.stream({
64
+ id: ADAPTER,
65
+ name: NAME,
66
+ unsupported: ["instructions", "speed", "timestamps"],
67
+ body: { from: fromRequest },
68
+ frames: (bytes, context) => GeminiGenerateContent.frames(bytes, context.request.mode),
69
+ initial: () => ({ chunks: [] }),
70
+ step,
71
+ finish,
72
+ });
73
+ export const model = (input) => SpeechModel.fromRoute({
74
+ id: ADAPTER,
75
+ provider: PROVIDER,
76
+ protocol,
77
+ baseURL: DEFAULT_BASE_URL,
78
+ // Only `gemini-3.1-flash-tts-preview` and later stream; earlier TTS models reject `streamGenerateContent`.
79
+ path: ({ request }) => GeminiGenerateContent.path(request.model.id, request.mode),
80
+ }, input);
81
+ export const GoogleSpeech = {
82
+ protocol,
83
+ model,
84
+ };