@opencode/ai 2.0.14 → 2.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +399 -56
- package/dist/experimental/evaluation-client.d.ts +1 -1
- package/dist/experimental/evaluation-client.js +39 -3
- package/dist/experimental/evaluation.d.ts +4 -4
- package/dist/experimental/evaluation.js +2 -2
- package/dist/experimental/system-one.d.ts +3 -3
- package/dist/experimental/system-one.js +40 -51
- package/dist/generation.d.ts +83 -0
- package/dist/generation.js +113 -0
- package/dist/image-client.d.ts +16 -7
- package/dist/image-client.js +29 -13
- package/dist/image.d.ts +1410 -81
- package/dist/image.js +97 -67
- package/dist/index.d.ts +17 -2
- package/dist/index.js +12 -1
- package/dist/llm.d.ts +9 -1
- package/dist/media-model.d.ts +44 -0
- package/dist/media-model.js +49 -0
- package/dist/media.d.ts +213 -0
- package/dist/media.js +227 -0
- package/dist/promise.d.ts +974 -0
- package/dist/promise.js +81 -0
- package/dist/protocols/alibaba-chat.d.ts +12 -0
- package/dist/protocols/alibaba-responses.d.ts +2 -2
- package/dist/protocols/anthropic-messages.js +8 -19
- package/dist/protocols/assemblyai-transcription.d.ts +40 -0
- package/dist/protocols/assemblyai-transcription.js +138 -0
- package/dist/protocols/bedrock-converse.d.ts +4 -4
- package/dist/protocols/bedrock-converse.js +6 -17
- package/dist/protocols/bfl-images.d.ts +32 -0
- package/dist/protocols/bfl-images.js +153 -0
- package/dist/protocols/cartesia-speech.d.ts +127 -0
- package/dist/protocols/cartesia-speech.js +126 -0
- package/dist/protocols/deepgram-speech.d.ts +119 -0
- package/dist/protocols/deepgram-speech.js +92 -0
- package/dist/protocols/deepgram-transcription.d.ts +25 -0
- package/dist/protocols/deepgram-transcription.js +129 -0
- package/dist/protocols/elevenlabs-speech.d.ts +122 -0
- package/dist/protocols/elevenlabs-speech.js +115 -0
- package/dist/protocols/fal-images.d.ts +24 -0
- package/dist/protocols/fal-images.js +114 -0
- package/dist/protocols/fal-video.d.ts +29 -0
- package/dist/protocols/fal-video.js +88 -0
- package/dist/protocols/gemini.d.ts +30 -9
- package/dist/protocols/gemini.js +45 -35
- package/dist/protocols/google-images.d.ts +9 -21
- package/dist/protocols/google-images.js +158 -133
- package/dist/protocols/google-speech.d.ts +130 -0
- package/dist/protocols/google-speech.js +84 -0
- package/dist/protocols/google-transcription.d.ts +173 -0
- package/dist/protocols/google-transcription.js +138 -0
- package/dist/protocols/google-video.d.ts +26 -0
- package/dist/protocols/google-video.js +158 -0
- package/dist/protocols/meta-images.d.ts +7 -12
- package/dist/protocols/meta-images.js +85 -66
- package/dist/protocols/meta-responses.d.ts +4 -4
- package/dist/protocols/meta-responses.js +1 -1
- package/dist/protocols/mistral-chat.js +7 -6
- package/dist/protocols/open-responses.d.ts +17 -9
- package/dist/protocols/open-responses.js +24 -14
- package/dist/protocols/openai-chat.d.ts +118 -1
- package/dist/protocols/openai-chat.js +125 -44
- package/dist/protocols/openai-compatible-chat.d.ts +12 -0
- package/dist/protocols/openai-compatible-responses.d.ts +2 -2
- package/dist/protocols/openai-images.d.ts +128 -18
- package/dist/protocols/openai-images.js +177 -154
- package/dist/protocols/openai-responses.d.ts +15 -15
- package/dist/protocols/openai-responses.js +5 -6
- package/dist/protocols/openai-speech.d.ts +116 -0
- package/dist/protocols/openai-speech.js +98 -0
- package/dist/protocols/openai-transcription.d.ts +207 -0
- package/dist/protocols/openai-transcription.js +190 -0
- package/dist/protocols/replicate-images.d.ts +28 -0
- package/dist/protocols/replicate-images.js +133 -0
- package/dist/protocols/runway-video.d.ts +38 -0
- package/dist/protocols/runway-video.js +146 -0
- package/dist/protocols/shared.d.ts +27 -17
- package/dist/protocols/shared.js +52 -35
- package/dist/protocols/stability-images.d.ts +38 -0
- package/dist/protocols/stability-images.js +148 -0
- package/dist/protocols/utils/bedrock-media.d.ts +2 -3
- package/dist/protocols/utils/bedrock-media.js +4 -4
- package/dist/protocols/utils/fal-queue.d.ts +28 -0
- package/dist/protocols/utils/fal-queue.js +69 -0
- package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
- package/dist/protocols/utils/gemini-generate-content.js +65 -0
- package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
- package/dist/protocols/utils/gemini-json-schema.js +76 -0
- package/dist/protocols/utils/media-input.d.ts +18 -0
- package/dist/protocols/utils/media-input.js +35 -0
- package/dist/protocols/utils/responses-compaction.js +6 -5
- package/dist/protocols/utils/speech-stream.d.ts +49 -0
- package/dist/protocols/utils/speech-stream.js +67 -0
- package/dist/protocols/utils/tool-schema.d.ts +2 -2
- package/dist/protocols/utils/tool-schema.js +40 -17
- package/dist/protocols/utils/tool-stream.d.ts +27 -3
- package/dist/protocols/xai-images.d.ts +9 -15
- package/dist/protocols/xai-images.js +75 -84
- package/dist/protocols/xai-responses.d.ts +2 -2
- package/dist/protocols/xai-video.d.ts +34 -0
- package/dist/protocols/xai-video.js +147 -0
- package/dist/protocols/zai-chat.d.ts +13 -1
- package/dist/protocols/zai-images.d.ts +9 -13
- package/dist/protocols/zai-images.js +59 -57
- package/dist/provider-error.js +3 -0
- package/dist/providers/alibaba.d.ts +14 -2
- package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
- package/dist/providers/amazon-bedrock.d.ts +2 -2
- package/dist/providers/assemblyai.d.ts +25 -0
- package/dist/providers/assemblyai.js +29 -0
- package/dist/providers/azure.d.ts +18 -6
- package/dist/providers/baseten.d.ts +24 -0
- package/dist/providers/black-forest-labs.d.ts +25 -0
- package/dist/providers/black-forest-labs.js +28 -0
- package/dist/providers/cartesia.d.ts +24 -0
- package/dist/providers/cartesia.js +22 -0
- package/dist/providers/cerebras.d.ts +24 -0
- package/dist/providers/cerebras.js +6 -1
- package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
- package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
- package/dist/providers/deepgram.d.ts +29 -0
- package/dist/providers/deepgram.js +31 -0
- package/dist/providers/deepinfra.d.ts +24 -0
- package/dist/providers/deepinfra.js +6 -1
- package/dist/providers/deepseek.d.ts +24 -0
- package/dist/providers/elevenlabs.d.ts +24 -0
- package/dist/providers/elevenlabs.js +28 -0
- package/dist/providers/fal.d.ts +29 -0
- package/dist/providers/fal.js +33 -0
- package/dist/providers/fireworks.d.ts +24 -0
- package/dist/providers/google-vertex-chat.d.ts +12 -0
- package/dist/providers/google-vertex-responses.d.ts +2 -2
- package/dist/providers/google-vertex.d.ts +10 -3
- package/dist/providers/google.d.ts +25 -3
- package/dist/providers/google.js +11 -2
- package/dist/providers/groq.d.ts +24 -0
- package/dist/providers/index.d.ts +10 -0
- package/dist/providers/index.js +10 -0
- package/dist/providers/meta.d.ts +14 -2
- package/dist/providers/minimax.d.ts +14 -2
- package/dist/providers/moonshot.d.ts +14 -2
- package/dist/providers/moonshot.js +3 -3
- package/dist/providers/openai-compatible-responses.d.ts +2 -2
- package/dist/providers/openai-compatible.d.ts +12 -0
- package/dist/providers/openai.d.ts +25 -3
- package/dist/providers/openai.js +10 -1
- package/dist/providers/openrouter.d.ts +67 -0
- package/dist/providers/openrouter.js +13 -1
- package/dist/providers/replicate.d.ts +25 -0
- package/dist/providers/replicate.js +22 -0
- package/dist/providers/runway.d.ts +24 -0
- package/dist/providers/runway.js +22 -0
- package/dist/providers/stability.d.ts +28 -0
- package/dist/providers/stability.js +23 -0
- package/dist/providers/togetherai.d.ts +24 -0
- package/dist/providers/vercel-ai-gateway.d.ts +41 -0
- package/dist/providers/vercel-ai-gateway.js +85 -0
- package/dist/providers/xai.d.ts +17 -0
- package/dist/providers/xai.js +5 -2
- package/dist/providers/zai-coding-plan.d.ts +15 -3
- package/dist/providers/zai.d.ts +13 -1
- package/dist/route/auth.d.ts +4 -1
- package/dist/route/auth.js +6 -0
- package/dist/route/client.d.ts +9 -1
- package/dist/route/endpoint.d.ts +10 -10
- package/dist/route/executor-service.d.ts +12 -0
- package/dist/route/executor-service.js +3 -0
- package/dist/route/executor.d.ts +4 -9
- package/dist/route/executor.js +3 -3
- package/dist/route/framing.d.ts +5 -1
- package/dist/route/framing.js +9 -0
- package/dist/route/index.d.ts +2 -0
- package/dist/route/index.js +2 -0
- package/dist/route/media-protocol.d.ts +158 -0
- package/dist/route/media-protocol.js +96 -0
- package/dist/route/media.d.ts +97 -0
- package/dist/route/media.js +236 -0
- package/dist/schema/errors.d.ts +13 -3
- package/dist/schema/errors.js +7 -0
- package/dist/schema/events.d.ts +557 -40
- package/dist/schema/events.js +35 -2
- package/dist/schema/messages.d.ts +95 -8
- package/dist/schema/messages.js +8 -6
- package/dist/schema/options.d.ts +6 -3
- package/dist/schema/options.js +6 -2
- package/dist/speech-client.d.ts +21 -0
- package/dist/speech-client.js +25 -0
- package/dist/speech.d.ts +1150 -0
- package/dist/speech.js +119 -0
- package/dist/testing.d.ts +72 -8
- package/dist/transcription-client.d.ts +28 -0
- package/dist/transcription-client.js +44 -0
- package/dist/transcription.d.ts +1504 -0
- package/dist/transcription.js +133 -0
- package/dist/utils/bytes.d.ts +1 -0
- package/dist/utils/bytes.js +10 -0
- package/dist/utils/media-type.d.ts +7 -0
- package/dist/utils/media-type.js +70 -0
- package/dist/utils/sanitize.js +3 -1
- package/dist/video-client.d.ts +28 -0
- package/dist/video-client.js +40 -0
- package/dist/video.d.ts +1359 -0
- package/dist/video.js +119 -0
- package/package.json +7 -3
- package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
- package/dist/protocols/utils/gemini-tool-schema.js +0 -103
- package/dist/protocols/utils/image-input.d.ts +0 -21
- package/dist/protocols/utils/image-input.js +0 -20
|
@@ -1,12 +1,18 @@
|
|
|
1
|
-
import { Effect,
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { ImageModel, ImageResponse } from "../image.js";
|
|
3
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
4
|
+
import { MediaRoute } from "../route/media.js";
|
|
5
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
6
6
|
import { ProviderShared } from "./shared.js";
|
|
7
|
-
import {
|
|
7
|
+
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
|
|
8
|
+
import { MediaInput } from "./utils/media-input.js";
|
|
8
9
|
const ADAPTER = "google-images";
|
|
10
|
+
const NAME = "Google Images";
|
|
11
|
+
const PROVIDER = ProviderID.make("google");
|
|
9
12
|
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// 2. Response schema
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
10
16
|
const GoogleUsage = Schema.StructWithRest(Schema.Struct({
|
|
11
17
|
cachedContentTokenCount: Schema.optional(Schema.Number),
|
|
12
18
|
thoughtsTokenCount: Schema.optional(Schema.Number),
|
|
@@ -41,141 +47,160 @@ const GoogleImageResponse = Schema.Struct({
|
|
|
41
47
|
responseId: Schema.optional(Schema.String),
|
|
42
48
|
promptFeedback: Schema.optional(Schema.Unknown),
|
|
43
49
|
});
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
};
|
|
50
|
-
const thinkingConfig = {
|
|
51
|
-
thinkingLevel,
|
|
52
|
-
includeThoughts,
|
|
53
|
-
};
|
|
50
|
+
// ---------------------------------------------------------------------------
|
|
51
|
+
// 5. Request body construction
|
|
52
|
+
// ---------------------------------------------------------------------------
|
|
53
|
+
const generationConfig = (request) => {
|
|
54
|
+
const { imageSize, thinkingLevel, includeThoughts, ...native } = request.providerOptions ?? {};
|
|
55
|
+
const imageConfig = { aspectRatio: request.aspectRatio, imageSize };
|
|
56
|
+
const thinkingConfig = { thinkingLevel, includeThoughts };
|
|
54
57
|
return (mergeJsonRecords({
|
|
55
58
|
responseModalities: ["IMAGE"],
|
|
56
|
-
imageConfig: Object.values(
|
|
57
|
-
seed,
|
|
59
|
+
imageConfig: Object.values(imageConfig).some((value) => value !== undefined) ? imageConfig : undefined,
|
|
60
|
+
seed: request.seed,
|
|
58
61
|
thinkingConfig: Object.values(thinkingConfig).some((value) => value !== undefined) ? thinkingConfig : undefined,
|
|
59
62
|
}, native) ?? { responseModalities: ["IMAGE"] });
|
|
60
63
|
};
|
|
61
|
-
const
|
|
62
|
-
if (
|
|
63
|
-
return
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
};
|
|
68
|
-
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
:
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
64
|
+
const fromRequest = Effect.fn("GoogleImages.fromRequest")(function* (request) {
|
|
65
|
+
if (request.n !== undefined && request.n > 1)
|
|
66
|
+
return yield* ProviderShared.unsupportedOperation({
|
|
67
|
+
operation: "image.n",
|
|
68
|
+
provider: PROVIDER,
|
|
69
|
+
route: ADAPTER,
|
|
70
|
+
message: `${NAME} generates one image per request; call it once per image instead of n=${request.n}`,
|
|
71
|
+
});
|
|
72
|
+
const parts = yield* Effect.forEach(request.images ?? [], (image) => GeminiGenerateContent.mediaPart(NAME, image));
|
|
73
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
74
|
+
contents: [{ role: "user", parts: [{ text: request.prompt }, ...parts] }],
|
|
75
|
+
generationConfig: generationConfig(request),
|
|
76
|
+
}, request.http?.body) ?? {});
|
|
77
|
+
});
|
|
78
|
+
// ---------------------------------------------------------------------------
|
|
79
|
+
// 6. Response decoding
|
|
80
|
+
// ---------------------------------------------------------------------------
|
|
81
|
+
const decodeResponse = Effect.fn("GoogleImages.decodeResponse")(function* (response) {
|
|
82
|
+
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, GoogleImageResponse)(response);
|
|
83
|
+
const decoded = output.value;
|
|
84
|
+
const candidates = decoded.candidates ?? [];
|
|
85
|
+
const candidateMetadata = candidates.map((candidate, candidateIndex) => ({
|
|
86
|
+
index: candidate.index ?? candidateIndex,
|
|
87
|
+
finishReason: candidate.finishReason,
|
|
88
|
+
finishMessage: candidate.finishMessage,
|
|
89
|
+
safetyRatings: candidate.safetyRatings,
|
|
90
|
+
citationMetadata: candidate.citationMetadata,
|
|
91
|
+
groundingMetadata: candidate.groundingMetadata,
|
|
92
|
+
parts: (candidate.content?.parts ?? []).map((part) => part.inlineData === undefined
|
|
93
|
+
? { type: "text", text: part.text, thought: part.thought, thoughtSignature: part.thoughtSignature }
|
|
94
|
+
: {
|
|
95
|
+
type: "inlineData",
|
|
96
|
+
mediaType: part.inlineData.mimeType,
|
|
97
|
+
thought: part.thought,
|
|
98
|
+
thoughtSignature: part.thoughtSignature,
|
|
99
|
+
}),
|
|
100
|
+
}));
|
|
101
|
+
// Thought parts are drafts; only non-thought inline data is a final image.
|
|
102
|
+
const encoded = candidates.flatMap((candidate, candidateIndex) => (candidate.content?.parts ?? []).flatMap((part, partIndex) => part.inlineData === undefined || part.thought === true
|
|
103
|
+
? []
|
|
104
|
+
: [
|
|
105
|
+
{
|
|
106
|
+
candidate,
|
|
107
|
+
candidateIndex,
|
|
108
|
+
partIndex,
|
|
109
|
+
inlineData: part.inlineData,
|
|
110
|
+
thoughtSignature: part.thoughtSignature,
|
|
111
|
+
},
|
|
112
|
+
]));
|
|
113
|
+
const images = yield* Effect.forEach(encoded, (item) => MediaInput.decodedAsset(output.invalid, `${NAME} candidate ${item.candidateIndex} part ${item.partIndex}`, item.inlineData.data, item.inlineData.mimeType, {
|
|
114
|
+
providerMetadata: {
|
|
115
|
+
google: {
|
|
116
|
+
candidateIndex: item.candidate.index ?? item.candidateIndex,
|
|
117
|
+
partIndex: item.partIndex,
|
|
118
|
+
finishReason: item.candidate.finishReason,
|
|
119
|
+
safetyRatings: item.candidate.safetyRatings,
|
|
120
|
+
citationMetadata: item.candidate.citationMetadata,
|
|
121
|
+
groundingMetadata: item.candidate.groundingMetadata,
|
|
122
|
+
thoughtSignature: item.thoughtSignature,
|
|
123
|
+
},
|
|
124
|
+
},
|
|
125
|
+
}));
|
|
126
|
+
if (images.length === 0) {
|
|
127
|
+
const finishReasons = candidates.flatMap((candidate) => candidate.finishReason === undefined ? [] : [candidate.finishReason]);
|
|
128
|
+
return yield* output.invalid(`${NAME} returned no final images${finishReasons.length === 0 ? "" : ` (finish reasons: ${finishReasons.join(", ")})`}; inspect body for prompt feedback and candidate details`);
|
|
129
|
+
}
|
|
130
|
+
// Candidates that stopped for a safety or policy reason are partial results, not a silent drop.
|
|
131
|
+
const notices = [
|
|
132
|
+
...(decoded.promptFeedback === undefined
|
|
133
|
+
? []
|
|
134
|
+
: [
|
|
135
|
+
{
|
|
136
|
+
type: "filtered",
|
|
137
|
+
message: `${NAME} reported prompt feedback`,
|
|
138
|
+
providerMetadata: { google: { promptFeedback: decoded.promptFeedback } },
|
|
128
139
|
},
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
inputTokens: usage.promptTokenCount,
|
|
144
|
-
outputTokens,
|
|
145
|
-
nonCachedInputTokens: ProviderShared.subtractTokens(usage.promptTokenCount, usage.cachedContentTokenCount),
|
|
146
|
-
cacheReadInputTokens: usage.cachedContentTokenCount,
|
|
147
|
-
reasoningTokens: usage.thoughtsTokenCount,
|
|
148
|
-
totalTokens: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
|
|
149
|
-
providerMetadata: { google: usage },
|
|
150
|
-
}),
|
|
151
|
-
providerMetadata: {
|
|
152
|
-
google: {
|
|
153
|
-
modelVersion: decoded.modelVersion,
|
|
154
|
-
responseId: decoded.responseId,
|
|
155
|
-
promptFeedback: decoded.promptFeedback,
|
|
156
|
-
candidates: candidateMetadata,
|
|
140
|
+
]),
|
|
141
|
+
...candidates.flatMap((candidate, index) => candidate.finishReason === undefined || candidate.finishReason === "STOP"
|
|
142
|
+
? []
|
|
143
|
+
: [
|
|
144
|
+
{
|
|
145
|
+
type: "filtered",
|
|
146
|
+
message: `${NAME} candidate ${candidate.index ?? index} finished with ${candidate.finishReason}${candidate.finishMessage === undefined ? "" : `: ${candidate.finishMessage}`}`,
|
|
147
|
+
providerMetadata: {
|
|
148
|
+
google: {
|
|
149
|
+
candidateIndex: candidate.index ?? index,
|
|
150
|
+
finishReason: candidate.finishReason,
|
|
151
|
+
finishMessage: candidate.finishMessage,
|
|
152
|
+
safetyRatings: candidate.safetyRatings,
|
|
153
|
+
},
|
|
157
154
|
},
|
|
158
155
|
},
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
}
|
|
156
|
+
]),
|
|
157
|
+
];
|
|
158
|
+
const usage = decoded.usageMetadata;
|
|
159
|
+
const outputTokens = usage?.candidatesTokenCount === undefined ? undefined : usage.candidatesTokenCount + (usage.thoughtsTokenCount ?? 0);
|
|
160
|
+
return new ImageResponse({
|
|
161
|
+
images,
|
|
162
|
+
notices: notices.length === 0 ? undefined : notices,
|
|
163
|
+
usage: usage === undefined
|
|
164
|
+
? undefined
|
|
165
|
+
: {
|
|
166
|
+
type: "tokens",
|
|
167
|
+
input: usage.promptTokenCount,
|
|
168
|
+
output: outputTokens,
|
|
169
|
+
total: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
|
|
170
|
+
details: {
|
|
171
|
+
reasoningTokens: usage.thoughtsTokenCount,
|
|
172
|
+
cacheReadInputTokens: usage.cachedContentTokenCount,
|
|
173
|
+
google: usage,
|
|
174
|
+
},
|
|
175
|
+
},
|
|
176
|
+
providerMetadata: {
|
|
177
|
+
google: {
|
|
178
|
+
modelVersion: decoded.modelVersion,
|
|
179
|
+
responseId: decoded.responseId,
|
|
180
|
+
promptFeedback: decoded.promptFeedback,
|
|
181
|
+
candidates: candidateMetadata,
|
|
182
|
+
},
|
|
183
|
+
},
|
|
184
|
+
});
|
|
185
|
+
});
|
|
186
|
+
// ---------------------------------------------------------------------------
|
|
187
|
+
// 7. Protocol and route
|
|
188
|
+
// ---------------------------------------------------------------------------
|
|
189
|
+
export const protocol = MediaProtocol.inline({
|
|
190
|
+
id: ADAPTER,
|
|
191
|
+
name: NAME,
|
|
192
|
+
unsupported: ["mask", "size", "format"],
|
|
193
|
+
body: { from: fromRequest },
|
|
194
|
+
response: { decode: decodeResponse },
|
|
195
|
+
});
|
|
196
|
+
export const model = (input) => ImageModel.fromRoute({
|
|
197
|
+
id: ADAPTER,
|
|
198
|
+
provider: PROVIDER,
|
|
199
|
+
protocol,
|
|
200
|
+
baseURL: DEFAULT_BASE_URL,
|
|
201
|
+
path: ({ request }) => `/models/${request.model.id}:generateContent`,
|
|
202
|
+
}, input);
|
|
179
203
|
export const GoogleImages = {
|
|
204
|
+
protocol,
|
|
180
205
|
model,
|
|
181
206
|
};
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
2
|
+
import { MediaRoute } from "../route/media.js";
|
|
3
|
+
import { SpeechModel, type SpeechRequestFor } from "../speech.js";
|
|
4
|
+
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
|
|
5
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
6
|
+
export declare const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
|
|
7
|
+
/** Style is directed in the text itself, and `speechConfig.multiSpeakerVoiceConfig` excludes `voice`. */
|
|
8
|
+
export type GoogleSpeechOptions = {
|
|
9
|
+
readonly temperature?: number;
|
|
10
|
+
readonly seed?: number;
|
|
11
|
+
readonly speechConfig?: {
|
|
12
|
+
readonly multiSpeakerVoiceConfig?: {
|
|
13
|
+
readonly speakerVoiceConfigs: ReadonlyArray<{
|
|
14
|
+
readonly speaker: string;
|
|
15
|
+
readonly voiceConfig: {
|
|
16
|
+
readonly prebuiltVoiceConfig: {
|
|
17
|
+
readonly voiceName: string;
|
|
18
|
+
};
|
|
19
|
+
};
|
|
20
|
+
}>;
|
|
21
|
+
};
|
|
22
|
+
};
|
|
23
|
+
} & Record<string, unknown>;
|
|
24
|
+
export type Request = SpeechRequestFor<GoogleSpeechOptions>;
|
|
25
|
+
interface State extends SpeechStream.Audio, GeminiGenerateContent.Metadata {
|
|
26
|
+
readonly mimeType?: string;
|
|
27
|
+
}
|
|
28
|
+
export declare const protocol: MediaProtocol.Streamed<Request, {
|
|
29
|
+
readonly type: "audio-delta";
|
|
30
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
31
|
+
} | {
|
|
32
|
+
readonly type: "timestamps";
|
|
33
|
+
readonly items: readonly {
|
|
34
|
+
readonly text: string;
|
|
35
|
+
readonly startSeconds: number;
|
|
36
|
+
readonly endSeconds: number;
|
|
37
|
+
}[];
|
|
38
|
+
} | {
|
|
39
|
+
readonly type: "finish";
|
|
40
|
+
readonly audio: import("../media.js").Asset;
|
|
41
|
+
readonly providerMetadata?: {
|
|
42
|
+
readonly [x: string]: {
|
|
43
|
+
readonly [x: string]: unknown;
|
|
44
|
+
};
|
|
45
|
+
} | undefined;
|
|
46
|
+
readonly usage?: {
|
|
47
|
+
readonly type: "tokens";
|
|
48
|
+
readonly input?: number | undefined;
|
|
49
|
+
readonly output?: number | undefined;
|
|
50
|
+
readonly total?: number | undefined;
|
|
51
|
+
readonly details?: {
|
|
52
|
+
readonly [x: string]: unknown;
|
|
53
|
+
} | undefined;
|
|
54
|
+
} | {
|
|
55
|
+
readonly type: "seconds";
|
|
56
|
+
readonly seconds: number;
|
|
57
|
+
} | {
|
|
58
|
+
readonly type: "characters";
|
|
59
|
+
readonly characters: number;
|
|
60
|
+
} | {
|
|
61
|
+
readonly type: "credits";
|
|
62
|
+
readonly credits: number;
|
|
63
|
+
} | {
|
|
64
|
+
readonly type: "compute";
|
|
65
|
+
readonly seconds: number;
|
|
66
|
+
} | undefined;
|
|
67
|
+
readonly notices?: readonly {
|
|
68
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
69
|
+
readonly message: string;
|
|
70
|
+
readonly providerMetadata?: {
|
|
71
|
+
readonly [x: string]: {
|
|
72
|
+
readonly [x: string]: unknown;
|
|
73
|
+
};
|
|
74
|
+
} | undefined;
|
|
75
|
+
}[] | undefined;
|
|
76
|
+
}, string, State>;
|
|
77
|
+
export declare const model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
|
|
78
|
+
export declare const GoogleSpeech: {
|
|
79
|
+
readonly protocol: MediaProtocol.Streamed<Request, {
|
|
80
|
+
readonly type: "audio-delta";
|
|
81
|
+
readonly chunk: Uint8Array<ArrayBufferLike>;
|
|
82
|
+
} | {
|
|
83
|
+
readonly type: "timestamps";
|
|
84
|
+
readonly items: readonly {
|
|
85
|
+
readonly text: string;
|
|
86
|
+
readonly startSeconds: number;
|
|
87
|
+
readonly endSeconds: number;
|
|
88
|
+
}[];
|
|
89
|
+
} | {
|
|
90
|
+
readonly type: "finish";
|
|
91
|
+
readonly audio: import("../media.js").Asset;
|
|
92
|
+
readonly providerMetadata?: {
|
|
93
|
+
readonly [x: string]: {
|
|
94
|
+
readonly [x: string]: unknown;
|
|
95
|
+
};
|
|
96
|
+
} | undefined;
|
|
97
|
+
readonly usage?: {
|
|
98
|
+
readonly type: "tokens";
|
|
99
|
+
readonly input?: number | undefined;
|
|
100
|
+
readonly output?: number | undefined;
|
|
101
|
+
readonly total?: number | undefined;
|
|
102
|
+
readonly details?: {
|
|
103
|
+
readonly [x: string]: unknown;
|
|
104
|
+
} | undefined;
|
|
105
|
+
} | {
|
|
106
|
+
readonly type: "seconds";
|
|
107
|
+
readonly seconds: number;
|
|
108
|
+
} | {
|
|
109
|
+
readonly type: "characters";
|
|
110
|
+
readonly characters: number;
|
|
111
|
+
} | {
|
|
112
|
+
readonly type: "credits";
|
|
113
|
+
readonly credits: number;
|
|
114
|
+
} | {
|
|
115
|
+
readonly type: "compute";
|
|
116
|
+
readonly seconds: number;
|
|
117
|
+
} | undefined;
|
|
118
|
+
readonly notices?: readonly {
|
|
119
|
+
readonly type: "other" | "moderated" | "filtered";
|
|
120
|
+
readonly message: string;
|
|
121
|
+
readonly providerMetadata?: {
|
|
122
|
+
readonly [x: string]: {
|
|
123
|
+
readonly [x: string]: unknown;
|
|
124
|
+
};
|
|
125
|
+
} | undefined;
|
|
126
|
+
}[] | undefined;
|
|
127
|
+
}, string, State>;
|
|
128
|
+
readonly model: (input: MediaRoute.ModelInput) => SpeechModel<GoogleSpeechOptions>;
|
|
129
|
+
};
|
|
130
|
+
export {};
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import { Effect, Schema } from "effect";
|
|
2
|
+
import { MediaProtocol } from "../route/media-protocol.js";
|
|
3
|
+
import { MediaRoute } from "../route/media.js";
|
|
4
|
+
import { ProviderID, mergeJsonRecords } from "../schema/index.js";
|
|
5
|
+
import { SpeechModel } from "../speech.js";
|
|
6
|
+
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js";
|
|
7
|
+
import { SpeechStream } from "./utils/speech-stream.js";
|
|
8
|
+
const ADAPTER = "google-speech";
|
|
9
|
+
const NAME = "Google Speech";
|
|
10
|
+
const PROVIDER = ProviderID.make("google");
|
|
11
|
+
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
|
|
12
|
+
const DEFAULT_SAMPLE_RATE = 24000;
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// 3. Streaming event schema
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
const GenerateContentChunk = GeminiGenerateContent.chunk(Schema.Struct({
|
|
17
|
+
text: Schema.optional(Schema.String),
|
|
18
|
+
inlineData: Schema.optional(Schema.Struct({ mimeType: Schema.String, data: Schema.Uint8ArrayFromBase64 })),
|
|
19
|
+
}));
|
|
20
|
+
const decodeChunk = MediaProtocol.decodeFrame(ADAPTER, NAME, GenerateContentChunk);
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
// 5. Request body construction
|
|
23
|
+
// ---------------------------------------------------------------------------
|
|
24
|
+
const fromRequest = Effect.fn("GoogleSpeech.fromRequest")(function* (request) {
|
|
25
|
+
if (request.format !== undefined && request.format !== "pcm")
|
|
26
|
+
return yield* SpeechStream.unsupportedFormat(PROVIDER, ADAPTER, `${NAME} only returns raw PCM; request format "pcm" or omit it, then wrap the samples yourself`);
|
|
27
|
+
const voiceName = SpeechStream.voiceID(request.voice);
|
|
28
|
+
return MediaProtocol.json(mergeJsonRecords({
|
|
29
|
+
contents: [{ role: "user", parts: [{ text: request.text }] }],
|
|
30
|
+
generationConfig: mergeJsonRecords({
|
|
31
|
+
responseModalities: ["AUDIO"],
|
|
32
|
+
speechConfig: {
|
|
33
|
+
voiceConfig: voiceName === undefined ? undefined : { prebuiltVoiceConfig: { voiceName } },
|
|
34
|
+
languageCode: request.language,
|
|
35
|
+
},
|
|
36
|
+
}, request.providerOptions),
|
|
37
|
+
}, request.http?.body) ?? {});
|
|
38
|
+
});
|
|
39
|
+
// ---------------------------------------------------------------------------
|
|
40
|
+
// 6. Stream parsing
|
|
41
|
+
// ---------------------------------------------------------------------------
|
|
42
|
+
const step = Effect.fn("GoogleSpeech.step")(function* (state, frame) {
|
|
43
|
+
const chunk = yield* decodeChunk(frame);
|
|
44
|
+
const blocked = GeminiGenerateContent.blocked(NAME, chunk, frame);
|
|
45
|
+
if (blocked !== undefined)
|
|
46
|
+
return yield* blocked;
|
|
47
|
+
const audio = (chunk.candidates?.[0]?.content?.parts ?? []).flatMap((part) => part.inlineData === undefined ? [] : [part.inlineData]);
|
|
48
|
+
const next = { ...GeminiGenerateContent.track(state, chunk), mimeType: state.mimeType ?? audio[0]?.mimeType };
|
|
49
|
+
return [next, audio.flatMap((part) => SpeechStream.delta(next, part.data)[1])];
|
|
50
|
+
});
|
|
51
|
+
const finish = (state) => {
|
|
52
|
+
const sampleRate = SpeechStream.sampleRate(state.mimeType) ?? DEFAULT_SAMPLE_RATE;
|
|
53
|
+
return SpeechStream.finish(ADAPTER, state, {
|
|
54
|
+
...SpeechStream.pcm("pcm_s16le", sampleRate, state.mimeType ?? `audio/L16;codec=pcm;rate=${sampleRate}`),
|
|
55
|
+
usage: GeminiGenerateContent.usage(state.usage),
|
|
56
|
+
providerMetadata: GeminiGenerateContent.providerMetadata(state),
|
|
57
|
+
detail: state.finishReason === undefined ? undefined : `finish reason: ${state.finishReason}`,
|
|
58
|
+
});
|
|
59
|
+
};
|
|
60
|
+
// ---------------------------------------------------------------------------
|
|
61
|
+
// 7. Protocol and route
|
|
62
|
+
// ---------------------------------------------------------------------------
|
|
63
|
+
export const protocol = MediaProtocol.stream({
|
|
64
|
+
id: ADAPTER,
|
|
65
|
+
name: NAME,
|
|
66
|
+
unsupported: ["instructions", "speed", "timestamps"],
|
|
67
|
+
body: { from: fromRequest },
|
|
68
|
+
frames: (bytes, context) => GeminiGenerateContent.frames(bytes, context.request.mode),
|
|
69
|
+
initial: () => ({ chunks: [] }),
|
|
70
|
+
step,
|
|
71
|
+
finish,
|
|
72
|
+
});
|
|
73
|
+
export const model = (input) => SpeechModel.fromRoute({
|
|
74
|
+
id: ADAPTER,
|
|
75
|
+
provider: PROVIDER,
|
|
76
|
+
protocol,
|
|
77
|
+
baseURL: DEFAULT_BASE_URL,
|
|
78
|
+
// Only `gemini-3.1-flash-tts-preview` and later stream; earlier TTS models reject `streamGenerateContent`.
|
|
79
|
+
path: ({ request }) => GeminiGenerateContent.path(request.model.id, request.mode),
|
|
80
|
+
}, input);
|
|
81
|
+
export const GoogleSpeech = {
|
|
82
|
+
protocol,
|
|
83
|
+
model,
|
|
84
|
+
};
|