@opencode/ai 2.0.15 → 2.0.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +441 -99
- package/dist/ai-client.d.ts +8 -0
- package/dist/ai-client.js +12 -0
- package/dist/experimental/evaluation-client.d.ts +3 -3
- package/dist/experimental/evaluation-client.js +1 -1
- package/dist/experimental/evaluation.js +1 -1
- package/dist/generation.d.ts +39 -29
- package/dist/generation.js +62 -36
- package/dist/image-client.d.ts +59 -17
- package/dist/image-client.js +16 -24
- package/dist/image.d.ts +394 -55
- package/dist/image.js +48 -57
- package/dist/index.d.ts +15 -2
- package/dist/index.js +10 -0
- package/dist/llm.d.ts +7 -5
- package/dist/llm.js +10 -4
- package/dist/media-client.d.ts +30 -0
- package/dist/media-client.js +51 -0
- package/dist/media-model.d.ts +43 -0
- package/dist/media-model.js +47 -0
- package/dist/media.d.ts +10 -9
- package/dist/media.js +11 -12
- package/dist/promise.d.ts +473 -15
- package/dist/promise.js +71 -13
- package/dist/protocols/alibaba-chat.d.ts +12 -0
- package/dist/protocols/alibaba-chat.js +4 -1
- package/dist/protocols/alibaba-messages.d.ts +1 -1
- package/dist/protocols/alibaba-messages.js +6 -4
- package/dist/protocols/alibaba-responses.d.ts +2 -2
- package/dist/protocols/anthropic-messages.d.ts +34 -34
- package/dist/protocols/anthropic-messages.js +14 -9
- package/dist/protocols/assemblyai-transcription.d.ts +41 -0
- package/dist/protocols/assemblyai-transcription.js +136 -0
- package/dist/protocols/bedrock-converse.d.ts +7 -0
- package/dist/protocols/bedrock-converse.js +39 -19
- package/dist/protocols/bfl-images.d.ts +38 -0
- package/dist/protocols/bfl-images.js +153 -0
- package/dist/protocols/cartesia-speech.d.ts +143 -0
- package/dist/protocols/cartesia-speech.js +120 -0
- package/dist/protocols/deepgram-speech.d.ts +135 -0
- package/dist/protocols/deepgram-speech.js +91 -0
- package/dist/protocols/deepgram-transcription.d.ts +26 -0
- package/dist/protocols/deepgram-transcription.js +125 -0
- package/dist/protocols/elevenlabs-speech.d.ts +138 -0
- package/dist/protocols/elevenlabs-speech.js +111 -0
- package/dist/protocols/fal-images.d.ts +25 -0
- package/dist/protocols/fal-images.js +104 -0
- package/dist/protocols/fal-video.d.ts +29 -0
- package/dist/protocols/fal-video.js +73 -0
- package/dist/protocols/gemini.d.ts +22 -22
- package/dist/protocols/gemini.js +19 -36
- package/dist/protocols/google-images.d.ts +3 -3
- package/dist/protocols/google-images.js +12 -34
- package/dist/protocols/google-speech.d.ts +146 -0
- package/dist/protocols/google-speech.js +88 -0
- package/dist/protocols/google-transcription.d.ts +174 -0
- package/dist/protocols/google-transcription.js +126 -0
- package/dist/protocols/google-video.d.ts +26 -0
- package/dist/protocols/google-video.js +142 -0
- package/dist/protocols/meta-images.d.ts +3 -4
- package/dist/protocols/meta-images.js +13 -29
- package/dist/protocols/meta-messages.d.ts +5 -5
- package/dist/protocols/meta-responses.d.ts +4 -4
- package/dist/protocols/meta-responses.js +6 -4
- package/dist/protocols/open-responses.d.ts +10 -9
- package/dist/protocols/open-responses.js +4 -9
- package/dist/protocols/openai-chat.d.ts +84 -0
- package/dist/protocols/openai-chat.js +28 -17
- package/dist/protocols/openai-compatible-chat.d.ts +12 -0
- package/dist/protocols/openai-compatible-responses.d.ts +2 -2
- package/dist/protocols/openai-images.d.ts +130 -7
- package/dist/protocols/openai-images.js +143 -91
- package/dist/protocols/openai-responses.d.ts +20 -20
- package/dist/protocols/openai-responses.js +31 -31
- package/dist/protocols/openai-speech.d.ts +136 -0
- package/dist/protocols/openai-speech.js +97 -0
- package/dist/protocols/openai-transcription.d.ts +211 -0
- package/dist/protocols/openai-transcription.js +202 -0
- package/dist/protocols/replicate-images.d.ts +28 -0
- package/dist/protocols/replicate-images.js +127 -0
- package/dist/protocols/runway-video.d.ts +38 -0
- package/dist/protocols/runway-video.js +140 -0
- package/dist/protocols/shared.d.ts +21 -17
- package/dist/protocols/shared.js +27 -36
- package/dist/protocols/stability-images.d.ts +39 -0
- package/dist/protocols/stability-images.js +136 -0
- package/dist/protocols/utils/fal-queue.d.ts +26 -0
- package/dist/protocols/utils/fal-queue.js +67 -0
- package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
- package/dist/protocols/utils/gemini-generate-content.js +65 -0
- package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
- package/dist/protocols/utils/gemini-json-schema.js +76 -0
- package/dist/protocols/utils/media-input.d.ts +23 -1
- package/dist/protocols/utils/media-input.js +40 -0
- package/dist/protocols/utils/responses-checkpoint.js +3 -7
- package/dist/protocols/utils/responses-compaction.d.ts +3 -1
- package/dist/protocols/utils/responses-compaction.js +16 -3
- package/dist/protocols/utils/speech-stream.d.ts +49 -0
- package/dist/protocols/utils/speech-stream.js +64 -0
- package/dist/protocols/utils/tool-schema.d.ts +2 -2
- package/dist/protocols/utils/tool-schema.js +62 -19
- package/dist/protocols/xai-images.d.ts +4 -4
- package/dist/protocols/xai-images.js +14 -40
- package/dist/protocols/xai-responses.d.ts +2 -2
- package/dist/protocols/xai-responses.js +1 -1
- package/dist/protocols/xai-video.d.ts +34 -0
- package/dist/protocols/xai-video.js +141 -0
- package/dist/protocols/zai-chat.d.ts +13 -1
- package/dist/protocols/zai-images.d.ts +2 -2
- package/dist/protocols/zai-images.js +11 -14
- package/dist/protocols/zai-messages.d.ts +1 -1
- package/dist/provider-error.js +10 -1
- package/dist/providers/alibaba.d.ts +15 -3
- package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
- package/dist/providers/amazon-bedrock.d.ts +2 -0
- package/dist/providers/amazon-bedrock.js +1 -0
- package/dist/providers/anthropic-compatible.d.ts +5 -5
- package/dist/providers/anthropic.d.ts +5 -5
- package/dist/providers/assemblyai.d.ts +25 -0
- package/dist/providers/assemblyai.js +24 -0
- package/dist/providers/azure.d.ts +20 -8
- package/dist/providers/azure.js +2 -2
- package/dist/providers/baseten.d.ts +24 -0
- package/dist/providers/black-forest-labs.d.ts +25 -0
- package/dist/providers/black-forest-labs.js +23 -0
- package/dist/providers/cartesia.d.ts +24 -0
- package/dist/providers/cartesia.js +17 -0
- package/dist/providers/cerebras.d.ts +24 -0
- package/dist/providers/cloudflare-ai-gateway.d.ts +42 -18
- package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
- package/dist/providers/deepgram.d.ts +29 -0
- package/dist/providers/deepgram.js +25 -0
- package/dist/providers/deepinfra.d.ts +24 -0
- package/dist/providers/deepseek.d.ts +24 -0
- package/dist/providers/elevenlabs.d.ts +24 -0
- package/dist/providers/elevenlabs.js +23 -0
- package/dist/providers/fal.d.ts +29 -0
- package/dist/providers/fal.js +26 -0
- package/dist/providers/fireworks.d.ts +24 -0
- package/dist/providers/google-vertex-chat.d.ts +12 -0
- package/dist/providers/google-vertex-messages.d.ts +5 -5
- package/dist/providers/google-vertex-responses.d.ts +2 -2
- package/dist/providers/google-vertex.d.ts +6 -6
- package/dist/providers/google.d.ts +21 -6
- package/dist/providers/google.js +13 -9
- package/dist/providers/groq.d.ts +24 -0
- package/dist/providers/index.d.ts +9 -0
- package/dist/providers/index.js +9 -0
- package/dist/providers/meta.d.ts +22 -10
- package/dist/providers/meta.js +4 -8
- package/dist/providers/minimax.d.ts +19 -7
- package/dist/providers/moonshot.d.ts +19 -7
- package/dist/providers/moonshot.js +3 -3
- package/dist/providers/openai-compatible-responses.d.ts +2 -2
- package/dist/providers/openai-compatible.d.ts +12 -0
- package/dist/providers/openai-options.d.ts +3 -9
- package/dist/providers/openai-options.js +4 -7
- package/dist/providers/openai.d.ts +33 -12
- package/dist/providers/openai.js +17 -9
- package/dist/providers/opencode-zen.js +1 -1
- package/dist/providers/openrouter.d.ts +54 -7
- package/dist/providers/openrouter.js +10 -5
- package/dist/providers/replicate.d.ts +25 -0
- package/dist/providers/replicate.js +17 -0
- package/dist/providers/runway.d.ts +24 -0
- package/dist/providers/runway.js +17 -0
- package/dist/providers/stability.d.ts +28 -0
- package/dist/providers/stability.js +18 -0
- package/dist/providers/togetherai.d.ts +24 -0
- package/dist/providers/typesafe-ai.js +1 -1
- package/dist/providers/vercel-ai-gateway.js +1 -1
- package/dist/providers/xai.d.ts +17 -0
- package/dist/providers/xai.js +7 -9
- package/dist/providers/zai-coding-plan.d.ts +16 -4
- package/dist/providers/zai.d.ts +13 -1
- package/dist/providers/zai.js +4 -8
- package/dist/route/auth.d.ts +5 -2
- package/dist/route/auth.js +20 -13
- package/dist/route/client.d.ts +9 -7
- package/dist/route/client.js +6 -8
- package/dist/route/endpoint.d.ts +1 -0
- package/dist/route/endpoint.js +2 -2
- package/dist/route/executor-service.d.ts +4 -2
- package/dist/route/executor-service.js +2 -1
- package/dist/route/executor.d.ts +3 -1
- package/dist/route/executor.js +7 -0
- package/dist/route/framing.d.ts +15 -2
- package/dist/route/framing.js +53 -4
- package/dist/route/index.d.ts +1 -1
- package/dist/route/media-protocol.d.ts +145 -18
- package/dist/route/media-protocol.js +96 -28
- package/dist/route/media.d.ts +54 -9
- package/dist/route/media.js +194 -38
- package/dist/route/protocol.d.ts +3 -1
- package/dist/schema/events.d.ts +0 -6
- package/dist/schema/messages.d.ts +0 -3
- package/dist/schema/options.d.ts +10 -6
- package/dist/schema/options.js +10 -5
- package/dist/speech-client.d.ts +66 -0
- package/dist/speech-client.js +21 -0
- package/dist/speech.d.ts +1307 -0
- package/dist/speech.js +124 -0
- package/dist/testing.d.ts +2 -2
- package/dist/transcription-client.d.ts +82 -0
- package/dist/transcription-client.js +13 -0
- package/dist/transcription.d.ts +1492 -0
- package/dist/transcription.js +127 -0
- package/dist/utils/bytes.d.ts +1 -0
- package/dist/utils/bytes.js +10 -0
- package/dist/utils/json.d.ts +4 -0
- package/dist/utils/json.js +4 -0
- package/dist/utils/media-type.d.ts +3 -1
- package/dist/utils/media-type.js +25 -2
- package/dist/video-client.d.ts +59 -0
- package/dist/video-client.js +20 -0
- package/dist/video.d.ts +1351 -0
- package/dist/video.js +118 -0
- package/package.json +3 -3
- package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
- package/dist/protocols/utils/gemini-tool-schema.js +0 -103
- package/dist/protocols/utils/meta-image.d.ts +0 -2
- package/dist/protocols/utils/meta-image.js +0 -13
- package/dist/protocols/utils/openai-image.d.ts +0 -5
- package/dist/protocols/utils/openai-image.js +0 -18
package/README.md
CHANGED
|
@@ -1,40 +1,38 @@
|
|
|
1
1
|
# @opencode/ai
|
|
2
2
|
|
|
3
|
-
Schema-first
|
|
3
|
+
Schema-first APIs for text, images, video, speech, and transcription, built with Effect.
|
|
4
4
|
|
|
5
5
|
```ts
|
|
6
|
-
import { Effect
|
|
7
|
-
import {
|
|
8
|
-
import { RequestExecutor } from "@opencode/ai/route"
|
|
6
|
+
import { Effect } from "effect"
|
|
7
|
+
import { AIClient, LLM } from "@opencode/ai"
|
|
9
8
|
import { OpenAI } from "@opencode/ai/providers"
|
|
10
9
|
|
|
11
10
|
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
|
12
11
|
|
|
13
|
-
const request = LLM.request({
|
|
14
|
-
model: openai.responses("gpt-4o-mini"), // `.chat(...)` selects the Chat Completions API instead
|
|
15
|
-
system: "You are concise.",
|
|
16
|
-
prompt: "Say hello in one short sentence.",
|
|
17
|
-
generation: { maxTokens: 40 },
|
|
18
|
-
})
|
|
19
|
-
|
|
20
12
|
const program = Effect.gen(function* () {
|
|
21
|
-
const response = yield*
|
|
13
|
+
const response = yield* LLM.generate({
|
|
14
|
+
model: openai.responses("gpt-4o-mini"), // `.chat(...)` selects the Chat Completions API instead
|
|
15
|
+
system: "You are concise.",
|
|
16
|
+
prompt: "Say hello in one short sentence.",
|
|
17
|
+
generation: { maxTokens: 40 },
|
|
18
|
+
})
|
|
22
19
|
console.log(response.text)
|
|
23
20
|
})
|
|
24
21
|
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
await Effect.runPromise(program.pipe(Effect.provide(llmLayer)))
|
|
22
|
+
// Every modality client plus the HTTP request executor; `AIClient.layerWith(executor)` swaps the executor.
|
|
23
|
+
await Effect.runPromise(program.pipe(Effect.provide(AIClient.layer)))
|
|
28
24
|
```
|
|
29
25
|
|
|
30
|
-
Run `
|
|
26
|
+
Run `LLM.stream(...)` instead of `generate` when you want incremental `LLMEvent`s. Both accept input or a prebuilt
|
|
27
|
+
`LLM.request(...)`. The event stream is provider-neutral — same shape across OpenAI Chat, OpenAI Responses,
|
|
28
|
+
Anthropic Messages, Gemini, Bedrock Converse, and any OpenAI-compatible deployment.
|
|
31
29
|
|
|
32
|
-
The same configured facade names image models. `Image.
|
|
33
|
-
returns `Media.Asset`s with lazily decoded bytes:
|
|
30
|
+
The same configured facade names image, video, speech, and transcription models. `Image.generate` resolves the
|
|
31
|
+
provider's image route from the model and returns `Media.Asset`s with lazily decoded bytes:
|
|
34
32
|
|
|
35
33
|
```ts
|
|
36
34
|
import { NodeFileSystem } from "@effect/platform-node"
|
|
37
|
-
import { Image,
|
|
35
|
+
import { Image, Media } from "@opencode/ai"
|
|
38
36
|
|
|
39
37
|
const image = Effect.gen(function* () {
|
|
40
38
|
const response = yield* Image.generate({
|
|
@@ -46,21 +44,38 @@ const image = Effect.gen(function* () {
|
|
|
46
44
|
yield* Media.write(response.image, "./garden.png")
|
|
47
45
|
})
|
|
48
46
|
|
|
49
|
-
// `
|
|
50
|
-
|
|
47
|
+
// `Media.file` / `Media.write` use the Effect `FileSystem` service; provide your platform's layer.
|
|
48
|
+
await Effect.runPromise(image.pipe(Effect.provide(AIClient.layer), Effect.provide(NodeFileSystem.layer)))
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Advanced: each client also has its own `layer`, which requires `RequestExecutor.Service`. Compose client layers with
|
|
52
|
+
`Layer.provideMerge`, not `Layer.provide`: `asset.bytes()`, `Media.write`, and Gemini's `media` output parts need the
|
|
53
|
+
executor too, and hiding it fails type-checking with `RequestExecutorService` left in the requirements.
|
|
54
|
+
|
|
55
|
+
To share a policy such as logging across every client, wrap the executor once with `RequestExecutor.middleware`:
|
|
51
56
|
|
|
52
|
-
|
|
57
|
+
```ts
|
|
58
|
+
import { RequestExecutor } from "@opencode/ai/route"
|
|
59
|
+
|
|
60
|
+
const logged = RequestExecutor.middleware((request, next) =>
|
|
61
|
+
Effect.log(`${request.method} ${request.url}`).pipe(Effect.andThen(next(request))),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
const everything = AIClient.layerWith(logged) // or AI.make({ layer: logged })
|
|
53
65
|
```
|
|
54
66
|
|
|
55
|
-
Prefer promises? `@opencode/ai/promise` exposes the same LLM and
|
|
67
|
+
Prefer promises? `@opencode/ai/promise` exposes the same LLM and media APIs over one managed runtime, plus asset
|
|
68
|
+
helpers; `ai.file` and `ai.write` load `node:fs/promises` on first use, so no Effect `FileSystem` is needed:
|
|
56
69
|
|
|
57
70
|
```ts
|
|
58
71
|
import { AI } from "@opencode/ai/promise"
|
|
59
72
|
|
|
60
73
|
const ai = AI.make()
|
|
61
|
-
const
|
|
74
|
+
const input = { model: openai.responses("gpt-4o-mini"), prompt: "Say hello." }
|
|
75
|
+
const text = await ai.llm.generate(input)
|
|
62
76
|
const generated = await ai.image.generate({ model: openai.image("gpt-image-2"), prompt: "A lighthouse" })
|
|
63
|
-
|
|
77
|
+
await ai.write(generated.image, "./lighthouse.png") // also ai.file(path), ai.bytes(asset), ai.base64(asset), ai.materialize(asset)
|
|
78
|
+
for await (const event of ai.llm.stream(ai.llm.request(input))) {
|
|
64
79
|
// LLMEvent
|
|
65
80
|
}
|
|
66
81
|
await ai.dispose()
|
|
@@ -322,10 +337,9 @@ and `moonshot/responses`; each exports `model(modelID, settings)`.
|
|
|
322
337
|
MiniMax defaults to its Messages API and reads `MINIMAX_API_KEY` when `apiKey` is omitted:
|
|
323
338
|
|
|
324
339
|
```ts
|
|
325
|
-
import { Effect
|
|
326
|
-
import {
|
|
340
|
+
import { Effect } from "effect"
|
|
341
|
+
import { AIClient, LLM } from "@opencode/ai"
|
|
327
342
|
import { MiniMax } from "@opencode/ai/providers"
|
|
328
|
-
import { RequestExecutor } from "@opencode/ai/route"
|
|
329
343
|
|
|
330
344
|
const minimax = MiniMax.configure({ apiKey: process.env.MINIMAX_API_KEY })
|
|
331
345
|
const request = LLM.request({
|
|
@@ -335,8 +349,7 @@ const request = LLM.request({
|
|
|
335
349
|
generation: { maxTokens: 1536 },
|
|
336
350
|
})
|
|
337
351
|
|
|
338
|
-
const
|
|
339
|
-
const response = await Effect.runPromise(LLMClient.generate(request).pipe(Effect.provide(layer)))
|
|
352
|
+
const response = await Effect.runPromise(LLM.generate(request).pipe(Effect.provide(AIClient.layer)))
|
|
340
353
|
console.log(response.text)
|
|
341
354
|
```
|
|
342
355
|
|
|
@@ -405,14 +418,14 @@ Use `Image.generate` for one-off generation or editing:
|
|
|
405
418
|
import { Image, Media } from "@opencode/ai"
|
|
406
419
|
|
|
407
420
|
const generation = Image.generate({
|
|
408
|
-
model: meta("muse-image-1.0"),
|
|
421
|
+
model: meta.image("muse-image-1.0"),
|
|
409
422
|
prompt: "A flat black square on a white background.",
|
|
410
423
|
n: 1,
|
|
411
424
|
providerOptions: { reasoningStrength: "low" },
|
|
412
425
|
})
|
|
413
426
|
|
|
414
427
|
const edit = Image.generate({
|
|
415
|
-
model: meta("muse-image-1.0"),
|
|
428
|
+
model: meta.image("muse-image-1.0"),
|
|
416
429
|
prompt: "Make the square purple.",
|
|
417
430
|
images: [Media.bytes(imageBytes, "image/webp")],
|
|
418
431
|
format: "png",
|
|
@@ -459,6 +472,25 @@ const program = Effect.gen(function* () {
|
|
|
459
472
|
})
|
|
460
473
|
```
|
|
461
474
|
|
|
475
|
+
Common fields are portable in shape, not in support. Unsupported fields fail with a typed `AIError` before any network
|
|
476
|
+
call rather than being dropped, so check this table before swapping only the `model`:
|
|
477
|
+
|
|
478
|
+
| Provider | `n` | `size` | `aspectRatio` | `seed` | `format` | `images` | `mask` |
|
|
479
|
+
| --------------------- | --- | --------- | ------------- | ------ | -------- | -------------------------------- | ------------------- |
|
|
480
|
+
| OpenAI | ✓¹ | ✓ | ✗ | ✗ | ✓ | ✓ | ✓ |
|
|
481
|
+
| Google (Gemini) | 1 | ✗ | ✓ | ✓ | ✗ | ✓ (no public URLs) | ✗ |
|
|
482
|
+
| xAI | ✓ | ✗ | ✓ | ✗ | ✗ | ✓ | ✗ |
|
|
483
|
+
| Z.ai | ✗ | ✓ | ✗ | ✗ | ✗ | ✗ | ✗ |
|
|
484
|
+
| Meta | ✓ | ✓ (hint) | ✗ | ✗ | ✓ | ✓ | ✗ |
|
|
485
|
+
| Black Forest Labs | 1 | per model | per model | ✓ | ✓ | per model (1–8) | `flux-pro-1.0-fill` |
|
|
486
|
+
| fal | ✓ | per model | per model | ✓ | ✓ | 1 (several on `/edit`, `/multi`) | ✓ |
|
|
487
|
+
| Replicate | ✗ | ✗ | ✗ | ✗ | ✗ | ✗ (use `providerOptions`) | ✗ |
|
|
488
|
+
| Stability `image` | 1 | ✗ | ✓ | ✓ | ✓ | 1 (not on `core`) | ✗ |
|
|
489
|
+
| Stability `upscale()` | ✗ | ✗ | ✗ | ✓ | ✓ | exactly 1 (required) | ✗ |
|
|
490
|
+
|
|
491
|
+
✓ lowers natively; ✗ fails whenever the field is set (including `n: 1`); `1` means `n > 1` fails. ¹ `Image.stream` on OpenAI generates one image. fal
|
|
492
|
+
rejects `size` and `aspectRatio` together; which one a fal or BFL model takes depends on the model.
|
|
493
|
+
|
|
462
494
|
`Media.Asset` is the one asset type shared by image requests, image responses, LLM messages, and tool results.
|
|
463
495
|
`asset.source` is the serializable `Media.Source` (`bytes`, `base64`, `url`, or `ref`); `asset.bytes()`,
|
|
464
496
|
`asset.base64()`, and `asset.dataUrl()` decode or download lazily and cache; `asset.materialize()` pulls a `url`
|
|
@@ -468,9 +500,8 @@ asset into owned bytes before the provider URL expires. Construct assets with `M
|
|
|
468
500
|
Pass ordered image inputs to the same method for editing, composition, or image-conditioned generation:
|
|
469
501
|
|
|
470
502
|
```ts
|
|
471
|
-
const
|
|
472
|
-
yield
|
|
473
|
-
Image.generate({
|
|
503
|
+
const composed = Effect.gen(function* () {
|
|
504
|
+
const response = yield* Image.generate({
|
|
474
505
|
model,
|
|
475
506
|
prompt: "Combine these product photos into one studio scene",
|
|
476
507
|
images: [
|
|
@@ -481,23 +512,25 @@ const response =
|
|
|
481
512
|
providerOptions,
|
|
482
513
|
http,
|
|
483
514
|
})
|
|
515
|
+
return response.images
|
|
516
|
+
})
|
|
484
517
|
```
|
|
485
518
|
|
|
486
519
|
`Media.ref(provider, id)` represents provider file handles such as OpenAI file IDs or Gemini Files URIs; routes
|
|
487
|
-
only forward refs that belong to their own provider
|
|
488
|
-
|
|
489
|
-
|
|
520
|
+
only forward refs that belong to their own provider (OpenAI, xAI, and Gemini images accept them). No shipped route
|
|
521
|
+
returns a ref yet, and `asset.bytes()` / `materialize()` on a ref fail by design. Raw strings are not accepted as
|
|
522
|
+
image inputs, avoiding ambiguity between base64, URLs, and provider IDs. Empty or omitted `images` uses text-to-image generation; a
|
|
523
|
+
non-empty array selects the provider's edit behavior (see the table above for routes that limit the count). OpenAI
|
|
490
524
|
uses multipart for byte/data-URL edits and its JSON reference body for URL or file-ID edits. The common `mask`
|
|
491
525
|
field selects inpainting; routes that cannot honor it fail with `UnsupportedOperation`:
|
|
492
526
|
|
|
493
527
|
```ts
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
})
|
|
528
|
+
const inpainted = Image.generate({
|
|
529
|
+
model: openai.image("gpt-image-2"),
|
|
530
|
+
prompt,
|
|
531
|
+
images: [Media.bytes(sourceBytes, "image/png")],
|
|
532
|
+
mask: Media.bytes(maskBytes, "image/png"),
|
|
533
|
+
})
|
|
501
534
|
```
|
|
502
535
|
|
|
503
536
|
On multipart requests, `http.body` can override option fields but not structural `model`, `prompt`, `image[]`,
|
|
@@ -508,31 +541,31 @@ not accept image inputs. These cases fail with a typed `AIError` before network
|
|
|
508
541
|
Provider-native image options belong to each request. Raw `http.body` fields have final precedence over them:
|
|
509
542
|
|
|
510
543
|
```ts
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
})
|
|
544
|
+
const medium = Image.generate({
|
|
545
|
+
model: openai.image("gpt-image-2"),
|
|
546
|
+
prompt,
|
|
547
|
+
providerOptions: { quality: "medium" },
|
|
548
|
+
http,
|
|
549
|
+
})
|
|
518
550
|
```
|
|
519
551
|
|
|
520
552
|
xAI image models use the same request API with xAI-native controls:
|
|
521
553
|
|
|
522
554
|
```ts
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
555
|
+
import { XAI } from "@opencode/ai/providers"
|
|
556
|
+
|
|
557
|
+
const xai = Image.generate({
|
|
558
|
+
model: XAI.configure({ apiKey }).image("any-model-id"),
|
|
559
|
+
prompt,
|
|
560
|
+
n: 2,
|
|
561
|
+
aspectRatio: "16:9",
|
|
562
|
+
providerOptions: {
|
|
563
|
+
resolution: "1k",
|
|
564
|
+
responseFormat: "b64_json",
|
|
565
|
+
future_option: true,
|
|
566
|
+
},
|
|
567
|
+
http,
|
|
568
|
+
})
|
|
536
569
|
```
|
|
537
570
|
|
|
538
571
|
Google's current Gemini image models use the same direct API:
|
|
@@ -542,7 +575,7 @@ import { Google } from "@opencode/ai/providers"
|
|
|
542
575
|
|
|
543
576
|
const googleProgram = Effect.gen(function* () {
|
|
544
577
|
const response = yield* Image.generate({
|
|
545
|
-
model: Google.configure({ apiKey })("any-model-id"),
|
|
578
|
+
model: Google.configure({ apiKey }).image("any-model-id"),
|
|
546
579
|
prompt: "A robot tending a rooftop garden",
|
|
547
580
|
aspectRatio: "16:9",
|
|
548
581
|
seed: 42,
|
|
@@ -567,23 +600,88 @@ their mapped aliases, and `http.body` is the final deep overlay. The selected mo
|
|
|
567
600
|
Z.ai image models infer open Z.ai-native options from the selected model:
|
|
568
601
|
|
|
569
602
|
```ts
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
603
|
+
import { ZAI } from "@opencode/ai/providers"
|
|
604
|
+
|
|
605
|
+
const zai = Image.generate({
|
|
606
|
+
model: ZAI.configure({ apiKey }).image("any-model-id"),
|
|
607
|
+
prompt,
|
|
608
|
+
providerOptions: {
|
|
609
|
+
quality: "hd",
|
|
610
|
+
userID: "user-123",
|
|
611
|
+
future_option: true,
|
|
612
|
+
},
|
|
613
|
+
http,
|
|
614
|
+
})
|
|
581
615
|
```
|
|
582
616
|
|
|
583
617
|
Z.ai does not include trustworthy MIME metadata for output URLs, so generated images use
|
|
584
618
|
`application/octet-stream` until materialized. Output URLs expire after 30 days; call `asset.materialize()` and
|
|
585
619
|
persist the bytes promptly if they must remain available.
|
|
586
620
|
|
|
621
|
+
### Partial images
|
|
622
|
+
|
|
623
|
+
OpenAI's GPT image models stream previews. `Image.stream` sends `stream: true` with `partialImages` (0–3, default 2)
|
|
624
|
+
and emits `image-partial` events before each final `image`; `Image.generate` keeps the plain JSON request:
|
|
625
|
+
|
|
626
|
+
```ts
|
|
627
|
+
import { Stream } from "effect"
|
|
628
|
+
import { ImageEvent } from "@opencode/ai"
|
|
629
|
+
|
|
630
|
+
const previews = Image.stream({
|
|
631
|
+
model: openai.image("gpt-image-2"),
|
|
632
|
+
prompt: "A lighthouse at dusk",
|
|
633
|
+
providerOptions: { partialImages: 2 },
|
|
634
|
+
}).pipe(Stream.runForEach((event) => (ImageEvent.is.imagePartial(event) ? showPreview(event.image) : Effect.void)))
|
|
635
|
+
```
|
|
636
|
+
|
|
637
|
+
The provider may send fewer previews than requested when the final image is ready first.
|
|
638
|
+
|
|
639
|
+
### Queued image providers
|
|
640
|
+
|
|
641
|
+
Black Forest Labs, fal, Replicate, and Stability's creative upscaler are submit-then-poll routes. `Image.generate`
|
|
642
|
+
and `Image.stream` poll for you (pass `{ poll }` to tune the interval and timeout); `Image.start` returns a
|
|
643
|
+
`Generation` whose `token` is serializable JSON for `Image.resume` in another process:
|
|
644
|
+
|
|
645
|
+
```ts
|
|
646
|
+
import { BlackForestLabs, Stability } from "@opencode/ai/providers"
|
|
647
|
+
|
|
648
|
+
const bfl = BlackForestLabs.configure({ apiKey: process.env.BFL_API_KEY })
|
|
649
|
+
|
|
650
|
+
const submit = Effect.gen(function* () {
|
|
651
|
+
const generation = yield* Image.start({ model: bfl.image("flux-2-pro"), prompt, size: "1024x768" })
|
|
652
|
+
persist({ provider: "black-forest-labs", modelID: "flux-2-pro", token: generation.token })
|
|
653
|
+
})
|
|
654
|
+
|
|
655
|
+
const finish = Effect.gen(function* () {
|
|
656
|
+
const saved = load()
|
|
657
|
+
const resumed = yield* Image.resume(bfl.image(saved.modelID), saved.token)
|
|
658
|
+
return yield* resumed.await({ poll: { interval: "2 seconds" } })
|
|
659
|
+
})
|
|
660
|
+
```
|
|
661
|
+
|
|
662
|
+
The token carries no route identity, so persist the provider and model ID alongside it: `resume` needs the model.
|
|
663
|
+
|
|
664
|
+
- **Black Forest Labs** — results are downloaded before returning, because `result.sample` expires in 10 minutes.
|
|
665
|
+
- **Replicate** — inputs are model-defined, so only `prompt` lowers: sizing, count, seed, format, and files go in
|
|
666
|
+
`providerOptions` under the model's names, with files as `Media.Asset` (data URLs up to 256 KB, larger by URL).
|
|
667
|
+
Outputs are removed an hour after the prediction completes. `Prefer: wait=60` in `headers` or `http.headers` holds
|
|
668
|
+
the submission open so a fast prediction costs one result read.
|
|
669
|
+
- **Stability** — `stability.image(id)` generates inline; `stability.upscale()` is the creative upscaler, queued:
|
|
670
|
+
|
|
671
|
+
```ts
|
|
672
|
+
const stability = Stability.configure({ apiKey: process.env.STABILITY_API_KEY })
|
|
673
|
+
const upscaled = Effect.gen(function* () {
|
|
674
|
+
const small = yield* Media.file("./small.png")
|
|
675
|
+
return yield* Image.generate(
|
|
676
|
+
{ model: stability.upscale(), prompt: "A lighthouse", images: [small] },
|
|
677
|
+
{ poll: { interval: "5 seconds" } },
|
|
678
|
+
)
|
|
679
|
+
})
|
|
680
|
+
```
|
|
681
|
+
|
|
682
|
+
Imagen is not available: Google shut it down on the Gemini API, and Vertex discontinued the Imagen 4 models on
|
|
683
|
+
2026-06-30. `Google.image(...)` uses Gemini-native image models.
|
|
684
|
+
|
|
587
685
|
Conversational image generation remains part of the LLM interaction. OpenAI Responses exposes it through its hosted image tool:
|
|
588
686
|
|
|
589
687
|
```ts
|
|
@@ -600,20 +698,257 @@ const program = Effect.gen(function* () {
|
|
|
600
698
|
})
|
|
601
699
|
```
|
|
602
700
|
|
|
603
|
-
The hosted result is represented as a provider-executed tool call and tool result
|
|
701
|
+
The hosted result is represented as a provider-executed tool call and a tool result whose content carries the generated image as a file. Gemini image-capable models instead emit a first-class `media` `LLMEvent` for inline image output (`response.message` then carries a `media` part). Retaining `response.message` preserves the generated image for continuation on both routes.
|
|
702
|
+
|
|
703
|
+
## Video generation
|
|
704
|
+
|
|
705
|
+
Video mirrors `Image` with one difference: every provider is asynchronous, so the route is a submit-then-poll
|
|
706
|
+
`Generation`. Models come from `.video(...)` selectors on the `Google` (Veo), `XAI`, `Fal`, and `Runway` facades.
|
|
707
|
+
Common fields (`frames`, `references`, `video`, `durationSeconds`, `aspectRatio`, `resolution`, `audio`, `n`, `seed`,
|
|
708
|
+
`negativePrompt`) lower natively or fail with a typed `AIError` before any network call; provider-native controls live
|
|
709
|
+
under `providerOptions`, inferred from the selected model.
|
|
710
|
+
|
|
711
|
+
```ts
|
|
712
|
+
import { Video } from "@opencode/ai"
|
|
713
|
+
import { Google, Runway } from "@opencode/ai/providers"
|
|
714
|
+
|
|
715
|
+
const google = Google.configure({ apiKey: process.env.GOOGLE_GENERATIVE_AI_API_KEY })
|
|
716
|
+
|
|
717
|
+
// Simple: submit and wait.
|
|
718
|
+
const program = Effect.gen(function* () {
|
|
719
|
+
const response = yield* Video.generate(
|
|
720
|
+
{
|
|
721
|
+
model: google.video("veo-3.1-generate-preview"),
|
|
722
|
+
prompt: "Panning wide shot of a calico kitten sleeping in the sunshine",
|
|
723
|
+
aspectRatio: "16:9",
|
|
724
|
+
resolution: "1080p",
|
|
725
|
+
durationSeconds: 8,
|
|
726
|
+
providerOptions: { personGeneration: "allow_adult" },
|
|
727
|
+
},
|
|
728
|
+
{ poll: { interval: "10 seconds", timeout: "10 minutes" } },
|
|
729
|
+
)
|
|
730
|
+
// Veo serves files for two days behind the API key. The asset knows the deadline (`expiresAt`) and carries the
|
|
731
|
+
// download credentials only on the live instance (`asset.headers`), never in `source` or JSON: materialize
|
|
732
|
+
// before persisting, or the persisted URL cannot be fetched again.
|
|
733
|
+
return yield* response.video.materialize()
|
|
734
|
+
})
|
|
735
|
+
|
|
736
|
+
// Explicit control: keep the handle, persist the token, resume elsewhere.
|
|
737
|
+
const controlled = Effect.gen(function* () {
|
|
738
|
+
const generation = yield* Video.start({ model: google.video("veo-3.1-generate-preview"), prompt })
|
|
739
|
+
generation.id // provider operation / task / request id
|
|
740
|
+
generation.status // "queued" | "running" | "completed" | "failed" | "cancelled" | "expired"
|
|
741
|
+
generation.token // route-owned JSON: `{ operation }`, `{ requestID }`, `{ taskID }`, or fal's follow-up URLs
|
|
742
|
+
// The token carries no route identity: persist the provider and model ID alongside it, since `resume` needs the model.
|
|
743
|
+
const saved = JSON.stringify(generation.token)
|
|
744
|
+
|
|
745
|
+
const resumed = yield* Video.resume(google.video("veo-3.1-generate-preview"), JSON.parse(saved))
|
|
746
|
+
return yield* resumed.await({ poll: { interval: "10 seconds" } })
|
|
747
|
+
})
|
|
748
|
+
|
|
749
|
+
// Progress as a stream: generation-queued | generation-progress | video | finish.
|
|
750
|
+
const events = Video.stream({ model: Runway.configure({ apiKey }).video("gen4.5"), prompt }, { poll })
|
|
751
|
+
```
|
|
752
|
+
|
|
753
|
+
Status polls, result fetches, cancels, and asset downloads all run through the same request executor with the route's
|
|
754
|
+
auth. `Generation.await` and `Generation.events` fail with a
|
|
755
|
+
`Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Failed,
|
|
756
|
+
cancelled, and expired generations fail typed with the provider's terminal document on `reason.body`; moderation
|
|
757
|
+
outcomes (Veo `raiMediaFilteredReasons`, xAI `respect_moderation`, Runway `SAFETY.*` codes) surface as `notices` when
|
|
758
|
+
a video is still returned and as a `ContentPolicy` reason when nothing is.
|
|
759
|
+
|
|
760
|
+
Provider notes:
|
|
761
|
+
|
|
762
|
+
- **Google Veo** takes inline bytes only (materialize `url` assets first); `frames.last` requires `frames.first`;
|
|
763
|
+
audio is always on, so `audio: false` fails typed; one video per request. Output URLs need the API key to
|
|
764
|
+
download, which the returned asset holds transiently (see above).
|
|
765
|
+
- **xAI** sends a `video` input to `/videos/edits`, or `/videos/extensions` with `providerOptions.mode: "extend"`.
|
|
766
|
+
`seed` and `negativePrompt` are not supported.
|
|
767
|
+
- **fal** endpoints are model-specific: `durationSeconds`, `references`, and `frames.last` fail typed and belong in
|
|
768
|
+
`providerOptions` under the model's own names (`duration: "8s"`, `end_image_url`, …). Auth is
|
|
769
|
+
`Authorization: Key <FAL_KEY>`.
|
|
770
|
+
- **Runway** expects pixel ratios in `aspectRatio` for most models (`"1280:720"`), pins `X-Runway-Version`, reports
|
|
771
|
+
`usage: { type: "credits" }`, and its output URLs expire after 24–48 hours.
|
|
772
|
+
|
|
773
|
+
The promise client exposes the same surface: `ai.video.start(...)` resolves to a handle with `await`, `events`,
|
|
774
|
+
`result`, `refresh`, `cancel`, and `token`; `ai.video.generate`, `ai.video.resume(model, token)`, and
|
|
775
|
+
`ai.video.stream` mirror the Effect API. The handle's `status` and `progress` are a snapshot from when it was
|
|
776
|
+
created; `refresh()` resolves to a new handle.
|
|
777
|
+
|
|
778
|
+
```ts
|
|
779
|
+
import { ai } from "@opencode/ai/promise"
|
|
780
|
+
|
|
781
|
+
const generation = await ai.video.start({ model, prompt })
|
|
782
|
+
for await (const event of generation.events({ poll: { interval: 10_000 } })) console.log(event.type)
|
|
783
|
+
const video = await generation.result({ signal })
|
|
784
|
+
await ai.write(video.video, "./kite.mp4")
|
|
785
|
+
```
|
|
786
|
+
|
|
787
|
+
## Speech generation
|
|
788
|
+
|
|
789
|
+
Speech (text-to-speech) is one request whose response is parsed incrementally, so every route supports both
|
|
790
|
+
`Speech.generate` (the whole file) and `Speech.stream` (audio chunks as they arrive). Models come from `.speech(...)`
|
|
791
|
+
selectors on the `OpenAI`, `Google` (Gemini TTS), `ElevenLabs`, `Cartesia`, and `Deepgram` facades. Common fields
|
|
792
|
+
(`voice`, `format`, `speed`, `language`, `instructions`, `timestamps`) lower natively or fail with a typed `AIError`
|
|
793
|
+
before any network call; provider-native controls live under `providerOptions`, inferred from the selected model.
|
|
794
|
+
|
|
795
|
+
```ts
|
|
796
|
+
import { Media, Speech, SpeechClient, SpeechEvent } from "@opencode/ai"
|
|
797
|
+
import { ElevenLabs, OpenAI } from "@opencode/ai/providers"
|
|
798
|
+
|
|
799
|
+
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
|
800
|
+
|
|
801
|
+
// The whole file, written to disk.
|
|
802
|
+
const program = Effect.gen(function* () {
|
|
803
|
+
const response = yield* Speech.generate({
|
|
804
|
+
model: openai.speech("gpt-4o-mini-tts"),
|
|
805
|
+
text: "Hello from OpenCode.",
|
|
806
|
+
voice: "coral",
|
|
807
|
+
format: "mp3",
|
|
808
|
+
instructions: "Warm and unhurried.",
|
|
809
|
+
})
|
|
810
|
+
response.audio // Media.Asset with bytes; headerless PCM carries info.encoding / sampleRate / channels
|
|
811
|
+
response.usage // undefined: OpenAI reports tokens only on SSE streams (Gemini: tokens; ElevenLabs: credits; Deepgram: characters)
|
|
812
|
+
yield* Media.write(response.audio, "hello.mp3")
|
|
813
|
+
})
|
|
814
|
+
|
|
815
|
+
// Chunks as they arrive: audio-delta* (interleaved with timestamps) then one finish carrying the assembled asset.
|
|
816
|
+
const events = Speech.stream({
|
|
817
|
+
model: ElevenLabs.configure({ apiKey }).speech("eleven_flash_v2_5"),
|
|
818
|
+
text: "Hello from OpenCode.",
|
|
819
|
+
voice: "JBFqnCBsd6RMkjVDRZzb",
|
|
820
|
+
format: "pcm",
|
|
821
|
+
timestamps: true,
|
|
822
|
+
}).pipe(
|
|
823
|
+
Stream.tap((event) => {
|
|
824
|
+
if (SpeechEvent.is.audioDelta(event)) return play(event.chunk)
|
|
825
|
+
if (SpeechEvent.is.timestamps(event)) return highlight(event.items) // { text, startSeconds, endSeconds }[]
|
|
826
|
+
return Effect.void
|
|
827
|
+
}),
|
|
828
|
+
)
|
|
829
|
+
```
|
|
830
|
+
|
|
831
|
+
`voice` is the provider's own identifier — a name on OpenAI and Gemini (`"coral"`, `"Kore"`), a voice id on
|
|
832
|
+
ElevenLabs and Cartesia. `{ id }` selects an OpenAI custom voice (`{ id: "voice_1234" }`) and means the same as the
|
|
833
|
+
plain string elsewhere. There is no cross-provider voice catalog. `format` is the container-level word (`mp3`, `wav`,
|
|
834
|
+
`pcm`, `opus`, `aac`, `flac`); sample rates and bitrates live under `providerOptions`, and a value the route cannot
|
|
835
|
+
produce fails as `UnsupportedOperation`. Streams buffer every chunk so `finish` can carry the whole clip.
|
|
836
|
+
|
|
837
|
+
Provider notes:
|
|
838
|
+
|
|
839
|
+
- **OpenAI** streams over SSE (`stream_format: "sse"`), which is also the only place it reports token usage; `tts-1`
|
|
840
|
+
and `tts-1-hd` do not support SSE and stream the raw audio body instead. `pcm` is 24 kHz 16-bit mono. `language`
|
|
841
|
+
and `timestamps` are not supported.
|
|
842
|
+
- **Gemini TTS** returns the provider's default output: WAV for Gemini 3.8 TTS `generate`, raw 16-bit PCM
|
|
843
|
+
(`audio/L16;codec=pcm;rate=24000`) otherwise. `pcm` is the only explicit `format` it accepts, and it fails typed on
|
|
844
|
+
Gemini 3.8 `generate`; the route never wraps PCM as WAV. Style is directed in the text, so `instructions` and
|
|
845
|
+
`speed` fail typed. Only `gemini-3.1-flash-tts-preview` and later support streaming. Two-speaker audio goes through
|
|
846
|
+
`providerOptions.speechConfig.multiSpeakerVoiceConfig`.
|
|
847
|
+
- **ElevenLabs** requires `voice` (the path voice id) and authenticates with `xi-api-key`. `format` maps to the
|
|
848
|
+
`output_format` query parameter (`mp3_44100_128`, `pcm_24000`, `wav_24000`, `opus_48000_64`);
|
|
849
|
+
`providerOptions.outputFormat` sets the exact string. WAV is only available from `generate`. `timestamps: true`
|
|
850
|
+
selects the `with-timestamps` endpoints and yields character-level alignment. `instructions` is not supported.
|
|
851
|
+
- **Cartesia** requires `voice` and pins `Cartesia-Version`. `generate` defaults to MP3 from `/tts/bytes`; streams
|
|
852
|
+
and `timestamps: true` (word-level) use `/tts/sse`, which only serves raw PCM. `providerOptions.sampleRate`,
|
|
853
|
+
`bitRate`, and `encoding` complete `output_format`. No usage is reported.
|
|
854
|
+
- **Deepgram** Aura's voice is the model id (`aura-2-thalia-en`), so `voice` and `language` fail typed. `format`
|
|
855
|
+
and `providerOptions` lower to query parameters (`encoding`, `container`, `sample_rate`, `bit_rate`); `pcm` is
|
|
856
|
+
`linear16` without a container. Auth is `Authorization: Token <DEEPGRAM_API_KEY>`.
|
|
857
|
+
|
|
858
|
+
The promise client mirrors the Effect API; `ai.speech.stream` is an `AsyncIterable`.
|
|
859
|
+
|
|
860
|
+
```ts
|
|
861
|
+
import { ai } from "@opencode/ai/promise"
|
|
862
|
+
|
|
863
|
+
const response = await ai.speech.generate({ model, text: "Hello from OpenCode.", voice: "coral" })
|
|
864
|
+
await ai.write(response.audio, "hello.mp3")
|
|
865
|
+
|
|
866
|
+
for await (const event of ai.speech.stream({ model, text: "Hello from OpenCode.", voice: "coral" })) {
|
|
867
|
+
if (event.type === "audio-delta") player.write(event.chunk)
|
|
868
|
+
}
|
|
869
|
+
```
|
|
870
|
+
|
|
871
|
+
## Transcription
|
|
872
|
+
|
|
873
|
+
Transcription (speech-to-text) is the one modality whose providers use every route kind: OpenAI and Gemini stream,
|
|
874
|
+
Deepgram answers inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream` work on all of
|
|
875
|
+
them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with `UnsupportedOperation`
|
|
876
|
+
elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`, `Deepgram`, and `AssemblyAI`
|
|
877
|
+
facades. Common fields (`language`, `prompt`, `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower
|
|
878
|
+
natively or fail with a typed `AIError` before any network call; a route may return more than asked.
|
|
879
|
+
|
|
880
|
+
```ts
|
|
881
|
+
import { Console, Effect, Stream } from "effect"
|
|
882
|
+
import { Media, Transcription, TranscriptionEvent } from "@opencode/ai"
|
|
883
|
+
import { AssemblyAI, Deepgram, OpenAI } from "@opencode/ai/providers"
|
|
884
|
+
|
|
885
|
+
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
|
886
|
+
|
|
887
|
+
const program = Effect.gen(function* () {
|
|
888
|
+
const audio = yield* Media.file("./call.mp3")
|
|
889
|
+
|
|
890
|
+
// Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0").
|
|
891
|
+
const response = yield* Transcription.generate({
|
|
892
|
+
model: Deepgram.configure({ apiKey }).transcription("nova-3"),
|
|
893
|
+
audio,
|
|
894
|
+
diarize: true,
|
|
895
|
+
timestamps: "word",
|
|
896
|
+
})
|
|
897
|
+
response.text // "Hello from OpenCode."
|
|
898
|
+
response.segments // [{ text, startSeconds, endSeconds, speaker: "0" }]
|
|
899
|
+
response.words // [{ text, startSeconds, endSeconds, speaker, confidence }]
|
|
900
|
+
response.language // the provider's own value, lowercased ("en", "english", "en_us")
|
|
901
|
+
|
|
902
|
+
// Text deltas as the model transcribes, then one finish carrying the whole transcript.
|
|
903
|
+
yield* Transcription.stream({ model: openai.transcription("gpt-4o-mini-transcribe"), audio }).pipe(
|
|
904
|
+
Stream.tap((event) => (TranscriptionEvent.is.textDelta(event) ? Console.log(event.delta) : Effect.void)),
|
|
905
|
+
Stream.runDrain,
|
|
906
|
+
)
|
|
907
|
+
|
|
908
|
+
// Queued: persist the token with the provider and model ID (the token alone cannot pick the model), resume, and await.
|
|
909
|
+
const model = AssemblyAI.configure({ apiKey }).transcription("universal-3-5-pro")
|
|
910
|
+
const generation = yield* Transcription.start({ model, audio })
|
|
911
|
+
const resumed = yield* Transcription.resume(model, JSON.parse(JSON.stringify(generation.token)))
|
|
912
|
+
const transcript = yield* resumed.await({ poll: { interval: "3 seconds" } })
|
|
913
|
+
})
|
|
914
|
+
```
|
|
915
|
+
|
|
916
|
+
Inline routes emit only `finish` from `stream` (no faked deltas); queued routes emit `generation-queued` /
|
|
917
|
+
`generation-progress` before it.
|
|
918
|
+
|
|
919
|
+
Provider notes:
|
|
920
|
+
|
|
921
|
+
- **OpenAI** takes inline audio only; `diarize` needs `gpt-4o-transcribe-diarize`, timestamps need `whisper-1`, and `whisper-1` does not stream.
|
|
922
|
+
- **Gemini** needs a transcribe model (`gemini-3.5-transcribe`); `prompt` and `speakers` fail typed.
|
|
923
|
+
- **Deepgram** detects the language unless `language` is set; vocabulary goes in `providerOptions.keyterm`.
|
|
924
|
+
- **AssemblyAI** uploads inline audio before submitting and is the only route that accepts `speakers`.
|
|
925
|
+
|
|
926
|
+
The promise client mirrors the Effect API:
|
|
927
|
+
|
|
928
|
+
```ts
|
|
929
|
+
const audio = await ai.file("./call.mp3")
|
|
930
|
+
const text = (await ai.transcription.generate({ model, audio })).text
|
|
931
|
+
for await (const event of ai.transcription.stream({ model, audio })) if (event.type === "text-delta") write(event.delta)
|
|
932
|
+
const generation = await ai.transcription.start({ model: assemblyai, audio })
|
|
933
|
+
const transcript = await generation.await({ poll: { interval: 3_000 } })
|
|
934
|
+
```
|
|
604
935
|
|
|
605
936
|
## Public API
|
|
606
937
|
|
|
607
938
|
- **`LLM.request({...})`** — build a provider-neutral `LLMRequest`. Accepts ergonomic inputs (`system: string`, `prompt: string`) that normalize into the canonical Schema classes.
|
|
608
|
-
- **`LLM.generate` / `LLM.stream`** —
|
|
939
|
+
- **`LLM.generate` / `LLM.stream`** — run direct input or an `LLMRequest` through `LLMClient` for one-import use.
|
|
609
940
|
- **`Message.user(...)` / `Message.assistant(...)` / `Message.tool(...)`** — message constructors from the canonical schema model.
|
|
610
941
|
- **`LanguageModel.make(...)` / `ToolCallPart.make(...)` / `ToolResultPart.make(...)` / `ToolDefinition.make(...)`** — model and tool-related constructors from the canonical schema model.
|
|
611
942
|
- **`LLMEvent.is.*`** — typed guards (`is.textDelta`, `is.toolCall`, `is.finish`, …) for filtering streams.
|
|
612
|
-
- **`Image.request` / `
|
|
943
|
+
- **`Image.request` / `generate` / `stream` / `start` / `resume`** — images over inline, streaming (partial previews), and queued routes through a provider-neutral request and response model.
|
|
613
944
|
- **`ImageClient`** — Effect service and layer for image execution, parallel to `LLMClient`.
|
|
614
945
|
- **`Media`** — the shared asset type (`Media.Asset`, `Media.Source`) and constructors used by messages, tool results, and media requests.
|
|
615
946
|
- **`Generation`** — provider-neutral handle for an in-flight media generation (`await`, `refresh`, `cancel`, `events`) used by queued media routes.
|
|
616
|
-
-
|
|
947
|
+
- **`Video.request` / `generate` / `stream` / `start` / `resume`** — queued video generation through a provider-neutral request; `VideoClient` is its Effect service and layer.
|
|
948
|
+
- **`Speech.request` / `Speech.generate` / `Speech.stream`** — text-to-speech through a provider-neutral request; `SpeechClient` is its Effect service and layer.
|
|
949
|
+
- **`Transcription.request` / `generate` / `stream` / `start` / `resume`** — speech-to-text over inline, streaming, and queued routes; `TranscriptionClient` is its Effect service and layer.
|
|
950
|
+
- **`AIClient.layer` / `AIClient.layerWith(executor)`** — every modality client plus the request executor in one layer.
|
|
951
|
+
- **`@opencode/ai/promise`** — `AI.make({ layer? })` and a default `ai` client exposing `llm`, `image`, `video`, `speech`, and `transcription` as Promise / `AsyncIterable` APIs, plus `file`, `write`, `bytes`, `base64`, and `materialize` for assets.
|
|
617
952
|
|
|
618
953
|
## Testing
|
|
619
954
|
|
|
@@ -676,11 +1011,13 @@ This is different from prompt caching, server-side history storage, or truncatio
|
|
|
676
1011
|
Prefer this operation, where supported, when the application owns compaction policy and durable context updates.
|
|
677
1012
|
|
|
678
1013
|
```ts
|
|
679
|
-
const
|
|
680
|
-
const
|
|
681
|
-
|
|
1014
|
+
const compacted = Effect.gen(function* () {
|
|
1015
|
+
const result = yield* LLMClient.compact(request)
|
|
1016
|
+
const next = LLMRequest.update(request, {
|
|
1017
|
+
messages: result.replacement,
|
|
1018
|
+
})
|
|
1019
|
+
return yield* LLMClient.generate(next)
|
|
682
1020
|
})
|
|
683
|
-
const response = yield * LLMClient.generate(next)
|
|
684
1021
|
```
|
|
685
1022
|
|
|
686
1023
|
`replacement` replaces the complete input window. Do not append it to the original transcript or extract only the encrypted item: the provider may retain additional messages in its output. Retained user and assistant messages remain ordinary messages with typed text, media, or reasoning parts, in their original order. Provider-specific message IDs, status, and phase use `providerMetadata`, not a raw output array hidden in an assistant message. Unsupported returned item types fail explicitly.
|
|
@@ -696,16 +1033,16 @@ The input must still fit the model's context window. Explicit compaction is not
|
|
|
696
1033
|
OpenAI Responses also exposes a separate, explicitly selected mechanism:
|
|
697
1034
|
|
|
698
1035
|
```ts
|
|
699
|
-
const
|
|
700
|
-
yield
|
|
701
|
-
LLMClient.compact(request, {
|
|
1036
|
+
const checkpoint = Effect.gen(function* () {
|
|
1037
|
+
const result = yield* LLMClient.compact(request, {
|
|
702
1038
|
mechanism: "trigger",
|
|
703
1039
|
webSocket, // Optional: without it, the request uses HTTP/SSE.
|
|
704
1040
|
})
|
|
705
1041
|
|
|
706
|
-
result.checkpoint // Successful encrypted CompactionPart.
|
|
707
|
-
result.responseID
|
|
708
|
-
result.usage
|
|
1042
|
+
result.checkpoint // Successful encrypted CompactionPart.
|
|
1043
|
+
result.responseID
|
|
1044
|
+
result.usage
|
|
1045
|
+
})
|
|
709
1046
|
```
|
|
710
1047
|
|
|
711
1048
|
This appends a native `compaction_trigger` control item to the full input and sends a normal Responses request, with tools and instructions retained, `stream: true`, `store: false`, and parallel tool calls enabled. It removes normal-answer text/output-format controls, forced tool choices, output-token/tool-call limits, and automatic `context_management`. Body overlays cannot replace `input` or supply `previous_response_id`/`conversation`; the complete canonical history is required for safe stateless replay. Request metadata, auth, headers, query parameters, service tier, and supported prompt-cache settings are preserved.
|
|
@@ -719,9 +1056,11 @@ The supplied WebSocket executor can reuse a compatible append baseline for the c
|
|
|
719
1056
|
Trigger support is separate from endpoint support. Only the OpenAI Responses route advertises it; Azure, xAI, Chat, and compatible Responses routes do not inherit it. Untyped calls still fail before sending: missing route capabilities return `UnsupportedOperation`, while unknown mechanism names and invalid inputs return `InvalidRequest`. Dynamic callers must narrow for the selected mechanism:
|
|
720
1057
|
|
|
721
1058
|
```ts
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
}
|
|
1059
|
+
const narrowed = Effect.gen(function* () {
|
|
1060
|
+
if (LLMClient.canCompact(request, { mechanism: "trigger" })) {
|
|
1061
|
+
const result = yield* LLMClient.compact(request, { mechanism: "trigger" })
|
|
1062
|
+
}
|
|
1063
|
+
})
|
|
725
1064
|
```
|
|
726
1065
|
|
|
727
1066
|
This capability describes protocol implementation, **not universal availability on OpenAI API deployments**. The host application owns subscription/deployment eligibility, OAuth, endpoint selection, and deployment-specific headers. Local protocol/socket tests do not establish live provider support.
|
|
@@ -730,9 +1069,10 @@ This capability describes protocol implementation, **not universal availability
|
|
|
730
1069
|
|
|
731
1070
|
`providerOptions.contextManagement` lets the provider decide when to compact during an ordinary `generate` or `stream` call. This is an advanced option for callers that own persistence and recovery: persist the complete assistant message, including its checkpoint, before continuing. Enabling the option does not provide durable checkpoint storage, interruption recovery, or model-switch policy. Keep the prior context until a successful checkpoint has been persisted.
|
|
732
1071
|
|
|
733
|
-
|
|
1072
|
+
Enable OpenAI compaction with typed provider options:
|
|
734
1073
|
|
|
735
1074
|
```ts
|
|
1075
|
+
import { Effect } from "effect"
|
|
736
1076
|
import { LLM, LLMClient, LLMRequest, Message } from "@opencode/ai"
|
|
737
1077
|
import { OpenAI } from "@opencode/ai/providers"
|
|
738
1078
|
|
|
@@ -743,9 +1083,11 @@ const request = LLM.request({
|
|
|
743
1083
|
contextManagement: [{ type: "compaction", compactThreshold: 200_000 }],
|
|
744
1084
|
},
|
|
745
1085
|
})
|
|
746
|
-
const
|
|
747
|
-
const
|
|
748
|
-
|
|
1086
|
+
const continued = Effect.gen(function* () {
|
|
1087
|
+
const response = yield* LLMClient.generate(request)
|
|
1088
|
+
return LLMRequest.update(request, {
|
|
1089
|
+
messages: [...request.messages, response.message, Message.user("Continue")],
|
|
1090
|
+
})
|
|
749
1091
|
})
|
|
750
1092
|
```
|
|
751
1093
|
|
|
@@ -973,7 +1315,7 @@ Compose a route with `Route.make({ protocol, endpoint, auth, framing, ... })`. T
|
|
|
973
1315
|
|
|
974
1316
|
## Effect
|
|
975
1317
|
|
|
976
|
-
This package is built on Effect. Public methods return `Effect` or `Stream`; provide `
|
|
1318
|
+
This package is built on Effect. Public methods return `Effect` or `Stream`; provide `AIClient.layer` (or `AIClient.layerWith(executor)`) for every modality, then import the provider/protocol modules for the routes you use. The example at `example/tutorial.ts` is a runnable walkthrough.
|
|
977
1319
|
|
|
978
1320
|
## See also
|
|
979
1321
|
|