@opencode/ai 2.0.15 → 2.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. package/README.md +286 -2
  2. package/dist/generation.d.ts +36 -22
  3. package/dist/generation.js +53 -24
  4. package/dist/image-client.d.ts +14 -7
  5. package/dist/image-client.js +24 -8
  6. package/dist/image.d.ts +398 -47
  7. package/dist/image.js +47 -45
  8. package/dist/index.d.ts +13 -1
  9. package/dist/index.js +9 -0
  10. package/dist/media-model.d.ts +44 -0
  11. package/dist/media-model.js +49 -0
  12. package/dist/media.d.ts +10 -9
  13. package/dist/media.js +9 -10
  14. package/dist/promise.d.ts +428 -8
  15. package/dist/promise.js +40 -3
  16. package/dist/protocols/alibaba-chat.d.ts +12 -0
  17. package/dist/protocols/alibaba-responses.d.ts +2 -2
  18. package/dist/protocols/anthropic-messages.js +1 -2
  19. package/dist/protocols/assemblyai-transcription.d.ts +40 -0
  20. package/dist/protocols/assemblyai-transcription.js +138 -0
  21. package/dist/protocols/bedrock-converse.js +5 -11
  22. package/dist/protocols/bfl-images.d.ts +32 -0
  23. package/dist/protocols/bfl-images.js +153 -0
  24. package/dist/protocols/cartesia-speech.d.ts +127 -0
  25. package/dist/protocols/cartesia-speech.js +126 -0
  26. package/dist/protocols/deepgram-speech.d.ts +119 -0
  27. package/dist/protocols/deepgram-speech.js +92 -0
  28. package/dist/protocols/deepgram-transcription.d.ts +25 -0
  29. package/dist/protocols/deepgram-transcription.js +129 -0
  30. package/dist/protocols/elevenlabs-speech.d.ts +122 -0
  31. package/dist/protocols/elevenlabs-speech.js +115 -0
  32. package/dist/protocols/fal-images.d.ts +24 -0
  33. package/dist/protocols/fal-images.js +114 -0
  34. package/dist/protocols/fal-video.d.ts +29 -0
  35. package/dist/protocols/fal-video.js +88 -0
  36. package/dist/protocols/gemini.d.ts +9 -9
  37. package/dist/protocols/gemini.js +8 -34
  38. package/dist/protocols/google-images.js +2 -14
  39. package/dist/protocols/google-speech.d.ts +130 -0
  40. package/dist/protocols/google-speech.js +84 -0
  41. package/dist/protocols/google-transcription.d.ts +173 -0
  42. package/dist/protocols/google-transcription.js +138 -0
  43. package/dist/protocols/google-video.d.ts +26 -0
  44. package/dist/protocols/google-video.js +158 -0
  45. package/dist/protocols/meta-images.js +2 -9
  46. package/dist/protocols/meta-responses.d.ts +4 -4
  47. package/dist/protocols/meta-responses.js +1 -1
  48. package/dist/protocols/open-responses.d.ts +6 -6
  49. package/dist/protocols/open-responses.js +1 -2
  50. package/dist/protocols/openai-chat.d.ts +84 -0
  51. package/dist/protocols/openai-chat.js +26 -14
  52. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  53. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  54. package/dist/protocols/openai-images.d.ts +124 -3
  55. package/dist/protocols/openai-images.js +107 -54
  56. package/dist/protocols/openai-responses.d.ts +15 -15
  57. package/dist/protocols/openai-responses.js +5 -6
  58. package/dist/protocols/openai-speech.d.ts +116 -0
  59. package/dist/protocols/openai-speech.js +98 -0
  60. package/dist/protocols/openai-transcription.d.ts +207 -0
  61. package/dist/protocols/openai-transcription.js +190 -0
  62. package/dist/protocols/replicate-images.d.ts +28 -0
  63. package/dist/protocols/replicate-images.js +133 -0
  64. package/dist/protocols/runway-video.d.ts +38 -0
  65. package/dist/protocols/runway-video.js +146 -0
  66. package/dist/protocols/shared.d.ts +13 -3
  67. package/dist/protocols/shared.js +23 -3
  68. package/dist/protocols/stability-images.d.ts +38 -0
  69. package/dist/protocols/stability-images.js +148 -0
  70. package/dist/protocols/utils/fal-queue.d.ts +28 -0
  71. package/dist/protocols/utils/fal-queue.js +69 -0
  72. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  73. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  74. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  75. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  76. package/dist/protocols/utils/media-input.d.ts +8 -0
  77. package/dist/protocols/utils/media-input.js +18 -0
  78. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  79. package/dist/protocols/utils/speech-stream.js +67 -0
  80. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  81. package/dist/protocols/utils/tool-schema.js +40 -17
  82. package/dist/protocols/xai-images.js +1 -12
  83. package/dist/protocols/xai-responses.d.ts +2 -2
  84. package/dist/protocols/xai-video.d.ts +34 -0
  85. package/dist/protocols/xai-video.js +147 -0
  86. package/dist/protocols/zai-chat.d.ts +13 -1
  87. package/dist/provider-error.js +3 -0
  88. package/dist/providers/alibaba.d.ts +14 -2
  89. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  90. package/dist/providers/assemblyai.d.ts +25 -0
  91. package/dist/providers/assemblyai.js +29 -0
  92. package/dist/providers/azure.d.ts +18 -6
  93. package/dist/providers/baseten.d.ts +24 -0
  94. package/dist/providers/black-forest-labs.d.ts +25 -0
  95. package/dist/providers/black-forest-labs.js +28 -0
  96. package/dist/providers/cartesia.d.ts +24 -0
  97. package/dist/providers/cartesia.js +22 -0
  98. package/dist/providers/cerebras.d.ts +24 -0
  99. package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
  100. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  101. package/dist/providers/deepgram.d.ts +29 -0
  102. package/dist/providers/deepgram.js +31 -0
  103. package/dist/providers/deepinfra.d.ts +24 -0
  104. package/dist/providers/deepseek.d.ts +24 -0
  105. package/dist/providers/elevenlabs.d.ts +24 -0
  106. package/dist/providers/elevenlabs.js +28 -0
  107. package/dist/providers/fal.d.ts +29 -0
  108. package/dist/providers/fal.js +33 -0
  109. package/dist/providers/fireworks.d.ts +24 -0
  110. package/dist/providers/google-vertex-chat.d.ts +12 -0
  111. package/dist/providers/google-vertex-responses.d.ts +2 -2
  112. package/dist/providers/google-vertex.d.ts +3 -3
  113. package/dist/providers/google.d.ts +18 -3
  114. package/dist/providers/google.js +11 -2
  115. package/dist/providers/groq.d.ts +24 -0
  116. package/dist/providers/index.d.ts +9 -0
  117. package/dist/providers/index.js +9 -0
  118. package/dist/providers/meta.d.ts +14 -2
  119. package/dist/providers/minimax.d.ts +14 -2
  120. package/dist/providers/moonshot.d.ts +14 -2
  121. package/dist/providers/moonshot.js +3 -3
  122. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  123. package/dist/providers/openai-compatible.d.ts +12 -0
  124. package/dist/providers/openai.d.ts +25 -3
  125. package/dist/providers/openai.js +10 -1
  126. package/dist/providers/openrouter.d.ts +48 -0
  127. package/dist/providers/replicate.d.ts +25 -0
  128. package/dist/providers/replicate.js +22 -0
  129. package/dist/providers/runway.d.ts +24 -0
  130. package/dist/providers/runway.js +22 -0
  131. package/dist/providers/stability.d.ts +28 -0
  132. package/dist/providers/stability.js +23 -0
  133. package/dist/providers/togetherai.d.ts +24 -0
  134. package/dist/providers/xai.d.ts +17 -0
  135. package/dist/providers/xai.js +5 -2
  136. package/dist/providers/zai-coding-plan.d.ts +15 -3
  137. package/dist/providers/zai.d.ts +13 -1
  138. package/dist/route/auth.d.ts +4 -1
  139. package/dist/route/auth.js +6 -0
  140. package/dist/route/framing.d.ts +5 -1
  141. package/dist/route/framing.js +9 -0
  142. package/dist/route/media-protocol.d.ts +116 -3
  143. package/dist/route/media-protocol.js +60 -4
  144. package/dist/route/media.d.ts +58 -7
  145. package/dist/route/media.js +201 -29
  146. package/dist/schema/events.d.ts +0 -6
  147. package/dist/schema/messages.d.ts +0 -3
  148. package/dist/schema/options.d.ts +4 -3
  149. package/dist/schema/options.js +3 -2
  150. package/dist/speech-client.d.ts +21 -0
  151. package/dist/speech-client.js +25 -0
  152. package/dist/speech.d.ts +1150 -0
  153. package/dist/speech.js +119 -0
  154. package/dist/transcription-client.d.ts +28 -0
  155. package/dist/transcription-client.js +44 -0
  156. package/dist/transcription.d.ts +1504 -0
  157. package/dist/transcription.js +133 -0
  158. package/dist/utils/bytes.d.ts +1 -0
  159. package/dist/utils/bytes.js +10 -0
  160. package/dist/utils/media-type.d.ts +1 -0
  161. package/dist/utils/media-type.js +22 -1
  162. package/dist/video-client.d.ts +28 -0
  163. package/dist/video-client.js +40 -0
  164. package/dist/video.d.ts +1359 -0
  165. package/dist/video.js +119 -0
  166. package/package.json +3 -3
  167. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  168. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
package/README.md CHANGED
@@ -584,6 +584,61 @@ Z.ai does not include trustworthy MIME metadata for output URLs, so generated im
584
584
  `application/octet-stream` until materialized. Output URLs expire after 30 days; call `asset.materialize()` and
585
585
  persist the bytes promptly if they must remain available.
586
586
 
587
+ ### Partial images
588
+
589
+ OpenAI's GPT image models stream previews. `Image.stream` sends `stream: true` with `partialImages` (0–3, default 2)
590
+ and emits `image-partial` events before each final `image`; `Image.generate` keeps the plain JSON request.
591
+ `dall-e-*` models do not stream and fail typed:
592
+
593
+ ```ts
594
+ yield *
595
+ Image.stream({
596
+ model: openai.image("gpt-image-2"),
597
+ prompt: "A lighthouse at dusk",
598
+ providerOptions: { partialImages: 2 },
599
+ }).pipe(Stream.runForEach((event) => (ImageEvent.is.imagePartial(event) ? showPreview(event.image) : Effect.void)))
600
+ ```
601
+
602
+ The provider may send fewer previews than requested when the final image is ready first.
603
+
604
+ ### Queued image providers
605
+
606
+ Black Forest Labs, fal, Replicate, and Stability's creative upscaler are submit-then-poll routes. `Image.generate`
607
+ and `Image.stream` poll for you (pass `{ poll }` to tune the interval and timeout); `Image.start` returns a
608
+ `Generation` whose `token` is serializable JSON for `Image.resume` in another process:
609
+
610
+ ```ts
611
+ import { BlackForestLabs, Stability } from "@opencode/ai/providers"
612
+
613
+ const bfl = BlackForestLabs.configure({ apiKey: process.env.BFL_API_KEY })
614
+
615
+ const generation = yield * Image.start({ model: bfl.image("flux-2-pro"), prompt, size: "1024x768" })
616
+ persist(generation.token)
617
+
618
+ const resumed = yield * Image.resume(bfl.image("flux-2-pro"), loadToken())
619
+ const response = yield * resumed.await({ poll: { interval: "2 seconds" } })
620
+ ```
621
+
622
+ - **Black Forest Labs** — results are downloaded before returning, because `result.sample` expires in 10 minutes.
623
+ - **Replicate** — inputs are model-defined, so only `prompt` lowers: sizing, count, seed, format, and files go in
624
+ `providerOptions` under the model's names, with files as `Media.Asset` (data URLs up to 256 KB, larger by URL).
625
+ Outputs are removed an hour after the prediction completes. `Prefer: wait=60` in `headers` or `http.headers` holds
626
+ the submission open so a fast prediction costs one result read.
627
+ - **Stability** — `stability.image(id)` generates inline; `stability.upscale()` is the creative upscaler, queued:
628
+
629
+ ```ts
630
+ const stability = Stability.configure({ apiKey: process.env.STABILITY_API_KEY })
631
+ const upscaled =
632
+ yield *
633
+ Image.generate(
634
+ { model: stability.upscale(), prompt: "A lighthouse", images: [yield * Media.file("./small.png")] },
635
+ { poll: { interval: "5 seconds" } },
636
+ )
637
+ ```
638
+
639
+ Imagen is not available: Google shut it down on the Gemini API, and Vertex discontinued the Imagen 4 models on
640
+ 2026-06-30. `Google.image(...)` uses Gemini-native image models.
641
+
587
642
  Conversational image generation remains part of the LLM interaction. OpenAI Responses exposes it through its hosted image tool:
588
643
 
589
644
  ```ts
@@ -602,6 +657,233 @@ const program = Effect.gen(function* () {
602
657
 
603
658
  The hosted result is represented as a provider-executed tool call and tool result, and the generated image is also emitted as a first-class `media` `LLMEvent` (`response.message` then carries a `media` part). Gemini image-capable models emit the same `media` event for inline image output. Retaining `response.message` preserves the generated image for continuation on both routes.
604
659
 
660
+ ## Video generation
661
+
662
+ Video mirrors `Image` with one difference: every provider is asynchronous, so the route is a submit-then-poll
663
+ `Generation`. Models come from `.video(...)` selectors on the `Google` (Veo), `XAI`, `Fal`, and `Runway` facades.
664
+ Common fields (`frames`, `references`, `video`, `durationSeconds`, `aspectRatio`, `resolution`, `audio`, `n`, `seed`,
665
+ `negativePrompt`) lower natively or fail with a typed `AIError` before any network call; provider-native controls live
666
+ under `providerOptions`, inferred from the selected model.
667
+
668
+ ```ts
669
+ import { Video, VideoClient } from "@opencode/ai"
670
+ import { Google } from "@opencode/ai/providers"
671
+
672
+ const google = Google.configure({ apiKey: process.env.GOOGLE_GENERATIVE_AI_API_KEY })
673
+
674
+ // Simple: submit and wait.
675
+ const program = Effect.gen(function* () {
676
+ const response = yield* Video.generate(
677
+ {
678
+ model: google.video("veo-3.1-generate-preview"),
679
+ prompt: "Panning wide shot of a calico kitten sleeping in the sunshine",
680
+ aspectRatio: "16:9",
681
+ resolution: "1080p",
682
+ durationSeconds: 8,
683
+ providerOptions: { personGeneration: "allow_adult" },
684
+ },
685
+ { poll: { interval: "10 seconds", timeout: "10 minutes" } },
686
+ )
687
+ // Veo serves files for two days behind the API key. The asset knows the deadline (`expiresAt`) and carries the
688
+ // download credentials only on the live instance (`asset.headers`), never in `source` or JSON: materialize
689
+ // before persisting, or the persisted URL cannot be fetched again.
690
+ return yield* response.video.materialize()
691
+ })
692
+
693
+ // Explicit control: keep the handle, persist the token, resume elsewhere.
694
+ const controlled = Effect.gen(function* () {
695
+ const generation = yield* Video.start({ model: google.video("veo-3.1-generate-preview"), prompt })
696
+ generation.id // provider operation / task / request id
697
+ generation.status // "queued" | "running" | "completed" | "failed" | "cancelled" | "expired"
698
+ generation.token // route-owned JSON: `{ operation }`, `{ requestID }`, `{ taskID }`, or fal's follow-up URLs
699
+ const saved = JSON.stringify(generation.token)
700
+
701
+ const resumed = yield* Video.resume(google.video("veo-3.1-generate-preview"), JSON.parse(saved))
702
+ return yield* resumed.await({ poll: { interval: "10 seconds" } })
703
+ })
704
+
705
+ // Progress as a stream: generation-queued | generation-progress | video | finish.
706
+ const events = Video.stream({ model: Runway.configure({ apiKey }).video("gen4.5"), prompt }, { poll })
707
+ ```
708
+
709
+ `VideoClient.layer` needs `RequestExecutor.Service`, and status polls, result fetches, cancels, and asset downloads
710
+ all run through the same executor with the route's auth. `Generation.await` and `Generation.events` fail with a
711
+ `Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Failed,
712
+ cancelled, and expired generations fail typed with the provider's terminal document on `reason.body`; moderation
713
+ outcomes (Veo `raiMediaFilteredReasons`, xAI `respect_moderation`, Runway `SAFETY.*` codes) surface as `notices` when
714
+ a video is still returned and as a `ContentPolicy` reason when nothing is.
715
+
716
+ Provider notes:
717
+
718
+ - **Google Veo** takes inline bytes only (materialize `url` assets first); `frames.last` requires `frames.first`;
719
+ audio is always on, so `audio: false` fails typed; one video per request. Output URLs need the API key to
720
+ download, which the returned asset holds transiently (see above).
721
+ - **xAI** sends a `video` input to `/videos/edits`, or `/videos/extensions` with `providerOptions.mode: "extend"`.
722
+ `seed` and `negativePrompt` are not supported.
723
+ - **fal** endpoints are model-specific: `durationSeconds`, `references`, and `frames.last` fail typed and belong in
724
+ `providerOptions` under the model's own names (`duration: "8s"`, `end_image_url`, …). Auth is
725
+ `Authorization: Key <FAL_KEY>`.
726
+ - **Runway** expects pixel ratios in `aspectRatio` for most models (`"1280:720"`), pins `X-Runway-Version`, reports
727
+ `usage: { type: "credits" }`, and its output URLs expire after 24–48 hours.
728
+
729
+ The promise client exposes the same surface: `ai.video.start(...)` resolves to a handle with `await`, `refresh`,
730
+ `cancel`, and `token`; `ai.video.generate`, `ai.video.resume(model, token)`, and `ai.video.stream` mirror the Effect
731
+ API.
732
+
733
+ ```ts
734
+ import { ai } from "@opencode/ai/promise"
735
+
736
+ const generation = await ai.video.start({ model, prompt })
737
+ const video = await generation.await({ poll: { interval: 10_000 }, signal })
738
+ ```
739
+
740
+ ## Speech generation
741
+
742
+ Speech (text-to-speech) is one request whose response is parsed incrementally, so every route supports both
743
+ `Speech.generate` (the whole file) and `Speech.stream` (audio chunks as they arrive). Models come from `.speech(...)`
744
+ selectors on the `OpenAI`, `Google` (Gemini TTS), `ElevenLabs`, `Cartesia`, and `Deepgram` facades. Common fields
745
+ (`voice`, `format`, `speed`, `language`, `instructions`, `timestamps`) lower natively or fail with a typed `AIError`
746
+ before any network call; provider-native controls live under `providerOptions`, inferred from the selected model.
747
+
748
+ ```ts
749
+ import { Media, Speech, SpeechClient, SpeechEvent } from "@opencode/ai"
750
+ import { ElevenLabs, OpenAI } from "@opencode/ai/providers"
751
+
752
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
753
+
754
+ // The whole file, written to disk.
755
+ const program = Effect.gen(function* () {
756
+ const response = yield* Speech.generate({
757
+ model: openai.speech("gpt-4o-mini-tts"),
758
+ text: "Hello from OpenCode.",
759
+ voice: "coral",
760
+ format: "mp3",
761
+ instructions: "Warm and unhurried.",
762
+ })
763
+ response.audio // Media.Asset with bytes; headerless PCM carries info.encoding / sampleRate / channels
764
+ response.usage // undefined: OpenAI reports tokens only on SSE streams (Gemini: tokens; ElevenLabs: credits; Deepgram: characters)
765
+ yield* Media.write(response.audio, "hello.mp3")
766
+ })
767
+
768
+ // Chunks as they arrive: audio-delta* (interleaved with timestamps) then one finish carrying the assembled asset.
769
+ const events = Speech.stream({
770
+ model: ElevenLabs.configure({ apiKey }).speech("eleven_flash_v2_5"),
771
+ text: "Hello from OpenCode.",
772
+ voice: "JBFqnCBsd6RMkjVDRZzb",
773
+ format: "pcm",
774
+ timestamps: true,
775
+ }).pipe(
776
+ Stream.tap((event) => {
777
+ if (SpeechEvent.is.audioDelta(event)) return play(event.chunk)
778
+ if (SpeechEvent.is.timestamps(event)) return highlight(event.items) // { text, startSeconds, endSeconds }[]
779
+ return Effect.void
780
+ }),
781
+ )
782
+ ```
783
+
784
+ `voice` is the provider's own identifier — a name on OpenAI and Gemini (`"coral"`, `"Kore"`), a voice id on
785
+ ElevenLabs and Cartesia. `{ id }` selects an OpenAI custom voice (`{ id: "voice_1234" }`) and means the same as the
786
+ plain string elsewhere. There is no cross-provider voice catalog. `format` is the container-level word (`mp3`, `wav`,
787
+ `pcm`, `opus`, `aac`, `flac`); sample rates and bitrates live under `providerOptions`, and a value the route cannot
788
+ produce fails as `UnsupportedOperation`. Streams buffer every chunk so `finish` can carry the whole clip.
789
+ `SpeechClient.layer` needs `RequestExecutor.Service`.
790
+
791
+ Provider notes:
792
+
793
+ - **OpenAI** streams over SSE (`stream_format: "sse"`), which is also the only place it reports token usage; `tts-1`
794
+ and `tts-1-hd` do not support SSE and stream the raw audio body instead. `pcm` is 24 kHz 16-bit mono. `language`
795
+ and `timestamps` are not supported.
796
+ - **Gemini TTS** returns raw 16-bit PCM only (`audio/L16;codec=pcm;rate=24000`), so any `format` other than `pcm`
797
+ fails typed; wrap the samples yourself. Style is directed in the text, so `instructions` and `speed` fail typed.
798
+ Only `gemini-3.1-flash-tts-preview` and later support streaming. Two-speaker audio goes through
799
+ `providerOptions.speechConfig.multiSpeakerVoiceConfig`.
800
+ - **ElevenLabs** requires `voice` (the path voice id) and authenticates with `xi-api-key`. `format` maps to the
801
+ `output_format` query parameter (`mp3_44100_128`, `pcm_24000`, `wav_24000`, `opus_48000_64`);
802
+ `providerOptions.outputFormat` sets the exact string. WAV is only available from `generate`. `timestamps: true`
803
+ selects the `with-timestamps` endpoints and yields character-level alignment. `instructions` is not supported.
804
+ - **Cartesia** requires `voice` and pins `Cartesia-Version`. `generate` defaults to MP3 from `/tts/bytes`; streams
805
+ and `timestamps: true` (word-level) use `/tts/sse`, which only serves raw PCM. `providerOptions.sampleRate`,
806
+ `bitRate`, and `encoding` complete `output_format`. No usage is reported.
807
+ - **Deepgram** Aura's voice is the model id (`aura-2-thalia-en`), so `voice` and `language` fail typed. `format`
808
+ and `providerOptions` lower to query parameters (`encoding`, `container`, `sample_rate`, `bit_rate`); `pcm` is
809
+ `linear16` without a container. Auth is `Authorization: Token <DEEPGRAM_API_KEY>`.
810
+
811
+ The promise client mirrors the Effect API; `ai.speech.stream` is an `AsyncIterable`.
812
+
813
+ ```ts
814
+ import { ai } from "@opencode/ai/promise"
815
+
816
+ const response = await ai.speech.generate({ model, text: "Hello from OpenCode.", voice: "coral" })
817
+ await Bun.write("hello.mp3", await ai.run(response.audio.bytes()))
818
+
819
+ for await (const event of ai.speech.stream({ model, text: "Hello from OpenCode.", voice: "coral" })) {
820
+ if (event.type === "audio-delta") player.write(event.chunk)
821
+ }
822
+ ```
823
+
824
+ ## Transcription
825
+
826
+ Transcription (speech-to-text) is the one modality whose providers use every route kind: OpenAI and Gemini stream,
827
+ Deepgram answers inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream` work on all of
828
+ them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with `UnsupportedOperation`
829
+ elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`, `Deepgram`, and `AssemblyAI`
830
+ facades. Common fields (`language`, `prompt`, `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower
831
+ natively or fail with a typed `AIError` before any network call; a route may return more than asked.
832
+
833
+ ```ts
834
+ import { Media, Transcription, TranscriptionEvent } from "@opencode/ai"
835
+ import { AssemblyAI, Deepgram, OpenAI } from "@opencode/ai/providers"
836
+
837
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
838
+
839
+ const program = Effect.gen(function* () {
840
+ const audio = yield* Media.file("./call.mp3")
841
+
842
+ // Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0").
843
+ const response = yield* Transcription.generate({
844
+ model: Deepgram.configure({ apiKey }).transcription("nova-3"),
845
+ audio,
846
+ diarize: true,
847
+ timestamps: "word",
848
+ })
849
+ response.text // "Hello from OpenCode."
850
+ response.segments // [{ text, startSeconds, endSeconds, speaker: "0" }]
851
+ response.words // [{ text, startSeconds, endSeconds, speaker, confidence }]
852
+ response.language // the provider's own value, lowercased ("en", "english", "en_us")
853
+
854
+ // Text deltas as the model transcribes, then one finish carrying the whole transcript.
855
+ yield* Transcription.stream({ model: openai.transcription("gpt-4o-mini-transcribe"), audio }).pipe(
856
+ Stream.tap((event) => (TranscriptionEvent.is.textDelta(event) ? Console.log(event.delta) : Effect.void)),
857
+ Stream.runDrain,
858
+ )
859
+
860
+ // Queued: persist the token, resume from another process, and await.
861
+ const model = AssemblyAI.configure({ apiKey }).transcription("universal-3-5-pro")
862
+ const generation = yield* Transcription.start({ model, audio })
863
+ const resumed = yield* Transcription.resume(model, JSON.parse(JSON.stringify(generation.token)))
864
+ const transcript = yield* resumed.await({ poll: { interval: "3 seconds" } })
865
+ })
866
+ ```
867
+
868
+ Inline routes emit only `finish` from `stream` (no faked deltas); queued routes emit `generation-queued` /
869
+ `generation-progress` before it. `TranscriptionClient.layer` needs `RequestExecutor.Service`.
870
+
871
+ Provider notes:
872
+
873
+ - **OpenAI** takes inline audio only; `diarize` needs `gpt-4o-transcribe-diarize`, timestamps need `whisper-1`, and `whisper-1` does not stream.
874
+ - **Gemini** needs a transcribe model (`gemini-3.5-transcribe`); `prompt` and `speakers` fail typed.
875
+ - **Deepgram** detects the language unless `language` is set; vocabulary goes in `providerOptions.keyterm`.
876
+ - **AssemblyAI** uploads inline audio before submitting and is the only route that accepts `speakers`.
877
+
878
+ The promise client mirrors the Effect API:
879
+
880
+ ```ts
881
+ const text = (await ai.transcription.generate({ model, audio })).text
882
+ for await (const event of ai.transcription.stream({ model, audio })) if (event.type === "text-delta") write(event.delta)
883
+ const generation = await ai.transcription.start({ model: assemblyai, audio })
884
+ const transcript = await generation.await({ poll: { interval: 3_000 } })
885
+ ```
886
+
605
887
  ## Public API
606
888
 
607
889
  - **`LLM.request({...})`** — build a provider-neutral `LLMRequest`. Accepts ergonomic inputs (`system: string`, `prompt: string`) that normalize into the canonical Schema classes.
@@ -609,11 +891,13 @@ The hosted result is represented as a provider-executed tool call and tool resul
609
891
  - **`Message.user(...)` / `Message.assistant(...)` / `Message.tool(...)`** — message constructors from the canonical schema model.
610
892
  - **`LanguageModel.make(...)` / `ToolCallPart.make(...)` / `ToolResultPart.make(...)` / `ToolDefinition.make(...)`** — model and tool-related constructors from the canonical schema model.
611
893
  - **`LLMEvent.is.*`** — typed guards (`is.textDelta`, `is.toolCall`, `is.finish`, …) for filtering streams.
612
- - **`Image.request` / `Image.generate` / `Image.stream`** — generate images through a provider-neutral image request and response model.
894
+ - **`Image.request` / `generate` / `stream` / `start` / `resume`** — images over inline, streaming (partial previews), and queued routes through a provider-neutral request and response model.
613
895
  - **`ImageClient`** — Effect service and layer for image execution, parallel to `LLMClient`.
614
896
  - **`Media`** — the shared asset type (`Media.Asset`, `Media.Source`) and constructors used by messages, tool results, and media requests.
615
897
  - **`Generation`** — provider-neutral handle for an in-flight media generation (`await`, `refresh`, `cancel`, `events`) used by queued media routes.
616
- - **`@opencode/ai/promise`** — `AI.make({ layer? })` and a default `ai` client exposing `llm` and `image` as Promise / `AsyncIterable` APIs.
898
+ - **`Speech.request` / `Speech.generate` / `Speech.stream`** — text-to-speech through a provider-neutral request; `SpeechClient` is its Effect service and layer.
899
+ - **`Transcription.request` / `generate` / `stream` / `start` / `resume`** — speech-to-text over inline, streaming, and queued routes; `TranscriptionClient` is its Effect service and layer.
900
+ - **`@opencode/ai/promise`** — `AI.make({ layer? })` and a default `ai` client exposing `llm`, `image`, `video`, `speech`, and `transcription` as Promise / `AsyncIterable` APIs.
617
901
 
618
902
  ## Testing
619
903
 
@@ -12,13 +12,13 @@ export interface Snapshot {
12
12
  readonly expiresAt?: number;
13
13
  }
14
14
  /**
15
- * Route-owned generation operations. `token` is the route's serializable handle (operation name, task id, response URL)
16
- * so a generation can be resumed from another process; its shape is opaque to `Generation`.
15
+ * Route-owned generation operations for one generation. The media route decodes its serializable token once (from the
16
+ * submission response or a `resume` input) and closes over it, so `Generation` never sees the token's shape.
17
17
  */
18
18
  export interface Route<Response> {
19
- readonly status: (token: unknown) => Effect.Effect<Snapshot, AIError>;
20
- readonly result: (token: unknown) => Effect.Effect<Response, AIError>;
21
- readonly cancel?: (token: unknown) => Effect.Effect<void, AIError>;
19
+ readonly status: Effect.Effect<Snapshot, AIError>;
20
+ readonly result: Effect.Effect<Response, AIError>;
21
+ readonly cancel?: Effect.Effect<void, AIError>;
22
22
  /** Provider polling hint (e.g. `openai-poll-after-ms`) that overrides the default interval for the next poll. */
23
23
  readonly pollHint?: (snapshot: Snapshot) => Duration.Duration | undefined;
24
24
  }
@@ -28,42 +28,56 @@ export interface Poll {
28
28
  /** Full override of the polling schedule; `interval` and `pollHint` are ignored when supplied. */
29
29
  readonly schedule?: Schedule.Schedule<unknown, Snapshot>;
30
30
  }
31
+ export interface AwaitOptions {
32
+ readonly poll?: Poll;
33
+ }
31
34
  export declare const DEFAULT_POLL_INTERVAL: Duration.Duration;
32
35
  export declare const DEFAULT_POLL_TIMEOUT: Duration.Duration;
33
- export type Event = {
34
- readonly type: "generation-queued";
35
- readonly id: string;
36
- readonly position?: number;
37
- } | {
38
- readonly type: "generation-progress";
39
- readonly id: string;
40
- readonly progress?: number;
41
- } | {
36
+ export declare const QueuedEvent: Schema.Struct<{
37
+ readonly type: Schema.tag<"generation-queued">;
38
+ readonly id: Schema.String;
39
+ readonly position: Schema.optional<Schema.Number>;
40
+ }>;
41
+ export declare const ProgressEvent: Schema.Struct<{
42
+ readonly type: Schema.tag<"generation-progress">;
43
+ readonly id: Schema.String;
44
+ readonly progress: Schema.optional<Schema.Number>;
45
+ }>;
46
+ export type Observation = Schema.Schema.Type<typeof QueuedEvent> | Schema.Schema.Type<typeof ProgressEvent>;
47
+ export type Event = Observation | {
42
48
  readonly type: "generation-finished";
43
49
  readonly id: string;
44
50
  readonly status: Status;
45
51
  };
46
52
  export declare class Generation<Response> {
47
53
  readonly route: Route<Response>;
54
+ /** Route-owned serializable JSON; pass it to the modality's `resume` from another process. */
48
55
  readonly token: unknown;
49
56
  readonly id: string;
50
57
  readonly status: Status;
51
58
  readonly progress?: number;
52
59
  readonly position?: number;
53
60
  readonly expiresAt?: number;
54
- constructor(route: Route<Response>, token: unknown, snapshot: Snapshot);
61
+ constructor(route: Route<Response>,
62
+ /** Route-owned serializable JSON; pass it to the modality's `resume` from another process. */
63
+ token: unknown, snapshot: Snapshot);
55
64
  get snapshot(): Snapshot;
56
65
  get terminal(): boolean;
57
66
  refresh(): Effect.Effect<Generation<Response>, AIError>;
67
+ /** Fetch the result without polling; non-completed terminal generations fail with the provider's terminal body. */
68
+ result(): Effect.Effect<Response, AIError>;
58
69
  /** Poll until the generation reaches a terminal status, then fetch the result. Fails with a `Timeout` reason on deadline. */
59
- await(options?: {
60
- readonly poll?: Poll;
61
- }): Effect.Effect<Response, AIError>;
70
+ await(options?: AwaitOptions): Effect.Effect<Response, AIError>;
62
71
  cancel(): Effect.Effect<void, AIError>;
63
- /** Status observations as a stream, ending after the first terminal observation. */
64
- events(options?: {
65
- readonly poll?: Poll;
66
- }): Stream.Stream<Event, AIError>;
72
+ /**
73
+ * Status observations as a stream, ending after the first terminal observation. Each poll is bounded by the time
74
+ * remaining until `poll.timeout`, so a hung status request fails the stream instead of stalling it. (`Stream.interruptWhen`
75
+ * would express this directly but deadlocks under `TestClock` when the source completes while the timer sleeps.)
76
+ */
77
+ events(options?: AwaitOptions): Stream.Stream<Event, AIError>;
78
+ private event;
79
+ private timeoutError;
67
80
  private poll;
68
81
  private schedule;
69
82
  }
83
+ export declare const resultEvents: <Response, A>(generation: Generation<Response>, expand: (response: Response) => ReadonlyArray<A>, options?: AwaitOptions) => Stream.Stream<Observation | A, AIError>;
@@ -1,8 +1,18 @@
1
- import { Duration, Effect, Schedule, Schema, Stream } from "effect";
1
+ import { Clock, Duration, Effect, Schedule, Schema, Stream } from "effect";
2
2
  import { AIError, TimeoutError } from "./schema/errors.js";
3
3
  export const Status = Schema.Literals(["queued", "running", "completed", "failed", "cancelled", "expired"]);
4
4
  export const DEFAULT_POLL_INTERVAL = Duration.seconds(5);
5
5
  export const DEFAULT_POLL_TIMEOUT = Duration.minutes(10);
6
+ export const QueuedEvent = Schema.Struct({
7
+ type: Schema.tag("generation-queued"),
8
+ id: Schema.String,
9
+ position: Schema.optional(Schema.Number),
10
+ }).annotate({ identifier: "Generation.Event.Queued" });
11
+ export const ProgressEvent = Schema.Struct({
12
+ type: Schema.tag("generation-progress"),
13
+ id: Schema.String,
14
+ progress: Schema.optional(Schema.Number),
15
+ }).annotate({ identifier: "Generation.Event.Progress" });
6
16
  const TERMINAL = new Set(["completed", "failed", "cancelled", "expired"]);
7
17
  export class Generation {
8
18
  route;
@@ -12,7 +22,9 @@ export class Generation {
12
22
  progress;
13
23
  position;
14
24
  expiresAt;
15
- constructor(route, token, snapshot) {
25
+ constructor(route,
26
+ /** Route-owned serializable JSON; pass it to the modality's `resume` from another process. */
27
+ token, snapshot) {
16
28
  this.route = route;
17
29
  this.token = token;
18
30
  this.id = snapshot.id;
@@ -34,7 +46,11 @@ export class Generation {
34
46
  return TERMINAL.has(this.status);
35
47
  }
36
48
  refresh() {
37
- return this.route.status(this.token).pipe(Effect.map((snapshot) => new Generation(this.route, this.token, snapshot)));
49
+ return this.route.status.pipe(Effect.map((snapshot) => new Generation(this.route, this.token, snapshot)));
50
+ }
51
+ /** Fetch the result without polling; non-completed terminal generations fail with the provider's terminal body. */
52
+ result() {
53
+ return this.route.result;
38
54
  }
39
55
  /** Poll until the generation reaches a terminal status, then fetch the result. Fails with a `Timeout` reason on deadline. */
40
56
  await(options) {
@@ -42,31 +58,43 @@ export class Generation {
42
58
  const settled = this.terminal ? Effect.succeed(this) : this.poll(options?.poll);
43
59
  return settled.pipe(
44
60
  // Non-completed terminal states also go through `result` so the route can surface its provider failure body.
45
- Effect.flatMap((generation) => generation.route.result(generation.token)), Effect.timeoutOrElse({
46
- duration: timeout,
47
- orElse: () => new AIError({
48
- reason: new TimeoutError({
49
- message: `Generation ${this.id} did not finish within ${Duration.format(timeout)}`,
50
- timeoutMs: Duration.toMillis(timeout),
51
- }),
52
- }),
53
- }));
61
+ Effect.flatMap((generation) => generation.result()), Effect.timeoutOrElse({ duration: timeout, orElse: () => this.timeoutError(timeout) }));
54
62
  }
55
63
  cancel() {
56
- return this.route.cancel?.(this.token) ?? Effect.void;
64
+ return this.route.cancel ?? Effect.void;
57
65
  }
58
- /** Status observations as a stream, ending after the first terminal observation. */
66
+ /**
67
+ * Status observations as a stream, ending after the first terminal observation. Each poll is bounded by the time
68
+ * remaining until `poll.timeout`, so a hung status request fails the stream instead of stalling it. (`Stream.interruptWhen`
69
+ * would express this directly but deadlocks under `TestClock` when the source completes while the timer sleeps.)
70
+ */
59
71
  events(options) {
60
- const observations = this.terminal
61
- ? Stream.make(this)
62
- : Stream.fromEffectSchedule(this.refresh(), this.schedule(options?.poll)).pipe(Stream.takeUntil((generation) => generation.terminal));
63
- return observations.pipe(Stream.map((generation) => {
64
- if (generation.terminal)
65
- return { type: "generation-finished", id: generation.id, status: generation.status };
66
- if (generation.status === "queued")
67
- return { type: "generation-queued", id: generation.id, position: generation.position };
68
- return { type: "generation-progress", id: generation.id, progress: generation.progress };
69
- }));
72
+ if (this.terminal)
73
+ return Stream.make(this.event());
74
+ const timeout = Duration.fromInputUnsafe(options?.poll?.timeout ?? DEFAULT_POLL_TIMEOUT);
75
+ return Stream.unwrap(Clock.currentTimeMillis.pipe(Effect.map((start) => {
76
+ const deadline = start + Duration.toMillis(timeout);
77
+ const refresh = Clock.currentTimeMillis.pipe(Effect.flatMap((now) => this.refresh().pipe(Effect.timeoutOrElse({
78
+ duration: Duration.millis(Math.max(0, deadline - now)),
79
+ orElse: () => this.timeoutError(timeout),
80
+ }))));
81
+ return Stream.fromEffectSchedule(refresh, this.schedule(options?.poll)).pipe(Stream.takeUntil((generation) => generation.terminal), Stream.map((generation) => generation.event()));
82
+ })));
83
+ }
84
+ event() {
85
+ if (this.terminal)
86
+ return { type: "generation-finished", id: this.id, status: this.status };
87
+ if (this.status === "queued")
88
+ return { type: "generation-queued", id: this.id, position: this.position };
89
+ return { type: "generation-progress", id: this.id, progress: this.progress };
90
+ }
91
+ timeoutError(timeout) {
92
+ return new AIError({
93
+ reason: new TimeoutError({
94
+ message: `Generation ${this.id} did not finish within ${Duration.format(timeout)}`,
95
+ timeoutMs: Duration.toMillis(timeout),
96
+ }),
97
+ });
70
98
  }
71
99
  poll(poll) {
72
100
  return this.refresh().pipe(Effect.repeat({ schedule: this.schedule(poll), until: (generation) => generation.terminal }));
@@ -82,3 +110,4 @@ export class Generation {
82
110
  return spaced.pipe(Schedule.modifyDelay((metadata) => Effect.succeed(pollHint(metadata.input.snapshot) ?? interval)));
83
111
  }
84
112
  }
113
+ export const resultEvents = (generation, expand, options) => generation.events(options).pipe(Stream.filter((event) => event.type !== "generation-finished"), Stream.concat(Stream.fromIterableEffect(Effect.map(generation.result(), expand))));
@@ -1,21 +1,28 @@
1
1
  import { Context, Effect, Layer, Stream } from "effect";
2
+ import type { AwaitOptions, Generation } from "./generation.js";
2
3
  import { RequestExecutor } from "./route/executor.js";
3
4
  import type { AIError } from "./schema/index.js";
4
- import { type ImageEvent, type ImageOptions, type ImageRequestFor, type ImageResponse } from "./image.js";
5
+ import { type ImageEvent, type ImageModel, type ImageOptions, type ImageRequestFor, type ImageResponse } from "./image.js";
5
6
  export interface Interface {
6
- readonly generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<ImageResponse, AIError>;
7
- readonly stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Stream.Stream<ImageEvent, AIError>;
7
+ readonly generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Effect.Effect<ImageResponse, AIError>;
8
+ readonly stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Stream.Stream<ImageEvent, AIError>;
9
+ readonly start: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<Generation<ImageResponse>, AIError>;
10
+ readonly resume: <Options extends ImageOptions>(model: ImageModel<Options>, token: unknown) => Effect.Effect<Generation<ImageResponse>, AIError>;
8
11
  }
9
12
  declare const Service_base: Context.ServiceClass<Service, "@opencode/ImageClient", Interface>;
10
13
  export declare class Service extends Service_base {
11
14
  }
12
- export declare const generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<ImageResponse, AIError, Service>;
13
- export declare const stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Stream.Stream<ImageEvent, AIError, Service>;
15
+ export declare const generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Effect.Effect<ImageResponse, AIError, Service>;
16
+ export declare const stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Stream.Stream<ImageEvent, AIError, Service>;
17
+ export declare const start: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<Generation<ImageResponse>, AIError, Service>;
18
+ export declare const resume: <Options extends ImageOptions>(model: ImageModel<Options>, token: unknown) => Effect.Effect<Generation<ImageResponse>, AIError, Service>;
14
19
  export declare const layer: Layer.Layer<Service, never, RequestExecutor.Service>;
15
20
  export declare const ImageClient: {
16
21
  readonly Service: typeof Service;
17
22
  readonly layer: Layer.Layer<Service, never, RequestExecutor.Service>;
18
- readonly generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<ImageResponse, AIError, Service>;
19
- readonly stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Stream.Stream<ImageEvent, AIError, Service>;
23
+ readonly generate: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Effect.Effect<ImageResponse, AIError, Service>;
24
+ readonly stream: <Options extends ImageOptions>(request: ImageRequestFor<Options>, options?: AwaitOptions) => Stream.Stream<ImageEvent, AIError, Service>;
25
+ readonly start: <Options extends ImageOptions>(request: ImageRequestFor<Options>) => Effect.Effect<Generation<ImageResponse>, AIError, Service>;
26
+ readonly resume: <Options extends ImageOptions>(model: ImageModel<Options>, token: unknown) => Effect.Effect<Generation<ImageResponse>, AIError, Service>;
20
27
  };
21
28
  export {};
@@ -1,23 +1,37 @@
1
1
  import { Context, Effect, Layer, Stream } from "effect";
2
2
  import { RequestExecutor } from "./route/executor.js";
3
+ import { MediaRoute } from "./route/media.js";
3
4
  import { responseEvents, } from "./image.js";
4
5
  export class Service extends Context.Service()("@opencode/ImageClient") {
5
6
  }
6
- export const generate = (request) => Effect.gen(function* () {
7
+ export const generate = (request, options) => Effect.gen(function* () {
7
8
  const client = yield* Service;
8
- return yield* client.generate(request);
9
+ return yield* client.generate(request, options);
9
10
  });
10
- export const stream = (request) => Stream.unwrap(Effect.gen(function* () {
11
+ export const stream = (request, options) => Stream.unwrap(Effect.gen(function* () {
11
12
  const client = yield* Service;
12
- return client.stream(request);
13
+ return client.stream(request, options);
13
14
  }));
15
+ export const start = (request) => Effect.gen(function* () {
16
+ const client = yield* Service;
17
+ return yield* client.start(request);
18
+ });
19
+ export const resume = (model, token) => Effect.gen(function* () {
20
+ const client = yield* Service;
21
+ return yield* client.resume(model, token);
22
+ });
14
23
  export const layer = Layer.effect(Service, Effect.gen(function* () {
15
24
  const executor = yield* RequestExecutor.Service;
16
- const generate = (request) => request.model.route.generate(request, executor.execute);
25
+ const dispatch = MediaRoute.dispatch({
26
+ modality: "image",
27
+ execute: executor.execute,
28
+ responseEvents,
29
+ });
17
30
  return Service.of({
18
- generate,
19
- // Inline routes have no partial frames yet; the stream is the completed response expanded into events.
20
- stream: (request) => Stream.unwrap(generate(request).pipe(Effect.map((response) => Stream.fromIterable(responseEvents(response))))),
31
+ start: (request) => dispatch.start(request.model.route, request),
32
+ resume: (model, token) => dispatch.resume(model.route, model, token),
33
+ generate: (request, options) => dispatch.generate(request.model.route, request, options),
34
+ stream: (request, options) => dispatch.stream(request.model.route, request, options),
21
35
  });
22
36
  }));
23
37
  export const ImageClient = {
@@ -25,4 +39,6 @@ export const ImageClient = {
25
39
  layer,
26
40
  generate,
27
41
  stream,
42
+ start,
43
+ resume,
28
44
  };