@opencode/ai 2.0.14 → 2.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +399 -56
  2. package/dist/experimental/evaluation-client.d.ts +1 -1
  3. package/dist/experimental/evaluation-client.js +39 -3
  4. package/dist/experimental/evaluation.d.ts +4 -4
  5. package/dist/experimental/evaluation.js +2 -2
  6. package/dist/experimental/system-one.d.ts +3 -3
  7. package/dist/experimental/system-one.js +40 -51
  8. package/dist/generation.d.ts +83 -0
  9. package/dist/generation.js +113 -0
  10. package/dist/image-client.d.ts +16 -7
  11. package/dist/image-client.js +29 -13
  12. package/dist/image.d.ts +1410 -81
  13. package/dist/image.js +97 -67
  14. package/dist/index.d.ts +17 -2
  15. package/dist/index.js +12 -1
  16. package/dist/llm.d.ts +9 -1
  17. package/dist/media-model.d.ts +44 -0
  18. package/dist/media-model.js +49 -0
  19. package/dist/media.d.ts +213 -0
  20. package/dist/media.js +227 -0
  21. package/dist/promise.d.ts +974 -0
  22. package/dist/promise.js +81 -0
  23. package/dist/protocols/alibaba-chat.d.ts +12 -0
  24. package/dist/protocols/alibaba-responses.d.ts +2 -2
  25. package/dist/protocols/anthropic-messages.js +8 -19
  26. package/dist/protocols/assemblyai-transcription.d.ts +40 -0
  27. package/dist/protocols/assemblyai-transcription.js +138 -0
  28. package/dist/protocols/bedrock-converse.d.ts +4 -4
  29. package/dist/protocols/bedrock-converse.js +6 -17
  30. package/dist/protocols/bfl-images.d.ts +32 -0
  31. package/dist/protocols/bfl-images.js +153 -0
  32. package/dist/protocols/cartesia-speech.d.ts +127 -0
  33. package/dist/protocols/cartesia-speech.js +126 -0
  34. package/dist/protocols/deepgram-speech.d.ts +119 -0
  35. package/dist/protocols/deepgram-speech.js +92 -0
  36. package/dist/protocols/deepgram-transcription.d.ts +25 -0
  37. package/dist/protocols/deepgram-transcription.js +129 -0
  38. package/dist/protocols/elevenlabs-speech.d.ts +122 -0
  39. package/dist/protocols/elevenlabs-speech.js +115 -0
  40. package/dist/protocols/fal-images.d.ts +24 -0
  41. package/dist/protocols/fal-images.js +114 -0
  42. package/dist/protocols/fal-video.d.ts +29 -0
  43. package/dist/protocols/fal-video.js +88 -0
  44. package/dist/protocols/gemini.d.ts +30 -9
  45. package/dist/protocols/gemini.js +45 -35
  46. package/dist/protocols/google-images.d.ts +9 -21
  47. package/dist/protocols/google-images.js +158 -133
  48. package/dist/protocols/google-speech.d.ts +130 -0
  49. package/dist/protocols/google-speech.js +84 -0
  50. package/dist/protocols/google-transcription.d.ts +173 -0
  51. package/dist/protocols/google-transcription.js +138 -0
  52. package/dist/protocols/google-video.d.ts +26 -0
  53. package/dist/protocols/google-video.js +158 -0
  54. package/dist/protocols/meta-images.d.ts +7 -12
  55. package/dist/protocols/meta-images.js +85 -66
  56. package/dist/protocols/meta-responses.d.ts +4 -4
  57. package/dist/protocols/meta-responses.js +1 -1
  58. package/dist/protocols/mistral-chat.js +7 -6
  59. package/dist/protocols/open-responses.d.ts +17 -9
  60. package/dist/protocols/open-responses.js +24 -14
  61. package/dist/protocols/openai-chat.d.ts +118 -1
  62. package/dist/protocols/openai-chat.js +125 -44
  63. package/dist/protocols/openai-compatible-chat.d.ts +12 -0
  64. package/dist/protocols/openai-compatible-responses.d.ts +2 -2
  65. package/dist/protocols/openai-images.d.ts +128 -18
  66. package/dist/protocols/openai-images.js +177 -154
  67. package/dist/protocols/openai-responses.d.ts +15 -15
  68. package/dist/protocols/openai-responses.js +5 -6
  69. package/dist/protocols/openai-speech.d.ts +116 -0
  70. package/dist/protocols/openai-speech.js +98 -0
  71. package/dist/protocols/openai-transcription.d.ts +207 -0
  72. package/dist/protocols/openai-transcription.js +190 -0
  73. package/dist/protocols/replicate-images.d.ts +28 -0
  74. package/dist/protocols/replicate-images.js +133 -0
  75. package/dist/protocols/runway-video.d.ts +38 -0
  76. package/dist/protocols/runway-video.js +146 -0
  77. package/dist/protocols/shared.d.ts +27 -17
  78. package/dist/protocols/shared.js +52 -35
  79. package/dist/protocols/stability-images.d.ts +38 -0
  80. package/dist/protocols/stability-images.js +148 -0
  81. package/dist/protocols/utils/bedrock-media.d.ts +2 -3
  82. package/dist/protocols/utils/bedrock-media.js +4 -4
  83. package/dist/protocols/utils/fal-queue.d.ts +28 -0
  84. package/dist/protocols/utils/fal-queue.js +69 -0
  85. package/dist/protocols/utils/gemini-generate-content.d.ts +65 -0
  86. package/dist/protocols/utils/gemini-generate-content.js +65 -0
  87. package/dist/protocols/utils/gemini-json-schema.d.ts +3 -0
  88. package/dist/protocols/utils/gemini-json-schema.js +76 -0
  89. package/dist/protocols/utils/media-input.d.ts +18 -0
  90. package/dist/protocols/utils/media-input.js +35 -0
  91. package/dist/protocols/utils/responses-compaction.js +6 -5
  92. package/dist/protocols/utils/speech-stream.d.ts +49 -0
  93. package/dist/protocols/utils/speech-stream.js +67 -0
  94. package/dist/protocols/utils/tool-schema.d.ts +2 -2
  95. package/dist/protocols/utils/tool-schema.js +40 -17
  96. package/dist/protocols/utils/tool-stream.d.ts +27 -3
  97. package/dist/protocols/xai-images.d.ts +9 -15
  98. package/dist/protocols/xai-images.js +75 -84
  99. package/dist/protocols/xai-responses.d.ts +2 -2
  100. package/dist/protocols/xai-video.d.ts +34 -0
  101. package/dist/protocols/xai-video.js +147 -0
  102. package/dist/protocols/zai-chat.d.ts +13 -1
  103. package/dist/protocols/zai-images.d.ts +9 -13
  104. package/dist/protocols/zai-images.js +59 -57
  105. package/dist/provider-error.js +3 -0
  106. package/dist/providers/alibaba.d.ts +14 -2
  107. package/dist/providers/amazon-bedrock-mantle.d.ts +14 -2
  108. package/dist/providers/amazon-bedrock.d.ts +2 -2
  109. package/dist/providers/assemblyai.d.ts +25 -0
  110. package/dist/providers/assemblyai.js +29 -0
  111. package/dist/providers/azure.d.ts +18 -6
  112. package/dist/providers/baseten.d.ts +24 -0
  113. package/dist/providers/black-forest-labs.d.ts +25 -0
  114. package/dist/providers/black-forest-labs.js +28 -0
  115. package/dist/providers/cartesia.d.ts +24 -0
  116. package/dist/providers/cartesia.js +22 -0
  117. package/dist/providers/cerebras.d.ts +24 -0
  118. package/dist/providers/cerebras.js +6 -1
  119. package/dist/providers/cloudflare-ai-gateway.d.ts +30 -6
  120. package/dist/providers/cloudflare-workers-ai.d.ts +24 -0
  121. package/dist/providers/deepgram.d.ts +29 -0
  122. package/dist/providers/deepgram.js +31 -0
  123. package/dist/providers/deepinfra.d.ts +24 -0
  124. package/dist/providers/deepinfra.js +6 -1
  125. package/dist/providers/deepseek.d.ts +24 -0
  126. package/dist/providers/elevenlabs.d.ts +24 -0
  127. package/dist/providers/elevenlabs.js +28 -0
  128. package/dist/providers/fal.d.ts +29 -0
  129. package/dist/providers/fal.js +33 -0
  130. package/dist/providers/fireworks.d.ts +24 -0
  131. package/dist/providers/google-vertex-chat.d.ts +12 -0
  132. package/dist/providers/google-vertex-responses.d.ts +2 -2
  133. package/dist/providers/google-vertex.d.ts +10 -3
  134. package/dist/providers/google.d.ts +25 -3
  135. package/dist/providers/google.js +11 -2
  136. package/dist/providers/groq.d.ts +24 -0
  137. package/dist/providers/index.d.ts +10 -0
  138. package/dist/providers/index.js +10 -0
  139. package/dist/providers/meta.d.ts +14 -2
  140. package/dist/providers/minimax.d.ts +14 -2
  141. package/dist/providers/moonshot.d.ts +14 -2
  142. package/dist/providers/moonshot.js +3 -3
  143. package/dist/providers/openai-compatible-responses.d.ts +2 -2
  144. package/dist/providers/openai-compatible.d.ts +12 -0
  145. package/dist/providers/openai.d.ts +25 -3
  146. package/dist/providers/openai.js +10 -1
  147. package/dist/providers/openrouter.d.ts +67 -0
  148. package/dist/providers/openrouter.js +13 -1
  149. package/dist/providers/replicate.d.ts +25 -0
  150. package/dist/providers/replicate.js +22 -0
  151. package/dist/providers/runway.d.ts +24 -0
  152. package/dist/providers/runway.js +22 -0
  153. package/dist/providers/stability.d.ts +28 -0
  154. package/dist/providers/stability.js +23 -0
  155. package/dist/providers/togetherai.d.ts +24 -0
  156. package/dist/providers/vercel-ai-gateway.d.ts +41 -0
  157. package/dist/providers/vercel-ai-gateway.js +85 -0
  158. package/dist/providers/xai.d.ts +17 -0
  159. package/dist/providers/xai.js +5 -2
  160. package/dist/providers/zai-coding-plan.d.ts +15 -3
  161. package/dist/providers/zai.d.ts +13 -1
  162. package/dist/route/auth.d.ts +4 -1
  163. package/dist/route/auth.js +6 -0
  164. package/dist/route/client.d.ts +9 -1
  165. package/dist/route/endpoint.d.ts +10 -10
  166. package/dist/route/executor-service.d.ts +12 -0
  167. package/dist/route/executor-service.js +3 -0
  168. package/dist/route/executor.d.ts +4 -9
  169. package/dist/route/executor.js +3 -3
  170. package/dist/route/framing.d.ts +5 -1
  171. package/dist/route/framing.js +9 -0
  172. package/dist/route/index.d.ts +2 -0
  173. package/dist/route/index.js +2 -0
  174. package/dist/route/media-protocol.d.ts +158 -0
  175. package/dist/route/media-protocol.js +96 -0
  176. package/dist/route/media.d.ts +97 -0
  177. package/dist/route/media.js +236 -0
  178. package/dist/schema/errors.d.ts +13 -3
  179. package/dist/schema/errors.js +7 -0
  180. package/dist/schema/events.d.ts +557 -40
  181. package/dist/schema/events.js +35 -2
  182. package/dist/schema/messages.d.ts +95 -8
  183. package/dist/schema/messages.js +8 -6
  184. package/dist/schema/options.d.ts +6 -3
  185. package/dist/schema/options.js +6 -2
  186. package/dist/speech-client.d.ts +21 -0
  187. package/dist/speech-client.js +25 -0
  188. package/dist/speech.d.ts +1150 -0
  189. package/dist/speech.js +119 -0
  190. package/dist/testing.d.ts +72 -8
  191. package/dist/transcription-client.d.ts +28 -0
  192. package/dist/transcription-client.js +44 -0
  193. package/dist/transcription.d.ts +1504 -0
  194. package/dist/transcription.js +133 -0
  195. package/dist/utils/bytes.d.ts +1 -0
  196. package/dist/utils/bytes.js +10 -0
  197. package/dist/utils/media-type.d.ts +7 -0
  198. package/dist/utils/media-type.js +70 -0
  199. package/dist/utils/sanitize.js +3 -1
  200. package/dist/video-client.d.ts +28 -0
  201. package/dist/video-client.js +40 -0
  202. package/dist/video.d.ts +1359 -0
  203. package/dist/video.js +119 -0
  204. package/package.json +7 -3
  205. package/dist/protocols/utils/gemini-tool-schema.d.ts +0 -2
  206. package/dist/protocols/utils/gemini-tool-schema.js +0 -103
  207. package/dist/protocols/utils/image-input.d.ts +0 -21
  208. package/dist/protocols/utils/image-input.js +0 -20
package/README.md CHANGED
@@ -8,10 +8,10 @@ import { LLM, LLMClient } from "@opencode/ai"
8
8
  import { RequestExecutor } from "@opencode/ai/route"
9
9
  import { OpenAI } from "@opencode/ai/providers"
10
10
 
11
- const model = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY }).responses("gpt-4o-mini")
11
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
12
12
 
13
13
  const request = LLM.request({
14
- model,
14
+ model: openai.responses("gpt-4o-mini"), // `.chat(...)` selects the Chat Completions API instead
15
15
  system: "You are concise.",
16
16
  prompt: "Say hello in one short sentence.",
17
17
  generation: { maxTokens: 40 },
@@ -29,6 +29,43 @@ await Effect.runPromise(program.pipe(Effect.provide(llmLayer)))
29
29
 
30
30
  Run `LLMClient.stream(request)` instead of `generate` when you want incremental `LLMEvent`s. The event stream is provider-neutral — same shape across OpenAI Chat, OpenAI Responses, Anthropic Messages, Gemini, Bedrock Converse, and any OpenAI-compatible deployment.
31
31
 
32
+ The same configured facade names image models. `Image.request` resolves the provider's image route from the ref and
33
+ returns `Media.Asset`s with lazily decoded bytes:
34
+
35
+ ```ts
36
+ import { NodeFileSystem } from "@effect/platform-node"
37
+ import { Image, ImageClient, Media } from "@opencode/ai"
38
+
39
+ const image = Effect.gen(function* () {
40
+ const response = yield* Image.generate({
41
+ model: openai.image("gpt-image-2"),
42
+ prompt: "A robot tending a rooftop garden",
43
+ size: "1024x1024",
44
+ providerOptions: { quality: "high" }, // typed per image model
45
+ })
46
+ yield* Media.write(response.image, "./garden.png")
47
+ })
48
+
49
+ // `asset.bytes()` / `Media.write` also need the executor, so merge it into the environment instead of hiding it.
50
+ const imageLayer = ImageClient.layer.pipe(Layer.provideMerge(RequestExecutor.fetchLayer))
51
+
52
+ await Effect.runPromise(image.pipe(Effect.provide(imageLayer), Effect.provide(NodeFileSystem.layer)))
53
+ ```
54
+
55
+ Prefer promises? `@opencode/ai/promise` exposes the same LLM and image APIs over one managed runtime:
56
+
57
+ ```ts
58
+ import { AI } from "@opencode/ai/promise"
59
+
60
+ const ai = AI.make()
61
+ const text = await ai.llm.generate({ model: openai.responses("gpt-4o-mini"), prompt: "Say hello." })
62
+ const generated = await ai.image.generate({ model: openai.image("gpt-image-2"), prompt: "A lighthouse" })
63
+ for await (const event of ai.llm.stream({ model: openai.responses("gpt-4o-mini"), prompt: "Stream hello." })) {
64
+ // LLMEvent
65
+ }
66
+ await ai.dispose()
67
+ ```
68
+
32
69
  ## Experimental evaluation
33
70
 
34
71
  Evaluation models compare shared state with typed choice, score, and boolean questions. The API is
@@ -41,7 +78,7 @@ import { TypeSafeAI } from "@opencode/ai/providers"
41
78
 
42
79
  const model = TypeSafeAI.configure().experimental.evaluation("jev-latest")
43
80
 
44
- const program = Evaluation.evaluate({
81
+ const program = Evaluation.run({
45
82
  model,
46
83
  state: "I was charged twice. Please refund the duplicate payment.",
47
84
  questions: {
@@ -66,7 +103,17 @@ console.log(response.answers.refund.probability)
66
103
  ```
67
104
 
68
105
  `TypeSafeAI` reads `TYPESAFE_API_KEY`. `OpenCodeZen` exposes the same selector and reads
69
- `OPENCODE_API_KEY`. The common API uses `boolean`; System One routes lower it to native `noul`.
106
+ `OPENCODE_API_KEY`. OpenRouter and Vercel AI Gateway use the same provider shape:
107
+
108
+ ```ts
109
+ import { OpenRouter, VercelAIGateway } from "@opencode/ai/providers"
110
+
111
+ OpenRouter.configure().experimental.evaluation("typesafe/jev-1.13")
112
+ VercelAIGateway.configure().experimental.evaluation("typesafe-ai/jev")
113
+ ```
114
+
115
+ OpenRouter reads `OPENROUTER_API_KEY`. Vercel reads `AI_GATEWAY_API_KEY`, then `VERCEL_OIDC_TOKEN`.
116
+ The common API uses `boolean`; System One routes lower it to native `noul`.
70
117
  Choice and score confidence plus score legends remain available in provider metadata, and the
71
118
  provider's rounded probabilities are returned unchanged.
72
119
 
@@ -355,23 +402,25 @@ citations or separate result blocks. Retain `response.message` for either API's
355
402
  Use `Image.generate` for one-off generation or editing:
356
403
 
357
404
  ```ts
358
- import { Image, ImageInput } from "@opencode/ai"
405
+ import { Image, Media } from "@opencode/ai"
359
406
 
360
407
  const generation = Image.generate({
361
- model: meta.image("muse-image-1.0"),
408
+ model: meta("muse-image-1.0"),
362
409
  prompt: "A flat black square on a white background.",
363
- options: { n: 1, reasoningStrength: "low" },
410
+ n: 1,
411
+ providerOptions: { reasoningStrength: "low" },
364
412
  })
365
413
 
366
414
  const edit = Image.generate({
367
- model: meta.image("muse-image-1.0"),
415
+ model: meta("muse-image-1.0"),
368
416
  prompt: "Make the square purple.",
369
- images: [ImageInput.bytes(imageBytes, "image/webp")],
370
- options: { outputFormat: "png", reasoningStrength: "low" },
417
+ images: [Media.bytes(imageBytes, "image/webp")],
418
+ format: "png",
419
+ providerOptions: { reasoningStrength: "low" },
371
420
  })
372
421
  ```
373
422
 
374
- The default image format is WEBP; `outputFormat` also accepts PNG/JPEG and `responseFormat: "url"`
423
+ The default image format is WEBP; `format` also accepts PNG/JPEG and `responseFormat: "url"`
375
424
  returns a signed URL. `size` is an aspect-ratio hint. For conversational images, select
376
425
  `meta.responses("muse-image-1.0")` with `tools: [Meta.imageGeneration({ reasoningStrength: "low" })]`.
377
426
  Generated images are provider-executed tool results with file content. Retain `response.message` to
@@ -382,29 +431,40 @@ Meta Responses is explicitly HTTP/SSE-only and does not use WebSockets, even whe
382
431
 
383
432
  ## Image generation
384
433
 
385
- Use `Image.generate` with an image model for direct asset generation:
434
+ Use `Image.generate` with an image model for direct asset generation. `Image.request` mirrors `LLM.request`: the
435
+ model comes from the facade's `.image(...)` selector (mirroring `.responses(...)`), common fields
436
+ (`images`, `mask`, `n`, `size`, `aspectRatio`, `seed`, `format`) lower natively or fail typed, and
437
+ `providerOptions` is inferred from the selected model:
386
438
 
387
439
  ```ts
388
- import { Image, ImageInput } from "@opencode/ai"
440
+ import { Image, Media } from "@opencode/ai"
389
441
  import { OpenAI } from "@opencode/ai/providers"
390
442
 
443
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
444
+
391
445
  const program = Effect.gen(function* () {
392
446
  const response = yield* Image.generate({
393
- model: OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY }).image("gpt-image-2"),
447
+ model: openai.image("gpt-image-2"),
394
448
  prompt: "A robot tending a rooftop garden",
395
- options: {
396
- n: 2,
397
- size: "1024x1024",
449
+ n: 2,
450
+ size: "1024x1024",
451
+ format: "webp",
452
+ providerOptions: {
398
453
  quality: "high", // inferred from the OpenAI image model
399
- outputFormat: "webp",
400
454
  future_option: true, // unknown native options pass through unchanged
401
455
  },
402
456
  })
403
457
 
404
- return response.images // GeneratedImage[] with owned bytes or a provider URL
458
+ return response.images // Media.Asset[] with owned bytes or a provider URL
405
459
  })
406
460
  ```
407
461
 
462
+ `Media.Asset` is the one asset type shared by image requests, image responses, LLM messages, and tool results.
463
+ `asset.source` is the serializable `Media.Source` (`bytes`, `base64`, `url`, or `ref`); `asset.bytes()`,
464
+ `asset.base64()`, and `asset.dataUrl()` decode or download lazily and cache; `asset.materialize()` pulls a `url`
465
+ asset into owned bytes before the provider URL expires. Construct assets with `Media.bytes`, `Media.base64`,
466
+ `Media.url`, `Media.ref(provider, id)`, `Media.fromDataUrl`, or `Media.file(path)`.
467
+
408
468
  Pass ordered image inputs to the same method for editing, composition, or image-conditioned generation:
409
469
 
410
470
  ```ts
@@ -414,49 +474,45 @@ const response =
414
474
  model,
415
475
  prompt: "Combine these product photos into one studio scene",
416
476
  images: [
417
- ImageInput.bytes(firstBytes, "image/png"),
418
- ImageInput.url("https://example.com/second.webp"),
419
- ImageInput.file("file_123"),
477
+ Media.bytes(firstBytes, "image/png"),
478
+ Media.url("https://example.com/second.webp"),
479
+ Media.ref("openai", "file_123"),
420
480
  ],
421
- options,
481
+ providerOptions,
422
482
  http,
423
483
  })
424
484
  ```
425
485
 
426
- `ImageInput.fileUri(uri, mediaType)` represents provider file URIs such as Gemini Files. Raw strings are not
427
- accepted as image inputs, avoiding ambiguity between base64, URLs, and provider IDs. Empty or omitted `images`
428
- uses text-to-image generation; a non-empty array selects the provider's edit behavior without enforcing provider
429
- image-count limits locally. `images` is the only common image-editing field. OpenAI uses multipart for byte/data-URL
430
- edits and its JSON reference body for URL or file-ID edits. Its provider-specific `options.mask` accepts an
431
- `ImageInput` for inpainting:
486
+ `Media.ref(provider, id)` represents provider file handles such as OpenAI file IDs or Gemini Files URIs; routes
487
+ only forward refs that belong to their own provider. Raw strings are not accepted as image inputs, avoiding
488
+ ambiguity between base64, URLs, and provider IDs. Empty or omitted `images` uses text-to-image generation; a
489
+ non-empty array selects the provider's edit behavior without enforcing provider image-count limits locally. OpenAI
490
+ uses multipart for byte/data-URL edits and its JSON reference body for URL or file-ID edits. The common `mask`
491
+ field selects inpainting; routes that cannot honor it fail with `UnsupportedOperation`:
432
492
 
433
493
  ```ts
434
494
  yield *
435
495
  Image.generate({
436
- model: OpenAI.configure({ apiKey }).image("gpt-image-2"),
496
+ model: openai.image("gpt-image-2"),
437
497
  prompt,
438
- images: [ImageInput.bytes(sourceBytes, "image/png")],
439
- options: { mask: ImageInput.bytes(maskBytes, "image/png") },
498
+ images: [Media.bytes(sourceBytes, "image/png")],
499
+ mask: Media.bytes(maskBytes, "image/png"),
440
500
  })
441
501
  ```
442
502
 
443
- The OpenAI adapter extracts this helper value into the edit request's native `mask` field rather than passing the
444
- tagged `ImageInput` object through as an ordinary option. On multipart requests, `http.body` can override option
445
- fields but not structural `model`, `prompt`, `image[]`, or `mask` fields, and the transport owns the multipart
446
- `Content-Type` boundary. For JSON requests, `http.body` remains the final raw-native overlay. Gemini does not fetch
447
- public HTTP URLs, and hosted Z.ai image generation does not accept image inputs. These cases fail with
448
- `InvalidRequest` before network I/O.
503
+ On multipart requests, `http.body` can override option fields but not structural `model`, `prompt`, `image[]`,
504
+ or `mask` fields, and the transport owns the multipart `Content-Type` boundary. For JSON requests, `http.body`
505
+ remains the final raw-native overlay. Gemini does not fetch public HTTP URLs, and hosted Z.ai image generation does
506
+ not accept image inputs. These cases fail with a typed `AIError` before network I/O.
449
507
 
450
508
  Provider-native image options belong to each request. Raw `http.body` fields have final precedence over them:
451
509
 
452
510
  ```ts
453
- const model = OpenAI.configure({ apiKey }).image("gpt-image-2")
454
-
455
511
  yield *
456
512
  Image.generate({
457
- model,
513
+ model: openai.image("gpt-image-2"),
458
514
  prompt,
459
- options: { quality: "medium" },
515
+ providerOptions: { quality: "medium" },
460
516
  http,
461
517
  })
462
518
  ```
@@ -466,11 +522,11 @@ xAI image models use the same request API with xAI-native controls:
466
522
  ```ts
467
523
  yield *
468
524
  Image.generate({
469
- model: XAI.configure({ apiKey }).image("any-model-id"),
525
+ model: XAI.configure({ apiKey })("any-model-id"),
470
526
  prompt,
471
- options: {
472
- n: 2,
473
- aspectRatio: "16:9",
527
+ n: 2,
528
+ aspectRatio: "16:9",
529
+ providerOptions: {
474
530
  resolution: "1k",
475
531
  responseFormat: "b64_json",
476
532
  future_option: true,
@@ -486,12 +542,12 @@ import { Google } from "@opencode/ai/providers"
486
542
 
487
543
  const googleProgram = Effect.gen(function* () {
488
544
  const response = yield* Image.generate({
489
- model: Google.configure({ apiKey }).image("any-model-id"),
545
+ model: Google.configure({ apiKey })("any-model-id"),
490
546
  prompt: "A robot tending a rooftop garden",
491
- options: {
492
- aspectRatio: "16:9",
547
+ aspectRatio: "16:9",
548
+ seed: 42,
549
+ providerOptions: {
493
550
  imageSize: "2K",
494
- seed: 42,
495
551
  thinkingLevel: "HIGH",
496
552
  includeThoughts: true,
497
553
  futureOption: true,
@@ -513,9 +569,9 @@ Z.ai image models infer open Z.ai-native options from the selected model:
513
569
  ```ts
514
570
  yield *
515
571
  Image.generate({
516
- model: ZAI.configure({ apiKey }).image("any-model-id"),
572
+ model: ZAI.configure({ apiKey })("any-model-id"),
517
573
  prompt,
518
- options: {
574
+ providerOptions: {
519
575
  quality: "hd",
520
576
  userID: "user-123",
521
577
  future_option: true,
@@ -525,8 +581,63 @@ yield *
525
581
  ```
526
582
 
527
583
  Z.ai does not include trustworthy MIME metadata for output URLs, so generated images use
528
- `application/octet-stream`. Output URLs expire after 30 days; download and persist them promptly if they must
529
- remain available.
584
+ `application/octet-stream` until materialized. Output URLs expire after 30 days; call `asset.materialize()` and
585
+ persist the bytes promptly if they must remain available.
586
+
587
+ ### Partial images
588
+
589
+ OpenAI's GPT image models stream previews. `Image.stream` sends `stream: true` with `partialImages` (0–3, default 2)
590
+ and emits `image-partial` events before each final `image`; `Image.generate` keeps the plain JSON request.
591
+ `dall-e-*` models do not stream and fail typed:
592
+
593
+ ```ts
594
+ yield *
595
+ Image.stream({
596
+ model: openai.image("gpt-image-2"),
597
+ prompt: "A lighthouse at dusk",
598
+ providerOptions: { partialImages: 2 },
599
+ }).pipe(Stream.runForEach((event) => (ImageEvent.is.imagePartial(event) ? showPreview(event.image) : Effect.void)))
600
+ ```
601
+
602
+ The provider may send fewer previews than requested when the final image is ready first.
603
+
604
+ ### Queued image providers
605
+
606
+ Black Forest Labs, fal, Replicate, and Stability's creative upscaler are submit-then-poll routes. `Image.generate`
607
+ and `Image.stream` poll for you (pass `{ poll }` to tune the interval and timeout); `Image.start` returns a
608
+ `Generation` whose `token` is serializable JSON for `Image.resume` in another process:
609
+
610
+ ```ts
611
+ import { BlackForestLabs, Stability } from "@opencode/ai/providers"
612
+
613
+ const bfl = BlackForestLabs.configure({ apiKey: process.env.BFL_API_KEY })
614
+
615
+ const generation = yield * Image.start({ model: bfl.image("flux-2-pro"), prompt, size: "1024x768" })
616
+ persist(generation.token)
617
+
618
+ const resumed = yield * Image.resume(bfl.image("flux-2-pro"), loadToken())
619
+ const response = yield * resumed.await({ poll: { interval: "2 seconds" } })
620
+ ```
621
+
622
+ - **Black Forest Labs** — results are downloaded before returning, because `result.sample` expires in 10 minutes.
623
+ - **Replicate** — inputs are model-defined, so only `prompt` lowers: sizing, count, seed, format, and files go in
624
+ `providerOptions` under the model's names, with files as `Media.Asset` (data URLs up to 256 KB, larger by URL).
625
+ Outputs are removed an hour after the prediction completes. `Prefer: wait=60` in `headers` or `http.headers` holds
626
+ the submission open so a fast prediction costs one result read.
627
+ - **Stability** — `stability.image(id)` generates inline; `stability.upscale()` is the creative upscaler, queued:
628
+
629
+ ```ts
630
+ const stability = Stability.configure({ apiKey: process.env.STABILITY_API_KEY })
631
+ const upscaled =
632
+ yield *
633
+ Image.generate(
634
+ { model: stability.upscale(), prompt: "A lighthouse", images: [yield * Media.file("./small.png")] },
635
+ { poll: { interval: "5 seconds" } },
636
+ )
637
+ ```
638
+
639
+ Imagen is not available: Google shut it down on the Gemini API, and Vertex discontinued the Imagen 4 models on
640
+ 2026-06-30. `Google.image(...)` uses Gemini-native image models.
530
641
 
531
642
  Conversational image generation remains part of the LLM interaction. OpenAI Responses exposes it through its hosted image tool:
532
643
 
@@ -544,7 +655,234 @@ const program = Effect.gen(function* () {
544
655
  })
545
656
  ```
546
657
 
547
- The hosted result is represented as a provider-executed tool call and tool result. Its image is a `file` content item with a data URI, so retaining `response.message` preserves the generated image for continuation.
658
+ The hosted result is represented as a provider-executed tool call and tool result, and the generated image is also emitted as a first-class `media` `LLMEvent` (`response.message` then carries a `media` part). Gemini image-capable models emit the same `media` event for inline image output. Retaining `response.message` preserves the generated image for continuation on both routes.
659
+
660
+ ## Video generation
661
+
662
+ Video mirrors `Image` with one difference: every provider is asynchronous, so the route is a submit-then-poll
663
+ `Generation`. Models come from `.video(...)` selectors on the `Google` (Veo), `XAI`, `Fal`, and `Runway` facades.
664
+ Common fields (`frames`, `references`, `video`, `durationSeconds`, `aspectRatio`, `resolution`, `audio`, `n`, `seed`,
665
+ `negativePrompt`) lower natively or fail with a typed `AIError` before any network call; provider-native controls live
666
+ under `providerOptions`, inferred from the selected model.
667
+
668
+ ```ts
669
+ import { Video, VideoClient } from "@opencode/ai"
670
+ import { Google } from "@opencode/ai/providers"
671
+
672
+ const google = Google.configure({ apiKey: process.env.GOOGLE_GENERATIVE_AI_API_KEY })
673
+
674
+ // Simple: submit and wait.
675
+ const program = Effect.gen(function* () {
676
+ const response = yield* Video.generate(
677
+ {
678
+ model: google.video("veo-3.1-generate-preview"),
679
+ prompt: "Panning wide shot of a calico kitten sleeping in the sunshine",
680
+ aspectRatio: "16:9",
681
+ resolution: "1080p",
682
+ durationSeconds: 8,
683
+ providerOptions: { personGeneration: "allow_adult" },
684
+ },
685
+ { poll: { interval: "10 seconds", timeout: "10 minutes" } },
686
+ )
687
+ // Veo serves files for two days behind the API key. The asset knows the deadline (`expiresAt`) and carries the
688
+ // download credentials only on the live instance (`asset.headers`), never in `source` or JSON: materialize
689
+ // before persisting, or the persisted URL cannot be fetched again.
690
+ return yield* response.video.materialize()
691
+ })
692
+
693
+ // Explicit control: keep the handle, persist the token, resume elsewhere.
694
+ const controlled = Effect.gen(function* () {
695
+ const generation = yield* Video.start({ model: google.video("veo-3.1-generate-preview"), prompt })
696
+ generation.id // provider operation / task / request id
697
+ generation.status // "queued" | "running" | "completed" | "failed" | "cancelled" | "expired"
698
+ generation.token // route-owned JSON: `{ operation }`, `{ requestID }`, `{ taskID }`, or fal's follow-up URLs
699
+ const saved = JSON.stringify(generation.token)
700
+
701
+ const resumed = yield* Video.resume(google.video("veo-3.1-generate-preview"), JSON.parse(saved))
702
+ return yield* resumed.await({ poll: { interval: "10 seconds" } })
703
+ })
704
+
705
+ // Progress as a stream: generation-queued | generation-progress | video | finish.
706
+ const events = Video.stream({ model: Runway.configure({ apiKey }).video("gen4.5"), prompt }, { poll })
707
+ ```
708
+
709
+ `VideoClient.layer` needs `RequestExecutor.Service`, and status polls, result fetches, cancels, and asset downloads
710
+ all run through the same executor with the route's auth. `Generation.await` and `Generation.events` fail with a
711
+ `Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Failed,
712
+ cancelled, and expired generations fail typed with the provider's terminal document on `reason.body`; moderation
713
+ outcomes (Veo `raiMediaFilteredReasons`, xAI `respect_moderation`, Runway `SAFETY.*` codes) surface as `notices` when
714
+ a video is still returned and as a `ContentPolicy` reason when nothing is.
715
+
716
+ Provider notes:
717
+
718
+ - **Google Veo** takes inline bytes only (materialize `url` assets first); `frames.last` requires `frames.first`;
719
+ audio is always on, so `audio: false` fails typed; one video per request. Output URLs need the API key to
720
+ download, which the returned asset holds transiently (see above).
721
+ - **xAI** sends a `video` input to `/videos/edits`, or `/videos/extensions` with `providerOptions.mode: "extend"`.
722
+ `seed` and `negativePrompt` are not supported.
723
+ - **fal** endpoints are model-specific: `durationSeconds`, `references`, and `frames.last` fail typed and belong in
724
+ `providerOptions` under the model's own names (`duration: "8s"`, `end_image_url`, …). Auth is
725
+ `Authorization: Key <FAL_KEY>`.
726
+ - **Runway** expects pixel ratios in `aspectRatio` for most models (`"1280:720"`), pins `X-Runway-Version`, reports
727
+ `usage: { type: "credits" }`, and its output URLs expire after 24–48 hours.
728
+
729
+ The promise client exposes the same surface: `ai.video.start(...)` resolves to a handle with `await`, `refresh`,
730
+ `cancel`, and `token`; `ai.video.generate`, `ai.video.resume(model, token)`, and `ai.video.stream` mirror the Effect
731
+ API.
732
+
733
+ ```ts
734
+ import { ai } from "@opencode/ai/promise"
735
+
736
+ const generation = await ai.video.start({ model, prompt })
737
+ const video = await generation.await({ poll: { interval: 10_000 }, signal })
738
+ ```
739
+
740
+ ## Speech generation
741
+
742
+ Speech (text-to-speech) is one request whose response is parsed incrementally, so every route supports both
743
+ `Speech.generate` (the whole file) and `Speech.stream` (audio chunks as they arrive). Models come from `.speech(...)`
744
+ selectors on the `OpenAI`, `Google` (Gemini TTS), `ElevenLabs`, `Cartesia`, and `Deepgram` facades. Common fields
745
+ (`voice`, `format`, `speed`, `language`, `instructions`, `timestamps`) lower natively or fail with a typed `AIError`
746
+ before any network call; provider-native controls live under `providerOptions`, inferred from the selected model.
747
+
748
+ ```ts
749
+ import { Media, Speech, SpeechClient, SpeechEvent } from "@opencode/ai"
750
+ import { ElevenLabs, OpenAI } from "@opencode/ai/providers"
751
+
752
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
753
+
754
+ // The whole file, written to disk.
755
+ const program = Effect.gen(function* () {
756
+ const response = yield* Speech.generate({
757
+ model: openai.speech("gpt-4o-mini-tts"),
758
+ text: "Hello from OpenCode.",
759
+ voice: "coral",
760
+ format: "mp3",
761
+ instructions: "Warm and unhurried.",
762
+ })
763
+ response.audio // Media.Asset with bytes; headerless PCM carries info.encoding / sampleRate / channels
764
+ response.usage // undefined: OpenAI reports tokens only on SSE streams (Gemini: tokens; ElevenLabs: credits; Deepgram: characters)
765
+ yield* Media.write(response.audio, "hello.mp3")
766
+ })
767
+
768
+ // Chunks as they arrive: audio-delta* (interleaved with timestamps) then one finish carrying the assembled asset.
769
+ const events = Speech.stream({
770
+ model: ElevenLabs.configure({ apiKey }).speech("eleven_flash_v2_5"),
771
+ text: "Hello from OpenCode.",
772
+ voice: "JBFqnCBsd6RMkjVDRZzb",
773
+ format: "pcm",
774
+ timestamps: true,
775
+ }).pipe(
776
+ Stream.tap((event) => {
777
+ if (SpeechEvent.is.audioDelta(event)) return play(event.chunk)
778
+ if (SpeechEvent.is.timestamps(event)) return highlight(event.items) // { text, startSeconds, endSeconds }[]
779
+ return Effect.void
780
+ }),
781
+ )
782
+ ```
783
+
784
+ `voice` is the provider's own identifier — a name on OpenAI and Gemini (`"coral"`, `"Kore"`), a voice id on
785
+ ElevenLabs and Cartesia. `{ id }` selects an OpenAI custom voice (`{ id: "voice_1234" }`) and means the same as the
786
+ plain string elsewhere. There is no cross-provider voice catalog. `format` is the container-level word (`mp3`, `wav`,
787
+ `pcm`, `opus`, `aac`, `flac`); sample rates and bitrates live under `providerOptions`, and a value the route cannot
788
+ produce fails as `UnsupportedOperation`. Streams buffer every chunk so `finish` can carry the whole clip.
789
+ `SpeechClient.layer` needs `RequestExecutor.Service`.
790
+
791
+ Provider notes:
792
+
793
+ - **OpenAI** streams over SSE (`stream_format: "sse"`), which is also the only place it reports token usage; `tts-1`
794
+ and `tts-1-hd` do not support SSE and stream the raw audio body instead. `pcm` is 24 kHz 16-bit mono. `language`
795
+ and `timestamps` are not supported.
796
+ - **Gemini TTS** returns raw 16-bit PCM only (`audio/L16;codec=pcm;rate=24000`), so any `format` other than `pcm`
797
+ fails typed; wrap the samples yourself. Style is directed in the text, so `instructions` and `speed` fail typed.
798
+ Only `gemini-3.1-flash-tts-preview` and later support streaming. Two-speaker audio goes through
799
+ `providerOptions.speechConfig.multiSpeakerVoiceConfig`.
800
+ - **ElevenLabs** requires `voice` (the path voice id) and authenticates with `xi-api-key`. `format` maps to the
801
+ `output_format` query parameter (`mp3_44100_128`, `pcm_24000`, `wav_24000`, `opus_48000_64`);
802
+ `providerOptions.outputFormat` sets the exact string. WAV is only available from `generate`. `timestamps: true`
803
+ selects the `with-timestamps` endpoints and yields character-level alignment. `instructions` is not supported.
804
+ - **Cartesia** requires `voice` and pins `Cartesia-Version`. `generate` defaults to MP3 from `/tts/bytes`; streams
805
+ and `timestamps: true` (word-level) use `/tts/sse`, which only serves raw PCM. `providerOptions.sampleRate`,
806
+ `bitRate`, and `encoding` complete `output_format`. No usage is reported.
807
+ - **Deepgram** Aura's voice is the model id (`aura-2-thalia-en`), so `voice` and `language` fail typed. `format`
808
+ and `providerOptions` lower to query parameters (`encoding`, `container`, `sample_rate`, `bit_rate`); `pcm` is
809
+ `linear16` without a container. Auth is `Authorization: Token <DEEPGRAM_API_KEY>`.
810
+
811
+ The promise client mirrors the Effect API; `ai.speech.stream` is an `AsyncIterable`.
812
+
813
+ ```ts
814
+ import { ai } from "@opencode/ai/promise"
815
+
816
+ const response = await ai.speech.generate({ model, text: "Hello from OpenCode.", voice: "coral" })
817
+ await Bun.write("hello.mp3", await ai.run(response.audio.bytes()))
818
+
819
+ for await (const event of ai.speech.stream({ model, text: "Hello from OpenCode.", voice: "coral" })) {
820
+ if (event.type === "audio-delta") player.write(event.chunk)
821
+ }
822
+ ```
823
+
824
+ ## Transcription
825
+
826
+ Transcription (speech-to-text) is the one modality whose providers use every route kind: OpenAI and Gemini stream,
827
+ Deepgram answers inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream` work on all of
828
+ them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with `UnsupportedOperation`
829
+ elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`, `Deepgram`, and `AssemblyAI`
830
+ facades. Common fields (`language`, `prompt`, `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower
831
+ natively or fail with a typed `AIError` before any network call; a route may return more than asked.
832
+
833
+ ```ts
834
+ import { Media, Transcription, TranscriptionEvent } from "@opencode/ai"
835
+ import { AssemblyAI, Deepgram, OpenAI } from "@opencode/ai/providers"
836
+
837
+ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
838
+
839
+ const program = Effect.gen(function* () {
840
+ const audio = yield* Media.file("./call.mp3")
841
+
842
+ // Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0").
843
+ const response = yield* Transcription.generate({
844
+ model: Deepgram.configure({ apiKey }).transcription("nova-3"),
845
+ audio,
846
+ diarize: true,
847
+ timestamps: "word",
848
+ })
849
+ response.text // "Hello from OpenCode."
850
+ response.segments // [{ text, startSeconds, endSeconds, speaker: "0" }]
851
+ response.words // [{ text, startSeconds, endSeconds, speaker, confidence }]
852
+ response.language // the provider's own value, lowercased ("en", "english", "en_us")
853
+
854
+ // Text deltas as the model transcribes, then one finish carrying the whole transcript.
855
+ yield* Transcription.stream({ model: openai.transcription("gpt-4o-mini-transcribe"), audio }).pipe(
856
+ Stream.tap((event) => (TranscriptionEvent.is.textDelta(event) ? Console.log(event.delta) : Effect.void)),
857
+ Stream.runDrain,
858
+ )
859
+
860
+ // Queued: persist the token, resume from another process, and await.
861
+ const model = AssemblyAI.configure({ apiKey }).transcription("universal-3-5-pro")
862
+ const generation = yield* Transcription.start({ model, audio })
863
+ const resumed = yield* Transcription.resume(model, JSON.parse(JSON.stringify(generation.token)))
864
+ const transcript = yield* resumed.await({ poll: { interval: "3 seconds" } })
865
+ })
866
+ ```
867
+
868
+ Inline routes emit only `finish` from `stream` (no faked deltas); queued routes emit `generation-queued` /
869
+ `generation-progress` before it. `TranscriptionClient.layer` needs `RequestExecutor.Service`.
870
+
871
+ Provider notes:
872
+
873
+ - **OpenAI** takes inline audio only; `diarize` needs `gpt-4o-transcribe-diarize`, timestamps need `whisper-1`, and `whisper-1` does not stream.
874
+ - **Gemini** needs a transcribe model (`gemini-3.5-transcribe`); `prompt` and `speakers` fail typed.
875
+ - **Deepgram** detects the language unless `language` is set; vocabulary goes in `providerOptions.keyterm`.
876
+ - **AssemblyAI** uploads inline audio before submitting and is the only route that accepts `speakers`.
877
+
878
+ The promise client mirrors the Effect API:
879
+
880
+ ```ts
881
+ const text = (await ai.transcription.generate({ model, audio })).text
882
+ for await (const event of ai.transcription.stream({ model, audio })) if (event.type === "text-delta") write(event.delta)
883
+ const generation = await ai.transcription.start({ model: assemblyai, audio })
884
+ const transcript = await generation.await({ poll: { interval: 3_000 } })
885
+ ```
548
886
 
549
887
  ## Public API
550
888
 
@@ -553,8 +891,13 @@ The hosted result is represented as a provider-executed tool call and tool resul
553
891
  - **`Message.user(...)` / `Message.assistant(...)` / `Message.tool(...)`** — message constructors from the canonical schema model.
554
892
  - **`LanguageModel.make(...)` / `ToolCallPart.make(...)` / `ToolResultPart.make(...)` / `ToolDefinition.make(...)`** — model and tool-related constructors from the canonical schema model.
555
893
  - **`LLMEvent.is.*`** — typed guards (`is.textDelta`, `is.toolCall`, `is.finish`, …) for filtering streams.
556
- - **`Image.generate({...})`** — generate images through a provider-neutral image request and response model.
894
+ - **`Image.request` / `generate` / `stream` / `start` / `resume`** — images over inline, streaming (partial previews), and queued routes through a provider-neutral request and response model.
557
895
  - **`ImageClient`** — Effect service and layer for image execution, parallel to `LLMClient`.
896
+ - **`Media`** — the shared asset type (`Media.Asset`, `Media.Source`) and constructors used by messages, tool results, and media requests.
897
+ - **`Generation`** — provider-neutral handle for an in-flight media generation (`await`, `refresh`, `cancel`, `events`) used by queued media routes.
898
+ - **`Speech.request` / `Speech.generate` / `Speech.stream`** — text-to-speech through a provider-neutral request; `SpeechClient` is its Effect service and layer.
899
+ - **`Transcription.request` / `generate` / `stream` / `start` / `resume`** — speech-to-text over inline, streaming, and queued routes; `TranscriptionClient` is its Effect service and layer.
900
+ - **`@opencode/ai/promise`** — `AI.make({ layer? })` and a default `ai` client exposing `llm`, `image`, `video`, `speech`, and `transcription` as Promise / `AsyncIterable` APIs.
558
901
 
559
902
  ## Testing
560
903
 
@@ -1,6 +1,6 @@
1
1
  import { Context, Effect, Layer } from "effect";
2
2
  import { RequestExecutor } from "../route/executor.js";
3
- import { type AIError } from "../schema/index.js";
3
+ import { AIError } from "../schema/index.js";
4
4
  import { type EvaluationOptions, type EvaluationQuestions, type EvaluationRequestFor, type EvaluationResponseFor } from "./evaluation.js";
5
5
  export type Execute = RequestExecutor.Interface["execute"];
6
6
  export interface Interface {