@opencode/ai 2.0.18 → 2.0.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/README.md +24 -12
  2. package/dist/cache-policy.js +1 -0
  3. package/dist/experimental/evaluation.d.ts +7 -0
  4. package/dist/experimental/evaluation.js +2 -0
  5. package/dist/experimental/system-one.js +10 -7
  6. package/dist/generation.d.ts +1 -1
  7. package/dist/generation.js +23 -18
  8. package/dist/index.d.ts +1 -1
  9. package/dist/index.js +1 -1
  10. package/dist/promise.d.ts +1 -1
  11. package/dist/promise.js +4 -3
  12. package/dist/protocols/alibaba-chat.d.ts +8 -2
  13. package/dist/protocols/deepgram-speech.js +9 -7
  14. package/dist/protocols/deepgram-transcription.js +4 -10
  15. package/dist/protocols/elevenlabs-transcription.d.ts +28 -0
  16. package/dist/protocols/elevenlabs-transcription.js +137 -0
  17. package/dist/protocols/gemini.js +1 -10
  18. package/dist/protocols/google-speech.js +9 -2
  19. package/dist/protocols/google-transcription.js +4 -0
  20. package/dist/protocols/google-video.js +11 -2
  21. package/dist/protocols/openai-chat.d.ts +53 -10
  22. package/dist/protocols/openai-chat.js +21 -8
  23. package/dist/protocols/openai-compatible-chat.d.ts +7 -2
  24. package/dist/protocols/openai-speech.js +6 -1
  25. package/dist/protocols/openai-transcription.js +1 -1
  26. package/dist/protocols/runway-video.js +2 -1
  27. package/dist/protocols/utils/gemini-generate-content.d.ts +6 -0
  28. package/dist/protocols/utils/gemini-generate-content.js +33 -0
  29. package/dist/protocols/utils/media-input.d.ts +4 -3
  30. package/dist/protocols/utils/media-input.js +5 -4
  31. package/dist/protocols/utils/speaker-turns.d.ts +3 -0
  32. package/dist/protocols/utils/speaker-turns.js +9 -0
  33. package/dist/protocols/utils/speech-stream.d.ts +1 -0
  34. package/dist/protocols/utils/speech-stream.js +1 -0
  35. package/dist/protocols/utils/tool-stream.d.ts +1 -0
  36. package/dist/protocols/utils/tool-stream.js +1 -1
  37. package/dist/protocols/xai-video.js +7 -1
  38. package/dist/protocols/zai-chat.d.ts +8 -2
  39. package/dist/provider-error.d.ts +5 -0
  40. package/dist/provider-error.js +46 -1
  41. package/dist/providers/alibaba.d.ts +7 -2
  42. package/dist/providers/amazon-bedrock-mantle.d.ts +7 -2
  43. package/dist/providers/anthropic-compatible.js +3 -2
  44. package/dist/providers/azure.d.ts +7 -2
  45. package/dist/providers/baseten.d.ts +14 -4
  46. package/dist/providers/cerebras.d.ts +14 -4
  47. package/dist/providers/cloudflare-ai-gateway.d.ts +14 -4
  48. package/dist/providers/cloudflare-workers-ai.d.ts +14 -4
  49. package/dist/providers/deepinfra.d.ts +14 -4
  50. package/dist/providers/deepseek.d.ts +14 -4
  51. package/dist/providers/elevenlabs.d.ts +5 -0
  52. package/dist/providers/elevenlabs.js +4 -0
  53. package/dist/providers/fireworks.d.ts +14 -4
  54. package/dist/providers/google-vertex-chat.d.ts +7 -2
  55. package/dist/providers/groq.d.ts +15 -4
  56. package/dist/providers/meta.d.ts +7 -2
  57. package/dist/providers/minimax.d.ts +7 -2
  58. package/dist/providers/moonshot.d.ts +7 -2
  59. package/dist/providers/openai-compatible.d.ts +7 -2
  60. package/dist/providers/openai.d.ts +7 -2
  61. package/dist/providers/openrouter.d.ts +27 -6
  62. package/dist/providers/togetherai.d.ts +14 -4
  63. package/dist/providers/xai.d.ts +7 -2
  64. package/dist/providers/xai.js +3 -3
  65. package/dist/providers/zai-coding-plan.d.ts +7 -2
  66. package/dist/providers/zai.d.ts +7 -2
  67. package/dist/route/media-protocol.d.ts +13 -3
  68. package/dist/route/media-protocol.js +16 -6
  69. package/dist/route/media.js +18 -4
  70. package/dist/transcription.d.ts +1 -1
  71. package/dist/transcription.js +1 -1
  72. package/package.json +3 -3
package/README.md CHANGED
@@ -129,8 +129,9 @@ VercelAIGateway.configure().experimental.evaluation("typesafe-ai/jev")
129
129
 
130
130
  OpenRouter reads `OPENROUTER_API_KEY`. Vercel reads `AI_GATEWAY_API_KEY`, then `VERCEL_OIDC_TOKEN`.
131
131
  The common API uses `boolean`; System One routes lower it to native `noul`.
132
- Choice and score confidence plus score legends remain available in provider metadata, and the
133
- provider's rounded probabilities are returned unchanged.
132
+ Choice and score answers include `confidence` when the provider returns it, such as
133
+ `response.answers.department.confidence`. Score legends remain available in provider metadata, and
134
+ the provider's rounded probabilities are returned unchanged.
134
135
 
135
136
  ## Alibaba Cloud Model Studio
136
137
 
@@ -752,7 +753,10 @@ const events = Video.stream({ model: Runway.configure({ apiKey }).video("gen4.5"
752
753
 
753
754
  Status polls, result fetches, cancels, and asset downloads all run through the same request executor with the route's
754
755
  auth. `Generation.await` and `Generation.events` fail with a
755
- `Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Failed,
756
+ `Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Status polls and result fetches retry transient
757
+ failures (rate limits, provider 5xx, network errors) with backoff that honors `retry-after`, always within
758
+ `poll.timeout`; submits and cancels never retry. Interrupting a wait (or aborting its `signal`) does not cancel the
759
+ provider job, which keeps running and billing: call `cancel()` to stop it. Failed,
756
760
  cancelled, and expired generations fail typed with the provider's terminal document on `reason.body`; moderation
757
761
  outcomes (Veo `raiMediaFilteredReasons`, xAI `respect_moderation`, Runway `SAFETY.*` codes) surface as `notices` when
758
762
  a video is still returned and as a `ContentPolicy` reason when nothing is.
@@ -773,7 +777,9 @@ Provider notes:
773
777
  The promise client exposes the same surface: `ai.video.start(...)` resolves to a handle with `await`, `events`,
774
778
  `result`, `refresh`, `cancel`, and `token`; `ai.video.generate`, `ai.video.resume(model, token)`, and
775
779
  `ai.video.stream` mirror the Effect API. The handle's `status` and `progress` are a snapshot from when it was
776
- created; `refresh()` resolves to a new handle.
780
+ created; `refresh()` resolves to a new handle. Every promise method and stream accepts `{ signal }`: like `fetch`,
781
+ aborting rejects the Promise or throws from the `for await` loop with `signal.reason` (an `AbortError` `DOMException`
782
+ unless `abort(reason)` passed one), while `break` stops a stream without throwing.
777
783
 
778
784
  ```ts
779
785
  import { ai } from "@opencode/ai/promise"
@@ -871,11 +877,12 @@ for await (const event of ai.speech.stream({ model, text: "Hello from OpenCode."
871
877
  ## Transcription
872
878
 
873
879
  Transcription (speech-to-text) is the one modality whose providers use every route kind: OpenAI and Gemini stream,
874
- Deepgram answers inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream` work on all of
875
- them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with `UnsupportedOperation`
876
- elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`, `Deepgram`, and `AssemblyAI`
877
- facades. Common fields (`language`, `prompt`, `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower
878
- natively or fail with a typed `AIError` before any network call; a route may return more than asked.
880
+ Deepgram and ElevenLabs answer inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream`
881
+ work on all of them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with
882
+ `UnsupportedOperation` elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`,
883
+ `Deepgram`, `ElevenLabs`, and `AssemblyAI` facades. Common fields (`language`, `prompt`,
884
+ `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower natively or fail with a typed `AIError` before
885
+ any network call; a route may return more than asked.
879
886
 
880
887
  ```ts
881
888
  import { Console, Effect, Stream } from "effect"
@@ -887,7 +894,7 @@ const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
887
894
  const program = Effect.gen(function* () {
888
895
  const audio = yield* Media.file("./call.mp3")
889
896
 
890
- // Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0").
897
+ // Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0", "speaker_0").
891
898
  const response = yield* Transcription.generate({
892
899
  model: Deepgram.configure({ apiKey }).transcription("nova-3"),
893
900
  audio,
@@ -897,7 +904,7 @@ const program = Effect.gen(function* () {
897
904
  response.text // "Hello from OpenCode."
898
905
  response.segments // [{ text, startSeconds, endSeconds, speaker: "0" }]
899
906
  response.words // [{ text, startSeconds, endSeconds, speaker, confidence }]
900
- response.language // the provider's own value, lowercased ("en", "english", "en_us")
907
+ response.language // the provider's own value, lowercased ("en", "eng", "english", "en_us")
901
908
 
902
909
  // Text deltas as the model transcribes, then one finish carrying the whole transcript.
903
910
  yield* Transcription.stream({ model: openai.transcription("gpt-4o-mini-transcribe"), audio }).pipe(
@@ -921,7 +928,12 @@ Provider notes:
921
928
  - **OpenAI** takes inline audio only; `diarize` needs `gpt-4o-transcribe-diarize`, timestamps need `whisper-1`, and `whisper-1` does not stream.
922
929
  - **Gemini** needs a transcribe model (`gemini-3.5-transcribe`); `prompt` and `speakers` fail typed.
923
930
  - **Deepgram** detects the language unless `language` is set; vocabulary goes in `providerOptions.keyterm`.
924
- - **AssemblyAI** uploads inline audio before submitting and is the only route that accepts `speakers`.
931
+ - **ElevenLabs** (`scribe_v2`) uploads inline audio as the multipart `file` and sends a URL as `source_url`. Words
932
+ always carry timestamps, and segments are speaker turns, so `diarize`, `timestamps: "segment"`, or `speakers` turns
933
+ on diarization. `speakers` is an upper bound (`num_speakers`); `prompt` fails typed (vocabulary goes in
934
+ `providerOptions.keyterms`), as do webhook delivery and per-channel output (`use_multi_channel` without
935
+ `multichannel_output_style: "combined"`).
936
+ - **AssemblyAI** uploads inline audio before submitting and treats `speakers` as the exact speaker count.
925
937
 
926
938
  The promise client mirrors the Effect API:
927
939
 
@@ -37,6 +37,7 @@ const resolve = (policy) => {
37
37
  // whole policy pass for these — emitting hints would be harmless but pointless.
38
38
  const RESPECTS_INLINE_HINTS = new Set([
39
39
  "anthropic-messages",
40
+ "anthropic-compatible-messages",
40
41
  "google-vertex-messages",
41
42
  "bedrock-converse",
42
43
  "openrouter",
@@ -54,12 +54,14 @@ export declare const ChoiceAnswer: Schema.Struct<{
54
54
  readonly type: Schema.Literal<"choice">;
55
55
  readonly choice: Schema.String;
56
56
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
57
+ readonly confidence: Schema.optional<Schema.Number>;
57
58
  }>;
58
59
  export type ChoiceAnswer = Schema.Schema.Type<typeof ChoiceAnswer>;
59
60
  export declare const ScoreAnswer: Schema.Struct<{
60
61
  readonly type: Schema.Literal<"score">;
61
62
  readonly score: Schema.Number;
62
63
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
64
+ readonly confidence: Schema.optional<Schema.Number>;
63
65
  }>;
64
66
  export type ScoreAnswer = Schema.Schema.Type<typeof ScoreAnswer>;
65
67
  export declare const BooleanAnswer: Schema.Struct<{
@@ -71,10 +73,12 @@ export declare const EvaluationAnswer: Schema.toTaggedUnion<"type", readonly [Sc
71
73
  readonly type: Schema.Literal<"choice">;
72
74
  readonly choice: Schema.String;
73
75
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
76
+ readonly confidence: Schema.optional<Schema.Number>;
74
77
  }>, Schema.Struct<{
75
78
  readonly type: Schema.Literal<"score">;
76
79
  readonly score: Schema.Number;
77
80
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
81
+ readonly confidence: Schema.optional<Schema.Number>;
78
82
  }>, Schema.Struct<{
79
83
  readonly type: Schema.Literal<"boolean">;
80
84
  readonly probability: Schema.Number;
@@ -87,6 +91,7 @@ export type AnswerFor<Question extends EvaluationQuestion> = Question extends {
87
91
  readonly type: "choice";
88
92
  readonly choice: Extract<keyof Criteria, string>;
89
93
  readonly probabilities?: Readonly<Record<Extract<keyof Criteria, string>, number>>;
94
+ readonly confidence?: number;
90
95
  } : Question extends {
91
96
  readonly type: "score";
92
97
  } ? ScoreAnswer : BooleanAnswer;
@@ -206,10 +211,12 @@ declare const EvaluationResponse_base: Schema.Class<EvaluationResponse, Schema.S
206
211
  readonly type: Schema.Literal<"choice">;
207
212
  readonly choice: Schema.String;
208
213
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
214
+ readonly confidence: Schema.optional<Schema.Number>;
209
215
  }>, Schema.Struct<{
210
216
  readonly type: Schema.Literal<"score">;
211
217
  readonly score: Schema.Number;
212
218
  readonly probabilities: Schema.optional<Schema.$Record<Schema.String, Schema.Number>>;
219
+ readonly confidence: Schema.optional<Schema.Number>;
213
220
  }>, Schema.Struct<{
214
221
  readonly type: Schema.Literal<"boolean">;
215
222
  readonly probability: Schema.Number;
@@ -33,11 +33,13 @@ export const ChoiceAnswer = Schema.Struct({
33
33
  type: Schema.Literal("choice"),
34
34
  choice: Schema.String,
35
35
  probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
36
+ confidence: Schema.optional(Probability),
36
37
  });
37
38
  export const ScoreAnswer = Schema.Struct({
38
39
  type: Schema.Literal("score"),
39
40
  score: Schema.Number,
40
41
  probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
42
+ confidence: Schema.optional(Probability),
41
43
  });
42
44
  export const BooleanAnswer = Schema.Struct({
43
45
  type: Schema.Literal("boolean"),
@@ -80,34 +80,37 @@ export const model = (cfg) => EvaluationModel.make({
80
80
  const fail = (message, cause, body) => new AIError({ reason: new InvalidProviderOutputError({ route: "system-one", message, body, http, cause }) });
81
81
  const text = yield* res.text.pipe(Effect.mapError((cause) => fail("Failed to read the System One response", cause)));
82
82
  const data = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(Response))(text).pipe(Effect.mapError((cause) => fail("System One returned an invalid response", cause, text)));
83
- const confidence = {};
84
83
  const legend = {};
85
84
  const answers = Object.fromEntries(Object.entries(data.answers).map(([id, answer]) => {
86
85
  if (answer.type === "noul")
87
86
  return [id, { type: "boolean", probability: answer.noul }];
88
87
  if (answer.type === "choice") {
89
- if (answer.confidence !== undefined)
90
- confidence[id] = answer.confidence;
91
88
  return [
92
89
  id,
93
90
  {
94
91
  type: "choice",
95
92
  choice: answer.choice,
96
93
  probabilities: answer.probabilities,
94
+ ...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
97
95
  },
98
96
  ];
99
97
  }
100
- if (answer.confidence !== undefined)
101
- confidence[id] = answer.confidence;
102
98
  if (answer.legend !== undefined)
103
99
  legend[id] = answer.legend;
104
- return [id, { type: "score", score: answer.score, probabilities: answer.probabilities }];
100
+ return [
101
+ id,
102
+ {
103
+ type: "score",
104
+ score: answer.score,
105
+ probabilities: answer.probabilities,
106
+ ...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
107
+ },
108
+ ];
105
109
  }));
106
110
  const meta = {
107
111
  ...(data.id === undefined ? {} : { responseId: data.id }),
108
112
  ...(data.provider === undefined ? {} : { provider: data.provider }),
109
113
  ...data.provider_metadata?.[cfg.providerMetadataKey],
110
- ...(Object.keys(confidence).length === 0 ? {} : { confidence }),
111
114
  ...(Object.keys(legend).length === 0 ? {} : { legend }),
112
115
  };
113
116
  return new EvaluationResponse({
@@ -72,8 +72,8 @@ export declare class Generation<Response> {
72
72
  */
73
73
  events(options?: AwaitOptions): Stream.Stream<Event, AIError>;
74
74
  private event;
75
- private timeoutError;
76
75
  private poll;
77
76
  private schedule;
78
77
  }
78
+ /** `events` followed by the expanded result, with the result fetch bounded by the same `poll.timeout` deadline. */
79
79
  export declare const resultEvents: <Response, A>(generation: Generation<Response>, expand: (response: Response) => ReadonlyArray<A>, options?: AwaitOptions) => Stream.Stream<Observation | A, AIError>;
@@ -56,7 +56,7 @@ export class Generation {
56
56
  const settled = this.terminal ? Effect.succeed(this) : this.poll(options?.poll);
57
57
  return settled.pipe(
58
58
  // Non-completed terminal states also go through `result` so the route can surface its provider failure body.
59
- Effect.flatMap((generation) => generation.result()), Effect.timeoutOrElse({ duration: timeout, orElse: () => this.timeoutError(timeout) }));
59
+ Effect.flatMap((generation) => generation.result()), Effect.timeoutOrElse({ duration: timeout, orElse: () => timeoutError(this.id, timeout) }));
60
60
  }
61
61
  cancel() {
62
62
  return this.route.cancel ?? Effect.void;
@@ -73,14 +73,7 @@ export class Generation {
73
73
  const timeout = Duration.fromInputUnsafe(options?.poll?.timeout ?? DEFAULT_POLL_TIMEOUT);
74
74
  return Stream.unwrap(Clock.currentTimeMillis.pipe(Effect.map((start) => {
75
75
  const deadline = start + Duration.toMillis(timeout);
76
- // Fail before polling once the deadline has passed: a fast status request could otherwise win the zero-budget
77
- // race and schedule another zero-delay poll.
78
- const refresh = Clock.currentTimeMillis.pipe(Effect.flatMap((now) => now >= deadline
79
- ? this.timeoutError(timeout)
80
- : this.refresh().pipe(Effect.timeoutOrElse({
81
- duration: Duration.millis(deadline - now),
82
- orElse: () => this.timeoutError(timeout),
83
- }))));
76
+ const refresh = within(this.refresh(), this.id, timeout, deadline);
84
77
  const schedule = this.schedule(options?.poll).pipe(Schedule.modifyDelay((meta) => Effect.succeed(Duration.min(meta.duration, Duration.millis(Math.max(0, deadline - meta.now))))));
85
78
  return Stream.fromEffectSchedule(refresh, schedule).pipe(Stream.takeUntil((generation) => generation.terminal), Stream.map((generation) => generation.event()));
86
79
  })));
@@ -92,14 +85,6 @@ export class Generation {
92
85
  return { type: "generation-queued", id: this.id, position: this.position };
93
86
  return { type: "generation-progress", id: this.id, progress: this.progress };
94
87
  }
95
- timeoutError(timeout) {
96
- return new AIError({
97
- reason: new TimeoutError({
98
- message: `Generation ${this.id} did not finish within ${Duration.format(timeout)}`,
99
- timeoutMs: Duration.toMillis(timeout),
100
- }),
101
- });
102
- }
103
88
  poll(poll) {
104
89
  return this.refresh().pipe(Effect.repeat({ schedule: this.schedule(poll), until: (generation) => generation.terminal }));
105
90
  }
@@ -107,4 +92,24 @@ export class Generation {
107
92
  return Schedule.spaced(poll?.interval ?? DEFAULT_POLL_INTERVAL);
108
93
  }
109
94
  }
110
- export const resultEvents = (generation, expand, options) => generation.events(options).pipe(Stream.filter((event) => event.type !== "generation-finished"), Stream.concat(Stream.fromIterableEffect(Effect.map(generation.result(), expand))));
95
+ /** `events` followed by the expanded result, with the result fetch bounded by the same `poll.timeout` deadline. */
96
+ export const resultEvents = (generation, expand, options) => {
97
+ const timeout = Duration.fromInputUnsafe(options?.poll?.timeout ?? DEFAULT_POLL_TIMEOUT);
98
+ return Stream.unwrap(Clock.currentTimeMillis.pipe(Effect.map((start) => generation.events(options).pipe(Stream.filter((event) => event.type !== "generation-finished"), Stream.concat(Stream.fromIterableEffect(within(generation.result(), generation.id, timeout, start + Duration.toMillis(timeout)).pipe(Effect.map(expand))))))));
99
+ };
100
+ /**
101
+ * Run `effect` within the time left until `deadline`. Fails before starting once the deadline has passed: a fast
102
+ * request could otherwise win the zero-budget race and schedule another zero-delay poll.
103
+ */
104
+ const within = (effect, id, timeout, deadline) => Clock.currentTimeMillis.pipe(Effect.flatMap((now) => now >= deadline
105
+ ? Effect.fail(timeoutError(id, timeout))
106
+ : effect.pipe(Effect.timeoutOrElse({
107
+ duration: Duration.millis(deadline - now),
108
+ orElse: () => Effect.fail(timeoutError(id, timeout)),
109
+ }))));
110
+ const timeoutError = (id, timeout) => new AIError({
111
+ reason: new TimeoutError({
112
+ message: `Generation ${id} did not finish within ${Duration.format(timeout)}`,
113
+ timeoutMs: Duration.toMillis(timeout),
114
+ }),
115
+ });
package/dist/index.d.ts CHANGED
@@ -4,7 +4,7 @@ export { ImageClient } from "./image-client.js";
4
4
  export { Auth } from "./route/auth.js";
5
5
  export { Provider } from "./provider.js";
6
6
  export { ProviderPackage } from "./provider-package.js";
7
- export { isContextOverflow, isContextOverflowFailure } from "./provider-error.js";
7
+ export { isContextOverflow, isContextOverflowFailure, isRetryable } from "./provider-error.js";
8
8
  export type { RouteLanguageModelInput, RouteRoutedLanguageModelInput, Interface as LLMClientShape, LLMClientService, } from "./route/client.js";
9
9
  export * from "./schema/index.js";
10
10
  export { ImageAspectRatio, ImageEvent, ImageModel, ImageModelSchema, ImageRequest, ImageResponse, ImageSize, } from "./image.js";
package/dist/index.js CHANGED
@@ -4,7 +4,7 @@ export { ImageClient } from "./image-client.js";
4
4
  export { Auth } from "./route/auth.js";
5
5
  export { Provider } from "./provider.js";
6
6
  export { ProviderPackage } from "./provider-package.js";
7
- export { isContextOverflow, isContextOverflowFailure } from "./provider-error.js";
7
+ export { isContextOverflow, isContextOverflowFailure, isRetryable } from "./provider-error.js";
8
8
  export * from "./schema/index.js";
9
9
  export { ImageAspectRatio, ImageEvent, ImageModel, ImageModelSchema, ImageRequest, ImageResponse, ImageSize, } from "./image.js";
10
10
  export { Image } from "./image.js";
package/dist/promise.d.ts CHANGED
@@ -30,7 +30,7 @@ export type GenerationHandle<Response> = Snapshot & {
30
30
  /** Serializable JSON; pass it back to `resume` from another process. */
31
31
  readonly token: unknown;
32
32
  readonly await: (options?: AwaitOptions & RunOptions) => Promise<Response>;
33
- /** Status observations until the first terminal one, polling like `await`; abort ends iteration without throwing. */
33
+ /** Status observations until the first terminal one, polling like `await`; abort throws `signal.reason`. */
34
34
  readonly events: (options?: AwaitOptions & RunOptions) => AsyncIterable<Event>;
35
35
  /** The result without polling; fails when the generation has not completed. */
36
36
  readonly result: (options?: RunOptions) => Promise<Response>;
package/dist/promise.js CHANGED
@@ -10,21 +10,22 @@ import { Speech } from "./speech.js";
10
10
  import { Transcription, } from "./transcription.js";
11
11
  import { fileMediaType } from "./utils/media-type.js";
12
12
  import { Video } from "./video.js";
13
+ // Fails with `signal.reason` so aborted calls reject and aborted streams throw like `fetch`: an `AbortError` by default.
13
14
  const abortEffect = (signal) => signal === undefined
14
15
  ? Effect.never
15
16
  : Effect.callback((resume) => {
16
17
  if (signal.aborted) {
17
- resume(Effect.void);
18
+ resume(Effect.fail(signal.reason));
18
19
  return;
19
20
  }
20
- const onAbort = () => resume(Effect.void);
21
+ const onAbort = () => resume(Effect.fail(signal.reason));
21
22
  signal.addEventListener("abort", onAbort, { once: true });
22
23
  return Effect.sync(() => signal.removeEventListener("abort", onAbort));
23
24
  });
24
25
  export const make = (options = {}) => {
25
26
  const runtime = ManagedRuntime.make(AIClient.layerWith(options.layer ?? RequestExecutor.fetchLayer));
26
27
  /** Run any package Effect (for example `LLMClient.compact(...)`) inside this runtime. */
27
- const run = (effect, options) => runtime.runPromise(effect, { signal: options?.signal });
28
+ const run = (effect, options) => runtime.runPromise(Effect.raceFirst(effect, abortEffect(options?.signal)));
28
29
  const iterate = (stream, options) => Stream.toAsyncIterable(Stream.unwrap(runtime.contextEffect.pipe(Effect.map((context) => stream.pipe(Stream.interruptWhen(abortEffect(options?.signal)), Stream.provideContext(context))))));
29
30
  const handle = (generation) => ({
30
31
  ...generation.snapshot,
@@ -90,12 +90,17 @@ export declare const protocol: Protocol<{
90
90
  readonly ttl?: string | undefined;
91
91
  } | undefined;
92
92
  readonly tool_calls?: readonly {
93
- readonly id: string;
94
- readonly type: "function";
95
93
  readonly function: {
96
94
  readonly name: string;
97
95
  readonly arguments: string;
98
96
  };
97
+ readonly id: string;
98
+ readonly type: "function";
99
+ readonly extra_content?: {
100
+ readonly google: {
101
+ readonly thought_signature: string;
102
+ };
103
+ } | undefined;
99
104
  }[] | undefined;
100
105
  readonly reasoning_details?: unknown;
101
106
  } | {
@@ -208,6 +213,7 @@ export declare const protocol: Protocol<{
208
213
  } | null | undefined;
209
214
  readonly id?: string | null | undefined;
210
215
  readonly index?: number | null | undefined;
216
+ readonly extra_content?: unknown;
211
217
  }[] | null | undefined;
212
218
  readonly reasoning_details?: unknown;
213
219
  } | null | undefined;
@@ -50,23 +50,25 @@ const fromRequest = Effect.fn("DeepgramSpeech.fromRequest")(function* (request)
50
50
  // ---------------------------------------------------------------------------
51
51
  // 6. Stream parsing
52
52
  // ---------------------------------------------------------------------------
53
+ /** Deepgram wraps raw encodings in WAV unless `container` is `none`, and defaults their sample rate per encoding. */
53
54
  const HEADERLESS_ENCODINGS = {
54
- linear16: "pcm_s16le",
55
- mulaw: "pcm_mulaw",
56
- alaw: "pcm_alaw",
55
+ linear16: { encoding: "pcm_s16le", sampleRate: 24000 },
56
+ mulaw: { encoding: "pcm_mulaw", sampleRate: 8000 },
57
+ alaw: { encoding: "pcm_alaw", sampleRate: 8000 },
57
58
  };
58
59
  const finish = (state, context) => {
59
60
  const headers = context.http.headers;
60
61
  const mediaType = headers["content-type"];
61
62
  const format = audioFormat(context.request);
62
- const encoding = HEADERLESS_ENCODINGS[format.encoding ?? ""];
63
+ const headerless = HEADERLESS_ENCODINGS[format.encoding ?? ""];
64
+ const container = format.container ?? (headerless === undefined ? undefined : "wav");
63
65
  const requestID = headers["dg-request-id"];
64
66
  const modelName = headers["dg-model-name"];
65
67
  return SpeechStream.finish(route, state, {
66
- ...(format.container === "none" && encoding !== undefined
67
- ? SpeechStream.pcm(encoding, SpeechStream.sampleRate(mediaType), mediaType)
68
+ ...(container === "none" && headerless !== undefined
69
+ ? SpeechStream.pcm(headerless.encoding, SpeechStream.sampleRate(mediaType) ?? context.request.providerOptions?.sampleRate ?? headerless.sampleRate, mediaType)
68
70
  : // Deepgram's default encoding is MP3; WAV is a container around any encoding.
69
- { mediaType, info: { format: format.container === "wav" ? "wav" : (format.encoding ?? "mp3") } }),
71
+ { mediaType, info: { format: container === "wav" ? "wav" : (format.encoding ?? "mp3") } }),
70
72
  usage: SpeechStream.headerUsage("characters", headers["dg-char-count"]),
71
73
  providerMetadata: requestID === undefined && modelName === undefined
72
74
  ? undefined
@@ -5,6 +5,7 @@ import { mergeJsonRecords } from "../schema/index.js";
5
5
  import { TranscriptionModel, TranscriptionResponse } from "../transcription.js";
6
6
  import { ProviderShared } from "./shared.js";
7
7
  import { MediaInput } from "./utils/media-input.js";
8
+ import { SpeakerTurns } from "./utils/speaker-turns.js";
8
9
  const route = MediaProtocol.identity({ id: "deepgram-transcription", name: "Deepgram", provider: "deepgram" });
9
10
  export const DEFAULT_BASE_URL = "https://api.deepgram.com";
10
11
  export const PATH = "/v1/listen";
@@ -63,15 +64,6 @@ const fromRequest = Effect.fn("DeepgramTranscription.fromRequest")(function* (re
63
64
  const decodeListen = route.decodeJson(ListenResponse);
64
65
  const speaker = (value) => (value === undefined ? undefined : String(value));
65
66
  const wordText = (word) => word.punctuated_word ?? word.word;
66
- // Utterances split on pauses, not speakers: the v2 diarizer labels a whole utterance with one speaker even when its
67
- // words change speaker, so segments split each utterance at speaker changes.
68
- const speakerTurns = (words) => words.reduce((turns, word) => {
69
- const last = turns.at(-1);
70
- if (last === undefined || last[0].speaker !== word.speaker)
71
- return [...turns, [word]];
72
- last.push(word);
73
- return turns;
74
- }, []);
75
67
  const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(function* (response) {
76
68
  const output = yield* decodeListen(response);
77
69
  const channel = output.value.results.channels[0];
@@ -82,6 +74,8 @@ const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(functio
82
74
  const requestID = output.value.metadata?.request_id;
83
75
  return new TranscriptionResponse({
84
76
  text: alternative.transcript,
77
+ // Utterances split on pauses, not speakers: the v2 diarizer labels a whole utterance with one speaker even when
78
+ // its words change speaker, so segments split each utterance at speaker changes.
85
79
  segments: output.value.results.utterances?.flatMap((utterance) => utterance.words === undefined || utterance.words.length === 0
86
80
  ? [
87
81
  {
@@ -91,7 +85,7 @@ const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(functio
91
85
  speaker: speaker(utterance.speaker),
92
86
  },
93
87
  ]
94
- : speakerTurns(utterance.words).map((turn) => ({
88
+ : SpeakerTurns.group(utterance.words, (word) => word.speaker).map((turn) => ({
95
89
  text: turn.map(wordText).join(" "),
96
90
  startSeconds: turn[0].start,
97
91
  endSeconds: turn[turn.length - 1].end,
@@ -0,0 +1,28 @@
1
+ import { MediaProtocol } from "../route/media-protocol.js";
2
+ import { MediaRoute } from "../route/media.js";
3
+ import { type OpenString } from "../schema/index.js";
4
+ import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js";
5
+ export declare const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
6
+ export declare const PATH = "/v1/speech-to-text";
7
+ export type ElevenLabsTranscriptionOptions = {
8
+ readonly tag_audio_events?: boolean;
9
+ readonly timestamps_granularity?: OpenString<"none" | "word" | "character">;
10
+ readonly diarization_threshold?: number;
11
+ readonly file_format?: OpenString<"pcm_s16le_16" | "other">;
12
+ readonly temperature?: number;
13
+ readonly seed?: number;
14
+ readonly keyterms?: ReadonlyArray<string>;
15
+ readonly no_verbatim?: boolean;
16
+ readonly detect_speaker_roles?: boolean;
17
+ readonly use_speaker_library?: boolean;
18
+ readonly entity_detection?: string | ReadonlyArray<string>;
19
+ readonly entity_redaction?: string | ReadonlyArray<string>;
20
+ readonly entity_redaction_mode?: OpenString<"redacted" | "entity_type" | "enumerated_entity_type">;
21
+ } & Record<string, unknown>;
22
+ export type Request = TranscriptionRequestFor<ElevenLabsTranscriptionOptions>;
23
+ export declare const protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
24
+ export declare const model: (input: MediaRoute.ModelInput) => TranscriptionModel<ElevenLabsTranscriptionOptions>;
25
+ export declare const ElevenLabsTranscription: {
26
+ readonly protocol: MediaProtocol.Inline<Request, TranscriptionResponse>;
27
+ readonly model: (input: MediaRoute.ModelInput) => TranscriptionModel<ElevenLabsTranscriptionOptions>;
28
+ };
@@ -0,0 +1,137 @@
1
+ import { Effect, Schema } from "effect";
2
+ import { MediaProtocol } from "../route/media-protocol.js";
3
+ import { MediaRoute } from "../route/media.js";
4
+ import { mergeJsonRecords } from "../schema/index.js";
5
+ import { TranscriptionModel, TranscriptionResponse } from "../transcription.js";
6
+ import { mediaTypeExtension } from "../utils/media-type.js";
7
+ import { ProviderShared, optionalNull } from "./shared.js";
8
+ import { MediaInput } from "./utils/media-input.js";
9
+ import { SpeakerTurns } from "./utils/speaker-turns.js";
10
+ const route = MediaProtocol.identity({
11
+ id: "elevenlabs-transcription",
12
+ name: "ElevenLabs Transcription",
13
+ provider: "elevenlabs",
14
+ });
15
+ export const DEFAULT_BASE_URL = "https://api.elevenlabs.io";
16
+ export const PATH = "/v1/speech-to-text";
17
+ // ---------------------------------------------------------------------------
18
+ // 2. Response schema
19
+ // ---------------------------------------------------------------------------
20
+ /** `type` is `word`, `spacing` (the whitespace between words), or `audio_event` (`(laughter)`). */
21
+ const Token = Schema.Struct({
22
+ text: Schema.String,
23
+ type: Schema.String,
24
+ start: optionalNull(Schema.Number),
25
+ end: optionalNull(Schema.Number),
26
+ speaker_id: optionalNull(Schema.String),
27
+ logprob: optionalNull(Schema.Number),
28
+ });
29
+ const Transcript = Schema.Struct({
30
+ language_code: optionalNull(Schema.String),
31
+ text: Schema.String,
32
+ words: optionalNull(Schema.Array(Token)),
33
+ transcription_id: optionalNull(Schema.String),
34
+ audio_duration_secs: optionalNull(Schema.Number),
35
+ });
36
+ // ---------------------------------------------------------------------------
37
+ // 5. Request body construction
38
+ // ---------------------------------------------------------------------------
39
+ /** Speaker turns are the only segments ElevenLabs can produce, and `num_speakers` only applies to diarization. */
40
+ const diarizes = (request) => request.diarize === true || request.timestamps === "segment" || request.speakers !== undefined;
41
+ const RESERVED_FORM_FIELDS = new Set([
42
+ "file",
43
+ "cloud_storage_url",
44
+ "source_url",
45
+ "model_id",
46
+ "language_code",
47
+ "diarize",
48
+ "num_speakers",
49
+ ]);
50
+ const validate = (request, overlay) => {
51
+ // Webhook requests return 202 with no transcript; the result arrives at a configured webhook instead.
52
+ if (overlay.webhook === true)
53
+ return Effect.fail(route.unsupported("transcription.webhook", `${route.name} does not deliver to webhooks`));
54
+ // Separate multichannel output replaces the transcript with one transcript per channel.
55
+ if (overlay.use_multi_channel === true && overlay.multichannel_output_style !== "combined")
56
+ return Effect.fail(route.unsupported("transcription.multichannel", `${route.name} returns a single transcript; set multichannel_output_style: "combined" to merge channels`));
57
+ if (overlay.timestamps_granularity === "none" && (request.timestamps === "word" || diarizes(request)))
58
+ return Effect.fail(route.unsupported("media.timestamps", `${route.name} cannot return word timestamps or speaker turns with timestamps_granularity: "none"`));
59
+ return Effect.void;
60
+ };
61
+ const fromRequest = Effect.fn("ElevenLabsTranscription.fromRequest")(function* (request) {
62
+ const overlay = mergeJsonRecords(request.providerOptions, request.http?.body) ?? {};
63
+ yield* validate(request, overlay);
64
+ const form = new FormData();
65
+ const url = ProviderShared.mediaUrl(request.audio);
66
+ if (url === undefined) {
67
+ const extension = mediaTypeExtension(request.audio.mediaType);
68
+ const audio = yield* MediaInput.inlineBytes(route.id, request.audio);
69
+ form.append("file", MediaInput.blob(audio, request.audio.mediaType), extension === undefined ? "audio" : `audio.${extension}`);
70
+ }
71
+ MediaInput.appendFields(form, {
72
+ model_id: request.model.id,
73
+ // `cloud_storage_url` is deprecated in favor of `source_url`, which accepts any hosted audio or video URL.
74
+ source_url: url,
75
+ language_code: request.language,
76
+ diarize: diarizes(request) ? true : undefined,
77
+ num_speakers: request.speakers,
78
+ }, { overlay, reserved: RESERVED_FORM_FIELDS, repeatArrays: "key" });
79
+ return MediaProtocol.multipart(form);
80
+ });
81
+ // ---------------------------------------------------------------------------
82
+ // 6. Response decoding
83
+ // ---------------------------------------------------------------------------
84
+ const decodeTranscript = route.decodeJson(Transcript);
85
+ const isTimedWord = (token) => token.type === "word" && typeof token.start === "number" && typeof token.end === "number";
86
+ /** Turn text keeps the provider's own spacing tokens, so languages written without spaces are not re-spaced. */
87
+ const speakerTurns = (tokens) => SpeakerTurns.group(tokens.filter((token) => token.type === "word" || token.type === "spacing"), (token) => token.speaker_id).flatMap((turn) => {
88
+ const words = turn.filter(isTimedWord);
89
+ if (words.length === 0)
90
+ return [];
91
+ return [
92
+ {
93
+ text: turn
94
+ .map((token) => token.text)
95
+ .join("")
96
+ .trim(),
97
+ startSeconds: words[0].start,
98
+ endSeconds: words[words.length - 1].end,
99
+ speaker: turn[0].speaker_id ?? undefined,
100
+ },
101
+ ];
102
+ });
103
+ const decodeResponse = Effect.fn("ElevenLabsTranscription.decodeResponse")(function* (response, context) {
104
+ const output = yield* decodeTranscript(response);
105
+ const transcript = output.value;
106
+ const tokens = transcript.words ?? [];
107
+ const duration = transcript.audio_duration_secs ?? undefined;
108
+ const transcriptionID = transcript.transcription_id ?? undefined;
109
+ return new TranscriptionResponse({
110
+ text: transcript.text,
111
+ segments: diarizes(context.request) ? speakerTurns(tokens) : undefined,
112
+ words: tokens.filter(isTimedWord).map((word) => ({
113
+ text: word.text,
114
+ startSeconds: word.start,
115
+ endSeconds: word.end,
116
+ speaker: word.speaker_id ?? undefined,
117
+ confidence: typeof word.logprob === "number" ? Math.exp(word.logprob) : undefined,
118
+ })),
119
+ language: transcript.language_code?.toLowerCase(),
120
+ durationSeconds: duration,
121
+ usage: duration === undefined ? undefined : { type: "seconds", seconds: duration },
122
+ providerMetadata: transcriptionID === undefined ? undefined : { elevenlabs: { transcriptionId: transcriptionID } },
123
+ });
124
+ });
125
+ // ---------------------------------------------------------------------------
126
+ // 7. Protocol and route
127
+ // ---------------------------------------------------------------------------
128
+ export const protocol = MediaProtocol.inline(route, {
129
+ unsupported: ["prompt"],
130
+ body: { from: fromRequest },
131
+ response: { decode: decodeResponse },
132
+ });
133
+ export const model = (input) => TranscriptionModel.fromRoute({ protocol, baseURL: DEFAULT_BASE_URL, path: PATH }, input);
134
+ export const ElevenLabsTranscription = {
135
+ protocol,
136
+ model,
137
+ };
@@ -435,16 +435,7 @@ const mapFinishReason = (finishReason, hasToolCalls) => {
435
435
  return hasToolCalls ? "tool-calls" : "stop";
436
436
  if (finishReason === "MAX_TOKENS")
437
437
  return "length";
438
- if (finishReason === "IMAGE_SAFETY" ||
439
- finishReason === "RECITATION" ||
440
- finishReason === "SAFETY" ||
441
- finishReason === "BLOCKLIST" ||
442
- finishReason === "PROHIBITED_CONTENT" ||
443
- finishReason === "SPII" ||
444
- finishReason === "MODEL_ARMOR" ||
445
- finishReason === "IMAGE_PROHIBITED_CONTENT" ||
446
- finishReason === "IMAGE_RECITATION" ||
447
- finishReason === "LANGUAGE")
438
+ if (GeminiGenerateContent.contentFiltered(finishReason))
448
439
  return "content-filter";
449
440
  if (finishReason === "MALFORMED_FUNCTION_CALL" ||
450
441
  finishReason === "UNEXPECTED_TOOL_CALL" ||