@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
import { default as OpenAI } from 'openai';
|
|
2
|
+
import { OpenAIBaseChatCompletionsTextAdapter } from '@tanstack/openai-base';
|
|
3
|
+
import { StructuredOutputOptions, StructuredOutputResult } from '@tanstack/ai/adapters';
|
|
4
|
+
import { ContentPart, Modality, ModelMessage, StreamChunk, TextOptions } from '@tanstack/ai';
|
|
5
|
+
import { ChatCompletionContentPart, ChatCompletionMessageParam } from 'openai/resources/chat/completions/completions';
|
|
6
|
+
import { BYTEPLUS_CHAT_MODELS, BytePlusChatModelToolCapabilitiesByName, ResolveInputModalities, ResolveProviderOptions } from '../model-meta.js';
|
|
7
|
+
import { BytePlusMessageMetadataByModality } from '../message-types.js';
|
|
8
|
+
import { BytePlusArkConfig } from '../utils/client.js';
|
|
9
|
+
type ResolveToolCapabilities<TModel extends string> = TModel extends keyof BytePlusChatModelToolCapabilitiesByName ? NonNullable<BytePlusChatModelToolCapabilitiesByName[TModel]> : readonly [];
|
|
10
|
+
/**
|
|
11
|
+
* Configuration for the BytePlus text adapter.
|
|
12
|
+
*/
|
|
13
|
+
export interface BytePlusTextConfig extends BytePlusArkConfig {
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* Re-export of the public provider options type.
|
|
17
|
+
*/
|
|
18
|
+
export type { BytePlusTextProviderOptions } from '../text/text-provider-options.js';
|
|
19
|
+
/**
|
|
20
|
+
* BytePlus ModelArk Text (Chat) Adapter
|
|
21
|
+
*
|
|
22
|
+
* Tree-shakeable adapter for the Seed / GLM / DeepSeek / gpt-oss chat models
|
|
23
|
+
* on BytePlus ModelArk. Ark serves an OpenAI-compatible Chat Completions
|
|
24
|
+
* endpoint, so this drives the OpenAI SDK against Ark's `baseURL` — the same
|
|
25
|
+
* pattern as `ai-groq` and `ai-grok`.
|
|
26
|
+
*
|
|
27
|
+
* Three Ark behaviours are handled on top of the shared base:
|
|
28
|
+
*
|
|
29
|
+
* 1. **`reasoning_content` deltas** — Ark streams reasoning under
|
|
30
|
+
* `delta.reasoning_content` rather than the OpenAI `reasoning` field.
|
|
31
|
+
* 2. **`encrypted_content` round-trip** — thinking-summary models emit an
|
|
32
|
+
* opaque signature over the reasoning trace. See
|
|
33
|
+
* {@link BytePlusTextAdapter.processStreamChunks} and
|
|
34
|
+
* {@link BytePlusTextAdapter.convertMessage}.
|
|
35
|
+
* 3. **Per-model structured-output gating** — only 10 of the 18 shipped chat
|
|
36
|
+
* models honour `response_format: json_schema` (glm-4-7 accepts it and then
|
|
37
|
+
* ignores the schema), and Ark rejects `json_object` everywhere, so there
|
|
38
|
+
* is no JSON-mode fallback.
|
|
39
|
+
*/
|
|
40
|
+
export declare class BytePlusTextAdapter<TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number], TProviderOptions extends Record<string, any> = ResolveProviderOptions<TModel>, TInputModalities extends ReadonlyArray<Modality> = ResolveInputModalities<TModel>, TToolCapabilities extends ReadonlyArray<string> = ResolveToolCapabilities<TModel>> extends OpenAIBaseChatCompletionsTextAdapter<TModel, TProviderOptions, TInputModalities, BytePlusMessageMetadataByModality, TToolCapabilities> {
|
|
41
|
+
readonly kind: "text";
|
|
42
|
+
readonly name: "byteplus";
|
|
43
|
+
constructor(config: BytePlusTextConfig, model: TModel);
|
|
44
|
+
/**
|
|
45
|
+
* Surfaces Ark's reasoning deltas. Thinking-enabled models stream the
|
|
46
|
+
* reasoning trace as `delta.reasoning_content` (the OpenAI chunk shape has
|
|
47
|
+
* no reasoning field); the base routes this hook through both `chatStream`
|
|
48
|
+
* and `structuredOutputStream`.
|
|
49
|
+
*/
|
|
50
|
+
protected extractReasoning(chunk: OpenAI.Chat.Completions.ChatCompletionChunk): {
|
|
51
|
+
text: string;
|
|
52
|
+
} | undefined;
|
|
53
|
+
/**
|
|
54
|
+
* Captures Ark's `encrypted_content` and attaches it to the reasoning
|
|
55
|
+
* step's `STEP_FINISHED` event as its `signature`.
|
|
56
|
+
*
|
|
57
|
+
* On a thinking-summary model Ark streams the whole blob as one dedicated
|
|
58
|
+
* chunk (empty `content` and `reasoning_content`) sitting between the
|
|
59
|
+
* reasoning deltas and the content deltas — so it is always captured before
|
|
60
|
+
* the base closes the reasoning lifecycle at the first content delta.
|
|
61
|
+
*
|
|
62
|
+
* `signature` is the framework's existing provider-signature seam: the chat
|
|
63
|
+
* engine stores it on the `ThinkingPart`, which
|
|
64
|
+
* `buildAssistantMessages` carries into `ModelMessage.thinking[].signature`,
|
|
65
|
+
* which {@link BytePlusTextAdapter.convertMessage} echoes back to Ark on the
|
|
66
|
+
* next turn. No base-class change is needed — this is the same round-trip
|
|
67
|
+
* Anthropic's thinking signatures use.
|
|
68
|
+
*
|
|
69
|
+
* Only `chatStream` is covered: `structuredOutputStream` drives the SDK
|
|
70
|
+
* directly in the base with no per-chunk seam, so a structured-output turn
|
|
71
|
+
* does not capture the blob. Ark accepts a following turn without it, so the
|
|
72
|
+
* consequence is a lost reasoning-cache hit, not a failed request.
|
|
73
|
+
*/
|
|
74
|
+
protected processStreamChunks(stream: AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk>, options: TextOptions, aguiState: {
|
|
75
|
+
runId: string;
|
|
76
|
+
threadId: string;
|
|
77
|
+
messageId: string;
|
|
78
|
+
hasEmittedRunStarted: boolean;
|
|
79
|
+
}): AsyncIterable<StreamChunk>;
|
|
80
|
+
/**
|
|
81
|
+
* Echoes a captured `encrypted_content` blob back on outgoing assistant
|
|
82
|
+
* messages so multi-turn conversations replay it verbatim, as Ark's
|
|
83
|
+
* thinking-summary docs require.
|
|
84
|
+
*
|
|
85
|
+
* The gate is `emitsEncryptedContent(this.model)` — the model being called
|
|
86
|
+
* now, not the provenance of the history. That guarantees a signature is
|
|
87
|
+
* never sent to a model that has no `encrypted_content` concept. It does
|
|
88
|
+
* NOT identify who produced the signature: `ModelMessage` carries no
|
|
89
|
+
* provider field, so a foreign signature (e.g. an Anthropic thinking
|
|
90
|
+
* signature in replayed cross-provider history) WILL be forwarded when the
|
|
91
|
+
* current model is a thinking-summary model. No shape guard is attempted —
|
|
92
|
+
* the blob is opaque and Ark is the only party that can validate it.
|
|
93
|
+
*
|
|
94
|
+
* Absence is never an error: a live probe confirmed Ark accepts a turn whose
|
|
95
|
+
* assistant message omits `encrypted_content`.
|
|
96
|
+
*/
|
|
97
|
+
protected convertMessage(message: ModelMessage): ChatCompletionMessageParam;
|
|
98
|
+
/**
|
|
99
|
+
* Adds the Ark-only content parts on top of the base's text/image handling:
|
|
100
|
+
* `video_url`, URL-addressed `input_audio`, and the extra `image_url`
|
|
101
|
+
* fields (`detail: 'xhigh'`, `image_pixel_limit`).
|
|
102
|
+
*/
|
|
103
|
+
protected convertContentPart(part: ContentPart): ChatCompletionContentPart | null;
|
|
104
|
+
/**
|
|
105
|
+
* Only the models in {@link BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS} accept
|
|
106
|
+
* `response_format: json_schema`; the rest reject it with a 400.
|
|
107
|
+
*
|
|
108
|
+
* Returning `false` for a rejecting model does not make structured output
|
|
109
|
+
* work — Ark has no `json_object` fallback to downgrade to. What it buys is
|
|
110
|
+
* keeping `response_format` out of the request the engine would otherwise
|
|
111
|
+
* build: with the hook false the engine takes its separate finalization
|
|
112
|
+
* path, and the guard in {@link BytePlusTextAdapter.structuredOutput} /
|
|
113
|
+
* {@link BytePlusTextAdapter.structuredOutputStream} stops that *before*
|
|
114
|
+
* any HTTP call. So a `chat({ outputSchema })` on a rejecting model fails
|
|
115
|
+
* loudly, named, without a schema Ark would 400 on ever leaving the
|
|
116
|
+
* process — rather than 400-ing on every turn, or (worse) parsing prose as
|
|
117
|
+
* if it were JSON.
|
|
118
|
+
*
|
|
119
|
+
* Tools without a schema are unaffected: `tools` alone never involves
|
|
120
|
+
* `response_format`.
|
|
121
|
+
*/
|
|
122
|
+
supportsCombinedToolsAndSchema(): boolean;
|
|
123
|
+
structuredOutput(options: StructuredOutputOptions<TProviderOptions>): Promise<StructuredOutputResult<unknown>>;
|
|
124
|
+
structuredOutputStream(options: StructuredOutputOptions<TProviderOptions>): AsyncIterable<StreamChunk>;
|
|
125
|
+
/**
|
|
126
|
+
* Explains why structured output is unavailable, or `undefined` when the
|
|
127
|
+
* model supports it. Ark rejects `response_format: json_object` on every
|
|
128
|
+
* model, so there is no JSON-mode fallback to degrade to — failing loud
|
|
129
|
+
* here beats a raw upstream 400.
|
|
130
|
+
*/
|
|
131
|
+
private structuredOutputUnsupportedMessage;
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Creates a BytePlus text adapter with an explicit API key.
|
|
135
|
+
*
|
|
136
|
+
* @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)
|
|
137
|
+
* @param apiKey - Your BytePlus Ark API key
|
|
138
|
+
* @param config - Optional additional configuration
|
|
139
|
+
*
|
|
140
|
+
* @example
|
|
141
|
+
* ```typescript
|
|
142
|
+
* const adapter = createBytePlusText('seed-2-0-lite-260428', 'ark-...')
|
|
143
|
+
* ```
|
|
144
|
+
*/
|
|
145
|
+
export declare function createBytePlusText<TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number]>(model: TModel, apiKey: string, config?: Omit<BytePlusTextConfig, 'apiKey'>): BytePlusTextAdapter<TModel>;
|
|
146
|
+
/**
|
|
147
|
+
* Creates a BytePlus text adapter with the API key read from `ARK_API_KEY`.
|
|
148
|
+
*
|
|
149
|
+
* @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)
|
|
150
|
+
* @param config - Optional configuration (excluding `apiKey`)
|
|
151
|
+
* @throws Error if `ARK_API_KEY` is not set
|
|
152
|
+
*
|
|
153
|
+
* @example
|
|
154
|
+
* ```typescript
|
|
155
|
+
* const adapter = byteplusText('seed-2-0-lite-260428')
|
|
156
|
+
*
|
|
157
|
+
* const stream = chat({
|
|
158
|
+
* adapter,
|
|
159
|
+
* messages: [{ role: 'user', content: 'Hello!' }],
|
|
160
|
+
* })
|
|
161
|
+
* ```
|
|
162
|
+
*/
|
|
163
|
+
export declare function byteplusText<TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number]>(model: TModel, config?: Omit<BytePlusTextConfig, 'apiKey'>): BytePlusTextAdapter<TModel>;
|
|
@@ -0,0 +1,347 @@
|
|
|
1
|
+
import { getBytePlusArkApiKeyFromEnv, withBytePlusArkDefaults } from "../utils/client.js";
|
|
2
|
+
import { BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, emitsEncryptedContent, supportsStructuredOutput } from "../model-meta.js";
|
|
3
|
+
import { EventType } from "@tanstack/ai";
|
|
4
|
+
import { generateId } from "@tanstack/ai-utils";
|
|
5
|
+
import OpenAI from "openai";
|
|
6
|
+
import { OpenAIBaseChatCompletionsTextAdapter } from "@tanstack/openai-base";
|
|
7
|
+
//#region src/adapters/text.ts
|
|
8
|
+
/**
|
|
9
|
+
* BytePlus ModelArk Text (Chat) Adapter
|
|
10
|
+
*
|
|
11
|
+
* Tree-shakeable adapter for the Seed / GLM / DeepSeek / gpt-oss chat models
|
|
12
|
+
* on BytePlus ModelArk. Ark serves an OpenAI-compatible Chat Completions
|
|
13
|
+
* endpoint, so this drives the OpenAI SDK against Ark's `baseURL` — the same
|
|
14
|
+
* pattern as `ai-groq` and `ai-grok`.
|
|
15
|
+
*
|
|
16
|
+
* Three Ark behaviours are handled on top of the shared base:
|
|
17
|
+
*
|
|
18
|
+
* 1. **`reasoning_content` deltas** — Ark streams reasoning under
|
|
19
|
+
* `delta.reasoning_content` rather than the OpenAI `reasoning` field.
|
|
20
|
+
* 2. **`encrypted_content` round-trip** — thinking-summary models emit an
|
|
21
|
+
* opaque signature over the reasoning trace. See
|
|
22
|
+
* {@link BytePlusTextAdapter.processStreamChunks} and
|
|
23
|
+
* {@link BytePlusTextAdapter.convertMessage}.
|
|
24
|
+
* 3. **Per-model structured-output gating** — only 10 of the 18 shipped chat
|
|
25
|
+
* models honour `response_format: json_schema` (glm-4-7 accepts it and then
|
|
26
|
+
* ignores the schema), and Ark rejects `json_object` everywhere, so there
|
|
27
|
+
* is no JSON-mode fallback.
|
|
28
|
+
*/
|
|
29
|
+
var BytePlusTextAdapter = class extends OpenAIBaseChatCompletionsTextAdapter {
|
|
30
|
+
kind = "text";
|
|
31
|
+
name = "byteplus";
|
|
32
|
+
constructor(config, model) {
|
|
33
|
+
super(model, "byteplus", new OpenAI(withBytePlusArkDefaults(config)));
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Surfaces Ark's reasoning deltas. Thinking-enabled models stream the
|
|
37
|
+
* reasoning trace as `delta.reasoning_content` (the OpenAI chunk shape has
|
|
38
|
+
* no reasoning field); the base routes this hook through both `chatStream`
|
|
39
|
+
* and `structuredOutputStream`.
|
|
40
|
+
*/
|
|
41
|
+
extractReasoning(chunk) {
|
|
42
|
+
const raw = (chunk.choices[0]?.delta)?.reasoning_content;
|
|
43
|
+
if (typeof raw === "string" && raw.length > 0) return { text: raw };
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Captures Ark's `encrypted_content` and attaches it to the reasoning
|
|
47
|
+
* step's `STEP_FINISHED` event as its `signature`.
|
|
48
|
+
*
|
|
49
|
+
* On a thinking-summary model Ark streams the whole blob as one dedicated
|
|
50
|
+
* chunk (empty `content` and `reasoning_content`) sitting between the
|
|
51
|
+
* reasoning deltas and the content deltas — so it is always captured before
|
|
52
|
+
* the base closes the reasoning lifecycle at the first content delta.
|
|
53
|
+
*
|
|
54
|
+
* `signature` is the framework's existing provider-signature seam: the chat
|
|
55
|
+
* engine stores it on the `ThinkingPart`, which
|
|
56
|
+
* `buildAssistantMessages` carries into `ModelMessage.thinking[].signature`,
|
|
57
|
+
* which {@link BytePlusTextAdapter.convertMessage} echoes back to Ark on the
|
|
58
|
+
* next turn. No base-class change is needed — this is the same round-trip
|
|
59
|
+
* Anthropic's thinking signatures use.
|
|
60
|
+
*
|
|
61
|
+
* Only `chatStream` is covered: `structuredOutputStream` drives the SDK
|
|
62
|
+
* directly in the base with no per-chunk seam, so a structured-output turn
|
|
63
|
+
* does not capture the blob. Ark accepts a following turn without it, so the
|
|
64
|
+
* consequence is a lost reasoning-cache hit, not a failed request.
|
|
65
|
+
*/
|
|
66
|
+
async *processStreamChunks(stream, options, aguiState) {
|
|
67
|
+
const captured = {};
|
|
68
|
+
for await (const event of super.processStreamChunks(captureEncryptedContent(stream, captured), options, aguiState)) {
|
|
69
|
+
if (event.type === EventType.STEP_FINISHED && captured.encryptedContent !== void 0 && event.signature === void 0) {
|
|
70
|
+
yield {
|
|
71
|
+
...event,
|
|
72
|
+
signature: captured.encryptedContent,
|
|
73
|
+
delta: event.delta ?? event.content ?? ""
|
|
74
|
+
};
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
yield event;
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Echoes a captured `encrypted_content` blob back on outgoing assistant
|
|
82
|
+
* messages so multi-turn conversations replay it verbatim, as Ark's
|
|
83
|
+
* thinking-summary docs require.
|
|
84
|
+
*
|
|
85
|
+
* The gate is `emitsEncryptedContent(this.model)` — the model being called
|
|
86
|
+
* now, not the provenance of the history. That guarantees a signature is
|
|
87
|
+
* never sent to a model that has no `encrypted_content` concept. It does
|
|
88
|
+
* NOT identify who produced the signature: `ModelMessage` carries no
|
|
89
|
+
* provider field, so a foreign signature (e.g. an Anthropic thinking
|
|
90
|
+
* signature in replayed cross-provider history) WILL be forwarded when the
|
|
91
|
+
* current model is a thinking-summary model. No shape guard is attempted —
|
|
92
|
+
* the blob is opaque and Ark is the only party that can validate it.
|
|
93
|
+
*
|
|
94
|
+
* Absence is never an error: a live probe confirmed Ark accepts a turn whose
|
|
95
|
+
* assistant message omits `encrypted_content`.
|
|
96
|
+
*/
|
|
97
|
+
convertMessage(message) {
|
|
98
|
+
const converted = super.convertMessage(message);
|
|
99
|
+
if (converted.role !== "assistant" || !emitsEncryptedContent(this.model)) return converted;
|
|
100
|
+
const encryptedContent = lastThinkingSignature(message);
|
|
101
|
+
if (encryptedContent === void 0) return converted;
|
|
102
|
+
return {
|
|
103
|
+
...converted,
|
|
104
|
+
encrypted_content: encryptedContent
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Adds the Ark-only content parts on top of the base's text/image handling:
|
|
109
|
+
* `video_url`, URL-addressed `input_audio`, and the extra `image_url`
|
|
110
|
+
* fields (`detail: 'xhigh'`, `image_pixel_limit`).
|
|
111
|
+
*/
|
|
112
|
+
convertContentPart(part) {
|
|
113
|
+
if (part.type === "image") {
|
|
114
|
+
const metadata = part.metadata;
|
|
115
|
+
return asChatContentPart({
|
|
116
|
+
type: "image_url",
|
|
117
|
+
image_url: {
|
|
118
|
+
url: toUrlOrDataUri(part.source),
|
|
119
|
+
detail: metadata?.detail ?? "auto",
|
|
120
|
+
...metadata?.image_pixel_limit && { image_pixel_limit: metadata.image_pixel_limit }
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
}
|
|
124
|
+
if (part.type === "video") {
|
|
125
|
+
const metadata = part.metadata;
|
|
126
|
+
return asChatContentPart({
|
|
127
|
+
type: "video_url",
|
|
128
|
+
video_url: {
|
|
129
|
+
url: toUrlOrDataUri(part.source),
|
|
130
|
+
...metadata?.fps !== void 0 && { fps: metadata.fps }
|
|
131
|
+
}
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
if (part.type === "audio") {
|
|
135
|
+
const metadata = part.metadata;
|
|
136
|
+
if (part.source.type === "url") return asChatContentPart({
|
|
137
|
+
type: "input_audio",
|
|
138
|
+
input_audio: { url: part.source.value }
|
|
139
|
+
});
|
|
140
|
+
const format = metadata?.format ?? audioFormatFromMimeType(part.source);
|
|
141
|
+
if (format === void 0) throw new Error(`Audio content part for ${this.name} has an unrecognised mimeType (${part.source.mimeType || "none"}). Set the container format explicitly via the part's metadata.format, or supply a URL source.`);
|
|
142
|
+
return asChatContentPart({
|
|
143
|
+
type: "input_audio",
|
|
144
|
+
input_audio: {
|
|
145
|
+
data: stripDataUriPrefix(part.source.value),
|
|
146
|
+
format
|
|
147
|
+
}
|
|
148
|
+
});
|
|
149
|
+
}
|
|
150
|
+
return super.convertContentPart(part);
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Only the models in {@link BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS} accept
|
|
154
|
+
* `response_format: json_schema`; the rest reject it with a 400.
|
|
155
|
+
*
|
|
156
|
+
* Returning `false` for a rejecting model does not make structured output
|
|
157
|
+
* work — Ark has no `json_object` fallback to downgrade to. What it buys is
|
|
158
|
+
* keeping `response_format` out of the request the engine would otherwise
|
|
159
|
+
* build: with the hook false the engine takes its separate finalization
|
|
160
|
+
* path, and the guard in {@link BytePlusTextAdapter.structuredOutput} /
|
|
161
|
+
* {@link BytePlusTextAdapter.structuredOutputStream} stops that *before*
|
|
162
|
+
* any HTTP call. So a `chat({ outputSchema })` on a rejecting model fails
|
|
163
|
+
* loudly, named, without a schema Ark would 400 on ever leaving the
|
|
164
|
+
* process — rather than 400-ing on every turn, or (worse) parsing prose as
|
|
165
|
+
* if it were JSON.
|
|
166
|
+
*
|
|
167
|
+
* Tools without a schema are unaffected: `tools` alone never involves
|
|
168
|
+
* `response_format`.
|
|
169
|
+
*/
|
|
170
|
+
supportsCombinedToolsAndSchema() {
|
|
171
|
+
return supportsStructuredOutput(this.model);
|
|
172
|
+
}
|
|
173
|
+
async structuredOutput(options) {
|
|
174
|
+
const unsupported = this.structuredOutputUnsupportedMessage();
|
|
175
|
+
if (unsupported) {
|
|
176
|
+
options.chatOptions.logger.errors(`${this.name}.structuredOutput unsupported model`, {
|
|
177
|
+
error: { message: unsupported },
|
|
178
|
+
source: `${this.name}.structuredOutput`
|
|
179
|
+
});
|
|
180
|
+
throw new Error(unsupported);
|
|
181
|
+
}
|
|
182
|
+
return await super.structuredOutput(options);
|
|
183
|
+
}
|
|
184
|
+
async *structuredOutputStream(options) {
|
|
185
|
+
const unsupported = this.structuredOutputUnsupportedMessage();
|
|
186
|
+
if (unsupported) {
|
|
187
|
+
const timestamp = Date.now();
|
|
188
|
+
const runId = generateId(this.name);
|
|
189
|
+
yield {
|
|
190
|
+
type: EventType.RUN_STARTED,
|
|
191
|
+
runId,
|
|
192
|
+
threadId: options.chatOptions.threadId ?? generateId(this.name),
|
|
193
|
+
model: options.chatOptions.model,
|
|
194
|
+
timestamp,
|
|
195
|
+
parentRunId: options.chatOptions.parentRunId
|
|
196
|
+
};
|
|
197
|
+
yield {
|
|
198
|
+
type: EventType.RUN_ERROR,
|
|
199
|
+
runId,
|
|
200
|
+
model: options.chatOptions.model,
|
|
201
|
+
timestamp,
|
|
202
|
+
message: unsupported,
|
|
203
|
+
code: "unsupported-structured-output",
|
|
204
|
+
error: {
|
|
205
|
+
message: unsupported,
|
|
206
|
+
code: "unsupported-structured-output"
|
|
207
|
+
}
|
|
208
|
+
};
|
|
209
|
+
options.chatOptions.logger.errors(`${this.name}.structuredOutputStream unsupported model`, {
|
|
210
|
+
error: { message: unsupported },
|
|
211
|
+
source: `${this.name}.structuredOutputStream`
|
|
212
|
+
});
|
|
213
|
+
return;
|
|
214
|
+
}
|
|
215
|
+
yield* super.structuredOutputStream(options);
|
|
216
|
+
}
|
|
217
|
+
/**
|
|
218
|
+
* Explains why structured output is unavailable, or `undefined` when the
|
|
219
|
+
* model supports it. Ark rejects `response_format: json_object` on every
|
|
220
|
+
* model, so there is no JSON-mode fallback to degrade to — failing loud
|
|
221
|
+
* here beats a raw upstream 400.
|
|
222
|
+
*/
|
|
223
|
+
structuredOutputUnsupportedMessage() {
|
|
224
|
+
if (supportsStructuredOutput(this.model)) return void 0;
|
|
225
|
+
return `BytePlus model ${this.model} does not support structured output — Ark rejects both response_format json_schema and json_object on it. Use one of: ${BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS.join(", ")}.`;
|
|
226
|
+
}
|
|
227
|
+
};
|
|
228
|
+
/**
|
|
229
|
+
* Passes Ark's chunks through untouched while recording the single
|
|
230
|
+
* `encrypted_content` blob a thinking-summary model emits.
|
|
231
|
+
*/
|
|
232
|
+
async function* captureEncryptedContent(stream, captured) {
|
|
233
|
+
for await (const chunk of stream) {
|
|
234
|
+
const blob = (chunk.choices[0]?.delta)?.encrypted_content;
|
|
235
|
+
if (typeof blob === "string" && blob.length > 0) captured.encryptedContent = blob;
|
|
236
|
+
yield chunk;
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* The blob to echo back for an assistant message: the last thinking step that
|
|
241
|
+
* carries a signature.
|
|
242
|
+
*/
|
|
243
|
+
function lastThinkingSignature(message) {
|
|
244
|
+
const thinking = message.thinking;
|
|
245
|
+
if (!thinking) return void 0;
|
|
246
|
+
for (let i = thinking.length - 1; i >= 0; i--) {
|
|
247
|
+
const signature = thinking[i]?.signature;
|
|
248
|
+
if (signature) return signature;
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
/**
|
|
252
|
+
* The one place the Ark content-part dialect meets the OpenAI SDK's request
|
|
253
|
+
* types.
|
|
254
|
+
*
|
|
255
|
+
* Ark's union is a superset of OpenAI's: `video_url` has no OpenAI arm at all,
|
|
256
|
+
* `input_audio` additionally accepts a `url`, and `image_url` carries
|
|
257
|
+
* `detail: 'xhigh'` and `image_pixel_limit`. `ChatCompletionContentPart` is a
|
|
258
|
+
* closed type alias in the SDK, so no interface augmentation can admit those
|
|
259
|
+
* arms and no narrowing can produce them — widening to `object` keeps this to
|
|
260
|
+
* a single downcast rather than spreading one through each branch of
|
|
261
|
+
* {@link BytePlusTextAdapter.convertContentPart}.
|
|
262
|
+
*/
|
|
263
|
+
function asChatContentPart(part) {
|
|
264
|
+
return part;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Renders a content source as the URL string Ark expects: URLs pass through,
|
|
268
|
+
* inline base64 becomes a `data:` URI.
|
|
269
|
+
*/
|
|
270
|
+
function toUrlOrDataUri(source) {
|
|
271
|
+
if (source.type !== "data" || source.value.startsWith("data:")) return source.value;
|
|
272
|
+
return `data:${source.mimeType || "application/octet-stream"};base64,${source.value}`;
|
|
273
|
+
}
|
|
274
|
+
/**
|
|
275
|
+
* Strips a `data:` prefix so inline audio is sent as bare base64.
|
|
276
|
+
*/
|
|
277
|
+
function stripDataUriPrefix(value) {
|
|
278
|
+
const comma = value.startsWith("data:") ? value.indexOf(",") : -1;
|
|
279
|
+
return comma === -1 ? value : value.slice(comma + 1);
|
|
280
|
+
}
|
|
281
|
+
var AUDIO_FORMAT_BY_MIME_SUBTYPE = {
|
|
282
|
+
mpeg: "mp3",
|
|
283
|
+
mp3: "mp3",
|
|
284
|
+
wav: "wav",
|
|
285
|
+
"x-wav": "wav",
|
|
286
|
+
wave: "wav",
|
|
287
|
+
ogg: "ogg",
|
|
288
|
+
flac: "flac",
|
|
289
|
+
"x-flac": "flac",
|
|
290
|
+
mp4: "m4a",
|
|
291
|
+
m4a: "m4a",
|
|
292
|
+
"x-m4a": "m4a",
|
|
293
|
+
aac: "aac",
|
|
294
|
+
pcm: "pcm",
|
|
295
|
+
l16: "pcm"
|
|
296
|
+
};
|
|
297
|
+
/**
|
|
298
|
+
* Maps an audio part's mimeType to Ark's container format token.
|
|
299
|
+
*/
|
|
300
|
+
function audioFormatFromMimeType(source) {
|
|
301
|
+
const mimeType = source.mimeType;
|
|
302
|
+
if (!mimeType) return void 0;
|
|
303
|
+
const subtype = mimeType.split(";")[0]?.split("/")[1]?.toLowerCase();
|
|
304
|
+
return subtype ? AUDIO_FORMAT_BY_MIME_SUBTYPE[subtype] : void 0;
|
|
305
|
+
}
|
|
306
|
+
/**
|
|
307
|
+
* Creates a BytePlus text adapter with an explicit API key.
|
|
308
|
+
*
|
|
309
|
+
* @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)
|
|
310
|
+
* @param apiKey - Your BytePlus Ark API key
|
|
311
|
+
* @param config - Optional additional configuration
|
|
312
|
+
*
|
|
313
|
+
* @example
|
|
314
|
+
* ```typescript
|
|
315
|
+
* const adapter = createBytePlusText('seed-2-0-lite-260428', 'ark-...')
|
|
316
|
+
* ```
|
|
317
|
+
*/
|
|
318
|
+
function createBytePlusText(model, apiKey, config) {
|
|
319
|
+
return new BytePlusTextAdapter({
|
|
320
|
+
apiKey,
|
|
321
|
+
...config
|
|
322
|
+
}, model);
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* Creates a BytePlus text adapter with the API key read from `ARK_API_KEY`.
|
|
326
|
+
*
|
|
327
|
+
* @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)
|
|
328
|
+
* @param config - Optional configuration (excluding `apiKey`)
|
|
329
|
+
* @throws Error if `ARK_API_KEY` is not set
|
|
330
|
+
*
|
|
331
|
+
* @example
|
|
332
|
+
* ```typescript
|
|
333
|
+
* const adapter = byteplusText('seed-2-0-lite-260428')
|
|
334
|
+
*
|
|
335
|
+
* const stream = chat({
|
|
336
|
+
* adapter,
|
|
337
|
+
* messages: [{ role: 'user', content: 'Hello!' }],
|
|
338
|
+
* })
|
|
339
|
+
* ```
|
|
340
|
+
*/
|
|
341
|
+
function byteplusText(model, config) {
|
|
342
|
+
return createBytePlusText(model, getBytePlusArkApiKeyFromEnv(), config);
|
|
343
|
+
}
|
|
344
|
+
//#endregion
|
|
345
|
+
export { BytePlusTextAdapter, byteplusText, createBytePlusText };
|
|
346
|
+
|
|
347
|
+
//# sourceMappingURL=text.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"text.js","names":[],"sources":["../../../src/adapters/text.ts"],"sourcesContent":["import OpenAI from 'openai'\nimport { EventType } from '@tanstack/ai'\nimport { OpenAIBaseChatCompletionsTextAdapter } from '@tanstack/openai-base'\nimport { generateId } from '@tanstack/ai-utils'\nimport {\n BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS,\n emitsEncryptedContent,\n supportsStructuredOutput,\n} from '../model-meta'\nimport {\n getBytePlusArkApiKeyFromEnv,\n withBytePlusArkDefaults,\n} from '../utils/client'\nimport type {\n StructuredOutputOptions,\n StructuredOutputResult,\n} from '@tanstack/ai/adapters'\nimport type {\n ContentPart,\n ContentPartSource,\n Modality,\n ModelMessage,\n StreamChunk,\n TextOptions,\n} from '@tanstack/ai'\nimport type {\n ChatCompletionContentPart,\n ChatCompletionMessageParam,\n} from 'openai/resources/chat/completions/completions'\nimport type {\n BYTEPLUS_CHAT_MODELS,\n BytePlusChatModelToolCapabilitiesByName,\n ResolveInputModalities,\n ResolveProviderOptions,\n} from '../model-meta'\nimport type {\n BytePlusAudioMetadata,\n BytePlusChatContentPart,\n BytePlusEncryptedContentFields,\n BytePlusImageMetadata,\n BytePlusInputAudioContentPart,\n BytePlusMessageMetadataByModality,\n BytePlusStreamDeltaExtras,\n BytePlusVideoMetadata,\n} from '../message-types'\nimport type { BytePlusArkConfig } from '../utils/client'\n\ntype ResolveToolCapabilities<TModel extends string> =\n TModel extends keyof BytePlusChatModelToolCapabilitiesByName\n ? NonNullable<BytePlusChatModelToolCapabilitiesByName[TModel]>\n : readonly []\n\n/**\n * Configuration for the BytePlus text adapter.\n */\nexport interface BytePlusTextConfig extends BytePlusArkConfig {}\n\n/**\n * Re-export of the public provider options type.\n */\nexport type { BytePlusTextProviderOptions } from '../text/text-provider-options'\n\n/**\n * BytePlus ModelArk Text (Chat) Adapter\n *\n * Tree-shakeable adapter for the Seed / GLM / DeepSeek / gpt-oss chat models\n * on BytePlus ModelArk. Ark serves an OpenAI-compatible Chat Completions\n * endpoint, so this drives the OpenAI SDK against Ark's `baseURL` — the same\n * pattern as `ai-groq` and `ai-grok`.\n *\n * Three Ark behaviours are handled on top of the shared base:\n *\n * 1. **`reasoning_content` deltas** — Ark streams reasoning under\n * `delta.reasoning_content` rather than the OpenAI `reasoning` field.\n * 2. **`encrypted_content` round-trip** — thinking-summary models emit an\n * opaque signature over the reasoning trace. See\n * {@link BytePlusTextAdapter.processStreamChunks} and\n * {@link BytePlusTextAdapter.convertMessage}.\n * 3. **Per-model structured-output gating** — only 10 of the 18 shipped chat\n * models honour `response_format: json_schema` (glm-4-7 accepts it and then\n * ignores the schema), and Ark rejects `json_object` everywhere, so there\n * is no JSON-mode fallback.\n */\nexport class BytePlusTextAdapter<\n TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number],\n // `Record<string, any>` (not `unknown`) mirrors the OpenAI/Groq/Grok text\n // adapters: the resolved provider options are an interface with no index\n // signature, assignable to `Record<string, any>` but not to\n // `Record<string, unknown>`. See issue #821.\n TProviderOptions extends Record<string, any> = ResolveProviderOptions<TModel>,\n TInputModalities extends ReadonlyArray<Modality> =\n ResolveInputModalities<TModel>,\n TToolCapabilities extends ReadonlyArray<string> =\n ResolveToolCapabilities<TModel>,\n> extends OpenAIBaseChatCompletionsTextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n BytePlusMessageMetadataByModality,\n TToolCapabilities\n> {\n override readonly kind = 'text' as const\n override readonly name = 'byteplus' as const\n\n constructor(config: BytePlusTextConfig, model: TModel) {\n super(model, 'byteplus', new OpenAI(withBytePlusArkDefaults(config)))\n }\n\n /**\n * Surfaces Ark's reasoning deltas. Thinking-enabled models stream the\n * reasoning trace as `delta.reasoning_content` (the OpenAI chunk shape has\n * no reasoning field); the base routes this hook through both `chatStream`\n * and `structuredOutputStream`.\n */\n protected override extractReasoning(\n chunk: OpenAI.Chat.Completions.ChatCompletionChunk,\n ): { text: string } | undefined {\n const delta = chunk.choices[0]?.delta as\n | BytePlusStreamDeltaExtras\n | undefined\n const raw = delta?.reasoning_content\n if (typeof raw === 'string' && raw.length > 0) {\n return { text: raw }\n }\n return undefined\n }\n\n /**\n * Captures Ark's `encrypted_content` and attaches it to the reasoning\n * step's `STEP_FINISHED` event as its `signature`.\n *\n * On a thinking-summary model Ark streams the whole blob as one dedicated\n * chunk (empty `content` and `reasoning_content`) sitting between the\n * reasoning deltas and the content deltas — so it is always captured before\n * the base closes the reasoning lifecycle at the first content delta.\n *\n * `signature` is the framework's existing provider-signature seam: the chat\n * engine stores it on the `ThinkingPart`, which\n * `buildAssistantMessages` carries into `ModelMessage.thinking[].signature`,\n * which {@link BytePlusTextAdapter.convertMessage} echoes back to Ark on the\n * next turn. No base-class change is needed — this is the same round-trip\n * Anthropic's thinking signatures use.\n *\n * Only `chatStream` is covered: `structuredOutputStream` drives the SDK\n * directly in the base with no per-chunk seam, so a structured-output turn\n * does not capture the blob. Ark accepts a following turn without it, so the\n * consequence is a lost reasoning-cache hit, not a failed request.\n */\n protected override async *processStreamChunks(\n stream: AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk>,\n options: TextOptions,\n aguiState: {\n runId: string\n threadId: string\n messageId: string\n hasEmittedRunStarted: boolean\n },\n ): AsyncIterable<StreamChunk> {\n const captured: { encryptedContent?: string } = {}\n\n for await (const event of super.processStreamChunks(\n captureEncryptedContent(stream, captured),\n options,\n aguiState,\n )) {\n if (\n event.type === EventType.STEP_FINISHED &&\n captured.encryptedContent !== undefined &&\n event.signature === undefined\n ) {\n // `delta` is stamped alongside the signature because the two consumers\n // read this event differently. `chat()`'s server agent loop accumulates\n // thinking ONLY from `STEP_FINISHED.delta` and then drops the whole\n // step — signature included — when the accumulated content is empty\n // (`finalizeCurrentThinkingStep`); the OpenAI base emits `content` but\n // never `delta`, so without this the blob never reaches the\n // continuation message. The client `StreamProcessor` can't double-count\n // it: it short-circuits STEP_FINISHED content once\n // `hasSeenReasoningEvents` is set, which the REASONING_MESSAGE_CONTENT\n // events preceding every STEP_FINISHED here always set.\n yield {\n ...event,\n signature: captured.encryptedContent,\n delta: event.delta ?? event.content ?? '',\n }\n continue\n }\n yield event\n }\n }\n\n /**\n * Echoes a captured `encrypted_content` blob back on outgoing assistant\n * messages so multi-turn conversations replay it verbatim, as Ark's\n * thinking-summary docs require.\n *\n * The gate is `emitsEncryptedContent(this.model)` — the model being called\n * now, not the provenance of the history. That guarantees a signature is\n * never sent to a model that has no `encrypted_content` concept. It does\n * NOT identify who produced the signature: `ModelMessage` carries no\n * provider field, so a foreign signature (e.g. an Anthropic thinking\n * signature in replayed cross-provider history) WILL be forwarded when the\n * current model is a thinking-summary model. No shape guard is attempted —\n * the blob is opaque and Ark is the only party that can validate it.\n *\n * Absence is never an error: a live probe confirmed Ark accepts a turn whose\n * assistant message omits `encrypted_content`.\n */\n protected override convertMessage(\n message: ModelMessage,\n ): ChatCompletionMessageParam {\n const converted = super.convertMessage(message)\n if (converted.role !== 'assistant' || !emitsEncryptedContent(this.model)) {\n return converted\n }\n\n const encryptedContent = lastThinkingSignature(message)\n if (encryptedContent === undefined) return converted\n\n // Intersection rather than a cast: `encrypted_content` is an Ark-only\n // field with no slot on the OpenAI message param, and the intersection is\n // still assignable to `ChatCompletionMessageParam`.\n const withEncrypted: typeof converted & BytePlusEncryptedContentFields = {\n ...converted,\n encrypted_content: encryptedContent,\n }\n return withEncrypted\n }\n\n /**\n * Adds the Ark-only content parts on top of the base's text/image handling:\n * `video_url`, URL-addressed `input_audio`, and the extra `image_url`\n * fields (`detail: 'xhigh'`, `image_pixel_limit`).\n */\n protected override convertContentPart(\n part: ContentPart,\n ): ChatCompletionContentPart | null {\n if (part.type === 'image') {\n const metadata = part.metadata as BytePlusImageMetadata | undefined\n return asChatContentPart({\n type: 'image_url',\n image_url: {\n url: toUrlOrDataUri(part.source),\n detail: metadata?.detail ?? 'auto',\n ...(metadata?.image_pixel_limit && {\n image_pixel_limit: metadata.image_pixel_limit,\n }),\n },\n })\n }\n\n if (part.type === 'video') {\n const metadata = part.metadata as BytePlusVideoMetadata | undefined\n return asChatContentPart({\n type: 'video_url',\n video_url: {\n url: toUrlOrDataUri(part.source),\n ...(metadata?.fps !== undefined && { fps: metadata.fps }),\n },\n })\n }\n\n if (part.type === 'audio') {\n const metadata = part.metadata as BytePlusAudioMetadata | undefined\n // Ark takes audio either by URL or as inline base64 with an explicit\n // container format; unlike images there is no data-URI form.\n if (part.source.type === 'url') {\n return asChatContentPart({\n type: 'input_audio',\n input_audio: { url: part.source.value },\n })\n }\n const format = metadata?.format ?? audioFormatFromMimeType(part.source)\n if (format === undefined) {\n throw new Error(\n `Audio content part for ${this.name} has an unrecognised mimeType ` +\n `(${part.source.mimeType || 'none'}). Set the container format ` +\n `explicitly via the part's metadata.format, or supply a URL source.`,\n )\n }\n return asChatContentPart({\n type: 'input_audio',\n input_audio: { data: stripDataUriPrefix(part.source.value), format },\n })\n }\n\n return super.convertContentPart(part)\n }\n\n /**\n * Only the models in {@link BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS} accept\n * `response_format: json_schema`; the rest reject it with a 400.\n *\n * Returning `false` for a rejecting model does not make structured output\n * work — Ark has no `json_object` fallback to downgrade to. What it buys is\n * keeping `response_format` out of the request the engine would otherwise\n * build: with the hook false the engine takes its separate finalization\n * path, and the guard in {@link BytePlusTextAdapter.structuredOutput} /\n * {@link BytePlusTextAdapter.structuredOutputStream} stops that *before*\n * any HTTP call. So a `chat({ outputSchema })` on a rejecting model fails\n * loudly, named, without a schema Ark would 400 on ever leaving the\n * process — rather than 400-ing on every turn, or (worse) parsing prose as\n * if it were JSON.\n *\n * Tools without a schema are unaffected: `tools` alone never involves\n * `response_format`.\n */\n override supportsCombinedToolsAndSchema(): boolean {\n return supportsStructuredOutput(this.model)\n }\n\n override async structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>> {\n const unsupported = this.structuredOutputUnsupportedMessage()\n if (unsupported) {\n options.chatOptions.logger.errors(\n `${this.name}.structuredOutput unsupported model`,\n {\n error: { message: unsupported },\n source: `${this.name}.structuredOutput`,\n },\n )\n throw new Error(unsupported)\n }\n return await super.structuredOutput(options)\n }\n\n override async *structuredOutputStream(\n options: StructuredOutputOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk> {\n const unsupported = this.structuredOutputUnsupportedMessage()\n if (unsupported) {\n // Mirror the base's contract: failures inside structuredOutputStream\n // surface as a RUN_STARTED → RUN_ERROR pair rather than a throw, so\n // consumers keep a single error-handling path.\n const timestamp = Date.now()\n const runId = generateId(this.name)\n yield {\n type: EventType.RUN_STARTED,\n runId,\n threadId: options.chatOptions.threadId ?? generateId(this.name),\n model: options.chatOptions.model,\n timestamp,\n parentRunId: options.chatOptions.parentRunId,\n }\n yield {\n type: EventType.RUN_ERROR,\n runId,\n model: options.chatOptions.model,\n timestamp,\n message: unsupported,\n code: 'unsupported-structured-output',\n error: { message: unsupported, code: 'unsupported-structured-output' },\n }\n options.chatOptions.logger.errors(\n `${this.name}.structuredOutputStream unsupported model`,\n {\n error: { message: unsupported },\n source: `${this.name}.structuredOutputStream`,\n },\n )\n return\n }\n yield* super.structuredOutputStream(options)\n }\n\n /**\n * Explains why structured output is unavailable, or `undefined` when the\n * model supports it. Ark rejects `response_format: json_object` on every\n * model, so there is no JSON-mode fallback to degrade to — failing loud\n * here beats a raw upstream 400.\n */\n private structuredOutputUnsupportedMessage(): string | undefined {\n if (supportsStructuredOutput(this.model)) return undefined\n return (\n `BytePlus model ${this.model} does not support structured output — Ark ` +\n `rejects both response_format json_schema and json_object on it. Use ` +\n `one of: ${BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS.join(', ')}.`\n )\n }\n}\n\n/**\n * Passes Ark's chunks through untouched while recording the single\n * `encrypted_content` blob a thinking-summary model emits.\n */\nasync function* captureEncryptedContent(\n stream: AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk>,\n captured: { encryptedContent?: string },\n): AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk> {\n for await (const chunk of stream) {\n const delta = chunk.choices[0]?.delta as\n | BytePlusStreamDeltaExtras\n | undefined\n const blob = delta?.encrypted_content\n if (typeof blob === 'string' && blob.length > 0) {\n captured.encryptedContent = blob\n }\n yield chunk\n }\n}\n\n/**\n * The blob to echo back for an assistant message: the last thinking step that\n * carries a signature.\n */\nfunction lastThinkingSignature(message: ModelMessage): string | undefined {\n const thinking = message.thinking\n if (!thinking) return undefined\n for (let i = thinking.length - 1; i >= 0; i--) {\n const signature = thinking[i]?.signature\n if (signature) return signature\n }\n return undefined\n}\n\n/**\n * The one place the Ark content-part dialect meets the OpenAI SDK's request\n * types.\n *\n * Ark's union is a superset of OpenAI's: `video_url` has no OpenAI arm at all,\n * `input_audio` additionally accepts a `url`, and `image_url` carries\n * `detail: 'xhigh'` and `image_pixel_limit`. `ChatCompletionContentPart` is a\n * closed type alias in the SDK, so no interface augmentation can admit those\n * arms and no narrowing can produce them — widening to `object` keeps this to\n * a single downcast rather than spreading one through each branch of\n * {@link BytePlusTextAdapter.convertContentPart}.\n */\nfunction asChatContentPart(\n part: BytePlusChatContentPart,\n): ChatCompletionContentPart {\n const arkPart: object = part\n return arkPart as ChatCompletionContentPart\n}\n\n/**\n * Renders a content source as the URL string Ark expects: URLs pass through,\n * inline base64 becomes a `data:` URI.\n */\nfunction toUrlOrDataUri(source: ContentPartSource): string {\n if (source.type !== 'data' || source.value.startsWith('data:')) {\n return source.value\n }\n // A missing mimeType would interpolate as \"data:undefined;base64,…\" and be\n // rejected, so fall back the same way the OpenAI base does.\n return `data:${source.mimeType || 'application/octet-stream'};base64,${source.value}`\n}\n\n/**\n * Strips a `data:` prefix so inline audio is sent as bare base64.\n */\nfunction stripDataUriPrefix(value: string): string {\n const comma = value.startsWith('data:') ? value.indexOf(',') : -1\n return comma === -1 ? value : value.slice(comma + 1)\n}\n\nconst AUDIO_FORMAT_BY_MIME_SUBTYPE: Record<\n string,\n NonNullable<BytePlusInputAudioContentPart['input_audio']['format']>\n> = {\n mpeg: 'mp3',\n mp3: 'mp3',\n wav: 'wav',\n 'x-wav': 'wav',\n wave: 'wav',\n ogg: 'ogg',\n flac: 'flac',\n 'x-flac': 'flac',\n mp4: 'm4a',\n m4a: 'm4a',\n 'x-m4a': 'm4a',\n aac: 'aac',\n pcm: 'pcm',\n l16: 'pcm',\n}\n\n/**\n * Maps an audio part's mimeType to Ark's container format token.\n */\nfunction audioFormatFromMimeType(\n source: ContentPartSource,\n):\n | NonNullable<BytePlusInputAudioContentPart['input_audio']['format']>\n | undefined {\n const mimeType = source.mimeType\n if (!mimeType) return undefined\n const subtype = mimeType.split(';')[0]?.split('/')[1]?.toLowerCase()\n return subtype ? AUDIO_FORMAT_BY_MIME_SUBTYPE[subtype] : undefined\n}\n\n/**\n * Creates a BytePlus text adapter with an explicit API key.\n *\n * @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)\n * @param apiKey - Your BytePlus Ark API key\n * @param config - Optional additional configuration\n *\n * @example\n * ```typescript\n * const adapter = createBytePlusText('seed-2-0-lite-260428', 'ark-...')\n * ```\n */\nexport function createBytePlusText<\n TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number],\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusTextConfig, 'apiKey'>,\n): BytePlusTextAdapter<TModel> {\n return new BytePlusTextAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a BytePlus text adapter with the API key read from `ARK_API_KEY`.\n *\n * @param model - The chat model id (e.g., `'seed-2-0-lite-260428'`)\n * @param config - Optional configuration (excluding `apiKey`)\n * @throws Error if `ARK_API_KEY` is not set\n *\n * @example\n * ```typescript\n * const adapter = byteplusText('seed-2-0-lite-260428')\n *\n * const stream = chat({\n * adapter,\n * messages: [{ role: 'user', content: 'Hello!' }],\n * })\n * ```\n */\nexport function byteplusText<\n TModel extends (typeof BYTEPLUS_CHAT_MODELS)[number],\n>(\n model: TModel,\n config?: Omit<BytePlusTextConfig, 'apiKey'>,\n): BytePlusTextAdapter<TModel> {\n const apiKey = getBytePlusArkApiKeyFromEnv()\n return createBytePlusText(model, apiKey, config)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmFA,IAAa,sBAAb,cAWU,qCAMR;CACA,OAAyB;CACzB,OAAyB;CAEzB,YAAY,QAA4B,OAAe;EACrD,MAAM,OAAO,YAAY,IAAI,OAAO,wBAAwB,MAAM,CAAC,CAAC;CACtE;;;;;;;CAQA,iBACE,OAC8B;EAI9B,MAAM,OAHQ,MAAM,QAAQ,EAAE,EAAE,MAAA,EAGb;EACnB,IAAI,OAAO,QAAQ,YAAY,IAAI,SAAS,GAC1C,OAAO,EAAE,MAAM,IAAI;CAGvB;;;;;;;;;;;;;;;;;;;;;;CAuBA,OAA0B,oBACxB,QACA,SACA,WAM4B;EAC5B,MAAM,WAA0C,CAAC;EAEjD,WAAW,MAAM,SAAS,MAAM,oBAC9B,wBAAwB,QAAQ,QAAQ,GACxC,SACA,SACF,GAAG;GACD,IACE,MAAM,SAAS,UAAU,iBACzB,SAAS,qBAAqB,KAAA,KAC9B,MAAM,cAAc,KAAA,GACpB;IAWA,MAAM;KACJ,GAAG;KACH,WAAW,SAAS;KACpB,OAAO,MAAM,SAAS,MAAM,WAAW;IACzC;IACA;GACF;GACA,MAAM;EACR;CACF;;;;;;;;;;;;;;;;;;CAmBA,eACE,SAC4B;EAC5B,MAAM,YAAY,MAAM,eAAe,OAAO;EAC9C,IAAI,UAAU,SAAS,eAAe,CAAC,sBAAsB,KAAK,KAAK,GACrE,OAAO;EAGT,MAAM,mBAAmB,sBAAsB,OAAO;EACtD,IAAI,qBAAqB,KAAA,GAAW,OAAO;EAS3C,OAAO;GAHL,GAAG;GACH,mBAAmB;EAEd;CACT;;;;;;CAOA,mBACE,MACkC;EAClC,IAAI,KAAK,SAAS,SAAS;GACzB,MAAM,WAAW,KAAK;GACtB,OAAO,kBAAkB;IACvB,MAAM;IACN,WAAW;KACT,KAAK,eAAe,KAAK,MAAM;KAC/B,QAAQ,UAAU,UAAU;KAC5B,GAAI,UAAU,qBAAqB,EACjC,mBAAmB,SAAS,kBAC9B;IACF;GACF,CAAC;EACH;EAEA,IAAI,KAAK,SAAS,SAAS;GACzB,MAAM,WAAW,KAAK;GACtB,OAAO,kBAAkB;IACvB,MAAM;IACN,WAAW;KACT,KAAK,eAAe,KAAK,MAAM;KAC/B,GAAI,UAAU,QAAQ,KAAA,KAAa,EAAE,KAAK,SAAS,IAAI;IACzD;GACF,CAAC;EACH;EAEA,IAAI,KAAK,SAAS,SAAS;GACzB,MAAM,WAAW,KAAK;GAGtB,IAAI,KAAK,OAAO,SAAS,OACvB,OAAO,kBAAkB;IACvB,MAAM;IACN,aAAa,EAAE,KAAK,KAAK,OAAO,MAAM;GACxC,CAAC;GAEH,MAAM,SAAS,UAAU,UAAU,wBAAwB,KAAK,MAAM;GACtE,IAAI,WAAW,KAAA,GACb,MAAM,IAAI,MACR,0BAA0B,KAAK,KAAK,iCAC9B,KAAK,OAAO,YAAY,OAAO,+FAEvC;GAEF,OAAO,kBAAkB;IACvB,MAAM;IACN,aAAa;KAAE,MAAM,mBAAmB,KAAK,OAAO,KAAK;KAAG;IAAO;GACrE,CAAC;EACH;EAEA,OAAO,MAAM,mBAAmB,IAAI;CACtC;;;;;;;;;;;;;;;;;;;CAoBA,iCAAmD;EACjD,OAAO,yBAAyB,KAAK,KAAK;CAC5C;CAEA,MAAe,iBACb,SAC0C;EAC1C,MAAM,cAAc,KAAK,mCAAmC;EAC5D,IAAI,aAAa;GACf,QAAQ,YAAY,OAAO,OACzB,GAAG,KAAK,KAAK,sCACb;IACE,OAAO,EAAE,SAAS,YAAY;IAC9B,QAAQ,GAAG,KAAK,KAAK;GACvB,CACF;GACA,MAAM,IAAI,MAAM,WAAW;EAC7B;EACA,OAAO,MAAM,MAAM,iBAAiB,OAAO;CAC7C;CAEA,OAAgB,uBACd,SAC4B;EAC5B,MAAM,cAAc,KAAK,mCAAmC;EAC5D,IAAI,aAAa;GAIf,MAAM,YAAY,KAAK,IAAI;GAC3B,MAAM,QAAQ,WAAW,KAAK,IAAI;GAClC,MAAM;IACJ,MAAM,UAAU;IAChB;IACA,UAAU,QAAQ,YAAY,YAAY,WAAW,KAAK,IAAI;IAC9D,OAAO,QAAQ,YAAY;IAC3B;IACA,aAAa,QAAQ,YAAY;GACnC;GACA,MAAM;IACJ,MAAM,UAAU;IAChB;IACA,OAAO,QAAQ,YAAY;IAC3B;IACA,SAAS;IACT,MAAM;IACN,OAAO;KAAE,SAAS;KAAa,MAAM;IAAgC;GACvE;GACA,QAAQ,YAAY,OAAO,OACzB,GAAG,KAAK,KAAK,4CACb;IACE,OAAO,EAAE,SAAS,YAAY;IAC9B,QAAQ,GAAG,KAAK,KAAK;GACvB,CACF;GACA;EACF;EACA,OAAO,MAAM,uBAAuB,OAAO;CAC7C;;;;;;;CAQA,qCAAiE;EAC/D,IAAI,yBAAyB,KAAK,KAAK,GAAG,OAAO,KAAA;EACjD,OACE,kBAAkB,KAAK,MAAM,wHAElB,uCAAuC,KAAK,IAAI,EAAE;CAEjE;AACF;;;;;AAMA,gBAAgB,wBACd,QACA,UAC4D;CAC5D,WAAW,MAAM,SAAS,QAAQ;EAIhC,MAAM,QAHQ,MAAM,QAAQ,EAAE,EAAE,MAAA,EAGZ;EACpB,IAAI,OAAO,SAAS,YAAY,KAAK,SAAS,GAC5C,SAAS,mBAAmB;EAE9B,MAAM;CACR;AACF;;;;;AAMA,SAAS,sBAAsB,SAA2C;CACxE,MAAM,WAAW,QAAQ;CACzB,IAAI,CAAC,UAAU,OAAO,KAAA;CACtB,KAAK,IAAI,IAAI,SAAS,SAAS,GAAG,KAAK,GAAG,KAAK;EAC7C,MAAM,YAAY,SAAS,EAAE,EAAE;EAC/B,IAAI,WAAW,OAAO;CACxB;AAEF;;;;;;;;;;;;;AAcA,SAAS,kBACP,MAC2B;CAE3B,OAAO;AACT;;;;;AAMA,SAAS,eAAe,QAAmC;CACzD,IAAI,OAAO,SAAS,UAAU,OAAO,MAAM,WAAW,OAAO,GAC3D,OAAO,OAAO;CAIhB,OAAO,QAAQ,OAAO,YAAY,2BAA2B,UAAU,OAAO;AAChF;;;;AAKA,SAAS,mBAAmB,OAAuB;CACjD,MAAM,QAAQ,MAAM,WAAW,OAAO,IAAI,MAAM,QAAQ,GAAG,IAAI;CAC/D,OAAO,UAAU,KAAK,QAAQ,MAAM,MAAM,QAAQ,CAAC;AACrD;AAEA,IAAM,+BAGF;CACF,MAAM;CACN,KAAK;CACL,KAAK;CACL,SAAS;CACT,MAAM;CACN,KAAK;CACL,MAAM;CACN,UAAU;CACV,KAAK;CACL,KAAK;CACL,SAAS;CACT,KAAK;CACL,KAAK;CACL,KAAK;AACP;;;;AAKA,SAAS,wBACP,QAGY;CACZ,MAAM,WAAW,OAAO;CACxB,IAAI,CAAC,UAAU,OAAO,KAAA;CACtB,MAAM,UAAU,SAAS,MAAM,GAAG,CAAC,CAAC,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC,EAAE,EAAE,YAAY;CACnE,OAAO,UAAU,6BAA6B,WAAW,KAAA;AAC3D;;;;;;;;;;;;;AAcA,SAAgB,mBAGd,OACA,QACA,QAC6B;CAC7B,OAAO,IAAI,oBAAoB;EAAE;EAAQ,GAAG;CAAO,GAAG,KAAK;AAC7D;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAGd,OACA,QAC6B;CAE7B,OAAO,mBAAmB,OADX,4BACkB,GAAQ,MAAM;AACjD"}
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters';
|
|
2
|
+
import { TranscriptionOptions, TranscriptionResult, TranscriptionWord } from '@tanstack/ai';
|
|
3
|
+
import { InternalLogger } from '@tanstack/ai/adapter-internals';
|
|
4
|
+
import { BytePlusVoiceConfig } from '../utils/client.js';
|
|
5
|
+
import { BytePlusTranscriptionModel } from '../model-meta.js';
|
|
6
|
+
import { BytePlusASRAudio, BytePlusASRRecognizeRequest, BytePlusASRRecognizeResponse } from '../audio/wire-types.js';
|
|
7
|
+
import { BytePlusTranscriptionProviderOptions } from '../audio/transcription-provider-options.js';
|
|
8
|
+
/**
|
|
9
|
+
* BytePlus-specific extension of `TranscriptionWord` carrying the per-word
|
|
10
|
+
* confidence Seed ASR returns. The cross-provider contract has no field for
|
|
11
|
+
* it, so callers who want it narrow the array — the same pattern the Grok
|
|
12
|
+
* adapter uses:
|
|
13
|
+
*
|
|
14
|
+
* ```ts
|
|
15
|
+
* const words = result.words as Array<BytePlusTranscriptionWord> | undefined
|
|
16
|
+
* ```
|
|
17
|
+
*/
|
|
18
|
+
export interface BytePlusTranscriptionWord extends TranscriptionWord {
|
|
19
|
+
/** Model confidence for the word, when Seed ASR returns one. */
|
|
20
|
+
confidence?: number;
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* BytePlus Seed Speech transcription (ASR) adapter.
|
|
24
|
+
*
|
|
25
|
+
* Talks to `POST {baseURL}/api/v3/auc/bigmodel/recognize/flash` — the
|
|
26
|
+
* synchronous "flash" endpoint, which returns the whole transcript in one
|
|
27
|
+
* response rather than requiring a submit/poll cycle. It accepts audio up to
|
|
28
|
+
* 2 hours long or 100 MB, either as a publicly reachable URL or as base64
|
|
29
|
+
* bytes.
|
|
30
|
+
*
|
|
31
|
+
* Two BytePlus-specific details:
|
|
32
|
+
*
|
|
33
|
+
* - The model is selected by the `X-Api-Resource-Id` header
|
|
34
|
+
* (`volc.seedasr.auc_turbo`), not by a `model` field in the body. The
|
|
35
|
+
* package's `seed-asr` model id exists to satisfy the SDK contract and to
|
|
36
|
+
* give logs a stable value.
|
|
37
|
+
* - Authentication uses `X-Api-Key` with the **Seed Speech** key, which is a
|
|
38
|
+
* different key from `ARK_API_KEY`.
|
|
39
|
+
*
|
|
40
|
+
* All timings on the wire are milliseconds; they are converted to seconds to
|
|
41
|
+
* match the cross-provider `TranscriptionResult`.
|
|
42
|
+
*
|
|
43
|
+
* @example
|
|
44
|
+
* ```ts
|
|
45
|
+
* const adapter = byteplusTranscription('seed-asr')
|
|
46
|
+
* const result = await generateTranscription({
|
|
47
|
+
* adapter,
|
|
48
|
+
* audio: 'https://example.com/interview.mp3',
|
|
49
|
+
* language: 'en-US',
|
|
50
|
+
* })
|
|
51
|
+
* ```
|
|
52
|
+
*/
|
|
53
|
+
export declare class BytePlusTranscriptionAdapter<TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel> extends BaseTranscriptionAdapter<TModel, BytePlusTranscriptionProviderOptions> {
|
|
54
|
+
readonly name: "byteplus";
|
|
55
|
+
private readonly apiKey;
|
|
56
|
+
private readonly baseURL;
|
|
57
|
+
private readonly defaultHeaders;
|
|
58
|
+
private readonly fetchImpl;
|
|
59
|
+
constructor(model: TModel, config: BytePlusVoiceConfig);
|
|
60
|
+
transcribe(options: TranscriptionOptions<BytePlusTranscriptionProviderOptions>): Promise<TranscriptionResult>;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Build the JSON body for `POST /api/v3/auc/bigmodel/recognize/flash`.
|
|
64
|
+
*
|
|
65
|
+
* `show_utterances` defaults to `true` so the response carries the
|
|
66
|
+
* per-utterance breakdown that populates `segments` and `words`.
|
|
67
|
+
*/
|
|
68
|
+
export declare function buildRecognizeRequestBody(options: {
|
|
69
|
+
audio: BytePlusASRAudio;
|
|
70
|
+
language: string | undefined;
|
|
71
|
+
modelOptions: BytePlusTranscriptionProviderOptions | undefined;
|
|
72
|
+
}): BytePlusASRRecognizeRequest;
|
|
73
|
+
/**
|
|
74
|
+
* Turn a recognition response into the transcript-shaped half of a
|
|
75
|
+
* `TranscriptionResult`. Wire timings are milliseconds; everything returned
|
|
76
|
+
* here is seconds.
|
|
77
|
+
*/
|
|
78
|
+
export declare function mapRecognizeResponse(data: BytePlusASRRecognizeResponse, text: string, logger?: InternalLogger): Omit<TranscriptionResult, 'id' | 'model'>;
|
|
79
|
+
/**
|
|
80
|
+
* Turn the cross-provider `audio` input into the endpoint's `audio` block.
|
|
81
|
+
*
|
|
82
|
+
* URLs are passed through untouched — Seed ASR fetches them itself, which
|
|
83
|
+
* avoids pulling large media through this process. Everything else is sent as
|
|
84
|
+
* base64 `data`, with the container inferred from the input's MIME type or
|
|
85
|
+
* filename when the caller didn't pin `audio_format`.
|
|
86
|
+
*/
|
|
87
|
+
export declare function normalizeAudioInput(audio: TranscriptionOptions['audio'], formatHint: string | undefined): Promise<BytePlusASRAudio>;
|
|
88
|
+
/**
|
|
89
|
+
* Creates a BytePlus Seed Speech transcription adapter with an explicit API
|
|
90
|
+
* key.
|
|
91
|
+
*
|
|
92
|
+
* The key is the **Seed Speech** key, not the Ark key used by the chat, image
|
|
93
|
+
* and video adapters.
|
|
94
|
+
*/
|
|
95
|
+
export declare function createBytePlusTranscription<TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel>(model: TModel, apiKey: string, config?: Omit<BytePlusVoiceConfig, 'apiKey'>): BytePlusTranscriptionAdapter<TModel>;
|
|
96
|
+
/**
|
|
97
|
+
* Creates a BytePlus Seed Speech transcription adapter, reading the API key
|
|
98
|
+
* from `BYTEPLUS_VOICE_API_KEY`.
|
|
99
|
+
*
|
|
100
|
+
* @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
|
|
101
|
+
*/
|
|
102
|
+
export declare function byteplusTranscription<TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel>(model: TModel, config?: Omit<BytePlusVoiceConfig, 'apiKey'>): BytePlusTranscriptionAdapter<TModel>;
|