@keo-ai/axiom 0.2.8 → 0.2.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -4
- package/dist/llm_provider/bailian.js +85 -34
- package/dist/llm_provider/models.d.ts +1 -1
- package/dist/llm_provider/models.js +2 -0
- package/dist/llm_provider/types.d.ts +9 -1
- package/dist/predict/predict.d.ts +2 -0
- package/dist/predict/predict.js +12 -4
- package/package.json +1 -1
- package/dist/predict/providers/bailian.d.ts +0 -27
- package/dist/predict/providers/bailian.js +0 -185
package/README.md
CHANGED
|
@@ -177,14 +177,17 @@ for await (const chunk of LLM.streamPredict({ model: 'qwen-max', prompt: '讲个
|
|
|
177
177
|
if (chunk.type === 'reasoning') {
|
|
178
178
|
process.stdout.write(chunk.delta); // 推理过程
|
|
179
179
|
}
|
|
180
|
-
if (chunk.type === 'finish'
|
|
180
|
+
if (chunk.type === 'finish') {
|
|
181
|
+
console.log('Finish:', chunk.finishReason, chunk.sawDone);
|
|
181
182
|
console.log('Token usage:', chunk.usage);
|
|
182
|
-
// { promptTokens, completionTokens, totalTokens, cachedPromptTokens? }
|
|
183
183
|
}
|
|
184
184
|
}
|
|
185
185
|
```
|
|
186
186
|
|
|
187
|
-
> 💡 流式调用自动启用 `stream_options: { include_usage: true }
|
|
187
|
+
> 💡 流式调用自动启用 `stream_options: { include_usage: true }`。`finish` 事件会返回 `finishReason`、
|
|
188
|
+
> `sawDone`、`sawFinishReason` 和 token 消耗统计。usage 可能与 `finish_reason` 在同一个 SSE chunk,也可能在独立的
|
|
189
|
+
> chunk(`choices: []`)中返回,两种情况均已兼容。若连接结束前既没有 `[DONE]` 也没有非空
|
|
190
|
+
> `finish_reason`,流会抛出异常,避免把提前 EOF 当作成功响应。
|
|
188
191
|
|
|
189
192
|
### 模型列表
|
|
190
193
|
|
|
@@ -193,12 +196,14 @@ for await (const chunk of LLM.streamPredict({ model: 'qwen-max', prompt: '讲个
|
|
|
193
196
|
| 模型 | 推理深度 | json 模式 | 说明 |
|
|
194
197
|
|------|----------|-----------|------|
|
|
195
198
|
| `qwen-max` | ❌ 不支持 | ✅ | **默认模型** |
|
|
196
|
-
| `qwen3.
|
|
199
|
+
| `qwen3.8-flash` | ✅ 映射支持 | ✅ | `low`/`medium`/`high` 映射为 `enable_thinking` |
|
|
200
|
+
| `qwen3.7-max` | ✅ 映射支持 | ✅ | 同上 |
|
|
197
201
|
| `qwen3.7-flash` | ✅ 映射支持 | ✅ | 同上 |
|
|
198
202
|
| `qwen-plus` | ✅ 映射支持 | ✅ | 同上 |
|
|
199
203
|
| `qwen-turbo` | ✅ 映射支持 | ✅ | 同上 |
|
|
200
204
|
| `deepseek-v4-pro` | ✅ 映射支持 | ✅ | 同上(百炼走「混合思考模式」,由 `enable_thinking` 控制) |
|
|
201
205
|
| `deepseek-v4-flash` | ✅ 映射支持 | ✅ | 同上 |
|
|
206
|
+
| `deepseek-v4.1-flash` | ✅ 映射支持 | ✅ | 同上(默认开启思考) |
|
|
202
207
|
| `kimi-k2.6` | ✅ 映射支持 | ✅ | `low`/`medium`/`high` 映射为 `extra_body.thinking.type` |
|
|
203
208
|
| `qwq-plus` | ❌ 不支持 | ✅ | 固定推理行为,不可调节 |
|
|
204
209
|
| `glm-5.1` | ❌ 不支持 | ❌ | 不支持推理参数,不支持 json 模式 |
|
|
@@ -224,6 +229,15 @@ try {
|
|
|
224
229
|
}
|
|
225
230
|
```
|
|
226
231
|
|
|
232
|
+
`responseFormat: 'json'` 时输出为空或无法 `JSON.parse`,抛出的 `Error.message` 会带上 `model`、`finish_reason`、
|
|
233
|
+
原文长度和模型原文,调用方直接打日志即可排查:`finish_reason=length` 表示被 `maxTokens` 截断,`stop` 表示模型本身输出了非法
|
|
234
|
+
JSON,`content_filter` 表示被内容审核拦截。
|
|
235
|
+
|
|
236
|
+
```
|
|
237
|
+
JSON parse failed: Unterminated string in JSON at position 1118 (model=qwen3.8-flash, finish_reason=length, raw_length=1118)
|
|
238
|
+
raw: {"classification":"normal","reason":"...
|
|
239
|
+
```
|
|
240
|
+
|
|
227
241
|
---
|
|
228
242
|
|
|
229
243
|
## Function Call Loop 模块
|
|
@@ -60,7 +60,16 @@ class BailianProvider {
|
|
|
60
60
|
// 通过 extra_body.enable_thinking 控制思考开关(low 关闭,medium/high 开启);
|
|
61
61
|
// Kimi(Moonshot)则通过 extra_body.thinking.type 实现(默认开启)。
|
|
62
62
|
if (body.reasoning_effort !== undefined) {
|
|
63
|
-
if ([
|
|
63
|
+
if ([
|
|
64
|
+
'qwen-plus',
|
|
65
|
+
'qwen-turbo',
|
|
66
|
+
'qwen3.8-flash',
|
|
67
|
+
'qwen3.7-max',
|
|
68
|
+
'qwen3.7-flash',
|
|
69
|
+
'deepseek-v4-pro',
|
|
70
|
+
'deepseek-v4-flash',
|
|
71
|
+
'deepseek-v4.1-flash',
|
|
72
|
+
].includes(body.model)) {
|
|
64
73
|
const { reasoning_effort } = body, rest = __rest(body, ["reasoning_effort"]);
|
|
65
74
|
return Object.assign(Object.assign({}, rest), { extra_body: Object.assign(Object.assign({}, ((_c = rest.extra_body) !== null && _c !== void 0 ? _c : {})), { enable_thinking: reasoning_effort !== 'low' }) });
|
|
66
75
|
}
|
|
@@ -82,7 +91,7 @@ class BailianProvider {
|
|
|
82
91
|
*/
|
|
83
92
|
generate(request) {
|
|
84
93
|
return __awaiter(this, void 0, void 0, function* () {
|
|
85
|
-
var _a, _b;
|
|
94
|
+
var _a, _b, _c;
|
|
86
95
|
const body = this.adaptRequest(Object.assign(Object.assign({}, this.buildRequestBody(request)), { stream: false }));
|
|
87
96
|
const data = yield (0, index_1.callChatCompletions)(this.config.baseUrl, this.config.apiKey, body, this.name);
|
|
88
97
|
const choice = data.choices[0];
|
|
@@ -98,6 +107,7 @@ class BailianProvider {
|
|
|
98
107
|
}
|
|
99
108
|
: undefined,
|
|
100
109
|
model: (_b = data.model) !== null && _b !== void 0 ? _b : body.model,
|
|
110
|
+
finishReason: (_c = choice.finish_reason) !== null && _c !== void 0 ? _c : undefined,
|
|
101
111
|
};
|
|
102
112
|
});
|
|
103
113
|
}
|
|
@@ -109,7 +119,7 @@ class BailianProvider {
|
|
|
109
119
|
*/
|
|
110
120
|
stream(request) {
|
|
111
121
|
return __asyncGenerator(this, arguments, function* stream_1() {
|
|
112
|
-
var _a
|
|
122
|
+
var _a;
|
|
113
123
|
const url = `${this.config.baseUrl}/chat/completions`;
|
|
114
124
|
const body = this.adaptRequest(Object.assign(Object.assign({}, this.buildRequestBody(request)), { stream: true, stream_options: { include_usage: true } }));
|
|
115
125
|
let response;
|
|
@@ -140,6 +150,53 @@ class BailianProvider {
|
|
|
140
150
|
// usage 可能与 finish_reason 在同一个 chunk,也可能在独立的 chunk(choices: [])中返回。
|
|
141
151
|
// 统一在此收集,流结束后一次性 yield finish 事件。
|
|
142
152
|
let streamUsage;
|
|
153
|
+
let finishReason;
|
|
154
|
+
let sawDone = false;
|
|
155
|
+
let sawFinishReason = false;
|
|
156
|
+
function* processSseLine(line) {
|
|
157
|
+
var _a, _b;
|
|
158
|
+
const trimmed = line.trim();
|
|
159
|
+
if (!trimmed || !trimmed.startsWith('data:'))
|
|
160
|
+
return;
|
|
161
|
+
const data = trimmed.slice(5).trimStart();
|
|
162
|
+
if (!data)
|
|
163
|
+
return;
|
|
164
|
+
if (data === '[DONE]') {
|
|
165
|
+
sawDone = true;
|
|
166
|
+
return;
|
|
167
|
+
}
|
|
168
|
+
let parsed;
|
|
169
|
+
try {
|
|
170
|
+
parsed = JSON.parse(data);
|
|
171
|
+
}
|
|
172
|
+
catch (cause) {
|
|
173
|
+
const message = cause instanceof Error ? cause.message : String(cause);
|
|
174
|
+
throw new Error(`[bailian] Malformed SSE JSON: ${message}`);
|
|
175
|
+
}
|
|
176
|
+
// 从任意 chunk 中收集 usage(包括 choices 为空的独立 usage chunk)
|
|
177
|
+
if (parsed.usage) {
|
|
178
|
+
streamUsage = {
|
|
179
|
+
promptTokens: parsed.usage.prompt_tokens,
|
|
180
|
+
completionTokens: parsed.usage.completion_tokens,
|
|
181
|
+
totalTokens: parsed.usage.total_tokens,
|
|
182
|
+
cachedPromptTokens: (_a = parsed.usage.prompt_tokens_details) === null || _a === void 0 ? void 0 : _a.cached_tokens,
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
const choice = (_b = parsed.choices) === null || _b === void 0 ? void 0 : _b[0];
|
|
186
|
+
if (!choice)
|
|
187
|
+
return;
|
|
188
|
+
const delta = choice.delta;
|
|
189
|
+
if (delta.reasoning_content) {
|
|
190
|
+
yield { type: 'reasoning', delta: delta.reasoning_content };
|
|
191
|
+
}
|
|
192
|
+
if (delta.content) {
|
|
193
|
+
yield { type: 'content', delta: delta.content };
|
|
194
|
+
}
|
|
195
|
+
if (choice.finish_reason) {
|
|
196
|
+
finishReason = choice.finish_reason;
|
|
197
|
+
sawFinishReason = true;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
143
200
|
try {
|
|
144
201
|
while (true) {
|
|
145
202
|
const { done, value } = yield __await(reader.read());
|
|
@@ -149,41 +206,31 @@ class BailianProvider {
|
|
|
149
206
|
const lines = buffer.split('\n');
|
|
150
207
|
buffer = (_a = lines.pop()) !== null && _a !== void 0 ? _a : '';
|
|
151
208
|
for (const line of lines) {
|
|
152
|
-
const
|
|
153
|
-
|
|
154
|
-
continue;
|
|
155
|
-
const data = trimmed.slice(6);
|
|
156
|
-
if (data === '[DONE]')
|
|
157
|
-
continue;
|
|
158
|
-
let parsed;
|
|
159
|
-
try {
|
|
160
|
-
parsed = JSON.parse(data);
|
|
161
|
-
}
|
|
162
|
-
catch (_d) {
|
|
163
|
-
continue;
|
|
164
|
-
}
|
|
165
|
-
// 从任意 chunk 中收集 usage(包括 choices 为空的独立 usage chunk)
|
|
166
|
-
if (parsed.usage) {
|
|
167
|
-
streamUsage = {
|
|
168
|
-
promptTokens: parsed.usage.prompt_tokens,
|
|
169
|
-
completionTokens: parsed.usage.completion_tokens,
|
|
170
|
-
totalTokens: parsed.usage.total_tokens,
|
|
171
|
-
cachedPromptTokens: (_b = parsed.usage.prompt_tokens_details) === null || _b === void 0 ? void 0 : _b.cached_tokens,
|
|
172
|
-
};
|
|
209
|
+
for (const chunk of processSseLine(line)) {
|
|
210
|
+
yield yield __await(chunk);
|
|
173
211
|
}
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
yield yield __await(
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
// TextDecoder 的流式模式可能仍持有不完整的多字节字符;EOF 时必须 flush。
|
|
215
|
+
buffer += decoder.decode();
|
|
216
|
+
if (buffer) {
|
|
217
|
+
const finalLines = buffer.split('\n');
|
|
218
|
+
for (const line of finalLines) {
|
|
219
|
+
for (const chunk of processSseLine(line)) {
|
|
220
|
+
yield yield __await(chunk);
|
|
183
221
|
}
|
|
184
222
|
}
|
|
185
223
|
}
|
|
186
|
-
|
|
224
|
+
if (!sawDone && !sawFinishReason) {
|
|
225
|
+
throw new Error(`[${this.name}] SSE stream ended before [DONE] or finish_reason`);
|
|
226
|
+
}
|
|
227
|
+
yield yield __await({
|
|
228
|
+
type: 'finish',
|
|
229
|
+
finishReason,
|
|
230
|
+
sawDone,
|
|
231
|
+
sawFinishReason,
|
|
232
|
+
usage: streamUsage,
|
|
233
|
+
});
|
|
187
234
|
}
|
|
188
235
|
finally {
|
|
189
236
|
reader.releaseLock();
|
|
@@ -212,6 +259,7 @@ class BailianProvider {
|
|
|
212
259
|
}
|
|
213
260
|
exports.BailianProvider = BailianProvider;
|
|
214
261
|
BailianProvider.SUPPORTED_MODELS = new Set([
|
|
262
|
+
'qwen3.8-flash',
|
|
215
263
|
'qwen3.7-max',
|
|
216
264
|
'qwen3.7-flash',
|
|
217
265
|
'qwen-plus',
|
|
@@ -219,12 +267,14 @@ BailianProvider.SUPPORTED_MODELS = new Set([
|
|
|
219
267
|
'qwq-plus',
|
|
220
268
|
'deepseek-v4-pro',
|
|
221
269
|
'deepseek-v4-flash',
|
|
270
|
+
'deepseek-v4.1-flash',
|
|
222
271
|
'kimi-k2.6',
|
|
223
272
|
'glm-5.1',
|
|
224
273
|
'qwen-vl-plus',
|
|
225
274
|
]);
|
|
226
275
|
BailianProvider.MODEL_CAPABILITIES = {
|
|
227
276
|
'qwen-max': { jsonMode: true, reasoningEffort: false },
|
|
277
|
+
'qwen3.8-flash': { jsonMode: true, reasoningEffort: true },
|
|
228
278
|
'qwen3.7-max': { jsonMode: true, reasoningEffort: true },
|
|
229
279
|
'qwen3.7-flash': { jsonMode: true, reasoningEffort: true },
|
|
230
280
|
'qwen-plus': { jsonMode: true, reasoningEffort: true },
|
|
@@ -234,6 +284,7 @@ BailianProvider.MODEL_CAPABILITIES = {
|
|
|
234
284
|
// 通过 adaptRequest 映射为 extra_body.enable_thinking(与 Qwen 同一分支)。
|
|
235
285
|
'deepseek-v4-pro': { jsonMode: true, reasoningEffort: true },
|
|
236
286
|
'deepseek-v4-flash': { jsonMode: true, reasoningEffort: true },
|
|
287
|
+
'deepseek-v4.1-flash': { jsonMode: true, reasoningEffort: true },
|
|
237
288
|
'kimi-k2.6': { jsonMode: true, reasoningEffort: true },
|
|
238
289
|
'glm-5.1': { jsonMode: false, reasoningEffort: false },
|
|
239
290
|
'qwen-vl-plus': { jsonMode: true, reasoningEffort: false },
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* 支持的模型枚举。项目层通过此枚举选择模型,Axiom 内部路由到对应 Provider。
|
|
3
3
|
*/
|
|
4
|
-
export type Model = 'qwen3.7-max' | 'qwen3.7-flash' | 'qwen-plus' | 'qwen-turbo' | 'qwq-plus' | 'deepseek-v4-pro' | 'deepseek-v4-flash' | 'kimi-k2.6' | 'glm-5.1' | 'qwen-vl-plus';
|
|
4
|
+
export type Model = 'qwen3.8-flash' | 'qwen3.7-max' | 'qwen3.7-flash' | 'qwen-plus' | 'qwen-turbo' | 'qwq-plus' | 'deepseek-v4-pro' | 'deepseek-v4-flash' | 'deepseek-v4.1-flash' | 'kimi-k2.6' | 'glm-5.1' | 'qwen-vl-plus';
|
|
5
5
|
/** 模型到 Provider 的映射配置 */
|
|
6
6
|
export interface ModelConfig {
|
|
7
7
|
readonly model: string;
|
|
@@ -6,6 +6,7 @@ exports.MODEL_REGISTRY = void 0;
|
|
|
6
6
|
* 当首选 Provider 失败时,Predictor 按此表顺序尝试下一个。
|
|
7
7
|
*/
|
|
8
8
|
exports.MODEL_REGISTRY = {
|
|
9
|
+
'qwen3.8-flash': [{ model: 'qwen3.8-flash', provider: 'bailian' }],
|
|
9
10
|
'qwen3.7-max': [{ model: 'qwen3.7-max', provider: 'bailian' }],
|
|
10
11
|
'qwen3.7-flash': [{ model: 'qwen3.7-flash', provider: 'bailian' }],
|
|
11
12
|
'qwen-plus': [{ model: 'qwen-plus', provider: 'bailian' }],
|
|
@@ -13,6 +14,7 @@ exports.MODEL_REGISTRY = {
|
|
|
13
14
|
'qwq-plus': [{ model: 'qwq-plus', provider: 'bailian' }],
|
|
14
15
|
'deepseek-v4-pro': [{ model: 'deepseek-v4-pro', provider: 'bailian' }],
|
|
15
16
|
'deepseek-v4-flash': [{ model: 'deepseek-v4-flash', provider: 'bailian' }],
|
|
17
|
+
'deepseek-v4.1-flash': [{ model: 'deepseek-v4.1-flash', provider: 'bailian' }],
|
|
16
18
|
'kimi-k2.6': [{ model: 'kimi-k2.6', provider: 'bailian' }],
|
|
17
19
|
'glm-5.1': [{ model: 'glm-5.1', provider: 'bailian' }],
|
|
18
20
|
'qwen-vl-plus': [{ model: 'qwen-vl-plus', provider: 'bailian' }],
|
|
@@ -40,6 +40,8 @@ export interface LLMResponse {
|
|
|
40
40
|
readonly reasoningContent?: string;
|
|
41
41
|
readonly usage?: TokenUsage;
|
|
42
42
|
readonly model: string;
|
|
43
|
+
/** 上游返回的生成结束原因,例如 `stop`、`length` 或 `content_filter`;Provider 未提供时为 undefined。 */
|
|
44
|
+
readonly finishReason?: string;
|
|
43
45
|
}
|
|
44
46
|
/**
|
|
45
47
|
* 流式输出片段。迭代器每次 yield 一个 chunk。
|
|
@@ -52,7 +54,13 @@ export type StreamChunk = {
|
|
|
52
54
|
readonly delta: string;
|
|
53
55
|
} | {
|
|
54
56
|
readonly type: 'finish';
|
|
55
|
-
|
|
57
|
+
/** 上游返回的生成结束原因,例如 `stop`、`length` 或 `content_filter`。 */
|
|
58
|
+
readonly finishReason?: string;
|
|
59
|
+
/** SSE 流是否收到显式的 `data: [DONE]` 终止标记;未提供表示 Provider 不支持该观测。 */
|
|
60
|
+
readonly sawDone?: boolean;
|
|
61
|
+
/** SSE 流是否收到非空的 `finish_reason`;未提供表示 Provider 不支持该观测。 */
|
|
62
|
+
readonly sawFinishReason?: boolean;
|
|
63
|
+
readonly usage?: LLMResponse['usage'];
|
|
56
64
|
};
|
|
57
65
|
/**
|
|
58
66
|
* Provider 配置。每个 Provider 实例需要一组连接参数。
|
|
@@ -26,6 +26,7 @@ export declare class LLM {
|
|
|
26
26
|
* @param config - 预测配置(模型、prompt、温度等)
|
|
27
27
|
* @returns 模型生成的完整响应及 token 消耗统计
|
|
28
28
|
* @throws Error - Provider 未配置、模型不存在、或所有 Provider 均失败时抛出
|
|
29
|
+
* @throws Error - `responseFormat: 'json'` 时输出为空或无法解析为 JSON;message 含 model、finish_reason、原文
|
|
29
30
|
*/
|
|
30
31
|
static predict(config: PredictConfig & {
|
|
31
32
|
responseFormat: 'json';
|
|
@@ -48,6 +49,7 @@ export declare class LLM {
|
|
|
48
49
|
* @param config - 预测配置(不含 prompt,因为由 messages 提供)
|
|
49
50
|
* @returns 模型生成的完整响应及 token 消耗统计
|
|
50
51
|
* @throws Error - Provider 未配置、模型不存在、或所有 Provider 均失败时抛出
|
|
52
|
+
* @throws Error - `responseFormat: 'json'` 时输出为空或无法解析为 JSON;message 含 model、finish_reason、原文
|
|
51
53
|
*/
|
|
52
54
|
static predictWithMessages(messages: ReadonlyArray<Message>, config: PredictWithMessagesConfig & {
|
|
53
55
|
responseFormat: 'json';
|
package/dist/predict/predict.js
CHANGED
|
@@ -69,16 +69,24 @@ function toLLMRequest(config, messages) {
|
|
|
69
69
|
};
|
|
70
70
|
}
|
|
71
71
|
function unwrapResponse(response, responseFormat) {
|
|
72
|
-
var _a;
|
|
72
|
+
var _a, _b;
|
|
73
73
|
const usage = response.usage;
|
|
74
74
|
if (responseFormat === 'json') {
|
|
75
|
+
// 原文与 finish_reason 写进 message:调用方只 catch 到异常时也能从日志判断是截断(length)还是模型输出非法 JSON(stop)
|
|
76
|
+
const context = `model=${response.model}, finish_reason=${(_a = response.finishReason) !== null && _a !== void 0 ? _a : 'unknown'}`;
|
|
75
77
|
if (!response.content) {
|
|
76
|
-
throw new Error(
|
|
78
|
+
throw new Error(`Empty response content when responseFormat is json (${context})`);
|
|
79
|
+
}
|
|
80
|
+
try {
|
|
81
|
+
return { content: JSON.parse(response.content), usage };
|
|
82
|
+
}
|
|
83
|
+
catch (cause) {
|
|
84
|
+
const message = cause instanceof Error ? cause.message : String(cause);
|
|
85
|
+
throw new Error(`JSON parse failed: ${message} (${context}, raw_length=${response.content.length})\nraw: ${response.content}`);
|
|
77
86
|
}
|
|
78
|
-
return { content: JSON.parse(response.content), usage };
|
|
79
87
|
}
|
|
80
88
|
// 默认按 text 返回(包括未设置 responseFormat 的情况)
|
|
81
|
-
return { content: (
|
|
89
|
+
return { content: (_b = response.content) !== null && _b !== void 0 ? _b : '', usage };
|
|
82
90
|
}
|
|
83
91
|
/**
|
|
84
92
|
* LLM 静态类。封装 Provider 连接、请求组装、故障转移和返回解析。
|
package/package.json
CHANGED
|
@@ -1,27 +0,0 @@
|
|
|
1
|
-
import type { LLMProvider } from '../llm';
|
|
2
|
-
import type { LLMRequest, LLMResponse, ProviderConfig, StreamChunk } from '../types';
|
|
3
|
-
/**
|
|
4
|
-
* 阿里云百炼(兼容 OpenAI 协议)Provider 实现。
|
|
5
|
-
* 支持非流式调用和流式 SSE 输出。
|
|
6
|
-
*/
|
|
7
|
-
export declare class BailianProvider implements LLMProvider {
|
|
8
|
-
readonly config: ProviderConfig;
|
|
9
|
-
readonly name = "bailian";
|
|
10
|
-
constructor(config: ProviderConfig);
|
|
11
|
-
supports(model: string): boolean;
|
|
12
|
-
/**
|
|
13
|
-
* 发送非流式 chat completion 请求到百炼服务。
|
|
14
|
-
* @param request - 标准化 LLM 请求
|
|
15
|
-
* @returns 解析后的模型响应(含 content、usage、model)
|
|
16
|
-
* @throws Error - 网络异常时抛 `[bailian] ...`;HTTP 非 2xx 时抛 `[bailian] HTTP {status}: ...`
|
|
17
|
-
*/
|
|
18
|
-
generate(request: LLMRequest): Promise<LLMResponse>;
|
|
19
|
-
/**
|
|
20
|
-
* 发送流式 chat completion 请求,解析 SSE 响应逐块返回。
|
|
21
|
-
* @param request - 标准化 LLM 请求
|
|
22
|
-
* @yields 内容片段(`content`)或结束标记(`finish`)
|
|
23
|
-
* @throws Error - 网络异常或 HTTP 错误时抛出
|
|
24
|
-
*/
|
|
25
|
-
stream(request: LLMRequest): AsyncGenerator<StreamChunk, void, unknown>;
|
|
26
|
-
private buildRequestBody;
|
|
27
|
-
}
|
|
@@ -1,185 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
var __awaiter = (this && this.__awaiter) || function (thisArg, _arguments, P, generator) {
|
|
3
|
-
function adopt(value) { return value instanceof P ? value : new P(function (resolve) { resolve(value); }); }
|
|
4
|
-
return new (P || (P = Promise))(function (resolve, reject) {
|
|
5
|
-
function fulfilled(value) { try { step(generator.next(value)); } catch (e) { reject(e); } }
|
|
6
|
-
function rejected(value) { try { step(generator["throw"](value)); } catch (e) { reject(e); } }
|
|
7
|
-
function step(result) { result.done ? resolve(result.value) : adopt(result.value).then(fulfilled, rejected); }
|
|
8
|
-
step((generator = generator.apply(thisArg, _arguments || [])).next());
|
|
9
|
-
});
|
|
10
|
-
};
|
|
11
|
-
var __await = (this && this.__await) || function (v) { return this instanceof __await ? (this.v = v, this) : new __await(v); }
|
|
12
|
-
var __asyncGenerator = (this && this.__asyncGenerator) || function (thisArg, _arguments, generator) {
|
|
13
|
-
if (!Symbol.asyncIterator) throw new TypeError("Symbol.asyncIterator is not defined.");
|
|
14
|
-
var g = generator.apply(thisArg, _arguments || []), i, q = [];
|
|
15
|
-
return i = Object.create((typeof AsyncIterator === "function" ? AsyncIterator : Object).prototype), verb("next"), verb("throw"), verb("return", awaitReturn), i[Symbol.asyncIterator] = function () { return this; }, i;
|
|
16
|
-
function awaitReturn(f) { return function (v) { return Promise.resolve(v).then(f, reject); }; }
|
|
17
|
-
function verb(n, f) { if (g[n]) { i[n] = function (v) { return new Promise(function (a, b) { q.push([n, v, a, b]) > 1 || resume(n, v); }); }; if (f) i[n] = f(i[n]); } }
|
|
18
|
-
function resume(n, v) { try { step(g[n](v)); } catch (e) { settle(q[0][3], e); } }
|
|
19
|
-
function step(r) { r.value instanceof __await ? Promise.resolve(r.value.v).then(fulfill, reject) : settle(q[0][2], r); }
|
|
20
|
-
function fulfill(value) { resume("next", value); }
|
|
21
|
-
function reject(value) { resume("throw", value); }
|
|
22
|
-
function settle(f, v) { if (f(v), q.shift(), q.length) resume(q[0][0], q[0][1]); }
|
|
23
|
-
};
|
|
24
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
|
-
exports.BailianProvider = void 0;
|
|
26
|
-
const models_1 = require("../models");
|
|
27
|
-
/**
|
|
28
|
-
* 阿里云百炼(兼容 OpenAI 协议)Provider 实现。
|
|
29
|
-
* 支持非流式调用和流式 SSE 输出。
|
|
30
|
-
*/
|
|
31
|
-
class BailianProvider {
|
|
32
|
-
constructor(config) {
|
|
33
|
-
this.config = config;
|
|
34
|
-
this.name = 'bailian';
|
|
35
|
-
}
|
|
36
|
-
supports(model) {
|
|
37
|
-
return Object.entries(models_1.MODEL_REGISTRY).some(([m, configs]) => m === model && configs.some((c) => c.provider === this.name));
|
|
38
|
-
}
|
|
39
|
-
/**
|
|
40
|
-
* 发送非流式 chat completion 请求到百炼服务。
|
|
41
|
-
* @param request - 标准化 LLM 请求
|
|
42
|
-
* @returns 解析后的模型响应(含 content、usage、model)
|
|
43
|
-
* @throws Error - 网络异常时抛 `[bailian] ...`;HTTP 非 2xx 时抛 `[bailian] HTTP {status}: ...`
|
|
44
|
-
*/
|
|
45
|
-
generate(request) {
|
|
46
|
-
return __awaiter(this, void 0, void 0, function* () {
|
|
47
|
-
const url = `${this.config.baseUrl}/chat/completions`;
|
|
48
|
-
const body = this.buildRequestBody(request);
|
|
49
|
-
let response;
|
|
50
|
-
try {
|
|
51
|
-
response = yield fetch(url, {
|
|
52
|
-
method: 'POST',
|
|
53
|
-
headers: {
|
|
54
|
-
'Content-Type': 'application/json',
|
|
55
|
-
Authorization: `Bearer ${this.config.apiKey}`,
|
|
56
|
-
},
|
|
57
|
-
body: JSON.stringify(Object.assign(Object.assign({}, body), { stream: false })),
|
|
58
|
-
});
|
|
59
|
-
}
|
|
60
|
-
catch (cause) {
|
|
61
|
-
throw new Error(`[${this.name}] ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
62
|
-
}
|
|
63
|
-
if (!response.ok) {
|
|
64
|
-
const text = yield response.text();
|
|
65
|
-
throw new Error(`[${this.name}] HTTP ${response.status}: ${text}`);
|
|
66
|
-
}
|
|
67
|
-
const data = (yield response.json());
|
|
68
|
-
const choice = data.choices[0];
|
|
69
|
-
if (!choice) {
|
|
70
|
-
throw new Error(`[${this.name}] No choice in response`);
|
|
71
|
-
}
|
|
72
|
-
return {
|
|
73
|
-
content: choice.message.content,
|
|
74
|
-
usage: data.usage
|
|
75
|
-
? {
|
|
76
|
-
promptTokens: data.usage.prompt_tokens,
|
|
77
|
-
completionTokens: data.usage.completion_tokens,
|
|
78
|
-
totalTokens: data.usage.total_tokens,
|
|
79
|
-
}
|
|
80
|
-
: undefined,
|
|
81
|
-
model: data.model,
|
|
82
|
-
};
|
|
83
|
-
});
|
|
84
|
-
}
|
|
85
|
-
/**
|
|
86
|
-
* 发送流式 chat completion 请求,解析 SSE 响应逐块返回。
|
|
87
|
-
* @param request - 标准化 LLM 请求
|
|
88
|
-
* @yields 内容片段(`content`)或结束标记(`finish`)
|
|
89
|
-
* @throws Error - 网络异常或 HTTP 错误时抛出
|
|
90
|
-
*/
|
|
91
|
-
stream(request) {
|
|
92
|
-
return __asyncGenerator(this, arguments, function* stream_1() {
|
|
93
|
-
var _a, _b;
|
|
94
|
-
const url = `${this.config.baseUrl}/chat/completions`;
|
|
95
|
-
const body = this.buildRequestBody(request);
|
|
96
|
-
let response;
|
|
97
|
-
try {
|
|
98
|
-
response = yield __await(fetch(url, {
|
|
99
|
-
method: 'POST',
|
|
100
|
-
headers: {
|
|
101
|
-
'Content-Type': 'application/json',
|
|
102
|
-
Authorization: `Bearer ${this.config.apiKey}`,
|
|
103
|
-
},
|
|
104
|
-
body: JSON.stringify(Object.assign(Object.assign({}, body), { stream: true })),
|
|
105
|
-
}));
|
|
106
|
-
}
|
|
107
|
-
catch (cause) {
|
|
108
|
-
throw new Error(`[${this.name}] ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
109
|
-
}
|
|
110
|
-
if (!response.ok) {
|
|
111
|
-
const text = yield __await(response.text());
|
|
112
|
-
throw new Error(`[${this.name}] HTTP ${response.status}: ${text}`);
|
|
113
|
-
}
|
|
114
|
-
if (!response.body) {
|
|
115
|
-
throw new Error(`[${this.name}] Response body is null`);
|
|
116
|
-
}
|
|
117
|
-
const reader = response.body.getReader();
|
|
118
|
-
const decoder = new TextDecoder();
|
|
119
|
-
let buffer = '';
|
|
120
|
-
try {
|
|
121
|
-
while (true) {
|
|
122
|
-
const { done, value } = yield __await(reader.read());
|
|
123
|
-
if (done)
|
|
124
|
-
break;
|
|
125
|
-
buffer += decoder.decode(value, { stream: true });
|
|
126
|
-
const lines = buffer.split('\n');
|
|
127
|
-
buffer = (_a = lines.pop()) !== null && _a !== void 0 ? _a : '';
|
|
128
|
-
for (const line of lines) {
|
|
129
|
-
const trimmed = line.trim();
|
|
130
|
-
if (!trimmed || !trimmed.startsWith('data: '))
|
|
131
|
-
continue;
|
|
132
|
-
const data = trimmed.slice(6);
|
|
133
|
-
if (data === '[DONE]')
|
|
134
|
-
continue;
|
|
135
|
-
let parsed;
|
|
136
|
-
try {
|
|
137
|
-
parsed = JSON.parse(data);
|
|
138
|
-
}
|
|
139
|
-
catch (_c) {
|
|
140
|
-
continue;
|
|
141
|
-
}
|
|
142
|
-
const choice = (_b = parsed.choices) === null || _b === void 0 ? void 0 : _b[0];
|
|
143
|
-
if (!choice)
|
|
144
|
-
continue;
|
|
145
|
-
const delta = choice.delta;
|
|
146
|
-
if (delta.content) {
|
|
147
|
-
yield yield __await({ type: 'content', delta: delta.content });
|
|
148
|
-
}
|
|
149
|
-
if (choice.finish_reason && parsed.usage) {
|
|
150
|
-
yield yield __await({
|
|
151
|
-
type: 'finish',
|
|
152
|
-
usage: {
|
|
153
|
-
promptTokens: parsed.usage.prompt_tokens,
|
|
154
|
-
completionTokens: parsed.usage.completion_tokens,
|
|
155
|
-
totalTokens: parsed.usage.total_tokens,
|
|
156
|
-
},
|
|
157
|
-
});
|
|
158
|
-
}
|
|
159
|
-
}
|
|
160
|
-
}
|
|
161
|
-
yield yield __await({ type: 'finish' });
|
|
162
|
-
}
|
|
163
|
-
finally {
|
|
164
|
-
reader.releaseLock();
|
|
165
|
-
}
|
|
166
|
-
});
|
|
167
|
-
}
|
|
168
|
-
buildRequestBody(request) {
|
|
169
|
-
var _a;
|
|
170
|
-
return {
|
|
171
|
-
model: (_a = request.model) !== null && _a !== void 0 ? _a : this.config.defaultModel,
|
|
172
|
-
messages: request.messages.map((m) => ({
|
|
173
|
-
role: m.role,
|
|
174
|
-
content: m.content,
|
|
175
|
-
})),
|
|
176
|
-
temperature: request.temperature,
|
|
177
|
-
max_tokens: request.maxTokens,
|
|
178
|
-
top_p: request.topP,
|
|
179
|
-
response_format: request.responseFormat
|
|
180
|
-
? { type: request.responseFormat === 'json' ? 'json_object' : 'text' }
|
|
181
|
-
: undefined,
|
|
182
|
-
};
|
|
183
|
-
}
|
|
184
|
-
}
|
|
185
|
-
exports.BailianProvider = BailianProvider;
|