@keo-ai/axiom 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -177,9 +177,15 @@ for await (const chunk of LLM.streamPredict({ model: 'qwen-max', prompt: '讲个
177
177
  if (chunk.type === 'reasoning') {
178
178
  process.stdout.write(chunk.delta); // 推理过程
179
179
  }
180
+ if (chunk.type === 'finish' && chunk.usage) {
181
+ console.log('Token usage:', chunk.usage);
182
+ // { promptTokens, completionTokens, totalTokens, cachedPromptTokens? }
183
+ }
180
184
  }
181
185
  ```
182
186
 
187
+ > 💡 流式调用自动启用 `stream_options: { include_usage: true }`,`finish` 事件携带完整的 token 消耗统计。usage 可能与 `finish_reason` 在同一个 SSE chunk,也可能在独立的 chunk(`choices: []`)中返回,两种情况均已兼容。
188
+
183
189
  ### 模型列表
184
190
 
185
191
  当前支持的模型(通过 `Model` 类型枚举):
@@ -328,6 +334,9 @@ for await (const chunk of stream) {
328
334
  break;
329
335
  case 'turn_end':
330
336
  console.log(`\n--- Turn ${chunk.turn} End ---`);
337
+ if (chunk.usage) {
338
+ console.log(`Token usage: prompt=${chunk.usage.promptTokens}, completion=${chunk.usage.completionTokens}`);
339
+ }
331
340
  break;
332
341
  }
333
342
  }
@@ -366,6 +375,8 @@ console.log('Turns:', result!.turns);
366
375
  | `tool_result` | tool 执行完成 | `callId`, `toolName`, `content`, `status`, `turn` |
367
376
  | `turn_end` | 一轮结束(tool 全部执行完或 content 直接返回) | `turn`, `usage?` |
368
377
 
378
+ > 💡 `turn_end` 的 `usage` 字段携带该轮 LLM 调用的 token 消耗统计(`promptTokens`、`completionTokens`、`totalTokens`、`cachedPromptTokens?`)。流式请求自动启用 `stream_options: { include_usage: true }`,确保 API 返回 usage 数据。
379
+
369
380
  #### 流式 vs 非流式的选择
370
381
 
371
382
  | 场景 | 推荐方式 |
@@ -118,6 +118,7 @@ function createLLMCaller(options) {
118
118
  tools: tools.length > 0 ? toOpenAITools(tools) : undefined,
119
119
  reasoning_effort: callOptions === null || callOptions === void 0 ? void 0 : callOptions.reasoningEffort,
120
120
  stream: true,
121
+ stream_options: { include_usage: true },
121
122
  });
122
123
  const url = `${options.baseUrl}/chat/completions`;
123
124
  let response;
@@ -148,7 +149,27 @@ function createLLMCaller(options) {
148
149
  const toolCallAccumulators = [];
149
150
  let fullContent = '';
150
151
  let fullReasoningContent = '';
151
- let hasYieldedFinish = false;
152
+ // 累积 usage:OpenAI 兼容 API 在 stream_options.include_usage 开启时,
153
+ // usage 可能与 finish_reason 在同一个 chunk,也可能在独立的 chunk(choices: [])中返回。
154
+ // 统一在此收集,流结束后合并到 finish 事件中 yield。
155
+ let streamUsage;
156
+ let hasFinishReason = false;
157
+ function buildToolCalls() {
158
+ const toolCalls = [];
159
+ for (const acc of toolCallAccumulators) {
160
+ if (acc.id && acc.type && acc.function.name) {
161
+ toolCalls.push({
162
+ id: acc.id,
163
+ type: acc.type,
164
+ function: {
165
+ name: acc.function.name,
166
+ arguments: acc.function.arguments,
167
+ },
168
+ });
169
+ }
170
+ }
171
+ return toolCalls;
172
+ }
152
173
  try {
153
174
  while (true) {
154
175
  const { done, value } = yield __await(reader.read());
@@ -171,7 +192,16 @@ function createLLMCaller(options) {
171
192
  catch (_g) {
172
193
  continue;
173
194
  }
174
- const choice = (_c = parsed.choices) === null || _c === void 0 ? void 0 : _c[0];
195
+ // 从任意 chunk 中收集 usage(包括 choices 为空的独立 usage chunk)
196
+ if (parsed.usage) {
197
+ streamUsage = {
198
+ promptTokens: parsed.usage.prompt_tokens,
199
+ completionTokens: parsed.usage.completion_tokens,
200
+ totalTokens: parsed.usage.total_tokens,
201
+ cachedPromptTokens: (_c = parsed.usage.prompt_tokens_details) === null || _c === void 0 ? void 0 : _c.cached_tokens,
202
+ };
203
+ }
204
+ const choice = (_d = parsed.choices) === null || _d === void 0 ? void 0 : _d[0];
175
205
  if (!choice)
176
206
  continue;
177
207
  const delta = choice.delta;
@@ -199,69 +229,29 @@ function createLLMCaller(options) {
199
229
  acc.id = tcDelta.id;
200
230
  if (tcDelta.type)
201
231
  acc.type = tcDelta.type;
202
- if ((_d = tcDelta.function) === null || _d === void 0 ? void 0 : _d.name) {
232
+ if ((_e = tcDelta.function) === null || _e === void 0 ? void 0 : _e.name) {
203
233
  acc.function.name = tcDelta.function.name;
204
234
  }
205
- if ((_e = tcDelta.function) === null || _e === void 0 ? void 0 : _e.arguments) {
235
+ if ((_f = tcDelta.function) === null || _f === void 0 ? void 0 : _f.arguments) {
206
236
  acc.function.arguments += tcDelta.function.arguments;
207
237
  }
208
238
  }
209
239
  }
210
- // 流结束
240
+ // 标记流结束(不立即 yield finish,等 usage 收集完毕)
211
241
  if (choice.finish_reason) {
212
- const toolCalls = [];
213
- for (const acc of toolCallAccumulators) {
214
- if (acc.id && acc.type && acc.function.name) {
215
- toolCalls.push({
216
- id: acc.id,
217
- type: acc.type,
218
- function: {
219
- name: acc.function.name,
220
- arguments: acc.function.arguments,
221
- },
222
- });
223
- }
224
- }
225
- yield yield __await({
226
- type: 'finish',
227
- content: fullContent || null,
228
- reasoningContent: fullReasoningContent || undefined,
229
- tool_calls: toolCalls.length > 0 ? toolCalls : undefined,
230
- usage: parsed.usage
231
- ? {
232
- promptTokens: parsed.usage.prompt_tokens,
233
- completionTokens: parsed.usage.completion_tokens,
234
- totalTokens: parsed.usage.total_tokens,
235
- cachedPromptTokens: (_f = parsed.usage.prompt_tokens_details) === null || _f === void 0 ? void 0 : _f.cached_tokens,
236
- }
237
- : undefined,
238
- });
239
- hasYieldedFinish = true;
240
- }
241
- }
242
- }
243
- // 兜底:如果流正常结束但没有收到 finish_reason,也 yield 一个 finish
244
- if (!hasYieldedFinish) {
245
- const toolCalls = [];
246
- for (const acc of toolCallAccumulators) {
247
- if (acc.id && acc.type && acc.function.name) {
248
- toolCalls.push({
249
- id: acc.id,
250
- type: acc.type,
251
- function: {
252
- name: acc.function.name,
253
- arguments: acc.function.arguments,
254
- },
255
- });
242
+ hasFinishReason = true;
256
243
  }
257
244
  }
258
- yield yield __await({
259
- type: 'finish',
260
- content: fullContent || null,
261
- reasoningContent: fullReasoningContent || undefined,
262
- tool_calls: toolCalls.length > 0 ? toolCalls : undefined,
263
- });
264
245
  }
246
+ // 流结束,统一 yield finish 事件(确保包含累积的 usage)
247
+ const toolCalls = buildToolCalls();
248
+ yield yield __await({
249
+ type: 'finish',
250
+ content: fullContent || null,
251
+ reasoningContent: fullReasoningContent || undefined,
252
+ tool_calls: toolCalls.length > 0 ? toolCalls : undefined,
253
+ usage: streamUsage,
254
+ });
265
255
  }
266
256
  finally {
267
257
  reader.releaseLock();
@@ -103,7 +103,7 @@ class BailianProvider {
103
103
  return __asyncGenerator(this, arguments, function* stream_1() {
104
104
  var _a, _b, _c;
105
105
  const url = `${this.config.baseUrl}/chat/completions`;
106
- const body = this.adaptRequest(Object.assign(Object.assign({}, this.buildRequestBody(request)), { stream: true }));
106
+ const body = this.adaptRequest(Object.assign(Object.assign({}, this.buildRequestBody(request)), { stream: true, stream_options: { include_usage: true } }));
107
107
  let response;
108
108
  try {
109
109
  response = yield __await(fetch(url, {
@@ -128,6 +128,10 @@ class BailianProvider {
128
128
  const reader = response.body.getReader();
129
129
  const decoder = new TextDecoder();
130
130
  let buffer = '';
131
+ // 累积 usage:OpenAI 兼容 API 在 stream_options.include_usage 开启时,
132
+ // usage 可能与 finish_reason 在同一个 chunk,也可能在独立的 chunk(choices: [])中返回。
133
+ // 统一在此收集,流结束后一次性 yield finish 事件。
134
+ let streamUsage;
131
135
  try {
132
136
  while (true) {
133
137
  const { done, value } = yield __await(reader.read());
@@ -150,7 +154,16 @@ class BailianProvider {
150
154
  catch (_d) {
151
155
  continue;
152
156
  }
153
- const choice = (_b = parsed.choices) === null || _b === void 0 ? void 0 : _b[0];
157
+ // 从任意 chunk 中收集 usage(包括 choices 为空的独立 usage chunk)
158
+ if (parsed.usage) {
159
+ streamUsage = {
160
+ promptTokens: parsed.usage.prompt_tokens,
161
+ completionTokens: parsed.usage.completion_tokens,
162
+ totalTokens: parsed.usage.total_tokens,
163
+ cachedPromptTokens: (_b = parsed.usage.prompt_tokens_details) === null || _b === void 0 ? void 0 : _b.cached_tokens,
164
+ };
165
+ }
166
+ const choice = (_c = parsed.choices) === null || _c === void 0 ? void 0 : _c[0];
154
167
  if (!choice)
155
168
  continue;
156
169
  const delta = choice.delta;
@@ -160,20 +173,9 @@ class BailianProvider {
160
173
  if (delta.content) {
161
174
  yield yield __await({ type: 'content', delta: delta.content });
162
175
  }
163
- if (choice.finish_reason && parsed.usage) {
164
- yield yield __await({
165
- type: 'finish',
166
- usage: {
167
- promptTokens: parsed.usage.prompt_tokens,
168
- completionTokens: parsed.usage.completion_tokens,
169
- totalTokens: parsed.usage.total_tokens,
170
- cachedPromptTokens: (_c = parsed.usage.prompt_tokens_details) === null || _c === void 0 ? void 0 : _c.cached_tokens,
171
- },
172
- });
173
- }
174
176
  }
175
177
  }
176
- yield yield __await({ type: 'finish' });
178
+ yield yield __await({ type: 'finish', usage: streamUsage });
177
179
  }
178
180
  finally {
179
181
  reader.releaseLock();
@@ -194,6 +196,8 @@ class BailianProvider {
194
196
  response_format: request.responseFormat
195
197
  ? { type: request.responseFormat === 'json' ? 'json_object' : 'text' }
196
198
  : undefined,
199
+ stream: request.stream,
200
+ stream_options: request.streamOptions,
197
201
  };
198
202
  }
199
203
  }
@@ -35,6 +35,10 @@ export interface OpenAIChatRequest {
35
35
  type: 'text' | 'json_object';
36
36
  };
37
37
  stream?: boolean;
38
+ /** 流式请求时是否在最后一个 chunk 中返回 usage 信息。OpenAI 兼容 API 默认不返回,需显式开启 */
39
+ stream_options?: {
40
+ include_usage: boolean;
41
+ };
38
42
  reasoning_effort?: 'low' | 'medium' | 'high';
39
43
  /** 百炼等平台的扩展参数 */
40
44
  extra_body?: Record<string, unknown>;
@@ -14,6 +14,9 @@ export interface LLMRequest {
14
14
  readonly maxTokens?: number;
15
15
  readonly topP?: number;
16
16
  readonly stream?: boolean;
17
+ readonly streamOptions?: {
18
+ include_usage: boolean;
19
+ };
17
20
  readonly model?: string;
18
21
  readonly responseFormat?: 'text' | 'json';
19
22
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@keo-ai/axiom",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "description": "基于 LLM 的预测与推理库,支持多 Provider 切换",
5
5
  "main": "dist/index.js",
6
6
  "types": "dist/index.d.ts",