agstack 1.24.1__tar.gz → 1.25.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agstack-1.24.1 → agstack-1.25.1}/PKG-INFO +2 -2
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/client.py +99 -8
- {agstack-1.24.1 → agstack-1.25.1}/agstack.egg-info/PKG-INFO +2 -2
- {agstack-1.24.1 → agstack-1.25.1}/agstack.egg-info/SOURCES.txt +1 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack.egg-info/requires.txt +1 -1
- {agstack-1.24.1 → agstack-1.25.1}/pyproject.toml +2 -2
- agstack-1.25.1/tests/test_llm_usage_callback.py +83 -0
- {agstack-1.24.1 → agstack-1.25.1}/LICENSE +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/README.md +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/cache/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/cache/base.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/cache/memory.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/cache/redis.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/config/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/config/logger.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/config/manager.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/config/types.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/contexts.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/decorators.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/events.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/exceptions.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/fastapi/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/fastapi/exception.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/fastapi/middleware.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/fastapi/offline.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/fastapi/sse.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/infra/db/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/infra/es/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/infra/kg/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/infra/mq/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/agent.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/context.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/event.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/exceptions.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/factory.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/flow.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/loader.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/agent_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/base.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/detect_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/echo_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/iterator_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/llm_chat_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/llm_embed_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/llm_rerank_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/python_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/subflow_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/switch_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/nodes/tool_node.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/records.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/registry.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/sandbox.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/state.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/tool.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/flow/trace.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/prompts.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/llm/token.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/messagebus/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/messagebus/base.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/messagebus/memory.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/messagebus/redis.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/schema.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/security/__init__.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/security/casbin.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/security/crypt.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack/status.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack.egg-info/dependency_links.txt +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/agstack.egg-info/top_level.txt +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/setup.cfg +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_cache_memory.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_cache_redis.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_flow_io.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_flow_iterator.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_flow_switch_subflow.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_messagebus_memory.py +0 -0
- {agstack-1.24.1 → agstack-1.25.1}/tests/test_messagebus_redis.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agstack
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.25.1
|
|
4
4
|
Summary: Production-ready toolkit for building FastAPI and LLM applications
|
|
5
5
|
Author-email: XtraVisions <gitadmin@xtravisions.com>, Chen Hao <chenhao@xtravisions.com>
|
|
6
6
|
Maintainer-email: XtraVisions <gitadmin@xtravisions.com>, Chen Hao <chenhao@xtravisions.com>
|
|
@@ -30,7 +30,7 @@ Requires-Dist: pydantic>=2.13.3
|
|
|
30
30
|
Requires-Dist: python-multipart>=0.0.26
|
|
31
31
|
Requires-Dist: requests>=2.32.5
|
|
32
32
|
Requires-Dist: RestrictedPython>=7.0
|
|
33
|
-
Requires-Dist: sqlobjects>=
|
|
33
|
+
Requires-Dist: sqlobjects>=2.0.0
|
|
34
34
|
Requires-Dist: tiktoken>=0.12.0
|
|
35
35
|
Requires-Dist: uvicorn>=0.46.0
|
|
36
36
|
Provides-Extra: mq
|
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
import logging
|
|
4
4
|
import time
|
|
5
|
-
from
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import TYPE_CHECKING, Any, AsyncIterator, Callable, Literal, overload
|
|
6
7
|
|
|
7
8
|
import httpx
|
|
8
9
|
from httpx import AsyncClient
|
|
@@ -24,6 +25,70 @@ from ..decorators import autoretry
|
|
|
24
25
|
logger = logging.getLogger(__name__)
|
|
25
26
|
|
|
26
27
|
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class UsageEvent:
|
|
30
|
+
"""单次 LLM 调用的用量事件
|
|
31
|
+
|
|
32
|
+
:param model: 模型名称
|
|
33
|
+
:param kind: 调用类型(chat / chat_stream / vision / embedding / rerank)
|
|
34
|
+
:param prompt_tokens: 输入 token 数(后端未返回 usage 时为 0)
|
|
35
|
+
:param completion_tokens: 输出 token 数
|
|
36
|
+
:param total_tokens: 总 token 数
|
|
37
|
+
:param duration_ms: 调用耗时(毫秒)
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
model: str
|
|
41
|
+
kind: str
|
|
42
|
+
prompt_tokens: int
|
|
43
|
+
completion_tokens: int
|
|
44
|
+
total_tokens: int
|
|
45
|
+
duration_ms: int
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
UsageCallback = Callable[[UsageEvent], None]
|
|
49
|
+
"""usage 回调类型:必须同步、快速返回且自行捕获所有异常(不阻塞、不影响主调用链)"""
|
|
50
|
+
|
|
51
|
+
_usage_callback: UsageCallback | None = None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def set_usage_callback(callback: UsageCallback | None) -> None:
|
|
55
|
+
"""注册全局 usage 回调(进程级单例,传 None 注销)
|
|
56
|
+
|
|
57
|
+
每次 LLM 调用(chat/chat_stream/vision/embed/rerank,含同步变体)完成后触发一次。
|
|
58
|
+
回调实现方应自行兜底异常;此处仍会捕获并告警,绝不向主调用链抛出。
|
|
59
|
+
"""
|
|
60
|
+
global _usage_callback
|
|
61
|
+
_usage_callback = callback
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _usage_field(usage: Any, key: str) -> int:
|
|
65
|
+
"""从 usage(对象/字典/None)中容错提取整数字段"""
|
|
66
|
+
if usage is None:
|
|
67
|
+
return 0
|
|
68
|
+
value = usage.get(key) if isinstance(usage, dict) else getattr(usage, key, None)
|
|
69
|
+
return int(value or 0)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _emit_usage(model: str, kind: str, usage: Any, duration_ms: int) -> None:
|
|
73
|
+
"""触发 usage 回调;usage 缺失时各分量记 0(保留 model/duration 供监控)"""
|
|
74
|
+
callback = _usage_callback
|
|
75
|
+
if callback is None:
|
|
76
|
+
return
|
|
77
|
+
try:
|
|
78
|
+
callback(
|
|
79
|
+
UsageEvent(
|
|
80
|
+
model=model,
|
|
81
|
+
kind=kind,
|
|
82
|
+
prompt_tokens=_usage_field(usage, "prompt_tokens"),
|
|
83
|
+
completion_tokens=_usage_field(usage, "completion_tokens"),
|
|
84
|
+
total_tokens=_usage_field(usage, "total_tokens"),
|
|
85
|
+
duration_ms=duration_ms,
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
except Exception:
|
|
89
|
+
logger.warning("Usage callback failed", exc_info=True)
|
|
90
|
+
|
|
91
|
+
|
|
27
92
|
class LLMError(AppException):
|
|
28
93
|
"""LLM 调用错误基类"""
|
|
29
94
|
|
|
@@ -140,10 +205,15 @@ class LLMClient:
|
|
|
140
205
|
"""
|
|
141
206
|
start = time.time()
|
|
142
207
|
model_name = model
|
|
208
|
+
# 内部调用类型标记(vision 经由 chat 转发时传入,不透传给推理后端)
|
|
209
|
+
usage_kind = kwargs.pop("usage_kind", "chat")
|
|
143
210
|
|
|
144
211
|
try:
|
|
145
212
|
if stream:
|
|
146
|
-
|
|
213
|
+
stream_kind = "chat_stream" if usage_kind == "chat" else usage_kind
|
|
214
|
+
return self._chat_stream(
|
|
215
|
+
messages, model_name, temperature, max_tokens, start, usage_kind=stream_kind, **kwargs
|
|
216
|
+
)
|
|
147
217
|
|
|
148
218
|
@autoretry(
|
|
149
219
|
logger,
|
|
@@ -171,6 +241,7 @@ class LLMClient:
|
|
|
171
241
|
logger.info(f"LLM: model={model_name}, tokens={usage.total_tokens}, duration={duration_ms}ms")
|
|
172
242
|
else:
|
|
173
243
|
logger.info(f"LLM: model={model_name}, duration={duration_ms}ms")
|
|
244
|
+
_emit_usage(model_name, usage_kind, usage, duration_ms)
|
|
174
245
|
|
|
175
246
|
return response
|
|
176
247
|
|
|
@@ -210,6 +281,7 @@ class LLMClient:
|
|
|
210
281
|
client = self._get_sync_client()
|
|
211
282
|
start = time.time()
|
|
212
283
|
model_name = model
|
|
284
|
+
usage_kind = kwargs.pop("usage_kind", "chat")
|
|
213
285
|
|
|
214
286
|
try:
|
|
215
287
|
response = client.chat.completions.create(
|
|
@@ -226,6 +298,7 @@ class LLMClient:
|
|
|
226
298
|
logger.info(f"LLM (sync): model={model_name}, tokens={usage.total_tokens}, duration={duration_ms}ms")
|
|
227
299
|
else:
|
|
228
300
|
logger.info(f"LLM (sync): model={model_name}, duration={duration_ms}ms")
|
|
301
|
+
_emit_usage(model_name, usage_kind, usage, duration_ms)
|
|
229
302
|
|
|
230
303
|
return response
|
|
231
304
|
|
|
@@ -250,10 +323,11 @@ class LLMClient:
|
|
|
250
323
|
temperature: float,
|
|
251
324
|
max_tokens: int | None,
|
|
252
325
|
start_time: float,
|
|
326
|
+
usage_kind: str = "chat_stream",
|
|
253
327
|
**kwargs: Any,
|
|
254
328
|
) -> AsyncIterator["ChatCompletionChunk"]:
|
|
255
329
|
"""流式响应"""
|
|
256
|
-
|
|
330
|
+
final_usage = None
|
|
257
331
|
|
|
258
332
|
try:
|
|
259
333
|
# noinspection PyTypeChecker
|
|
@@ -269,15 +343,17 @@ class LLMClient:
|
|
|
269
343
|
)
|
|
270
344
|
|
|
271
345
|
async for chunk in stream:
|
|
272
|
-
# 收集 token
|
|
346
|
+
# 收集 token 统计(usage 通常在末尾 chunk 返回)
|
|
273
347
|
if chunk.usage:
|
|
274
|
-
|
|
348
|
+
final_usage = chunk.usage
|
|
275
349
|
|
|
276
350
|
yield chunk
|
|
277
351
|
|
|
278
352
|
# 记录指标
|
|
279
353
|
duration_ms = int((time.time() - start_time) * 1000)
|
|
354
|
+
total_tokens = final_usage.total_tokens if final_usage else 0
|
|
280
355
|
logger.info(f"LLM stream: model={model}, tokens={total_tokens}, duration={duration_ms}ms")
|
|
356
|
+
_emit_usage(model, usage_kind, final_usage, duration_ms)
|
|
281
357
|
|
|
282
358
|
except APITimeoutError as e:
|
|
283
359
|
logger.error(f"LLM stream timeout: {e}")
|
|
@@ -294,6 +370,7 @@ class LLMClient:
|
|
|
294
370
|
:param model: 模型名称
|
|
295
371
|
:return: 向量列表
|
|
296
372
|
"""
|
|
373
|
+
start = time.time()
|
|
297
374
|
|
|
298
375
|
@autoretry(
|
|
299
376
|
logger,
|
|
@@ -309,6 +386,7 @@ class LLMClient:
|
|
|
309
386
|
|
|
310
387
|
try:
|
|
311
388
|
response = await _call()
|
|
389
|
+
_emit_usage(model, "embedding", getattr(response, "usage", None), int((time.time() - start) * 1000))
|
|
312
390
|
data = getattr(response, "data", None)
|
|
313
391
|
if data:
|
|
314
392
|
return [item.embedding for item in data]
|
|
@@ -326,9 +404,11 @@ class LLMClient:
|
|
|
326
404
|
:return: 向量列表
|
|
327
405
|
"""
|
|
328
406
|
client = self._get_sync_client()
|
|
407
|
+
start = time.time()
|
|
329
408
|
|
|
330
409
|
try:
|
|
331
410
|
response = client.embeddings.create(model=model, input=texts)
|
|
411
|
+
_emit_usage(model, "embedding", getattr(response, "usage", None), int((time.time() - start) * 1000))
|
|
332
412
|
data = getattr(response, "data", None)
|
|
333
413
|
if data:
|
|
334
414
|
return [item.embedding for item in data]
|
|
@@ -395,9 +475,9 @@ class LLMClient:
|
|
|
395
475
|
|
|
396
476
|
# bypass type check of Literal param `stream`
|
|
397
477
|
if stream:
|
|
398
|
-
return await self.chat(messages, model=model, stream=True, **kwargs)
|
|
478
|
+
return await self.chat(messages, model=model, stream=True, usage_kind="vision", **kwargs)
|
|
399
479
|
|
|
400
|
-
return await self.chat(messages, model=model, stream=False, **kwargs)
|
|
480
|
+
return await self.chat(messages, model=model, stream=False, usage_kind="vision", **kwargs)
|
|
401
481
|
|
|
402
482
|
def vision_sync(
|
|
403
483
|
self,
|
|
@@ -431,7 +511,7 @@ class LLMClient:
|
|
|
431
511
|
{"role": "user", "content": content} # type: ignore[list-item]
|
|
432
512
|
]
|
|
433
513
|
|
|
434
|
-
return self.chat_sync(messages, model=model, **kwargs)
|
|
514
|
+
return self.chat_sync(messages, model=model, usage_kind="vision", **kwargs)
|
|
435
515
|
|
|
436
516
|
async def rerank(
|
|
437
517
|
self,
|
|
@@ -450,6 +530,8 @@ class LLMClient:
|
|
|
450
530
|
"""
|
|
451
531
|
from httpx import ConnectTimeout, TimeoutException
|
|
452
532
|
|
|
533
|
+
start = time.time()
|
|
534
|
+
|
|
453
535
|
@autoretry(
|
|
454
536
|
logger,
|
|
455
537
|
retries=3,
|
|
@@ -475,6 +557,10 @@ class LLMClient:
|
|
|
475
557
|
response.raise_for_status()
|
|
476
558
|
data = response.json()
|
|
477
559
|
|
|
560
|
+
# 上报用量(仅后端返回 usage 时)
|
|
561
|
+
if usage := data.get("usage"):
|
|
562
|
+
_emit_usage(model, "rerank", usage, int((time.time() - start) * 1000))
|
|
563
|
+
|
|
478
564
|
# 解析响应
|
|
479
565
|
results = []
|
|
480
566
|
for item in data.get("results", []):
|
|
@@ -519,6 +605,7 @@ class LLMClient:
|
|
|
519
605
|
import requests
|
|
520
606
|
|
|
521
607
|
session = self._get_sync_http_session()
|
|
608
|
+
start = time.time()
|
|
522
609
|
|
|
523
610
|
try:
|
|
524
611
|
response = session.post(
|
|
@@ -535,6 +622,10 @@ class LLMClient:
|
|
|
535
622
|
response.raise_for_status()
|
|
536
623
|
data = response.json()
|
|
537
624
|
|
|
625
|
+
# 上报用量(仅后端返回 usage 时)
|
|
626
|
+
if usage := data.get("usage"):
|
|
627
|
+
_emit_usage(model, "rerank", usage, int((time.time() - start) * 1000))
|
|
628
|
+
|
|
538
629
|
results = []
|
|
539
630
|
for item in data.get("results", []):
|
|
540
631
|
index = item.get("index")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agstack
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.25.1
|
|
4
4
|
Summary: Production-ready toolkit for building FastAPI and LLM applications
|
|
5
5
|
Author-email: XtraVisions <gitadmin@xtravisions.com>, Chen Hao <chenhao@xtravisions.com>
|
|
6
6
|
Maintainer-email: XtraVisions <gitadmin@xtravisions.com>, Chen Hao <chenhao@xtravisions.com>
|
|
@@ -30,7 +30,7 @@ Requires-Dist: pydantic>=2.13.3
|
|
|
30
30
|
Requires-Dist: python-multipart>=0.0.26
|
|
31
31
|
Requires-Dist: requests>=2.32.5
|
|
32
32
|
Requires-Dist: RestrictedPython>=7.0
|
|
33
|
-
Requires-Dist: sqlobjects>=
|
|
33
|
+
Requires-Dist: sqlobjects>=2.0.0
|
|
34
34
|
Requires-Dist: tiktoken>=0.12.0
|
|
35
35
|
Requires-Dist: uvicorn>=0.46.0
|
|
36
36
|
Provides-Extra: mq
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "agstack"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.25.1"
|
|
4
4
|
description = "Production-ready toolkit for building FastAPI and LLM applications"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -49,7 +49,7 @@ dependencies = [
|
|
|
49
49
|
"python-multipart>=0.0.26",
|
|
50
50
|
"requests>=2.32.5",
|
|
51
51
|
"RestrictedPython>=7.0",
|
|
52
|
-
"sqlobjects>=
|
|
52
|
+
"sqlobjects>=2.0.0",
|
|
53
53
|
"tiktoken>=0.12.0",
|
|
54
54
|
"uvicorn>=0.46.0",
|
|
55
55
|
]
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Copyright (c) 2020-2026 XtraVisions, All rights reserved.
|
|
2
|
+
|
|
3
|
+
"""LLM usage 回调钩子测试(纯内存,不依赖真实推理后端)"""
|
|
4
|
+
|
|
5
|
+
from types import SimpleNamespace
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from agstack.llm import client as llm_client
|
|
10
|
+
from agstack.llm.client import UsageEvent, _emit_usage, set_usage_callback
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@pytest.fixture(autouse=True)
|
|
14
|
+
def _reset_callback():
|
|
15
|
+
"""每个用例结束后注销全局回调,避免用例间串扰"""
|
|
16
|
+
yield
|
|
17
|
+
set_usage_callback(None)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_emit_with_object_usage():
|
|
21
|
+
"""openai 响应对象形式的 usage 正常提取三个分量"""
|
|
22
|
+
events: list[UsageEvent] = []
|
|
23
|
+
set_usage_callback(events.append)
|
|
24
|
+
|
|
25
|
+
usage = SimpleNamespace(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
|
26
|
+
_emit_usage("qwen3", "chat", usage, 1200)
|
|
27
|
+
|
|
28
|
+
assert events == [
|
|
29
|
+
UsageEvent(
|
|
30
|
+
model="qwen3", kind="chat", prompt_tokens=100, completion_tokens=50, total_tokens=150, duration_ms=1200
|
|
31
|
+
)
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_emit_with_dict_usage():
|
|
36
|
+
"""rerank 的 JSON 字典形式 usage 同样可提取,缺失字段记 0"""
|
|
37
|
+
events: list[UsageEvent] = []
|
|
38
|
+
set_usage_callback(events.append)
|
|
39
|
+
|
|
40
|
+
_emit_usage("bge-reranker", "rerank", {"total_tokens": 320}, 80)
|
|
41
|
+
|
|
42
|
+
assert events[0].total_tokens == 320
|
|
43
|
+
assert events[0].prompt_tokens == 0
|
|
44
|
+
assert events[0].completion_tokens == 0
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_emit_with_none_usage():
|
|
48
|
+
"""后端未返回 usage 时仍触发事件,各分量为 0(保留 model/duration 供监控)"""
|
|
49
|
+
events: list[UsageEvent] = []
|
|
50
|
+
set_usage_callback(events.append)
|
|
51
|
+
|
|
52
|
+
_emit_usage("qwen3", "chat_stream", None, 500)
|
|
53
|
+
|
|
54
|
+
assert events[0].total_tokens == 0
|
|
55
|
+
assert events[0].model == "qwen3"
|
|
56
|
+
assert events[0].duration_ms == 500
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_callback_exception_swallowed():
|
|
60
|
+
"""回调抛异常不得影响主调用链"""
|
|
61
|
+
|
|
62
|
+
def _bad_callback(_evt: UsageEvent) -> None:
|
|
63
|
+
raise RuntimeError("boom")
|
|
64
|
+
|
|
65
|
+
set_usage_callback(_bad_callback)
|
|
66
|
+
_emit_usage("qwen3", "chat", None, 10) # 不应抛出
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_no_callback_noop():
|
|
70
|
+
"""未注册回调时静默跳过"""
|
|
71
|
+
assert llm_client._usage_callback is None
|
|
72
|
+
_emit_usage("qwen3", "chat", None, 10) # 不应抛出
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_unregister_callback():
|
|
76
|
+
"""传 None 注销回调后不再触发"""
|
|
77
|
+
events: list[UsageEvent] = []
|
|
78
|
+
set_usage_callback(events.append)
|
|
79
|
+
set_usage_callback(None)
|
|
80
|
+
|
|
81
|
+
_emit_usage("qwen3", "chat", None, 10)
|
|
82
|
+
|
|
83
|
+
assert events == []
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|