tina-python 0.6.8__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. {tina_python-0.6.8/tina_python.egg-info → tina_python-0.7.0}/PKG-INFO +10 -7
  2. {tina_python-0.6.8 → tina_python-0.7.0}/README.md +4 -4
  3. {tina_python-0.6.8 → tina_python-0.7.0}/pyproject.toml +8 -3
  4. {tina_python-0.6.8 → tina_python-0.7.0}/setup.py +6 -1
  5. {tina_python-0.6.8 → tina_python-0.7.0}/tina/__init__.py +3 -1
  6. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/agent.py +13 -0
  7. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/agent_runtime.py +97 -51
  8. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/context_manager.py +8 -3
  9. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/events.py +105 -1
  10. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/executor.py +154 -43
  11. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/tools.py +130 -14
  12. {tina_python-0.6.8 → tina_python-0.7.0}/tina/core/error.py +54 -0
  13. {tina_python-0.6.8 → tina_python-0.7.0}/tina/llm/__init__.py +4 -3
  14. {tina_python-0.6.8 → tina_python-0.7.0}/tina/llm/base_api.py +14 -11
  15. {tina_python-0.6.8 → tina_python-0.7.0}/tina/llm/base_multimodal_api.py +6 -2
  16. tina_python-0.7.0/tina/llm/files_api.py +544 -0
  17. {tina_python-0.6.8 → tina_python-0.7.0}/tina/mcp/client.py +55 -5
  18. tina_python-0.7.0/tina/mcp/mcp_tools_executor.py +108 -0
  19. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/__init__.py +3 -0
  20. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/deepseek.py +31 -4
  21. tina_python-0.7.0/tina/utils/multi_agent/__init__.py +26 -0
  22. tina_python-0.7.0/tina/utils/multi_agent/environment.py +3 -0
  23. tina_python-0.7.0/tina/utils/multi_agent/message.py +3 -0
  24. tina_python-0.7.0/tina/utils/multi_agent/message_bus.py +3 -0
  25. tina_python-0.7.0/tina/utils/multi_agent/web/__init__.py +3 -0
  26. tina_python-0.7.0/tina/utils/multi_agent/web/server.py +3 -0
  27. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/multimodal_formatter.py +10 -0
  28. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/output_parser.py +258 -27
  29. tina_python-0.7.0/tina/utils/tui/__init__.py +41 -0
  30. tina_python-0.7.0/tina/utils/tui/context_manager.py +3 -0
  31. tina_python-0.7.0/tina/utils/tui/token.py +3 -0
  32. tina_python-0.7.0/tina/utils/tui/tui.py +3 -0
  33. tina_python-0.7.0/tina/utils/url.py +32 -0
  34. tina_python-0.7.0/tina/utils/usage.py +120 -0
  35. {tina_python-0.6.8 → tina_python-0.7.0/tina_python.egg-info}/PKG-INFO +10 -7
  36. {tina_python-0.6.8 → tina_python-0.7.0}/tina_python.egg-info/SOURCES.txt +9 -3
  37. {tina_python-0.6.8 → tina_python-0.7.0}/tina_python.egg-info/requires.txt +6 -2
  38. tina_python-0.6.8/tina/llm/BaseNewAPI.py +0 -5
  39. tina_python-0.6.8/tina/llm/ollama_api.py +0 -81
  40. tina_python-0.6.8/tina/mcp/mcp_tools_executor.py +0 -88
  41. tina_python-0.6.8/tina/utils/timer.py +0 -59
  42. tina_python-0.6.8/tina/utils/tui/__init__.py +0 -34
  43. tina_python-0.6.8/tina/utils/tui/context_manager.py +0 -603
  44. tina_python-0.6.8/tina/utils/tui/token.py +0 -89
  45. tina_python-0.6.8/tina/utils/tui/tui.py +0 -1545
  46. {tina_python-0.6.8 → tina_python-0.7.0}/LICENSE +0 -0
  47. {tina_python-0.6.8 → tina_python-0.7.0}/MANIFEST.in +0 -0
  48. {tina_python-0.6.8 → tina_python-0.7.0}/setup.cfg +0 -0
  49. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/__init__.py +0 -0
  50. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/__init__.py +0 -0
  51. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/agent_response.py +0 -0
  52. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/keyword_actions.py +0 -0
  53. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/prompt.py +0 -0
  54. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/core/state.py +0 -0
  55. {tina_python-0.6.8 → tina_python-0.7.0}/tina/agent/multimodal_agent.py +0 -0
  56. {tina_python-0.6.8 → tina_python-0.7.0}/tina/core/__init__.py +0 -0
  57. {tina_python-0.6.8 → tina_python-0.7.0}/tina/core/logger.py +0 -0
  58. {tina_python-0.6.8 → tina_python-0.7.0}/tina/mcp/__init__.py +0 -0
  59. {tina_python-0.6.8 → tina_python-0.7.0}/tina/py.typed +0 -0
  60. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/agent_worker.py +0 -0
  61. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/doc_parser.py +0 -0
  62. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/env_reader.py +0 -0
  63. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/run_agent_in_cli.py +0 -0
  64. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/system_tools.py +0 -0
  65. {tina_python-0.6.8 → tina_python-0.7.0}/tina/utils/type_mapper.py +0 -0
  66. {tina_python-0.6.8 → tina_python-0.7.0}/tina_python.egg-info/dependency_links.txt +0 -0
  67. {tina_python-0.6.8 → tina_python-0.7.0}/tina_python.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tina-python
3
- Version: 0.6.8
3
+ Version: 0.7.0
4
4
  Summary: tina is in your computer!
5
5
  Home-page: https://gitee.com/wang-churi/tina
6
6
  Author: 王出日
@@ -18,7 +18,9 @@ License-File: LICENSE
18
18
  Requires-Dist: httpx
19
19
  Requires-Dist: python-dotenv
20
20
  Provides-Extra: tui
21
- Requires-Dist: textual>=0.80; extra == "tui"
21
+ Requires-Dist: tina-tui>=0.1.0; extra == "tui"
22
+ Provides-Extra: multi-agent
23
+ Requires-Dist: tina-multi-agent>=0.1.0; extra == "multi-agent"
22
24
  Provides-Extra: mcp
23
25
  Requires-Dist: mcp>=2.0; extra == "mcp"
24
26
  Provides-Extra: test
@@ -26,7 +28,8 @@ Requires-Dist: pytest>=9.0; extra == "test"
26
28
  Requires-Dist: pytest-asyncio>=1.4; extra == "test"
27
29
  Requires-Dist: pytest-cov>=7.0; extra == "test"
28
30
  Provides-Extra: all
29
- Requires-Dist: textual>=0.80; extra == "all"
31
+ Requires-Dist: tina-tui>=0.1.0; extra == "all"
32
+ Requires-Dist: tina-multi-agent>=0.1.0; extra == "all"
30
33
  Requires-Dist: mcp>=2.0; extra == "all"
31
34
  Dynamic: author
32
35
  Dynamic: home-page
@@ -1030,16 +1033,16 @@ print(message)
1030
1033
 
1031
1034
  ### 6.5 终端界面 TUI(可选扩展)
1032
1035
 
1033
- `tina` 自带一个基于 [Textual](https://textual.textualize.io/) 的终端界面。它的定位是**可选扩展**:方便你测试自己写的 Agent,也可以直接当作自己的 TUI 来用。核心不依赖它,需要额外安装:
1036
+ 终端界面由配套包 **`tina-tui`**(基于 [Textual](https://textual.textualize.io/))提供,作为**可选扩展**单独发行:方便你测试自己写的 Agent,也可以直接当作自己的 TUI 来用。核心不依赖它,需要额外安装:
1034
1037
 
1035
1038
  ```bash
1036
- pip install tina-python[tui]
1039
+ pip install tina-python[tui] # 等价于安装 tina-tui
1037
1040
  ```
1038
1041
 
1039
1042
  ```python
1040
1043
  from tina import Agent, Tools
1041
1044
  from tina.llm import BaseAPI
1042
- from tina.utils.tui import run_agent_in_tui
1045
+ from tina_tui import run_agent_in_tui
1043
1046
 
1044
1047
  agent = Agent(llm=BaseAPI(), tools=Tools())
1045
1048
 
@@ -1073,7 +1076,7 @@ run_agent_in_tui(agent, max_tokens=32000)
1073
1076
 
1074
1077
  快捷键:`Ctrl+↑/↓` 跳转消息、`Ctrl+Home/End` 首/末条、`Ctrl+R` 折叠思考、`Ctrl+T` 折叠工具与结果、`Ctrl+O` 查看完整工具参数、`Esc` 关闭弹窗/取消确认 / 打断本轮回复。
1075
1078
 
1076
- > 界面是可选的:数据层 `TuiMessageStore`(渲染块列表)和 `TokenCounter`(token 计数)**不依赖 textual**,可以单独导入,用来接入你自己的界面。
1079
+ > 界面是可选的:数据层 `TuiMessageStore`(渲染块列表)和 `TokenCounter`(token 计数)在 `tina_tui` 里**不依赖 textual**,可以单独导入,用来接入你自己的界面。
1077
1080
 
1078
1081
  完整说明(安装、指令、快捷键、自建界面)见 [`docs/tui.md`](./docs/tui.md)。
1079
1082
 
@@ -995,16 +995,16 @@ print(message)
995
995
 
996
996
  ### 6.5 终端界面 TUI(可选扩展)
997
997
 
998
- `tina` 自带一个基于 [Textual](https://textual.textualize.io/) 的终端界面。它的定位是**可选扩展**:方便你测试自己写的 Agent,也可以直接当作自己的 TUI 来用。核心不依赖它,需要额外安装:
998
+ 终端界面由配套包 **`tina-tui`**(基于 [Textual](https://textual.textualize.io/))提供,作为**可选扩展**单独发行:方便你测试自己写的 Agent,也可以直接当作自己的 TUI 来用。核心不依赖它,需要额外安装:
999
999
 
1000
1000
  ```bash
1001
- pip install tina-python[tui]
1001
+ pip install tina-python[tui] # 等价于安装 tina-tui
1002
1002
  ```
1003
1003
 
1004
1004
  ```python
1005
1005
  from tina import Agent, Tools
1006
1006
  from tina.llm import BaseAPI
1007
- from tina.utils.tui import run_agent_in_tui
1007
+ from tina_tui import run_agent_in_tui
1008
1008
 
1009
1009
  agent = Agent(llm=BaseAPI(), tools=Tools())
1010
1010
 
@@ -1038,7 +1038,7 @@ run_agent_in_tui(agent, max_tokens=32000)
1038
1038
 
1039
1039
  快捷键:`Ctrl+↑/↓` 跳转消息、`Ctrl+Home/End` 首/末条、`Ctrl+R` 折叠思考、`Ctrl+T` 折叠工具与结果、`Ctrl+O` 查看完整工具参数、`Esc` 关闭弹窗/取消确认 / 打断本轮回复。
1040
1040
 
1041
- > 界面是可选的:数据层 `TuiMessageStore`(渲染块列表)和 `TokenCounter`(token 计数)**不依赖 textual**,可以单独导入,用来接入你自己的界面。
1041
+ > 界面是可选的:数据层 `TuiMessageStore`(渲染块列表)和 `TokenCounter`(token 计数)在 `tina_tui` 里**不依赖 textual**,可以单独导入,用来接入你自己的界面。
1042
1042
 
1043
1043
  完整说明(安装、指令、快捷键、自建界面)见 [`docs/tui.md`](./docs/tui.md)。
1044
1044
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "tina-python"
7
- version = "0.6.8"
7
+ version = "0.7.0"
8
8
  description = "tina is in your computer!"
9
9
  readme = { file = "README.md", content-type = "text/markdown" }
10
10
  requires-python = ">=3.10"
@@ -29,7 +29,10 @@ GitHub = "https://github.com/XIMOCY/tina.git"
29
29
 
30
30
  [project.optional-dependencies]
31
31
  tui = [
32
- "textual>=0.80",
32
+ "tina-tui>=0.1.0",
33
+ ]
34
+ multi-agent = [
35
+ "tina-multi-agent>=0.1.0",
33
36
  ]
34
37
  mcp = [
35
38
  "mcp>=2.0",
@@ -40,12 +43,14 @@ test = [
40
43
  "pytest-cov>=7.0",
41
44
  ]
42
45
  all = [
43
- "textual>=0.80",
46
+ "tina-tui>=0.1.0",
47
+ "tina-multi-agent>=0.1.0",
44
48
  "mcp>=2.0",
45
49
  ]
46
50
 
47
51
  [tool.pytest.ini_options]
48
52
  testpaths = ["test"]
53
+ pythonpath = ["packages/tina-tui", "packages/tina-multi-agent"]
49
54
  asyncio_mode = "auto"
50
55
  markers = [
51
56
  "slow: 需要 LLM 实际调用的测试(默认跳过)",
@@ -2,10 +2,15 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="tina-python",
5
- version="0.6.8",
5
+ version="0.7.0",
6
6
  packages=find_packages(),
7
7
  package_data={"tina": ["py.typed"]},
8
8
  install_requires=["httpx", "python-dotenv"],
9
+ extras_require={
10
+ "tui": ["tina-tui>=0.1.0"],
11
+ "multi-agent": ["tina-multi-agent>=0.1.0"],
12
+ "mcp": ["mcp>=2.0"],
13
+ },
9
14
  description="tina is in your computer!",
10
15
  long_description=open("README.md", encoding="utf-8").read(),
11
16
  long_description_content_type="text/markdown",
@@ -8,6 +8,7 @@ from tina.agent import (
8
8
  AgentState,
9
9
  AgentEvents,
10
10
  )
11
+ from tina.llm.files_api import FilesAPI
11
12
 
12
13
  __all__ = [
13
14
  "Agent",
@@ -18,5 +19,6 @@ __all__ = [
18
19
  "MultimodalAgent",
19
20
  "AgentState",
20
21
  "AgentEvents",
22
+ "FilesAPI",
21
23
  ]
22
- __version__ = "0.6.8"
24
+ __version__ = "0.7.0"
@@ -339,6 +339,19 @@ class Agent:
339
339
  """
340
340
  self.events.add_on_turn_end_handler(func)
341
341
 
342
+ def before_llm_call(self):
343
+ """
344
+ 在每一次 LLM 调用之前触发(含工具循环的每一轮),不需要参数。
345
+ 这是安全的上下文注入点。
346
+ """
347
+ return self.events.before_llm_call()
348
+
349
+ def add_before_llm_call_handler(self, func: callable | list[callable]):
350
+ """
351
+ 在每一次 LLM 调用之前触发(含工具循环的每一轮),不需要参数。
352
+ """
353
+ self.events.add_before_llm_call_handler(func)
354
+
342
355
  def _mcp_to_tools(self, MCP):
343
356
  """如果传入了MCP,则将MCP的工具集加入到当前的工具集中"""
344
357
  try:
@@ -6,6 +6,7 @@ from .context_manager import BaseContextManager
6
6
  from typing import Generator
7
7
  from .state import AgentState
8
8
  from .events import AgentEvents
9
+ from ...utils.usage import Usage
9
10
 
10
11
 
11
12
  class BaseAgentRuntime:
@@ -103,28 +104,57 @@ class BaseAgentRuntime:
103
104
  return ""
104
105
  return self.keyword_actions.flush_visible()
105
106
 
106
- def _emit_visible_content_chunk(self, content: str = "", usage=None) -> dict | None:
107
+ def _emit_visible_content_chunk(
108
+ self, content: str = "", usage=None, source: dict = None
109
+ ) -> dict | None:
107
110
  """
108
111
  组装并触发可见 content chunk。content 应为已过滤文本。
109
- 无可见内容且无 usage 时不发射。
112
+
113
+ source 存在时就地复用它(保留 usage / timing 等字段),只覆盖 content;
114
+ 无可见内容、无 usage、无 timing 时不发射。
110
115
  """
111
- if not content and usage is None:
116
+ has_timing = bool(source) and source.get("timing") is not None
117
+ if not content and usage is None and not has_timing:
112
118
  return None
113
- chunk = {"role": "assistant", "content": content or ""}
114
- if usage is not None:
115
- chunk["usage"] = usage
119
+ if source is not None:
120
+ chunk = source
121
+ chunk["role"] = chunk.get("role") or "assistant"
122
+ chunk["content"] = content or ""
123
+ else:
124
+ chunk = {"role": "assistant", "content": content or ""}
125
+ if usage is not None:
126
+ chunk["usage"] = usage
116
127
  self.events.trigger_on_stream_chunk(chunk)
117
128
  return chunk
118
129
 
119
- async def _aemit_visible_content_chunk(self, content: str = "", usage=None) -> dict | None:
120
- if not content and usage is None:
130
+ async def _aemit_visible_content_chunk(
131
+ self, content: str = "", usage=None, source: dict = None
132
+ ) -> dict | None:
133
+ has_timing = bool(source) and source.get("timing") is not None
134
+ if not content and usage is None and not has_timing:
121
135
  return None
122
- chunk = {"role": "assistant", "content": content or ""}
123
- if usage is not None:
124
- chunk["usage"] = usage
136
+ if source is not None:
137
+ chunk = source
138
+ chunk["role"] = chunk.get("role") or "assistant"
139
+ chunk["content"] = content or ""
140
+ else:
141
+ chunk = {"role": "assistant", "content": content or ""}
142
+ if usage is not None:
143
+ chunk["usage"] = usage
125
144
  await self.events.atrigger_on_stream_chunk(chunk)
126
145
  return chunk
127
146
 
147
+ def _record_usage(self, usage_raw) -> None:
148
+ """把一次 LLM 返回的 usage 归一化后通过 on_usage 事件广播"""
149
+ usage = Usage.from_raw(usage_raw)
150
+ if usage is not None:
151
+ self.events.trigger_on_usage(usage)
152
+
153
+ async def _arecord_usage(self, usage_raw) -> None:
154
+ usage = Usage.from_raw(usage_raw)
155
+ if usage is not None:
156
+ await self.events.atrigger_on_usage(usage)
157
+
128
158
  def _check_keyword_actions(self, text: str):
129
159
  """assistant 正文凑齐后触发(含 tool_calls 前的中间段);匹配用原文。"""
130
160
  if self.keyword_actions is not None and text:
@@ -209,6 +239,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
209
239
  super().run_prediction_no_stream(instruction, temperature, top_p, top_k, min_p)
210
240
  counter = 0
211
241
  while counter < self.max_tool_loop:
242
+ self.events.trigger_before_llm_call()
212
243
  self.state = AgentState.THINKING
213
244
  llm_response = self.llm.predict_no_stream(
214
245
  messages=self.context_manager.get_messages(),
@@ -216,6 +247,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
216
247
  tools=self.tools.get_tools_for_llm(),
217
248
  top_p=top_p,
218
249
  )
250
+ self._record_usage(llm_response.get("usage"))
219
251
 
220
252
  if "tool_calls" in llm_response:
221
253
  self.state = AgentState.TOOL_CALLING
@@ -253,6 +285,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
253
285
  super().run_prediction_stream(instruction, temperature, top_p, top_k, min_p)
254
286
  counter = 0
255
287
  while counter < self.max_tool_loop:
288
+ self.events.trigger_before_llm_call()
256
289
  tool_called = False
257
290
  llm_response = self.llm.predict_stream(
258
291
  messages=self.context_manager.get_messages(),
@@ -270,6 +303,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
270
303
  for chunk in llm_response:
271
304
  if chunk.get("content") is None:
272
305
  chunk["content"] = ""
306
+ self._record_usage(chunk.get("usage"))
273
307
 
274
308
  if "tool_name" in chunk or "tool_arguments" in chunk:
275
309
  self.state = AgentState.TOOL_CALLING
@@ -315,15 +349,16 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
315
349
  else:
316
350
  content = chunk.get("content", "")
317
351
  usage = chunk.get("usage")
318
- if content or usage is not None:
319
- if content:
320
- content_parts.append(content)
321
- visible = (
322
- self._filter_stream_content(content) if content else ""
323
- )
324
- emitted = self._emit_visible_content_chunk(visible, usage)
325
- if emitted is not None:
326
- yield emitted
352
+ if content:
353
+ content_parts.append(content)
354
+ visible = (
355
+ self._filter_stream_content(content) if content else ""
356
+ )
357
+ emitted = self._emit_visible_content_chunk(
358
+ visible, usage, source=chunk
359
+ )
360
+ if emitted is not None:
361
+ yield emitted
327
362
 
328
363
  whole_content = "".join(content_parts)
329
364
  if whole_content:
@@ -357,6 +392,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
357
392
  )
358
393
  counter = 0
359
394
  while counter < self.max_tool_loop:
395
+ await self.events.atrigger_before_llm_call()
360
396
  self.state = AgentState.THINKING
361
397
  llm_result = await self.llm.apredict(
362
398
  messages=self.context_manager.get_messages(),
@@ -364,6 +400,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
364
400
  tools=self.tools.get_tools_for_llm(),
365
401
  top_p=top_p,
366
402
  )
403
+ await self._arecord_usage(llm_result.get("usage"))
367
404
  if "tool_calls" in llm_result:
368
405
  self.state = AgentState.TOOL_CALLING
369
406
  _content = llm_result.get("content") or ""
@@ -405,6 +442,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
405
442
  )
406
443
  counter = 0
407
444
  while counter < self.max_tool_loop:
445
+ await self.events.atrigger_before_llm_call()
408
446
  tool_called = False
409
447
  llm_response = await self.llm.apredict(
410
448
  messages=self.context_manager.get_messages(),
@@ -421,6 +459,7 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
421
459
  async for chunk in llm_response:
422
460
  if chunk.get("content") is None:
423
461
  chunk["content"] = ""
462
+ await self._arecord_usage(chunk.get("usage"))
424
463
 
425
464
  if "tool_name" in chunk or "tool_arguments" in chunk:
426
465
  await self.events.atrigger_on_stream_chunk(chunk)
@@ -468,17 +507,16 @@ class ToolCallingAgentRuntime(BaseAgentRuntime):
468
507
  else:
469
508
  content = chunk.get("content", "")
470
509
  usage = chunk.get("usage")
471
- if content or usage is not None:
472
- if content:
473
- content_parts.append(content)
474
- visible = (
475
- self._filter_stream_content(content) if content else ""
476
- )
477
- emitted = await self._aemit_visible_content_chunk(
478
- visible, usage
479
- )
480
- if emitted is not None:
481
- yield emitted
510
+ if content:
511
+ content_parts.append(content)
512
+ visible = (
513
+ self._filter_stream_content(content) if content else ""
514
+ )
515
+ emitted = await self._aemit_visible_content_chunk(
516
+ visible, usage, source=chunk
517
+ )
518
+ if emitted is not None:
519
+ yield emitted
482
520
 
483
521
  whole_content = "".join(content_parts)
484
522
  if whole_content:
@@ -542,6 +580,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
542
580
  )
543
581
  counter = 0
544
582
  while counter < self.max_tool_loop:
583
+ self.events.trigger_before_llm_call()
545
584
  self.state = AgentState.THINKING
546
585
  llm_response = self.llm.predict_no_stream(
547
586
  messages=self.context_manager.get_messages(),
@@ -551,6 +590,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
551
590
  top_k=top_k,
552
591
  min_p=min_p,
553
592
  )
593
+ self._record_usage(llm_response.get("usage"))
554
594
 
555
595
  if "tool_calls" in llm_response:
556
596
  self.state = AgentState.TOOL_CALLING
@@ -599,6 +639,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
599
639
  )
600
640
  counter = 0
601
641
  while counter < self.max_tool_loop:
642
+ self.events.trigger_before_llm_call()
602
643
 
603
644
  tool_called = False
604
645
  llm_response = self.llm.predict_stream(
@@ -617,6 +658,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
617
658
  for chunk in llm_response:
618
659
  if chunk.get("content") is None:
619
660
  chunk["content"] = ""
661
+ self._record_usage(chunk.get("usage"))
620
662
 
621
663
  if "tool_name" in chunk or "tool_arguments" in chunk:
622
664
  self.events.trigger_on_stream_chunk(chunk)
@@ -665,15 +707,16 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
665
707
  else:
666
708
  content = chunk.get("content", "")
667
709
  usage = chunk.get("usage")
668
- if content or usage is not None:
669
- if content:
670
- content_parts.append(content)
671
- visible = (
672
- self._filter_stream_content(content) if content else ""
673
- )
674
- emitted = self._emit_visible_content_chunk(visible, usage)
675
- if emitted is not None:
676
- yield emitted
710
+ if content:
711
+ content_parts.append(content)
712
+ visible = (
713
+ self._filter_stream_content(content) if content else ""
714
+ )
715
+ emitted = self._emit_visible_content_chunk(
716
+ visible, usage, source=chunk
717
+ )
718
+ if emitted is not None:
719
+ yield emitted
677
720
 
678
721
  whole_content = "".join(content_parts)
679
722
  if whole_content:
@@ -719,6 +762,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
719
762
  )
720
763
  counter = 0
721
764
  while counter < self.max_tool_loop:
765
+ await self.events.atrigger_before_llm_call()
722
766
  self.state = AgentState.THINKING
723
767
  llm_result = await self.llm.apredict(
724
768
  messages=self.context_manager.get_messages(),
@@ -728,6 +772,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
728
772
  top_k=top_k,
729
773
  min_p=min_p,
730
774
  )
775
+ await self._arecord_usage(llm_result.get("usage"))
731
776
  if "tool_calls" in llm_result:
732
777
  self.state = AgentState.TOOL_CALLING
733
778
  _content = llm_result.get("content") or ""
@@ -777,6 +822,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
777
822
  )
778
823
  counter = 0
779
824
  while counter < self.max_tool_loop:
825
+ await self.events.atrigger_before_llm_call()
780
826
  tool_called = False
781
827
  llm_response = await self.llm.apredict(
782
828
  messages=self.context_manager.get_messages(),
@@ -793,6 +839,7 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
793
839
  async for chunk in llm_response:
794
840
  if chunk.get("content") is None:
795
841
  chunk["content"] = ""
842
+ await self._arecord_usage(chunk.get("usage"))
796
843
 
797
844
  if "tool_name" in chunk or "tool_arguments" in chunk:
798
845
  await self.events.atrigger_on_stream_chunk(chunk)
@@ -840,17 +887,16 @@ class ToolCallingMutilemodalAgentRuntime(BaseAgentRuntime):
840
887
  else:
841
888
  content = chunk.get("content", "")
842
889
  usage = chunk.get("usage")
843
- if content or usage is not None:
844
- if content:
845
- content_parts.append(content)
846
- visible = (
847
- self._filter_stream_content(content) if content else ""
848
- )
849
- emitted = await self._aemit_visible_content_chunk(
850
- visible, usage
851
- )
852
- if emitted is not None:
853
- yield emitted
890
+ if content:
891
+ content_parts.append(content)
892
+ visible = (
893
+ self._filter_stream_content(content) if content else ""
894
+ )
895
+ emitted = await self._aemit_visible_content_chunk(
896
+ visible, usage, source=chunk
897
+ )
898
+ if emitted is not None:
899
+ yield emitted
854
900
 
855
901
  whole_content = "".join(content_parts)
856
902
  if whole_content:
@@ -323,6 +323,7 @@ class MultimodalContextManager(ContextManager):
323
323
  image: str | list[str] = None,
324
324
  audio: str | list[str] = None,
325
325
  url: str | list[str] = None,
326
+ file_id: str | list[str] = None,
326
327
  ) -> list[dict[str, Any]]:
327
328
  user_content = build_multimodal_message(
328
329
  input_text=instruction,
@@ -330,6 +331,7 @@ class MultimodalContextManager(ContextManager):
330
331
  input_audio=audio,
331
332
  input_url=url,
332
333
  role="user",
334
+ input_file_id=file_id,
333
335
  )
334
336
  if user_content is None:
335
337
  return self.messages
@@ -395,12 +397,13 @@ class MultimodalContextManager(ContextManager):
395
397
  pending_multimodal.append((item["tool_name"], multimodal_result))
396
398
 
397
399
  # 全部 tool 消息写入完毕后,再追加多模态 user 消息
398
- for tool_name, (images, audios, urls) in pending_multimodal:
400
+ for tool_name, (images, audios, urls, file_ids) in pending_multimodal:
399
401
  self.add_user_message(
400
402
  instruction=f"工具{tool_name}的结果",
401
403
  image=images,
402
404
  audio=audios,
403
405
  url=urls,
406
+ file_id=file_ids,
404
407
  )
405
408
 
406
409
  self.limit_messages()
@@ -429,16 +432,18 @@ class MultimodalContextManager(ContextManager):
429
432
  if not values:
430
433
  return None
431
434
 
432
- images, audios, urls = [], [], []
435
+ images, audios, urls, file_ids = [], [], [], []
433
436
  if multimodal_type == "image":
434
437
  images = values
435
438
  elif multimodal_type == "audio":
436
439
  audios = values
437
440
  elif multimodal_type == "url":
438
441
  urls = values
442
+ elif multimodal_type == "file_id":
443
+ file_ids = values
439
444
  else:
440
445
  return None
441
- return images, audios, urls
446
+ return images, audios, urls, file_ids
442
447
 
443
448
  def add_tool_call_result(
444
449
  self, tool_result: str | None, tool_call_id: str, tool_call: dict[str, Any]