mmsp 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. mmsp-0.5.0/PKG-INFO +356 -0
  2. mmsp-0.5.0/README.md +330 -0
  3. mmsp-0.5.0/mmsp/__init__.py +45 -0
  4. mmsp-0.5.0/mmsp/abort_signal.py +135 -0
  5. mmsp-0.5.0/mmsp/ant_messages/__init__.py +18 -0
  6. mmsp-0.5.0/mmsp/ant_messages/client.py +395 -0
  7. mmsp-0.5.0/mmsp/anthropic_official/__init__.py +18 -0
  8. mmsp-0.5.0/mmsp/anthropic_official/client.py +491 -0
  9. mmsp-0.5.0/mmsp/auto_client.py +281 -0
  10. mmsp-0.5.0/mmsp/base_client.py +378 -0
  11. mmsp-0.5.0/mmsp/deepseek_official/__init__.py +18 -0
  12. mmsp-0.5.0/mmsp/deepseek_official/client.py +384 -0
  13. mmsp-0.5.0/mmsp/errors.py +123 -0
  14. mmsp-0.5.0/mmsp/gemini_generate_content/__init__.py +18 -0
  15. mmsp-0.5.0/mmsp/gemini_generate_content/client.py +668 -0
  16. mmsp-0.5.0/mmsp/gemini_official/__init__.py +18 -0
  17. mmsp-0.5.0/mmsp/gemini_official/client.py +674 -0
  18. mmsp-0.5.0/mmsp/integration/__init__.py +14 -0
  19. mmsp-0.5.0/mmsp/integration/playground.py +3306 -0
  20. mmsp-0.5.0/mmsp/integration/tracer.py +1812 -0
  21. mmsp-0.5.0/mmsp/legacy.py +78 -0
  22. mmsp-0.5.0/mmsp/minimax_official/__init__.py +18 -0
  23. mmsp-0.5.0/mmsp/minimax_official/client.py +323 -0
  24. mmsp-0.5.0/mmsp/moonshot_official/__init__.py +18 -0
  25. mmsp-0.5.0/mmsp/moonshot_official/client.py +412 -0
  26. mmsp-0.5.0/mmsp/openai_chat/__init__.py +18 -0
  27. mmsp-0.5.0/mmsp/openai_chat/client.py +379 -0
  28. mmsp-0.5.0/mmsp/openai_chat_vllm_adapter/__init__.py +4 -0
  29. mmsp-0.5.0/mmsp/openai_chat_vllm_adapter/client.py +122 -0
  30. mmsp-0.5.0/mmsp/openai_embedding/__init__.py +18 -0
  31. mmsp-0.5.0/mmsp/openai_embedding/client.py +102 -0
  32. mmsp-0.5.0/mmsp/openai_official/__init__.py +18 -0
  33. mmsp-0.5.0/mmsp/openai_official/client.py +426 -0
  34. mmsp-0.5.0/mmsp/openai_responses/__init__.py +18 -0
  35. mmsp-0.5.0/mmsp/openai_responses/client.py +411 -0
  36. mmsp-0.5.0/mmsp/registry.py +846 -0
  37. mmsp-0.5.0/mmsp/stream_items.py +199 -0
  38. mmsp-0.5.0/mmsp/types.py +258 -0
  39. mmsp-0.5.0/mmsp/utils.py +305 -0
  40. mmsp-0.5.0/mmsp/zai_official/__init__.py +18 -0
  41. mmsp-0.5.0/mmsp/zai_official/client.py +406 -0
  42. mmsp-0.5.0/pyproject.toml +49 -0
mmsp-0.5.0/PKG-INFO ADDED
@@ -0,0 +1,356 @@
1
+ Metadata-Version: 2.4
2
+ Name: mmsp
3
+ Version: 0.5.0
4
+ Summary: MMSP, the Model Message Stream Protocol: one message format and one streaming grammar for every model provider, in Python and TypeScript.
5
+ Keywords: mmsp,llm,stream,gemini,claude,gpt
6
+ Author: PrismShadow
7
+ License-Expression: Apache-2.0
8
+ Requires-Dist: google-genai>=2.23.0
9
+ Requires-Dist: anthropic[bedrock]>=0.87.0
10
+ Requires-Dist: flask>=3.0.0
11
+ Requires-Dist: openai>=2.30.0
12
+ Requires-Dist: httpx>=0.27.0
13
+ Requires-Dist: httpx[socks] ; extra == 'dev'
14
+ Requires-Dist: pytest>=8.4.2 ; extra == 'dev'
15
+ Requires-Dist: pytest-asyncio>=0.23.0 ; extra == 'dev'
16
+ Requires-Dist: pytest-xdist>=3.6.0 ; extra == 'dev'
17
+ Requires-Dist: pytest-rerunfailures>=15.0 ; extra == 'dev'
18
+ Requires-Dist: ruff>=0.14.3 ; extra == 'dev'
19
+ Requires-Dist: pillow>=10.0.0 ; extra == 'dev'
20
+ Requires-Python: >=3.11
21
+ Project-URL: Homepage, https://github.com/Prism-Shadow/model-message-stream-protocol
22
+ Project-URL: Repository, https://github.com/Prism-Shadow/model-message-stream-protocol
23
+ Project-URL: Issues, https://github.com/Prism-Shadow/model-message-stream-protocol/issues
24
+ Provides-Extra: dev
25
+ Description-Content-Type: text/markdown
26
+
27
+ # MMSP Python Implementation
28
+
29
+ This document demonstrates how to use `AutoLLMClient` for unified LLM interactions in MMSP.
30
+
31
+ ## Building
32
+
33
+ ```bash
34
+ make install # Install dependencies
35
+ make build # Build Python package
36
+ make lint # Run ruff linter
37
+ make test # Run tests
38
+ ```
39
+
40
+ ## AutoLLMClient Overview
41
+
42
+ `AutoLLMClient` is a stateful client that automatically routes requests to the appropriate model-specific implementation. It maintains conversation history and provides a unified interface for different LLM providers.
43
+
44
+ ### Initialization
45
+
46
+ Create a client by specifying the model name:
47
+
48
+ ```python
49
+ from mmsp import AutoLLMClient
50
+
51
+ # The official OpenAI client, named by the model id's family
52
+ client = AutoLLMClient(model="gpt-5.5")
53
+
54
+ # The same, spelled out, with the key given in code
55
+ client = AutoLLMClient(model="gpt-5.5", client_type="openai-official", api_key="your-openai-api-key")
56
+
57
+ # A compatible client, for any endpoint that serves OpenAI Chat Completions
58
+ client = AutoLLMClient(
59
+ model="custom-model", client_type="openai-chat", base_url="http://127.0.0.1:8000/v1/", api_key="none"
60
+ )
61
+
62
+ # Gemini on Google Vertex AI: the service-account JSON key is the API key
63
+ client = AutoLLMClient(model="gemini-3.8-flash", api_key=open("service-account.json").read())
64
+ ```
65
+
66
+ `client_type` names one of the official clients (`openai-official`, `anthropic-official`, `gemini-official`, `zai-official`, `moonshot-official`, `deepseek-official`, `minimax-official`) or one of the compatible clients (`openai-responses`, `openai-chat`, `openai-chat-vllm-adapter`, `openai-embedding`, `ant-messages`, `gemini-generate-content`). It may be omitted for a model id that begins with a known family (`gpt-`, `text-embedding-`, `claude-`, `gemini-`, `glm-`, `kimi-`, `deepseek-`, `minimax-`), which names its official client; any other id raises and asks for one.
67
+
68
+ A Vertex AI service-account key is served through generateContent, because Vertex AI's Interactions endpoint serves none of the Gemini models; any other Gemini key uses the Interactions API. `client_type="gemini-generate-content"` names generateContent explicitly, for gateways that proxy it.
69
+
70
+ ## Core Methods
71
+
72
+ ### streaming_response
73
+
74
+ Stateless method that requires passing the full message history on each call:
75
+
76
+ ```python
77
+ import asyncio
78
+ from mmsp import AutoLLMClient
79
+
80
+
81
+ async def main():
82
+ client = AutoLLMClient(model="gpt-5.5")
83
+
84
+ async for event in client.streaming_response(
85
+ messages=[{"role": "user", "content_items": [{"type": "text.done", "text": "Hello!"}]}], config={}
86
+ ):
87
+ print(event)
88
+
89
+
90
+ asyncio.run(main())
91
+ ```
92
+
93
+ Both streaming methods yield `delta` events, each carrying exactly one content item, then exactly one `stop` event, always last, carrying `usage_metadata` and `finish_reason`. Each item streams as one or more `.delta` fragments (`text.delta`, `tool_call.delta`, …) followed by its complete `.done` item (`text.done`, `tool_call.done`, …); items never interleave.
94
+
95
+ ### streaming_response_stateful
96
+
97
+ Stateful method that maintains conversation history internally:
98
+
99
+ ```python
100
+ import asyncio
101
+ from mmsp import AutoLLMClient
102
+
103
+
104
+ async def main():
105
+ client = AutoLLMClient(model="gpt-5.5")
106
+
107
+ # First message
108
+ async for event in client.streaming_response_stateful(
109
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "My name is Alice"}]}, config={}
110
+ ):
111
+ print(event)
112
+
113
+ # Second message - history is maintained automatically
114
+ async for event in client.streaming_response_stateful(
115
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "What's my name?"}]}, config={}
116
+ ):
117
+ print(event)
118
+
119
+
120
+ asyncio.run(main())
121
+ ```
122
+
123
+ ### get_history
124
+
125
+ Retrieve the conversation history:
126
+
127
+ ```python
128
+ # Get all messages in the conversation
129
+ history = client.get_history()
130
+ print(f"Total messages: {len(history)}")
131
+
132
+ for msg in history:
133
+ print(f"Role: {msg['role']}")
134
+ print(f"Content: {msg['content_items']}")
135
+ ```
136
+
137
+ ### clear_history
138
+
139
+ Clear the conversation history:
140
+
141
+ ```python
142
+ # Clear all conversation history
143
+ client.clear_history()
144
+
145
+ # Verify history is empty
146
+ assert len(client.get_history()) == 0
147
+ ```
148
+
149
+ ### set_history
150
+
151
+ Replace the conversation history with a copy of the provided list:
152
+
153
+ ```python
154
+ # Save current history
155
+ saved_history = client.get_history()
156
+
157
+ # ... do other things, then restore
158
+ client.set_history(saved_history)
159
+
160
+ # Verify history was replaced
161
+ assert len(client.get_history()) == len(saved_history)
162
+ ```
163
+
164
+ ## Tool Calling
165
+
166
+ When using tools, you must handle `tool_call_id` correctly:
167
+
168
+ ```python
169
+ import asyncio
170
+ import json
171
+ from mmsp import AutoLLMClient
172
+
173
+
174
+ def get_weather(location: str) -> str:
175
+ """Mock function to get weather."""
176
+ return f"Temperature in {location}: 22°C"
177
+
178
+
179
+ async def main():
180
+ # Define tool
181
+ weather_function = {
182
+ "name": "get_weather",
183
+ "description": "Gets the current weather for a given location.",
184
+ "parameters": {
185
+ "type": "object",
186
+ "properties": {"location": {"type": "string", "description": "The city name"}},
187
+ "required": ["location"],
188
+ },
189
+ }
190
+
191
+ client = AutoLLMClient(model="gpt-5.5")
192
+ config = {"tools": [weather_function]}
193
+
194
+ # User asks about weather
195
+ events = []
196
+ async for event in client.streaming_response_stateful(
197
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "What's the weather in London?"}]},
198
+ config=config,
199
+ ):
200
+ events.append(event)
201
+
202
+ # Read the complete call from its tool_call.done item; tool_call.delta items are fragments
203
+ tool_call = None
204
+ for event in events:
205
+ for item in event["content_items"]:
206
+ if item["type"] == "tool_call.done":
207
+ tool_call = item
208
+ break
209
+
210
+ if tool_call:
211
+ break
212
+
213
+ # Execute function and send result back with tool_call_id
214
+ if tool_call:
215
+ result = get_weather(**tool_call["arguments"])
216
+
217
+ # IMPORTANT: Include tool_call_id in the tool response
218
+ async for event in client.streaming_response_stateful(
219
+ message={
220
+ "role": "user",
221
+ "content_items": [
222
+ {
223
+ "type": "tool_result.done",
224
+ "text": result,
225
+ "tool_call_id": tool_call["tool_call_id"], # Required for tool responses
226
+ }
227
+ ],
228
+ },
229
+ config=config,
230
+ ):
231
+ print(event)
232
+
233
+
234
+ asyncio.run(main())
235
+ ```
236
+
237
+ ## Message Format
238
+
239
+ ### UniMessage Structure
240
+
241
+ ```python
242
+ {
243
+ "role": "user" | "assistant",
244
+ "content_items": [
245
+ {"type": "text.done", "text": "Hello"},
246
+ {"type": "image_url.done", "image_url": "https://..."},
247
+ {
248
+ "type": "tool_call.done",
249
+ "name": "get_weather",
250
+ "arguments": {"location": "London"},
251
+ "tool_call_id": "call_abc123",
252
+ },
253
+ ],
254
+ }
255
+ ```
256
+
257
+ Messages hold complete items only, typed with a `.done` suffix. Item types without the suffix, saved before 0.5.0, are still accepted with a deprecation warning until 0.6.0; `normalize_legacy_messages(messages)` converts stored messages.
258
+
259
+ ### Tool Response with tool_call_id
260
+
261
+ When responding to a tool call, include the `tool_call_id` in the result content item:
262
+
263
+ ```python
264
+ {
265
+ "role": "user",
266
+ "content_items": [
267
+ {
268
+ "type": "tool_result.done",
269
+ "text": "London is 22°C today.",
270
+ "tool_call_id": "call_abc123", # From the tool_call.done item
271
+ }
272
+ ],
273
+ }
274
+ ```
275
+
276
+ ## Configuration Options
277
+
278
+ ```python
279
+ from mmsp import PromptCaching, ThinkingLevel
280
+
281
+ config = {
282
+ "max_tokens": 500,
283
+ "temperature": 1.0,
284
+ "tools": [tool_definition],
285
+ "thinking_summary": True,
286
+ "thinking_level": ThinkingLevel.HIGH,
287
+ "tool_choice": "auto", # "auto", "required", "none", or ["tool_name"]
288
+ "system_prompt": "You are a helpful assistant",
289
+ "prompt_caching": PromptCaching.ENABLE,
290
+ "trace_id": "agent1/conversation_001", # Optional: save conversation trace
291
+ }
292
+ ```
293
+
294
+ ## Conversation Tracing
295
+
296
+ MMSP provides a built-in `Tracer` to save and browse conversation history. When you specify a `trace_id` in the config, conversations are automatically saved to both JSON and TXT formats.
297
+
298
+ ### Basic Usage
299
+
300
+ ```python
301
+ from mmsp import AutoLLMClient
302
+
303
+ client = AutoLLMClient(model="gpt-5.5")
304
+
305
+ # Add trace_id to config
306
+ config = {"trace_id": "agent1/conversation_001"}
307
+
308
+ async for event in client.streaming_response_stateful(
309
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "Hello"}]}, config=config
310
+ ):
311
+ pass # Conversation is automatically saved
312
+ ```
313
+
314
+ The default cache directory is `cache`, you can change it by setting `MMSP_CACHE_DIR` environment variable.
315
+
316
+ This creates two files in the `cache` directory:
317
+ - `cache/agent1/conversation_001.json` - Structured data with full history and config
318
+ - `cache/agent1/conversation_001.txt` - Human-readable conversation format
319
+
320
+ ### Browsing Traces with Web Interface
321
+
322
+ Start a web server to browse and view saved conversations:
323
+
324
+ ```python
325
+ from mmsp.integration.tracer import Tracer
326
+
327
+ # Start web server
328
+ Tracer("path/to/cache").start_web_server(host="127.0.0.1", port=25750)
329
+ ```
330
+
331
+ Or use the CLI:
332
+
333
+ ```bash
334
+ python -m mmsp.integration.tracer --cache_dir ./cache --host 127.0.0.1 --port 25750
335
+ ```
336
+
337
+ Then visit `http://127.0.0.1:25750` in your browser to browse saved conversations.
338
+
339
+ ### Test with Playground
340
+
341
+ Start a web server to test with the playground:
342
+
343
+ ```python
344
+ from mmsp.integration.playground import start_playground_server
345
+
346
+ start_playground_server()
347
+ ```
348
+
349
+ Or use the CLI:
350
+
351
+ ```bash
352
+ python -m mmsp.integration.playground --host 127.0.0.1 --port 25751
353
+ ```
354
+
355
+ Then visit `http://127.0.0.1:25751` in your browser to test with the playground.
356
+ The integrated tracer is available at `http://127.0.0.1:25751/tracer/`.
mmsp-0.5.0/README.md ADDED
@@ -0,0 +1,330 @@
1
+ # MMSP Python Implementation
2
+
3
+ This document demonstrates how to use `AutoLLMClient` for unified LLM interactions in MMSP.
4
+
5
+ ## Building
6
+
7
+ ```bash
8
+ make install # Install dependencies
9
+ make build # Build Python package
10
+ make lint # Run ruff linter
11
+ make test # Run tests
12
+ ```
13
+
14
+ ## AutoLLMClient Overview
15
+
16
+ `AutoLLMClient` is a stateful client that automatically routes requests to the appropriate model-specific implementation. It maintains conversation history and provides a unified interface for different LLM providers.
17
+
18
+ ### Initialization
19
+
20
+ Create a client by specifying the model name:
21
+
22
+ ```python
23
+ from mmsp import AutoLLMClient
24
+
25
+ # The official OpenAI client, named by the model id's family
26
+ client = AutoLLMClient(model="gpt-5.5")
27
+
28
+ # The same, spelled out, with the key given in code
29
+ client = AutoLLMClient(model="gpt-5.5", client_type="openai-official", api_key="your-openai-api-key")
30
+
31
+ # A compatible client, for any endpoint that serves OpenAI Chat Completions
32
+ client = AutoLLMClient(
33
+ model="custom-model", client_type="openai-chat", base_url="http://127.0.0.1:8000/v1/", api_key="none"
34
+ )
35
+
36
+ # Gemini on Google Vertex AI: the service-account JSON key is the API key
37
+ client = AutoLLMClient(model="gemini-3.8-flash", api_key=open("service-account.json").read())
38
+ ```
39
+
40
+ `client_type` names one of the official clients (`openai-official`, `anthropic-official`, `gemini-official`, `zai-official`, `moonshot-official`, `deepseek-official`, `minimax-official`) or one of the compatible clients (`openai-responses`, `openai-chat`, `openai-chat-vllm-adapter`, `openai-embedding`, `ant-messages`, `gemini-generate-content`). It may be omitted for a model id that begins with a known family (`gpt-`, `text-embedding-`, `claude-`, `gemini-`, `glm-`, `kimi-`, `deepseek-`, `minimax-`), which names its official client; any other id raises and asks for one.
41
+
42
+ A Vertex AI service-account key is served through generateContent, because Vertex AI's Interactions endpoint serves none of the Gemini models; any other Gemini key uses the Interactions API. `client_type="gemini-generate-content"` names generateContent explicitly, for gateways that proxy it.
43
+
44
+ ## Core Methods
45
+
46
+ ### streaming_response
47
+
48
+ Stateless method that requires passing the full message history on each call:
49
+
50
+ ```python
51
+ import asyncio
52
+ from mmsp import AutoLLMClient
53
+
54
+
55
+ async def main():
56
+ client = AutoLLMClient(model="gpt-5.5")
57
+
58
+ async for event in client.streaming_response(
59
+ messages=[{"role": "user", "content_items": [{"type": "text.done", "text": "Hello!"}]}], config={}
60
+ ):
61
+ print(event)
62
+
63
+
64
+ asyncio.run(main())
65
+ ```
66
+
67
+ Both streaming methods yield `delta` events, each carrying exactly one content item, then exactly one `stop` event, always last, carrying `usage_metadata` and `finish_reason`. Each item streams as one or more `.delta` fragments (`text.delta`, `tool_call.delta`, …) followed by its complete `.done` item (`text.done`, `tool_call.done`, …); items never interleave.
68
+
69
+ ### streaming_response_stateful
70
+
71
+ Stateful method that maintains conversation history internally:
72
+
73
+ ```python
74
+ import asyncio
75
+ from mmsp import AutoLLMClient
76
+
77
+
78
+ async def main():
79
+ client = AutoLLMClient(model="gpt-5.5")
80
+
81
+ # First message
82
+ async for event in client.streaming_response_stateful(
83
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "My name is Alice"}]}, config={}
84
+ ):
85
+ print(event)
86
+
87
+ # Second message - history is maintained automatically
88
+ async for event in client.streaming_response_stateful(
89
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "What's my name?"}]}, config={}
90
+ ):
91
+ print(event)
92
+
93
+
94
+ asyncio.run(main())
95
+ ```
96
+
97
+ ### get_history
98
+
99
+ Retrieve the conversation history:
100
+
101
+ ```python
102
+ # Get all messages in the conversation
103
+ history = client.get_history()
104
+ print(f"Total messages: {len(history)}")
105
+
106
+ for msg in history:
107
+ print(f"Role: {msg['role']}")
108
+ print(f"Content: {msg['content_items']}")
109
+ ```
110
+
111
+ ### clear_history
112
+
113
+ Clear the conversation history:
114
+
115
+ ```python
116
+ # Clear all conversation history
117
+ client.clear_history()
118
+
119
+ # Verify history is empty
120
+ assert len(client.get_history()) == 0
121
+ ```
122
+
123
+ ### set_history
124
+
125
+ Replace the conversation history with a copy of the provided list:
126
+
127
+ ```python
128
+ # Save current history
129
+ saved_history = client.get_history()
130
+
131
+ # ... do other things, then restore
132
+ client.set_history(saved_history)
133
+
134
+ # Verify history was replaced
135
+ assert len(client.get_history()) == len(saved_history)
136
+ ```
137
+
138
+ ## Tool Calling
139
+
140
+ When using tools, you must handle `tool_call_id` correctly:
141
+
142
+ ```python
143
+ import asyncio
144
+ import json
145
+ from mmsp import AutoLLMClient
146
+
147
+
148
+ def get_weather(location: str) -> str:
149
+ """Mock function to get weather."""
150
+ return f"Temperature in {location}: 22°C"
151
+
152
+
153
+ async def main():
154
+ # Define tool
155
+ weather_function = {
156
+ "name": "get_weather",
157
+ "description": "Gets the current weather for a given location.",
158
+ "parameters": {
159
+ "type": "object",
160
+ "properties": {"location": {"type": "string", "description": "The city name"}},
161
+ "required": ["location"],
162
+ },
163
+ }
164
+
165
+ client = AutoLLMClient(model="gpt-5.5")
166
+ config = {"tools": [weather_function]}
167
+
168
+ # User asks about weather
169
+ events = []
170
+ async for event in client.streaming_response_stateful(
171
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "What's the weather in London?"}]},
172
+ config=config,
173
+ ):
174
+ events.append(event)
175
+
176
+ # Read the complete call from its tool_call.done item; tool_call.delta items are fragments
177
+ tool_call = None
178
+ for event in events:
179
+ for item in event["content_items"]:
180
+ if item["type"] == "tool_call.done":
181
+ tool_call = item
182
+ break
183
+
184
+ if tool_call:
185
+ break
186
+
187
+ # Execute function and send result back with tool_call_id
188
+ if tool_call:
189
+ result = get_weather(**tool_call["arguments"])
190
+
191
+ # IMPORTANT: Include tool_call_id in the tool response
192
+ async for event in client.streaming_response_stateful(
193
+ message={
194
+ "role": "user",
195
+ "content_items": [
196
+ {
197
+ "type": "tool_result.done",
198
+ "text": result,
199
+ "tool_call_id": tool_call["tool_call_id"], # Required for tool responses
200
+ }
201
+ ],
202
+ },
203
+ config=config,
204
+ ):
205
+ print(event)
206
+
207
+
208
+ asyncio.run(main())
209
+ ```
210
+
211
+ ## Message Format
212
+
213
+ ### UniMessage Structure
214
+
215
+ ```python
216
+ {
217
+ "role": "user" | "assistant",
218
+ "content_items": [
219
+ {"type": "text.done", "text": "Hello"},
220
+ {"type": "image_url.done", "image_url": "https://..."},
221
+ {
222
+ "type": "tool_call.done",
223
+ "name": "get_weather",
224
+ "arguments": {"location": "London"},
225
+ "tool_call_id": "call_abc123",
226
+ },
227
+ ],
228
+ }
229
+ ```
230
+
231
+ Messages hold complete items only, typed with a `.done` suffix. Item types without the suffix, saved before 0.5.0, are still accepted with a deprecation warning until 0.6.0; `normalize_legacy_messages(messages)` converts stored messages.
232
+
233
+ ### Tool Response with tool_call_id
234
+
235
+ When responding to a tool call, include the `tool_call_id` in the result content item:
236
+
237
+ ```python
238
+ {
239
+ "role": "user",
240
+ "content_items": [
241
+ {
242
+ "type": "tool_result.done",
243
+ "text": "London is 22°C today.",
244
+ "tool_call_id": "call_abc123", # From the tool_call.done item
245
+ }
246
+ ],
247
+ }
248
+ ```
249
+
250
+ ## Configuration Options
251
+
252
+ ```python
253
+ from mmsp import PromptCaching, ThinkingLevel
254
+
255
+ config = {
256
+ "max_tokens": 500,
257
+ "temperature": 1.0,
258
+ "tools": [tool_definition],
259
+ "thinking_summary": True,
260
+ "thinking_level": ThinkingLevel.HIGH,
261
+ "tool_choice": "auto", # "auto", "required", "none", or ["tool_name"]
262
+ "system_prompt": "You are a helpful assistant",
263
+ "prompt_caching": PromptCaching.ENABLE,
264
+ "trace_id": "agent1/conversation_001", # Optional: save conversation trace
265
+ }
266
+ ```
267
+
268
+ ## Conversation Tracing
269
+
270
+ MMSP provides a built-in `Tracer` to save and browse conversation history. When you specify a `trace_id` in the config, conversations are automatically saved to both JSON and TXT formats.
271
+
272
+ ### Basic Usage
273
+
274
+ ```python
275
+ from mmsp import AutoLLMClient
276
+
277
+ client = AutoLLMClient(model="gpt-5.5")
278
+
279
+ # Add trace_id to config
280
+ config = {"trace_id": "agent1/conversation_001"}
281
+
282
+ async for event in client.streaming_response_stateful(
283
+ message={"role": "user", "content_items": [{"type": "text.done", "text": "Hello"}]}, config=config
284
+ ):
285
+ pass # Conversation is automatically saved
286
+ ```
287
+
288
+ The default cache directory is `cache`, you can change it by setting `MMSP_CACHE_DIR` environment variable.
289
+
290
+ This creates two files in the `cache` directory:
291
+ - `cache/agent1/conversation_001.json` - Structured data with full history and config
292
+ - `cache/agent1/conversation_001.txt` - Human-readable conversation format
293
+
294
+ ### Browsing Traces with Web Interface
295
+
296
+ Start a web server to browse and view saved conversations:
297
+
298
+ ```python
299
+ from mmsp.integration.tracer import Tracer
300
+
301
+ # Start web server
302
+ Tracer("path/to/cache").start_web_server(host="127.0.0.1", port=25750)
303
+ ```
304
+
305
+ Or use the CLI:
306
+
307
+ ```bash
308
+ python -m mmsp.integration.tracer --cache_dir ./cache --host 127.0.0.1 --port 25750
309
+ ```
310
+
311
+ Then visit `http://127.0.0.1:25750` in your browser to browse saved conversations.
312
+
313
+ ### Test with Playground
314
+
315
+ Start a web server to test with the playground:
316
+
317
+ ```python
318
+ from mmsp.integration.playground import start_playground_server
319
+
320
+ start_playground_server()
321
+ ```
322
+
323
+ Or use the CLI:
324
+
325
+ ```bash
326
+ python -m mmsp.integration.playground --host 127.0.0.1 --port 25751
327
+ ```
328
+
329
+ Then visit `http://127.0.0.1:25751` in your browser to test with the playground.
330
+ The integrated tracer is available at `http://127.0.0.1:25751/tracer/`.