mmsp 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mmsp-0.5.0/PKG-INFO +356 -0
- mmsp-0.5.0/README.md +330 -0
- mmsp-0.5.0/mmsp/__init__.py +45 -0
- mmsp-0.5.0/mmsp/abort_signal.py +135 -0
- mmsp-0.5.0/mmsp/ant_messages/__init__.py +18 -0
- mmsp-0.5.0/mmsp/ant_messages/client.py +395 -0
- mmsp-0.5.0/mmsp/anthropic_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/anthropic_official/client.py +491 -0
- mmsp-0.5.0/mmsp/auto_client.py +281 -0
- mmsp-0.5.0/mmsp/base_client.py +378 -0
- mmsp-0.5.0/mmsp/deepseek_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/deepseek_official/client.py +384 -0
- mmsp-0.5.0/mmsp/errors.py +123 -0
- mmsp-0.5.0/mmsp/gemini_generate_content/__init__.py +18 -0
- mmsp-0.5.0/mmsp/gemini_generate_content/client.py +668 -0
- mmsp-0.5.0/mmsp/gemini_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/gemini_official/client.py +674 -0
- mmsp-0.5.0/mmsp/integration/__init__.py +14 -0
- mmsp-0.5.0/mmsp/integration/playground.py +3306 -0
- mmsp-0.5.0/mmsp/integration/tracer.py +1812 -0
- mmsp-0.5.0/mmsp/legacy.py +78 -0
- mmsp-0.5.0/mmsp/minimax_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/minimax_official/client.py +323 -0
- mmsp-0.5.0/mmsp/moonshot_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/moonshot_official/client.py +412 -0
- mmsp-0.5.0/mmsp/openai_chat/__init__.py +18 -0
- mmsp-0.5.0/mmsp/openai_chat/client.py +379 -0
- mmsp-0.5.0/mmsp/openai_chat_vllm_adapter/__init__.py +4 -0
- mmsp-0.5.0/mmsp/openai_chat_vllm_adapter/client.py +122 -0
- mmsp-0.5.0/mmsp/openai_embedding/__init__.py +18 -0
- mmsp-0.5.0/mmsp/openai_embedding/client.py +102 -0
- mmsp-0.5.0/mmsp/openai_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/openai_official/client.py +426 -0
- mmsp-0.5.0/mmsp/openai_responses/__init__.py +18 -0
- mmsp-0.5.0/mmsp/openai_responses/client.py +411 -0
- mmsp-0.5.0/mmsp/registry.py +846 -0
- mmsp-0.5.0/mmsp/stream_items.py +199 -0
- mmsp-0.5.0/mmsp/types.py +258 -0
- mmsp-0.5.0/mmsp/utils.py +305 -0
- mmsp-0.5.0/mmsp/zai_official/__init__.py +18 -0
- mmsp-0.5.0/mmsp/zai_official/client.py +406 -0
- mmsp-0.5.0/pyproject.toml +49 -0
mmsp-0.5.0/PKG-INFO
ADDED
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mmsp
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: MMSP, the Model Message Stream Protocol: one message format and one streaming grammar for every model provider, in Python and TypeScript.
|
|
5
|
+
Keywords: mmsp,llm,stream,gemini,claude,gpt
|
|
6
|
+
Author: PrismShadow
|
|
7
|
+
License-Expression: Apache-2.0
|
|
8
|
+
Requires-Dist: google-genai>=2.23.0
|
|
9
|
+
Requires-Dist: anthropic[bedrock]>=0.87.0
|
|
10
|
+
Requires-Dist: flask>=3.0.0
|
|
11
|
+
Requires-Dist: openai>=2.30.0
|
|
12
|
+
Requires-Dist: httpx>=0.27.0
|
|
13
|
+
Requires-Dist: httpx[socks] ; extra == 'dev'
|
|
14
|
+
Requires-Dist: pytest>=8.4.2 ; extra == 'dev'
|
|
15
|
+
Requires-Dist: pytest-asyncio>=0.23.0 ; extra == 'dev'
|
|
16
|
+
Requires-Dist: pytest-xdist>=3.6.0 ; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest-rerunfailures>=15.0 ; extra == 'dev'
|
|
18
|
+
Requires-Dist: ruff>=0.14.3 ; extra == 'dev'
|
|
19
|
+
Requires-Dist: pillow>=10.0.0 ; extra == 'dev'
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Project-URL: Homepage, https://github.com/Prism-Shadow/model-message-stream-protocol
|
|
22
|
+
Project-URL: Repository, https://github.com/Prism-Shadow/model-message-stream-protocol
|
|
23
|
+
Project-URL: Issues, https://github.com/Prism-Shadow/model-message-stream-protocol/issues
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# MMSP Python Implementation
|
|
28
|
+
|
|
29
|
+
This document demonstrates how to use `AutoLLMClient` for unified LLM interactions in MMSP.
|
|
30
|
+
|
|
31
|
+
## Building
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
make install # Install dependencies
|
|
35
|
+
make build # Build Python package
|
|
36
|
+
make lint # Run ruff linter
|
|
37
|
+
make test # Run tests
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## AutoLLMClient Overview
|
|
41
|
+
|
|
42
|
+
`AutoLLMClient` is a stateful client that automatically routes requests to the appropriate model-specific implementation. It maintains conversation history and provides a unified interface for different LLM providers.
|
|
43
|
+
|
|
44
|
+
### Initialization
|
|
45
|
+
|
|
46
|
+
Create a client by specifying the model name:
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from mmsp import AutoLLMClient
|
|
50
|
+
|
|
51
|
+
# The official OpenAI client, named by the model id's family
|
|
52
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
53
|
+
|
|
54
|
+
# The same, spelled out, with the key given in code
|
|
55
|
+
client = AutoLLMClient(model="gpt-5.5", client_type="openai-official", api_key="your-openai-api-key")
|
|
56
|
+
|
|
57
|
+
# A compatible client, for any endpoint that serves OpenAI Chat Completions
|
|
58
|
+
client = AutoLLMClient(
|
|
59
|
+
model="custom-model", client_type="openai-chat", base_url="http://127.0.0.1:8000/v1/", api_key="none"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
# Gemini on Google Vertex AI: the service-account JSON key is the API key
|
|
63
|
+
client = AutoLLMClient(model="gemini-3.8-flash", api_key=open("service-account.json").read())
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
`client_type` names one of the official clients (`openai-official`, `anthropic-official`, `gemini-official`, `zai-official`, `moonshot-official`, `deepseek-official`, `minimax-official`) or one of the compatible clients (`openai-responses`, `openai-chat`, `openai-chat-vllm-adapter`, `openai-embedding`, `ant-messages`, `gemini-generate-content`). It may be omitted for a model id that begins with a known family (`gpt-`, `text-embedding-`, `claude-`, `gemini-`, `glm-`, `kimi-`, `deepseek-`, `minimax-`), which names its official client; any other id raises and asks for one.
|
|
67
|
+
|
|
68
|
+
A Vertex AI service-account key is served through generateContent, because Vertex AI's Interactions endpoint serves none of the Gemini models; any other Gemini key uses the Interactions API. `client_type="gemini-generate-content"` names generateContent explicitly, for gateways that proxy it.
|
|
69
|
+
|
|
70
|
+
## Core Methods
|
|
71
|
+
|
|
72
|
+
### streaming_response
|
|
73
|
+
|
|
74
|
+
Stateless method that requires passing the full message history on each call:
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import asyncio
|
|
78
|
+
from mmsp import AutoLLMClient
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
async def main():
|
|
82
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
83
|
+
|
|
84
|
+
async for event in client.streaming_response(
|
|
85
|
+
messages=[{"role": "user", "content_items": [{"type": "text.done", "text": "Hello!"}]}], config={}
|
|
86
|
+
):
|
|
87
|
+
print(event)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
asyncio.run(main())
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Both streaming methods yield `delta` events, each carrying exactly one content item, then exactly one `stop` event, always last, carrying `usage_metadata` and `finish_reason`. Each item streams as one or more `.delta` fragments (`text.delta`, `tool_call.delta`, …) followed by its complete `.done` item (`text.done`, `tool_call.done`, …); items never interleave.
|
|
94
|
+
|
|
95
|
+
### streaming_response_stateful
|
|
96
|
+
|
|
97
|
+
Stateful method that maintains conversation history internally:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
import asyncio
|
|
101
|
+
from mmsp import AutoLLMClient
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
async def main():
|
|
105
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
106
|
+
|
|
107
|
+
# First message
|
|
108
|
+
async for event in client.streaming_response_stateful(
|
|
109
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "My name is Alice"}]}, config={}
|
|
110
|
+
):
|
|
111
|
+
print(event)
|
|
112
|
+
|
|
113
|
+
# Second message - history is maintained automatically
|
|
114
|
+
async for event in client.streaming_response_stateful(
|
|
115
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "What's my name?"}]}, config={}
|
|
116
|
+
):
|
|
117
|
+
print(event)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
asyncio.run(main())
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
### get_history
|
|
124
|
+
|
|
125
|
+
Retrieve the conversation history:
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
# Get all messages in the conversation
|
|
129
|
+
history = client.get_history()
|
|
130
|
+
print(f"Total messages: {len(history)}")
|
|
131
|
+
|
|
132
|
+
for msg in history:
|
|
133
|
+
print(f"Role: {msg['role']}")
|
|
134
|
+
print(f"Content: {msg['content_items']}")
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
### clear_history
|
|
138
|
+
|
|
139
|
+
Clear the conversation history:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
# Clear all conversation history
|
|
143
|
+
client.clear_history()
|
|
144
|
+
|
|
145
|
+
# Verify history is empty
|
|
146
|
+
assert len(client.get_history()) == 0
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
### set_history
|
|
150
|
+
|
|
151
|
+
Replace the conversation history with a copy of the provided list:
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
# Save current history
|
|
155
|
+
saved_history = client.get_history()
|
|
156
|
+
|
|
157
|
+
# ... do other things, then restore
|
|
158
|
+
client.set_history(saved_history)
|
|
159
|
+
|
|
160
|
+
# Verify history was replaced
|
|
161
|
+
assert len(client.get_history()) == len(saved_history)
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Tool Calling
|
|
165
|
+
|
|
166
|
+
When using tools, you must handle `tool_call_id` correctly:
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
import asyncio
|
|
170
|
+
import json
|
|
171
|
+
from mmsp import AutoLLMClient
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def get_weather(location: str) -> str:
|
|
175
|
+
"""Mock function to get weather."""
|
|
176
|
+
return f"Temperature in {location}: 22°C"
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
async def main():
|
|
180
|
+
# Define tool
|
|
181
|
+
weather_function = {
|
|
182
|
+
"name": "get_weather",
|
|
183
|
+
"description": "Gets the current weather for a given location.",
|
|
184
|
+
"parameters": {
|
|
185
|
+
"type": "object",
|
|
186
|
+
"properties": {"location": {"type": "string", "description": "The city name"}},
|
|
187
|
+
"required": ["location"],
|
|
188
|
+
},
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
192
|
+
config = {"tools": [weather_function]}
|
|
193
|
+
|
|
194
|
+
# User asks about weather
|
|
195
|
+
events = []
|
|
196
|
+
async for event in client.streaming_response_stateful(
|
|
197
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "What's the weather in London?"}]},
|
|
198
|
+
config=config,
|
|
199
|
+
):
|
|
200
|
+
events.append(event)
|
|
201
|
+
|
|
202
|
+
# Read the complete call from its tool_call.done item; tool_call.delta items are fragments
|
|
203
|
+
tool_call = None
|
|
204
|
+
for event in events:
|
|
205
|
+
for item in event["content_items"]:
|
|
206
|
+
if item["type"] == "tool_call.done":
|
|
207
|
+
tool_call = item
|
|
208
|
+
break
|
|
209
|
+
|
|
210
|
+
if tool_call:
|
|
211
|
+
break
|
|
212
|
+
|
|
213
|
+
# Execute function and send result back with tool_call_id
|
|
214
|
+
if tool_call:
|
|
215
|
+
result = get_weather(**tool_call["arguments"])
|
|
216
|
+
|
|
217
|
+
# IMPORTANT: Include tool_call_id in the tool response
|
|
218
|
+
async for event in client.streaming_response_stateful(
|
|
219
|
+
message={
|
|
220
|
+
"role": "user",
|
|
221
|
+
"content_items": [
|
|
222
|
+
{
|
|
223
|
+
"type": "tool_result.done",
|
|
224
|
+
"text": result,
|
|
225
|
+
"tool_call_id": tool_call["tool_call_id"], # Required for tool responses
|
|
226
|
+
}
|
|
227
|
+
],
|
|
228
|
+
},
|
|
229
|
+
config=config,
|
|
230
|
+
):
|
|
231
|
+
print(event)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
asyncio.run(main())
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
## Message Format
|
|
238
|
+
|
|
239
|
+
### UniMessage Structure
|
|
240
|
+
|
|
241
|
+
```python
|
|
242
|
+
{
|
|
243
|
+
"role": "user" | "assistant",
|
|
244
|
+
"content_items": [
|
|
245
|
+
{"type": "text.done", "text": "Hello"},
|
|
246
|
+
{"type": "image_url.done", "image_url": "https://..."},
|
|
247
|
+
{
|
|
248
|
+
"type": "tool_call.done",
|
|
249
|
+
"name": "get_weather",
|
|
250
|
+
"arguments": {"location": "London"},
|
|
251
|
+
"tool_call_id": "call_abc123",
|
|
252
|
+
},
|
|
253
|
+
],
|
|
254
|
+
}
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
Messages hold complete items only, typed with a `.done` suffix. Item types without the suffix, saved before 0.5.0, are still accepted with a deprecation warning until 0.6.0; `normalize_legacy_messages(messages)` converts stored messages.
|
|
258
|
+
|
|
259
|
+
### Tool Response with tool_call_id
|
|
260
|
+
|
|
261
|
+
When responding to a tool call, include the `tool_call_id` in the result content item:
|
|
262
|
+
|
|
263
|
+
```python
|
|
264
|
+
{
|
|
265
|
+
"role": "user",
|
|
266
|
+
"content_items": [
|
|
267
|
+
{
|
|
268
|
+
"type": "tool_result.done",
|
|
269
|
+
"text": "London is 22°C today.",
|
|
270
|
+
"tool_call_id": "call_abc123", # From the tool_call.done item
|
|
271
|
+
}
|
|
272
|
+
],
|
|
273
|
+
}
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
## Configuration Options
|
|
277
|
+
|
|
278
|
+
```python
|
|
279
|
+
from mmsp import PromptCaching, ThinkingLevel
|
|
280
|
+
|
|
281
|
+
config = {
|
|
282
|
+
"max_tokens": 500,
|
|
283
|
+
"temperature": 1.0,
|
|
284
|
+
"tools": [tool_definition],
|
|
285
|
+
"thinking_summary": True,
|
|
286
|
+
"thinking_level": ThinkingLevel.HIGH,
|
|
287
|
+
"tool_choice": "auto", # "auto", "required", "none", or ["tool_name"]
|
|
288
|
+
"system_prompt": "You are a helpful assistant",
|
|
289
|
+
"prompt_caching": PromptCaching.ENABLE,
|
|
290
|
+
"trace_id": "agent1/conversation_001", # Optional: save conversation trace
|
|
291
|
+
}
|
|
292
|
+
```
|
|
293
|
+
|
|
294
|
+
## Conversation Tracing
|
|
295
|
+
|
|
296
|
+
MMSP provides a built-in `Tracer` to save and browse conversation history. When you specify a `trace_id` in the config, conversations are automatically saved to both JSON and TXT formats.
|
|
297
|
+
|
|
298
|
+
### Basic Usage
|
|
299
|
+
|
|
300
|
+
```python
|
|
301
|
+
from mmsp import AutoLLMClient
|
|
302
|
+
|
|
303
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
304
|
+
|
|
305
|
+
# Add trace_id to config
|
|
306
|
+
config = {"trace_id": "agent1/conversation_001"}
|
|
307
|
+
|
|
308
|
+
async for event in client.streaming_response_stateful(
|
|
309
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "Hello"}]}, config=config
|
|
310
|
+
):
|
|
311
|
+
pass # Conversation is automatically saved
|
|
312
|
+
```
|
|
313
|
+
|
|
314
|
+
The default cache directory is `cache`, you can change it by setting `MMSP_CACHE_DIR` environment variable.
|
|
315
|
+
|
|
316
|
+
This creates two files in the `cache` directory:
|
|
317
|
+
- `cache/agent1/conversation_001.json` - Structured data with full history and config
|
|
318
|
+
- `cache/agent1/conversation_001.txt` - Human-readable conversation format
|
|
319
|
+
|
|
320
|
+
### Browsing Traces with Web Interface
|
|
321
|
+
|
|
322
|
+
Start a web server to browse and view saved conversations:
|
|
323
|
+
|
|
324
|
+
```python
|
|
325
|
+
from mmsp.integration.tracer import Tracer
|
|
326
|
+
|
|
327
|
+
# Start web server
|
|
328
|
+
Tracer("path/to/cache").start_web_server(host="127.0.0.1", port=25750)
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
Or use the CLI:
|
|
332
|
+
|
|
333
|
+
```bash
|
|
334
|
+
python -m mmsp.integration.tracer --cache_dir ./cache --host 127.0.0.1 --port 25750
|
|
335
|
+
```
|
|
336
|
+
|
|
337
|
+
Then visit `http://127.0.0.1:25750` in your browser to browse saved conversations.
|
|
338
|
+
|
|
339
|
+
### Test with Playground
|
|
340
|
+
|
|
341
|
+
Start a web server to test with the playground:
|
|
342
|
+
|
|
343
|
+
```python
|
|
344
|
+
from mmsp.integration.playground import start_playground_server
|
|
345
|
+
|
|
346
|
+
start_playground_server()
|
|
347
|
+
```
|
|
348
|
+
|
|
349
|
+
Or use the CLI:
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
python -m mmsp.integration.playground --host 127.0.0.1 --port 25751
|
|
353
|
+
```
|
|
354
|
+
|
|
355
|
+
Then visit `http://127.0.0.1:25751` in your browser to test with the playground.
|
|
356
|
+
The integrated tracer is available at `http://127.0.0.1:25751/tracer/`.
|
mmsp-0.5.0/README.md
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
# MMSP Python Implementation
|
|
2
|
+
|
|
3
|
+
This document demonstrates how to use `AutoLLMClient` for unified LLM interactions in MMSP.
|
|
4
|
+
|
|
5
|
+
## Building
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
make install # Install dependencies
|
|
9
|
+
make build # Build Python package
|
|
10
|
+
make lint # Run ruff linter
|
|
11
|
+
make test # Run tests
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## AutoLLMClient Overview
|
|
15
|
+
|
|
16
|
+
`AutoLLMClient` is a stateful client that automatically routes requests to the appropriate model-specific implementation. It maintains conversation history and provides a unified interface for different LLM providers.
|
|
17
|
+
|
|
18
|
+
### Initialization
|
|
19
|
+
|
|
20
|
+
Create a client by specifying the model name:
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
from mmsp import AutoLLMClient
|
|
24
|
+
|
|
25
|
+
# The official OpenAI client, named by the model id's family
|
|
26
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
27
|
+
|
|
28
|
+
# The same, spelled out, with the key given in code
|
|
29
|
+
client = AutoLLMClient(model="gpt-5.5", client_type="openai-official", api_key="your-openai-api-key")
|
|
30
|
+
|
|
31
|
+
# A compatible client, for any endpoint that serves OpenAI Chat Completions
|
|
32
|
+
client = AutoLLMClient(
|
|
33
|
+
model="custom-model", client_type="openai-chat", base_url="http://127.0.0.1:8000/v1/", api_key="none"
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
# Gemini on Google Vertex AI: the service-account JSON key is the API key
|
|
37
|
+
client = AutoLLMClient(model="gemini-3.8-flash", api_key=open("service-account.json").read())
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
`client_type` names one of the official clients (`openai-official`, `anthropic-official`, `gemini-official`, `zai-official`, `moonshot-official`, `deepseek-official`, `minimax-official`) or one of the compatible clients (`openai-responses`, `openai-chat`, `openai-chat-vllm-adapter`, `openai-embedding`, `ant-messages`, `gemini-generate-content`). It may be omitted for a model id that begins with a known family (`gpt-`, `text-embedding-`, `claude-`, `gemini-`, `glm-`, `kimi-`, `deepseek-`, `minimax-`), which names its official client; any other id raises and asks for one.
|
|
41
|
+
|
|
42
|
+
A Vertex AI service-account key is served through generateContent, because Vertex AI's Interactions endpoint serves none of the Gemini models; any other Gemini key uses the Interactions API. `client_type="gemini-generate-content"` names generateContent explicitly, for gateways that proxy it.
|
|
43
|
+
|
|
44
|
+
## Core Methods
|
|
45
|
+
|
|
46
|
+
### streaming_response
|
|
47
|
+
|
|
48
|
+
Stateless method that requires passing the full message history on each call:
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
import asyncio
|
|
52
|
+
from mmsp import AutoLLMClient
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
async def main():
|
|
56
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
57
|
+
|
|
58
|
+
async for event in client.streaming_response(
|
|
59
|
+
messages=[{"role": "user", "content_items": [{"type": "text.done", "text": "Hello!"}]}], config={}
|
|
60
|
+
):
|
|
61
|
+
print(event)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
asyncio.run(main())
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Both streaming methods yield `delta` events, each carrying exactly one content item, then exactly one `stop` event, always last, carrying `usage_metadata` and `finish_reason`. Each item streams as one or more `.delta` fragments (`text.delta`, `tool_call.delta`, …) followed by its complete `.done` item (`text.done`, `tool_call.done`, …); items never interleave.
|
|
68
|
+
|
|
69
|
+
### streaming_response_stateful
|
|
70
|
+
|
|
71
|
+
Stateful method that maintains conversation history internally:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import asyncio
|
|
75
|
+
from mmsp import AutoLLMClient
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
async def main():
|
|
79
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
80
|
+
|
|
81
|
+
# First message
|
|
82
|
+
async for event in client.streaming_response_stateful(
|
|
83
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "My name is Alice"}]}, config={}
|
|
84
|
+
):
|
|
85
|
+
print(event)
|
|
86
|
+
|
|
87
|
+
# Second message - history is maintained automatically
|
|
88
|
+
async for event in client.streaming_response_stateful(
|
|
89
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "What's my name?"}]}, config={}
|
|
90
|
+
):
|
|
91
|
+
print(event)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
asyncio.run(main())
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### get_history
|
|
98
|
+
|
|
99
|
+
Retrieve the conversation history:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
# Get all messages in the conversation
|
|
103
|
+
history = client.get_history()
|
|
104
|
+
print(f"Total messages: {len(history)}")
|
|
105
|
+
|
|
106
|
+
for msg in history:
|
|
107
|
+
print(f"Role: {msg['role']}")
|
|
108
|
+
print(f"Content: {msg['content_items']}")
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### clear_history
|
|
112
|
+
|
|
113
|
+
Clear the conversation history:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
# Clear all conversation history
|
|
117
|
+
client.clear_history()
|
|
118
|
+
|
|
119
|
+
# Verify history is empty
|
|
120
|
+
assert len(client.get_history()) == 0
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
### set_history
|
|
124
|
+
|
|
125
|
+
Replace the conversation history with a copy of the provided list:
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
# Save current history
|
|
129
|
+
saved_history = client.get_history()
|
|
130
|
+
|
|
131
|
+
# ... do other things, then restore
|
|
132
|
+
client.set_history(saved_history)
|
|
133
|
+
|
|
134
|
+
# Verify history was replaced
|
|
135
|
+
assert len(client.get_history()) == len(saved_history)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Tool Calling
|
|
139
|
+
|
|
140
|
+
When using tools, you must handle `tool_call_id` correctly:
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
import asyncio
|
|
144
|
+
import json
|
|
145
|
+
from mmsp import AutoLLMClient
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def get_weather(location: str) -> str:
|
|
149
|
+
"""Mock function to get weather."""
|
|
150
|
+
return f"Temperature in {location}: 22°C"
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
async def main():
|
|
154
|
+
# Define tool
|
|
155
|
+
weather_function = {
|
|
156
|
+
"name": "get_weather",
|
|
157
|
+
"description": "Gets the current weather for a given location.",
|
|
158
|
+
"parameters": {
|
|
159
|
+
"type": "object",
|
|
160
|
+
"properties": {"location": {"type": "string", "description": "The city name"}},
|
|
161
|
+
"required": ["location"],
|
|
162
|
+
},
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
166
|
+
config = {"tools": [weather_function]}
|
|
167
|
+
|
|
168
|
+
# User asks about weather
|
|
169
|
+
events = []
|
|
170
|
+
async for event in client.streaming_response_stateful(
|
|
171
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "What's the weather in London?"}]},
|
|
172
|
+
config=config,
|
|
173
|
+
):
|
|
174
|
+
events.append(event)
|
|
175
|
+
|
|
176
|
+
# Read the complete call from its tool_call.done item; tool_call.delta items are fragments
|
|
177
|
+
tool_call = None
|
|
178
|
+
for event in events:
|
|
179
|
+
for item in event["content_items"]:
|
|
180
|
+
if item["type"] == "tool_call.done":
|
|
181
|
+
tool_call = item
|
|
182
|
+
break
|
|
183
|
+
|
|
184
|
+
if tool_call:
|
|
185
|
+
break
|
|
186
|
+
|
|
187
|
+
# Execute function and send result back with tool_call_id
|
|
188
|
+
if tool_call:
|
|
189
|
+
result = get_weather(**tool_call["arguments"])
|
|
190
|
+
|
|
191
|
+
# IMPORTANT: Include tool_call_id in the tool response
|
|
192
|
+
async for event in client.streaming_response_stateful(
|
|
193
|
+
message={
|
|
194
|
+
"role": "user",
|
|
195
|
+
"content_items": [
|
|
196
|
+
{
|
|
197
|
+
"type": "tool_result.done",
|
|
198
|
+
"text": result,
|
|
199
|
+
"tool_call_id": tool_call["tool_call_id"], # Required for tool responses
|
|
200
|
+
}
|
|
201
|
+
],
|
|
202
|
+
},
|
|
203
|
+
config=config,
|
|
204
|
+
):
|
|
205
|
+
print(event)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
asyncio.run(main())
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
## Message Format
|
|
212
|
+
|
|
213
|
+
### UniMessage Structure
|
|
214
|
+
|
|
215
|
+
```python
|
|
216
|
+
{
|
|
217
|
+
"role": "user" | "assistant",
|
|
218
|
+
"content_items": [
|
|
219
|
+
{"type": "text.done", "text": "Hello"},
|
|
220
|
+
{"type": "image_url.done", "image_url": "https://..."},
|
|
221
|
+
{
|
|
222
|
+
"type": "tool_call.done",
|
|
223
|
+
"name": "get_weather",
|
|
224
|
+
"arguments": {"location": "London"},
|
|
225
|
+
"tool_call_id": "call_abc123",
|
|
226
|
+
},
|
|
227
|
+
],
|
|
228
|
+
}
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
Messages hold complete items only, typed with a `.done` suffix. Item types without the suffix, saved before 0.5.0, are still accepted with a deprecation warning until 0.6.0; `normalize_legacy_messages(messages)` converts stored messages.
|
|
232
|
+
|
|
233
|
+
### Tool Response with tool_call_id
|
|
234
|
+
|
|
235
|
+
When responding to a tool call, include the `tool_call_id` in the result content item:
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
{
|
|
239
|
+
"role": "user",
|
|
240
|
+
"content_items": [
|
|
241
|
+
{
|
|
242
|
+
"type": "tool_result.done",
|
|
243
|
+
"text": "London is 22°C today.",
|
|
244
|
+
"tool_call_id": "call_abc123", # From the tool_call.done item
|
|
245
|
+
}
|
|
246
|
+
],
|
|
247
|
+
}
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
## Configuration Options
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
from mmsp import PromptCaching, ThinkingLevel
|
|
254
|
+
|
|
255
|
+
config = {
|
|
256
|
+
"max_tokens": 500,
|
|
257
|
+
"temperature": 1.0,
|
|
258
|
+
"tools": [tool_definition],
|
|
259
|
+
"thinking_summary": True,
|
|
260
|
+
"thinking_level": ThinkingLevel.HIGH,
|
|
261
|
+
"tool_choice": "auto", # "auto", "required", "none", or ["tool_name"]
|
|
262
|
+
"system_prompt": "You are a helpful assistant",
|
|
263
|
+
"prompt_caching": PromptCaching.ENABLE,
|
|
264
|
+
"trace_id": "agent1/conversation_001", # Optional: save conversation trace
|
|
265
|
+
}
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
## Conversation Tracing
|
|
269
|
+
|
|
270
|
+
MMSP provides a built-in `Tracer` to save and browse conversation history. When you specify a `trace_id` in the config, conversations are automatically saved to both JSON and TXT formats.
|
|
271
|
+
|
|
272
|
+
### Basic Usage
|
|
273
|
+
|
|
274
|
+
```python
|
|
275
|
+
from mmsp import AutoLLMClient
|
|
276
|
+
|
|
277
|
+
client = AutoLLMClient(model="gpt-5.5")
|
|
278
|
+
|
|
279
|
+
# Add trace_id to config
|
|
280
|
+
config = {"trace_id": "agent1/conversation_001"}
|
|
281
|
+
|
|
282
|
+
async for event in client.streaming_response_stateful(
|
|
283
|
+
message={"role": "user", "content_items": [{"type": "text.done", "text": "Hello"}]}, config=config
|
|
284
|
+
):
|
|
285
|
+
pass # Conversation is automatically saved
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
The default cache directory is `cache`, you can change it by setting `MMSP_CACHE_DIR` environment variable.
|
|
289
|
+
|
|
290
|
+
This creates two files in the `cache` directory:
|
|
291
|
+
- `cache/agent1/conversation_001.json` - Structured data with full history and config
|
|
292
|
+
- `cache/agent1/conversation_001.txt` - Human-readable conversation format
|
|
293
|
+
|
|
294
|
+
### Browsing Traces with Web Interface
|
|
295
|
+
|
|
296
|
+
Start a web server to browse and view saved conversations:
|
|
297
|
+
|
|
298
|
+
```python
|
|
299
|
+
from mmsp.integration.tracer import Tracer
|
|
300
|
+
|
|
301
|
+
# Start web server
|
|
302
|
+
Tracer("path/to/cache").start_web_server(host="127.0.0.1", port=25750)
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
Or use the CLI:
|
|
306
|
+
|
|
307
|
+
```bash
|
|
308
|
+
python -m mmsp.integration.tracer --cache_dir ./cache --host 127.0.0.1 --port 25750
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
Then visit `http://127.0.0.1:25750` in your browser to browse saved conversations.
|
|
312
|
+
|
|
313
|
+
### Test with Playground
|
|
314
|
+
|
|
315
|
+
Start a web server to test with the playground:
|
|
316
|
+
|
|
317
|
+
```python
|
|
318
|
+
from mmsp.integration.playground import start_playground_server
|
|
319
|
+
|
|
320
|
+
start_playground_server()
|
|
321
|
+
```
|
|
322
|
+
|
|
323
|
+
Or use the CLI:
|
|
324
|
+
|
|
325
|
+
```bash
|
|
326
|
+
python -m mmsp.integration.playground --host 127.0.0.1 --port 25751
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
Then visit `http://127.0.0.1:25751` in your browser to test with the playground.
|
|
330
|
+
The integrated tracer is available at `http://127.0.0.1:25751/tracer/`.
|