agenthub-python 0.3.0__py3-none-any.whl → 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agenthub/auto_client.py CHANGED
@@ -54,10 +54,10 @@ class AutoLLMClient(LLMClient):
54
54
  from .claude4_6 import Claude4_6Client
55
55
 
56
56
  return Claude4_6Client(model=model, api_key=api_key, base_url=base_url)
57
- elif "gpt-5.4" in client_type: # e.g., gpt-5.4
58
- from .gpt5_4 import GPT5_4Client
57
+ elif "gpt-5.4" in client_type or "gpt-5.5" in client_type: # e.g., gpt-5.5
58
+ from .gpt5_5 import GPT5_5Client
59
59
 
60
- return GPT5_4Client(model=model, api_key=api_key, base_url=base_url)
60
+ return GPT5_5Client(model=model, api_key=api_key, base_url=base_url)
61
61
  elif "glm-5" in client_type:
62
62
  from .glm5 import GLM5Client
63
63
 
@@ -73,7 +73,7 @@ class AutoLLMClient(LLMClient):
73
73
  else:
74
74
  raise ValueError(
75
75
  f"{client_type} is not supported. "
76
- "Supported client types: gemini-3, claude-4-6, gpt-5.4, glm-5, kimi-k2.5, qwen3."
76
+ "Supported client types: gemini-3, claude-4-6, gpt-5.4, gpt-5.5, glm-5, kimi-k2.5, qwen3."
77
77
  )
78
78
 
79
79
  def transform_uni_config_to_model_config(self, config: UniConfig) -> Any:
@@ -126,3 +126,7 @@ class AutoLLMClient(LLMClient):
126
126
  def get_history(self) -> list[UniMessage]:
127
127
  """Get history from the underlying client."""
128
128
  return self._client.get_history()
129
+
130
+ def set_history(self, history: list[UniMessage]) -> None:
131
+ """Set history in the underlying client."""
132
+ self._client.set_history(history)
agenthub/base_client.py CHANGED
@@ -12,6 +12,7 @@
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
14
 
15
+ import time
15
16
  from abc import ABC, abstractmethod
16
17
  from typing import Any, AsyncIterator
17
18
 
@@ -84,6 +85,7 @@ class LLMClient(ABC):
84
85
  content_items: list[ContentItem] = []
85
86
  usage_metadata: UsageMetadata | None = None
86
87
  finish_reason: FinishReason | None = None
88
+ created_at: int | None = None
87
89
 
88
90
  for event in events:
89
91
  # Merge content_items from all events
@@ -119,12 +121,14 @@ class LLMClient(ABC):
119
121
 
120
122
  usage_metadata = event.get("usage_metadata") # usage_metadata is taken from the last event
121
123
  finish_reason = event.get("finish_reason") # finish_reason is taken from the last event
124
+ created_at = event.get("created_at") # created_at is taken from the last event
122
125
 
123
126
  return {
124
127
  "role": "assistant",
125
128
  "content_items": content_items,
126
129
  "usage_metadata": usage_metadata,
127
130
  "finish_reason": finish_reason,
131
+ "created_at": created_at,
128
132
  }
129
133
 
130
134
  @abstractmethod
@@ -167,13 +171,29 @@ class LLMClient(ABC):
167
171
  Yields:
168
172
  Universal events from the streaming response
169
173
  """
174
+ # Stamp any messages that don't yet have a created_at timestamp
175
+ for msg in messages:
176
+ if "created_at" not in msg:
177
+ msg["created_at"] = int(time.time() * 1000)
178
+
170
179
  last_event: UniEvent | None = None
180
+ events = []
171
181
  async for event in self._streaming_response_internal(messages, config):
182
+ event["created_at"] = int(time.time() * 1000)
172
183
  last_event = event
184
+ events.append(event)
173
185
  yield event
174
186
 
175
187
  self._validate_last_event(last_event)
176
188
 
189
+ # Save history to file if trace_id is specified
190
+ if config.get("trace_id") and events:
191
+ from .integration.tracer import Tracer
192
+
193
+ assistant_message = self.concat_uni_events_to_uni_message(events)
194
+ tracer = Tracer()
195
+ tracer.save_history(self._model, messages + [assistant_message], config["trace_id"], config)
196
+
177
197
  async def streaming_response_stateful(
178
198
  self,
179
199
  message: UniMessage,
@@ -203,18 +223,12 @@ class LLMClient(ABC):
203
223
  yield event
204
224
 
205
225
  # Only update history after successful inference
226
+ # temp_messages[-1] is the user message, now stamped with created_at by streaming_response
206
227
  if events:
207
228
  assistant_message = self.concat_uni_events_to_uni_message(events)
208
- self._history.append(message)
229
+ self._history.append(temp_messages[-1])
209
230
  self._history.append(assistant_message)
210
231
 
211
- # Save history to file if trace_id is specified
212
- if config.get("trace_id"):
213
- from .integration.tracer import Tracer
214
-
215
- tracer = Tracer()
216
- tracer.save_history(self._model, self._history, config["trace_id"], config)
217
-
218
232
  @staticmethod
219
233
  def _validate_last_event(last_event: UniEvent | None) -> None:
220
234
  """Validate that the last event has usage_metadata and finish_reason.
@@ -244,3 +258,11 @@ class LLMClient(ABC):
244
258
  def get_history(self) -> list[UniMessage]:
245
259
  """Get the current message history."""
246
260
  return self._history.copy()
261
+
262
+ def set_history(self, history: list[UniMessage]) -> None:
263
+ """Replace the message history with a copy of the provided history.
264
+
265
+ Args:
266
+ history: List of universal message dictionaries to set as the new history
267
+ """
268
+ self._history = list(history)
@@ -171,6 +171,15 @@ class Claude4_6Client(LLMClient):
171
171
  if config.get("tool_choice") is not None:
172
172
  claude_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
173
173
 
174
+ # Add cache_control if prompt caching is enabled
175
+ # TODO: wait for bedrock to support cache_control in config
176
+ if not self._use_bedrock:
177
+ prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
178
+ if prompt_caching == PromptCaching.ENABLE:
179
+ claude_config["cache_control"] = {"type": "ephemeral"}
180
+ elif prompt_caching == PromptCaching.ENHANCE:
181
+ claude_config["cache_control"] = {"type": "ephemeral", "ttl": "1h"}
182
+
174
183
  return claude_config
175
184
 
176
185
  async def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> list[BetaMessageParam]:
@@ -334,18 +343,20 @@ class Claude4_6Client(LLMClient):
334
343
  # Use unified message conversion
335
344
  claude_messages = await self.transform_uni_message_to_model_input(messages)
336
345
 
337
- # Add cache_control to last user message's last item if enabled
338
- prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
339
- if prompt_caching != PromptCaching.DISABLE and claude_messages:
340
- try:
341
- last_user_message = next(filter(lambda x: x["role"] == "user", claude_messages[::-1]))
342
- last_content_item = last_user_message["content"][-1]
343
- last_content_item["cache_control"] = {
344
- "type": "ephemeral",
345
- "ttl": "1h" if prompt_caching == PromptCaching.ENHANCE else "5m",
346
- }
347
- except StopIteration:
348
- pass
346
+ # Add cache_control to last user message's last item if using bedrock and enabled prompt caching
347
+ # TODO: remove after bedrock supports cache_control in config
348
+ if self._use_bedrock:
349
+ prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
350
+ if prompt_caching != PromptCaching.DISABLE and claude_messages:
351
+ try:
352
+ last_user_message = next(filter(lambda x: x["role"] == "user", claude_messages[::-1]))
353
+ last_content_item = last_user_message["content"][-1]
354
+ last_content_item["cache_control"] = {
355
+ "type": "ephemeral",
356
+ "ttl": "1h" if prompt_caching == PromptCaching.ENHANCE else "5m",
357
+ }
358
+ except StopIteration:
359
+ pass
349
360
 
350
361
  # Stream generate
351
362
  partial_tool_call = {}
@@ -147,6 +147,44 @@ class Gemini3Client(LLMClient):
147
147
  if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
148
148
  raise ValueError("prompt_caching must be ENABLE for Gemini 3.")
149
149
 
150
+ if config.get("image_config") is not None:
151
+ config_params["image_config"] = types.ImageConfig(**config["image_config"])
152
+
153
+ # tts config
154
+ if "tts" in self._model.lower():
155
+ config_params["response_modalities"] = ["AUDIO"]
156
+ tts_config = config.get("tts_config") or [{"voice": "Kore"}]
157
+ if len(tts_config) not in (1, 2):
158
+ raise ValueError("tts_config must contain 1 or 2 entries.")
159
+
160
+ if len(tts_config) == 1:
161
+ config_params["speech_config"] = types.SpeechConfig(
162
+ voice_config=types.VoiceConfig(
163
+ prebuilt_voice_config=types.PrebuiltVoiceConfig(voice_name=tts_config[0]["voice"])
164
+ )
165
+ )
166
+ else:
167
+ speaker_voice_configs = []
168
+ for speaker_config in tts_config:
169
+ speaker = speaker_config.get("speaker")
170
+ if not speaker:
171
+ raise ValueError("speaker is required when tts_config has 2 entries.")
172
+
173
+ speaker_voice_configs.append(
174
+ types.SpeakerVoiceConfig(
175
+ speaker=speaker,
176
+ voice_config=types.VoiceConfig(
177
+ prebuilt_voice_config=types.PrebuiltVoiceConfig(voice_name=speaker_config["voice"])
178
+ ),
179
+ )
180
+ )
181
+
182
+ config_params["speech_config"] = types.SpeechConfig(
183
+ multi_speaker_voice_config=types.MultiSpeakerVoiceConfig(
184
+ speaker_voice_configs=speaker_voice_configs
185
+ )
186
+ )
187
+
150
188
  return types.GenerateContentConfig(**config_params) if config_params else None
151
189
 
152
190
  async def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> list[types.Content]:
@@ -170,10 +208,18 @@ class Gemini3Client(LLMClient):
170
208
  image_url = item["image_url"]
171
209
  image_data = await self._get_image_bytes_and_mime_type(image_url)
172
210
  parts.append(types.Part.from_bytes(**image_data))
211
+ elif item["type"] == "inline_data":
212
+ inline_data = types.Blob(data=item["data"], mime_type=item["mime_type"])
213
+ parts.append(types.Part(inline_data=inline_data, thought_signature=item.get("signature")))
173
214
  elif item["type"] == "thinking":
174
215
  parts.append(
175
216
  types.Part(text=item["thinking"], thought=True, thought_signature=item.get("signature"))
176
217
  )
218
+ elif item["type"] == "inline_thinking":
219
+ inline_data = types.Blob(data=item["data"], mime_type=item["mime_type"])
220
+ parts.append(
221
+ types.Part(inline_data=inline_data, thought=True, thought_signature=item.get("signature"))
222
+ )
177
223
  elif item["type"] == "tool_call":
178
224
  function_call = types.FunctionCall(name=item["name"], args=item["arguments"])
179
225
  parts.append(types.Part(function_call=function_call, thought_signature=item.get("signature")))
@@ -232,9 +278,28 @@ class Gemini3Client(LLMClient):
232
278
  "signature": part.thought_signature,
233
279
  }
234
280
  )
235
- elif part.text is not None and part.thought:
281
+ elif part.thought:
282
+ if part.text is not None:
283
+ content_items.append(
284
+ {"type": "thinking", "thinking": part.text, "signature": part.thought_signature}
285
+ )
286
+ elif part.inline_data is not None:
287
+ content_items.append(
288
+ {
289
+ "type": "inline_thinking",
290
+ "data": part.inline_data.data,
291
+ "mime_type": part.inline_data.mime_type,
292
+ "signature": part.thought_signature,
293
+ }
294
+ )
295
+ elif part.inline_data is not None:
236
296
  content_items.append(
237
- {"type": "thinking", "thinking": part.text, "signature": part.thought_signature}
297
+ {
298
+ "type": "inline_data",
299
+ "data": part.inline_data.data,
300
+ "mime_type": part.inline_data.mime_type,
301
+ "signature": part.thought_signature,
302
+ }
238
303
  )
239
304
  elif part.text is not None:
240
305
  content_items.append({"type": "text", "text": part.text, "signature": part.thought_signature})
@@ -278,6 +343,15 @@ class Gemini3Client(LLMClient):
278
343
  # Use unified config conversion
279
344
  gemini_config = self.transform_uni_config_to_model_config(config)
280
345
 
346
+ # check if all items are text for tts model
347
+ if "tts" in self._model.lower():
348
+ invalid_item = next(
349
+ (item for message in messages for item in message["content_items"] if item["type"] != "text"),
350
+ None,
351
+ )
352
+ if invalid_item is not None:
353
+ raise ValueError(f"Gemini TTS only supports text input, got content item type={invalid_item['type']}.")
354
+
281
355
  # Use unified message conversion
282
356
  contents = await self.transform_uni_message_to_model_input(messages)
283
357
 
agenthub/glm5/client.py CHANGED
@@ -206,9 +206,9 @@ class GLM5Client(LLMClient):
206
206
  content_items.append(
207
207
  {
208
208
  "type": "partial_tool_call",
209
- "name": tool_call.function.name,
210
- "arguments": tool_call.function.arguments,
211
- "tool_call_id": tool_call.id,
209
+ "name": tool_call.function.name or "",
210
+ "arguments": tool_call.function.arguments or "",
211
+ "tool_call_id": tool_call.id or "",
212
212
  }
213
213
  )
214
214
 
@@ -12,7 +12,7 @@
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
14
 
15
- from .client import GPT5_4Client
15
+ from .client import GPT5_5Client
16
16
 
17
17
 
18
- __all__ = ["GPT5_4Client"]
18
+ __all__ = ["GPT5_5Client"]
@@ -24,6 +24,7 @@ from ..types import (
24
24
  EventType,
25
25
  FinishReason,
26
26
  PartialContentItem,
27
+ PromptCaching,
27
28
  ThinkingLevel,
28
29
  ToolChoice,
29
30
  UniConfig,
@@ -33,11 +34,11 @@ from ..types import (
33
34
  )
34
35
 
35
36
 
36
- class GPT5_4Client(LLMClient):
37
- """GPT-5.4-specific LLM client implementation."""
37
+ class GPT5_5Client(LLMClient):
38
+ """GPT-5.5-specific LLM client implementation."""
38
39
 
39
40
  def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
40
- """Initialize GPT-5.4 client with model and API key."""
41
+ """Initialize GPT-5.5 client with model and API key."""
41
42
  self._model = model
42
43
  api_key = api_key or os.getenv("OPENAI_API_KEY")
43
44
  base_url = base_url or os.getenv("OPENAI_BASE_URL")
@@ -86,7 +87,7 @@ class GPT5_4Client(LLMClient):
86
87
  openai_config["max_output_tokens"] = config["max_tokens"]
87
88
 
88
89
  if config.get("temperature") is not None and config["temperature"] != 1.0:
89
- raise ValueError("GPT-5.4 does not support setting temperature.")
90
+ raise ValueError("GPT-5.5 does not support setting temperature.")
90
91
 
91
92
  if config.get("thinking_level") is not None:
92
93
  openai_config["reasoning"] = {"effort": self._convert_thinking_level_to_effort(config["thinking_level"])}
@@ -99,6 +100,9 @@ class GPT5_4Client(LLMClient):
99
100
  if config.get("tool_choice") is not None:
100
101
  openai_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
101
102
 
103
+ if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
104
+ raise ValueError("prompt_caching must be ENABLE for GPT-5.5.")
105
+
102
106
  return openai_config
103
107
 
104
108
  def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> ResponseInputParam:
@@ -213,13 +217,14 @@ class GPT5_4Client(LLMClient):
213
217
  "tool_call_id": model_output.item.call_id,
214
218
  }
215
219
  )
216
- elif model_output.item.type == "reasoning":
217
- event_type = "delta"
218
- signature = {
219
- "id": model_output.item.id,
220
- "encrypted_content": model_output.item.encrypted_content,
221
- }
222
- content_items.append({"type": "thinking", "thinking": "", "signature": json.dumps(signature)})
220
+ # adding the following thinking item leads to 400 invalid request error, why?
221
+ # elif model_output.item.type == "reasoning":
222
+ # event_type = "delta"
223
+ # signature = {
224
+ # "id": model_output.item.id,
225
+ # "encrypted_content": model_output.item.encrypted_content,
226
+ # }
227
+ # content_items.append({"type": "thinking", "thinking": "", "signature": json.dumps(signature)})
223
228
  elif model_output.item.type == "message":
224
229
  if hasattr(model_output.item, "phase"):
225
230
  event_type = "delta"
@@ -311,7 +316,6 @@ class GPT5_4Client(LLMClient):
311
316
  # Stream generate
312
317
  partial_tool_call = {}
313
318
  stream = await self._client.responses.create(**openai_config, input=input_list, stream=True)
314
-
315
319
  async for event in stream:
316
320
  event = self.transform_model_output_to_uni_event(event)
317
321
  if event["event_type"] == "start":
@@ -110,11 +110,13 @@ def create_chat_app() -> Flask:
110
110
  class="px-3 py-2 border border-gray-300 rounded-md text-sm focus:ring-2 focus:ring-blue-500 focus:border-blue-500"
111
111
  />
112
112
  <datalist id="modelList">
113
- <option value="gpt-5.2">GPT 5.2</option>
113
+ <option value="gpt-5.5">GPT 5.5</option>
114
114
  <option value="gemini-3-flash-preview">Gemini 3 Flash</option>
115
115
  <option value="claude-sonnet-4-6">Claude Sonnet 4.6</option>
116
116
  <option value="kimi-k2.5">Kimi K2.5</option>
117
117
  <option value="glm-5">GLM 5</option>
118
+ <option value="gemini-3.1-flash-image-preview">Gemini 3.1 Flash Image (Nano Banana 2)</option>
119
+ <option value="gemini-3.1-flash-tts-preview">Gemini 3.1 Flash TTS</option>
118
120
  </datalist>
119
121
  </div>
120
122
  <div class="flex flex-col">
@@ -191,6 +193,7 @@ def create_chat_app() -> Flask:
191
193
  let isStreaming = false;
192
194
  let sessionId = Math.random().toString(36).substring(7);
193
195
  let selectedImages = [];
196
+ let lastMessageTimestamp = null;
194
197
 
195
198
  function escapeHtml(text) {
196
199
  const div = document.createElement('div');
@@ -198,6 +201,76 @@ def create_chat_app() -> Flask:
198
201
  return div.innerHTML;
199
202
  }
200
203
 
204
+ function formatTimestamp(ms) {
205
+ if (!ms) return '';
206
+ const d = new Date(ms);
207
+ const pad = n => n.toString().padStart(2, '0');
208
+ return `${d.getFullYear()}-${pad(d.getMonth()+1)}-${pad(d.getDate())} ${pad(d.getHours())}:${pad(d.getMinutes())}:${pad(d.getSeconds())}`;
209
+ }
210
+
211
+ function pcmBase64ToWavDataUrl(pcmBase64, sampleRate = 24000, channels = 1, bitsPerSample = 16) {
212
+ const binary = atob(pcmBase64);
213
+ const pcmBytes = new Uint8Array(binary.length);
214
+ for (let i = 0; i < binary.length; i++) {
215
+ pcmBytes[i] = binary.charCodeAt(i);
216
+ }
217
+
218
+ const header = new ArrayBuffer(44);
219
+ const view = new DataView(header);
220
+ const byteRate = sampleRate * channels * bitsPerSample / 8;
221
+ const blockAlign = channels * bitsPerSample / 8;
222
+
223
+ const writeString = (offset, value) => {
224
+ for (let i = 0; i < value.length; i++) {
225
+ view.setUint8(offset + i, value.charCodeAt(i));
226
+ }
227
+ };
228
+
229
+ writeString(0, 'RIFF');
230
+ view.setUint32(4, 36 + pcmBytes.length, true);
231
+ writeString(8, 'WAVE');
232
+ writeString(12, 'fmt ');
233
+ view.setUint32(16, 16, true);
234
+ view.setUint16(20, 1, true);
235
+ view.setUint16(22, channels, true);
236
+ view.setUint32(24, sampleRate, true);
237
+ view.setUint32(28, byteRate, true);
238
+ view.setUint16(32, blockAlign, true);
239
+ view.setUint16(34, bitsPerSample, true);
240
+ writeString(36, 'data');
241
+ view.setUint32(40, pcmBytes.length, true);
242
+
243
+ const wavBytes = new Uint8Array(44 + pcmBytes.length);
244
+ wavBytes.set(new Uint8Array(header), 0);
245
+ wavBytes.set(pcmBytes, 44);
246
+
247
+ let wavBinary = '';
248
+ const chunkSize = 0x8000;
249
+ for (let i = 0; i < wavBytes.length; i += chunkSize) {
250
+ wavBinary += String.fromCharCode(...wavBytes.subarray(i, i + chunkSize));
251
+ }
252
+ return `data:audio/wav;base64,${btoa(wavBinary)}`;
253
+ }
254
+
255
+ function renderInlineData(item) {
256
+ const mimeType = (item.mime_type || '').toLowerCase();
257
+ if (mimeType.startsWith('image/')) {
258
+ return `<div class="mb-3"><img src="data:${mimeType || 'image/png'};base64,${item.data}" class="max-w-xs rounded border border-gray-300"></div>`;
259
+ }
260
+
261
+ const audioMimeTypes = ['audio/wav', 'audio/x-wav', 'audio/mpeg', 'audio/mp3', 'audio/ogg', 'audio/webm', 'audio/flac', 'audio/aac', 'audio/mp4'];
262
+ const isAudio = !mimeType || mimeType === 'application/octet-stream' || mimeType.startsWith('audio/');
263
+ if (!isAudio) {
264
+ return `<div class="mb-3 rounded border border-gray-300 bg-gray-50 px-3 py-2 text-xs text-gray-600">Inline data: ${escapeHtml(item.mime_type || 'application/octet-stream')}</div>`;
265
+ }
266
+
267
+ const playableMimeType = audioMimeTypes.includes(mimeType);
268
+ const audioSrc = playableMimeType
269
+ ? `data:${mimeType || 'application/octet-stream'};base64,${item.data}`
270
+ : pcmBase64ToWavDataUrl(item.data);
271
+ return `<div class="mb-3"><audio controls preload="metadata" class="max-w-xs"><source src="${audioSrc}" type="${playableMimeType ? mimeType : 'audio/wav'}"></audio></div>`;
272
+ }
273
+
201
274
  function handleImageSelect(event) {
202
275
  const files = event.target.files;
203
276
  if (!files || files.length === 0) return;
@@ -302,7 +375,7 @@ def create_chat_app() -> Flask:
302
375
  return config;
303
376
  }
304
377
 
305
- function addMessageCard(role, content, metadata = null, images = []) {
378
+ function addMessageCard(role, content, metadata = null, images = [], timestamp = null, tookMs = null) {
306
379
  const container = document.getElementById('messagesContainer');
307
380
 
308
381
  if (container.children.length === 1 && container.children[0].className.includes('text-center')) {
@@ -316,6 +389,10 @@ def create_chat_app() -> Flask:
316
389
  let html = `
317
390
  <div class="flex justify-between items-center mb-3">
318
391
  <span class="font-semibold text-sm uppercase ${isUser ? 'text-blue-600' : 'text-green-600'}">${role}</span>
392
+ <div class="flex items-center gap-2">
393
+ <span class="msg-took text-xs text-gray-400">${tookMs !== null ? 'Took ' + tookMs + ' ms' : ''}</span>
394
+ <span class="text-xs text-gray-400 msg-timestamp">${timestamp ? formatTimestamp(timestamp) : ''}</span>
395
+ </div>
319
396
  </div>
320
397
  `;
321
398
 
@@ -350,6 +427,7 @@ def create_chat_app() -> Flask:
350
427
  async function sendMessage() {
351
428
  const input = document.getElementById('messageInput');
352
429
  const sendButton = document.getElementById('sendButton');
430
+ const container = document.getElementById('messagesContainer');
353
431
  const message = input.value.trim();
354
432
 
355
433
  if ((!message && selectedImages.length === 0) || isStreaming) return;
@@ -363,7 +441,9 @@ def create_chat_app() -> Flask:
363
441
  selectedImages = [];
364
442
  updateImagePreview();
365
443
 
366
- addMessageCard('user', message, null, currentImages);
444
+ const userSendTime = Date.now();
445
+ const timeSinceLastResponse = lastMessageTimestamp !== null ? userSendTime - lastMessageTimestamp : null;
446
+ addMessageCard('user', message, null, currentImages, userSendTime, timeSinceLastResponse);
367
447
 
368
448
  const assistantCard = addMessageCard('assistant', '');
369
449
  const contentDiv = assistantCard.querySelector('.message-content');
@@ -402,14 +482,19 @@ def create_chat_app() -> Flask:
402
482
  let fullToolName = '';
403
483
  let fullToolArgs = '';
404
484
  let metadata = null;
485
+ let lastCreatedAt = null;
486
+ let buffer = '';
405
487
 
406
488
  while (true) {
407
489
  const { done, value } = await reader.read();
408
490
  if (done) break;
409
491
 
410
492
  const chunk = decoder.decode(value);
411
- const lines = chunk.split('\\n');
493
+ buffer += chunk;
494
+ if (!buffer.endsWith('\\n\\n')) continue;
412
495
 
496
+ const lines = buffer.split('\\n');
497
+ buffer = '';
413
498
  for (const line of lines) {
414
499
  if (line.startsWith('data: ')) {
415
500
  const data = line.slice(6);
@@ -439,6 +524,9 @@ def create_chat_app() -> Flask:
439
524
  contentDiv.insertBefore(thinkingContainer, textContainer || contentDiv.firstChild);
440
525
  }
441
526
  thinkingContainer.textContent = `💭 ${fullThinking}`;
527
+ } else if (item.type === 'inline_thinking') {
528
+ // Ignore thinking inline data
529
+ continue;
442
530
  } else if (item.type === 'partial_tool_call') {
443
531
  fullToolName += item.name || '';
444
532
  fullToolArgs += item.arguments || '';
@@ -454,9 +542,16 @@ def create_chat_app() -> Flask:
454
542
  toolResultDiv.className = 'bg-green-50 p-3 rounded-md border-l-4 border-green-500 mb-2';
455
543
  toolResultDiv.innerHTML = `<strong class="text-sm">✅ Tool Result:</strong><br><div class="mt-1 text-xs whitespace-pre-wrap">${escapeHtml(item.text)}</div>`;
456
544
  contentDiv.appendChild(toolResultDiv);
545
+ } else if (item.type === 'inline_data') {
546
+ const inlineDataDiv = document.createElement('div');
547
+ inlineDataDiv.innerHTML = renderInlineData(item);
548
+ if (inlineDataDiv.firstChild) {
549
+ contentDiv.appendChild(inlineDataDiv.firstChild);
550
+ }
457
551
  }
458
552
  }
459
553
 
554
+ container.scrollTop = container.scrollHeight;
460
555
  if (event.usage_metadata) {
461
556
  const usage = event.usage_metadata;
462
557
  const inputTokens = (usage.cached_tokens || 0) + (usage.prompt_tokens || 0);
@@ -474,6 +569,9 @@ def create_chat_app() -> Flask:
474
569
  metadata = metadata || {};
475
570
  metadata.finish_reason = event.finish_reason;
476
571
  }
572
+ if (event.created_at) {
573
+ lastCreatedAt = event.created_at;
574
+ }
477
575
  } catch (e) {
478
576
  console.error('Error parsing event:', e);
479
577
  }
@@ -481,6 +579,21 @@ def create_chat_app() -> Flask:
481
579
  }
482
580
  }
483
581
 
582
+ if (lastCreatedAt) {
583
+ const timestampEl = assistantCard.querySelector('.msg-timestamp');
584
+ if (timestampEl) {
585
+ timestampEl.textContent = formatTimestamp(lastCreatedAt);
586
+ }
587
+ }
588
+
589
+ const endTime = Date.now();
590
+ const responseTimeMs = endTime - userSendTime;
591
+ lastMessageTimestamp = endTime;
592
+ const tookEl = assistantCard.querySelector('.msg-took');
593
+ if (tookEl) {
594
+ tookEl.textContent = `Took ${responseTimeMs} ms`;
595
+ }
596
+
484
597
  if (metadata) {
485
598
  let metadataHtml = '<div class="flex justify-end gap-3 mt-3 pt-3 border-t border-gray-200 text-xs text-gray-500">';
486
599
  const parts = [];
@@ -500,6 +613,7 @@ def create_chat_app() -> Flask:
500
613
  } catch (error) {
501
614
  contentDiv.textContent = `Error: ${error.message}`;
502
615
  console.error('Error:', error);
616
+ lastMessageTimestamp = Date.now();
503
617
  }
504
618
 
505
619
  isStreaming = false;
@@ -518,6 +632,7 @@ def create_chat_app() -> Flask:
518
632
  })
519
633
  }).then(() => {
520
634
  sessionId = Math.random().toString(36).substring(7);
635
+ lastMessageTimestamp = null;
521
636
  const container = document.getElementById('messagesContainer');
522
637
  container.innerHTML = `
523
638
  <div class="text-center text-gray-500 py-10">
@@ -543,6 +658,7 @@ def create_chat_app() -> Flask:
543
658
  this.style.height = 'auto';
544
659
  this.style.height = Math.min(this.scrollHeight, 200) + 'px';
545
660
  });
661
+
546
662
  </script>
547
663
  </body>
548
664
  </html>
@@ -569,7 +685,7 @@ def create_chat_app() -> Flask:
569
685
  try:
570
686
  # Get or create client for this session
571
687
  if session_id not in _session_clients:
572
- model = config.get("model") or "gpt-5.2"
688
+ model = config.get("model") or "gpt-5.5"
573
689
  _session_clients[session_id] = AutoLLMClient(model=model)
574
690
 
575
691
  client = _session_clients[session_id]