monkeybot 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (178) hide show
  1. monkeybot/__init__.py +3 -0
  2. monkeybot/cli/__init__.py +3 -0
  3. monkeybot/cli/__main__.py +8 -0
  4. monkeybot/cli/audio_io.py +8 -0
  5. monkeybot/cli/gateway_manager.py +17 -0
  6. monkeybot/cli/main.py +22 -0
  7. monkeybot/cli/push_to_talk.py +12 -0
  8. monkeybot/cli/realtime_client.py +13 -0
  9. monkeybot/core/__init__.py +19 -0
  10. monkeybot/core/attachments/__init__.py +22 -0
  11. monkeybot/core/attachments/catalog.py +62 -0
  12. monkeybot/core/attachments/config.py +52 -0
  13. monkeybot/core/attachments/freeze.py +158 -0
  14. monkeybot/core/attachments/resolve.py +70 -0
  15. monkeybot/core/attachments/store.py +180 -0
  16. monkeybot/core/attachments/text.py +72 -0
  17. monkeybot/core/attachments/tools.py +54 -0
  18. monkeybot/core/bootstrap.py +242 -0
  19. monkeybot/core/config/__init__.py +71 -0
  20. monkeybot/core/config/realtime_config.py +150 -0
  21. monkeybot/core/config/runtime_env.py +262 -0
  22. monkeybot/core/config/settings.py +341 -0
  23. monkeybot/core/config/validation.py +249 -0
  24. monkeybot/core/config/yaml_loader.py +45 -0
  25. monkeybot/core/context/__init__.py +781 -0
  26. monkeybot/core/context/campaign_context.py +8 -0
  27. monkeybot/core/context/common.py +14 -0
  28. monkeybot/core/context/curator.py +255 -0
  29. monkeybot/core/context/epoch.py +226 -0
  30. monkeybot/core/context/memory_prompt.py +222 -0
  31. monkeybot/core/context/tool_output_policy.py +270 -0
  32. monkeybot/core/context/tool_result_ingress.py +290 -0
  33. monkeybot/core/context/tool_shapers.py +361 -0
  34. monkeybot/core/hooks/__init__.py +261 -0
  35. monkeybot/core/llm/__init__.py +4 -0
  36. monkeybot/core/llm/provider.py +296 -0
  37. monkeybot/core/llm/realtime_provider.py +203 -0
  38. monkeybot/core/llm/usage.py +57 -0
  39. monkeybot/core/logging_utils.py +24 -0
  40. monkeybot/core/mcp/__init__.py +1 -0
  41. monkeybot/core/mcp/mcp_client.py +1215 -0
  42. monkeybot/core/mcp/ports_mcp.py +109 -0
  43. monkeybot/core/memory/__init__.py +24 -0
  44. monkeybot/core/memory/hook.py +413 -0
  45. monkeybot/core/memory/index_format.py +104 -0
  46. monkeybot/core/memory/integrity.py +180 -0
  47. monkeybot/core/memory/organizer.py +270 -0
  48. monkeybot/core/memory/storage_ops.py +139 -0
  49. monkeybot/core/memory/subsystem.py +91 -0
  50. monkeybot/core/messages/__init__.py +16 -0
  51. monkeybot/core/messages/convert_provider.py +41 -0
  52. monkeybot/core/messages/tool_integrity.py +262 -0
  53. monkeybot/core/messages/transform_context.py +84 -0
  54. monkeybot/core/path_safety.py +11 -0
  55. monkeybot/core/persistence/__init__.py +17 -0
  56. monkeybot/core/persistence/backends.py +236 -0
  57. monkeybot/core/persistence/db.py +28 -0
  58. monkeybot/core/persistence/durable_runs.py +286 -0
  59. monkeybot/core/persistence/firestore.py +658 -0
  60. monkeybot/core/persistence/firestore_scheduled_loops.py +336 -0
  61. monkeybot/core/persistence/history.py +156 -0
  62. monkeybot/core/persistence/postgres.py +895 -0
  63. monkeybot/core/persistence/runs.py +76 -0
  64. monkeybot/core/persistence/scheduled_loops.py +435 -0
  65. monkeybot/core/persistence/session_turn_locks.py +94 -0
  66. monkeybot/core/persistence/sqlite.py +218 -0
  67. monkeybot/core/persistence/sqlite_backend.py +74 -0
  68. monkeybot/core/persistence/thread_summary.py +61 -0
  69. monkeybot/core/persistence/transcript.py +194 -0
  70. monkeybot/core/persistence/usage.py +149 -0
  71. monkeybot/core/prompts/__init__.py +1 -0
  72. monkeybot/core/prompts/harness_prompt.py +197 -0
  73. monkeybot/core/prompts/prompt.py +215 -0
  74. monkeybot/core/runtime/__init__.py +1 -0
  75. monkeybot/core/runtime/context_budget.py +267 -0
  76. monkeybot/core/runtime/events.py +819 -0
  77. monkeybot/core/runtime/input_admission.py +154 -0
  78. monkeybot/core/runtime/loop.py +2374 -0
  79. monkeybot/core/runtime/provider_stream_mapper.py +159 -0
  80. monkeybot/core/runtime/realtime_loop.py +654 -0
  81. monkeybot/core/runtime/utterance_buffer.py +179 -0
  82. monkeybot/core/subagents/__init__.py +1 -0
  83. monkeybot/core/subagents/subagent_proto.py +331 -0
  84. monkeybot/core/subagents/subagent_worker.py +441 -0
  85. monkeybot/core/subagents/worker_pool.py +403 -0
  86. monkeybot/core/testing/__init__.py +1 -0
  87. monkeybot/core/testing/mocks_provider.py +86 -0
  88. monkeybot/core/testing/mocks_realtime_provider.py +137 -0
  89. monkeybot/core/tools/__init__.py +1 -0
  90. monkeybot/core/tools/core_tool_executor.py +1548 -0
  91. monkeybot/core/tools/inspector.py +226 -0
  92. monkeybot/core/tools/loop_inspector.py +45 -0
  93. monkeybot/core/tools/patch.py +480 -0
  94. monkeybot/core/tools/permission.py +284 -0
  95. monkeybot/core/tools/sandbox_executor.py +255 -0
  96. monkeybot/core/tools/spill_inventory.py +35 -0
  97. monkeybot/core/tools/terminal.py +381 -0
  98. monkeybot/core/tools/text_normalize.py +25 -0
  99. monkeybot/core/tools/types.py +33 -0
  100. monkeybot/core/tools/workspace_service.py +710 -0
  101. monkeybot/core/tools/workspace_tools.py +116 -0
  102. monkeybot/core/types/__init__.py +1 -0
  103. monkeybot/core/types/content_blocks.py +644 -0
  104. monkeybot/core/types/interfaces.py +156 -0
  105. monkeybot/core/types/types_tools.py +29 -0
  106. monkeybot/core/workspace/__init__.py +8 -0
  107. monkeybot/core/workspace/factory.py +45 -0
  108. monkeybot/core/workspace/gcs.py +130 -0
  109. monkeybot/core/workspace/local.py +162 -0
  110. monkeybot/core/workspace/protocol.py +45 -0
  111. monkeybot/core/workspace/s3.py +151 -0
  112. monkeybot/core/workspace_layout.py +27 -0
  113. monkeybot/gateway/__init__.py +1 -0
  114. monkeybot/gateway/bootstrap.py +18 -0
  115. monkeybot/gateway/main.py +47 -0
  116. monkeybot/gateway/realtime/__init__.py +31 -0
  117. monkeybot/gateway/realtime/app.py +321 -0
  118. monkeybot/gateway/realtime/deps.py +52 -0
  119. monkeybot/gateway/realtime/errors.py +81 -0
  120. monkeybot/gateway/realtime/guardrails.py +88 -0
  121. monkeybot/gateway/realtime/manager.py +77 -0
  122. monkeybot/gateway/realtime/metrics.py +144 -0
  123. monkeybot/gateway/realtime/routes.py +864 -0
  124. monkeybot/gateway/realtime/session.py +232 -0
  125. monkeybot/gateway/realtime/wire.py +412 -0
  126. monkeybot/gateway/realtime_main.py +49 -0
  127. monkeybot/gateway/sse/__init__.py +1 -0
  128. monkeybot/gateway/sse/app.py +733 -0
  129. monkeybot/gateway/sse/loop_port.py +31 -0
  130. monkeybot/gateway/sse/models.py +177 -0
  131. monkeybot/gateway/sse/reply_body.py +91 -0
  132. monkeybot/gateway/sse/routes.py +1101 -0
  133. monkeybot/gateway/sse/scheduler_routes.py +200 -0
  134. monkeybot/gateway/sse/scheduler_wiring.py +96 -0
  135. monkeybot/gateway/sse/session_bus.py +226 -0
  136. monkeybot/gateway/sse/sse.py +46 -0
  137. monkeybot/gateway/sse/workspace_layout.py +7 -0
  138. monkeybot/observability/__init__.py +220 -0
  139. monkeybot/observability/_state.py +10 -0
  140. monkeybot/observability/instrumentation.py +153 -0
  141. monkeybot/observability/propagation.py +65 -0
  142. monkeybot/observability/spans.py +455 -0
  143. monkeybot/providers/__init__.py +19 -0
  144. monkeybot/providers/_openai_compat.py +450 -0
  145. monkeybot/providers/_utils.py +473 -0
  146. monkeybot/providers/bedrock.py +145 -0
  147. monkeybot/providers/claude.py +125 -0
  148. monkeybot/providers/gemini.py +677 -0
  149. monkeybot/providers/gemini_live.py +398 -0
  150. monkeybot/providers/huggingface.py +129 -0
  151. monkeybot/providers/nvidia.py +104 -0
  152. monkeybot/providers/ollama.py +152 -0
  153. monkeybot/providers/openai.py +127 -0
  154. monkeybot/providers/pricing.py +60 -0
  155. monkeybot/providers/sampling.py +44 -0
  156. monkeybot/providers/vertex_claude.py +148 -0
  157. monkeybot/scaffold/__init__.py +33 -0
  158. monkeybot/scheduler/__init__.py +13 -0
  159. monkeybot/scheduler/__main__.py +4 -0
  160. monkeybot/scheduler/engine.py +333 -0
  161. monkeybot/scheduler/http_invoker.py +61 -0
  162. monkeybot/scheduler/interval.py +77 -0
  163. monkeybot/scheduler/tick_result.py +34 -0
  164. monkeybot/scheduler/worker.py +87 -0
  165. monkeybot/subagents/__init__.py +1 -0
  166. monkeybot/subagents/worker/__init__.py +1 -0
  167. monkeybot/subagents/worker/__main__.py +22 -0
  168. monkeybot/web_search/__init__.py +82 -0
  169. monkeybot/web_search/backends/__init__.py +5 -0
  170. monkeybot/web_search/backends/duckduckgo.py +32 -0
  171. monkeybot/web_search/backends/firecrawl.py +43 -0
  172. monkeybot/web_search/backends/tavily.py +45 -0
  173. monkeybot/web_search/protocol.py +25 -0
  174. monkeybot/web_search/tool.py +56 -0
  175. monkeybot-2.1.1.dist-info/METADATA +318 -0
  176. monkeybot-2.1.1.dist-info/RECORD +178 -0
  177. monkeybot-2.1.1.dist-info/WHEEL +4 -0
  178. monkeybot-2.1.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,677 @@
1
+ """Vertex Gemini streaming via the official ``google-genai`` SDK (no LangChain in this module)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ import os
8
+ from collections.abc import AsyncIterator, Sequence
9
+ from typing import Any
10
+
11
+ from monkeybot.core.llm.provider import (
12
+ Done,
13
+ GroundingEvent,
14
+ Message,
15
+ ProviderEvent,
16
+ TextDelta,
17
+ ThinkingDelta,
18
+ ToolCall,
19
+ UsageEvent,
20
+ )
21
+ from monkeybot.core.logging_utils import kv
22
+ from monkeybot.core.types.content_blocks import (
23
+ File,
24
+ Image,
25
+ Text,
26
+ Thinking,
27
+ ToolRequest,
28
+ ToolResponse,
29
+ )
30
+ from monkeybot.core.types.interfaces import LLMError
31
+ from monkeybot.core.types.types_tools import ToolDef
32
+ from monkeybot.providers.sampling import resolve_model_sampling
33
+
34
+ THOUGHT_SIGNATURE_KEY = "thoughtSignature"
35
+ SYNTHETIC_THOUGHT_SIGNATURE = "skip_thought_signature_validator"
36
+
37
+ _log = logging.getLogger(__name__)
38
+
39
+ # Keep this in sync when onboarding another Vertex Gemini model that must use "global".
40
+ _GLOBAL_VERTEX_MODEL_IDS = frozenset({"gemini-3-flash-preview"})
41
+
42
+
43
+ def _normalize_vertex_model(model: str) -> str:
44
+ """Return a ``model`` value accepted by ``google.genai`` for Vertex (see SDK docstring).
45
+
46
+ Accepts bare ids (``gemini-2.5-flash``), ``models/...``, ``publishers/...``, ``google/...``,
47
+ or a full ``projects/.../locations/.../publishers/google/models/...`` resource name.
48
+ Strips a mistaken ``models/`` prefix when using the Vertex client (Vertex uses bare ids
49
+ or publisher paths, not the AI Studio ``models/`` prefix).
50
+ """
51
+ m = (model or "").strip()
52
+ if not m:
53
+ raise LLMError("MODEL_NAME (model parameter) is empty.")
54
+ if m.startswith("projects/"):
55
+ return m
56
+ if m.startswith("publishers/") or m.startswith("google/"):
57
+ return m
58
+ if m.startswith("models/"):
59
+ return m[len("models/") :].strip()
60
+ return m
61
+
62
+
63
+ def _location_from_full_vertex_model(model: str) -> str | None:
64
+ """If ``model`` is a full Vertex resource, return the ``locations/{loc}`` segment."""
65
+ if not model.startswith("projects/") or "/locations/" not in model:
66
+ return None
67
+ try:
68
+ idx = model.index("/locations/") + len("/locations/")
69
+ return model[idx:].split("/", 1)[0].strip() or None
70
+ except (ValueError, IndexError):
71
+ return None
72
+
73
+
74
+ def _vertex_project_and_location(model_param: str) -> tuple[str, str]:
75
+ """Resolve project id and API location for ``genai.Client(vertexai=True, ...)``.
76
+
77
+ Some preview model ids are not published under regional
78
+ endpoints like ``us-central1``; Vertex serves them from ``global`` unless you override
79
+ ``VERTEX_AI_LOCATION`` / ``GOOGLE_CLOUD_LOCATION``.
80
+ """
81
+ project = (
82
+ os.environ.get("GCP_PROJECT_ID")
83
+ or os.environ.get("VERTEX_AI_PROJECT_ID")
84
+ or os.environ.get("GOOGLE_CLOUD_PROJECT")
85
+ )
86
+ if not project or not str(project).strip():
87
+ raise LLMError(
88
+ "Set VERTEX_AI_PROJECT_ID, GCP_PROJECT_ID, or GOOGLE_CLOUD_PROJECT for Vertex Gemini."
89
+ )
90
+ explicit = os.environ.get("VERTEX_AI_LOCATION") or os.environ.get("GOOGLE_CLOUD_LOCATION")
91
+ if explicit and str(explicit).strip():
92
+ return str(project).strip(), str(explicit).strip()
93
+
94
+ embedded = _location_from_full_vertex_model(model_param)
95
+ if embedded:
96
+ return str(project).strip(), embedded
97
+
98
+ tail = model_param.split("/")[-1]
99
+ if tail.lower() in _GLOBAL_VERTEX_MODEL_IDS:
100
+ return str(project).strip(), "global"
101
+
102
+ return str(project).strip(), "us-central1"
103
+
104
+
105
+ _THINKING_DISABLED = 0
106
+ """Sentinel passed as ``thinking_budget`` to explicitly disable Gemini extended thinking.
107
+
108
+ Distinct from ``-1`` (the "omit ThinkingConfig" default) so auxiliary calls can
109
+ send ``ThinkingConfig(thinking_budget=0)`` to the API, which instructs the server
110
+ not to think rather than leaving it to the server's default for the model.
111
+ """
112
+
113
+
114
+ def _suppress_thinking_for_auxiliary_call(
115
+ messages: Sequence[Message],
116
+ tools: Sequence[ToolDef],
117
+ ) -> bool:
118
+ """True when this call is a known auxiliary job that must not use extended thinking.
119
+
120
+ Context curation and history summarization use ``tools=()`` and a fixed 2-message
121
+ shape. Extended thinking on preview models can stall the stream for 10s+ with no
122
+ visible output, burning the curator timeout before ``Done`` is received.
123
+ """
124
+ if tools:
125
+ return False
126
+ if len(messages) != 2 or messages[0].role != "system":
127
+ return False
128
+ sys_txt = "".join(b.text for b in messages[0].content if isinstance(b, Text))
129
+ markers = (
130
+ "You narrow context for another assistant",
131
+ "You compress prior agent conversation turns",
132
+ )
133
+ return any(m in sys_txt for m in markers)
134
+
135
+
136
+ def _resolve_thinking_budget(
137
+ configured: int | None,
138
+ *,
139
+ override: int | None,
140
+ messages: Sequence[Message],
141
+ tools: Sequence[ToolDef],
142
+ ) -> int:
143
+ if override is not None:
144
+ thinking_budget = int(override)
145
+ elif configured is not None:
146
+ thinking_budget = int(configured)
147
+ else:
148
+ thinking_budget = int(os.environ.get("MODEL_THINKING_BUDGET", "-1"))
149
+ if _suppress_thinking_for_auxiliary_call(messages, tools):
150
+ return _THINKING_DISABLED
151
+ return thinking_budget
152
+
153
+
154
+ def _split_system_and_rest(messages: Sequence[Message]) -> tuple[str, list[Message]]:
155
+ systems: list[str] = []
156
+ rest: list[Message] = []
157
+ for m in messages:
158
+ if m.role == "system":
159
+ texts = [b.text for b in m.content if isinstance(b, Text)]
160
+ systems.append("\n\n".join(texts))
161
+ else:
162
+ rest.append(m)
163
+ joined = "\n\n".join(s for s in systems if s).strip()
164
+ return joined, rest
165
+
166
+
167
+ def _flatten_tool_response_result(block: ToolResponse) -> str:
168
+ parts: list[str] = []
169
+ for b in block.result:
170
+ if isinstance(b, Text):
171
+ parts.append(b.text)
172
+ elif isinstance(b, (Image, File)):
173
+ continue
174
+ else:
175
+ raise LLMError(
176
+ "Cannot replay tool result to Vertex: unsupported result block "
177
+ f"{type(b).__name__}"
178
+ )
179
+ return "".join(parts)
180
+
181
+
182
+ def _media_parts_from_blocks(blocks: Sequence[object]) -> list[Any]:
183
+ import base64
184
+
185
+ from google.genai import types
186
+
187
+ parts: list[Any] = []
188
+ for b in blocks:
189
+ if isinstance(b, (Image, File)):
190
+ parts.append(
191
+ types.Part(
192
+ inline_data=types.Blob(
193
+ mime_type=b.mime_type,
194
+ data=base64.b64decode(b.data),
195
+ )
196
+ )
197
+ )
198
+ return parts
199
+
200
+
201
+ def _is_user_loop_boundary(message: Message) -> bool:
202
+ """A user turn that carries at least one non-ToolResponse block opens an active loop."""
203
+ if message.role != "user":
204
+ return False
205
+ return any(not isinstance(b, ToolResponse) for b in message.content)
206
+
207
+
208
+ def _active_loop_start_index(messages: Sequence[Message]) -> int | None:
209
+ """Return the index of the most recent user-loop boundary, or ``None``.
210
+
211
+ Goose's contract: only messages from this index onward must replay
212
+ ``thoughtSignature`` to satisfy Vertex Gemini's strict validation.
213
+ """
214
+ for i in range(len(messages) - 1, -1, -1):
215
+ if _is_user_loop_boundary(messages[i]):
216
+ return i
217
+ return None
218
+
219
+
220
+ def _signature_from_metadata(metadata: dict[str, object] | None) -> str | None:
221
+ if not metadata:
222
+ return None
223
+ sig = metadata.get(THOUGHT_SIGNATURE_KEY)
224
+ return sig if isinstance(sig, str) and sig else None
225
+
226
+
227
+ def _normalize_signature(value: Any) -> str | None:
228
+ """Coerce SDK ``thought_signature`` (bytes or str) to a non-empty Python string.
229
+
230
+ The google-genai pydantic model normalizes wire signatures to ``bytes``
231
+ using base64 decoding for string inputs. Round-tripping requires preserving
232
+ the original textual representation; we standardise on base64 strings when
233
+ we receive bytes that aren't valid UTF-8.
234
+ """
235
+ if value is None:
236
+ return None
237
+ if isinstance(value, bytes):
238
+ try:
239
+ decoded = value.decode("utf-8")
240
+ except UnicodeDecodeError:
241
+ import base64
242
+
243
+ decoded = base64.b64encode(value).decode("ascii")
244
+ return decoded or None
245
+ if isinstance(value, str):
246
+ return value or None
247
+ return None
248
+
249
+
250
+ def _messages_to_contents(rest: Sequence[Message]) -> list[Any]:
251
+ """Build ``google.genai.types.Content`` list from harness messages (no system rows).
252
+
253
+ Vertex Gemini 2.5+ enforces ``thoughtSignature`` round-trip on the active
254
+ conversation loop. This builder mirrors Goose's contract:
255
+
256
+ - Signatures are re-attached to ``functionCall`` and reasoning parts in the
257
+ active loop (from the last user-loop boundary onward).
258
+ - Earlier turns drop signatures (Vertex would reject them as stale).
259
+ - If the active loop's first model tool call has no captured signature, a
260
+ synthetic placeholder (``skip_thought_signature_validator``) is inserted
261
+ so Vertex skips strict validation rather than 400-ing.
262
+ - ``Thinking`` blocks are round-tripped as ``Part(text, thought=True,
263
+ thought_signature=...)`` only inside the active loop.
264
+ """
265
+ from google.genai import types
266
+
267
+ for m in rest:
268
+ if m.role != "user":
269
+ continue
270
+ for block in m.content:
271
+ if isinstance(block, ToolResponse) and not str(block.tool_name or "").strip():
272
+ raise LLMError(
273
+ "Cannot replay tool result to Vertex: empty tool_name "
274
+ f"(tool_call_id={block.id!r})."
275
+ )
276
+
277
+ active_start = _active_loop_start_index(rest)
278
+
279
+ contents: list[Any] = []
280
+ for idx, m in enumerate(rest):
281
+ gemini_role = "user" if m.role == "user" else "model"
282
+ in_active_loop = active_start is not None and idx >= active_start
283
+ needs_synthetic_for_first_model_tool_call = in_active_loop and m.role != "user"
284
+ parts: list[Any] = []
285
+ for block in m.content:
286
+ if isinstance(block, Text):
287
+ parts.append(types.Part(text=block.text))
288
+ elif isinstance(block, (Image, File)):
289
+ parts.extend(_media_parts_from_blocks([block]))
290
+ elif isinstance(block, Thinking):
291
+ if not in_active_loop:
292
+ continue
293
+ kwargs: dict[str, Any] = {"text": block.thinking, "thought": True}
294
+ if block.signature:
295
+ kwargs["thought_signature"] = block.signature.encode("utf-8")
296
+ parts.append(types.Part(**kwargs))
297
+ elif isinstance(block, ToolRequest):
298
+ fc_kwargs: dict[str, Any] = {
299
+ "name": block.name,
300
+ "args": dict(block.args),
301
+ "id": block.id,
302
+ }
303
+ part_kwargs: dict[str, Any] = {
304
+ "function_call": types.FunctionCall(**fc_kwargs),
305
+ }
306
+ if in_active_loop:
307
+ sig = _signature_from_metadata(block.metadata)
308
+ if sig is None and needs_synthetic_for_first_model_tool_call:
309
+ sig = SYNTHETIC_THOUGHT_SIGNATURE
310
+ if sig is not None:
311
+ # SDK validator base64-decodes string inputs; pass raw bytes
312
+ # so the literal signature survives the round-trip.
313
+ part_kwargs["thought_signature"] = sig.encode("utf-8")
314
+ needs_synthetic_for_first_model_tool_call = False
315
+ parts.append(types.Part(**part_kwargs))
316
+ elif isinstance(block, ToolResponse):
317
+ parts.append(
318
+ types.Part(
319
+ function_response=types.FunctionResponse(
320
+ name=block.tool_name,
321
+ response={"result": _flatten_tool_response_result(block)},
322
+ )
323
+ )
324
+ )
325
+ parts.extend(_media_parts_from_blocks(block.result))
326
+ else:
327
+ raise LLMError(
328
+ "Cannot replay message to Vertex Gemini: unsupported block "
329
+ f"{type(block).__name__}"
330
+ )
331
+ if not parts:
332
+ parts.append(types.Part(text=""))
333
+ contents.append(types.Content(role=gemini_role, parts=parts))
334
+ return contents
335
+
336
+
337
+ def _tool_defs_to_declarations(tools: Sequence[ToolDef]) -> list[Any]:
338
+ from google.genai import types
339
+
340
+ out: list[Any] = []
341
+ for t in tools:
342
+ schema = dict(t.input_schema) if t.input_schema else {"type": "object"}
343
+ out.append(
344
+ types.FunctionDeclaration(
345
+ name=t.name,
346
+ description=t.description or "",
347
+ parameters_json_schema=schema,
348
+ )
349
+ )
350
+ return out
351
+
352
+
353
+ def _grounding_metadata_to_dict(gm: Any) -> dict[str, Any] | None:
354
+ """Flatten Vertex ``GroundingMetadata`` into a small, wire-friendly dict.
355
+
356
+ Only carries what a UI needs for citations/search-suggestion display: source
357
+ chunks (title/uri) and web search queries. Returns ``None`` when there is
358
+ nothing worth surfacing.
359
+ """
360
+ if gm is None:
361
+ return None
362
+ chunks: list[dict[str, str]] = []
363
+ for chunk in getattr(gm, "grounding_chunks", None) or []:
364
+ web = getattr(chunk, "web", None)
365
+ if web is None:
366
+ continue
367
+ title = str(getattr(web, "title", "") or "")
368
+ uri = str(getattr(web, "uri", "") or "")
369
+ if title or uri:
370
+ chunks.append({"title": title, "uri": uri})
371
+ queries = [str(q) for q in (getattr(gm, "web_search_queries", None) or [])]
372
+ if not chunks and not queries:
373
+ return None
374
+ out: dict[str, Any] = {}
375
+ if chunks:
376
+ out["sources"] = chunks
377
+ if queries:
378
+ out["search_queries"] = queries
379
+ return out
380
+
381
+
382
+ def _merge_function_call_args(existing: dict[str, Any], fc: Any) -> dict[str, Any]:
383
+ merged = dict(existing)
384
+ if isinstance(getattr(fc, "args", None), dict):
385
+ merged.update(fc.args)
386
+ partial = getattr(fc, "partial_args", None)
387
+ if isinstance(partial, dict):
388
+ merged.update(partial)
389
+ elif isinstance(partial, str) and partial.strip():
390
+ try: # noqa: SIM105 — preserve streaming merge semantics (spec: verbatim)
391
+ merged.update(json.loads(partial))
392
+ except json.JSONDecodeError:
393
+ pass
394
+ return merged
395
+
396
+
397
+ def _usage_from_response(um: Any) -> UsageEvent | None:
398
+ if um is None:
399
+ return None
400
+ inp = int(getattr(um, "prompt_token_count", 0) or 0)
401
+ out = int(getattr(um, "candidates_token_count", 0) or 0)
402
+ # Vertex Gemini implicit caching reports a single read count; there is no
403
+ # separate cache-creation cost, so creation is 0 and the total equals the read.
404
+ cache_read = int(getattr(um, "cached_content_token_count", 0) or 0)
405
+ return UsageEvent(
406
+ input_tokens=inp,
407
+ output_tokens=out,
408
+ cached_tokens=cache_read,
409
+ cache_read_tokens=cache_read,
410
+ cache_creation_tokens=0,
411
+ )
412
+
413
+
414
+ class GeminiProvider:
415
+ def __init__(
416
+ self,
417
+ *,
418
+ supports_streaming: bool = True,
419
+ temperature: float | None = None,
420
+ max_tokens: int | None = None,
421
+ max_output_tokens: int | None = None,
422
+ thinking_budget: int | None = None,
423
+ api_key: str | None = None,
424
+ ) -> None:
425
+ """Gemini streaming provider (Vertex AI or Google AI Studio).
426
+
427
+ When ``api_key`` is provided, the provider connects to Google AI Studio
428
+ (``generativelanguage.googleapis.com``). Otherwise it connects to Vertex AI
429
+ using application default credentials or the project/location environment variables.
430
+
431
+ ``max_output_tokens`` is a backward-compatible alias for ``max_tokens``.
432
+
433
+ Native ``google_search`` grounding is opt-in per call via the
434
+ ``vertex_google_search`` keyword on :meth:`stream` and
435
+ :meth:`count_input_tokens` (Gemini-only; the runtime loop passes it only
436
+ when ``provider.name == \"gemini\"``).
437
+ """
438
+ if max_output_tokens is not None and max_tokens is not None and max_output_tokens != max_tokens:
439
+ raise ValueError("pass only one of max_tokens or max_output_tokens")
440
+ effective_max_tokens = max_tokens if max_tokens is not None else max_output_tokens
441
+ self._supports_streaming = supports_streaming
442
+ sampling = resolve_model_sampling(temperature=temperature, max_tokens=effective_max_tokens)
443
+ self._temperature = sampling.temperature
444
+ self._max_tokens = sampling.max_tokens
445
+ self._thinking_budget = thinking_budget
446
+ self._api_key = api_key
447
+
448
+ def _client(self, model_param: str) -> Any:
449
+ """Create a google-genai client for Vertex AI or Google AI Studio."""
450
+ from google import genai
451
+
452
+ if self._api_key:
453
+ return genai.Client(api_key=self._api_key)
454
+ project, location = _vertex_project_and_location(model_param)
455
+ return genai.Client(vertexai=True, project=project, location=location)
456
+
457
+ @property
458
+ def name(self) -> str:
459
+ return "gemini"
460
+
461
+ @property
462
+ def supports_streaming(self) -> bool:
463
+ return self._supports_streaming
464
+
465
+ async def count_input_tokens(
466
+ self,
467
+ messages: Sequence[Message],
468
+ tools: Sequence[ToolDef],
469
+ *,
470
+ model: str,
471
+ thinking_budget: int | None = None,
472
+ vertex_google_search: bool = False,
473
+ ) -> int:
474
+ model_param = _normalize_vertex_model(model)
475
+ try:
476
+ from google.genai import types
477
+ except ImportError as exc:
478
+ raise LLMError(
479
+ "google-genai is required for GeminiProvider. Install with: uv sync (monkeybot dependencies)."
480
+ ) from exc
481
+
482
+ temperature = float(self._temperature)
483
+ max_tokens = int(self._max_tokens)
484
+ thinking_budget = _resolve_thinking_budget(
485
+ self._thinking_budget,
486
+ override=thinking_budget,
487
+ messages=messages,
488
+ tools=tools,
489
+ )
490
+
491
+ system_instruction, rest = _split_system_and_rest(messages)
492
+ contents = _messages_to_contents(rest)
493
+ decls = _tool_defs_to_declarations(tools)
494
+
495
+ count_cfg_kwargs: dict[str, Any] = {}
496
+ if system_instruction:
497
+ count_cfg_kwargs["system_instruction"] = system_instruction
498
+ count_tools: list[Any] = []
499
+ if decls:
500
+ count_tools.append(types.Tool(function_declarations=decls))
501
+ if vertex_google_search:
502
+ count_tools.append(types.Tool(google_search=types.GoogleSearch()))
503
+ if count_tools:
504
+ count_cfg_kwargs["tools"] = count_tools
505
+
506
+ gen_cfg_kwargs: dict[str, Any] = {
507
+ "temperature": temperature,
508
+ "max_output_tokens": max_tokens,
509
+ }
510
+ if thinking_budget != -1:
511
+ gen_cfg_kwargs["thinking_config"] = types.ThinkingConfig(thinking_budget=thinking_budget)
512
+ count_cfg_kwargs["generation_config"] = types.GenerationConfig(**gen_cfg_kwargs)
513
+
514
+ ct_cfg = types.CountTokensConfig(**count_cfg_kwargs)
515
+ client = self._client(model_param)
516
+ resp = await client.aio.models.count_tokens(
517
+ model=model_param,
518
+ contents=contents,
519
+ config=ct_cfg,
520
+ )
521
+ return int(resp.total_tokens or 0)
522
+
523
+ async def stream(
524
+ self,
525
+ messages: Sequence[Message],
526
+ tools: Sequence[ToolDef],
527
+ *,
528
+ model: str,
529
+ thinking_budget: int | None = None,
530
+ vertex_google_search: bool = False,
531
+ ) -> AsyncIterator[ProviderEvent]:
532
+ model_param = _normalize_vertex_model(model)
533
+ try:
534
+ from google.genai import types
535
+ except ImportError as exc:
536
+ raise LLMError(
537
+ "google-genai is required for GeminiProvider. Install with: uv sync (monkeybot dependencies)."
538
+ ) from exc
539
+
540
+ temperature = float(self._temperature)
541
+ max_tokens = int(self._max_tokens)
542
+ thinking_budget = _resolve_thinking_budget(
543
+ self._thinking_budget,
544
+ override=thinking_budget,
545
+ messages=messages,
546
+ tools=tools,
547
+ )
548
+
549
+ system_instruction, rest = _split_system_and_rest(messages)
550
+ contents = _messages_to_contents(rest)
551
+
552
+ cfg_kwargs: dict[str, Any] = {
553
+ "temperature": temperature,
554
+ "max_output_tokens": max_tokens,
555
+ }
556
+ if system_instruction:
557
+ cfg_kwargs["system_instruction"] = system_instruction
558
+ # -1 means "omit ThinkingConfig entirely" (model default).
559
+ # 0 means explicitly disable thinking (sent to API).
560
+ # Any other positive value sets a token budget.
561
+ if thinking_budget != -1:
562
+ cfg_kwargs["thinking_config"] = types.ThinkingConfig(thinking_budget=thinking_budget)
563
+
564
+ decls = _tool_defs_to_declarations(tools)
565
+ stream_tools: list[Any] = []
566
+ if decls:
567
+ stream_tools.append(types.Tool(function_declarations=decls))
568
+ if vertex_google_search:
569
+ stream_tools.append(types.Tool(google_search=types.GoogleSearch()))
570
+ if stream_tools:
571
+ cfg_kwargs["tools"] = stream_tools
572
+
573
+ config = types.GenerateContentConfig(**cfg_kwargs)
574
+
575
+ client = self._client(model_param)
576
+
577
+ pending_tools: dict[str, ToolCall] = {}
578
+ last_usage: Any = None
579
+ last_signature: str | None = None
580
+ grounding_meta: dict[str, Any] | None = None
581
+ truncated = False
582
+
583
+ try:
584
+ stream_it = await client.aio.models.generate_content_stream(
585
+ model=model_param,
586
+ contents=contents,
587
+ config=config,
588
+ )
589
+ async for resp in stream_it:
590
+ if getattr(resp, "usage_metadata", None) is not None:
591
+ last_usage = resp.usage_metadata
592
+
593
+ for cand in resp.candidates or []:
594
+ fr = getattr(cand, "finish_reason", None)
595
+ if fr is not None:
596
+ fr_name = str(getattr(fr, "name", fr)).upper()
597
+ if "MAX_TOKEN" in fr_name:
598
+ truncated = True
599
+ cand_gm = _grounding_metadata_to_dict(getattr(cand, "grounding_metadata", None))
600
+ if cand_gm is not None:
601
+ if grounding_meta is not None and cand_gm != grounding_meta:
602
+ _log.debug(
603
+ "grounding metadata replaced by later candidate %s",
604
+ kv(provider="gemini", model=model),
605
+ )
606
+ grounding_meta = cand_gm
607
+ content = getattr(cand, "content", None)
608
+ if content is None or not content.parts:
609
+ continue
610
+ for part in content.parts:
611
+ part_sig = _normalize_signature(getattr(part, "thought_signature", None))
612
+ if part_sig:
613
+ last_signature = part_sig
614
+
615
+ is_thought = bool(getattr(part, "thought", False))
616
+ if is_thought:
617
+ txt = getattr(part, "text", None) or ""
618
+ if txt:
619
+ yield ThinkingDelta(
620
+ text=txt,
621
+ signature=last_signature,
622
+ )
623
+ continue
624
+
625
+ if part.text:
626
+ yield TextDelta(text=part.text)
627
+
628
+ fc = part.function_call
629
+ if fc is None:
630
+ continue
631
+ name = str(getattr(fc, "name", "") or "")
632
+ if not name:
633
+ continue
634
+ fid = str(getattr(fc, "id", "") or "")
635
+ key = fid if fid else f"anon:{name}"
636
+ prev = pending_tools.get(key)
637
+ prev_args: dict[str, Any] = dict(prev.args) if prev else {}
638
+ merged_args = _merge_function_call_args(prev_args, fc)
639
+ call_id = fid if fid else key
640
+
641
+ # Per Goose: prefer the part's own signature, else carry forward
642
+ # from the most recent signed part in this stream.
643
+ effective_sig = part_sig or last_signature
644
+ prev_meta = dict(prev.metadata) if prev and prev.metadata else {}
645
+ if effective_sig:
646
+ prev_meta[THOUGHT_SIGNATURE_KEY] = effective_sig
647
+
648
+ pending_tools[key] = ToolCall(
649
+ call_id=call_id,
650
+ name=name,
651
+ args=merged_args,
652
+ metadata=prev_meta or None,
653
+ )
654
+ except LLMError:
655
+ raise
656
+ except Exception as exc:
657
+ _log.warning(
658
+ "Gemini stream error %s",
659
+ kv(provider="gemini", model=model, n_messages=len(messages), n_tools=len(tools)),
660
+ exc_info=True,
661
+ )
662
+ raise LLMError(str(exc)) from exc
663
+
664
+ if grounding_meta is not None:
665
+ yield GroundingEvent(
666
+ sources=list(grounding_meta.get("sources", [])),
667
+ search_queries=list(grounding_meta.get("search_queries", [])),
668
+ )
669
+
670
+ ev = _usage_from_response(last_usage)
671
+ if ev is not None:
672
+ yield ev
673
+
674
+ for call_id in sorted(pending_tools.keys()):
675
+ yield pending_tools[call_id]
676
+
677
+ yield Done(truncated=truncated)