monkeybot 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- monkeybot/__init__.py +3 -0
- monkeybot/cli/__init__.py +3 -0
- monkeybot/cli/__main__.py +8 -0
- monkeybot/cli/audio_io.py +8 -0
- monkeybot/cli/gateway_manager.py +17 -0
- monkeybot/cli/main.py +22 -0
- monkeybot/cli/push_to_talk.py +12 -0
- monkeybot/cli/realtime_client.py +13 -0
- monkeybot/core/__init__.py +19 -0
- monkeybot/core/attachments/__init__.py +22 -0
- monkeybot/core/attachments/catalog.py +62 -0
- monkeybot/core/attachments/config.py +52 -0
- monkeybot/core/attachments/freeze.py +158 -0
- monkeybot/core/attachments/resolve.py +70 -0
- monkeybot/core/attachments/store.py +180 -0
- monkeybot/core/attachments/text.py +72 -0
- monkeybot/core/attachments/tools.py +54 -0
- monkeybot/core/bootstrap.py +242 -0
- monkeybot/core/config/__init__.py +71 -0
- monkeybot/core/config/realtime_config.py +150 -0
- monkeybot/core/config/runtime_env.py +262 -0
- monkeybot/core/config/settings.py +341 -0
- monkeybot/core/config/validation.py +249 -0
- monkeybot/core/config/yaml_loader.py +45 -0
- monkeybot/core/context/__init__.py +781 -0
- monkeybot/core/context/campaign_context.py +8 -0
- monkeybot/core/context/common.py +14 -0
- monkeybot/core/context/curator.py +255 -0
- monkeybot/core/context/epoch.py +226 -0
- monkeybot/core/context/memory_prompt.py +222 -0
- monkeybot/core/context/tool_output_policy.py +270 -0
- monkeybot/core/context/tool_result_ingress.py +290 -0
- monkeybot/core/context/tool_shapers.py +361 -0
- monkeybot/core/hooks/__init__.py +261 -0
- monkeybot/core/llm/__init__.py +4 -0
- monkeybot/core/llm/provider.py +296 -0
- monkeybot/core/llm/realtime_provider.py +203 -0
- monkeybot/core/llm/usage.py +57 -0
- monkeybot/core/logging_utils.py +24 -0
- monkeybot/core/mcp/__init__.py +1 -0
- monkeybot/core/mcp/mcp_client.py +1215 -0
- monkeybot/core/mcp/ports_mcp.py +109 -0
- monkeybot/core/memory/__init__.py +24 -0
- monkeybot/core/memory/hook.py +413 -0
- monkeybot/core/memory/index_format.py +104 -0
- monkeybot/core/memory/integrity.py +180 -0
- monkeybot/core/memory/organizer.py +270 -0
- monkeybot/core/memory/storage_ops.py +139 -0
- monkeybot/core/memory/subsystem.py +91 -0
- monkeybot/core/messages/__init__.py +16 -0
- monkeybot/core/messages/convert_provider.py +41 -0
- monkeybot/core/messages/tool_integrity.py +262 -0
- monkeybot/core/messages/transform_context.py +84 -0
- monkeybot/core/path_safety.py +11 -0
- monkeybot/core/persistence/__init__.py +17 -0
- monkeybot/core/persistence/backends.py +236 -0
- monkeybot/core/persistence/db.py +28 -0
- monkeybot/core/persistence/durable_runs.py +286 -0
- monkeybot/core/persistence/firestore.py +658 -0
- monkeybot/core/persistence/firestore_scheduled_loops.py +336 -0
- monkeybot/core/persistence/history.py +156 -0
- monkeybot/core/persistence/postgres.py +895 -0
- monkeybot/core/persistence/runs.py +76 -0
- monkeybot/core/persistence/scheduled_loops.py +435 -0
- monkeybot/core/persistence/session_turn_locks.py +94 -0
- monkeybot/core/persistence/sqlite.py +218 -0
- monkeybot/core/persistence/sqlite_backend.py +74 -0
- monkeybot/core/persistence/thread_summary.py +61 -0
- monkeybot/core/persistence/transcript.py +194 -0
- monkeybot/core/persistence/usage.py +149 -0
- monkeybot/core/prompts/__init__.py +1 -0
- monkeybot/core/prompts/harness_prompt.py +197 -0
- monkeybot/core/prompts/prompt.py +215 -0
- monkeybot/core/runtime/__init__.py +1 -0
- monkeybot/core/runtime/context_budget.py +267 -0
- monkeybot/core/runtime/events.py +819 -0
- monkeybot/core/runtime/input_admission.py +154 -0
- monkeybot/core/runtime/loop.py +2374 -0
- monkeybot/core/runtime/provider_stream_mapper.py +159 -0
- monkeybot/core/runtime/realtime_loop.py +654 -0
- monkeybot/core/runtime/utterance_buffer.py +179 -0
- monkeybot/core/subagents/__init__.py +1 -0
- monkeybot/core/subagents/subagent_proto.py +331 -0
- monkeybot/core/subagents/subagent_worker.py +441 -0
- monkeybot/core/subagents/worker_pool.py +403 -0
- monkeybot/core/testing/__init__.py +1 -0
- monkeybot/core/testing/mocks_provider.py +86 -0
- monkeybot/core/testing/mocks_realtime_provider.py +137 -0
- monkeybot/core/tools/__init__.py +1 -0
- monkeybot/core/tools/core_tool_executor.py +1548 -0
- monkeybot/core/tools/inspector.py +226 -0
- monkeybot/core/tools/loop_inspector.py +45 -0
- monkeybot/core/tools/patch.py +480 -0
- monkeybot/core/tools/permission.py +284 -0
- monkeybot/core/tools/sandbox_executor.py +255 -0
- monkeybot/core/tools/spill_inventory.py +35 -0
- monkeybot/core/tools/terminal.py +381 -0
- monkeybot/core/tools/text_normalize.py +25 -0
- monkeybot/core/tools/types.py +33 -0
- monkeybot/core/tools/workspace_service.py +710 -0
- monkeybot/core/tools/workspace_tools.py +116 -0
- monkeybot/core/types/__init__.py +1 -0
- monkeybot/core/types/content_blocks.py +644 -0
- monkeybot/core/types/interfaces.py +156 -0
- monkeybot/core/types/types_tools.py +29 -0
- monkeybot/core/workspace/__init__.py +8 -0
- monkeybot/core/workspace/factory.py +45 -0
- monkeybot/core/workspace/gcs.py +130 -0
- monkeybot/core/workspace/local.py +162 -0
- monkeybot/core/workspace/protocol.py +45 -0
- monkeybot/core/workspace/s3.py +151 -0
- monkeybot/core/workspace_layout.py +27 -0
- monkeybot/gateway/__init__.py +1 -0
- monkeybot/gateway/bootstrap.py +18 -0
- monkeybot/gateway/main.py +47 -0
- monkeybot/gateway/realtime/__init__.py +31 -0
- monkeybot/gateway/realtime/app.py +321 -0
- monkeybot/gateway/realtime/deps.py +52 -0
- monkeybot/gateway/realtime/errors.py +81 -0
- monkeybot/gateway/realtime/guardrails.py +88 -0
- monkeybot/gateway/realtime/manager.py +77 -0
- monkeybot/gateway/realtime/metrics.py +144 -0
- monkeybot/gateway/realtime/routes.py +864 -0
- monkeybot/gateway/realtime/session.py +232 -0
- monkeybot/gateway/realtime/wire.py +412 -0
- monkeybot/gateway/realtime_main.py +49 -0
- monkeybot/gateway/sse/__init__.py +1 -0
- monkeybot/gateway/sse/app.py +733 -0
- monkeybot/gateway/sse/loop_port.py +31 -0
- monkeybot/gateway/sse/models.py +177 -0
- monkeybot/gateway/sse/reply_body.py +91 -0
- monkeybot/gateway/sse/routes.py +1101 -0
- monkeybot/gateway/sse/scheduler_routes.py +200 -0
- monkeybot/gateway/sse/scheduler_wiring.py +96 -0
- monkeybot/gateway/sse/session_bus.py +226 -0
- monkeybot/gateway/sse/sse.py +46 -0
- monkeybot/gateway/sse/workspace_layout.py +7 -0
- monkeybot/observability/__init__.py +220 -0
- monkeybot/observability/_state.py +10 -0
- monkeybot/observability/instrumentation.py +153 -0
- monkeybot/observability/propagation.py +65 -0
- monkeybot/observability/spans.py +455 -0
- monkeybot/providers/__init__.py +19 -0
- monkeybot/providers/_openai_compat.py +450 -0
- monkeybot/providers/_utils.py +473 -0
- monkeybot/providers/bedrock.py +145 -0
- monkeybot/providers/claude.py +125 -0
- monkeybot/providers/gemini.py +677 -0
- monkeybot/providers/gemini_live.py +398 -0
- monkeybot/providers/huggingface.py +129 -0
- monkeybot/providers/nvidia.py +104 -0
- monkeybot/providers/ollama.py +152 -0
- monkeybot/providers/openai.py +127 -0
- monkeybot/providers/pricing.py +60 -0
- monkeybot/providers/sampling.py +44 -0
- monkeybot/providers/vertex_claude.py +148 -0
- monkeybot/scaffold/__init__.py +33 -0
- monkeybot/scheduler/__init__.py +13 -0
- monkeybot/scheduler/__main__.py +4 -0
- monkeybot/scheduler/engine.py +333 -0
- monkeybot/scheduler/http_invoker.py +61 -0
- monkeybot/scheduler/interval.py +77 -0
- monkeybot/scheduler/tick_result.py +34 -0
- monkeybot/scheduler/worker.py +87 -0
- monkeybot/subagents/__init__.py +1 -0
- monkeybot/subagents/worker/__init__.py +1 -0
- monkeybot/subagents/worker/__main__.py +22 -0
- monkeybot/web_search/__init__.py +82 -0
- monkeybot/web_search/backends/__init__.py +5 -0
- monkeybot/web_search/backends/duckduckgo.py +32 -0
- monkeybot/web_search/backends/firecrawl.py +43 -0
- monkeybot/web_search/backends/tavily.py +45 -0
- monkeybot/web_search/protocol.py +25 -0
- monkeybot/web_search/tool.py +56 -0
- monkeybot-2.1.1.dist-info/METADATA +318 -0
- monkeybot-2.1.1.dist-info/RECORD +178 -0
- monkeybot-2.1.1.dist-info/WHEEL +4 -0
- monkeybot-2.1.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,677 @@
|
|
|
1
|
+
"""Vertex Gemini streaming via the official ``google-genai`` SDK (no LangChain in this module)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
from collections.abc import AsyncIterator, Sequence
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from monkeybot.core.llm.provider import (
|
|
12
|
+
Done,
|
|
13
|
+
GroundingEvent,
|
|
14
|
+
Message,
|
|
15
|
+
ProviderEvent,
|
|
16
|
+
TextDelta,
|
|
17
|
+
ThinkingDelta,
|
|
18
|
+
ToolCall,
|
|
19
|
+
UsageEvent,
|
|
20
|
+
)
|
|
21
|
+
from monkeybot.core.logging_utils import kv
|
|
22
|
+
from monkeybot.core.types.content_blocks import (
|
|
23
|
+
File,
|
|
24
|
+
Image,
|
|
25
|
+
Text,
|
|
26
|
+
Thinking,
|
|
27
|
+
ToolRequest,
|
|
28
|
+
ToolResponse,
|
|
29
|
+
)
|
|
30
|
+
from monkeybot.core.types.interfaces import LLMError
|
|
31
|
+
from monkeybot.core.types.types_tools import ToolDef
|
|
32
|
+
from monkeybot.providers.sampling import resolve_model_sampling
|
|
33
|
+
|
|
34
|
+
THOUGHT_SIGNATURE_KEY = "thoughtSignature"
|
|
35
|
+
SYNTHETIC_THOUGHT_SIGNATURE = "skip_thought_signature_validator"
|
|
36
|
+
|
|
37
|
+
_log = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
# Keep this in sync when onboarding another Vertex Gemini model that must use "global".
|
|
40
|
+
_GLOBAL_VERTEX_MODEL_IDS = frozenset({"gemini-3-flash-preview"})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _normalize_vertex_model(model: str) -> str:
|
|
44
|
+
"""Return a ``model`` value accepted by ``google.genai`` for Vertex (see SDK docstring).
|
|
45
|
+
|
|
46
|
+
Accepts bare ids (``gemini-2.5-flash``), ``models/...``, ``publishers/...``, ``google/...``,
|
|
47
|
+
or a full ``projects/.../locations/.../publishers/google/models/...`` resource name.
|
|
48
|
+
Strips a mistaken ``models/`` prefix when using the Vertex client (Vertex uses bare ids
|
|
49
|
+
or publisher paths, not the AI Studio ``models/`` prefix).
|
|
50
|
+
"""
|
|
51
|
+
m = (model or "").strip()
|
|
52
|
+
if not m:
|
|
53
|
+
raise LLMError("MODEL_NAME (model parameter) is empty.")
|
|
54
|
+
if m.startswith("projects/"):
|
|
55
|
+
return m
|
|
56
|
+
if m.startswith("publishers/") or m.startswith("google/"):
|
|
57
|
+
return m
|
|
58
|
+
if m.startswith("models/"):
|
|
59
|
+
return m[len("models/") :].strip()
|
|
60
|
+
return m
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _location_from_full_vertex_model(model: str) -> str | None:
|
|
64
|
+
"""If ``model`` is a full Vertex resource, return the ``locations/{loc}`` segment."""
|
|
65
|
+
if not model.startswith("projects/") or "/locations/" not in model:
|
|
66
|
+
return None
|
|
67
|
+
try:
|
|
68
|
+
idx = model.index("/locations/") + len("/locations/")
|
|
69
|
+
return model[idx:].split("/", 1)[0].strip() or None
|
|
70
|
+
except (ValueError, IndexError):
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _vertex_project_and_location(model_param: str) -> tuple[str, str]:
|
|
75
|
+
"""Resolve project id and API location for ``genai.Client(vertexai=True, ...)``.
|
|
76
|
+
|
|
77
|
+
Some preview model ids are not published under regional
|
|
78
|
+
endpoints like ``us-central1``; Vertex serves them from ``global`` unless you override
|
|
79
|
+
``VERTEX_AI_LOCATION`` / ``GOOGLE_CLOUD_LOCATION``.
|
|
80
|
+
"""
|
|
81
|
+
project = (
|
|
82
|
+
os.environ.get("GCP_PROJECT_ID")
|
|
83
|
+
or os.environ.get("VERTEX_AI_PROJECT_ID")
|
|
84
|
+
or os.environ.get("GOOGLE_CLOUD_PROJECT")
|
|
85
|
+
)
|
|
86
|
+
if not project or not str(project).strip():
|
|
87
|
+
raise LLMError(
|
|
88
|
+
"Set VERTEX_AI_PROJECT_ID, GCP_PROJECT_ID, or GOOGLE_CLOUD_PROJECT for Vertex Gemini."
|
|
89
|
+
)
|
|
90
|
+
explicit = os.environ.get("VERTEX_AI_LOCATION") or os.environ.get("GOOGLE_CLOUD_LOCATION")
|
|
91
|
+
if explicit and str(explicit).strip():
|
|
92
|
+
return str(project).strip(), str(explicit).strip()
|
|
93
|
+
|
|
94
|
+
embedded = _location_from_full_vertex_model(model_param)
|
|
95
|
+
if embedded:
|
|
96
|
+
return str(project).strip(), embedded
|
|
97
|
+
|
|
98
|
+
tail = model_param.split("/")[-1]
|
|
99
|
+
if tail.lower() in _GLOBAL_VERTEX_MODEL_IDS:
|
|
100
|
+
return str(project).strip(), "global"
|
|
101
|
+
|
|
102
|
+
return str(project).strip(), "us-central1"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
_THINKING_DISABLED = 0
|
|
106
|
+
"""Sentinel passed as ``thinking_budget`` to explicitly disable Gemini extended thinking.
|
|
107
|
+
|
|
108
|
+
Distinct from ``-1`` (the "omit ThinkingConfig" default) so auxiliary calls can
|
|
109
|
+
send ``ThinkingConfig(thinking_budget=0)`` to the API, which instructs the server
|
|
110
|
+
not to think rather than leaving it to the server's default for the model.
|
|
111
|
+
"""
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _suppress_thinking_for_auxiliary_call(
|
|
115
|
+
messages: Sequence[Message],
|
|
116
|
+
tools: Sequence[ToolDef],
|
|
117
|
+
) -> bool:
|
|
118
|
+
"""True when this call is a known auxiliary job that must not use extended thinking.
|
|
119
|
+
|
|
120
|
+
Context curation and history summarization use ``tools=()`` and a fixed 2-message
|
|
121
|
+
shape. Extended thinking on preview models can stall the stream for 10s+ with no
|
|
122
|
+
visible output, burning the curator timeout before ``Done`` is received.
|
|
123
|
+
"""
|
|
124
|
+
if tools:
|
|
125
|
+
return False
|
|
126
|
+
if len(messages) != 2 or messages[0].role != "system":
|
|
127
|
+
return False
|
|
128
|
+
sys_txt = "".join(b.text for b in messages[0].content if isinstance(b, Text))
|
|
129
|
+
markers = (
|
|
130
|
+
"You narrow context for another assistant",
|
|
131
|
+
"You compress prior agent conversation turns",
|
|
132
|
+
)
|
|
133
|
+
return any(m in sys_txt for m in markers)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _resolve_thinking_budget(
|
|
137
|
+
configured: int | None,
|
|
138
|
+
*,
|
|
139
|
+
override: int | None,
|
|
140
|
+
messages: Sequence[Message],
|
|
141
|
+
tools: Sequence[ToolDef],
|
|
142
|
+
) -> int:
|
|
143
|
+
if override is not None:
|
|
144
|
+
thinking_budget = int(override)
|
|
145
|
+
elif configured is not None:
|
|
146
|
+
thinking_budget = int(configured)
|
|
147
|
+
else:
|
|
148
|
+
thinking_budget = int(os.environ.get("MODEL_THINKING_BUDGET", "-1"))
|
|
149
|
+
if _suppress_thinking_for_auxiliary_call(messages, tools):
|
|
150
|
+
return _THINKING_DISABLED
|
|
151
|
+
return thinking_budget
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _split_system_and_rest(messages: Sequence[Message]) -> tuple[str, list[Message]]:
|
|
155
|
+
systems: list[str] = []
|
|
156
|
+
rest: list[Message] = []
|
|
157
|
+
for m in messages:
|
|
158
|
+
if m.role == "system":
|
|
159
|
+
texts = [b.text for b in m.content if isinstance(b, Text)]
|
|
160
|
+
systems.append("\n\n".join(texts))
|
|
161
|
+
else:
|
|
162
|
+
rest.append(m)
|
|
163
|
+
joined = "\n\n".join(s for s in systems if s).strip()
|
|
164
|
+
return joined, rest
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _flatten_tool_response_result(block: ToolResponse) -> str:
|
|
168
|
+
parts: list[str] = []
|
|
169
|
+
for b in block.result:
|
|
170
|
+
if isinstance(b, Text):
|
|
171
|
+
parts.append(b.text)
|
|
172
|
+
elif isinstance(b, (Image, File)):
|
|
173
|
+
continue
|
|
174
|
+
else:
|
|
175
|
+
raise LLMError(
|
|
176
|
+
"Cannot replay tool result to Vertex: unsupported result block "
|
|
177
|
+
f"{type(b).__name__}"
|
|
178
|
+
)
|
|
179
|
+
return "".join(parts)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _media_parts_from_blocks(blocks: Sequence[object]) -> list[Any]:
|
|
183
|
+
import base64
|
|
184
|
+
|
|
185
|
+
from google.genai import types
|
|
186
|
+
|
|
187
|
+
parts: list[Any] = []
|
|
188
|
+
for b in blocks:
|
|
189
|
+
if isinstance(b, (Image, File)):
|
|
190
|
+
parts.append(
|
|
191
|
+
types.Part(
|
|
192
|
+
inline_data=types.Blob(
|
|
193
|
+
mime_type=b.mime_type,
|
|
194
|
+
data=base64.b64decode(b.data),
|
|
195
|
+
)
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
return parts
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _is_user_loop_boundary(message: Message) -> bool:
|
|
202
|
+
"""A user turn that carries at least one non-ToolResponse block opens an active loop."""
|
|
203
|
+
if message.role != "user":
|
|
204
|
+
return False
|
|
205
|
+
return any(not isinstance(b, ToolResponse) for b in message.content)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _active_loop_start_index(messages: Sequence[Message]) -> int | None:
|
|
209
|
+
"""Return the index of the most recent user-loop boundary, or ``None``.
|
|
210
|
+
|
|
211
|
+
Goose's contract: only messages from this index onward must replay
|
|
212
|
+
``thoughtSignature`` to satisfy Vertex Gemini's strict validation.
|
|
213
|
+
"""
|
|
214
|
+
for i in range(len(messages) - 1, -1, -1):
|
|
215
|
+
if _is_user_loop_boundary(messages[i]):
|
|
216
|
+
return i
|
|
217
|
+
return None
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _signature_from_metadata(metadata: dict[str, object] | None) -> str | None:
|
|
221
|
+
if not metadata:
|
|
222
|
+
return None
|
|
223
|
+
sig = metadata.get(THOUGHT_SIGNATURE_KEY)
|
|
224
|
+
return sig if isinstance(sig, str) and sig else None
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _normalize_signature(value: Any) -> str | None:
|
|
228
|
+
"""Coerce SDK ``thought_signature`` (bytes or str) to a non-empty Python string.
|
|
229
|
+
|
|
230
|
+
The google-genai pydantic model normalizes wire signatures to ``bytes``
|
|
231
|
+
using base64 decoding for string inputs. Round-tripping requires preserving
|
|
232
|
+
the original textual representation; we standardise on base64 strings when
|
|
233
|
+
we receive bytes that aren't valid UTF-8.
|
|
234
|
+
"""
|
|
235
|
+
if value is None:
|
|
236
|
+
return None
|
|
237
|
+
if isinstance(value, bytes):
|
|
238
|
+
try:
|
|
239
|
+
decoded = value.decode("utf-8")
|
|
240
|
+
except UnicodeDecodeError:
|
|
241
|
+
import base64
|
|
242
|
+
|
|
243
|
+
decoded = base64.b64encode(value).decode("ascii")
|
|
244
|
+
return decoded or None
|
|
245
|
+
if isinstance(value, str):
|
|
246
|
+
return value or None
|
|
247
|
+
return None
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _messages_to_contents(rest: Sequence[Message]) -> list[Any]:
|
|
251
|
+
"""Build ``google.genai.types.Content`` list from harness messages (no system rows).
|
|
252
|
+
|
|
253
|
+
Vertex Gemini 2.5+ enforces ``thoughtSignature`` round-trip on the active
|
|
254
|
+
conversation loop. This builder mirrors Goose's contract:
|
|
255
|
+
|
|
256
|
+
- Signatures are re-attached to ``functionCall`` and reasoning parts in the
|
|
257
|
+
active loop (from the last user-loop boundary onward).
|
|
258
|
+
- Earlier turns drop signatures (Vertex would reject them as stale).
|
|
259
|
+
- If the active loop's first model tool call has no captured signature, a
|
|
260
|
+
synthetic placeholder (``skip_thought_signature_validator``) is inserted
|
|
261
|
+
so Vertex skips strict validation rather than 400-ing.
|
|
262
|
+
- ``Thinking`` blocks are round-tripped as ``Part(text, thought=True,
|
|
263
|
+
thought_signature=...)`` only inside the active loop.
|
|
264
|
+
"""
|
|
265
|
+
from google.genai import types
|
|
266
|
+
|
|
267
|
+
for m in rest:
|
|
268
|
+
if m.role != "user":
|
|
269
|
+
continue
|
|
270
|
+
for block in m.content:
|
|
271
|
+
if isinstance(block, ToolResponse) and not str(block.tool_name or "").strip():
|
|
272
|
+
raise LLMError(
|
|
273
|
+
"Cannot replay tool result to Vertex: empty tool_name "
|
|
274
|
+
f"(tool_call_id={block.id!r})."
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
active_start = _active_loop_start_index(rest)
|
|
278
|
+
|
|
279
|
+
contents: list[Any] = []
|
|
280
|
+
for idx, m in enumerate(rest):
|
|
281
|
+
gemini_role = "user" if m.role == "user" else "model"
|
|
282
|
+
in_active_loop = active_start is not None and idx >= active_start
|
|
283
|
+
needs_synthetic_for_first_model_tool_call = in_active_loop and m.role != "user"
|
|
284
|
+
parts: list[Any] = []
|
|
285
|
+
for block in m.content:
|
|
286
|
+
if isinstance(block, Text):
|
|
287
|
+
parts.append(types.Part(text=block.text))
|
|
288
|
+
elif isinstance(block, (Image, File)):
|
|
289
|
+
parts.extend(_media_parts_from_blocks([block]))
|
|
290
|
+
elif isinstance(block, Thinking):
|
|
291
|
+
if not in_active_loop:
|
|
292
|
+
continue
|
|
293
|
+
kwargs: dict[str, Any] = {"text": block.thinking, "thought": True}
|
|
294
|
+
if block.signature:
|
|
295
|
+
kwargs["thought_signature"] = block.signature.encode("utf-8")
|
|
296
|
+
parts.append(types.Part(**kwargs))
|
|
297
|
+
elif isinstance(block, ToolRequest):
|
|
298
|
+
fc_kwargs: dict[str, Any] = {
|
|
299
|
+
"name": block.name,
|
|
300
|
+
"args": dict(block.args),
|
|
301
|
+
"id": block.id,
|
|
302
|
+
}
|
|
303
|
+
part_kwargs: dict[str, Any] = {
|
|
304
|
+
"function_call": types.FunctionCall(**fc_kwargs),
|
|
305
|
+
}
|
|
306
|
+
if in_active_loop:
|
|
307
|
+
sig = _signature_from_metadata(block.metadata)
|
|
308
|
+
if sig is None and needs_synthetic_for_first_model_tool_call:
|
|
309
|
+
sig = SYNTHETIC_THOUGHT_SIGNATURE
|
|
310
|
+
if sig is not None:
|
|
311
|
+
# SDK validator base64-decodes string inputs; pass raw bytes
|
|
312
|
+
# so the literal signature survives the round-trip.
|
|
313
|
+
part_kwargs["thought_signature"] = sig.encode("utf-8")
|
|
314
|
+
needs_synthetic_for_first_model_tool_call = False
|
|
315
|
+
parts.append(types.Part(**part_kwargs))
|
|
316
|
+
elif isinstance(block, ToolResponse):
|
|
317
|
+
parts.append(
|
|
318
|
+
types.Part(
|
|
319
|
+
function_response=types.FunctionResponse(
|
|
320
|
+
name=block.tool_name,
|
|
321
|
+
response={"result": _flatten_tool_response_result(block)},
|
|
322
|
+
)
|
|
323
|
+
)
|
|
324
|
+
)
|
|
325
|
+
parts.extend(_media_parts_from_blocks(block.result))
|
|
326
|
+
else:
|
|
327
|
+
raise LLMError(
|
|
328
|
+
"Cannot replay message to Vertex Gemini: unsupported block "
|
|
329
|
+
f"{type(block).__name__}"
|
|
330
|
+
)
|
|
331
|
+
if not parts:
|
|
332
|
+
parts.append(types.Part(text=""))
|
|
333
|
+
contents.append(types.Content(role=gemini_role, parts=parts))
|
|
334
|
+
return contents
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _tool_defs_to_declarations(tools: Sequence[ToolDef]) -> list[Any]:
|
|
338
|
+
from google.genai import types
|
|
339
|
+
|
|
340
|
+
out: list[Any] = []
|
|
341
|
+
for t in tools:
|
|
342
|
+
schema = dict(t.input_schema) if t.input_schema else {"type": "object"}
|
|
343
|
+
out.append(
|
|
344
|
+
types.FunctionDeclaration(
|
|
345
|
+
name=t.name,
|
|
346
|
+
description=t.description or "",
|
|
347
|
+
parameters_json_schema=schema,
|
|
348
|
+
)
|
|
349
|
+
)
|
|
350
|
+
return out
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _grounding_metadata_to_dict(gm: Any) -> dict[str, Any] | None:
|
|
354
|
+
"""Flatten Vertex ``GroundingMetadata`` into a small, wire-friendly dict.
|
|
355
|
+
|
|
356
|
+
Only carries what a UI needs for citations/search-suggestion display: source
|
|
357
|
+
chunks (title/uri) and web search queries. Returns ``None`` when there is
|
|
358
|
+
nothing worth surfacing.
|
|
359
|
+
"""
|
|
360
|
+
if gm is None:
|
|
361
|
+
return None
|
|
362
|
+
chunks: list[dict[str, str]] = []
|
|
363
|
+
for chunk in getattr(gm, "grounding_chunks", None) or []:
|
|
364
|
+
web = getattr(chunk, "web", None)
|
|
365
|
+
if web is None:
|
|
366
|
+
continue
|
|
367
|
+
title = str(getattr(web, "title", "") or "")
|
|
368
|
+
uri = str(getattr(web, "uri", "") or "")
|
|
369
|
+
if title or uri:
|
|
370
|
+
chunks.append({"title": title, "uri": uri})
|
|
371
|
+
queries = [str(q) for q in (getattr(gm, "web_search_queries", None) or [])]
|
|
372
|
+
if not chunks and not queries:
|
|
373
|
+
return None
|
|
374
|
+
out: dict[str, Any] = {}
|
|
375
|
+
if chunks:
|
|
376
|
+
out["sources"] = chunks
|
|
377
|
+
if queries:
|
|
378
|
+
out["search_queries"] = queries
|
|
379
|
+
return out
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _merge_function_call_args(existing: dict[str, Any], fc: Any) -> dict[str, Any]:
|
|
383
|
+
merged = dict(existing)
|
|
384
|
+
if isinstance(getattr(fc, "args", None), dict):
|
|
385
|
+
merged.update(fc.args)
|
|
386
|
+
partial = getattr(fc, "partial_args", None)
|
|
387
|
+
if isinstance(partial, dict):
|
|
388
|
+
merged.update(partial)
|
|
389
|
+
elif isinstance(partial, str) and partial.strip():
|
|
390
|
+
try: # noqa: SIM105 — preserve streaming merge semantics (spec: verbatim)
|
|
391
|
+
merged.update(json.loads(partial))
|
|
392
|
+
except json.JSONDecodeError:
|
|
393
|
+
pass
|
|
394
|
+
return merged
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _usage_from_response(um: Any) -> UsageEvent | None:
|
|
398
|
+
if um is None:
|
|
399
|
+
return None
|
|
400
|
+
inp = int(getattr(um, "prompt_token_count", 0) or 0)
|
|
401
|
+
out = int(getattr(um, "candidates_token_count", 0) or 0)
|
|
402
|
+
# Vertex Gemini implicit caching reports a single read count; there is no
|
|
403
|
+
# separate cache-creation cost, so creation is 0 and the total equals the read.
|
|
404
|
+
cache_read = int(getattr(um, "cached_content_token_count", 0) or 0)
|
|
405
|
+
return UsageEvent(
|
|
406
|
+
input_tokens=inp,
|
|
407
|
+
output_tokens=out,
|
|
408
|
+
cached_tokens=cache_read,
|
|
409
|
+
cache_read_tokens=cache_read,
|
|
410
|
+
cache_creation_tokens=0,
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
class GeminiProvider:
|
|
415
|
+
def __init__(
|
|
416
|
+
self,
|
|
417
|
+
*,
|
|
418
|
+
supports_streaming: bool = True,
|
|
419
|
+
temperature: float | None = None,
|
|
420
|
+
max_tokens: int | None = None,
|
|
421
|
+
max_output_tokens: int | None = None,
|
|
422
|
+
thinking_budget: int | None = None,
|
|
423
|
+
api_key: str | None = None,
|
|
424
|
+
) -> None:
|
|
425
|
+
"""Gemini streaming provider (Vertex AI or Google AI Studio).
|
|
426
|
+
|
|
427
|
+
When ``api_key`` is provided, the provider connects to Google AI Studio
|
|
428
|
+
(``generativelanguage.googleapis.com``). Otherwise it connects to Vertex AI
|
|
429
|
+
using application default credentials or the project/location environment variables.
|
|
430
|
+
|
|
431
|
+
``max_output_tokens`` is a backward-compatible alias for ``max_tokens``.
|
|
432
|
+
|
|
433
|
+
Native ``google_search`` grounding is opt-in per call via the
|
|
434
|
+
``vertex_google_search`` keyword on :meth:`stream` and
|
|
435
|
+
:meth:`count_input_tokens` (Gemini-only; the runtime loop passes it only
|
|
436
|
+
when ``provider.name == \"gemini\"``).
|
|
437
|
+
"""
|
|
438
|
+
if max_output_tokens is not None and max_tokens is not None and max_output_tokens != max_tokens:
|
|
439
|
+
raise ValueError("pass only one of max_tokens or max_output_tokens")
|
|
440
|
+
effective_max_tokens = max_tokens if max_tokens is not None else max_output_tokens
|
|
441
|
+
self._supports_streaming = supports_streaming
|
|
442
|
+
sampling = resolve_model_sampling(temperature=temperature, max_tokens=effective_max_tokens)
|
|
443
|
+
self._temperature = sampling.temperature
|
|
444
|
+
self._max_tokens = sampling.max_tokens
|
|
445
|
+
self._thinking_budget = thinking_budget
|
|
446
|
+
self._api_key = api_key
|
|
447
|
+
|
|
448
|
+
def _client(self, model_param: str) -> Any:
|
|
449
|
+
"""Create a google-genai client for Vertex AI or Google AI Studio."""
|
|
450
|
+
from google import genai
|
|
451
|
+
|
|
452
|
+
if self._api_key:
|
|
453
|
+
return genai.Client(api_key=self._api_key)
|
|
454
|
+
project, location = _vertex_project_and_location(model_param)
|
|
455
|
+
return genai.Client(vertexai=True, project=project, location=location)
|
|
456
|
+
|
|
457
|
+
@property
|
|
458
|
+
def name(self) -> str:
|
|
459
|
+
return "gemini"
|
|
460
|
+
|
|
461
|
+
@property
|
|
462
|
+
def supports_streaming(self) -> bool:
|
|
463
|
+
return self._supports_streaming
|
|
464
|
+
|
|
465
|
+
async def count_input_tokens(
|
|
466
|
+
self,
|
|
467
|
+
messages: Sequence[Message],
|
|
468
|
+
tools: Sequence[ToolDef],
|
|
469
|
+
*,
|
|
470
|
+
model: str,
|
|
471
|
+
thinking_budget: int | None = None,
|
|
472
|
+
vertex_google_search: bool = False,
|
|
473
|
+
) -> int:
|
|
474
|
+
model_param = _normalize_vertex_model(model)
|
|
475
|
+
try:
|
|
476
|
+
from google.genai import types
|
|
477
|
+
except ImportError as exc:
|
|
478
|
+
raise LLMError(
|
|
479
|
+
"google-genai is required for GeminiProvider. Install with: uv sync (monkeybot dependencies)."
|
|
480
|
+
) from exc
|
|
481
|
+
|
|
482
|
+
temperature = float(self._temperature)
|
|
483
|
+
max_tokens = int(self._max_tokens)
|
|
484
|
+
thinking_budget = _resolve_thinking_budget(
|
|
485
|
+
self._thinking_budget,
|
|
486
|
+
override=thinking_budget,
|
|
487
|
+
messages=messages,
|
|
488
|
+
tools=tools,
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
system_instruction, rest = _split_system_and_rest(messages)
|
|
492
|
+
contents = _messages_to_contents(rest)
|
|
493
|
+
decls = _tool_defs_to_declarations(tools)
|
|
494
|
+
|
|
495
|
+
count_cfg_kwargs: dict[str, Any] = {}
|
|
496
|
+
if system_instruction:
|
|
497
|
+
count_cfg_kwargs["system_instruction"] = system_instruction
|
|
498
|
+
count_tools: list[Any] = []
|
|
499
|
+
if decls:
|
|
500
|
+
count_tools.append(types.Tool(function_declarations=decls))
|
|
501
|
+
if vertex_google_search:
|
|
502
|
+
count_tools.append(types.Tool(google_search=types.GoogleSearch()))
|
|
503
|
+
if count_tools:
|
|
504
|
+
count_cfg_kwargs["tools"] = count_tools
|
|
505
|
+
|
|
506
|
+
gen_cfg_kwargs: dict[str, Any] = {
|
|
507
|
+
"temperature": temperature,
|
|
508
|
+
"max_output_tokens": max_tokens,
|
|
509
|
+
}
|
|
510
|
+
if thinking_budget != -1:
|
|
511
|
+
gen_cfg_kwargs["thinking_config"] = types.ThinkingConfig(thinking_budget=thinking_budget)
|
|
512
|
+
count_cfg_kwargs["generation_config"] = types.GenerationConfig(**gen_cfg_kwargs)
|
|
513
|
+
|
|
514
|
+
ct_cfg = types.CountTokensConfig(**count_cfg_kwargs)
|
|
515
|
+
client = self._client(model_param)
|
|
516
|
+
resp = await client.aio.models.count_tokens(
|
|
517
|
+
model=model_param,
|
|
518
|
+
contents=contents,
|
|
519
|
+
config=ct_cfg,
|
|
520
|
+
)
|
|
521
|
+
return int(resp.total_tokens or 0)
|
|
522
|
+
|
|
523
|
+
async def stream(
|
|
524
|
+
self,
|
|
525
|
+
messages: Sequence[Message],
|
|
526
|
+
tools: Sequence[ToolDef],
|
|
527
|
+
*,
|
|
528
|
+
model: str,
|
|
529
|
+
thinking_budget: int | None = None,
|
|
530
|
+
vertex_google_search: bool = False,
|
|
531
|
+
) -> AsyncIterator[ProviderEvent]:
|
|
532
|
+
model_param = _normalize_vertex_model(model)
|
|
533
|
+
try:
|
|
534
|
+
from google.genai import types
|
|
535
|
+
except ImportError as exc:
|
|
536
|
+
raise LLMError(
|
|
537
|
+
"google-genai is required for GeminiProvider. Install with: uv sync (monkeybot dependencies)."
|
|
538
|
+
) from exc
|
|
539
|
+
|
|
540
|
+
temperature = float(self._temperature)
|
|
541
|
+
max_tokens = int(self._max_tokens)
|
|
542
|
+
thinking_budget = _resolve_thinking_budget(
|
|
543
|
+
self._thinking_budget,
|
|
544
|
+
override=thinking_budget,
|
|
545
|
+
messages=messages,
|
|
546
|
+
tools=tools,
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
system_instruction, rest = _split_system_and_rest(messages)
|
|
550
|
+
contents = _messages_to_contents(rest)
|
|
551
|
+
|
|
552
|
+
cfg_kwargs: dict[str, Any] = {
|
|
553
|
+
"temperature": temperature,
|
|
554
|
+
"max_output_tokens": max_tokens,
|
|
555
|
+
}
|
|
556
|
+
if system_instruction:
|
|
557
|
+
cfg_kwargs["system_instruction"] = system_instruction
|
|
558
|
+
# -1 means "omit ThinkingConfig entirely" (model default).
|
|
559
|
+
# 0 means explicitly disable thinking (sent to API).
|
|
560
|
+
# Any other positive value sets a token budget.
|
|
561
|
+
if thinking_budget != -1:
|
|
562
|
+
cfg_kwargs["thinking_config"] = types.ThinkingConfig(thinking_budget=thinking_budget)
|
|
563
|
+
|
|
564
|
+
decls = _tool_defs_to_declarations(tools)
|
|
565
|
+
stream_tools: list[Any] = []
|
|
566
|
+
if decls:
|
|
567
|
+
stream_tools.append(types.Tool(function_declarations=decls))
|
|
568
|
+
if vertex_google_search:
|
|
569
|
+
stream_tools.append(types.Tool(google_search=types.GoogleSearch()))
|
|
570
|
+
if stream_tools:
|
|
571
|
+
cfg_kwargs["tools"] = stream_tools
|
|
572
|
+
|
|
573
|
+
config = types.GenerateContentConfig(**cfg_kwargs)
|
|
574
|
+
|
|
575
|
+
client = self._client(model_param)
|
|
576
|
+
|
|
577
|
+
pending_tools: dict[str, ToolCall] = {}
|
|
578
|
+
last_usage: Any = None
|
|
579
|
+
last_signature: str | None = None
|
|
580
|
+
grounding_meta: dict[str, Any] | None = None
|
|
581
|
+
truncated = False
|
|
582
|
+
|
|
583
|
+
try:
|
|
584
|
+
stream_it = await client.aio.models.generate_content_stream(
|
|
585
|
+
model=model_param,
|
|
586
|
+
contents=contents,
|
|
587
|
+
config=config,
|
|
588
|
+
)
|
|
589
|
+
async for resp in stream_it:
|
|
590
|
+
if getattr(resp, "usage_metadata", None) is not None:
|
|
591
|
+
last_usage = resp.usage_metadata
|
|
592
|
+
|
|
593
|
+
for cand in resp.candidates or []:
|
|
594
|
+
fr = getattr(cand, "finish_reason", None)
|
|
595
|
+
if fr is not None:
|
|
596
|
+
fr_name = str(getattr(fr, "name", fr)).upper()
|
|
597
|
+
if "MAX_TOKEN" in fr_name:
|
|
598
|
+
truncated = True
|
|
599
|
+
cand_gm = _grounding_metadata_to_dict(getattr(cand, "grounding_metadata", None))
|
|
600
|
+
if cand_gm is not None:
|
|
601
|
+
if grounding_meta is not None and cand_gm != grounding_meta:
|
|
602
|
+
_log.debug(
|
|
603
|
+
"grounding metadata replaced by later candidate %s",
|
|
604
|
+
kv(provider="gemini", model=model),
|
|
605
|
+
)
|
|
606
|
+
grounding_meta = cand_gm
|
|
607
|
+
content = getattr(cand, "content", None)
|
|
608
|
+
if content is None or not content.parts:
|
|
609
|
+
continue
|
|
610
|
+
for part in content.parts:
|
|
611
|
+
part_sig = _normalize_signature(getattr(part, "thought_signature", None))
|
|
612
|
+
if part_sig:
|
|
613
|
+
last_signature = part_sig
|
|
614
|
+
|
|
615
|
+
is_thought = bool(getattr(part, "thought", False))
|
|
616
|
+
if is_thought:
|
|
617
|
+
txt = getattr(part, "text", None) or ""
|
|
618
|
+
if txt:
|
|
619
|
+
yield ThinkingDelta(
|
|
620
|
+
text=txt,
|
|
621
|
+
signature=last_signature,
|
|
622
|
+
)
|
|
623
|
+
continue
|
|
624
|
+
|
|
625
|
+
if part.text:
|
|
626
|
+
yield TextDelta(text=part.text)
|
|
627
|
+
|
|
628
|
+
fc = part.function_call
|
|
629
|
+
if fc is None:
|
|
630
|
+
continue
|
|
631
|
+
name = str(getattr(fc, "name", "") or "")
|
|
632
|
+
if not name:
|
|
633
|
+
continue
|
|
634
|
+
fid = str(getattr(fc, "id", "") or "")
|
|
635
|
+
key = fid if fid else f"anon:{name}"
|
|
636
|
+
prev = pending_tools.get(key)
|
|
637
|
+
prev_args: dict[str, Any] = dict(prev.args) if prev else {}
|
|
638
|
+
merged_args = _merge_function_call_args(prev_args, fc)
|
|
639
|
+
call_id = fid if fid else key
|
|
640
|
+
|
|
641
|
+
# Per Goose: prefer the part's own signature, else carry forward
|
|
642
|
+
# from the most recent signed part in this stream.
|
|
643
|
+
effective_sig = part_sig or last_signature
|
|
644
|
+
prev_meta = dict(prev.metadata) if prev and prev.metadata else {}
|
|
645
|
+
if effective_sig:
|
|
646
|
+
prev_meta[THOUGHT_SIGNATURE_KEY] = effective_sig
|
|
647
|
+
|
|
648
|
+
pending_tools[key] = ToolCall(
|
|
649
|
+
call_id=call_id,
|
|
650
|
+
name=name,
|
|
651
|
+
args=merged_args,
|
|
652
|
+
metadata=prev_meta or None,
|
|
653
|
+
)
|
|
654
|
+
except LLMError:
|
|
655
|
+
raise
|
|
656
|
+
except Exception as exc:
|
|
657
|
+
_log.warning(
|
|
658
|
+
"Gemini stream error %s",
|
|
659
|
+
kv(provider="gemini", model=model, n_messages=len(messages), n_tools=len(tools)),
|
|
660
|
+
exc_info=True,
|
|
661
|
+
)
|
|
662
|
+
raise LLMError(str(exc)) from exc
|
|
663
|
+
|
|
664
|
+
if grounding_meta is not None:
|
|
665
|
+
yield GroundingEvent(
|
|
666
|
+
sources=list(grounding_meta.get("sources", [])),
|
|
667
|
+
search_queries=list(grounding_meta.get("search_queries", [])),
|
|
668
|
+
)
|
|
669
|
+
|
|
670
|
+
ev = _usage_from_response(last_usage)
|
|
671
|
+
if ev is not None:
|
|
672
|
+
yield ev
|
|
673
|
+
|
|
674
|
+
for call_id in sorted(pending_tools.keys()):
|
|
675
|
+
yield pending_tools[call_id]
|
|
676
|
+
|
|
677
|
+
yield Done(truncated=truncated)
|