monkeybot 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (178) hide show
  1. monkeybot/__init__.py +3 -0
  2. monkeybot/cli/__init__.py +3 -0
  3. monkeybot/cli/__main__.py +8 -0
  4. monkeybot/cli/audio_io.py +8 -0
  5. monkeybot/cli/gateway_manager.py +17 -0
  6. monkeybot/cli/main.py +22 -0
  7. monkeybot/cli/push_to_talk.py +12 -0
  8. monkeybot/cli/realtime_client.py +13 -0
  9. monkeybot/core/__init__.py +19 -0
  10. monkeybot/core/attachments/__init__.py +22 -0
  11. monkeybot/core/attachments/catalog.py +62 -0
  12. monkeybot/core/attachments/config.py +52 -0
  13. monkeybot/core/attachments/freeze.py +158 -0
  14. monkeybot/core/attachments/resolve.py +70 -0
  15. monkeybot/core/attachments/store.py +180 -0
  16. monkeybot/core/attachments/text.py +72 -0
  17. monkeybot/core/attachments/tools.py +54 -0
  18. monkeybot/core/bootstrap.py +242 -0
  19. monkeybot/core/config/__init__.py +71 -0
  20. monkeybot/core/config/realtime_config.py +150 -0
  21. monkeybot/core/config/runtime_env.py +262 -0
  22. monkeybot/core/config/settings.py +341 -0
  23. monkeybot/core/config/validation.py +249 -0
  24. monkeybot/core/config/yaml_loader.py +45 -0
  25. monkeybot/core/context/__init__.py +781 -0
  26. monkeybot/core/context/campaign_context.py +8 -0
  27. monkeybot/core/context/common.py +14 -0
  28. monkeybot/core/context/curator.py +255 -0
  29. monkeybot/core/context/epoch.py +226 -0
  30. monkeybot/core/context/memory_prompt.py +222 -0
  31. monkeybot/core/context/tool_output_policy.py +270 -0
  32. monkeybot/core/context/tool_result_ingress.py +290 -0
  33. monkeybot/core/context/tool_shapers.py +361 -0
  34. monkeybot/core/hooks/__init__.py +261 -0
  35. monkeybot/core/llm/__init__.py +4 -0
  36. monkeybot/core/llm/provider.py +296 -0
  37. monkeybot/core/llm/realtime_provider.py +203 -0
  38. monkeybot/core/llm/usage.py +57 -0
  39. monkeybot/core/logging_utils.py +24 -0
  40. monkeybot/core/mcp/__init__.py +1 -0
  41. monkeybot/core/mcp/mcp_client.py +1215 -0
  42. monkeybot/core/mcp/ports_mcp.py +109 -0
  43. monkeybot/core/memory/__init__.py +24 -0
  44. monkeybot/core/memory/hook.py +413 -0
  45. monkeybot/core/memory/index_format.py +104 -0
  46. monkeybot/core/memory/integrity.py +180 -0
  47. monkeybot/core/memory/organizer.py +270 -0
  48. monkeybot/core/memory/storage_ops.py +139 -0
  49. monkeybot/core/memory/subsystem.py +91 -0
  50. monkeybot/core/messages/__init__.py +16 -0
  51. monkeybot/core/messages/convert_provider.py +41 -0
  52. monkeybot/core/messages/tool_integrity.py +262 -0
  53. monkeybot/core/messages/transform_context.py +84 -0
  54. monkeybot/core/path_safety.py +11 -0
  55. monkeybot/core/persistence/__init__.py +17 -0
  56. monkeybot/core/persistence/backends.py +236 -0
  57. monkeybot/core/persistence/db.py +28 -0
  58. monkeybot/core/persistence/durable_runs.py +286 -0
  59. monkeybot/core/persistence/firestore.py +658 -0
  60. monkeybot/core/persistence/firestore_scheduled_loops.py +336 -0
  61. monkeybot/core/persistence/history.py +156 -0
  62. monkeybot/core/persistence/postgres.py +895 -0
  63. monkeybot/core/persistence/runs.py +76 -0
  64. monkeybot/core/persistence/scheduled_loops.py +435 -0
  65. monkeybot/core/persistence/session_turn_locks.py +94 -0
  66. monkeybot/core/persistence/sqlite.py +218 -0
  67. monkeybot/core/persistence/sqlite_backend.py +74 -0
  68. monkeybot/core/persistence/thread_summary.py +61 -0
  69. monkeybot/core/persistence/transcript.py +194 -0
  70. monkeybot/core/persistence/usage.py +149 -0
  71. monkeybot/core/prompts/__init__.py +1 -0
  72. monkeybot/core/prompts/harness_prompt.py +197 -0
  73. monkeybot/core/prompts/prompt.py +215 -0
  74. monkeybot/core/runtime/__init__.py +1 -0
  75. monkeybot/core/runtime/context_budget.py +267 -0
  76. monkeybot/core/runtime/events.py +819 -0
  77. monkeybot/core/runtime/input_admission.py +154 -0
  78. monkeybot/core/runtime/loop.py +2374 -0
  79. monkeybot/core/runtime/provider_stream_mapper.py +159 -0
  80. monkeybot/core/runtime/realtime_loop.py +654 -0
  81. monkeybot/core/runtime/utterance_buffer.py +179 -0
  82. monkeybot/core/subagents/__init__.py +1 -0
  83. monkeybot/core/subagents/subagent_proto.py +331 -0
  84. monkeybot/core/subagents/subagent_worker.py +441 -0
  85. monkeybot/core/subagents/worker_pool.py +403 -0
  86. monkeybot/core/testing/__init__.py +1 -0
  87. monkeybot/core/testing/mocks_provider.py +86 -0
  88. monkeybot/core/testing/mocks_realtime_provider.py +137 -0
  89. monkeybot/core/tools/__init__.py +1 -0
  90. monkeybot/core/tools/core_tool_executor.py +1548 -0
  91. monkeybot/core/tools/inspector.py +226 -0
  92. monkeybot/core/tools/loop_inspector.py +45 -0
  93. monkeybot/core/tools/patch.py +480 -0
  94. monkeybot/core/tools/permission.py +284 -0
  95. monkeybot/core/tools/sandbox_executor.py +255 -0
  96. monkeybot/core/tools/spill_inventory.py +35 -0
  97. monkeybot/core/tools/terminal.py +381 -0
  98. monkeybot/core/tools/text_normalize.py +25 -0
  99. monkeybot/core/tools/types.py +33 -0
  100. monkeybot/core/tools/workspace_service.py +710 -0
  101. monkeybot/core/tools/workspace_tools.py +116 -0
  102. monkeybot/core/types/__init__.py +1 -0
  103. monkeybot/core/types/content_blocks.py +644 -0
  104. monkeybot/core/types/interfaces.py +156 -0
  105. monkeybot/core/types/types_tools.py +29 -0
  106. monkeybot/core/workspace/__init__.py +8 -0
  107. monkeybot/core/workspace/factory.py +45 -0
  108. monkeybot/core/workspace/gcs.py +130 -0
  109. monkeybot/core/workspace/local.py +162 -0
  110. monkeybot/core/workspace/protocol.py +45 -0
  111. monkeybot/core/workspace/s3.py +151 -0
  112. monkeybot/core/workspace_layout.py +27 -0
  113. monkeybot/gateway/__init__.py +1 -0
  114. monkeybot/gateway/bootstrap.py +18 -0
  115. monkeybot/gateway/main.py +47 -0
  116. monkeybot/gateway/realtime/__init__.py +31 -0
  117. monkeybot/gateway/realtime/app.py +321 -0
  118. monkeybot/gateway/realtime/deps.py +52 -0
  119. monkeybot/gateway/realtime/errors.py +81 -0
  120. monkeybot/gateway/realtime/guardrails.py +88 -0
  121. monkeybot/gateway/realtime/manager.py +77 -0
  122. monkeybot/gateway/realtime/metrics.py +144 -0
  123. monkeybot/gateway/realtime/routes.py +864 -0
  124. monkeybot/gateway/realtime/session.py +232 -0
  125. monkeybot/gateway/realtime/wire.py +412 -0
  126. monkeybot/gateway/realtime_main.py +49 -0
  127. monkeybot/gateway/sse/__init__.py +1 -0
  128. monkeybot/gateway/sse/app.py +733 -0
  129. monkeybot/gateway/sse/loop_port.py +31 -0
  130. monkeybot/gateway/sse/models.py +177 -0
  131. monkeybot/gateway/sse/reply_body.py +91 -0
  132. monkeybot/gateway/sse/routes.py +1101 -0
  133. monkeybot/gateway/sse/scheduler_routes.py +200 -0
  134. monkeybot/gateway/sse/scheduler_wiring.py +96 -0
  135. monkeybot/gateway/sse/session_bus.py +226 -0
  136. monkeybot/gateway/sse/sse.py +46 -0
  137. monkeybot/gateway/sse/workspace_layout.py +7 -0
  138. monkeybot/observability/__init__.py +220 -0
  139. monkeybot/observability/_state.py +10 -0
  140. monkeybot/observability/instrumentation.py +153 -0
  141. monkeybot/observability/propagation.py +65 -0
  142. monkeybot/observability/spans.py +455 -0
  143. monkeybot/providers/__init__.py +19 -0
  144. monkeybot/providers/_openai_compat.py +450 -0
  145. monkeybot/providers/_utils.py +473 -0
  146. monkeybot/providers/bedrock.py +145 -0
  147. monkeybot/providers/claude.py +125 -0
  148. monkeybot/providers/gemini.py +677 -0
  149. monkeybot/providers/gemini_live.py +398 -0
  150. monkeybot/providers/huggingface.py +129 -0
  151. monkeybot/providers/nvidia.py +104 -0
  152. monkeybot/providers/ollama.py +152 -0
  153. monkeybot/providers/openai.py +127 -0
  154. monkeybot/providers/pricing.py +60 -0
  155. monkeybot/providers/sampling.py +44 -0
  156. monkeybot/providers/vertex_claude.py +148 -0
  157. monkeybot/scaffold/__init__.py +33 -0
  158. monkeybot/scheduler/__init__.py +13 -0
  159. monkeybot/scheduler/__main__.py +4 -0
  160. monkeybot/scheduler/engine.py +333 -0
  161. monkeybot/scheduler/http_invoker.py +61 -0
  162. monkeybot/scheduler/interval.py +77 -0
  163. monkeybot/scheduler/tick_result.py +34 -0
  164. monkeybot/scheduler/worker.py +87 -0
  165. monkeybot/subagents/__init__.py +1 -0
  166. monkeybot/subagents/worker/__init__.py +1 -0
  167. monkeybot/subagents/worker/__main__.py +22 -0
  168. monkeybot/web_search/__init__.py +82 -0
  169. monkeybot/web_search/backends/__init__.py +5 -0
  170. monkeybot/web_search/backends/duckduckgo.py +32 -0
  171. monkeybot/web_search/backends/firecrawl.py +43 -0
  172. monkeybot/web_search/backends/tavily.py +45 -0
  173. monkeybot/web_search/protocol.py +25 -0
  174. monkeybot/web_search/tool.py +56 -0
  175. monkeybot-2.1.1.dist-info/METADATA +318 -0
  176. monkeybot-2.1.1.dist-info/RECORD +178 -0
  177. monkeybot-2.1.1.dist-info/WHEEL +4 -0
  178. monkeybot-2.1.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,290 @@
1
+ """Sanitize, cap, and redact tool/MCP payloads before they enter chat history."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ import os
8
+ import re
9
+ from typing import Any
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ _DEFAULT_TOOL_RESULT_MAX_CHARS = 32_768
14
+ _DEFAULT_SUMMARY_TOOL_RESULT_MAX_CHARS = 8_192
15
+ # Only applies to denylisted blob keys (data/base64/…), not arbitrary text fields
16
+ # like browser ``tree``. Whole-result size is handled by spill + TOOL_RESULT_MAX_CHARS.
17
+ _DEFAULT_BLOB_JSON_FIELD_MAX_CHARS = 512
18
+ _MIN_BASE64_RUN = 800
19
+
20
+ _DATA_URL_RE = re.compile(
21
+ r"data:(?:image|audio|video|application)/[\w.+-]+;base64,[\sA-Za-z0-9+/=]{200,}",
22
+ re.IGNORECASE,
23
+ )
24
+ _LONG_B64_RE = re.compile(
25
+ r"(?<![A-Za-z0-9+/=])(?:[A-Za-z0-9+/]{4}){200,}(?:[A-Za-z0-9+/]{0,3}={0,3})(?![A-Za-z0-9+/=])"
26
+ )
27
+
28
+ _REDACT_JSON_KEYS = frozenset(
29
+ {
30
+ "data",
31
+ "base64",
32
+ "image_data",
33
+ "content_base64",
34
+ "blob",
35
+ "binary",
36
+ "image",
37
+ "screenshot",
38
+ "audio",
39
+ "bytes",
40
+ "payload",
41
+ }
42
+ )
43
+
44
+ # Workspace tools return faithful text/JSON; skip ingress JSON field redaction (see CoreToolExecutor).
45
+ SANITIZE_SKIP_TOOL_NAMES = frozenset(
46
+ {
47
+ "read_file",
48
+ "write_file",
49
+ "replace_in_file",
50
+ "glob",
51
+ "search_memory",
52
+ "list_skills",
53
+ }
54
+ )
55
+
56
+
57
+ def skip_tool_result_sanitize(tool_name: str) -> bool:
58
+ """True when tool results must not pass through ``sanitize_tool_result_text``."""
59
+ return tool_name in SANITIZE_SKIP_TOOL_NAMES
60
+
61
+
62
+ def _int_from_env(name: str, default: int) -> int:
63
+ raw = os.environ.get(name, "").strip()
64
+ if not raw:
65
+ return default
66
+ try:
67
+ return max(0, int(raw))
68
+ except ValueError:
69
+ logger.warning("invalid env var %s=%r; using default %s", name, raw, default)
70
+ return default
71
+
72
+
73
+ def sanitize_enabled_from_env() -> bool:
74
+ raw = os.environ.get("MONKEYBOT_TOOL_RESULT_SANITIZE", "1").strip().lower()
75
+ return raw not in ("0", "false", "no", "off")
76
+
77
+
78
+ def tool_result_max_chars_from_env() -> int:
79
+ return _int_from_env("MONKEYBOT_TOOL_RESULT_MAX_CHARS", _DEFAULT_TOOL_RESULT_MAX_CHARS)
80
+
81
+
82
+ def summary_tool_result_max_chars_from_env() -> int:
83
+ return _int_from_env(
84
+ "MONKEYBOT_SUMMARY_TOOL_RESULT_MAX_CHARS",
85
+ _DEFAULT_SUMMARY_TOOL_RESULT_MAX_CHARS,
86
+ )
87
+
88
+
89
+ def json_field_max_chars_from_env() -> int:
90
+ """Max length for *denylisted* JSON blob fields before omission.
91
+
92
+ Does **not** apply to ordinary text fields (``tree``, ``result``, etc.).
93
+ Env: ``MONKEYBOT_TOOL_RESULT_JSON_FIELD_MAX`` (legacy name kept).
94
+ """
95
+ return _int_from_env(
96
+ "MONKEYBOT_TOOL_RESULT_JSON_FIELD_MAX", _DEFAULT_BLOB_JSON_FIELD_MAX_CHARS
97
+ )
98
+
99
+
100
+ def _looks_like_base64(value: str) -> bool:
101
+ """Return True when *value* is likely base64, including short JSON field payloads."""
102
+ if not value:
103
+ return False
104
+ if _is_plausible_base64_run(value):
105
+ return True
106
+ stripped = value.strip()
107
+ if len(stripped) < 16:
108
+ return False
109
+ if any(
110
+ ch not in "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r"
111
+ for ch in stripped
112
+ ):
113
+ return False
114
+ sample = stripped[:512]
115
+ if len(set(sample.strip("="))) <= 2:
116
+ return False
117
+ if stripped.rstrip().endswith("=") or any(ch in "+/" for ch in sample):
118
+ return True
119
+ return len(set(sample)) >= 6
120
+
121
+
122
+ def _is_plausible_base64_run(text: str) -> bool:
123
+ if len(text) < _MIN_BASE64_RUN:
124
+ return False
125
+ sample = text[:4096]
126
+ if len(set(sample.strip("="))) <= 2:
127
+ return False
128
+ if any(ch in "+/" for ch in sample):
129
+ return True
130
+ if sample.rstrip().endswith("="):
131
+ return True
132
+ allowed = sum(1 for ch in sample if ch.isalnum() or ch in "+/=\n\r")
133
+ return allowed / max(1, len(sample)) >= 0.98 and len(set(sample)) >= 8
134
+
135
+
136
+ def _replace_long_b64(match: re.Match[str]) -> str:
137
+ run = match.group(0)
138
+ if not _is_plausible_base64_run(run):
139
+ return run
140
+ return f"[omitted ~{len(run)} char base64 run]"
141
+
142
+
143
+ def _should_redact_blob_field(key: str, value: str, *, max_field: int) -> bool:
144
+ """True only for denylisted blob keys that look like binary/base64 or are oversized."""
145
+ if key.lower() not in _REDACT_JSON_KEYS:
146
+ return False
147
+ if _looks_like_base64(value):
148
+ return True
149
+ return max_field > 0 and len(value) > max_field
150
+
151
+
152
+ def _redact_json_value(key: str, value: Any, *, max_field: int) -> Any:
153
+ if isinstance(value, str):
154
+ if _should_redact_blob_field(key, value, max_field=max_field):
155
+ return f"[omitted {len(value)} chars from {key!r}]"
156
+ return value
157
+ if isinstance(value, dict):
158
+ return {k: _redact_json_value(str(k), v, max_field=max_field) for k, v in value.items()}
159
+ if isinstance(value, list):
160
+ return [_redact_json_value(key, item, max_field=max_field) for item in value]
161
+ return value
162
+
163
+
164
+ def _redact_json_text(text: str, *, max_field: int) -> str:
165
+ stripped = text.lstrip()
166
+ if not stripped.startswith(("{", "[")):
167
+ return text
168
+ try:
169
+ parsed = json.loads(text)
170
+ except json.JSONDecodeError:
171
+ return text
172
+ redacted = _redact_json_value("", parsed, max_field=max_field)
173
+ try:
174
+ return json.dumps(redacted, ensure_ascii=False, default=str)
175
+ except TypeError:
176
+ return text
177
+
178
+
179
+ def sanitize_tool_result_text(text: str, *, enabled: bool | None = None) -> str:
180
+ """Strip embedded binary/base64 payloads from tool result strings.
181
+
182
+ Ordinary long text fields (e.g. browser ``tree``) are left intact; size is
183
+ controlled later by spill (``MONKEYBOT_SPILL_MIN_CHARS``) and
184
+ ``MONKEYBOT_TOOL_RESULT_MAX_CHARS``. Only denylisted blob keys are
185
+ field-redacted here.
186
+ """
187
+ if not text:
188
+ return text
189
+ if enabled is None:
190
+ enabled = sanitize_enabled_from_env()
191
+ if not enabled:
192
+ return text
193
+
194
+ max_field = json_field_max_chars_from_env()
195
+ out = _DATA_URL_RE.sub(
196
+ lambda m: f"data:...;base64,[omitted {len(m.group(0))} chars]",
197
+ text,
198
+ )
199
+ out = _LONG_B64_RE.sub(_replace_long_b64, out)
200
+ return _redact_json_text(out, max_field=max_field)
201
+
202
+
203
+ def truncate_with_note(text: str, max_chars: int, *, label: str = "tool result") -> str:
204
+ """Hard-cap ``text`` with a trailing omission note."""
205
+ if max_chars <= 0 or len(text) <= max_chars:
206
+ return text
207
+ omitted = len(text) - max_chars
208
+ return f"{text[:max_chars]}\n[{label} truncated — {omitted} chars omitted]"
209
+
210
+
211
+ def cap_tool_result_text(text: str, *, max_chars: int | None = None) -> str:
212
+ """Apply the inline history hard ceiling after sanitization/spill."""
213
+ limit = tool_result_max_chars_from_env() if max_chars is None else max_chars
214
+ return truncate_with_note(text, limit, label="tool result")
215
+
216
+
217
+ def summarize_tool_result_text(text: str, *, max_chars: int | None = None) -> str:
218
+ """Sanitize and cap text fed into context summarization."""
219
+ limit = summary_tool_result_max_chars_from_env() if max_chars is None else max_chars
220
+ cleaned = sanitize_tool_result_text(text)
221
+ return truncate_with_note(cleaned, limit, label="summarized tool result")
222
+
223
+
224
+ def format_mcp_binary_block(
225
+ *,
226
+ kind: str,
227
+ data: object,
228
+ mime: str | None,
229
+ ) -> str:
230
+ """Placeholder for MCP image/audio/resource blocks."""
231
+ size = len(str(data or ""))
232
+ mime_label = mime or f"{kind}/*"
233
+ return (
234
+ f"[{kind} omitted from tool text: mime={mime_label}, "
235
+ f"~{size} base64 chars — use a vision-capable provider or a text-returning tool]"
236
+ )
237
+
238
+
239
+ def dump_model_or_str(obj: Any) -> str:
240
+ """JSON-dump ``obj.model_dump()`` when available and serializable, else ``str(obj)``."""
241
+ md = getattr(obj, "model_dump", None)
242
+ if callable(md):
243
+ try:
244
+ dumped = md(mode="python", by_alias=True)
245
+ return json.dumps(dumped, default=str)
246
+ except TypeError as exc:
247
+ logger.debug("model_dump not JSON-serializable, falling back to str(): %s", exc)
248
+ return str(obj)
249
+
250
+
251
+ def format_mcp_content_block(block: Any, *, sanitize: bool = True) -> str:
252
+ """Turn one MCP content block into a safe string."""
253
+ blk_type = getattr(block, "type", None)
254
+ if blk_type == "text":
255
+ txt = getattr(block, "text", None)
256
+ if txt is None:
257
+ return ""
258
+ text = str(txt)
259
+ return sanitize_tool_result_text(text) if sanitize else text
260
+ if blk_type == "image":
261
+ data = getattr(block, "data", None) or ""
262
+ mime = getattr(block, "mimeType", None) or getattr(block, "mime_type", None) or "image/png"
263
+ return format_mcp_binary_block(kind="image", data=data, mime=str(mime))
264
+ if blk_type == "audio":
265
+ data = getattr(block, "data", None) or ""
266
+ mime = getattr(block, "mimeType", None) or getattr(block, "mime_type", None) or "audio/*"
267
+ return format_mcp_binary_block(kind="audio", data=data, mime=str(mime))
268
+ if blk_type == "resource":
269
+ uri = getattr(block, "uri", None) or getattr(block, "resource", None) or ""
270
+ mime = getattr(block, "mimeType", None) or getattr(block, "mime_type", None) or "resource"
271
+ return f"[resource omitted from tool text: uri={uri!s}, mime={mime}]"
272
+
273
+ text = dump_model_or_str(block)
274
+ return sanitize_tool_result_text(text) if sanitize else text
275
+
276
+
277
+ __all__ = [
278
+ "SANITIZE_SKIP_TOOL_NAMES",
279
+ "cap_tool_result_text",
280
+ "dump_model_or_str",
281
+ "format_mcp_binary_block",
282
+ "format_mcp_content_block",
283
+ "sanitize_enabled_from_env",
284
+ "sanitize_tool_result_text",
285
+ "skip_tool_result_sanitize",
286
+ "summarize_tool_result_text",
287
+ "summary_tool_result_max_chars_from_env",
288
+ "tool_result_max_chars_from_env",
289
+ "truncate_with_note",
290
+ ]
@@ -0,0 +1,361 @@
1
+ """Content-aware shaping for large tool outputs (no ML, pure Python)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ import os
8
+ import re
9
+ from collections.abc import Sequence
10
+ from typing import Any, Literal
11
+
12
+ from monkeybot.core.context.common import ContextPressureTier, text_from_blocks
13
+ from monkeybot.core.context.tool_output_policy import ToolOutputBudget, resolve_tool_budget
14
+ from monkeybot.core.llm.provider import Message
15
+ from monkeybot.core.logging_utils import kv
16
+ from monkeybot.core.types.content_blocks import ContentBlock, Text, ToolResponse
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ ContentType = Literal["json", "logs", "code", "prose"]
21
+
22
+ _DEFAULT_LOG_HEAD_LINES = 40
23
+ _DEFAULT_LOG_TAIL_LINES = 40
24
+ _DEFAULT_MAX_ARRAY_ITEMS = 50
25
+ _DEFAULT_ERROR_PATTERNS = (
26
+ r"(?i)\berror\b",
27
+ r"(?i)\bfatal\b",
28
+ r"(?i)traceback",
29
+ r"(?i)\bexception\b",
30
+ r"(?i)\bfailed\b",
31
+ )
32
+ _CODE_MARKERS = re.compile(
33
+ r"(^\s*(def |class |import |from .+ import |function |const |let |var |#include ))",
34
+ re.MULTILINE,
35
+ )
36
+
37
+
38
+ def _int_from_env(name: str, default: int) -> int:
39
+ raw = os.environ.get(name, "").strip()
40
+ if not raw:
41
+ return default
42
+ try:
43
+ return max(1, int(raw))
44
+ except ValueError:
45
+ logger.warning(
46
+ "invalid env var value %s",
47
+ kv(name=name, value=raw, default=default),
48
+ )
49
+ return default
50
+
51
+
52
+ def log_head_lines_from_env() -> int:
53
+ return _int_from_env("MONKEYBOT_SHAPE_LOG_HEAD_LINES", _DEFAULT_LOG_HEAD_LINES)
54
+
55
+
56
+ def log_tail_lines_from_env() -> int:
57
+ return _int_from_env("MONKEYBOT_SHAPE_LOG_TAIL_LINES", _DEFAULT_LOG_TAIL_LINES)
58
+
59
+
60
+ def max_array_items_from_env() -> int:
61
+ return _int_from_env("MONKEYBOT_SHAPE_MAX_ARRAY_ITEMS", _DEFAULT_MAX_ARRAY_ITEMS)
62
+
63
+
64
+ def classify_content(text: str, *, tool_name: str, hint: str | None = None) -> ContentType:
65
+ """Route tool text to a shaper category."""
66
+ if hint and hint != "auto":
67
+ return hint # type: ignore[return-value]
68
+
69
+ stripped = text.lstrip()
70
+ if tool_name in ("run_command",):
71
+ return "logs"
72
+ if tool_name in ("web_search", "search_memory", "task") and stripped.startswith(("{", "[")):
73
+ return "json"
74
+ if tool_name in ("read_file", "write_file", "replace_in_file", "glob"):
75
+ return "code"
76
+
77
+ if stripped.startswith(("{", "[")):
78
+ try:
79
+ json.loads(text)
80
+ return "json"
81
+ except json.JSONDecodeError:
82
+ pass
83
+
84
+ lines = text.splitlines()
85
+ if len(lines) >= 8:
86
+ duplicate_run = sum(
87
+ 1 for i in range(1, len(lines)) if lines[i] == lines[i - 1] and lines[i].strip()
88
+ )
89
+ if duplicate_run >= max(3, len(lines) // 10):
90
+ return "logs"
91
+ if any(_CODE_MARKERS.search(line) for line in lines[:20]):
92
+ return "code"
93
+
94
+ if len(lines) >= 30 and not stripped.startswith("{"):
95
+ return "logs"
96
+
97
+ return "prose"
98
+
99
+
100
+ def _compiled_keep_patterns(patterns: tuple[str, ...]) -> tuple[re.Pattern[str], ...]:
101
+ src = patterns or _DEFAULT_ERROR_PATTERNS
102
+ out: list[re.Pattern[str]] = []
103
+ for pat in src:
104
+ try:
105
+ out.append(re.compile(pat))
106
+ except re.error:
107
+ continue
108
+ return tuple(out)
109
+
110
+
111
+ def _line_matches_keep(line: str, patterns: tuple[re.Pattern[str], ...]) -> bool:
112
+ return any(rx.search(line) for rx in patterns)
113
+
114
+
115
+ def shape_logs(
116
+ text: str,
117
+ *,
118
+ max_lines: int | None,
119
+ keep_patterns: tuple[str, ...],
120
+ collapse_repeated: bool,
121
+ ) -> str:
122
+ lines = text.splitlines()
123
+ if len(lines) <= 1:
124
+ return text
125
+
126
+ compiled = _compiled_keep_patterns(keep_patterns)
127
+ if collapse_repeated:
128
+ collapsed: list[str] = []
129
+ prev: str | None = None
130
+ repeat_count = 0
131
+ for line in lines:
132
+ if line == prev and line.strip():
133
+ repeat_count += 1
134
+ continue
135
+ if repeat_count > 0 and prev is not None:
136
+ collapsed.append(f"{prev} … (repeated {repeat_count + 1}×)")
137
+ repeat_count = 0
138
+ if prev is not None:
139
+ collapsed.append(prev)
140
+ prev = line
141
+ if repeat_count > 0 and prev is not None:
142
+ collapsed.append(f"{prev} … (repeated {repeat_count + 1}×)")
143
+ elif prev is not None:
144
+ collapsed.append(prev)
145
+ lines = collapsed
146
+
147
+ cap = max_lines
148
+ if cap is None or len(lines) <= cap:
149
+ return "\n".join(lines)
150
+
151
+ head_n = min(log_head_lines_from_env(), max(1, cap // 2))
152
+ tail_n = min(log_tail_lines_from_env(), max(1, cap - head_n))
153
+ keep_indices = set(range(min(head_n, len(lines))))
154
+ keep_indices.update(range(max(0, len(lines) - tail_n), len(lines)))
155
+ for i, line in enumerate(lines):
156
+ if _line_matches_keep(line, compiled):
157
+ keep_indices.add(i)
158
+
159
+ if len(keep_indices) >= cap:
160
+ ordered = sorted(keep_indices)
161
+ selected = [lines[i] for i in ordered[:cap]]
162
+ omitted = len(lines) - len(selected)
163
+ if omitted > 0:
164
+ selected.append(f"… (+{omitted} more lines omitted by log shaper)")
165
+ return "\n".join(selected)
166
+
167
+ ordered = sorted(keep_indices)
168
+ selected = [lines[i] for i in ordered]
169
+ omitted = len(lines) - len(ordered)
170
+ if omitted > 0:
171
+ insert_at = len(selected)
172
+ for i in range(len(lines)):
173
+ if i not in keep_indices:
174
+ insert_at = min(insert_at, len(selected))
175
+ break
176
+ marker = f"… (+{omitted} lines omitted by log shaper)"
177
+ mid = len(selected) // 2
178
+ selected = [*selected[:mid], marker, *selected[mid:]]
179
+ return "\n".join(selected)
180
+
181
+
182
+ def _json_dedup_key(item: Any) -> str:
183
+ try:
184
+ return json.dumps(item, sort_keys=True, default=str)
185
+ except TypeError:
186
+ return repr(item)
187
+
188
+
189
+ def _json_row_is_anomaly(item: Any) -> bool:
190
+ if isinstance(item, dict):
191
+ blob = json.dumps(item, sort_keys=True).lower()
192
+ return any(k in blob for k in ("error", "exception", "fail", "fatal"))
193
+ if isinstance(item, str):
194
+ lower = item.lower()
195
+ return any(k in lower for k in ("error", "exception", "fail", "fatal"))
196
+ return False
197
+
198
+
199
+ def _shape_json_value(value: Any, *, max_array_items: int, depth: int = 0) -> Any:
200
+ if depth > 6:
201
+ return value
202
+ if isinstance(value, list):
203
+ seen: set[str] = set()
204
+ kept: list[Any] = []
205
+ anomalies: list[Any] = []
206
+ for item in value:
207
+ if _json_row_is_anomaly(item):
208
+ anomalies.append(_shape_json_value(item, max_array_items=max_array_items, depth=depth + 1))
209
+ continue
210
+ key = _json_dedup_key(item)
211
+ if key in seen:
212
+ continue
213
+ seen.add(key)
214
+ kept.append(_shape_json_value(item, max_array_items=max_array_items, depth=depth + 1))
215
+ merged = anomalies + kept
216
+ if len(merged) > max_array_items:
217
+ trimmed = merged[:max_array_items]
218
+ trimmed.append(f"… (+{len(merged) - max_array_items} more items)")
219
+ return trimmed
220
+ return merged
221
+ if isinstance(value, dict):
222
+ out: dict[str, Any] = {}
223
+ for k, v in value.items():
224
+ if v is None or v == "" or v == [] or v == {}:
225
+ continue
226
+ out[str(k)] = _shape_json_value(v, max_array_items=max_array_items, depth=depth + 1)
227
+ return out
228
+ return value
229
+
230
+
231
+ def shape_json(text: str, *, max_array_items: int | None) -> str:
232
+ try:
233
+ parsed = json.loads(text)
234
+ except json.JSONDecodeError:
235
+ return text
236
+ cap = max_array_items if max_array_items is not None else max_array_items_from_env()
237
+ shaped = _shape_json_value(parsed, max_array_items=cap)
238
+ return json.dumps(shaped, indent=2, ensure_ascii=False, default=str)
239
+
240
+
241
+ def exceeds_tool_output_budget(text: str, *, tool_name: str, budget: ToolOutputBudget) -> bool:
242
+ """True when source text exceeds configured per-tool caps (safe to shape at append)."""
243
+ if not text.strip():
244
+ return False
245
+ lines = text.splitlines()
246
+ if budget.max_output_lines is not None and len(lines) > budget.max_output_lines:
247
+ return True
248
+ if budget.max_array_items is not None and tool_name in ("web_search", "search_memory", "task"):
249
+ stripped = text.lstrip()
250
+ if stripped.startswith(("[", "{")):
251
+ try:
252
+ parsed = json.loads(text)
253
+ except json.JSONDecodeError:
254
+ return False
255
+ if isinstance(parsed, list) and len(parsed) > budget.max_array_items:
256
+ return True
257
+ if budget.collapse_repeated and len(lines) > 30:
258
+ repeats = sum(
259
+ 1 for i in range(1, len(lines)) if lines[i] == lines[i - 1] and lines[i].strip()
260
+ )
261
+ if repeats >= 5:
262
+ return True
263
+ return False
264
+
265
+
266
+ def shape_tool_text(
267
+ text: str,
268
+ *,
269
+ tool_name: str,
270
+ budget: ToolOutputBudget | None,
271
+ pressure_tier: ContextPressureTier | None = None,
272
+ ) -> str:
273
+ """Apply content-aware shaping when policy or context pressure warrants it."""
274
+ if not text or not text.strip():
275
+ return text
276
+
277
+ has_policy = budget is not None
278
+ pressure_shape = pressure_tier in ("moderate", "aggressive")
279
+ if not has_policy and not pressure_shape:
280
+ return text
281
+
282
+ hint = budget.content_type if budget is not None else None
283
+ content_type = classify_content(text, tool_name=tool_name, hint=hint)
284
+
285
+ if content_type == "code":
286
+ return text
287
+ if content_type == "prose" and not pressure_shape:
288
+ return text
289
+
290
+ if content_type == "json":
291
+ cap = budget.max_array_items if budget is not None else None
292
+ return shape_json(text, max_array_items=cap)
293
+
294
+ if content_type == "logs":
295
+ return shape_logs(
296
+ text,
297
+ max_lines=budget.max_output_lines if budget is not None else None,
298
+ keep_patterns=budget.keep_patterns if budget is not None else (),
299
+ collapse_repeated=bool(budget and budget.collapse_repeated),
300
+ )
301
+
302
+ return text
303
+
304
+
305
+ def shape_messages_tool_results(
306
+ messages: Sequence[Message],
307
+ *,
308
+ protect_recent: int,
309
+ pressure_tier: ContextPressureTier | None,
310
+ ) -> list[Message]:
311
+ """Shape tool results in older history rows when context pressure is elevated.
312
+
313
+ The most recent ``protect_recent`` messages are left untouched so the model
314
+ always sees fresh tool output verbatim.
315
+ """
316
+ if pressure_tier not in ("moderate", "aggressive"):
317
+ return list(messages)
318
+ cutoff = max(0, len(messages) - protect_recent)
319
+ out: list[Message] = []
320
+ for i, msg in enumerate(messages):
321
+ if i >= cutoff or msg.role != "user":
322
+ out.append(msg)
323
+ continue
324
+ new_content: list[ContentBlock] = []
325
+ changed = False
326
+ for block in msg.content:
327
+ if not isinstance(block, ToolResponse) or block.is_error:
328
+ new_content.append(block)
329
+ continue
330
+ text = text_from_blocks(list(block.result))
331
+ budget = resolve_tool_budget(block.tool_name)
332
+ shaped = shape_tool_text(
333
+ text,
334
+ tool_name=block.tool_name,
335
+ budget=budget,
336
+ pressure_tier=pressure_tier,
337
+ )
338
+ if shaped == text:
339
+ new_content.append(block)
340
+ continue
341
+ changed = True
342
+ new_content.append(
343
+ ToolResponse(
344
+ id=block.id,
345
+ tool_name=block.tool_name,
346
+ result=[Text(text=shaped)],
347
+ is_error=block.is_error,
348
+ )
349
+ )
350
+ out.append(Message(role=msg.role, content=new_content) if changed else msg)
351
+ return out
352
+
353
+
354
+ __all__ = [
355
+ "classify_content",
356
+ "exceeds_tool_output_budget",
357
+ "shape_json",
358
+ "shape_logs",
359
+ "shape_messages_tool_results",
360
+ "shape_tool_text",
361
+ ]