monkeybot 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- monkeybot/__init__.py +3 -0
- monkeybot/cli/__init__.py +3 -0
- monkeybot/cli/__main__.py +8 -0
- monkeybot/cli/audio_io.py +8 -0
- monkeybot/cli/gateway_manager.py +17 -0
- monkeybot/cli/main.py +22 -0
- monkeybot/cli/push_to_talk.py +12 -0
- monkeybot/cli/realtime_client.py +13 -0
- monkeybot/core/__init__.py +19 -0
- monkeybot/core/attachments/__init__.py +22 -0
- monkeybot/core/attachments/catalog.py +62 -0
- monkeybot/core/attachments/config.py +52 -0
- monkeybot/core/attachments/freeze.py +158 -0
- monkeybot/core/attachments/resolve.py +70 -0
- monkeybot/core/attachments/store.py +180 -0
- monkeybot/core/attachments/text.py +72 -0
- monkeybot/core/attachments/tools.py +54 -0
- monkeybot/core/bootstrap.py +242 -0
- monkeybot/core/config/__init__.py +71 -0
- monkeybot/core/config/realtime_config.py +150 -0
- monkeybot/core/config/runtime_env.py +262 -0
- monkeybot/core/config/settings.py +341 -0
- monkeybot/core/config/validation.py +249 -0
- monkeybot/core/config/yaml_loader.py +45 -0
- monkeybot/core/context/__init__.py +781 -0
- monkeybot/core/context/campaign_context.py +8 -0
- monkeybot/core/context/common.py +14 -0
- monkeybot/core/context/curator.py +255 -0
- monkeybot/core/context/epoch.py +226 -0
- monkeybot/core/context/memory_prompt.py +222 -0
- monkeybot/core/context/tool_output_policy.py +270 -0
- monkeybot/core/context/tool_result_ingress.py +290 -0
- monkeybot/core/context/tool_shapers.py +361 -0
- monkeybot/core/hooks/__init__.py +261 -0
- monkeybot/core/llm/__init__.py +4 -0
- monkeybot/core/llm/provider.py +296 -0
- monkeybot/core/llm/realtime_provider.py +203 -0
- monkeybot/core/llm/usage.py +57 -0
- monkeybot/core/logging_utils.py +24 -0
- monkeybot/core/mcp/__init__.py +1 -0
- monkeybot/core/mcp/mcp_client.py +1215 -0
- monkeybot/core/mcp/ports_mcp.py +109 -0
- monkeybot/core/memory/__init__.py +24 -0
- monkeybot/core/memory/hook.py +413 -0
- monkeybot/core/memory/index_format.py +104 -0
- monkeybot/core/memory/integrity.py +180 -0
- monkeybot/core/memory/organizer.py +270 -0
- monkeybot/core/memory/storage_ops.py +139 -0
- monkeybot/core/memory/subsystem.py +91 -0
- monkeybot/core/messages/__init__.py +16 -0
- monkeybot/core/messages/convert_provider.py +41 -0
- monkeybot/core/messages/tool_integrity.py +262 -0
- monkeybot/core/messages/transform_context.py +84 -0
- monkeybot/core/path_safety.py +11 -0
- monkeybot/core/persistence/__init__.py +17 -0
- monkeybot/core/persistence/backends.py +236 -0
- monkeybot/core/persistence/db.py +28 -0
- monkeybot/core/persistence/durable_runs.py +286 -0
- monkeybot/core/persistence/firestore.py +658 -0
- monkeybot/core/persistence/firestore_scheduled_loops.py +336 -0
- monkeybot/core/persistence/history.py +156 -0
- monkeybot/core/persistence/postgres.py +895 -0
- monkeybot/core/persistence/runs.py +76 -0
- monkeybot/core/persistence/scheduled_loops.py +435 -0
- monkeybot/core/persistence/session_turn_locks.py +94 -0
- monkeybot/core/persistence/sqlite.py +218 -0
- monkeybot/core/persistence/sqlite_backend.py +74 -0
- monkeybot/core/persistence/thread_summary.py +61 -0
- monkeybot/core/persistence/transcript.py +194 -0
- monkeybot/core/persistence/usage.py +149 -0
- monkeybot/core/prompts/__init__.py +1 -0
- monkeybot/core/prompts/harness_prompt.py +197 -0
- monkeybot/core/prompts/prompt.py +215 -0
- monkeybot/core/runtime/__init__.py +1 -0
- monkeybot/core/runtime/context_budget.py +267 -0
- monkeybot/core/runtime/events.py +819 -0
- monkeybot/core/runtime/input_admission.py +154 -0
- monkeybot/core/runtime/loop.py +2374 -0
- monkeybot/core/runtime/provider_stream_mapper.py +159 -0
- monkeybot/core/runtime/realtime_loop.py +654 -0
- monkeybot/core/runtime/utterance_buffer.py +179 -0
- monkeybot/core/subagents/__init__.py +1 -0
- monkeybot/core/subagents/subagent_proto.py +331 -0
- monkeybot/core/subagents/subagent_worker.py +441 -0
- monkeybot/core/subagents/worker_pool.py +403 -0
- monkeybot/core/testing/__init__.py +1 -0
- monkeybot/core/testing/mocks_provider.py +86 -0
- monkeybot/core/testing/mocks_realtime_provider.py +137 -0
- monkeybot/core/tools/__init__.py +1 -0
- monkeybot/core/tools/core_tool_executor.py +1548 -0
- monkeybot/core/tools/inspector.py +226 -0
- monkeybot/core/tools/loop_inspector.py +45 -0
- monkeybot/core/tools/patch.py +480 -0
- monkeybot/core/tools/permission.py +284 -0
- monkeybot/core/tools/sandbox_executor.py +255 -0
- monkeybot/core/tools/spill_inventory.py +35 -0
- monkeybot/core/tools/terminal.py +381 -0
- monkeybot/core/tools/text_normalize.py +25 -0
- monkeybot/core/tools/types.py +33 -0
- monkeybot/core/tools/workspace_service.py +710 -0
- monkeybot/core/tools/workspace_tools.py +116 -0
- monkeybot/core/types/__init__.py +1 -0
- monkeybot/core/types/content_blocks.py +644 -0
- monkeybot/core/types/interfaces.py +156 -0
- monkeybot/core/types/types_tools.py +29 -0
- monkeybot/core/workspace/__init__.py +8 -0
- monkeybot/core/workspace/factory.py +45 -0
- monkeybot/core/workspace/gcs.py +130 -0
- monkeybot/core/workspace/local.py +162 -0
- monkeybot/core/workspace/protocol.py +45 -0
- monkeybot/core/workspace/s3.py +151 -0
- monkeybot/core/workspace_layout.py +27 -0
- monkeybot/gateway/__init__.py +1 -0
- monkeybot/gateway/bootstrap.py +18 -0
- monkeybot/gateway/main.py +47 -0
- monkeybot/gateway/realtime/__init__.py +31 -0
- monkeybot/gateway/realtime/app.py +321 -0
- monkeybot/gateway/realtime/deps.py +52 -0
- monkeybot/gateway/realtime/errors.py +81 -0
- monkeybot/gateway/realtime/guardrails.py +88 -0
- monkeybot/gateway/realtime/manager.py +77 -0
- monkeybot/gateway/realtime/metrics.py +144 -0
- monkeybot/gateway/realtime/routes.py +864 -0
- monkeybot/gateway/realtime/session.py +232 -0
- monkeybot/gateway/realtime/wire.py +412 -0
- monkeybot/gateway/realtime_main.py +49 -0
- monkeybot/gateway/sse/__init__.py +1 -0
- monkeybot/gateway/sse/app.py +733 -0
- monkeybot/gateway/sse/loop_port.py +31 -0
- monkeybot/gateway/sse/models.py +177 -0
- monkeybot/gateway/sse/reply_body.py +91 -0
- monkeybot/gateway/sse/routes.py +1101 -0
- monkeybot/gateway/sse/scheduler_routes.py +200 -0
- monkeybot/gateway/sse/scheduler_wiring.py +96 -0
- monkeybot/gateway/sse/session_bus.py +226 -0
- monkeybot/gateway/sse/sse.py +46 -0
- monkeybot/gateway/sse/workspace_layout.py +7 -0
- monkeybot/observability/__init__.py +220 -0
- monkeybot/observability/_state.py +10 -0
- monkeybot/observability/instrumentation.py +153 -0
- monkeybot/observability/propagation.py +65 -0
- monkeybot/observability/spans.py +455 -0
- monkeybot/providers/__init__.py +19 -0
- monkeybot/providers/_openai_compat.py +450 -0
- monkeybot/providers/_utils.py +473 -0
- monkeybot/providers/bedrock.py +145 -0
- monkeybot/providers/claude.py +125 -0
- monkeybot/providers/gemini.py +677 -0
- monkeybot/providers/gemini_live.py +398 -0
- monkeybot/providers/huggingface.py +129 -0
- monkeybot/providers/nvidia.py +104 -0
- monkeybot/providers/ollama.py +152 -0
- monkeybot/providers/openai.py +127 -0
- monkeybot/providers/pricing.py +60 -0
- monkeybot/providers/sampling.py +44 -0
- monkeybot/providers/vertex_claude.py +148 -0
- monkeybot/scaffold/__init__.py +33 -0
- monkeybot/scheduler/__init__.py +13 -0
- monkeybot/scheduler/__main__.py +4 -0
- monkeybot/scheduler/engine.py +333 -0
- monkeybot/scheduler/http_invoker.py +61 -0
- monkeybot/scheduler/interval.py +77 -0
- monkeybot/scheduler/tick_result.py +34 -0
- monkeybot/scheduler/worker.py +87 -0
- monkeybot/subagents/__init__.py +1 -0
- monkeybot/subagents/worker/__init__.py +1 -0
- monkeybot/subagents/worker/__main__.py +22 -0
- monkeybot/web_search/__init__.py +82 -0
- monkeybot/web_search/backends/__init__.py +5 -0
- monkeybot/web_search/backends/duckduckgo.py +32 -0
- monkeybot/web_search/backends/firecrawl.py +43 -0
- monkeybot/web_search/backends/tavily.py +45 -0
- monkeybot/web_search/protocol.py +25 -0
- monkeybot/web_search/tool.py +56 -0
- monkeybot-2.1.1.dist-info/METADATA +318 -0
- monkeybot-2.1.1.dist-info/RECORD +178 -0
- monkeybot-2.1.1.dist-info/WHEEL +4 -0
- monkeybot-2.1.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,710 @@
|
|
|
1
|
+
"""Safe repo-scoped read / write / replace / glob / grep for workspace API and agent tools."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import fnmatch
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import TypedDict
|
|
12
|
+
|
|
13
|
+
from monkeybot.core.tools.text_normalize import normalize_unicode_punctuation
|
|
14
|
+
|
|
15
|
+
# Directories skipped when walking for grep: noisy, large, or not source content.
|
|
16
|
+
_GREP_IGNORE_DIRS = frozenset(
|
|
17
|
+
{
|
|
18
|
+
".git",
|
|
19
|
+
".hg",
|
|
20
|
+
".svn",
|
|
21
|
+
"node_modules",
|
|
22
|
+
"__pycache__",
|
|
23
|
+
".venv",
|
|
24
|
+
"venv",
|
|
25
|
+
".mypy_cache",
|
|
26
|
+
".pytest_cache",
|
|
27
|
+
".ruff_cache",
|
|
28
|
+
".tox",
|
|
29
|
+
"dist",
|
|
30
|
+
"build",
|
|
31
|
+
".next",
|
|
32
|
+
}
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ReadFileResult(TypedDict):
|
|
37
|
+
ok: bool
|
|
38
|
+
path: str
|
|
39
|
+
content: str
|
|
40
|
+
start_line: int
|
|
41
|
+
end_line: int
|
|
42
|
+
total_lines: int
|
|
43
|
+
truncated: bool
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class WriteFileResult(TypedDict):
|
|
47
|
+
ok: bool
|
|
48
|
+
path: str
|
|
49
|
+
bytes: int
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class ReplaceResult(TypedDict):
|
|
53
|
+
ok: bool
|
|
54
|
+
path: str
|
|
55
|
+
replacements: int
|
|
56
|
+
bytes: int
|
|
57
|
+
match_mode: str
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class DeleteResult(TypedDict):
|
|
61
|
+
ok: bool
|
|
62
|
+
path: str
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class GrepMatch(TypedDict):
|
|
66
|
+
path: str
|
|
67
|
+
line: int
|
|
68
|
+
text: str
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class GlobResult(TypedDict):
|
|
72
|
+
ok: bool
|
|
73
|
+
root: str
|
|
74
|
+
pattern: str
|
|
75
|
+
paths: list[str]
|
|
76
|
+
count: int
|
|
77
|
+
truncated: bool
|
|
78
|
+
duration_ms: int
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class GrepResult(TypedDict):
|
|
82
|
+
ok: bool
|
|
83
|
+
root: str
|
|
84
|
+
pattern: str
|
|
85
|
+
matches: list[GrepMatch]
|
|
86
|
+
match_count: int
|
|
87
|
+
files_scanned: int
|
|
88
|
+
truncated: bool
|
|
89
|
+
duration_ms: int
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass
|
|
93
|
+
class WorkspaceSettings:
|
|
94
|
+
"""Defaults for workspace file limits (override via env in callers or pass explicit settings)."""
|
|
95
|
+
|
|
96
|
+
WORKSPACE_READ_MAX_LINES: int = 50000
|
|
97
|
+
WORKSPACE_READ_DEFAULT_LINES: int = 20000
|
|
98
|
+
WORKSPACE_SPILL_READ_MAX_LINES: int = 50_000
|
|
99
|
+
WORKSPACE_WRITE_MAX_BYTES: int = 8_000_000
|
|
100
|
+
WORKSPACE_GLOB_MAX_PATHS: int = 2000
|
|
101
|
+
WORKSPACE_GLOB_TIMEOUT_SEC: float = 20.0
|
|
102
|
+
WORKSPACE_GREP_MAX_MATCHES: int = 500
|
|
103
|
+
WORKSPACE_GREP_MAX_FILES: int = 5000
|
|
104
|
+
WORKSPACE_GREP_MAX_FILE_BYTES: int = 512_000
|
|
105
|
+
# If set (repo-relative POSIX prefix, no leading slash), write_file / replace_in_file
|
|
106
|
+
# only allow paths under repo_root / this prefix (e.g. sync-backed agent memory).
|
|
107
|
+
WORKSPACE_WRITE_SCOPE_REL: str | None = None
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class WorkspaceError(Exception):
|
|
111
|
+
"""Logical error for workspace operations (maps to HTTP 400)."""
|
|
112
|
+
|
|
113
|
+
def __init__(self, message: str, code: str = "workspace_error") -> None:
|
|
114
|
+
super().__init__(message)
|
|
115
|
+
self.code = code
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _coerce_workspace_settings(settings: object | None) -> WorkspaceSettings:
|
|
119
|
+
"""Accept WorkspaceSettings, any object with same attribute names, or None."""
|
|
120
|
+
if settings is None:
|
|
121
|
+
return WorkspaceSettings()
|
|
122
|
+
if isinstance(settings, WorkspaceSettings):
|
|
123
|
+
return settings
|
|
124
|
+
out = WorkspaceSettings()
|
|
125
|
+
for field in (
|
|
126
|
+
"WORKSPACE_READ_MAX_LINES",
|
|
127
|
+
"WORKSPACE_READ_DEFAULT_LINES",
|
|
128
|
+
"WORKSPACE_SPILL_READ_MAX_LINES",
|
|
129
|
+
"WORKSPACE_WRITE_MAX_BYTES",
|
|
130
|
+
"WORKSPACE_GLOB_MAX_PATHS",
|
|
131
|
+
"WORKSPACE_GLOB_TIMEOUT_SEC",
|
|
132
|
+
"WORKSPACE_GREP_MAX_MATCHES",
|
|
133
|
+
"WORKSPACE_GREP_MAX_FILES",
|
|
134
|
+
"WORKSPACE_GREP_MAX_FILE_BYTES",
|
|
135
|
+
"WORKSPACE_WRITE_SCOPE_REL",
|
|
136
|
+
):
|
|
137
|
+
val = getattr(settings, field, None)
|
|
138
|
+
if val is not None:
|
|
139
|
+
setattr(out, field, val)
|
|
140
|
+
return out
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _is_disproportionate_match(search: str, old_string: str) -> bool:
|
|
144
|
+
"""Reject fuzzy spans that are much larger than the caller's old_string."""
|
|
145
|
+
old_lines = old_string.split("\n")
|
|
146
|
+
search_lines = search.split("\n")
|
|
147
|
+
old_n = len(old_lines)
|
|
148
|
+
search_n = len(search_lines)
|
|
149
|
+
if search_n >= max(old_n + 3, old_n * 2):
|
|
150
|
+
return True
|
|
151
|
+
if old_n == 1:
|
|
152
|
+
return False
|
|
153
|
+
return len(search.strip()) > max(len(old_string.strip()) + 500, len(old_string.strip()) * 4)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _collect_exact_spans(text: str, needle: str) -> list[tuple[int, int]]:
|
|
157
|
+
spans: list[tuple[int, int]] = []
|
|
158
|
+
start = 0
|
|
159
|
+
while True:
|
|
160
|
+
idx = text.find(needle, start)
|
|
161
|
+
if idx == -1:
|
|
162
|
+
break
|
|
163
|
+
spans.append((idx, idx + len(needle)))
|
|
164
|
+
start = idx + max(len(needle), 1)
|
|
165
|
+
return spans
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _line_trimmed_spans(content: str, find: str) -> list[tuple[int, int]]:
|
|
169
|
+
original_lines = content.split("\n")
|
|
170
|
+
search_lines = find.split("\n")
|
|
171
|
+
if search_lines and search_lines[-1] == "":
|
|
172
|
+
search_lines = search_lines[:-1]
|
|
173
|
+
if not search_lines:
|
|
174
|
+
return []
|
|
175
|
+
spans: list[tuple[int, int]] = []
|
|
176
|
+
for i in range(0, len(original_lines) - len(search_lines) + 1):
|
|
177
|
+
if all(
|
|
178
|
+
original_lines[i + j].strip() == search_lines[j].strip()
|
|
179
|
+
for j in range(len(search_lines))
|
|
180
|
+
):
|
|
181
|
+
start = sum(len(original_lines[k]) + 1 for k in range(i))
|
|
182
|
+
end = start + sum(len(original_lines[i + j]) for j in range(len(search_lines)))
|
|
183
|
+
if len(search_lines) > 1:
|
|
184
|
+
end += len(search_lines) - 1
|
|
185
|
+
spans.append((start, end))
|
|
186
|
+
return spans
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _whitespace_normalized_spans(content: str, find: str) -> list[tuple[int, int]]:
|
|
190
|
+
def norm(s: str) -> str:
|
|
191
|
+
return re.sub(r"\s+", " ", s).strip()
|
|
192
|
+
|
|
193
|
+
normalized_find = norm(find)
|
|
194
|
+
spans: list[tuple[int, int]] = []
|
|
195
|
+
lines = content.split("\n")
|
|
196
|
+
for i, line in enumerate(lines):
|
|
197
|
+
if norm(line) == normalized_find:
|
|
198
|
+
start = sum(len(lines[k]) + 1 for k in range(i))
|
|
199
|
+
spans.append((start, start + len(line)))
|
|
200
|
+
find_lines = find.split("\n")
|
|
201
|
+
if len(find_lines) > 1:
|
|
202
|
+
for i in range(0, len(lines) - len(find_lines) + 1):
|
|
203
|
+
block = "\n".join(lines[i : i + len(find_lines)])
|
|
204
|
+
if norm(block) == normalized_find:
|
|
205
|
+
start = sum(len(lines[k]) + 1 for k in range(i))
|
|
206
|
+
spans.append((start, start + len(block)))
|
|
207
|
+
return spans
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _indentation_flexible_spans(content: str, find: str) -> list[tuple[int, int]]:
|
|
211
|
+
def deindent(text: str) -> str:
|
|
212
|
+
lines = text.split("\n")
|
|
213
|
+
nonempty = [ln for ln in lines if ln.strip()]
|
|
214
|
+
if not nonempty:
|
|
215
|
+
return text
|
|
216
|
+
min_indent = min(len(ln) - len(ln.lstrip()) for ln in nonempty)
|
|
217
|
+
return "\n".join(
|
|
218
|
+
ln if not ln.strip() else ln[min_indent:] for ln in lines
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
normalized_find = deindent(find)
|
|
222
|
+
content_lines = content.split("\n")
|
|
223
|
+
find_lines = find.split("\n")
|
|
224
|
+
spans: list[tuple[int, int]] = []
|
|
225
|
+
for i in range(0, len(content_lines) - len(find_lines) + 1):
|
|
226
|
+
block = "\n".join(content_lines[i : i + len(find_lines)])
|
|
227
|
+
if deindent(block) == normalized_find:
|
|
228
|
+
start = sum(len(content_lines[k]) + 1 for k in range(i))
|
|
229
|
+
spans.append((start, start + len(block)))
|
|
230
|
+
return spans
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _unicode_normalized_spans(content: str, find: str) -> list[tuple[int, int]]:
|
|
234
|
+
"""Match after stripping and mapping smart quotes/dashes (shared with apply_patch)."""
|
|
235
|
+
original_lines = content.split("\n")
|
|
236
|
+
search_lines = find.split("\n")
|
|
237
|
+
if search_lines and search_lines[-1] == "":
|
|
238
|
+
search_lines = search_lines[:-1]
|
|
239
|
+
if not search_lines:
|
|
240
|
+
return []
|
|
241
|
+
spans: list[tuple[int, int]] = []
|
|
242
|
+
for i in range(0, len(original_lines) - len(search_lines) + 1):
|
|
243
|
+
if all(
|
|
244
|
+
normalize_unicode_punctuation(original_lines[i + j].strip())
|
|
245
|
+
== normalize_unicode_punctuation(search_lines[j].strip())
|
|
246
|
+
for j in range(len(search_lines))
|
|
247
|
+
):
|
|
248
|
+
start = sum(len(original_lines[k]) + 1 for k in range(i))
|
|
249
|
+
end = start + sum(len(original_lines[i + j]) for j in range(len(search_lines)))
|
|
250
|
+
if len(search_lines) > 1:
|
|
251
|
+
end += len(search_lines) - 1
|
|
252
|
+
spans.append((start, end))
|
|
253
|
+
return spans
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _find_replace_span(
|
|
257
|
+
text: str,
|
|
258
|
+
old_string: str,
|
|
259
|
+
*,
|
|
260
|
+
replace_all: bool,
|
|
261
|
+
) -> tuple[list[tuple[int, int]], str]:
|
|
262
|
+
"""Return (spans, match_mode). Raises :class:`WorkspaceError` if missing or ambiguous."""
|
|
263
|
+
exact = _collect_exact_spans(text, old_string)
|
|
264
|
+
if exact:
|
|
265
|
+
if len(exact) > 1 and not replace_all:
|
|
266
|
+
raise WorkspaceError(
|
|
267
|
+
"old_string is not unique; widen the snippet or set replace_all=true",
|
|
268
|
+
code="ambiguous_replace",
|
|
269
|
+
)
|
|
270
|
+
return (exact if replace_all else exact[:1]), "exact"
|
|
271
|
+
|
|
272
|
+
for mode, finder in (
|
|
273
|
+
("line_trimmed", _line_trimmed_spans),
|
|
274
|
+
("whitespace_normalized", _whitespace_normalized_spans),
|
|
275
|
+
("indentation_flexible", _indentation_flexible_spans),
|
|
276
|
+
("unicode_normalized", _unicode_normalized_spans),
|
|
277
|
+
):
|
|
278
|
+
spans = finder(text, old_string)
|
|
279
|
+
if not spans:
|
|
280
|
+
continue
|
|
281
|
+
kept: list[tuple[int, int]] = []
|
|
282
|
+
for start, end in spans:
|
|
283
|
+
matched = text[start:end]
|
|
284
|
+
if _is_disproportionate_match(matched, old_string):
|
|
285
|
+
continue
|
|
286
|
+
kept.append((start, end))
|
|
287
|
+
if not kept:
|
|
288
|
+
continue
|
|
289
|
+
if len(kept) > 1 and not replace_all:
|
|
290
|
+
raise WorkspaceError(
|
|
291
|
+
"old_string is not unique; widen the snippet or set replace_all=true",
|
|
292
|
+
code="ambiguous_replace",
|
|
293
|
+
)
|
|
294
|
+
return (kept if replace_all else kept[:1]), mode
|
|
295
|
+
|
|
296
|
+
raise WorkspaceError("old_string not found in file", code="not_found_replace")
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _apply_replace_spans(
|
|
300
|
+
text: str,
|
|
301
|
+
spans: list[tuple[int, int]],
|
|
302
|
+
new_string: str,
|
|
303
|
+
) -> str:
|
|
304
|
+
"""Apply replacements from end to start so offsets stay valid."""
|
|
305
|
+
out = text
|
|
306
|
+
for start, end in sorted(spans, key=lambda s: s[0], reverse=True):
|
|
307
|
+
out = out[:start] + new_string + out[end:]
|
|
308
|
+
return out
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
class WorkspaceFileService:
|
|
312
|
+
"""All paths are repo-relative POSIX strings; resolved under ``repo_root``."""
|
|
313
|
+
|
|
314
|
+
def __init__(self, repo_root: Path, settings: object | None = None) -> None:
|
|
315
|
+
self._root = Path(repo_root).resolve()
|
|
316
|
+
self._settings = _coerce_workspace_settings(settings)
|
|
317
|
+
|
|
318
|
+
@property
|
|
319
|
+
def repo_root(self) -> Path:
|
|
320
|
+
return self._root
|
|
321
|
+
|
|
322
|
+
@staticmethod
|
|
323
|
+
def _normalize_rel_segments(rel: str, *, label: str) -> tuple[str, ...]:
|
|
324
|
+
"""Split a repo-relative path, normalize ``.`` / ``..``, forbid absolute paths."""
|
|
325
|
+
stack: list[str] = []
|
|
326
|
+
for part in rel.replace("\\", "/").split("/"):
|
|
327
|
+
if not part or part == ".":
|
|
328
|
+
continue
|
|
329
|
+
if part == "..":
|
|
330
|
+
if not stack:
|
|
331
|
+
raise WorkspaceError(
|
|
332
|
+
f"Invalid {label}: path escapes workspace root",
|
|
333
|
+
code="invalid_path",
|
|
334
|
+
)
|
|
335
|
+
stack.pop()
|
|
336
|
+
else:
|
|
337
|
+
stack.append(part)
|
|
338
|
+
return tuple(stack)
|
|
339
|
+
|
|
340
|
+
def _join_under_root(self, segments: tuple[str, ...]) -> Path:
|
|
341
|
+
return self._root.joinpath(*segments) if segments else self._root
|
|
342
|
+
|
|
343
|
+
def _resolve_under_root(self, rel: str, *, label: str = "path") -> Path:
|
|
344
|
+
if rel is None or not str(rel).strip():
|
|
345
|
+
raise WorkspaceError(f"{label} is required", code="missing_path")
|
|
346
|
+
s = str(rel).strip().replace("\\", "/")
|
|
347
|
+
if s.startswith("~") or s.startswith("/"):
|
|
348
|
+
raise WorkspaceError(f"Invalid {label}: absolute or home not allowed", code="invalid_path")
|
|
349
|
+
segs = self._normalize_rel_segments(s.lstrip("/"), label=label)
|
|
350
|
+
# Lexical join (no final .resolve()) so symlinks under the workspace may point outside
|
|
351
|
+
# the physical root — e.g. demo agent ``workspace/data`` when linked to sibling dirs.
|
|
352
|
+
return self._join_under_root(segs)
|
|
353
|
+
|
|
354
|
+
def resolve_workspace_path(self, rel: str, *, label: str = "path") -> Path:
|
|
355
|
+
"""Public path preflight: repo-relative → absolute path under the workspace root."""
|
|
356
|
+
return self._resolve_under_root(rel, label=label)
|
|
357
|
+
|
|
358
|
+
def _resolve_root_dir(self, rel: str | None) -> Path:
|
|
359
|
+
if rel is None or not str(rel).strip() or str(rel).strip() in (".", "./"):
|
|
360
|
+
return self._root
|
|
361
|
+
s = str(rel).strip().replace("\\", "/")
|
|
362
|
+
if s.startswith("~") or s.startswith("/"):
|
|
363
|
+
raise WorkspaceError("Invalid root: absolute or home not allowed", code="invalid_path")
|
|
364
|
+
segs = self._normalize_rel_segments(s.lstrip("/"), label="root")
|
|
365
|
+
return self._join_under_root(segs)
|
|
366
|
+
|
|
367
|
+
def _write_scope_root(self) -> Path | None:
|
|
368
|
+
rel = self._settings.WORKSPACE_WRITE_SCOPE_REL
|
|
369
|
+
if rel is None:
|
|
370
|
+
return None
|
|
371
|
+
s = str(rel).strip().replace("\\", "/").lstrip("/")
|
|
372
|
+
if not s or ".." in s:
|
|
373
|
+
return None
|
|
374
|
+
return (self._root / s).resolve()
|
|
375
|
+
|
|
376
|
+
def _require_under_write_scope(self, fp: Path) -> None:
|
|
377
|
+
scope_root = self._write_scope_root()
|
|
378
|
+
if scope_root is None:
|
|
379
|
+
return
|
|
380
|
+
monkeybot_root = (self._root / ".monkeybot").resolve()
|
|
381
|
+
try:
|
|
382
|
+
fp.resolve().relative_to(monkeybot_root)
|
|
383
|
+
return
|
|
384
|
+
except ValueError:
|
|
385
|
+
pass
|
|
386
|
+
try:
|
|
387
|
+
fp.resolve().relative_to(scope_root)
|
|
388
|
+
except ValueError:
|
|
389
|
+
raise WorkspaceError(
|
|
390
|
+
f"Writes are limited to {self._settings.WORKSPACE_WRITE_SCOPE_REL!r} (repo-relative)",
|
|
391
|
+
code="write_outside_scope",
|
|
392
|
+
) from None
|
|
393
|
+
|
|
394
|
+
def list_directory(self, path: str | None) -> list[dict[str, str]]:
|
|
395
|
+
"""List immediate children of a repo-relative directory (dirs first, then files).
|
|
396
|
+
|
|
397
|
+
``path`` may be ``None``, empty, or ``"."`` for the workspace root.
|
|
398
|
+
"""
|
|
399
|
+
base = self._resolve_root_dir(path)
|
|
400
|
+
if not base.is_dir():
|
|
401
|
+
raise WorkspaceError(f"Not a directory: {path or '.'}", code="not_found")
|
|
402
|
+
try:
|
|
403
|
+
children = list(base.iterdir())
|
|
404
|
+
except OSError as e:
|
|
405
|
+
raise WorkspaceError(f"List failed: {e}", code="list_failed") from e
|
|
406
|
+
|
|
407
|
+
def sort_key(p: Path) -> tuple[int, str]:
|
|
408
|
+
return (0 if p.is_dir() else 1, p.name.lower())
|
|
409
|
+
|
|
410
|
+
out: list[dict[str, str]] = []
|
|
411
|
+
for ch in sorted(children, key=sort_key):
|
|
412
|
+
if ch.name in (".", ".."):
|
|
413
|
+
continue
|
|
414
|
+
full = base / ch.name
|
|
415
|
+
try:
|
|
416
|
+
rel_p = full.relative_to(self._root).as_posix()
|
|
417
|
+
except ValueError:
|
|
418
|
+
continue
|
|
419
|
+
try:
|
|
420
|
+
kind = "dir" if full.is_dir() else "file"
|
|
421
|
+
except OSError:
|
|
422
|
+
continue
|
|
423
|
+
out.append({"name": ch.name, "path": rel_p, "kind": kind})
|
|
424
|
+
return out
|
|
425
|
+
|
|
426
|
+
def read_file(
|
|
427
|
+
self,
|
|
428
|
+
path: str,
|
|
429
|
+
*,
|
|
430
|
+
offset: int = 1,
|
|
431
|
+
limit: int | None = None,
|
|
432
|
+
max_lines_cap: int | None = None,
|
|
433
|
+
) -> ReadFileResult:
|
|
434
|
+
if offset < 1:
|
|
435
|
+
raise WorkspaceError("offset must be >= 1", code="invalid_offset")
|
|
436
|
+
max_lines = (
|
|
437
|
+
max_lines_cap
|
|
438
|
+
if max_lines_cap is not None
|
|
439
|
+
else self._settings.WORKSPACE_READ_MAX_LINES
|
|
440
|
+
)
|
|
441
|
+
if limit is None:
|
|
442
|
+
limit = self._settings.WORKSPACE_READ_DEFAULT_LINES
|
|
443
|
+
if limit < 1 or limit > max_lines:
|
|
444
|
+
raise WorkspaceError(
|
|
445
|
+
f"limit must be between 1 and {max_lines}",
|
|
446
|
+
code="invalid_limit",
|
|
447
|
+
)
|
|
448
|
+
fp = self._resolve_under_root(path)
|
|
449
|
+
if not fp.is_file():
|
|
450
|
+
raise WorkspaceError(f"Not a file: {path}", code="not_found")
|
|
451
|
+
text = fp.read_text(encoding="utf-8", errors="replace")
|
|
452
|
+
lines = text.splitlines()
|
|
453
|
+
total = len(lines)
|
|
454
|
+
start_idx = offset - 1
|
|
455
|
+
if start_idx > total:
|
|
456
|
+
return {
|
|
457
|
+
"ok": True,
|
|
458
|
+
"path": self._as_repo_rel(fp),
|
|
459
|
+
"content": "",
|
|
460
|
+
"start_line": offset,
|
|
461
|
+
"end_line": offset - 1,
|
|
462
|
+
"total_lines": total,
|
|
463
|
+
"truncated": False,
|
|
464
|
+
}
|
|
465
|
+
end_idx = min(start_idx + limit, total)
|
|
466
|
+
chunk = lines[start_idx:end_idx]
|
|
467
|
+
width = max(6, len(str(end_idx)))
|
|
468
|
+
numbered = "\n".join(f"{start_idx + 1 + i:{width}d}|{chunk[i]}" for i in range(len(chunk)))
|
|
469
|
+
return {
|
|
470
|
+
"ok": True,
|
|
471
|
+
"path": self._as_repo_rel(fp),
|
|
472
|
+
"content": numbered,
|
|
473
|
+
"start_line": start_idx + 1,
|
|
474
|
+
"end_line": end_idx,
|
|
475
|
+
"total_lines": total,
|
|
476
|
+
"truncated": end_idx < total,
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
def write_file(self, path: str, content: str) -> WriteFileResult:
|
|
480
|
+
if content is None:
|
|
481
|
+
content = ""
|
|
482
|
+
raw = content.encode("utf-8")
|
|
483
|
+
if len(raw) > self._settings.WORKSPACE_WRITE_MAX_BYTES:
|
|
484
|
+
raise WorkspaceError(
|
|
485
|
+
f"Content exceeds WORKSPACE_WRITE_MAX_BYTES ({self._settings.WORKSPACE_WRITE_MAX_BYTES})",
|
|
486
|
+
code="payload_too_large",
|
|
487
|
+
)
|
|
488
|
+
fp = self._resolve_under_root(path)
|
|
489
|
+
self._require_under_write_scope(fp)
|
|
490
|
+
fp.parent.mkdir(parents=True, exist_ok=True)
|
|
491
|
+
tmp = fp.with_suffix(fp.suffix + ".workspace_tmp")
|
|
492
|
+
try:
|
|
493
|
+
tmp.write_bytes(raw)
|
|
494
|
+
tmp.replace(fp)
|
|
495
|
+
except OSError as e:
|
|
496
|
+
if tmp.exists():
|
|
497
|
+
try:
|
|
498
|
+
tmp.unlink()
|
|
499
|
+
except OSError:
|
|
500
|
+
pass
|
|
501
|
+
raise WorkspaceError(f"Write failed: {e}", code="write_failed") from e
|
|
502
|
+
return {
|
|
503
|
+
"ok": True,
|
|
504
|
+
"path": self._as_repo_rel(fp),
|
|
505
|
+
"bytes": len(raw),
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
def delete_file(self, path: str) -> DeleteResult:
|
|
509
|
+
fp = self._resolve_under_root(path)
|
|
510
|
+
self._require_under_write_scope(fp)
|
|
511
|
+
if not fp.exists():
|
|
512
|
+
raise WorkspaceError(f"Not a file: {path}", code="not_found")
|
|
513
|
+
if fp.is_dir():
|
|
514
|
+
raise WorkspaceError(f"Path is a directory, not a file: {path}", code="is_directory")
|
|
515
|
+
if not fp.is_file():
|
|
516
|
+
raise WorkspaceError(f"Not a file: {path}", code="not_found")
|
|
517
|
+
try:
|
|
518
|
+
fp.unlink()
|
|
519
|
+
except OSError as e:
|
|
520
|
+
raise WorkspaceError(f"Delete failed: {e}", code="delete_failed") from e
|
|
521
|
+
return {"ok": True, "path": self._as_repo_rel(fp)}
|
|
522
|
+
|
|
523
|
+
def replace_in_file(
|
|
524
|
+
self,
|
|
525
|
+
path: str,
|
|
526
|
+
old_string: str,
|
|
527
|
+
new_string: str,
|
|
528
|
+
*,
|
|
529
|
+
replace_all: bool = False,
|
|
530
|
+
) -> ReplaceResult:
|
|
531
|
+
if old_string is None:
|
|
532
|
+
old_string = ""
|
|
533
|
+
if new_string is None:
|
|
534
|
+
new_string = ""
|
|
535
|
+
if old_string == new_string:
|
|
536
|
+
raise WorkspaceError(
|
|
537
|
+
"No changes to apply: old_string and new_string are identical.",
|
|
538
|
+
code="no_change",
|
|
539
|
+
)
|
|
540
|
+
if old_string == "":
|
|
541
|
+
raise WorkspaceError(
|
|
542
|
+
"old_string cannot be empty when editing an existing file. "
|
|
543
|
+
"Provide the exact text to replace, or use write_file for a full rewrite.",
|
|
544
|
+
code="empty_old_string",
|
|
545
|
+
)
|
|
546
|
+
fp = self._resolve_under_root(path)
|
|
547
|
+
self._require_under_write_scope(fp)
|
|
548
|
+
if not fp.is_file():
|
|
549
|
+
raise WorkspaceError(f"Not a file: {path}", code="not_found")
|
|
550
|
+
text = fp.read_text(encoding="utf-8", errors="replace")
|
|
551
|
+
spans, match_mode = _find_replace_span(text, old_string, replace_all=replace_all)
|
|
552
|
+
new_text = _apply_replace_spans(text, spans, new_string)
|
|
553
|
+
raw = new_text.encode("utf-8")
|
|
554
|
+
if len(raw) > self._settings.WORKSPACE_WRITE_MAX_BYTES:
|
|
555
|
+
raise WorkspaceError(
|
|
556
|
+
f"Result exceeds WORKSPACE_WRITE_MAX_BYTES ({self._settings.WORKSPACE_WRITE_MAX_BYTES})",
|
|
557
|
+
code="payload_too_large",
|
|
558
|
+
)
|
|
559
|
+
tmp = fp.with_suffix(fp.suffix + ".workspace_tmp")
|
|
560
|
+
try:
|
|
561
|
+
tmp.write_bytes(raw)
|
|
562
|
+
tmp.replace(fp)
|
|
563
|
+
except OSError as e:
|
|
564
|
+
if tmp.exists():
|
|
565
|
+
try:
|
|
566
|
+
tmp.unlink()
|
|
567
|
+
except OSError:
|
|
568
|
+
pass
|
|
569
|
+
raise WorkspaceError(f"Write failed: {e}", code="write_failed") from e
|
|
570
|
+
return {
|
|
571
|
+
"ok": True,
|
|
572
|
+
"path": self._as_repo_rel(fp),
|
|
573
|
+
"replacements": len(spans),
|
|
574
|
+
"bytes": len(raw),
|
|
575
|
+
"match_mode": match_mode,
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
def glob_paths(self, pattern: str, root: str | None = None) -> GlobResult:
|
|
579
|
+
if not pattern or not pattern.strip():
|
|
580
|
+
raise WorkspaceError("pattern is required", code="missing_pattern")
|
|
581
|
+
pattern = pattern.strip()
|
|
582
|
+
if pattern.startswith("/") or ".." in pattern:
|
|
583
|
+
raise WorkspaceError("Invalid glob pattern", code="invalid_pattern")
|
|
584
|
+
base = self._resolve_root_dir(root)
|
|
585
|
+
deadline = time.monotonic() + self._settings.WORKSPACE_GLOB_TIMEOUT_SEC
|
|
586
|
+
max_paths = self._settings.WORKSPACE_GLOB_MAX_PATHS
|
|
587
|
+
paths: list[str] = []
|
|
588
|
+
truncated = False
|
|
589
|
+
t0 = time.monotonic()
|
|
590
|
+
try:
|
|
591
|
+
for p in base.glob(pattern):
|
|
592
|
+
if time.monotonic() > deadline:
|
|
593
|
+
truncated = True
|
|
594
|
+
break
|
|
595
|
+
if not p.is_file():
|
|
596
|
+
continue
|
|
597
|
+
try:
|
|
598
|
+
p.resolve().relative_to(self._root)
|
|
599
|
+
except ValueError:
|
|
600
|
+
continue
|
|
601
|
+
paths.append(self._as_repo_rel(p))
|
|
602
|
+
if len(paths) >= max_paths:
|
|
603
|
+
truncated = True
|
|
604
|
+
break
|
|
605
|
+
except OSError as e:
|
|
606
|
+
raise WorkspaceError(f"Glob failed: {e}", code="glob_failed") from e
|
|
607
|
+
paths.sort()
|
|
608
|
+
duration_ms = int((time.monotonic() - t0) * 1000)
|
|
609
|
+
return {
|
|
610
|
+
"ok": True,
|
|
611
|
+
"root": self._as_repo_rel(base) if base != self._root else ".",
|
|
612
|
+
"pattern": pattern,
|
|
613
|
+
"paths": paths,
|
|
614
|
+
"count": len(paths),
|
|
615
|
+
"truncated": truncated,
|
|
616
|
+
"duration_ms": duration_ms,
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
def grep(
|
|
620
|
+
self,
|
|
621
|
+
pattern: str,
|
|
622
|
+
*,
|
|
623
|
+
root: str | None = None,
|
|
624
|
+
ignore_case: bool = False,
|
|
625
|
+
file_glob: str | None = None,
|
|
626
|
+
max_matches: int | None = None,
|
|
627
|
+
) -> GrepResult:
|
|
628
|
+
if not pattern or not str(pattern).strip():
|
|
629
|
+
raise WorkspaceError("pattern is required", code="missing_pattern")
|
|
630
|
+
flags = re.IGNORECASE if ignore_case else 0
|
|
631
|
+
try:
|
|
632
|
+
regex = re.compile(pattern.strip(), flags)
|
|
633
|
+
except re.error as e:
|
|
634
|
+
raise WorkspaceError(f"Invalid regex: {e}", code="invalid_regex") from e
|
|
635
|
+
base = self._resolve_root_dir(root)
|
|
636
|
+
max_m = max_matches if max_matches is not None else self._settings.WORKSPACE_GREP_MAX_MATCHES
|
|
637
|
+
max_files = self._settings.WORKSPACE_GREP_MAX_FILES
|
|
638
|
+
max_file_bytes = self._settings.WORKSPACE_GREP_MAX_FILE_BYTES
|
|
639
|
+
matches: list[GrepMatch] = []
|
|
640
|
+
files_scanned = 0
|
|
641
|
+
truncated = False
|
|
642
|
+
t0 = time.monotonic()
|
|
643
|
+
for dirpath, dirnames, filenames in os.walk(base):
|
|
644
|
+
dirnames[:] = sorted(d for d in dirnames if d not in _GREP_IGNORE_DIRS)
|
|
645
|
+
for name in sorted(filenames):
|
|
646
|
+
if len(matches) >= max_m:
|
|
647
|
+
truncated = True
|
|
648
|
+
break
|
|
649
|
+
if files_scanned >= max_files:
|
|
650
|
+
truncated = True
|
|
651
|
+
break
|
|
652
|
+
fp = Path(dirpath) / name
|
|
653
|
+
try:
|
|
654
|
+
fp.resolve().relative_to(self._root)
|
|
655
|
+
except ValueError:
|
|
656
|
+
continue
|
|
657
|
+
rel = self._as_repo_rel(fp)
|
|
658
|
+
if file_glob and not fnmatch.fnmatch(fp.name, file_glob):
|
|
659
|
+
continue
|
|
660
|
+
try:
|
|
661
|
+
st = fp.stat()
|
|
662
|
+
except OSError:
|
|
663
|
+
continue
|
|
664
|
+
if st.st_size > max_file_bytes:
|
|
665
|
+
continue
|
|
666
|
+
files_scanned += 1
|
|
667
|
+
try:
|
|
668
|
+
data = fp.read_bytes()
|
|
669
|
+
except OSError:
|
|
670
|
+
continue
|
|
671
|
+
if b"\x00" in data[:8192]:
|
|
672
|
+
continue
|
|
673
|
+
text = data.decode("utf-8", errors="replace")
|
|
674
|
+
for line_no, line in enumerate(text.splitlines(), start=1):
|
|
675
|
+
if len(matches) >= max_m:
|
|
676
|
+
truncated = True
|
|
677
|
+
break
|
|
678
|
+
if regex.search(line):
|
|
679
|
+
matches.append(
|
|
680
|
+
{
|
|
681
|
+
"path": rel,
|
|
682
|
+
"line": line_no,
|
|
683
|
+
"text": line[:2000],
|
|
684
|
+
}
|
|
685
|
+
)
|
|
686
|
+
if truncated:
|
|
687
|
+
break
|
|
688
|
+
if truncated:
|
|
689
|
+
break
|
|
690
|
+
duration_ms = int((time.monotonic() - t0) * 1000)
|
|
691
|
+
return {
|
|
692
|
+
"ok": True,
|
|
693
|
+
"root": self._as_repo_rel(base) if base != self._root else ".",
|
|
694
|
+
"pattern": pattern.strip(),
|
|
695
|
+
"matches": matches,
|
|
696
|
+
"match_count": len(matches),
|
|
697
|
+
"files_scanned": files_scanned,
|
|
698
|
+
"truncated": truncated,
|
|
699
|
+
"duration_ms": duration_ms,
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
def _as_repo_rel(self, p: Path) -> str:
|
|
703
|
+
try:
|
|
704
|
+
return p.relative_to(self._root).as_posix()
|
|
705
|
+
except ValueError:
|
|
706
|
+
pass
|
|
707
|
+
try:
|
|
708
|
+
return p.resolve().relative_to(self._root).as_posix()
|
|
709
|
+
except ValueError as e:
|
|
710
|
+
raise WorkspaceError("path escapes workspace root", code="path_escape") from e
|