millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,738 @@
|
|
|
1
|
+
"""Deterministic Python implementations of Pi's ``grep`` and ``find`` tools.
|
|
2
|
+
|
|
3
|
+
The public operations intentionally stay independent from Millforge's tool
|
|
4
|
+
descriptors. They are a source-attributed behavioral port of Pi 0.79.6 with
|
|
5
|
+
the filesystem-search adaptations specified by Millforge Spec 11 section 7.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
import errno
|
|
12
|
+
import math
|
|
13
|
+
import os
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
import re
|
|
16
|
+
import stat
|
|
17
|
+
import sys
|
|
18
|
+
from typing import Iterator, cast
|
|
19
|
+
|
|
20
|
+
from pathspec import PathSpec
|
|
21
|
+
|
|
22
|
+
from .contracts import (
|
|
23
|
+
PiCompatErrorKind,
|
|
24
|
+
PiCompatOperationResult,
|
|
25
|
+
PiCompatSideEffectState,
|
|
26
|
+
)
|
|
27
|
+
from .paths import _PathValidationError, _is_absolute_cwd, resolve_to_cwd
|
|
28
|
+
from .truncation import (
|
|
29
|
+
DEFAULT_MAX_BYTES,
|
|
30
|
+
GREP_MAX_LINE_LENGTH,
|
|
31
|
+
sanitize_text,
|
|
32
|
+
truncate_head,
|
|
33
|
+
truncate_line,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_DEFAULT_GREP_LIMIT = 100
|
|
38
|
+
_DEFAULT_FIND_LIMIT = 1_000
|
|
39
|
+
_MAX_NOTICE_BYTES = 2_048
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class _TraversalEntry:
|
|
44
|
+
path: Path
|
|
45
|
+
relative_path: str
|
|
46
|
+
kind: str
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class _IgnoreSpec:
|
|
51
|
+
base_relative_path: str
|
|
52
|
+
spec: PathSpec
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class _GlobMatcher:
|
|
57
|
+
spec: PathSpec
|
|
58
|
+
matches_path: bool
|
|
59
|
+
matches_nothing: bool = False
|
|
60
|
+
|
|
61
|
+
def matches(self, relative_path: str) -> bool:
|
|
62
|
+
if self.matches_nothing:
|
|
63
|
+
return False
|
|
64
|
+
candidate = (
|
|
65
|
+
relative_path if self.matches_path else relative_path.rsplit("/", 1)[-1]
|
|
66
|
+
)
|
|
67
|
+
return self.spec.match_file(candidate)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class _InvalidSearchArgument(ValueError):
|
|
71
|
+
"""Raised for a regex or glob that cannot be evaluated by this port."""
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class _RootSearchError(OSError):
|
|
75
|
+
"""Raised when an explicit search root cannot be accessed."""
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class _SearchTraversal:
|
|
79
|
+
"""Walk one search root while collecting descendant read failures."""
|
|
80
|
+
|
|
81
|
+
def __init__(self) -> None:
|
|
82
|
+
self.unreadable_count = 0
|
|
83
|
+
self._unreadable_paths: set[str] = set()
|
|
84
|
+
|
|
85
|
+
def walk(self, root: Path) -> Iterator[_TraversalEntry]:
|
|
86
|
+
yield from self._walk_directory(
|
|
87
|
+
root,
|
|
88
|
+
relative_directory="",
|
|
89
|
+
inherited_ignore_specs=(),
|
|
90
|
+
include_directory=False,
|
|
91
|
+
is_root=True,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
def mark_unreadable(self, path: Path) -> None:
|
|
95
|
+
key = os.fspath(path)
|
|
96
|
+
if key not in self._unreadable_paths:
|
|
97
|
+
self._unreadable_paths.add(key)
|
|
98
|
+
self.unreadable_count += 1
|
|
99
|
+
|
|
100
|
+
def was_unreadable(self, path: Path) -> bool:
|
|
101
|
+
return os.fspath(path) in self._unreadable_paths
|
|
102
|
+
|
|
103
|
+
def _walk_directory(
|
|
104
|
+
self,
|
|
105
|
+
directory: Path,
|
|
106
|
+
*,
|
|
107
|
+
relative_directory: str,
|
|
108
|
+
inherited_ignore_specs: tuple[_IgnoreSpec, ...],
|
|
109
|
+
include_directory: bool,
|
|
110
|
+
is_root: bool,
|
|
111
|
+
) -> Iterator[_TraversalEntry]:
|
|
112
|
+
try:
|
|
113
|
+
entries = sorted(
|
|
114
|
+
os.scandir(directory),
|
|
115
|
+
key=lambda entry: (entry.name.casefold(), entry.name),
|
|
116
|
+
)
|
|
117
|
+
except OSError as exc:
|
|
118
|
+
if is_root:
|
|
119
|
+
raise _RootSearchError(*exc.args) from exc
|
|
120
|
+
self.mark_unreadable(directory)
|
|
121
|
+
return
|
|
122
|
+
|
|
123
|
+
ignore_specs = inherited_ignore_specs
|
|
124
|
+
ignore_file = directory / ".gitignore"
|
|
125
|
+
try:
|
|
126
|
+
ignore_stat = ignore_file.stat()
|
|
127
|
+
except FileNotFoundError:
|
|
128
|
+
ignore_stat = None
|
|
129
|
+
except OSError:
|
|
130
|
+
self.mark_unreadable(ignore_file)
|
|
131
|
+
ignore_stat = None
|
|
132
|
+
|
|
133
|
+
if ignore_stat is not None and stat.S_ISREG(ignore_stat.st_mode):
|
|
134
|
+
try:
|
|
135
|
+
ignore_text = ignore_file.read_text(encoding="utf-8", errors="replace")
|
|
136
|
+
except OSError:
|
|
137
|
+
self.mark_unreadable(ignore_file)
|
|
138
|
+
else:
|
|
139
|
+
try:
|
|
140
|
+
ignore_spec = PathSpec.from_lines(
|
|
141
|
+
"gitwildmatch", ignore_text.splitlines()
|
|
142
|
+
)
|
|
143
|
+
except (TypeError, ValueError, re.error):
|
|
144
|
+
# Git ignore files are intentionally forgiving. An invalid
|
|
145
|
+
# local rule cannot make the search operation itself fail.
|
|
146
|
+
ignore_spec = None
|
|
147
|
+
if ignore_spec is not None:
|
|
148
|
+
ignore_specs = (
|
|
149
|
+
*ignore_specs,
|
|
150
|
+
_IgnoreSpec(relative_directory, ignore_spec),
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
if include_directory:
|
|
154
|
+
yield _TraversalEntry(directory, relative_directory, "directory")
|
|
155
|
+
|
|
156
|
+
for entry in entries:
|
|
157
|
+
child_path = Path(entry.path)
|
|
158
|
+
relative_path = _join_relative_path(relative_directory, entry.name)
|
|
159
|
+
if self.was_unreadable(child_path):
|
|
160
|
+
continue
|
|
161
|
+
|
|
162
|
+
try:
|
|
163
|
+
is_symlink = entry.is_symlink()
|
|
164
|
+
is_directory = not is_symlink and entry.is_dir(follow_symlinks=False)
|
|
165
|
+
is_file = not is_symlink and entry.is_file(follow_symlinks=False)
|
|
166
|
+
except OSError:
|
|
167
|
+
self.mark_unreadable(child_path)
|
|
168
|
+
continue
|
|
169
|
+
|
|
170
|
+
if _is_ignored(relative_path, is_directory, ignore_specs):
|
|
171
|
+
continue
|
|
172
|
+
|
|
173
|
+
if is_symlink:
|
|
174
|
+
yield _TraversalEntry(child_path, relative_path, "symlink")
|
|
175
|
+
elif is_directory:
|
|
176
|
+
yield from self._walk_directory(
|
|
177
|
+
child_path,
|
|
178
|
+
relative_directory=relative_path,
|
|
179
|
+
inherited_ignore_specs=ignore_specs,
|
|
180
|
+
include_directory=True,
|
|
181
|
+
is_root=False,
|
|
182
|
+
)
|
|
183
|
+
elif is_file:
|
|
184
|
+
yield _TraversalEntry(child_path, relative_path, "file")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def execute_grep(
|
|
188
|
+
*,
|
|
189
|
+
cwd: Path,
|
|
190
|
+
pattern: str,
|
|
191
|
+
path: str | None = None,
|
|
192
|
+
glob: str | None = None,
|
|
193
|
+
ignoreCase: bool | None = None,
|
|
194
|
+
literal: bool | None = None,
|
|
195
|
+
context: int | float | None = None,
|
|
196
|
+
limit: int | float | None = None,
|
|
197
|
+
) -> PiCompatOperationResult:
|
|
198
|
+
"""Search non-ignored regular files using Pi's model-visible grep format."""
|
|
199
|
+
|
|
200
|
+
if not _is_absolute_cwd(cwd):
|
|
201
|
+
return _error_result("cwd must be an absolute path", "invalid_arguments")
|
|
202
|
+
if "\x00" in str(cwd):
|
|
203
|
+
return _error_result("cwd must not contain NUL bytes", "invalid_arguments")
|
|
204
|
+
invalid = _validate_grep_arguments(
|
|
205
|
+
pattern=pattern,
|
|
206
|
+
path=path,
|
|
207
|
+
glob=glob,
|
|
208
|
+
ignore_case=ignoreCase,
|
|
209
|
+
literal=literal,
|
|
210
|
+
context=context,
|
|
211
|
+
limit=limit,
|
|
212
|
+
)
|
|
213
|
+
if invalid is not None:
|
|
214
|
+
return _error_result(invalid, "invalid_arguments")
|
|
215
|
+
|
|
216
|
+
try:
|
|
217
|
+
search_root = _resolve_search_path(cwd, path)
|
|
218
|
+
root_kind = _search_root_kind(search_root, find_requires_directory=False)
|
|
219
|
+
matcher = _compile_grep_matcher(pattern, bool(ignoreCase), bool(literal))
|
|
220
|
+
glob_matcher = _compile_glob(glob) if glob else None
|
|
221
|
+
except _PathValidationError as exc:
|
|
222
|
+
return _error_result(str(exc), "invalid_arguments")
|
|
223
|
+
except _InvalidSearchArgument as exc:
|
|
224
|
+
return _error_result(str(exc), "invalid_arguments")
|
|
225
|
+
except OSError as exc:
|
|
226
|
+
return _root_error_result(
|
|
227
|
+
search_root if "search_root" in locals() else cwd, exc
|
|
228
|
+
)
|
|
229
|
+
except UnicodeError:
|
|
230
|
+
return _error_result(
|
|
231
|
+
"path cannot be encoded by this filesystem", "invalid_arguments"
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
effective_limit = max(1, limit if limit is not None else _DEFAULT_GREP_LIMIT)
|
|
235
|
+
context_value = context if context is not None and context > 0 else 0
|
|
236
|
+
raw_lines: list[str] = []
|
|
237
|
+
match_count = 0
|
|
238
|
+
match_limit_reached = False
|
|
239
|
+
lines_truncated = False
|
|
240
|
+
traversal = _SearchTraversal()
|
|
241
|
+
|
|
242
|
+
def process_file(
|
|
243
|
+
file_path: Path, display_path: str, *, explicit_root: bool = False
|
|
244
|
+
) -> None:
|
|
245
|
+
nonlocal lines_truncated, match_count, match_limit_reached
|
|
246
|
+
if glob_matcher is not None and not glob_matcher.matches(display_path):
|
|
247
|
+
return
|
|
248
|
+
try:
|
|
249
|
+
data = _read_search_file(file_path)
|
|
250
|
+
except OSError as exc:
|
|
251
|
+
if explicit_root:
|
|
252
|
+
raise _RootSearchError(*exc.args) from exc
|
|
253
|
+
traversal.mark_unreadable(file_path)
|
|
254
|
+
return
|
|
255
|
+
if data is None:
|
|
256
|
+
return
|
|
257
|
+
|
|
258
|
+
matching_lines, context_source_lines = _split_grep_lines(
|
|
259
|
+
data, context_value > 0
|
|
260
|
+
)
|
|
261
|
+
for line_number, line in enumerate(matching_lines, start=1):
|
|
262
|
+
if not matcher.search(line):
|
|
263
|
+
continue
|
|
264
|
+
if match_count >= effective_limit:
|
|
265
|
+
match_limit_reached = True
|
|
266
|
+
return
|
|
267
|
+
|
|
268
|
+
match_count += 1
|
|
269
|
+
if context_value == 0:
|
|
270
|
+
rendered, was_truncated = truncate_line(line)
|
|
271
|
+
lines_truncated = lines_truncated or was_truncated
|
|
272
|
+
raw_lines.append(f"{display_path}:{line_number}: {rendered}")
|
|
273
|
+
else:
|
|
274
|
+
for current_line_number in _context_line_numbers(
|
|
275
|
+
line_number, len(context_source_lines), context_value
|
|
276
|
+
):
|
|
277
|
+
source_line = _context_source_line(
|
|
278
|
+
context_source_lines, current_line_number
|
|
279
|
+
)
|
|
280
|
+
rendered, was_truncated = truncate_line(source_line)
|
|
281
|
+
lines_truncated = lines_truncated or was_truncated
|
|
282
|
+
if current_line_number == line_number:
|
|
283
|
+
raw_lines.append(
|
|
284
|
+
f"{display_path}:{_format_number(current_line_number)}: "
|
|
285
|
+
f"{rendered}"
|
|
286
|
+
)
|
|
287
|
+
else:
|
|
288
|
+
raw_lines.append(
|
|
289
|
+
f"{display_path}-{_format_number(current_line_number)}- "
|
|
290
|
+
f"{rendered}"
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
if match_count >= effective_limit:
|
|
294
|
+
match_limit_reached = True
|
|
295
|
+
return
|
|
296
|
+
|
|
297
|
+
if root_kind == "file":
|
|
298
|
+
try:
|
|
299
|
+
process_file(search_root, search_root.name, explicit_root=True)
|
|
300
|
+
except _RootSearchError as exc:
|
|
301
|
+
return _root_error_result(search_root, exc)
|
|
302
|
+
else:
|
|
303
|
+
try:
|
|
304
|
+
for entry in traversal.walk(search_root):
|
|
305
|
+
if match_limit_reached:
|
|
306
|
+
break
|
|
307
|
+
if entry.kind != "file":
|
|
308
|
+
continue
|
|
309
|
+
process_file(entry.path, entry.relative_path)
|
|
310
|
+
except _RootSearchError as exc:
|
|
311
|
+
return _root_error_result(search_root, exc)
|
|
312
|
+
|
|
313
|
+
rendered_output, byte_truncated = _render_search_output(
|
|
314
|
+
raw_lines,
|
|
315
|
+
empty_text="No matches found",
|
|
316
|
+
match_or_result_limit_reached=match_limit_reached,
|
|
317
|
+
match_or_result_notice=(
|
|
318
|
+
f"{_format_number(effective_limit)} matches limit reached. "
|
|
319
|
+
f"Use limit={_format_number(effective_limit * 2)} for more, or refine pattern"
|
|
320
|
+
),
|
|
321
|
+
long_line_notice=(
|
|
322
|
+
f"Some lines truncated to {GREP_MAX_LINE_LENGTH} chars. Use read tool to see full lines"
|
|
323
|
+
if lines_truncated
|
|
324
|
+
else None
|
|
325
|
+
),
|
|
326
|
+
unreadable_count=traversal.unreadable_count,
|
|
327
|
+
)
|
|
328
|
+
return _success_result(
|
|
329
|
+
rendered_output,
|
|
330
|
+
truncated=(
|
|
331
|
+
match_limit_reached
|
|
332
|
+
or byte_truncated
|
|
333
|
+
or lines_truncated
|
|
334
|
+
or traversal.unreadable_count > 0
|
|
335
|
+
),
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def execute_find(
|
|
340
|
+
*,
|
|
341
|
+
cwd: Path,
|
|
342
|
+
pattern: str,
|
|
343
|
+
path: str | None = None,
|
|
344
|
+
limit: int | float | None = None,
|
|
345
|
+
) -> PiCompatOperationResult:
|
|
346
|
+
"""Find non-ignored paths using Pi's relative-path result format."""
|
|
347
|
+
|
|
348
|
+
if not _is_absolute_cwd(cwd):
|
|
349
|
+
return _error_result("cwd must be an absolute path", "invalid_arguments")
|
|
350
|
+
if "\x00" in str(cwd):
|
|
351
|
+
return _error_result("cwd must not contain NUL bytes", "invalid_arguments")
|
|
352
|
+
invalid = _validate_find_arguments(pattern=pattern, path=path, limit=limit)
|
|
353
|
+
if invalid is not None:
|
|
354
|
+
return _error_result(invalid, "invalid_arguments")
|
|
355
|
+
|
|
356
|
+
try:
|
|
357
|
+
search_root = _resolve_search_path(cwd, path)
|
|
358
|
+
_search_root_kind(search_root, find_requires_directory=True)
|
|
359
|
+
matcher = _compile_glob(pattern)
|
|
360
|
+
except _PathValidationError as exc:
|
|
361
|
+
return _error_result(str(exc), "invalid_arguments")
|
|
362
|
+
except _InvalidSearchArgument as exc:
|
|
363
|
+
return _error_result(str(exc), "invalid_arguments")
|
|
364
|
+
except OSError as exc:
|
|
365
|
+
return _root_error_result(
|
|
366
|
+
search_root if "search_root" in locals() else cwd, exc
|
|
367
|
+
)
|
|
368
|
+
except UnicodeError:
|
|
369
|
+
return _error_result(
|
|
370
|
+
"path cannot be encoded by this filesystem", "invalid_arguments"
|
|
371
|
+
)
|
|
372
|
+
|
|
373
|
+
effective_limit = limit if limit is not None else _DEFAULT_FIND_LIMIT
|
|
374
|
+
raw_lines: list[str] = []
|
|
375
|
+
result_limit_reached = False
|
|
376
|
+
traversal = _SearchTraversal()
|
|
377
|
+
|
|
378
|
+
try:
|
|
379
|
+
for entry in traversal.walk(search_root):
|
|
380
|
+
if result_limit_reached:
|
|
381
|
+
break
|
|
382
|
+
if not matcher.matches(entry.relative_path):
|
|
383
|
+
continue
|
|
384
|
+
|
|
385
|
+
if entry.kind == "file" and not _can_open_for_read(entry.path):
|
|
386
|
+
traversal.mark_unreadable(entry.path)
|
|
387
|
+
continue
|
|
388
|
+
|
|
389
|
+
raw_lines.append(
|
|
390
|
+
f"{entry.relative_path}/"
|
|
391
|
+
if entry.kind == "directory"
|
|
392
|
+
else entry.relative_path
|
|
393
|
+
)
|
|
394
|
+
if effective_limit != 0 and len(raw_lines) >= effective_limit:
|
|
395
|
+
result_limit_reached = True
|
|
396
|
+
except _RootSearchError as exc:
|
|
397
|
+
return _root_error_result(search_root, exc)
|
|
398
|
+
|
|
399
|
+
if effective_limit == 0 and raw_lines:
|
|
400
|
+
result_limit_reached = True
|
|
401
|
+
|
|
402
|
+
rendered_output, byte_truncated = _render_search_output(
|
|
403
|
+
raw_lines,
|
|
404
|
+
empty_text="No files found matching pattern",
|
|
405
|
+
match_or_result_limit_reached=result_limit_reached,
|
|
406
|
+
match_or_result_notice=(
|
|
407
|
+
f"{_format_number(effective_limit)} results limit reached. "
|
|
408
|
+
f"Use limit={_format_number(effective_limit * 2)} for more, or refine pattern"
|
|
409
|
+
),
|
|
410
|
+
long_line_notice=None,
|
|
411
|
+
unreadable_count=traversal.unreadable_count,
|
|
412
|
+
)
|
|
413
|
+
return _success_result(
|
|
414
|
+
rendered_output,
|
|
415
|
+
truncated=result_limit_reached
|
|
416
|
+
or byte_truncated
|
|
417
|
+
or traversal.unreadable_count > 0,
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _validate_grep_arguments(
|
|
422
|
+
*,
|
|
423
|
+
pattern: object,
|
|
424
|
+
path: object,
|
|
425
|
+
glob: object,
|
|
426
|
+
ignore_case: object,
|
|
427
|
+
literal: object,
|
|
428
|
+
context: object,
|
|
429
|
+
limit: object,
|
|
430
|
+
) -> str | None:
|
|
431
|
+
if not isinstance(pattern, str):
|
|
432
|
+
return "grep pattern must be a string"
|
|
433
|
+
if path is not None and not isinstance(path, str):
|
|
434
|
+
return "grep path must be a string"
|
|
435
|
+
if isinstance(path, str) and "\x00" in path:
|
|
436
|
+
return "grep path must not contain NUL bytes"
|
|
437
|
+
if glob is not None and not isinstance(glob, str):
|
|
438
|
+
return "grep glob must be a string"
|
|
439
|
+
if ignore_case is not None and not isinstance(ignore_case, bool):
|
|
440
|
+
return "grep ignoreCase must be a boolean"
|
|
441
|
+
if literal is not None and not isinstance(literal, bool):
|
|
442
|
+
return "grep literal must be a boolean"
|
|
443
|
+
if not _is_json_number_or_none(context):
|
|
444
|
+
return "grep context must be a number"
|
|
445
|
+
if not _is_json_number_or_none(limit):
|
|
446
|
+
return "grep limit must be a number"
|
|
447
|
+
return None
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def _validate_find_arguments(
|
|
451
|
+
*, pattern: object, path: object, limit: object
|
|
452
|
+
) -> str | None:
|
|
453
|
+
if not isinstance(pattern, str):
|
|
454
|
+
return "find pattern must be a string"
|
|
455
|
+
if path is not None and not isinstance(path, str):
|
|
456
|
+
return "find path must be a string"
|
|
457
|
+
if isinstance(path, str) and "\x00" in path:
|
|
458
|
+
return "find path must not contain NUL bytes"
|
|
459
|
+
if not _is_json_number_or_none(limit):
|
|
460
|
+
return "find limit must be a number"
|
|
461
|
+
numeric_limit = cast(int | float | None, limit)
|
|
462
|
+
if numeric_limit is not None and (
|
|
463
|
+
numeric_limit < 0
|
|
464
|
+
or (isinstance(numeric_limit, float) and not numeric_limit.is_integer())
|
|
465
|
+
or numeric_limit > sys.maxsize
|
|
466
|
+
):
|
|
467
|
+
return "find limit must be a non-negative integer"
|
|
468
|
+
return None
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def _is_json_number_or_none(value: object) -> bool:
|
|
472
|
+
if value is None:
|
|
473
|
+
return True
|
|
474
|
+
if isinstance(value, bool):
|
|
475
|
+
return False
|
|
476
|
+
if isinstance(value, int):
|
|
477
|
+
return True
|
|
478
|
+
return isinstance(value, float) and math.isfinite(value)
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def _resolve_search_path(cwd: Path, supplied_path: str | None) -> Path:
|
|
482
|
+
if not cwd.is_absolute():
|
|
483
|
+
raise _InvalidSearchArgument("cwd must be an absolute path")
|
|
484
|
+
return resolve_to_cwd(supplied_path or ".", cwd)
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _search_root_kind(path: Path, *, find_requires_directory: bool) -> str:
|
|
488
|
+
try:
|
|
489
|
+
path_stat = path.stat()
|
|
490
|
+
except FileNotFoundError as exc:
|
|
491
|
+
raise FileNotFoundError(*exc.args) from exc
|
|
492
|
+
|
|
493
|
+
if stat.S_ISDIR(path_stat.st_mode):
|
|
494
|
+
return "directory"
|
|
495
|
+
if stat.S_ISREG(path_stat.st_mode) and not find_requires_directory:
|
|
496
|
+
return "file"
|
|
497
|
+
if find_requires_directory:
|
|
498
|
+
raise _InvalidSearchArgument(f"Find path must be a directory: {path}")
|
|
499
|
+
raise _InvalidSearchArgument(f"Grep path must be a file or directory: {path}")
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _compile_grep_matcher(
|
|
503
|
+
pattern: str, ignore_case: bool, literal: bool
|
|
504
|
+
) -> re.Pattern[str]:
|
|
505
|
+
source = re.escape(pattern) if literal else pattern
|
|
506
|
+
try:
|
|
507
|
+
return re.compile(source, re.IGNORECASE if ignore_case else 0)
|
|
508
|
+
except re.error as exc:
|
|
509
|
+
raise _InvalidSearchArgument(f"Invalid regex pattern: {exc}") from exc
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _compile_glob(pattern: str) -> _GlobMatcher:
|
|
513
|
+
_validate_glob_character_classes(pattern)
|
|
514
|
+
matches_path = "/" in pattern
|
|
515
|
+
if pattern.endswith("/"):
|
|
516
|
+
return _GlobMatcher(
|
|
517
|
+
PathSpec.from_lines("gitwildmatch", ()),
|
|
518
|
+
matches_path,
|
|
519
|
+
matches_nothing=True,
|
|
520
|
+
)
|
|
521
|
+
pattern_for_match = pattern or "*"
|
|
522
|
+
if pattern_for_match.startswith("!"):
|
|
523
|
+
pattern_for_match = f"\\{pattern_for_match}"
|
|
524
|
+
effective_pattern = pattern_for_match
|
|
525
|
+
if (
|
|
526
|
+
matches_path
|
|
527
|
+
and not pattern_for_match.startswith(("/", "**/"))
|
|
528
|
+
and pattern_for_match != "**"
|
|
529
|
+
):
|
|
530
|
+
effective_pattern = f"**/{pattern_for_match}"
|
|
531
|
+
try:
|
|
532
|
+
spec = PathSpec.from_lines("gitwildmatch", (effective_pattern,))
|
|
533
|
+
except (TypeError, ValueError, re.error) as exc:
|
|
534
|
+
raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}") from exc
|
|
535
|
+
patterns = tuple(spec.patterns)
|
|
536
|
+
if not patterns:
|
|
537
|
+
raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}")
|
|
538
|
+
return _GlobMatcher(spec, matches_path)
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _validate_glob_character_classes(pattern: str) -> None:
|
|
542
|
+
escaped = False
|
|
543
|
+
in_character_class = False
|
|
544
|
+
for character in pattern:
|
|
545
|
+
if escaped:
|
|
546
|
+
escaped = False
|
|
547
|
+
continue
|
|
548
|
+
if character == "\\":
|
|
549
|
+
escaped = True
|
|
550
|
+
elif character == "[":
|
|
551
|
+
in_character_class = True
|
|
552
|
+
elif character == "]" and in_character_class:
|
|
553
|
+
in_character_class = False
|
|
554
|
+
if in_character_class:
|
|
555
|
+
raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}")
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _is_ignored(
|
|
559
|
+
relative_path: str,
|
|
560
|
+
is_directory: bool,
|
|
561
|
+
ignore_specs: tuple[_IgnoreSpec, ...],
|
|
562
|
+
) -> bool:
|
|
563
|
+
ignored = False
|
|
564
|
+
for ignore_spec in ignore_specs:
|
|
565
|
+
candidate = _relative_to_ignore_base(
|
|
566
|
+
relative_path, ignore_spec.base_relative_path
|
|
567
|
+
)
|
|
568
|
+
if candidate is None:
|
|
569
|
+
continue
|
|
570
|
+
if is_directory:
|
|
571
|
+
candidate = f"{candidate}/"
|
|
572
|
+
|
|
573
|
+
decision: bool | None = None
|
|
574
|
+
for pattern in ignore_spec.spec.patterns:
|
|
575
|
+
if pattern.include is not None and pattern.match_file(candidate):
|
|
576
|
+
decision = bool(pattern.include)
|
|
577
|
+
if decision is not None:
|
|
578
|
+
ignored = decision
|
|
579
|
+
return ignored
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
def _relative_to_ignore_base(relative_path: str, base_relative_path: str) -> str | None:
|
|
583
|
+
if not base_relative_path:
|
|
584
|
+
return relative_path
|
|
585
|
+
prefix = f"{base_relative_path}/"
|
|
586
|
+
if relative_path.startswith(prefix):
|
|
587
|
+
return relative_path[len(prefix) :]
|
|
588
|
+
return None
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _join_relative_path(parent: str, child: str) -> str:
|
|
592
|
+
return f"{parent}/{child}" if parent else child
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
def _read_search_file(path: Path) -> bytes | None:
|
|
596
|
+
with path.open("rb") as handle:
|
|
597
|
+
first_bytes = handle.read(8 * 1024)
|
|
598
|
+
if b"\x00" in first_bytes:
|
|
599
|
+
return None
|
|
600
|
+
return first_bytes + handle.read()
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def _can_open_for_read(path: Path) -> bool:
|
|
604
|
+
try:
|
|
605
|
+
with path.open("rb"):
|
|
606
|
+
return True
|
|
607
|
+
except OSError:
|
|
608
|
+
return False
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def _split_grep_lines(
|
|
612
|
+
data: bytes, include_context_lines: bool
|
|
613
|
+
) -> tuple[list[str], list[str]]:
|
|
614
|
+
text = data.decode("utf-8", errors="replace")
|
|
615
|
+
matching_text = text.replace("\r\n", "\n").replace("\r", "")
|
|
616
|
+
matching_lines = _split_without_terminal_empty_line(matching_text)
|
|
617
|
+
|
|
618
|
+
if not include_context_lines:
|
|
619
|
+
return matching_lines, matching_lines
|
|
620
|
+
|
|
621
|
+
context_text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
622
|
+
if not context_text:
|
|
623
|
+
return matching_lines, []
|
|
624
|
+
return matching_lines, context_text.split("\n")
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def _split_without_terminal_empty_line(text: str) -> list[str]:
|
|
628
|
+
if not text:
|
|
629
|
+
return []
|
|
630
|
+
lines = text.split("\n")
|
|
631
|
+
if text.endswith("\n"):
|
|
632
|
+
lines.pop()
|
|
633
|
+
return lines
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
def _context_line_numbers(
|
|
637
|
+
line_number: int, line_count: int, context: int | float
|
|
638
|
+
) -> Iterator[int | float]:
|
|
639
|
+
"""Yield the same sequence as Pi's JavaScript context loop."""
|
|
640
|
+
|
|
641
|
+
start = max(1, line_number - context)
|
|
642
|
+
end = min(line_count, line_number + context)
|
|
643
|
+
if isinstance(context, float) and not context.is_integer():
|
|
644
|
+
current: int | float = start
|
|
645
|
+
while current <= end:
|
|
646
|
+
yield current
|
|
647
|
+
current += 1
|
|
648
|
+
return
|
|
649
|
+
yield from range(int(start), int(end) + 1)
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _context_source_line(lines: list[str], line_number: int | float) -> str:
|
|
653
|
+
if isinstance(line_number, float) and not line_number.is_integer():
|
|
654
|
+
return ""
|
|
655
|
+
index = int(line_number) - 1
|
|
656
|
+
return lines[index] if 0 <= index < len(lines) else ""
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def _render_search_output(
|
|
660
|
+
raw_lines: list[str],
|
|
661
|
+
*,
|
|
662
|
+
empty_text: str,
|
|
663
|
+
match_or_result_limit_reached: bool,
|
|
664
|
+
match_or_result_notice: str,
|
|
665
|
+
long_line_notice: str | None,
|
|
666
|
+
unreadable_count: int,
|
|
667
|
+
) -> tuple[str, bool]:
|
|
668
|
+
raw_output = "\n".join(raw_lines)
|
|
669
|
+
if raw_lines:
|
|
670
|
+
output, byte_truncated = _truncate_head_to_bytes(raw_output)
|
|
671
|
+
else:
|
|
672
|
+
output = empty_text
|
|
673
|
+
byte_truncated = False
|
|
674
|
+
|
|
675
|
+
notices: list[str] = []
|
|
676
|
+
if match_or_result_limit_reached:
|
|
677
|
+
notices.append(match_or_result_notice)
|
|
678
|
+
if byte_truncated:
|
|
679
|
+
notices.append("50.0KB limit reached")
|
|
680
|
+
if long_line_notice is not None:
|
|
681
|
+
notices.append(long_line_notice)
|
|
682
|
+
if unreadable_count:
|
|
683
|
+
notices.append(f"Skipped {unreadable_count} unreadable path(s)")
|
|
684
|
+
if notices:
|
|
685
|
+
notice_text = _truncate_utf8(". ".join(notices), _MAX_NOTICE_BYTES)
|
|
686
|
+
output += f"\n\n[{notice_text}]"
|
|
687
|
+
return output, byte_truncated
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _truncate_head_to_bytes(content: str) -> tuple[str, bool]:
|
|
691
|
+
truncation = truncate_head(
|
|
692
|
+
content, max_lines=2**53 - 1, max_bytes=DEFAULT_MAX_BYTES
|
|
693
|
+
)
|
|
694
|
+
return truncation.content, truncation.truncated
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
def _truncate_utf8(text: str, max_bytes: int) -> str:
|
|
698
|
+
text = sanitize_text(text)
|
|
699
|
+
encoded = text.encode("utf-8")
|
|
700
|
+
if len(encoded) <= max_bytes:
|
|
701
|
+
return text
|
|
702
|
+
return encoded[:max_bytes].decode("utf-8", errors="ignore")
|
|
703
|
+
|
|
704
|
+
|
|
705
|
+
def _format_number(value: int | float) -> str:
|
|
706
|
+
if isinstance(value, float) and value.is_integer():
|
|
707
|
+
return str(int(value))
|
|
708
|
+
return str(value)
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
def _success_result(model_text: str, *, truncated: bool) -> PiCompatOperationResult:
|
|
712
|
+
return PiCompatOperationResult(
|
|
713
|
+
model_text=sanitize_text(model_text),
|
|
714
|
+
truncated=truncated,
|
|
715
|
+
error_kind=None,
|
|
716
|
+
exit_code=None,
|
|
717
|
+
changed_path=None,
|
|
718
|
+
side_effect_state=PiCompatSideEffectState("not_attempted"),
|
|
719
|
+
)
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def _error_result(model_text: str, error_kind: str) -> PiCompatOperationResult:
|
|
723
|
+
return PiCompatOperationResult(
|
|
724
|
+
model_text=sanitize_text(model_text),
|
|
725
|
+
truncated=False,
|
|
726
|
+
error_kind=PiCompatErrorKind(error_kind),
|
|
727
|
+
exit_code=None,
|
|
728
|
+
changed_path=None,
|
|
729
|
+
side_effect_state=PiCompatSideEffectState("not_attempted"),
|
|
730
|
+
)
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def _root_error_result(path: Path, exc: OSError) -> PiCompatOperationResult:
|
|
734
|
+
if isinstance(exc, FileNotFoundError) or exc.errno == errno.ENOENT:
|
|
735
|
+
return _error_result(f"Path not found: {path}", "not_found")
|
|
736
|
+
if isinstance(exc, PermissionError) or exc.errno in {errno.EACCES, errno.EPERM}:
|
|
737
|
+
return _error_result(f"Permission denied: {path}", "permission_denied")
|
|
738
|
+
return _error_result(f"Unable to access path: {path}", "io_error")
|