millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,738 @@
1
+ """Deterministic Python implementations of Pi's ``grep`` and ``find`` tools.
2
+
3
+ The public operations intentionally stay independent from Millforge's tool
4
+ descriptors. They are a source-attributed behavioral port of Pi 0.79.6 with
5
+ the filesystem-search adaptations specified by Millforge Spec 11 section 7.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ import errno
12
+ import math
13
+ import os
14
+ from pathlib import Path
15
+ import re
16
+ import stat
17
+ import sys
18
+ from typing import Iterator, cast
19
+
20
+ from pathspec import PathSpec
21
+
22
+ from .contracts import (
23
+ PiCompatErrorKind,
24
+ PiCompatOperationResult,
25
+ PiCompatSideEffectState,
26
+ )
27
+ from .paths import _PathValidationError, _is_absolute_cwd, resolve_to_cwd
28
+ from .truncation import (
29
+ DEFAULT_MAX_BYTES,
30
+ GREP_MAX_LINE_LENGTH,
31
+ sanitize_text,
32
+ truncate_head,
33
+ truncate_line,
34
+ )
35
+
36
+
37
+ _DEFAULT_GREP_LIMIT = 100
38
+ _DEFAULT_FIND_LIMIT = 1_000
39
+ _MAX_NOTICE_BYTES = 2_048
40
+
41
+
42
+ @dataclass(frozen=True)
43
+ class _TraversalEntry:
44
+ path: Path
45
+ relative_path: str
46
+ kind: str
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class _IgnoreSpec:
51
+ base_relative_path: str
52
+ spec: PathSpec
53
+
54
+
55
+ @dataclass(frozen=True)
56
+ class _GlobMatcher:
57
+ spec: PathSpec
58
+ matches_path: bool
59
+ matches_nothing: bool = False
60
+
61
+ def matches(self, relative_path: str) -> bool:
62
+ if self.matches_nothing:
63
+ return False
64
+ candidate = (
65
+ relative_path if self.matches_path else relative_path.rsplit("/", 1)[-1]
66
+ )
67
+ return self.spec.match_file(candidate)
68
+
69
+
70
+ class _InvalidSearchArgument(ValueError):
71
+ """Raised for a regex or glob that cannot be evaluated by this port."""
72
+
73
+
74
+ class _RootSearchError(OSError):
75
+ """Raised when an explicit search root cannot be accessed."""
76
+
77
+
78
+ class _SearchTraversal:
79
+ """Walk one search root while collecting descendant read failures."""
80
+
81
+ def __init__(self) -> None:
82
+ self.unreadable_count = 0
83
+ self._unreadable_paths: set[str] = set()
84
+
85
+ def walk(self, root: Path) -> Iterator[_TraversalEntry]:
86
+ yield from self._walk_directory(
87
+ root,
88
+ relative_directory="",
89
+ inherited_ignore_specs=(),
90
+ include_directory=False,
91
+ is_root=True,
92
+ )
93
+
94
+ def mark_unreadable(self, path: Path) -> None:
95
+ key = os.fspath(path)
96
+ if key not in self._unreadable_paths:
97
+ self._unreadable_paths.add(key)
98
+ self.unreadable_count += 1
99
+
100
+ def was_unreadable(self, path: Path) -> bool:
101
+ return os.fspath(path) in self._unreadable_paths
102
+
103
+ def _walk_directory(
104
+ self,
105
+ directory: Path,
106
+ *,
107
+ relative_directory: str,
108
+ inherited_ignore_specs: tuple[_IgnoreSpec, ...],
109
+ include_directory: bool,
110
+ is_root: bool,
111
+ ) -> Iterator[_TraversalEntry]:
112
+ try:
113
+ entries = sorted(
114
+ os.scandir(directory),
115
+ key=lambda entry: (entry.name.casefold(), entry.name),
116
+ )
117
+ except OSError as exc:
118
+ if is_root:
119
+ raise _RootSearchError(*exc.args) from exc
120
+ self.mark_unreadable(directory)
121
+ return
122
+
123
+ ignore_specs = inherited_ignore_specs
124
+ ignore_file = directory / ".gitignore"
125
+ try:
126
+ ignore_stat = ignore_file.stat()
127
+ except FileNotFoundError:
128
+ ignore_stat = None
129
+ except OSError:
130
+ self.mark_unreadable(ignore_file)
131
+ ignore_stat = None
132
+
133
+ if ignore_stat is not None and stat.S_ISREG(ignore_stat.st_mode):
134
+ try:
135
+ ignore_text = ignore_file.read_text(encoding="utf-8", errors="replace")
136
+ except OSError:
137
+ self.mark_unreadable(ignore_file)
138
+ else:
139
+ try:
140
+ ignore_spec = PathSpec.from_lines(
141
+ "gitwildmatch", ignore_text.splitlines()
142
+ )
143
+ except (TypeError, ValueError, re.error):
144
+ # Git ignore files are intentionally forgiving. An invalid
145
+ # local rule cannot make the search operation itself fail.
146
+ ignore_spec = None
147
+ if ignore_spec is not None:
148
+ ignore_specs = (
149
+ *ignore_specs,
150
+ _IgnoreSpec(relative_directory, ignore_spec),
151
+ )
152
+
153
+ if include_directory:
154
+ yield _TraversalEntry(directory, relative_directory, "directory")
155
+
156
+ for entry in entries:
157
+ child_path = Path(entry.path)
158
+ relative_path = _join_relative_path(relative_directory, entry.name)
159
+ if self.was_unreadable(child_path):
160
+ continue
161
+
162
+ try:
163
+ is_symlink = entry.is_symlink()
164
+ is_directory = not is_symlink and entry.is_dir(follow_symlinks=False)
165
+ is_file = not is_symlink and entry.is_file(follow_symlinks=False)
166
+ except OSError:
167
+ self.mark_unreadable(child_path)
168
+ continue
169
+
170
+ if _is_ignored(relative_path, is_directory, ignore_specs):
171
+ continue
172
+
173
+ if is_symlink:
174
+ yield _TraversalEntry(child_path, relative_path, "symlink")
175
+ elif is_directory:
176
+ yield from self._walk_directory(
177
+ child_path,
178
+ relative_directory=relative_path,
179
+ inherited_ignore_specs=ignore_specs,
180
+ include_directory=True,
181
+ is_root=False,
182
+ )
183
+ elif is_file:
184
+ yield _TraversalEntry(child_path, relative_path, "file")
185
+
186
+
187
+ def execute_grep(
188
+ *,
189
+ cwd: Path,
190
+ pattern: str,
191
+ path: str | None = None,
192
+ glob: str | None = None,
193
+ ignoreCase: bool | None = None,
194
+ literal: bool | None = None,
195
+ context: int | float | None = None,
196
+ limit: int | float | None = None,
197
+ ) -> PiCompatOperationResult:
198
+ """Search non-ignored regular files using Pi's model-visible grep format."""
199
+
200
+ if not _is_absolute_cwd(cwd):
201
+ return _error_result("cwd must be an absolute path", "invalid_arguments")
202
+ if "\x00" in str(cwd):
203
+ return _error_result("cwd must not contain NUL bytes", "invalid_arguments")
204
+ invalid = _validate_grep_arguments(
205
+ pattern=pattern,
206
+ path=path,
207
+ glob=glob,
208
+ ignore_case=ignoreCase,
209
+ literal=literal,
210
+ context=context,
211
+ limit=limit,
212
+ )
213
+ if invalid is not None:
214
+ return _error_result(invalid, "invalid_arguments")
215
+
216
+ try:
217
+ search_root = _resolve_search_path(cwd, path)
218
+ root_kind = _search_root_kind(search_root, find_requires_directory=False)
219
+ matcher = _compile_grep_matcher(pattern, bool(ignoreCase), bool(literal))
220
+ glob_matcher = _compile_glob(glob) if glob else None
221
+ except _PathValidationError as exc:
222
+ return _error_result(str(exc), "invalid_arguments")
223
+ except _InvalidSearchArgument as exc:
224
+ return _error_result(str(exc), "invalid_arguments")
225
+ except OSError as exc:
226
+ return _root_error_result(
227
+ search_root if "search_root" in locals() else cwd, exc
228
+ )
229
+ except UnicodeError:
230
+ return _error_result(
231
+ "path cannot be encoded by this filesystem", "invalid_arguments"
232
+ )
233
+
234
+ effective_limit = max(1, limit if limit is not None else _DEFAULT_GREP_LIMIT)
235
+ context_value = context if context is not None and context > 0 else 0
236
+ raw_lines: list[str] = []
237
+ match_count = 0
238
+ match_limit_reached = False
239
+ lines_truncated = False
240
+ traversal = _SearchTraversal()
241
+
242
+ def process_file(
243
+ file_path: Path, display_path: str, *, explicit_root: bool = False
244
+ ) -> None:
245
+ nonlocal lines_truncated, match_count, match_limit_reached
246
+ if glob_matcher is not None and not glob_matcher.matches(display_path):
247
+ return
248
+ try:
249
+ data = _read_search_file(file_path)
250
+ except OSError as exc:
251
+ if explicit_root:
252
+ raise _RootSearchError(*exc.args) from exc
253
+ traversal.mark_unreadable(file_path)
254
+ return
255
+ if data is None:
256
+ return
257
+
258
+ matching_lines, context_source_lines = _split_grep_lines(
259
+ data, context_value > 0
260
+ )
261
+ for line_number, line in enumerate(matching_lines, start=1):
262
+ if not matcher.search(line):
263
+ continue
264
+ if match_count >= effective_limit:
265
+ match_limit_reached = True
266
+ return
267
+
268
+ match_count += 1
269
+ if context_value == 0:
270
+ rendered, was_truncated = truncate_line(line)
271
+ lines_truncated = lines_truncated or was_truncated
272
+ raw_lines.append(f"{display_path}:{line_number}: {rendered}")
273
+ else:
274
+ for current_line_number in _context_line_numbers(
275
+ line_number, len(context_source_lines), context_value
276
+ ):
277
+ source_line = _context_source_line(
278
+ context_source_lines, current_line_number
279
+ )
280
+ rendered, was_truncated = truncate_line(source_line)
281
+ lines_truncated = lines_truncated or was_truncated
282
+ if current_line_number == line_number:
283
+ raw_lines.append(
284
+ f"{display_path}:{_format_number(current_line_number)}: "
285
+ f"{rendered}"
286
+ )
287
+ else:
288
+ raw_lines.append(
289
+ f"{display_path}-{_format_number(current_line_number)}- "
290
+ f"{rendered}"
291
+ )
292
+
293
+ if match_count >= effective_limit:
294
+ match_limit_reached = True
295
+ return
296
+
297
+ if root_kind == "file":
298
+ try:
299
+ process_file(search_root, search_root.name, explicit_root=True)
300
+ except _RootSearchError as exc:
301
+ return _root_error_result(search_root, exc)
302
+ else:
303
+ try:
304
+ for entry in traversal.walk(search_root):
305
+ if match_limit_reached:
306
+ break
307
+ if entry.kind != "file":
308
+ continue
309
+ process_file(entry.path, entry.relative_path)
310
+ except _RootSearchError as exc:
311
+ return _root_error_result(search_root, exc)
312
+
313
+ rendered_output, byte_truncated = _render_search_output(
314
+ raw_lines,
315
+ empty_text="No matches found",
316
+ match_or_result_limit_reached=match_limit_reached,
317
+ match_or_result_notice=(
318
+ f"{_format_number(effective_limit)} matches limit reached. "
319
+ f"Use limit={_format_number(effective_limit * 2)} for more, or refine pattern"
320
+ ),
321
+ long_line_notice=(
322
+ f"Some lines truncated to {GREP_MAX_LINE_LENGTH} chars. Use read tool to see full lines"
323
+ if lines_truncated
324
+ else None
325
+ ),
326
+ unreadable_count=traversal.unreadable_count,
327
+ )
328
+ return _success_result(
329
+ rendered_output,
330
+ truncated=(
331
+ match_limit_reached
332
+ or byte_truncated
333
+ or lines_truncated
334
+ or traversal.unreadable_count > 0
335
+ ),
336
+ )
337
+
338
+
339
+ def execute_find(
340
+ *,
341
+ cwd: Path,
342
+ pattern: str,
343
+ path: str | None = None,
344
+ limit: int | float | None = None,
345
+ ) -> PiCompatOperationResult:
346
+ """Find non-ignored paths using Pi's relative-path result format."""
347
+
348
+ if not _is_absolute_cwd(cwd):
349
+ return _error_result("cwd must be an absolute path", "invalid_arguments")
350
+ if "\x00" in str(cwd):
351
+ return _error_result("cwd must not contain NUL bytes", "invalid_arguments")
352
+ invalid = _validate_find_arguments(pattern=pattern, path=path, limit=limit)
353
+ if invalid is not None:
354
+ return _error_result(invalid, "invalid_arguments")
355
+
356
+ try:
357
+ search_root = _resolve_search_path(cwd, path)
358
+ _search_root_kind(search_root, find_requires_directory=True)
359
+ matcher = _compile_glob(pattern)
360
+ except _PathValidationError as exc:
361
+ return _error_result(str(exc), "invalid_arguments")
362
+ except _InvalidSearchArgument as exc:
363
+ return _error_result(str(exc), "invalid_arguments")
364
+ except OSError as exc:
365
+ return _root_error_result(
366
+ search_root if "search_root" in locals() else cwd, exc
367
+ )
368
+ except UnicodeError:
369
+ return _error_result(
370
+ "path cannot be encoded by this filesystem", "invalid_arguments"
371
+ )
372
+
373
+ effective_limit = limit if limit is not None else _DEFAULT_FIND_LIMIT
374
+ raw_lines: list[str] = []
375
+ result_limit_reached = False
376
+ traversal = _SearchTraversal()
377
+
378
+ try:
379
+ for entry in traversal.walk(search_root):
380
+ if result_limit_reached:
381
+ break
382
+ if not matcher.matches(entry.relative_path):
383
+ continue
384
+
385
+ if entry.kind == "file" and not _can_open_for_read(entry.path):
386
+ traversal.mark_unreadable(entry.path)
387
+ continue
388
+
389
+ raw_lines.append(
390
+ f"{entry.relative_path}/"
391
+ if entry.kind == "directory"
392
+ else entry.relative_path
393
+ )
394
+ if effective_limit != 0 and len(raw_lines) >= effective_limit:
395
+ result_limit_reached = True
396
+ except _RootSearchError as exc:
397
+ return _root_error_result(search_root, exc)
398
+
399
+ if effective_limit == 0 and raw_lines:
400
+ result_limit_reached = True
401
+
402
+ rendered_output, byte_truncated = _render_search_output(
403
+ raw_lines,
404
+ empty_text="No files found matching pattern",
405
+ match_or_result_limit_reached=result_limit_reached,
406
+ match_or_result_notice=(
407
+ f"{_format_number(effective_limit)} results limit reached. "
408
+ f"Use limit={_format_number(effective_limit * 2)} for more, or refine pattern"
409
+ ),
410
+ long_line_notice=None,
411
+ unreadable_count=traversal.unreadable_count,
412
+ )
413
+ return _success_result(
414
+ rendered_output,
415
+ truncated=result_limit_reached
416
+ or byte_truncated
417
+ or traversal.unreadable_count > 0,
418
+ )
419
+
420
+
421
+ def _validate_grep_arguments(
422
+ *,
423
+ pattern: object,
424
+ path: object,
425
+ glob: object,
426
+ ignore_case: object,
427
+ literal: object,
428
+ context: object,
429
+ limit: object,
430
+ ) -> str | None:
431
+ if not isinstance(pattern, str):
432
+ return "grep pattern must be a string"
433
+ if path is not None and not isinstance(path, str):
434
+ return "grep path must be a string"
435
+ if isinstance(path, str) and "\x00" in path:
436
+ return "grep path must not contain NUL bytes"
437
+ if glob is not None and not isinstance(glob, str):
438
+ return "grep glob must be a string"
439
+ if ignore_case is not None and not isinstance(ignore_case, bool):
440
+ return "grep ignoreCase must be a boolean"
441
+ if literal is not None and not isinstance(literal, bool):
442
+ return "grep literal must be a boolean"
443
+ if not _is_json_number_or_none(context):
444
+ return "grep context must be a number"
445
+ if not _is_json_number_or_none(limit):
446
+ return "grep limit must be a number"
447
+ return None
448
+
449
+
450
+ def _validate_find_arguments(
451
+ *, pattern: object, path: object, limit: object
452
+ ) -> str | None:
453
+ if not isinstance(pattern, str):
454
+ return "find pattern must be a string"
455
+ if path is not None and not isinstance(path, str):
456
+ return "find path must be a string"
457
+ if isinstance(path, str) and "\x00" in path:
458
+ return "find path must not contain NUL bytes"
459
+ if not _is_json_number_or_none(limit):
460
+ return "find limit must be a number"
461
+ numeric_limit = cast(int | float | None, limit)
462
+ if numeric_limit is not None and (
463
+ numeric_limit < 0
464
+ or (isinstance(numeric_limit, float) and not numeric_limit.is_integer())
465
+ or numeric_limit > sys.maxsize
466
+ ):
467
+ return "find limit must be a non-negative integer"
468
+ return None
469
+
470
+
471
+ def _is_json_number_or_none(value: object) -> bool:
472
+ if value is None:
473
+ return True
474
+ if isinstance(value, bool):
475
+ return False
476
+ if isinstance(value, int):
477
+ return True
478
+ return isinstance(value, float) and math.isfinite(value)
479
+
480
+
481
+ def _resolve_search_path(cwd: Path, supplied_path: str | None) -> Path:
482
+ if not cwd.is_absolute():
483
+ raise _InvalidSearchArgument("cwd must be an absolute path")
484
+ return resolve_to_cwd(supplied_path or ".", cwd)
485
+
486
+
487
+ def _search_root_kind(path: Path, *, find_requires_directory: bool) -> str:
488
+ try:
489
+ path_stat = path.stat()
490
+ except FileNotFoundError as exc:
491
+ raise FileNotFoundError(*exc.args) from exc
492
+
493
+ if stat.S_ISDIR(path_stat.st_mode):
494
+ return "directory"
495
+ if stat.S_ISREG(path_stat.st_mode) and not find_requires_directory:
496
+ return "file"
497
+ if find_requires_directory:
498
+ raise _InvalidSearchArgument(f"Find path must be a directory: {path}")
499
+ raise _InvalidSearchArgument(f"Grep path must be a file or directory: {path}")
500
+
501
+
502
+ def _compile_grep_matcher(
503
+ pattern: str, ignore_case: bool, literal: bool
504
+ ) -> re.Pattern[str]:
505
+ source = re.escape(pattern) if literal else pattern
506
+ try:
507
+ return re.compile(source, re.IGNORECASE if ignore_case else 0)
508
+ except re.error as exc:
509
+ raise _InvalidSearchArgument(f"Invalid regex pattern: {exc}") from exc
510
+
511
+
512
+ def _compile_glob(pattern: str) -> _GlobMatcher:
513
+ _validate_glob_character_classes(pattern)
514
+ matches_path = "/" in pattern
515
+ if pattern.endswith("/"):
516
+ return _GlobMatcher(
517
+ PathSpec.from_lines("gitwildmatch", ()),
518
+ matches_path,
519
+ matches_nothing=True,
520
+ )
521
+ pattern_for_match = pattern or "*"
522
+ if pattern_for_match.startswith("!"):
523
+ pattern_for_match = f"\\{pattern_for_match}"
524
+ effective_pattern = pattern_for_match
525
+ if (
526
+ matches_path
527
+ and not pattern_for_match.startswith(("/", "**/"))
528
+ and pattern_for_match != "**"
529
+ ):
530
+ effective_pattern = f"**/{pattern_for_match}"
531
+ try:
532
+ spec = PathSpec.from_lines("gitwildmatch", (effective_pattern,))
533
+ except (TypeError, ValueError, re.error) as exc:
534
+ raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}") from exc
535
+ patterns = tuple(spec.patterns)
536
+ if not patterns:
537
+ raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}")
538
+ return _GlobMatcher(spec, matches_path)
539
+
540
+
541
+ def _validate_glob_character_classes(pattern: str) -> None:
542
+ escaped = False
543
+ in_character_class = False
544
+ for character in pattern:
545
+ if escaped:
546
+ escaped = False
547
+ continue
548
+ if character == "\\":
549
+ escaped = True
550
+ elif character == "[":
551
+ in_character_class = True
552
+ elif character == "]" and in_character_class:
553
+ in_character_class = False
554
+ if in_character_class:
555
+ raise _InvalidSearchArgument(f"Invalid glob pattern: {pattern}")
556
+
557
+
558
+ def _is_ignored(
559
+ relative_path: str,
560
+ is_directory: bool,
561
+ ignore_specs: tuple[_IgnoreSpec, ...],
562
+ ) -> bool:
563
+ ignored = False
564
+ for ignore_spec in ignore_specs:
565
+ candidate = _relative_to_ignore_base(
566
+ relative_path, ignore_spec.base_relative_path
567
+ )
568
+ if candidate is None:
569
+ continue
570
+ if is_directory:
571
+ candidate = f"{candidate}/"
572
+
573
+ decision: bool | None = None
574
+ for pattern in ignore_spec.spec.patterns:
575
+ if pattern.include is not None and pattern.match_file(candidate):
576
+ decision = bool(pattern.include)
577
+ if decision is not None:
578
+ ignored = decision
579
+ return ignored
580
+
581
+
582
+ def _relative_to_ignore_base(relative_path: str, base_relative_path: str) -> str | None:
583
+ if not base_relative_path:
584
+ return relative_path
585
+ prefix = f"{base_relative_path}/"
586
+ if relative_path.startswith(prefix):
587
+ return relative_path[len(prefix) :]
588
+ return None
589
+
590
+
591
+ def _join_relative_path(parent: str, child: str) -> str:
592
+ return f"{parent}/{child}" if parent else child
593
+
594
+
595
+ def _read_search_file(path: Path) -> bytes | None:
596
+ with path.open("rb") as handle:
597
+ first_bytes = handle.read(8 * 1024)
598
+ if b"\x00" in first_bytes:
599
+ return None
600
+ return first_bytes + handle.read()
601
+
602
+
603
+ def _can_open_for_read(path: Path) -> bool:
604
+ try:
605
+ with path.open("rb"):
606
+ return True
607
+ except OSError:
608
+ return False
609
+
610
+
611
+ def _split_grep_lines(
612
+ data: bytes, include_context_lines: bool
613
+ ) -> tuple[list[str], list[str]]:
614
+ text = data.decode("utf-8", errors="replace")
615
+ matching_text = text.replace("\r\n", "\n").replace("\r", "")
616
+ matching_lines = _split_without_terminal_empty_line(matching_text)
617
+
618
+ if not include_context_lines:
619
+ return matching_lines, matching_lines
620
+
621
+ context_text = text.replace("\r\n", "\n").replace("\r", "\n")
622
+ if not context_text:
623
+ return matching_lines, []
624
+ return matching_lines, context_text.split("\n")
625
+
626
+
627
+ def _split_without_terminal_empty_line(text: str) -> list[str]:
628
+ if not text:
629
+ return []
630
+ lines = text.split("\n")
631
+ if text.endswith("\n"):
632
+ lines.pop()
633
+ return lines
634
+
635
+
636
+ def _context_line_numbers(
637
+ line_number: int, line_count: int, context: int | float
638
+ ) -> Iterator[int | float]:
639
+ """Yield the same sequence as Pi's JavaScript context loop."""
640
+
641
+ start = max(1, line_number - context)
642
+ end = min(line_count, line_number + context)
643
+ if isinstance(context, float) and not context.is_integer():
644
+ current: int | float = start
645
+ while current <= end:
646
+ yield current
647
+ current += 1
648
+ return
649
+ yield from range(int(start), int(end) + 1)
650
+
651
+
652
+ def _context_source_line(lines: list[str], line_number: int | float) -> str:
653
+ if isinstance(line_number, float) and not line_number.is_integer():
654
+ return ""
655
+ index = int(line_number) - 1
656
+ return lines[index] if 0 <= index < len(lines) else ""
657
+
658
+
659
+ def _render_search_output(
660
+ raw_lines: list[str],
661
+ *,
662
+ empty_text: str,
663
+ match_or_result_limit_reached: bool,
664
+ match_or_result_notice: str,
665
+ long_line_notice: str | None,
666
+ unreadable_count: int,
667
+ ) -> tuple[str, bool]:
668
+ raw_output = "\n".join(raw_lines)
669
+ if raw_lines:
670
+ output, byte_truncated = _truncate_head_to_bytes(raw_output)
671
+ else:
672
+ output = empty_text
673
+ byte_truncated = False
674
+
675
+ notices: list[str] = []
676
+ if match_or_result_limit_reached:
677
+ notices.append(match_or_result_notice)
678
+ if byte_truncated:
679
+ notices.append("50.0KB limit reached")
680
+ if long_line_notice is not None:
681
+ notices.append(long_line_notice)
682
+ if unreadable_count:
683
+ notices.append(f"Skipped {unreadable_count} unreadable path(s)")
684
+ if notices:
685
+ notice_text = _truncate_utf8(". ".join(notices), _MAX_NOTICE_BYTES)
686
+ output += f"\n\n[{notice_text}]"
687
+ return output, byte_truncated
688
+
689
+
690
+ def _truncate_head_to_bytes(content: str) -> tuple[str, bool]:
691
+ truncation = truncate_head(
692
+ content, max_lines=2**53 - 1, max_bytes=DEFAULT_MAX_BYTES
693
+ )
694
+ return truncation.content, truncation.truncated
695
+
696
+
697
+ def _truncate_utf8(text: str, max_bytes: int) -> str:
698
+ text = sanitize_text(text)
699
+ encoded = text.encode("utf-8")
700
+ if len(encoded) <= max_bytes:
701
+ return text
702
+ return encoded[:max_bytes].decode("utf-8", errors="ignore")
703
+
704
+
705
+ def _format_number(value: int | float) -> str:
706
+ if isinstance(value, float) and value.is_integer():
707
+ return str(int(value))
708
+ return str(value)
709
+
710
+
711
+ def _success_result(model_text: str, *, truncated: bool) -> PiCompatOperationResult:
712
+ return PiCompatOperationResult(
713
+ model_text=sanitize_text(model_text),
714
+ truncated=truncated,
715
+ error_kind=None,
716
+ exit_code=None,
717
+ changed_path=None,
718
+ side_effect_state=PiCompatSideEffectState("not_attempted"),
719
+ )
720
+
721
+
722
+ def _error_result(model_text: str, error_kind: str) -> PiCompatOperationResult:
723
+ return PiCompatOperationResult(
724
+ model_text=sanitize_text(model_text),
725
+ truncated=False,
726
+ error_kind=PiCompatErrorKind(error_kind),
727
+ exit_code=None,
728
+ changed_path=None,
729
+ side_effect_state=PiCompatSideEffectState("not_attempted"),
730
+ )
731
+
732
+
733
+ def _root_error_result(path: Path, exc: OSError) -> PiCompatOperationResult:
734
+ if isinstance(exc, FileNotFoundError) or exc.errno == errno.ENOENT:
735
+ return _error_result(f"Path not found: {path}", "not_found")
736
+ if isinstance(exc, PermissionError) or exc.errno in {errno.EACCES, errno.EPERM}:
737
+ return _error_result(f"Permission denied: {path}", "permission_denied")
738
+ return _error_result(f"Unable to access path: {path}", "io_error")