sidegraph 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sidegraph/__init__.py +37 -0
- sidegraph/anchoring.py +246 -0
- sidegraph/bootstrap/__init__.py +49 -0
- sidegraph/bootstrap/apply.py +603 -0
- sidegraph/bootstrap/catalog.py +92 -0
- sidegraph/bootstrap/cli.py +827 -0
- sidegraph/bootstrap/integrations.py +184 -0
- sidegraph/bootstrap/model.py +277 -0
- sidegraph/bootstrap/planner.py +400 -0
- sidegraph/bootstrap/proof.py +103 -0
- sidegraph/bootstrap/review.py +331 -0
- sidegraph/bootstrap/scan.py +289 -0
- sidegraph/capture.py +1794 -0
- sidegraph/cli.py +1902 -0
- sidegraph/config.py +148 -0
- sidegraph/doc_import.py +2099 -0
- sidegraph/doctor.py +1429 -0
- sidegraph/domains.py +902 -0
- sidegraph/engine/__init__.py +7 -0
- sidegraph/engine/reader.py +353 -0
- sidegraph/gitio.py +572 -0
- sidegraph/host/__init__.py +7 -0
- sidegraph/host/hooks.py +770 -0
- sidegraph/importer.py +239 -0
- sidegraph/okf.py +471 -0
- sidegraph/profiles.py +459 -0
- sidegraph/retrieval.py +1657 -0
- sidegraph/schema.py +386 -0
- sidegraph/server.py +2608 -0
- sidegraph/store.py +3363 -0
- sidegraph/sync.py +885 -0
- sidegraph/verify.py +1040 -0
- sidegraph/viz/__init__.py +4 -0
- sidegraph/viz/assets/vis-network.min.js +33 -0
- sidegraph/viz/model.py +248 -0
- sidegraph/viz/render.py +110 -0
- sidegraph/viz/template.html +131 -0
- sidegraph-0.1.0.dist-info/METADATA +392 -0
- sidegraph-0.1.0.dist-info/RECORD +42 -0
- sidegraph-0.1.0.dist-info/WHEEL +4 -0
- sidegraph-0.1.0.dist-info/entry_points.txt +18 -0
- sidegraph-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""Interactive, side-effect-free review of bootstrap candidates."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import difflib
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import shlex
|
|
9
|
+
import subprocess
|
|
10
|
+
import tempfile
|
|
11
|
+
from collections.abc import Callable, Mapping
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from time import monotonic as _monotonic
|
|
14
|
+
from typing import TYPE_CHECKING, Any
|
|
15
|
+
|
|
16
|
+
from sidegraph.bootstrap.model import (
|
|
17
|
+
WARNING_CONSEQUENCES,
|
|
18
|
+
BootstrapCandidate,
|
|
19
|
+
BootstrapPlan,
|
|
20
|
+
EditableCandidate,
|
|
21
|
+
EditResult,
|
|
22
|
+
ReviewAction,
|
|
23
|
+
ReviewedCandidate,
|
|
24
|
+
ReviewResult,
|
|
25
|
+
)
|
|
26
|
+
from sidegraph.bootstrap.planner import replan_edited_candidate
|
|
27
|
+
from sidegraph.capture import redact
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from sidegraph.bootstrap.catalog import CanonicalCatalog
|
|
31
|
+
from sidegraph.engine.reader import GraphifyReader
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _editable(candidate: BootstrapCandidate) -> EditableCandidate:
|
|
35
|
+
"""Return the only fields an editor may see, redacting again at this boundary."""
|
|
36
|
+
title, _ = redact(candidate.title)
|
|
37
|
+
context, _ = redact(candidate.context)
|
|
38
|
+
choice, _ = redact(candidate.choice)
|
|
39
|
+
rejected = None
|
|
40
|
+
if candidate.rejected is not None:
|
|
41
|
+
rejected, _ = redact(candidate.rejected)
|
|
42
|
+
consequences = None
|
|
43
|
+
if candidate.consequences is not None:
|
|
44
|
+
consequences, _ = redact(candidate.consequences)
|
|
45
|
+
return EditableCandidate(
|
|
46
|
+
title=title,
|
|
47
|
+
context=context,
|
|
48
|
+
choice=choice,
|
|
49
|
+
rejected=rejected,
|
|
50
|
+
consequences=consequences,
|
|
51
|
+
kind=candidate.kind,
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def render_candidate(candidate: BootstrapCandidate) -> str:
|
|
56
|
+
"""Render every reviewable field through the redaction boundary."""
|
|
57
|
+
editable = _editable(candidate)
|
|
58
|
+
source, _ = redact(candidate.ref)
|
|
59
|
+
lines = [
|
|
60
|
+
f"title: {editable.title}",
|
|
61
|
+
f"source: {source}",
|
|
62
|
+
f"context: {editable.context}",
|
|
63
|
+
f"choice: {editable.choice}",
|
|
64
|
+
f"rejected: {editable.rejected or '-'}",
|
|
65
|
+
f"consequences: {editable.consequences or '-'}",
|
|
66
|
+
]
|
|
67
|
+
if candidate.warnings:
|
|
68
|
+
for warning in candidate.warnings:
|
|
69
|
+
lines.append(f"warning: {warning.value} — {WARNING_CONSEQUENCES[warning]}")
|
|
70
|
+
else:
|
|
71
|
+
lines.append("warnings: none")
|
|
72
|
+
if candidate.anchors:
|
|
73
|
+
for anchor in candidate.anchors:
|
|
74
|
+
name, _ = redact(anchor.descriptor.name)
|
|
75
|
+
tier = anchor.tier if anchor.tier is not None else "-"
|
|
76
|
+
lines.append(f"anchor: {name} ({anchor.status}, tier {tier})")
|
|
77
|
+
else:
|
|
78
|
+
lines.append("anchor: -")
|
|
79
|
+
return "\n".join(lines)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def render_redacted_diff(before: BootstrapCandidate, after: BootstrapCandidate) -> str:
|
|
83
|
+
"""Render an editor-safe unified diff without source-only candidate fields."""
|
|
84
|
+
old_json = json.dumps(_editable(before).model_dump(mode="json"), indent=2, sort_keys=True)
|
|
85
|
+
new_json = json.dumps(_editable(after).model_dump(mode="json"), indent=2, sort_keys=True)
|
|
86
|
+
old = old_json.splitlines(keepends=True)
|
|
87
|
+
new = new_json.splitlines(keepends=True)
|
|
88
|
+
return "".join(difflib.unified_diff(old, new, fromfile="before", tofile="after"))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _redact_json_text(value: object) -> object:
|
|
92
|
+
"""Remove secrets from parsed editor JSON before validation can render an error."""
|
|
93
|
+
if isinstance(value, str):
|
|
94
|
+
return redact(value)[0]
|
|
95
|
+
if isinstance(value, list):
|
|
96
|
+
return [_redact_json_text(item) for item in value]
|
|
97
|
+
if isinstance(value, dict):
|
|
98
|
+
return {
|
|
99
|
+
redact(key)[0] if isinstance(key, str) else key: _redact_json_text(item)
|
|
100
|
+
for key, item in value.items()
|
|
101
|
+
}
|
|
102
|
+
return value
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _load_redacted_editable(path: Path) -> EditableCandidate:
|
|
106
|
+
try:
|
|
107
|
+
parsed = json.loads(path.read_text(encoding="utf-8"))
|
|
108
|
+
except json.JSONDecodeError as error:
|
|
109
|
+
raise ValueError("edited JSON is invalid") from error
|
|
110
|
+
return EditableCandidate.model_validate(_redact_json_text(parsed))
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class EditRejected(ValueError):
|
|
114
|
+
"""Edited content failed to parse or validate. Carries the REDACTED buffer so the next
|
|
115
|
+
edit of the same candidate can reopen the user's own text instead of a fresh render."""
|
|
116
|
+
|
|
117
|
+
def __init__(self, message: str, buffer: str) -> None:
|
|
118
|
+
super().__init__(message)
|
|
119
|
+
self.buffer = buffer
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def edit_candidate(
|
|
123
|
+
candidate: BootstrapCandidate,
|
|
124
|
+
*,
|
|
125
|
+
editor: str,
|
|
126
|
+
reader: GraphifyReader | None = None,
|
|
127
|
+
catalog: CanonicalCatalog | None = None,
|
|
128
|
+
run_editor: Callable[..., subprocess.CompletedProcess[Any]] = subprocess.run,
|
|
129
|
+
seed: str | None = None,
|
|
130
|
+
) -> EditResult:
|
|
131
|
+
"""Edit redacted fields and replan; this function never writes canonical state."""
|
|
132
|
+
argv = shlex.split(editor)
|
|
133
|
+
if not argv:
|
|
134
|
+
raise ValueError("editor command is empty")
|
|
135
|
+
|
|
136
|
+
fd, name = tempfile.mkstemp(prefix="sidegraph-bootstrap-", suffix=".json")
|
|
137
|
+
path = Path(name)
|
|
138
|
+
try:
|
|
139
|
+
os.fchmod(fd, 0o600)
|
|
140
|
+
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
|
141
|
+
if seed is not None:
|
|
142
|
+
handle.write(seed)
|
|
143
|
+
else:
|
|
144
|
+
json.dump(_editable(candidate).model_dump(mode="json"), handle, indent=2)
|
|
145
|
+
handle.write("\n")
|
|
146
|
+
completed = run_editor([*argv, str(path)])
|
|
147
|
+
if completed.returncode != 0:
|
|
148
|
+
raise subprocess.CalledProcessError(completed.returncode, [*argv, str(path)])
|
|
149
|
+
try:
|
|
150
|
+
changed = _load_redacted_editable(path)
|
|
151
|
+
except ValueError as error:
|
|
152
|
+
raw = path.read_text(encoding="utf-8")
|
|
153
|
+
# Redact here explicitly: on a JSONDecodeError nothing was parsed, so
|
|
154
|
+
# _redact_json_text never ran over this text.
|
|
155
|
+
raise EditRejected(str(error), redact(raw)[0]) from error
|
|
156
|
+
replanned = replan_edited_candidate(candidate, changed, reader=reader, catalog=catalog)
|
|
157
|
+
diff = render_redacted_diff(candidate, replanned)
|
|
158
|
+
return EditResult(candidate=replanned, diff=diff)
|
|
159
|
+
finally:
|
|
160
|
+
path.unlink(missing_ok=True)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _batch_numbers(command: str, candidate_count: int, *, first: int) -> set[int] | None:
|
|
164
|
+
"""Parse `batch N,M`. Returns None when the command is not an unambiguous selection of
|
|
165
|
+
still-unreviewed candidates — the caller reprompts rather than assuming an intent."""
|
|
166
|
+
prefix = "batch "
|
|
167
|
+
if not command.startswith(prefix):
|
|
168
|
+
return None
|
|
169
|
+
values = [value.strip() for value in command.removeprefix(prefix).split(",")]
|
|
170
|
+
# isdigit() alone accepts Unicode digits (e.g. '²') that int() cannot parse and raises
|
|
171
|
+
# ValueError on; require plain ASCII digits so no input can escape as an exception. This
|
|
172
|
+
# also rejects non-ASCII decimal digits (e.g. Arabic-Indic '٣') that int() could parse —
|
|
173
|
+
# a deliberate choice, not a side effect: candidate numbers are indices this CLI prints in
|
|
174
|
+
# ASCII, so accepting another digit script would be a silent-conversion surface, not a
|
|
175
|
+
# feature.
|
|
176
|
+
if not values or any(not (value.isascii() and value.isdigit()) for value in values):
|
|
177
|
+
return None
|
|
178
|
+
numbers = {int(value) for value in values}
|
|
179
|
+
if (
|
|
180
|
+
len(numbers) != len(values)
|
|
181
|
+
or not numbers
|
|
182
|
+
or any(n < first or n > candidate_count for n in numbers)
|
|
183
|
+
):
|
|
184
|
+
return None
|
|
185
|
+
return numbers
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _edit_failure(detail: str) -> str:
|
|
189
|
+
"""Redact and clip an edit-failure message; the candidate is untouched either way. The
|
|
190
|
+
whole message is clipped to at most 200 characters, not just the detail before the
|
|
191
|
+
suffix is appended."""
|
|
192
|
+
redacted, _ = redact(detail)
|
|
193
|
+
suffix = "; candidate unchanged"
|
|
194
|
+
budget = 200 - len(suffix)
|
|
195
|
+
if len(redacted) > budget:
|
|
196
|
+
redacted = f"{redacted[: budget - 3]}..."
|
|
197
|
+
return f"{redacted}{suffix}"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _confirm_edited_candidate(read_line: Callable[[str], str]) -> ReviewAction:
|
|
201
|
+
actions = {
|
|
202
|
+
"accept": ReviewAction.ACCEPT,
|
|
203
|
+
"proposed": ReviewAction.KEEP_PROPOSED,
|
|
204
|
+
"cancel": ReviewAction.SKIP,
|
|
205
|
+
}
|
|
206
|
+
while True:
|
|
207
|
+
confirmation = read_line("Edited candidate: accept, proposed, or cancel? ").strip().lower()
|
|
208
|
+
if confirmation in actions:
|
|
209
|
+
return actions[confirmation]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def review_plan(
|
|
213
|
+
plan: BootstrapPlan,
|
|
214
|
+
*,
|
|
215
|
+
reader: GraphifyReader | None = None,
|
|
216
|
+
catalog: CanonicalCatalog | None = None,
|
|
217
|
+
read_line: Callable[[str], str] = input,
|
|
218
|
+
write_line: Callable[[str], None] = print,
|
|
219
|
+
environ: Mapping[str, str] = os.environ,
|
|
220
|
+
run_editor: Callable[..., subprocess.CompletedProcess[Any]] = subprocess.run,
|
|
221
|
+
monotonic: Callable[[], float] = _monotonic,
|
|
222
|
+
) -> ReviewResult:
|
|
223
|
+
"""Review candidates in plan order using explicit, run-local decisions only."""
|
|
224
|
+
items: list[ReviewedCandidate] = []
|
|
225
|
+
candidates = plan.candidates
|
|
226
|
+
for number, candidate in enumerate(candidates, start=1):
|
|
227
|
+
action_started = monotonic()
|
|
228
|
+
seed: str | None = None
|
|
229
|
+
while True:
|
|
230
|
+
write_line(f"\nCandidate {number}/{len(candidates)} [{candidate.key}]")
|
|
231
|
+
write_line(render_candidate(candidate))
|
|
232
|
+
command = (
|
|
233
|
+
read_line(
|
|
234
|
+
f"Candidate {number}/{len(candidates)}: "
|
|
235
|
+
"accept (a), proposed (p), skip (s), edit (e), "
|
|
236
|
+
"or batch 1,3? "
|
|
237
|
+
)
|
|
238
|
+
.strip()
|
|
239
|
+
.lower()
|
|
240
|
+
)
|
|
241
|
+
batch = _batch_numbers(command, len(candidates), first=number)
|
|
242
|
+
if command.startswith("batch"):
|
|
243
|
+
if batch is None:
|
|
244
|
+
example = (
|
|
245
|
+
f"batch {number}"
|
|
246
|
+
if number == len(candidates)
|
|
247
|
+
else f"batch {number},{len(candidates)}"
|
|
248
|
+
)
|
|
249
|
+
write_line(
|
|
250
|
+
f"batch needs distinct candidate numbers between {number} and "
|
|
251
|
+
f"{len(candidates)}, for example: {example}"
|
|
252
|
+
)
|
|
253
|
+
continue
|
|
254
|
+
remaining = candidates[number - 1 :]
|
|
255
|
+
elapsed = max(0.0, monotonic() - action_started)
|
|
256
|
+
elapsed_per_candidate = elapsed / len(remaining)
|
|
257
|
+
items.extend(
|
|
258
|
+
ReviewedCandidate(
|
|
259
|
+
candidate=remaining_candidate,
|
|
260
|
+
action=(
|
|
261
|
+
ReviewAction.ACCEPT if index in batch else ReviewAction.KEEP_PROPOSED
|
|
262
|
+
),
|
|
263
|
+
action_elapsed_seconds=elapsed_per_candidate,
|
|
264
|
+
)
|
|
265
|
+
for index, remaining_candidate in enumerate(remaining, start=number)
|
|
266
|
+
)
|
|
267
|
+
return ReviewResult(
|
|
268
|
+
items=tuple(items),
|
|
269
|
+
elapsed_seconds=sum(item.action_elapsed_seconds for item in items),
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
edited_action = False
|
|
273
|
+
if command == "a":
|
|
274
|
+
action = ReviewAction.ACCEPT
|
|
275
|
+
elif command == "p":
|
|
276
|
+
action = ReviewAction.KEEP_PROPOSED
|
|
277
|
+
elif command == "e":
|
|
278
|
+
editor = environ.get("VISUAL") or environ.get("EDITOR")
|
|
279
|
+
if not editor:
|
|
280
|
+
write_line(
|
|
281
|
+
"Set VISUAL or EDITOR, then rerun sidegraph-bootstrap --candidate "
|
|
282
|
+
f"{candidate.key}"
|
|
283
|
+
)
|
|
284
|
+
action = ReviewAction.KEEP_PROPOSED
|
|
285
|
+
else:
|
|
286
|
+
try:
|
|
287
|
+
edited = edit_candidate(
|
|
288
|
+
candidate,
|
|
289
|
+
editor=editor,
|
|
290
|
+
reader=reader,
|
|
291
|
+
catalog=catalog,
|
|
292
|
+
run_editor=run_editor,
|
|
293
|
+
seed=seed,
|
|
294
|
+
)
|
|
295
|
+
except subprocess.CalledProcessError as error:
|
|
296
|
+
write_line(
|
|
297
|
+
_edit_failure(f"edit cancelled by the editor (exit {error.returncode})")
|
|
298
|
+
)
|
|
299
|
+
seed = None # cancel means cancel: discard the buffer
|
|
300
|
+
continue
|
|
301
|
+
except OSError as error:
|
|
302
|
+
write_line(_edit_failure(f"editor step failed: {error.strerror or error}"))
|
|
303
|
+
seed = None # transient failure: reopen a fresh render, not stale text
|
|
304
|
+
continue
|
|
305
|
+
except ValueError as error:
|
|
306
|
+
# EditRejected carries the redacted buffer; a plain ValueError does not.
|
|
307
|
+
seed = getattr(error, "buffer", None)
|
|
308
|
+
write_line(_edit_failure(f"edited candidate was not valid: {error}"))
|
|
309
|
+
continue
|
|
310
|
+
write_line(edited.diff)
|
|
311
|
+
action = _confirm_edited_candidate(read_line)
|
|
312
|
+
if action != ReviewAction.SKIP:
|
|
313
|
+
candidate = edited.candidate
|
|
314
|
+
edited_action = True
|
|
315
|
+
else:
|
|
316
|
+
action = ReviewAction.SKIP
|
|
317
|
+
break
|
|
318
|
+
seed = None
|
|
319
|
+
elapsed = max(0.0, monotonic() - action_started)
|
|
320
|
+
items.append(
|
|
321
|
+
ReviewedCandidate(
|
|
322
|
+
candidate=candidate,
|
|
323
|
+
action=action,
|
|
324
|
+
edited=edited_action,
|
|
325
|
+
action_elapsed_seconds=elapsed,
|
|
326
|
+
)
|
|
327
|
+
)
|
|
328
|
+
return ReviewResult(
|
|
329
|
+
items=tuple(items),
|
|
330
|
+
elapsed_seconds=sum(item.action_elapsed_seconds for item in items),
|
|
331
|
+
)
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""Read-only, repository-confined source discovery for Bootstrap previews."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from fnmatch import fnmatchcase
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from sidegraph.bootstrap.model import Exclusion, ScanResult
|
|
9
|
+
from sidegraph.profiles import FlowProfile
|
|
10
|
+
|
|
11
|
+
DEFAULT_MAX_BYTES = 512_000
|
|
12
|
+
EXCLUDED_DIRS = frozenset(
|
|
13
|
+
{
|
|
14
|
+
".git",
|
|
15
|
+
".sidegraph",
|
|
16
|
+
".venv",
|
|
17
|
+
"venv",
|
|
18
|
+
"node_modules",
|
|
19
|
+
"vendor",
|
|
20
|
+
"dist",
|
|
21
|
+
"build",
|
|
22
|
+
"target",
|
|
23
|
+
"graphify-out",
|
|
24
|
+
"_build",
|
|
25
|
+
"coverage",
|
|
26
|
+
}
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _resolve(path: Path) -> Path | None:
|
|
31
|
+
try:
|
|
32
|
+
return path.resolve(strict=True)
|
|
33
|
+
except OSError:
|
|
34
|
+
return None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _inside(root: Path, candidate: Path) -> bool:
|
|
38
|
+
try:
|
|
39
|
+
candidate.resolve(strict=True).relative_to(root.resolve(strict=True))
|
|
40
|
+
except (OSError, ValueError):
|
|
41
|
+
return False
|
|
42
|
+
return True
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _display_path(root: Path, path: Path) -> str:
|
|
46
|
+
try:
|
|
47
|
+
return path.absolute().relative_to(root.absolute()).as_posix()
|
|
48
|
+
except ValueError:
|
|
49
|
+
return f"<outside-repository>/{path.name}"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _lexical_excluded_directory(root: Path, path: Path) -> Path | None:
|
|
53
|
+
try:
|
|
54
|
+
parts = path.absolute().relative_to(root.absolute()).parts
|
|
55
|
+
except ValueError:
|
|
56
|
+
return None
|
|
57
|
+
candidate = root.absolute()
|
|
58
|
+
for index, part in enumerate(parts, start=1):
|
|
59
|
+
candidate = candidate / part
|
|
60
|
+
if part in EXCLUDED_DIRS and candidate.is_dir():
|
|
61
|
+
return root.absolute().joinpath(*parts[:index])
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _excluded_directory(root: Path, path: Path) -> Path | None:
|
|
66
|
+
lexical = _lexical_excluded_directory(root, path)
|
|
67
|
+
if lexical is not None:
|
|
68
|
+
return lexical
|
|
69
|
+
resolved = _resolve(path)
|
|
70
|
+
resolved_root = _resolve(root)
|
|
71
|
+
if resolved is None or resolved_root is None:
|
|
72
|
+
return None
|
|
73
|
+
if _lexical_excluded_directory(resolved_root, resolved) is not None:
|
|
74
|
+
return path.absolute()
|
|
75
|
+
return None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _accept_file(
|
|
79
|
+
root: Path, path: Path, *, included: set[Path], max_bytes: int
|
|
80
|
+
) -> tuple[str | None, Exclusion | None]:
|
|
81
|
+
rel = _display_path(root, path)
|
|
82
|
+
if not _inside(root, path):
|
|
83
|
+
return None, Exclusion(path=rel, reason="outside-repository")
|
|
84
|
+
try:
|
|
85
|
+
size = path.stat().st_size
|
|
86
|
+
except OSError:
|
|
87
|
+
return None, Exclusion(path=rel, reason="unreadable")
|
|
88
|
+
resolved = _resolve(path)
|
|
89
|
+
if resolved is None:
|
|
90
|
+
return None, Exclusion(path=rel, reason="outside-repository")
|
|
91
|
+
override = resolved in included
|
|
92
|
+
if size > max_bytes and not override:
|
|
93
|
+
return None, Exclusion(path=rel, reason="over-size-limit")
|
|
94
|
+
try:
|
|
95
|
+
with path.open("rb") as handle:
|
|
96
|
+
prefix = handle.read(4096)
|
|
97
|
+
if b"\0" in prefix:
|
|
98
|
+
return None, Exclusion(path=rel, reason="binary")
|
|
99
|
+
path.read_text(encoding="utf-8")
|
|
100
|
+
except (OSError, UnicodeDecodeError):
|
|
101
|
+
return None, Exclusion(path=rel, reason="unreadable")
|
|
102
|
+
return rel, None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _discover_directory(
|
|
106
|
+
root: Path, directory: Path, visited: set[Path] | None = None
|
|
107
|
+
) -> tuple[list[Path], list[Exclusion]]:
|
|
108
|
+
if not _inside(root, directory):
|
|
109
|
+
return [directory], []
|
|
110
|
+
excluded = _excluded_directory(root, directory)
|
|
111
|
+
if excluded is not None:
|
|
112
|
+
return [], [Exclusion(path=_display_path(root, excluded), reason="excluded-directory")]
|
|
113
|
+
if visited is None:
|
|
114
|
+
visited = set()
|
|
115
|
+
resolved = _resolve(directory)
|
|
116
|
+
if resolved is None:
|
|
117
|
+
return [directory], []
|
|
118
|
+
if resolved in visited:
|
|
119
|
+
return [], []
|
|
120
|
+
visited.add(resolved)
|
|
121
|
+
|
|
122
|
+
files: list[Path] = []
|
|
123
|
+
exclusions: list[Exclusion] = []
|
|
124
|
+
try:
|
|
125
|
+
children = sorted(directory.iterdir(), key=lambda child: child.name)
|
|
126
|
+
except OSError:
|
|
127
|
+
return [], [Exclusion(path=_display_path(root, directory), reason="unreadable")]
|
|
128
|
+
|
|
129
|
+
for child in children:
|
|
130
|
+
try:
|
|
131
|
+
is_directory = child.is_dir()
|
|
132
|
+
except OSError:
|
|
133
|
+
files.append(child)
|
|
134
|
+
continue
|
|
135
|
+
if is_directory:
|
|
136
|
+
nested_files, nested_exclusions = _discover_directory(root, child, visited)
|
|
137
|
+
files.extend(nested_files)
|
|
138
|
+
exclusions.extend(nested_exclusions)
|
|
139
|
+
else:
|
|
140
|
+
files.append(child)
|
|
141
|
+
return files, exclusions
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _discover_profile_pattern(
|
|
145
|
+
root: Path,
|
|
146
|
+
directory: Path,
|
|
147
|
+
pattern: tuple[str, ...],
|
|
148
|
+
index: int = 0,
|
|
149
|
+
visited: set[tuple[Path, int]] | None = None,
|
|
150
|
+
) -> tuple[list[Path], list[Exclusion]]:
|
|
151
|
+
if not _inside(root, directory):
|
|
152
|
+
return [directory], []
|
|
153
|
+
excluded = _excluded_directory(root, directory)
|
|
154
|
+
if excluded is not None:
|
|
155
|
+
return [], [Exclusion(path=_display_path(root, excluded), reason="excluded-directory")]
|
|
156
|
+
if visited is None:
|
|
157
|
+
visited = set()
|
|
158
|
+
resolved = _resolve(directory)
|
|
159
|
+
if resolved is None:
|
|
160
|
+
return [directory], []
|
|
161
|
+
state = (resolved, index)
|
|
162
|
+
if state in visited:
|
|
163
|
+
return [], []
|
|
164
|
+
visited.add(state)
|
|
165
|
+
|
|
166
|
+
if index == len(pattern):
|
|
167
|
+
return [directory], []
|
|
168
|
+
try:
|
|
169
|
+
children = sorted(directory.iterdir(), key=lambda child: child.name)
|
|
170
|
+
except OSError:
|
|
171
|
+
return [], [Exclusion(path=_display_path(root, directory), reason="unreadable")]
|
|
172
|
+
|
|
173
|
+
files: list[Path] = []
|
|
174
|
+
exclusions: list[Exclusion] = []
|
|
175
|
+
part = pattern[index]
|
|
176
|
+
if part == "**":
|
|
177
|
+
nested_files, nested_exclusions = _discover_profile_pattern(
|
|
178
|
+
root, directory, pattern, index + 1, visited
|
|
179
|
+
)
|
|
180
|
+
files.extend(nested_files)
|
|
181
|
+
exclusions.extend(nested_exclusions)
|
|
182
|
+
for child in children:
|
|
183
|
+
try:
|
|
184
|
+
is_directory = child.is_dir()
|
|
185
|
+
except OSError:
|
|
186
|
+
is_directory = False
|
|
187
|
+
if is_directory:
|
|
188
|
+
nested_files, nested_exclusions = _discover_profile_pattern(
|
|
189
|
+
root, child, pattern, index, visited
|
|
190
|
+
)
|
|
191
|
+
files.extend(nested_files)
|
|
192
|
+
exclusions.extend(nested_exclusions)
|
|
193
|
+
return files, exclusions
|
|
194
|
+
|
|
195
|
+
for child in children:
|
|
196
|
+
if not fnmatchcase(child.name, part):
|
|
197
|
+
continue
|
|
198
|
+
if index == len(pattern) - 1:
|
|
199
|
+
files.append(child)
|
|
200
|
+
continue
|
|
201
|
+
try:
|
|
202
|
+
is_directory = child.is_dir()
|
|
203
|
+
except OSError:
|
|
204
|
+
is_directory = False
|
|
205
|
+
if is_directory:
|
|
206
|
+
nested_files, nested_exclusions = _discover_profile_pattern(
|
|
207
|
+
root, child, pattern, index + 1, visited
|
|
208
|
+
)
|
|
209
|
+
files.extend(nested_files)
|
|
210
|
+
exclusions.extend(nested_exclusions)
|
|
211
|
+
return files, exclusions
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _normalise_path(root: Path, path: Path) -> Path:
|
|
215
|
+
return path if path.is_absolute() else root / path
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def scan_sources(
|
|
219
|
+
root: Path,
|
|
220
|
+
profile: FlowProfile,
|
|
221
|
+
explicit_paths: tuple[Path, ...] = (),
|
|
222
|
+
included_files: tuple[Path, ...] = (),
|
|
223
|
+
max_bytes: int = DEFAULT_MAX_BYTES,
|
|
224
|
+
) -> ScanResult:
|
|
225
|
+
"""Discover eligible profile and explicit sources without reading outside ``root``."""
|
|
226
|
+
root = root.absolute()
|
|
227
|
+
supplied = tuple(_normalise_path(root, path) for path in explicit_paths)
|
|
228
|
+
included_paths = tuple(_normalise_path(root, path) for path in included_files)
|
|
229
|
+
included: set[Path] = set()
|
|
230
|
+
for path in included_paths:
|
|
231
|
+
resolved = _resolve(path)
|
|
232
|
+
if resolved is not None:
|
|
233
|
+
included.add(resolved)
|
|
234
|
+
candidates: list[Path] = []
|
|
235
|
+
exclusions: list[Exclusion] = []
|
|
236
|
+
|
|
237
|
+
for pattern in profile.ingest_globs:
|
|
238
|
+
discovered_files, pattern_exclusions = _discover_profile_pattern(
|
|
239
|
+
root, root, Path(pattern).parts
|
|
240
|
+
)
|
|
241
|
+
candidates.extend(discovered_files)
|
|
242
|
+
exclusions.extend(pattern_exclusions)
|
|
243
|
+
for path in supplied:
|
|
244
|
+
try:
|
|
245
|
+
is_directory = path.is_dir()
|
|
246
|
+
except OSError:
|
|
247
|
+
is_directory = False
|
|
248
|
+
if is_directory:
|
|
249
|
+
discovered_files, directory_exclusions = _discover_directory(root, path)
|
|
250
|
+
candidates.extend(discovered_files)
|
|
251
|
+
exclusions.extend(directory_exclusions)
|
|
252
|
+
else:
|
|
253
|
+
candidates.append(path)
|
|
254
|
+
candidates.extend(included_paths)
|
|
255
|
+
|
|
256
|
+
accepted_files: set[str] = set()
|
|
257
|
+
for path in sorted(set(candidates), key=lambda candidate: candidate.as_posix()):
|
|
258
|
+
if not _inside(root, path):
|
|
259
|
+
exclusions.append(
|
|
260
|
+
Exclusion(path=_display_path(root, path), reason="outside-repository")
|
|
261
|
+
)
|
|
262
|
+
continue
|
|
263
|
+
resolved = _resolve(path)
|
|
264
|
+
if resolved is None:
|
|
265
|
+
exclusions.append(
|
|
266
|
+
Exclusion(path=_display_path(root, path), reason="outside-repository")
|
|
267
|
+
)
|
|
268
|
+
continue
|
|
269
|
+
excluded = _excluded_directory(root, path)
|
|
270
|
+
if excluded is not None and resolved not in included:
|
|
271
|
+
exclusions.append(
|
|
272
|
+
Exclusion(path=_display_path(root, path), reason="excluded-directory")
|
|
273
|
+
)
|
|
274
|
+
continue
|
|
275
|
+
accepted, exclusion = _accept_file(root, path, included=included, max_bytes=max_bytes)
|
|
276
|
+
if accepted is not None:
|
|
277
|
+
accepted_files.add(accepted)
|
|
278
|
+
if exclusion is not None:
|
|
279
|
+
exclusions.append(exclusion)
|
|
280
|
+
|
|
281
|
+
unique_exclusions = {(item.path, item.reason): item for item in exclusions}
|
|
282
|
+
return ScanResult(
|
|
283
|
+
root=root.resolve().as_posix(),
|
|
284
|
+
files=tuple(sorted(accepted_files)),
|
|
285
|
+
exclusions=tuple(
|
|
286
|
+
item for _, item in sorted(unique_exclusions.items(), key=lambda entry: entry[0])
|
|
287
|
+
),
|
|
288
|
+
max_bytes=max_bytes,
|
|
289
|
+
)
|