hugpy-tools 0.2.0a0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hugpy_tools/__init__.py +84 -0
- hugpy_tools/data.py +74 -0
- hugpy_tools/fs.py +231 -0
- hugpy_tools/hashkit.py +35 -0
- hugpy_tools/paths.py +80 -0
- hugpy_tools/py.typed +0 -0
- hugpy_tools/search/README.md +173 -0
- hugpy_tools/search/__init__.py +761 -0
- hugpy_tools/text.py +90 -0
- hugpy_tools/timekit.py +33 -0
- hugpy_tools/web.py +421 -0
- hugpy_tools-0.2.0a0.dist-info/METADATA +93 -0
- hugpy_tools-0.2.0a0.dist-info/RECORD +16 -0
- hugpy_tools-0.2.0a0.dist-info/WHEEL +5 -0
- hugpy_tools-0.2.0a0.dist-info/licenses/LICENSE +41 -0
- hugpy_tools-0.2.0a0.dist-info/top_level.txt +1 -0
hugpy_tools/__init__.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""hugpy-tools: a slim, stdlib-only capability suite for autonomous agents.
|
|
2
|
+
|
|
3
|
+
Extracted from the sprawl of ``abstract_utilities`` / ``abstract_webtools`` and
|
|
4
|
+
rebuilt with no third-party runtime dependency, so a machine with only Python
|
|
5
|
+
can use it. It imports NO other ``hugpy_*`` package (foundation layer).
|
|
6
|
+
|
|
7
|
+
Modules:
|
|
8
|
+
|
|
9
|
+
paths path confinement (jail) + path-part helpers + filename sanitising
|
|
10
|
+
fs safe file/dir ops: encoding-detecting read, line-range read,
|
|
11
|
+
atomic write, exact-string edit, list/tree, rich file info
|
|
12
|
+
hashkit sha256 (text/file, streamed) + a cheap size+head quick hash
|
|
13
|
+
text approximate token counting, token/line chunking, unified diffs
|
|
14
|
+
data JSON/TOML/YAML read + atomic JSON write (safe dumping)
|
|
15
|
+
timekit UTC ISO / epoch conversions
|
|
16
|
+
web assess_webpage / prescreen_webpage — assessManager-parity webpage
|
|
17
|
+
assessment over urllib + html.parser (opt-in JS render)
|
|
18
|
+
search reserved namespace (owned elsewhere; not implemented here)
|
|
19
|
+
|
|
20
|
+
Every ``fs`` function takes an already-confined absolute path — call
|
|
21
|
+
``paths.confine(root, path)`` first to enforce a jail. The web fetch never
|
|
22
|
+
follows a redirect off http(s) and disables ``file://`` / ``ftp://``.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from hugpy_tools.data import read_data, safe_json_dumps, write_json
|
|
27
|
+
from hugpy_tools.fs import (
|
|
28
|
+
atomic_write,
|
|
29
|
+
detect_encoding,
|
|
30
|
+
edit_replace,
|
|
31
|
+
file_info,
|
|
32
|
+
list_dir,
|
|
33
|
+
read_lines,
|
|
34
|
+
read_text,
|
|
35
|
+
tree,
|
|
36
|
+
)
|
|
37
|
+
from hugpy_tools.hashkit import quick_hash, sha256_file, sha256_text
|
|
38
|
+
from hugpy_tools.paths import PathEscape, confine, file_parts, sanitize_filename
|
|
39
|
+
from hugpy_tools.text import (
|
|
40
|
+
chunk_by_lines,
|
|
41
|
+
chunk_by_tokens,
|
|
42
|
+
count_tokens,
|
|
43
|
+
unified_diff,
|
|
44
|
+
)
|
|
45
|
+
from hugpy_tools.timekit import epoch_to_iso, iso_to_epoch, now_epoch, now_iso
|
|
46
|
+
from hugpy_tools.web import assess_webpage, prescreen_webpage
|
|
47
|
+
|
|
48
|
+
__all__ = [
|
|
49
|
+
# paths
|
|
50
|
+
"PathEscape",
|
|
51
|
+
"confine",
|
|
52
|
+
"file_parts",
|
|
53
|
+
"sanitize_filename",
|
|
54
|
+
# fs
|
|
55
|
+
"atomic_write",
|
|
56
|
+
"detect_encoding",
|
|
57
|
+
"edit_replace",
|
|
58
|
+
"file_info",
|
|
59
|
+
"list_dir",
|
|
60
|
+
"read_lines",
|
|
61
|
+
"read_text",
|
|
62
|
+
"tree",
|
|
63
|
+
# hashkit
|
|
64
|
+
"quick_hash",
|
|
65
|
+
"sha256_file",
|
|
66
|
+
"sha256_text",
|
|
67
|
+
# text
|
|
68
|
+
"chunk_by_lines",
|
|
69
|
+
"chunk_by_tokens",
|
|
70
|
+
"count_tokens",
|
|
71
|
+
"unified_diff",
|
|
72
|
+
# data
|
|
73
|
+
"read_data",
|
|
74
|
+
"safe_json_dumps",
|
|
75
|
+
"write_json",
|
|
76
|
+
# timekit
|
|
77
|
+
"epoch_to_iso",
|
|
78
|
+
"iso_to_epoch",
|
|
79
|
+
"now_epoch",
|
|
80
|
+
"now_iso",
|
|
81
|
+
# web
|
|
82
|
+
"assess_webpage",
|
|
83
|
+
"prescreen_webpage",
|
|
84
|
+
]
|
hugpy_tools/data.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Structured-data read/write: JSON always, TOML via stdlib ``tomllib``, YAML
|
|
2
|
+
only when PyYAML happens to be installed. Writing is JSON-only on purpose
|
|
3
|
+
(stdlib has no TOML/YAML serializer); the write path is atomic.
|
|
4
|
+
|
|
5
|
+
Every format degrades honestly: an unavailable parser raises a clear error
|
|
6
|
+
naming exactly what is missing, never a silent wrong-format guess. Ported in
|
|
7
|
+
spirit from abstract_utilities.json_utils (safe_dump / safe_read) but slimmed
|
|
8
|
+
to the two calls an agent needs.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
|
|
15
|
+
from . import fs
|
|
16
|
+
|
|
17
|
+
# Map a lowercase extension to a logical format.
|
|
18
|
+
_EXT_FORMAT = {
|
|
19
|
+
".json": "json",
|
|
20
|
+
".toml": "toml",
|
|
21
|
+
".yaml": "yaml",
|
|
22
|
+
".yml": "yaml",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _format_for(path: str, explicit: str | None) -> str:
|
|
27
|
+
if explicit:
|
|
28
|
+
return explicit.lower()
|
|
29
|
+
return _EXT_FORMAT.get(os.path.splitext(path)[1].lower(), "json")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_data(path: str, fmt: str | None = None):
|
|
33
|
+
"""Parse a JSON/TOML/YAML file to a Python object. Format is inferred from
|
|
34
|
+
the extension unless ``fmt`` is given. Raises ``RuntimeError`` when a
|
|
35
|
+
parser is unavailable, ``ValueError`` on an unknown format."""
|
|
36
|
+
fmt = _format_for(path, fmt)
|
|
37
|
+
if fmt == "json":
|
|
38
|
+
with open(path, "rb") as fh:
|
|
39
|
+
return json.loads(fh.read().decode("utf-8", errors="replace"))
|
|
40
|
+
if fmt == "toml":
|
|
41
|
+
try:
|
|
42
|
+
import tomllib # stdlib >= 3.11
|
|
43
|
+
except ModuleNotFoundError as exc:
|
|
44
|
+
raise RuntimeError(
|
|
45
|
+
"TOML reading needs Python 3.11+ (stdlib 'tomllib'); this "
|
|
46
|
+
"interpreter is older. Install the 'tomli' backport and read "
|
|
47
|
+
"it yourself, or upgrade Python.") from exc
|
|
48
|
+
with open(path, "rb") as fh:
|
|
49
|
+
return tomllib.load(fh)
|
|
50
|
+
if fmt == "yaml":
|
|
51
|
+
try:
|
|
52
|
+
import yaml # optional third-party
|
|
53
|
+
except ModuleNotFoundError as exc:
|
|
54
|
+
raise RuntimeError(
|
|
55
|
+
"YAML reading needs PyYAML, which is not installed "
|
|
56
|
+
"(hugpy_tools is stdlib-only). Install 'PyYAML' to enable "
|
|
57
|
+
"YAML.") from exc
|
|
58
|
+
with open(path, "rb") as fh:
|
|
59
|
+
return yaml.safe_load(fh)
|
|
60
|
+
raise ValueError("unknown data format %r (use json/toml/yaml)" % fmt)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def safe_json_dumps(obj, indent: int = 2, sort_keys: bool = False) -> str:
|
|
64
|
+
"""``json.dumps`` that never explodes on a non-serialisable value: anything
|
|
65
|
+
the encoder cannot handle is stringified via ``default=str`` and non-ASCII
|
|
66
|
+
is preserved (``ensure_ascii=False``)."""
|
|
67
|
+
return json.dumps(obj, indent=indent, sort_keys=sort_keys,
|
|
68
|
+
ensure_ascii=False, default=str)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def write_json(path: str, obj, indent: int = 2, sort_keys: bool = False) -> dict:
|
|
72
|
+
"""Serialise ``obj`` to JSON and write it atomically. Returns
|
|
73
|
+
``{path, bytes, mode}`` (from :func:`fs.atomic_write`)."""
|
|
74
|
+
return fs.atomic_write(path, safe_json_dumps(obj, indent, sort_keys) + "\n")
|
hugpy_tools/fs.py
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Safe file/directory operations (stdlib only).
|
|
2
|
+
|
|
3
|
+
Every function here takes an ALREADY-CONFINED absolute path — confinement is
|
|
4
|
+
the caller's job (see :func:`paths.confine`). That split keeps this module a
|
|
5
|
+
pure file-ops library with no policy of its own, which is exactly what a
|
|
6
|
+
standalone package wants.
|
|
7
|
+
|
|
8
|
+
Highlights ported from abstract_utilities:
|
|
9
|
+
* encoding detection on read (BOM sniff -> utf-8 -> cp1252 fallback)
|
|
10
|
+
* atomic write (temp file in the same dir + ``os.replace``)
|
|
11
|
+
* exact string edit with an occurrence-count contract
|
|
12
|
+
* directory listing + bounded recursive tree + rich file info
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import tempfile
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
|
|
20
|
+
from . import hashkit
|
|
21
|
+
|
|
22
|
+
# BOM -> declared encoding
|
|
23
|
+
_BOMS = (
|
|
24
|
+
(b"\xef\xbb\xbf", "utf-8-sig"),
|
|
25
|
+
(b"\xff\xfe\x00\x00", "utf-32-le"),
|
|
26
|
+
(b"\x00\x00\xfe\xff", "utf-32-be"),
|
|
27
|
+
(b"\xff\xfe", "utf-16-le"),
|
|
28
|
+
(b"\xfe\xff", "utf-16-be"),
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def detect_encoding(data: bytes) -> str:
|
|
33
|
+
"""Best-effort encoding guess for ``data`` using only stdlib: honour a BOM,
|
|
34
|
+
then prefer strict utf-8, then fall back to cp1252 (a superset of latin-1
|
|
35
|
+
that decodes any byte). Never raises."""
|
|
36
|
+
for bom, enc in _BOMS:
|
|
37
|
+
if data.startswith(bom):
|
|
38
|
+
return enc
|
|
39
|
+
try:
|
|
40
|
+
data.decode("utf-8")
|
|
41
|
+
return "utf-8"
|
|
42
|
+
except UnicodeDecodeError:
|
|
43
|
+
return "cp1252"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def read_text(path: str, max_bytes: int | None = None) -> dict:
|
|
47
|
+
"""Read a text file with encoding detection. Returns
|
|
48
|
+
``{path, encoding, bytes, truncated, text}``."""
|
|
49
|
+
with open(path, "rb") as fh:
|
|
50
|
+
if max_bytes is None:
|
|
51
|
+
raw = fh.read()
|
|
52
|
+
truncated = False
|
|
53
|
+
else:
|
|
54
|
+
raw = fh.read(max_bytes + 1)
|
|
55
|
+
truncated = len(raw) > max_bytes
|
|
56
|
+
raw = raw[:max_bytes]
|
|
57
|
+
enc = detect_encoding(raw)
|
|
58
|
+
return {
|
|
59
|
+
"path": path,
|
|
60
|
+
"encoding": enc,
|
|
61
|
+
"bytes": len(raw),
|
|
62
|
+
"truncated": truncated,
|
|
63
|
+
"text": raw.decode(enc, errors="replace"),
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def read_lines(path: str, start: int = 1, end: int | None = None,
|
|
68
|
+
number: bool = False) -> dict:
|
|
69
|
+
"""Read a 1-based inclusive line range. ``end=None`` reads to EOF.
|
|
70
|
+
Returns ``{path, start, end, total_lines, returned, text}``."""
|
|
71
|
+
start = max(1, int(start))
|
|
72
|
+
with open(path, "rb") as fh:
|
|
73
|
+
raw = fh.read()
|
|
74
|
+
enc = detect_encoding(raw)
|
|
75
|
+
lines = raw.decode(enc, errors="replace").splitlines()
|
|
76
|
+
total = len(lines)
|
|
77
|
+
last = total if end is None else min(int(end), total)
|
|
78
|
+
chosen = lines[start - 1:last] if start <= total else []
|
|
79
|
+
if number:
|
|
80
|
+
width = len(str(last))
|
|
81
|
+
body = "\n".join("%*d\t%s" % (width, start + i, ln)
|
|
82
|
+
for i, ln in enumerate(chosen))
|
|
83
|
+
else:
|
|
84
|
+
body = "\n".join(chosen)
|
|
85
|
+
return {
|
|
86
|
+
"path": path,
|
|
87
|
+
"start": start,
|
|
88
|
+
"end": last,
|
|
89
|
+
"total_lines": total,
|
|
90
|
+
"returned": len(chosen),
|
|
91
|
+
"text": body,
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def atomic_write(path: str, content: str, append: bool = False,
|
|
96
|
+
encoding: str = "utf-8") -> dict:
|
|
97
|
+
"""Write ``content`` durably. Overwrite is atomic (temp file in the same
|
|
98
|
+
directory + ``os.replace``, so a reader never sees a half-written file).
|
|
99
|
+
Append opens the real file directly (append has no atomic swap). Returns
|
|
100
|
+
``{path, bytes, mode}``."""
|
|
101
|
+
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
|
102
|
+
data = content.encode(encoding, errors="replace")
|
|
103
|
+
if append:
|
|
104
|
+
with open(path, "ab") as fh:
|
|
105
|
+
fh.write(data)
|
|
106
|
+
else:
|
|
107
|
+
directory = os.path.dirname(path) or "."
|
|
108
|
+
fd, tmp = tempfile.mkstemp(dir=directory, prefix=".tmp-", suffix=".part")
|
|
109
|
+
try:
|
|
110
|
+
with os.fdopen(fd, "wb") as fh:
|
|
111
|
+
fh.write(data)
|
|
112
|
+
os.replace(tmp, path)
|
|
113
|
+
except BaseException:
|
|
114
|
+
try:
|
|
115
|
+
os.unlink(tmp)
|
|
116
|
+
except OSError:
|
|
117
|
+
pass
|
|
118
|
+
raise
|
|
119
|
+
return {"path": path, "bytes": len(data),
|
|
120
|
+
"mode": "append" if append else "overwrite"}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def edit_replace(path: str, old: str, new: str, count: int | None = None) -> dict:
|
|
124
|
+
"""Exact string replacement in a text file. By default every occurrence is
|
|
125
|
+
replaced; pass ``count`` to bound it. Raises ``ValueError`` if ``old`` is
|
|
126
|
+
empty or not found. Returns ``{path, replaced, bytes}``; the write is
|
|
127
|
+
atomic."""
|
|
128
|
+
if old == "":
|
|
129
|
+
raise ValueError("old must be a non-empty string")
|
|
130
|
+
info = read_text(path)
|
|
131
|
+
text = info["text"]
|
|
132
|
+
occurrences = text.count(old)
|
|
133
|
+
if occurrences == 0:
|
|
134
|
+
raise ValueError("old string not found in %s" % path)
|
|
135
|
+
replaced = occurrences if count is None else min(occurrences, int(count))
|
|
136
|
+
text = text.replace(old, new, -1 if count is None else int(count))
|
|
137
|
+
out = atomic_write(path, text, encoding="utf-8"
|
|
138
|
+
if info["encoding"] in ("utf-8", "utf-8-sig", "cp1252")
|
|
139
|
+
else info["encoding"])
|
|
140
|
+
return {"path": path, "replaced": replaced, "bytes": out["bytes"]}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _iso(ts: float) -> str:
|
|
144
|
+
return datetime.fromtimestamp(ts, tz=timezone.utc).isoformat()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def file_info(path: str, hash_files: bool = False) -> dict:
|
|
148
|
+
"""Rich stat for one path. For a regular file with ``hash_files`` the
|
|
149
|
+
sha256 is included."""
|
|
150
|
+
st = os.lstat(path)
|
|
151
|
+
is_link = os.path.islink(path)
|
|
152
|
+
real = os.path.realpath(path)
|
|
153
|
+
is_dir = os.path.isdir(real)
|
|
154
|
+
info = {
|
|
155
|
+
"path": path,
|
|
156
|
+
"name": os.path.basename(path.rstrip(os.sep)) or path,
|
|
157
|
+
"type": "dir" if is_dir else ("symlink" if is_link and not os.path.exists(real) else "file"),
|
|
158
|
+
"is_dir": is_dir,
|
|
159
|
+
"is_symlink": is_link,
|
|
160
|
+
"size": st.st_size,
|
|
161
|
+
"mtime": _iso(st.st_mtime),
|
|
162
|
+
}
|
|
163
|
+
if hash_files and not is_dir and os.path.isfile(real):
|
|
164
|
+
info["sha256"] = hashkit.sha256_file(real)
|
|
165
|
+
return info
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def list_dir(path: str, glob: str | None = None, files_only: bool = False,
|
|
169
|
+
dirs_only: bool = False, limit: int = 500) -> dict:
|
|
170
|
+
"""List the immediate entries of a directory (sorted: dirs first, then
|
|
171
|
+
files, each alphabetical). Optional fnmatch ``glob`` on the entry name.
|
|
172
|
+
Returns ``{path, count, truncated, entries:[{name,type,size,mtime}]}``."""
|
|
173
|
+
import fnmatch
|
|
174
|
+
names = sorted(os.listdir(path))
|
|
175
|
+
entries = []
|
|
176
|
+
for name in names:
|
|
177
|
+
full = os.path.join(path, name)
|
|
178
|
+
is_dir = os.path.isdir(full)
|
|
179
|
+
if files_only and is_dir:
|
|
180
|
+
continue
|
|
181
|
+
if dirs_only and not is_dir:
|
|
182
|
+
continue
|
|
183
|
+
if glob and not fnmatch.fnmatch(name, glob):
|
|
184
|
+
continue
|
|
185
|
+
try:
|
|
186
|
+
st = os.lstat(full)
|
|
187
|
+
size, mtime = st.st_size, _iso(st.st_mtime)
|
|
188
|
+
except OSError:
|
|
189
|
+
size, mtime = None, None
|
|
190
|
+
entries.append({"name": name,
|
|
191
|
+
"type": "dir" if is_dir else "file",
|
|
192
|
+
"size": size, "mtime": mtime})
|
|
193
|
+
entries.sort(key=lambda e: (e["type"] != "dir", e["name"]))
|
|
194
|
+
truncated = len(entries) > limit
|
|
195
|
+
return {"path": path, "count": len(entries), "truncated": truncated,
|
|
196
|
+
"entries": entries[:limit]}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def tree(path: str, max_depth: int = 3, max_entries: int = 500,
|
|
200
|
+
show_files: bool = True) -> dict:
|
|
201
|
+
"""Bounded recursive directory tree rendered as indented text. Skips
|
|
202
|
+
symlinked directories (never recurses out through a link). Returns
|
|
203
|
+
``{path, entries, truncated, text}`` where ``entries`` is the count
|
|
204
|
+
emitted."""
|
|
205
|
+
lines: list[str] = []
|
|
206
|
+
state = {"n": 0, "truncated": False}
|
|
207
|
+
|
|
208
|
+
def walk(cur: str, depth: int, prefix: str) -> None:
|
|
209
|
+
if depth > max_depth or state["truncated"]:
|
|
210
|
+
return
|
|
211
|
+
try:
|
|
212
|
+
names = sorted(os.listdir(cur))
|
|
213
|
+
except OSError:
|
|
214
|
+
return
|
|
215
|
+
dirs = [n for n in names if os.path.isdir(os.path.join(cur, n))
|
|
216
|
+
and not os.path.islink(os.path.join(cur, n))]
|
|
217
|
+
files = [n for n in names if not os.path.isdir(os.path.join(cur, n))]
|
|
218
|
+
ordered = [(n, True) for n in dirs] + \
|
|
219
|
+
([(n, False) for n in files] if show_files else [])
|
|
220
|
+
for name, is_dir in ordered:
|
|
221
|
+
if state["n"] >= max_entries:
|
|
222
|
+
state["truncated"] = True
|
|
223
|
+
return
|
|
224
|
+
state["n"] += 1
|
|
225
|
+
lines.append("%s%s%s" % (prefix, name, "/" if is_dir else ""))
|
|
226
|
+
if is_dir:
|
|
227
|
+
walk(os.path.join(cur, name), depth + 1, prefix + " ")
|
|
228
|
+
|
|
229
|
+
walk(path, 1, "")
|
|
230
|
+
return {"path": path, "entries": state["n"],
|
|
231
|
+
"truncated": state["truncated"], "text": "\n".join(lines)}
|
hugpy_tools/hashkit.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Content hashing helpers (stdlib ``hashlib`` only).
|
|
2
|
+
|
|
3
|
+
Ported from abstract_utilities.hash_utils (full_hash/quick_hash) but written
|
|
4
|
+
to take an already-confined path and stream in bounded chunks so a large file
|
|
5
|
+
never loads whole into memory.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
|
|
11
|
+
_CHUNK = 1024 * 1024 # 1 MiB streaming window
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def sha256_text(text: str) -> str:
|
|
15
|
+
return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def sha256_file(path: str) -> str:
|
|
19
|
+
"""Full SHA-256 of a file, streamed in 1 MiB windows."""
|
|
20
|
+
h = hashlib.sha256()
|
|
21
|
+
with open(path, "rb") as fh:
|
|
22
|
+
for block in iter(lambda: fh.read(_CHUNK), b""):
|
|
23
|
+
h.update(block)
|
|
24
|
+
return h.hexdigest()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def quick_hash(path: str, bytes_to_read: int = 16384) -> str:
|
|
28
|
+
"""Cheap identity hash: size + first ``bytes_to_read`` bytes. Good for
|
|
29
|
+
dedup pre-filtering, not a cryptographic guarantee."""
|
|
30
|
+
import os
|
|
31
|
+
h = hashlib.sha256()
|
|
32
|
+
h.update(str(os.path.getsize(path)).encode())
|
|
33
|
+
with open(path, "rb") as fh:
|
|
34
|
+
h.update(fh.read(max(0, int(bytes_to_read))))
|
|
35
|
+
return h.hexdigest()
|
hugpy_tools/paths.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Path confinement and path-part helpers (stdlib only).
|
|
2
|
+
|
|
3
|
+
`confine` is the jail primitive ported from the agent's tools/fs.py: a path
|
|
4
|
+
(absolute or root-relative) is resolved through ``os.path.realpath`` and must
|
|
5
|
+
land under the realpath of ``root``. This rejects ``..`` escapes AND symlink
|
|
6
|
+
escapes in one check. For writes the *parent* directory is confined (the
|
|
7
|
+
target file may not exist yet, but its directory does), then the basename is
|
|
8
|
+
re-attached — so a dangling path with a symlinked ancestor cannot slip through.
|
|
9
|
+
|
|
10
|
+
Fail-closed: any ambiguity about where a path lands is a ``PathEscape``, never
|
|
11
|
+
a guess. Everything here is pure stdlib so the module ports cleanly into a
|
|
12
|
+
standalone package.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class PathEscape(ValueError):
|
|
21
|
+
"""A path resolved outside the confinement root."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def confine(root: str, path: str, for_write: bool = False) -> str:
|
|
25
|
+
"""Resolve ``path`` to a real absolute path confined under ``root``.
|
|
26
|
+
|
|
27
|
+
Raises :class:`PathEscape` on any escape (``..``, absolute-outside, or a
|
|
28
|
+
symlink pointing out of the jail).
|
|
29
|
+
"""
|
|
30
|
+
real_root = os.path.realpath(root)
|
|
31
|
+
candidate = path if os.path.isabs(path) else os.path.join(real_root, path)
|
|
32
|
+
if for_write:
|
|
33
|
+
parent = os.path.realpath(os.path.dirname(candidate) or real_root)
|
|
34
|
+
resolved = os.path.join(parent, os.path.basename(candidate))
|
|
35
|
+
else:
|
|
36
|
+
resolved = os.path.realpath(candidate)
|
|
37
|
+
if resolved != real_root and not resolved.startswith(real_root + os.sep):
|
|
38
|
+
raise PathEscape("path %r escapes the root (%s)" % (path, real_root))
|
|
39
|
+
return resolved
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def file_parts(path: str) -> dict:
|
|
43
|
+
"""Decompose ``path`` into its named components (dir/base/name/ext plus the
|
|
44
|
+
two enclosing directory levels), never raising on a bare or odd path."""
|
|
45
|
+
path = str(path)
|
|
46
|
+
dirname = os.path.dirname(path)
|
|
47
|
+
basename = os.path.basename(path)
|
|
48
|
+
filename, ext = os.path.splitext(basename)
|
|
49
|
+
parent_dirname = os.path.dirname(dirname)
|
|
50
|
+
super_dirname = os.path.dirname(parent_dirname)
|
|
51
|
+
return {
|
|
52
|
+
"file_path": path,
|
|
53
|
+
"dirname": dirname,
|
|
54
|
+
"basename": basename,
|
|
55
|
+
"filename": filename,
|
|
56
|
+
"ext": ext,
|
|
57
|
+
"dirbase": os.path.basename(dirname),
|
|
58
|
+
"parent_dirname": parent_dirname,
|
|
59
|
+
"parent_dirbase": os.path.basename(parent_dirname),
|
|
60
|
+
"super_dirname": super_dirname,
|
|
61
|
+
"super_dirbase": os.path.basename(super_dirname),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
_UNSAFE_NAME = re.compile(r"[^A-Za-z0-9._-]+")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def sanitize_filename(name: str, replacement: str = "_", max_len: int = 255) -> str:
|
|
69
|
+
"""Reduce ``name`` to a safe single path component: strip directory
|
|
70
|
+
separators, collapse unsafe characters, and bound the length. Never
|
|
71
|
+
returns an empty string (falls back to ``"file"``)."""
|
|
72
|
+
name = os.path.basename(str(name)).strip()
|
|
73
|
+
name = _UNSAFE_NAME.sub(replacement, name).strip("._-" + replacement)
|
|
74
|
+
if not name:
|
|
75
|
+
name = "file"
|
|
76
|
+
if len(name) > max_len:
|
|
77
|
+
stem, ext = os.path.splitext(name)
|
|
78
|
+
keep = max_len - len(ext)
|
|
79
|
+
name = (stem[:keep] if keep > 0 else stem[:max_len]) + (ext if keep > 0 else "")
|
|
80
|
+
return name
|
hugpy_tools/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# `hugpy_tools.search` — fast, self-contained local content search
|
|
2
|
+
|
|
3
|
+
Stdlib-only. Imports nothing from the rest of hugpy. Give it roots + terms; it
|
|
4
|
+
returns line-level hits. It replaces the need for `abstract_search` /
|
|
5
|
+
`abstract_utilities` in the agent's filesystem search path: same behaviour and
|
|
6
|
+
API, correct include/exclude, materially faster.
|
|
7
|
+
|
|
8
|
+
## Why
|
|
9
|
+
|
|
10
|
+
`abstract_search` shells out to Unix `find` (a subprocess per query, no
|
|
11
|
+
directory pruning) and its include/exclude filters are wrong (see
|
|
12
|
+
"abstract_search bugs" at the bottom). This module walks with `os.scandir`,
|
|
13
|
+
prunes excluded directories at walk time (they are never entered), filters by
|
|
14
|
+
extension before any `stat`/`open`, and matches literals on bytes — decoding
|
|
15
|
+
only the lines it needs.
|
|
16
|
+
|
|
17
|
+
## NO implicit filtering
|
|
18
|
+
|
|
19
|
+
A search engine must not allow/exclude anything silently. **With no user
|
|
20
|
+
filters, you get EVERYTHING under the roots** — dotfiles, `node_modules`,
|
|
21
|
+
`.venv`, binaries. There is no default extension whitelist, no default excluded
|
|
22
|
+
dirs, no default hidden-file skipping.
|
|
23
|
+
|
|
24
|
+
Opt into the old conveniences EXPLICITLY with named **presets**:
|
|
25
|
+
|
|
26
|
+
| preset | applies |
|
|
27
|
+
|---|---|
|
|
28
|
+
| `"code"` | `CODE_TEXT_EXTS` allow-list + `EXCLUDE_NOISE_DIRS` + skip hidden |
|
|
29
|
+
| `"noise"` | `EXCLUDE_NOISE_DIRS` + `BINARY_EXTS` excludes + skip hidden |
|
|
30
|
+
| `"text"` | `BINARY_EXTS` + `EXCLUDE_NOISE_DIRS` excludes + skip hidden |
|
|
31
|
+
|
|
32
|
+
`CODE_TEXT_EXTS` / `EXCLUDE_NOISE_DIRS` / `BINARY_EXTS` are exported; pass them
|
|
33
|
+
as filter kwargs if you want.
|
|
34
|
+
|
|
35
|
+
**Preset merge = UNION (never be surprised).** Presets and your kwargs are
|
|
36
|
+
unioned field-by-field. Good for excludes; but the allow-whitelist is unioned
|
|
37
|
+
too, so a whitelist preset **broadens**: `preset="code"` + `allowed_exts=[".rs"]`
|
|
38
|
+
matches code/text **and** `.rs`, not just `.rs`. To **narrow** by extension use
|
|
39
|
+
`preset="noise"` (excludes only) + your own `allowed_exts`, or no preset.
|
|
40
|
+
Everything applied shows in the report's `effective_filters`.
|
|
41
|
+
|
|
42
|
+
Nothing is dropped silently. Every `find_content` / `get_files_and_dirs` /
|
|
43
|
+
`search_content` / `collect` records **`last_report()`**: `effective_filters`
|
|
44
|
+
(exactly what applied, incl. presets), `scanned`, `matched` (search) and
|
|
45
|
+
`skipped` counts by reason (`binary` / `unreadable` / `too_large` /
|
|
46
|
+
`symlink_loop`). Content search still skips binary files (a NUL-byte match is
|
|
47
|
+
meaningless) — but reports them. Pass `report=True` for `(result, report)`
|
|
48
|
+
inline.
|
|
49
|
+
|
|
50
|
+
## Include / exclude semantics (when you DO filter)
|
|
51
|
+
|
|
52
|
+
- **Exclude always wins.** Anything matched by an exclude is gone, even if an
|
|
53
|
+
include also matches it.
|
|
54
|
+
- **Include narrows.** No include ⇒ everything not excluded is in scope.
|
|
55
|
+
- **Directory excludes** match an exact **path segment** and are **pruned**
|
|
56
|
+
during the walk: `build` drops `.../build/...` but never `.../rebuild/...`,
|
|
57
|
+
and the excluded subtree is never descended.
|
|
58
|
+
- **Globs** (`include_globs` / `exclude_globs`, aliases `allowed_patterns` /
|
|
59
|
+
`exclude_patterns`) match with `fnmatch` against **both** the root-relative
|
|
60
|
+
path **and** the basename, so `**/mct/*.py` and `*session*` both work (`*`
|
|
61
|
+
crosses `/`; a leading `**/` is normalised to `*`).
|
|
62
|
+
- **Extensions** (`exts` / `exclude_exts`) compare lower-cased, with a leading
|
|
63
|
+
dot.
|
|
64
|
+
- **Case:** name/glob/extension matching is case-insensitive. Literal content
|
|
65
|
+
matching is case-insensitive unless `case_sensitive=True`.
|
|
66
|
+
- **Symlinks** are not followed by default (least surprising; loops avoided).
|
|
67
|
+
`max_depth` / `max_bytes` are explicit params (defaults: unbounded, no cap).
|
|
68
|
+
|
|
69
|
+
`__init__.py` is never treated specially (the old `abstract_search` dropped
|
|
70
|
+
every `__init__.py` by default — one of the bugs this replaced).
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from hugpy_tools.search import get_files_and_dirs, find_content, last_report
|
|
74
|
+
|
|
75
|
+
files = get_files_and_dirs(directory="/srv/app")[1] # EVERYTHING
|
|
76
|
+
files = get_files_and_dirs(directory="/srv/app", preset="code")[1] # code/text, no noise
|
|
77
|
+
hits, rep = find_content(directory="/srv/app", strings=["x"], preset="noise", report=True)
|
|
78
|
+
rep["effective_filters"]; rep["skipped"] # or last_report()
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Examples
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
from hugpy_tools.search import find_content, iter_files, make_filters, read_any_file
|
|
85
|
+
|
|
86
|
+
# 1. Enumerate .py files (node_modules etc. only pruned if you ask — preset/exclude).
|
|
87
|
+
files = list(iter_files("/srv/app", make_filters(allowed_exts=[".py"], preset="noise")))
|
|
88
|
+
|
|
89
|
+
# 2. Literal search (case-insensitive). Returns
|
|
90
|
+
# [{"file_path": str, "lines": [{"line": int, "content": str}, ...]}, ...]
|
|
91
|
+
hits = find_content(directory="/srv/app", strings=["assure_model_key"])
|
|
92
|
+
|
|
93
|
+
# 3. Intersection: files containing BOTH terms (total_strings defaults True).
|
|
94
|
+
hits = find_content(directory="/srv/app", strings=["nginx", "ssl"])
|
|
95
|
+
|
|
96
|
+
# 4. Path-scoped: only .py under any mct/ directory.
|
|
97
|
+
hits = find_content(directory="/srv/app", strings=["submit_pull"],
|
|
98
|
+
allowed_patterns=["**/mct/*.py"])
|
|
99
|
+
|
|
100
|
+
# 5. Regex.
|
|
101
|
+
hits = find_content(directory="/srv/app", strings=[r"def\s+\w+\("], regex=True)
|
|
102
|
+
|
|
103
|
+
# 6. Read one file as text (binary/oversize -> ValueError; missing -> FileNotFoundError).
|
|
104
|
+
text = read_any_file("/srv/app/README.md")
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### Directive helpers used by the steward
|
|
108
|
+
|
|
109
|
+
`get_file_filters(root, **kw) -> (dirs, cfg, allowed, include_files, recursive)`
|
|
110
|
+
and `get_files_and_dirs(*, directory, cfg, recursive) -> (dirs, files)` mirror
|
|
111
|
+
`abstract_search`'s shapes so callers unpack them unchanged. `all` / `any` /
|
|
112
|
+
`none` directive logic lives in the caller (e.g. hugpy-agent's
|
|
113
|
+
`mct/session.py`): `all` = intersection, `any` = union, `none` = a file-level
|
|
114
|
+
veto (a file with an excluded term is dropped, not merely demoted).
|
|
115
|
+
|
|
116
|
+
## Speed
|
|
117
|
+
|
|
118
|
+
`use_rg` (default **False**) enables an optional ripgrep prefilter. It is a
|
|
119
|
+
*pure* prefilter — ripgrep only narrows which files the Python line-matcher
|
|
120
|
+
opens, so results are byte-for-byte identical to the pure-Python path (tested).
|
|
121
|
+
It is off by default because the pure-Python pruned walk benchmarked **faster**
|
|
122
|
+
on real trees (ripgrep run with `-uuu` rescans ignored/venv subtrees the walk
|
|
123
|
+
prunes, and its subprocess cost rarely pays off); turn it on only for very large
|
|
124
|
+
trees scanned with a highly selective term.
|
|
125
|
+
|
|
126
|
+
## Benchmark (this machine, 2026-09-24)
|
|
127
|
+
|
|
128
|
+
Enumerate `*.py` (excl. node_modules/build), literal `import`, regex
|
|
129
|
+
`def\s+\w+\(`. Wall time, best-of-2.
|
|
130
|
+
|
|
131
|
+
| tree | op | hugpy_tools | abstract_search | speedup |
|
|
132
|
+
|---|---|---|---|---|
|
|
133
|
+
| `/srv/hugpy/src/hugpy` | enumerate | 10.6 ms (n=1133) | 210 ms (n=863) | ~20x |
|
|
134
|
+
| | literal | 86 ms (n=1106) | 422 ms (n=860) | ~5x |
|
|
135
|
+
| | regex | 43 ms (n=1011) | 266 ms (n=820) | ~6x |
|
|
136
|
+
| `/srv/pyit/dev` | enumerate | 35 ms (n=3347) | 455 ms (n=4566) | ~13x |
|
|
137
|
+
| | literal | 114 ms (n=3205) | 1000 ms (n=4329) | ~9x |
|
|
138
|
+
| | regex | 108 ms (n=1914) | 853 ms (n=3377) | ~8x |
|
|
139
|
+
|
|
140
|
+
`abstract_utilities` delegates its search to `abstract_search`, so its numbers
|
|
141
|
+
track `abstract_search`'s (1010 ms literal on `/srv/pyit/dev`).
|
|
142
|
+
|
|
143
|
+
**Count differences are `abstract_search` bugs, not misses:**
|
|
144
|
+
- `hugpy_tools` finds the `__init__.py` files `abstract_search` drops (its
|
|
145
|
+
default `__init__*` exclude-pattern + `__init__` exclude-dir): 270 real source
|
|
146
|
+
files on the first tree, ~1010 on the second.
|
|
147
|
+
- `abstract_search` returns thousands of `.venv` / `site-packages` /
|
|
148
|
+
hidden-backup files (2229 on `/srv/pyit/dev`) that `hugpy_tools` prunes by
|
|
149
|
+
default — and pays for descending them.
|
|
150
|
+
|
|
151
|
+
## abstract_search bugs (for reference; not fixed here)
|
|
152
|
+
|
|
153
|
+
Paths under `/srv/pyit/dev/abstract_search/src/abstract_search/`:
|
|
154
|
+
- `find_collect.py:89-91` — excluded dirs become `find ... ! -path '*d*'`: a
|
|
155
|
+
**substring** match on the full path (so `build` also drops `rebuild`,
|
|
156
|
+
`prebuilder`) that is a result **filter, not a `-prune`**, so `find` still
|
|
157
|
+
descends the excluded tree (slow).
|
|
158
|
+
- `find_collect.py:93-99` & `filters.py:296-301` — include/exclude *patterns*
|
|
159
|
+
match the **basename only** (`-name` / `fnmatch(name, …)`), so path globs like
|
|
160
|
+
`**/mct/session.py` match nothing.
|
|
161
|
+
- `constants.py:98,102` — defaults exclude the `__init__` dir **and** the
|
|
162
|
+
`__init__*` pattern, silently dropping every `__init__.py`.
|
|
163
|
+
- `filters.py:154-169` (`ensure_patterns`) — a pattern with no `*`/`?` is
|
|
164
|
+
rewritten to a prefix/suffix glob, so an intended exact match (`session.py`)
|
|
165
|
+
becomes `session.py*`.
|
|
166
|
+
- `find_content.py:35-44,186` (`_normalize`) — final line matching strips `//`
|
|
167
|
+
comments and everything after them, so a term inside a comment or after
|
|
168
|
+
`http://` never matches, even though the file passed the prefilter.
|
|
169
|
+
- `find_content.py:103,173` — each matched file is read from disk **twice**
|
|
170
|
+
(once in `getPaths`, once in the match loop).
|
|
171
|
+
- `find_content.py:108` (`getPaths`) — literal prefilter uses
|
|
172
|
+
`tot_strings not in og_content`: **case-sensitive**, which is why callers had
|
|
173
|
+
to retry every term in both cases.
|