hugpy-tools 0.2.0a0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,84 @@
1
+ """hugpy-tools: a slim, stdlib-only capability suite for autonomous agents.
2
+
3
+ Extracted from the sprawl of ``abstract_utilities`` / ``abstract_webtools`` and
4
+ rebuilt with no third-party runtime dependency, so a machine with only Python
5
+ can use it. It imports NO other ``hugpy_*`` package (foundation layer).
6
+
7
+ Modules:
8
+
9
+ paths path confinement (jail) + path-part helpers + filename sanitising
10
+ fs safe file/dir ops: encoding-detecting read, line-range read,
11
+ atomic write, exact-string edit, list/tree, rich file info
12
+ hashkit sha256 (text/file, streamed) + a cheap size+head quick hash
13
+ text approximate token counting, token/line chunking, unified diffs
14
+ data JSON/TOML/YAML read + atomic JSON write (safe dumping)
15
+ timekit UTC ISO / epoch conversions
16
+ web assess_webpage / prescreen_webpage — assessManager-parity webpage
17
+ assessment over urllib + html.parser (opt-in JS render)
18
+ search reserved namespace (owned elsewhere; not implemented here)
19
+
20
+ Every ``fs`` function takes an already-confined absolute path — call
21
+ ``paths.confine(root, path)`` first to enforce a jail. The web fetch never
22
+ follows a redirect off http(s) and disables ``file://`` / ``ftp://``.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from hugpy_tools.data import read_data, safe_json_dumps, write_json
27
+ from hugpy_tools.fs import (
28
+ atomic_write,
29
+ detect_encoding,
30
+ edit_replace,
31
+ file_info,
32
+ list_dir,
33
+ read_lines,
34
+ read_text,
35
+ tree,
36
+ )
37
+ from hugpy_tools.hashkit import quick_hash, sha256_file, sha256_text
38
+ from hugpy_tools.paths import PathEscape, confine, file_parts, sanitize_filename
39
+ from hugpy_tools.text import (
40
+ chunk_by_lines,
41
+ chunk_by_tokens,
42
+ count_tokens,
43
+ unified_diff,
44
+ )
45
+ from hugpy_tools.timekit import epoch_to_iso, iso_to_epoch, now_epoch, now_iso
46
+ from hugpy_tools.web import assess_webpage, prescreen_webpage
47
+
48
+ __all__ = [
49
+ # paths
50
+ "PathEscape",
51
+ "confine",
52
+ "file_parts",
53
+ "sanitize_filename",
54
+ # fs
55
+ "atomic_write",
56
+ "detect_encoding",
57
+ "edit_replace",
58
+ "file_info",
59
+ "list_dir",
60
+ "read_lines",
61
+ "read_text",
62
+ "tree",
63
+ # hashkit
64
+ "quick_hash",
65
+ "sha256_file",
66
+ "sha256_text",
67
+ # text
68
+ "chunk_by_lines",
69
+ "chunk_by_tokens",
70
+ "count_tokens",
71
+ "unified_diff",
72
+ # data
73
+ "read_data",
74
+ "safe_json_dumps",
75
+ "write_json",
76
+ # timekit
77
+ "epoch_to_iso",
78
+ "iso_to_epoch",
79
+ "now_epoch",
80
+ "now_iso",
81
+ # web
82
+ "assess_webpage",
83
+ "prescreen_webpage",
84
+ ]
hugpy_tools/data.py ADDED
@@ -0,0 +1,74 @@
1
+ """Structured-data read/write: JSON always, TOML via stdlib ``tomllib``, YAML
2
+ only when PyYAML happens to be installed. Writing is JSON-only on purpose
3
+ (stdlib has no TOML/YAML serializer); the write path is atomic.
4
+
5
+ Every format degrades honestly: an unavailable parser raises a clear error
6
+ naming exactly what is missing, never a silent wrong-format guess. Ported in
7
+ spirit from abstract_utilities.json_utils (safe_dump / safe_read) but slimmed
8
+ to the two calls an agent needs.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import os
14
+
15
+ from . import fs
16
+
17
+ # Map a lowercase extension to a logical format.
18
+ _EXT_FORMAT = {
19
+ ".json": "json",
20
+ ".toml": "toml",
21
+ ".yaml": "yaml",
22
+ ".yml": "yaml",
23
+ }
24
+
25
+
26
+ def _format_for(path: str, explicit: str | None) -> str:
27
+ if explicit:
28
+ return explicit.lower()
29
+ return _EXT_FORMAT.get(os.path.splitext(path)[1].lower(), "json")
30
+
31
+
32
+ def read_data(path: str, fmt: str | None = None):
33
+ """Parse a JSON/TOML/YAML file to a Python object. Format is inferred from
34
+ the extension unless ``fmt`` is given. Raises ``RuntimeError`` when a
35
+ parser is unavailable, ``ValueError`` on an unknown format."""
36
+ fmt = _format_for(path, fmt)
37
+ if fmt == "json":
38
+ with open(path, "rb") as fh:
39
+ return json.loads(fh.read().decode("utf-8", errors="replace"))
40
+ if fmt == "toml":
41
+ try:
42
+ import tomllib # stdlib >= 3.11
43
+ except ModuleNotFoundError as exc:
44
+ raise RuntimeError(
45
+ "TOML reading needs Python 3.11+ (stdlib 'tomllib'); this "
46
+ "interpreter is older. Install the 'tomli' backport and read "
47
+ "it yourself, or upgrade Python.") from exc
48
+ with open(path, "rb") as fh:
49
+ return tomllib.load(fh)
50
+ if fmt == "yaml":
51
+ try:
52
+ import yaml # optional third-party
53
+ except ModuleNotFoundError as exc:
54
+ raise RuntimeError(
55
+ "YAML reading needs PyYAML, which is not installed "
56
+ "(hugpy_tools is stdlib-only). Install 'PyYAML' to enable "
57
+ "YAML.") from exc
58
+ with open(path, "rb") as fh:
59
+ return yaml.safe_load(fh)
60
+ raise ValueError("unknown data format %r (use json/toml/yaml)" % fmt)
61
+
62
+
63
+ def safe_json_dumps(obj, indent: int = 2, sort_keys: bool = False) -> str:
64
+ """``json.dumps`` that never explodes on a non-serialisable value: anything
65
+ the encoder cannot handle is stringified via ``default=str`` and non-ASCII
66
+ is preserved (``ensure_ascii=False``)."""
67
+ return json.dumps(obj, indent=indent, sort_keys=sort_keys,
68
+ ensure_ascii=False, default=str)
69
+
70
+
71
+ def write_json(path: str, obj, indent: int = 2, sort_keys: bool = False) -> dict:
72
+ """Serialise ``obj`` to JSON and write it atomically. Returns
73
+ ``{path, bytes, mode}`` (from :func:`fs.atomic_write`)."""
74
+ return fs.atomic_write(path, safe_json_dumps(obj, indent, sort_keys) + "\n")
hugpy_tools/fs.py ADDED
@@ -0,0 +1,231 @@
1
+ """Safe file/directory operations (stdlib only).
2
+
3
+ Every function here takes an ALREADY-CONFINED absolute path — confinement is
4
+ the caller's job (see :func:`paths.confine`). That split keeps this module a
5
+ pure file-ops library with no policy of its own, which is exactly what a
6
+ standalone package wants.
7
+
8
+ Highlights ported from abstract_utilities:
9
+ * encoding detection on read (BOM sniff -> utf-8 -> cp1252 fallback)
10
+ * atomic write (temp file in the same dir + ``os.replace``)
11
+ * exact string edit with an occurrence-count contract
12
+ * directory listing + bounded recursive tree + rich file info
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import tempfile
18
+ from datetime import datetime, timezone
19
+
20
+ from . import hashkit
21
+
22
+ # BOM -> declared encoding
23
+ _BOMS = (
24
+ (b"\xef\xbb\xbf", "utf-8-sig"),
25
+ (b"\xff\xfe\x00\x00", "utf-32-le"),
26
+ (b"\x00\x00\xfe\xff", "utf-32-be"),
27
+ (b"\xff\xfe", "utf-16-le"),
28
+ (b"\xfe\xff", "utf-16-be"),
29
+ )
30
+
31
+
32
+ def detect_encoding(data: bytes) -> str:
33
+ """Best-effort encoding guess for ``data`` using only stdlib: honour a BOM,
34
+ then prefer strict utf-8, then fall back to cp1252 (a superset of latin-1
35
+ that decodes any byte). Never raises."""
36
+ for bom, enc in _BOMS:
37
+ if data.startswith(bom):
38
+ return enc
39
+ try:
40
+ data.decode("utf-8")
41
+ return "utf-8"
42
+ except UnicodeDecodeError:
43
+ return "cp1252"
44
+
45
+
46
+ def read_text(path: str, max_bytes: int | None = None) -> dict:
47
+ """Read a text file with encoding detection. Returns
48
+ ``{path, encoding, bytes, truncated, text}``."""
49
+ with open(path, "rb") as fh:
50
+ if max_bytes is None:
51
+ raw = fh.read()
52
+ truncated = False
53
+ else:
54
+ raw = fh.read(max_bytes + 1)
55
+ truncated = len(raw) > max_bytes
56
+ raw = raw[:max_bytes]
57
+ enc = detect_encoding(raw)
58
+ return {
59
+ "path": path,
60
+ "encoding": enc,
61
+ "bytes": len(raw),
62
+ "truncated": truncated,
63
+ "text": raw.decode(enc, errors="replace"),
64
+ }
65
+
66
+
67
+ def read_lines(path: str, start: int = 1, end: int | None = None,
68
+ number: bool = False) -> dict:
69
+ """Read a 1-based inclusive line range. ``end=None`` reads to EOF.
70
+ Returns ``{path, start, end, total_lines, returned, text}``."""
71
+ start = max(1, int(start))
72
+ with open(path, "rb") as fh:
73
+ raw = fh.read()
74
+ enc = detect_encoding(raw)
75
+ lines = raw.decode(enc, errors="replace").splitlines()
76
+ total = len(lines)
77
+ last = total if end is None else min(int(end), total)
78
+ chosen = lines[start - 1:last] if start <= total else []
79
+ if number:
80
+ width = len(str(last))
81
+ body = "\n".join("%*d\t%s" % (width, start + i, ln)
82
+ for i, ln in enumerate(chosen))
83
+ else:
84
+ body = "\n".join(chosen)
85
+ return {
86
+ "path": path,
87
+ "start": start,
88
+ "end": last,
89
+ "total_lines": total,
90
+ "returned": len(chosen),
91
+ "text": body,
92
+ }
93
+
94
+
95
+ def atomic_write(path: str, content: str, append: bool = False,
96
+ encoding: str = "utf-8") -> dict:
97
+ """Write ``content`` durably. Overwrite is atomic (temp file in the same
98
+ directory + ``os.replace``, so a reader never sees a half-written file).
99
+ Append opens the real file directly (append has no atomic swap). Returns
100
+ ``{path, bytes, mode}``."""
101
+ os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
102
+ data = content.encode(encoding, errors="replace")
103
+ if append:
104
+ with open(path, "ab") as fh:
105
+ fh.write(data)
106
+ else:
107
+ directory = os.path.dirname(path) or "."
108
+ fd, tmp = tempfile.mkstemp(dir=directory, prefix=".tmp-", suffix=".part")
109
+ try:
110
+ with os.fdopen(fd, "wb") as fh:
111
+ fh.write(data)
112
+ os.replace(tmp, path)
113
+ except BaseException:
114
+ try:
115
+ os.unlink(tmp)
116
+ except OSError:
117
+ pass
118
+ raise
119
+ return {"path": path, "bytes": len(data),
120
+ "mode": "append" if append else "overwrite"}
121
+
122
+
123
+ def edit_replace(path: str, old: str, new: str, count: int | None = None) -> dict:
124
+ """Exact string replacement in a text file. By default every occurrence is
125
+ replaced; pass ``count`` to bound it. Raises ``ValueError`` if ``old`` is
126
+ empty or not found. Returns ``{path, replaced, bytes}``; the write is
127
+ atomic."""
128
+ if old == "":
129
+ raise ValueError("old must be a non-empty string")
130
+ info = read_text(path)
131
+ text = info["text"]
132
+ occurrences = text.count(old)
133
+ if occurrences == 0:
134
+ raise ValueError("old string not found in %s" % path)
135
+ replaced = occurrences if count is None else min(occurrences, int(count))
136
+ text = text.replace(old, new, -1 if count is None else int(count))
137
+ out = atomic_write(path, text, encoding="utf-8"
138
+ if info["encoding"] in ("utf-8", "utf-8-sig", "cp1252")
139
+ else info["encoding"])
140
+ return {"path": path, "replaced": replaced, "bytes": out["bytes"]}
141
+
142
+
143
+ def _iso(ts: float) -> str:
144
+ return datetime.fromtimestamp(ts, tz=timezone.utc).isoformat()
145
+
146
+
147
+ def file_info(path: str, hash_files: bool = False) -> dict:
148
+ """Rich stat for one path. For a regular file with ``hash_files`` the
149
+ sha256 is included."""
150
+ st = os.lstat(path)
151
+ is_link = os.path.islink(path)
152
+ real = os.path.realpath(path)
153
+ is_dir = os.path.isdir(real)
154
+ info = {
155
+ "path": path,
156
+ "name": os.path.basename(path.rstrip(os.sep)) or path,
157
+ "type": "dir" if is_dir else ("symlink" if is_link and not os.path.exists(real) else "file"),
158
+ "is_dir": is_dir,
159
+ "is_symlink": is_link,
160
+ "size": st.st_size,
161
+ "mtime": _iso(st.st_mtime),
162
+ }
163
+ if hash_files and not is_dir and os.path.isfile(real):
164
+ info["sha256"] = hashkit.sha256_file(real)
165
+ return info
166
+
167
+
168
+ def list_dir(path: str, glob: str | None = None, files_only: bool = False,
169
+ dirs_only: bool = False, limit: int = 500) -> dict:
170
+ """List the immediate entries of a directory (sorted: dirs first, then
171
+ files, each alphabetical). Optional fnmatch ``glob`` on the entry name.
172
+ Returns ``{path, count, truncated, entries:[{name,type,size,mtime}]}``."""
173
+ import fnmatch
174
+ names = sorted(os.listdir(path))
175
+ entries = []
176
+ for name in names:
177
+ full = os.path.join(path, name)
178
+ is_dir = os.path.isdir(full)
179
+ if files_only and is_dir:
180
+ continue
181
+ if dirs_only and not is_dir:
182
+ continue
183
+ if glob and not fnmatch.fnmatch(name, glob):
184
+ continue
185
+ try:
186
+ st = os.lstat(full)
187
+ size, mtime = st.st_size, _iso(st.st_mtime)
188
+ except OSError:
189
+ size, mtime = None, None
190
+ entries.append({"name": name,
191
+ "type": "dir" if is_dir else "file",
192
+ "size": size, "mtime": mtime})
193
+ entries.sort(key=lambda e: (e["type"] != "dir", e["name"]))
194
+ truncated = len(entries) > limit
195
+ return {"path": path, "count": len(entries), "truncated": truncated,
196
+ "entries": entries[:limit]}
197
+
198
+
199
+ def tree(path: str, max_depth: int = 3, max_entries: int = 500,
200
+ show_files: bool = True) -> dict:
201
+ """Bounded recursive directory tree rendered as indented text. Skips
202
+ symlinked directories (never recurses out through a link). Returns
203
+ ``{path, entries, truncated, text}`` where ``entries`` is the count
204
+ emitted."""
205
+ lines: list[str] = []
206
+ state = {"n": 0, "truncated": False}
207
+
208
+ def walk(cur: str, depth: int, prefix: str) -> None:
209
+ if depth > max_depth or state["truncated"]:
210
+ return
211
+ try:
212
+ names = sorted(os.listdir(cur))
213
+ except OSError:
214
+ return
215
+ dirs = [n for n in names if os.path.isdir(os.path.join(cur, n))
216
+ and not os.path.islink(os.path.join(cur, n))]
217
+ files = [n for n in names if not os.path.isdir(os.path.join(cur, n))]
218
+ ordered = [(n, True) for n in dirs] + \
219
+ ([(n, False) for n in files] if show_files else [])
220
+ for name, is_dir in ordered:
221
+ if state["n"] >= max_entries:
222
+ state["truncated"] = True
223
+ return
224
+ state["n"] += 1
225
+ lines.append("%s%s%s" % (prefix, name, "/" if is_dir else ""))
226
+ if is_dir:
227
+ walk(os.path.join(cur, name), depth + 1, prefix + " ")
228
+
229
+ walk(path, 1, "")
230
+ return {"path": path, "entries": state["n"],
231
+ "truncated": state["truncated"], "text": "\n".join(lines)}
hugpy_tools/hashkit.py ADDED
@@ -0,0 +1,35 @@
1
+ """Content hashing helpers (stdlib ``hashlib`` only).
2
+
3
+ Ported from abstract_utilities.hash_utils (full_hash/quick_hash) but written
4
+ to take an already-confined path and stream in bounded chunks so a large file
5
+ never loads whole into memory.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+
11
+ _CHUNK = 1024 * 1024 # 1 MiB streaming window
12
+
13
+
14
+ def sha256_text(text: str) -> str:
15
+ return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()
16
+
17
+
18
+ def sha256_file(path: str) -> str:
19
+ """Full SHA-256 of a file, streamed in 1 MiB windows."""
20
+ h = hashlib.sha256()
21
+ with open(path, "rb") as fh:
22
+ for block in iter(lambda: fh.read(_CHUNK), b""):
23
+ h.update(block)
24
+ return h.hexdigest()
25
+
26
+
27
+ def quick_hash(path: str, bytes_to_read: int = 16384) -> str:
28
+ """Cheap identity hash: size + first ``bytes_to_read`` bytes. Good for
29
+ dedup pre-filtering, not a cryptographic guarantee."""
30
+ import os
31
+ h = hashlib.sha256()
32
+ h.update(str(os.path.getsize(path)).encode())
33
+ with open(path, "rb") as fh:
34
+ h.update(fh.read(max(0, int(bytes_to_read))))
35
+ return h.hexdigest()
hugpy_tools/paths.py ADDED
@@ -0,0 +1,80 @@
1
+ """Path confinement and path-part helpers (stdlib only).
2
+
3
+ `confine` is the jail primitive ported from the agent's tools/fs.py: a path
4
+ (absolute or root-relative) is resolved through ``os.path.realpath`` and must
5
+ land under the realpath of ``root``. This rejects ``..`` escapes AND symlink
6
+ escapes in one check. For writes the *parent* directory is confined (the
7
+ target file may not exist yet, but its directory does), then the basename is
8
+ re-attached — so a dangling path with a symlinked ancestor cannot slip through.
9
+
10
+ Fail-closed: any ambiguity about where a path lands is a ``PathEscape``, never
11
+ a guess. Everything here is pure stdlib so the module ports cleanly into a
12
+ standalone package.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import re
18
+
19
+
20
+ class PathEscape(ValueError):
21
+ """A path resolved outside the confinement root."""
22
+
23
+
24
+ def confine(root: str, path: str, for_write: bool = False) -> str:
25
+ """Resolve ``path`` to a real absolute path confined under ``root``.
26
+
27
+ Raises :class:`PathEscape` on any escape (``..``, absolute-outside, or a
28
+ symlink pointing out of the jail).
29
+ """
30
+ real_root = os.path.realpath(root)
31
+ candidate = path if os.path.isabs(path) else os.path.join(real_root, path)
32
+ if for_write:
33
+ parent = os.path.realpath(os.path.dirname(candidate) or real_root)
34
+ resolved = os.path.join(parent, os.path.basename(candidate))
35
+ else:
36
+ resolved = os.path.realpath(candidate)
37
+ if resolved != real_root and not resolved.startswith(real_root + os.sep):
38
+ raise PathEscape("path %r escapes the root (%s)" % (path, real_root))
39
+ return resolved
40
+
41
+
42
+ def file_parts(path: str) -> dict:
43
+ """Decompose ``path`` into its named components (dir/base/name/ext plus the
44
+ two enclosing directory levels), never raising on a bare or odd path."""
45
+ path = str(path)
46
+ dirname = os.path.dirname(path)
47
+ basename = os.path.basename(path)
48
+ filename, ext = os.path.splitext(basename)
49
+ parent_dirname = os.path.dirname(dirname)
50
+ super_dirname = os.path.dirname(parent_dirname)
51
+ return {
52
+ "file_path": path,
53
+ "dirname": dirname,
54
+ "basename": basename,
55
+ "filename": filename,
56
+ "ext": ext,
57
+ "dirbase": os.path.basename(dirname),
58
+ "parent_dirname": parent_dirname,
59
+ "parent_dirbase": os.path.basename(parent_dirname),
60
+ "super_dirname": super_dirname,
61
+ "super_dirbase": os.path.basename(super_dirname),
62
+ }
63
+
64
+
65
+ _UNSAFE_NAME = re.compile(r"[^A-Za-z0-9._-]+")
66
+
67
+
68
+ def sanitize_filename(name: str, replacement: str = "_", max_len: int = 255) -> str:
69
+ """Reduce ``name`` to a safe single path component: strip directory
70
+ separators, collapse unsafe characters, and bound the length. Never
71
+ returns an empty string (falls back to ``"file"``)."""
72
+ name = os.path.basename(str(name)).strip()
73
+ name = _UNSAFE_NAME.sub(replacement, name).strip("._-" + replacement)
74
+ if not name:
75
+ name = "file"
76
+ if len(name) > max_len:
77
+ stem, ext = os.path.splitext(name)
78
+ keep = max_len - len(ext)
79
+ name = (stem[:keep] if keep > 0 else stem[:max_len]) + (ext if keep > 0 else "")
80
+ return name
hugpy_tools/py.typed ADDED
File without changes
@@ -0,0 +1,173 @@
1
+ # `hugpy_tools.search` — fast, self-contained local content search
2
+
3
+ Stdlib-only. Imports nothing from the rest of hugpy. Give it roots + terms; it
4
+ returns line-level hits. It replaces the need for `abstract_search` /
5
+ `abstract_utilities` in the agent's filesystem search path: same behaviour and
6
+ API, correct include/exclude, materially faster.
7
+
8
+ ## Why
9
+
10
+ `abstract_search` shells out to Unix `find` (a subprocess per query, no
11
+ directory pruning) and its include/exclude filters are wrong (see
12
+ "abstract_search bugs" at the bottom). This module walks with `os.scandir`,
13
+ prunes excluded directories at walk time (they are never entered), filters by
14
+ extension before any `stat`/`open`, and matches literals on bytes — decoding
15
+ only the lines it needs.
16
+
17
+ ## NO implicit filtering
18
+
19
+ A search engine must not allow/exclude anything silently. **With no user
20
+ filters, you get EVERYTHING under the roots** — dotfiles, `node_modules`,
21
+ `.venv`, binaries. There is no default extension whitelist, no default excluded
22
+ dirs, no default hidden-file skipping.
23
+
24
+ Opt into the old conveniences EXPLICITLY with named **presets**:
25
+
26
+ | preset | applies |
27
+ |---|---|
28
+ | `"code"` | `CODE_TEXT_EXTS` allow-list + `EXCLUDE_NOISE_DIRS` + skip hidden |
29
+ | `"noise"` | `EXCLUDE_NOISE_DIRS` + `BINARY_EXTS` excludes + skip hidden |
30
+ | `"text"` | `BINARY_EXTS` + `EXCLUDE_NOISE_DIRS` excludes + skip hidden |
31
+
32
+ `CODE_TEXT_EXTS` / `EXCLUDE_NOISE_DIRS` / `BINARY_EXTS` are exported; pass them
33
+ as filter kwargs if you want.
34
+
35
+ **Preset merge = UNION (never be surprised).** Presets and your kwargs are
36
+ unioned field-by-field. Good for excludes; but the allow-whitelist is unioned
37
+ too, so a whitelist preset **broadens**: `preset="code"` + `allowed_exts=[".rs"]`
38
+ matches code/text **and** `.rs`, not just `.rs`. To **narrow** by extension use
39
+ `preset="noise"` (excludes only) + your own `allowed_exts`, or no preset.
40
+ Everything applied shows in the report's `effective_filters`.
41
+
42
+ Nothing is dropped silently. Every `find_content` / `get_files_and_dirs` /
43
+ `search_content` / `collect` records **`last_report()`**: `effective_filters`
44
+ (exactly what applied, incl. presets), `scanned`, `matched` (search) and
45
+ `skipped` counts by reason (`binary` / `unreadable` / `too_large` /
46
+ `symlink_loop`). Content search still skips binary files (a NUL-byte match is
47
+ meaningless) — but reports them. Pass `report=True` for `(result, report)`
48
+ inline.
49
+
50
+ ## Include / exclude semantics (when you DO filter)
51
+
52
+ - **Exclude always wins.** Anything matched by an exclude is gone, even if an
53
+ include also matches it.
54
+ - **Include narrows.** No include ⇒ everything not excluded is in scope.
55
+ - **Directory excludes** match an exact **path segment** and are **pruned**
56
+ during the walk: `build` drops `.../build/...` but never `.../rebuild/...`,
57
+ and the excluded subtree is never descended.
58
+ - **Globs** (`include_globs` / `exclude_globs`, aliases `allowed_patterns` /
59
+ `exclude_patterns`) match with `fnmatch` against **both** the root-relative
60
+ path **and** the basename, so `**/mct/*.py` and `*session*` both work (`*`
61
+ crosses `/`; a leading `**/` is normalised to `*`).
62
+ - **Extensions** (`exts` / `exclude_exts`) compare lower-cased, with a leading
63
+ dot.
64
+ - **Case:** name/glob/extension matching is case-insensitive. Literal content
65
+ matching is case-insensitive unless `case_sensitive=True`.
66
+ - **Symlinks** are not followed by default (least surprising; loops avoided).
67
+ `max_depth` / `max_bytes` are explicit params (defaults: unbounded, no cap).
68
+
69
+ `__init__.py` is never treated specially (the old `abstract_search` dropped
70
+ every `__init__.py` by default — one of the bugs this replaced).
71
+
72
+ ```python
73
+ from hugpy_tools.search import get_files_and_dirs, find_content, last_report
74
+
75
+ files = get_files_and_dirs(directory="/srv/app")[1] # EVERYTHING
76
+ files = get_files_and_dirs(directory="/srv/app", preset="code")[1] # code/text, no noise
77
+ hits, rep = find_content(directory="/srv/app", strings=["x"], preset="noise", report=True)
78
+ rep["effective_filters"]; rep["skipped"] # or last_report()
79
+ ```
80
+
81
+ ## Examples
82
+
83
+ ```python
84
+ from hugpy_tools.search import find_content, iter_files, make_filters, read_any_file
85
+
86
+ # 1. Enumerate .py files (node_modules etc. only pruned if you ask — preset/exclude).
87
+ files = list(iter_files("/srv/app", make_filters(allowed_exts=[".py"], preset="noise")))
88
+
89
+ # 2. Literal search (case-insensitive). Returns
90
+ # [{"file_path": str, "lines": [{"line": int, "content": str}, ...]}, ...]
91
+ hits = find_content(directory="/srv/app", strings=["assure_model_key"])
92
+
93
+ # 3. Intersection: files containing BOTH terms (total_strings defaults True).
94
+ hits = find_content(directory="/srv/app", strings=["nginx", "ssl"])
95
+
96
+ # 4. Path-scoped: only .py under any mct/ directory.
97
+ hits = find_content(directory="/srv/app", strings=["submit_pull"],
98
+ allowed_patterns=["**/mct/*.py"])
99
+
100
+ # 5. Regex.
101
+ hits = find_content(directory="/srv/app", strings=[r"def\s+\w+\("], regex=True)
102
+
103
+ # 6. Read one file as text (binary/oversize -> ValueError; missing -> FileNotFoundError).
104
+ text = read_any_file("/srv/app/README.md")
105
+ ```
106
+
107
+ ### Directive helpers used by the steward
108
+
109
+ `get_file_filters(root, **kw) -> (dirs, cfg, allowed, include_files, recursive)`
110
+ and `get_files_and_dirs(*, directory, cfg, recursive) -> (dirs, files)` mirror
111
+ `abstract_search`'s shapes so callers unpack them unchanged. `all` / `any` /
112
+ `none` directive logic lives in the caller (e.g. hugpy-agent's
113
+ `mct/session.py`): `all` = intersection, `any` = union, `none` = a file-level
114
+ veto (a file with an excluded term is dropped, not merely demoted).
115
+
116
+ ## Speed
117
+
118
+ `use_rg` (default **False**) enables an optional ripgrep prefilter. It is a
119
+ *pure* prefilter — ripgrep only narrows which files the Python line-matcher
120
+ opens, so results are byte-for-byte identical to the pure-Python path (tested).
121
+ It is off by default because the pure-Python pruned walk benchmarked **faster**
122
+ on real trees (ripgrep run with `-uuu` rescans ignored/venv subtrees the walk
123
+ prunes, and its subprocess cost rarely pays off); turn it on only for very large
124
+ trees scanned with a highly selective term.
125
+
126
+ ## Benchmark (this machine, 2026-09-24)
127
+
128
+ Enumerate `*.py` (excl. node_modules/build), literal `import`, regex
129
+ `def\s+\w+\(`. Wall time, best-of-2.
130
+
131
+ | tree | op | hugpy_tools | abstract_search | speedup |
132
+ |---|---|---|---|---|
133
+ | `/srv/hugpy/src/hugpy` | enumerate | 10.6 ms (n=1133) | 210 ms (n=863) | ~20x |
134
+ | | literal | 86 ms (n=1106) | 422 ms (n=860) | ~5x |
135
+ | | regex | 43 ms (n=1011) | 266 ms (n=820) | ~6x |
136
+ | `/srv/pyit/dev` | enumerate | 35 ms (n=3347) | 455 ms (n=4566) | ~13x |
137
+ | | literal | 114 ms (n=3205) | 1000 ms (n=4329) | ~9x |
138
+ | | regex | 108 ms (n=1914) | 853 ms (n=3377) | ~8x |
139
+
140
+ `abstract_utilities` delegates its search to `abstract_search`, so its numbers
141
+ track `abstract_search`'s (1010 ms literal on `/srv/pyit/dev`).
142
+
143
+ **Count differences are `abstract_search` bugs, not misses:**
144
+ - `hugpy_tools` finds the `__init__.py` files `abstract_search` drops (its
145
+ default `__init__*` exclude-pattern + `__init__` exclude-dir): 270 real source
146
+ files on the first tree, ~1010 on the second.
147
+ - `abstract_search` returns thousands of `.venv` / `site-packages` /
148
+ hidden-backup files (2229 on `/srv/pyit/dev`) that `hugpy_tools` prunes by
149
+ default — and pays for descending them.
150
+
151
+ ## abstract_search bugs (for reference; not fixed here)
152
+
153
+ Paths under `/srv/pyit/dev/abstract_search/src/abstract_search/`:
154
+ - `find_collect.py:89-91` — excluded dirs become `find ... ! -path '*d*'`: a
155
+ **substring** match on the full path (so `build` also drops `rebuild`,
156
+ `prebuilder`) that is a result **filter, not a `-prune`**, so `find` still
157
+ descends the excluded tree (slow).
158
+ - `find_collect.py:93-99` & `filters.py:296-301` — include/exclude *patterns*
159
+ match the **basename only** (`-name` / `fnmatch(name, …)`), so path globs like
160
+ `**/mct/session.py` match nothing.
161
+ - `constants.py:98,102` — defaults exclude the `__init__` dir **and** the
162
+ `__init__*` pattern, silently dropping every `__init__.py`.
163
+ - `filters.py:154-169` (`ensure_patterns`) — a pattern with no `*`/`?` is
164
+ rewritten to a prefix/suffix glob, so an intended exact match (`session.py`)
165
+ becomes `session.py*`.
166
+ - `find_content.py:35-44,186` (`_normalize`) — final line matching strips `//`
167
+ comments and everything after them, so a term inside a comment or after
168
+ `http://` never matches, even though the file passed the prefilter.
169
+ - `find_content.py:103,173` — each matched file is read from disk **twice**
170
+ (once in `getPaths`, once in the match loop).
171
+ - `find_content.py:108` (`getPaths`) — literal prefilter uses
172
+ `tot_strings not in og_content`: **case-sensitive**, which is why callers had
173
+ to retry every term in both cases.