hugpy-tools 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,41 @@
1
+ hugpy — Source-Available License
2
+
3
+ Copyright (c) 2026 putkoff (hugpy.ai). All rights reserved.
4
+
5
+ Permission is granted, free of charge, to use this software ("hugpy") for
6
+ personal and non-commercial purposes, and for time-limited commercial
7
+ evaluation, subject to the following conditions:
8
+
9
+ 1. Non-commercial use means use by an individual for personal purposes, or
10
+ use by a non-profit or educational institution for its own internal
11
+ purposes. Any use by, for, or on behalf of a for-profit business or in
12
+ connection with revenue-generating activity is commercial use — including
13
+ internal business use, use in producing goods or services, and use on
14
+ paid engagements.
15
+
16
+ 2. Commercial use requires a commercial license from the copyright holder.
17
+ Exception: a business may evaluate the software internally for up to
18
+ thirty (30) days free of charge; continued use after that requires a
19
+ commercial license.
20
+
21
+ 3. Redistribution of this software, in source or binary form, modified or
22
+ unmodified, is not permitted without prior written permission from the
23
+ copyright holder. Downloading the software from an official distribution
24
+ channel (PyPI, npm, hugpy.ai) is not redistribution.
25
+
26
+ 4. Modification for personal use or internal evaluation is permitted;
27
+ distribution of modified versions is not.
28
+
29
+ 5. This notice must be retained in all copies or substantial portions of
30
+ the software.
31
+
32
+ 6. Any use outside these terms automatically terminates this license.
33
+
34
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
35
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
36
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
37
+ COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY ARISING
38
+ FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
39
+ IN THE SOFTWARE.
40
+
41
+ For commercial licensing or redistribution permission: https://hugpy.ai
@@ -0,0 +1,93 @@
1
+ Metadata-Version: 2.4
2
+ Name: hugpy-tools
3
+ Version: 0.2.2
4
+ Summary: Hugpy tools: a slim, stdlib-only capability suite for autonomous agents (safe file/path ops, text chunking and diffing, hashing, structured-data read/write, and urllib+html.parser webpage assessment)
5
+ Author-email: putkoff <support@hugpy.ai>
6
+ License-Expression: LicenseRef-Proprietary
7
+ Project-URL: Homepage, https://hugpy.ai
8
+ Project-URL: Documentation, https://github.com/hugpy/hugpy/blob/main/py/foundation/hugpy_tools/README.md
9
+ Project-URL: Repository, https://github.com/hugpy/hugpy
10
+ Project-URL: Source, https://github.com/hugpy/hugpy/tree/main/py/foundation/hugpy_tools
11
+ Project-URL: Issues, https://github.com/hugpy/hugpy/issues
12
+ Project-URL: Changelog, https://github.com/hugpy/hugpy/releases
13
+ Project-URL: Architecture, https://github.com/hugpy/hugpy/blob/main/PARTITION.md
14
+ Keywords: hugpy,agent,tools,filesystem,webpage,assessment
15
+ Classifier: Development Status :: 3 - Alpha
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3 :: Only
20
+ Classifier: Topic :: Software Development :: Libraries
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Provides-Extra: toml
25
+ Requires-Dist: tomli; python_version < "3.11" and extra == "toml"
26
+ Provides-Extra: yaml
27
+ Requires-Dist: PyYAML; extra == "yaml"
28
+ Provides-Extra: render
29
+ Requires-Dist: playwright; extra == "render"
30
+ Provides-Extra: test
31
+ Requires-Dist: pytest>=8; extra == "test"
32
+ Requires-Dist: pytest-timeout; extra == "test"
33
+ Dynamic: license-file
34
+
35
+ # hugpy-tools
36
+
37
+ `hugpy_tools` — a slim, **stdlib-only** capability suite for autonomous agents,
38
+ extracted from `abstract_utilities` / `abstract_webtools` and rebuilt with no
39
+ third-party runtime dependency. Foundation layer: it imports no other `hugpy_*`
40
+ package, so it can be lifted out as its own distribution unchanged.
41
+
42
+ Ownership and allowed dependencies are declared in `py/partition.toml`; see
43
+ `PARTITION.md` at the workspace root.
44
+
45
+ Allowed Python dependencies inside the ecosystem: **none** (stdlib only).
46
+
47
+ ## Built-in tools
48
+
49
+ Every `fs` function takes an **already-confined** absolute path — call
50
+ `paths.confine(root, path)` first to enforce a workspace jail (rejects `..`,
51
+ absolute-outside, and symlink escapes in one check).
52
+
53
+ | Module | Function | What it does |
54
+ | --- | --- | --- |
55
+ | `paths` | `confine(root, path, for_write=False)` | Resolve a path under `root`; raise `PathEscape` on any escape. |
56
+ | `paths` | `file_parts(path)` | Decompose into dir/base/name/ext + two enclosing dir levels. |
57
+ | `paths` | `sanitize_filename(name)` | Reduce to one safe path component, length-bounded. |
58
+ | `fs` | `read_text(path, max_bytes=None)` | Read with encoding detection (BOM → utf-8 → cp1252). |
59
+ | `fs` | `read_lines(path, start, end, number=False)` | 1-based inclusive line-range read. |
60
+ | `fs` | `atomic_write(path, content, append=False)` | Durable write (temp file + `os.replace`); append supported. |
61
+ | `fs` | `edit_replace(path, old, new, count=None)` | Exact string replace with an occurrence-count contract; atomic. |
62
+ | `fs` | `file_info(path, hash_files=False)` | Size, mtime, type, symlink flag, optional sha256. |
63
+ | `fs` | `list_dir(path, glob=None, ...)` | Immediate entries (dirs first), optional fnmatch. |
64
+ | `fs` | `tree(path, max_depth=3, max_entries=500)` | Bounded recursive tree text (never follows dir symlinks). |
65
+ | `hashkit` | `sha256_text` / `sha256_file` / `quick_hash` | Content hashing (files streamed in 1 MiB windows). |
66
+ | `text` | `count_tokens(text)` | Dependency-free approximate BPE token count. |
67
+ | `text` | `chunk_by_lines` / `chunk_by_tokens` | Chunk on line or token budgets (paragraph-aware). |
68
+ | `text` | `unified_diff(before, after)` | Unified diff between two texts. |
69
+ | `data` | `read_data(path, fmt=None)` | Parse JSON always, TOML via stdlib `tomllib`, YAML if PyYAML present. |
70
+ | `data` | `write_json(path, obj)` / `safe_json_dumps(obj)` | Atomic JSON write / non-exploding dumps. |
71
+ | `timekit` | `now_iso` / `now_epoch` / `epoch_to_iso` / `iso_to_epoch` | UTC time conversions. |
72
+ | `web` | `assess_webpage(url, ...)` | assessManager-parity structured page assessment over urllib + `html.parser`. |
73
+ | `web` | `prescreen_webpage(url, ...)` | Cheap title + description + lede relevance pre-screen. |
74
+
75
+ ### Web assessment
76
+
77
+ `assess_webpage` returns the same dict shape as
78
+ `abstract_webtools.assessManager.assess_webpage`
79
+ (`{url, title, description, text, metadata, jsonld, links, truncated, render}`)
80
+ plus additive keys `final_url`, `status`, `canonical`, `lang`, `headings`, and
81
+ `error`. The fetch is http(s)-only, refuses redirects to other schemes,
82
+ disables `file://` / `ftp://`, caps the body, times out, handles gzip/deflate,
83
+ and detects charset (Content-Type → BOM → `<meta charset>` → utf-8 → cp1252).
84
+
85
+ JS rendering is opt-in and lazy: `force_render=True` (or an automatic fall-back
86
+ when the cheap fetch yields almost no text) drives a headless browser **only if
87
+ Playwright or Selenium is installed**, and raises a clear "install X" error
88
+ otherwise. The default path never needs a browser.
89
+
90
+ ## Layer
91
+
92
+ Foundation. `hugpy_tools` imports no other `hugpy_*` package. The `search`
93
+ subpackage is owned by a separate work-stream.
@@ -0,0 +1,59 @@
1
+ # hugpy-tools
2
+
3
+ `hugpy_tools` — a slim, **stdlib-only** capability suite for autonomous agents,
4
+ extracted from `abstract_utilities` / `abstract_webtools` and rebuilt with no
5
+ third-party runtime dependency. Foundation layer: it imports no other `hugpy_*`
6
+ package, so it can be lifted out as its own distribution unchanged.
7
+
8
+ Ownership and allowed dependencies are declared in `py/partition.toml`; see
9
+ `PARTITION.md` at the workspace root.
10
+
11
+ Allowed Python dependencies inside the ecosystem: **none** (stdlib only).
12
+
13
+ ## Built-in tools
14
+
15
+ Every `fs` function takes an **already-confined** absolute path — call
16
+ `paths.confine(root, path)` first to enforce a workspace jail (rejects `..`,
17
+ absolute-outside, and symlink escapes in one check).
18
+
19
+ | Module | Function | What it does |
20
+ | --- | --- | --- |
21
+ | `paths` | `confine(root, path, for_write=False)` | Resolve a path under `root`; raise `PathEscape` on any escape. |
22
+ | `paths` | `file_parts(path)` | Decompose into dir/base/name/ext + two enclosing dir levels. |
23
+ | `paths` | `sanitize_filename(name)` | Reduce to one safe path component, length-bounded. |
24
+ | `fs` | `read_text(path, max_bytes=None)` | Read with encoding detection (BOM → utf-8 → cp1252). |
25
+ | `fs` | `read_lines(path, start, end, number=False)` | 1-based inclusive line-range read. |
26
+ | `fs` | `atomic_write(path, content, append=False)` | Durable write (temp file + `os.replace`); append supported. |
27
+ | `fs` | `edit_replace(path, old, new, count=None)` | Exact string replace with an occurrence-count contract; atomic. |
28
+ | `fs` | `file_info(path, hash_files=False)` | Size, mtime, type, symlink flag, optional sha256. |
29
+ | `fs` | `list_dir(path, glob=None, ...)` | Immediate entries (dirs first), optional fnmatch. |
30
+ | `fs` | `tree(path, max_depth=3, max_entries=500)` | Bounded recursive tree text (never follows dir symlinks). |
31
+ | `hashkit` | `sha256_text` / `sha256_file` / `quick_hash` | Content hashing (files streamed in 1 MiB windows). |
32
+ | `text` | `count_tokens(text)` | Dependency-free approximate BPE token count. |
33
+ | `text` | `chunk_by_lines` / `chunk_by_tokens` | Chunk on line or token budgets (paragraph-aware). |
34
+ | `text` | `unified_diff(before, after)` | Unified diff between two texts. |
35
+ | `data` | `read_data(path, fmt=None)` | Parse JSON always, TOML via stdlib `tomllib`, YAML if PyYAML present. |
36
+ | `data` | `write_json(path, obj)` / `safe_json_dumps(obj)` | Atomic JSON write / non-exploding dumps. |
37
+ | `timekit` | `now_iso` / `now_epoch` / `epoch_to_iso` / `iso_to_epoch` | UTC time conversions. |
38
+ | `web` | `assess_webpage(url, ...)` | assessManager-parity structured page assessment over urllib + `html.parser`. |
39
+ | `web` | `prescreen_webpage(url, ...)` | Cheap title + description + lede relevance pre-screen. |
40
+
41
+ ### Web assessment
42
+
43
+ `assess_webpage` returns the same dict shape as
44
+ `abstract_webtools.assessManager.assess_webpage`
45
+ (`{url, title, description, text, metadata, jsonld, links, truncated, render}`)
46
+ plus additive keys `final_url`, `status`, `canonical`, `lang`, `headings`, and
47
+ `error`. The fetch is http(s)-only, refuses redirects to other schemes,
48
+ disables `file://` / `ftp://`, caps the body, times out, handles gzip/deflate,
49
+ and detects charset (Content-Type → BOM → `<meta charset>` → utf-8 → cp1252).
50
+
51
+ JS rendering is opt-in and lazy: `force_render=True` (or an automatic fall-back
52
+ when the cheap fetch yields almost no text) drives a headless browser **only if
53
+ Playwright or Selenium is installed**, and raises a clear "install X" error
54
+ otherwise. The default path never needs a browser.
55
+
56
+ ## Layer
57
+
58
+ Foundation. `hugpy_tools` imports no other `hugpy_*` package. The `search`
59
+ subpackage is owned by a separate work-stream.
@@ -0,0 +1,65 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "setuptools-scm>=8"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "hugpy-tools"
7
+ dynamic = ["version"]
8
+ description = "Hugpy tools: a slim, stdlib-only capability suite for autonomous agents (safe file/path ops, text chunking and diffing, hashing, structured-data read/write, and urllib+html.parser webpage assessment)"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "LicenseRef-Proprietary"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "putkoff", email = "support@hugpy.ai" }]
14
+ keywords = [
15
+ "hugpy",
16
+ "agent",
17
+ "tools",
18
+ "filesystem",
19
+ "webpage",
20
+ "assessment",
21
+ ]
22
+ classifiers = [
23
+ "Development Status :: 3 - Alpha",
24
+ "Intended Audience :: Developers",
25
+ "Operating System :: OS Independent",
26
+ "Programming Language :: Python :: 3",
27
+ "Programming Language :: Python :: 3 :: Only",
28
+ "Topic :: Software Development :: Libraries",
29
+ ]
30
+ # Deliberately dependency-free: stdlib only, so a machine with just Python can
31
+ # use it and no other hugpy_* package is imported (foundation layer).
32
+ dependencies = []
33
+
34
+ [project.urls]
35
+ Homepage = "https://hugpy.ai"
36
+ Documentation = "https://github.com/hugpy/hugpy/blob/main/py/foundation/hugpy_tools/README.md"
37
+ Repository = "https://github.com/hugpy/hugpy"
38
+ Source = "https://github.com/hugpy/hugpy/tree/main/py/foundation/hugpy_tools"
39
+ Issues = "https://github.com/hugpy/hugpy/issues"
40
+ Changelog = "https://github.com/hugpy/hugpy/releases"
41
+ Architecture = "https://github.com/hugpy/hugpy/blob/main/PARTITION.md"
42
+
43
+ [project.optional-dependencies]
44
+ # TOML reading uses stdlib tomllib on 3.11+; on 3.10 install the backport.
45
+ toml = ["tomli; python_version < '3.11'"]
46
+ # YAML reading is opt-in; the default never imports it.
47
+ yaml = ["PyYAML"]
48
+ # assess_webpage(force_render=True) can drive a headless browser if present.
49
+ render = ["playwright"]
50
+ test = ["pytest>=8", "pytest-timeout"]
51
+
52
+ [tool.setuptools.packages.find]
53
+ where = ["src"]
54
+
55
+ [tool.setuptools.package-data]
56
+ hugpy_tools = ["py.typed", "**/*.json", "**/*.md", "**/*.txt"]
57
+
58
+ # ---------------------------------------------------------------------------
59
+ # Lockstep workspace version (2026-09-22): every in-tree hugpy-* distribution
60
+ # takes ONE version from the workspace git tag (vX.Y.Z at the repo root). A
61
+ # checkout without git metadata builds as 0.0.0+unknown, which central refuses
62
+ # to advertise to workers. (Identical policy to every sibling package.)
63
+ [tool.setuptools_scm]
64
+ root = "../../.."
65
+ fallback_version = "0.0.0+unknown"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,84 @@
1
+ """hugpy-tools: a slim, stdlib-only capability suite for autonomous agents.
2
+
3
+ Extracted from the sprawl of ``abstract_utilities`` / ``abstract_webtools`` and
4
+ rebuilt with no third-party runtime dependency, so a machine with only Python
5
+ can use it. It imports NO other ``hugpy_*`` package (foundation layer).
6
+
7
+ Modules:
8
+
9
+ paths path confinement (jail) + path-part helpers + filename sanitising
10
+ fs safe file/dir ops: encoding-detecting read, line-range read,
11
+ atomic write, exact-string edit, list/tree, rich file info
12
+ hashkit sha256 (text/file, streamed) + a cheap size+head quick hash
13
+ text approximate token counting, token/line chunking, unified diffs
14
+ data JSON/TOML/YAML read + atomic JSON write (safe dumping)
15
+ timekit UTC ISO / epoch conversions
16
+ web assess_webpage / prescreen_webpage — assessManager-parity webpage
17
+ assessment over urllib + html.parser (opt-in JS render)
18
+ search reserved namespace (owned elsewhere; not implemented here)
19
+
20
+ Every ``fs`` function takes an already-confined absolute path — call
21
+ ``paths.confine(root, path)`` first to enforce a jail. The web fetch never
22
+ follows a redirect off http(s) and disables ``file://`` / ``ftp://``.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from hugpy_tools.data import read_data, safe_json_dumps, write_json
27
+ from hugpy_tools.fs import (
28
+ atomic_write,
29
+ detect_encoding,
30
+ edit_replace,
31
+ file_info,
32
+ list_dir,
33
+ read_lines,
34
+ read_text,
35
+ tree,
36
+ )
37
+ from hugpy_tools.hashkit import quick_hash, sha256_file, sha256_text
38
+ from hugpy_tools.paths import PathEscape, confine, file_parts, sanitize_filename
39
+ from hugpy_tools.text import (
40
+ chunk_by_lines,
41
+ chunk_by_tokens,
42
+ count_tokens,
43
+ unified_diff,
44
+ )
45
+ from hugpy_tools.timekit import epoch_to_iso, iso_to_epoch, now_epoch, now_iso
46
+ from hugpy_tools.web import assess_webpage, prescreen_webpage
47
+
48
+ __all__ = [
49
+ # paths
50
+ "PathEscape",
51
+ "confine",
52
+ "file_parts",
53
+ "sanitize_filename",
54
+ # fs
55
+ "atomic_write",
56
+ "detect_encoding",
57
+ "edit_replace",
58
+ "file_info",
59
+ "list_dir",
60
+ "read_lines",
61
+ "read_text",
62
+ "tree",
63
+ # hashkit
64
+ "quick_hash",
65
+ "sha256_file",
66
+ "sha256_text",
67
+ # text
68
+ "chunk_by_lines",
69
+ "chunk_by_tokens",
70
+ "count_tokens",
71
+ "unified_diff",
72
+ # data
73
+ "read_data",
74
+ "safe_json_dumps",
75
+ "write_json",
76
+ # timekit
77
+ "epoch_to_iso",
78
+ "iso_to_epoch",
79
+ "now_epoch",
80
+ "now_iso",
81
+ # web
82
+ "assess_webpage",
83
+ "prescreen_webpage",
84
+ ]
@@ -0,0 +1,74 @@
1
+ """Structured-data read/write: JSON always, TOML via stdlib ``tomllib``, YAML
2
+ only when PyYAML happens to be installed. Writing is JSON-only on purpose
3
+ (stdlib has no TOML/YAML serializer); the write path is atomic.
4
+
5
+ Every format degrades honestly: an unavailable parser raises a clear error
6
+ naming exactly what is missing, never a silent wrong-format guess. Ported in
7
+ spirit from abstract_utilities.json_utils (safe_dump / safe_read) but slimmed
8
+ to the two calls an agent needs.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import os
14
+
15
+ from . import fs
16
+
17
+ # Map a lowercase extension to a logical format.
18
+ _EXT_FORMAT = {
19
+ ".json": "json",
20
+ ".toml": "toml",
21
+ ".yaml": "yaml",
22
+ ".yml": "yaml",
23
+ }
24
+
25
+
26
+ def _format_for(path: str, explicit: str | None) -> str:
27
+ if explicit:
28
+ return explicit.lower()
29
+ return _EXT_FORMAT.get(os.path.splitext(path)[1].lower(), "json")
30
+
31
+
32
+ def read_data(path: str, fmt: str | None = None):
33
+ """Parse a JSON/TOML/YAML file to a Python object. Format is inferred from
34
+ the extension unless ``fmt`` is given. Raises ``RuntimeError`` when a
35
+ parser is unavailable, ``ValueError`` on an unknown format."""
36
+ fmt = _format_for(path, fmt)
37
+ if fmt == "json":
38
+ with open(path, "rb") as fh:
39
+ return json.loads(fh.read().decode("utf-8", errors="replace"))
40
+ if fmt == "toml":
41
+ try:
42
+ import tomllib # stdlib >= 3.11
43
+ except ModuleNotFoundError as exc:
44
+ raise RuntimeError(
45
+ "TOML reading needs Python 3.11+ (stdlib 'tomllib'); this "
46
+ "interpreter is older. Install the 'tomli' backport and read "
47
+ "it yourself, or upgrade Python.") from exc
48
+ with open(path, "rb") as fh:
49
+ return tomllib.load(fh)
50
+ if fmt == "yaml":
51
+ try:
52
+ import yaml # optional third-party
53
+ except ModuleNotFoundError as exc:
54
+ raise RuntimeError(
55
+ "YAML reading needs PyYAML, which is not installed "
56
+ "(hugpy_tools is stdlib-only). Install 'PyYAML' to enable "
57
+ "YAML.") from exc
58
+ with open(path, "rb") as fh:
59
+ return yaml.safe_load(fh)
60
+ raise ValueError("unknown data format %r (use json/toml/yaml)" % fmt)
61
+
62
+
63
+ def safe_json_dumps(obj, indent: int = 2, sort_keys: bool = False) -> str:
64
+ """``json.dumps`` that never explodes on a non-serialisable value: anything
65
+ the encoder cannot handle is stringified via ``default=str`` and non-ASCII
66
+ is preserved (``ensure_ascii=False``)."""
67
+ return json.dumps(obj, indent=indent, sort_keys=sort_keys,
68
+ ensure_ascii=False, default=str)
69
+
70
+
71
+ def write_json(path: str, obj, indent: int = 2, sort_keys: bool = False) -> dict:
72
+ """Serialise ``obj`` to JSON and write it atomically. Returns
73
+ ``{path, bytes, mode}`` (from :func:`fs.atomic_write`)."""
74
+ return fs.atomic_write(path, safe_json_dumps(obj, indent, sort_keys) + "\n")
@@ -0,0 +1,231 @@
1
+ """Safe file/directory operations (stdlib only).
2
+
3
+ Every function here takes an ALREADY-CONFINED absolute path — confinement is
4
+ the caller's job (see :func:`paths.confine`). That split keeps this module a
5
+ pure file-ops library with no policy of its own, which is exactly what a
6
+ standalone package wants.
7
+
8
+ Highlights ported from abstract_utilities:
9
+ * encoding detection on read (BOM sniff -> utf-8 -> cp1252 fallback)
10
+ * atomic write (temp file in the same dir + ``os.replace``)
11
+ * exact string edit with an occurrence-count contract
12
+ * directory listing + bounded recursive tree + rich file info
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import tempfile
18
+ from datetime import datetime, timezone
19
+
20
+ from . import hashkit
21
+
22
+ # BOM -> declared encoding
23
+ _BOMS = (
24
+ (b"\xef\xbb\xbf", "utf-8-sig"),
25
+ (b"\xff\xfe\x00\x00", "utf-32-le"),
26
+ (b"\x00\x00\xfe\xff", "utf-32-be"),
27
+ (b"\xff\xfe", "utf-16-le"),
28
+ (b"\xfe\xff", "utf-16-be"),
29
+ )
30
+
31
+
32
+ def detect_encoding(data: bytes) -> str:
33
+ """Best-effort encoding guess for ``data`` using only stdlib: honour a BOM,
34
+ then prefer strict utf-8, then fall back to cp1252 (a superset of latin-1
35
+ that decodes any byte). Never raises."""
36
+ for bom, enc in _BOMS:
37
+ if data.startswith(bom):
38
+ return enc
39
+ try:
40
+ data.decode("utf-8")
41
+ return "utf-8"
42
+ except UnicodeDecodeError:
43
+ return "cp1252"
44
+
45
+
46
+ def read_text(path: str, max_bytes: int | None = None) -> dict:
47
+ """Read a text file with encoding detection. Returns
48
+ ``{path, encoding, bytes, truncated, text}``."""
49
+ with open(path, "rb") as fh:
50
+ if max_bytes is None:
51
+ raw = fh.read()
52
+ truncated = False
53
+ else:
54
+ raw = fh.read(max_bytes + 1)
55
+ truncated = len(raw) > max_bytes
56
+ raw = raw[:max_bytes]
57
+ enc = detect_encoding(raw)
58
+ return {
59
+ "path": path,
60
+ "encoding": enc,
61
+ "bytes": len(raw),
62
+ "truncated": truncated,
63
+ "text": raw.decode(enc, errors="replace"),
64
+ }
65
+
66
+
67
+ def read_lines(path: str, start: int = 1, end: int | None = None,
68
+ number: bool = False) -> dict:
69
+ """Read a 1-based inclusive line range. ``end=None`` reads to EOF.
70
+ Returns ``{path, start, end, total_lines, returned, text}``."""
71
+ start = max(1, int(start))
72
+ with open(path, "rb") as fh:
73
+ raw = fh.read()
74
+ enc = detect_encoding(raw)
75
+ lines = raw.decode(enc, errors="replace").splitlines()
76
+ total = len(lines)
77
+ last = total if end is None else min(int(end), total)
78
+ chosen = lines[start - 1:last] if start <= total else []
79
+ if number:
80
+ width = len(str(last))
81
+ body = "\n".join("%*d\t%s" % (width, start + i, ln)
82
+ for i, ln in enumerate(chosen))
83
+ else:
84
+ body = "\n".join(chosen)
85
+ return {
86
+ "path": path,
87
+ "start": start,
88
+ "end": last,
89
+ "total_lines": total,
90
+ "returned": len(chosen),
91
+ "text": body,
92
+ }
93
+
94
+
95
+ def atomic_write(path: str, content: str, append: bool = False,
96
+ encoding: str = "utf-8") -> dict:
97
+ """Write ``content`` durably. Overwrite is atomic (temp file in the same
98
+ directory + ``os.replace``, so a reader never sees a half-written file).
99
+ Append opens the real file directly (append has no atomic swap). Returns
100
+ ``{path, bytes, mode}``."""
101
+ os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
102
+ data = content.encode(encoding, errors="replace")
103
+ if append:
104
+ with open(path, "ab") as fh:
105
+ fh.write(data)
106
+ else:
107
+ directory = os.path.dirname(path) or "."
108
+ fd, tmp = tempfile.mkstemp(dir=directory, prefix=".tmp-", suffix=".part")
109
+ try:
110
+ with os.fdopen(fd, "wb") as fh:
111
+ fh.write(data)
112
+ os.replace(tmp, path)
113
+ except BaseException:
114
+ try:
115
+ os.unlink(tmp)
116
+ except OSError:
117
+ pass
118
+ raise
119
+ return {"path": path, "bytes": len(data),
120
+ "mode": "append" if append else "overwrite"}
121
+
122
+
123
+ def edit_replace(path: str, old: str, new: str, count: int | None = None) -> dict:
124
+ """Exact string replacement in a text file. By default every occurrence is
125
+ replaced; pass ``count`` to bound it. Raises ``ValueError`` if ``old`` is
126
+ empty or not found. Returns ``{path, replaced, bytes}``; the write is
127
+ atomic."""
128
+ if old == "":
129
+ raise ValueError("old must be a non-empty string")
130
+ info = read_text(path)
131
+ text = info["text"]
132
+ occurrences = text.count(old)
133
+ if occurrences == 0:
134
+ raise ValueError("old string not found in %s" % path)
135
+ replaced = occurrences if count is None else min(occurrences, int(count))
136
+ text = text.replace(old, new, -1 if count is None else int(count))
137
+ out = atomic_write(path, text, encoding="utf-8"
138
+ if info["encoding"] in ("utf-8", "utf-8-sig", "cp1252")
139
+ else info["encoding"])
140
+ return {"path": path, "replaced": replaced, "bytes": out["bytes"]}
141
+
142
+
143
+ def _iso(ts: float) -> str:
144
+ return datetime.fromtimestamp(ts, tz=timezone.utc).isoformat()
145
+
146
+
147
+ def file_info(path: str, hash_files: bool = False) -> dict:
148
+ """Rich stat for one path. For a regular file with ``hash_files`` the
149
+ sha256 is included."""
150
+ st = os.lstat(path)
151
+ is_link = os.path.islink(path)
152
+ real = os.path.realpath(path)
153
+ is_dir = os.path.isdir(real)
154
+ info = {
155
+ "path": path,
156
+ "name": os.path.basename(path.rstrip(os.sep)) or path,
157
+ "type": "dir" if is_dir else ("symlink" if is_link and not os.path.exists(real) else "file"),
158
+ "is_dir": is_dir,
159
+ "is_symlink": is_link,
160
+ "size": st.st_size,
161
+ "mtime": _iso(st.st_mtime),
162
+ }
163
+ if hash_files and not is_dir and os.path.isfile(real):
164
+ info["sha256"] = hashkit.sha256_file(real)
165
+ return info
166
+
167
+
168
+ def list_dir(path: str, glob: str | None = None, files_only: bool = False,
169
+ dirs_only: bool = False, limit: int = 500) -> dict:
170
+ """List the immediate entries of a directory (sorted: dirs first, then
171
+ files, each alphabetical). Optional fnmatch ``glob`` on the entry name.
172
+ Returns ``{path, count, truncated, entries:[{name,type,size,mtime}]}``."""
173
+ import fnmatch
174
+ names = sorted(os.listdir(path))
175
+ entries = []
176
+ for name in names:
177
+ full = os.path.join(path, name)
178
+ is_dir = os.path.isdir(full)
179
+ if files_only and is_dir:
180
+ continue
181
+ if dirs_only and not is_dir:
182
+ continue
183
+ if glob and not fnmatch.fnmatch(name, glob):
184
+ continue
185
+ try:
186
+ st = os.lstat(full)
187
+ size, mtime = st.st_size, _iso(st.st_mtime)
188
+ except OSError:
189
+ size, mtime = None, None
190
+ entries.append({"name": name,
191
+ "type": "dir" if is_dir else "file",
192
+ "size": size, "mtime": mtime})
193
+ entries.sort(key=lambda e: (e["type"] != "dir", e["name"]))
194
+ truncated = len(entries) > limit
195
+ return {"path": path, "count": len(entries), "truncated": truncated,
196
+ "entries": entries[:limit]}
197
+
198
+
199
+ def tree(path: str, max_depth: int = 3, max_entries: int = 500,
200
+ show_files: bool = True) -> dict:
201
+ """Bounded recursive directory tree rendered as indented text. Skips
202
+ symlinked directories (never recurses out through a link). Returns
203
+ ``{path, entries, truncated, text}`` where ``entries`` is the count
204
+ emitted."""
205
+ lines: list[str] = []
206
+ state = {"n": 0, "truncated": False}
207
+
208
+ def walk(cur: str, depth: int, prefix: str) -> None:
209
+ if depth > max_depth or state["truncated"]:
210
+ return
211
+ try:
212
+ names = sorted(os.listdir(cur))
213
+ except OSError:
214
+ return
215
+ dirs = [n for n in names if os.path.isdir(os.path.join(cur, n))
216
+ and not os.path.islink(os.path.join(cur, n))]
217
+ files = [n for n in names if not os.path.isdir(os.path.join(cur, n))]
218
+ ordered = [(n, True) for n in dirs] + \
219
+ ([(n, False) for n in files] if show_files else [])
220
+ for name, is_dir in ordered:
221
+ if state["n"] >= max_entries:
222
+ state["truncated"] = True
223
+ return
224
+ state["n"] += 1
225
+ lines.append("%s%s%s" % (prefix, name, "/" if is_dir else ""))
226
+ if is_dir:
227
+ walk(os.path.join(cur, name), depth + 1, prefix + " ")
228
+
229
+ walk(path, 1, "")
230
+ return {"path": path, "entries": state["n"],
231
+ "truncated": state["truncated"], "text": "\n".join(lines)}