hugpy-tools 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hugpy_tools-0.2.2/LICENSE +41 -0
- hugpy_tools-0.2.2/PKG-INFO +93 -0
- hugpy_tools-0.2.2/README.md +59 -0
- hugpy_tools-0.2.2/pyproject.toml +65 -0
- hugpy_tools-0.2.2/setup.cfg +4 -0
- hugpy_tools-0.2.2/src/hugpy_tools/__init__.py +84 -0
- hugpy_tools-0.2.2/src/hugpy_tools/data.py +74 -0
- hugpy_tools-0.2.2/src/hugpy_tools/fs.py +231 -0
- hugpy_tools-0.2.2/src/hugpy_tools/hashkit.py +35 -0
- hugpy_tools-0.2.2/src/hugpy_tools/paths.py +80 -0
- hugpy_tools-0.2.2/src/hugpy_tools/py.typed +0 -0
- hugpy_tools-0.2.2/src/hugpy_tools/search/README.md +173 -0
- hugpy_tools-0.2.2/src/hugpy_tools/search/__init__.py +761 -0
- hugpy_tools-0.2.2/src/hugpy_tools/text.py +90 -0
- hugpy_tools-0.2.2/src/hugpy_tools/timekit.py +33 -0
- hugpy_tools-0.2.2/src/hugpy_tools/web.py +421 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/PKG-INFO +93 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/SOURCES.txt +26 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/dependency_links.txt +1 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/requires.txt +15 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/scm_file_list.json +23 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/scm_version.json +8 -0
- hugpy_tools-0.2.2/src/hugpy_tools.egg-info/top_level.txt +1 -0
- hugpy_tools-0.2.2/tests/test_fs_and_paths.py +196 -0
- hugpy_tools-0.2.2/tests/test_public_api.py +38 -0
- hugpy_tools-0.2.2/tests/test_search.py +247 -0
- hugpy_tools-0.2.2/tests/test_text_data_hash_time.py +104 -0
- hugpy_tools-0.2.2/tests/test_web.py +163 -0
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
hugpy — Source-Available License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 putkoff (hugpy.ai). All rights reserved.
|
|
4
|
+
|
|
5
|
+
Permission is granted, free of charge, to use this software ("hugpy") for
|
|
6
|
+
personal and non-commercial purposes, and for time-limited commercial
|
|
7
|
+
evaluation, subject to the following conditions:
|
|
8
|
+
|
|
9
|
+
1. Non-commercial use means use by an individual for personal purposes, or
|
|
10
|
+
use by a non-profit or educational institution for its own internal
|
|
11
|
+
purposes. Any use by, for, or on behalf of a for-profit business or in
|
|
12
|
+
connection with revenue-generating activity is commercial use — including
|
|
13
|
+
internal business use, use in producing goods or services, and use on
|
|
14
|
+
paid engagements.
|
|
15
|
+
|
|
16
|
+
2. Commercial use requires a commercial license from the copyright holder.
|
|
17
|
+
Exception: a business may evaluate the software internally for up to
|
|
18
|
+
thirty (30) days free of charge; continued use after that requires a
|
|
19
|
+
commercial license.
|
|
20
|
+
|
|
21
|
+
3. Redistribution of this software, in source or binary form, modified or
|
|
22
|
+
unmodified, is not permitted without prior written permission from the
|
|
23
|
+
copyright holder. Downloading the software from an official distribution
|
|
24
|
+
channel (PyPI, npm, hugpy.ai) is not redistribution.
|
|
25
|
+
|
|
26
|
+
4. Modification for personal use or internal evaluation is permitted;
|
|
27
|
+
distribution of modified versions is not.
|
|
28
|
+
|
|
29
|
+
5. This notice must be retained in all copies or substantial portions of
|
|
30
|
+
the software.
|
|
31
|
+
|
|
32
|
+
6. Any use outside these terms automatically terminates this license.
|
|
33
|
+
|
|
34
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
35
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
36
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
37
|
+
COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY ARISING
|
|
38
|
+
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
|
|
39
|
+
IN THE SOFTWARE.
|
|
40
|
+
|
|
41
|
+
For commercial licensing or redistribution permission: https://hugpy.ai
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hugpy-tools
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: Hugpy tools: a slim, stdlib-only capability suite for autonomous agents (safe file/path ops, text chunking and diffing, hashing, structured-data read/write, and urllib+html.parser webpage assessment)
|
|
5
|
+
Author-email: putkoff <support@hugpy.ai>
|
|
6
|
+
License-Expression: LicenseRef-Proprietary
|
|
7
|
+
Project-URL: Homepage, https://hugpy.ai
|
|
8
|
+
Project-URL: Documentation, https://github.com/hugpy/hugpy/blob/main/py/foundation/hugpy_tools/README.md
|
|
9
|
+
Project-URL: Repository, https://github.com/hugpy/hugpy
|
|
10
|
+
Project-URL: Source, https://github.com/hugpy/hugpy/tree/main/py/foundation/hugpy_tools
|
|
11
|
+
Project-URL: Issues, https://github.com/hugpy/hugpy/issues
|
|
12
|
+
Project-URL: Changelog, https://github.com/hugpy/hugpy/releases
|
|
13
|
+
Project-URL: Architecture, https://github.com/hugpy/hugpy/blob/main/PARTITION.md
|
|
14
|
+
Keywords: hugpy,agent,tools,filesystem,webpage,assessment
|
|
15
|
+
Classifier: Development Status :: 3 - Alpha
|
|
16
|
+
Classifier: Intended Audience :: Developers
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Provides-Extra: toml
|
|
25
|
+
Requires-Dist: tomli; python_version < "3.11" and extra == "toml"
|
|
26
|
+
Provides-Extra: yaml
|
|
27
|
+
Requires-Dist: PyYAML; extra == "yaml"
|
|
28
|
+
Provides-Extra: render
|
|
29
|
+
Requires-Dist: playwright; extra == "render"
|
|
30
|
+
Provides-Extra: test
|
|
31
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
32
|
+
Requires-Dist: pytest-timeout; extra == "test"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# hugpy-tools
|
|
36
|
+
|
|
37
|
+
`hugpy_tools` — a slim, **stdlib-only** capability suite for autonomous agents,
|
|
38
|
+
extracted from `abstract_utilities` / `abstract_webtools` and rebuilt with no
|
|
39
|
+
third-party runtime dependency. Foundation layer: it imports no other `hugpy_*`
|
|
40
|
+
package, so it can be lifted out as its own distribution unchanged.
|
|
41
|
+
|
|
42
|
+
Ownership and allowed dependencies are declared in `py/partition.toml`; see
|
|
43
|
+
`PARTITION.md` at the workspace root.
|
|
44
|
+
|
|
45
|
+
Allowed Python dependencies inside the ecosystem: **none** (stdlib only).
|
|
46
|
+
|
|
47
|
+
## Built-in tools
|
|
48
|
+
|
|
49
|
+
Every `fs` function takes an **already-confined** absolute path — call
|
|
50
|
+
`paths.confine(root, path)` first to enforce a workspace jail (rejects `..`,
|
|
51
|
+
absolute-outside, and symlink escapes in one check).
|
|
52
|
+
|
|
53
|
+
| Module | Function | What it does |
|
|
54
|
+
| --- | --- | --- |
|
|
55
|
+
| `paths` | `confine(root, path, for_write=False)` | Resolve a path under `root`; raise `PathEscape` on any escape. |
|
|
56
|
+
| `paths` | `file_parts(path)` | Decompose into dir/base/name/ext + two enclosing dir levels. |
|
|
57
|
+
| `paths` | `sanitize_filename(name)` | Reduce to one safe path component, length-bounded. |
|
|
58
|
+
| `fs` | `read_text(path, max_bytes=None)` | Read with encoding detection (BOM → utf-8 → cp1252). |
|
|
59
|
+
| `fs` | `read_lines(path, start, end, number=False)` | 1-based inclusive line-range read. |
|
|
60
|
+
| `fs` | `atomic_write(path, content, append=False)` | Durable write (temp file + `os.replace`); append supported. |
|
|
61
|
+
| `fs` | `edit_replace(path, old, new, count=None)` | Exact string replace with an occurrence-count contract; atomic. |
|
|
62
|
+
| `fs` | `file_info(path, hash_files=False)` | Size, mtime, type, symlink flag, optional sha256. |
|
|
63
|
+
| `fs` | `list_dir(path, glob=None, ...)` | Immediate entries (dirs first), optional fnmatch. |
|
|
64
|
+
| `fs` | `tree(path, max_depth=3, max_entries=500)` | Bounded recursive tree text (never follows dir symlinks). |
|
|
65
|
+
| `hashkit` | `sha256_text` / `sha256_file` / `quick_hash` | Content hashing (files streamed in 1 MiB windows). |
|
|
66
|
+
| `text` | `count_tokens(text)` | Dependency-free approximate BPE token count. |
|
|
67
|
+
| `text` | `chunk_by_lines` / `chunk_by_tokens` | Chunk on line or token budgets (paragraph-aware). |
|
|
68
|
+
| `text` | `unified_diff(before, after)` | Unified diff between two texts. |
|
|
69
|
+
| `data` | `read_data(path, fmt=None)` | Parse JSON always, TOML via stdlib `tomllib`, YAML if PyYAML present. |
|
|
70
|
+
| `data` | `write_json(path, obj)` / `safe_json_dumps(obj)` | Atomic JSON write / non-exploding dumps. |
|
|
71
|
+
| `timekit` | `now_iso` / `now_epoch` / `epoch_to_iso` / `iso_to_epoch` | UTC time conversions. |
|
|
72
|
+
| `web` | `assess_webpage(url, ...)` | assessManager-parity structured page assessment over urllib + `html.parser`. |
|
|
73
|
+
| `web` | `prescreen_webpage(url, ...)` | Cheap title + description + lede relevance pre-screen. |
|
|
74
|
+
|
|
75
|
+
### Web assessment
|
|
76
|
+
|
|
77
|
+
`assess_webpage` returns the same dict shape as
|
|
78
|
+
`abstract_webtools.assessManager.assess_webpage`
|
|
79
|
+
(`{url, title, description, text, metadata, jsonld, links, truncated, render}`)
|
|
80
|
+
plus additive keys `final_url`, `status`, `canonical`, `lang`, `headings`, and
|
|
81
|
+
`error`. The fetch is http(s)-only, refuses redirects to other schemes,
|
|
82
|
+
disables `file://` / `ftp://`, caps the body, times out, handles gzip/deflate,
|
|
83
|
+
and detects charset (Content-Type → BOM → `<meta charset>` → utf-8 → cp1252).
|
|
84
|
+
|
|
85
|
+
JS rendering is opt-in and lazy: `force_render=True` (or an automatic fall-back
|
|
86
|
+
when the cheap fetch yields almost no text) drives a headless browser **only if
|
|
87
|
+
Playwright or Selenium is installed**, and raises a clear "install X" error
|
|
88
|
+
otherwise. The default path never needs a browser.
|
|
89
|
+
|
|
90
|
+
## Layer
|
|
91
|
+
|
|
92
|
+
Foundation. `hugpy_tools` imports no other `hugpy_*` package. The `search`
|
|
93
|
+
subpackage is owned by a separate work-stream.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# hugpy-tools
|
|
2
|
+
|
|
3
|
+
`hugpy_tools` — a slim, **stdlib-only** capability suite for autonomous agents,
|
|
4
|
+
extracted from `abstract_utilities` / `abstract_webtools` and rebuilt with no
|
|
5
|
+
third-party runtime dependency. Foundation layer: it imports no other `hugpy_*`
|
|
6
|
+
package, so it can be lifted out as its own distribution unchanged.
|
|
7
|
+
|
|
8
|
+
Ownership and allowed dependencies are declared in `py/partition.toml`; see
|
|
9
|
+
`PARTITION.md` at the workspace root.
|
|
10
|
+
|
|
11
|
+
Allowed Python dependencies inside the ecosystem: **none** (stdlib only).
|
|
12
|
+
|
|
13
|
+
## Built-in tools
|
|
14
|
+
|
|
15
|
+
Every `fs` function takes an **already-confined** absolute path — call
|
|
16
|
+
`paths.confine(root, path)` first to enforce a workspace jail (rejects `..`,
|
|
17
|
+
absolute-outside, and symlink escapes in one check).
|
|
18
|
+
|
|
19
|
+
| Module | Function | What it does |
|
|
20
|
+
| --- | --- | --- |
|
|
21
|
+
| `paths` | `confine(root, path, for_write=False)` | Resolve a path under `root`; raise `PathEscape` on any escape. |
|
|
22
|
+
| `paths` | `file_parts(path)` | Decompose into dir/base/name/ext + two enclosing dir levels. |
|
|
23
|
+
| `paths` | `sanitize_filename(name)` | Reduce to one safe path component, length-bounded. |
|
|
24
|
+
| `fs` | `read_text(path, max_bytes=None)` | Read with encoding detection (BOM → utf-8 → cp1252). |
|
|
25
|
+
| `fs` | `read_lines(path, start, end, number=False)` | 1-based inclusive line-range read. |
|
|
26
|
+
| `fs` | `atomic_write(path, content, append=False)` | Durable write (temp file + `os.replace`); append supported. |
|
|
27
|
+
| `fs` | `edit_replace(path, old, new, count=None)` | Exact string replace with an occurrence-count contract; atomic. |
|
|
28
|
+
| `fs` | `file_info(path, hash_files=False)` | Size, mtime, type, symlink flag, optional sha256. |
|
|
29
|
+
| `fs` | `list_dir(path, glob=None, ...)` | Immediate entries (dirs first), optional fnmatch. |
|
|
30
|
+
| `fs` | `tree(path, max_depth=3, max_entries=500)` | Bounded recursive tree text (never follows dir symlinks). |
|
|
31
|
+
| `hashkit` | `sha256_text` / `sha256_file` / `quick_hash` | Content hashing (files streamed in 1 MiB windows). |
|
|
32
|
+
| `text` | `count_tokens(text)` | Dependency-free approximate BPE token count. |
|
|
33
|
+
| `text` | `chunk_by_lines` / `chunk_by_tokens` | Chunk on line or token budgets (paragraph-aware). |
|
|
34
|
+
| `text` | `unified_diff(before, after)` | Unified diff between two texts. |
|
|
35
|
+
| `data` | `read_data(path, fmt=None)` | Parse JSON always, TOML via stdlib `tomllib`, YAML if PyYAML present. |
|
|
36
|
+
| `data` | `write_json(path, obj)` / `safe_json_dumps(obj)` | Atomic JSON write / non-exploding dumps. |
|
|
37
|
+
| `timekit` | `now_iso` / `now_epoch` / `epoch_to_iso` / `iso_to_epoch` | UTC time conversions. |
|
|
38
|
+
| `web` | `assess_webpage(url, ...)` | assessManager-parity structured page assessment over urllib + `html.parser`. |
|
|
39
|
+
| `web` | `prescreen_webpage(url, ...)` | Cheap title + description + lede relevance pre-screen. |
|
|
40
|
+
|
|
41
|
+
### Web assessment
|
|
42
|
+
|
|
43
|
+
`assess_webpage` returns the same dict shape as
|
|
44
|
+
`abstract_webtools.assessManager.assess_webpage`
|
|
45
|
+
(`{url, title, description, text, metadata, jsonld, links, truncated, render}`)
|
|
46
|
+
plus additive keys `final_url`, `status`, `canonical`, `lang`, `headings`, and
|
|
47
|
+
`error`. The fetch is http(s)-only, refuses redirects to other schemes,
|
|
48
|
+
disables `file://` / `ftp://`, caps the body, times out, handles gzip/deflate,
|
|
49
|
+
and detects charset (Content-Type → BOM → `<meta charset>` → utf-8 → cp1252).
|
|
50
|
+
|
|
51
|
+
JS rendering is opt-in and lazy: `force_render=True` (or an automatic fall-back
|
|
52
|
+
when the cheap fetch yields almost no text) drives a headless browser **only if
|
|
53
|
+
Playwright or Selenium is installed**, and raises a clear "install X" error
|
|
54
|
+
otherwise. The default path never needs a browser.
|
|
55
|
+
|
|
56
|
+
## Layer
|
|
57
|
+
|
|
58
|
+
Foundation. `hugpy_tools` imports no other `hugpy_*` package. The `search`
|
|
59
|
+
subpackage is owned by a separate work-stream.
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "setuptools-scm>=8"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hugpy-tools"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Hugpy tools: a slim, stdlib-only capability suite for autonomous agents (safe file/path ops, text chunking and diffing, hashing, structured-data read/write, and urllib+html.parser webpage assessment)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "LicenseRef-Proprietary"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "putkoff", email = "support@hugpy.ai" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"hugpy",
|
|
16
|
+
"agent",
|
|
17
|
+
"tools",
|
|
18
|
+
"filesystem",
|
|
19
|
+
"webpage",
|
|
20
|
+
"assessment",
|
|
21
|
+
]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 3 - Alpha",
|
|
24
|
+
"Intended Audience :: Developers",
|
|
25
|
+
"Operating System :: OS Independent",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
28
|
+
"Topic :: Software Development :: Libraries",
|
|
29
|
+
]
|
|
30
|
+
# Deliberately dependency-free: stdlib only, so a machine with just Python can
|
|
31
|
+
# use it and no other hugpy_* package is imported (foundation layer).
|
|
32
|
+
dependencies = []
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://hugpy.ai"
|
|
36
|
+
Documentation = "https://github.com/hugpy/hugpy/blob/main/py/foundation/hugpy_tools/README.md"
|
|
37
|
+
Repository = "https://github.com/hugpy/hugpy"
|
|
38
|
+
Source = "https://github.com/hugpy/hugpy/tree/main/py/foundation/hugpy_tools"
|
|
39
|
+
Issues = "https://github.com/hugpy/hugpy/issues"
|
|
40
|
+
Changelog = "https://github.com/hugpy/hugpy/releases"
|
|
41
|
+
Architecture = "https://github.com/hugpy/hugpy/blob/main/PARTITION.md"
|
|
42
|
+
|
|
43
|
+
[project.optional-dependencies]
|
|
44
|
+
# TOML reading uses stdlib tomllib on 3.11+; on 3.10 install the backport.
|
|
45
|
+
toml = ["tomli; python_version < '3.11'"]
|
|
46
|
+
# YAML reading is opt-in; the default never imports it.
|
|
47
|
+
yaml = ["PyYAML"]
|
|
48
|
+
# assess_webpage(force_render=True) can drive a headless browser if present.
|
|
49
|
+
render = ["playwright"]
|
|
50
|
+
test = ["pytest>=8", "pytest-timeout"]
|
|
51
|
+
|
|
52
|
+
[tool.setuptools.packages.find]
|
|
53
|
+
where = ["src"]
|
|
54
|
+
|
|
55
|
+
[tool.setuptools.package-data]
|
|
56
|
+
hugpy_tools = ["py.typed", "**/*.json", "**/*.md", "**/*.txt"]
|
|
57
|
+
|
|
58
|
+
# ---------------------------------------------------------------------------
|
|
59
|
+
# Lockstep workspace version (2026-09-22): every in-tree hugpy-* distribution
|
|
60
|
+
# takes ONE version from the workspace git tag (vX.Y.Z at the repo root). A
|
|
61
|
+
# checkout without git metadata builds as 0.0.0+unknown, which central refuses
|
|
62
|
+
# to advertise to workers. (Identical policy to every sibling package.)
|
|
63
|
+
[tool.setuptools_scm]
|
|
64
|
+
root = "../../.."
|
|
65
|
+
fallback_version = "0.0.0+unknown"
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""hugpy-tools: a slim, stdlib-only capability suite for autonomous agents.
|
|
2
|
+
|
|
3
|
+
Extracted from the sprawl of ``abstract_utilities`` / ``abstract_webtools`` and
|
|
4
|
+
rebuilt with no third-party runtime dependency, so a machine with only Python
|
|
5
|
+
can use it. It imports NO other ``hugpy_*`` package (foundation layer).
|
|
6
|
+
|
|
7
|
+
Modules:
|
|
8
|
+
|
|
9
|
+
paths path confinement (jail) + path-part helpers + filename sanitising
|
|
10
|
+
fs safe file/dir ops: encoding-detecting read, line-range read,
|
|
11
|
+
atomic write, exact-string edit, list/tree, rich file info
|
|
12
|
+
hashkit sha256 (text/file, streamed) + a cheap size+head quick hash
|
|
13
|
+
text approximate token counting, token/line chunking, unified diffs
|
|
14
|
+
data JSON/TOML/YAML read + atomic JSON write (safe dumping)
|
|
15
|
+
timekit UTC ISO / epoch conversions
|
|
16
|
+
web assess_webpage / prescreen_webpage — assessManager-parity webpage
|
|
17
|
+
assessment over urllib + html.parser (opt-in JS render)
|
|
18
|
+
search reserved namespace (owned elsewhere; not implemented here)
|
|
19
|
+
|
|
20
|
+
Every ``fs`` function takes an already-confined absolute path — call
|
|
21
|
+
``paths.confine(root, path)`` first to enforce a jail. The web fetch never
|
|
22
|
+
follows a redirect off http(s) and disables ``file://`` / ``ftp://``.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from hugpy_tools.data import read_data, safe_json_dumps, write_json
|
|
27
|
+
from hugpy_tools.fs import (
|
|
28
|
+
atomic_write,
|
|
29
|
+
detect_encoding,
|
|
30
|
+
edit_replace,
|
|
31
|
+
file_info,
|
|
32
|
+
list_dir,
|
|
33
|
+
read_lines,
|
|
34
|
+
read_text,
|
|
35
|
+
tree,
|
|
36
|
+
)
|
|
37
|
+
from hugpy_tools.hashkit import quick_hash, sha256_file, sha256_text
|
|
38
|
+
from hugpy_tools.paths import PathEscape, confine, file_parts, sanitize_filename
|
|
39
|
+
from hugpy_tools.text import (
|
|
40
|
+
chunk_by_lines,
|
|
41
|
+
chunk_by_tokens,
|
|
42
|
+
count_tokens,
|
|
43
|
+
unified_diff,
|
|
44
|
+
)
|
|
45
|
+
from hugpy_tools.timekit import epoch_to_iso, iso_to_epoch, now_epoch, now_iso
|
|
46
|
+
from hugpy_tools.web import assess_webpage, prescreen_webpage
|
|
47
|
+
|
|
48
|
+
__all__ = [
|
|
49
|
+
# paths
|
|
50
|
+
"PathEscape",
|
|
51
|
+
"confine",
|
|
52
|
+
"file_parts",
|
|
53
|
+
"sanitize_filename",
|
|
54
|
+
# fs
|
|
55
|
+
"atomic_write",
|
|
56
|
+
"detect_encoding",
|
|
57
|
+
"edit_replace",
|
|
58
|
+
"file_info",
|
|
59
|
+
"list_dir",
|
|
60
|
+
"read_lines",
|
|
61
|
+
"read_text",
|
|
62
|
+
"tree",
|
|
63
|
+
# hashkit
|
|
64
|
+
"quick_hash",
|
|
65
|
+
"sha256_file",
|
|
66
|
+
"sha256_text",
|
|
67
|
+
# text
|
|
68
|
+
"chunk_by_lines",
|
|
69
|
+
"chunk_by_tokens",
|
|
70
|
+
"count_tokens",
|
|
71
|
+
"unified_diff",
|
|
72
|
+
# data
|
|
73
|
+
"read_data",
|
|
74
|
+
"safe_json_dumps",
|
|
75
|
+
"write_json",
|
|
76
|
+
# timekit
|
|
77
|
+
"epoch_to_iso",
|
|
78
|
+
"iso_to_epoch",
|
|
79
|
+
"now_epoch",
|
|
80
|
+
"now_iso",
|
|
81
|
+
# web
|
|
82
|
+
"assess_webpage",
|
|
83
|
+
"prescreen_webpage",
|
|
84
|
+
]
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Structured-data read/write: JSON always, TOML via stdlib ``tomllib``, YAML
|
|
2
|
+
only when PyYAML happens to be installed. Writing is JSON-only on purpose
|
|
3
|
+
(stdlib has no TOML/YAML serializer); the write path is atomic.
|
|
4
|
+
|
|
5
|
+
Every format degrades honestly: an unavailable parser raises a clear error
|
|
6
|
+
naming exactly what is missing, never a silent wrong-format guess. Ported in
|
|
7
|
+
spirit from abstract_utilities.json_utils (safe_dump / safe_read) but slimmed
|
|
8
|
+
to the two calls an agent needs.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
|
|
15
|
+
from . import fs
|
|
16
|
+
|
|
17
|
+
# Map a lowercase extension to a logical format.
|
|
18
|
+
_EXT_FORMAT = {
|
|
19
|
+
".json": "json",
|
|
20
|
+
".toml": "toml",
|
|
21
|
+
".yaml": "yaml",
|
|
22
|
+
".yml": "yaml",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _format_for(path: str, explicit: str | None) -> str:
|
|
27
|
+
if explicit:
|
|
28
|
+
return explicit.lower()
|
|
29
|
+
return _EXT_FORMAT.get(os.path.splitext(path)[1].lower(), "json")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_data(path: str, fmt: str | None = None):
|
|
33
|
+
"""Parse a JSON/TOML/YAML file to a Python object. Format is inferred from
|
|
34
|
+
the extension unless ``fmt`` is given. Raises ``RuntimeError`` when a
|
|
35
|
+
parser is unavailable, ``ValueError`` on an unknown format."""
|
|
36
|
+
fmt = _format_for(path, fmt)
|
|
37
|
+
if fmt == "json":
|
|
38
|
+
with open(path, "rb") as fh:
|
|
39
|
+
return json.loads(fh.read().decode("utf-8", errors="replace"))
|
|
40
|
+
if fmt == "toml":
|
|
41
|
+
try:
|
|
42
|
+
import tomllib # stdlib >= 3.11
|
|
43
|
+
except ModuleNotFoundError as exc:
|
|
44
|
+
raise RuntimeError(
|
|
45
|
+
"TOML reading needs Python 3.11+ (stdlib 'tomllib'); this "
|
|
46
|
+
"interpreter is older. Install the 'tomli' backport and read "
|
|
47
|
+
"it yourself, or upgrade Python.") from exc
|
|
48
|
+
with open(path, "rb") as fh:
|
|
49
|
+
return tomllib.load(fh)
|
|
50
|
+
if fmt == "yaml":
|
|
51
|
+
try:
|
|
52
|
+
import yaml # optional third-party
|
|
53
|
+
except ModuleNotFoundError as exc:
|
|
54
|
+
raise RuntimeError(
|
|
55
|
+
"YAML reading needs PyYAML, which is not installed "
|
|
56
|
+
"(hugpy_tools is stdlib-only). Install 'PyYAML' to enable "
|
|
57
|
+
"YAML.") from exc
|
|
58
|
+
with open(path, "rb") as fh:
|
|
59
|
+
return yaml.safe_load(fh)
|
|
60
|
+
raise ValueError("unknown data format %r (use json/toml/yaml)" % fmt)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def safe_json_dumps(obj, indent: int = 2, sort_keys: bool = False) -> str:
|
|
64
|
+
"""``json.dumps`` that never explodes on a non-serialisable value: anything
|
|
65
|
+
the encoder cannot handle is stringified via ``default=str`` and non-ASCII
|
|
66
|
+
is preserved (``ensure_ascii=False``)."""
|
|
67
|
+
return json.dumps(obj, indent=indent, sort_keys=sort_keys,
|
|
68
|
+
ensure_ascii=False, default=str)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def write_json(path: str, obj, indent: int = 2, sort_keys: bool = False) -> dict:
|
|
72
|
+
"""Serialise ``obj`` to JSON and write it atomically. Returns
|
|
73
|
+
``{path, bytes, mode}`` (from :func:`fs.atomic_write`)."""
|
|
74
|
+
return fs.atomic_write(path, safe_json_dumps(obj, indent, sort_keys) + "\n")
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Safe file/directory operations (stdlib only).
|
|
2
|
+
|
|
3
|
+
Every function here takes an ALREADY-CONFINED absolute path — confinement is
|
|
4
|
+
the caller's job (see :func:`paths.confine`). That split keeps this module a
|
|
5
|
+
pure file-ops library with no policy of its own, which is exactly what a
|
|
6
|
+
standalone package wants.
|
|
7
|
+
|
|
8
|
+
Highlights ported from abstract_utilities:
|
|
9
|
+
* encoding detection on read (BOM sniff -> utf-8 -> cp1252 fallback)
|
|
10
|
+
* atomic write (temp file in the same dir + ``os.replace``)
|
|
11
|
+
* exact string edit with an occurrence-count contract
|
|
12
|
+
* directory listing + bounded recursive tree + rich file info
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import tempfile
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
|
|
20
|
+
from . import hashkit
|
|
21
|
+
|
|
22
|
+
# BOM -> declared encoding
|
|
23
|
+
_BOMS = (
|
|
24
|
+
(b"\xef\xbb\xbf", "utf-8-sig"),
|
|
25
|
+
(b"\xff\xfe\x00\x00", "utf-32-le"),
|
|
26
|
+
(b"\x00\x00\xfe\xff", "utf-32-be"),
|
|
27
|
+
(b"\xff\xfe", "utf-16-le"),
|
|
28
|
+
(b"\xfe\xff", "utf-16-be"),
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def detect_encoding(data: bytes) -> str:
|
|
33
|
+
"""Best-effort encoding guess for ``data`` using only stdlib: honour a BOM,
|
|
34
|
+
then prefer strict utf-8, then fall back to cp1252 (a superset of latin-1
|
|
35
|
+
that decodes any byte). Never raises."""
|
|
36
|
+
for bom, enc in _BOMS:
|
|
37
|
+
if data.startswith(bom):
|
|
38
|
+
return enc
|
|
39
|
+
try:
|
|
40
|
+
data.decode("utf-8")
|
|
41
|
+
return "utf-8"
|
|
42
|
+
except UnicodeDecodeError:
|
|
43
|
+
return "cp1252"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def read_text(path: str, max_bytes: int | None = None) -> dict:
|
|
47
|
+
"""Read a text file with encoding detection. Returns
|
|
48
|
+
``{path, encoding, bytes, truncated, text}``."""
|
|
49
|
+
with open(path, "rb") as fh:
|
|
50
|
+
if max_bytes is None:
|
|
51
|
+
raw = fh.read()
|
|
52
|
+
truncated = False
|
|
53
|
+
else:
|
|
54
|
+
raw = fh.read(max_bytes + 1)
|
|
55
|
+
truncated = len(raw) > max_bytes
|
|
56
|
+
raw = raw[:max_bytes]
|
|
57
|
+
enc = detect_encoding(raw)
|
|
58
|
+
return {
|
|
59
|
+
"path": path,
|
|
60
|
+
"encoding": enc,
|
|
61
|
+
"bytes": len(raw),
|
|
62
|
+
"truncated": truncated,
|
|
63
|
+
"text": raw.decode(enc, errors="replace"),
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def read_lines(path: str, start: int = 1, end: int | None = None,
|
|
68
|
+
number: bool = False) -> dict:
|
|
69
|
+
"""Read a 1-based inclusive line range. ``end=None`` reads to EOF.
|
|
70
|
+
Returns ``{path, start, end, total_lines, returned, text}``."""
|
|
71
|
+
start = max(1, int(start))
|
|
72
|
+
with open(path, "rb") as fh:
|
|
73
|
+
raw = fh.read()
|
|
74
|
+
enc = detect_encoding(raw)
|
|
75
|
+
lines = raw.decode(enc, errors="replace").splitlines()
|
|
76
|
+
total = len(lines)
|
|
77
|
+
last = total if end is None else min(int(end), total)
|
|
78
|
+
chosen = lines[start - 1:last] if start <= total else []
|
|
79
|
+
if number:
|
|
80
|
+
width = len(str(last))
|
|
81
|
+
body = "\n".join("%*d\t%s" % (width, start + i, ln)
|
|
82
|
+
for i, ln in enumerate(chosen))
|
|
83
|
+
else:
|
|
84
|
+
body = "\n".join(chosen)
|
|
85
|
+
return {
|
|
86
|
+
"path": path,
|
|
87
|
+
"start": start,
|
|
88
|
+
"end": last,
|
|
89
|
+
"total_lines": total,
|
|
90
|
+
"returned": len(chosen),
|
|
91
|
+
"text": body,
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def atomic_write(path: str, content: str, append: bool = False,
|
|
96
|
+
encoding: str = "utf-8") -> dict:
|
|
97
|
+
"""Write ``content`` durably. Overwrite is atomic (temp file in the same
|
|
98
|
+
directory + ``os.replace``, so a reader never sees a half-written file).
|
|
99
|
+
Append opens the real file directly (append has no atomic swap). Returns
|
|
100
|
+
``{path, bytes, mode}``."""
|
|
101
|
+
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
|
102
|
+
data = content.encode(encoding, errors="replace")
|
|
103
|
+
if append:
|
|
104
|
+
with open(path, "ab") as fh:
|
|
105
|
+
fh.write(data)
|
|
106
|
+
else:
|
|
107
|
+
directory = os.path.dirname(path) or "."
|
|
108
|
+
fd, tmp = tempfile.mkstemp(dir=directory, prefix=".tmp-", suffix=".part")
|
|
109
|
+
try:
|
|
110
|
+
with os.fdopen(fd, "wb") as fh:
|
|
111
|
+
fh.write(data)
|
|
112
|
+
os.replace(tmp, path)
|
|
113
|
+
except BaseException:
|
|
114
|
+
try:
|
|
115
|
+
os.unlink(tmp)
|
|
116
|
+
except OSError:
|
|
117
|
+
pass
|
|
118
|
+
raise
|
|
119
|
+
return {"path": path, "bytes": len(data),
|
|
120
|
+
"mode": "append" if append else "overwrite"}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def edit_replace(path: str, old: str, new: str, count: int | None = None) -> dict:
|
|
124
|
+
"""Exact string replacement in a text file. By default every occurrence is
|
|
125
|
+
replaced; pass ``count`` to bound it. Raises ``ValueError`` if ``old`` is
|
|
126
|
+
empty or not found. Returns ``{path, replaced, bytes}``; the write is
|
|
127
|
+
atomic."""
|
|
128
|
+
if old == "":
|
|
129
|
+
raise ValueError("old must be a non-empty string")
|
|
130
|
+
info = read_text(path)
|
|
131
|
+
text = info["text"]
|
|
132
|
+
occurrences = text.count(old)
|
|
133
|
+
if occurrences == 0:
|
|
134
|
+
raise ValueError("old string not found in %s" % path)
|
|
135
|
+
replaced = occurrences if count is None else min(occurrences, int(count))
|
|
136
|
+
text = text.replace(old, new, -1 if count is None else int(count))
|
|
137
|
+
out = atomic_write(path, text, encoding="utf-8"
|
|
138
|
+
if info["encoding"] in ("utf-8", "utf-8-sig", "cp1252")
|
|
139
|
+
else info["encoding"])
|
|
140
|
+
return {"path": path, "replaced": replaced, "bytes": out["bytes"]}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _iso(ts: float) -> str:
|
|
144
|
+
return datetime.fromtimestamp(ts, tz=timezone.utc).isoformat()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def file_info(path: str, hash_files: bool = False) -> dict:
|
|
148
|
+
"""Rich stat for one path. For a regular file with ``hash_files`` the
|
|
149
|
+
sha256 is included."""
|
|
150
|
+
st = os.lstat(path)
|
|
151
|
+
is_link = os.path.islink(path)
|
|
152
|
+
real = os.path.realpath(path)
|
|
153
|
+
is_dir = os.path.isdir(real)
|
|
154
|
+
info = {
|
|
155
|
+
"path": path,
|
|
156
|
+
"name": os.path.basename(path.rstrip(os.sep)) or path,
|
|
157
|
+
"type": "dir" if is_dir else ("symlink" if is_link and not os.path.exists(real) else "file"),
|
|
158
|
+
"is_dir": is_dir,
|
|
159
|
+
"is_symlink": is_link,
|
|
160
|
+
"size": st.st_size,
|
|
161
|
+
"mtime": _iso(st.st_mtime),
|
|
162
|
+
}
|
|
163
|
+
if hash_files and not is_dir and os.path.isfile(real):
|
|
164
|
+
info["sha256"] = hashkit.sha256_file(real)
|
|
165
|
+
return info
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def list_dir(path: str, glob: str | None = None, files_only: bool = False,
|
|
169
|
+
dirs_only: bool = False, limit: int = 500) -> dict:
|
|
170
|
+
"""List the immediate entries of a directory (sorted: dirs first, then
|
|
171
|
+
files, each alphabetical). Optional fnmatch ``glob`` on the entry name.
|
|
172
|
+
Returns ``{path, count, truncated, entries:[{name,type,size,mtime}]}``."""
|
|
173
|
+
import fnmatch
|
|
174
|
+
names = sorted(os.listdir(path))
|
|
175
|
+
entries = []
|
|
176
|
+
for name in names:
|
|
177
|
+
full = os.path.join(path, name)
|
|
178
|
+
is_dir = os.path.isdir(full)
|
|
179
|
+
if files_only and is_dir:
|
|
180
|
+
continue
|
|
181
|
+
if dirs_only and not is_dir:
|
|
182
|
+
continue
|
|
183
|
+
if glob and not fnmatch.fnmatch(name, glob):
|
|
184
|
+
continue
|
|
185
|
+
try:
|
|
186
|
+
st = os.lstat(full)
|
|
187
|
+
size, mtime = st.st_size, _iso(st.st_mtime)
|
|
188
|
+
except OSError:
|
|
189
|
+
size, mtime = None, None
|
|
190
|
+
entries.append({"name": name,
|
|
191
|
+
"type": "dir" if is_dir else "file",
|
|
192
|
+
"size": size, "mtime": mtime})
|
|
193
|
+
entries.sort(key=lambda e: (e["type"] != "dir", e["name"]))
|
|
194
|
+
truncated = len(entries) > limit
|
|
195
|
+
return {"path": path, "count": len(entries), "truncated": truncated,
|
|
196
|
+
"entries": entries[:limit]}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def tree(path: str, max_depth: int = 3, max_entries: int = 500,
|
|
200
|
+
show_files: bool = True) -> dict:
|
|
201
|
+
"""Bounded recursive directory tree rendered as indented text. Skips
|
|
202
|
+
symlinked directories (never recurses out through a link). Returns
|
|
203
|
+
``{path, entries, truncated, text}`` where ``entries`` is the count
|
|
204
|
+
emitted."""
|
|
205
|
+
lines: list[str] = []
|
|
206
|
+
state = {"n": 0, "truncated": False}
|
|
207
|
+
|
|
208
|
+
def walk(cur: str, depth: int, prefix: str) -> None:
|
|
209
|
+
if depth > max_depth or state["truncated"]:
|
|
210
|
+
return
|
|
211
|
+
try:
|
|
212
|
+
names = sorted(os.listdir(cur))
|
|
213
|
+
except OSError:
|
|
214
|
+
return
|
|
215
|
+
dirs = [n for n in names if os.path.isdir(os.path.join(cur, n))
|
|
216
|
+
and not os.path.islink(os.path.join(cur, n))]
|
|
217
|
+
files = [n for n in names if not os.path.isdir(os.path.join(cur, n))]
|
|
218
|
+
ordered = [(n, True) for n in dirs] + \
|
|
219
|
+
([(n, False) for n in files] if show_files else [])
|
|
220
|
+
for name, is_dir in ordered:
|
|
221
|
+
if state["n"] >= max_entries:
|
|
222
|
+
state["truncated"] = True
|
|
223
|
+
return
|
|
224
|
+
state["n"] += 1
|
|
225
|
+
lines.append("%s%s%s" % (prefix, name, "/" if is_dir else ""))
|
|
226
|
+
if is_dir:
|
|
227
|
+
walk(os.path.join(cur, name), depth + 1, prefix + " ")
|
|
228
|
+
|
|
229
|
+
walk(path, 1, "")
|
|
230
|
+
return {"path": path, "entries": state["n"],
|
|
231
|
+
"truncated": state["truncated"], "text": "\n".join(lines)}
|