awstorage 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awstorage/__init__.py +97 -0
- awstorage/_fs.py +222 -0
- awstorage/catalog.py +279 -0
- awstorage/classify.py +232 -0
- awstorage/cli.py +487 -0
- awstorage/collectors/__init__.py +38 -0
- awstorage/collectors/journal.py +83 -0
- awstorage/collectors/podman.py +222 -0
- awstorage/diff.py +53 -0
- awstorage/graph.py +87 -0
- awstorage/node_run.py +184 -0
- awstorage/policy.py +458 -0
- awstorage/remote.py +376 -0
- awstorage/report.py +69 -0
- awstorage-0.1.0.dist-info/METADATA +89 -0
- awstorage-0.1.0.dist-info/RECORD +20 -0
- awstorage-0.1.0.dist-info/WHEEL +5 -0
- awstorage-0.1.0.dist-info/entry_points.txt +2 -0
- awstorage-0.1.0.dist-info/licenses/LICENSE +202 -0
- awstorage-0.1.0.dist-info/top_level.txt +1 -0
awstorage/__init__.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""awstorage -- every drive on every node, indexed, classified and diffed.
|
|
2
|
+
|
|
3
|
+
Scan a root, get a SNAPSHOT: every directory tree down to a bounded depth with
|
|
4
|
+
its bytes, file count and newest mtime, plus the largest files. Classify each
|
|
5
|
+
tree (cache / build-temp / model-weights / dataset / backup / repo / logs /
|
|
6
|
+
service-state / media / unknown) and mark whether it is RE-FETCHABLE. Store
|
|
7
|
+
snapshots in a catalog and DIFF them, so "what grew since last week" is a
|
|
8
|
+
query rather than a memory. Turn the classification into PROPOSALS under a
|
|
9
|
+
written policy, and APPLY only what the policy pre-approves or a human
|
|
10
|
+
approved -- dry-run by default, ledgered, and never outside a declared root.
|
|
11
|
+
|
|
12
|
+
import awstorage
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
snap = awstorage.scan(Path("E:/"), max_depth=3, time_budget_s=120)
|
|
16
|
+
awstorage.classify_snapshot(snap) # heuristics, offline
|
|
17
|
+
cat = awstorage.Catalog(Path("inventory.db"))
|
|
18
|
+
sid = cat.put_snapshot(snap)
|
|
19
|
+
for row in awstorage.rank(snap)[:20]:
|
|
20
|
+
print(row["bytes"], row["path"], row["cls"], row["refetchable"])
|
|
21
|
+
props = awstorage.propose(snap, awstorage.default_policy())
|
|
22
|
+
awstorage.apply(props[0], roots=[Path("E:/")], dry_run=True)
|
|
23
|
+
|
|
24
|
+
Three rules it exists to enforce:
|
|
25
|
+
|
|
26
|
+
**Measure before you delete.** A `du` answers one question once. A snapshot
|
|
27
|
+
answers the same question every week, and the diff between two is the
|
|
28
|
+
question you actually had ("what filled the disk?").
|
|
29
|
+
|
|
30
|
+
**Re-fetchable is a property, not a guess.** A tree is re-fetchable when a
|
|
31
|
+
known producer can recreate it (a package cache, a build output, a model
|
|
32
|
+
weight that lives on a mirror). The classifier records WHY it decided that,
|
|
33
|
+
and whether a heuristic or a model decided -- so a wrong call is auditable.
|
|
34
|
+
|
|
35
|
+
**Deleting is a policy, not a mood.** `apply` refuses anything outside the
|
|
36
|
+
declared roots, anything whose fingerprint changed since the scan, and any
|
|
37
|
+
class the policy does not pre-approve unless a human approved that proposal.
|
|
38
|
+
It writes a ledger row either way. A dry run is the default.
|
|
39
|
+
|
|
40
|
+
The package is stdlib-only and speaks to nothing. Fleet integration (a scanner
|
|
41
|
+
per node, a catalog behind an API, a GUI, an autonomous steward, model-backed
|
|
42
|
+
classification) lives in the platform that imports it -- awstorage works alone.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
from ._fs import SCHEMA_VERSION, ScanError, scan
|
|
48
|
+
from .catalog import Catalog
|
|
49
|
+
from .classify import (
|
|
50
|
+
CLASSES,
|
|
51
|
+
Classifier,
|
|
52
|
+
HeuristicClassifier,
|
|
53
|
+
LLMClassifier,
|
|
54
|
+
classify_snapshot,
|
|
55
|
+
classify_tree,
|
|
56
|
+
)
|
|
57
|
+
from .diff import diff_snapshots
|
|
58
|
+
from .graph import to_graph
|
|
59
|
+
from .policy import (
|
|
60
|
+
ApplyRefused,
|
|
61
|
+
Proposal,
|
|
62
|
+
apply,
|
|
63
|
+
default_policy,
|
|
64
|
+
list_quarantine,
|
|
65
|
+
propose,
|
|
66
|
+
purge_quarantine,
|
|
67
|
+
revert,
|
|
68
|
+
)
|
|
69
|
+
from .report import rank, summarize
|
|
70
|
+
|
|
71
|
+
__version__ = "0.1.0"
|
|
72
|
+
|
|
73
|
+
__all__ = [
|
|
74
|
+
"SCHEMA_VERSION",
|
|
75
|
+
"ScanError",
|
|
76
|
+
"scan",
|
|
77
|
+
"Catalog",
|
|
78
|
+
"CLASSES",
|
|
79
|
+
"Classifier",
|
|
80
|
+
"HeuristicClassifier",
|
|
81
|
+
"LLMClassifier",
|
|
82
|
+
"classify_snapshot",
|
|
83
|
+
"classify_tree",
|
|
84
|
+
"diff_snapshots",
|
|
85
|
+
"to_graph",
|
|
86
|
+
"ApplyRefused",
|
|
87
|
+
"Proposal",
|
|
88
|
+
"apply",
|
|
89
|
+
"default_policy",
|
|
90
|
+
"list_quarantine",
|
|
91
|
+
"propose",
|
|
92
|
+
"purge_quarantine",
|
|
93
|
+
"revert",
|
|
94
|
+
"rank",
|
|
95
|
+
"summarize",
|
|
96
|
+
"__version__",
|
|
97
|
+
]
|
awstorage/_fs.py
ADDED
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""Bounded, resumable directory scanning -> a snapshot dict.
|
|
2
|
+
|
|
3
|
+
A snapshot is a plain dict (JSON-serialisable) so it can cross any wire:
|
|
4
|
+
|
|
5
|
+
{"schema": 1, "node": "<hostname>", "root": "E:/", "taken_at": "...",
|
|
6
|
+
"max_depth": 3, "time_budget_s": 120, "truncated": false,
|
|
7
|
+
"trees": [ {"path": "E:/Caches", "depth": 1, "bytes": 157_000_000_000,
|
|
8
|
+
"files": 12000, "dirs": 340, "newest_mtime": 1756000000.0,
|
|
9
|
+
"oldest_mtime": 1700000000.0, "fingerprint": "sha1:..."} ],
|
|
10
|
+
"top_files": [ {"path": ..., "bytes": ..., "mtime": ...} ],
|
|
11
|
+
"errors": ["E:/System Volume Information: PermissionError"],
|
|
12
|
+
"elapsed_s": 41.2}
|
|
13
|
+
|
|
14
|
+
`trees` holds every directory at depth <= max_depth (depth 0 is the root),
|
|
15
|
+
each with the AGGREGATE of everything beneath it -- so a parent's bytes
|
|
16
|
+
already include its children's. Consumers that want exclusive sizes subtract.
|
|
17
|
+
|
|
18
|
+
The fingerprint is sha1 over (bytes, files, newest_mtime): cheap, and enough
|
|
19
|
+
to notice that a tree changed between the scan and an apply. It is NOT a
|
|
20
|
+
content hash; awstorage never reads file contents.
|
|
21
|
+
|
|
22
|
+
Bounded on purpose: a scan that runs forever on a 4 TB drive is one nobody
|
|
23
|
+
runs twice. `time_budget_s` stops the walk and marks the snapshot TRUNCATED
|
|
24
|
+
(a real field, printed by every consumer) rather than returning a total that
|
|
25
|
+
looks complete and is not.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import hashlib
|
|
31
|
+
import os
|
|
32
|
+
import socket
|
|
33
|
+
import time
|
|
34
|
+
from datetime import datetime, timezone
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
from typing import Iterable
|
|
37
|
+
|
|
38
|
+
SCHEMA_VERSION = 1
|
|
39
|
+
|
|
40
|
+
# Directories that are never worth descending into: either they lie about size
|
|
41
|
+
# (reparse/junction targets, /proc) or they are the OS's own business.
|
|
42
|
+
DEFAULT_SKIP_NAMES = frozenset(
|
|
43
|
+
{
|
|
44
|
+
"$recycle.bin",
|
|
45
|
+
"system volume information",
|
|
46
|
+
"proc",
|
|
47
|
+
"sys",
|
|
48
|
+
"dev",
|
|
49
|
+
"run",
|
|
50
|
+
}
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ScanError(Exception):
|
|
55
|
+
"""The scan could not run at all (root missing, unreadable, not a dir)."""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _now_iso() -> str:
|
|
59
|
+
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _norm(p: Path) -> str:
|
|
63
|
+
# Forward slashes everywhere so the same tree has ONE spelling in the
|
|
64
|
+
# catalog whether it was scanned from Windows, WSL, or a Linux node.
|
|
65
|
+
return str(p).replace("\\", "/")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def fingerprint(bytes_: int, files: int, newest_mtime: float) -> str:
|
|
69
|
+
h = hashlib.sha1(f"{bytes_}|{files}|{int(newest_mtime)}".encode())
|
|
70
|
+
return "sha1:" + h.hexdigest()[:16]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def scan(
|
|
74
|
+
root: Path | str,
|
|
75
|
+
*,
|
|
76
|
+
max_depth: int = 3,
|
|
77
|
+
time_budget_s: float = 300.0,
|
|
78
|
+
top_files: int = 50,
|
|
79
|
+
skip_names: Iterable[str] = DEFAULT_SKIP_NAMES,
|
|
80
|
+
node: str | None = None,
|
|
81
|
+
follow_symlinks: bool = False,
|
|
82
|
+
) -> dict:
|
|
83
|
+
"""Walk `root` and return a snapshot dict. Never raises past the root check.
|
|
84
|
+
|
|
85
|
+
Per-entry errors (permission denied, vanished file) are recorded in
|
|
86
|
+
`errors` and the walk continues; only an unusable ROOT raises ScanError.
|
|
87
|
+
"""
|
|
88
|
+
root_p = Path(root)
|
|
89
|
+
if not root_p.exists():
|
|
90
|
+
raise ScanError(f"root does not exist: {root}")
|
|
91
|
+
if not root_p.is_dir():
|
|
92
|
+
raise ScanError(f"root is not a directory: {root}")
|
|
93
|
+
skip = {s.lower() for s in skip_names}
|
|
94
|
+
t0 = time.monotonic()
|
|
95
|
+
deadline = t0 + float(time_budget_s)
|
|
96
|
+
|
|
97
|
+
# Aggregates keyed by the directory path (string), for dirs at depth <= max_depth.
|
|
98
|
+
agg: dict[str, dict] = {}
|
|
99
|
+
errors: list[str] = []
|
|
100
|
+
biggest: list[tuple[int, float, str]] = [] # (bytes, mtime, path) kept small
|
|
101
|
+
truncated = False
|
|
102
|
+
|
|
103
|
+
def account(dir_key: str, size: int, mtime: float, is_file: bool) -> None:
|
|
104
|
+
a = agg[dir_key]
|
|
105
|
+
a["bytes"] += size
|
|
106
|
+
if is_file:
|
|
107
|
+
a["files"] += 1
|
|
108
|
+
else:
|
|
109
|
+
a["dirs"] += 1
|
|
110
|
+
if mtime > a["newest_mtime"]:
|
|
111
|
+
a["newest_mtime"] = mtime
|
|
112
|
+
if mtime and (a["oldest_mtime"] == 0.0 or mtime < a["oldest_mtime"]):
|
|
113
|
+
a["oldest_mtime"] = mtime
|
|
114
|
+
|
|
115
|
+
def ensure(path_str: str, depth: int) -> None:
|
|
116
|
+
if path_str not in agg:
|
|
117
|
+
agg[path_str] = {
|
|
118
|
+
"path": path_str,
|
|
119
|
+
"depth": depth,
|
|
120
|
+
"bytes": 0,
|
|
121
|
+
"files": 0,
|
|
122
|
+
"dirs": 0,
|
|
123
|
+
"newest_mtime": 0.0,
|
|
124
|
+
"oldest_mtime": 0.0,
|
|
125
|
+
"git": False,
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
root_key = _norm(root_p)
|
|
129
|
+
ensure(root_key, 0)
|
|
130
|
+
# Stack of (dir path, depth, ancestors-at-or-below-max-depth keys)
|
|
131
|
+
stack: list[tuple[str, int, tuple[str, ...]]] = [(str(root_p), 0, (root_key,))]
|
|
132
|
+
|
|
133
|
+
while stack:
|
|
134
|
+
if time.monotonic() >= deadline:
|
|
135
|
+
truncated = True
|
|
136
|
+
break
|
|
137
|
+
d, depth, owners = stack.pop()
|
|
138
|
+
try:
|
|
139
|
+
it = os.scandir(d)
|
|
140
|
+
except OSError as exc:
|
|
141
|
+
errors.append(f"{_norm(Path(d))}: {type(exc).__name__}")
|
|
142
|
+
continue
|
|
143
|
+
with it:
|
|
144
|
+
for e in it:
|
|
145
|
+
try:
|
|
146
|
+
if e.is_symlink() and not follow_symlinks:
|
|
147
|
+
continue
|
|
148
|
+
if e.is_dir(follow_symlinks=follow_symlinks):
|
|
149
|
+
if e.name == ".git" and owners[-1] in agg:
|
|
150
|
+
# A git working tree: its build outputs may be TRACKED
|
|
151
|
+
# (tenant repos `git add -f` their dist/), so the policy
|
|
152
|
+
# must not auto-delete inside one. Recorded, not decided.
|
|
153
|
+
agg[owners[-1]]["git"] = True
|
|
154
|
+
if e.name.lower() in skip:
|
|
155
|
+
continue
|
|
156
|
+
child_depth = depth + 1
|
|
157
|
+
child_key = _norm(Path(e.path))
|
|
158
|
+
for k in owners:
|
|
159
|
+
account(k, 0, 0.0, is_file=False)
|
|
160
|
+
if child_depth <= max_depth:
|
|
161
|
+
ensure(child_key, child_depth)
|
|
162
|
+
stack.append((e.path, child_depth, owners + (child_key,)))
|
|
163
|
+
else:
|
|
164
|
+
stack.append((e.path, child_depth, owners))
|
|
165
|
+
else:
|
|
166
|
+
st = e.stat(follow_symlinks=follow_symlinks)
|
|
167
|
+
size = int(st.st_size)
|
|
168
|
+
mtime = float(st.st_mtime)
|
|
169
|
+
for k in owners:
|
|
170
|
+
account(k, size, mtime, is_file=True)
|
|
171
|
+
if top_files > 0:
|
|
172
|
+
biggest.append((size, mtime, _norm(Path(e.path))))
|
|
173
|
+
if len(biggest) > top_files * 4:
|
|
174
|
+
biggest.sort(reverse=True)
|
|
175
|
+
del biggest[top_files:]
|
|
176
|
+
except OSError as exc:
|
|
177
|
+
errors.append(f"{_norm(Path(e.path))}: {type(exc).__name__}")
|
|
178
|
+
|
|
179
|
+
biggest.sort(reverse=True)
|
|
180
|
+
trees = []
|
|
181
|
+
for a in sorted(agg.values(), key=lambda x: (x["depth"], x["path"])):
|
|
182
|
+
a = dict(a)
|
|
183
|
+
a["fingerprint"] = fingerprint(a["bytes"], a["files"], a["newest_mtime"])
|
|
184
|
+
trees.append(a)
|
|
185
|
+
|
|
186
|
+
return {
|
|
187
|
+
"schema": SCHEMA_VERSION,
|
|
188
|
+
"node": node or socket.gethostname(),
|
|
189
|
+
"root": root_key,
|
|
190
|
+
"taken_at": _now_iso(),
|
|
191
|
+
"max_depth": int(max_depth),
|
|
192
|
+
"time_budget_s": float(time_budget_s),
|
|
193
|
+
"truncated": truncated,
|
|
194
|
+
"trees": trees,
|
|
195
|
+
"top_files": [
|
|
196
|
+
{"path": p, "bytes": b, "mtime": m} for b, m, p in biggest[:top_files]
|
|
197
|
+
],
|
|
198
|
+
"errors": errors[:200],
|
|
199
|
+
"error_count": len(errors),
|
|
200
|
+
"elapsed_s": round(time.monotonic() - t0, 2),
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def exclusive_bytes(snapshot: dict) -> dict[str, int]:
|
|
205
|
+
"""Bytes owned by each tree EXCLUDING its scanned children (depth-aware)."""
|
|
206
|
+
by_depth: dict[int, list[dict]] = {}
|
|
207
|
+
for t in snapshot["trees"]:
|
|
208
|
+
by_depth.setdefault(t["depth"], []).append(t)
|
|
209
|
+
children: dict[str, int] = {}
|
|
210
|
+
for depth, trees in by_depth.items():
|
|
211
|
+
if depth == 0:
|
|
212
|
+
continue
|
|
213
|
+
parents = by_depth.get(depth - 1, [])
|
|
214
|
+
for t in trees:
|
|
215
|
+
# The parent is the depth-1 tree whose path is the longest prefix of ours.
|
|
216
|
+
cand = [p["path"] for p in parents
|
|
217
|
+
if t["path"].startswith(p["path"].rstrip("/") + "/")]
|
|
218
|
+
if cand:
|
|
219
|
+
parent = max(cand, key=len)
|
|
220
|
+
children[parent] = children.get(parent, 0) + t["bytes"]
|
|
221
|
+
return {t["path"]: max(0, t["bytes"] - children.get(t["path"], 0))
|
|
222
|
+
for t in snapshot["trees"]}
|
awstorage/catalog.py
ADDED
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
"""SQLite catalog: snapshots, trees, proposals, ledger.
|
|
2
|
+
|
|
3
|
+
One file, stdlib sqlite3, WAL mode. The catalog is the thing a fleet posts
|
|
4
|
+
scans INTO and a GUI reads OUT of; the package only needs it to answer "what
|
|
5
|
+
did this root look like last time" and "what did we decide to do about it".
|
|
6
|
+
|
|
7
|
+
Every apply writes a LEDGER row whether it acted, dry-ran, or refused. A
|
|
8
|
+
storage tool that deletes without a ledger is indistinguishable from a bug.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import sqlite3
|
|
15
|
+
from datetime import datetime, timezone
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
_DDL = """
|
|
19
|
+
CREATE TABLE IF NOT EXISTS snapshots (
|
|
20
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
21
|
+
node TEXT NOT NULL,
|
|
22
|
+
root TEXT NOT NULL,
|
|
23
|
+
taken_at TEXT NOT NULL,
|
|
24
|
+
schema INTEGER NOT NULL,
|
|
25
|
+
truncated INTEGER NOT NULL DEFAULT 0,
|
|
26
|
+
total_bytes INTEGER NOT NULL DEFAULT 0,
|
|
27
|
+
tree_count INTEGER NOT NULL DEFAULT 0,
|
|
28
|
+
meta TEXT NOT NULL DEFAULT '{}'
|
|
29
|
+
);
|
|
30
|
+
CREATE INDEX IF NOT EXISTS snapshots_node_root ON snapshots(node, root, taken_at);
|
|
31
|
+
CREATE TABLE IF NOT EXISTS trees (
|
|
32
|
+
snapshot_id INTEGER NOT NULL REFERENCES snapshots(id) ON DELETE CASCADE,
|
|
33
|
+
path TEXT NOT NULL,
|
|
34
|
+
depth INTEGER NOT NULL,
|
|
35
|
+
bytes INTEGER NOT NULL,
|
|
36
|
+
files INTEGER NOT NULL,
|
|
37
|
+
dirs INTEGER NOT NULL,
|
|
38
|
+
newest_mtime REAL NOT NULL,
|
|
39
|
+
fingerprint TEXT NOT NULL,
|
|
40
|
+
cls TEXT,
|
|
41
|
+
refetchable INTEGER,
|
|
42
|
+
confidence REAL,
|
|
43
|
+
reason TEXT,
|
|
44
|
+
source TEXT,
|
|
45
|
+
PRIMARY KEY (snapshot_id, path)
|
|
46
|
+
);
|
|
47
|
+
CREATE TABLE IF NOT EXISTS top_files (
|
|
48
|
+
snapshot_id INTEGER NOT NULL REFERENCES snapshots(id) ON DELETE CASCADE,
|
|
49
|
+
path TEXT NOT NULL,
|
|
50
|
+
bytes INTEGER NOT NULL,
|
|
51
|
+
mtime REAL NOT NULL
|
|
52
|
+
);
|
|
53
|
+
CREATE TABLE IF NOT EXISTS proposals (
|
|
54
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
55
|
+
created_at TEXT NOT NULL,
|
|
56
|
+
snapshot_id INTEGER REFERENCES snapshots(id) ON DELETE SET NULL,
|
|
57
|
+
node TEXT NOT NULL,
|
|
58
|
+
path TEXT NOT NULL,
|
|
59
|
+
action TEXT NOT NULL,
|
|
60
|
+
bytes INTEGER NOT NULL,
|
|
61
|
+
cls TEXT NOT NULL,
|
|
62
|
+
policy_rule TEXT,
|
|
63
|
+
auto INTEGER NOT NULL DEFAULT 0,
|
|
64
|
+
status TEXT NOT NULL DEFAULT 'proposed',
|
|
65
|
+
fingerprint TEXT,
|
|
66
|
+
note TEXT
|
|
67
|
+
);
|
|
68
|
+
CREATE TABLE IF NOT EXISTS ledger (
|
|
69
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
70
|
+
at TEXT NOT NULL,
|
|
71
|
+
proposal_id INTEGER,
|
|
72
|
+
node TEXT NOT NULL,
|
|
73
|
+
path TEXT NOT NULL,
|
|
74
|
+
action TEXT NOT NULL,
|
|
75
|
+
outcome TEXT NOT NULL,
|
|
76
|
+
bytes INTEGER NOT NULL DEFAULT 0,
|
|
77
|
+
detail TEXT
|
|
78
|
+
);
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
_STATUSES = {"proposed", "approved", "rejected", "applied", "expired", "snoozed"}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _now() -> str:
|
|
85
|
+
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class Catalog:
|
|
89
|
+
def __init__(self, path: Path | str) -> None:
|
|
90
|
+
self.path = Path(path)
|
|
91
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
92
|
+
self._db = sqlite3.connect(str(self.path))
|
|
93
|
+
self._db.row_factory = sqlite3.Row
|
|
94
|
+
self._db.execute("PRAGMA journal_mode=WAL")
|
|
95
|
+
self._db.execute("PRAGMA foreign_keys=ON")
|
|
96
|
+
self._db.executescript(_DDL)
|
|
97
|
+
# v1.1: trees.git (scanner saw a .git). Additive, so an older catalog opens.
|
|
98
|
+
cols = {r[1] for r in self._db.execute("PRAGMA table_info(trees)")}
|
|
99
|
+
if "git" not in cols:
|
|
100
|
+
with self._db:
|
|
101
|
+
self._db.execute("ALTER TABLE trees ADD COLUMN git INTEGER NOT NULL DEFAULT 0")
|
|
102
|
+
|
|
103
|
+
def close(self) -> None:
|
|
104
|
+
self._db.close()
|
|
105
|
+
|
|
106
|
+
# -- snapshots -----------------------------------------------------------------
|
|
107
|
+
|
|
108
|
+
def put_snapshot(self, snap: dict) -> int:
|
|
109
|
+
trees = snap.get("trees", [])
|
|
110
|
+
root_row = next((t for t in trees if t["depth"] == 0), None)
|
|
111
|
+
total = int(root_row["bytes"]) if root_row else sum(t["bytes"] for t in trees)
|
|
112
|
+
meta = {k: snap.get(k) for k in ("max_depth", "time_budget_s", "elapsed_s",
|
|
113
|
+
"error_count", "classifier")}
|
|
114
|
+
with self._db:
|
|
115
|
+
cur = self._db.execute(
|
|
116
|
+
"INSERT INTO snapshots(node, root, taken_at, schema, truncated, total_bytes,"
|
|
117
|
+
" tree_count, meta) VALUES (?,?,?,?,?,?,?,?)",
|
|
118
|
+
(snap["node"], snap["root"], snap["taken_at"], int(snap["schema"]),
|
|
119
|
+
1 if snap.get("truncated") else 0, total, len(trees), json.dumps(meta)),
|
|
120
|
+
)
|
|
121
|
+
sid = int(cur.lastrowid)
|
|
122
|
+
self._db.executemany(
|
|
123
|
+
"INSERT INTO trees(snapshot_id, path, depth, bytes, files, dirs, newest_mtime,"
|
|
124
|
+
" fingerprint, cls, refetchable, confidence, reason, source, git)"
|
|
125
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
126
|
+
[(sid, t["path"], t["depth"], t["bytes"], t["files"], t["dirs"],
|
|
127
|
+
t["newest_mtime"], t["fingerprint"], t.get("cls"),
|
|
128
|
+
None if t.get("refetchable") is None else int(bool(t["refetchable"])),
|
|
129
|
+
t.get("confidence"), t.get("reason"), t.get("source"),
|
|
130
|
+
1 if t.get("git") else 0) for t in trees],
|
|
131
|
+
)
|
|
132
|
+
self._db.executemany(
|
|
133
|
+
"INSERT INTO top_files VALUES (?,?,?,?)",
|
|
134
|
+
[(sid, f["path"], f["bytes"], f["mtime"]) for f in snap.get("top_files", [])],
|
|
135
|
+
)
|
|
136
|
+
return sid
|
|
137
|
+
|
|
138
|
+
def list_snapshots(self, node: str | None = None, root: str | None = None,
|
|
139
|
+
limit: int = 50) -> list[dict]:
|
|
140
|
+
q = "SELECT * FROM snapshots"
|
|
141
|
+
cond, args = [], []
|
|
142
|
+
if node:
|
|
143
|
+
cond.append("node = ?")
|
|
144
|
+
args.append(node)
|
|
145
|
+
if root:
|
|
146
|
+
cond.append("root = ?")
|
|
147
|
+
args.append(root)
|
|
148
|
+
if cond:
|
|
149
|
+
q += " WHERE " + " AND ".join(cond)
|
|
150
|
+
q += " ORDER BY taken_at DESC, id DESC LIMIT ?"
|
|
151
|
+
args.append(int(limit))
|
|
152
|
+
return [dict(r) for r in self._db.execute(q, args)]
|
|
153
|
+
|
|
154
|
+
def get_snapshot(self, sid: int) -> dict | None:
|
|
155
|
+
row = self._db.execute("SELECT * FROM snapshots WHERE id = ?", (sid,)).fetchone()
|
|
156
|
+
if not row:
|
|
157
|
+
return None
|
|
158
|
+
snap = dict(row)
|
|
159
|
+
snap["meta"] = json.loads(snap.get("meta") or "{}")
|
|
160
|
+
snap["truncated"] = bool(snap["truncated"])
|
|
161
|
+
snap["trees"] = []
|
|
162
|
+
for t in self._db.execute(
|
|
163
|
+
"SELECT * FROM trees WHERE snapshot_id = ? ORDER BY depth, path", (sid,)
|
|
164
|
+
):
|
|
165
|
+
d = dict(t)
|
|
166
|
+
d.pop("snapshot_id", None)
|
|
167
|
+
if d.get("refetchable") is not None:
|
|
168
|
+
d["refetchable"] = bool(d["refetchable"])
|
|
169
|
+
d["git"] = bool(d.get("git"))
|
|
170
|
+
snap["trees"].append(d)
|
|
171
|
+
snap["top_files"] = [
|
|
172
|
+
{"path": r["path"], "bytes": r["bytes"], "mtime": r["mtime"]}
|
|
173
|
+
for r in self._db.execute(
|
|
174
|
+
"SELECT path, bytes, mtime FROM top_files WHERE snapshot_id = ?"
|
|
175
|
+
" ORDER BY bytes DESC", (sid,))
|
|
176
|
+
]
|
|
177
|
+
return snap
|
|
178
|
+
|
|
179
|
+
def latest_pair(self, node: str, root: str) -> tuple[dict | None, dict | None]:
|
|
180
|
+
"""(newest, previous) snapshots for a node+root -- the diff's usual inputs."""
|
|
181
|
+
rows = self.list_snapshots(node=node, root=root, limit=2)
|
|
182
|
+
newest = self.get_snapshot(rows[0]["id"]) if rows else None
|
|
183
|
+
prev = self.get_snapshot(rows[1]["id"]) if len(rows) > 1 else None
|
|
184
|
+
return newest, prev
|
|
185
|
+
|
|
186
|
+
def drop_snapshot(self, sid: int) -> None:
|
|
187
|
+
with self._db:
|
|
188
|
+
self._db.execute("DELETE FROM snapshots WHERE id = ?", (sid,))
|
|
189
|
+
|
|
190
|
+
# -- proposals / ledger --------------------------------------------------------
|
|
191
|
+
|
|
192
|
+
def put_proposals(self, proposals: list) -> list[int]:
|
|
193
|
+
ids = []
|
|
194
|
+
with self._db:
|
|
195
|
+
for p in proposals:
|
|
196
|
+
d = p if isinstance(p, dict) else p.__dict__
|
|
197
|
+
cur = self._db.execute(
|
|
198
|
+
"INSERT INTO proposals(created_at, snapshot_id, node, path, action, bytes,"
|
|
199
|
+
" cls, policy_rule, auto, status, fingerprint, note)"
|
|
200
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
201
|
+
(_now(), d.get("snapshot_id"), d["node"], d["path"], d["action"],
|
|
202
|
+
int(d["bytes"]), d["cls"], d.get("policy_rule"),
|
|
203
|
+
1 if d.get("auto") else 0, d.get("status", "proposed"),
|
|
204
|
+
d.get("fingerprint"), d.get("note")),
|
|
205
|
+
)
|
|
206
|
+
ids.append(int(cur.lastrowid))
|
|
207
|
+
return ids
|
|
208
|
+
|
|
209
|
+
def list_proposals(self, status: str | None = None, node: str | None = None,
|
|
210
|
+
limit: int = 200) -> list[dict]:
|
|
211
|
+
q = "SELECT * FROM proposals"
|
|
212
|
+
cond, args = [], []
|
|
213
|
+
if status:
|
|
214
|
+
cond.append("status = ?")
|
|
215
|
+
args.append(status)
|
|
216
|
+
if node:
|
|
217
|
+
cond.append("node = ?")
|
|
218
|
+
args.append(node)
|
|
219
|
+
if cond:
|
|
220
|
+
q += " WHERE " + " AND ".join(cond)
|
|
221
|
+
q += " ORDER BY bytes DESC, id DESC LIMIT ?"
|
|
222
|
+
args.append(int(limit))
|
|
223
|
+
out = []
|
|
224
|
+
for r in self._db.execute(q, args):
|
|
225
|
+
d = dict(r)
|
|
226
|
+
d["auto"] = bool(d["auto"])
|
|
227
|
+
out.append(d)
|
|
228
|
+
return out
|
|
229
|
+
|
|
230
|
+
def get_proposal(self, pid: int) -> dict | None:
|
|
231
|
+
r = self._db.execute("SELECT * FROM proposals WHERE id = ?", (pid,)).fetchone()
|
|
232
|
+
if not r:
|
|
233
|
+
return None
|
|
234
|
+
d = dict(r)
|
|
235
|
+
d["auto"] = bool(d["auto"])
|
|
236
|
+
return d
|
|
237
|
+
|
|
238
|
+
def set_status(self, pid: int, status: str, note: str | None = None) -> None:
|
|
239
|
+
if status not in _STATUSES:
|
|
240
|
+
raise ValueError(f"unknown proposal status {status!r}")
|
|
241
|
+
with self._db:
|
|
242
|
+
if note is None:
|
|
243
|
+
self._db.execute("UPDATE proposals SET status = ? WHERE id = ?", (status, pid))
|
|
244
|
+
else:
|
|
245
|
+
self._db.execute("UPDATE proposals SET status = ?, note = ? WHERE id = ?",
|
|
246
|
+
(status, note, pid))
|
|
247
|
+
|
|
248
|
+
def ledger(self, *, proposal_id: int | None, node: str, path: str, action: str,
|
|
249
|
+
outcome: str, bytes_: int = 0, detail: str | None = None) -> int:
|
|
250
|
+
with self._db:
|
|
251
|
+
cur = self._db.execute(
|
|
252
|
+
"INSERT INTO ledger(at, proposal_id, node, path, action, outcome, bytes, detail)"
|
|
253
|
+
" VALUES (?,?,?,?,?,?,?,?)",
|
|
254
|
+
(_now(), proposal_id, node, path, action, outcome, int(bytes_), detail),
|
|
255
|
+
)
|
|
256
|
+
return int(cur.lastrowid)
|
|
257
|
+
|
|
258
|
+
def list_ledger(self, limit: int = 200) -> list[dict]:
|
|
259
|
+
return [dict(r) for r in self._db.execute(
|
|
260
|
+
"SELECT * FROM ledger ORDER BY id DESC LIMIT ?", (int(limit),))]
|
|
261
|
+
|
|
262
|
+
def totals(self) -> dict:
|
|
263
|
+
"""Cross-node roll-up for a dashboard: newest snapshot per (node, root)."""
|
|
264
|
+
rows = self._db.execute(
|
|
265
|
+
"SELECT s.* FROM snapshots s JOIN ("
|
|
266
|
+
" SELECT node, root, MAX(taken_at) AS t FROM snapshots GROUP BY node, root"
|
|
267
|
+
") m ON s.node = m.node AND s.root = m.root AND s.taken_at = m.t"
|
|
268
|
+
" ORDER BY s.node, s.root"
|
|
269
|
+
).fetchall()
|
|
270
|
+
out = {"roots": [], "total_bytes": 0, "nodes": set()}
|
|
271
|
+
for r in rows:
|
|
272
|
+
d = dict(r)
|
|
273
|
+
d["truncated"] = bool(d["truncated"])
|
|
274
|
+
out["roots"].append({k: d[k] for k in ("id", "node", "root", "taken_at",
|
|
275
|
+
"total_bytes", "truncated")})
|
|
276
|
+
out["total_bytes"] += int(d["total_bytes"])
|
|
277
|
+
out["nodes"].add(d["node"])
|
|
278
|
+
out["nodes"] = sorted(out["nodes"])
|
|
279
|
+
return out
|