@inneranimalmedia/agentsam-sdk 1.8.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/DEVELOPMENT.md +1 -1
- package/docs/RELEASES.iam-mirror.md +20 -0
- package/docs/RELEASES.md +4 -1
- package/package.json +2 -2
- package/python/agentsam_sdk/cli.py +57 -7
- package/python/agentsam_sdk/data/d1_bloat.py +279 -99
- package/python/agentsam_sdk/repository/__init__.py +26 -0
- package/python/agentsam_sdk/repository/inspect.py +496 -0
- package/python/docs/gaps.md +1 -1
- package/python/tests/test_d1_bloat.py +60 -24
- package/python/tests/test_repository_inspect.py +84 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Repository-level audits (inventory, scan_bloat, inspect).
|
|
2
|
+
|
|
3
|
+
Lazy exports so `python -m agentsam_sdk.repository.inspect` does not warn.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
__all__ = ["inventory", "scan_bloat", "inspect"]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def __getattr__(name: str) -> Any:
|
|
14
|
+
if name == "inventory":
|
|
15
|
+
from agentsam_sdk.repository import inventory as mod
|
|
16
|
+
|
|
17
|
+
return mod
|
|
18
|
+
if name == "scan_bloat":
|
|
19
|
+
from agentsam_sdk.repository import scan_bloat as mod
|
|
20
|
+
|
|
21
|
+
return mod
|
|
22
|
+
if name == "inspect":
|
|
23
|
+
from agentsam_sdk.repository import inspect as mod
|
|
24
|
+
|
|
25
|
+
return mod
|
|
26
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
"""agentsam_sdk.repository.inspect — repo walk (size + dates) + optional content dupes.
|
|
2
|
+
|
|
3
|
+
Reusable library: no print(), no sys.exit(), no hardcoded tenant/workspace ids.
|
|
4
|
+
Root path is always an explicit parameter.
|
|
5
|
+
|
|
6
|
+
from agentsam_sdk.repository.inspect import walk_repo, summarize, find_dupes, build_report
|
|
7
|
+
|
|
8
|
+
CLI:
|
|
9
|
+
python3 scripts/repo_inspect.py --text
|
|
10
|
+
python3 -m agentsam_sdk.repository.inspect --json --dupes
|
|
11
|
+
agentsam repository inspect --repo-root . --format json --dupes
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import hashlib
|
|
16
|
+
import os
|
|
17
|
+
import subprocess
|
|
18
|
+
from collections import defaultdict
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Iterable
|
|
22
|
+
|
|
23
|
+
SCHEMA_VERSION = 1
|
|
24
|
+
TOOL_NAME = "repository.inspect"
|
|
25
|
+
HASH_CHUNK = 1024 * 1024
|
|
26
|
+
|
|
27
|
+
DEFAULT_SKIP_DIR_NAMES = frozenset(
|
|
28
|
+
{
|
|
29
|
+
".git",
|
|
30
|
+
".scratch",
|
|
31
|
+
".venv",
|
|
32
|
+
".venv_agentsam",
|
|
33
|
+
"venv",
|
|
34
|
+
"node_modules",
|
|
35
|
+
"__pycache__",
|
|
36
|
+
".pytest_cache",
|
|
37
|
+
".mypy_cache",
|
|
38
|
+
".ruff_cache",
|
|
39
|
+
".turbo",
|
|
40
|
+
".next",
|
|
41
|
+
"dist",
|
|
42
|
+
"build",
|
|
43
|
+
"coverage",
|
|
44
|
+
".wrangler",
|
|
45
|
+
"vendor",
|
|
46
|
+
"captures",
|
|
47
|
+
"architecture-map",
|
|
48
|
+
}
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class RepoRootError(Exception):
|
|
53
|
+
"""Raised when a git toplevel cannot be resolved."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def utc_now() -> datetime:
|
|
57
|
+
return datetime.now(timezone.utc)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def iso_from_unix(ts: float | int | None) -> str | None:
|
|
61
|
+
if ts is None:
|
|
62
|
+
return None
|
|
63
|
+
try:
|
|
64
|
+
return datetime.fromtimestamp(float(ts), tz=timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
65
|
+
except (OSError, OverflowError, ValueError):
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def parse_since(raw: str | None, *, now_unix: int | None = None) -> int | None:
|
|
70
|
+
"""Return min mtime_unix, or None for no filter. Raises ValueError on bad input."""
|
|
71
|
+
if raw is None or str(raw).strip() == "":
|
|
72
|
+
return None
|
|
73
|
+
s = str(raw).strip().lower()
|
|
74
|
+
now = int(now_unix if now_unix is not None else utc_now().timestamp())
|
|
75
|
+
if s.endswith("d") and s[:-1].isdigit():
|
|
76
|
+
return now - int(s[:-1]) * 86400
|
|
77
|
+
if s.endswith("h") and s[:-1].isdigit():
|
|
78
|
+
return now - int(s[:-1]) * 3600
|
|
79
|
+
if s.isdigit():
|
|
80
|
+
return now - int(s)
|
|
81
|
+
raise ValueError(f"bad --since {raw!r} (use Nd, Nh, or seconds)")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def find_repo_root(start: Path | None = None) -> Path:
|
|
85
|
+
"""Resolve git toplevel from start (default: cwd). Raises RepoRootError."""
|
|
86
|
+
here = (start or Path.cwd()).resolve()
|
|
87
|
+
try:
|
|
88
|
+
out = subprocess.check_output(
|
|
89
|
+
["git", "rev-parse", "--show-toplevel"],
|
|
90
|
+
cwd=here,
|
|
91
|
+
text=True,
|
|
92
|
+
stderr=subprocess.DEVNULL,
|
|
93
|
+
).strip()
|
|
94
|
+
if out:
|
|
95
|
+
return Path(out).resolve()
|
|
96
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
97
|
+
pass
|
|
98
|
+
for p in [here, *here.parents]:
|
|
99
|
+
if (p / ".git").exists():
|
|
100
|
+
return p
|
|
101
|
+
raise RepoRootError(f"not inside a git repository (start={here})")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def git_head(repo_root: Path) -> dict[str, Any]:
|
|
105
|
+
def _run(*args: str) -> str:
|
|
106
|
+
try:
|
|
107
|
+
return subprocess.check_output(
|
|
108
|
+
["git", *args],
|
|
109
|
+
cwd=repo_root,
|
|
110
|
+
text=True,
|
|
111
|
+
stderr=subprocess.DEVNULL,
|
|
112
|
+
).strip()
|
|
113
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
114
|
+
return ""
|
|
115
|
+
|
|
116
|
+
sha = _run("rev-parse", "HEAD")
|
|
117
|
+
branch = _run("branch", "--show-current") or _run("rev-parse", "--abbrev-ref", "HEAD")
|
|
118
|
+
subject = _run("log", "-1", "--format=%s")
|
|
119
|
+
author_unix = _run("log", "-1", "--format=%at")
|
|
120
|
+
author_unix_i = int(author_unix) if author_unix.isdigit() else None
|
|
121
|
+
return {
|
|
122
|
+
"branch": branch or None,
|
|
123
|
+
"head_sha": sha if len(sha) == 40 else (sha or None),
|
|
124
|
+
"head_subject": subject or None,
|
|
125
|
+
"head_author_unix": author_unix_i,
|
|
126
|
+
"head_author_iso": iso_from_unix(author_unix_i),
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def file_times(st: os.stat_result) -> dict[str, Any]:
|
|
131
|
+
mtime = int(st.st_mtime)
|
|
132
|
+
ctime = int(st.st_ctime)
|
|
133
|
+
birth = None
|
|
134
|
+
birth_raw = getattr(st, "st_birthtime", None)
|
|
135
|
+
if birth_raw is not None:
|
|
136
|
+
try:
|
|
137
|
+
birth = int(birth_raw)
|
|
138
|
+
except (TypeError, ValueError):
|
|
139
|
+
birth = None
|
|
140
|
+
return {
|
|
141
|
+
"size_bytes": int(st.st_size),
|
|
142
|
+
"mtime_unix": mtime,
|
|
143
|
+
"mtime_iso": iso_from_unix(mtime),
|
|
144
|
+
"ctime_unix": ctime,
|
|
145
|
+
"ctime_iso": iso_from_unix(ctime),
|
|
146
|
+
"birth_unix": birth,
|
|
147
|
+
"birth_iso": iso_from_unix(birth),
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def walk_repo(
|
|
152
|
+
root: Path,
|
|
153
|
+
*,
|
|
154
|
+
skip_dir_names: Iterable[str] | None = None,
|
|
155
|
+
respect_gitignore: bool = True,
|
|
156
|
+
follow_symlinks: bool = False,
|
|
157
|
+
errors: list[dict[str, str]] | None = None,
|
|
158
|
+
) -> list[dict[str, Any]]:
|
|
159
|
+
"""Walk root and return file rows (path relative to root, sizes, dates).
|
|
160
|
+
|
|
161
|
+
respect_gitignore=True applies DEFAULT_SKIP_DIR_NAMES (heavy/generated dirs).
|
|
162
|
+
Full .gitignore parsing is not implemented — pass skip_dir_names to customize.
|
|
163
|
+
Non-fatal OSErrors append to errors when provided; otherwise they are skipped.
|
|
164
|
+
"""
|
|
165
|
+
root = Path(root).resolve()
|
|
166
|
+
skip = set(DEFAULT_SKIP_DIR_NAMES if respect_gitignore else ())
|
|
167
|
+
if skip_dir_names is not None:
|
|
168
|
+
skip |= {str(s) for s in skip_dir_names}
|
|
169
|
+
|
|
170
|
+
rows: list[dict[str, Any]] = []
|
|
171
|
+
err_sink = errors if errors is not None else []
|
|
172
|
+
|
|
173
|
+
for dirpath, dirnames, filenames in os.walk(root, topdown=True, followlinks=follow_symlinks):
|
|
174
|
+
dirnames[:] = sorted(
|
|
175
|
+
d for d in dirnames if d not in skip and not d.startswith(".cache")
|
|
176
|
+
)
|
|
177
|
+
base = Path(dirpath)
|
|
178
|
+
for name in filenames:
|
|
179
|
+
if name == ".DS_Store":
|
|
180
|
+
continue
|
|
181
|
+
path = base / name
|
|
182
|
+
try:
|
|
183
|
+
if path.is_symlink() and not follow_symlinks:
|
|
184
|
+
continue
|
|
185
|
+
st = path.stat()
|
|
186
|
+
except OSError as e:
|
|
187
|
+
err_sink.append({"path": str(path), "error": f"stat:{e}"})
|
|
188
|
+
continue
|
|
189
|
+
if not path.is_file():
|
|
190
|
+
continue
|
|
191
|
+
try:
|
|
192
|
+
rel = path.relative_to(root).as_posix()
|
|
193
|
+
except ValueError as e:
|
|
194
|
+
err_sink.append({"path": str(path), "error": f"relative:{e}"})
|
|
195
|
+
continue
|
|
196
|
+
top = rel.split("/", 1)[0] if "/" in rel else "(root)"
|
|
197
|
+
ext = path.suffix.lower() or "(none)"
|
|
198
|
+
rows.append(
|
|
199
|
+
{
|
|
200
|
+
"path": rel,
|
|
201
|
+
"top_dir": top,
|
|
202
|
+
"ext": ext,
|
|
203
|
+
**file_times(st),
|
|
204
|
+
}
|
|
205
|
+
)
|
|
206
|
+
return rows
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def summarize(
|
|
210
|
+
files: list[dict[str, Any]],
|
|
211
|
+
*,
|
|
212
|
+
recent_n: int = 50,
|
|
213
|
+
largest_n: int = 30,
|
|
214
|
+
since_unix: int | None = None,
|
|
215
|
+
) -> dict[str, Any]:
|
|
216
|
+
"""Rollups + recent/largest slices from walk_repo rows."""
|
|
217
|
+
by_dir: dict[str, dict[str, int]] = defaultdict(lambda: {"files": 0, "bytes": 0})
|
|
218
|
+
total_bytes = 0
|
|
219
|
+
for f in files:
|
|
220
|
+
total_bytes += int(f.get("size_bytes") or 0)
|
|
221
|
+
bucket = by_dir[str(f.get("top_dir") or "(root)")]
|
|
222
|
+
bucket["files"] += 1
|
|
223
|
+
bucket["bytes"] += int(f.get("size_bytes") or 0)
|
|
224
|
+
|
|
225
|
+
filtered = files
|
|
226
|
+
if since_unix is not None:
|
|
227
|
+
filtered = [f for f in files if int(f.get("mtime_unix") or 0) >= since_unix]
|
|
228
|
+
|
|
229
|
+
recent = sorted(filtered, key=lambda r: (-int(r.get("mtime_unix") or 0), r.get("path") or ""))[
|
|
230
|
+
: max(1, recent_n)
|
|
231
|
+
]
|
|
232
|
+
largest = sorted(files, key=lambda r: (-int(r.get("size_bytes") or 0), r.get("path") or ""))[
|
|
233
|
+
: max(1, largest_n)
|
|
234
|
+
]
|
|
235
|
+
top_dirs = sorted(
|
|
236
|
+
({"top_dir": k, "files": v["files"], "bytes": v["bytes"]} for k, v in by_dir.items()),
|
|
237
|
+
key=lambda r: (-r["bytes"], r["top_dir"]),
|
|
238
|
+
)
|
|
239
|
+
return {
|
|
240
|
+
"file_count": len(files),
|
|
241
|
+
"total_bytes": total_bytes,
|
|
242
|
+
"since_unix": since_unix,
|
|
243
|
+
"recent_count": len(recent),
|
|
244
|
+
"by_top_dir": top_dirs,
|
|
245
|
+
"recent": recent,
|
|
246
|
+
"largest": largest,
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def sha256_file(path: Path, *, chunk_size: int = HASH_CHUNK) -> str:
|
|
251
|
+
h = hashlib.sha256()
|
|
252
|
+
with path.open("rb") as fh:
|
|
253
|
+
while True:
|
|
254
|
+
chunk = fh.read(chunk_size)
|
|
255
|
+
if not chunk:
|
|
256
|
+
break
|
|
257
|
+
h.update(chunk)
|
|
258
|
+
return h.hexdigest()
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def find_dupes(
|
|
262
|
+
files: list[dict[str, Any]],
|
|
263
|
+
*,
|
|
264
|
+
repo_root: Path,
|
|
265
|
+
errors: list[dict[str, str]] | None = None,
|
|
266
|
+
warnings: list[str] | None = None,
|
|
267
|
+
) -> list[dict[str, Any]]:
|
|
268
|
+
"""True content duplicates: same size_bytes + SHA-256.
|
|
269
|
+
|
|
270
|
+
Returns groups sorted by size_bytes descending. Each group:
|
|
271
|
+
size_bytes, sha256, count, wasted_bytes (= size * (count-1)), paths[]
|
|
272
|
+
"""
|
|
273
|
+
root = Path(repo_root).resolve()
|
|
274
|
+
err_sink = errors if errors is not None else []
|
|
275
|
+
warn_sink = warnings if warnings is not None else []
|
|
276
|
+
|
|
277
|
+
by_size: dict[int, list[dict[str, Any]]] = defaultdict(list)
|
|
278
|
+
for f in files:
|
|
279
|
+
by_size[int(f.get("size_bytes") or 0)].append(f)
|
|
280
|
+
|
|
281
|
+
groups: list[dict[str, Any]] = []
|
|
282
|
+
for size, bucket in by_size.items():
|
|
283
|
+
if len(bucket) < 2:
|
|
284
|
+
continue
|
|
285
|
+
by_hash: dict[str, list[str]] = defaultdict(list)
|
|
286
|
+
for f in bucket:
|
|
287
|
+
rel = str(f.get("path") or "")
|
|
288
|
+
abs_path = root / rel
|
|
289
|
+
try:
|
|
290
|
+
digest = sha256_file(abs_path)
|
|
291
|
+
except OSError as e:
|
|
292
|
+
msg = f"hash skip {rel}: {e}"
|
|
293
|
+
warn_sink.append(msg)
|
|
294
|
+
err_sink.append({"path": rel, "error": f"hash:{e}"})
|
|
295
|
+
continue
|
|
296
|
+
by_hash[digest].append(rel)
|
|
297
|
+
|
|
298
|
+
for digest, paths in by_hash.items():
|
|
299
|
+
if len(paths) < 2:
|
|
300
|
+
continue
|
|
301
|
+
paths_sorted = sorted(paths)
|
|
302
|
+
count = len(paths_sorted)
|
|
303
|
+
groups.append(
|
|
304
|
+
{
|
|
305
|
+
"size_bytes": size,
|
|
306
|
+
"sha256": digest,
|
|
307
|
+
"count": count,
|
|
308
|
+
"wasted_bytes": size * (count - 1),
|
|
309
|
+
"paths": paths_sorted,
|
|
310
|
+
}
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
groups.sort(key=lambda g: (-int(g["size_bytes"]), -int(g["count"]), g["sha256"]))
|
|
314
|
+
return groups
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def build_report(
|
|
318
|
+
repo_root: Path,
|
|
319
|
+
*,
|
|
320
|
+
recent_n: int = 50,
|
|
321
|
+
largest_n: int = 30,
|
|
322
|
+
since_unix: int | None = None,
|
|
323
|
+
include_all: bool = False,
|
|
324
|
+
include_dupes: bool = False,
|
|
325
|
+
skip_dir_names: Iterable[str] | None = None,
|
|
326
|
+
respect_gitignore: bool = True,
|
|
327
|
+
) -> dict[str, Any]:
|
|
328
|
+
"""Full jq-stable report dict (summary/recent/largest keys preserved)."""
|
|
329
|
+
root = Path(repo_root).resolve()
|
|
330
|
+
walk_errors: list[dict[str, str]] = []
|
|
331
|
+
files = walk_repo(
|
|
332
|
+
root,
|
|
333
|
+
skip_dir_names=skip_dir_names,
|
|
334
|
+
respect_gitignore=respect_gitignore,
|
|
335
|
+
errors=walk_errors,
|
|
336
|
+
)
|
|
337
|
+
rollup = summarize(
|
|
338
|
+
files,
|
|
339
|
+
recent_n=recent_n,
|
|
340
|
+
largest_n=largest_n,
|
|
341
|
+
since_unix=since_unix,
|
|
342
|
+
)
|
|
343
|
+
now = int(utc_now().timestamp())
|
|
344
|
+
report: dict[str, Any] = {
|
|
345
|
+
"schema_version": SCHEMA_VERSION,
|
|
346
|
+
"tool": "repo_inspect",
|
|
347
|
+
"repo_root": str(root),
|
|
348
|
+
"repo_name": root.name,
|
|
349
|
+
"generated_at_unix": now,
|
|
350
|
+
"generated_at_iso": iso_from_unix(now),
|
|
351
|
+
"git": git_head(root),
|
|
352
|
+
"summary": {
|
|
353
|
+
"file_count": rollup["file_count"],
|
|
354
|
+
"total_bytes": rollup["total_bytes"],
|
|
355
|
+
"since_unix": rollup["since_unix"],
|
|
356
|
+
"recent_count": rollup["recent_count"],
|
|
357
|
+
"by_top_dir": rollup["by_top_dir"],
|
|
358
|
+
},
|
|
359
|
+
"recent": rollup["recent"],
|
|
360
|
+
"largest": rollup["largest"],
|
|
361
|
+
"jq": {
|
|
362
|
+
"summary": ".summary",
|
|
363
|
+
"recent": ".recent[] | {path, size_bytes, mtime_iso}",
|
|
364
|
+
"largest": ".largest[] | {path, size_bytes}",
|
|
365
|
+
"by_dir": ".summary.by_top_dir[]",
|
|
366
|
+
"changed_today": f".recent[] | select(.mtime_unix >= {now - 86400})",
|
|
367
|
+
"duplicates": ".duplicates[] | {size_bytes, wasted_bytes, count, paths}",
|
|
368
|
+
},
|
|
369
|
+
}
|
|
370
|
+
if include_all:
|
|
371
|
+
report["files"] = sorted(files, key=lambda r: r["path"])
|
|
372
|
+
if walk_errors:
|
|
373
|
+
report["walk_errors"] = walk_errors
|
|
374
|
+
|
|
375
|
+
if include_dupes:
|
|
376
|
+
hash_errors: list[dict[str, str]] = []
|
|
377
|
+
hash_warnings: list[str] = []
|
|
378
|
+
dupes = find_dupes(files, repo_root=root, errors=hash_errors, warnings=hash_warnings)
|
|
379
|
+
report["duplicates"] = dupes
|
|
380
|
+
report["summary"]["duplicate_groups"] = len(dupes)
|
|
381
|
+
report["summary"]["duplicate_wasted_bytes"] = sum(int(g["wasted_bytes"]) for g in dupes)
|
|
382
|
+
if hash_warnings:
|
|
383
|
+
report["hash_warnings"] = hash_warnings
|
|
384
|
+
if hash_errors:
|
|
385
|
+
report["hash_errors"] = hash_errors
|
|
386
|
+
return report
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def render_text(report: dict[str, Any]) -> str:
|
|
390
|
+
g = report.get("git") or {}
|
|
391
|
+
s = report.get("summary") or {}
|
|
392
|
+
lines = [
|
|
393
|
+
f"repo_inspect {report.get('repo_name')} @ {report.get('generated_at_iso')}",
|
|
394
|
+
f"git {g.get('branch') or '?'} {str(g.get('head_sha') or '')[:12]} {g.get('head_subject') or ''}",
|
|
395
|
+
f"files {int(s.get('file_count') or 0):,} bytes {int(s.get('total_bytes') or 0):,}",
|
|
396
|
+
"",
|
|
397
|
+
"recent (mtime):",
|
|
398
|
+
]
|
|
399
|
+
for f in (report.get("recent") or [])[:25]:
|
|
400
|
+
lines.append(f" {f.get('mtime_iso')} {int(f.get('size_bytes') or 0):>10,} {f.get('path')}")
|
|
401
|
+
lines.append("")
|
|
402
|
+
lines.append("largest:")
|
|
403
|
+
for f in (report.get("largest") or [])[:15]:
|
|
404
|
+
lines.append(f" {int(f.get('size_bytes') or 0):>12,} {f.get('path')}")
|
|
405
|
+
lines.append("")
|
|
406
|
+
lines.append("top dirs:")
|
|
407
|
+
for d in (s.get("by_top_dir") or [])[:12]:
|
|
408
|
+
lines.append(
|
|
409
|
+
f" {int(d.get('bytes') or 0):>12,} {int(d.get('files') or 0):>6} files {d.get('top_dir')}"
|
|
410
|
+
)
|
|
411
|
+
dupes = report.get("duplicates")
|
|
412
|
+
if isinstance(dupes, list):
|
|
413
|
+
lines.append("")
|
|
414
|
+
wasted = int(s.get("duplicate_wasted_bytes") or 0)
|
|
415
|
+
lines.append(f"duplicates: {len(dupes)} group(s), wasted {wasted:,} bytes")
|
|
416
|
+
for gdup in dupes[:20]:
|
|
417
|
+
lines.append(
|
|
418
|
+
f" size={int(gdup.get('size_bytes') or 0):,} "
|
|
419
|
+
f"count={int(gdup.get('count') or 0)} "
|
|
420
|
+
f"wasted={int(gdup.get('wasted_bytes') or 0):,} "
|
|
421
|
+
f"sha256={str(gdup.get('sha256') or '')[:12]}…"
|
|
422
|
+
)
|
|
423
|
+
for pth in (gdup.get("paths") or [])[:8]:
|
|
424
|
+
lines.append(f" {pth}")
|
|
425
|
+
lines.append("")
|
|
426
|
+
lines.append("jq: python3 scripts/repo_inspect.py --json | jq '.recent[0:10]'")
|
|
427
|
+
lines.append("dupes: python3 scripts/repo_inspect.py --json --dupes | jq '.duplicates'")
|
|
428
|
+
return "\n".join(lines) + "\n"
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def main_cli(argv: list[str] | None = None) -> int:
|
|
432
|
+
"""Argparse entry for `python -m agentsam_sdk.repository.inspect` and shims."""
|
|
433
|
+
import argparse
|
|
434
|
+
import json
|
|
435
|
+
import sys
|
|
436
|
+
|
|
437
|
+
p = argparse.ArgumentParser(description="Canonical repo file inspect (size + dates)")
|
|
438
|
+
p.add_argument("--repo-root", default=None, help="Repo root (default: git toplevel)")
|
|
439
|
+
p.add_argument("--json", action="store_true", help="Emit JSON (default when not --text)")
|
|
440
|
+
p.add_argument("--text", action="store_true", help="Human briefing on stdout")
|
|
441
|
+
p.add_argument("--recent", type=int, default=50, help="How many recent files (mtime)")
|
|
442
|
+
p.add_argument("--largest", type=int, default=30, help="How many largest files")
|
|
443
|
+
p.add_argument("--since", default=None, help="Only recent[] after window (e.g. 7d, 24h)")
|
|
444
|
+
p.add_argument("--all", action="store_true", help="Include full files[] array")
|
|
445
|
+
p.add_argument(
|
|
446
|
+
"--dupes",
|
|
447
|
+
action="store_true",
|
|
448
|
+
help="SHA-256 content duplicate groups (expensive; off by default)",
|
|
449
|
+
)
|
|
450
|
+
p.add_argument(
|
|
451
|
+
"--out",
|
|
452
|
+
default=None,
|
|
453
|
+
help="Write JSON to path (also prints text/json to stdout per flags)",
|
|
454
|
+
)
|
|
455
|
+
args = p.parse_args(argv)
|
|
456
|
+
|
|
457
|
+
try:
|
|
458
|
+
repo = Path(args.repo_root).resolve() if args.repo_root else find_repo_root()
|
|
459
|
+
except RepoRootError as e:
|
|
460
|
+
print(str(e), file=sys.stderr)
|
|
461
|
+
return 2
|
|
462
|
+
|
|
463
|
+
try:
|
|
464
|
+
since_unix = parse_since(args.since)
|
|
465
|
+
except ValueError as e:
|
|
466
|
+
print(str(e), file=sys.stderr)
|
|
467
|
+
return 2
|
|
468
|
+
|
|
469
|
+
report = build_report(
|
|
470
|
+
repo,
|
|
471
|
+
recent_n=max(1, args.recent),
|
|
472
|
+
largest_n=max(1, args.largest),
|
|
473
|
+
since_unix=since_unix,
|
|
474
|
+
include_all=bool(args.all),
|
|
475
|
+
include_dupes=bool(args.dupes),
|
|
476
|
+
)
|
|
477
|
+
|
|
478
|
+
if args.out:
|
|
479
|
+
out_path = Path(args.out)
|
|
480
|
+
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
481
|
+
out_path.write_text(json.dumps(report, indent=2, sort_keys=False) + "\n", encoding="utf-8")
|
|
482
|
+
|
|
483
|
+
if args.dupes:
|
|
484
|
+
for w in report.get("hash_warnings") or []:
|
|
485
|
+
print(f"warning: {w}", file=sys.stderr)
|
|
486
|
+
|
|
487
|
+
if args.text and not args.json:
|
|
488
|
+
sys.stdout.write(render_text(report))
|
|
489
|
+
else:
|
|
490
|
+
json.dump(report, sys.stdout, indent=2, sort_keys=False)
|
|
491
|
+
sys.stdout.write("\n")
|
|
492
|
+
return 0
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
if __name__ == "__main__":
|
|
496
|
+
raise SystemExit(main_cli())
|
package/python/docs/gaps.md
CHANGED
|
@@ -13,7 +13,7 @@ Host tools (jq, wrangler, Python): see `docs/tooling.md` + `scripts/check-host-t
|
|
|
13
13
|
|
|
14
14
|
| Legacy script | SDK target | Status |
|
|
15
15
|
|---|---|---|
|
|
16
|
-
| `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported.**
|
|
16
|
+
| `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported + aligned (2026-08).** Database-scoped: `--quick` = all tables COUNT(*); `--full` = text LENGTH + briefing. Removed `QUICK_TABLE_RE` / `SKIP_COL_RE` / `--count-only`. Deferred: `--email`/Resend (CLI/ops layer). |
|
|
17
17
|
| `scripts/run-d1-bloat-audit.sh` | `agentsam data d1-bloat` via CLI | **Shimmed**, see below — legacy `npm run audit:d1-bloat*` scripts still work unchanged. GCP-fallback/nohup wrapper behavior not reimplemented in the SDK itself (that's operational, not tool logic); still available via the legacy `.sh`. |
|
|
18
18
|
| `scripts/walk_agentsam_tables.py` | `agentsam_sdk.data.agentsam_walk` | **Ported, condensed.** Schema/indexes/FKs/row-count/freshness/capability-grouping all present. Not byte-for-byte: duplicate-table detection and some staleness heuristics from the 801-line original are deferred. |
|
|
19
19
|
| `scripts/d1_schema_audit.py` | folded into `agentsam_walk` (schema slice) | **Partially folded.** The capability-grouping + schema dump is covered by `agentsam_walk`. NOT ported: the per-feature markdown chunking into 14 separate `db/agentsam-*.md` files, and the curated `TABLE_META` purpose annotations (760+ lines of hand-written table descriptions) -- that's product documentation content, not audit logic, and belongs in a follow-up pass, not this one. Also note: the legacy script hardcoded a D1 database id as a fallback default (`D1_DATABASE_ID = os.environ.get("D1_DATABASE_ID", "cf87b717-...")`) -- **do not carry that forward**; the new adapter has no such fallback (HARD LAW). |
|
|
@@ -2,42 +2,73 @@
|
|
|
2
2
|
import unittest
|
|
3
3
|
|
|
4
4
|
from agentsam_sdk.data.d1_bloat import (
|
|
5
|
-
ColStat,
|
|
5
|
+
ColStat,
|
|
6
|
+
TableStat,
|
|
7
|
+
_build_briefing,
|
|
8
|
+
_build_doing_well,
|
|
9
|
+
_build_findings,
|
|
10
|
+
_fmt_bytes,
|
|
11
|
+
_pick_measure_columns,
|
|
6
12
|
)
|
|
7
13
|
|
|
8
14
|
|
|
9
|
-
class
|
|
15
|
+
class TestMeasureColumnPicking(unittest.TestCase):
|
|
10
16
|
def test_picks_json_and_body_columns(self):
|
|
11
|
-
cols = [
|
|
12
|
-
|
|
17
|
+
cols = [
|
|
18
|
+
("id", "INTEGER"),
|
|
19
|
+
("input_json", "TEXT"),
|
|
20
|
+
("output_json", "TEXT"),
|
|
21
|
+
("created_at", "TEXT"),
|
|
22
|
+
]
|
|
23
|
+
picked = _pick_measure_columns(cols)
|
|
13
24
|
self.assertIn("input_json", picked)
|
|
14
25
|
self.assertIn("output_json", picked)
|
|
26
|
+
# id is INTEGER — never measured; created_at is TEXT but not payload-ish
|
|
27
|
+
# when preferred names exist, only preferred are kept
|
|
15
28
|
self.assertNotIn("id", picked)
|
|
16
29
|
self.assertNotIn("created_at", picked)
|
|
17
30
|
|
|
18
|
-
def
|
|
31
|
+
def test_falls_back_to_any_text_when_no_preferred(self):
|
|
32
|
+
cols = [("tenant_id", "TEXT"), ("workspace_id", "TEXT"), ("label", "TEXT")]
|
|
33
|
+
picked = _pick_measure_columns(cols)
|
|
34
|
+
# No SKIP_COL_RE — database-scoped; without preferred names, all text cols qualify
|
|
35
|
+
self.assertIn("tenant_id", picked)
|
|
36
|
+
self.assertIn("workspace_id", picked)
|
|
37
|
+
self.assertIn("label", picked)
|
|
38
|
+
|
|
39
|
+
def test_prefers_metadata_over_ids_when_mixed(self):
|
|
19
40
|
cols = [("tenant_id", "TEXT"), ("workspace_id", "TEXT"), ("metadata", "TEXT")]
|
|
20
|
-
picked =
|
|
21
|
-
self.
|
|
22
|
-
self.assertNotIn("workspace_id", picked)
|
|
23
|
-
self.assertIn("metadata", picked)
|
|
41
|
+
picked = _pick_measure_columns(cols)
|
|
42
|
+
self.assertEqual(picked, ["metadata"])
|
|
24
43
|
|
|
25
44
|
|
|
26
|
-
class
|
|
27
|
-
def
|
|
28
|
-
big = TableStat(
|
|
29
|
-
|
|
45
|
+
class TestFindings(unittest.TestCase):
|
|
46
|
+
def test_full_flags_large_table_high_severity(self):
|
|
47
|
+
big = TableStat(
|
|
48
|
+
name="agentsam_tool_call_log",
|
|
49
|
+
row_count=250_000,
|
|
50
|
+
text_bytes=18_874_368,
|
|
51
|
+
est_bytes=18_874_368,
|
|
52
|
+
columns=[ColStat(name="output_json", bytes=12_582_912)],
|
|
53
|
+
)
|
|
30
54
|
small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
|
|
31
|
-
|
|
32
|
-
names = {f["table"] for f in
|
|
55
|
+
findings = _build_findings([big, small], "full")
|
|
56
|
+
names = {f["table"] for f in findings}
|
|
33
57
|
self.assertIn("agentsam_tool_call_log", names)
|
|
34
|
-
big_flag = next(f for f in
|
|
58
|
+
big_flag = next(f for f in findings if f["table"] == "agentsam_tool_call_log")
|
|
35
59
|
self.assertEqual(big_flag["severity"], "high")
|
|
36
60
|
|
|
37
|
-
def
|
|
61
|
+
def test_full_empty_table_not_flagged(self):
|
|
38
62
|
small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
|
|
39
|
-
|
|
40
|
-
self.assertEqual(
|
|
63
|
+
findings = _build_findings([small], "full")
|
|
64
|
+
self.assertEqual(findings, [])
|
|
65
|
+
|
|
66
|
+
def test_quick_flags_high_row_count(self):
|
|
67
|
+
big = TableStat(name="otlp_traces", row_count=120_000, est_bytes=120_000 * 120)
|
|
68
|
+
findings = _build_findings([big], "quick")
|
|
69
|
+
self.assertEqual(len(findings), 1)
|
|
70
|
+
self.assertEqual(findings[0]["severity"], "high")
|
|
71
|
+
self.assertIn("--full", findings[0]["why"])
|
|
41
72
|
|
|
42
73
|
|
|
43
74
|
class TestFormatting(unittest.TestCase):
|
|
@@ -46,11 +77,16 @@ class TestFormatting(unittest.TestCase):
|
|
|
46
77
|
self.assertEqual(_fmt_bytes(2048), "2.0 KB")
|
|
47
78
|
self.assertEqual(_fmt_bytes(5 * 1024 * 1024), "5.00 MB")
|
|
48
79
|
|
|
49
|
-
def
|
|
50
|
-
stats = [
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
80
|
+
def test_briefing_includes_table_and_verdict(self):
|
|
81
|
+
stats = [
|
|
82
|
+
TableStat(name="agentsam_memory", row_count=40, text_bytes=12000, est_bytes=12000)
|
|
83
|
+
]
|
|
84
|
+
findings = _build_findings(stats, "quick")
|
|
85
|
+
well = _build_doing_well(stats, findings)
|
|
86
|
+
md = _build_briefing(stats, findings, well, "inneranimalmedia-business", "1.2 MB", "quick", 1)
|
|
87
|
+
self.assertIn("D1 health", md)
|
|
88
|
+
self.assertIn("database-scoped", md)
|
|
89
|
+
self.assertIn("Verdict", md)
|
|
54
90
|
|
|
55
91
|
|
|
56
92
|
if __name__ == "__main__":
|