@inneranimalmedia/agentsam-sdk 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/DEVELOPMENT.md +25 -5
- package/README.md +2 -0
- package/docs/RELEASES.iam-mirror.md +20 -0
- package/docs/RELEASES.md +9 -0
- package/package.json +9 -4
- package/protocol/README.md +51 -0
- package/protocol/dual-repo-sync.md +35 -0
- package/python/README.md +12 -0
- package/python/agentsam_sdk/__init__.py +9 -0
- package/python/agentsam_sdk/cli.py +262 -0
- package/python/agentsam_sdk/data/__init__.py +0 -0
- package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
- package/python/agentsam_sdk/data/d1_adapter.py +124 -0
- package/python/agentsam_sdk/data/d1_bloat.py +445 -0
- package/python/agentsam_sdk/repository/__init__.py +26 -0
- package/python/agentsam_sdk/repository/__main__.py +3 -0
- package/python/agentsam_sdk/repository/inspect.py +496 -0
- package/python/agentsam_sdk/repository/inventory.py +351 -0
- package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
- package/python/agentsam_sdk/runtime/__init__.py +0 -0
- package/python/agentsam_sdk/runtime/contract.py +105 -0
- package/python/docs/gaps.md +63 -0
- package/python/docs/tooling.md +67 -0
- package/python/protocol/README.md +51 -0
- package/python/protocol/dual-repo-sync.md +35 -0
- package/python/pyproject.toml +16 -0
- package/python/scripts/check-host-tooling.sh +65 -0
- package/python/tests/__init__.py +0 -0
- package/python/tests/fixtures/sample_tables.json +17 -0
- package/python/tests/fixtures.py +95 -0
- package/python/tests/test_agentsam_walk.py +31 -0
- package/python/tests/test_contract.py +32 -0
- package/python/tests/test_d1_bloat.py +93 -0
- package/python/tests/test_repository_inspect.py +84 -0
- package/python/tests/test_repository_inventory.py +53 -0
- package/python/tests/test_scan_bloat.py +31 -0
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
"""agentsam_sdk.repository.inspect — repo walk (size + dates) + optional content dupes.
|
|
2
|
+
|
|
3
|
+
Reusable library: no print(), no sys.exit(), no hardcoded tenant/workspace ids.
|
|
4
|
+
Root path is always an explicit parameter.
|
|
5
|
+
|
|
6
|
+
from agentsam_sdk.repository.inspect import walk_repo, summarize, find_dupes, build_report
|
|
7
|
+
|
|
8
|
+
CLI:
|
|
9
|
+
python3 scripts/repo_inspect.py --text
|
|
10
|
+
python3 -m agentsam_sdk.repository.inspect --json --dupes
|
|
11
|
+
agentsam repository inspect --repo-root . --format json --dupes
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import hashlib
|
|
16
|
+
import os
|
|
17
|
+
import subprocess
|
|
18
|
+
from collections import defaultdict
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Iterable
|
|
22
|
+
|
|
23
|
+
SCHEMA_VERSION = 1
|
|
24
|
+
TOOL_NAME = "repository.inspect"
|
|
25
|
+
HASH_CHUNK = 1024 * 1024
|
|
26
|
+
|
|
27
|
+
DEFAULT_SKIP_DIR_NAMES = frozenset(
|
|
28
|
+
{
|
|
29
|
+
".git",
|
|
30
|
+
".scratch",
|
|
31
|
+
".venv",
|
|
32
|
+
".venv_agentsam",
|
|
33
|
+
"venv",
|
|
34
|
+
"node_modules",
|
|
35
|
+
"__pycache__",
|
|
36
|
+
".pytest_cache",
|
|
37
|
+
".mypy_cache",
|
|
38
|
+
".ruff_cache",
|
|
39
|
+
".turbo",
|
|
40
|
+
".next",
|
|
41
|
+
"dist",
|
|
42
|
+
"build",
|
|
43
|
+
"coverage",
|
|
44
|
+
".wrangler",
|
|
45
|
+
"vendor",
|
|
46
|
+
"captures",
|
|
47
|
+
"architecture-map",
|
|
48
|
+
}
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class RepoRootError(Exception):
|
|
53
|
+
"""Raised when a git toplevel cannot be resolved."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def utc_now() -> datetime:
|
|
57
|
+
return datetime.now(timezone.utc)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def iso_from_unix(ts: float | int | None) -> str | None:
|
|
61
|
+
if ts is None:
|
|
62
|
+
return None
|
|
63
|
+
try:
|
|
64
|
+
return datetime.fromtimestamp(float(ts), tz=timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
65
|
+
except (OSError, OverflowError, ValueError):
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def parse_since(raw: str | None, *, now_unix: int | None = None) -> int | None:
|
|
70
|
+
"""Return min mtime_unix, or None for no filter. Raises ValueError on bad input."""
|
|
71
|
+
if raw is None or str(raw).strip() == "":
|
|
72
|
+
return None
|
|
73
|
+
s = str(raw).strip().lower()
|
|
74
|
+
now = int(now_unix if now_unix is not None else utc_now().timestamp())
|
|
75
|
+
if s.endswith("d") and s[:-1].isdigit():
|
|
76
|
+
return now - int(s[:-1]) * 86400
|
|
77
|
+
if s.endswith("h") and s[:-1].isdigit():
|
|
78
|
+
return now - int(s[:-1]) * 3600
|
|
79
|
+
if s.isdigit():
|
|
80
|
+
return now - int(s)
|
|
81
|
+
raise ValueError(f"bad --since {raw!r} (use Nd, Nh, or seconds)")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def find_repo_root(start: Path | None = None) -> Path:
|
|
85
|
+
"""Resolve git toplevel from start (default: cwd). Raises RepoRootError."""
|
|
86
|
+
here = (start or Path.cwd()).resolve()
|
|
87
|
+
try:
|
|
88
|
+
out = subprocess.check_output(
|
|
89
|
+
["git", "rev-parse", "--show-toplevel"],
|
|
90
|
+
cwd=here,
|
|
91
|
+
text=True,
|
|
92
|
+
stderr=subprocess.DEVNULL,
|
|
93
|
+
).strip()
|
|
94
|
+
if out:
|
|
95
|
+
return Path(out).resolve()
|
|
96
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
97
|
+
pass
|
|
98
|
+
for p in [here, *here.parents]:
|
|
99
|
+
if (p / ".git").exists():
|
|
100
|
+
return p
|
|
101
|
+
raise RepoRootError(f"not inside a git repository (start={here})")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def git_head(repo_root: Path) -> dict[str, Any]:
|
|
105
|
+
def _run(*args: str) -> str:
|
|
106
|
+
try:
|
|
107
|
+
return subprocess.check_output(
|
|
108
|
+
["git", *args],
|
|
109
|
+
cwd=repo_root,
|
|
110
|
+
text=True,
|
|
111
|
+
stderr=subprocess.DEVNULL,
|
|
112
|
+
).strip()
|
|
113
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
114
|
+
return ""
|
|
115
|
+
|
|
116
|
+
sha = _run("rev-parse", "HEAD")
|
|
117
|
+
branch = _run("branch", "--show-current") or _run("rev-parse", "--abbrev-ref", "HEAD")
|
|
118
|
+
subject = _run("log", "-1", "--format=%s")
|
|
119
|
+
author_unix = _run("log", "-1", "--format=%at")
|
|
120
|
+
author_unix_i = int(author_unix) if author_unix.isdigit() else None
|
|
121
|
+
return {
|
|
122
|
+
"branch": branch or None,
|
|
123
|
+
"head_sha": sha if len(sha) == 40 else (sha or None),
|
|
124
|
+
"head_subject": subject or None,
|
|
125
|
+
"head_author_unix": author_unix_i,
|
|
126
|
+
"head_author_iso": iso_from_unix(author_unix_i),
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def file_times(st: os.stat_result) -> dict[str, Any]:
|
|
131
|
+
mtime = int(st.st_mtime)
|
|
132
|
+
ctime = int(st.st_ctime)
|
|
133
|
+
birth = None
|
|
134
|
+
birth_raw = getattr(st, "st_birthtime", None)
|
|
135
|
+
if birth_raw is not None:
|
|
136
|
+
try:
|
|
137
|
+
birth = int(birth_raw)
|
|
138
|
+
except (TypeError, ValueError):
|
|
139
|
+
birth = None
|
|
140
|
+
return {
|
|
141
|
+
"size_bytes": int(st.st_size),
|
|
142
|
+
"mtime_unix": mtime,
|
|
143
|
+
"mtime_iso": iso_from_unix(mtime),
|
|
144
|
+
"ctime_unix": ctime,
|
|
145
|
+
"ctime_iso": iso_from_unix(ctime),
|
|
146
|
+
"birth_unix": birth,
|
|
147
|
+
"birth_iso": iso_from_unix(birth),
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def walk_repo(
|
|
152
|
+
root: Path,
|
|
153
|
+
*,
|
|
154
|
+
skip_dir_names: Iterable[str] | None = None,
|
|
155
|
+
respect_gitignore: bool = True,
|
|
156
|
+
follow_symlinks: bool = False,
|
|
157
|
+
errors: list[dict[str, str]] | None = None,
|
|
158
|
+
) -> list[dict[str, Any]]:
|
|
159
|
+
"""Walk root and return file rows (path relative to root, sizes, dates).
|
|
160
|
+
|
|
161
|
+
respect_gitignore=True applies DEFAULT_SKIP_DIR_NAMES (heavy/generated dirs).
|
|
162
|
+
Full .gitignore parsing is not implemented — pass skip_dir_names to customize.
|
|
163
|
+
Non-fatal OSErrors append to errors when provided; otherwise they are skipped.
|
|
164
|
+
"""
|
|
165
|
+
root = Path(root).resolve()
|
|
166
|
+
skip = set(DEFAULT_SKIP_DIR_NAMES if respect_gitignore else ())
|
|
167
|
+
if skip_dir_names is not None:
|
|
168
|
+
skip |= {str(s) for s in skip_dir_names}
|
|
169
|
+
|
|
170
|
+
rows: list[dict[str, Any]] = []
|
|
171
|
+
err_sink = errors if errors is not None else []
|
|
172
|
+
|
|
173
|
+
for dirpath, dirnames, filenames in os.walk(root, topdown=True, followlinks=follow_symlinks):
|
|
174
|
+
dirnames[:] = sorted(
|
|
175
|
+
d for d in dirnames if d not in skip and not d.startswith(".cache")
|
|
176
|
+
)
|
|
177
|
+
base = Path(dirpath)
|
|
178
|
+
for name in filenames:
|
|
179
|
+
if name == ".DS_Store":
|
|
180
|
+
continue
|
|
181
|
+
path = base / name
|
|
182
|
+
try:
|
|
183
|
+
if path.is_symlink() and not follow_symlinks:
|
|
184
|
+
continue
|
|
185
|
+
st = path.stat()
|
|
186
|
+
except OSError as e:
|
|
187
|
+
err_sink.append({"path": str(path), "error": f"stat:{e}"})
|
|
188
|
+
continue
|
|
189
|
+
if not path.is_file():
|
|
190
|
+
continue
|
|
191
|
+
try:
|
|
192
|
+
rel = path.relative_to(root).as_posix()
|
|
193
|
+
except ValueError as e:
|
|
194
|
+
err_sink.append({"path": str(path), "error": f"relative:{e}"})
|
|
195
|
+
continue
|
|
196
|
+
top = rel.split("/", 1)[0] if "/" in rel else "(root)"
|
|
197
|
+
ext = path.suffix.lower() or "(none)"
|
|
198
|
+
rows.append(
|
|
199
|
+
{
|
|
200
|
+
"path": rel,
|
|
201
|
+
"top_dir": top,
|
|
202
|
+
"ext": ext,
|
|
203
|
+
**file_times(st),
|
|
204
|
+
}
|
|
205
|
+
)
|
|
206
|
+
return rows
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def summarize(
|
|
210
|
+
files: list[dict[str, Any]],
|
|
211
|
+
*,
|
|
212
|
+
recent_n: int = 50,
|
|
213
|
+
largest_n: int = 30,
|
|
214
|
+
since_unix: int | None = None,
|
|
215
|
+
) -> dict[str, Any]:
|
|
216
|
+
"""Rollups + recent/largest slices from walk_repo rows."""
|
|
217
|
+
by_dir: dict[str, dict[str, int]] = defaultdict(lambda: {"files": 0, "bytes": 0})
|
|
218
|
+
total_bytes = 0
|
|
219
|
+
for f in files:
|
|
220
|
+
total_bytes += int(f.get("size_bytes") or 0)
|
|
221
|
+
bucket = by_dir[str(f.get("top_dir") or "(root)")]
|
|
222
|
+
bucket["files"] += 1
|
|
223
|
+
bucket["bytes"] += int(f.get("size_bytes") or 0)
|
|
224
|
+
|
|
225
|
+
filtered = files
|
|
226
|
+
if since_unix is not None:
|
|
227
|
+
filtered = [f for f in files if int(f.get("mtime_unix") or 0) >= since_unix]
|
|
228
|
+
|
|
229
|
+
recent = sorted(filtered, key=lambda r: (-int(r.get("mtime_unix") or 0), r.get("path") or ""))[
|
|
230
|
+
: max(1, recent_n)
|
|
231
|
+
]
|
|
232
|
+
largest = sorted(files, key=lambda r: (-int(r.get("size_bytes") or 0), r.get("path") or ""))[
|
|
233
|
+
: max(1, largest_n)
|
|
234
|
+
]
|
|
235
|
+
top_dirs = sorted(
|
|
236
|
+
({"top_dir": k, "files": v["files"], "bytes": v["bytes"]} for k, v in by_dir.items()),
|
|
237
|
+
key=lambda r: (-r["bytes"], r["top_dir"]),
|
|
238
|
+
)
|
|
239
|
+
return {
|
|
240
|
+
"file_count": len(files),
|
|
241
|
+
"total_bytes": total_bytes,
|
|
242
|
+
"since_unix": since_unix,
|
|
243
|
+
"recent_count": len(recent),
|
|
244
|
+
"by_top_dir": top_dirs,
|
|
245
|
+
"recent": recent,
|
|
246
|
+
"largest": largest,
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def sha256_file(path: Path, *, chunk_size: int = HASH_CHUNK) -> str:
|
|
251
|
+
h = hashlib.sha256()
|
|
252
|
+
with path.open("rb") as fh:
|
|
253
|
+
while True:
|
|
254
|
+
chunk = fh.read(chunk_size)
|
|
255
|
+
if not chunk:
|
|
256
|
+
break
|
|
257
|
+
h.update(chunk)
|
|
258
|
+
return h.hexdigest()
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def find_dupes(
|
|
262
|
+
files: list[dict[str, Any]],
|
|
263
|
+
*,
|
|
264
|
+
repo_root: Path,
|
|
265
|
+
errors: list[dict[str, str]] | None = None,
|
|
266
|
+
warnings: list[str] | None = None,
|
|
267
|
+
) -> list[dict[str, Any]]:
|
|
268
|
+
"""True content duplicates: same size_bytes + SHA-256.
|
|
269
|
+
|
|
270
|
+
Returns groups sorted by size_bytes descending. Each group:
|
|
271
|
+
size_bytes, sha256, count, wasted_bytes (= size * (count-1)), paths[]
|
|
272
|
+
"""
|
|
273
|
+
root = Path(repo_root).resolve()
|
|
274
|
+
err_sink = errors if errors is not None else []
|
|
275
|
+
warn_sink = warnings if warnings is not None else []
|
|
276
|
+
|
|
277
|
+
by_size: dict[int, list[dict[str, Any]]] = defaultdict(list)
|
|
278
|
+
for f in files:
|
|
279
|
+
by_size[int(f.get("size_bytes") or 0)].append(f)
|
|
280
|
+
|
|
281
|
+
groups: list[dict[str, Any]] = []
|
|
282
|
+
for size, bucket in by_size.items():
|
|
283
|
+
if len(bucket) < 2:
|
|
284
|
+
continue
|
|
285
|
+
by_hash: dict[str, list[str]] = defaultdict(list)
|
|
286
|
+
for f in bucket:
|
|
287
|
+
rel = str(f.get("path") or "")
|
|
288
|
+
abs_path = root / rel
|
|
289
|
+
try:
|
|
290
|
+
digest = sha256_file(abs_path)
|
|
291
|
+
except OSError as e:
|
|
292
|
+
msg = f"hash skip {rel}: {e}"
|
|
293
|
+
warn_sink.append(msg)
|
|
294
|
+
err_sink.append({"path": rel, "error": f"hash:{e}"})
|
|
295
|
+
continue
|
|
296
|
+
by_hash[digest].append(rel)
|
|
297
|
+
|
|
298
|
+
for digest, paths in by_hash.items():
|
|
299
|
+
if len(paths) < 2:
|
|
300
|
+
continue
|
|
301
|
+
paths_sorted = sorted(paths)
|
|
302
|
+
count = len(paths_sorted)
|
|
303
|
+
groups.append(
|
|
304
|
+
{
|
|
305
|
+
"size_bytes": size,
|
|
306
|
+
"sha256": digest,
|
|
307
|
+
"count": count,
|
|
308
|
+
"wasted_bytes": size * (count - 1),
|
|
309
|
+
"paths": paths_sorted,
|
|
310
|
+
}
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
groups.sort(key=lambda g: (-int(g["size_bytes"]), -int(g["count"]), g["sha256"]))
|
|
314
|
+
return groups
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def build_report(
|
|
318
|
+
repo_root: Path,
|
|
319
|
+
*,
|
|
320
|
+
recent_n: int = 50,
|
|
321
|
+
largest_n: int = 30,
|
|
322
|
+
since_unix: int | None = None,
|
|
323
|
+
include_all: bool = False,
|
|
324
|
+
include_dupes: bool = False,
|
|
325
|
+
skip_dir_names: Iterable[str] | None = None,
|
|
326
|
+
respect_gitignore: bool = True,
|
|
327
|
+
) -> dict[str, Any]:
|
|
328
|
+
"""Full jq-stable report dict (summary/recent/largest keys preserved)."""
|
|
329
|
+
root = Path(repo_root).resolve()
|
|
330
|
+
walk_errors: list[dict[str, str]] = []
|
|
331
|
+
files = walk_repo(
|
|
332
|
+
root,
|
|
333
|
+
skip_dir_names=skip_dir_names,
|
|
334
|
+
respect_gitignore=respect_gitignore,
|
|
335
|
+
errors=walk_errors,
|
|
336
|
+
)
|
|
337
|
+
rollup = summarize(
|
|
338
|
+
files,
|
|
339
|
+
recent_n=recent_n,
|
|
340
|
+
largest_n=largest_n,
|
|
341
|
+
since_unix=since_unix,
|
|
342
|
+
)
|
|
343
|
+
now = int(utc_now().timestamp())
|
|
344
|
+
report: dict[str, Any] = {
|
|
345
|
+
"schema_version": SCHEMA_VERSION,
|
|
346
|
+
"tool": "repo_inspect",
|
|
347
|
+
"repo_root": str(root),
|
|
348
|
+
"repo_name": root.name,
|
|
349
|
+
"generated_at_unix": now,
|
|
350
|
+
"generated_at_iso": iso_from_unix(now),
|
|
351
|
+
"git": git_head(root),
|
|
352
|
+
"summary": {
|
|
353
|
+
"file_count": rollup["file_count"],
|
|
354
|
+
"total_bytes": rollup["total_bytes"],
|
|
355
|
+
"since_unix": rollup["since_unix"],
|
|
356
|
+
"recent_count": rollup["recent_count"],
|
|
357
|
+
"by_top_dir": rollup["by_top_dir"],
|
|
358
|
+
},
|
|
359
|
+
"recent": rollup["recent"],
|
|
360
|
+
"largest": rollup["largest"],
|
|
361
|
+
"jq": {
|
|
362
|
+
"summary": ".summary",
|
|
363
|
+
"recent": ".recent[] | {path, size_bytes, mtime_iso}",
|
|
364
|
+
"largest": ".largest[] | {path, size_bytes}",
|
|
365
|
+
"by_dir": ".summary.by_top_dir[]",
|
|
366
|
+
"changed_today": f".recent[] | select(.mtime_unix >= {now - 86400})",
|
|
367
|
+
"duplicates": ".duplicates[] | {size_bytes, wasted_bytes, count, paths}",
|
|
368
|
+
},
|
|
369
|
+
}
|
|
370
|
+
if include_all:
|
|
371
|
+
report["files"] = sorted(files, key=lambda r: r["path"])
|
|
372
|
+
if walk_errors:
|
|
373
|
+
report["walk_errors"] = walk_errors
|
|
374
|
+
|
|
375
|
+
if include_dupes:
|
|
376
|
+
hash_errors: list[dict[str, str]] = []
|
|
377
|
+
hash_warnings: list[str] = []
|
|
378
|
+
dupes = find_dupes(files, repo_root=root, errors=hash_errors, warnings=hash_warnings)
|
|
379
|
+
report["duplicates"] = dupes
|
|
380
|
+
report["summary"]["duplicate_groups"] = len(dupes)
|
|
381
|
+
report["summary"]["duplicate_wasted_bytes"] = sum(int(g["wasted_bytes"]) for g in dupes)
|
|
382
|
+
if hash_warnings:
|
|
383
|
+
report["hash_warnings"] = hash_warnings
|
|
384
|
+
if hash_errors:
|
|
385
|
+
report["hash_errors"] = hash_errors
|
|
386
|
+
return report
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def render_text(report: dict[str, Any]) -> str:
|
|
390
|
+
g = report.get("git") or {}
|
|
391
|
+
s = report.get("summary") or {}
|
|
392
|
+
lines = [
|
|
393
|
+
f"repo_inspect {report.get('repo_name')} @ {report.get('generated_at_iso')}",
|
|
394
|
+
f"git {g.get('branch') or '?'} {str(g.get('head_sha') or '')[:12]} {g.get('head_subject') or ''}",
|
|
395
|
+
f"files {int(s.get('file_count') or 0):,} bytes {int(s.get('total_bytes') or 0):,}",
|
|
396
|
+
"",
|
|
397
|
+
"recent (mtime):",
|
|
398
|
+
]
|
|
399
|
+
for f in (report.get("recent") or [])[:25]:
|
|
400
|
+
lines.append(f" {f.get('mtime_iso')} {int(f.get('size_bytes') or 0):>10,} {f.get('path')}")
|
|
401
|
+
lines.append("")
|
|
402
|
+
lines.append("largest:")
|
|
403
|
+
for f in (report.get("largest") or [])[:15]:
|
|
404
|
+
lines.append(f" {int(f.get('size_bytes') or 0):>12,} {f.get('path')}")
|
|
405
|
+
lines.append("")
|
|
406
|
+
lines.append("top dirs:")
|
|
407
|
+
for d in (s.get("by_top_dir") or [])[:12]:
|
|
408
|
+
lines.append(
|
|
409
|
+
f" {int(d.get('bytes') or 0):>12,} {int(d.get('files') or 0):>6} files {d.get('top_dir')}"
|
|
410
|
+
)
|
|
411
|
+
dupes = report.get("duplicates")
|
|
412
|
+
if isinstance(dupes, list):
|
|
413
|
+
lines.append("")
|
|
414
|
+
wasted = int(s.get("duplicate_wasted_bytes") or 0)
|
|
415
|
+
lines.append(f"duplicates: {len(dupes)} group(s), wasted {wasted:,} bytes")
|
|
416
|
+
for gdup in dupes[:20]:
|
|
417
|
+
lines.append(
|
|
418
|
+
f" size={int(gdup.get('size_bytes') or 0):,} "
|
|
419
|
+
f"count={int(gdup.get('count') or 0)} "
|
|
420
|
+
f"wasted={int(gdup.get('wasted_bytes') or 0):,} "
|
|
421
|
+
f"sha256={str(gdup.get('sha256') or '')[:12]}…"
|
|
422
|
+
)
|
|
423
|
+
for pth in (gdup.get("paths") or [])[:8]:
|
|
424
|
+
lines.append(f" {pth}")
|
|
425
|
+
lines.append("")
|
|
426
|
+
lines.append("jq: python3 scripts/repo_inspect.py --json | jq '.recent[0:10]'")
|
|
427
|
+
lines.append("dupes: python3 scripts/repo_inspect.py --json --dupes | jq '.duplicates'")
|
|
428
|
+
return "\n".join(lines) + "\n"
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def main_cli(argv: list[str] | None = None) -> int:
|
|
432
|
+
"""Argparse entry for `python -m agentsam_sdk.repository.inspect` and shims."""
|
|
433
|
+
import argparse
|
|
434
|
+
import json
|
|
435
|
+
import sys
|
|
436
|
+
|
|
437
|
+
p = argparse.ArgumentParser(description="Canonical repo file inspect (size + dates)")
|
|
438
|
+
p.add_argument("--repo-root", default=None, help="Repo root (default: git toplevel)")
|
|
439
|
+
p.add_argument("--json", action="store_true", help="Emit JSON (default when not --text)")
|
|
440
|
+
p.add_argument("--text", action="store_true", help="Human briefing on stdout")
|
|
441
|
+
p.add_argument("--recent", type=int, default=50, help="How many recent files (mtime)")
|
|
442
|
+
p.add_argument("--largest", type=int, default=30, help="How many largest files")
|
|
443
|
+
p.add_argument("--since", default=None, help="Only recent[] after window (e.g. 7d, 24h)")
|
|
444
|
+
p.add_argument("--all", action="store_true", help="Include full files[] array")
|
|
445
|
+
p.add_argument(
|
|
446
|
+
"--dupes",
|
|
447
|
+
action="store_true",
|
|
448
|
+
help="SHA-256 content duplicate groups (expensive; off by default)",
|
|
449
|
+
)
|
|
450
|
+
p.add_argument(
|
|
451
|
+
"--out",
|
|
452
|
+
default=None,
|
|
453
|
+
help="Write JSON to path (also prints text/json to stdout per flags)",
|
|
454
|
+
)
|
|
455
|
+
args = p.parse_args(argv)
|
|
456
|
+
|
|
457
|
+
try:
|
|
458
|
+
repo = Path(args.repo_root).resolve() if args.repo_root else find_repo_root()
|
|
459
|
+
except RepoRootError as e:
|
|
460
|
+
print(str(e), file=sys.stderr)
|
|
461
|
+
return 2
|
|
462
|
+
|
|
463
|
+
try:
|
|
464
|
+
since_unix = parse_since(args.since)
|
|
465
|
+
except ValueError as e:
|
|
466
|
+
print(str(e), file=sys.stderr)
|
|
467
|
+
return 2
|
|
468
|
+
|
|
469
|
+
report = build_report(
|
|
470
|
+
repo,
|
|
471
|
+
recent_n=max(1, args.recent),
|
|
472
|
+
largest_n=max(1, args.largest),
|
|
473
|
+
since_unix=since_unix,
|
|
474
|
+
include_all=bool(args.all),
|
|
475
|
+
include_dupes=bool(args.dupes),
|
|
476
|
+
)
|
|
477
|
+
|
|
478
|
+
if args.out:
|
|
479
|
+
out_path = Path(args.out)
|
|
480
|
+
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
481
|
+
out_path.write_text(json.dumps(report, indent=2, sort_keys=False) + "\n", encoding="utf-8")
|
|
482
|
+
|
|
483
|
+
if args.dupes:
|
|
484
|
+
for w in report.get("hash_warnings") or []:
|
|
485
|
+
print(f"warning: {w}", file=sys.stderr)
|
|
486
|
+
|
|
487
|
+
if args.text and not args.json:
|
|
488
|
+
sys.stdout.write(render_text(report))
|
|
489
|
+
else:
|
|
490
|
+
json.dump(report, sys.stdout, indent=2, sort_keys=False)
|
|
491
|
+
sys.stdout.write("\n")
|
|
492
|
+
return 0
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
if __name__ == "__main__":
|
|
496
|
+
raise SystemExit(main_cli())
|