@inneranimalmedia/agentsam-sdk 1.7.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/DEVELOPMENT.md +25 -5
  2. package/README.md +2 -0
  3. package/docs/RELEASES.iam-mirror.md +20 -0
  4. package/docs/RELEASES.md +9 -0
  5. package/package.json +9 -4
  6. package/protocol/README.md +51 -0
  7. package/protocol/dual-repo-sync.md +35 -0
  8. package/python/README.md +12 -0
  9. package/python/agentsam_sdk/__init__.py +9 -0
  10. package/python/agentsam_sdk/cli.py +262 -0
  11. package/python/agentsam_sdk/data/__init__.py +0 -0
  12. package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
  13. package/python/agentsam_sdk/data/d1_adapter.py +124 -0
  14. package/python/agentsam_sdk/data/d1_bloat.py +445 -0
  15. package/python/agentsam_sdk/repository/__init__.py +26 -0
  16. package/python/agentsam_sdk/repository/__main__.py +3 -0
  17. package/python/agentsam_sdk/repository/inspect.py +496 -0
  18. package/python/agentsam_sdk/repository/inventory.py +351 -0
  19. package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
  20. package/python/agentsam_sdk/runtime/__init__.py +0 -0
  21. package/python/agentsam_sdk/runtime/contract.py +105 -0
  22. package/python/docs/gaps.md +63 -0
  23. package/python/docs/tooling.md +67 -0
  24. package/python/protocol/README.md +51 -0
  25. package/python/protocol/dual-repo-sync.md +35 -0
  26. package/python/pyproject.toml +16 -0
  27. package/python/scripts/check-host-tooling.sh +65 -0
  28. package/python/tests/__init__.py +0 -0
  29. package/python/tests/fixtures/sample_tables.json +17 -0
  30. package/python/tests/fixtures.py +95 -0
  31. package/python/tests/test_agentsam_walk.py +31 -0
  32. package/python/tests/test_contract.py +32 -0
  33. package/python/tests/test_d1_bloat.py +93 -0
  34. package/python/tests/test_repository_inspect.py +84 -0
  35. package/python/tests/test_repository_inventory.py +53 -0
  36. package/python/tests/test_scan_bloat.py +31 -0
@@ -0,0 +1,496 @@
1
+ """agentsam_sdk.repository.inspect — repo walk (size + dates) + optional content dupes.
2
+
3
+ Reusable library: no print(), no sys.exit(), no hardcoded tenant/workspace ids.
4
+ Root path is always an explicit parameter.
5
+
6
+ from agentsam_sdk.repository.inspect import walk_repo, summarize, find_dupes, build_report
7
+
8
+ CLI:
9
+ python3 scripts/repo_inspect.py --text
10
+ python3 -m agentsam_sdk.repository.inspect --json --dupes
11
+ agentsam repository inspect --repo-root . --format json --dupes
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import hashlib
16
+ import os
17
+ import subprocess
18
+ from collections import defaultdict
19
+ from datetime import datetime, timezone
20
+ from pathlib import Path
21
+ from typing import Any, Iterable
22
+
23
+ SCHEMA_VERSION = 1
24
+ TOOL_NAME = "repository.inspect"
25
+ HASH_CHUNK = 1024 * 1024
26
+
27
+ DEFAULT_SKIP_DIR_NAMES = frozenset(
28
+ {
29
+ ".git",
30
+ ".scratch",
31
+ ".venv",
32
+ ".venv_agentsam",
33
+ "venv",
34
+ "node_modules",
35
+ "__pycache__",
36
+ ".pytest_cache",
37
+ ".mypy_cache",
38
+ ".ruff_cache",
39
+ ".turbo",
40
+ ".next",
41
+ "dist",
42
+ "build",
43
+ "coverage",
44
+ ".wrangler",
45
+ "vendor",
46
+ "captures",
47
+ "architecture-map",
48
+ }
49
+ )
50
+
51
+
52
+ class RepoRootError(Exception):
53
+ """Raised when a git toplevel cannot be resolved."""
54
+
55
+
56
+ def utc_now() -> datetime:
57
+ return datetime.now(timezone.utc)
58
+
59
+
60
+ def iso_from_unix(ts: float | int | None) -> str | None:
61
+ if ts is None:
62
+ return None
63
+ try:
64
+ return datetime.fromtimestamp(float(ts), tz=timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
65
+ except (OSError, OverflowError, ValueError):
66
+ return None
67
+
68
+
69
+ def parse_since(raw: str | None, *, now_unix: int | None = None) -> int | None:
70
+ """Return min mtime_unix, or None for no filter. Raises ValueError on bad input."""
71
+ if raw is None or str(raw).strip() == "":
72
+ return None
73
+ s = str(raw).strip().lower()
74
+ now = int(now_unix if now_unix is not None else utc_now().timestamp())
75
+ if s.endswith("d") and s[:-1].isdigit():
76
+ return now - int(s[:-1]) * 86400
77
+ if s.endswith("h") and s[:-1].isdigit():
78
+ return now - int(s[:-1]) * 3600
79
+ if s.isdigit():
80
+ return now - int(s)
81
+ raise ValueError(f"bad --since {raw!r} (use Nd, Nh, or seconds)")
82
+
83
+
84
+ def find_repo_root(start: Path | None = None) -> Path:
85
+ """Resolve git toplevel from start (default: cwd). Raises RepoRootError."""
86
+ here = (start or Path.cwd()).resolve()
87
+ try:
88
+ out = subprocess.check_output(
89
+ ["git", "rev-parse", "--show-toplevel"],
90
+ cwd=here,
91
+ text=True,
92
+ stderr=subprocess.DEVNULL,
93
+ ).strip()
94
+ if out:
95
+ return Path(out).resolve()
96
+ except (subprocess.CalledProcessError, FileNotFoundError):
97
+ pass
98
+ for p in [here, *here.parents]:
99
+ if (p / ".git").exists():
100
+ return p
101
+ raise RepoRootError(f"not inside a git repository (start={here})")
102
+
103
+
104
+ def git_head(repo_root: Path) -> dict[str, Any]:
105
+ def _run(*args: str) -> str:
106
+ try:
107
+ return subprocess.check_output(
108
+ ["git", *args],
109
+ cwd=repo_root,
110
+ text=True,
111
+ stderr=subprocess.DEVNULL,
112
+ ).strip()
113
+ except (subprocess.CalledProcessError, FileNotFoundError):
114
+ return ""
115
+
116
+ sha = _run("rev-parse", "HEAD")
117
+ branch = _run("branch", "--show-current") or _run("rev-parse", "--abbrev-ref", "HEAD")
118
+ subject = _run("log", "-1", "--format=%s")
119
+ author_unix = _run("log", "-1", "--format=%at")
120
+ author_unix_i = int(author_unix) if author_unix.isdigit() else None
121
+ return {
122
+ "branch": branch or None,
123
+ "head_sha": sha if len(sha) == 40 else (sha or None),
124
+ "head_subject": subject or None,
125
+ "head_author_unix": author_unix_i,
126
+ "head_author_iso": iso_from_unix(author_unix_i),
127
+ }
128
+
129
+
130
+ def file_times(st: os.stat_result) -> dict[str, Any]:
131
+ mtime = int(st.st_mtime)
132
+ ctime = int(st.st_ctime)
133
+ birth = None
134
+ birth_raw = getattr(st, "st_birthtime", None)
135
+ if birth_raw is not None:
136
+ try:
137
+ birth = int(birth_raw)
138
+ except (TypeError, ValueError):
139
+ birth = None
140
+ return {
141
+ "size_bytes": int(st.st_size),
142
+ "mtime_unix": mtime,
143
+ "mtime_iso": iso_from_unix(mtime),
144
+ "ctime_unix": ctime,
145
+ "ctime_iso": iso_from_unix(ctime),
146
+ "birth_unix": birth,
147
+ "birth_iso": iso_from_unix(birth),
148
+ }
149
+
150
+
151
+ def walk_repo(
152
+ root: Path,
153
+ *,
154
+ skip_dir_names: Iterable[str] | None = None,
155
+ respect_gitignore: bool = True,
156
+ follow_symlinks: bool = False,
157
+ errors: list[dict[str, str]] | None = None,
158
+ ) -> list[dict[str, Any]]:
159
+ """Walk root and return file rows (path relative to root, sizes, dates).
160
+
161
+ respect_gitignore=True applies DEFAULT_SKIP_DIR_NAMES (heavy/generated dirs).
162
+ Full .gitignore parsing is not implemented — pass skip_dir_names to customize.
163
+ Non-fatal OSErrors append to errors when provided; otherwise they are skipped.
164
+ """
165
+ root = Path(root).resolve()
166
+ skip = set(DEFAULT_SKIP_DIR_NAMES if respect_gitignore else ())
167
+ if skip_dir_names is not None:
168
+ skip |= {str(s) for s in skip_dir_names}
169
+
170
+ rows: list[dict[str, Any]] = []
171
+ err_sink = errors if errors is not None else []
172
+
173
+ for dirpath, dirnames, filenames in os.walk(root, topdown=True, followlinks=follow_symlinks):
174
+ dirnames[:] = sorted(
175
+ d for d in dirnames if d not in skip and not d.startswith(".cache")
176
+ )
177
+ base = Path(dirpath)
178
+ for name in filenames:
179
+ if name == ".DS_Store":
180
+ continue
181
+ path = base / name
182
+ try:
183
+ if path.is_symlink() and not follow_symlinks:
184
+ continue
185
+ st = path.stat()
186
+ except OSError as e:
187
+ err_sink.append({"path": str(path), "error": f"stat:{e}"})
188
+ continue
189
+ if not path.is_file():
190
+ continue
191
+ try:
192
+ rel = path.relative_to(root).as_posix()
193
+ except ValueError as e:
194
+ err_sink.append({"path": str(path), "error": f"relative:{e}"})
195
+ continue
196
+ top = rel.split("/", 1)[0] if "/" in rel else "(root)"
197
+ ext = path.suffix.lower() or "(none)"
198
+ rows.append(
199
+ {
200
+ "path": rel,
201
+ "top_dir": top,
202
+ "ext": ext,
203
+ **file_times(st),
204
+ }
205
+ )
206
+ return rows
207
+
208
+
209
+ def summarize(
210
+ files: list[dict[str, Any]],
211
+ *,
212
+ recent_n: int = 50,
213
+ largest_n: int = 30,
214
+ since_unix: int | None = None,
215
+ ) -> dict[str, Any]:
216
+ """Rollups + recent/largest slices from walk_repo rows."""
217
+ by_dir: dict[str, dict[str, int]] = defaultdict(lambda: {"files": 0, "bytes": 0})
218
+ total_bytes = 0
219
+ for f in files:
220
+ total_bytes += int(f.get("size_bytes") or 0)
221
+ bucket = by_dir[str(f.get("top_dir") or "(root)")]
222
+ bucket["files"] += 1
223
+ bucket["bytes"] += int(f.get("size_bytes") or 0)
224
+
225
+ filtered = files
226
+ if since_unix is not None:
227
+ filtered = [f for f in files if int(f.get("mtime_unix") or 0) >= since_unix]
228
+
229
+ recent = sorted(filtered, key=lambda r: (-int(r.get("mtime_unix") or 0), r.get("path") or ""))[
230
+ : max(1, recent_n)
231
+ ]
232
+ largest = sorted(files, key=lambda r: (-int(r.get("size_bytes") or 0), r.get("path") or ""))[
233
+ : max(1, largest_n)
234
+ ]
235
+ top_dirs = sorted(
236
+ ({"top_dir": k, "files": v["files"], "bytes": v["bytes"]} for k, v in by_dir.items()),
237
+ key=lambda r: (-r["bytes"], r["top_dir"]),
238
+ )
239
+ return {
240
+ "file_count": len(files),
241
+ "total_bytes": total_bytes,
242
+ "since_unix": since_unix,
243
+ "recent_count": len(recent),
244
+ "by_top_dir": top_dirs,
245
+ "recent": recent,
246
+ "largest": largest,
247
+ }
248
+
249
+
250
+ def sha256_file(path: Path, *, chunk_size: int = HASH_CHUNK) -> str:
251
+ h = hashlib.sha256()
252
+ with path.open("rb") as fh:
253
+ while True:
254
+ chunk = fh.read(chunk_size)
255
+ if not chunk:
256
+ break
257
+ h.update(chunk)
258
+ return h.hexdigest()
259
+
260
+
261
+ def find_dupes(
262
+ files: list[dict[str, Any]],
263
+ *,
264
+ repo_root: Path,
265
+ errors: list[dict[str, str]] | None = None,
266
+ warnings: list[str] | None = None,
267
+ ) -> list[dict[str, Any]]:
268
+ """True content duplicates: same size_bytes + SHA-256.
269
+
270
+ Returns groups sorted by size_bytes descending. Each group:
271
+ size_bytes, sha256, count, wasted_bytes (= size * (count-1)), paths[]
272
+ """
273
+ root = Path(repo_root).resolve()
274
+ err_sink = errors if errors is not None else []
275
+ warn_sink = warnings if warnings is not None else []
276
+
277
+ by_size: dict[int, list[dict[str, Any]]] = defaultdict(list)
278
+ for f in files:
279
+ by_size[int(f.get("size_bytes") or 0)].append(f)
280
+
281
+ groups: list[dict[str, Any]] = []
282
+ for size, bucket in by_size.items():
283
+ if len(bucket) < 2:
284
+ continue
285
+ by_hash: dict[str, list[str]] = defaultdict(list)
286
+ for f in bucket:
287
+ rel = str(f.get("path") or "")
288
+ abs_path = root / rel
289
+ try:
290
+ digest = sha256_file(abs_path)
291
+ except OSError as e:
292
+ msg = f"hash skip {rel}: {e}"
293
+ warn_sink.append(msg)
294
+ err_sink.append({"path": rel, "error": f"hash:{e}"})
295
+ continue
296
+ by_hash[digest].append(rel)
297
+
298
+ for digest, paths in by_hash.items():
299
+ if len(paths) < 2:
300
+ continue
301
+ paths_sorted = sorted(paths)
302
+ count = len(paths_sorted)
303
+ groups.append(
304
+ {
305
+ "size_bytes": size,
306
+ "sha256": digest,
307
+ "count": count,
308
+ "wasted_bytes": size * (count - 1),
309
+ "paths": paths_sorted,
310
+ }
311
+ )
312
+
313
+ groups.sort(key=lambda g: (-int(g["size_bytes"]), -int(g["count"]), g["sha256"]))
314
+ return groups
315
+
316
+
317
+ def build_report(
318
+ repo_root: Path,
319
+ *,
320
+ recent_n: int = 50,
321
+ largest_n: int = 30,
322
+ since_unix: int | None = None,
323
+ include_all: bool = False,
324
+ include_dupes: bool = False,
325
+ skip_dir_names: Iterable[str] | None = None,
326
+ respect_gitignore: bool = True,
327
+ ) -> dict[str, Any]:
328
+ """Full jq-stable report dict (summary/recent/largest keys preserved)."""
329
+ root = Path(repo_root).resolve()
330
+ walk_errors: list[dict[str, str]] = []
331
+ files = walk_repo(
332
+ root,
333
+ skip_dir_names=skip_dir_names,
334
+ respect_gitignore=respect_gitignore,
335
+ errors=walk_errors,
336
+ )
337
+ rollup = summarize(
338
+ files,
339
+ recent_n=recent_n,
340
+ largest_n=largest_n,
341
+ since_unix=since_unix,
342
+ )
343
+ now = int(utc_now().timestamp())
344
+ report: dict[str, Any] = {
345
+ "schema_version": SCHEMA_VERSION,
346
+ "tool": "repo_inspect",
347
+ "repo_root": str(root),
348
+ "repo_name": root.name,
349
+ "generated_at_unix": now,
350
+ "generated_at_iso": iso_from_unix(now),
351
+ "git": git_head(root),
352
+ "summary": {
353
+ "file_count": rollup["file_count"],
354
+ "total_bytes": rollup["total_bytes"],
355
+ "since_unix": rollup["since_unix"],
356
+ "recent_count": rollup["recent_count"],
357
+ "by_top_dir": rollup["by_top_dir"],
358
+ },
359
+ "recent": rollup["recent"],
360
+ "largest": rollup["largest"],
361
+ "jq": {
362
+ "summary": ".summary",
363
+ "recent": ".recent[] | {path, size_bytes, mtime_iso}",
364
+ "largest": ".largest[] | {path, size_bytes}",
365
+ "by_dir": ".summary.by_top_dir[]",
366
+ "changed_today": f".recent[] | select(.mtime_unix >= {now - 86400})",
367
+ "duplicates": ".duplicates[] | {size_bytes, wasted_bytes, count, paths}",
368
+ },
369
+ }
370
+ if include_all:
371
+ report["files"] = sorted(files, key=lambda r: r["path"])
372
+ if walk_errors:
373
+ report["walk_errors"] = walk_errors
374
+
375
+ if include_dupes:
376
+ hash_errors: list[dict[str, str]] = []
377
+ hash_warnings: list[str] = []
378
+ dupes = find_dupes(files, repo_root=root, errors=hash_errors, warnings=hash_warnings)
379
+ report["duplicates"] = dupes
380
+ report["summary"]["duplicate_groups"] = len(dupes)
381
+ report["summary"]["duplicate_wasted_bytes"] = sum(int(g["wasted_bytes"]) for g in dupes)
382
+ if hash_warnings:
383
+ report["hash_warnings"] = hash_warnings
384
+ if hash_errors:
385
+ report["hash_errors"] = hash_errors
386
+ return report
387
+
388
+
389
+ def render_text(report: dict[str, Any]) -> str:
390
+ g = report.get("git") or {}
391
+ s = report.get("summary") or {}
392
+ lines = [
393
+ f"repo_inspect {report.get('repo_name')} @ {report.get('generated_at_iso')}",
394
+ f"git {g.get('branch') or '?'} {str(g.get('head_sha') or '')[:12]} {g.get('head_subject') or ''}",
395
+ f"files {int(s.get('file_count') or 0):,} bytes {int(s.get('total_bytes') or 0):,}",
396
+ "",
397
+ "recent (mtime):",
398
+ ]
399
+ for f in (report.get("recent") or [])[:25]:
400
+ lines.append(f" {f.get('mtime_iso')} {int(f.get('size_bytes') or 0):>10,} {f.get('path')}")
401
+ lines.append("")
402
+ lines.append("largest:")
403
+ for f in (report.get("largest") or [])[:15]:
404
+ lines.append(f" {int(f.get('size_bytes') or 0):>12,} {f.get('path')}")
405
+ lines.append("")
406
+ lines.append("top dirs:")
407
+ for d in (s.get("by_top_dir") or [])[:12]:
408
+ lines.append(
409
+ f" {int(d.get('bytes') or 0):>12,} {int(d.get('files') or 0):>6} files {d.get('top_dir')}"
410
+ )
411
+ dupes = report.get("duplicates")
412
+ if isinstance(dupes, list):
413
+ lines.append("")
414
+ wasted = int(s.get("duplicate_wasted_bytes") or 0)
415
+ lines.append(f"duplicates: {len(dupes)} group(s), wasted {wasted:,} bytes")
416
+ for gdup in dupes[:20]:
417
+ lines.append(
418
+ f" size={int(gdup.get('size_bytes') or 0):,} "
419
+ f"count={int(gdup.get('count') or 0)} "
420
+ f"wasted={int(gdup.get('wasted_bytes') or 0):,} "
421
+ f"sha256={str(gdup.get('sha256') or '')[:12]}…"
422
+ )
423
+ for pth in (gdup.get("paths") or [])[:8]:
424
+ lines.append(f" {pth}")
425
+ lines.append("")
426
+ lines.append("jq: python3 scripts/repo_inspect.py --json | jq '.recent[0:10]'")
427
+ lines.append("dupes: python3 scripts/repo_inspect.py --json --dupes | jq '.duplicates'")
428
+ return "\n".join(lines) + "\n"
429
+
430
+
431
+ def main_cli(argv: list[str] | None = None) -> int:
432
+ """Argparse entry for `python -m agentsam_sdk.repository.inspect` and shims."""
433
+ import argparse
434
+ import json
435
+ import sys
436
+
437
+ p = argparse.ArgumentParser(description="Canonical repo file inspect (size + dates)")
438
+ p.add_argument("--repo-root", default=None, help="Repo root (default: git toplevel)")
439
+ p.add_argument("--json", action="store_true", help="Emit JSON (default when not --text)")
440
+ p.add_argument("--text", action="store_true", help="Human briefing on stdout")
441
+ p.add_argument("--recent", type=int, default=50, help="How many recent files (mtime)")
442
+ p.add_argument("--largest", type=int, default=30, help="How many largest files")
443
+ p.add_argument("--since", default=None, help="Only recent[] after window (e.g. 7d, 24h)")
444
+ p.add_argument("--all", action="store_true", help="Include full files[] array")
445
+ p.add_argument(
446
+ "--dupes",
447
+ action="store_true",
448
+ help="SHA-256 content duplicate groups (expensive; off by default)",
449
+ )
450
+ p.add_argument(
451
+ "--out",
452
+ default=None,
453
+ help="Write JSON to path (also prints text/json to stdout per flags)",
454
+ )
455
+ args = p.parse_args(argv)
456
+
457
+ try:
458
+ repo = Path(args.repo_root).resolve() if args.repo_root else find_repo_root()
459
+ except RepoRootError as e:
460
+ print(str(e), file=sys.stderr)
461
+ return 2
462
+
463
+ try:
464
+ since_unix = parse_since(args.since)
465
+ except ValueError as e:
466
+ print(str(e), file=sys.stderr)
467
+ return 2
468
+
469
+ report = build_report(
470
+ repo,
471
+ recent_n=max(1, args.recent),
472
+ largest_n=max(1, args.largest),
473
+ since_unix=since_unix,
474
+ include_all=bool(args.all),
475
+ include_dupes=bool(args.dupes),
476
+ )
477
+
478
+ if args.out:
479
+ out_path = Path(args.out)
480
+ out_path.parent.mkdir(parents=True, exist_ok=True)
481
+ out_path.write_text(json.dumps(report, indent=2, sort_keys=False) + "\n", encoding="utf-8")
482
+
483
+ if args.dupes:
484
+ for w in report.get("hash_warnings") or []:
485
+ print(f"warning: {w}", file=sys.stderr)
486
+
487
+ if args.text and not args.json:
488
+ sys.stdout.write(render_text(report))
489
+ else:
490
+ json.dump(report, sys.stdout, indent=2, sort_keys=False)
491
+ sys.stdout.write("\n")
492
+ return 0
493
+
494
+
495
+ if __name__ == "__main__":
496
+ raise SystemExit(main_cli())