@inneranimalmedia/agentsam-sdk 1.8.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ """Repository-level audits (inventory, scan_bloat, inspect).
2
+
3
+ Lazy exports so `python -m agentsam_sdk.repository.inspect` does not warn.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from typing import Any
9
+
10
+ __all__ = ["inventory", "scan_bloat", "inspect"]
11
+
12
+
13
+ def __getattr__(name: str) -> Any:
14
+ if name == "inventory":
15
+ from agentsam_sdk.repository import inventory as mod
16
+
17
+ return mod
18
+ if name == "scan_bloat":
19
+ from agentsam_sdk.repository import scan_bloat as mod
20
+
21
+ return mod
22
+ if name == "inspect":
23
+ from agentsam_sdk.repository import inspect as mod
24
+
25
+ return mod
26
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,496 @@
1
+ """agentsam_sdk.repository.inspect — repo walk (size + dates) + optional content dupes.
2
+
3
+ Reusable library: no print(), no sys.exit(), no hardcoded tenant/workspace ids.
4
+ Root path is always an explicit parameter.
5
+
6
+ from agentsam_sdk.repository.inspect import walk_repo, summarize, find_dupes, build_report
7
+
8
+ CLI:
9
+ python3 scripts/repo_inspect.py --text
10
+ python3 -m agentsam_sdk.repository.inspect --json --dupes
11
+ agentsam repository inspect --repo-root . --format json --dupes
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import hashlib
16
+ import os
17
+ import subprocess
18
+ from collections import defaultdict
19
+ from datetime import datetime, timezone
20
+ from pathlib import Path
21
+ from typing import Any, Iterable
22
+
23
+ SCHEMA_VERSION = 1
24
+ TOOL_NAME = "repository.inspect"
25
+ HASH_CHUNK = 1024 * 1024
26
+
27
+ DEFAULT_SKIP_DIR_NAMES = frozenset(
28
+ {
29
+ ".git",
30
+ ".scratch",
31
+ ".venv",
32
+ ".venv_agentsam",
33
+ "venv",
34
+ "node_modules",
35
+ "__pycache__",
36
+ ".pytest_cache",
37
+ ".mypy_cache",
38
+ ".ruff_cache",
39
+ ".turbo",
40
+ ".next",
41
+ "dist",
42
+ "build",
43
+ "coverage",
44
+ ".wrangler",
45
+ "vendor",
46
+ "captures",
47
+ "architecture-map",
48
+ }
49
+ )
50
+
51
+
52
+ class RepoRootError(Exception):
53
+ """Raised when a git toplevel cannot be resolved."""
54
+
55
+
56
+ def utc_now() -> datetime:
57
+ return datetime.now(timezone.utc)
58
+
59
+
60
+ def iso_from_unix(ts: float | int | None) -> str | None:
61
+ if ts is None:
62
+ return None
63
+ try:
64
+ return datetime.fromtimestamp(float(ts), tz=timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
65
+ except (OSError, OverflowError, ValueError):
66
+ return None
67
+
68
+
69
+ def parse_since(raw: str | None, *, now_unix: int | None = None) -> int | None:
70
+ """Return min mtime_unix, or None for no filter. Raises ValueError on bad input."""
71
+ if raw is None or str(raw).strip() == "":
72
+ return None
73
+ s = str(raw).strip().lower()
74
+ now = int(now_unix if now_unix is not None else utc_now().timestamp())
75
+ if s.endswith("d") and s[:-1].isdigit():
76
+ return now - int(s[:-1]) * 86400
77
+ if s.endswith("h") and s[:-1].isdigit():
78
+ return now - int(s[:-1]) * 3600
79
+ if s.isdigit():
80
+ return now - int(s)
81
+ raise ValueError(f"bad --since {raw!r} (use Nd, Nh, or seconds)")
82
+
83
+
84
+ def find_repo_root(start: Path | None = None) -> Path:
85
+ """Resolve git toplevel from start (default: cwd). Raises RepoRootError."""
86
+ here = (start or Path.cwd()).resolve()
87
+ try:
88
+ out = subprocess.check_output(
89
+ ["git", "rev-parse", "--show-toplevel"],
90
+ cwd=here,
91
+ text=True,
92
+ stderr=subprocess.DEVNULL,
93
+ ).strip()
94
+ if out:
95
+ return Path(out).resolve()
96
+ except (subprocess.CalledProcessError, FileNotFoundError):
97
+ pass
98
+ for p in [here, *here.parents]:
99
+ if (p / ".git").exists():
100
+ return p
101
+ raise RepoRootError(f"not inside a git repository (start={here})")
102
+
103
+
104
+ def git_head(repo_root: Path) -> dict[str, Any]:
105
+ def _run(*args: str) -> str:
106
+ try:
107
+ return subprocess.check_output(
108
+ ["git", *args],
109
+ cwd=repo_root,
110
+ text=True,
111
+ stderr=subprocess.DEVNULL,
112
+ ).strip()
113
+ except (subprocess.CalledProcessError, FileNotFoundError):
114
+ return ""
115
+
116
+ sha = _run("rev-parse", "HEAD")
117
+ branch = _run("branch", "--show-current") or _run("rev-parse", "--abbrev-ref", "HEAD")
118
+ subject = _run("log", "-1", "--format=%s")
119
+ author_unix = _run("log", "-1", "--format=%at")
120
+ author_unix_i = int(author_unix) if author_unix.isdigit() else None
121
+ return {
122
+ "branch": branch or None,
123
+ "head_sha": sha if len(sha) == 40 else (sha or None),
124
+ "head_subject": subject or None,
125
+ "head_author_unix": author_unix_i,
126
+ "head_author_iso": iso_from_unix(author_unix_i),
127
+ }
128
+
129
+
130
+ def file_times(st: os.stat_result) -> dict[str, Any]:
131
+ mtime = int(st.st_mtime)
132
+ ctime = int(st.st_ctime)
133
+ birth = None
134
+ birth_raw = getattr(st, "st_birthtime", None)
135
+ if birth_raw is not None:
136
+ try:
137
+ birth = int(birth_raw)
138
+ except (TypeError, ValueError):
139
+ birth = None
140
+ return {
141
+ "size_bytes": int(st.st_size),
142
+ "mtime_unix": mtime,
143
+ "mtime_iso": iso_from_unix(mtime),
144
+ "ctime_unix": ctime,
145
+ "ctime_iso": iso_from_unix(ctime),
146
+ "birth_unix": birth,
147
+ "birth_iso": iso_from_unix(birth),
148
+ }
149
+
150
+
151
+ def walk_repo(
152
+ root: Path,
153
+ *,
154
+ skip_dir_names: Iterable[str] | None = None,
155
+ respect_gitignore: bool = True,
156
+ follow_symlinks: bool = False,
157
+ errors: list[dict[str, str]] | None = None,
158
+ ) -> list[dict[str, Any]]:
159
+ """Walk root and return file rows (path relative to root, sizes, dates).
160
+
161
+ respect_gitignore=True applies DEFAULT_SKIP_DIR_NAMES (heavy/generated dirs).
162
+ Full .gitignore parsing is not implemented — pass skip_dir_names to customize.
163
+ Non-fatal OSErrors append to errors when provided; otherwise they are skipped.
164
+ """
165
+ root = Path(root).resolve()
166
+ skip = set(DEFAULT_SKIP_DIR_NAMES if respect_gitignore else ())
167
+ if skip_dir_names is not None:
168
+ skip |= {str(s) for s in skip_dir_names}
169
+
170
+ rows: list[dict[str, Any]] = []
171
+ err_sink = errors if errors is not None else []
172
+
173
+ for dirpath, dirnames, filenames in os.walk(root, topdown=True, followlinks=follow_symlinks):
174
+ dirnames[:] = sorted(
175
+ d for d in dirnames if d not in skip and not d.startswith(".cache")
176
+ )
177
+ base = Path(dirpath)
178
+ for name in filenames:
179
+ if name == ".DS_Store":
180
+ continue
181
+ path = base / name
182
+ try:
183
+ if path.is_symlink() and not follow_symlinks:
184
+ continue
185
+ st = path.stat()
186
+ except OSError as e:
187
+ err_sink.append({"path": str(path), "error": f"stat:{e}"})
188
+ continue
189
+ if not path.is_file():
190
+ continue
191
+ try:
192
+ rel = path.relative_to(root).as_posix()
193
+ except ValueError as e:
194
+ err_sink.append({"path": str(path), "error": f"relative:{e}"})
195
+ continue
196
+ top = rel.split("/", 1)[0] if "/" in rel else "(root)"
197
+ ext = path.suffix.lower() or "(none)"
198
+ rows.append(
199
+ {
200
+ "path": rel,
201
+ "top_dir": top,
202
+ "ext": ext,
203
+ **file_times(st),
204
+ }
205
+ )
206
+ return rows
207
+
208
+
209
+ def summarize(
210
+ files: list[dict[str, Any]],
211
+ *,
212
+ recent_n: int = 50,
213
+ largest_n: int = 30,
214
+ since_unix: int | None = None,
215
+ ) -> dict[str, Any]:
216
+ """Rollups + recent/largest slices from walk_repo rows."""
217
+ by_dir: dict[str, dict[str, int]] = defaultdict(lambda: {"files": 0, "bytes": 0})
218
+ total_bytes = 0
219
+ for f in files:
220
+ total_bytes += int(f.get("size_bytes") or 0)
221
+ bucket = by_dir[str(f.get("top_dir") or "(root)")]
222
+ bucket["files"] += 1
223
+ bucket["bytes"] += int(f.get("size_bytes") or 0)
224
+
225
+ filtered = files
226
+ if since_unix is not None:
227
+ filtered = [f for f in files if int(f.get("mtime_unix") or 0) >= since_unix]
228
+
229
+ recent = sorted(filtered, key=lambda r: (-int(r.get("mtime_unix") or 0), r.get("path") or ""))[
230
+ : max(1, recent_n)
231
+ ]
232
+ largest = sorted(files, key=lambda r: (-int(r.get("size_bytes") or 0), r.get("path") or ""))[
233
+ : max(1, largest_n)
234
+ ]
235
+ top_dirs = sorted(
236
+ ({"top_dir": k, "files": v["files"], "bytes": v["bytes"]} for k, v in by_dir.items()),
237
+ key=lambda r: (-r["bytes"], r["top_dir"]),
238
+ )
239
+ return {
240
+ "file_count": len(files),
241
+ "total_bytes": total_bytes,
242
+ "since_unix": since_unix,
243
+ "recent_count": len(recent),
244
+ "by_top_dir": top_dirs,
245
+ "recent": recent,
246
+ "largest": largest,
247
+ }
248
+
249
+
250
+ def sha256_file(path: Path, *, chunk_size: int = HASH_CHUNK) -> str:
251
+ h = hashlib.sha256()
252
+ with path.open("rb") as fh:
253
+ while True:
254
+ chunk = fh.read(chunk_size)
255
+ if not chunk:
256
+ break
257
+ h.update(chunk)
258
+ return h.hexdigest()
259
+
260
+
261
+ def find_dupes(
262
+ files: list[dict[str, Any]],
263
+ *,
264
+ repo_root: Path,
265
+ errors: list[dict[str, str]] | None = None,
266
+ warnings: list[str] | None = None,
267
+ ) -> list[dict[str, Any]]:
268
+ """True content duplicates: same size_bytes + SHA-256.
269
+
270
+ Returns groups sorted by size_bytes descending. Each group:
271
+ size_bytes, sha256, count, wasted_bytes (= size * (count-1)), paths[]
272
+ """
273
+ root = Path(repo_root).resolve()
274
+ err_sink = errors if errors is not None else []
275
+ warn_sink = warnings if warnings is not None else []
276
+
277
+ by_size: dict[int, list[dict[str, Any]]] = defaultdict(list)
278
+ for f in files:
279
+ by_size[int(f.get("size_bytes") or 0)].append(f)
280
+
281
+ groups: list[dict[str, Any]] = []
282
+ for size, bucket in by_size.items():
283
+ if len(bucket) < 2:
284
+ continue
285
+ by_hash: dict[str, list[str]] = defaultdict(list)
286
+ for f in bucket:
287
+ rel = str(f.get("path") or "")
288
+ abs_path = root / rel
289
+ try:
290
+ digest = sha256_file(abs_path)
291
+ except OSError as e:
292
+ msg = f"hash skip {rel}: {e}"
293
+ warn_sink.append(msg)
294
+ err_sink.append({"path": rel, "error": f"hash:{e}"})
295
+ continue
296
+ by_hash[digest].append(rel)
297
+
298
+ for digest, paths in by_hash.items():
299
+ if len(paths) < 2:
300
+ continue
301
+ paths_sorted = sorted(paths)
302
+ count = len(paths_sorted)
303
+ groups.append(
304
+ {
305
+ "size_bytes": size,
306
+ "sha256": digest,
307
+ "count": count,
308
+ "wasted_bytes": size * (count - 1),
309
+ "paths": paths_sorted,
310
+ }
311
+ )
312
+
313
+ groups.sort(key=lambda g: (-int(g["size_bytes"]), -int(g["count"]), g["sha256"]))
314
+ return groups
315
+
316
+
317
+ def build_report(
318
+ repo_root: Path,
319
+ *,
320
+ recent_n: int = 50,
321
+ largest_n: int = 30,
322
+ since_unix: int | None = None,
323
+ include_all: bool = False,
324
+ include_dupes: bool = False,
325
+ skip_dir_names: Iterable[str] | None = None,
326
+ respect_gitignore: bool = True,
327
+ ) -> dict[str, Any]:
328
+ """Full jq-stable report dict (summary/recent/largest keys preserved)."""
329
+ root = Path(repo_root).resolve()
330
+ walk_errors: list[dict[str, str]] = []
331
+ files = walk_repo(
332
+ root,
333
+ skip_dir_names=skip_dir_names,
334
+ respect_gitignore=respect_gitignore,
335
+ errors=walk_errors,
336
+ )
337
+ rollup = summarize(
338
+ files,
339
+ recent_n=recent_n,
340
+ largest_n=largest_n,
341
+ since_unix=since_unix,
342
+ )
343
+ now = int(utc_now().timestamp())
344
+ report: dict[str, Any] = {
345
+ "schema_version": SCHEMA_VERSION,
346
+ "tool": "repo_inspect",
347
+ "repo_root": str(root),
348
+ "repo_name": root.name,
349
+ "generated_at_unix": now,
350
+ "generated_at_iso": iso_from_unix(now),
351
+ "git": git_head(root),
352
+ "summary": {
353
+ "file_count": rollup["file_count"],
354
+ "total_bytes": rollup["total_bytes"],
355
+ "since_unix": rollup["since_unix"],
356
+ "recent_count": rollup["recent_count"],
357
+ "by_top_dir": rollup["by_top_dir"],
358
+ },
359
+ "recent": rollup["recent"],
360
+ "largest": rollup["largest"],
361
+ "jq": {
362
+ "summary": ".summary",
363
+ "recent": ".recent[] | {path, size_bytes, mtime_iso}",
364
+ "largest": ".largest[] | {path, size_bytes}",
365
+ "by_dir": ".summary.by_top_dir[]",
366
+ "changed_today": f".recent[] | select(.mtime_unix >= {now - 86400})",
367
+ "duplicates": ".duplicates[] | {size_bytes, wasted_bytes, count, paths}",
368
+ },
369
+ }
370
+ if include_all:
371
+ report["files"] = sorted(files, key=lambda r: r["path"])
372
+ if walk_errors:
373
+ report["walk_errors"] = walk_errors
374
+
375
+ if include_dupes:
376
+ hash_errors: list[dict[str, str]] = []
377
+ hash_warnings: list[str] = []
378
+ dupes = find_dupes(files, repo_root=root, errors=hash_errors, warnings=hash_warnings)
379
+ report["duplicates"] = dupes
380
+ report["summary"]["duplicate_groups"] = len(dupes)
381
+ report["summary"]["duplicate_wasted_bytes"] = sum(int(g["wasted_bytes"]) for g in dupes)
382
+ if hash_warnings:
383
+ report["hash_warnings"] = hash_warnings
384
+ if hash_errors:
385
+ report["hash_errors"] = hash_errors
386
+ return report
387
+
388
+
389
+ def render_text(report: dict[str, Any]) -> str:
390
+ g = report.get("git") or {}
391
+ s = report.get("summary") or {}
392
+ lines = [
393
+ f"repo_inspect {report.get('repo_name')} @ {report.get('generated_at_iso')}",
394
+ f"git {g.get('branch') or '?'} {str(g.get('head_sha') or '')[:12]} {g.get('head_subject') or ''}",
395
+ f"files {int(s.get('file_count') or 0):,} bytes {int(s.get('total_bytes') or 0):,}",
396
+ "",
397
+ "recent (mtime):",
398
+ ]
399
+ for f in (report.get("recent") or [])[:25]:
400
+ lines.append(f" {f.get('mtime_iso')} {int(f.get('size_bytes') or 0):>10,} {f.get('path')}")
401
+ lines.append("")
402
+ lines.append("largest:")
403
+ for f in (report.get("largest") or [])[:15]:
404
+ lines.append(f" {int(f.get('size_bytes') or 0):>12,} {f.get('path')}")
405
+ lines.append("")
406
+ lines.append("top dirs:")
407
+ for d in (s.get("by_top_dir") or [])[:12]:
408
+ lines.append(
409
+ f" {int(d.get('bytes') or 0):>12,} {int(d.get('files') or 0):>6} files {d.get('top_dir')}"
410
+ )
411
+ dupes = report.get("duplicates")
412
+ if isinstance(dupes, list):
413
+ lines.append("")
414
+ wasted = int(s.get("duplicate_wasted_bytes") or 0)
415
+ lines.append(f"duplicates: {len(dupes)} group(s), wasted {wasted:,} bytes")
416
+ for gdup in dupes[:20]:
417
+ lines.append(
418
+ f" size={int(gdup.get('size_bytes') or 0):,} "
419
+ f"count={int(gdup.get('count') or 0)} "
420
+ f"wasted={int(gdup.get('wasted_bytes') or 0):,} "
421
+ f"sha256={str(gdup.get('sha256') or '')[:12]}…"
422
+ )
423
+ for pth in (gdup.get("paths") or [])[:8]:
424
+ lines.append(f" {pth}")
425
+ lines.append("")
426
+ lines.append("jq: python3 scripts/repo_inspect.py --json | jq '.recent[0:10]'")
427
+ lines.append("dupes: python3 scripts/repo_inspect.py --json --dupes | jq '.duplicates'")
428
+ return "\n".join(lines) + "\n"
429
+
430
+
431
+ def main_cli(argv: list[str] | None = None) -> int:
432
+ """Argparse entry for `python -m agentsam_sdk.repository.inspect` and shims."""
433
+ import argparse
434
+ import json
435
+ import sys
436
+
437
+ p = argparse.ArgumentParser(description="Canonical repo file inspect (size + dates)")
438
+ p.add_argument("--repo-root", default=None, help="Repo root (default: git toplevel)")
439
+ p.add_argument("--json", action="store_true", help="Emit JSON (default when not --text)")
440
+ p.add_argument("--text", action="store_true", help="Human briefing on stdout")
441
+ p.add_argument("--recent", type=int, default=50, help="How many recent files (mtime)")
442
+ p.add_argument("--largest", type=int, default=30, help="How many largest files")
443
+ p.add_argument("--since", default=None, help="Only recent[] after window (e.g. 7d, 24h)")
444
+ p.add_argument("--all", action="store_true", help="Include full files[] array")
445
+ p.add_argument(
446
+ "--dupes",
447
+ action="store_true",
448
+ help="SHA-256 content duplicate groups (expensive; off by default)",
449
+ )
450
+ p.add_argument(
451
+ "--out",
452
+ default=None,
453
+ help="Write JSON to path (also prints text/json to stdout per flags)",
454
+ )
455
+ args = p.parse_args(argv)
456
+
457
+ try:
458
+ repo = Path(args.repo_root).resolve() if args.repo_root else find_repo_root()
459
+ except RepoRootError as e:
460
+ print(str(e), file=sys.stderr)
461
+ return 2
462
+
463
+ try:
464
+ since_unix = parse_since(args.since)
465
+ except ValueError as e:
466
+ print(str(e), file=sys.stderr)
467
+ return 2
468
+
469
+ report = build_report(
470
+ repo,
471
+ recent_n=max(1, args.recent),
472
+ largest_n=max(1, args.largest),
473
+ since_unix=since_unix,
474
+ include_all=bool(args.all),
475
+ include_dupes=bool(args.dupes),
476
+ )
477
+
478
+ if args.out:
479
+ out_path = Path(args.out)
480
+ out_path.parent.mkdir(parents=True, exist_ok=True)
481
+ out_path.write_text(json.dumps(report, indent=2, sort_keys=False) + "\n", encoding="utf-8")
482
+
483
+ if args.dupes:
484
+ for w in report.get("hash_warnings") or []:
485
+ print(f"warning: {w}", file=sys.stderr)
486
+
487
+ if args.text and not args.json:
488
+ sys.stdout.write(render_text(report))
489
+ else:
490
+ json.dump(report, sys.stdout, indent=2, sort_keys=False)
491
+ sys.stdout.write("\n")
492
+ return 0
493
+
494
+
495
+ if __name__ == "__main__":
496
+ raise SystemExit(main_cli())
@@ -13,7 +13,7 @@ Host tools (jq, wrangler, Python): see `docs/tooling.md` + `scripts/check-host-t
13
13
 
14
14
  | Legacy script | SDK target | Status |
15
15
  |---|---|---|
16
- | `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported.** Full parity on scan/flag/render logic. Deferred: `--email`/Resend delivery (kept module free of a RESEND_API_KEY dep; wire at CLI/ops layer if still wanted). |
16
+ | `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported + aligned (2026-08).** Database-scoped: `--quick` = all tables COUNT(*); `--full` = text LENGTH + briefing. Removed `QUICK_TABLE_RE` / `SKIP_COL_RE` / `--count-only`. Deferred: `--email`/Resend (CLI/ops layer). |
17
17
  | `scripts/run-d1-bloat-audit.sh` | `agentsam data d1-bloat` via CLI | **Shimmed**, see below — legacy `npm run audit:d1-bloat*` scripts still work unchanged. GCP-fallback/nohup wrapper behavior not reimplemented in the SDK itself (that's operational, not tool logic); still available via the legacy `.sh`. |
18
18
  | `scripts/walk_agentsam_tables.py` | `agentsam_sdk.data.agentsam_walk` | **Ported, condensed.** Schema/indexes/FKs/row-count/freshness/capability-grouping all present. Not byte-for-byte: duplicate-table detection and some staleness heuristics from the 801-line original are deferred. |
19
19
  | `scripts/d1_schema_audit.py` | folded into `agentsam_walk` (schema slice) | **Partially folded.** The capability-grouping + schema dump is covered by `agentsam_walk`. NOT ported: the per-feature markdown chunking into 14 separate `db/agentsam-*.md` files, and the curated `TABLE_META` purpose annotations (760+ lines of hand-written table descriptions) -- that's product documentation content, not audit logic, and belongs in a follow-up pass, not this one. Also note: the legacy script hardcoded a D1 database id as a fallback default (`D1_DATABASE_ID = os.environ.get("D1_DATABASE_ID", "cf87b717-...")`) -- **do not carry that forward**; the new adapter has no such fallback (HARD LAW). |
@@ -2,42 +2,73 @@
2
2
  import unittest
3
3
 
4
4
  from agentsam_sdk.data.d1_bloat import (
5
- ColStat, TableStat, _pick_bloat_columns, _flag_suspicious, _render_markdown, _fmt_bytes,
5
+ ColStat,
6
+ TableStat,
7
+ _build_briefing,
8
+ _build_doing_well,
9
+ _build_findings,
10
+ _fmt_bytes,
11
+ _pick_measure_columns,
6
12
  )
7
13
 
8
14
 
9
- class TestBloatColumnPicking(unittest.TestCase):
15
+ class TestMeasureColumnPicking(unittest.TestCase):
10
16
  def test_picks_json_and_body_columns(self):
11
- cols = [("id", "INTEGER"), ("input_json", "TEXT"), ("output_json", "TEXT"), ("created_at", "TEXT")]
12
- picked = _pick_bloat_columns(cols)
17
+ cols = [
18
+ ("id", "INTEGER"),
19
+ ("input_json", "TEXT"),
20
+ ("output_json", "TEXT"),
21
+ ("created_at", "TEXT"),
22
+ ]
23
+ picked = _pick_measure_columns(cols)
13
24
  self.assertIn("input_json", picked)
14
25
  self.assertIn("output_json", picked)
26
+ # id is INTEGER — never measured; created_at is TEXT but not payload-ish
27
+ # when preferred names exist, only preferred are kept
15
28
  self.assertNotIn("id", picked)
16
29
  self.assertNotIn("created_at", picked)
17
30
 
18
- def test_skips_id_and_fk_like_columns(self):
31
+ def test_falls_back_to_any_text_when_no_preferred(self):
32
+ cols = [("tenant_id", "TEXT"), ("workspace_id", "TEXT"), ("label", "TEXT")]
33
+ picked = _pick_measure_columns(cols)
34
+ # No SKIP_COL_RE — database-scoped; without preferred names, all text cols qualify
35
+ self.assertIn("tenant_id", picked)
36
+ self.assertIn("workspace_id", picked)
37
+ self.assertIn("label", picked)
38
+
39
+ def test_prefers_metadata_over_ids_when_mixed(self):
19
40
  cols = [("tenant_id", "TEXT"), ("workspace_id", "TEXT"), ("metadata", "TEXT")]
20
- picked = _pick_bloat_columns(cols)
21
- self.assertNotIn("tenant_id", picked)
22
- self.assertNotIn("workspace_id", picked)
23
- self.assertIn("metadata", picked)
41
+ picked = _pick_measure_columns(cols)
42
+ self.assertEqual(picked, ["metadata"])
24
43
 
25
44
 
26
- class TestFlagging(unittest.TestCase):
27
- def test_flags_large_table_high_severity(self):
28
- big = TableStat(name="agentsam_tool_call_log", row_count=250_000, text_bytes=18_874_368,
29
- est_bytes=18_874_368, columns=[ColStat(name="output_json", bytes=12_582_912)])
45
+ class TestFindings(unittest.TestCase):
46
+ def test_full_flags_large_table_high_severity(self):
47
+ big = TableStat(
48
+ name="agentsam_tool_call_log",
49
+ row_count=250_000,
50
+ text_bytes=18_874_368,
51
+ est_bytes=18_874_368,
52
+ columns=[ColStat(name="output_json", bytes=12_582_912)],
53
+ )
30
54
  small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
31
- flags = _flag_suspicious([big, small])
32
- names = {f["table"] for f in flags}
55
+ findings = _build_findings([big, small], "full")
56
+ names = {f["table"] for f in findings}
33
57
  self.assertIn("agentsam_tool_call_log", names)
34
- big_flag = next(f for f in flags if f["table"] == "agentsam_tool_call_log")
58
+ big_flag = next(f for f in findings if f["table"] == "agentsam_tool_call_log")
35
59
  self.assertEqual(big_flag["severity"], "high")
36
60
 
37
- def test_empty_table_not_flagged(self):
61
+ def test_full_empty_table_not_flagged(self):
38
62
  small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
39
- flags = _flag_suspicious([small])
40
- self.assertEqual(flags, [])
63
+ findings = _build_findings([small], "full")
64
+ self.assertEqual(findings, [])
65
+
66
+ def test_quick_flags_high_row_count(self):
67
+ big = TableStat(name="otlp_traces", row_count=120_000, est_bytes=120_000 * 120)
68
+ findings = _build_findings([big], "quick")
69
+ self.assertEqual(len(findings), 1)
70
+ self.assertEqual(findings[0]["severity"], "high")
71
+ self.assertIn("--full", findings[0]["why"])
41
72
 
42
73
 
43
74
  class TestFormatting(unittest.TestCase):
@@ -46,11 +77,16 @@ class TestFormatting(unittest.TestCase):
46
77
  self.assertEqual(_fmt_bytes(2048), "2.0 KB")
47
78
  self.assertEqual(_fmt_bytes(5 * 1024 * 1024), "5.00 MB")
48
79
 
49
- def test_render_markdown_includes_table_names(self):
50
- stats = [TableStat(name="agentsam_memory", row_count=40, text_bytes=12000, est_bytes=12000)]
51
- md = _render_markdown(stats, "1.2 MB", "quick", 10, 1)
52
- self.assertIn("agentsam_memory", md)
53
- self.assertIn("D1 bloat audit", md)
80
+ def test_briefing_includes_table_and_verdict(self):
81
+ stats = [
82
+ TableStat(name="agentsam_memory", row_count=40, text_bytes=12000, est_bytes=12000)
83
+ ]
84
+ findings = _build_findings(stats, "quick")
85
+ well = _build_doing_well(stats, findings)
86
+ md = _build_briefing(stats, findings, well, "inneranimalmedia-business", "1.2 MB", "quick", 1)
87
+ self.assertIn("D1 health", md)
88
+ self.assertIn("database-scoped", md)
89
+ self.assertIn("Verdict", md)
54
90
 
55
91
 
56
92
  if __name__ == "__main__":