@inneranimalmedia/agentsam-sdk 1.7.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/DEVELOPMENT.md +25 -5
  2. package/README.md +2 -0
  3. package/docs/RELEASES.iam-mirror.md +20 -0
  4. package/docs/RELEASES.md +9 -0
  5. package/package.json +9 -4
  6. package/protocol/README.md +51 -0
  7. package/protocol/dual-repo-sync.md +35 -0
  8. package/python/README.md +12 -0
  9. package/python/agentsam_sdk/__init__.py +9 -0
  10. package/python/agentsam_sdk/cli.py +262 -0
  11. package/python/agentsam_sdk/data/__init__.py +0 -0
  12. package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
  13. package/python/agentsam_sdk/data/d1_adapter.py +124 -0
  14. package/python/agentsam_sdk/data/d1_bloat.py +445 -0
  15. package/python/agentsam_sdk/repository/__init__.py +26 -0
  16. package/python/agentsam_sdk/repository/__main__.py +3 -0
  17. package/python/agentsam_sdk/repository/inspect.py +496 -0
  18. package/python/agentsam_sdk/repository/inventory.py +351 -0
  19. package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
  20. package/python/agentsam_sdk/runtime/__init__.py +0 -0
  21. package/python/agentsam_sdk/runtime/contract.py +105 -0
  22. package/python/docs/gaps.md +63 -0
  23. package/python/docs/tooling.md +67 -0
  24. package/python/protocol/README.md +51 -0
  25. package/python/protocol/dual-repo-sync.md +35 -0
  26. package/python/pyproject.toml +16 -0
  27. package/python/scripts/check-host-tooling.sh +65 -0
  28. package/python/tests/__init__.py +0 -0
  29. package/python/tests/fixtures/sample_tables.json +17 -0
  30. package/python/tests/fixtures.py +95 -0
  31. package/python/tests/test_agentsam_walk.py +31 -0
  32. package/python/tests/test_contract.py +32 -0
  33. package/python/tests/test_d1_bloat.py +93 -0
  34. package/python/tests/test_repository_inspect.py +84 -0
  35. package/python/tests/test_repository_inventory.py +53 -0
  36. package/python/tests/test_scan_bloat.py +31 -0
@@ -0,0 +1,351 @@
1
+ """agentsam_sdk.repository.inventory — repo file counts + sizes by category.
2
+
3
+ Port of the battle-tested scanner (formerly scripts/repo-size-inventory.py on
4
+ main). Read-only. No secrets, no D1.
5
+
6
+ JSON is jq-friendly. Examples (host `jq` required for the pipe examples):
7
+
8
+ agentsam repository inventory --repo-root .. --format json \\
9
+ | jq '.data.categories[] | select(.id==\"docs\")'
10
+
11
+ agentsam repository inventory --repo-root .. --output-dir /tmp/inv --format json
12
+ jq '.totals' /tmp/inv/repository-inventory.json
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import os
18
+ from collections import Counter, defaultdict
19
+ from dataclasses import dataclass, field
20
+ from pathlib import Path
21
+ from typing import Any, Iterable
22
+
23
+ from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
24
+
25
+ TOOL_NAME = "repository.inventory"
26
+
27
+ DEFAULT_SKIP_DIR_NAMES = frozenset(
28
+ {
29
+ ".git",
30
+ "node_modules",
31
+ ".wrangler",
32
+ "dist",
33
+ "coverage",
34
+ "__pycache__",
35
+ ".venv",
36
+ "venv",
37
+ ".venv_agentsam",
38
+ ".turbo",
39
+ ".next",
40
+ ".cache",
41
+ ".scratch",
42
+ "build",
43
+ }
44
+ )
45
+
46
+ # First path segment → category id
47
+ CATEGORY_RULES: list[tuple[str, tuple[str, ...]]] = [
48
+ ("worker_src", ("src",)),
49
+ ("dashboard", ("dashboard",)),
50
+ ("migrations", ("migrations",)),
51
+ ("docs", ("docs",)),
52
+ ("plans", ("plans",)),
53
+ ("scripts", ("scripts",)),
54
+ ("tests", ("tests", "test", "e2e")),
55
+ ("services", ("services",)),
56
+ ("supabase", ("supabase",)),
57
+ ("product_manifests", ("product-manifests",)),
58
+ ("static_assets", ("static", "public", "assets")),
59
+ ("artifacts", ("artifacts", ".scratch")),
60
+ ("vendor", ("vendor",)),
61
+ ("tools", ("tools",)),
62
+ ("local_venvs", (".venv_agentsam", ".venv", "venv")),
63
+ ("cms", ("cms-editor", "studio-cms")),
64
+ ("config_cursor", (".cursor", ".agents", ".claude", ".codex", ".githooks")),
65
+ ("ci", (".github",)),
66
+ ("agentsam_sdk_pkg", ("agentsam-sdk", "agentsam_sdk")),
67
+ ]
68
+
69
+ CATEGORY_LABELS = {
70
+ "worker_src": "Worker src/",
71
+ "dashboard": "Dashboard SPA",
72
+ "migrations": "D1 migrations",
73
+ "docs": "Docs",
74
+ "plans": "Plans",
75
+ "scripts": "Scripts",
76
+ "tests": "Tests",
77
+ "services": "Services / satellites",
78
+ "supabase": "Supabase",
79
+ "product_manifests": "Product manifests",
80
+ "static_assets": "Static / public assets",
81
+ "artifacts": "Artifacts / scratch dumps",
82
+ "vendor": "Vendor copies",
83
+ "tools": "Tools / offline utilities",
84
+ "local_venvs": "Local Python venvs",
85
+ "cms": "CMS editor packages",
86
+ "config_cursor": "Cursor / agent config",
87
+ "ci": "CI (.github)",
88
+ "agentsam_sdk_pkg": "agentsam-sdk package",
89
+ "root_misc": "Repo root files",
90
+ "other": "Other paths",
91
+ }
92
+
93
+
94
+ @dataclass
95
+ class Bucket:
96
+ id: str
97
+ label: str
98
+ file_count: int = 0
99
+ bytes: int = 0
100
+ top_files: list[dict] = field(default_factory=list)
101
+
102
+ def add(self, size: int) -> None:
103
+ self.file_count += 1
104
+ self.bytes += size
105
+
106
+
107
+ def human_bytes(n: int) -> str:
108
+ if n < 1024:
109
+ return f"{n} B"
110
+ units = ["KiB", "MiB", "GiB", "TiB"]
111
+ x = float(n)
112
+ for u in units:
113
+ x /= 1024.0
114
+ if x < 1024.0:
115
+ return f"{x:.2f} {u}"
116
+ return f"{x:.2f} PiB"
117
+
118
+
119
+ def categorize(rel: Path) -> str:
120
+ parts = rel.parts
121
+ if not parts:
122
+ return "root_misc"
123
+ first = parts[0]
124
+ for cat_id, prefixes in CATEGORY_RULES:
125
+ if first in prefixes:
126
+ return cat_id
127
+ if len(parts) == 1:
128
+ return "root_misc"
129
+ return "other"
130
+
131
+
132
+ def should_skip_dir(name: str, skip_names: frozenset[str]) -> bool:
133
+ return name in skip_names or name.endswith(".bak")
134
+
135
+
136
+ def iter_files(
137
+ root: Path,
138
+ skip_names: frozenset[str],
139
+ follow_symlinks: bool,
140
+ ) -> Iterable[tuple[Path, int]]:
141
+ for dirpath, dirnames, filenames in os.walk(
142
+ root, topdown=True, followlinks=follow_symlinks
143
+ ):
144
+ dirnames[:] = [d for d in dirnames if not should_skip_dir(d, skip_names)]
145
+ base = Path(dirpath)
146
+ for name in filenames:
147
+ if name.endswith(".bak") or name.endswith(".pyc"):
148
+ continue
149
+ path = base / name
150
+ try:
151
+ if path.is_symlink() and not follow_symlinks:
152
+ continue
153
+ st = path.stat()
154
+ except (OSError, ValueError):
155
+ continue
156
+ if not path.is_file():
157
+ continue
158
+ yield path, int(st.st_size)
159
+
160
+
161
+ def scan(
162
+ root: Path,
163
+ *,
164
+ skip_names: frozenset[str],
165
+ top_n: int,
166
+ min_bytes: int,
167
+ follow_symlinks: bool,
168
+ by_ext: bool,
169
+ ) -> dict[str, Any]:
170
+ buckets: dict[str, Bucket] = {}
171
+ ext_counts: Counter = Counter()
172
+ ext_bytes: dict[str, int] = defaultdict(int)
173
+ top_level: Counter = Counter()
174
+ largest: list[tuple[int, str, str]] = []
175
+
176
+ def bucket(cat_id: str) -> Bucket:
177
+ if cat_id not in buckets:
178
+ buckets[cat_id] = Bucket(
179
+ id=cat_id, label=CATEGORY_LABELS.get(cat_id, cat_id)
180
+ )
181
+ return buckets[cat_id]
182
+
183
+ file_total = 0
184
+ byte_total = 0
185
+
186
+ for path, size in iter_files(root, skip_names, follow_symlinks):
187
+ try:
188
+ rel = path.relative_to(root)
189
+ except ValueError:
190
+ continue
191
+ cat = categorize(rel)
192
+ bucket(cat).add(size)
193
+ file_total += 1
194
+ byte_total += size
195
+
196
+ top = rel.parts[0] if rel.parts else "."
197
+ top_level[top] += 1
198
+
199
+ ext = path.suffix.lower() or "(none)"
200
+ ext_counts[ext] += 1
201
+ if by_ext:
202
+ ext_bytes[ext] += size
203
+
204
+ if size >= min_bytes:
205
+ largest.append((size, str(rel).replace("\\", "/"), cat))
206
+
207
+ largest.sort(key=lambda t: t[0], reverse=True)
208
+ top_files = [
209
+ {"path": p, "bytes": s, "human": human_bytes(s), "category": c}
210
+ for s, p, c in largest[: max(0, top_n)]
211
+ ]
212
+
213
+ cats = sorted(buckets.values(), key=lambda b: b.bytes, reverse=True)
214
+ categories = [
215
+ {
216
+ "id": b.id,
217
+ "label": b.label,
218
+ "file_count": b.file_count,
219
+ "bytes": b.bytes,
220
+ "human": human_bytes(b.bytes),
221
+ "pct_bytes": round(100.0 * b.bytes / byte_total, 2) if byte_total else 0.0,
222
+ }
223
+ for b in cats
224
+ ]
225
+
226
+ out: dict[str, Any] = {
227
+ "ok": True,
228
+ "tool": TOOL_NAME,
229
+ "repo_root": str(root),
230
+ "file_total": file_total,
231
+ "totals": {
232
+ "file_count": file_total,
233
+ "bytes": byte_total,
234
+ "human": human_bytes(byte_total),
235
+ "categories": len(categories),
236
+ },
237
+ "skipped_dir_names": sorted(skip_names),
238
+ "categories": categories,
239
+ "largest_files": top_files,
240
+ # Backward-compatible stub fields (counts only)
241
+ "by_extension": dict(ext_counts.most_common(40)),
242
+ "by_top_level_dir": dict(top_level.most_common(40)),
243
+ }
244
+ if by_ext:
245
+ out["by_extension_detail"] = [
246
+ {
247
+ "ext": k,
248
+ "file_count": ext_counts[k],
249
+ "bytes": ext_bytes[k],
250
+ "human": human_bytes(ext_bytes[k]),
251
+ }
252
+ for k in sorted(ext_bytes, key=lambda e: ext_bytes[e], reverse=True)
253
+ ]
254
+ return out
255
+
256
+
257
+ def _markdown(report: dict[str, Any]) -> str:
258
+ lines = [
259
+ "# Repository inventory",
260
+ "",
261
+ f"- **repo_root:** `{report['repo_root']}`",
262
+ f"- **files:** {report['file_total']:,}",
263
+ f"- **bytes:** {report['totals']['human']} ({report['totals']['bytes']:,})",
264
+ "",
265
+ "## By category",
266
+ "",
267
+ "| Category | Files | Size | % |",
268
+ "|---|---:|---:|---:|",
269
+ ]
270
+ for c in report["categories"]:
271
+ lines.append(
272
+ f"| {c['label']} | {c['file_count']:,} | {c['human']} | {c['pct_bytes']:.1f}% |"
273
+ )
274
+ if report.get("largest_files"):
275
+ lines += ["", "## Largest files", ""]
276
+ for f in report["largest_files"]:
277
+ lines.append(f"- `{f['path']}` — {f['human']} (`{f['category']}`)")
278
+ lines += ["", "## By extension (top)", "", "| Ext | Count |", "|-----|-------|"]
279
+ for e, c in list(report.get("by_extension", {}).items())[:30]:
280
+ lines.append(f"| `{e}` | {c} |")
281
+ return "\n".join(lines) + "\n"
282
+
283
+
284
+ def run(tool_input: ToolInput) -> ToolResult:
285
+ started = start_timer()
286
+ p = tool_input.params
287
+ repo_root = Path(p.get("repo_root", ".")).expanduser().resolve()
288
+ output_dir = tool_input.output_path()
289
+
290
+ if not repo_root.exists():
291
+ result = ToolResult(
292
+ ok=False,
293
+ tool=TOOL_NAME,
294
+ mode=tool_input.mode,
295
+ request_id=tool_input.request_id,
296
+ started_at=started,
297
+ finished_at=start_timer(),
298
+ summary=f"repo_root does not exist: {repo_root}",
299
+ error="repo_root_not_found",
300
+ )
301
+ write_receipt(result, output_dir)
302
+ return result
303
+
304
+ skip = set(p.get("skip_dir_names") or DEFAULT_SKIP_DIR_NAMES)
305
+ if p.get("include_node_modules"):
306
+ skip.discard("node_modules")
307
+ if p.get("include_venvs"):
308
+ skip.discard(".venv")
309
+ skip.discard("venv")
310
+ skip.discard(".venv_agentsam")
311
+ if p.get("include_git"):
312
+ skip.discard(".git")
313
+ if p.get("include_dist"):
314
+ skip.discard("dist")
315
+ skip.discard(".wrangler")
316
+ skip.discard("build")
317
+
318
+ report = scan(
319
+ repo_root,
320
+ skip_names=frozenset(skip),
321
+ top_n=int(p.get("top", 20)),
322
+ min_bytes=int(p.get("min_bytes", 0)),
323
+ follow_symlinks=bool(p.get("follow_symlinks", False)),
324
+ by_ext=bool(p.get("by_ext", True)),
325
+ )
326
+
327
+ artifacts: list[str] = []
328
+ if output_dir:
329
+ output_dir.mkdir(parents=True, exist_ok=True)
330
+ json_path = output_dir / "repository-inventory.json"
331
+ md_path = output_dir / "repository-inventory.md"
332
+ json_path.write_text(json.dumps(report, indent=2), encoding="utf-8")
333
+ md_path.write_text(_markdown(report), encoding="utf-8")
334
+ artifacts = [str(json_path), str(md_path)]
335
+
336
+ result = ToolResult(
337
+ ok=True,
338
+ tool=TOOL_NAME,
339
+ mode=tool_input.mode,
340
+ request_id=tool_input.request_id,
341
+ started_at=started,
342
+ finished_at=start_timer(),
343
+ summary=(
344
+ f"{report['file_total']} files · {report['totals']['human']} "
345
+ f"under {repo_root}"
346
+ ),
347
+ data=report,
348
+ artifacts=artifacts,
349
+ )
350
+ write_receipt(result, output_dir)
351
+ return result
@@ -0,0 +1,173 @@
1
+ """agentsam_sdk.repository.scan_bloat — per-file source bloat inventory.
2
+
3
+ Companion to repository.inventory (category rollups). This tool lists the
4
+ largest runtime source files under a root (default: cwd) with size / lines /
5
+ est. tokens — for refactor targeting and agent context budgeting.
6
+
7
+ Read-only. No secrets, no D1.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import os
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
17
+
18
+ TOOL_NAME = "repository.scan_bloat"
19
+
20
+ DEFAULT_EXTS = frozenset({".js", ".ts", ".jsx", ".tsx", ".mjs", ".cjs"})
21
+ DEFAULT_EXCLUDE_DIRS = frozenset(
22
+ {
23
+ "node_modules",
24
+ ".git",
25
+ "dist",
26
+ "build",
27
+ ".wrangler",
28
+ ".next",
29
+ ".turbo",
30
+ "coverage",
31
+ ".cache",
32
+ "out",
33
+ ".vercel",
34
+ "__pycache__",
35
+ ".venv",
36
+ "venv",
37
+ }
38
+ )
39
+ CHARS_PER_TOKEN = 3.7
40
+
41
+
42
+ def scan(
43
+ root: str | Path,
44
+ *,
45
+ exts: set[str] | frozenset[str] = DEFAULT_EXTS,
46
+ exclude_dirs: set[str] | frozenset[str] = DEFAULT_EXCLUDE_DIRS,
47
+ ) -> list[dict[str, Any]]:
48
+ """Walk root and return file stats sorted by size_bytes desc."""
49
+ root_path = Path(root).resolve()
50
+ results: list[dict[str, Any]] = []
51
+ if root_path.is_file():
52
+ # Allow a single-file root for smoke tests
53
+ if root_path.suffix in exts:
54
+ results.append(_stat_file(root_path, root_path.parent))
55
+ return results
56
+
57
+ for dirpath, dirnames, filenames in os.walk(root_path):
58
+ dirnames[:] = [
59
+ d for d in dirnames if d not in exclude_dirs and not d.startswith(".")
60
+ ]
61
+ for fname in filenames:
62
+ ext = os.path.splitext(fname)[1]
63
+ if ext not in exts:
64
+ continue
65
+ fpath = Path(dirpath) / fname
66
+ try:
67
+ results.append(_stat_file(fpath, root_path))
68
+ except (OSError, UnicodeDecodeError):
69
+ continue
70
+ results.sort(key=lambda r: r["size_bytes"], reverse=True)
71
+ return results
72
+
73
+
74
+ def _stat_file(fpath: Path, root: Path) -> dict[str, Any]:
75
+ size_bytes = fpath.stat().st_size
76
+ content = fpath.read_text(encoding="utf-8", errors="replace")
77
+ line_count = content.count("\n") + (1 if content else 0)
78
+ return {
79
+ "path": str(fpath.relative_to(root)),
80
+ "size_bytes": size_bytes,
81
+ "size_kb": round(size_bytes / 1024, 1),
82
+ "lines": line_count,
83
+ "est_tokens": round(size_bytes / CHARS_PER_TOKEN),
84
+ "bytes_per_line": round(size_bytes / line_count, 1) if line_count else 0,
85
+ }
86
+
87
+
88
+ def human_table(files: list[dict[str, Any]], *, scanned: int, total_kb: float, total_tokens: int) -> str:
89
+ """Plain-text table for terminal / markdown CLI mode."""
90
+ if not files:
91
+ return "No files matched.\n"
92
+ path_w = min(max(len(r["path"]) for r in files), 70)
93
+ header = f"{'SIZE':>9} {'LINES':>7} {'~TOKENS':>8} {'B/LINE':>7} PATH"
94
+ lines = [header, "-" * len(header)]
95
+ for r in files:
96
+ path = r["path"] if len(r["path"]) <= path_w else "…" + r["path"][-(path_w - 1) :]
97
+ lines.append(
98
+ f"{r['size_kb']:>8.1f}KB {r['lines']:>7} {r['est_tokens']:>8} "
99
+ f"{r['bytes_per_line']:>7} {path}"
100
+ )
101
+ lines.append("-" * len(header))
102
+ lines.append(
103
+ f"Scanned {scanned} files, {total_kb:.1f}KB total, ~{total_tokens:,} est. tokens"
104
+ )
105
+ return "\n".join(lines) + "\n"
106
+
107
+
108
+ def run(tool_input: ToolInput) -> ToolResult:
109
+ started = start_timer()
110
+ p = tool_input.params
111
+ root = str(p.get("root") or p.get("repo_root") or ".")
112
+ top = max(1, int(p.get("top", 30)))
113
+ min_kb = float(p.get("min_kb", 0) or 0)
114
+ ext_raw = p.get("ext") or ",".join(sorted(DEFAULT_EXTS))
115
+ exts = {
116
+ e if str(e).startswith(".") else f".{e}"
117
+ for e in str(ext_raw).split(",")
118
+ if str(e).strip()
119
+ }
120
+ exclude = set(DEFAULT_EXCLUDE_DIRS)
121
+ extra = p.get("exclude") or ""
122
+ if extra:
123
+ exclude |= {d.strip() for d in str(extra).split(",") if d.strip()}
124
+
125
+ try:
126
+ all_files = scan(root, exts=exts, exclude_dirs=exclude)
127
+ filtered = [r for r in all_files if r["size_kb"] >= min_kb][:top]
128
+ total_kb = round(sum(r["size_kb"] for r in all_files), 1)
129
+ total_tokens = sum(r["est_tokens"] for r in all_files)
130
+ data = {
131
+ "ok": True,
132
+ "root": str(Path(root).resolve()),
133
+ "file_count": len(all_files),
134
+ "total_kb": total_kb,
135
+ "total_est_tokens": total_tokens,
136
+ "files": filtered,
137
+ }
138
+ artifacts: list[str] = []
139
+ output_dir = tool_input.output_path()
140
+ if output_dir:
141
+ output_dir.mkdir(parents=True, exist_ok=True)
142
+ out_json = output_dir / "scan-bloat.json"
143
+ out_json.write_text(json.dumps(data, indent=2), encoding="utf-8")
144
+ artifacts.append(str(out_json))
145
+
146
+ result = ToolResult(
147
+ ok=True,
148
+ tool=TOOL_NAME,
149
+ mode=tool_input.mode or "read-only",
150
+ request_id=tool_input.request_id,
151
+ started_at=started,
152
+ finished_at=start_timer(),
153
+ summary=(
154
+ f"Scanned {len(all_files)} files, {total_kb}KB total, "
155
+ f"top {len(filtered)} ≥{min_kb}KB."
156
+ ),
157
+ data=data,
158
+ artifacts=artifacts,
159
+ )
160
+ except Exception as e: # noqa: BLE001 — surfaced in ToolResult
161
+ result = ToolResult(
162
+ ok=False,
163
+ tool=TOOL_NAME,
164
+ mode=tool_input.mode or "read-only",
165
+ request_id=tool_input.request_id,
166
+ started_at=started,
167
+ finished_at=start_timer(),
168
+ summary="scan_bloat failed",
169
+ error=str(e)[:500],
170
+ )
171
+
172
+ write_receipt(result, tool_input.output_path())
173
+ return result
File without changes
@@ -0,0 +1,105 @@
1
+ """Shared tool contract: ToolInput / ToolResult / receipts.
2
+
3
+ HARD LAW (see inneranimalmedia AGENTS.md): never hardcode identity, repo,
4
+ workspace, or tenant values anywhere in this package -- not in shipped code,
5
+ patches, examples, or fallback defaults. All identity-shaped values
6
+ (database names, account ids, wrangler config paths) must come from the
7
+ caller (CLI flag) or the environment. Generic placeholders only in examples
8
+ (e.g. "owner/repo-name", "$D1_DATABASE_NAME").
9
+
10
+ Every tool module in agentsam_sdk exposes a `run(tool_input: ToolInput) ->
11
+ ToolResult` function (or is wrapped to look like one from cli.py). Modes are
12
+ read-only by default -- any tool that can write must accept an explicit
13
+ `write=True` and should say so loudly in its ToolResult.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import os
19
+ import time
20
+ import uuid
21
+ from dataclasses import dataclass, field, asdict
22
+ from pathlib import Path
23
+ from typing import Any, Optional
24
+
25
+
26
+ HARD_LAW_NOTE = (
27
+ "Never hardcode identity/repo/workspace/tenant values. Use env vars or "
28
+ "explicit CLI args; fall back to generic placeholders only in docs."
29
+ )
30
+
31
+
32
+ @dataclass
33
+ class ToolInput:
34
+ """Normalized input every agentsam_sdk tool accepts.
35
+
36
+ mode: tool-specific sub-mode (e.g. "quick" | "full" for d1_bloat)
37
+ params: tool-specific keyword args
38
+ output_dir: where json/markdown output + receipt get written
39
+ write: True only for tools that mutate state; default False (read-only)
40
+ """
41
+
42
+ mode: str = "default"
43
+ params: dict[str, Any] = field(default_factory=dict)
44
+ output_dir: Optional[str] = None
45
+ write: bool = False
46
+ request_id: str = field(default_factory=lambda: uuid.uuid4().hex[:12])
47
+
48
+ def output_path(self) -> Optional[Path]:
49
+ return Path(self.output_dir) if self.output_dir else None
50
+
51
+
52
+ @dataclass
53
+ class ToolResult:
54
+ """Normalized output every agentsam_sdk tool returns."""
55
+
56
+ ok: bool
57
+ tool: str
58
+ mode: str
59
+ request_id: str
60
+ started_at: float
61
+ finished_at: float
62
+ summary: str
63
+ data: dict[str, Any] = field(default_factory=dict)
64
+ artifacts: list[str] = field(default_factory=list)
65
+ error: Optional[str] = None
66
+
67
+ @property
68
+ def duration_s(self) -> float:
69
+ return round(self.finished_at - self.started_at, 3)
70
+
71
+ def to_dict(self) -> dict[str, Any]:
72
+ d = asdict(self)
73
+ d["duration_s"] = self.duration_s
74
+ return d
75
+
76
+
77
+ def new_receipt_path(output_dir: Path, tool: str, request_id: str) -> Path:
78
+ output_dir.mkdir(parents=True, exist_ok=True)
79
+ return output_dir / f"receipt-{tool}-{request_id}.json"
80
+
81
+
82
+ def write_receipt(result: ToolResult, output_dir: Optional[Path]) -> Optional[str]:
83
+ """Write a receipt (contract rule: every tool call gets one). Returns path or None."""
84
+ if output_dir is None:
85
+ return None
86
+ path = new_receipt_path(output_dir, result.tool, result.request_id)
87
+ payload = result.to_dict()
88
+ payload["written_at_unixepoch"] = int(time.time())
89
+ path.write_text(json.dumps(payload, indent=2, default=str), encoding="utf-8")
90
+ return str(path)
91
+
92
+
93
+ def require_env(*names: str) -> dict[str, str]:
94
+ """Fetch required env vars; raise with a clear message (never invent values)."""
95
+ missing = [n for n in names if not os.environ.get(n)]
96
+ if missing:
97
+ raise RuntimeError(
98
+ f"Missing required env var(s): {', '.join(missing)}. "
99
+ f"{HARD_LAW_NOTE}"
100
+ )
101
+ return {n: os.environ[n] for n in names}
102
+
103
+
104
+ def start_timer() -> float:
105
+ return time.time()
@@ -0,0 +1,63 @@
1
+ # agentsam-sdk — D1 audit port: status
2
+
3
+ Tracks the file list from `plans/active/CLAUDE-AGENTSAM-SDK-D1-AUDIT-PORT-HANDOFF-2026-08.md`.
4
+
5
+ Context: `feature/agentsam-sdk-scaffold` had no `agentsam-sdk/` package at all
6
+ when this pass started (no pyproject, no CLI, no docs/gaps.md) -- the
7
+ handoff doc assumed a scaffold that hadn't actually landed. This pass builds
8
+ the minimal scaffold from zero, then ports P0.
9
+
10
+ Host tools (jq, wrangler, Python): see `docs/tooling.md` + `scripts/check-host-tooling.sh`.
11
+
12
+ ## P0 — whole-D1 / agentsam walkers
13
+
14
+ | Legacy script | SDK target | Status |
15
+ |---|---|---|
16
+ | `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported + aligned (2026-08).** Database-scoped: `--quick` = all tables COUNT(*); `--full` = text LENGTH + briefing. Removed `QUICK_TABLE_RE` / `SKIP_COL_RE` / `--count-only`. Deferred: `--email`/Resend (CLI/ops layer). |
17
+ | `scripts/run-d1-bloat-audit.sh` | `agentsam data d1-bloat` via CLI | **Shimmed**, see below — legacy `npm run audit:d1-bloat*` scripts still work unchanged. GCP-fallback/nohup wrapper behavior not reimplemented in the SDK itself (that's operational, not tool logic); still available via the legacy `.sh`. |
18
+ | `scripts/walk_agentsam_tables.py` | `agentsam_sdk.data.agentsam_walk` | **Ported, condensed.** Schema/indexes/FKs/row-count/freshness/capability-grouping all present. Not byte-for-byte: duplicate-table detection and some staleness heuristics from the 801-line original are deferred. |
19
+ | `scripts/d1_schema_audit.py` | folded into `agentsam_walk` (schema slice) | **Partially folded.** The capability-grouping + schema dump is covered by `agentsam_walk`. NOT ported: the per-feature markdown chunking into 14 separate `db/agentsam-*.md` files, and the curated `TABLE_META` purpose annotations (760+ lines of hand-written table descriptions) -- that's product documentation content, not audit logic, and belongs in a follow-up pass, not this one. Also note: the legacy script hardcoded a D1 database id as a fallback default (`D1_DATABASE_ID = os.environ.get("D1_DATABASE_ID", "cf87b717-...")`) -- **do not carry that forward**; the new adapter has no such fallback (HARD LAW). |
20
+
21
+ ## P1 — agentsam quality / wiring (not started this pass)
22
+
23
+ `audit_agentsam_full.py`, `agentsam_db_deep_audit.py`, `audit_agentsam_tables.py`,
24
+ `audit_agentsam_schema.py`, `audit_agentsam_table_usage.py`,
25
+ `agentsam_cms_d1_table_audit.py`, `agentsam_audit.py` — all deferred. Reason:
26
+ scope boundary for this pass was P0 only, per the handoff doc's file list.
27
+
28
+ ## P2 — repository / history / readiness
29
+
30
+ | Legacy script | SDK target | Status |
31
+ |---|---|---|
32
+ | `scripts/repo-size-inventory.py` (main) / `repo_inventory.py` | `agentsam_sdk.repository.inventory` | **Ported** (category + byte sizes + largest files + extension rollups; jq-friendly JSON; stub fields `by_extension` / `by_top_level_dir` retained). |
33
+ | `tools/scan_bloat.py` (IAM shim) | `agentsam_sdk.repository.scan_bloat` | **Ported.** Per-file KB/lines/est. tokens; CLI `agentsam repository scan-bloat`. D1: `agentsam_scripts.slug=scan_bloat` (not `agentsam_commands`). |
34
+ | `repo_cleanup_classify.py`, `repo-cleanup.py` | `agentsam_sdk.repository.cleanup_plan` | **Not started** |
35
+ | `audit_dead_code.py` | `agentsam_sdk.repository.dead_paths` | **Not started** |
36
+ | `audit_hardcoded_identity.py`, `guard-no-hardcoded-identity.sh` | `agentsam_sdk.readiness.boundaries` | **Not started** |
37
+ | `audit_migration_chain.py` | `agentsam_sdk.history.timeline` / `.supersession` | **Not started** |
38
+ | leftover `d1_*_audit.py` | `agentsam_sdk.data.d1` adapters | **Not started** |
39
+
40
+ ## Out of scope (per handoff doc, unchanged)
41
+
42
+ `scripts/embed_*`, `scripts/ingest_*` (product Vectorize/pgvector pipelines);
43
+ dashboard/Worker `client_fs`/ExecOS path-propose lane; inventing new D1
44
+ tables for audit output (output stays json+markdown under `--output-dir`,
45
+ R2 later if wanted).
46
+
47
+ ## Contract compliance
48
+
49
+ - [x] `runtime/contract.py`: `ToolInput`/`ToolResult` + receipt, every tool call writes one.
50
+ - [x] Read-only default; `ToolInput.write` exists but nothing in this pass sets it True — no silent production writes.
51
+ - [x] D1 access only via `data/d1_adapter.py` (wraps `wrangler d1 execute`); no other module shells out to wrangler or hits the CF API directly.
52
+ - [x] No hardcoded `au_*`/`ws_*`/`tenant_*`/database-id values anywhere — `D1Adapter.from_env` raises rather than guessing.
53
+ - [x] Output: json + markdown from `d1_bloat`, `agentsam_walk`, and `repository.inventory`. No other formats claimed.
54
+ - [x] Tests use fixtures/pure functions only — `python3 -m unittest discover -s tests` needs no live D1 or network.
55
+ - [x] Receipts use `written_at_unixepoch` (unixepoch integer per AGENTS.md); tool-level timestamps otherwise use `time.time()` floats for duration math, not stored as durable rows.
56
+ - [x] `scripts/run-d1-bloat-audit.sh` untouched and still works (legacy path preserved) — see shim note above.
57
+ - [x] Host tooling documented: `docs/tooling.md` (jq + wrangler + env); `scripts/check-host-tooling.sh`.
58
+
59
+ ## Known limitations to fix before this is "done" for P0
60
+
61
+ 1. `agentsam_walk` is a condensed reimplementation, not byte-identical to the 801-line original — needs a side-by-side diff review against real D1 output before calling P0 fully closed.
62
+ 2. No live-D1 smoke test has been run yet in this pass (would need `AGENTSAM_D1_DB_NAME` + Cloudflare creds on the operator machine) — see README "D1 audits" section for the command to run one.
63
+ 3. `d1_schema_audit.py`'s curated `TABLE_META` prose (table-by-table purpose descriptions) is real, hand-maintained documentation value that this port does NOT carry over. Flagging so it isn't silently lost.