@inneranimalmedia/agentsam-sdk 1.7.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/DEVELOPMENT.md +25 -5
  2. package/README.md +2 -0
  3. package/docs/RELEASES.iam-mirror.md +20 -0
  4. package/docs/RELEASES.md +9 -0
  5. package/package.json +9 -4
  6. package/protocol/README.md +51 -0
  7. package/protocol/dual-repo-sync.md +35 -0
  8. package/python/README.md +12 -0
  9. package/python/agentsam_sdk/__init__.py +9 -0
  10. package/python/agentsam_sdk/cli.py +262 -0
  11. package/python/agentsam_sdk/data/__init__.py +0 -0
  12. package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
  13. package/python/agentsam_sdk/data/d1_adapter.py +124 -0
  14. package/python/agentsam_sdk/data/d1_bloat.py +445 -0
  15. package/python/agentsam_sdk/repository/__init__.py +26 -0
  16. package/python/agentsam_sdk/repository/__main__.py +3 -0
  17. package/python/agentsam_sdk/repository/inspect.py +496 -0
  18. package/python/agentsam_sdk/repository/inventory.py +351 -0
  19. package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
  20. package/python/agentsam_sdk/runtime/__init__.py +0 -0
  21. package/python/agentsam_sdk/runtime/contract.py +105 -0
  22. package/python/docs/gaps.md +63 -0
  23. package/python/docs/tooling.md +67 -0
  24. package/python/protocol/README.md +51 -0
  25. package/python/protocol/dual-repo-sync.md +35 -0
  26. package/python/pyproject.toml +16 -0
  27. package/python/scripts/check-host-tooling.sh +65 -0
  28. package/python/tests/__init__.py +0 -0
  29. package/python/tests/fixtures/sample_tables.json +17 -0
  30. package/python/tests/fixtures.py +95 -0
  31. package/python/tests/test_agentsam_walk.py +31 -0
  32. package/python/tests/test_contract.py +32 -0
  33. package/python/tests/test_d1_bloat.py +93 -0
  34. package/python/tests/test_repository_inspect.py +84 -0
  35. package/python/tests/test_repository_inventory.py +53 -0
  36. package/python/tests/test_scan_bloat.py +31 -0
@@ -0,0 +1,124 @@
1
+ """D1 adapter -- the only place this package shells out to wrangler.
2
+
3
+ Every other data/*.py module must go through D1Adapter, never call
4
+ subprocess/urllib against D1 directly (contract rule: "D1 access only via
5
+ an adapter"). Resolves database name / wrangler config from env or explicit
6
+ args -- never a hardcoded database id or account id (HARD LAW).
7
+
8
+ Env vars (all optional overrides; wrangler itself resolves the account via
9
+ CLOUDFLARE_API_TOKEN / `wrangler login` state and the database id via the
10
+ wrangler config file's [[d1_databases]] binding -- so this adapter does not
11
+ need to know the raw D1 database id at all):
12
+
13
+ AGENTSAM_D1_DB_NAME D1 database name (required -- no default;
14
+ pass --db or set this)
15
+ AGENTSAM_WRANGLER_CONFIG path to wrangler config, default "wrangler.toml"
16
+ AGENTSAM_REPO_ROOT repo root wrangler runs from, default cwd
17
+ CLOUDFLARE_API_TOKEN passed straight through to the wrangler subprocess
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import os
23
+ import re
24
+ import subprocess
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import Optional
28
+
29
+
30
+ class D1AdapterError(RuntimeError):
31
+ pass
32
+
33
+
34
+ @dataclass
35
+ class D1Adapter:
36
+ db_name: str
37
+ wrangler_config: str = "wrangler.toml"
38
+ repo_root: Optional[Path] = None
39
+ remote: bool = True
40
+ timeout_s: int = 120
41
+
42
+ @classmethod
43
+ def from_env(
44
+ cls,
45
+ db_name: Optional[str] = None,
46
+ wrangler_config: Optional[str] = None,
47
+ repo_root: Optional[str] = None,
48
+ ) -> "D1Adapter":
49
+ name = db_name or os.environ.get("AGENTSAM_D1_DB_NAME")
50
+ if not name:
51
+ raise D1AdapterError(
52
+ "No D1 database name given -- pass --db or set "
53
+ "AGENTSAM_D1_DB_NAME. Refusing to guess/hardcode one."
54
+ )
55
+ cfg = wrangler_config or os.environ.get("AGENTSAM_WRANGLER_CONFIG", "wrangler.toml")
56
+ root = Path(repo_root or os.environ.get("AGENTSAM_REPO_ROOT") or Path.cwd())
57
+ return cls(db_name=name, wrangler_config=cfg, repo_root=root)
58
+
59
+ def _run_wrangler(self, args: list[str]) -> subprocess.CompletedProcess:
60
+ cmd = ["npx", "wrangler", *args]
61
+ env = os.environ.copy()
62
+ return subprocess.run(
63
+ cmd,
64
+ cwd=str(self.repo_root) if self.repo_root else None,
65
+ capture_output=True,
66
+ text=True,
67
+ timeout=self.timeout_s,
68
+ env=env,
69
+ )
70
+
71
+ def query(self, sql: str) -> list[dict]:
72
+ args = ["d1", "execute", self.db_name]
73
+ if self.remote:
74
+ args.append("--remote")
75
+ args += ["-c", self.wrangler_config, "--json", "--command", sql]
76
+ proc = self._run_wrangler(args)
77
+ raw = (proc.stdout or "").strip()
78
+ if not raw:
79
+ raise D1AdapterError((proc.stderr or "empty wrangler output")[:400])
80
+ try:
81
+ data = json.loads(raw)
82
+ except json.JSONDecodeError as e:
83
+ raise D1AdapterError(f"non-JSON wrangler output: {e}") from e
84
+ if isinstance(data, dict) and data.get("error"):
85
+ raise D1AdapterError(str(data["error"])[:400])
86
+ if isinstance(data, list) and data:
87
+ return data[0].get("results") or []
88
+ return []
89
+
90
+ def database_size(self) -> Optional[str]:
91
+ args = ["d1", "info", self.db_name, "-c", self.wrangler_config]
92
+ proc = self._run_wrangler(args)
93
+ m = re.search(r"database_size\s*\│\s*([^\│]+)", proc.stdout or "")
94
+ return m.group(1).strip() if m else None
95
+
96
+ def list_tables(self, like: Optional[str] = None) -> list[str]:
97
+ sql = (
98
+ "SELECT name FROM sqlite_master WHERE type='table' "
99
+ "AND name NOT LIKE 'sqlite_%' AND name NOT LIKE '_cf_%'"
100
+ )
101
+ if like:
102
+ safe = like.replace("'", "")
103
+ sql += f" AND name LIKE '{safe}'"
104
+ sql += " ORDER BY name"
105
+ rows = self.query(sql)
106
+ return [r["name"] for r in rows]
107
+
108
+ def table_columns(self, table: str) -> list[tuple[str, str]]:
109
+ safe = table.replace("'", "")
110
+ rows = self.query(f"SELECT name, type FROM pragma_table_info('{safe}')")
111
+ return [(r["name"], r.get("type") or "TEXT") for r in rows]
112
+
113
+ def table_indexes(self, table: str) -> list[dict]:
114
+ safe = table.replace('"', "")
115
+ return self.query(f'PRAGMA index_list("{safe}")')
116
+
117
+ def foreign_keys(self, table: str) -> list[dict]:
118
+ safe = table.replace('"', "")
119
+ return self.query(f'PRAGMA foreign_key_list("{safe}")')
120
+
121
+ def row_count(self, table: str) -> int:
122
+ safe = table.replace('"', "")
123
+ rows = self.query(f'SELECT COUNT(*) AS rc FROM "{safe}"')
124
+ return int(rows[0].get("rc") or 0) if rows else 0
@@ -0,0 +1,445 @@
1
+ """agentsam_sdk.data.d1_bloat -- port of scripts/d1_bloat_audit.py.
2
+
3
+ Database-scoped D1 size audit (not tenant/workspace). Walks one CF D1 database.
4
+
5
+ Modes:
6
+ quick — every user table + COUNT(*) only
7
+ full — same tables + SUM(LENGTH(...)) on text-ish columns + briefing
8
+
9
+ D1 remote has no dbstat; sizes are LENGTH estimates, not exact page bytes.
10
+
11
+ Deferred from the legacy script (see docs/gaps.md): --email/Resend delivery.
12
+ This module writes json + markdown only.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import re
18
+ from concurrent.futures import ThreadPoolExecutor, as_completed
19
+ from dataclasses import asdict, dataclass, field
20
+ from datetime import datetime, timezone
21
+ from typing import Any, Optional
22
+
23
+ from agentsam_sdk.data.d1_adapter import D1Adapter, D1AdapterError
24
+ from agentsam_sdk.runtime.contract import ToolInput, ToolResult, start_timer, write_receipt
25
+
26
+ TOOL_NAME = "data.d1_bloat"
27
+
28
+ # Prefer payload-ish TEXT columns when measuring LENGTH (full mode).
29
+ BLOAT_COL_RE = re.compile(
30
+ r"(body|content|value|markdown|_json\b|schema|payload|output|prompt|message|"
31
+ r"text|config|metadata|description|notes|script|summary|arguments|result|"
32
+ r"attributes|events|resource|handler|input_|output_|sql\b|embedding|merged_)",
33
+ re.I,
34
+ )
35
+
36
+ ROLLUP_HINTS: dict[str, str] = {
37
+ "agentsam_tool_call_log": "Archive/purge output_json + input_json >30d; keep output_summary + ids.",
38
+ "agentsam_tool_chain": "result_json dominates — rollup to R2 or truncate; keep summaries.",
39
+ "agentsam_tool_cache": "Enforce TTL + max rows; output_json must not grow unbounded.",
40
+ "agentsam_mcp_tool_execution": "Archive old output_json; mirror tool_call_log policy.",
41
+ "agentsam_execution_steps": "Archive input/output JSON after workflow completes.",
42
+ "agentsam_workflow_runs": "Move step_results_json to R2; D1 row = pointer + status + cost.",
43
+ "agentsam_webhook_events": "Rollup payload_json; retain type + ts + external id.",
44
+ "agentsam_scripts": "body must stay empty; canonical source in R2 (source_stored=r2:…).",
45
+ "agentsam_skill": "Large SKILL.md → R2; D1 = metadata + retrieval_strategy=r2.",
46
+ "agentsam_memory": "Archive stale prose; vectors live in Supabase/Vectorize, not D1.",
47
+ "agentsam_rules_document": "body_markdown → R2; D1 = trigger + key + short summary.",
48
+ "agentsam_cron_runs": "Trim metadata_json on old runs.",
49
+ "agentsam_hook_execution": "Archive payload_json; keep hook id + status.",
50
+ "agentsam_eval_runs": "Cap grader notes; long artifacts → Supabase eval tables.",
51
+ "otlp_traces": "Retention on attributes_json; sample or export off-D1.",
52
+ "terminal_history_archive_431": "Archive scrollback to R2 or cap rows per connection.",
53
+ "system_health_snapshots": "Shorten retention or aggregate further.",
54
+ }
55
+
56
+
57
+ @dataclass
58
+ class ColStat:
59
+ name: str
60
+ bytes: int
61
+ max_len: int = 0
62
+
63
+
64
+ @dataclass
65
+ class TableStat:
66
+ name: str
67
+ row_count: int = 0
68
+ text_bytes: int = 0
69
+ est_bytes: int = 0
70
+ columns: list[ColStat] = field(default_factory=list)
71
+ error: Optional[str] = None
72
+
73
+ @property
74
+ def rollup_hint(self) -> Optional[str]:
75
+ if self.name in ROLLUP_HINTS:
76
+ return ROLLUP_HINTS[self.name]
77
+ if self.text_bytes > 500_000 and any(
78
+ c.name
79
+ in (
80
+ "body",
81
+ "content_markdown",
82
+ "value",
83
+ "output_json",
84
+ "result_json",
85
+ "payload_json",
86
+ )
87
+ for c in self.columns
88
+ ):
89
+ return "Large text/JSON in D1 — prefer R2 pointer + vector lanes for search."
90
+ return None
91
+
92
+
93
+ def _pick_measure_columns(cols: list[tuple[str, str]], max_cols: int = 12) -> list[str]:
94
+ """TEXT/BLOB/JSON columns to LENGTH-scan. Prefer payload-ish names; else any text cols."""
95
+ textish = [
96
+ n
97
+ for n, typ in cols
98
+ if (typ or "TEXT").upper() in ("TEXT", "BLOB", "JSON") or "CHAR" in (typ or "").upper()
99
+ ]
100
+ preferred = [n for n in textish if BLOAT_COL_RE.search(n)]
101
+ chosen = preferred if preferred else textish
102
+ return chosen[:max_cols]
103
+
104
+
105
+ def _scan_table(adapter: D1Adapter, table: str, analyze_text: bool) -> TableStat:
106
+ stat = TableStat(name=table)
107
+ try:
108
+ if not analyze_text:
109
+ stat.row_count = adapter.row_count(table)
110
+ stat.est_bytes = stat.row_count * 120
111
+ return stat
112
+
113
+ cols = adapter.table_columns(table)
114
+ measure = _pick_measure_columns(cols)
115
+ if measure:
116
+ parts = [f'SUM(LENGTH(COALESCE("{c}", \'\'))) AS "{c}"' for c in measure]
117
+ max_parts = [f'MAX(LENGTH(COALESCE("{c}", \'\'))) AS "m_{c}"' for c in measure]
118
+ sql = f'SELECT COUNT(*) AS rc, {", ".join(parts + max_parts)} FROM "{table}"'
119
+ row = adapter.query(sql)[0]
120
+ stat.row_count = int(row.get("rc") or 0)
121
+ for c in measure:
122
+ b = int(row.get(c) or 0)
123
+ if b:
124
+ stat.columns.append(
125
+ ColStat(name=c, bytes=b, max_len=int(row.get(f"m_{c}") or 0))
126
+ )
127
+ stat.text_bytes = sum(c.bytes for c in stat.columns)
128
+ stat.est_bytes = stat.text_bytes if stat.text_bytes else stat.row_count * 120
129
+ else:
130
+ stat.row_count = adapter.row_count(table)
131
+ stat.est_bytes = stat.row_count * 120
132
+ except D1AdapterError as e:
133
+ stat.error = str(e)[:200]
134
+ except Exception as e: # noqa: BLE001 -- surfaced in receipt, not swallowed
135
+ stat.error = str(e)[:200]
136
+ return stat
137
+
138
+
139
+ def _fmt_bytes(n: int) -> str:
140
+ if n >= 1024 * 1024:
141
+ return f"{n / 1024 / 1024:.2f} MB"
142
+ if n >= 1024:
143
+ return f"{n / 1024:.1f} KB"
144
+ return f"{n} B"
145
+
146
+
147
+ def _build_findings(stats: list[TableStat], mode: str) -> list[dict[str, Any]]:
148
+ findings: list[dict[str, Any]] = []
149
+ if mode == "quick":
150
+ ranked = sorted(stats, key=lambda s: s.row_count, reverse=True)
151
+ for s in ranked[:40]:
152
+ if s.error:
153
+ findings.append(
154
+ {
155
+ "severity": "medium",
156
+ "table": s.name,
157
+ "rows": s.row_count,
158
+ "est_bytes": 0,
159
+ "est_human": "—",
160
+ "why": f"scan_error: {s.error[:120]}",
161
+ "next_steps": ["Re-run --full for this table after fixing access."],
162
+ }
163
+ )
164
+ continue
165
+ if s.row_count < 50_000:
166
+ continue
167
+ sev = "high" if s.row_count >= 100_000 else "medium"
168
+ findings.append(
169
+ {
170
+ "severity": sev,
171
+ "table": s.name,
172
+ "rows": s.row_count,
173
+ "est_bytes": s.est_bytes,
174
+ "est_human": _fmt_bytes(s.est_bytes),
175
+ "why": f"High row count ({s.row_count:,}) — run --full to size text/JSON.",
176
+ "next_steps": [
177
+ f"agentsam data d1-bloat --full --prefix {s.name}",
178
+ "Confirm retention / archive policy for this table.",
179
+ ],
180
+ }
181
+ )
182
+ return findings
183
+
184
+ total_text = sum(s.text_bytes for s in stats) or 1
185
+ ranked = sorted(stats, key=lambda s: s.text_bytes or s.est_bytes, reverse=True)
186
+ for s in ranked[:80]:
187
+ est = s.text_bytes or s.est_bytes
188
+ reasons: list[str] = []
189
+ if est >= 5_000_000:
190
+ reasons.append(f"text_est>={_fmt_bytes(5_000_000)}")
191
+ if s.row_count >= 100_000:
192
+ reasons.append(f"rows>={s.row_count:,}")
193
+ if est >= 1_000_000 and (est / total_text) >= 0.08:
194
+ reasons.append(f"share>={100 * est / total_text:.0f}% of scanned text")
195
+ if s.rollup_hint and est >= 500_000:
196
+ reasons.append("known_rollup_candidate")
197
+ if s.error:
198
+ reasons.append(f"scan_error:{s.error[:80]}")
199
+ if not reasons:
200
+ continue
201
+ steps = []
202
+ if s.rollup_hint:
203
+ steps.append(s.rollup_hint)
204
+ else:
205
+ steps.append("Inspect top text columns; archive or move large payloads to R2.")
206
+ steps.append("Cap writers at source (result ceilings / retention).")
207
+ findings.append(
208
+ {
209
+ "severity": "high"
210
+ if est >= 5_000_000 or s.row_count >= 100_000
211
+ else "medium",
212
+ "table": s.name,
213
+ "rows": s.row_count,
214
+ "est_bytes": est,
215
+ "est_human": _fmt_bytes(est),
216
+ "why": "; ".join(reasons),
217
+ "next_steps": steps,
218
+ "top_columns": [
219
+ {"name": c.name, "bytes": c.bytes, "human": _fmt_bytes(c.bytes)}
220
+ for c in sorted(s.columns, key=lambda x: x.bytes, reverse=True)[:5]
221
+ ],
222
+ }
223
+ )
224
+ return findings
225
+
226
+
227
+ def _build_doing_well(
228
+ stats: list[TableStat], findings: list[dict[str, Any]]
229
+ ) -> list[dict[str, str]]:
230
+ flagged = {f["table"] for f in findings}
231
+ well: list[dict[str, str]] = []
232
+ small = [s for s in stats if not s.error and s.row_count < 1_000 and s.name not in flagged]
233
+ if small:
234
+ well.append(
235
+ {
236
+ "area": "small_tables",
237
+ "why": f"{len(small)} tables under 1k rows and not flagged — fine for registry/config.",
238
+ }
239
+ )
240
+ empty = sum(1 for s in stats if not s.error and s.row_count == 0)
241
+ if empty:
242
+ well.append(
243
+ {
244
+ "area": "empty_tables",
245
+ "why": f"{empty} empty tables (candidates to drop later, not urgent).",
246
+ }
247
+ )
248
+ if not findings:
249
+ well.append(
250
+ {"area": "no_flags", "why": "No high/medium bloat heuristics fired on this pass."}
251
+ )
252
+ return well
253
+
254
+
255
+ def _build_briefing(
256
+ stats: list[TableStat],
257
+ findings: list[dict[str, Any]],
258
+ doing_well: list[dict[str, str]],
259
+ db_name: str,
260
+ db_size: Optional[str],
261
+ mode: str,
262
+ table_total: int,
263
+ ) -> str:
264
+ high = sum(1 for f in findings if f["severity"] == "high")
265
+ med = sum(1 for f in findings if f["severity"] == "medium")
266
+ verdict = "needs_attention" if high or med else "healthy"
267
+ lines = [
268
+ f"# D1 health — {db_name}",
269
+ "",
270
+ f"- **Generated:** {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M UTC')}",
271
+ f"- **Mode:** {mode} (database-scoped; not tenant/workspace)",
272
+ f"- **Reported DB size:** {db_size or 'unknown'}",
273
+ f"- **Tables:** {table_total} scanned",
274
+ f"- **Verdict:** {verdict} ({high} high / {med} medium findings)",
275
+ "",
276
+ ]
277
+ if mode == "quick":
278
+ lines.append(
279
+ "> Quick = row counts only. Run `--full` for text/JSON size estimates and rollup guidance."
280
+ )
281
+ lines.append("")
282
+ else:
283
+ lines.append(
284
+ f"> Full text estimate (scanned columns): {_fmt_bytes(sum(s.text_bytes for s in stats))}."
285
+ )
286
+ lines.append(
287
+ "> D1 remote has no `dbstat` — sizes are `SUM(LENGTH(...))`, not exact page bytes."
288
+ )
289
+ lines.append("")
290
+
291
+ lines.append("## What's fine")
292
+ lines.append("")
293
+ if doing_well:
294
+ for w in doing_well:
295
+ lines.append(f"- **{w['area']}:** {w['why']}")
296
+ else:
297
+ lines.append("- (nothing notable)")
298
+ lines.append("")
299
+
300
+ lines.append("## What's bloated / needs attention")
301
+ lines.append("")
302
+ if not findings:
303
+ lines.append("- None flagged on this pass.")
304
+ else:
305
+ for i, f in enumerate(findings[:25], 1):
306
+ lines.append(
307
+ f"{i}. **`{f['table']}`** [{f['severity']}] — "
308
+ f"rows={f['rows']:,}, est={f['est_human']}"
309
+ )
310
+ lines.append(f" - Why: {f['why']}")
311
+ for step in f.get("next_steps") or []:
312
+ lines.append(f" - Next: {step}")
313
+ lines.append("")
314
+
315
+ lines.append("## Next steps (global)")
316
+ lines.append("")
317
+ if mode == "quick":
318
+ lines.append("1. `agentsam data d1-bloat --full --format json` for size + column detail.")
319
+ lines.append("2. Prioritize tables with ≥100k rows from the inventory above.")
320
+ else:
321
+ lines.append("1. Act on high findings first (archive / R2 / writer caps).")
322
+ lines.append("2. Re-run `--quick` weekly for row-count drift; `--full` after large ingest.")
323
+ lines.append("3. Prefer D1 for pointers + state; R2 for bytes; Vectorize/Supabase for search.")
324
+ lines.append("")
325
+ return "\n".join(lines)
326
+
327
+
328
+ def run(tool_input: ToolInput) -> ToolResult:
329
+ started = start_timer()
330
+ tool_input.assert_read_only()
331
+ p = tool_input.params
332
+ mode = tool_input.mode if tool_input.mode in ("quick", "full") else "quick"
333
+ prefix = p.get("prefix")
334
+ workers = int(p.get("workers", 6))
335
+ top = int(p.get("top", 40))
336
+ analyze_text = mode == "full"
337
+
338
+ output_dir = tool_input.output_path()
339
+
340
+ try:
341
+ adapter = D1Adapter.from_env(
342
+ db_name=p.get("db"),
343
+ wrangler_config=p.get("config"),
344
+ repo_root=p.get("repo_root"),
345
+ )
346
+ all_tables = adapter.list_tables()
347
+ tables = all_tables
348
+ if prefix:
349
+ tables = [t for t in all_tables if t.lower().startswith(str(prefix).lower())]
350
+
351
+ db_size = adapter.database_size()
352
+ stats: list[TableStat] = []
353
+ with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
354
+ futures = {
355
+ pool.submit(_scan_table, adapter, t, analyze_text): t for t in tables
356
+ }
357
+ for fut in as_completed(futures):
358
+ try:
359
+ stats.append(fut.result())
360
+ except Exception as e: # noqa: BLE001
361
+ stats.append(TableStat(name=futures[fut], error=str(e)))
362
+
363
+ if mode == "quick":
364
+ stats.sort(key=lambda s: s.row_count, reverse=True)
365
+ else:
366
+ stats.sort(key=lambda s: s.text_bytes or s.est_bytes, reverse=True)
367
+
368
+ findings = _build_findings(stats, mode)
369
+ doing_well = _build_doing_well(stats, findings)
370
+ high = sum(1 for f in findings if f["severity"] == "high")
371
+ med = sum(1 for f in findings if f["severity"] == "medium")
372
+ verdict = "needs_attention" if high or med else "healthy"
373
+ md = _build_briefing(
374
+ stats, findings, doing_well, adapter.db_name, db_size, mode, len(tables)
375
+ )
376
+
377
+ json_payload = {
378
+ "database": adapter.db_name,
379
+ "scope": "database",
380
+ "mode": mode,
381
+ "verdict": verdict,
382
+ "database_size": db_size,
383
+ "tables_total": len(all_tables),
384
+ "tables_scanned": len(tables),
385
+ "estimated_text_bytes": sum(s.text_bytes for s in stats),
386
+ "doing_well": doing_well,
387
+ "findings": findings,
388
+ "next_steps_global": (
389
+ [
390
+ "agentsam data d1-bloat --full --format json",
391
+ "Act on tables with ≥100k rows first",
392
+ ]
393
+ if mode == "quick"
394
+ else [
395
+ "Act on high findings (archive / R2 / writer caps)",
396
+ "Re-run --quick weekly for row drift",
397
+ ]
398
+ ),
399
+ "tables": [
400
+ {
401
+ **{k: v for k, v in asdict(s).items() if k != "columns"},
402
+ "columns": [asdict(c) for c in s.columns],
403
+ "rollup_hint": s.rollup_hint,
404
+ }
405
+ for s in stats[:top]
406
+ ],
407
+ }
408
+
409
+ artifacts: list[str] = []
410
+ if output_dir:
411
+ output_dir.mkdir(parents=True, exist_ok=True)
412
+ (output_dir / "d1-bloat.md").write_text(md, encoding="utf-8")
413
+ (output_dir / "d1-bloat.json").write_text(
414
+ json.dumps(json_payload, indent=2), encoding="utf-8"
415
+ )
416
+ artifacts = [str(output_dir / "d1-bloat.md"), str(output_dir / "d1-bloat.json")]
417
+
418
+ result = ToolResult(
419
+ ok=True,
420
+ tool=TOOL_NAME,
421
+ mode=mode,
422
+ request_id=tool_input.request_id,
423
+ started_at=started,
424
+ finished_at=start_timer(),
425
+ summary=(
426
+ f"Scanned {len(tables)}/{len(all_tables)} tables "
427
+ f"(mode={mode}, verdict={verdict}, findings={len(findings)})."
428
+ ),
429
+ data=json_payload,
430
+ artifacts=artifacts,
431
+ )
432
+ except D1AdapterError as e:
433
+ result = ToolResult(
434
+ ok=False,
435
+ tool=TOOL_NAME,
436
+ mode=mode,
437
+ request_id=tool_input.request_id,
438
+ started_at=started,
439
+ finished_at=start_timer(),
440
+ summary="D1 adapter error",
441
+ error=str(e),
442
+ )
443
+
444
+ write_receipt(result, output_dir)
445
+ return result
@@ -0,0 +1,26 @@
1
+ """Repository-level audits (inventory, scan_bloat, inspect).
2
+
3
+ Lazy exports so `python -m agentsam_sdk.repository.inspect` does not warn.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from typing import Any
9
+
10
+ __all__ = ["inventory", "scan_bloat", "inspect"]
11
+
12
+
13
+ def __getattr__(name: str) -> Any:
14
+ if name == "inventory":
15
+ from agentsam_sdk.repository import inventory as mod
16
+
17
+ return mod
18
+ if name == "scan_bloat":
19
+ from agentsam_sdk.repository import scan_bloat as mod
20
+
21
+ return mod
22
+ if name == "inspect":
23
+ from agentsam_sdk.repository import inspect as mod
24
+
25
+ return mod
26
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,3 @@
1
+ from agentsam_sdk.repository.inventory import main
2
+
3
+ raise SystemExit(main())