@inneranimalmedia/agentsam-sdk 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/DEVELOPMENT.md +24 -4
  2. package/README.md +2 -0
  3. package/docs/RELEASES.md +6 -0
  4. package/package.json +8 -3
  5. package/protocol/README.md +51 -0
  6. package/protocol/dual-repo-sync.md +35 -0
  7. package/python/README.md +12 -0
  8. package/python/agentsam_sdk/__init__.py +9 -0
  9. package/python/agentsam_sdk/cli.py +212 -0
  10. package/python/agentsam_sdk/data/__init__.py +0 -0
  11. package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
  12. package/python/agentsam_sdk/data/d1_adapter.py +124 -0
  13. package/python/agentsam_sdk/data/d1_bloat.py +265 -0
  14. package/python/agentsam_sdk/repository/__init__.py +0 -0
  15. package/python/agentsam_sdk/repository/__main__.py +3 -0
  16. package/python/agentsam_sdk/repository/inventory.py +351 -0
  17. package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
  18. package/python/agentsam_sdk/runtime/__init__.py +0 -0
  19. package/python/agentsam_sdk/runtime/contract.py +105 -0
  20. package/python/docs/gaps.md +63 -0
  21. package/python/docs/tooling.md +67 -0
  22. package/python/protocol/README.md +51 -0
  23. package/python/protocol/dual-repo-sync.md +35 -0
  24. package/python/pyproject.toml +16 -0
  25. package/python/scripts/check-host-tooling.sh +65 -0
  26. package/python/tests/__init__.py +0 -0
  27. package/python/tests/fixtures/sample_tables.json +17 -0
  28. package/python/tests/fixtures.py +95 -0
  29. package/python/tests/test_agentsam_walk.py +31 -0
  30. package/python/tests/test_contract.py +32 -0
  31. package/python/tests/test_d1_bloat.py +57 -0
  32. package/python/tests/test_repository_inventory.py +53 -0
  33. package/python/tests/test_scan_bloat.py +31 -0
@@ -0,0 +1,124 @@
1
+ """D1 adapter -- the only place this package shells out to wrangler.
2
+
3
+ Every other data/*.py module must go through D1Adapter, never call
4
+ subprocess/urllib against D1 directly (contract rule: "D1 access only via
5
+ an adapter"). Resolves database name / wrangler config from env or explicit
6
+ args -- never a hardcoded database id or account id (HARD LAW).
7
+
8
+ Env vars (all optional overrides; wrangler itself resolves the account via
9
+ CLOUDFLARE_API_TOKEN / `wrangler login` state and the database id via the
10
+ wrangler config file's [[d1_databases]] binding -- so this adapter does not
11
+ need to know the raw D1 database id at all):
12
+
13
+ AGENTSAM_D1_DB_NAME D1 database name (required -- no default;
14
+ pass --db or set this)
15
+ AGENTSAM_WRANGLER_CONFIG path to wrangler config, default "wrangler.toml"
16
+ AGENTSAM_REPO_ROOT repo root wrangler runs from, default cwd
17
+ CLOUDFLARE_API_TOKEN passed straight through to the wrangler subprocess
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import os
23
+ import re
24
+ import subprocess
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import Optional
28
+
29
+
30
+ class D1AdapterError(RuntimeError):
31
+ pass
32
+
33
+
34
+ @dataclass
35
+ class D1Adapter:
36
+ db_name: str
37
+ wrangler_config: str = "wrangler.toml"
38
+ repo_root: Optional[Path] = None
39
+ remote: bool = True
40
+ timeout_s: int = 120
41
+
42
+ @classmethod
43
+ def from_env(
44
+ cls,
45
+ db_name: Optional[str] = None,
46
+ wrangler_config: Optional[str] = None,
47
+ repo_root: Optional[str] = None,
48
+ ) -> "D1Adapter":
49
+ name = db_name or os.environ.get("AGENTSAM_D1_DB_NAME")
50
+ if not name:
51
+ raise D1AdapterError(
52
+ "No D1 database name given -- pass --db or set "
53
+ "AGENTSAM_D1_DB_NAME. Refusing to guess/hardcode one."
54
+ )
55
+ cfg = wrangler_config or os.environ.get("AGENTSAM_WRANGLER_CONFIG", "wrangler.toml")
56
+ root = Path(repo_root or os.environ.get("AGENTSAM_REPO_ROOT") or Path.cwd())
57
+ return cls(db_name=name, wrangler_config=cfg, repo_root=root)
58
+
59
+ def _run_wrangler(self, args: list[str]) -> subprocess.CompletedProcess:
60
+ cmd = ["npx", "wrangler", *args]
61
+ env = os.environ.copy()
62
+ return subprocess.run(
63
+ cmd,
64
+ cwd=str(self.repo_root) if self.repo_root else None,
65
+ capture_output=True,
66
+ text=True,
67
+ timeout=self.timeout_s,
68
+ env=env,
69
+ )
70
+
71
+ def query(self, sql: str) -> list[dict]:
72
+ args = ["d1", "execute", self.db_name]
73
+ if self.remote:
74
+ args.append("--remote")
75
+ args += ["-c", self.wrangler_config, "--json", "--command", sql]
76
+ proc = self._run_wrangler(args)
77
+ raw = (proc.stdout or "").strip()
78
+ if not raw:
79
+ raise D1AdapterError((proc.stderr or "empty wrangler output")[:400])
80
+ try:
81
+ data = json.loads(raw)
82
+ except json.JSONDecodeError as e:
83
+ raise D1AdapterError(f"non-JSON wrangler output: {e}") from e
84
+ if isinstance(data, dict) and data.get("error"):
85
+ raise D1AdapterError(str(data["error"])[:400])
86
+ if isinstance(data, list) and data:
87
+ return data[0].get("results") or []
88
+ return []
89
+
90
+ def database_size(self) -> Optional[str]:
91
+ args = ["d1", "info", self.db_name, "-c", self.wrangler_config]
92
+ proc = self._run_wrangler(args)
93
+ m = re.search(r"database_size\s*\│\s*([^\│]+)", proc.stdout or "")
94
+ return m.group(1).strip() if m else None
95
+
96
+ def list_tables(self, like: Optional[str] = None) -> list[str]:
97
+ sql = (
98
+ "SELECT name FROM sqlite_master WHERE type='table' "
99
+ "AND name NOT LIKE 'sqlite_%' AND name NOT LIKE '_cf_%'"
100
+ )
101
+ if like:
102
+ safe = like.replace("'", "")
103
+ sql += f" AND name LIKE '{safe}'"
104
+ sql += " ORDER BY name"
105
+ rows = self.query(sql)
106
+ return [r["name"] for r in rows]
107
+
108
+ def table_columns(self, table: str) -> list[tuple[str, str]]:
109
+ safe = table.replace("'", "")
110
+ rows = self.query(f"SELECT name, type FROM pragma_table_info('{safe}')")
111
+ return [(r["name"], r.get("type") or "TEXT") for r in rows]
112
+
113
+ def table_indexes(self, table: str) -> list[dict]:
114
+ safe = table.replace('"', "")
115
+ return self.query(f'PRAGMA index_list("{safe}")')
116
+
117
+ def foreign_keys(self, table: str) -> list[dict]:
118
+ safe = table.replace('"', "")
119
+ return self.query(f'PRAGMA foreign_key_list("{safe}")')
120
+
121
+ def row_count(self, table: str) -> int:
122
+ safe = table.replace('"', "")
123
+ rows = self.query(f'SELECT COUNT(*) AS rc FROM "{safe}"')
124
+ return int(rows[0].get("rc") or 0) if rows else 0
@@ -0,0 +1,265 @@
1
+ """agentsam_sdk.data.d1_bloat -- port of scripts/d1_bloat_audit.py.
2
+
3
+ Finds largest / text-heavy D1 tables via row counts + SUM(LENGTH(text_col))
4
+ (D1 remote has no dbstat, so this is an estimate, not exact page bytes).
5
+
6
+ Deferred from the legacy script (see docs/gaps.md): --email/Resend delivery.
7
+ This module writes json + markdown only; wire up email at the CLI/ops layer
8
+ if still wanted -- keeps this module free of a RESEND_API_KEY dependency.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import re
13
+ from concurrent.futures import ThreadPoolExecutor, as_completed
14
+ from dataclasses import dataclass, field, asdict
15
+ from datetime import datetime, timezone
16
+ from pathlib import Path
17
+ from typing import Any, Optional
18
+
19
+ from agentsam_sdk.data.d1_adapter import D1Adapter, D1AdapterError
20
+ from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
21
+
22
+ TOOL_NAME = "data.d1_bloat"
23
+
24
+ BLOAT_COL_RE = re.compile(
25
+ r"(body|content|value|markdown|_json\b|schema|payload|output|prompt|message|"
26
+ r"text|config|metadata|description|notes|script|summary|arguments|result|"
27
+ r"attributes|events|resource|handler|input_|output_|sql\b|embedding|merged_)",
28
+ re.I,
29
+ )
30
+ SKIP_COL_RE = re.compile(
31
+ r"(^id$|_id$|_at$|_at_epoch$|_hash$|_key$|_uuid$|_ref$|_url$|_path$|_email$|"
32
+ r"_slug$|_name$|_type$|_status$|_mode$|_token$|tenant_id|workspace_id|user_id)",
33
+ re.I,
34
+ )
35
+ QUICK_TABLE_RE = re.compile(
36
+ r"^(agentsam_|otlp_|system_health|deployment|terminal_|cms_|worker_analytics|"
37
+ r"ai_api_test|dashboard_versions|semantic_search)",
38
+ re.I,
39
+ )
40
+
41
+ # Table-name -> rollup guidance. Naming convention hints, not identity/secrets;
42
+ # safe to ship. Extend freely -- this is documentation, not config.
43
+ ROLLUP_HINTS: dict[str, str] = {
44
+ "agentsam_tool_call_log": "Archive/purge output_json + input_json >30d; keep output_summary + ids.",
45
+ "agentsam_tool_chain": "result_json dominates -- rollup to object storage or truncate JSON.",
46
+ "agentsam_tool_cache": "Cache table -- enforce TTL + max rows.",
47
+ "agentsam_workflow_runs": "Move step_results_json to R2 artifact; D1 row = pointer + status + cost.",
48
+ "agentsam_scripts": "body must stay empty; canonical source in R2 (source_stored=r2:...).",
49
+ "agentsam_skill": "Large SKILL.md -> R2; D1 = metadata + retrieval_strategy=r2.",
50
+ "agentsam_memory": "value is prose -- OK for pinned rows; vectors live in Supabase/Vectorize.",
51
+ "otlp_traces": "Retention policy on attributes_json; sample or export to observability backend.",
52
+ }
53
+
54
+
55
+ @dataclass
56
+ class ColStat:
57
+ name: str
58
+ bytes: int
59
+ max_len: int = 0
60
+
61
+
62
+ @dataclass
63
+ class TableStat:
64
+ name: str
65
+ row_count: int = 0
66
+ text_bytes: int = 0
67
+ est_bytes: int = 0
68
+ columns: list[ColStat] = field(default_factory=list)
69
+ error: Optional[str] = None
70
+
71
+ @property
72
+ def rollup_hint(self) -> Optional[str]:
73
+ if self.name in ROLLUP_HINTS:
74
+ return ROLLUP_HINTS[self.name]
75
+ if self.text_bytes > 500_000 and any(
76
+ c.name in ("body", "content_markdown", "value", "output_json", "result_json", "payload_json")
77
+ for c in self.columns
78
+ ):
79
+ return "Large text/JSON in D1 -- prefer R2 pointer + vector lanes for search."
80
+ return None
81
+
82
+
83
+ def _pick_bloat_columns(cols: list[tuple[str, str]], max_cols: int = 8) -> list[str]:
84
+ out: list[str] = []
85
+ for name, typ in cols:
86
+ if typ not in ("TEXT", "BLOB", "JSON"):
87
+ continue
88
+ if SKIP_COL_RE.search(name):
89
+ continue
90
+ if BLOAT_COL_RE.search(name):
91
+ out.append(name)
92
+ return out[:max_cols]
93
+
94
+
95
+ def _scan_table(adapter: D1Adapter, table: str, analyze_text: bool) -> TableStat:
96
+ stat = TableStat(name=table)
97
+ try:
98
+ cols = adapter.table_columns(table)
99
+ bloat_cols = _pick_bloat_columns(cols) if analyze_text else []
100
+ if bloat_cols:
101
+ parts = [f'SUM(LENGTH(COALESCE("{c}", \'\'))) AS "{c}"' for c in bloat_cols]
102
+ max_parts = [f'MAX(LENGTH(COALESCE("{c}", \'\'))) AS "m_{c}"' for c in bloat_cols]
103
+ sql = f'SELECT COUNT(*) AS rc, {", ".join(parts + max_parts)} FROM "{table}"'
104
+ row = adapter.query(sql)[0]
105
+ stat.row_count = int(row.get("rc") or 0)
106
+ for c in bloat_cols:
107
+ b = int(row.get(c) or 0)
108
+ if b:
109
+ stat.columns.append(ColStat(name=c, bytes=b, max_len=int(row.get(f"m_{c}") or 0)))
110
+ stat.text_bytes = sum(c.bytes for c in stat.columns)
111
+ stat.est_bytes = stat.text_bytes if stat.text_bytes else stat.row_count * 120
112
+ else:
113
+ stat.row_count = adapter.row_count(table)
114
+ stat.est_bytes = stat.row_count * 120
115
+ except D1AdapterError as e:
116
+ stat.error = str(e)[:200]
117
+ except Exception as e: # noqa: BLE001 -- surfaced in receipt, not swallowed
118
+ stat.error = str(e)[:200]
119
+ return stat
120
+
121
+
122
+ def _fmt_bytes(n: int) -> str:
123
+ if n >= 1024 * 1024:
124
+ return f"{n / 1024 / 1024:.2f} MB"
125
+ if n >= 1024:
126
+ return f"{n / 1024:.1f} KB"
127
+ return f"{n} B"
128
+
129
+
130
+ def _flag_suspicious(stats: list[TableStat]) -> list[dict[str, Any]]:
131
+ flags: list[dict[str, Any]] = []
132
+ ranked = sorted(stats, key=lambda s: s.text_bytes or s.est_bytes, reverse=True)
133
+ total_text = sum(s.text_bytes for s in stats) or 1
134
+ for s in ranked[:80]:
135
+ est = s.text_bytes or s.est_bytes
136
+ reasons: list[str] = []
137
+ if est >= 5_000_000:
138
+ reasons.append(f"text_est>={_fmt_bytes(5_000_000)}")
139
+ if s.row_count >= 100_000:
140
+ reasons.append(f"rows>={s.row_count:,}")
141
+ if est >= 1_000_000 and (est / total_text) >= 0.08:
142
+ reasons.append(f"share>={100 * est / total_text:.0f}% of scanned text")
143
+ if s.rollup_hint and est >= 500_000:
144
+ reasons.append("known_rollup_candidate")
145
+ if s.error:
146
+ reasons.append(f"scan_error:{s.error[:80]}")
147
+ if reasons:
148
+ flags.append({
149
+ "table": s.name, "rows": s.row_count, "est_bytes": est,
150
+ "est_human": _fmt_bytes(est),
151
+ "severity": "high" if est >= 5_000_000 or s.row_count >= 100_000 else "medium",
152
+ "reasons": reasons, "hint": s.rollup_hint,
153
+ })
154
+ return flags
155
+
156
+
157
+ def _render_markdown(stats: list[TableStat], db_size, mode, table_total, scanned) -> str:
158
+ lines = [
159
+ "# D1 bloat audit",
160
+ "",
161
+ f"- **Generated:** {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M UTC')}",
162
+ f"- **Reported DB size:** {db_size or 'unknown'}",
163
+ f"- **Mode:** {mode}",
164
+ f"- **Tables in DB:** {table_total}",
165
+ f"- **Tables scanned:** {scanned}",
166
+ f"- **Estimated text payload (scanned):** {_fmt_bytes(sum(s.text_bytes for s in stats))}",
167
+ "",
168
+ "> D1 remote has no `dbstat`. Sizes are `SUM(LENGTH(text_col))` estimates.",
169
+ "",
170
+ "## Top tables by estimated text bytes",
171
+ "",
172
+ "| Rank | Table | Rows | Text est. | Top columns | Rollup hint |",
173
+ "|------|-------|------|-----------|-------------|-------------|",
174
+ ]
175
+ ranked = sorted(stats, key=lambda s: s.est_bytes, reverse=True)
176
+ for i, s in enumerate(ranked[:40], 1):
177
+ top_cols = ", ".join(
178
+ f"`{c.name}` {_fmt_bytes(c.bytes)}"
179
+ for c in sorted(s.columns, key=lambda x: x.bytes, reverse=True)[:3]
180
+ )
181
+ hint = (s.rollup_hint or "").replace("|", "/")[:80]
182
+ err = f" ⚠ {s.error}" if s.error else ""
183
+ lines.append(
184
+ f"| {i} | `{s.name}` | {s.row_count:,} | {_fmt_bytes(s.text_bytes or s.est_bytes)} | "
185
+ f"{top_cols or '—'} | {hint or '—'}{err} |"
186
+ )
187
+ lines.append("")
188
+ return "\n".join(lines)
189
+
190
+
191
+ def run(tool_input: ToolInput) -> ToolResult:
192
+ started = start_timer()
193
+ p = tool_input.params
194
+ mode = tool_input.mode if tool_input.mode in ("quick", "full") else "quick"
195
+ prefix = p.get("prefix")
196
+ workers = int(p.get("workers", 6))
197
+ count_only = bool(p.get("count_only", False))
198
+ top = int(p.get("top", 40))
199
+
200
+ output_dir = tool_input.output_path()
201
+
202
+ try:
203
+ adapter = D1Adapter.from_env(
204
+ db_name=p.get("db"), wrangler_config=p.get("config"), repo_root=p.get("repo_root")
205
+ )
206
+ all_tables = adapter.list_tables()
207
+ if prefix:
208
+ tables = [t for t in all_tables if t.lower().startswith(prefix.lower())]
209
+ elif mode == "quick":
210
+ tables = [t for t in all_tables if QUICK_TABLE_RE.match(t)]
211
+ else:
212
+ tables = all_tables
213
+
214
+ db_size = adapter.database_size()
215
+ stats: list[TableStat] = []
216
+ with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
217
+ futures = {pool.submit(_scan_table, adapter, t, not count_only): t for t in tables}
218
+ for fut in as_completed(futures):
219
+ try:
220
+ stats.append(fut.result())
221
+ except Exception as e: # noqa: BLE001
222
+ stats.append(TableStat(name=futures[fut], error=str(e)))
223
+
224
+ stats.sort(key=lambda s: s.est_bytes, reverse=True)
225
+ flags = _flag_suspicious(stats)
226
+ md = _render_markdown(stats, db_size, mode, len(all_tables), len(tables))
227
+
228
+ artifacts: list[str] = []
229
+ json_payload = {
230
+ "database": adapter.db_name,
231
+ "database_size": db_size,
232
+ "mode": mode,
233
+ "tables_total": len(all_tables),
234
+ "tables_scanned": len(tables),
235
+ "estimated_text_bytes": sum(s.text_bytes for s in stats),
236
+ "flags": flags,
237
+ "tables": [
238
+ {**{k: v for k, v in asdict(s).items() if k != "columns"},
239
+ "columns": [asdict(c) for c in s.columns], "rollup_hint": s.rollup_hint}
240
+ for s in stats[:top]
241
+ ],
242
+ }
243
+ if output_dir:
244
+ output_dir.mkdir(parents=True, exist_ok=True)
245
+ (output_dir / "d1-bloat.md").write_text(md, encoding="utf-8")
246
+ (output_dir / "d1-bloat.json").write_text(
247
+ __import__("json").dumps(json_payload, indent=2), encoding="utf-8"
248
+ )
249
+ artifacts = [str(output_dir / "d1-bloat.md"), str(output_dir / "d1-bloat.json")]
250
+
251
+ result = ToolResult(
252
+ ok=True, tool=TOOL_NAME, mode=mode, request_id=tool_input.request_id,
253
+ started_at=started, finished_at=start_timer(),
254
+ summary=f"Scanned {len(tables)}/{len(all_tables)} tables, {len(flags)} flagged.",
255
+ data=json_payload, artifacts=artifacts,
256
+ )
257
+ except D1AdapterError as e:
258
+ result = ToolResult(
259
+ ok=False, tool=TOOL_NAME, mode=mode, request_id=tool_input.request_id,
260
+ started_at=started, finished_at=start_timer(),
261
+ summary="D1 adapter error", error=str(e),
262
+ )
263
+
264
+ write_receipt(result, output_dir)
265
+ return result
File without changes
@@ -0,0 +1,3 @@
1
+ from agentsam_sdk.repository.inventory import main
2
+
3
+ raise SystemExit(main())