@inneranimalmedia/agentsam-sdk 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/DEVELOPMENT.md +25 -5
- package/README.md +2 -0
- package/docs/RELEASES.iam-mirror.md +20 -0
- package/docs/RELEASES.md +9 -0
- package/package.json +9 -4
- package/protocol/README.md +51 -0
- package/protocol/dual-repo-sync.md +35 -0
- package/python/README.md +12 -0
- package/python/agentsam_sdk/__init__.py +9 -0
- package/python/agentsam_sdk/cli.py +262 -0
- package/python/agentsam_sdk/data/__init__.py +0 -0
- package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
- package/python/agentsam_sdk/data/d1_adapter.py +124 -0
- package/python/agentsam_sdk/data/d1_bloat.py +445 -0
- package/python/agentsam_sdk/repository/__init__.py +26 -0
- package/python/agentsam_sdk/repository/__main__.py +3 -0
- package/python/agentsam_sdk/repository/inspect.py +496 -0
- package/python/agentsam_sdk/repository/inventory.py +351 -0
- package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
- package/python/agentsam_sdk/runtime/__init__.py +0 -0
- package/python/agentsam_sdk/runtime/contract.py +105 -0
- package/python/docs/gaps.md +63 -0
- package/python/docs/tooling.md +67 -0
- package/python/protocol/README.md +51 -0
- package/python/protocol/dual-repo-sync.md +35 -0
- package/python/pyproject.toml +16 -0
- package/python/scripts/check-host-tooling.sh +65 -0
- package/python/tests/__init__.py +0 -0
- package/python/tests/fixtures/sample_tables.json +17 -0
- package/python/tests/fixtures.py +95 -0
- package/python/tests/test_agentsam_walk.py +31 -0
- package/python/tests/test_contract.py +32 -0
- package/python/tests/test_d1_bloat.py +93 -0
- package/python/tests/test_repository_inspect.py +84 -0
- package/python/tests/test_repository_inventory.py +53 -0
- package/python/tests/test_scan_bloat.py +31 -0
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
"""agentsam_sdk.repository.inventory — repo file counts + sizes by category.
|
|
2
|
+
|
|
3
|
+
Port of the battle-tested scanner (formerly scripts/repo-size-inventory.py on
|
|
4
|
+
main). Read-only. No secrets, no D1.
|
|
5
|
+
|
|
6
|
+
JSON is jq-friendly. Examples (host `jq` required for the pipe examples):
|
|
7
|
+
|
|
8
|
+
agentsam repository inventory --repo-root .. --format json \\
|
|
9
|
+
| jq '.data.categories[] | select(.id==\"docs\")'
|
|
10
|
+
|
|
11
|
+
agentsam repository inventory --repo-root .. --output-dir /tmp/inv --format json
|
|
12
|
+
jq '.totals' /tmp/inv/repository-inventory.json
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
from collections import Counter, defaultdict
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Iterable
|
|
22
|
+
|
|
23
|
+
from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
|
|
24
|
+
|
|
25
|
+
TOOL_NAME = "repository.inventory"
|
|
26
|
+
|
|
27
|
+
DEFAULT_SKIP_DIR_NAMES = frozenset(
|
|
28
|
+
{
|
|
29
|
+
".git",
|
|
30
|
+
"node_modules",
|
|
31
|
+
".wrangler",
|
|
32
|
+
"dist",
|
|
33
|
+
"coverage",
|
|
34
|
+
"__pycache__",
|
|
35
|
+
".venv",
|
|
36
|
+
"venv",
|
|
37
|
+
".venv_agentsam",
|
|
38
|
+
".turbo",
|
|
39
|
+
".next",
|
|
40
|
+
".cache",
|
|
41
|
+
".scratch",
|
|
42
|
+
"build",
|
|
43
|
+
}
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# First path segment → category id
|
|
47
|
+
CATEGORY_RULES: list[tuple[str, tuple[str, ...]]] = [
|
|
48
|
+
("worker_src", ("src",)),
|
|
49
|
+
("dashboard", ("dashboard",)),
|
|
50
|
+
("migrations", ("migrations",)),
|
|
51
|
+
("docs", ("docs",)),
|
|
52
|
+
("plans", ("plans",)),
|
|
53
|
+
("scripts", ("scripts",)),
|
|
54
|
+
("tests", ("tests", "test", "e2e")),
|
|
55
|
+
("services", ("services",)),
|
|
56
|
+
("supabase", ("supabase",)),
|
|
57
|
+
("product_manifests", ("product-manifests",)),
|
|
58
|
+
("static_assets", ("static", "public", "assets")),
|
|
59
|
+
("artifacts", ("artifacts", ".scratch")),
|
|
60
|
+
("vendor", ("vendor",)),
|
|
61
|
+
("tools", ("tools",)),
|
|
62
|
+
("local_venvs", (".venv_agentsam", ".venv", "venv")),
|
|
63
|
+
("cms", ("cms-editor", "studio-cms")),
|
|
64
|
+
("config_cursor", (".cursor", ".agents", ".claude", ".codex", ".githooks")),
|
|
65
|
+
("ci", (".github",)),
|
|
66
|
+
("agentsam_sdk_pkg", ("agentsam-sdk", "agentsam_sdk")),
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
CATEGORY_LABELS = {
|
|
70
|
+
"worker_src": "Worker src/",
|
|
71
|
+
"dashboard": "Dashboard SPA",
|
|
72
|
+
"migrations": "D1 migrations",
|
|
73
|
+
"docs": "Docs",
|
|
74
|
+
"plans": "Plans",
|
|
75
|
+
"scripts": "Scripts",
|
|
76
|
+
"tests": "Tests",
|
|
77
|
+
"services": "Services / satellites",
|
|
78
|
+
"supabase": "Supabase",
|
|
79
|
+
"product_manifests": "Product manifests",
|
|
80
|
+
"static_assets": "Static / public assets",
|
|
81
|
+
"artifacts": "Artifacts / scratch dumps",
|
|
82
|
+
"vendor": "Vendor copies",
|
|
83
|
+
"tools": "Tools / offline utilities",
|
|
84
|
+
"local_venvs": "Local Python venvs",
|
|
85
|
+
"cms": "CMS editor packages",
|
|
86
|
+
"config_cursor": "Cursor / agent config",
|
|
87
|
+
"ci": "CI (.github)",
|
|
88
|
+
"agentsam_sdk_pkg": "agentsam-sdk package",
|
|
89
|
+
"root_misc": "Repo root files",
|
|
90
|
+
"other": "Other paths",
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass
|
|
95
|
+
class Bucket:
|
|
96
|
+
id: str
|
|
97
|
+
label: str
|
|
98
|
+
file_count: int = 0
|
|
99
|
+
bytes: int = 0
|
|
100
|
+
top_files: list[dict] = field(default_factory=list)
|
|
101
|
+
|
|
102
|
+
def add(self, size: int) -> None:
|
|
103
|
+
self.file_count += 1
|
|
104
|
+
self.bytes += size
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def human_bytes(n: int) -> str:
|
|
108
|
+
if n < 1024:
|
|
109
|
+
return f"{n} B"
|
|
110
|
+
units = ["KiB", "MiB", "GiB", "TiB"]
|
|
111
|
+
x = float(n)
|
|
112
|
+
for u in units:
|
|
113
|
+
x /= 1024.0
|
|
114
|
+
if x < 1024.0:
|
|
115
|
+
return f"{x:.2f} {u}"
|
|
116
|
+
return f"{x:.2f} PiB"
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def categorize(rel: Path) -> str:
|
|
120
|
+
parts = rel.parts
|
|
121
|
+
if not parts:
|
|
122
|
+
return "root_misc"
|
|
123
|
+
first = parts[0]
|
|
124
|
+
for cat_id, prefixes in CATEGORY_RULES:
|
|
125
|
+
if first in prefixes:
|
|
126
|
+
return cat_id
|
|
127
|
+
if len(parts) == 1:
|
|
128
|
+
return "root_misc"
|
|
129
|
+
return "other"
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def should_skip_dir(name: str, skip_names: frozenset[str]) -> bool:
|
|
133
|
+
return name in skip_names or name.endswith(".bak")
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def iter_files(
|
|
137
|
+
root: Path,
|
|
138
|
+
skip_names: frozenset[str],
|
|
139
|
+
follow_symlinks: bool,
|
|
140
|
+
) -> Iterable[tuple[Path, int]]:
|
|
141
|
+
for dirpath, dirnames, filenames in os.walk(
|
|
142
|
+
root, topdown=True, followlinks=follow_symlinks
|
|
143
|
+
):
|
|
144
|
+
dirnames[:] = [d for d in dirnames if not should_skip_dir(d, skip_names)]
|
|
145
|
+
base = Path(dirpath)
|
|
146
|
+
for name in filenames:
|
|
147
|
+
if name.endswith(".bak") or name.endswith(".pyc"):
|
|
148
|
+
continue
|
|
149
|
+
path = base / name
|
|
150
|
+
try:
|
|
151
|
+
if path.is_symlink() and not follow_symlinks:
|
|
152
|
+
continue
|
|
153
|
+
st = path.stat()
|
|
154
|
+
except (OSError, ValueError):
|
|
155
|
+
continue
|
|
156
|
+
if not path.is_file():
|
|
157
|
+
continue
|
|
158
|
+
yield path, int(st.st_size)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def scan(
|
|
162
|
+
root: Path,
|
|
163
|
+
*,
|
|
164
|
+
skip_names: frozenset[str],
|
|
165
|
+
top_n: int,
|
|
166
|
+
min_bytes: int,
|
|
167
|
+
follow_symlinks: bool,
|
|
168
|
+
by_ext: bool,
|
|
169
|
+
) -> dict[str, Any]:
|
|
170
|
+
buckets: dict[str, Bucket] = {}
|
|
171
|
+
ext_counts: Counter = Counter()
|
|
172
|
+
ext_bytes: dict[str, int] = defaultdict(int)
|
|
173
|
+
top_level: Counter = Counter()
|
|
174
|
+
largest: list[tuple[int, str, str]] = []
|
|
175
|
+
|
|
176
|
+
def bucket(cat_id: str) -> Bucket:
|
|
177
|
+
if cat_id not in buckets:
|
|
178
|
+
buckets[cat_id] = Bucket(
|
|
179
|
+
id=cat_id, label=CATEGORY_LABELS.get(cat_id, cat_id)
|
|
180
|
+
)
|
|
181
|
+
return buckets[cat_id]
|
|
182
|
+
|
|
183
|
+
file_total = 0
|
|
184
|
+
byte_total = 0
|
|
185
|
+
|
|
186
|
+
for path, size in iter_files(root, skip_names, follow_symlinks):
|
|
187
|
+
try:
|
|
188
|
+
rel = path.relative_to(root)
|
|
189
|
+
except ValueError:
|
|
190
|
+
continue
|
|
191
|
+
cat = categorize(rel)
|
|
192
|
+
bucket(cat).add(size)
|
|
193
|
+
file_total += 1
|
|
194
|
+
byte_total += size
|
|
195
|
+
|
|
196
|
+
top = rel.parts[0] if rel.parts else "."
|
|
197
|
+
top_level[top] += 1
|
|
198
|
+
|
|
199
|
+
ext = path.suffix.lower() or "(none)"
|
|
200
|
+
ext_counts[ext] += 1
|
|
201
|
+
if by_ext:
|
|
202
|
+
ext_bytes[ext] += size
|
|
203
|
+
|
|
204
|
+
if size >= min_bytes:
|
|
205
|
+
largest.append((size, str(rel).replace("\\", "/"), cat))
|
|
206
|
+
|
|
207
|
+
largest.sort(key=lambda t: t[0], reverse=True)
|
|
208
|
+
top_files = [
|
|
209
|
+
{"path": p, "bytes": s, "human": human_bytes(s), "category": c}
|
|
210
|
+
for s, p, c in largest[: max(0, top_n)]
|
|
211
|
+
]
|
|
212
|
+
|
|
213
|
+
cats = sorted(buckets.values(), key=lambda b: b.bytes, reverse=True)
|
|
214
|
+
categories = [
|
|
215
|
+
{
|
|
216
|
+
"id": b.id,
|
|
217
|
+
"label": b.label,
|
|
218
|
+
"file_count": b.file_count,
|
|
219
|
+
"bytes": b.bytes,
|
|
220
|
+
"human": human_bytes(b.bytes),
|
|
221
|
+
"pct_bytes": round(100.0 * b.bytes / byte_total, 2) if byte_total else 0.0,
|
|
222
|
+
}
|
|
223
|
+
for b in cats
|
|
224
|
+
]
|
|
225
|
+
|
|
226
|
+
out: dict[str, Any] = {
|
|
227
|
+
"ok": True,
|
|
228
|
+
"tool": TOOL_NAME,
|
|
229
|
+
"repo_root": str(root),
|
|
230
|
+
"file_total": file_total,
|
|
231
|
+
"totals": {
|
|
232
|
+
"file_count": file_total,
|
|
233
|
+
"bytes": byte_total,
|
|
234
|
+
"human": human_bytes(byte_total),
|
|
235
|
+
"categories": len(categories),
|
|
236
|
+
},
|
|
237
|
+
"skipped_dir_names": sorted(skip_names),
|
|
238
|
+
"categories": categories,
|
|
239
|
+
"largest_files": top_files,
|
|
240
|
+
# Backward-compatible stub fields (counts only)
|
|
241
|
+
"by_extension": dict(ext_counts.most_common(40)),
|
|
242
|
+
"by_top_level_dir": dict(top_level.most_common(40)),
|
|
243
|
+
}
|
|
244
|
+
if by_ext:
|
|
245
|
+
out["by_extension_detail"] = [
|
|
246
|
+
{
|
|
247
|
+
"ext": k,
|
|
248
|
+
"file_count": ext_counts[k],
|
|
249
|
+
"bytes": ext_bytes[k],
|
|
250
|
+
"human": human_bytes(ext_bytes[k]),
|
|
251
|
+
}
|
|
252
|
+
for k in sorted(ext_bytes, key=lambda e: ext_bytes[e], reverse=True)
|
|
253
|
+
]
|
|
254
|
+
return out
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _markdown(report: dict[str, Any]) -> str:
|
|
258
|
+
lines = [
|
|
259
|
+
"# Repository inventory",
|
|
260
|
+
"",
|
|
261
|
+
f"- **repo_root:** `{report['repo_root']}`",
|
|
262
|
+
f"- **files:** {report['file_total']:,}",
|
|
263
|
+
f"- **bytes:** {report['totals']['human']} ({report['totals']['bytes']:,})",
|
|
264
|
+
"",
|
|
265
|
+
"## By category",
|
|
266
|
+
"",
|
|
267
|
+
"| Category | Files | Size | % |",
|
|
268
|
+
"|---|---:|---:|---:|",
|
|
269
|
+
]
|
|
270
|
+
for c in report["categories"]:
|
|
271
|
+
lines.append(
|
|
272
|
+
f"| {c['label']} | {c['file_count']:,} | {c['human']} | {c['pct_bytes']:.1f}% |"
|
|
273
|
+
)
|
|
274
|
+
if report.get("largest_files"):
|
|
275
|
+
lines += ["", "## Largest files", ""]
|
|
276
|
+
for f in report["largest_files"]:
|
|
277
|
+
lines.append(f"- `{f['path']}` — {f['human']} (`{f['category']}`)")
|
|
278
|
+
lines += ["", "## By extension (top)", "", "| Ext | Count |", "|-----|-------|"]
|
|
279
|
+
for e, c in list(report.get("by_extension", {}).items())[:30]:
|
|
280
|
+
lines.append(f"| `{e}` | {c} |")
|
|
281
|
+
return "\n".join(lines) + "\n"
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def run(tool_input: ToolInput) -> ToolResult:
|
|
285
|
+
started = start_timer()
|
|
286
|
+
p = tool_input.params
|
|
287
|
+
repo_root = Path(p.get("repo_root", ".")).expanduser().resolve()
|
|
288
|
+
output_dir = tool_input.output_path()
|
|
289
|
+
|
|
290
|
+
if not repo_root.exists():
|
|
291
|
+
result = ToolResult(
|
|
292
|
+
ok=False,
|
|
293
|
+
tool=TOOL_NAME,
|
|
294
|
+
mode=tool_input.mode,
|
|
295
|
+
request_id=tool_input.request_id,
|
|
296
|
+
started_at=started,
|
|
297
|
+
finished_at=start_timer(),
|
|
298
|
+
summary=f"repo_root does not exist: {repo_root}",
|
|
299
|
+
error="repo_root_not_found",
|
|
300
|
+
)
|
|
301
|
+
write_receipt(result, output_dir)
|
|
302
|
+
return result
|
|
303
|
+
|
|
304
|
+
skip = set(p.get("skip_dir_names") or DEFAULT_SKIP_DIR_NAMES)
|
|
305
|
+
if p.get("include_node_modules"):
|
|
306
|
+
skip.discard("node_modules")
|
|
307
|
+
if p.get("include_venvs"):
|
|
308
|
+
skip.discard(".venv")
|
|
309
|
+
skip.discard("venv")
|
|
310
|
+
skip.discard(".venv_agentsam")
|
|
311
|
+
if p.get("include_git"):
|
|
312
|
+
skip.discard(".git")
|
|
313
|
+
if p.get("include_dist"):
|
|
314
|
+
skip.discard("dist")
|
|
315
|
+
skip.discard(".wrangler")
|
|
316
|
+
skip.discard("build")
|
|
317
|
+
|
|
318
|
+
report = scan(
|
|
319
|
+
repo_root,
|
|
320
|
+
skip_names=frozenset(skip),
|
|
321
|
+
top_n=int(p.get("top", 20)),
|
|
322
|
+
min_bytes=int(p.get("min_bytes", 0)),
|
|
323
|
+
follow_symlinks=bool(p.get("follow_symlinks", False)),
|
|
324
|
+
by_ext=bool(p.get("by_ext", True)),
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
artifacts: list[str] = []
|
|
328
|
+
if output_dir:
|
|
329
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
330
|
+
json_path = output_dir / "repository-inventory.json"
|
|
331
|
+
md_path = output_dir / "repository-inventory.md"
|
|
332
|
+
json_path.write_text(json.dumps(report, indent=2), encoding="utf-8")
|
|
333
|
+
md_path.write_text(_markdown(report), encoding="utf-8")
|
|
334
|
+
artifacts = [str(json_path), str(md_path)]
|
|
335
|
+
|
|
336
|
+
result = ToolResult(
|
|
337
|
+
ok=True,
|
|
338
|
+
tool=TOOL_NAME,
|
|
339
|
+
mode=tool_input.mode,
|
|
340
|
+
request_id=tool_input.request_id,
|
|
341
|
+
started_at=started,
|
|
342
|
+
finished_at=start_timer(),
|
|
343
|
+
summary=(
|
|
344
|
+
f"{report['file_total']} files · {report['totals']['human']} "
|
|
345
|
+
f"under {repo_root}"
|
|
346
|
+
),
|
|
347
|
+
data=report,
|
|
348
|
+
artifacts=artifacts,
|
|
349
|
+
)
|
|
350
|
+
write_receipt(result, output_dir)
|
|
351
|
+
return result
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""agentsam_sdk.repository.scan_bloat — per-file source bloat inventory.
|
|
2
|
+
|
|
3
|
+
Companion to repository.inventory (category rollups). This tool lists the
|
|
4
|
+
largest runtime source files under a root (default: cwd) with size / lines /
|
|
5
|
+
est. tokens — for refactor targeting and agent context budgeting.
|
|
6
|
+
|
|
7
|
+
Read-only. No secrets, no D1.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
|
|
17
|
+
|
|
18
|
+
TOOL_NAME = "repository.scan_bloat"
|
|
19
|
+
|
|
20
|
+
DEFAULT_EXTS = frozenset({".js", ".ts", ".jsx", ".tsx", ".mjs", ".cjs"})
|
|
21
|
+
DEFAULT_EXCLUDE_DIRS = frozenset(
|
|
22
|
+
{
|
|
23
|
+
"node_modules",
|
|
24
|
+
".git",
|
|
25
|
+
"dist",
|
|
26
|
+
"build",
|
|
27
|
+
".wrangler",
|
|
28
|
+
".next",
|
|
29
|
+
".turbo",
|
|
30
|
+
"coverage",
|
|
31
|
+
".cache",
|
|
32
|
+
"out",
|
|
33
|
+
".vercel",
|
|
34
|
+
"__pycache__",
|
|
35
|
+
".venv",
|
|
36
|
+
"venv",
|
|
37
|
+
}
|
|
38
|
+
)
|
|
39
|
+
CHARS_PER_TOKEN = 3.7
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def scan(
|
|
43
|
+
root: str | Path,
|
|
44
|
+
*,
|
|
45
|
+
exts: set[str] | frozenset[str] = DEFAULT_EXTS,
|
|
46
|
+
exclude_dirs: set[str] | frozenset[str] = DEFAULT_EXCLUDE_DIRS,
|
|
47
|
+
) -> list[dict[str, Any]]:
|
|
48
|
+
"""Walk root and return file stats sorted by size_bytes desc."""
|
|
49
|
+
root_path = Path(root).resolve()
|
|
50
|
+
results: list[dict[str, Any]] = []
|
|
51
|
+
if root_path.is_file():
|
|
52
|
+
# Allow a single-file root for smoke tests
|
|
53
|
+
if root_path.suffix in exts:
|
|
54
|
+
results.append(_stat_file(root_path, root_path.parent))
|
|
55
|
+
return results
|
|
56
|
+
|
|
57
|
+
for dirpath, dirnames, filenames in os.walk(root_path):
|
|
58
|
+
dirnames[:] = [
|
|
59
|
+
d for d in dirnames if d not in exclude_dirs and not d.startswith(".")
|
|
60
|
+
]
|
|
61
|
+
for fname in filenames:
|
|
62
|
+
ext = os.path.splitext(fname)[1]
|
|
63
|
+
if ext not in exts:
|
|
64
|
+
continue
|
|
65
|
+
fpath = Path(dirpath) / fname
|
|
66
|
+
try:
|
|
67
|
+
results.append(_stat_file(fpath, root_path))
|
|
68
|
+
except (OSError, UnicodeDecodeError):
|
|
69
|
+
continue
|
|
70
|
+
results.sort(key=lambda r: r["size_bytes"], reverse=True)
|
|
71
|
+
return results
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _stat_file(fpath: Path, root: Path) -> dict[str, Any]:
|
|
75
|
+
size_bytes = fpath.stat().st_size
|
|
76
|
+
content = fpath.read_text(encoding="utf-8", errors="replace")
|
|
77
|
+
line_count = content.count("\n") + (1 if content else 0)
|
|
78
|
+
return {
|
|
79
|
+
"path": str(fpath.relative_to(root)),
|
|
80
|
+
"size_bytes": size_bytes,
|
|
81
|
+
"size_kb": round(size_bytes / 1024, 1),
|
|
82
|
+
"lines": line_count,
|
|
83
|
+
"est_tokens": round(size_bytes / CHARS_PER_TOKEN),
|
|
84
|
+
"bytes_per_line": round(size_bytes / line_count, 1) if line_count else 0,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def human_table(files: list[dict[str, Any]], *, scanned: int, total_kb: float, total_tokens: int) -> str:
|
|
89
|
+
"""Plain-text table for terminal / markdown CLI mode."""
|
|
90
|
+
if not files:
|
|
91
|
+
return "No files matched.\n"
|
|
92
|
+
path_w = min(max(len(r["path"]) for r in files), 70)
|
|
93
|
+
header = f"{'SIZE':>9} {'LINES':>7} {'~TOKENS':>8} {'B/LINE':>7} PATH"
|
|
94
|
+
lines = [header, "-" * len(header)]
|
|
95
|
+
for r in files:
|
|
96
|
+
path = r["path"] if len(r["path"]) <= path_w else "…" + r["path"][-(path_w - 1) :]
|
|
97
|
+
lines.append(
|
|
98
|
+
f"{r['size_kb']:>8.1f}KB {r['lines']:>7} {r['est_tokens']:>8} "
|
|
99
|
+
f"{r['bytes_per_line']:>7} {path}"
|
|
100
|
+
)
|
|
101
|
+
lines.append("-" * len(header))
|
|
102
|
+
lines.append(
|
|
103
|
+
f"Scanned {scanned} files, {total_kb:.1f}KB total, ~{total_tokens:,} est. tokens"
|
|
104
|
+
)
|
|
105
|
+
return "\n".join(lines) + "\n"
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def run(tool_input: ToolInput) -> ToolResult:
|
|
109
|
+
started = start_timer()
|
|
110
|
+
p = tool_input.params
|
|
111
|
+
root = str(p.get("root") or p.get("repo_root") or ".")
|
|
112
|
+
top = max(1, int(p.get("top", 30)))
|
|
113
|
+
min_kb = float(p.get("min_kb", 0) or 0)
|
|
114
|
+
ext_raw = p.get("ext") or ",".join(sorted(DEFAULT_EXTS))
|
|
115
|
+
exts = {
|
|
116
|
+
e if str(e).startswith(".") else f".{e}"
|
|
117
|
+
for e in str(ext_raw).split(",")
|
|
118
|
+
if str(e).strip()
|
|
119
|
+
}
|
|
120
|
+
exclude = set(DEFAULT_EXCLUDE_DIRS)
|
|
121
|
+
extra = p.get("exclude") or ""
|
|
122
|
+
if extra:
|
|
123
|
+
exclude |= {d.strip() for d in str(extra).split(",") if d.strip()}
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
all_files = scan(root, exts=exts, exclude_dirs=exclude)
|
|
127
|
+
filtered = [r for r in all_files if r["size_kb"] >= min_kb][:top]
|
|
128
|
+
total_kb = round(sum(r["size_kb"] for r in all_files), 1)
|
|
129
|
+
total_tokens = sum(r["est_tokens"] for r in all_files)
|
|
130
|
+
data = {
|
|
131
|
+
"ok": True,
|
|
132
|
+
"root": str(Path(root).resolve()),
|
|
133
|
+
"file_count": len(all_files),
|
|
134
|
+
"total_kb": total_kb,
|
|
135
|
+
"total_est_tokens": total_tokens,
|
|
136
|
+
"files": filtered,
|
|
137
|
+
}
|
|
138
|
+
artifacts: list[str] = []
|
|
139
|
+
output_dir = tool_input.output_path()
|
|
140
|
+
if output_dir:
|
|
141
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
142
|
+
out_json = output_dir / "scan-bloat.json"
|
|
143
|
+
out_json.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
|
144
|
+
artifacts.append(str(out_json))
|
|
145
|
+
|
|
146
|
+
result = ToolResult(
|
|
147
|
+
ok=True,
|
|
148
|
+
tool=TOOL_NAME,
|
|
149
|
+
mode=tool_input.mode or "read-only",
|
|
150
|
+
request_id=tool_input.request_id,
|
|
151
|
+
started_at=started,
|
|
152
|
+
finished_at=start_timer(),
|
|
153
|
+
summary=(
|
|
154
|
+
f"Scanned {len(all_files)} files, {total_kb}KB total, "
|
|
155
|
+
f"top {len(filtered)} ≥{min_kb}KB."
|
|
156
|
+
),
|
|
157
|
+
data=data,
|
|
158
|
+
artifacts=artifacts,
|
|
159
|
+
)
|
|
160
|
+
except Exception as e: # noqa: BLE001 — surfaced in ToolResult
|
|
161
|
+
result = ToolResult(
|
|
162
|
+
ok=False,
|
|
163
|
+
tool=TOOL_NAME,
|
|
164
|
+
mode=tool_input.mode or "read-only",
|
|
165
|
+
request_id=tool_input.request_id,
|
|
166
|
+
started_at=started,
|
|
167
|
+
finished_at=start_timer(),
|
|
168
|
+
summary="scan_bloat failed",
|
|
169
|
+
error=str(e)[:500],
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
write_receipt(result, tool_input.output_path())
|
|
173
|
+
return result
|
|
File without changes
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Shared tool contract: ToolInput / ToolResult / receipts.
|
|
2
|
+
|
|
3
|
+
HARD LAW (see inneranimalmedia AGENTS.md): never hardcode identity, repo,
|
|
4
|
+
workspace, or tenant values anywhere in this package -- not in shipped code,
|
|
5
|
+
patches, examples, or fallback defaults. All identity-shaped values
|
|
6
|
+
(database names, account ids, wrangler config paths) must come from the
|
|
7
|
+
caller (CLI flag) or the environment. Generic placeholders only in examples
|
|
8
|
+
(e.g. "owner/repo-name", "$D1_DATABASE_NAME").
|
|
9
|
+
|
|
10
|
+
Every tool module in agentsam_sdk exposes a `run(tool_input: ToolInput) ->
|
|
11
|
+
ToolResult` function (or is wrapped to look like one from cli.py). Modes are
|
|
12
|
+
read-only by default -- any tool that can write must accept an explicit
|
|
13
|
+
`write=True` and should say so loudly in its ToolResult.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import time
|
|
20
|
+
import uuid
|
|
21
|
+
from dataclasses import dataclass, field, asdict
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, Optional
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
HARD_LAW_NOTE = (
|
|
27
|
+
"Never hardcode identity/repo/workspace/tenant values. Use env vars or "
|
|
28
|
+
"explicit CLI args; fall back to generic placeholders only in docs."
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ToolInput:
|
|
34
|
+
"""Normalized input every agentsam_sdk tool accepts.
|
|
35
|
+
|
|
36
|
+
mode: tool-specific sub-mode (e.g. "quick" | "full" for d1_bloat)
|
|
37
|
+
params: tool-specific keyword args
|
|
38
|
+
output_dir: where json/markdown output + receipt get written
|
|
39
|
+
write: True only for tools that mutate state; default False (read-only)
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
mode: str = "default"
|
|
43
|
+
params: dict[str, Any] = field(default_factory=dict)
|
|
44
|
+
output_dir: Optional[str] = None
|
|
45
|
+
write: bool = False
|
|
46
|
+
request_id: str = field(default_factory=lambda: uuid.uuid4().hex[:12])
|
|
47
|
+
|
|
48
|
+
def output_path(self) -> Optional[Path]:
|
|
49
|
+
return Path(self.output_dir) if self.output_dir else None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class ToolResult:
|
|
54
|
+
"""Normalized output every agentsam_sdk tool returns."""
|
|
55
|
+
|
|
56
|
+
ok: bool
|
|
57
|
+
tool: str
|
|
58
|
+
mode: str
|
|
59
|
+
request_id: str
|
|
60
|
+
started_at: float
|
|
61
|
+
finished_at: float
|
|
62
|
+
summary: str
|
|
63
|
+
data: dict[str, Any] = field(default_factory=dict)
|
|
64
|
+
artifacts: list[str] = field(default_factory=list)
|
|
65
|
+
error: Optional[str] = None
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def duration_s(self) -> float:
|
|
69
|
+
return round(self.finished_at - self.started_at, 3)
|
|
70
|
+
|
|
71
|
+
def to_dict(self) -> dict[str, Any]:
|
|
72
|
+
d = asdict(self)
|
|
73
|
+
d["duration_s"] = self.duration_s
|
|
74
|
+
return d
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def new_receipt_path(output_dir: Path, tool: str, request_id: str) -> Path:
|
|
78
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
return output_dir / f"receipt-{tool}-{request_id}.json"
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def write_receipt(result: ToolResult, output_dir: Optional[Path]) -> Optional[str]:
|
|
83
|
+
"""Write a receipt (contract rule: every tool call gets one). Returns path or None."""
|
|
84
|
+
if output_dir is None:
|
|
85
|
+
return None
|
|
86
|
+
path = new_receipt_path(output_dir, result.tool, result.request_id)
|
|
87
|
+
payload = result.to_dict()
|
|
88
|
+
payload["written_at_unixepoch"] = int(time.time())
|
|
89
|
+
path.write_text(json.dumps(payload, indent=2, default=str), encoding="utf-8")
|
|
90
|
+
return str(path)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def require_env(*names: str) -> dict[str, str]:
|
|
94
|
+
"""Fetch required env vars; raise with a clear message (never invent values)."""
|
|
95
|
+
missing = [n for n in names if not os.environ.get(n)]
|
|
96
|
+
if missing:
|
|
97
|
+
raise RuntimeError(
|
|
98
|
+
f"Missing required env var(s): {', '.join(missing)}. "
|
|
99
|
+
f"{HARD_LAW_NOTE}"
|
|
100
|
+
)
|
|
101
|
+
return {n: os.environ[n] for n in names}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def start_timer() -> float:
|
|
105
|
+
return time.time()
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# agentsam-sdk — D1 audit port: status
|
|
2
|
+
|
|
3
|
+
Tracks the file list from `plans/active/CLAUDE-AGENTSAM-SDK-D1-AUDIT-PORT-HANDOFF-2026-08.md`.
|
|
4
|
+
|
|
5
|
+
Context: `feature/agentsam-sdk-scaffold` had no `agentsam-sdk/` package at all
|
|
6
|
+
when this pass started (no pyproject, no CLI, no docs/gaps.md) -- the
|
|
7
|
+
handoff doc assumed a scaffold that hadn't actually landed. This pass builds
|
|
8
|
+
the minimal scaffold from zero, then ports P0.
|
|
9
|
+
|
|
10
|
+
Host tools (jq, wrangler, Python): see `docs/tooling.md` + `scripts/check-host-tooling.sh`.
|
|
11
|
+
|
|
12
|
+
## P0 — whole-D1 / agentsam walkers
|
|
13
|
+
|
|
14
|
+
| Legacy script | SDK target | Status |
|
|
15
|
+
|---|---|---|
|
|
16
|
+
| `scripts/d1_bloat_audit.py` | `agentsam_sdk.data.d1_bloat` | **Ported + aligned (2026-08).** Database-scoped: `--quick` = all tables COUNT(*); `--full` = text LENGTH + briefing. Removed `QUICK_TABLE_RE` / `SKIP_COL_RE` / `--count-only`. Deferred: `--email`/Resend (CLI/ops layer). |
|
|
17
|
+
| `scripts/run-d1-bloat-audit.sh` | `agentsam data d1-bloat` via CLI | **Shimmed**, see below — legacy `npm run audit:d1-bloat*` scripts still work unchanged. GCP-fallback/nohup wrapper behavior not reimplemented in the SDK itself (that's operational, not tool logic); still available via the legacy `.sh`. |
|
|
18
|
+
| `scripts/walk_agentsam_tables.py` | `agentsam_sdk.data.agentsam_walk` | **Ported, condensed.** Schema/indexes/FKs/row-count/freshness/capability-grouping all present. Not byte-for-byte: duplicate-table detection and some staleness heuristics from the 801-line original are deferred. |
|
|
19
|
+
| `scripts/d1_schema_audit.py` | folded into `agentsam_walk` (schema slice) | **Partially folded.** The capability-grouping + schema dump is covered by `agentsam_walk`. NOT ported: the per-feature markdown chunking into 14 separate `db/agentsam-*.md` files, and the curated `TABLE_META` purpose annotations (760+ lines of hand-written table descriptions) -- that's product documentation content, not audit logic, and belongs in a follow-up pass, not this one. Also note: the legacy script hardcoded a D1 database id as a fallback default (`D1_DATABASE_ID = os.environ.get("D1_DATABASE_ID", "cf87b717-...")`) -- **do not carry that forward**; the new adapter has no such fallback (HARD LAW). |
|
|
20
|
+
|
|
21
|
+
## P1 — agentsam quality / wiring (not started this pass)
|
|
22
|
+
|
|
23
|
+
`audit_agentsam_full.py`, `agentsam_db_deep_audit.py`, `audit_agentsam_tables.py`,
|
|
24
|
+
`audit_agentsam_schema.py`, `audit_agentsam_table_usage.py`,
|
|
25
|
+
`agentsam_cms_d1_table_audit.py`, `agentsam_audit.py` — all deferred. Reason:
|
|
26
|
+
scope boundary for this pass was P0 only, per the handoff doc's file list.
|
|
27
|
+
|
|
28
|
+
## P2 — repository / history / readiness
|
|
29
|
+
|
|
30
|
+
| Legacy script | SDK target | Status |
|
|
31
|
+
|---|---|---|
|
|
32
|
+
| `scripts/repo-size-inventory.py` (main) / `repo_inventory.py` | `agentsam_sdk.repository.inventory` | **Ported** (category + byte sizes + largest files + extension rollups; jq-friendly JSON; stub fields `by_extension` / `by_top_level_dir` retained). |
|
|
33
|
+
| `tools/scan_bloat.py` (IAM shim) | `agentsam_sdk.repository.scan_bloat` | **Ported.** Per-file KB/lines/est. tokens; CLI `agentsam repository scan-bloat`. D1: `agentsam_scripts.slug=scan_bloat` (not `agentsam_commands`). |
|
|
34
|
+
| `repo_cleanup_classify.py`, `repo-cleanup.py` | `agentsam_sdk.repository.cleanup_plan` | **Not started** |
|
|
35
|
+
| `audit_dead_code.py` | `agentsam_sdk.repository.dead_paths` | **Not started** |
|
|
36
|
+
| `audit_hardcoded_identity.py`, `guard-no-hardcoded-identity.sh` | `agentsam_sdk.readiness.boundaries` | **Not started** |
|
|
37
|
+
| `audit_migration_chain.py` | `agentsam_sdk.history.timeline` / `.supersession` | **Not started** |
|
|
38
|
+
| leftover `d1_*_audit.py` | `agentsam_sdk.data.d1` adapters | **Not started** |
|
|
39
|
+
|
|
40
|
+
## Out of scope (per handoff doc, unchanged)
|
|
41
|
+
|
|
42
|
+
`scripts/embed_*`, `scripts/ingest_*` (product Vectorize/pgvector pipelines);
|
|
43
|
+
dashboard/Worker `client_fs`/ExecOS path-propose lane; inventing new D1
|
|
44
|
+
tables for audit output (output stays json+markdown under `--output-dir`,
|
|
45
|
+
R2 later if wanted).
|
|
46
|
+
|
|
47
|
+
## Contract compliance
|
|
48
|
+
|
|
49
|
+
- [x] `runtime/contract.py`: `ToolInput`/`ToolResult` + receipt, every tool call writes one.
|
|
50
|
+
- [x] Read-only default; `ToolInput.write` exists but nothing in this pass sets it True — no silent production writes.
|
|
51
|
+
- [x] D1 access only via `data/d1_adapter.py` (wraps `wrangler d1 execute`); no other module shells out to wrangler or hits the CF API directly.
|
|
52
|
+
- [x] No hardcoded `au_*`/`ws_*`/`tenant_*`/database-id values anywhere — `D1Adapter.from_env` raises rather than guessing.
|
|
53
|
+
- [x] Output: json + markdown from `d1_bloat`, `agentsam_walk`, and `repository.inventory`. No other formats claimed.
|
|
54
|
+
- [x] Tests use fixtures/pure functions only — `python3 -m unittest discover -s tests` needs no live D1 or network.
|
|
55
|
+
- [x] Receipts use `written_at_unixepoch` (unixepoch integer per AGENTS.md); tool-level timestamps otherwise use `time.time()` floats for duration math, not stored as durable rows.
|
|
56
|
+
- [x] `scripts/run-d1-bloat-audit.sh` untouched and still works (legacy path preserved) — see shim note above.
|
|
57
|
+
- [x] Host tooling documented: `docs/tooling.md` (jq + wrangler + env); `scripts/check-host-tooling.sh`.
|
|
58
|
+
|
|
59
|
+
## Known limitations to fix before this is "done" for P0
|
|
60
|
+
|
|
61
|
+
1. `agentsam_walk` is a condensed reimplementation, not byte-identical to the 801-line original — needs a side-by-side diff review against real D1 output before calling P0 fully closed.
|
|
62
|
+
2. No live-D1 smoke test has been run yet in this pass (would need `AGENTSAM_D1_DB_NAME` + Cloudflare creds on the operator machine) — see README "D1 audits" section for the command to run one.
|
|
63
|
+
3. `d1_schema_audit.py`'s curated `TABLE_META` prose (table-by-table purpose descriptions) is real, hand-maintained documentation value that this port does NOT carry over. Flagging so it isn't silently lost.
|