@inneranimalmedia/agentsam-sdk 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/DEVELOPMENT.md +24 -4
  2. package/README.md +2 -0
  3. package/docs/RELEASES.md +6 -0
  4. package/package.json +8 -3
  5. package/protocol/README.md +51 -0
  6. package/protocol/dual-repo-sync.md +35 -0
  7. package/python/README.md +12 -0
  8. package/python/agentsam_sdk/__init__.py +9 -0
  9. package/python/agentsam_sdk/cli.py +212 -0
  10. package/python/agentsam_sdk/data/__init__.py +0 -0
  11. package/python/agentsam_sdk/data/agentsam_walk.py +157 -0
  12. package/python/agentsam_sdk/data/d1_adapter.py +124 -0
  13. package/python/agentsam_sdk/data/d1_bloat.py +265 -0
  14. package/python/agentsam_sdk/repository/__init__.py +0 -0
  15. package/python/agentsam_sdk/repository/__main__.py +3 -0
  16. package/python/agentsam_sdk/repository/inventory.py +351 -0
  17. package/python/agentsam_sdk/repository/scan_bloat.py +173 -0
  18. package/python/agentsam_sdk/runtime/__init__.py +0 -0
  19. package/python/agentsam_sdk/runtime/contract.py +105 -0
  20. package/python/docs/gaps.md +63 -0
  21. package/python/docs/tooling.md +67 -0
  22. package/python/protocol/README.md +51 -0
  23. package/python/protocol/dual-repo-sync.md +35 -0
  24. package/python/pyproject.toml +16 -0
  25. package/python/scripts/check-host-tooling.sh +65 -0
  26. package/python/tests/__init__.py +0 -0
  27. package/python/tests/fixtures/sample_tables.json +17 -0
  28. package/python/tests/fixtures.py +95 -0
  29. package/python/tests/test_agentsam_walk.py +31 -0
  30. package/python/tests/test_contract.py +32 -0
  31. package/python/tests/test_d1_bloat.py +57 -0
  32. package/python/tests/test_repository_inventory.py +53 -0
  33. package/python/tests/test_scan_bloat.py +31 -0
@@ -0,0 +1,67 @@
1
+ # Host tooling for agentsam-sdk
2
+
3
+ Python package is **stdlib-only** (`pip install -e .` adds no third-party deps).
4
+ These **host** tools are expected on the operator/CI machine:
5
+
6
+ | Tool | Required for | Install (macOS) | Check |
7
+ |------|--------------|-----------------|-------|
8
+ | **Python ≥ 3.10** | all modules + unittest | Homebrew / pyenv | `python3 --version` |
9
+ | **jq** | filtering inventory / audit JSON on the CLI | `brew install jq` | `jq --version` |
10
+ | **wrangler** (Cloudflare CLI) | `data.d1_bloat`, `data.agentsam_walk` via `D1Adapter` | `npm i -g wrangler` (or repo local) | `wrangler --version` |
11
+ | **CLOUDFLARE_API_TOKEN** (env) | remote D1 execute | dashboard API token | `test -n "$CLOUDFLARE_API_TOKEN"` |
12
+ | **AGENTSAM_D1_DB_NAME** (env) | D1 tools when `--db` omitted | set in shell / `.env.cloudflare` loader | — |
13
+
14
+ Optional:
15
+
16
+ | Tool | Why |
17
+ |------|-----|
18
+ | `git` | repo-root detection in some callers |
19
+ | `node` / `npm` | IAM `npm run inventory:repo-size*` shims, wrangler install |
20
+
21
+ ## Quick host check
22
+
23
+ From `agentsam-sdk/`:
24
+
25
+ ```bash
26
+ ./scripts/check-host-tooling.sh
27
+ # or:
28
+ ./scripts/check-host-tooling.sh --require-d1 # also needs wrangler + token + db name
29
+ ```
30
+
31
+ ## jq recipes (inventory)
32
+
33
+ ```bash
34
+ # category row for docs
35
+ agentsam repository inventory --repo-root .. --format json \
36
+ | jq '.data.categories[] | select(.id=="docs")'
37
+
38
+ # top 10 largest files
39
+ agentsam repository inventory --repo-root .. --format json \
40
+ | jq '.data.largest_files[:10]'
41
+
42
+ # totals only
43
+ jq '.totals' /tmp/inv/repository-inventory.json
44
+
45
+ # extensions over 100 files
46
+ jq '.by_extension | to_entries | map(select(.value > 100))' \
47
+ /tmp/inv/repository-inventory.json
48
+ ```
49
+
50
+ ## jq recipes (per-file source bloat)
51
+
52
+ ```bash
53
+ agentsam repository scan-bloat --root src --min-kb 10 --top 50 --json-envelope \
54
+ | jq -r '.files[] | "\(.size_kb)KB\t\(.path)"'
55
+ ```
56
+
57
+ ## jq recipes (D1 audits)
58
+
59
+ ```bash
60
+ agentsam data d1-bloat --db "$AGENTSAM_D1_DB_NAME" --quick --format json \
61
+ --output-dir /tmp/d1-bloat
62
+ jq '.data.tables[:10] | .[] | {name, rows}' /tmp/d1-bloat/*.json 2>/dev/null \
63
+ || jq '.' /tmp/d1-bloat/receipt-*.json | head
64
+ ```
65
+
66
+ Exact JSON shapes vary by tool — prefer reading the written `*.json` under
67
+ `--output-dir` over scraping stdout when chaining in CI.
@@ -0,0 +1,51 @@
1
+ # agentsam-sdk protocol (LOCKED)
2
+
3
+ Every Agent Sam **tool / feature SDK surface** is dual-homed. There is no
4
+ “Python audits only in IAM” vs “JS CLI only on npm” split for productized
5
+ tooling.
6
+
7
+ | Home | Path | Role |
8
+ |------|------|------|
9
+ | **Main platform repo** | `inneranimalmedia/agentsam-sdk/` | Source of truth while building; ships with platform deploy |
10
+ | **Published SDK repo** | `github.com/SamPrimeaux/agentsam-sdk` → npm `@inneranimalmedia/agentsam-sdk` | Same contract + modules for external / CLI consumers |
11
+
12
+ ## Rules
13
+
14
+ 1. **Author once, land twice.** New `agentsam_sdk.*` tools (data, repository,
15
+ readiness, history, code-intel, …) land in the monorepo package **and** are
16
+ mirrored into the published `agentsam-sdk` repo in the same change set / PR
17
+ pair. No silent one-sided land.
18
+ 2. **Contract is shared.** `runtime/contract` (`ToolInput` / `ToolResult` /
19
+ receipts, unixepoch, no hardcoded identity) is the protocol. Language may
20
+ differ (Python stdlib vs JS), but CLI verbs and JSON shapes must stay
21
+ aligned — see `protocol/dual-repo-sync.md`.
22
+ 3. **No drift.** Publishing npm without updating the monorepo copy (or the
23
+ reverse) is a protocol violation. Gate: version bump + changelog note that
24
+ lists mirrored paths.
25
+ 4. **Ship is two-sided.**
26
+ - Platform: Mac `npm run deploy:full` / `deploy:fast` (or GCP `ship:remote`)
27
+ when the monorepo copy or Worker wiring changed.
28
+ - SDK: **manual** npm publish / version bump on `agentsam-sdk` after the
29
+ mirror lands (operator-owned — do not assume CI auto-publishes).
30
+ 5. **Inspiration ≠ copy-paste secrets.** In-app surfaces (e.g. `src/core/code-indexer.js`,
31
+ AST-RAG Phase 1/2) are the reference implementations to port into portable
32
+ SDK modules — strip platform-only bindings, keep adapters for D1 / git /
33
+ Hyperdrive.
34
+
35
+ ## Layout (both homes)
36
+
37
+ ```
38
+ agentsam-sdk/
39
+ protocol/ ← this law
40
+ python/agentsam_sdk ← Python portable tools (stdlib-first audits / inventory)
41
+ src/ ← JS CLI / scaffold (published npm entry)
42
+ packages/ ← optional workspace packages (not always in npm tarball)
43
+ agentsam-shell-kit/ ← @inneranimalmedia/agentsam-shell-kit (private until ready)
44
+ docs/gaps.md ← port status vs IAM scripts + in-app indexers
45
+ ```
46
+
47
+ **npm identity (LOCKED):** root publishable package is always
48
+ `@inneranimalmedia/agentsam-sdk`. Do not overwrite root `package.json` with a
49
+ workspace kit name. Fold UI kits under `packages/*`.
50
+
51
+ Exact folder names may evolve; the dual-home + root-identity rules do not.
@@ -0,0 +1,35 @@
1
+ # Dual-repo sync checklist
2
+
3
+ Use this every time you add or change an `agentsam_sdk` tool/feature.
4
+
5
+ ## Before coding
6
+
7
+ - [ ] Name the module (`agentsam_sdk.<domain>.<tool>`) and CLI verb
8
+ - [ ] Confirm it belongs in SDK (portable) vs platform-only Worker hot path
9
+ - [ ] Note in-app inspiration path (if any), e.g. `src/core/code-indexer.js`
10
+
11
+ ## Land
12
+
13
+ - [ ] Implement + tests in **`inneranimalmedia/agentsam-sdk/`**
14
+ - [ ] Mirror the same module/CLI/docs into **`agentsam-sdk`** (npm repo)
15
+ - [ ] Update `docs/gaps.md` (or equivalent) in **both** trees
16
+ - [ ] Update `protocol/` only in monorepo if law changed; copy README blurb to npm
17
+
18
+ ## Publish / deploy (operator)
19
+
20
+ - [ ] Platform: commit/push IAM → deploy by host (`deploy:full` / `ship:remote`)
21
+ - [ ] SDK: bump `package.json` version in npm repo → `npm publish` (manual)
22
+ - [ ] Record versions next to each other (PR description or receipt):
23
+ `iam@<sha>` ↔ `@inneranimalmedia/agentsam-sdk@<semver>`
24
+
25
+ ## Drift signals (fail the PR)
26
+
27
+ - Module exists only in one repo
28
+ - CLI flag / JSON field renamed on one side only
29
+ - README claims “this is not the npm package” / “audits don’t belong here”
30
+ - npm publish without a same-day IAM mirror commit (or documented lag ticket)
31
+
32
+ ## jq / host tools
33
+
34
+ Host tooling (`jq`, `wrangler`, Python ≥3.10) is documented in
35
+ `docs/tooling.md`. Keep that file mirrored when recipes change.
@@ -0,0 +1,16 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "agentsam-sdk"
7
+ version = "0.1.0"
8
+ description = "Stdlib-only D1/repository audit toolkit for inneranimalmedia (Agent Sam)."
9
+ requires-python = ">=3.10"
10
+ dependencies = []
11
+
12
+ [project.scripts]
13
+ agentsam = "agentsam_sdk.cli:main"
14
+
15
+ [tool.setuptools.packages.find]
16
+ include = ["agentsam_sdk*"]
@@ -0,0 +1,65 @@
1
+ #!/usr/bin/env bash
2
+ # Verify host tools for agentsam-sdk (stdlib Python + jq + optional wrangler/D1).
3
+ set -euo pipefail
4
+
5
+ REQUIRE_D1=0
6
+ for arg in "$@"; do
7
+ case "$arg" in
8
+ --require-d1) REQUIRE_D1=1 ;;
9
+ -h|--help)
10
+ echo "Usage: $0 [--require-d1]"
11
+ exit 0
12
+ ;;
13
+ esac
14
+ done
15
+
16
+ fail=0
17
+ ok() { printf ' OK %s\n' "$1"; }
18
+ bad() { printf ' MISSING %s\n' "$1"; fail=1; }
19
+
20
+ echo "agentsam-sdk host tooling"
21
+
22
+ if command -v python3 >/dev/null 2>&1; then
23
+ pyv=$(python3 -c 'import sys; print("%d.%d"%sys.version_info[:2])')
24
+ # shellcheck disable=SC2072
25
+ if python3 -c 'import sys; raise SystemExit(0 if sys.version_info >= (3,10) else 1)'; then
26
+ ok "python3 ($pyv)"
27
+ else
28
+ bad "python3 >= 3.10 (found $pyv)"
29
+ fi
30
+ else
31
+ bad "python3"
32
+ fi
33
+
34
+ if command -v jq >/dev/null 2>&1; then
35
+ ok "jq ($(jq --version 2>&1))"
36
+ else
37
+ bad "jq (brew install jq) — needed for JSON filters in docs/tooling.md"
38
+ fi
39
+
40
+ if [[ "$REQUIRE_D1" -eq 1 ]]; then
41
+ if command -v wrangler >/dev/null 2>&1 || command -v npx >/dev/null 2>&1; then
42
+ ok "wrangler/npx available"
43
+ else
44
+ bad "wrangler or npx"
45
+ fi
46
+ if [[ -n "${CLOUDFLARE_API_TOKEN:-}" ]]; then
47
+ ok "CLOUDFLARE_API_TOKEN set"
48
+ else
49
+ bad "CLOUDFLARE_API_TOKEN"
50
+ fi
51
+ if [[ -n "${AGENTSAM_D1_DB_NAME:-}" ]]; then
52
+ ok "AGENTSAM_D1_DB_NAME=${AGENTSAM_D1_DB_NAME}"
53
+ else
54
+ bad "AGENTSAM_D1_DB_NAME (or pass --db on each command)"
55
+ fi
56
+ else
57
+ echo " (skip D1 env — pass --require-d1 to enforce)"
58
+ fi
59
+
60
+ if [[ "$fail" -ne 0 ]]; then
61
+ echo "FAIL: install missing tools (see docs/tooling.md)"
62
+ exit 1
63
+ fi
64
+ echo "PASS"
65
+ exit 0
File without changes
@@ -0,0 +1,17 @@
1
+ {
2
+ "agentsam_tool_call_log": {
3
+ "columns": [["id","INTEGER"],["input_json","TEXT"],["output_json","TEXT"],["created_at","TEXT"]],
4
+ "row_count": 250000,
5
+ "text_lengths": {"input_json": 6291456, "output_json": 12582912}
6
+ },
7
+ "agentsam_memory": {
8
+ "columns": [["id","INTEGER"],["key","TEXT"],["value","TEXT"],["updated_at","TEXT"]],
9
+ "row_count": 40,
10
+ "text_lengths": {"value": 12000}
11
+ },
12
+ "cms_pages": {
13
+ "columns": [["id","INTEGER"],["slug","TEXT"],["title","TEXT"]],
14
+ "row_count": 0,
15
+ "text_lengths": {}
16
+ }
17
+ }
@@ -0,0 +1,95 @@
1
+ """Stub D1 shape shared by data/* tests -- no live D1 required.
2
+
3
+ Mirrors a small slice of the real schema (a couple agentsam_* tables plus
4
+ one bloated cms_ table) so d1_bloat / agentsam_walk logic can be exercised
5
+ without wrangler or a network call.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ FAKE_TABLES = [
10
+ "agentsam_tool_call_log",
11
+ "agentsam_workflow_runs",
12
+ "cms_pages",
13
+ ]
14
+
15
+ FAKE_COLUMNS: dict[str, list[tuple[str, str]]] = {
16
+ "agentsam_tool_call_log": [
17
+ ("id", "TEXT"),
18
+ ("tool_name", "TEXT"),
19
+ ("input_json", "TEXT"),
20
+ ("output_json", "TEXT"),
21
+ ("created_at", "INTEGER"),
22
+ ],
23
+ "agentsam_workflow_runs": [
24
+ ("id", "TEXT"),
25
+ ("step_results_json", "TEXT"),
26
+ ("status", "TEXT"),
27
+ ("updated_at", "INTEGER"),
28
+ ],
29
+ "cms_pages": [
30
+ ("id", "TEXT"),
31
+ ("body", "TEXT"),
32
+ ("slug", "TEXT"),
33
+ ],
34
+ }
35
+
36
+ FAKE_ROW_COUNTS = {
37
+ "agentsam_tool_call_log": 150_000,
38
+ "agentsam_workflow_runs": 40,
39
+ "cms_pages": 12,
40
+ }
41
+
42
+
43
+ class FakeD1Adapter:
44
+ """Drop-in stand-in for agentsam_sdk.data.d1_adapter.D1Adapter.
45
+
46
+ Constructed the same way tests need it (via a classmethod matching
47
+ from_env's call signature) and implements only the methods d1_bloat.py /
48
+ agentsam_walk.py actually call.
49
+ """
50
+
51
+ def __init__(self, db_name: str = "fake-db"):
52
+ self.db_name = db_name
53
+
54
+ @classmethod
55
+ def from_env(cls, db_name=None, wrangler_config=None, repo_root=None):
56
+ return cls(db_name=db_name or "fake-db")
57
+
58
+ def list_tables(self, like=None):
59
+ if not like:
60
+ return list(FAKE_TABLES)
61
+ prefix = like.rstrip("%")
62
+ return [t for t in FAKE_TABLES if t.startswith(prefix)]
63
+
64
+ def database_size(self):
65
+ return "12.3 MB"
66
+
67
+ def table_columns(self, table):
68
+ return list(FAKE_COLUMNS.get(table, []))
69
+
70
+ def table_indexes(self, table):
71
+ return []
72
+
73
+ def foreign_keys(self, table):
74
+ return []
75
+
76
+ def row_count(self, table):
77
+ return FAKE_ROW_COUNTS.get(table, 0)
78
+
79
+ def query(self, sql):
80
+ # Only d1_bloat's SUM(LENGTH(...)) aggregate query shape is needed.
81
+ rc = FAKE_ROW_COUNTS.get(self._table_from_sql(sql), 0)
82
+ row = {"rc": rc}
83
+ for col in FAKE_COLUMNS.get(self._table_from_sql(sql), []):
84
+ name = col[0]
85
+ if name in sql:
86
+ row[name] = 200 * rc
87
+ row[f"m_{name}"] = 200
88
+ return [row]
89
+
90
+ @staticmethod
91
+ def _table_from_sql(sql: str) -> str:
92
+ for t in FAKE_TABLES:
93
+ if f'"{t}"' in sql:
94
+ return t
95
+ return ""
@@ -0,0 +1,31 @@
1
+ """Unit tests for data.agentsam_walk -- pure-function logic, no live D1."""
2
+ import unittest
3
+
4
+ from agentsam_sdk.data.agentsam_walk import _capability_for, TableWalk, _render_markdown
5
+
6
+
7
+ class TestCapabilityHeuristic(unittest.TestCase):
8
+ def test_tool_chain_maps_to_tools_commands_mcp(self):
9
+ self.assertEqual(_capability_for("agentsam_tool_chain"), "tools_commands_mcp")
10
+
11
+ def test_workflow_runs_maps_to_workflow_dag(self):
12
+ self.assertEqual(_capability_for("agentsam_workflow_runs"), "workflow_dag")
13
+
14
+ def test_unknown_table_is_uncategorized(self):
15
+ self.assertEqual(_capability_for("zzz_totally_unmatched_xyz"), "uncategorized")
16
+
17
+
18
+ class TestRenderMarkdown(unittest.TestCase):
19
+ def test_groups_by_capability(self):
20
+ walks = [
21
+ TableWalk(name="agentsam_tool_chain", capability="tools_commands_mcp", row_count=10),
22
+ TableWalk(name="agentsam_memory", capability="memory_rag", row_count=5),
23
+ ]
24
+ md = _render_markdown(walks, "agentsam_")
25
+ self.assertIn("tools_commands_mcp", md)
26
+ self.assertIn("memory_rag", md)
27
+ self.assertIn("agentsam_tool_chain", md)
28
+
29
+
30
+ if __name__ == "__main__":
31
+ unittest.main()
@@ -0,0 +1,32 @@
1
+ """Unit tests for runtime.contract -- receipts, no I/O beyond a tmp dir."""
2
+ import tempfile
3
+ import unittest
4
+ from pathlib import Path
5
+
6
+ from agentsam_sdk.runtime.contract import ToolInput, ToolResult, write_receipt, start_timer
7
+
8
+
9
+ class TestReceipts(unittest.TestCase):
10
+ def test_write_receipt_creates_file(self):
11
+ with tempfile.TemporaryDirectory() as tmp:
12
+ ti = ToolInput(output_dir=tmp)
13
+ started = start_timer()
14
+ result = ToolResult(
15
+ ok=True, tool="data.d1_bloat", mode="quick", request_id=ti.request_id,
16
+ started_at=started, finished_at=start_timer(), summary="test run",
17
+ )
18
+ path = write_receipt(result, ti.output_path())
19
+ self.assertIsNotNone(path)
20
+ self.assertTrue(Path(path).exists())
21
+
22
+ def test_write_receipt_noop_without_output_dir(self):
23
+ ti = ToolInput()
24
+ result = ToolResult(
25
+ ok=True, tool="data.d1_bloat", mode="quick", request_id=ti.request_id,
26
+ started_at=0.0, finished_at=1.0, summary="test",
27
+ )
28
+ self.assertIsNone(write_receipt(result, ti.output_path()))
29
+
30
+
31
+ if __name__ == "__main__":
32
+ unittest.main()
@@ -0,0 +1,57 @@
1
+ """Unit tests for data.d1_bloat -- pure-function parsing, no live D1/network."""
2
+ import unittest
3
+
4
+ from agentsam_sdk.data.d1_bloat import (
5
+ ColStat, TableStat, _pick_bloat_columns, _flag_suspicious, _render_markdown, _fmt_bytes,
6
+ )
7
+
8
+
9
+ class TestBloatColumnPicking(unittest.TestCase):
10
+ def test_picks_json_and_body_columns(self):
11
+ cols = [("id", "INTEGER"), ("input_json", "TEXT"), ("output_json", "TEXT"), ("created_at", "TEXT")]
12
+ picked = _pick_bloat_columns(cols)
13
+ self.assertIn("input_json", picked)
14
+ self.assertIn("output_json", picked)
15
+ self.assertNotIn("id", picked)
16
+ self.assertNotIn("created_at", picked)
17
+
18
+ def test_skips_id_and_fk_like_columns(self):
19
+ cols = [("tenant_id", "TEXT"), ("workspace_id", "TEXT"), ("metadata", "TEXT")]
20
+ picked = _pick_bloat_columns(cols)
21
+ self.assertNotIn("tenant_id", picked)
22
+ self.assertNotIn("workspace_id", picked)
23
+ self.assertIn("metadata", picked)
24
+
25
+
26
+ class TestFlagging(unittest.TestCase):
27
+ def test_flags_large_table_high_severity(self):
28
+ big = TableStat(name="agentsam_tool_call_log", row_count=250_000, text_bytes=18_874_368,
29
+ est_bytes=18_874_368, columns=[ColStat(name="output_json", bytes=12_582_912)])
30
+ small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
31
+ flags = _flag_suspicious([big, small])
32
+ names = {f["table"] for f in flags}
33
+ self.assertIn("agentsam_tool_call_log", names)
34
+ big_flag = next(f for f in flags if f["table"] == "agentsam_tool_call_log")
35
+ self.assertEqual(big_flag["severity"], "high")
36
+
37
+ def test_empty_table_not_flagged(self):
38
+ small = TableStat(name="cms_pages", row_count=0, text_bytes=0, est_bytes=0)
39
+ flags = _flag_suspicious([small])
40
+ self.assertEqual(flags, [])
41
+
42
+
43
+ class TestFormatting(unittest.TestCase):
44
+ def test_fmt_bytes_scales(self):
45
+ self.assertEqual(_fmt_bytes(500), "500 B")
46
+ self.assertEqual(_fmt_bytes(2048), "2.0 KB")
47
+ self.assertEqual(_fmt_bytes(5 * 1024 * 1024), "5.00 MB")
48
+
49
+ def test_render_markdown_includes_table_names(self):
50
+ stats = [TableStat(name="agentsam_memory", row_count=40, text_bytes=12000, est_bytes=12000)]
51
+ md = _render_markdown(stats, "1.2 MB", "quick", 10, 1)
52
+ self.assertIn("agentsam_memory", md)
53
+ self.assertIn("D1 bloat audit", md)
54
+
55
+
56
+ if __name__ == "__main__":
57
+ unittest.main()
@@ -0,0 +1,53 @@
1
+ import unittest
2
+ from tempfile import TemporaryDirectory
3
+ from pathlib import Path
4
+
5
+ from agentsam_sdk.repository import inventory
6
+ from agentsam_sdk.runtime.contract import ToolInput
7
+
8
+
9
+ class TestRepositoryInventory(unittest.TestCase):
10
+ def test_counts_files_by_category_size_and_extension(self):
11
+ with TemporaryDirectory() as repo, TemporaryDirectory() as out:
12
+ root = Path(repo)
13
+ (root / "src").mkdir()
14
+ (root / "src" / "a.py").write_text("x" * 100)
15
+ (root / "src" / "b.py").write_text("y" * 50)
16
+ (root / "docs").mkdir()
17
+ (root / "docs" / "c.md").write_text("z" * 20)
18
+ (root / "node_modules").mkdir()
19
+ (root / "node_modules" / "skip.js").write_text("skip me" * 1000)
20
+
21
+ ti = ToolInput(
22
+ params={"repo_root": str(root), "top": 5, "by_ext": True},
23
+ output_dir=out,
24
+ )
25
+ result = inventory.run(ti)
26
+
27
+ self.assertTrue(result.ok)
28
+ self.assertEqual(result.data["file_total"], 3) # node_modules excluded
29
+ self.assertEqual(result.data["by_extension"][".py"], 2)
30
+ self.assertEqual(result.data["by_top_level_dir"]["src"], 2)
31
+ ids = {c["id"] for c in result.data["categories"]}
32
+ self.assertIn("worker_src", ids)
33
+ self.assertIn("docs", ids)
34
+ self.assertGreater(result.data["totals"]["bytes"], 0)
35
+ self.assertTrue((Path(out) / "repository-inventory.json").exists())
36
+ self.assertTrue((Path(out) / "repository-inventory.md").exists())
37
+ # largest list prefers bigger src file
38
+ self.assertEqual(result.data["largest_files"][0]["path"], "src/a.py")
39
+
40
+ def test_missing_repo_root_is_reported_not_raised(self):
41
+ ti = ToolInput(params={"repo_root": "/definitely/does/not/exist/xyz"})
42
+ result = inventory.run(ti)
43
+ self.assertFalse(result.ok)
44
+ self.assertEqual(result.error, "repo_root_not_found")
45
+
46
+ def test_categorize_helpers(self):
47
+ self.assertEqual(inventory.categorize(Path("dashboard/App.tsx")), "dashboard")
48
+ self.assertEqual(inventory.categorize(Path("README.md")), "root_misc")
49
+ self.assertIn("KiB", inventory.human_bytes(2048))
50
+
51
+
52
+ if __name__ == "__main__":
53
+ unittest.main()
@@ -0,0 +1,31 @@
1
+ """Unit tests for repository.scan_bloat — no network."""
2
+ from __future__ import annotations
3
+
4
+ from pathlib import Path
5
+
6
+ from agentsam_sdk.repository import scan_bloat
7
+ from agentsam_sdk.runtime.contract import ToolInput
8
+
9
+
10
+ def test_scan_ranks_by_size(tmp_path: Path) -> None:
11
+ (tmp_path / "small.js").write_text("x=1\n", encoding="utf-8")
12
+ (tmp_path / "big.js").write_text("x=1\n" * 500, encoding="utf-8")
13
+ (tmp_path / "skip.py").write_text("print(1)\n" * 200, encoding="utf-8")
14
+ rows = scan_bloat.scan(tmp_path)
15
+ assert len(rows) == 2
16
+ assert rows[0]["path"] == "big.js"
17
+ assert rows[0]["size_bytes"] > rows[1]["size_bytes"]
18
+
19
+
20
+ def test_run_json_envelope(tmp_path: Path) -> None:
21
+ (tmp_path / "a.ts").write_text("// " + ("n" * 2000) + "\n", encoding="utf-8")
22
+ result = scan_bloat.run(
23
+ ToolInput(
24
+ mode="read-only",
25
+ params={"root": str(tmp_path), "top": 5, "min_kb": 0},
26
+ )
27
+ )
28
+ assert result.ok
29
+ assert result.tool == "repository.scan_bloat"
30
+ assert result.data["file_count"] == 1
31
+ assert result.data["files"][0]["path"] == "a.ts"