@theglitchking/babel-fish 2.0.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/project-map/generate.py +258 -22
- package/.claude/project-map/grader.py +77 -1
- package/.claude/project-map/mine-sessions.py +141 -23
- package/.claude/project-map/test_generate.py +267 -0
- package/.claude/project-map/test_grader.py +164 -0
- package/.claude/project-map/test_mine_sessions.py +206 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/.claude-plugin/plugin.json +1 -1
- package/.githooks/pre-commit +2 -1
- package/CHANGELOG.md +248 -0
- package/README.md +9 -2
- package/checksums.json +2 -2
- package/hooks/session-start.js +34 -1
- package/package.json +13 -4
- package/.claude/project-map/PROJECT_MAP.md +0 -61
- package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
- package/.claude/project-map/checksums.json +0 -6
- package/.claude/project-map/learned-vocabulary.json +0 -1
- package/.claude/project-map/reports/install-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-score.json +0 -7
- package/.claude/project-map/sections/01-vocabulary.md +0 -8
- package/.claude/project-map/sections/02-service-topology.md +0 -6
- package/.claude/project-map/sections/03-environment.md +0 -6
- package/.claude/project-map/sections/04-api-routes.md +0 -6
- package/.claude/project-map/sections/05-data-models.md +0 -4
- package/.claude/project-map/sections/06-schemas.md +0 -4
- package/.claude/project-map/sections/07-services.md +0 -6
- package/.claude/project-map/sections/08-background-jobs.md +0 -5
- package/.claude/project-map/sections/09-frontend-features.md +0 -4
- package/.claude/project-map/sections/10-tools-commands.md +0 -8
- package/.claude/project-map/sections/11-migrations.md +0 -4
- package/.claude/project-map/sections/12-import-chains.md +0 -7
- package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
- package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
- package/.claude/project-map/sections/15-auth-config.md +0 -6
- package/.claude/project-map/sections/16-infra-profile.md +0 -13
- package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
- package/.claude/project-map/sections/18-dead-code.md +0 -9
- package/.claude/project-map/sections/19-doc-pointers.md +0 -5
- package/.claude/project-map/stack.json +0 -12
- package/.claude/rules/operational-runbook.md +0 -40
- package/.claude/rules/project-vocabulary.md +0 -25
- package/.claude/settings.json +0 -6
- package/.claude/settings.local.json +0 -6
- package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Regression tests for grader.py — run: python3 .claude/project-map/test_grader.py
|
|
3
|
+
|
|
4
|
+
Guards the fixes from issue #9. The headline defect there was that the grader
|
|
5
|
+
scored a completely empty map and a populated one identically (97.0% both), so
|
|
6
|
+
these assert the SIGNAL that separates them, not the score — deliberately, since
|
|
7
|
+
folding usefulness into the weighted score would fail legitimately sparse repos.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import importlib.util
|
|
12
|
+
import shutil
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
HERE = Path(__file__).parent
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load_grader(map_dir: Path):
|
|
21
|
+
spec = importlib.util.spec_from_file_location("grader_under_test", HERE / "grader.py")
|
|
22
|
+
g = importlib.util.module_from_spec(spec)
|
|
23
|
+
argv, sys.argv = sys.argv, ["grader.py"]
|
|
24
|
+
try:
|
|
25
|
+
spec.loader.exec_module(g)
|
|
26
|
+
finally:
|
|
27
|
+
sys.argv = argv
|
|
28
|
+
g.MAP_DIR = map_dir
|
|
29
|
+
g.SECTIONS_DIR = map_dir / "sections"
|
|
30
|
+
g.PROJECT_ROOT = map_dir
|
|
31
|
+
return g
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def write_map(map_dir: Path, *, vocab: int, scanned: int, extracted: int = 0) -> None:
|
|
35
|
+
(map_dir / "sections").mkdir(parents=True, exist_ok=True)
|
|
36
|
+
(map_dir / "PROJECT_MAP.md").write_text(f"""## Stats
|
|
37
|
+
|
|
38
|
+
| Metric | Count |
|
|
39
|
+
|--------|-------|
|
|
40
|
+
| API Routes | {extracted} |
|
|
41
|
+
| Data Models | 0 |
|
|
42
|
+
| Schemas/DTOs | 0 |
|
|
43
|
+
| Frontend Features | 0 |
|
|
44
|
+
| Vocabulary Entries | {vocab} |
|
|
45
|
+
|
|
46
|
+
### Source files scanned
|
|
47
|
+
|
|
48
|
+
| Language | Files |
|
|
49
|
+
|----------|-------|
|
|
50
|
+
| python | {scanned} |
|
|
51
|
+
|
|
52
|
+
## Section Index
|
|
53
|
+
""")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_empty_vocabulary_warns(tmp):
|
|
57
|
+
write_map(tmp, vocab=0, scanned=0)
|
|
58
|
+
w = load_grader(tmp).usefulness_warnings()
|
|
59
|
+
assert any("Vocabulary is empty" in x for x in w), w
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_populated_vocabulary_does_not_warn(tmp):
|
|
63
|
+
write_map(tmp, vocab=10, scanned=3)
|
|
64
|
+
w = load_grader(tmp).usefulness_warnings()
|
|
65
|
+
assert not any("Vocabulary is empty" in x for x in w), w
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_many_files_no_extraction_warns(tmp):
|
|
69
|
+
write_map(tmp, vocab=7, scanned=47, extracted=0)
|
|
70
|
+
w = load_grader(tmp).usefulness_warnings()
|
|
71
|
+
assert any("Scanned 47" in x for x in w), w
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_few_files_no_extraction_is_quiet(tmp):
|
|
75
|
+
"""This repo: 3 CLI shims yielding no routes is correct, not a defect."""
|
|
76
|
+
write_map(tmp, vocab=10, scanned=3, extracted=0)
|
|
77
|
+
w = load_grader(tmp).usefulness_warnings()
|
|
78
|
+
assert not any("Scanned" in x for x in w), w
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_warnings_do_not_touch_the_score(tmp):
|
|
82
|
+
"""Usefulness is reported, never scored — scoring it would fail sparse repos."""
|
|
83
|
+
g = load_grader(tmp)
|
|
84
|
+
write_map(tmp, vocab=0, scanned=99)
|
|
85
|
+
for name in dir(g):
|
|
86
|
+
if name.startswith("grade_"):
|
|
87
|
+
r = getattr(g, name)()
|
|
88
|
+
assert 0.0 <= r.raw_score <= 100.0
|
|
89
|
+
assert g.usefulness_warnings(), "expected warnings for an empty map"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_populated_chains_not_misread_as_empty(tmp):
|
|
93
|
+
"""`or '_No' in content` matched _Note / _Nothing anywhere in the file and
|
|
94
|
+
scored a populated section as an acceptable empty one at a flat 80%."""
|
|
95
|
+
(tmp / "sections").mkdir(parents=True, exist_ok=True)
|
|
96
|
+
(tmp / "sections" / "12-import-chains.md").write_text(
|
|
97
|
+
"# Section 12\n\n_Note: partial._\n\n```\nsrc/a.py -> src/b.py\n```\n")
|
|
98
|
+
g = load_grader(tmp)
|
|
99
|
+
r = g.grade_import_chains()
|
|
100
|
+
assert "No chains traced" not in r.details, (
|
|
101
|
+
f"a populated section was scored as empty: {r.details}")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_stub_chains_still_recognised(tmp):
|
|
105
|
+
(tmp / "sections").mkdir(parents=True, exist_ok=True)
|
|
106
|
+
(tmp / "sections" / "12-import-chains.md").write_text(
|
|
107
|
+
"# Section 12\n\n_No import chains traced._\n")
|
|
108
|
+
r = load_grader(tmp).grade_import_chains()
|
|
109
|
+
assert "No chains traced" in r.details, r.details
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def test_empty_vocab_section_reaches_greenfield_branch(tmp):
|
|
113
|
+
"""generate.py used to emit a placeholder TABLE ROW, which parsed as a valid
|
|
114
|
+
entry with a neutral location — scoring an empty vocabulary 100% and leaving
|
|
115
|
+
this branch permanently unreachable."""
|
|
116
|
+
(tmp / "sections").mkdir(parents=True, exist_ok=True)
|
|
117
|
+
(tmp / "sections" / "01-vocabulary.md").write_text(
|
|
118
|
+
"# Section 01\n\n| Alias | Type | Location | Notes |\n|---|---|---|---|\n\n"
|
|
119
|
+
"_No vocabulary generated yet — add source code to populate._\n")
|
|
120
|
+
r = load_grader(tmp).grade_vocabulary_accuracy()
|
|
121
|
+
assert r.raw_score == 85.0, f"greenfield branch not reached: {r.raw_score} {r.details}"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_generate_emits_prose_not_a_row_for_empty_vocab():
|
|
125
|
+
spec = importlib.util.spec_from_file_location("gen", HERE / "generate.py")
|
|
126
|
+
gen = importlib.util.module_from_spec(spec)
|
|
127
|
+
argv, sys.argv = sys.argv, ["generate.py"]
|
|
128
|
+
try:
|
|
129
|
+
spec.loader.exec_module(gen)
|
|
130
|
+
finally:
|
|
131
|
+
sys.argv = argv
|
|
132
|
+
out = gen.build_vocabulary_section([])
|
|
133
|
+
rows = [l for l in out.splitlines()
|
|
134
|
+
if l.strip().startswith("|") and "Alias" not in l and set(l.strip()) - set("|-: ")]
|
|
135
|
+
assert not rows, f"empty vocabulary must not emit a table row, got {rows}"
|
|
136
|
+
assert "No vocabulary generated yet" in out
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def main() -> int:
|
|
140
|
+
import inspect
|
|
141
|
+
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
|
142
|
+
failed = []
|
|
143
|
+
base = Path(tempfile.mkdtemp(prefix="grader-test-"))
|
|
144
|
+
try:
|
|
145
|
+
for fn in tests:
|
|
146
|
+
d = base / fn.__name__
|
|
147
|
+
d.mkdir(parents=True)
|
|
148
|
+
try:
|
|
149
|
+
if "tmp" in inspect.signature(fn).parameters:
|
|
150
|
+
fn(d)
|
|
151
|
+
else:
|
|
152
|
+
fn()
|
|
153
|
+
print(f" ok {fn.__name__}")
|
|
154
|
+
except Exception as e:
|
|
155
|
+
failed.append(fn.__name__)
|
|
156
|
+
print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
|
|
157
|
+
finally:
|
|
158
|
+
shutil.rmtree(base, ignore_errors=True)
|
|
159
|
+
print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
|
|
160
|
+
return 1 if failed else 0
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
if __name__ == "__main__":
|
|
164
|
+
sys.exit(main())
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Regression tests for mine-sessions.py — run: python3 .claude/project-map/test_mine_sessions.py
|
|
3
|
+
|
|
4
|
+
Same shape as test_generate.py: stdlib assert + __main__, no framework.
|
|
5
|
+
|
|
6
|
+
Every test here corresponds to a defect that shipped and produced NO error —
|
|
7
|
+
the miner exited 0 and reported "Extracted 0 alias(es)" for its entire
|
|
8
|
+
existence. Silent-zero is the failure mode these guard against.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import importlib.util
|
|
13
|
+
import json
|
|
14
|
+
import shutil
|
|
15
|
+
import sys
|
|
16
|
+
import tempfile
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
HERE = Path(__file__).parent
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load_miner(project_root: Path, cursor_dir: Path):
|
|
24
|
+
spec = importlib.util.spec_from_file_location("miner_under_test", HERE / "mine-sessions.py")
|
|
25
|
+
mod = importlib.util.module_from_spec(spec)
|
|
26
|
+
argv, sys.argv = sys.argv, ["mine-sessions.py"]
|
|
27
|
+
try:
|
|
28
|
+
spec.loader.exec_module(mod)
|
|
29
|
+
finally:
|
|
30
|
+
sys.argv = argv
|
|
31
|
+
mod.PROJECT_ROOT = project_root
|
|
32
|
+
mod.MINE_CURSOR = cursor_dir / ".mine-cursor.json"
|
|
33
|
+
return mod
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def write_transcript(path: Path, entries: list[dict]) -> None:
|
|
37
|
+
path.write_text("\n".join(json.dumps(e) for e in entries) + "\n")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def user_msg(text: str, **extra) -> dict:
|
|
41
|
+
"""A genuine user turn, in the real nested shape Claude Code writes."""
|
|
42
|
+
return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
43
|
+
"message": {"role": "user", "content": [{"type": "text", "text": text}]}, **extra}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def tool_msg(name: str, inp: dict) -> dict:
|
|
47
|
+
return {"type": "assistant", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
48
|
+
"message": {"role": "assistant",
|
|
49
|
+
"content": [{"type": "tool_use", "name": name, "input": inp}]}}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def tool_result() -> dict:
|
|
53
|
+
"""Tool results come back with role=user — the collision that broke pairing."""
|
|
54
|
+
return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
55
|
+
"message": {"role": "user",
|
|
56
|
+
"content": [{"type": "tool_result", "content": "ok"}]}}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ── Defect 1: JSONL nesting ──────────────────────────────────────────────────
|
|
60
|
+
|
|
61
|
+
def test_reads_nested_message_content(m, root):
|
|
62
|
+
msg = user_msg("open the settings page")
|
|
63
|
+
assert m.msg_role(msg) == "user", "role must be read from message.role"
|
|
64
|
+
content = m.msg_content(msg)
|
|
65
|
+
assert isinstance(content, list) and content[0]["text"] == "open the settings page"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_still_reads_flat_content(m, root):
|
|
69
|
+
"""Older/third-party transcripts must not regress."""
|
|
70
|
+
flat = {"role": "user", "content": [{"type": "text", "text": "hello there"}]}
|
|
71
|
+
assert m.msg_role(flat) == "user"
|
|
72
|
+
assert m.msg_content(flat)[0]["text"] == "hello there"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_extracts_tool_paths_from_nested(m, root):
|
|
76
|
+
paths = m.extract_file_paths_from_tool_calls(
|
|
77
|
+
[tool_msg("Read", {"file_path": str(root / "src/app.py")})])
|
|
78
|
+
assert paths == ["src/app.py"], paths
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ── Defect 2: regex quantifiers ──────────────────────────────────────────────
|
|
82
|
+
|
|
83
|
+
def test_the_x_page_pattern_matches(m, root):
|
|
84
|
+
"""'{2,40?}' compiled fine and matched nothing — no error, just silence."""
|
|
85
|
+
got = m.extract_user_phrases("please look at the settings page")
|
|
86
|
+
assert "settings" in got, got
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_x_feature_pattern_matches(m, root):
|
|
90
|
+
got = m.extract_user_phrases("the billing workflow is broken")
|
|
91
|
+
assert any("billing" in g for g in got), got
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ── Defect 3: tool_result / user-turn collision ──────────────────────────────
|
|
95
|
+
|
|
96
|
+
def test_tool_result_is_not_a_user_turn(m, root):
|
|
97
|
+
assert m.is_user_turn(user_msg("the deals page")) is True
|
|
98
|
+
assert m.is_user_turn(tool_result()) is False, (
|
|
99
|
+
"tool results carry role=user; treating them as user turns closed the "
|
|
100
|
+
"pairing window on the assistant's own output"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_meta_and_sidechain_are_not_user_turns(m, root):
|
|
105
|
+
assert m.is_user_turn(user_msg("skill body text", isMeta=True)) is False
|
|
106
|
+
assert m.is_user_turn(user_msg("subagent text", isSidechain=True)) is False
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ── Defect 4: Bash-mediated file access ──────────────────────────────────────
|
|
110
|
+
|
|
111
|
+
def test_bash_paths_extracted_when_file_exists(m, root):
|
|
112
|
+
(root / "src").mkdir(parents=True, exist_ok=True)
|
|
113
|
+
(root / "src/app.py").write_text("# app\n")
|
|
114
|
+
got = m.extract_paths_from_bash("sed -n '1,40p' src/app.py", root)
|
|
115
|
+
assert got == ["src/app.py"], got
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_bash_paths_ignore_nonexistent(m, root):
|
|
119
|
+
got = m.extract_paths_from_bash("cat totally/made/up.py && ls -la", root)
|
|
120
|
+
assert got == [], f"only real files may become aliases, got {got}"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# ── Defect 6: session discovery ──────────────────────────────────────────────
|
|
124
|
+
|
|
125
|
+
def test_slug_keeps_leading_separator(m, root, monkey_home):
|
|
126
|
+
""".lstrip('-') meant the exact match never hit, so every lookup fell
|
|
127
|
+
through to a fuzzy substring match that ALSO ran additively. A decoy
|
|
128
|
+
sharing the project name proves the exact path is used and that another
|
|
129
|
+
project's transcripts are not swept in — "kentro" matches four real
|
|
130
|
+
directories on this machine."""
|
|
131
|
+
projects = monkey_home / ".claude" / "projects"
|
|
132
|
+
slug = str(root).replace("/", "-")
|
|
133
|
+
(projects / slug).mkdir(parents=True)
|
|
134
|
+
write_transcript(projects / slug / "s.jsonl", [user_msg("hi there")])
|
|
135
|
+
|
|
136
|
+
decoy = projects / (slug + "-other-project")
|
|
137
|
+
decoy.mkdir(parents=True)
|
|
138
|
+
write_transcript(decoy / "d.jsonl", [user_msg("decoy transcript")])
|
|
139
|
+
|
|
140
|
+
found = m.find_session_files(root)
|
|
141
|
+
assert len(found) == 1, f"expected only the exact match, got {found}"
|
|
142
|
+
assert found[0].parent.name == slug, found[0]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
# ── End to end ───────────────────────────────────────────────────────────────
|
|
146
|
+
|
|
147
|
+
def test_mines_alias_to_path(m, root, monkey_home):
|
|
148
|
+
"""The test that fails if any defect returns."""
|
|
149
|
+
(root / "src").mkdir(parents=True, exist_ok=True)
|
|
150
|
+
(root / "src/deals.py").write_text("# deals\n")
|
|
151
|
+
slug = str(root).replace("/", "-")
|
|
152
|
+
d = monkey_home / ".claude" / "projects" / slug
|
|
153
|
+
d.mkdir(parents=True)
|
|
154
|
+
|
|
155
|
+
entries = []
|
|
156
|
+
for _ in range(6): # clear MIN_SCORE = 5.0 at weight 1.0
|
|
157
|
+
entries.append(user_msg("update the deals page please"))
|
|
158
|
+
entries.append(tool_msg("Read", {"file_path": str(root / "src/deals.py")}))
|
|
159
|
+
entries.append(tool_result())
|
|
160
|
+
write_transcript(d / "s.jsonl", entries)
|
|
161
|
+
|
|
162
|
+
miner = m.SessionMiner(root)
|
|
163
|
+
miner.mine(m.find_session_files(root))
|
|
164
|
+
res = miner.results()
|
|
165
|
+
assert res, "mined nothing from a transcript containing 6 clear pairings"
|
|
166
|
+
assert "deals" in res, list(res)
|
|
167
|
+
assert "src/deals.py" in res["deals"]["targets"], res["deals"]
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_junk_phrases_filtered(m, root):
|
|
171
|
+
got = m.extract_user_phrases('he said "that, if not" and "total documents:"')
|
|
172
|
+
assert not any("," in g or ":" in g for g in got), got
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def main() -> int:
|
|
176
|
+
import inspect
|
|
177
|
+
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
|
178
|
+
failed = []
|
|
179
|
+
tmp = Path(tempfile.mkdtemp(prefix="mine-test-"))
|
|
180
|
+
try:
|
|
181
|
+
for fn in tests:
|
|
182
|
+
root = tmp / fn.__name__ / "repo"
|
|
183
|
+
home = tmp / fn.__name__ / "home"
|
|
184
|
+
root.mkdir(parents=True); home.mkdir(parents=True)
|
|
185
|
+
m = load_miner(root, root)
|
|
186
|
+
params = inspect.signature(fn).parameters
|
|
187
|
+
kwargs = {}
|
|
188
|
+
needs_home = "monkey_home" in params
|
|
189
|
+
if needs_home:
|
|
190
|
+
# find_session_files() resolves ~/.claude/projects via Path.home()
|
|
191
|
+
m.Path.home = staticmethod(lambda: home)
|
|
192
|
+
kwargs["monkey_home"] = home
|
|
193
|
+
try:
|
|
194
|
+
fn(m, root, **kwargs)
|
|
195
|
+
print(f" ok {fn.__name__}")
|
|
196
|
+
except Exception as e:
|
|
197
|
+
failed.append(fn.__name__)
|
|
198
|
+
print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
|
|
199
|
+
finally:
|
|
200
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
201
|
+
print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
|
|
202
|
+
return 1 if failed else 0
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
if __name__ == "__main__":
|
|
206
|
+
sys.exit(main())
|
|
@@ -6,13 +6,13 @@
|
|
|
6
6
|
},
|
|
7
7
|
"metadata": {
|
|
8
8
|
"description": "Official marketplace for babel-fish - Codebase introspection and vocabulary translation for AI coding assistants",
|
|
9
|
-
"version": "2.0
|
|
9
|
+
"version": "2.4.0"
|
|
10
10
|
},
|
|
11
11
|
"plugins": [
|
|
12
12
|
{
|
|
13
13
|
"name": "babel-fish",
|
|
14
14
|
"description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
|
|
15
|
-
"version": "2.0
|
|
15
|
+
"version": "2.4.0",
|
|
16
16
|
"author": {
|
|
17
17
|
"name": "TheGlitchKing"
|
|
18
18
|
},
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "babel-fish",
|
|
3
3
|
"description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
|
|
4
|
-
"version": "2.0
|
|
4
|
+
"version": "2.4.0",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TheGlitchKing",
|
|
7
7
|
"email": "theglitchking@users.noreply.github.com"
|
package/.githooks/pre-commit
CHANGED
|
@@ -14,7 +14,8 @@ if echo "$STAGED_FILES" | grep -qE "$EXTENSIONS_PATTERN" 2>/dev/null; then
|
|
|
14
14
|
echo "[codebase-mapper] Regenerating project map..."
|
|
15
15
|
if $PYTHON "$MAP_SCRIPT" 2>/dev/null; then
|
|
16
16
|
git add .claude/project-map/PROJECT_MAP.md .claude/project-map/checksums.json \
|
|
17
|
-
.claude/project-map/sections/*.md .claude/project-map/learned-vocabulary.json
|
|
17
|
+
.claude/project-map/sections/*.md .claude/project-map/learned-vocabulary.json \
|
|
18
|
+
.claude/project-map/glossary.json 2>/dev/null || true
|
|
18
19
|
fi
|
|
19
20
|
fi
|
|
20
21
|
fi
|
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,254 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [2.4.0] - 2026-09-04
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- **`glossary.json` — a structured vocabulary artifact for machine consumers**
|
|
10
|
+
([#10](https://github.com/TheGlitchKing/babel-fish/issues/10)). Written to
|
|
11
|
+
`.claude/project-map/glossary.json` on every map build and staged by the
|
|
12
|
+
pre-commit hook. Consumers read this instead of parsing `01-vocabulary.md`.
|
|
13
|
+
|
|
14
|
+
The markdown was never a good parsing target: its `Notes` column mixes
|
|
15
|
+
descriptions with metadata, and pipe-escaping leaks into values — this repo's
|
|
16
|
+
own output contained `(auto \| nudge \| off)`. The JSON carries the real
|
|
17
|
+
string.
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- **The glossary contract described a format babel-fish has never emitted.**
|
|
22
|
+
`glossary-contract.md` v1.0 specified bullet entries
|
|
23
|
+
(`- **key** → \`path\` — desc`) and stated that non-conforming bullets are
|
|
24
|
+
ignored; the generator has always written a markdown table. A consumer built
|
|
25
|
+
strictly to that spec would extract **zero entries**. Rewritten (v2.0) around
|
|
26
|
+
`glossary.json`, with the markdown documented as human-facing output that is
|
|
27
|
+
not parsed.
|
|
28
|
+
|
|
29
|
+
- **Both sides of that contract documented a directory neither produces.** The
|
|
30
|
+
contract, the integration guide and the README referred to `.babel-fish/`;
|
|
31
|
+
babel-fish writes `.claude/project-map/`. The same error is mirrored in
|
|
32
|
+
semantic-memory's `smart-middle-activation.md` and `corpora-json.md`, where it
|
|
33
|
+
would have made Phase 3.1.0 find nothing — silently. Corrected here and filed
|
|
34
|
+
there as
|
|
35
|
+
[semantic-memory#28](https://github.com/the-glitch-kingdom/semantic-memory/issues/28).
|
|
36
|
+
|
|
37
|
+
- **`--project-root` crashed the generator.** `write_glossary()` computed the
|
|
38
|
+
source path with `SECTIONS_DIR.relative_to(PROJECT_ROOT)`, but `SECTIONS_DIR`
|
|
39
|
+
is bound to the script's own location, so pointing `--project-root` elsewhere
|
|
40
|
+
raised `ValueError`. Found by a test written for the new artifact.
|
|
41
|
+
|
|
42
|
+
- `integration-with-semantic-memory.md` no longer describes a setup that does
|
|
43
|
+
not exist. semantic-memory 1.5.1 ships no `translate` verbs, no `project-map`
|
|
44
|
+
corpus and no glossary reader; the guide now says so at the top instead of
|
|
45
|
+
giving instructions for it.
|
|
46
|
+
|
|
47
|
+
### Note
|
|
48
|
+
|
|
49
|
+
The consumer is unbuilt, so nothing was pinned to the old format. Issue #10
|
|
50
|
+
originally advised caution about "a breaking change to a format a downstream
|
|
51
|
+
consumer pins to" — checking semantic-memory's source showed zero
|
|
52
|
+
`translate`/`reverse_translate`/`list_vocabulary` verbs in its 164-entry tool
|
|
53
|
+
surface and no `glossary` string in `src/`. That freed the format choice
|
|
54
|
+
entirely.
|
|
55
|
+
|
|
56
|
+
## [2.3.0] - 2026-09-04
|
|
57
|
+
|
|
58
|
+
### Fixed
|
|
59
|
+
|
|
60
|
+
- **The grader could not tell a useless map from a good one**
|
|
61
|
+
([#9](https://github.com/TheGlitchKing/babel-fish/issues/9)). All seven graded
|
|
62
|
+
categories measure *form*, and `generate.py` always emits well-formed output,
|
|
63
|
+
so the completely empty pre-2.1.0 map scored **97.0% PASS** — identical
|
|
64
|
+
category-for-category to the populated map. Verified against the real artifact
|
|
65
|
+
recovered from git, not a reconstruction.
|
|
66
|
+
|
|
67
|
+
Rather than reweighting, usefulness is now reported as **warnings that never
|
|
68
|
+
touch the score**, so no existing install flips from pass to fail:
|
|
69
|
+
|
|
70
|
+
- `generate.py` records *inputs* beside outputs (`### Source files scanned`).
|
|
71
|
+
"0 routes" cannot be judged alone; "0 routes from 47 Python files" can. The
|
|
72
|
+
counts already existed in `main()` and were being discarded.
|
|
73
|
+
- An empty vocabulary warns — that is what babel-fish is for, so zero entries
|
|
74
|
+
means the map gave you nothing. This fires on the real pre-2.1.0 map and is
|
|
75
|
+
the signal that would have surfaced #6 at install time.
|
|
76
|
+
- Ten or more source files scanned with nothing extracted warns separately,
|
|
77
|
+
catching a parser that does not fit the stack. Small repos stay quiet.
|
|
78
|
+
- A populated-sections count prints as an explicit diagnostic.
|
|
79
|
+
|
|
80
|
+
- **The greenfield branch in `grade_vocabulary_accuracy()` was unreachable.**
|
|
81
|
+
`build_vocabulary_section()` emitted a placeholder *table row* for an empty
|
|
82
|
+
vocabulary, which parsed as a valid entry whose blank location counts as
|
|
83
|
+
neutral — scoring 1/1 = 100% and stepping straight over the `if not rows`
|
|
84
|
+
branch written to award 85%. It now emits prose.
|
|
85
|
+
|
|
86
|
+
- **`or '_No' in content` matched too much.** Any italicised word beginning
|
|
87
|
+
"No" (`_Note`, `_Nothing`) anywhere in `12-import-chains.md` scored a
|
|
88
|
+
populated section as an acceptable empty one at a flat 80%.
|
|
89
|
+
|
|
90
|
+
### Fixed (packaging)
|
|
91
|
+
|
|
92
|
+
- **The npm tarball shipped local and generated files.** `files` listed
|
|
93
|
+
`.claude/` wholesale, and npm does **not** honour `.gitignore` for paths named
|
|
94
|
+
there — so every release carried this repo's own generated project map,
|
|
95
|
+
another plugin's local state (`.claude/.semantic-memory/`), five plugins'
|
|
96
|
+
update caches, ~150 kB of `__pycache__` bytecode, and
|
|
97
|
+
`.claude/settings.local.json`. `files` now lists only what
|
|
98
|
+
`.claude/install.sh` actually copies plus the test suites. Also anchored
|
|
99
|
+
`checksums.json` to `./checksums.json`: a bare filename in `files` globs at
|
|
100
|
+
any depth, so it was matching `.claude/project-map/checksums.json` too.
|
|
101
|
+
|
|
102
|
+
Tarball: 76 files / 140.9 kB → **33 files / 65.6 kB**. Verified by installing
|
|
103
|
+
from the packed tarball into a scratch project.
|
|
104
|
+
|
|
105
|
+
### Added
|
|
106
|
+
|
|
107
|
+
- 9 tests in `test_grader.py`, 7 of which fail against the previous code.
|
|
108
|
+
`npm test` now runs three suites (8 + 12 + 9 = 29).
|
|
109
|
+
- `architecture/grading-semantics.md` — what the score means, what it
|
|
110
|
+
deliberately omits, and the measurements behind that choice.
|
|
111
|
+
|
|
112
|
+
### Not done, deliberately
|
|
113
|
+
|
|
114
|
+
Issue #9 originally proposed scoring section completeness on populated content.
|
|
115
|
+
Measured and withdrawn: it fails the *correct* map too (85.2%), because 19
|
|
116
|
+
sections is aspirational — a plugin repo can never populate routes, models,
|
|
117
|
+
schemas or migrations, so `populated/19` tops out near 10/19 on a perfect map.
|
|
118
|
+
It would fail every legitimately sparse repo, a worse failure than the one it
|
|
119
|
+
fixes. The 90% threshold and the category weights are unchanged.
|
|
120
|
+
|
|
121
|
+
## [2.2.0] - 2026-09-04
|
|
122
|
+
|
|
123
|
+
### Fixed
|
|
124
|
+
|
|
125
|
+
- **Session vocabulary mining has never worked**
|
|
126
|
+
([#7](https://github.com/TheGlitchKing/babel-fish/issues/7)). The issue
|
|
127
|
+
reported that `mine-sessions.py` has no caller. It also had six defects, each
|
|
128
|
+
sufficient on its own to make it extract nothing — it exited 0 reporting
|
|
129
|
+
"Extracted 0 alias(es)" for its entire existence, which is why they survived.
|
|
130
|
+
|
|
131
|
+
1. **Wrong JSONL nesting.** Read `msg['content']`; Claude Code writes
|
|
132
|
+
`msg['message']['content']`. Measured on a real transcript: 0 vs 157
|
|
133
|
+
`tool_use` blocks, 0 vs 14 user messages. Both halves of the pairing were
|
|
134
|
+
empty.
|
|
135
|
+
2. **Invalid regex quantifiers.** `{2,40?}` and `{2,30?}` are malformed brace
|
|
136
|
+
expressions that Python silently treats as literals, so both patterns
|
|
137
|
+
compiled and matched nothing — including the one implementing this
|
|
138
|
+
feature's own README example, "the numbers page".
|
|
139
|
+
3. **Tool results collide with user turns.** Results arrive as `role: "user"`
|
|
140
|
+
(167 of 181 in one transcript), so the pairing window closed on the
|
|
141
|
+
assistant's own output.
|
|
142
|
+
4. **Bash file access was invisible.** A real session ran 148 Bash calls
|
|
143
|
+
against 2 Read and 2 Edit; only 4 of 157 tool calls qualified.
|
|
144
|
+
5. **Injected text was mined as user speech.** Skill and slash-command bodies
|
|
145
|
+
arrive in the user slot, and taught the miner aliases from the injected
|
|
146
|
+
documents themselves (`block_index_edits`, `refactor authentication
|
|
147
|
+
system` — the latter from a skill's worked example). Now skipped via
|
|
148
|
+
`isMeta` / `isSidechain`.
|
|
149
|
+
6. **Session discovery never matched exactly.** `.lstrip('-')` stripped the
|
|
150
|
+
leading separator that `~/.claude/projects/` slugs keep, so every lookup
|
|
151
|
+
fell through to a fuzzy substring match that also ran additively — and a
|
|
152
|
+
name like `kentro` matches four unrelated projects, whose aliases would be
|
|
153
|
+
attributed to this repo.
|
|
154
|
+
|
|
155
|
+
Verified against 220 MB of transcripts for a real product repo: 0 aliases
|
|
156
|
+
before, 170 after, reading like genuine domain vocabulary (`sign-up` →
|
|
157
|
+
`payments.py`, `pricing` → `subscription_gate.py`).
|
|
158
|
+
|
|
159
|
+
- **`README.md` claimed mining happened "automatically".** It did not — nothing
|
|
160
|
+
called the miner. Now true, and documented with its two real caveats.
|
|
161
|
+
|
|
162
|
+
### Added
|
|
163
|
+
|
|
164
|
+
- **Mining runs at session start.** `hooks/session-start.js` spawns the miner
|
|
165
|
+
detached with output discarded and nothing awaited; it cannot delay or fail a
|
|
166
|
+
session, and no-ops when Python or the script is absent.
|
|
167
|
+
- **Incremental cursor** (`.mine-cursor.json`). Not only a cost guard:
|
|
168
|
+
`merge_learned()` adds scores, so re-mining a counted transcript inflates it
|
|
169
|
+
without bound. `--all` forces a full re-mine.
|
|
170
|
+
- 12 tests in `test_mine_sessions.py`, one per defect plus an end-to-end mine.
|
|
171
|
+
All 12 fail against the previous miner. `npm test` runs both suites (20).
|
|
172
|
+
- Docs: `architecture/session-vocabulary-mining.md` (including the transcript
|
|
173
|
+
shape assumptions the miner depends on but does not control) and
|
|
174
|
+
`troubleshooting/learned-vocabulary-empty.md`.
|
|
175
|
+
|
|
176
|
+
### Known issues
|
|
177
|
+
|
|
178
|
+
- Aliases land one session late: SessionStart mines transcripts through the
|
|
179
|
+
previous session, since the current one isn't written yet.
|
|
180
|
+
- Phrase quality is heuristic. Filtering drops clause-like candidates, but a
|
|
181
|
+
quoted string in a user message can still become an alias.
|
|
182
|
+
|
|
183
|
+
## [2.1.1] - 2026-09-04
|
|
184
|
+
|
|
185
|
+
### Fixed
|
|
186
|
+
|
|
187
|
+
- **Section 19 listed generated hit-em-with-the-docs reports.**
|
|
188
|
+
`.documentation/reports/` holds timestamped audit output, so every `hewtd
|
|
189
|
+
maintain` wrote a new filename, which changed the doc path set, moved the
|
|
190
|
+
checksum and forced a full map regeneration — the exact churn the path-only
|
|
191
|
+
doc hash exists to prevent, reintroduced through a directory that was
|
|
192
|
+
gitignored but never excluded from the doc walk. `reports` joins `archive` in
|
|
193
|
+
`DOC_SKIP_DIRS`. Regression test added; it fails without the fix.
|
|
194
|
+
|
|
195
|
+
## [2.1.0] - 2026-09-04
|
|
196
|
+
|
|
197
|
+
### Fixed
|
|
198
|
+
|
|
199
|
+
- **Plugin and skill repositories no longer generate an empty project map**
|
|
200
|
+
([#6](https://github.com/TheGlitchKing/babel-fish/issues/6)). Run babel-fish
|
|
201
|
+
against a repo of markdown skills, slash commands and bash scripts and every
|
|
202
|
+
one of the 19 sections came back a "none detected" stub. Two causes, both
|
|
203
|
+
fixed:
|
|
204
|
+
|
|
205
|
+
- The checksum was blind to the files that define such a repo.
|
|
206
|
+
`collect_watched_files()` returned 11 files for babel-fish's own repository,
|
|
207
|
+
with no `.md` and no `.sh`, so editing a `SKILL.md` left the checksum
|
|
208
|
+
bit-identical and `is_unchanged()` exited before parsing. Skill and command
|
|
209
|
+
manifests are now matched by path glob (`skills/*/SKILL.md`,
|
|
210
|
+
`commands/*.md`), and `.sh` joins `WATCHED_EXTENSIONS`.
|
|
211
|
+
- Nothing read those manifests. `SkillParser` now feeds skill and command
|
|
212
|
+
frontmatter into section 01 (vocabulary) and section 10 (tools) — the two
|
|
213
|
+
sections they already fit. No new sections, no renumbering.
|
|
214
|
+
|
|
215
|
+
Measured on this repository: 0 vocabulary entries to 10, sections 2,589 bytes
|
|
216
|
+
to 4,389. On `hit-em-with-the-docs`, an unrelated plugin repo: 0 to 30.
|
|
217
|
+
|
|
218
|
+
- **Section 19 missed `.documentation/` trees and went stale silently.** Doc
|
|
219
|
+
directories were never watched, so adding a document did not move the
|
|
220
|
+
checksum and the pointer list rotted until an unrelated source file happened
|
|
221
|
+
to change. Doc paths are now hashed **without** mtime: adding, renaming or
|
|
222
|
+
deleting a document refreshes section 19, while editing one does not force a
|
|
223
|
+
full regeneration. `.documentation` joins the doc directories, and generated
|
|
224
|
+
navigation (`INDEX.md`, `REGISTRY.md`) plus `archive/` are excluded — without
|
|
225
|
+
that, a 15-domain tree contributes 32 nav files and crowds every real
|
|
226
|
+
document out of the 30-entry cap.
|
|
227
|
+
|
|
228
|
+
- **`checksums.json` was stale**, so the documented curl installer aborted with
|
|
229
|
+
`CHECKSUM MISMATCH` for everyone. `.claude/install.sh` was edited in `da8d2f7`
|
|
230
|
+
without regenerating the manifest.
|
|
231
|
+
|
|
232
|
+
### Added
|
|
233
|
+
|
|
234
|
+
- First tests in the repository: `npm test` runs an 8-test regression suite over
|
|
235
|
+
a fixture repo shaped like #6. Stdlib `assert`, no framework. Verified to fail
|
|
236
|
+
7/8 against the pre-fix generator rather than merely passing after it.
|
|
237
|
+
- `.documentation/` docs for the watch set, the skill parser contract, and an
|
|
238
|
+
empty/stale map troubleshooting guide; operational runbook gained the
|
|
239
|
+
corresponding gotchas.
|
|
240
|
+
|
|
241
|
+
### Known issues
|
|
242
|
+
|
|
243
|
+
- `grader.py` scores a completely empty map at 97.0% PASS, the same as a fully
|
|
244
|
+
populated one — "Vocabulary Accuracy" is 100% on zero entries because
|
|
245
|
+
0/0 = 100. It measures well-formedness, not usefulness, and must not be used
|
|
246
|
+
to confirm an extractor fix. Left unchanged here: adding a floor would fail
|
|
247
|
+
existing installs that currently pass.
|
|
248
|
+
- `01-vocabulary.md` is emitted as a markdown table, while
|
|
249
|
+
`.documentation/api/glossary-contract.md` specifies `- **key** → \`path\``
|
|
250
|
+
bullets. A consumer implemented strictly to that contract extracts zero
|
|
251
|
+
entries. Predates this release; which side moves is undecided.
|
|
252
|
+
|
|
5
253
|
## [2.0.3] - 2026-06-08
|
|
6
254
|
|
|
7
255
|
### Fixed
|