@theglitchking/babel-fish 2.0.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/.claude/project-map/generate.py +258 -22
  2. package/.claude/project-map/grader.py +77 -1
  3. package/.claude/project-map/mine-sessions.py +141 -23
  4. package/.claude/project-map/test_generate.py +267 -0
  5. package/.claude/project-map/test_grader.py +164 -0
  6. package/.claude/project-map/test_mine_sessions.py +206 -0
  7. package/.claude-plugin/marketplace.json +2 -2
  8. package/.claude-plugin/plugin.json +1 -1
  9. package/.githooks/pre-commit +2 -1
  10. package/CHANGELOG.md +248 -0
  11. package/README.md +9 -2
  12. package/checksums.json +2 -2
  13. package/hooks/session-start.js +34 -1
  14. package/package.json +13 -4
  15. package/.claude/project-map/PROJECT_MAP.md +0 -61
  16. package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
  17. package/.claude/project-map/checksums.json +0 -6
  18. package/.claude/project-map/learned-vocabulary.json +0 -1
  19. package/.claude/project-map/reports/install-report.md +0 -54
  20. package/.claude/project-map/reports/iteration-01-report.md +0 -54
  21. package/.claude/project-map/reports/iteration-01-score.json +0 -7
  22. package/.claude/project-map/sections/01-vocabulary.md +0 -8
  23. package/.claude/project-map/sections/02-service-topology.md +0 -6
  24. package/.claude/project-map/sections/03-environment.md +0 -6
  25. package/.claude/project-map/sections/04-api-routes.md +0 -6
  26. package/.claude/project-map/sections/05-data-models.md +0 -4
  27. package/.claude/project-map/sections/06-schemas.md +0 -4
  28. package/.claude/project-map/sections/07-services.md +0 -6
  29. package/.claude/project-map/sections/08-background-jobs.md +0 -5
  30. package/.claude/project-map/sections/09-frontend-features.md +0 -4
  31. package/.claude/project-map/sections/10-tools-commands.md +0 -8
  32. package/.claude/project-map/sections/11-migrations.md +0 -4
  33. package/.claude/project-map/sections/12-import-chains.md +0 -7
  34. package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
  35. package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
  36. package/.claude/project-map/sections/15-auth-config.md +0 -6
  37. package/.claude/project-map/sections/16-infra-profile.md +0 -13
  38. package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
  39. package/.claude/project-map/sections/18-dead-code.md +0 -9
  40. package/.claude/project-map/sections/19-doc-pointers.md +0 -5
  41. package/.claude/project-map/stack.json +0 -12
  42. package/.claude/rules/operational-runbook.md +0 -40
  43. package/.claude/rules/project-vocabulary.md +0 -25
  44. package/.claude/settings.json +0 -6
  45. package/.claude/settings.local.json +0 -6
  46. package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
@@ -0,0 +1,164 @@
1
+ #!/usr/bin/env python3
2
+ """Regression tests for grader.py — run: python3 .claude/project-map/test_grader.py
3
+
4
+ Guards the fixes from issue #9. The headline defect there was that the grader
5
+ scored a completely empty map and a populated one identically (97.0% both), so
6
+ these assert the SIGNAL that separates them, not the score — deliberately, since
7
+ folding usefulness into the weighted score would fail legitimately sparse repos.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import importlib.util
12
+ import shutil
13
+ import sys
14
+ import tempfile
15
+ from pathlib import Path
16
+
17
+ HERE = Path(__file__).parent
18
+
19
+
20
+ def load_grader(map_dir: Path):
21
+ spec = importlib.util.spec_from_file_location("grader_under_test", HERE / "grader.py")
22
+ g = importlib.util.module_from_spec(spec)
23
+ argv, sys.argv = sys.argv, ["grader.py"]
24
+ try:
25
+ spec.loader.exec_module(g)
26
+ finally:
27
+ sys.argv = argv
28
+ g.MAP_DIR = map_dir
29
+ g.SECTIONS_DIR = map_dir / "sections"
30
+ g.PROJECT_ROOT = map_dir
31
+ return g
32
+
33
+
34
+ def write_map(map_dir: Path, *, vocab: int, scanned: int, extracted: int = 0) -> None:
35
+ (map_dir / "sections").mkdir(parents=True, exist_ok=True)
36
+ (map_dir / "PROJECT_MAP.md").write_text(f"""## Stats
37
+
38
+ | Metric | Count |
39
+ |--------|-------|
40
+ | API Routes | {extracted} |
41
+ | Data Models | 0 |
42
+ | Schemas/DTOs | 0 |
43
+ | Frontend Features | 0 |
44
+ | Vocabulary Entries | {vocab} |
45
+
46
+ ### Source files scanned
47
+
48
+ | Language | Files |
49
+ |----------|-------|
50
+ | python | {scanned} |
51
+
52
+ ## Section Index
53
+ """)
54
+
55
+
56
+ def test_empty_vocabulary_warns(tmp):
57
+ write_map(tmp, vocab=0, scanned=0)
58
+ w = load_grader(tmp).usefulness_warnings()
59
+ assert any("Vocabulary is empty" in x for x in w), w
60
+
61
+
62
+ def test_populated_vocabulary_does_not_warn(tmp):
63
+ write_map(tmp, vocab=10, scanned=3)
64
+ w = load_grader(tmp).usefulness_warnings()
65
+ assert not any("Vocabulary is empty" in x for x in w), w
66
+
67
+
68
+ def test_many_files_no_extraction_warns(tmp):
69
+ write_map(tmp, vocab=7, scanned=47, extracted=0)
70
+ w = load_grader(tmp).usefulness_warnings()
71
+ assert any("Scanned 47" in x for x in w), w
72
+
73
+
74
+ def test_few_files_no_extraction_is_quiet(tmp):
75
+ """This repo: 3 CLI shims yielding no routes is correct, not a defect."""
76
+ write_map(tmp, vocab=10, scanned=3, extracted=0)
77
+ w = load_grader(tmp).usefulness_warnings()
78
+ assert not any("Scanned" in x for x in w), w
79
+
80
+
81
+ def test_warnings_do_not_touch_the_score(tmp):
82
+ """Usefulness is reported, never scored — scoring it would fail sparse repos."""
83
+ g = load_grader(tmp)
84
+ write_map(tmp, vocab=0, scanned=99)
85
+ for name in dir(g):
86
+ if name.startswith("grade_"):
87
+ r = getattr(g, name)()
88
+ assert 0.0 <= r.raw_score <= 100.0
89
+ assert g.usefulness_warnings(), "expected warnings for an empty map"
90
+
91
+
92
+ def test_populated_chains_not_misread_as_empty(tmp):
93
+ """`or '_No' in content` matched _Note / _Nothing anywhere in the file and
94
+ scored a populated section as an acceptable empty one at a flat 80%."""
95
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
96
+ (tmp / "sections" / "12-import-chains.md").write_text(
97
+ "# Section 12\n\n_Note: partial._\n\n```\nsrc/a.py -> src/b.py\n```\n")
98
+ g = load_grader(tmp)
99
+ r = g.grade_import_chains()
100
+ assert "No chains traced" not in r.details, (
101
+ f"a populated section was scored as empty: {r.details}")
102
+
103
+
104
+ def test_stub_chains_still_recognised(tmp):
105
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
106
+ (tmp / "sections" / "12-import-chains.md").write_text(
107
+ "# Section 12\n\n_No import chains traced._\n")
108
+ r = load_grader(tmp).grade_import_chains()
109
+ assert "No chains traced" in r.details, r.details
110
+
111
+
112
+ def test_empty_vocab_section_reaches_greenfield_branch(tmp):
113
+ """generate.py used to emit a placeholder TABLE ROW, which parsed as a valid
114
+ entry with a neutral location — scoring an empty vocabulary 100% and leaving
115
+ this branch permanently unreachable."""
116
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
117
+ (tmp / "sections" / "01-vocabulary.md").write_text(
118
+ "# Section 01\n\n| Alias | Type | Location | Notes |\n|---|---|---|---|\n\n"
119
+ "_No vocabulary generated yet — add source code to populate._\n")
120
+ r = load_grader(tmp).grade_vocabulary_accuracy()
121
+ assert r.raw_score == 85.0, f"greenfield branch not reached: {r.raw_score} {r.details}"
122
+
123
+
124
+ def test_generate_emits_prose_not_a_row_for_empty_vocab():
125
+ spec = importlib.util.spec_from_file_location("gen", HERE / "generate.py")
126
+ gen = importlib.util.module_from_spec(spec)
127
+ argv, sys.argv = sys.argv, ["generate.py"]
128
+ try:
129
+ spec.loader.exec_module(gen)
130
+ finally:
131
+ sys.argv = argv
132
+ out = gen.build_vocabulary_section([])
133
+ rows = [l for l in out.splitlines()
134
+ if l.strip().startswith("|") and "Alias" not in l and set(l.strip()) - set("|-: ")]
135
+ assert not rows, f"empty vocabulary must not emit a table row, got {rows}"
136
+ assert "No vocabulary generated yet" in out
137
+
138
+
139
+ def main() -> int:
140
+ import inspect
141
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
142
+ failed = []
143
+ base = Path(tempfile.mkdtemp(prefix="grader-test-"))
144
+ try:
145
+ for fn in tests:
146
+ d = base / fn.__name__
147
+ d.mkdir(parents=True)
148
+ try:
149
+ if "tmp" in inspect.signature(fn).parameters:
150
+ fn(d)
151
+ else:
152
+ fn()
153
+ print(f" ok {fn.__name__}")
154
+ except Exception as e:
155
+ failed.append(fn.__name__)
156
+ print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
157
+ finally:
158
+ shutil.rmtree(base, ignore_errors=True)
159
+ print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
160
+ return 1 if failed else 0
161
+
162
+
163
+ if __name__ == "__main__":
164
+ sys.exit(main())
@@ -0,0 +1,206 @@
1
+ #!/usr/bin/env python3
2
+ """Regression tests for mine-sessions.py — run: python3 .claude/project-map/test_mine_sessions.py
3
+
4
+ Same shape as test_generate.py: stdlib assert + __main__, no framework.
5
+
6
+ Every test here corresponds to a defect that shipped and produced NO error —
7
+ the miner exited 0 and reported "Extracted 0 alias(es)" for its entire
8
+ existence. Silent-zero is the failure mode these guard against.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import importlib.util
13
+ import json
14
+ import shutil
15
+ import sys
16
+ import tempfile
17
+ from datetime import datetime, timezone
18
+ from pathlib import Path
19
+
20
+ HERE = Path(__file__).parent
21
+
22
+
23
+ def load_miner(project_root: Path, cursor_dir: Path):
24
+ spec = importlib.util.spec_from_file_location("miner_under_test", HERE / "mine-sessions.py")
25
+ mod = importlib.util.module_from_spec(spec)
26
+ argv, sys.argv = sys.argv, ["mine-sessions.py"]
27
+ try:
28
+ spec.loader.exec_module(mod)
29
+ finally:
30
+ sys.argv = argv
31
+ mod.PROJECT_ROOT = project_root
32
+ mod.MINE_CURSOR = cursor_dir / ".mine-cursor.json"
33
+ return mod
34
+
35
+
36
+ def write_transcript(path: Path, entries: list[dict]) -> None:
37
+ path.write_text("\n".join(json.dumps(e) for e in entries) + "\n")
38
+
39
+
40
+ def user_msg(text: str, **extra) -> dict:
41
+ """A genuine user turn, in the real nested shape Claude Code writes."""
42
+ return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
43
+ "message": {"role": "user", "content": [{"type": "text", "text": text}]}, **extra}
44
+
45
+
46
+ def tool_msg(name: str, inp: dict) -> dict:
47
+ return {"type": "assistant", "timestamp": datetime.now(timezone.utc).isoformat(),
48
+ "message": {"role": "assistant",
49
+ "content": [{"type": "tool_use", "name": name, "input": inp}]}}
50
+
51
+
52
+ def tool_result() -> dict:
53
+ """Tool results come back with role=user — the collision that broke pairing."""
54
+ return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
55
+ "message": {"role": "user",
56
+ "content": [{"type": "tool_result", "content": "ok"}]}}
57
+
58
+
59
+ # ── Defect 1: JSONL nesting ──────────────────────────────────────────────────
60
+
61
+ def test_reads_nested_message_content(m, root):
62
+ msg = user_msg("open the settings page")
63
+ assert m.msg_role(msg) == "user", "role must be read from message.role"
64
+ content = m.msg_content(msg)
65
+ assert isinstance(content, list) and content[0]["text"] == "open the settings page"
66
+
67
+
68
+ def test_still_reads_flat_content(m, root):
69
+ """Older/third-party transcripts must not regress."""
70
+ flat = {"role": "user", "content": [{"type": "text", "text": "hello there"}]}
71
+ assert m.msg_role(flat) == "user"
72
+ assert m.msg_content(flat)[0]["text"] == "hello there"
73
+
74
+
75
+ def test_extracts_tool_paths_from_nested(m, root):
76
+ paths = m.extract_file_paths_from_tool_calls(
77
+ [tool_msg("Read", {"file_path": str(root / "src/app.py")})])
78
+ assert paths == ["src/app.py"], paths
79
+
80
+
81
+ # ── Defect 2: regex quantifiers ──────────────────────────────────────────────
82
+
83
+ def test_the_x_page_pattern_matches(m, root):
84
+ """'{2,40?}' compiled fine and matched nothing — no error, just silence."""
85
+ got = m.extract_user_phrases("please look at the settings page")
86
+ assert "settings" in got, got
87
+
88
+
89
+ def test_x_feature_pattern_matches(m, root):
90
+ got = m.extract_user_phrases("the billing workflow is broken")
91
+ assert any("billing" in g for g in got), got
92
+
93
+
94
+ # ── Defect 3: tool_result / user-turn collision ──────────────────────────────
95
+
96
+ def test_tool_result_is_not_a_user_turn(m, root):
97
+ assert m.is_user_turn(user_msg("the deals page")) is True
98
+ assert m.is_user_turn(tool_result()) is False, (
99
+ "tool results carry role=user; treating them as user turns closed the "
100
+ "pairing window on the assistant's own output"
101
+ )
102
+
103
+
104
+ def test_meta_and_sidechain_are_not_user_turns(m, root):
105
+ assert m.is_user_turn(user_msg("skill body text", isMeta=True)) is False
106
+ assert m.is_user_turn(user_msg("subagent text", isSidechain=True)) is False
107
+
108
+
109
+ # ── Defect 4: Bash-mediated file access ──────────────────────────────────────
110
+
111
+ def test_bash_paths_extracted_when_file_exists(m, root):
112
+ (root / "src").mkdir(parents=True, exist_ok=True)
113
+ (root / "src/app.py").write_text("# app\n")
114
+ got = m.extract_paths_from_bash("sed -n '1,40p' src/app.py", root)
115
+ assert got == ["src/app.py"], got
116
+
117
+
118
+ def test_bash_paths_ignore_nonexistent(m, root):
119
+ got = m.extract_paths_from_bash("cat totally/made/up.py && ls -la", root)
120
+ assert got == [], f"only real files may become aliases, got {got}"
121
+
122
+
123
+ # ── Defect 6: session discovery ──────────────────────────────────────────────
124
+
125
+ def test_slug_keeps_leading_separator(m, root, monkey_home):
126
+ """.lstrip('-') meant the exact match never hit, so every lookup fell
127
+ through to a fuzzy substring match that ALSO ran additively. A decoy
128
+ sharing the project name proves the exact path is used and that another
129
+ project's transcripts are not swept in — "kentro" matches four real
130
+ directories on this machine."""
131
+ projects = monkey_home / ".claude" / "projects"
132
+ slug = str(root).replace("/", "-")
133
+ (projects / slug).mkdir(parents=True)
134
+ write_transcript(projects / slug / "s.jsonl", [user_msg("hi there")])
135
+
136
+ decoy = projects / (slug + "-other-project")
137
+ decoy.mkdir(parents=True)
138
+ write_transcript(decoy / "d.jsonl", [user_msg("decoy transcript")])
139
+
140
+ found = m.find_session_files(root)
141
+ assert len(found) == 1, f"expected only the exact match, got {found}"
142
+ assert found[0].parent.name == slug, found[0]
143
+
144
+
145
+ # ── End to end ───────────────────────────────────────────────────────────────
146
+
147
+ def test_mines_alias_to_path(m, root, monkey_home):
148
+ """The test that fails if any defect returns."""
149
+ (root / "src").mkdir(parents=True, exist_ok=True)
150
+ (root / "src/deals.py").write_text("# deals\n")
151
+ slug = str(root).replace("/", "-")
152
+ d = monkey_home / ".claude" / "projects" / slug
153
+ d.mkdir(parents=True)
154
+
155
+ entries = []
156
+ for _ in range(6): # clear MIN_SCORE = 5.0 at weight 1.0
157
+ entries.append(user_msg("update the deals page please"))
158
+ entries.append(tool_msg("Read", {"file_path": str(root / "src/deals.py")}))
159
+ entries.append(tool_result())
160
+ write_transcript(d / "s.jsonl", entries)
161
+
162
+ miner = m.SessionMiner(root)
163
+ miner.mine(m.find_session_files(root))
164
+ res = miner.results()
165
+ assert res, "mined nothing from a transcript containing 6 clear pairings"
166
+ assert "deals" in res, list(res)
167
+ assert "src/deals.py" in res["deals"]["targets"], res["deals"]
168
+
169
+
170
+ def test_junk_phrases_filtered(m, root):
171
+ got = m.extract_user_phrases('he said "that, if not" and "total documents:"')
172
+ assert not any("," in g or ":" in g for g in got), got
173
+
174
+
175
+ def main() -> int:
176
+ import inspect
177
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
178
+ failed = []
179
+ tmp = Path(tempfile.mkdtemp(prefix="mine-test-"))
180
+ try:
181
+ for fn in tests:
182
+ root = tmp / fn.__name__ / "repo"
183
+ home = tmp / fn.__name__ / "home"
184
+ root.mkdir(parents=True); home.mkdir(parents=True)
185
+ m = load_miner(root, root)
186
+ params = inspect.signature(fn).parameters
187
+ kwargs = {}
188
+ needs_home = "monkey_home" in params
189
+ if needs_home:
190
+ # find_session_files() resolves ~/.claude/projects via Path.home()
191
+ m.Path.home = staticmethod(lambda: home)
192
+ kwargs["monkey_home"] = home
193
+ try:
194
+ fn(m, root, **kwargs)
195
+ print(f" ok {fn.__name__}")
196
+ except Exception as e:
197
+ failed.append(fn.__name__)
198
+ print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
199
+ finally:
200
+ shutil.rmtree(tmp, ignore_errors=True)
201
+ print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
202
+ return 1 if failed else 0
203
+
204
+
205
+ if __name__ == "__main__":
206
+ sys.exit(main())
@@ -6,13 +6,13 @@
6
6
  },
7
7
  "metadata": {
8
8
  "description": "Official marketplace for babel-fish - Codebase introspection and vocabulary translation for AI coding assistants",
9
- "version": "2.0.3"
9
+ "version": "2.4.0"
10
10
  },
11
11
  "plugins": [
12
12
  {
13
13
  "name": "babel-fish",
14
14
  "description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
15
- "version": "2.0.3",
15
+ "version": "2.4.0",
16
16
  "author": {
17
17
  "name": "TheGlitchKing"
18
18
  },
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "babel-fish",
3
3
  "description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
4
- "version": "2.0.3",
4
+ "version": "2.4.0",
5
5
  "author": {
6
6
  "name": "TheGlitchKing",
7
7
  "email": "theglitchking@users.noreply.github.com"
@@ -14,7 +14,8 @@ if echo "$STAGED_FILES" | grep -qE "$EXTENSIONS_PATTERN" 2>/dev/null; then
14
14
  echo "[codebase-mapper] Regenerating project map..."
15
15
  if $PYTHON "$MAP_SCRIPT" 2>/dev/null; then
16
16
  git add .claude/project-map/PROJECT_MAP.md .claude/project-map/checksums.json \
17
- .claude/project-map/sections/*.md .claude/project-map/learned-vocabulary.json 2>/dev/null || true
17
+ .claude/project-map/sections/*.md .claude/project-map/learned-vocabulary.json \
18
+ .claude/project-map/glossary.json 2>/dev/null || true
18
19
  fi
19
20
  fi
20
21
  fi
package/CHANGELOG.md CHANGED
@@ -2,6 +2,254 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
+ ## [2.4.0] - 2026-09-04
6
+
7
+ ### Added
8
+
9
+ - **`glossary.json` — a structured vocabulary artifact for machine consumers**
10
+ ([#10](https://github.com/TheGlitchKing/babel-fish/issues/10)). Written to
11
+ `.claude/project-map/glossary.json` on every map build and staged by the
12
+ pre-commit hook. Consumers read this instead of parsing `01-vocabulary.md`.
13
+
14
+ The markdown was never a good parsing target: its `Notes` column mixes
15
+ descriptions with metadata, and pipe-escaping leaks into values — this repo's
16
+ own output contained `(auto \| nudge \| off)`. The JSON carries the real
17
+ string.
18
+
19
+ ### Fixed
20
+
21
+ - **The glossary contract described a format babel-fish has never emitted.**
22
+ `glossary-contract.md` v1.0 specified bullet entries
23
+ (`- **key** → \`path\` — desc`) and stated that non-conforming bullets are
24
+ ignored; the generator has always written a markdown table. A consumer built
25
+ strictly to that spec would extract **zero entries**. Rewritten (v2.0) around
26
+ `glossary.json`, with the markdown documented as human-facing output that is
27
+ not parsed.
28
+
29
+ - **Both sides of that contract documented a directory neither produces.** The
30
+ contract, the integration guide and the README referred to `.babel-fish/`;
31
+ babel-fish writes `.claude/project-map/`. The same error is mirrored in
32
+ semantic-memory's `smart-middle-activation.md` and `corpora-json.md`, where it
33
+ would have made Phase 3.1.0 find nothing — silently. Corrected here and filed
34
+ there as
35
+ [semantic-memory#28](https://github.com/the-glitch-kingdom/semantic-memory/issues/28).
36
+
37
+ - **`--project-root` crashed the generator.** `write_glossary()` computed the
38
+ source path with `SECTIONS_DIR.relative_to(PROJECT_ROOT)`, but `SECTIONS_DIR`
39
+ is bound to the script's own location, so pointing `--project-root` elsewhere
40
+ raised `ValueError`. Found by a test written for the new artifact.
41
+
42
+ - `integration-with-semantic-memory.md` no longer describes a setup that does
43
+ not exist. semantic-memory 1.5.1 ships no `translate` verbs, no `project-map`
44
+ corpus and no glossary reader; the guide now says so at the top instead of
45
+ giving instructions for it.
46
+
47
+ ### Note
48
+
49
+ The consumer is unbuilt, so nothing was pinned to the old format. Issue #10
50
+ originally advised caution about "a breaking change to a format a downstream
51
+ consumer pins to" — checking semantic-memory's source showed zero
52
+ `translate`/`reverse_translate`/`list_vocabulary` verbs in its 164-entry tool
53
+ surface and no `glossary` string in `src/`. That freed the format choice
54
+ entirely.
55
+
56
+ ## [2.3.0] - 2026-09-04
57
+
58
+ ### Fixed
59
+
60
+ - **The grader could not tell a useless map from a good one**
61
+ ([#9](https://github.com/TheGlitchKing/babel-fish/issues/9)). All seven graded
62
+ categories measure *form*, and `generate.py` always emits well-formed output,
63
+ so the completely empty pre-2.1.0 map scored **97.0% PASS** — identical
64
+ category-for-category to the populated map. Verified against the real artifact
65
+ recovered from git, not a reconstruction.
66
+
67
+ Rather than reweighting, usefulness is now reported as **warnings that never
68
+ touch the score**, so no existing install flips from pass to fail:
69
+
70
+ - `generate.py` records *inputs* beside outputs (`### Source files scanned`).
71
+ "0 routes" cannot be judged alone; "0 routes from 47 Python files" can. The
72
+ counts already existed in `main()` and were being discarded.
73
+ - An empty vocabulary warns — that is what babel-fish is for, so zero entries
74
+ means the map gave you nothing. This fires on the real pre-2.1.0 map and is
75
+ the signal that would have surfaced #6 at install time.
76
+ - Ten or more source files scanned with nothing extracted warns separately,
77
+ catching a parser that does not fit the stack. Small repos stay quiet.
78
+ - A populated-sections count prints as an explicit diagnostic.
79
+
80
+ - **The greenfield branch in `grade_vocabulary_accuracy()` was unreachable.**
81
+ `build_vocabulary_section()` emitted a placeholder *table row* for an empty
82
+ vocabulary, which parsed as a valid entry whose blank location counts as
83
+ neutral — scoring 1/1 = 100% and stepping straight over the `if not rows`
84
+ branch written to award 85%. It now emits prose.
85
+
86
+ - **`or '_No' in content` matched too much.** Any italicised word beginning
87
+ "No" (`_Note`, `_Nothing`) anywhere in `12-import-chains.md` scored a
88
+ populated section as an acceptable empty one at a flat 80%.
89
+
90
+ ### Fixed (packaging)
91
+
92
+ - **The npm tarball shipped local and generated files.** `files` listed
93
+ `.claude/` wholesale, and npm does **not** honour `.gitignore` for paths named
94
+ there — so every release carried this repo's own generated project map,
95
+ another plugin's local state (`.claude/.semantic-memory/`), five plugins'
96
+ update caches, ~150 kB of `__pycache__` bytecode, and
97
+ `.claude/settings.local.json`. `files` now lists only what
98
+ `.claude/install.sh` actually copies plus the test suites. Also anchored
99
+ `checksums.json` to `./checksums.json`: a bare filename in `files` globs at
100
+ any depth, so it was matching `.claude/project-map/checksums.json` too.
101
+
102
+ Tarball: 76 files / 140.9 kB → **33 files / 65.6 kB**. Verified by installing
103
+ from the packed tarball into a scratch project.
104
+
105
+ ### Added
106
+
107
+ - 9 tests in `test_grader.py`, 7 of which fail against the previous code.
108
+ `npm test` now runs three suites (8 + 12 + 9 = 29).
109
+ - `architecture/grading-semantics.md` — what the score means, what it
110
+ deliberately omits, and the measurements behind that choice.
111
+
112
+ ### Not done, deliberately
113
+
114
+ Issue #9 originally proposed scoring section completeness on populated content.
115
+ Measured and withdrawn: it fails the *correct* map too (85.2%), because 19
116
+ sections is aspirational — a plugin repo can never populate routes, models,
117
+ schemas or migrations, so `populated/19` tops out near 10/19 on a perfect map.
118
+ It would fail every legitimately sparse repo, a worse failure than the one it
119
+ fixes. The 90% threshold and the category weights are unchanged.
120
+
121
+ ## [2.2.0] - 2026-09-04
122
+
123
+ ### Fixed
124
+
125
+ - **Session vocabulary mining has never worked**
126
+ ([#7](https://github.com/TheGlitchKing/babel-fish/issues/7)). The issue
127
+ reported that `mine-sessions.py` has no caller. It also had six defects, each
128
+ sufficient on its own to make it extract nothing — it exited 0 reporting
129
+ "Extracted 0 alias(es)" for its entire existence, which is why they survived.
130
+
131
+ 1. **Wrong JSONL nesting.** Read `msg['content']`; Claude Code writes
132
+ `msg['message']['content']`. Measured on a real transcript: 0 vs 157
133
+ `tool_use` blocks, 0 vs 14 user messages. Both halves of the pairing were
134
+ empty.
135
+ 2. **Invalid regex quantifiers.** `{2,40?}` and `{2,30?}` are malformed brace
136
+ expressions that Python silently treats as literals, so both patterns
137
+ compiled and matched nothing — including the one implementing this
138
+ feature's own README example, "the numbers page".
139
+ 3. **Tool results collide with user turns.** Results arrive as `role: "user"`
140
+ (167 of 181 in one transcript), so the pairing window closed on the
141
+ assistant's own output.
142
+ 4. **Bash file access was invisible.** A real session ran 148 Bash calls
143
+ against 2 Read and 2 Edit; only 4 of 157 tool calls qualified.
144
+ 5. **Injected text was mined as user speech.** Skill and slash-command bodies
145
+ arrive in the user slot, and taught the miner aliases from the injected
146
+ documents themselves (`block_index_edits`, `refactor authentication
147
+ system` — the latter from a skill's worked example). Now skipped via
148
+ `isMeta` / `isSidechain`.
149
+ 6. **Session discovery never matched exactly.** `.lstrip('-')` stripped the
150
+ leading separator that `~/.claude/projects/` slugs keep, so every lookup
151
+ fell through to a fuzzy substring match that also ran additively — and a
152
+ name like `kentro` matches four unrelated projects, whose aliases would be
153
+ attributed to this repo.
154
+
155
+ Verified against 220 MB of transcripts for a real product repo: 0 aliases
156
+ before, 170 after, reading like genuine domain vocabulary (`sign-up` →
157
+ `payments.py`, `pricing` → `subscription_gate.py`).
158
+
159
+ - **`README.md` claimed mining happened "automatically".** It did not — nothing
160
+ called the miner. Now true, and documented with its two real caveats.
161
+
162
+ ### Added
163
+
164
+ - **Mining runs at session start.** `hooks/session-start.js` spawns the miner
165
+ detached with output discarded and nothing awaited; it cannot delay or fail a
166
+ session, and no-ops when Python or the script is absent.
167
+ - **Incremental cursor** (`.mine-cursor.json`). Not only a cost guard:
168
+ `merge_learned()` adds scores, so re-mining a counted transcript inflates it
169
+ without bound. `--all` forces a full re-mine.
170
+ - 12 tests in `test_mine_sessions.py`, one per defect plus an end-to-end mine.
171
+ All 12 fail against the previous miner. `npm test` runs both suites (20).
172
+ - Docs: `architecture/session-vocabulary-mining.md` (including the transcript
173
+ shape assumptions the miner depends on but does not control) and
174
+ `troubleshooting/learned-vocabulary-empty.md`.
175
+
176
+ ### Known issues
177
+
178
+ - Aliases land one session late: SessionStart mines transcripts through the
179
+ previous session, since the current one isn't written yet.
180
+ - Phrase quality is heuristic. Filtering drops clause-like candidates, but a
181
+ quoted string in a user message can still become an alias.
182
+
183
+ ## [2.1.1] - 2026-09-04
184
+
185
+ ### Fixed
186
+
187
+ - **Section 19 listed generated hit-em-with-the-docs reports.**
188
+ `.documentation/reports/` holds timestamped audit output, so every `hewtd
189
+ maintain` wrote a new filename, which changed the doc path set, moved the
190
+ checksum and forced a full map regeneration — the exact churn the path-only
191
+ doc hash exists to prevent, reintroduced through a directory that was
192
+ gitignored but never excluded from the doc walk. `reports` joins `archive` in
193
+ `DOC_SKIP_DIRS`. Regression test added; it fails without the fix.
194
+
195
+ ## [2.1.0] - 2026-09-04
196
+
197
+ ### Fixed
198
+
199
+ - **Plugin and skill repositories no longer generate an empty project map**
200
+ ([#6](https://github.com/TheGlitchKing/babel-fish/issues/6)). Run babel-fish
201
+ against a repo of markdown skills, slash commands and bash scripts and every
202
+ one of the 19 sections came back a "none detected" stub. Two causes, both
203
+ fixed:
204
+
205
+ - The checksum was blind to the files that define such a repo.
206
+ `collect_watched_files()` returned 11 files for babel-fish's own repository,
207
+ with no `.md` and no `.sh`, so editing a `SKILL.md` left the checksum
208
+ bit-identical and `is_unchanged()` exited before parsing. Skill and command
209
+ manifests are now matched by path glob (`skills/*/SKILL.md`,
210
+ `commands/*.md`), and `.sh` joins `WATCHED_EXTENSIONS`.
211
+ - Nothing read those manifests. `SkillParser` now feeds skill and command
212
+ frontmatter into section 01 (vocabulary) and section 10 (tools) — the two
213
+ sections they already fit. No new sections, no renumbering.
214
+
215
+ Measured on this repository: 0 vocabulary entries to 10, sections 2,589 bytes
216
+ to 4,389. On `hit-em-with-the-docs`, an unrelated plugin repo: 0 to 30.
217
+
218
+ - **Section 19 missed `.documentation/` trees and went stale silently.** Doc
219
+ directories were never watched, so adding a document did not move the
220
+ checksum and the pointer list rotted until an unrelated source file happened
221
+ to change. Doc paths are now hashed **without** mtime: adding, renaming or
222
+ deleting a document refreshes section 19, while editing one does not force a
223
+ full regeneration. `.documentation` joins the doc directories, and generated
224
+ navigation (`INDEX.md`, `REGISTRY.md`) plus `archive/` are excluded — without
225
+ that, a 15-domain tree contributes 32 nav files and crowds every real
226
+ document out of the 30-entry cap.
227
+
228
+ - **`checksums.json` was stale**, so the documented curl installer aborted with
229
+ `CHECKSUM MISMATCH` for everyone. `.claude/install.sh` was edited in `da8d2f7`
230
+ without regenerating the manifest.
231
+
232
+ ### Added
233
+
234
+ - First tests in the repository: `npm test` runs an 8-test regression suite over
235
+ a fixture repo shaped like #6. Stdlib `assert`, no framework. Verified to fail
236
+ 7/8 against the pre-fix generator rather than merely passing after it.
237
+ - `.documentation/` docs for the watch set, the skill parser contract, and an
238
+ empty/stale map troubleshooting guide; operational runbook gained the
239
+ corresponding gotchas.
240
+
241
+ ### Known issues
242
+
243
+ - `grader.py` scores a completely empty map at 97.0% PASS, the same as a fully
244
+ populated one — "Vocabulary Accuracy" is 100% on zero entries because
245
+ 0/0 = 100. It measures well-formedness, not usefulness, and must not be used
246
+ to confirm an extractor fix. Left unchanged here: adding a floor would fail
247
+ existing installs that currently pass.
248
+ - `01-vocabulary.md` is emitted as a markdown table, while
249
+ `.documentation/api/glossary-contract.md` specifies `- **key** → \`path\``
250
+ bullets. A consumer implemented strictly to that contract extracts zero
251
+ entries. Predates this release; which side moves is undecided.
252
+
5
253
  ## [2.0.3] - 2026-06-08
6
254
 
7
255
  ### Fixed