@theglitchking/babel-fish 2.0.3 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/.claude/project-map/generate.py +206 -21
  2. package/.claude/project-map/grader.py +77 -1
  3. package/.claude/project-map/mine-sessions.py +141 -23
  4. package/.claude/project-map/test_generate.py +203 -0
  5. package/.claude/project-map/test_grader.py +164 -0
  6. package/.claude/project-map/test_mine_sessions.py +206 -0
  7. package/.claude-plugin/marketplace.json +2 -2
  8. package/.claude-plugin/plugin.json +1 -1
  9. package/CHANGELOG.md +197 -0
  10. package/README.md +9 -2
  11. package/checksums.json +2 -2
  12. package/hooks/session-start.js +34 -1
  13. package/package.json +13 -4
  14. package/.claude/project-map/PROJECT_MAP.md +0 -61
  15. package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
  16. package/.claude/project-map/checksums.json +0 -6
  17. package/.claude/project-map/learned-vocabulary.json +0 -1
  18. package/.claude/project-map/reports/install-report.md +0 -54
  19. package/.claude/project-map/reports/iteration-01-report.md +0 -54
  20. package/.claude/project-map/reports/iteration-01-score.json +0 -7
  21. package/.claude/project-map/sections/01-vocabulary.md +0 -8
  22. package/.claude/project-map/sections/02-service-topology.md +0 -6
  23. package/.claude/project-map/sections/03-environment.md +0 -6
  24. package/.claude/project-map/sections/04-api-routes.md +0 -6
  25. package/.claude/project-map/sections/05-data-models.md +0 -4
  26. package/.claude/project-map/sections/06-schemas.md +0 -4
  27. package/.claude/project-map/sections/07-services.md +0 -6
  28. package/.claude/project-map/sections/08-background-jobs.md +0 -5
  29. package/.claude/project-map/sections/09-frontend-features.md +0 -4
  30. package/.claude/project-map/sections/10-tools-commands.md +0 -8
  31. package/.claude/project-map/sections/11-migrations.md +0 -4
  32. package/.claude/project-map/sections/12-import-chains.md +0 -7
  33. package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
  34. package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
  35. package/.claude/project-map/sections/15-auth-config.md +0 -6
  36. package/.claude/project-map/sections/16-infra-profile.md +0 -13
  37. package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
  38. package/.claude/project-map/sections/18-dead-code.md +0 -9
  39. package/.claude/project-map/sections/19-doc-pointers.md +0 -5
  40. package/.claude/project-map/stack.json +0 -12
  41. package/.claude/rules/operational-runbook.md +0 -40
  42. package/.claude/rules/project-vocabulary.md +0 -25
  43. package/.claude/settings.json +0 -6
  44. package/.claude/settings.local.json +0 -6
  45. package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
@@ -0,0 +1,206 @@
1
+ #!/usr/bin/env python3
2
+ """Regression tests for mine-sessions.py — run: python3 .claude/project-map/test_mine_sessions.py
3
+
4
+ Same shape as test_generate.py: stdlib assert + __main__, no framework.
5
+
6
+ Every test here corresponds to a defect that shipped and produced NO error —
7
+ the miner exited 0 and reported "Extracted 0 alias(es)" for its entire
8
+ existence. Silent-zero is the failure mode these guard against.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import importlib.util
13
+ import json
14
+ import shutil
15
+ import sys
16
+ import tempfile
17
+ from datetime import datetime, timezone
18
+ from pathlib import Path
19
+
20
+ HERE = Path(__file__).parent
21
+
22
+
23
+ def load_miner(project_root: Path, cursor_dir: Path):
24
+ spec = importlib.util.spec_from_file_location("miner_under_test", HERE / "mine-sessions.py")
25
+ mod = importlib.util.module_from_spec(spec)
26
+ argv, sys.argv = sys.argv, ["mine-sessions.py"]
27
+ try:
28
+ spec.loader.exec_module(mod)
29
+ finally:
30
+ sys.argv = argv
31
+ mod.PROJECT_ROOT = project_root
32
+ mod.MINE_CURSOR = cursor_dir / ".mine-cursor.json"
33
+ return mod
34
+
35
+
36
+ def write_transcript(path: Path, entries: list[dict]) -> None:
37
+ path.write_text("\n".join(json.dumps(e) for e in entries) + "\n")
38
+
39
+
40
+ def user_msg(text: str, **extra) -> dict:
41
+ """A genuine user turn, in the real nested shape Claude Code writes."""
42
+ return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
43
+ "message": {"role": "user", "content": [{"type": "text", "text": text}]}, **extra}
44
+
45
+
46
+ def tool_msg(name: str, inp: dict) -> dict:
47
+ return {"type": "assistant", "timestamp": datetime.now(timezone.utc).isoformat(),
48
+ "message": {"role": "assistant",
49
+ "content": [{"type": "tool_use", "name": name, "input": inp}]}}
50
+
51
+
52
+ def tool_result() -> dict:
53
+ """Tool results come back with role=user — the collision that broke pairing."""
54
+ return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
55
+ "message": {"role": "user",
56
+ "content": [{"type": "tool_result", "content": "ok"}]}}
57
+
58
+
59
+ # ── Defect 1: JSONL nesting ──────────────────────────────────────────────────
60
+
61
+ def test_reads_nested_message_content(m, root):
62
+ msg = user_msg("open the settings page")
63
+ assert m.msg_role(msg) == "user", "role must be read from message.role"
64
+ content = m.msg_content(msg)
65
+ assert isinstance(content, list) and content[0]["text"] == "open the settings page"
66
+
67
+
68
+ def test_still_reads_flat_content(m, root):
69
+ """Older/third-party transcripts must not regress."""
70
+ flat = {"role": "user", "content": [{"type": "text", "text": "hello there"}]}
71
+ assert m.msg_role(flat) == "user"
72
+ assert m.msg_content(flat)[0]["text"] == "hello there"
73
+
74
+
75
+ def test_extracts_tool_paths_from_nested(m, root):
76
+ paths = m.extract_file_paths_from_tool_calls(
77
+ [tool_msg("Read", {"file_path": str(root / "src/app.py")})])
78
+ assert paths == ["src/app.py"], paths
79
+
80
+
81
+ # ── Defect 2: regex quantifiers ──────────────────────────────────────────────
82
+
83
+ def test_the_x_page_pattern_matches(m, root):
84
+ """'{2,40?}' compiled fine and matched nothing — no error, just silence."""
85
+ got = m.extract_user_phrases("please look at the settings page")
86
+ assert "settings" in got, got
87
+
88
+
89
+ def test_x_feature_pattern_matches(m, root):
90
+ got = m.extract_user_phrases("the billing workflow is broken")
91
+ assert any("billing" in g for g in got), got
92
+
93
+
94
+ # ── Defect 3: tool_result / user-turn collision ──────────────────────────────
95
+
96
+ def test_tool_result_is_not_a_user_turn(m, root):
97
+ assert m.is_user_turn(user_msg("the deals page")) is True
98
+ assert m.is_user_turn(tool_result()) is False, (
99
+ "tool results carry role=user; treating them as user turns closed the "
100
+ "pairing window on the assistant's own output"
101
+ )
102
+
103
+
104
+ def test_meta_and_sidechain_are_not_user_turns(m, root):
105
+ assert m.is_user_turn(user_msg("skill body text", isMeta=True)) is False
106
+ assert m.is_user_turn(user_msg("subagent text", isSidechain=True)) is False
107
+
108
+
109
+ # ── Defect 4: Bash-mediated file access ──────────────────────────────────────
110
+
111
+ def test_bash_paths_extracted_when_file_exists(m, root):
112
+ (root / "src").mkdir(parents=True, exist_ok=True)
113
+ (root / "src/app.py").write_text("# app\n")
114
+ got = m.extract_paths_from_bash("sed -n '1,40p' src/app.py", root)
115
+ assert got == ["src/app.py"], got
116
+
117
+
118
+ def test_bash_paths_ignore_nonexistent(m, root):
119
+ got = m.extract_paths_from_bash("cat totally/made/up.py && ls -la", root)
120
+ assert got == [], f"only real files may become aliases, got {got}"
121
+
122
+
123
+ # ── Defect 6: session discovery ──────────────────────────────────────────────
124
+
125
+ def test_slug_keeps_leading_separator(m, root, monkey_home):
126
+ """.lstrip('-') meant the exact match never hit, so every lookup fell
127
+ through to a fuzzy substring match that ALSO ran additively. A decoy
128
+ sharing the project name proves the exact path is used and that another
129
+ project's transcripts are not swept in — "kentro" matches four real
130
+ directories on this machine."""
131
+ projects = monkey_home / ".claude" / "projects"
132
+ slug = str(root).replace("/", "-")
133
+ (projects / slug).mkdir(parents=True)
134
+ write_transcript(projects / slug / "s.jsonl", [user_msg("hi there")])
135
+
136
+ decoy = projects / (slug + "-other-project")
137
+ decoy.mkdir(parents=True)
138
+ write_transcript(decoy / "d.jsonl", [user_msg("decoy transcript")])
139
+
140
+ found = m.find_session_files(root)
141
+ assert len(found) == 1, f"expected only the exact match, got {found}"
142
+ assert found[0].parent.name == slug, found[0]
143
+
144
+
145
+ # ── End to end ───────────────────────────────────────────────────────────────
146
+
147
+ def test_mines_alias_to_path(m, root, monkey_home):
148
+ """The test that fails if any defect returns."""
149
+ (root / "src").mkdir(parents=True, exist_ok=True)
150
+ (root / "src/deals.py").write_text("# deals\n")
151
+ slug = str(root).replace("/", "-")
152
+ d = monkey_home / ".claude" / "projects" / slug
153
+ d.mkdir(parents=True)
154
+
155
+ entries = []
156
+ for _ in range(6): # clear MIN_SCORE = 5.0 at weight 1.0
157
+ entries.append(user_msg("update the deals page please"))
158
+ entries.append(tool_msg("Read", {"file_path": str(root / "src/deals.py")}))
159
+ entries.append(tool_result())
160
+ write_transcript(d / "s.jsonl", entries)
161
+
162
+ miner = m.SessionMiner(root)
163
+ miner.mine(m.find_session_files(root))
164
+ res = miner.results()
165
+ assert res, "mined nothing from a transcript containing 6 clear pairings"
166
+ assert "deals" in res, list(res)
167
+ assert "src/deals.py" in res["deals"]["targets"], res["deals"]
168
+
169
+
170
+ def test_junk_phrases_filtered(m, root):
171
+ got = m.extract_user_phrases('he said "that, if not" and "total documents:"')
172
+ assert not any("," in g or ":" in g for g in got), got
173
+
174
+
175
+ def main() -> int:
176
+ import inspect
177
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
178
+ failed = []
179
+ tmp = Path(tempfile.mkdtemp(prefix="mine-test-"))
180
+ try:
181
+ for fn in tests:
182
+ root = tmp / fn.__name__ / "repo"
183
+ home = tmp / fn.__name__ / "home"
184
+ root.mkdir(parents=True); home.mkdir(parents=True)
185
+ m = load_miner(root, root)
186
+ params = inspect.signature(fn).parameters
187
+ kwargs = {}
188
+ needs_home = "monkey_home" in params
189
+ if needs_home:
190
+ # find_session_files() resolves ~/.claude/projects via Path.home()
191
+ m.Path.home = staticmethod(lambda: home)
192
+ kwargs["monkey_home"] = home
193
+ try:
194
+ fn(m, root, **kwargs)
195
+ print(f" ok {fn.__name__}")
196
+ except Exception as e:
197
+ failed.append(fn.__name__)
198
+ print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
199
+ finally:
200
+ shutil.rmtree(tmp, ignore_errors=True)
201
+ print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
202
+ return 1 if failed else 0
203
+
204
+
205
+ if __name__ == "__main__":
206
+ sys.exit(main())
@@ -6,13 +6,13 @@
6
6
  },
7
7
  "metadata": {
8
8
  "description": "Official marketplace for babel-fish - Codebase introspection and vocabulary translation for AI coding assistants",
9
- "version": "2.0.3"
9
+ "version": "2.3.0"
10
10
  },
11
11
  "plugins": [
12
12
  {
13
13
  "name": "babel-fish",
14
14
  "description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
15
- "version": "2.0.3",
15
+ "version": "2.3.0",
16
16
  "author": {
17
17
  "name": "TheGlitchKing"
18
18
  },
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "babel-fish",
3
3
  "description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
4
- "version": "2.0.3",
4
+ "version": "2.3.0",
5
5
  "author": {
6
6
  "name": "TheGlitchKing",
7
7
  "email": "theglitchking@users.noreply.github.com"
package/CHANGELOG.md CHANGED
@@ -2,6 +2,203 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
+ ## [2.3.0] - 2026-09-04
6
+
7
+ ### Fixed
8
+
9
+ - **The grader could not tell a useless map from a good one**
10
+ ([#9](https://github.com/TheGlitchKing/babel-fish/issues/9)). All seven graded
11
+ categories measure *form*, and `generate.py` always emits well-formed output,
12
+ so the completely empty pre-2.1.0 map scored **97.0% PASS** — identical
13
+ category-for-category to the populated map. Verified against the real artifact
14
+ recovered from git, not a reconstruction.
15
+
16
+ Rather than reweighting, usefulness is now reported as **warnings that never
17
+ touch the score**, so no existing install flips from pass to fail:
18
+
19
+ - `generate.py` records *inputs* beside outputs (`### Source files scanned`).
20
+ "0 routes" cannot be judged alone; "0 routes from 47 Python files" can. The
21
+ counts already existed in `main()` and were being discarded.
22
+ - An empty vocabulary warns — that is what babel-fish is for, so zero entries
23
+ means the map gave you nothing. This fires on the real pre-2.1.0 map and is
24
+ the signal that would have surfaced #6 at install time.
25
+ - Ten or more source files scanned with nothing extracted warns separately,
26
+ catching a parser that does not fit the stack. Small repos stay quiet.
27
+ - A populated-sections count prints as an explicit diagnostic.
28
+
29
+ - **The greenfield branch in `grade_vocabulary_accuracy()` was unreachable.**
30
+ `build_vocabulary_section()` emitted a placeholder *table row* for an empty
31
+ vocabulary, which parsed as a valid entry whose blank location counts as
32
+ neutral — scoring 1/1 = 100% and stepping straight over the `if not rows`
33
+ branch written to award 85%. It now emits prose.
34
+
35
+ - **`or '_No' in content` matched too much.** Any italicised word beginning
36
+ "No" (`_Note`, `_Nothing`) anywhere in `12-import-chains.md` scored a
37
+ populated section as an acceptable empty one at a flat 80%.
38
+
39
+ ### Fixed (packaging)
40
+
41
+ - **The npm tarball shipped local and generated files.** `files` listed
42
+ `.claude/` wholesale, and npm does **not** honour `.gitignore` for paths named
43
+ there — so every release carried this repo's own generated project map,
44
+ another plugin's local state (`.claude/.semantic-memory/`), five plugins'
45
+ update caches, ~150 kB of `__pycache__` bytecode, and
46
+ `.claude/settings.local.json`. `files` now lists only what
47
+ `.claude/install.sh` actually copies plus the test suites. Also anchored
48
+ `checksums.json` to `./checksums.json`: a bare filename in `files` globs at
49
+ any depth, so it was matching `.claude/project-map/checksums.json` too.
50
+
51
+ Tarball: 76 files / 140.9 kB → **33 files / 65.6 kB**. Verified by installing
52
+ from the packed tarball into a scratch project.
53
+
54
+ ### Added
55
+
56
+ - 9 tests in `test_grader.py`, 7 of which fail against the previous code.
57
+ `npm test` now runs three suites (8 + 12 + 9 = 29).
58
+ - `architecture/grading-semantics.md` — what the score means, what it
59
+ deliberately omits, and the measurements behind that choice.
60
+
61
+ ### Not done, deliberately
62
+
63
+ Issue #9 originally proposed scoring section completeness on populated content.
64
+ Measured and withdrawn: it fails the *correct* map too (85.2%), because 19
65
+ sections is aspirational — a plugin repo can never populate routes, models,
66
+ schemas or migrations, so `populated/19` tops out near 10/19 on a perfect map.
67
+ It would fail every legitimately sparse repo, a worse failure than the one it
68
+ fixes. The 90% threshold and the category weights are unchanged.
69
+
70
+ ## [2.2.0] - 2026-09-04
71
+
72
+ ### Fixed
73
+
74
+ - **Session vocabulary mining has never worked**
75
+ ([#7](https://github.com/TheGlitchKing/babel-fish/issues/7)). The issue
76
+ reported that `mine-sessions.py` has no caller. It also had six defects, each
77
+ sufficient on its own to make it extract nothing — it exited 0 reporting
78
+ "Extracted 0 alias(es)" for its entire existence, which is why they survived.
79
+
80
+ 1. **Wrong JSONL nesting.** Read `msg['content']`; Claude Code writes
81
+ `msg['message']['content']`. Measured on a real transcript: 0 vs 157
82
+ `tool_use` blocks, 0 vs 14 user messages. Both halves of the pairing were
83
+ empty.
84
+ 2. **Invalid regex quantifiers.** `{2,40?}` and `{2,30?}` are malformed brace
85
+ expressions that Python silently treats as literals, so both patterns
86
+ compiled and matched nothing — including the one implementing this
87
+ feature's own README example, "the numbers page".
88
+ 3. **Tool results collide with user turns.** Results arrive as `role: "user"`
89
+ (167 of 181 in one transcript), so the pairing window closed on the
90
+ assistant's own output.
91
+ 4. **Bash file access was invisible.** A real session ran 148 Bash calls
92
+ against 2 Read and 2 Edit; only 4 of 157 tool calls qualified.
93
+ 5. **Injected text was mined as user speech.** Skill and slash-command bodies
94
+ arrive in the user slot, and taught the miner aliases from the injected
95
+ documents themselves (`block_index_edits`, `refactor authentication
96
+ system` — the latter from a skill's worked example). Now skipped via
97
+ `isMeta` / `isSidechain`.
98
+ 6. **Session discovery never matched exactly.** `.lstrip('-')` stripped the
99
+ leading separator that `~/.claude/projects/` slugs keep, so every lookup
100
+ fell through to a fuzzy substring match that also ran additively — and a
101
+ name like `kentro` matches four unrelated projects, whose aliases would be
102
+ attributed to this repo.
103
+
104
+ Verified against 220 MB of transcripts for a real product repo: 0 aliases
105
+ before, 170 after, reading like genuine domain vocabulary (`sign-up` →
106
+ `payments.py`, `pricing` → `subscription_gate.py`).
107
+
108
+ - **`README.md` claimed mining happened "automatically".** It did not — nothing
109
+ called the miner. Now true, and documented with its two real caveats.
110
+
111
+ ### Added
112
+
113
+ - **Mining runs at session start.** `hooks/session-start.js` spawns the miner
114
+ detached with output discarded and nothing awaited; it cannot delay or fail a
115
+ session, and no-ops when Python or the script is absent.
116
+ - **Incremental cursor** (`.mine-cursor.json`). Not only a cost guard:
117
+ `merge_learned()` adds scores, so re-mining a counted transcript inflates it
118
+ without bound. `--all` forces a full re-mine.
119
+ - 12 tests in `test_mine_sessions.py`, one per defect plus an end-to-end mine.
120
+ All 12 fail against the previous miner. `npm test` runs both suites (20).
121
+ - Docs: `architecture/session-vocabulary-mining.md` (including the transcript
122
+ shape assumptions the miner depends on but does not control) and
123
+ `troubleshooting/learned-vocabulary-empty.md`.
124
+
125
+ ### Known issues
126
+
127
+ - Aliases land one session late: SessionStart mines transcripts through the
128
+ previous session, since the current one isn't written yet.
129
+ - Phrase quality is heuristic. Filtering drops clause-like candidates, but a
130
+ quoted string in a user message can still become an alias.
131
+
132
+ ## [2.1.1] - 2026-09-04
133
+
134
+ ### Fixed
135
+
136
+ - **Section 19 listed generated hit-em-with-the-docs reports.**
137
+ `.documentation/reports/` holds timestamped audit output, so every `hewtd
138
+ maintain` wrote a new filename, which changed the doc path set, moved the
139
+ checksum and forced a full map regeneration — the exact churn the path-only
140
+ doc hash exists to prevent, reintroduced through a directory that was
141
+ gitignored but never excluded from the doc walk. `reports` joins `archive` in
142
+ `DOC_SKIP_DIRS`. Regression test added; it fails without the fix.
143
+
144
+ ## [2.1.0] - 2026-09-04
145
+
146
+ ### Fixed
147
+
148
+ - **Plugin and skill repositories no longer generate an empty project map**
149
+ ([#6](https://github.com/TheGlitchKing/babel-fish/issues/6)). Run babel-fish
150
+ against a repo of markdown skills, slash commands and bash scripts and every
151
+ one of the 19 sections came back a "none detected" stub. Two causes, both
152
+ fixed:
153
+
154
+ - The checksum was blind to the files that define such a repo.
155
+ `collect_watched_files()` returned 11 files for babel-fish's own repository,
156
+ with no `.md` and no `.sh`, so editing a `SKILL.md` left the checksum
157
+ bit-identical and `is_unchanged()` exited before parsing. Skill and command
158
+ manifests are now matched by path glob (`skills/*/SKILL.md`,
159
+ `commands/*.md`), and `.sh` joins `WATCHED_EXTENSIONS`.
160
+ - Nothing read those manifests. `SkillParser` now feeds skill and command
161
+ frontmatter into section 01 (vocabulary) and section 10 (tools) — the two
162
+ sections they already fit. No new sections, no renumbering.
163
+
164
+ Measured on this repository: 0 vocabulary entries to 10, sections 2,589 bytes
165
+ to 4,389. On `hit-em-with-the-docs`, an unrelated plugin repo: 0 to 30.
166
+
167
+ - **Section 19 missed `.documentation/` trees and went stale silently.** Doc
168
+ directories were never watched, so adding a document did not move the
169
+ checksum and the pointer list rotted until an unrelated source file happened
170
+ to change. Doc paths are now hashed **without** mtime: adding, renaming or
171
+ deleting a document refreshes section 19, while editing one does not force a
172
+ full regeneration. `.documentation` joins the doc directories, and generated
173
+ navigation (`INDEX.md`, `REGISTRY.md`) plus `archive/` are excluded — without
174
+ that, a 15-domain tree contributes 32 nav files and crowds every real
175
+ document out of the 30-entry cap.
176
+
177
+ - **`checksums.json` was stale**, so the documented curl installer aborted with
178
+ `CHECKSUM MISMATCH` for everyone. `.claude/install.sh` was edited in `da8d2f7`
179
+ without regenerating the manifest.
180
+
181
+ ### Added
182
+
183
+ - First tests in the repository: `npm test` runs an 8-test regression suite over
184
+ a fixture repo shaped like #6. Stdlib `assert`, no framework. Verified to fail
185
+ 7/8 against the pre-fix generator rather than merely passing after it.
186
+ - `.documentation/` docs for the watch set, the skill parser contract, and an
187
+ empty/stale map troubleshooting guide; operational runbook gained the
188
+ corresponding gotchas.
189
+
190
+ ### Known issues
191
+
192
+ - `grader.py` scores a completely empty map at 97.0% PASS, the same as a fully
193
+ populated one — "Vocabulary Accuracy" is 100% on zero entries because
194
+ 0/0 = 100. It measures well-formedness, not usefulness, and must not be used
195
+ to confirm an extractor fix. Left unchanged here: adding a floor would fail
196
+ existing installs that currently pass.
197
+ - `01-vocabulary.md` is emitted as a markdown table, while
198
+ `.documentation/api/glossary-contract.md` specifies `- **key** → \`path\``
199
+ bullets. A consumer implemented strictly to that contract extracts zero
200
+ entries. Predates this release; which side moves is undecided.
201
+
5
202
  ## [2.0.3] - 2026-06-08
6
203
 
7
204
  ### Fixed
package/README.md CHANGED
@@ -9,7 +9,7 @@
9
9
  [![GitHub: TheGlitchKing/babel-fish](https://img.shields.io/badge/GitHub-TheGlitchKing%2Fbabel--fish-blue)](https://github.com/TheGlitchKing/babel-fish)
10
10
 
11
11
  > [!NOTE]
12
- > **Pairs with [`semantic-memory`](https://github.com/TheGlitchKing/semantic-sidekick) (formerly `semantic-sidekick`).** When both are installed, semantic-memory consumes babel-fish's auto-generated `.babel-fish/` output as a `project-map` corpus AND extracts `01-vocabulary.md` into a structured `glossary.json` side-channel. That gives your AI a deterministic `translate("deals page") → "features/deal-pipeline/DealPipeline.tsx"` MCP verb instead of relying on semantic-search-luck. See [`docs/glossary-contract.md`](./docs/glossary-contract.md) for the producer/consumer data contract and [`docs/integration-with-semantic-memory.md`](./docs/integration-with-semantic-memory.md) for the setup walkthrough. babel-fish standalone behavior is unchanged — semantic-memory is purely additive.
12
+ > **Pairs with [`semantic-memory`](https://github.com/TheGlitchKing/semantic-sidekick) (formerly `semantic-sidekick`).** When both are installed, semantic-memory consumes babel-fish's auto-generated `.babel-fish/` output as a `project-map` corpus AND extracts `01-vocabulary.md` into a structured `glossary.json` side-channel. That gives your AI a deterministic `translate("deals page") → "features/deal-pipeline/DealPipeline.tsx"` MCP verb instead of relying on semantic-search-luck. See [`.documentation/api/glossary-contract.md`](./.documentation/api/glossary-contract.md) for the producer/consumer data contract and [`.documentation/quickstart/integration-with-semantic-memory.md`](./.documentation/quickstart/integration-with-semantic-memory.md) for the setup walkthrough. babel-fish standalone behavior is unchanged — semantic-memory is purely additive.
13
13
 
14
14
  ---
15
15
 
@@ -200,7 +200,14 @@ python .claude/project-map/grader.py
200
200
 
201
201
  ## Learned Vocabulary
202
202
 
203
- Every AI session is mined for vocabulary. When you say "the numbers page" and the AI opens `DealAnalyzerV2.tsx`, that alias is recorded with a score (frequency × recency). Aliases with a score ≥ 5 appear in `17-learned-vocabulary.md` automatically.
203
+ Every AI session is mined for vocabulary. When you say "the numbers page" and the AI opens `DealAnalyzerV2.tsx`, that alias is recorded with a score (frequency × recency). Aliases with a score ≥ 5 appear in `17-learned-vocabulary.md`.
204
+
205
+ Mining runs automatically at session start *(2.2.0+)*, detached and best-effort — it never delays or blocks a session. Only transcripts changed since the last run are read.
206
+
207
+ Two things to expect:
208
+
209
+ - **Aliases land one session late.** A session's transcript isn't written until it ends, so what you say today is mined at the *next* session start.
210
+ - **Operational sessions mine little.** The miner learns feature nouns — "the deals page", "the billing workflow". A session spent on refactoring or releases contains few of those and will correctly yield nothing. See [learned vocabulary is empty](./.documentation/troubleshooting/learned-vocabulary-empty.md).
204
211
 
205
212
  Run the miner manually:
206
213
 
package/checksums.json CHANGED
@@ -1,4 +1,4 @@
1
1
  {
2
- "install_sh": "3fa520851ef80139d792ade3e6008e4356678153f39414f79c1920bf2a69dab1",
3
- "note": "SHA256 of .claude/install.sh — verified by the remote installer before execution"
2
+ "install_sh": "565ad35c264c8e683ade82f441ab5a9f8feb3921e0b4180623bb038b23452518",
3
+ "note": "SHA256 of .claude/install.sh \u2014 verified by the remote installer before execution"
4
4
  }
@@ -1,11 +1,44 @@
1
1
  #!/usr/bin/env node
2
2
  // babel-fish SessionStart hook. Runs the runtime's update check per
3
- // policy (off / nudge / auto). No plugin-specific .mcp.json reconcile.
3
+ // policy (off / nudge / auto), then kicks off session vocabulary mining.
4
4
 
5
5
  import { runSessionStart } from "@theglitchking/claude-plugin-runtime";
6
+ import { spawn, spawnSync } from "node:child_process";
7
+ import { existsSync } from "node:fs";
8
+ import { join } from "node:path";
9
+
10
+ // Mining is best-effort: it must never delay or fail session start, so it is
11
+ // detached with output discarded and nothing is awaited. It reads only
12
+ // transcripts changed since the last run (see mine-sessions.py's cursor).
13
+ //
14
+ // Timing: SessionStart sees transcripts through the PREVIOUS session — the
15
+ // current one isn't written yet — so an alias lands one session after it is
16
+ // first used. Accepted; a SessionEnd hook would be exact but is a second hook
17
+ // for one session of latency.
18
+ function mineVocabulary(cwd) {
19
+ try {
20
+ const script = join(cwd, ".claude", "project-map", "mine-sessions.py");
21
+ if (!existsSync(script)) return; // marketplace install that skipped the copy
22
+
23
+ const python = ["python3", "python"].find(
24
+ (bin) => spawnSync(bin, ["--version"], { stdio: "ignore" }).status === 0
25
+ );
26
+ if (!python) return; // no interpreter — nothing to do, and nothing to say
27
+
28
+ spawn(python, [script], {
29
+ cwd,
30
+ detached: true,
31
+ stdio: "ignore",
32
+ }).unref();
33
+ } catch {
34
+ // never surface: a mining failure must not affect session start
35
+ }
36
+ }
6
37
 
7
38
  await runSessionStart({
8
39
  packageName: "@theglitchking/babel-fish",
9
40
  pluginName: "babel-fish",
10
41
  configFile: "babel-fish.json",
11
42
  });
43
+
44
+ mineVocabulary(process.cwd());
package/package.json CHANGED
@@ -1,13 +1,14 @@
1
1
  {
2
2
  "name": "@theglitchking/babel-fish",
3
- "version": "2.0.3",
3
+ "version": "2.3.0",
4
4
  "description": "Gives your AI coding assistant instant, accurate knowledge of every route, model, service, feature, and infrastructure element in your codebase.",
5
5
  "type": "module",
6
6
  "bin": {
7
7
  "babel-fish": "bin/babel-fish.js"
8
8
  },
9
9
  "scripts": {
10
- "postinstall": "node scripts/link-skills.js"
10
+ "postinstall": "node scripts/link-skills.js",
11
+ "test": "python3 .claude/project-map/test_generate.py && python3 .claude/project-map/test_mine_sessions.py && python3 .claude/project-map/test_grader.py"
11
12
  },
12
13
  "files": [
13
14
  "bin/",
@@ -15,11 +16,19 @@
15
16
  "hooks/",
16
17
  "scripts/link-skills.js",
17
18
  ".claude-plugin/",
18
- ".claude/",
19
+ ".claude/install.sh",
20
+ ".claude/scripts/",
21
+ ".claude/templates/",
22
+ ".claude/project-map/generate.py",
23
+ ".claude/project-map/grader.py",
24
+ ".claude/project-map/mine-sessions.py",
25
+ ".claude/project-map/test_generate.py",
26
+ ".claude/project-map/test_mine_sessions.py",
27
+ ".claude/project-map/test_grader.py",
19
28
  ".githooks/",
20
29
  "skills/",
21
30
  "install.sh",
22
- "checksums.json",
31
+ "./checksums.json",
23
32
  "LICENSE",
24
33
  "README.md",
25
34
  "CHANGELOG.md"
@@ -1,61 +0,0 @@
1
- # babel-fish — Project Map
2
-
3
- > Auto-generated by `generate.py` on 2026-04-01 20:29. Do not edit manually.
4
-
5
-
6
- ## Stats
7
-
8
- | Metric | Count |
9
- |--------|-------|
10
- | API Routes | 0 |
11
- | Data Models | 0 |
12
- | Schemas/DTOs | 0 |
13
- | Frontend Features | 0 |
14
- | Migrations | 0 |
15
- | Docker Services | 0 |
16
- | Vocabulary Entries | 0 |
17
- | Stack | unknown / unknown |
18
-
19
- ## Section Index
20
-
21
- | # | Section | Size | When to Read |
22
- |---|---------|------|--------------|
23
- | [01](sections/01-vocabulary.md) | 01 Vocabulary | 0.3 KB | Any task — start here if you don't know where the code lives |
24
- | [02](sections/02-service-topology.md) | 02 Service Topology | 0.1 KB | Debugging connectivity, adding a service, understanding ports |
25
- | [03](sections/03-environment.md) | 03 Environment | 0.1 KB | Environment setup, missing vars, config issues |
26
- | [04](sections/04-api-routes.md) | 04 Api Routes | 0.1 KB | Adding/editing API endpoints, checking what routes exist |
27
- | [05](sections/05-data-models.md) | 05 Data Models | 0.1 KB | Changing database schema, adding fields, understanding relations |
28
- | [06](sections/06-schemas.md) | 06 Schemas | 0.1 KB | Adding DTOs, changing request/response shapes |
29
- | [07](sections/07-services.md) | 07 Services | 0.1 KB | Adding service logic, understanding service boundaries |
30
- | [08](sections/08-background-jobs.md) | 08 Background Jobs | 0.1 KB | Working with background jobs, queues, scheduled tasks |
31
- | [09](sections/09-frontend-features.md) | 09 Frontend Features | 0.1 KB | Frontend feature work, understanding UI structure |
32
- | [10](sections/10-tools-commands.md) | 10 Tools Commands | 0.3 KB | Available commands, scripts, developer tooling |
33
- | [11](sections/11-migrations.md) | 11 Migrations | 0.1 KB | Database migrations, schema history |
34
- | [12](sections/12-import-chains.md) | 12 Import Chains | 0.2 KB | Tracing data flow from HTTP request to DB |
35
- | [13](sections/13-frontend-backend-map.md) | 13 Frontend Backend Map | 0.1 KB | Understanding which frontend calls which backend endpoint |
36
- | [14](sections/14-reverse-proxy.md) | 14 Reverse Proxy | 0.1 KB | Proxy routing, nginx/caddy config |
37
- | [15](sections/15-auth-config.md) | 15 Auth Config | 0.1 KB | Auth flow, sessions, permissions |
38
- | [16](sections/16-infra-profile.md) | 16 Infra Profile | 0.3 KB | Infrastructure overview, tech stack summary |
39
- | [17](sections/17-learned-vocabulary.md) | 17 Learned Vocabulary | 0.2 KB | Vocabulary learned from past sessions |
40
- | [18](sections/18-dead-code.md) | 18 Dead Code | 0.2 KB | Dead code review |
41
- | [19](sections/19-doc-pointers.md) | 19 Doc Pointers | 0.1 KB | Finding documentation, READMEs, wikis |
42
-
43
- ## Quick Routing
44
-
45
- | Task | Read Sections |
46
- |------|--------------|
47
- | Feature / UX work | 01 → 09 → 04 |
48
- | Add model or field | 05 → 06 → 12 |
49
- | Troubleshoot error | 02 → 03 → 14 |
50
- | Infrastructure / scaling | 16 → 02 |
51
- | Auth / security | 15 → 19 |
52
- | What tools exist | 10 |
53
- | Background jobs | 08 |
54
- | Migration history | 11 |
55
-
56
- ## Regenerate
57
-
58
- ```bash
59
- python .claude/project-map/generate.py # skip if unchanged
60
- python .claude/project-map/generate.py --force # always regenerate
61
- ```
@@ -1,6 +0,0 @@
1
- {
2
- "input_hash": "01b723fba1eeffe29b8a71be2fbd81197a22fc0e4fec0eb10a096361b4b3d054",
3
- "generated_at": "2026-04-01T20:29:02.877707",
4
- "route_count": 0,
5
- "model_count": 0
6
- }